{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":9988,"databundleVersionId":868324,"sourceType":"competition"}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install segmentation-models","metadata":{"execution":{"iopub.status.busy":"2024-07-13T23:35:30.635654Z","iopub.execute_input":"2024-07-13T23:35:30.636000Z","iopub.status.idle":"2024-07-13T23:35:45.276802Z","shell.execute_reply.started":"2024-07-13T23:35:30.635973Z","shell.execute_reply":"2024-07-13T23:35:45.275607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.environ[\"SM_FRAMEWORK\"] = \"tf.keras\"\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-13T23:35:45.279383Z","iopub.execute_input":"2024-07-13T23:35:45.280199Z","iopub.status.idle":"2024-07-13T23:35:45.285213Z","shell.execute_reply.started":"2024-07-13T23:35:45.280156Z","shell.execute_reply":"2024-07-13T23:35:45.284092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport time\nimport tensorflow as tf\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.optimizers import Adam\nimport segmentation_models as sm\nfrom segmentation_models import Unet\nfrom segmentation_models.losses import DiceLoss\nfrom segmentation_models.metrics import FScore\nfrom tensorflow.keras.applications.resnet import preprocess_input\nfrom tensorflow.keras.callbacks import Callback\n","metadata":{"execution":{"iopub.status.busy":"2024-07-13T23:35:45.286741Z","iopub.execute_input":"2024-07-13T23:35:45.287176Z","iopub.status.idle":"2024-07-13T23:35:57.618336Z","shell.execute_reply.started":"2024-07-13T23:35:45.287142Z","shell.execute_reply":"2024-07-13T23:35:57.617353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#завантаження даних\ndf = pd.read_csv('/kaggle/input/airbus-ship-detection/train_ship_segmentations_v2.csv')\nimage_dir = '/kaggle/input/airbus-ship-detection/train_v2'\ndf","metadata":{"execution":{"iopub.status.busy":"2024-07-13T23:35:57.620625Z","iopub.execute_input":"2024-07-13T23:35:57.621153Z","iopub.status.idle":"2024-07-13T23:35:58.618191Z","shell.execute_reply.started":"2024-07-13T23:35:57.621126Z","shell.execute_reply":"2024-07-13T23:35:58.617114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# групування EncodedPixels у список та зменшення загальної кількості зображень \n\ndf['EncodedPixels'] = df['EncodedPixels'].astype(str) \n\ndf_listed = df.groupby('ImageId')['EncodedPixels'].apply(lambda x: x.tolist()).reset_index()\n\ngrouped = df_listed.groupby(df_listed['EncodedPixels'].apply(lambda x: x == ['nan']))\n\nnan_group = grouped.filter(lambda x: x['EncodedPixels'].iloc[0] == ['nan'])\nnot_nan_group = grouped.filter(lambda x: x['EncodedPixels'].iloc[0] != ['nan'])\n\nreduced_not_nan_group = not_nan_group.sample(1500)\nreduced_nan_group = nan_group.sample(1500)\n\ndf_balanced = pd.concat([reduced_not_nan_group, reduced_nan_group])\ndf_balanced","metadata":{"execution":{"iopub.status.busy":"2024-07-13T23:35:58.619545Z","iopub.execute_input":"2024-07-13T23:35:58.619877Z","iopub.status.idle":"2024-07-13T23:36:03.902265Z","shell.execute_reply.started":"2024-07-13T23:35:58.619849Z","shell.execute_reply":"2024-07-13T23:36:03.901329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# візуалізація співвідношення\ndf_balanced_to_viz = df_balanced.copy()\n\ndf_balanced_to_viz['has_ship'] = df_balanced_to_viz['EncodedPixels'].apply(lambda x: 0 if x == ['nan'] else 1)\n\n\ngrouped_df_to_viz = df_balanced_to_viz.groupby('ImageId')['has_ship'].max().reset_index()\n\n\nsns.countplot(x='has_ship', data=grouped_df_to_viz)\nplt.xlabel('Has Ship')\nplt.ylabel('Count')\nplt.title('Distribution of Images with/without Ships')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-13T23:36:03.903382Z","iopub.execute_input":"2024-07-13T23:36:03.903680Z","iopub.status.idle":"2024-07-13T23:36:04.181502Z","shell.execute_reply.started":"2024-07-13T23:36:03.903655Z","shell.execute_reply":"2024-07-13T23:36:04.180556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# створення генератора даних \n\nclass ShipDataset(Sequence):\n    def __init__(self, df, image_dir, image_shape=(768,768), batch_size=32, shuffle=True, transform=None, preprocessing_fn=None):\n        self.df = df\n        self.image_dir = image_dir\n        self.shape = image_shape\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.transform = transform\n        self.preprocessing_fn = preprocessing_fn\n        \n        #метод ініціалізаціі індексів та перемішування при необхідності\n        self.on_epoch_end()\n    \n    def __len__(self):\n        #повертає кількість батчів на одну епоху\n        return int(np.ceil(len(self.df) / self.batch_size))\n    \n    def __getitem__(self, idx):\n        #генерування одного батчу даних\n        \n        batch_indices = self.indices[idx * self.batch_size:(idx + 1) * self.batch_size]\n        batch_df = self.df.iloc[batch_indices]\n        \n        images, masks = [], []\n        for _, row in batch_df.iterrows():\n            image, mask = self.load_image_and_mask(row)\n            images.append(image)\n            masks.append(mask)\n        \n        images = np.array(images)\n        masks = np.array(masks)\n        \n        #зміна типу та додавання додаткового виміру для масок\n        masks = masks.astype(np.float32)\n        masks = np.expand_dims(masks, axis=-1)\n        \n        if self.transform:\n            images = np.array([self.transform(image=image)['image'] for image in images])\n        \n        if self.preprocessing_fn:\n            images = self.preprocessing_fn(images)\n        \n        return images, masks\n    \n    def on_epoch_end(self):\n        self.indices = np.arange(len(self.df))\n        \n        if self.shuffle:\n            np.random.shuffle(self.indices)\n    \n    def load_image_and_mask(self, row):\n        image_name = row['ImageId']\n        image_path = os.path.join(self.image_dir, image_name)\n        image = img_to_array(load_img(image_path, target_size=self.shape))\n        \n        rles = row['EncodedPixels']\n        mask = self.combine_rle_masks(rles, self.shape)\n        \n        return image, mask\n    \n    def rle_to_mask(self, rle, shape):\n        \n        #конвертація RLE рядка у маску\n        \n        mask = np.zeros(shape[0] * shape[1], dtype=np.uint8)\n\n        if rle == 'nan':\n            return mask.reshape(shape)\n\n        rle_nums = list(map(int, rle.split()))\n        starts = rle_nums[0::2]\n        lengths = rle_nums[1::2]\n        starts = [start - 1 for start in starts]\n\n        for start, length in zip(starts, lengths):\n            mask[start:start + length] = 1\n\n        return mask.reshape(shape).T\n\n    def combine_rle_masks(self, rles, shape):\n        \n        # комбінування масок в одну \n        \n        combined_mask = np.zeros(shape, dtype=np.uint8)\n\n        for rle in rles:\n            if rle != 'nan':\n                mask = self.rle_to_mask(rle, shape)\n                combined_mask = np.maximum(combined_mask, mask)\n\n        return combined_mask","metadata":{"execution":{"iopub.status.busy":"2024-07-13T23:36:04.182758Z","iopub.execute_input":"2024-07-13T23:36:04.183048Z","iopub.status.idle":"2024-07-13T23:36:04.201955Z","shell.execute_reply.started":"2024-07-13T23:36:04.183024Z","shell.execute_reply":"2024-07-13T23:36:04.200950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#візуалізація зображень та іх масок \n\nvizualize_dataset = ShipDataset(df_balanced, image_dir, batch_size=5, image_shape=(768, 768))\n\nimages, masks = vizualize_dataset[0]\n\nplt.figure(figsize=(15, 10))\nfor i in range(len(images)):\n    plt.subplot(5, 2, 2*i + 1)\n    plt.imshow(images[i].astype('uint8'))\n    plt.title('Image')\n    plt.axis('off')\n\n    plt.subplot(5, 2, 2*i + 2)\n    plt.imshow(masks[i], cmap='gray')\n    plt.title('Mask')\n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-13T23:36:04.203350Z","iopub.execute_input":"2024-07-13T23:36:04.203693Z","iopub.status.idle":"2024-07-13T23:36:05.814288Z","shell.execute_reply.started":"2024-07-13T23:36:04.203666Z","shell.execute_reply":"2024-07-13T23:36:05.813240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# розділення даних на тренувальні та валідаційні\n\ntrain_df, val_df = train_test_split(df_balanced, train_size=0.8, shuffle=True)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2024-07-13T23:36:05.815889Z","iopub.execute_input":"2024-07-13T23:36:05.816531Z","iopub.status.idle":"2024-07-13T23:36:05.834158Z","shell.execute_reply.started":"2024-07-13T23:36:05.816494Z","shell.execute_reply":"2024-07-13T23:36:05.833264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# створення датасетів\ntrain_dataset = ShipDataset(train_df, image_dir, image_shape=(768,768), batch_size=16, preprocessing_fn=preprocess_input)\nval_dataset = ShipDataset(val_df, image_dir, image_shape=(768,768), batch_size=16, preprocessing_fn=preprocess_input)","metadata":{"execution":{"iopub.status.busy":"2024-07-13T23:36:05.837532Z","iopub.execute_input":"2024-07-13T23:36:05.837819Z","iopub.status.idle":"2024-07-13T23:36:05.843031Z","shell.execute_reply.started":"2024-07-13T23:36:05.837794Z","shell.execute_reply":"2024-07-13T23:36:05.842059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#створення моделі\nmodel = sm.Unet('resnet18',\n                input_shape=(768, 768, 3),\n                classes=1,\n                activation='sigmoid')\n\nfor layer in model.layers:\n    if 'decoder' not in layer.name:\n        layer.trainable = False\n    if layer.name in ['final_conv', 'sigmoid']:\n        layer.trainable = True\n\noptimizer = Adam(learning_rate=0.0001)\n\nmodel.compile(optimizer=optimizer, loss=DiceLoss(), metrics=[FScore()])","metadata":{"execution":{"iopub.status.busy":"2024-07-13T23:36:05.844136Z","iopub.execute_input":"2024-07-13T23:36:05.844464Z","iopub.status.idle":"2024-07-13T23:36:08.020097Z","shell.execute_reply.started":"2024-07-13T23:36:05.844431Z","shell.execute_reply":"2024-07-13T23:36:08.019257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-07-13T23:36:08.021370Z","iopub.execute_input":"2024-07-13T23:36:08.021752Z","iopub.status.idle":"2024-07-13T23:36:08.201760Z","shell.execute_reply.started":"2024-07-13T23:36:08.021717Z","shell.execute_reply":"2024-07-13T23:36:08.200280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in model.layers:\n    print(f'{layer.name}: Trainable={layer.trainable}')","metadata":{"execution":{"iopub.status.busy":"2024-07-13T23:36:08.204486Z","iopub.execute_input":"2024-07-13T23:36:08.204799Z","iopub.status.idle":"2024-07-13T23:36:08.211511Z","shell.execute_reply.started":"2024-07-13T23:36:08.204774Z","shell.execute_reply":"2024-07-13T23:36:08.210428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# створення класу для відстежування процессу навчання\n\nclass TrainingStats(Callback):\n    def on_epoch_begin(self, epoch, logs=None):\n        self.epoch_start_time = time.time()\n\n    def on_epoch_end(self, epoch, logs=None):\n        epoch_time = time.time() - self.epoch_start_time\n        train_loss = logs.get('loss')\n        val_loss = logs.get('val_loss')\n        train_recall = logs.get('f1-score')\n        val_recall = logs.get('val_f1-score')\n        print(f\"\\n Epoch {epoch + 1}/{self.params['epochs']}, \"\n              f\"Train Loss: {train_loss:.4f}, Val Loss: {val_loss:.4f}, \"\n              f\"Train score: {train_recall:.4f}, Val score: {val_recall:.4f}, \"\n              f\"Epoch Time: {epoch_time:.2f}\")","metadata":{"execution":{"iopub.status.busy":"2024-07-13T23:36:08.212900Z","iopub.execute_input":"2024-07-13T23:36:08.213236Z","iopub.status.idle":"2024-07-13T23:36:08.220865Z","shell.execute_reply.started":"2024-07-13T23:36:08.213211Z","shell.execute_reply":"2024-07-13T23:36:08.219848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore', category=Warning)\n\nhistory = model.fit(train_dataset, validation_data=val_dataset, epochs=10, callbacks=[TrainingStats()])\n\nmodel.save('saved_model.keras')","metadata":{"_kg_hide-output":false,"execution":{"iopub.status.busy":"2024-07-13T23:36:08.222047Z","iopub.execute_input":"2024-07-13T23:36:08.222392Z","iopub.status.idle":"2024-07-14T00:08:50.265076Z","shell.execute_reply.started":"2024-07-13T23:36:08.222363Z","shell.execute_reply":"2024-07-14T00:08:50.264139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Візуалізація історії тренування моделі\n\ndef plot_training(history):\n    f1_score = history.history['f1-score']\n    val_f1_score = history.history['val_f1-score']\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n    epochs = range(1, len(f1_score) + 1)\n\n    plt.figure(figsize=(12, 5))\n\n    # Plot F1-score\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs, f1_score, 'r', label='Training F1-score')\n    plt.plot(epochs, val_f1_score, 'b', label='Validation F1-score')\n    plt.title('Training and validation F1-score')\n    plt.xlabel('Epochs')\n    plt.ylabel('F1-score')\n    plt.legend()\n\n    # Plot loss\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs, loss, 'r', label='Training loss')\n    plt.plot(epochs, val_loss, 'b', label='Validation loss')\n    plt.title('Training and validation loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend()\n\n    plt.tight_layout()\n    plt.show()\n\nplot_training(history)","metadata":{"execution":{"iopub.status.busy":"2024-07-14T00:08:50.266420Z","iopub.execute_input":"2024-07-14T00:08:50.266718Z","iopub.status.idle":"2024-07-14T00:08:50.822695Z","shell.execute_reply.started":"2024-07-14T00:08:50.266685Z","shell.execute_reply":"2024-07-14T00:08:50.821748Z"},"trusted":true},"execution_count":null,"outputs":[]}]}