{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"GAUSSIAN_NOISE = 0.1\nUPSAMPLE_MODE = 'SIMPLE'\n# number of validation images to use\nVALID_IMG_COUNT = 1500\n# maximum number of training images\nMAX_TRAIN_IMAGES = 10000\nBASE_MODEL='RESNET52' # ['VGG16', 'RESNET52']\nIMG_SIZE = (299, 299) # [(224, 224), (384, 384), (512, 512), (640, 640)]\nBATCH_SIZE = 64 # [1, 8, 16, 24]\nDROPOUT = 0.5\nDENSE_COUNT = 86\nLEARN_RATE = 1e-4\nRGB_FLIP = 1 # should rgb be flipped when rendering images","metadata":{"_uuid":"301a5d939c566d1487a049bb2554d09b592b18b1","execution":{"iopub.status.busy":"2023-08-24T00:28:07.104848Z","iopub.execute_input":"2023-08-24T00:28:07.105198Z","iopub.status.idle":"2023-08-24T00:28:07.111251Z","shell.execute_reply.started":"2023-08-24T00:28:07.105137Z","shell.execute_reply":"2023-08-24T00:28:07.110198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom skimage.io import imread\nimport matplotlib.pyplot as plt\nfrom skimage.segmentation import mark_boundaries\nfrom skimage.util.montage import montage2d as montage\nmontage_rgb = lambda x: np.stack([montage(x[:, :, :, i]) for i in range(x.shape[3])], -1)\nship_dir = '../input'\ntrain_image_dir = os.path.join(ship_dir, 'train_v2')\ntest_image_dir = os.path.join(ship_dir, 'test_v2')\nimport gc; gc.enable() # memory is tight","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-24T00:28:09.326391Z","iopub.execute_input":"2023-08-24T00:28:09.326745Z","iopub.status.idle":"2023-08-24T00:28:10.325914Z","shell.execute_reply.started":"2023-08-24T00:28:09.326683Z","shell.execute_reply":"2023-08-24T00:28:10.325034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"masks = pd.read_csv(os.path.join('../input/',\n                                 'train_ship_segmentations_v2.csv'))\nprint(masks.shape[0], 'masks found')\nprint(masks['ImageId'].value_counts().shape[0])\nmasks['path'] = masks['ImageId'].map(lambda x: os.path.join(train_image_dir, x))\nmasks.head()","metadata":{"_uuid":"3ca7119188fbb4c6540d9df55f5833b55435287e","execution":{"iopub.status.busy":"2023-08-24T00:28:13.433933Z","iopub.execute_input":"2023-08-24T00:28:13.434309Z","iopub.status.idle":"2023-08-24T00:28:15.428092Z","shell.execute_reply.started":"2023-08-24T00:28:13.434242Z","shell.execute_reply":"2023-08-24T00:28:15.427265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nmasks['ships'] = masks['EncodedPixels'].map(lambda c_row: 1 if isinstance(c_row, str) else 0)\nunique_img_ids = masks.groupby('ImageId').agg({'ships': 'sum'}).reset_index()\nunique_img_ids['has_ship'] = unique_img_ids['ships'].map(lambda x: 1.0 if x>0 else 0.0)\nunique_img_ids['has_ship_vec'] = unique_img_ids['has_ship'].map(lambda x: [x])\nmasks.drop(['ships'], axis=1, inplace=True)\ntrain_ids, valid_ids = train_test_split(unique_img_ids, \n                 test_size = 0.3, \n                 stratify = unique_img_ids['ships'])\ntrain_df = pd.merge(masks, train_ids)\nvalid_df = pd.merge(masks, valid_ids)\nprint(train_df.shape[0], 'training masks')\nprint(valid_df.shape[0], 'validation masks')","metadata":{"_uuid":"871720221ac25f7f9408bfe01aeb4ccb95edbd1f","execution":{"iopub.status.busy":"2023-08-24T00:28:23.050424Z","iopub.execute_input":"2023-08-24T00:28:23.050780Z","iopub.status.idle":"2023-08-24T00:28:24.342902Z","shell.execute_reply.started":"2023-08-24T00:28:23.050721Z","shell.execute_reply":"2023-08-24T00:28:24.341764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.sample(min(MAX_TRAIN_IMAGES, train_df.shape[0])) # limit size of training set (otherwise it takes too long)","metadata":{"_uuid":"fa2154d3621497e7497731e5ef6c6c72f27ad483","execution":{"iopub.status.busy":"2023-08-24T00:28:26.590860Z","iopub.execute_input":"2023-08-24T00:28:26.593809Z","iopub.status.idle":"2023-08-24T00:28:26.659827Z","shell.execute_reply.started":"2023-08-24T00:28:26.591226Z","shell.execute_reply":"2023-08-24T00:28:26.658782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[['ships', 'has_ship']].hist()","metadata":{"_uuid":"2612fa47c7e9fdcaa7aa720c4e15fc86fd65d69a","execution":{"iopub.status.busy":"2023-08-24T00:28:28.989875Z","iopub.execute_input":"2023-08-24T00:28:28.990255Z","iopub.status.idle":"2023-08-24T00:28:29.494275Z","shell.execute_reply.started":"2023-08-24T00:28:28.990175Z","shell.execute_reply":"2023-08-24T00:28:29.492954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\nif BASE_MODEL=='VGG16':\n    from keras.applications.vgg16 import VGG16 as PTModel, preprocess_input\nelif BASE_MODEL=='RESNET52':\n    from keras.applications.resnet50 import ResNet50 as PTModel, preprocess_input\nelse:\n    raise ValueError('Unknown model: {}'.format(BASE_MODEL))","metadata":{"_uuid":"ccbc1747ffb0f2942cc99bc27e4afc3262ba9f94","execution":{"iopub.status.busy":"2023-08-24T00:28:32.911327Z","iopub.execute_input":"2023-08-24T00:28:32.911693Z","iopub.status.idle":"2023-08-24T00:28:33.232925Z","shell.execute_reply.started":"2023-08-24T00:28:32.911629Z","shell.execute_reply":"2023-08-24T00:28:33.232180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\ndg_args = dict(featurewise_center = False, \n                  samplewise_center = False,\n                  rotation_range = 45, \n                  width_shift_range = 0.1, \n                  height_shift_range = 0.1, \n                  shear_range = 0.01,\n                  zoom_range = [0.9, 1.25],  \n                  brightness_range = [0.5, 1.5],\n                  horizontal_flip = True, \n                  vertical_flip = True,\n                  fill_mode = 'reflect',\n                   data_format = 'channels_last',\n              preprocessing_function = preprocess_input)\nvalid_args = dict(fill_mode = 'reflect',\n                   data_format = 'channels_last',\n                  preprocessing_function = preprocess_input)\n\ncore_idg = ImageDataGenerator(**dg_args)\nvalid_idg = ImageDataGenerator(**valid_args)","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2023-08-24T00:28:37.894671Z","iopub.execute_input":"2023-08-24T00:28:37.895043Z","iopub.status.idle":"2023-08-24T00:28:37.903759Z","shell.execute_reply.started":"2023-08-24T00:28:37.894968Z","shell.execute_reply":"2023-08-24T00:28:37.902591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def flow_from_dataframe(img_data_gen, in_df, path_col, y_col, **dflow_args):\n    base_dir = os.path.dirname(in_df[path_col].values[0])\n    print('## Ignore next message from keras, values are replaced anyways')\n    df_gen = img_data_gen.flow_from_directory(base_dir, \n                                     class_mode = 'sparse',\n                                    **dflow_args)\n    df_gen.filenames = in_df[path_col].values\n    df_gen.classes = np.stack(in_df[y_col].values)\n    df_gen.samples = in_df.shape[0]\n    df_gen.n = in_df.shape[0]\n    df_gen._set_index_array()\n    df_gen.directory = '' # since we have the full path\n    print('Reinserting dataframe: {} images'.format(in_df.shape[0]))\n    return df_gen","metadata":{"_uuid":"c5cc31e3117fdfa923b2acb1e6542a7007f4f955","execution":{"iopub.status.busy":"2023-08-24T00:28:57.274534Z","iopub.execute_input":"2023-08-24T00:28:57.274899Z","iopub.status.idle":"2023-08-24T00:28:57.283877Z","shell.execute_reply.started":"2023-08-24T00:28:57.274840Z","shell.execute_reply":"2023-08-24T00:28:57.282543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_gen = flow_from_dataframe(core_idg, train_df, \n                             path_col = 'path',\n                            y_col = 'has_ship_vec', \n                            target_size = IMG_SIZE,\n                             color_mode = 'rgb',\n                            batch_size = BATCH_SIZE)\n\n# used a fixed dataset for evaluating the algorithm\nvalid_x, valid_y = next(flow_from_dataframe(valid_idg, \n                               valid_df, \n                             path_col = 'path',\n                            y_col = 'has_ship_vec', \n                            target_size = IMG_SIZE,\n                             color_mode = 'rgb',\n                            batch_size = VALID_IMG_COUNT)) # one big batch\nprint(valid_x.shape, valid_y.shape)","metadata":{"_uuid":"67136e1743dbbc4e07dba0d69f79231603f31d93","execution":{"iopub.status.busy":"2023-08-24T00:29:00.878676Z","iopub.execute_input":"2023-08-24T00:29:00.879009Z","iopub.status.idle":"2023-08-24T00:32:32.180942Z","shell.execute_reply.started":"2023-08-24T00:29:00.878948Z","shell.execute_reply":"2023-08-24T00:32:32.180121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_x, t_y = next(train_gen)\nprint('x', t_x.shape, t_x.dtype, t_x.min(), t_x.max())\nprint('y', t_y.shape, t_y.dtype, t_y.min(), t_y.max())\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize = (20, 10))\nax1.imshow(montage_rgb((t_x-t_x.min())/(t_x.max()-t_x.min()))[:, :, ::RGB_FLIP], cmap='gray')\nax1.set_title('images')\nax2.plot(t_y)\nax2.set_title('ships')","metadata":{"_uuid":"6122ccb9e58bfac6fa5e11c86121e78d9e5151b1","execution":{"iopub.status.busy":"2023-08-24T00:33:47.702046Z","iopub.execute_input":"2023-08-24T00:33:47.702381Z","iopub.status.idle":"2023-08-24T00:33:51.950757Z","shell.execute_reply.started":"2023-08-24T00:33:47.702318Z","shell.execute_reply":"2023-08-24T00:33:51.949671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model\n","metadata":{"_uuid":"ba08494eb9736ec3556b7c879143cdcdea89febf"}},{"cell_type":"code","source":"base_pretrained_model = PTModel(input_shape =  t_x.shape[1:], \n                              include_top = False, \n                                weights = 'imagenet')\nbase_pretrained_model.trainable = False","metadata":{"_uuid":"9fb62d14ec059f7f92d82e07b35822169c77112d","execution":{"iopub.status.busy":"2023-08-24T00:34:04.079732Z","iopub.execute_input":"2023-08-24T00:34:04.080106Z","iopub.status.idle":"2023-08-24T00:39:34.063817Z","shell.execute_reply.started":"2023-08-24T00:34:04.080040Z","shell.execute_reply":"2023-08-24T00:39:34.062893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Setup the Subsequent Layers\nHere we setup the rest of the model which we will actually be training","metadata":{"_uuid":"0bc7c0ee177697838d22aed4de046e7542027a10"}},{"cell_type":"code","source":"from keras import models, layers\nfrom keras.optimizers import Adam\nimg_in = layers.Input(t_x.shape[1:], name='Image_RGB_In')\nimg_noise = layers.GaussianNoise(GAUSSIAN_NOISE)(img_in)\npt_features = base_pretrained_model(img_noise)\npt_depth = base_pretrained_model.get_output_shape_at(0)[-1]\nbn_features = layers.BatchNormalization()(pt_features)\nfeature_dropout = layers.SpatialDropout2D(DROPOUT)(bn_features)\ngmp_dr = layers.GlobalMaxPooling2D()(feature_dropout)\ndr_steps = layers.Dropout(DROPOUT)(layers.Dense(DENSE_COUNT, activation = 'relu')(gmp_dr))\nout_layer = layers.Dense(1, activation = 'sigmoid')(dr_steps)\n\nship_model = models.Model(inputs = [img_in], outputs = [out_layer], name = 'full_model')\n\nship_model.compile(optimizer = Adam(lr=LEARN_RATE), \n                   loss = 'binary_crossentropy',\n                   metrics = ['binary_accuracy'])\n\nship_model.summary()","metadata":{"_uuid":"2687377309d3cbbab1197f4eccd2b50ab996f5a6","execution":{"iopub.status.busy":"2023-08-24T00:45:49.474417Z","iopub.execute_input":"2023-08-24T00:45:49.474779Z","iopub.status.idle":"2023-08-24T00:45:54.790476Z","shell.execute_reply.started":"2023-08-24T00:45:49.474713Z","shell.execute_reply":"2023-08-24T00:45:54.789492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.callbacks import ModelCheckpoint, LearningRateScheduler, EarlyStopping, ReduceLROnPlateau\nweight_path=\"{}_weights.best.hdf5\".format('boat_detector')\n\ncheckpoint = ModelCheckpoint(weight_path, monitor='val_loss', verbose=1, \n                             save_best_only=True, mode='min', save_weights_only = True)\n\nreduceLROnPlat = ReduceLROnPlateau(monitor='val_loss', factor=0.8, patience=10, verbose=1, mode='auto', epsilon=0.0001, cooldown=5, min_lr=0.0001)\nearly = EarlyStopping(monitor=\"val_loss\", \n                      mode=\"min\", \n                      patience=10) # probably needs to be more patient, but kaggle time is limited\ncallbacks_list = [checkpoint, early, reduceLROnPlat]","metadata":{"_uuid":"7282d18de3aff1cee12ff89b7d511a391702814f","execution":{"iopub.status.busy":"2023-08-24T00:45:58.190285Z","iopub.execute_input":"2023-08-24T00:45:58.190636Z","iopub.status.idle":"2023-08-24T00:45:58.199808Z","shell.execute_reply.started":"2023-08-24T00:45:58.190560Z","shell.execute_reply":"2023-08-24T00:45:58.198890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_gen.batch_size = BATCH_SIZE\nship_model.fit_generator(train_gen, \n                         steps_per_epoch=train_gen.n//BATCH_SIZE,\n                      validation_data=(valid_x, valid_y), \n                      epochs=25, \n                      callbacks=callbacks_list,\n                      workers=3)","metadata":{"_uuid":"5b67d808c0b8c7e28bff41e6d3858ff6f09dd626","execution":{"iopub.status.busy":"2023-08-24T00:46:02.665895Z","iopub.execute_input":"2023-08-24T00:46:02.666250Z","iopub.status.idle":"2023-08-24T02:50:44.049354Z","shell.execute_reply.started":"2023-08-24T00:46:02.666183Z","shell.execute_reply":"2023-08-24T02:50:44.048366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ship_model.load_weights(weight_path)\nship_model.save('full_ship_model.h5')","metadata":{"_uuid":"a168c8b1af446b800f6129104906003ededd61c4","execution":{"iopub.status.busy":"2023-08-24T02:58:40.898068Z","iopub.execute_input":"2023-08-24T02:58:40.898744Z","iopub.status.idle":"2023-08-24T02:58:42.444819Z","shell.execute_reply.started":"2023-08-24T02:58:40.898438Z","shell.execute_reply":"2023-08-24T02:58:42.443921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Run the test data\nWe use the sample_submission file as the basis for loading and running the images.","metadata":{"_uuid":"17edb177402ae51651692511827a7e9d60646533"}},{"cell_type":"code","source":"test_paths = os.listdir(test_image_dir)\nprint(len(test_paths), 'test images found')\nsubmission_df = pd.read_csv('../input/sample_submission_v2.csv')\nsubmission_df['path'] = submission_df['ImageId'].map(lambda x: os.path.join(test_image_dir, x))","metadata":{"_uuid":"4911811f267f9f3397a58902da9e75c6f261ad40","execution":{"iopub.status.busy":"2023-08-24T02:58:45.157649Z","iopub.execute_input":"2023-08-24T02:58:45.158002Z","iopub.status.idle":"2023-08-24T02:58:45.484168Z","shell.execute_reply.started":"2023-08-24T02:58:45.157941Z","shell.execute_reply":"2023-08-24T02:58:45.483271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Setup Test Data Generator\nWe use the same generator as before to read and preprocess images","metadata":{"_uuid":"5b3eea954bd883d598c6ac80167dfd4353b0f558"}},{"cell_type":"code","source":"test_gen = flow_from_dataframe(valid_idg, \n                               submission_df, \n                             path_col = 'path',\n                            y_col = 'ImageId', \n                            target_size = IMG_SIZE,\n                             color_mode = 'rgb',\n                            batch_size = BATCH_SIZE, \n                              shuffle = False)","metadata":{"_uuid":"f75595679ba8606fd1ac15645b8612e117db0b30","execution":{"iopub.status.busy":"2023-08-24T02:58:48.459258Z","iopub.execute_input":"2023-08-24T02:58:48.459606Z","iopub.status.idle":"2023-08-24T02:59:13.467628Z","shell.execute_reply.started":"2023-08-24T02:58:48.459526Z","shell.execute_reply":"2023-08-24T02:59:13.466720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score, accuracy_score\nimport matplotlib.pyplot as plt\n\n# Obtener las predicciones del modelo en el conjunto de validación\nvalid_pred = ship_model.predict(valid_x)\n\n# Redondear las predicciones a 0 o 1 (dependiendo de un umbral)\nthreshold = 0.5  # Puedes ajustar este umbral según tus necesidades\nvalid_pred_rounded = (valid_pred > threshold).astype(int)\n\n# Calcular el F1 Score\nf1 = f1_score(valid_y, valid_pred_rounded)\n\n# Calcular el Accuracy\naccuracy = accuracy_score(valid_y, valid_pred_rounded)\n\nprint(f'F1 Score: {f1}')\nprint(f'Accuracy: {accuracy}')\n\n# Gráfica de las métricas durante el entrenamiento\nhistory = ship_model.history.history\nplt.figure(figsize=(12, 4))\nplt.subplot(1, 2, 1)\nplt.plot(history['binary_accuracy'], label='Training Accuracy')\nplt.plot(history['val_binary_accuracy'], label='Validation Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(history['loss'], label='Training Loss')\nplt.plot(history['val_loss'], label='Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-24T02:59:24.591132Z","iopub.execute_input":"2023-08-24T02:59:24.591542Z","iopub.status.idle":"2023-08-24T02:59:42.581983Z","shell.execute_reply.started":"2023-08-24T02:59:24.591475Z","shell.execute_reply":"2023-08-24T02:59:42.580887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, m_axs = plt.subplots(3, 2, figsize = (20, 30))\nfor (ax1, ax2), (t_x, c_img_names) in zip(m_axs, test_gen):\n    t_y = ship_model.predict(t_x)\n    t_stack = ((t_x-t_x.min())/(t_x.max()-t_x.min()))[:, :, :, ::RGB_FLIP]\n    ax1.imshow(montage_rgb(t_stack))\n    ax1.set_title('images')\n    alpha_stack = np.tile(np.expand_dims(np.expand_dims(t_y, -1), -1), [1, t_stack.shape[1], t_stack.shape[2], 1])\n    rgba_stack = np.concatenate([t_stack, alpha_stack], -1)\n    ax2.imshow(montage_rgb(rgba_stack))\n    ax2.set_title('ships')\nfig.savefig('test_predictions.png')","metadata":{"_uuid":"73ef7b3b2a74bf64968c79b4005075d4f0e23143","execution":{"iopub.status.busy":"2023-08-24T03:00:48.976907Z","iopub.execute_input":"2023-08-24T03:00:48.977255Z","iopub.status.idle":"2023-08-24T03:01:02.524442Z","shell.execute_reply.started":"2023-08-24T03:00:48.977193Z","shell.execute_reply":"2023-08-24T03:01:02.520801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The Scores","metadata":{"_uuid":"4a942dbb4a939d73526dbb35745402b35785fdf4"}}]}