{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:11:26.094852Z","iopub.execute_input":"2022-03-27T10:11:26.095134Z","iopub.status.idle":"2022-03-27T10:11:26.104425Z","shell.execute_reply.started":"2022-03-27T10:11:26.095082Z","shell.execute_reply":"2022-03-27T10:11:26.101222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/background-removed-happywhale-dataset/seg_train.csv')\ndf_train","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:11:27.633911Z","iopub.execute_input":"2022-03-27T10:11:27.634372Z","iopub.status.idle":"2022-03-27T10:11:27.772251Z","shell.execute_reply.started":"2022-03-27T10:11:27.634333Z","shell.execute_reply":"2022-03-27T10:11:27.771541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:11:37.558397Z","iopub.execute_input":"2022-03-27T10:11:37.55865Z","iopub.status.idle":"2022-03-27T10:11:37.65936Z","shell.execute_reply.started":"2022-03-27T10:11:37.558621Z","shell.execute_reply":"2022-03-27T10:11:37.658566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_corr = df_train.corr\ndf_corr","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:11:40.053011Z","iopub.execute_input":"2022-03-27T10:11:40.053609Z","iopub.status.idle":"2022-03-27T10:11:40.063333Z","shell.execute_reply.started":"2022-03-27T10:11:40.053568Z","shell.execute_reply":"2022-03-27T10:11:40.062468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['box'] = df_train['box'].fillna(0)\ndf_train","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:11:43.70058Z","iopub.execute_input":"2022-03-27T10:11:43.700832Z","iopub.status.idle":"2022-03-27T10:11:43.721681Z","shell.execute_reply.started":"2022-03-27T10:11:43.700803Z","shell.execute_reply":"2022-03-27T10:11:43.720981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"image\"].replace('.png')\ndf_train[\"image\"]","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:14:18.227478Z","iopub.execute_input":"2022-03-27T10:14:18.227747Z","iopub.status.idle":"2022-03-27T10:14:18.239614Z","shell.execute_reply.started":"2022-03-27T10:14:18.227717Z","shell.execute_reply":"2022-03-27T10:14:18.238825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nimport pandas as pd\nDATA_DIR  = '../input/happy-whale-and-dolphin/'\nTRAIN_DIR = DATA_DIR + 'train_images/'\n# TEST_DIR  = DATA_DIR + 'resized_test_images/'\n\n\n\nlabel_encoder = LabelEncoder()\n\n# Load Train Data\ntrain_df = pd.read_csv('../input/background-removed-happywhale-dataset/seg_train.csv')\ntrain_df['Id'] = train_df['image'].apply(lambda x: f'{TRAIN_DIR}{x}')\n\n# Adjust typos in \"species\" column from Andrada's kernel\ntrain_df[\"species\"] = train_df[\"species\"].replace([\"bottlenose_dolpin\", \"kiler_whale\",\n                                             \"beluga\", \n                                             \"globis\", \"pilot_whale\"],\n                                            [\"bottlenose_dolphin\", \"killer_whale\",\n                                             \"beluga_whale\", \n                                             \"short_finned_pilot_whale\", \"short_finned_pilot_whale\"])\n\n\n# Set a specific label to be able to perform stratification\n#train_df['stratify_label'] = train_df['individual_id']\n\ntrain_df['target_value']  = label_encoder.fit_transform(train_df['individual_id'] )\n\n# Summary\nprint(f'train_df: {train_df.shape}')\ntrain_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:55:11.036937Z","iopub.execute_input":"2022-03-27T10:55:11.037631Z","iopub.status.idle":"2022-03-27T10:55:12.191979Z","shell.execute_reply.started":"2022-03-27T10:55:11.037539Z","shell.execute_reply":"2022-03-27T10:55:12.191209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['box'] = train_df['box'].astype(\"string\")\ntrain_df.to_csv('train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:10:07.317449Z","iopub.execute_input":"2022-03-27T09:10:07.317703Z","iopub.status.idle":"2022-03-27T09:10:07.589258Z","shell.execute_reply.started":"2022-03-27T09:10:07.31767Z","shell.execute_reply":"2022-03-27T09:10:07.587634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info","metadata":{"execution":{"iopub.status.busy":"2022-03-27T09:10:21.915134Z","iopub.execute_input":"2022-03-27T09:10:21.915679Z","iopub.status.idle":"2022-03-27T09:10:21.926278Z","shell.execute_reply.started":"2022-03-27T09:10:21.915642Z","shell.execute_reply":"2022-03-27T09:10:21.925351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.layers import AveragePooling2D\nfrom tensorflow.keras.layers import Dropout\nfrom tensorflow.keras.layers import Flatten\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.layers import Input\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.applications.resnet50 import preprocess_input\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom sklearn.metrics import classification_report\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport argparse\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:55:18.453682Z","iopub.execute_input":"2022-03-27T10:55:18.453939Z","iopub.status.idle":"2022-03-27T10:55:23.797177Z","shell.execute_reply.started":"2022-03-27T10:55:18.453911Z","shell.execute_reply":"2022-03-27T10:55:23.796296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"box\"] = train_df[\"box\"].fillna(0)\ntrain_df[\"box\"] = train_df[\"box\"].astype(\"string\")","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:55:30.355232Z","iopub.execute_input":"2022-03-27T10:55:30.355782Z","iopub.status.idle":"2022-03-27T10:55:30.377705Z","shell.execute_reply.started":"2022-03-27T10:55:30.355743Z","shell.execute_reply":"2022-03-27T10:55:30.377017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"box\"] = train_df[\"box\"].fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:54:40.038031Z","iopub.execute_input":"2022-03-27T10:54:40.038508Z","iopub.status.idle":"2022-03-27T10:54:40.092551Z","shell.execute_reply.started":"2022-03-27T10:54:40.038471Z","shell.execute_reply":"2022-03-27T10:54:40.090888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndatagen = ImageDataGenerator(\n    featurewise_center=False,\n    samplewise_center=False,\n    featurewise_std_normalization=False,\n    samplewise_std_normalization=False,\n    preprocessing_function=preprocess_input,\n    validation_split=0.3)\n\n\ntrain_generator=datagen.flow_from_dataframe(\ndataframe=train_df,\n# directory=\"../input/happy-whale-and-dolphin/train_images\",\nx_col=\"Id\",\ny_col=[\"individual_id\",\"box\"],\nsubset=\"training\",\nbatch_size=256,\n# seed=2022,\nshuffle=True,\nclass_mode=\"multi_output\",\ntarget_size=(512,512))\n\nvalid_generator=datagen.flow_from_dataframe(\ndataframe=train_df,\n# directory=\"../input/happy-whale-and-dolphin/train_images\",\nx_col=\"Id\",\ny_col=[\"individual_id\",\"box\"],\nsubset=\"validation\",\nbatch_size=256,\n# seed=2022,\nshuffle=True,\nclass_mode=\"multi_output\",\ntarget_size=(512,512))","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:55:36.603236Z","iopub.execute_input":"2022-03-27T10:55:36.603914Z","iopub.status.idle":"2022-03-27T10:56:31.585885Z","shell.execute_reply.started":"2022-03-27T10:55:36.603878Z","shell.execute_reply":"2022-03-27T10:56:31.584325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimg = cv2.imread('../input/background-removed-happywhale-dataset/seg_img/00021adfb725ed.png')\nimg.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:25:27.086658Z","iopub.execute_input":"2022-03-27T10:25:27.087193Z","iopub.status.idle":"2022-03-27T10:25:27.395607Z","shell.execute_reply.started":"2022-03-27T10:25:27.087156Z","shell.execute_reply":"2022-03-27T10:25:27.394943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = ResNet50(weights=\"imagenet\", include_top=False, input_shape=(512,512, 3))\n\n# Construct the head of the model that will be placed on top of the base model\nhead_model = base_model.output\nhead_model = AveragePooling2D(pool_size=(3,3))(head_model)\nhead_model = Flatten(name=\"flatten\")(head_model)\n# head_model = Dropout(0.5)(head_model)\nhead_model = Dense(512, activation=\"relu\")(head_model)\nhead_model = Dropout(0.5)(head_model)\nhead_model = Dense(15587, activation=\"softmax\")(head_model)\n\n# Place the head FC model on top of the base model (this will become the actual model we will train)\nmodel = Model(inputs=base_model.input, outputs=head_model)","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:56:34.825684Z","iopub.execute_input":"2022-03-27T10:56:34.82594Z","iopub.status.idle":"2022-03-27T10:56:38.944301Z","shell.execute_reply.started":"2022-03-27T10:56:34.825913Z","shell.execute_reply":"2022-03-27T10:56:38.943574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in base_model.layers:\n    layer.trainable = False\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:56:40.847704Z","iopub.execute_input":"2022-03-27T10:56:40.848207Z","iopub.status.idle":"2022-03-27T10:56:40.958938Z","shell.execute_reply.started":"2022-03-27T10:56:40.84816Z","shell.execute_reply":"2022-03-27T10:56:40.95825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"opt = Adam(learning_rate=0.001)\nmodel.compile(loss=\"sparse_categorical_crossentropy\", optimizer=opt, metrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:56:46.510904Z","iopub.execute_input":"2022-03-27T10:56:46.511151Z","iopub.status.idle":"2022-03-27T10:56:46.529362Z","shell.execute_reply.started":"2022-03-27T10:56:46.511124Z","shell.execute_reply":"2022-03-27T10:56:46.528652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"H = model.fit(\n    train_generator,\n    steps_per_epoch=train_generator.samples // train_generator.batch_size,\n#     callbacks=[early_stopping],\n    validation_data=valid_generator,\n    validation_steps=valid_generator.samples // valid_generator.batch_size,\n    epochs=4)\n","metadata":{"execution":{"iopub.status.busy":"2022-03-27T10:58:17.730681Z","iopub.execute_input":"2022-03-27T10:58:17.73094Z","iopub.status.idle":"2022-03-27T10:58:35.52371Z","shell.execute_reply.started":"2022-03-27T10:58:17.730913Z","shell.execute_reply":"2022-03-27T10:58:35.522481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}