{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Implementation 1\n## ResNet50","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nfrom sklearn.metrics import confusion_matrix, classification_report\nfrom sklearn.model_selection import train_test_split\nfrom pathlib import Path\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport cv2\nfrom PIL import Image\nfrom keras import backend as K\nfrom keras.models import load_model\n# from keras.preprocessing.image import img_to_array\nfrom tensorflow.keras.optimizers import Adam, RMSprop\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","scrolled":true,"execution":{"iopub.status.busy":"2023-06-06T07:04:11.989507Z","iopub.execute_input":"2023-06-06T07:04:11.989806Z","iopub.status.idle":"2023-06-06T07:04:23.225363Z","shell.execute_reply.started":"2023-06-06T07:04:11.989775Z","shell.execute_reply":"2023-06-06T07:04:23.224234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# parse input\ntrain_csv = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/train.csv')\ntrain_csv.head()\n\ntrain = '/kaggle/input/happy-whale-and-dolphin/train_images'\ntrain_csv['image']  = train_csv['image'].apply(lambda x : train + '/'+ x)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:06:46.823289Z","iopub.execute_input":"2023-06-06T07:06:46.824017Z","iopub.status.idle":"2023-06-06T07:06:46.918817Z","shell.execute_reply.started":"2023-06-06T07:06:46.823976Z","shell.execute_reply":"2023-06-06T07:06:46.917702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get paths to images, drop classes with less than 1000 images to train on\n# train_csv['species'].value_counts()\n# cat = train_csv['species'].value_counts().loc[lambda x : x < 1000].index.tolist()\n# for i in cat:\n#     drop = train_csv.loc[train_csv['species'] == i, 'species'].index.values.tolist()\n#     train_csv = train_csv.drop(drop, axis = 0)\n# train_csv.reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:06:47.642551Z","iopub.execute_input":"2023-06-06T07:06:47.642937Z","iopub.status.idle":"2023-06-06T07:06:47.650961Z","shell.execute_reply.started":"2023-06-06T07:06:47.642904Z","shell.execute_reply":"2023-06-06T07:06:47.649820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # get 000 samples from each class\n# samples = []\n# for i in train_csv['species'].unique():\n#     x = train_csv.query('species == @i')\n#     samples.append(x.sample(1000, random_state = 1))\n# train_csv = pd.concat(samples, axis = 0).sample(frac = 1.0, random_state = 1).reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:06:48.402424Z","iopub.execute_input":"2023-06-06T07:06:48.403135Z","iopub.status.idle":"2023-06-06T07:06:48.408140Z","shell.execute_reply.started":"2023-06-06T07:06:48.403097Z","shell.execute_reply":"2023-06-06T07:06:48.406790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv['species'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:06:49.245019Z","iopub.execute_input":"2023-06-06T07:06:49.246774Z","iopub.status.idle":"2023-06-06T07:06:49.262634Z","shell.execute_reply.started":"2023-06-06T07:06:49.246723Z","shell.execute_reply":"2023-06-06T07:06:49.261235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df , test_df = train_test_split(train_csv, test_size = 0.30, shuffle = True, random_state = 1)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:06:55.190553Z","iopub.execute_input":"2023-06-06T07:06:55.190923Z","iopub.status.idle":"2023-06-06T07:06:55.216140Z","shell.execute_reply.started":"2023-06-06T07:06:55.190890Z","shell.execute_reply":"2023-06-06T07:06:55.214892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training, validation, test sets\n\ntrain_gen = tf.keras.preprocessing.image.ImageDataGenerator(preprocessing_function = tf.keras.applications.mobilenet_v2.preprocess_input,validation_split = 0.2)\ntest_gen = tf.keras.preprocessing.image.ImageDataGenerator(preprocessing_function = tf.keras.applications.mobilenet_v2.preprocess_input)\n\ntrain_image = train_gen.flow_from_dataframe(\n    dataframe = train_df,\n    x_col = 'image',\n    y_col = 'species',\n    target_size = (150, 150),\n    color_mode ='rgb',\n    class_mode = 'categorical',\n    batch_size = 32,\n    shuffle = True,\n    seed = 42,\n    subset = 'training'\n)\nval_image = train_gen.flow_from_dataframe(\n    dataframe = train_df,\n    x_col = 'image',\n    y_col = 'species',\n    target_size = (150, 150),\n    color_mode ='rgb',\n    class_mode = 'categorical',\n    batch_size = 32,\n    shuffle = True,\n    seed = 42,\n    subset = 'validation'\n)\ntest_image = test_gen.flow_from_dataframe(\n    dataframe = test_df,\n    x_col = 'image',\n    y_col = 'species',\n    target_size = (150, 150),\n    color_mode ='rgb',\n    class_mode = 'categorical',\n    batch_size = 32,\n    shuffle = False\n)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:07:11.060491Z","iopub.execute_input":"2023-06-06T07:07:11.061510Z","iopub.status.idle":"2023-06-06T07:09:14.502941Z","shell.execute_reply.started":"2023-06-06T07:07:11.061426Z","shell.execute_reply":"2023-06-06T07:09:14.501775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get ResNet50 model\n\nfrom tensorflow.keras.optimizers import SGD\nResNet50_model = tf.keras.applications.ResNet50(weights='imagenet', include_top=False, input_shape=(150,150,3), classes=13)\n\nfor layers in ResNet50_model.layers:\n    layers.trainable=True\n\nopt = SGD(lr=0.01,momentum=0.7)\nresnet50_x = tf.keras.layers.Flatten()(ResNet50_model.output)\nresnet50_x = tf.keras.layers.Dense(256,activation='relu')(resnet50_x)\nresnet50_x = tf.keras.layers.Dense(30,activation='softmax')(resnet50_x)\nresnet50_x_final_model = tf.keras.Model(inputs=ResNet50_model.input, outputs=resnet50_x)\nresnet50_x_final_model.compile(loss = 'categorical_crossentropy', optimizer= opt, metrics=['acc'])\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:09:57.782581Z","iopub.execute_input":"2023-06-06T07:09:57.783213Z","iopub.status.idle":"2023-06-06T07:10:04.311843Z","shell.execute_reply.started":"2023-06-06T07:09:57.783174Z","shell.execute_reply":"2023-06-06T07:10:04.310765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train resnet50 model\n\n# print(train_image.class_indices)\n\nresnet_classifier = resnet50_x_final_model.fit(train_image,\n                            steps_per_epoch=(train_image.samples//32),\n                            epochs = 5,\n                            validation_data=val_image,\n                            validation_steps=(val_image.samples//32),\n                            batch_size = 32,\n                            verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:10:17.727336Z","iopub.execute_input":"2023-06-06T07:10:17.728061Z","iopub.status.idle":"2023-06-06T08:54:35.245015Z","shell.execute_reply.started":"2023-06-06T07:10:17.728019Z","shell.execute_reply":"2023-06-06T08:54:35.230477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the entire model as a SavedModel.\n!mkdir -p saved_model\nresnet50_x_final_model.save('saved_model/my_model')","metadata":{"execution":{"iopub.status.busy":"2023-06-06T08:54:40.544319Z","iopub.execute_input":"2023-06-06T08:54:40.545041Z","iopub.status.idle":"2023-06-06T08:55:11.868068Z","shell.execute_reply.started":"2023-06-06T08:54:40.545000Z","shell.execute_reply":"2023-06-06T08:55:11.866832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Click here to <a href=\"/kaggle/working/saved_model\"> download the model </a> - not working for some reason","metadata":{}},{"cell_type":"code","source":"!cd /kaggle/working\n!ls\n!tar -czvf my_work.zip saved_model","metadata":{"execution":{"iopub.status.busy":"2023-06-06T09:27:27.468974Z","iopub.execute_input":"2023-06-06T09:27:27.469390Z","iopub.status.idle":"2023-06-06T09:27:30.727129Z","shell.execute_reply.started":"2023-06-06T09:27:27.469353Z","shell.execute_reply":"2023-06-06T09:27:30.725623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = resnet50_x_final_model.evaluate(test_image, verbose = 1)\npred = np.argmax(resnet50_x_final_model.predict(test_image), axis = 1)\nclass_names = list(test_image.class_indices.keys())\ncm = confusion_matrix(test_image.labels, pred, labels = np.arange(30))\nclr = classification_report(test_image.labels, pred, labels = np.arange(30),target_names = class_names)\nprint(f'\\nTest Accuracy : {round(results[1], 4)*100}%\\n')\nplt.figure(figsize = (10,10))\nsns.heatmap(cm, annot = True, fmt = 'g', vmin = 0, cbar = False)\nplt.xticks(ticks = np.arange(30) + 0.5, labels = class_names, rotation = 90)\nplt.yticks(ticks = np.arange(30) + 0.5, labels = class_names, rotation = 0)\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()\nprint(f'classification Report------------>\\n{clr}')","metadata":{"execution":{"iopub.status.busy":"2023-06-06T08:59:07.150645Z","iopub.execute_input":"2023-06-06T08:59:07.151260Z","iopub.status.idle":"2023-06-06T09:26:08.479495Z","shell.execute_reply.started":"2023-06-06T08:59:07.151221Z","shell.execute_reply":"2023-06-06T09:26:08.476155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Implementation 2\n## Inception V3\n","metadata":{}},{"cell_type":"code","source":"# parse input\ntrain_csv = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/train.csv')\ntrain_csv.head()\n\ntrain = '/kaggle/input/happy-whale-and-dolphin/train_images'\ntrain_csv['image']  = train_csv['image'].apply(lambda x : train + '/'+ x)\ntrain_csv['species'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:04:23.721576Z","iopub.status.idle":"2023-06-06T07:04:23.722064Z","shell.execute_reply.started":"2023-06-06T07:04:23.721811Z","shell.execute_reply":"2023-06-06T07:04:23.721836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training, validation, test sets\ntrain_df , test_df = train_test_split(train_csv, test_size = 0.30, shuffle = True, random_state = 1)\ntrain_df\n\ntrain_gen = tf.keras.preprocessing.image.ImageDataGenerator(preprocessing_function = tf.keras.applications.mobilenet_v2.preprocess_input,validation_split = 0.2)\ntest_gen = tf.keras.preprocessing.image.ImageDataGenerator(preprocessing_function = tf.keras.applications.mobilenet_v2.preprocess_input)\n\n# image size 150x150 instead of 299x299\n\ntrain_image = train_gen.flow_from_dataframe(\n    dataframe = train_df,\n    x_col = 'image',\n    y_col = 'species',\n    target_size = (150, 150),\n    color_mode ='rgb',\n    class_mode = 'categorical',\n    batch_size = 32,\n    shuffle = True,\n    seed = 42,\n    subset = 'training'\n)\nval_image = train_gen.flow_from_dataframe(\n    dataframe = train_df,\n    x_col = 'image',\n    y_col = 'species',\n    target_size = (150, 150),\n    color_mode ='rgb',\n    class_mode = 'categorical',\n    batch_size = 32,\n    shuffle = True,\n    seed = 42,\n    subset = 'validation'\n)\ntest_image = test_gen.flow_from_dataframe(\n    dataframe = test_df,\n    x_col = 'image',\n    y_col = 'species',\n    target_size = (150, 150),\n    color_mode ='rgb',\n    class_mode = 'categorical',\n    batch_size = 32,\n    shuffle = False\n)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:04:23.723956Z","iopub.status.idle":"2023-06-06T07:04:23.724515Z","shell.execute_reply.started":"2023-06-06T07:04:23.724217Z","shell.execute_reply":"2023-06-06T07:04:23.724246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import the inception model  \nfrom tensorflow.keras.applications.inception_v3 import InceptionV3\n\nbase_model = InceptionV3(\n    include_top = False,\n    weights = \"imagenet\",\n    input_shape = None)\nx = base_model.output\nx = tf.keras.layers.GlobalAveragePooling2D()(x)\nx = tf.keras.layers.Dense(512, activation='relu')(x)\npredictions = tf.keras.layers.Dense(30, activation='softmax')(x)\n\nmodel = tf.keras.Model(inputs = base_model.input, outputs = predictions)\n\nfor layer in model.layers[:249]:\n    layer.trainable = False\n\n    if layer.name.startswith('batch_normalization'):\n        layer.trainable = True\n\nfor layer in model.layers[249:]:\n    layer.trainable = True\n\n# compile\nmodel.compile(\n    optimizer = Adam(),\n    loss = 'categorical_crossentropy',\n    metrics = [\"accuracy\"]\n)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:04:23.726355Z","iopub.status.idle":"2023-06-06T07:04:23.726896Z","shell.execute_reply.started":"2023-06-06T07:04:23.726623Z","shell.execute_reply":"2023-06-06T07:04:23.726650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ninception_v3_classifier = model.fit(train_image,\n                            steps_per_epoch=(train_image.samples//32),\n                            epochs = 12,\n                            validation_data=val_image,\n                            validation_steps=(val_image.samples//32),\n                            batch_size = 32,\n                            verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:04:23.728496Z","iopub.status.idle":"2023-06-06T07:04:23.729405Z","shell.execute_reply.started":"2023-06-06T07:04:23.729140Z","shell.execute_reply":"2023-06-06T07:04:23.729167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = model.evaluate(test_image, verbose = 1)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:04:23.732152Z","iopub.status.idle":"2023-06-06T07:04:23.732679Z","shell.execute_reply.started":"2023-06-06T07:04:23.732393Z","shell.execute_reply":"2023-06-06T07:04:23.732419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = np.argmax(model.predict(test_image), axis = 1)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:04:23.734493Z","iopub.status.idle":"2023-06-06T07:04:23.734983Z","shell.execute_reply.started":"2023-06-06T07:04:23.734726Z","shell.execute_reply":"2023-06-06T07:04:23.734752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = list(test_image.class_indices.keys())\ncm = confusion_matrix(test_image.labels, pred, labels = np.arange(30))\nclr = classification_report(test_image.labels, pred, labels = np.arange(30),target_names = class_names)\nprint(f'\\nTest Accuracy : {round(results[1], 4)*100}%\\n')\nplt.figure(figsize = (10,10))\nsns.heatmap(cm, annot = True, fmt = 'g', vmin = 0, cbar = False)\nplt.xticks(ticks = np.arange(13) + 0.5, labels = class_names, rotation = 90)\nplt.yticks(ticks = np.arange(13) + 0.5, labels = class_names, rotation = 0)\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()\nprint(f'classification Report------------>\\n{clr}')","metadata":{"execution":{"iopub.status.busy":"2023-06-06T07:04:23.737284Z","iopub.status.idle":"2023-06-06T07:04:23.738429Z","shell.execute_reply.started":"2023-06-06T07:04:23.738158Z","shell.execute_reply":"2023-06-06T07:04:23.738186Z"},"trusted":true},"execution_count":null,"outputs":[]}]}