{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Simple convolutional neural network \n\nData by @RDizzl3 https://www.kaggle.com/c/happy-whale-and-dolphin/discussion/304686\n\nWIP ","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport cv2\nimport glob\nimport matplotlib.pyplot as plt\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-08T09:14:07.580450Z","iopub.execute_input":"2022-02-08T09:14:07.581292Z","iopub.status.idle":"2022-02-08T09:14:07.849343Z","shell.execute_reply.started":"2022-02-08T09:14:07.581245Z","shell.execute_reply":"2022-02-08T09:14:07.848570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Dense,Conv2D,MaxPooling2D,Flatten,Reshape,Dropout\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import EarlyStopping","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:07.851014Z","iopub.execute_input":"2022-02-08T09:14:07.851348Z","iopub.status.idle":"2022-02-08T09:14:12.974496Z","shell.execute_reply.started":"2022-02-08T09:14:07.851310Z","shell.execute_reply":"2022-02-08T09:14:12.973731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndf = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:12.975978Z","iopub.execute_input":"2022-02-08T09:14:12.976266Z","iopub.status.idle":"2022-02-08T09:14:13.066291Z","shell.execute_reply.started":"2022-02-08T09:14:12.976226Z","shell.execute_reply":"2022-02-08T09:14:13.065573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:13.069247Z","iopub.execute_input":"2022-02-08T09:14:13.069666Z","iopub.status.idle":"2022-02-08T09:14:13.087575Z","shell.execute_reply.started":"2022-02-08T09:14:13.069623Z","shell.execute_reply":"2022-02-08T09:14:13.086861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:13.089728Z","iopub.execute_input":"2022-02-08T09:14:13.090452Z","iopub.status.idle":"2022-02-08T09:14:13.123165Z","shell.execute_reply.started":"2022-02-08T09:14:13.090415Z","shell.execute_reply":"2022-02-08T09:14:13.122415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the train image file names\n# Use the * as a wild card this will tell glob get us all images in this directory\nimage_path = '/kaggle/input/jpeg-happywhale-128x128/train_images-128-128/train_images-128-128/*'\ntrain_filenames = glob.glob(image_path)","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:13.124528Z","iopub.execute_input":"2022-02-08T09:14:13.124822Z","iopub.status.idle":"2022-02-08T09:14:14.363635Z","shell.execute_reply.started":"2022-02-08T09:14:13.124786Z","shell.execute_reply":"2022-02-08T09:14:14.362916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get a single image path\nimage_path = train_filenames[0]\nimg = cv2.imread(image_path)\n\n# Check the shape - should be (128, 128, 3)\nimg.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:14.365012Z","iopub.execute_input":"2022-02-08T09:14:14.365268Z","iopub.status.idle":"2022-02-08T09:14:14.389459Z","shell.execute_reply.started":"2022-02-08T09:14:14.365234Z","shell.execute_reply":"2022-02-08T09:14:14.388783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the image\nplt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:14.390785Z","iopub.execute_input":"2022-02-08T09:14:14.391263Z","iopub.status.idle":"2022-02-08T09:14:14.634417Z","shell.execute_reply.started":"2022-02-08T09:14:14.391227Z","shell.execute_reply":"2022-02-08T09:14:14.633735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:14.635319Z","iopub.execute_input":"2022-02-08T09:14:14.635563Z","iopub.status.idle":"2022-02-08T09:14:14.644818Z","shell.execute_reply.started":"2022-02-08T09:14:14.635523Z","shell.execute_reply":"2022-02-08T09:14:14.644119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_df = df[['image','individual_id']]","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:14.646314Z","iopub.execute_input":"2022-02-08T09:14:14.646757Z","iopub.status.idle":"2022-02-08T09:14:14.655530Z","shell.execute_reply.started":"2022-02-08T09:14:14.646720Z","shell.execute_reply":"2022-02-08T09:14:14.654836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:14.656940Z","iopub.execute_input":"2022-02-08T09:14:14.657220Z","iopub.status.idle":"2022-02-08T09:14:14.666862Z","shell.execute_reply.started":"2022-02-08T09:14:14.657188Z","shell.execute_reply":"2022-02-08T09:14:14.666121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks = [\n            EarlyStopping(patience = 10,monitor=\"accuracy\",mode=\"max\"),    \n            ]","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:14.668218Z","iopub.execute_input":"2022-02-08T09:14:14.668523Z","iopub.status.idle":"2022-02-08T09:14:14.675152Z","shell.execute_reply.started":"2022-02-08T09:14:14.668489Z","shell.execute_reply":"2022-02-08T09:14:14.674320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_augmentation = keras.Sequential(\n    [\n        keras.layers.experimental.preprocessing.RandomFlip(\"horizontal\"),\n        keras.layers.experimental.preprocessing.RandomRotation(0.1),\n    ]\n)","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:14.676510Z","iopub.execute_input":"2022-02-08T09:14:14.676796Z","iopub.status.idle":"2022-02-08T09:14:17.080955Z","shell.execute_reply.started":"2022-02-08T09:14:14.676758Z","shell.execute_reply":"2022-02-08T09:14:17.079124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale=1./255, \n                                   shear_range=0.2,\n                                   zoom_range=0.2, \n                                   horizontal_flip=True,\n                                  width_shift_range=0.1,\n                                  height_shift_range=0.1\n                                   )","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:17.084493Z","iopub.execute_input":"2022-02-08T09:14:17.084935Z","iopub.status.idle":"2022-02-08T09:14:17.088779Z","shell.execute_reply.started":"2022-02-08T09:14:17.084901Z","shell.execute_reply":"2022-02-08T09:14:17.088100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_datagen = ImageDataGenerator(rescale=1./255,validation_split=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:17.090992Z","iopub.execute_input":"2022-02-08T09:14:17.091404Z","iopub.status.idle":"2022-02-08T09:14:17.102012Z","shell.execute_reply.started":"2022-02-08T09:14:17.091367Z","shell.execute_reply":"2022-02-08T09:14:17.101409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = '/kaggle/input/jpeg-happywhale-128x128/train_images-128-128/train_images-128-128/'","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:17.104669Z","iopub.execute_input":"2022-02-08T09:14:17.105406Z","iopub.status.idle":"2022-02-08T09:14:17.110577Z","shell.execute_reply.started":"2022-02-08T09:14:17.105377Z","shell.execute_reply":"2022-02-08T09:14:17.109851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:17.111507Z","iopub.execute_input":"2022-02-08T09:14:17.111891Z","iopub.status.idle":"2022-02-08T09:14:17.126670Z","shell.execute_reply.started":"2022-02-08T09:14:17.111838Z","shell.execute_reply":"2022-02-08T09:14:17.125934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:17.127841Z","iopub.execute_input":"2022-02-08T09:14:17.128527Z","iopub.status.idle":"2022-02-08T09:14:17.138118Z","shell.execute_reply.started":"2022-02-08T09:14:17.128375Z","shell.execute_reply":"2022-02-08T09:14:17.137389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:17.139334Z","iopub.execute_input":"2022-02-08T09:14:17.140084Z","iopub.status.idle":"2022-02-08T09:14:17.144833Z","shell.execute_reply.started":"2022-02-08T09:14:17.140036Z","shell.execute_reply":"2022-02-08T09:14:17.144102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"individuals = list(df['individual_id'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:17.145810Z","iopub.execute_input":"2022-02-08T09:14:17.146415Z","iopub.status.idle":"2022-02-08T09:14:17.159764Z","shell.execute_reply.started":"2022-02-08T09:14:17.146379Z","shell.execute_reply":"2022-02-08T09:14:17.158967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_set = train_datagen.flow_from_dataframe(\n                                                train_df,\n                                                train_dir,\n                                                seed=101,                                                 \n                                                target_size=(64, 64),\n                                                labels = individuals,\n                                                batch_size=32,\n                                                x_col='image',\n                                                y_col='individual_id',\n                                                class_mode='categorical',\n                                                validation_split=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:14:17.161793Z","iopub.execute_input":"2022-02-08T09:14:17.162444Z","iopub.status.idle":"2022-02-08T09:15:25.759161Z","shell.execute_reply.started":"2022-02-08T09:14:17.162409Z","shell.execute_reply":"2022-02-08T09:15:25.758418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_set = validation_datagen.flow_from_dataframe(\n                                                train_df,\n                                                train_dir,\n                                                seed=101,                                                 \n                                                target_size=(64, 64),\n                                                labels = individuals,\n                                                batch_size=32,\n                                                x_col='image',\n                                                y_col='individual_id',\n                                                class_mode='categorical',\n                                                subset = \"validation\"\n                                                )","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:15:25.760263Z","iopub.execute_input":"2022-02-08T09:15:25.761004Z","iopub.status.idle":"2022-02-08T09:16:07.144732Z","shell.execute_reply.started":"2022-02-08T09:15:25.760963Z","shell.execute_reply":"2022-02-08T09:16:07.143977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = keras.applications.Xception(\n    weights='imagenet',  # Load weights pre-trained on ImageNet.\n    input_shape=(150, 150, 3),\n    include_top=False)  # Do not include the ImageNet classifier at the top.","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:16:07.146039Z","iopub.execute_input":"2022-02-08T09:16:07.146479Z","iopub.status.idle":"2022-02-08T09:16:08.911366Z","shell.execute_reply.started":"2022-02-08T09:16:07.146441Z","shell.execute_reply":"2022-02-08T09:16:08.910594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model.trainable = False","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:16:08.912631Z","iopub.execute_input":"2022-02-08T09:16:08.912904Z","iopub.status.idle":"2022-02-08T09:16:08.921941Z","shell.execute_reply.started":"2022-02-08T09:16:08.912865Z","shell.execute_reply":"2022-02-08T09:16:08.920955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs = keras.Input(shape=(150, 150, 3))\n# We make sure that the base_model is running in inference mode here,\n# by passing `training=False`. This is important for fine-tuning, as you will\n# learn in a few paragraphs.\nx = base_model(inputs, training=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:16:08.923517Z","iopub.execute_input":"2022-02-08T09:16:08.924086Z","iopub.status.idle":"2022-02-08T09:16:09.219986Z","shell.execute_reply.started":"2022-02-08T09:16:08.924034Z","shell.execute_reply":"2022-02-08T09:16:09.219305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = data_augmentation(inputs)  # Apply random data augmentation\nx = tf.keras.applications.xception.preprocess_input(x)","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:16:09.221375Z","iopub.execute_input":"2022-02-08T09:16:09.221652Z","iopub.status.idle":"2022-02-08T09:16:09.361185Z","shell.execute_reply.started":"2022-02-08T09:16:09.221615Z","shell.execute_reply":"2022-02-08T09:16:09.360427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The base model contains batchnorm layers. We want to keep them in inference mode\n# when we unfreeze the base model for fine-tuning, so we make sure that the\n# base_model is running in inference mode here.\nx = base_model(x, training=False)\nx = keras.layers.GlobalAveragePooling2D()(x)\nx = keras.layers.Dropout(0.2)(x)  # Regularize with dropout\noutputs = keras.layers.Dense(15587)(x)\nmodel = keras.Model(inputs, outputs)\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:16:09.362396Z","iopub.execute_input":"2022-02-08T09:16:09.363132Z","iopub.status.idle":"2022-02-08T09:16:09.669322Z","shell.execute_reply.started":"2022-02-08T09:16:09.363089Z","shell.execute_reply":"2022-02-08T09:16:09.668623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=keras.optimizers.Adam(),\n              loss=keras.losses.CategoricalCrossentropy(from_logits=True),\n              metrics=[keras.metrics.Accuracy()])\nmodel.fit(training_set, epochs=10, validation_data=validation_set)","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:16:09.670577Z","iopub.execute_input":"2022-02-08T09:16:09.670815Z","iopub.status.idle":"2022-02-08T09:45:27.064205Z","shell.execute_reply.started":"2022-02-08T09:16:09.670781Z","shell.execute_reply":"2022-02-08T09:45:27.063522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Unfreeze the base model\nbase_model.trainable = True\n\n# It's important to recompile your model after you make any changes\n# to the `trainable` attribute of any inner layer, so that your changes\n# are take into account\nmodel.compile(optimizer=keras.optimizers.Adam(1e-5),  # Very low learning rate\n              loss=keras.losses.CategoricalCrossentropy(from_logits=True),\n              metrics=[keras.metrics.Accuracy()])\n\n# Train end-to-end. Be careful to stop before you overfit!\nhistory = model.fit(training_set, epochs=10,validation_data=validation_set)","metadata":{"execution":{"iopub.status.busy":"2022-02-08T09:45:27.065398Z","iopub.execute_input":"2022-02-08T09:45:27.065963Z","iopub.status.idle":"2022-02-08T10:14:12.485048Z","shell.execute_reply.started":"2022-02-08T09:45:27.065924Z","shell.execute_reply":"2022-02-08T10:14:12.484357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']","metadata":{"execution":{"iopub.status.busy":"2022-02-08T10:14:12.486657Z","iopub.execute_input":"2022-02-08T10:14:12.486912Z","iopub.status.idle":"2022-02-08T10:14:12.491214Z","shell.execute_reply.started":"2022-02-08T10:14:12.486878Z","shell.execute_reply":"2022-02-08T10:14:12.490284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nlosses = pd.DataFrame(history.history)","metadata":{"execution":{"iopub.status.busy":"2022-02-08T10:14:12.492539Z","iopub.execute_input":"2022-02-08T10:14:12.492933Z","iopub.status.idle":"2022-02-08T10:14:12.503756Z","shell.execute_reply.started":"2022-02-08T10:14:12.492898Z","shell.execute_reply":"2022-02-08T10:14:12.503093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize=(8,6))\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nlosses['loss'].plot(label='Training Loss')\nlosses['val_loss'].plot(label='Validation Loss')\nplt.legend(loc='center')","metadata":{"execution":{"iopub.status.busy":"2022-02-08T10:14:12.504899Z","iopub.execute_input":"2022-02-08T10:14:12.505252Z","iopub.status.idle":"2022-02-08T10:14:12.736550Z","shell.execute_reply.started":"2022-02-08T10:14:12.505217Z","shell.execute_reply":"2022-02-08T10:14:12.735904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss, accuracy = model.evaluate(validation_set)\nprint('Test accuracy :', accuracy)","metadata":{"execution":{"iopub.status.busy":"2022-02-08T10:14:12.737784Z","iopub.execute_input":"2022-02-08T10:14:12.738024Z","iopub.status.idle":"2022-02-08T10:14:33.286034Z","shell.execute_reply.started":"2022-02-08T10:14:12.737991Z","shell.execute_reply":"2022-02-08T10:14:33.285144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It doesn't seem to work!","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}