{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Nelson VICEL--FARAH\n### Antoine ZELLMEYER","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2022-07-10T07:25:51.623553Z","iopub.execute_input":"2022-07-10T07:25:51.624323Z","iopub.status.idle":"2022-07-10T07:25:51.660526Z","shell.execute_reply.started":"2022-07-10T07:25:51.624224Z","shell.execute_reply":"2022-07-10T07:25:51.659301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\nSEED = 123\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:25:52.098623Z","iopub.execute_input":"2022-07-10T07:25:52.099224Z","iopub.status.idle":"2022-07-10T07:25:57.051983Z","shell.execute_reply.started":"2022-07-10T07:25:52.099190Z","shell.execute_reply":"2022-07-10T07:25:57.050521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar xzf /kaggle/input/navire-2022-libre/ships.tgz","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:25:57.054898Z","iopub.execute_input":"2022-07-10T07:25:57.055736Z","iopub.status.idle":"2022-07-10T07:26:00.697974Z","shell.execute_reply.started":"2022-07-10T07:25:57.055685Z","shell.execute_reply":"2022-07-10T07:26:00.696295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/working","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:26:00.700548Z","iopub.execute_input":"2022-07-10T07:26:00.701513Z","iopub.status.idle":"2022-07-10T07:26:01.523190Z","shell.execute_reply.started":"2022-07-10T07:26:00.701462Z","shell.execute_reply":"2022-07-10T07:26:01.521787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Chargement du dataset","metadata":{}},{"cell_type":"code","source":"ds_train = tf.keras.preprocessing.image_dataset_from_directory(\n    \"/kaggle/working/ships32\",\n    validation_split=0.1,\n    labels='inferred',\n    subset=\"training\",\n    image_size=(32,32),\n    seed = SEED,\n    batch_size = 32,\n    color_mode=\"rgb\"\n)\n\nds_val = tf.keras.preprocessing.image_dataset_from_directory(\n    \"/kaggle/working/ships32\",\n    validation_split=0.1,\n    labels='inferred',\n    subset=\"validation\",\n    image_size=(32,32),\n    seed = SEED,\n    batch_size = 32,\n    color_mode=\"rgb\"\n)\n\n# On normalise les pixels entre 0 et 1 \nds_train = ds_train.map(lambda x, y: (x/255.0, y))\nds_val = ds_val.map(lambda x, y: (x/255.0, y))","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:31:42.541511Z","iopub.execute_input":"2022-07-10T07:31:42.542463Z","iopub.status.idle":"2022-07-10T07:31:45.806317Z","shell.execute_reply.started":"2022-07-10T07:31:42.542424Z","shell.execute_reply":"2022-07-10T07:31:45.804503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Réseau de neurones\n\nResNet50 est un réseau qui a d'excellentes performences sur le dataset CIFAR10 qui ressemble plus ou moins à notre configuration vu qu'il consiste à classer des images de 32x32 pixels. Nous décidons donc de le réutiliser et de l'adapter à notre cas en prenant en considération les 13 classes de bateaux.\n\nDe plus, avec la regularisation l2, nous nous aussurons une répartition équilibrée des poids entre ResNet et la couche de sortie.\n\nEtant donné qu'on utilise ResNet avec les poids de ImageNet qui est un dataset d'images de taille 224 x 224 pixels, nous rapportons les images de notre datatset de bateaux à cette taille via une couche d'UpScaling.","metadata":{}},{"cell_type":"code","source":"from keras.models import Model\nfrom keras.layers import *\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.applications import ResNet50\n\noriginal_dim = (32, 32, 3)\n\ninput = Input(original_dim)\ninput_tensor = tf.keras.layers.UpSampling2D(size=(7,7))(input)\n\nrn50_model = ResNet50(include_top = False,\n                          weights = \"imagenet\",\n                          input_tensor = input_tensor,\n                          pooling = 'max')\n\nx = rn50_model.output\nx = Dense(13, activation='softmax', kernel_regularizer=\"l2\")(x)\n\nmodel = Model(rn50_model.input, outputs = x)\n\n\n# Pas d'entrainement sur les poids de ResNet pour l'instant\nfor layer in rn50_model.layers:\n    layer.trainable=False\n\n\nmodel.compile(loss='sparse_categorical_crossentropy', optimizer=Adam(learning_rate=0.001), metrics=['accuracy'])\nmodel.build(input_shape=(32,32,3))\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T08:27:56.644081Z","iopub.execute_input":"2022-07-10T08:27:56.644458Z","iopub.status.idle":"2022-07-10T08:27:58.248542Z","shell.execute_reply.started":"2022-07-10T08:27:56.644426Z","shell.execute_reply":"2022-07-10T08:27:58.247398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Préparation de l'entraînement\n\nOn rajoute :\n- une fonction de checkpoint qui sauvegarde les poids du modèles au moment où il performe le mieux.\n- une fonction qui réduit le learning rate quand la convergence ralentit\n- on prévoit l'entraînement de poids de façon à contrebalancer le déséquilibre des données","metadata":{}},{"cell_type":"code","source":"from sklearn.utils import class_weight\n\nmodel_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n    filepath=\"/kaggle/working/checkpoint\",\n    save_weights_only=True,\n    monitor='val_accuracy',\n    mode='max',\n    save_best_only=True,\n    verbose=1)\n\nrlronp=tf.keras.callbacks.ReduceLROnPlateau(monitor=\"val_loss\",\n                                            factor=0.5,\n                                            patience=4,\n                                            verbose=1)\n\nsample = np.array([y.ref() for x,y in ds_train])\nclass_weights = dict(enumerate(class_weight.compute_class_weight(class_weight = \"balanced\", classes = np.unique(sample), y = sample)))","metadata":{"execution":{"iopub.status.busy":"2022-07-10T08:28:11.878545Z","iopub.execute_input":"2022-07-10T08:28:11.879273Z","iopub.status.idle":"2022-07-10T08:29:01.305410Z","shell.execute_reply.started":"2022-07-10T08:28:11.879235Z","shell.execute_reply":"2022-07-10T08:29:01.304176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pré-entraînement sur quelques époques\n\nDans cette partie nous n'entraînons que les poids qui ne sont pas déjà pré-entraînés. Nous entraînerons l'ensemble du modèle dans la partie suivante.","metadata":{}},{"cell_type":"code","source":"pre_epochs = 10\n\nhistory = model.fit(ds_train, validation_data=ds_val, epochs=pre_epochs, callbacks=[model_checkpoint_callback, rlronp], class_weight=class_weights)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T08:29:01.308238Z","iopub.execute_input":"2022-07-10T08:29:01.308665Z","iopub.status.idle":"2022-07-10T08:45:45.975762Z","shell.execute_reply.started":"2022-07-10T08:29:01.308627Z","shell.execute_reply":"2022-07-10T08:45:45.974645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Entraînement (ResNet50 tuning)","metadata":{}},{"cell_type":"code","source":"# On reprend l'entraînement au meilleur moment\nmodel.load_weights(\"/kaggle/working/checkpoint\")\n\nfor layer in rn50_model.layers:\n    layer.trainable=True\n    \nmodel.compile(optimizer=Adam(learning_rate=0.0001), \n              loss='sparse_categorical_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-07-10T08:45:45.978922Z","iopub.execute_input":"2022-07-10T08:45:45.979797Z","iopub.status.idle":"2022-07-10T08:45:46.657724Z","shell.execute_reply.started":"2022-07-10T08:45:45.979756Z","shell.execute_reply":"2022-07-10T08:45:46.656775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_history = model.fit(ds_train,\n                        validation_data=ds_val,\n                        epochs=15,\n                        callbacks=[model_checkpoint_callback, rlronp],\n                        class_weight=class_weights)\n\nhistory.history = {k: history.history.get(k, 0) + new_history.history.get(k, 0) for k in set(history.history) | set(new_history.history)}","metadata":{"execution":{"iopub.status.busy":"2022-07-10T08:45:46.660161Z","iopub.execute_input":"2022-07-10T08:45:46.660850Z","iopub.status.idle":"2022-07-10T09:49:02.839461Z","shell.execute_reply.started":"2022-07-10T08:45:46.660802Z","shell.execute_reply":"2022-07-10T09:49:02.838106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Trace de l'entraînement","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfig, ax = plt.subplots(1,2, figsize=(12,6))\n\nax[0].plot(history.history[\"loss\"], label=\"train loss\")\nax[0].plot(history.history[\"val_loss\"], label=\"val loss\")\nax[0].axvline(x=pre_epochs, c=\"red\", linestyle=\"--\", label=\"resnet50 tuning\")\nax[0].legend()\nax[0].set_title(\"loss\")\n\nax[1].plot(history.history[\"accuracy\"], label=\"train accuracy\")\nax[1].plot(history.history[\"val_accuracy\"], label=\"val accuracy\")\nax[1].axvline(x=pre_epochs, c=\"red\", linestyle=\"--\", label=\"resnet50 tuning\")\nax[1].legend()\nax[1].set_title(\"accuracy\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T09:49:02.842186Z","iopub.execute_input":"2022-07-10T09:49:02.842682Z","iopub.status.idle":"2022-07-10T09:49:03.302798Z","shell.execute_reply.started":"2022-07-10T09:49:02.842632Z","shell.execute_reply":"2022-07-10T09:49:03.301631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# On conserve les meilleurs poids","metadata":{}},{"cell_type":"code","source":"model.load_weights(\"/kaggle/working/checkpoint\")","metadata":{"execution":{"iopub.status.busy":"2022-07-10T09:49:03.304739Z","iopub.execute_input":"2022-07-10T09:49:03.305801Z","iopub.status.idle":"2022-07-10T09:49:05.348628Z","shell.execute_reply.started":"2022-07-10T09:49:03.305750Z","shell.execute_reply":"2022-07-10T09:49:05.347416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Matrice de confusion","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\n\npredictions = np.array([])\nlabels =  np.array([])\nfor x, y in ds_val:\n    predictions = np.concatenate([predictions, model.predict(x).argmax(axis=-1)], axis=0)\n    labels = np.concatenate([labels, y], axis=0)\n\nplt.figure(figsize=(8,8))\nplt.title(\"Matrice de confusion\")\nsns.heatmap(tf.math.confusion_matrix(labels=labels, predictions=predictions).numpy(), annot=True, cmap='Oranges', fmt='g',)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-10T09:49:05.353555Z","iopub.execute_input":"2022-07-10T09:49:05.354664Z","iopub.status.idle":"2022-07-10T09:49:27.156161Z","shell.execute_reply.started":"2022-07-10T09:49:05.354606Z","shell.execute_reply":"2022-07-10T09:49:27.154799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"La matrice de confusion nous laisse penser que le modèle est globalement précis puisque qu'il y a beaucoup plus de vrais positifs que de faux negatifs/positifs. Cela transparait par le contraste entre la diagonale et le reste de la matrice.","metadata":{}},{"cell_type":"markdown","source":"## Résultat à soumettre","metadata":{"execution":{"iopub.execute_input":"2022-05-02T16:53:05.081503Z","iopub.status.busy":"2022-05-02T16:53:05.081093Z","iopub.status.idle":"2022-05-02T16:53:05.087029Z","shell.execute_reply":"2022-05-02T16:53:05.085976Z","shell.execute_reply.started":"2022-05-02T16:53:05.081444Z"}}},{"cell_type":"code","source":"X_test = np.load('/kaggle/working/ships_competition.npz', allow_pickle=True)[\"X\"]\nX_test = X_test.astype('float32') / 255.","metadata":{"execution":{"iopub.status.busy":"2022-07-10T09:49:27.158213Z","iopub.execute_input":"2022-07-10T09:49:27.158708Z","iopub.status.idle":"2022-07-10T09:49:27.187889Z","shell.execute_reply.started":"2022-07-10T09:49:27.158664Z","shell.execute_reply":"2022-07-10T09:49:27.186645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res = model.predict(X_test).argmax(axis=1)\ndf = pd.DataFrame({\"Category\":res})\ndf.to_csv(\"reco_nav.csv\", index_label=\"Id\")","metadata":{"execution":{"iopub.status.busy":"2022-07-10T09:49:27.189946Z","iopub.execute_input":"2022-07-10T09:49:27.190450Z","iopub.status.idle":"2022-07-10T09:49:32.393961Z","shell.execute_reply.started":"2022-07-10T09:49:27.190396Z","shell.execute_reply":"2022-07-10T09:49:32.392513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head reco_nav.csv","metadata":{"execution":{"iopub.status.busy":"2022-07-10T09:49:32.396038Z","iopub.execute_input":"2022-07-10T09:49:32.398103Z","iopub.status.idle":"2022-07-10T09:49:33.461234Z","shell.execute_reply.started":"2022-07-10T09:49:32.398046Z","shell.execute_reply":"2022-07-10T09:49:33.459814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.chdir(r'/kaggle/working')\nfrom IPython.display import FileLink\nFileLink(r'reco_nav.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T09:49:50.317427Z","iopub.execute_input":"2022-07-10T09:49:50.318331Z","iopub.status.idle":"2022-07-10T09:49:50.328861Z","shell.execute_reply.started":"2022-07-10T09:49:50.318278Z","shell.execute_reply":"2022-07-10T09:49:50.327271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf ships32/","metadata":{"execution":{"iopub.status.busy":"2022-07-09T23:31:30.236152Z","iopub.execute_input":"2022-07-09T23:31:30.236536Z","iopub.status.idle":"2022-07-09T23:31:32.105531Z","shell.execute_reply.started":"2022-07-09T23:31:30.236504Z","shell.execute_reply":"2022-07-09T23:31:32.104196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}