{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **HERBARIUM 2020 FGVC9 COMPETITION**\n- Only included a VGG16 transfer learning model to predict, not using any further metadata apart from **file_name**.\n- Models included to benchmarking:\n    - VGG16 + manual tunning of last layers\n    - VGG16 + AutoML","metadata":{}},{"cell_type":"code","source":"import os\n\n# Data working packages\nimport numpy as np\nimport pandas as pd\nimport json, codecs\n\n# Model building packages\nimport tensorflow as tf\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential, Model\nfrom keras.layers import Dense, Dropout, Flatten, Input\nfrom keras.applications.vgg16 import VGG16\nfrom keras.optimizers import Adam\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nimport keras_tuner as kt\n\n# Plotting packages\nfrom matplotlib import pyplot as plt\n\n# Set paths\nPATH = '/kaggle/input/herbarium-2022-fgvc9'\nTRAIN_IMG = os.path.join(PATH, \"train_images\")\nTEST_IMG = os.path.join(PATH, \"test_images\")","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:33:19.631900Z","iopub.execute_input":"2023-03-17T12:33:19.632371Z","iopub.status.idle":"2023-03-17T12:33:19.641787Z","shell.execute_reply.started":"2023-03-17T12:33:19.632323Z","shell.execute_reply":"2023-03-17T12:33:19.640219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **PIPELINE:**\n1. [Load metadata and create an unified train and test metadata dataframe](#1.-LOAD-METADATA)\n2. [Set up image generator](#2.-DATA-GENERATOR)\n3. [Model building](#3.-MODEL-BUILDING)\n    - [VGG16 + manual tunning](#3.1-VGG16-+-manual-tunning)\n    - [VGG16 + AutoML](#3.2-VGG16-+-AutoML)\n4. [Model prediction](#4.-MODEL-PREDICTION)\n5. [Submission](#5.-OUTPUT-SUBMISSION)","metadata":{}},{"cell_type":"markdown","source":"## 1. LOAD METADATA","metadata":{}},{"cell_type":"code","source":"# sample submission data\nsubmission_sample = pd.read_csv('../input/herbarium-2022-fgvc9/sample_submission.csv')\nsubmission_sample.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:33:19.643726Z","iopub.execute_input":"2023-03-17T12:33:19.644641Z","iopub.status.idle":"2023-03-17T12:33:19.720040Z","shell.execute_reply.started":"2023-03-17T12:33:19.644600Z","shell.execute_reply":"2023-03-17T12:33:19.718751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# .json with metadata of the images and read them as dictionaries\nwith codecs.open(\"../input/herbarium-2022-fgvc9/train_metadata.json\", \n                 'r', encoding='utf-8', errors='ignore') as file:\n    meta_train = json.load(file)\n    \nwith codecs.open(\"../input/herbarium-2022-fgvc9/test_metadata.json\", \n                 'r', encoding='utf-8', errors='ignore') as file:\n    meta_test = json.load(file)","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:33:19.722045Z","iopub.execute_input":"2023-03-17T12:33:19.722443Z","iopub.status.idle":"2023-03-17T12:33:28.472836Z","shell.execute_reply.started":"2023-03-17T12:33:19.722406Z","shell.execute_reply":"2023-03-17T12:33:28.471393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Get the important info of each metadata file\n#### TRAIN","metadata":{}},{"cell_type":"code","source":"meta_train.keys()","metadata":{"execution":{"iopub.status.busy":"2023-03-16T11:23:09.680195Z","iopub.execute_input":"2023-03-16T11:23:09.680612Z","iopub.status.idle":"2023-03-16T11:23:09.688561Z","shell.execute_reply.started":"2023-03-16T11:23:09.680572Z","shell.execute_reply":"2023-03-16T11:23:09.687149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extract annotations of the metadata as df\nann_train = pd.DataFrame(meta_train['annotations'])\nprint(ann_train.shape)\nann_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:33:28.475030Z","iopub.execute_input":"2023-03-17T12:33:28.475761Z","iopub.status.idle":"2023-03-17T12:33:29.745265Z","shell.execute_reply.started":"2023-03-17T12:33:28.475708Z","shell.execute_reply":"2023-03-17T12:33:29.744267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extract images of the metadata as df\nim_train = pd.DataFrame(meta_train['images'])\nprint(im_train.shape)\nim_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:33:29.746985Z","iopub.execute_input":"2023-03-17T12:33:29.747408Z","iopub.status.idle":"2023-03-17T12:33:30.702541Z","shell.execute_reply.started":"2023-03-17T12:33:29.747366Z","shell.execute_reply":"2023-03-17T12:33:30.701184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extract categories of the metadata as df\ncat_train = pd.DataFrame(meta_train['categories'])\nprint(cat_train.shape)\ncat_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:33:30.704392Z","iopub.execute_input":"2023-03-17T12:33:30.705568Z","iopub.status.idle":"2023-03-17T12:33:30.748987Z","shell.execute_reply.started":"2023-03-17T12:33:30.705519Z","shell.execute_reply":"2023-03-17T12:33:30.747584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extract genera of the metadata as df\ngen_train = pd.DataFrame(meta_train['genera'])\nprint(gen_train.shape)\ngen_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:33:30.750988Z","iopub.execute_input":"2023-03-17T12:33:30.751409Z","iopub.status.idle":"2023-03-17T12:33:30.770273Z","shell.execute_reply.started":"2023-03-17T12:33:30.751367Z","shell.execute_reply":"2023-03-17T12:33:30.768907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extract institutions of the metadata as df\nins_train = pd.DataFrame(meta_train['institutions'])\nprint(ins_train.shape)\nins_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:33:30.772043Z","iopub.execute_input":"2023-03-17T12:33:30.772566Z","iopub.status.idle":"2023-03-17T12:33:30.788152Z","shell.execute_reply.started":"2023-03-17T12:33:30.772510Z","shell.execute_reply":"2023-03-17T12:33:30.786656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extract distances of the metadata as df\ndis_train = pd.DataFrame(meta_train['distances'])\nprint(dis_train.shape)\ndis_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:33:30.792243Z","iopub.execute_input":"2023-03-17T12:33:30.792716Z","iopub.status.idle":"2023-03-17T12:33:35.154765Z","shell.execute_reply.started":"2023-03-17T12:33:30.792652Z","shell.execute_reply":"2023-03-17T12:33:35.153535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extract distances of the metadata as df\nli_train = pd.DataFrame(meta_train['license'])\nprint(li_train.shape)\nli_train","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:33:35.156566Z","iopub.execute_input":"2023-03-17T12:33:35.156971Z","iopub.status.idle":"2023-03-17T12:33:35.172167Z","shell.execute_reply.started":"2023-03-17T12:33:35.156933Z","shell.execute_reply":"2023-03-17T12:33:35.170690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After explore the whole metadata, the most interesting keys are:\n- **Annotations**: with the *genus_id*, institution_id, *category_id* and *image_id*\n- **Images**: *image_id*, file_name and licence\n- **Categories**: *category_id*, scientificName, family, genus, species, authors","metadata":{}},{"cell_type":"code","source":"## Unify all train metadata by corresponding id\ndf_train = ann_train.merge(im_train, on='image_id', how='outer')\ndf_train = df_train.merge(cat_train, on='category_id', how='outer')\ndf_train","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:33:35.173959Z","iopub.execute_input":"2023-03-17T12:33:35.174385Z","iopub.status.idle":"2023-03-17T12:33:36.669146Z","shell.execute_reply.started":"2023-03-17T12:33:35.174344Z","shell.execute_reply":"2023-03-17T12:33:36.667666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### TEST","metadata":{}},{"cell_type":"code","source":"df_test = pd.DataFrame(meta_test)\ndf_test","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:33:36.671076Z","iopub.execute_input":"2023-03-17T12:33:36.671671Z","iopub.status.idle":"2023-03-17T12:33:36.929626Z","shell.execute_reply.started":"2023-03-17T12:33:36.671603Z","shell.execute_reply":"2023-03-17T12:33:36.928299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. DATA GENERATOR","metadata":{}},{"cell_type":"markdown","source":"### BARE IN MIND!!\n1. Using **only 0.01%** of the whole data to reduce time consumption on each split\n2. Also, to achieve better performance, there will be **only 10 categories of each class**\n3. **Remove next code cell** this to execute the original problem","metadata":{}},{"cell_type":"code","source":"# only select category_id 0-10 to reduce complexity (getting 661 images)\ndf_train = df_train[(df_train.category_id < 11)]\n\n# Get only 0.001% of each test data split (getting 21 images)\ndf_test = df_test.sample(frac=0.0001, random_state=27)","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:38:29.811170Z","iopub.execute_input":"2023-03-17T12:38:29.812549Z","iopub.status.idle":"2023-03-17T12:38:29.833933Z","shell.execute_reply.started":"2023-03-17T12:38:29.812477Z","shell.execute_reply":"2023-03-17T12:38:29.832060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Transform data to introduce in the model","metadata":{}},{"cell_type":"code","source":"# create a label encoder for the target variable and divide \ndata = df_train[['file_name','category_id']].copy()\ntest_data = df_test[['file_name']].copy()\n\n# principal y output\nle1 = LabelEncoder()\ndata['category_id'] = le1.fit_transform(data['category_id'])\n\n# secondary y outputs to enhace the model\n#data = df_train[['file_name','category_id','family','genus_id']].copy()\n#le2 = LabelEncoder()\n#le3 = LabelEncoder()\n#data['family'] = le2.fit_transform(data['family'])\n#data['genus_id'] = le3.fit_transform(data['genus_id'])\n\n# ImageDataGenerator().flow_from_dataframe(class_mode = 'sparse') requires string labels\ndata['category_id'] = data['category_id'].astype('str')\ndata","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:38:32.746610Z","iopub.execute_input":"2023-03-17T12:38:32.748499Z","iopub.status.idle":"2023-03-17T12:38:32.776413Z","shell.execute_reply.started":"2023-03-17T12:38:32.748423Z","shell.execute_reply":"2023-03-17T12:38:32.774852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Divide train and val data","metadata":{}},{"cell_type":"code","source":"train_data, val_data = train_test_split(data, test_size=0.2, random_state=42)\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:38:36.644599Z","iopub.execute_input":"2023-03-17T12:38:36.645084Z","iopub.status.idle":"2023-03-17T12:38:36.673399Z","shell.execute_reply.started":"2023-03-17T12:38:36.645041Z","shell.execute_reply":"2023-03-17T12:38:36.671548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Create the data generator for train and validation - category_id as output","metadata":{}},{"cell_type":"code","source":"# set up the data generators \ndatagen = ImageDataGenerator(rescale=1./255)\n\ntrain_generator = datagen.flow_from_dataframe(\n    dataframe=train_data,\n    directory=TRAIN_IMG, \n    x_col='file_name',\n    y_col='category_id',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='sparse')\n\nval_generator = datagen.flow_from_dataframe(\n    dataframe=val_data,\n    directory=TRAIN_IMG,\n    x_col='file_name',\n    y_col='category_id',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='sparse')","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:38:41.241365Z","iopub.execute_input":"2023-03-17T12:38:41.241905Z","iopub.status.idle":"2023-03-17T12:38:43.792842Z","shell.execute_reply.started":"2023-03-17T12:38:41.241859Z","shell.execute_reply":"2023-03-17T12:38:43.790992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Create the data generator for test","metadata":{}},{"cell_type":"code","source":"# set up the data generator for the test data\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\ntest_generator = test_datagen.flow_from_dataframe(\n    dataframe=test_data,\n    directory=TEST_IMG,\n    x_col='file_name',\n    y_col=None,\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode=None,\n    shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:40:49.047451Z","iopub.execute_input":"2023-03-17T12:40:49.047967Z","iopub.status.idle":"2023-03-17T12:40:49.291809Z","shell.execute_reply.started":"2023-03-17T12:40:49.047922Z","shell.execute_reply":"2023-03-17T12:40:49.290671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#STEP_SIZE_TRAIN=train_generator.n//train_generator.batch_size\n#STEP_SIZE_VALID=val_generator.n//val_generator.batch_size\n#STEP_SIZE_TEST=test_generator.n//test_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2023-03-16T11:23:18.489444Z","iopub.execute_input":"2023-03-16T11:23:18.490702Z","iopub.status.idle":"2023-03-16T11:23:18.495866Z","shell.execute_reply.started":"2023-03-16T11:23:18.490657Z","shell.execute_reply":"2023-03-16T11:23:18.494638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. MODEL BUILDING","metadata":{}},{"cell_type":"markdown","source":"####  Load the pre-trained [VGG16 Model](https://keras.io/api/applications/vgg/)","metadata":{}},{"cell_type":"code","source":"vgg16 = VGG16(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n\n# freeze the pre-trained layers\nfor layer in vgg16.layers:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:40:54.433317Z","iopub.execute_input":"2023-03-17T12:40:54.433968Z","iopub.status.idle":"2023-03-17T12:40:55.769072Z","shell.execute_reply.started":"2023-03-17T12:40:54.433908Z","shell.execute_reply":"2023-03-17T12:40:55.766747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.1 VGG16 + manual tunning\nOnly file_name metadata","metadata":{}},{"cell_type":"code","source":"x = vgg16.output\nx = tf.keras.layers.Flatten()(x)\n\nx = Dense(1024, activation=\"relu\", name='manual_dense_1')(x)\nx = Dropout(0.5)(x)\noutput = Dense(len(le1.classes_), activation=\"softmax\")(x)\n\nmodel = Model(inputs=vgg16.input, outputs=output)\n\n## ESENTIAL TO USE SPARSER_CATEGORICAL_CROSSENTROPY DUE TO OTHERWISE: \n##The error arose because 'categorical_crossentropy' works on one-hot encoded target, \n##while 'sparse_categorical_crossentropy' works on integer target.\nmodel.compile(loss='sparse_categorical_crossentropy', \n               optimizer=tf.keras.optimizers.Adam(lr=0.001), \n               metrics=['accuracy'])\n\nhistory = model.fit(train_generator,\n                    #steps_per_epoch=STEP_SIZE_TRAIN,\n                    epochs = 10,\n                    validation_data = val_generator,\n                    verbose = 0)\n                    #,\n                    #validation_steps=STEP_SIZE_VALID)\nhistory","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:42:44.465483Z","iopub.execute_input":"2023-03-17T12:42:44.465999Z","iopub.status.idle":"2023-03-17T12:51:30.442588Z","shell.execute_reply.started":"2023-03-17T12:42:44.465951Z","shell.execute_reply":"2023-03-17T12:51:30.440981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'validation'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-16T11:33:32.147163Z","iopub.execute_input":"2023-03-16T11:33:32.148114Z","iopub.status.idle":"2023-03-16T11:33:32.380115Z","shell.execute_reply.started":"2023-03-16T11:33:32.148059Z","shell.execute_reply":"2023-03-16T11:33:32.378692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m_eval_dict = model.evaluate(val_generator,\n                           return_dict=True)\nm_eval_dict","metadata":{"execution":{"iopub.status.busy":"2023-03-16T11:33:32.382177Z","iopub.execute_input":"2023-03-16T11:33:32.383259Z","iopub.status.idle":"2023-03-16T11:33:43.083271Z","shell.execute_reply.started":"2023-03-16T11:33:32.383218Z","shell.execute_reply":"2023-03-16T11:33:43.081923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.2 VGG16 + AutoML\nOnly file_name metadata","metadata":{}},{"cell_type":"code","source":"def model_builder(hp):\n    \"\"\"\n    Builds the model and sets up the hyperparameters to tune.\n    :params:\n    hp - Keras tuner object\n    :returns:\n    model - with hyperparameters to tune\n    \"\"\"\n    x = vgg16.output\n    x = tf.keras.layers.Flatten(input_shape=(28, 28))(x)\n\n    hp_units = hp.Int('units', min_value=32, max_value=1024, step=32)\n    x = Dense(units=hp_units, activation='relu', name='tuned_dense')(x)\n    \n    x = Dropout(0.5)(x)\n    output = Dense(len(le1.classes_), activation=\"softmax\")(x)\n    \n    model = Model(inputs=vgg16.input, outputs=output)\n\n    hp_learning_rate = hp.Choice('learning_rate', values=[1e-2, 1e-3, 1e-4])\n\n    model.compile(loss='sparse_categorical_crossentropy', \n                  optimizer=tf.keras.optimizers.Adam(lr=hp_learning_rate), \n                  metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-03-16T11:33:43.086951Z","iopub.execute_input":"2023-03-16T11:33:43.087353Z","iopub.status.idle":"2023-03-16T11:33:43.097754Z","shell.execute_reply.started":"2023-03-16T11:33:43.087314Z","shell.execute_reply":"2023-03-16T11:33:43.096353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tuner = kt.Hyperband(model_builder,\n                     objective='val_accuracy',\n                     max_epochs=10,\n                     factor=3,\n                     directory='kt_dir',\n                     project_name='kt_hyperband')\n\ntuner.search_space_summary()\n\nstop_early = tf.keras.callbacks.EarlyStopping(monitor='val_accuracy', patience=20)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T13:36:34.492297Z","iopub.execute_input":"2023-03-16T13:36:34.492790Z","iopub.status.idle":"2023-03-16T13:36:34.535817Z","shell.execute_reply.started":"2023-03-16T13:36:34.492747Z","shell.execute_reply":"2023-03-16T13:36:34.534003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tuner.search(train_generator, callbacks=[stop_early], validation_data=val_generator)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T13:38:22.114138Z","iopub.execute_input":"2023-03-16T13:38:22.115445Z","iopub.status.idle":"2023-03-16T13:39:10.161167Z","shell.execute_reply.started":"2023-03-16T13:38:22.115377Z","shell.execute_reply":"2023-03-16T13:39:10.158348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_hps=tuner.get_best_hyperparameters()[0]\nprint(f\"\"\"\nThe hyperparameter search is complete. \nThe optimal number of units in the hidden\nlayer is {best_hps.get('units')} and the \noptimal learning rate for the optimizer\nis {best_hps.get('learning_rate')}.\n\"\"\")","metadata":{"execution":{"iopub.status.busy":"2023-03-16T13:09:43.205618Z","iopub.execute_input":"2023-03-16T13:09:43.209308Z","iopub.status.idle":"2023-03-16T13:09:43.220767Z","shell.execute_reply.started":"2023-03-16T13:09:43.209245Z","shell.execute_reply":"2023-03-16T13:09:43.219331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model with VGG16 + last layer with best hyperparams\nh_model = tuner.hypermodel.build(best_hps)\nh_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-16T13:09:43.232202Z","iopub.execute_input":"2023-03-16T13:09:43.233394Z","iopub.status.idle":"2023-03-16T13:09:43.436765Z","shell.execute_reply.started":"2023-03-16T13:09:43.233330Z","shell.execute_reply":"2023-03-16T13:09:43.435388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"h_history = h_model.fit(train_generator,\n                    #steps_per_epoch=STEP_SIZE_TRAIN,\n                    epochs=10,\n                    validation_data=val_generator,\n                    verbose = 0)\n                    #,\n                    #validation_steps=STEP_SIZE_VALID)\nh_history","metadata":{"execution":{"iopub.status.busy":"2023-03-16T13:09:43.438588Z","iopub.execute_input":"2023-03-16T13:09:43.439068Z","iopub.status.idle":"2023-03-16T13:19:52.078003Z","shell.execute_reply.started":"2023-03-16T13:09:43.439016Z","shell.execute_reply":"2023-03-16T13:19:52.077068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(h_history.history['accuracy'])\nplt.plot(h_history.history['val_accuracy'])\nplt.title('tunned model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'validation'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-16T13:19:52.079314Z","iopub.execute_input":"2023-03-16T13:19:52.080323Z","iopub.status.idle":"2023-03-16T13:19:52.317116Z","shell.execute_reply.started":"2023-03-16T13:19:52.080266Z","shell.execute_reply":"2023-03-16T13:19:52.313368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"h_eval_dict = h_model.evaluate(val_generator,\n                               return_dict=True)\nh_eval_dict","metadata":{"execution":{"iopub.status.busy":"2023-03-16T13:19:52.319047Z","iopub.execute_input":"2023-03-16T13:19:52.319554Z","iopub.status.idle":"2023-03-16T13:20:03.026666Z","shell.execute_reply.started":"2023-03-16T13:19:52.319505Z","shell.execute_reply":"2023-03-16T13:20:03.025190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. MODEL PREDICTION","metadata":{}},{"cell_type":"markdown","source":"### Benchmarking of models to select the best","metadata":{}},{"cell_type":"code","source":"# Print results of the baseline and hypertuned model\nprint('Manual model achieved in validation data a loss of: {} and an accuracy of: {}'.format(m_eval_dict['loss'], m_eval_dict['accuracy']))\nprint('Tunned model achieved in validation data a loss of: {} and an accuracy of: {}'.format(h_eval_dict['loss'], h_eval_dict['accuracy']))","metadata":{"execution":{"iopub.status.busy":"2023-03-16T13:31:20.166716Z","iopub.execute_input":"2023-03-16T13:31:20.167182Z","iopub.status.idle":"2023-03-16T13:31:20.173900Z","shell.execute_reply.started":"2023-03-16T13:31:20.167138Z","shell.execute_reply":"2023-03-16T13:31:20.172610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mymodels = [m_eval_dict, h_eval_dict]\ndf_models = pd.DataFrame.from_records(mymodels).fillna(0)\ndf_models.index = ['Manual model', 'Tunned model']\ndf_models","metadata":{"execution":{"iopub.status.busy":"2023-03-16T13:31:29.405569Z","iopub.execute_input":"2023-03-16T13:31:29.405980Z","iopub.status.idle":"2023-03-16T13:31:29.422446Z","shell.execute_reply.started":"2023-03-16T13:31:29.405941Z","shell.execute_reply":"2023-03-16T13:31:29.420964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select best model\nbest_model = df_models[['accuracy']].idxmax()[0]\nbest_model\nif best_model == 'Manual model':\n    best_model = model\nelse:\n    best_model = h_model","metadata":{"execution":{"iopub.status.busy":"2023-03-16T13:33:38.520187Z","iopub.execute_input":"2023-03-16T13:33:38.520896Z","iopub.status.idle":"2023-03-16T13:33:38.528220Z","shell.execute_reply.started":"2023-03-16T13:33:38.520847Z","shell.execute_reply":"2023-03-16T13:33:38.527250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Predict with the model that better result had in the benchmarking","metadata":{}},{"cell_type":"code","source":"test_generator.reset()\npred = best_model.predict(test_generator,\n                    steps=len(test_data),\n                    verbose=1)\n\npredicted_class_indices = np.argmax(pred,axis=1)\nlabels = le1.classes_\npredictions = [labels[k] for k in predicted_class_indices]","metadata":{"execution":{"iopub.status.busy":"2023-03-16T13:33:42.768252Z","iopub.execute_input":"2023-03-16T13:33:42.769379Z","iopub.status.idle":"2023-03-16T13:33:49.188680Z","shell.execute_reply.started":"2023-03-16T13:33:42.769330Z","shell.execute_reply":"2023-03-16T13:33:49.187366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dataframe of predictions\ndf_test.reset_index(drop=True, inplace=True)\n\ndf_pred = pd.DataFrame(predictions, columns=['Prediction'])\ndf_pred.reset_index(drop=True, inplace=True)\n\ndf_pred = pd.concat([df_test['image_id'], df_pred['Prediction']], axis=1, keys=['Image_id', 'Prediction'])\ndf_pred","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. OUTPUT SUBMISSION\n\n**Only possible to send output if the 100% of the test data is predicted**!!","metadata":{}},{"cell_type":"markdown","source":"### Create the dataframe to send the submission_sample","metadata":{}},{"cell_type":"code","source":"#submission_sample['Predicted'] = predictions\n#submission_sample.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T13:20:03.063339Z","iopub.status.idle":"2023-03-16T13:20:03.063761Z","shell.execute_reply.started":"2023-03-16T13:20:03.063554Z","shell.execute_reply":"2023-03-16T13:20:03.063575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# send to evaluate\n#submission_sample.to_csv('submission.csv', index = False) ","metadata":{"execution":{"iopub.status.busy":"2023-03-16T13:20:03.064877Z","iopub.status.idle":"2023-03-16T13:20:03.065548Z","shell.execute_reply.started":"2023-03-16T13:20:03.065319Z","shell.execute_reply":"2023-03-16T13:20:03.065350Z"},"trusted":true},"execution_count":null,"outputs":[]}]}