{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport cv2\nfrom tensorflow import keras\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-11T09:16:22.469483Z","iopub.status.idle":"2021-12-11T09:16:22.469946Z","shell.execute_reply.started":"2021-12-11T09:16:22.469736Z","shell.execute_reply":"2021-12-11T09:16:22.469757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q tensorflow==2.3.0 # Use 2.3.0 for built-in EfficientNet\n!pip install -q git+https://github.com/keras-team/keras-tuner@master # Use github head for newly added TPU support\n!pip install -q cloud-tpu-client # Needed for sync TPU version\n!pip install -U tensorflow-gcs-config==2.3.0 # Needed for using private dataset","metadata":{"execution":{"iopub.status.busy":"2021-12-11T09:16:22.471313Z","iopub.status.idle":"2021-12-11T09:16:22.471669Z","shell.execute_reply.started":"2021-12-11T09:16:22.471465Z","shell.execute_reply":"2021-12-11T09:16:22.471486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # Sync TPU version\n    from cloud_tpu_client import Client\n    c = Client()\n    c.configure_tpu_version(tf.__version__, restart_type='ifNeeded')\n    \n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n    \n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy() # default distribution strategy in Tensorflow. Works on CPU and single GPU.\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T09:16:22.472749Z","iopub.status.idle":"2021-12-11T09:16:22.4733Z","shell.execute_reply.started":"2021-12-11T09:16:22.47309Z","shell.execute_reply":"2021-12-11T09:16:22.473114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data parameter\ninput_dir = os.path.join('..', 'input')\noutput_dir = os.path.join('..', 'output')","metadata":{"execution":{"iopub.status.busy":"2021-12-11T09:16:22.474383Z","iopub.status.idle":"2021-12-11T09:16:22.474745Z","shell.execute_reply.started":"2021-12-11T09:16:22.474545Z","shell.execute_reply":"2021-12-11T09:16:22.474566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_dir = os.path.join(input_dir, 'landmark-recognition-2020')\ntrain_dir = os.path.join(dataset_dir, 'train')\ntrain_labelmap_dir = os.path.join(dataset_dir, 'train.csv')\ntest_dir = os.path.join(dataset_dir, 'test')\ntest_labelmap_dir = os.path.join(dataset_dir, 'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-11T09:16:22.475876Z","iopub.status.idle":"2021-12-11T09:16:22.476203Z","shell.execute_reply.started":"2021-12-11T09:16:22.476029Z","shell.execute_reply":"2021-12-11T09:16:22.47605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(train_labelmap_dir)\ntest_df = pd.read_csv(test_labelmap_dir)\nnum_data = len(train_df)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T09:16:22.477098Z","iopub.status.idle":"2021-12-11T09:16:22.477433Z","shell.execute_reply.started":"2021-12-11T09:16:22.477257Z","shell.execute_reply":"2021-12-11T09:16:22.477278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"The number of train images is :\", train_df.shape[0])\nprint(\"The number of test images is :\", test_df.shape[0])\nprint('The total number of images is :', train_df.shape[0]+test_df.shape[0])","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:28:36.056885Z","iopub.execute_input":"2021-12-10T17:28:36.057168Z","iopub.status.idle":"2021-12-10T17:28:36.064368Z","shell.execute_reply.started":"2021-12-10T17:28:36.057138Z","shell.execute_reply":"2021-12-10T17:28:36.063518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"landmark = train_df.landmark_id.value_counts()\nlandmark_df = pd.DataFrame({'landmark_id':landmark.index, 'frequency':landmark.values})#.head(30)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:28:36.948788Z","iopub.execute_input":"2021-12-10T17:28:36.949109Z","iopub.status.idle":"2021-12-10T17:28:36.990176Z","shell.execute_reply.started":"2021-12-10T17:28:36.94908Z","shell.execute_reply":"2021-12-10T17:28:36.988868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"landmark_df.tail()","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:28:37.915981Z","iopub.execute_input":"2021-12-10T17:28:37.917192Z","iopub.status.idle":"2021-12-10T17:28:37.926857Z","shell.execute_reply.started":"2021-12-10T17:28:37.917136Z","shell.execute_reply":"2021-12-10T17:28:37.925974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"landmark_df.reset_index(inplace=True)\nprint(\"Amount of classes with less than 5 trainning samples:\", (landmark_df['frequency'].between(0,4,True)).sum())\nprint(\"Amount of classes with between 5 and 10 training samples:\", (landmark_df['frequency'].between(5,10,True)).sum())\n\nprint(landmark_df)\nplt.figure(figsize=(20,6))\nplt.hist(train_df.landmark_id, bins=81312,log=True);\nplt.title('Images per class', fontsize=16)\nplt.xlabel('Class number')\nplt.ylabel('Number of images')\nplt.show()\n\nplt.figure(figsize = (10, 8))\nplt.title('Landmark ID Distribuition')\nsns.distplot(landmark_df['landmark_id'])\nplt.show()\n\nprint(\"The data distribution will affect the training process negatively, since some classes have a very large number of samples when compared with other classes. \\n For example the largest class contains 6272 images where there are 4749 classes which contain only 2 images, meaning that the largest class will have higher impact \\n when trying to do some predictions after training the model. \\n in other words, when trying to predict a sample from the small classes, 99% of the times, the classifier will predict it as it belongs to the large class, which is refered to as generalization. \") ","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:30:10.183941Z","iopub.execute_input":"2021-12-10T17:30:10.184228Z","iopub.status.idle":"2021-12-10T17:31:29.431647Z","shell.execute_reply.started":"2021-12-10T17:30:10.184201Z","shell.execute_reply":"2021-12-10T17:31:29.430087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize different images","metadata":{}},{"cell_type":"code","source":"### Visualize random images from the dataset\n\ndef get_image_from_number(num):\n    fname, label = train_df.loc[num,:]\n    fname = fname + \".jpg\"\n    f1 = fname[0]\n    f2 = fname[1]\n    f3 = fname[2]\n    path = os.path.join(f1,f2,f3,fname)\n    im = cv2.imread(os.path.join(train_dir,path))\n    RGB_img = cv2.cvtColor(im, cv2.COLOR_BGR2RGB)\n    return RGB_img, label\n\n\n# fig=plt.figure(figsize=(16, 8))\n\n# columns = 4\n# rows = 1\n# for i in range(1, columns*rows +1):\n#     n = np.random.randint(num_data)\n#     img, lbl = get_image_from_number(n)\n#     fig.add_subplot(rows, columns, i)\n#     plt.imshow(img)\n#     plt.title(\"Label = \" + str(lbl))\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:31:32.323274Z","iopub.execute_input":"2021-12-10T17:31:32.324052Z","iopub.status.idle":"2021-12-10T17:31:32.330051Z","shell.execute_reply.started":"2021-12-10T17:31:32.324013Z","shell.execute_reply":"2021-12-10T17:31:32.329433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Adding two extra columns to the data-frame, where filename will be used to retreive the images when needed, and the \"label\" is the integer version of the \"landmark_id\"\ntrain_df[\"filename\"] = train_df.id.str[0]+\"/\"+train_df.id.str[1]+\"/\"+train_df.id.str[2]+\"/\"+train_df.id+\".jpg\"\ntrain_df[\"label\"] = train_df.landmark_id.astype(str)\n\n#Due to the limited resources, i will be keeping only the top 1000 classes of the data.\nfrom collections import Counter\n\nnumber_of_classes_to_keep = 1000\n\nc = train_df.landmark_id.values\ncount = Counter(c).most_common(number_of_classes_to_keep)\nkeep_labels = [i[0] for i in count]\ntrain_keep = train_df[train_df.landmark_id.isin(keep_labels)]\n\n#Since we took the 1000 most common classes, we will reset the index of the samples in addition to shuffling them.\ntrain_keep = train_keep.sample(frac=1).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:31:34.933009Z","iopub.execute_input":"2021-12-10T17:31:34.933418Z","iopub.status.idle":"2021-12-10T17:31:43.360949Z","shell.execute_reply.started":"2021-12-10T17:31:34.933373Z","shell.execute_reply":"2021-12-10T17:31:43.360058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n# detect and init the TPU\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\n# instantiate a distribution strategy\ntpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training model","metadata":{}},{"cell_type":"markdown","source":"Importing the needed libraries","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import VGG19\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom tensorflow.keras.models import load_model\nfrom tensorflow import keras\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dropout\nfrom sklearn.utils import class_weight","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:31:47.862245Z","iopub.execute_input":"2021-12-10T17:31:47.862638Z","iopub.status.idle":"2021-12-10T17:31:47.870264Z","shell.execute_reply.started":"2021-12-10T17:31:47.862592Z","shell.execute_reply":"2021-12-10T17:31:47.869084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The used settings**","metadata":{}},{"cell_type":"code","source":"val_split = 0.2 #The percentage of the validation data out of the entire training data\nbatch_size = 32 \nlearning_rate = 0.005 # The learning rate\n# opt = tf.keras.optimizers.SGD(lr=learning_rate, momentum=0.9) #The optimizor\nopt = keras.optimizers.Adagrad(learning_rate = learning_rate, initial_accumulator_value=0.01, epsilon=1e-07)\nepochs = 30 # The number of epochs ","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:31:54.671544Z","iopub.execute_input":"2021-12-10T17:31:54.671866Z","iopub.status.idle":"2021-12-10T17:31:54.68191Z","shell.execute_reply.started":"2021-12-10T17:31:54.67182Z","shell.execute_reply":"2021-12-10T17:31:54.681143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The data generation and preprocessing step, before the training","metadata":{}},{"cell_type":"code","source":"datagen = ImageDataGenerator(validation_split=val_split,\n                             rescale=1.0/255.0,\n                             rotation_range=40,\n                             shear_range=0.2,\n                             zoom_range=0.2,\n                             horizontal_flip=True\n                            )\n\ntrain_datagen = datagen.flow_from_dataframe(\n    train_keep, # Pandas dataframe containing the filepaths relative to directory (or absolute paths if directory is None) and classes label\n    directory=train_dir + \"/\",\n    x_col=\"filename\",\n    y_col=\"label\",\n    weight_col=None,\n    target_size=(224, 224),\n    color_mode=\"rgb\",\n    classes=None,\n    class_mode=\"categorical\",\n    batch_size=batch_size,\n    shuffle=True,\n    subset=\"training\",\n    interpolation=\"nearest\",\n    validate_filenames=False)\n\nval_datagen = datagen.flow_from_dataframe(\n    train_keep, # Pandas dataframe containing the filepaths relative to directory (or absolute paths if directory is None) and classes label\n    directory=train_dir + \"/\",\n    x_col=\"filename\",\n    y_col=\"label\",\n    weight_col=None,\n    target_size=(224, 224),\n    color_mode=\"rgb\",\n    classes=None,\n    class_mode=\"categorical\",\n    batch_size=batch_size,\n    shuffle=True,\n    subset=\"validation\",\n    interpolation=\"nearest\",\n    validate_filenames=False)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:31:56.096209Z","iopub.execute_input":"2021-12-10T17:31:56.096633Z","iopub.status.idle":"2021-12-10T17:31:58.026714Z","shell.execute_reply.started":"2021-12-10T17:31:56.096604Z","shell.execute_reply":"2021-12-10T17:31:58.025678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The network architecture used to train the model","metadata":{}},{"cell_type":"markdown","source":"Preparing the class weights to be fed to the network when the training starts\n","metadata":{}},{"cell_type":"code","source":"class_weights = class_weight.compute_class_weight('balanced',\n                                                 np.unique(train_keep.landmark_id),\n                                                 train_keep.landmark_id)\nclass_weights = dict(enumerate(class_weights))","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:32:04.909213Z","iopub.execute_input":"2021-12-10T17:32:04.909991Z","iopub.status.idle":"2021-12-10T17:32:05.033693Z","shell.execute_reply.started":"2021-12-10T17:32:04.909946Z","shell.execute_reply":"2021-12-10T17:32:05.032972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training the model","metadata":{}},{"cell_type":"code","source":"def create_model():\n    model = Sequential()\n    model.add(Input(shape=(224,224,3)))\n    model.add(BatchNormalization())\n    model.add(Conv2D(64, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(Conv2D(64, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(MaxPooling2D())\n    model.add(Conv2D(128, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(Conv2D(128, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(MaxPooling2D())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(MaxPooling2D())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(MaxPooling2D())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(MaxPooling2D())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(Conv2D(256, kernel_size = (3,3), padding = \"same\"))\n    model.add(BatchNormalization())\n    model.add(MaxPooling2D())\n    model.add(Flatten())\n    model.add(Dense(4096, activation = \"relu\"))\n    model.add(Dropout(0.2))\n    model.add(Dense(4096, activation = \"relu\"))\n    model.add(Dropout(0.2))\n    model.add(Dense(number_of_classes_to_keep, activation=\"softmax\"))\n\n    return model\n    ","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:38:24.684351Z","iopub.execute_input":"2021-12-10T17:38:24.685022Z","iopub.status.idle":"2021-12-10T17:38:24.708078Z","shell.execute_reply.started":"2021-12-10T17:38:24.684984Z","shell.execute_reply":"2021-12-10T17:38:24.707261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_steps = int(len(train_keep)*(1-val_split))//batch_size\nval_steps = int(len(train_keep)*val_split)//batch_size\n\nmodel_checkpoint = ModelCheckpoint(\"best_model.h5\", save_best_only=True, verbose=1)\n\nwith tpu_strategy.scope():\n    model = create_model()\n    model.compile(\n        optimizer=opt,\n        loss = 'categorical_crossentropy', \n        metrics=['categorical_accuracy']\n    )\n    history = model.fit(train_datagen, \n                        steps_per_epoch=train_steps,\n                        epochs=epochs,\n                        validation_data=val_datagen,\n                        validation_steps=val_steps,\n                        class_weight=class_weights,\n                        callbacks=[model_checkpoint])\n\nmodel.save(\"model.h5\")","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:38:45.678642Z","iopub.execute_input":"2021-12-10T17:38:45.679264Z","iopub.status.idle":"2021-12-10T17:38:57.144781Z","shell.execute_reply.started":"2021-12-10T17:38:45.679227Z","shell.execute_reply":"2021-12-10T17:38:57.143668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Validate model","metadata":{}},{"cell_type":"markdown","source":"Plotting the accuracy/loss metrices ","metadata":{}},{"cell_type":"code","source":"print(history.history.keys())\nplt.plot(history.history['categorical_accuracy'])\nplt.plot(history.history['val_categorical_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()\n# summarize history for loss\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:27:41.330236Z","iopub.status.idle":"2021-12-10T17:27:41.330698Z","shell.execute_reply.started":"2021-12-10T17:27:41.330449Z","shell.execute_reply":"2021-12-10T17:27:41.330472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Doing a prediction on the validation data, just to gen an insight on how the model is performing.","metadata":{}},{"cell_type":"code","source":"predict = model.predict(val_datagen, val_steps,verbose=1)\n\ngood_preds = []\nbad_preds = []\n\nval_filenames = val_datagen.filenames\nlabel_map = (val_datagen.class_indices)\n#label_categories = to_categorical(np.asarray(labels)) \ncla = np.argmax(predict, axis=1)\nlabel_map = list(map(int, label_map.keys()))\nval_label = val_datagen.labels\n\nfor idx, res in enumerate(predict):\n    #print(\"image_id: \", val_filenames[idx], \", class predict: \", label_map[cla[idx]], \"class: \", label_map[val_label[idx]])\n    \n    if label_map[cla[idx]] != label_map[val_label[idx]]:\n        bad_preds.append([val_filenames[idx], label_map[cla[idx]], label_map[val_label[idx]], res[cla[idx]]])\n    else:\n        good_preds.append([val_filenames[idx], label_map[cla[idx]], label_map[val_label[idx]], res[cla[idx]]])\nprint(\"wrong predictions: \", len(bad_preds), \" right predictions: \", len(good_preds), \" acc: \", np.round(100*(len(predict)-len(bad_preds))/len(predict),2))","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:27:41.333318Z","iopub.status.idle":"2021-12-10T17:27:41.33367Z","shell.execute_reply.started":"2021-12-10T17:27:41.33351Z","shell.execute_reply":"2021-12-10T17:27:41.333526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### plot some of the best predictions\nfig=plt.figure(figsize=(16, 8))\n\ngood_preds = np.array(good_preds)\ngood_preds = np.array(sorted(good_preds, key = lambda x: x[3], reverse=True))\n#print(good_preds.shape)\n\ncolumns = 4\nrows = 1\nfor i in range(1, columns*rows +1):\n    n = good_preds[i,0]\n    #print(n)\n    img = cv2.imread(os.path.join(train_dir,n))\n    RGB_img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    lbl = good_preds[i,2]\n    fig.add_subplot(rows, columns, i)\n    plt.imshow(RGB_img)\n    lbl2 = good_preds[i,1]\n    plt.title(\"Label = \" + str(lbl) + \" \\nClassified:\" + str(lbl2) + \" \\nConfidence:\" + str(good_preds[i,3]))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:27:41.335236Z","iopub.status.idle":"2021-12-10T17:27:41.33598Z","shell.execute_reply.started":"2021-12-10T17:27:41.335652Z","shell.execute_reply":"2021-12-10T17:27:41.335676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### plot the worst predictions\n\nfig=plt.figure(figsize=(16, 8))\n\nbad_preds = np.array(bad_preds)\nbad_preds = np.array(sorted(bad_preds, key = lambda x: x[3], reverse=True))\n#print(bad_preds.shape)\n\ncolumns = 4\nrows = 1\nfor i in range(1, columns*rows +1):\n    n = bad_preds[i,0]\n    #print(n)\n    img = cv2.imread(os.path.join(train_dir,n))\n    RGB_img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    lbl = bad_preds[i,2]\n    fig.add_subplot(rows, columns, i)\n    plt.imshow(RGB_img)\n    lbl2 = bad_preds[i,1]\n    plt.title(\"Label = \" + str(lbl) + \" \\nClassified:\" + str(lbl2) + \" \\nConfidence:\" + str(good_preds[i,3]))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:27:41.337119Z","iopub.status.idle":"2021-12-10T17:27:41.337907Z","shell.execute_reply.started":"2021-12-10T17:27:41.337593Z","shell.execute_reply":"2021-12-10T17:27:41.33762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explaining the predictions","metadata":{}},{"cell_type":"code","source":"train_val = train_keep.landmark_id.value_counts()\ntrain_keep_df = pd.DataFrame({'landmark_id':train_val.index, 'frequency':train_val.values})\ntrain_keep_df.reset_index(inplace=True)\n\nprint(\"Top 5 training classes with most data:\")\nfor i in range(5):\n    print(\"label:\", train_keep_df.landmark_id[i], \"has\", train_keep_df.frequency[i], \"instances in training set\" )\n\ntrain_keep_df.set_index(\"landmark_id\", inplace = True)\nprint(\"\\nTop 5 classes with the worst prediction\")\n    \nfor i in range(5):\n    label = bad_preds[i, 2]\n    #print(label)\n    label_counts = train_keep_df.loc[int(label)]\n    #print(label_counts)\n    print(\"label:\", label, \"has\", label_counts[\"frequency\"], \"instances in training set\" )\n    \n    \nprint(\"\\nTop 5 classes with the best prediction\")\nfor i in range(5):\n    label = good_preds[i, 2]\n    #print(label)\n    label_counts = train_keep_df.loc[int(label)]\n    #print(label_counts)\n    print(\"label:\", label, \"has\", label_counts[\"frequency\"], \"instances in training set\" )","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:27:41.339342Z","iopub.status.idle":"2021-12-10T17:27:41.340194Z","shell.execute_reply.started":"2021-12-10T17:27:41.339911Z","shell.execute_reply":"2021-12-10T17:27:41.339939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Usnig the trained model to predict the given test data","metadata":{}},{"cell_type":"markdown","source":"Preprocessing the test data before doing the predection","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/input/landmark-recognition-2020/sample_submission.csv\")\nsub[\"filename\"] = sub.id.str[0]+\"/\"+sub.id.str[1]+\"/\"+sub.id.str[2]+\"/\"+sub.id+\".jpg\"\n\nbest_model = load_model(\"best_model.h5\")\n\ntest_gen = ImageDataGenerator(rescale=1.0/255.0).flow_from_dataframe(\n    sub,\n    directory=\"/kaggle/input/landmark-recognition-2020/test/\",\n    x_col=\"filename\",\n    y_col=None,\n    weight_col=None,\n    target_size=(224, 224),\n    color_mode=\"rgb\",\n    classes=None,\n    class_mode=None,\n    batch_size=1,\n    shuffle=True,\n    subset=None,\n    interpolation=\"nearest\",\n    validate_filenames=False)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:27:41.341496Z","iopub.status.idle":"2021-12-10T17:27:41.341983Z","shell.execute_reply.started":"2021-12-10T17:27:41.341705Z","shell.execute_reply":"2021-12-10T17:27:41.341728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Performing the prediction","metadata":{}},{"cell_type":"code","source":"y_pred_one_hot = best_model.predict_generator(test_gen, verbose=1, steps=len(sub))\ny_pred = np.argmax(y_pred_one_hot, axis=-1)\ny_prob = np.max(y_pred_one_hot, axis=-1)\ny_uniq = np.unique(train_keep.landmark_id.values)\ny_pred = [y_uniq[Y] for Y in y_pred]","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:27:41.343712Z","iopub.status.idle":"2021-12-10T17:27:41.344223Z","shell.execute_reply.started":"2021-12-10T17:27:41.343958Z","shell.execute_reply":"2021-12-10T17:27:41.343992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Submitting the prediction results","metadata":{}},{"cell_type":"code","source":"for i in range(len(sub)):\n    sub.loc[i, \"landmarks\"] = str(y_pred[i])+\" \"+str(y_prob[i])\nsub = sub.drop(columns=\"filename\")\nsub.to_csv(\"submission.csv\", index=False)\n","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:27:41.345434Z","iopub.status.idle":"2021-12-10T17:27:41.345918Z","shell.execute_reply.started":"2021-12-10T17:27:41.345642Z","shell.execute_reply":"2021-12-10T17:27:41.345666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing best and worst classifications from the test data","metadata":{}},{"cell_type":"code","source":"df = pd.DataFrame()\nfor i in range(len(sub)):    \n    df.loc[i, \"id\"] = sub.loc[i,\"id\"]\n    df.loc[i, \"class\"]=str(y_pred[i])\n    df.loc[i,\"prob\"]=y_prob[i]\ndf.sort_values(by=['prob'], inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:27:41.347327Z","iopub.status.idle":"2021-12-10T17:27:41.347782Z","shell.execute_reply.started":"2021-12-10T17:27:41.347536Z","shell.execute_reply":"2021-12-10T17:27:41.34756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Showing the 4 classifications with high confidence**","metadata":{}},{"cell_type":"code","source":"good_predections= df.tail(4).reset_index()\nfig=plt.figure(figsize=(16, 8))\ncolumns = 4\nrows = 1\n\nfor i in range(0, columns*rows ):\n    image_path = str(test_dir+\"/\"+good_predections[\"id\"][i][0]+\"/\"+good_predections[\"id\"][i][1]+\"/\"+good_predections[\"id\"][i][2]+\"/\"+good_predections[\"id\"][i]+\".jpg\")\n    fig.add_subplot(rows, columns, i+1)\n    image = cv2.imread(image_path)\n    RGB_img = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(RGB_img)\n    plt.title(\" \\nClassified as:\" + str(good_predections[\"class\"][i]) + \" \\nConfidence:\" + str(good_predections[\"prob\"][i]))\n\n","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:27:41.349094Z","iopub.status.idle":"2021-12-10T17:27:41.349557Z","shell.execute_reply.started":"2021-12-10T17:27:41.349307Z","shell.execute_reply":"2021-12-10T17:27:41.34933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Showing the 4 classifications with low confidence**","metadata":{}},{"cell_type":"code","source":"bad_predections= df.head(4).reset_index()\nfig=plt.figure(figsize=(16, 8))\ncolumns = 4\nrows = 1\n\nfor i in range(0, columns*rows ):\n    image_path = str(test_dir+\"/\"+bad_predections[\"id\"][i][0]+\"/\"+bad_predections[\"id\"][i][1]+\"/\"+bad_predections[\"id\"][i][2]+\"/\"+bad_predections[\"id\"][i]+\".jpg\")\n    fig.add_subplot(rows, columns, i+1)\n    image = cv2.imread(image_path,cv2.COLOR_BGR2RGB)\n    RGB_img = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(RGB_img)\n    plt.title(\" \\nClassified as:\" + str(bad_predections[\"class\"][i]) + \" \\nConfidence:\" + str(bad_predections[\"prob\"][i]))\n\n    ","metadata":{"execution":{"iopub.status.busy":"2021-12-10T17:27:41.351055Z","iopub.status.idle":"2021-12-10T17:27:41.351514Z","shell.execute_reply.started":"2021-12-10T17:27:41.351268Z","shell.execute_reply":"2021-12-10T17:27:41.351291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}