{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Intro\nWelcome to the [Google Landmark Recognition 2021](https://www.kaggle.com/c/landmark-recognition-2021) compedition\n![](https://storage.googleapis.com/kaggle-competitions/kaggle/29762/logos/header.png)\n\nThis notebook will give you a guideline to start step by step with this compedition. We focus on:\n* the underlying structure of the data,\n* a data generator to load the image data on demand during the prediction process.\n\nWe use a simple model with a pretrained model on a subset of the train data to clarify the workflow. Additionally we recommend to use the power of GPU.\n\n\n<span style=\"color: royalblue;\">Please vote the notebook up if it helps you. Feel free to leave a comment above the notebook. Thank you. </span>","metadata":{}},{"cell_type":"markdown","source":"# Libraries\nWe use some standard python packages and the libraries of scikit learn and keras. ","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\n\nfrom sklearn.model_selection import train_test_split\n\nfrom keras.utils import to_categorical, Sequence\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten\nfrom keras.optimizers import RMSprop,Adam\nfrom keras.applications import VGG19, VGG16, ResNet50\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:50:05.113597Z","iopub.execute_input":"2021-08-14T07:50:05.113926Z","iopub.status.idle":"2021-08-14T07:50:05.121162Z","shell.execute_reply.started":"2021-08-14T07:50:05.113898Z","shell.execute_reply":"2021-08-14T07:50:05.119877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Path","metadata":{}},{"cell_type":"code","source":"path = '/kaggle/input/landmark-recognition-2021/'\nos.listdir(path)","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:50:07.20898Z","iopub.execute_input":"2021-08-14T07:50:07.209311Z","iopub.status.idle":"2021-08-14T07:50:07.219628Z","shell.execute_reply.started":"2021-08-14T07:50:07.209284Z","shell.execute_reply":"2021-08-14T07:50:07.218434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(path+'train.csv')\nsamp_subm = pd.read_csv(path+'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:50:09.436513Z","iopub.execute_input":"2021-08-14T07:50:09.436895Z","iopub.status.idle":"2021-08-14T07:50:11.014966Z","shell.execute_reply.started":"2021-08-14T07:50:09.436864Z","shell.execute_reply":"2021-08-14T07:50:11.014041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Functions","metadata":{}},{"cell_type":"code","source":"def plot_examples(landmark_id=1):\n    \"\"\" Plot 5 examples of images with the same landmark_id \"\"\"\n    \n    fig, axs = plt.subplots(1, 5, figsize=(25, 12))\n    fig.subplots_adjust(hspace = .2, wspace=.2)\n    axs = axs.ravel()\n    for i in range(5):\n        idx = train_data[train_data['landmark_id']==landmark_id].index[i]\n        image_id = train_data.loc[idx, 'id']\n        file = image_id+'.jpg'\n        subpath = '/'.join([char for char in image_id[0:3]])\n        img = cv2.imread(path+'train/'+subpath+'/'+file)\n        axs[i].imshow(img)\n        axs[i].set_title('landmark_id: '+str(landmark_id))\n        axs[i].set_xticklabels([])\n        axs[i].set_yticklabels([])","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:50:12.435012Z","iopub.execute_input":"2021-08-14T07:50:12.435349Z","iopub.status.idle":"2021-08-14T07:50:12.444051Z","shell.execute_reply.started":"2021-08-14T07:50:12.435321Z","shell.execute_reply":"2021-08-14T07:50:12.443088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Overview\nFirst we look on the size of the dataset:","metadata":{}},{"cell_type":"code","source":"print('Samples train:', len(train_data))\nprint('Samples test:', len(samp_subm))","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:50:15.229036Z","iopub.execute_input":"2021-08-14T07:50:15.229399Z","iopub.status.idle":"2021-08-14T07:50:15.235809Z","shell.execute_reply.started":"2021-08-14T07:50:15.229369Z","shell.execute_reply":"2021-08-14T07:50:15.234638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:50:51.779908Z","iopub.execute_input":"2021-08-14T07:50:51.7803Z","iopub.status.idle":"2021-08-14T07:50:51.789924Z","shell.execute_reply.started":"2021-08-14T07:50:51.780271Z","shell.execute_reply":"2021-08-14T07:50:51.789056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 81313 unique classes:","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:51:34.633814Z","iopub.execute_input":"2021-08-14T07:51:34.634214Z"}}},{"cell_type":"code","source":"len(train_data['landmark_id'].unique())","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:51:19.661403Z","iopub.execute_input":"2021-08-14T07:51:19.661743Z","iopub.status.idle":"2021-08-14T07:51:19.685459Z","shell.execute_reply.started":"2021-08-14T07:51:19.661714Z","shell.execute_reply":"2021-08-14T07:51:19.684517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For each test image, we have to predict one landmark label and a corresponding confidence score. ","metadata":{}},{"cell_type":"code","source":"samp_subm.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:50:42.426273Z","iopub.execute_input":"2021-08-14T07:50:42.426631Z","iopub.status.idle":"2021-08-14T07:50:42.435837Z","shell.execute_reply.started":"2021-08-14T07:50:42.426601Z","shell.execute_reply":"2021-08-14T07:50:42.434758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Find Image\nWe consider the first image of the train data set and plot it. The first 3 characters ares used for the subpath which is the location of the image. ","metadata":{}},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:52:57.200579Z","iopub.execute_input":"2021-08-14T07:52:57.200906Z","iopub.status.idle":"2021-08-14T07:52:57.210524Z","shell.execute_reply.started":"2021-08-14T07:52:57.200878Z","shell.execute_reply":"2021-08-14T07:52:57.209554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = train_data.loc[0, 'id']\nfile = image_id+'.jpg'\nsubpath = '/'.join([char for char in image_id[0:3]]) ","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:52:58.394221Z","iopub.execute_input":"2021-08-14T07:52:58.394535Z","iopub.status.idle":"2021-08-14T07:52:58.409972Z","shell.execute_reply.started":"2021-08-14T07:52:58.394506Z","shell.execute_reply":"2021-08-14T07:52:58.40911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Is the file located in the subpath?","metadata":{}},{"cell_type":"code","source":"file in os.listdir(path+'train/'+subpath)","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:53:01.038661Z","iopub.execute_input":"2021-08-14T07:53:01.039007Z","iopub.status.idle":"2021-08-14T07:53:01.091742Z","shell.execute_reply.started":"2021-08-14T07:53:01.038977Z","shell.execute_reply":"2021-08-14T07:53:01.091014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plot the image:","metadata":{}},{"cell_type":"code","source":"img = cv2.imread(path+'train/'+subpath+'/'+file)\nplt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:53:03.462064Z","iopub.execute_input":"2021-08-14T07:53:03.46239Z","iopub.status.idle":"2021-08-14T07:53:03.673226Z","shell.execute_reply.started":"2021-08-14T07:53:03.462362Z","shell.execute_reply":"2021-08-14T07:53:03.672359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Look on the image shape:","metadata":{}},{"cell_type":"code","source":"img.shape","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:53:06.362253Z","iopub.execute_input":"2021-08-14T07:53:06.362566Z","iopub.status.idle":"2021-08-14T07:53:06.369002Z","shell.execute_reply.started":"2021-08-14T07:53:06.362538Z","shell.execute_reply":"2021-08-14T07:53:06.367974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot Some Examples\nWe plot some examples of images with the same **landmark_id** in a row.","metadata":{}},{"cell_type":"code","source":"plot_examples(landmark_id = 138982)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-14T07:53:09.381187Z","iopub.execute_input":"2021-08-14T07:53:09.381583Z","iopub.status.idle":"2021-08-14T07:53:10.597656Z","shell.execute_reply.started":"2021-08-14T07:53:09.38155Z","shell.execute_reply":"2021-08-14T07:53:10.596901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_examples(landmark_id = 126637)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-14T07:53:16.394327Z","iopub.execute_input":"2021-08-14T07:53:16.394699Z","iopub.status.idle":"2021-08-14T07:53:17.645337Z","shell.execute_reply.started":"2021-08-14T07:53:16.39467Z","shell.execute_reply":"2021-08-14T07:53:17.639966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_examples(landmark_id = 83144)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-14T07:53:21.085074Z","iopub.execute_input":"2021-08-14T07:53:21.085436Z","iopub.status.idle":"2021-08-14T07:53:22.138105Z","shell.execute_reply.started":"2021-08-14T07:53:21.085408Z","shell.execute_reply":"2021-08-14T07:53:22.136925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split Data\nWe define train, validation and test data.","metadata":{}},{"cell_type":"code","source":"list_IDs_train, list_IDs_val = train_test_split(list(train_data.index)[:500000], test_size=0.33, random_state=2021)\nlist_IDs_test = list(samp_subm.index)","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:53:24.127382Z","iopub.execute_input":"2021-08-14T07:53:24.127693Z","iopub.status.idle":"2021-08-14T07:53:24.414201Z","shell.execute_reply.started":"2021-08-14T07:53:24.127666Z","shell.execute_reply":"2021-08-14T07:53:24.413251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number train samples:', len(list_IDs_train))\nprint('Number val samples:', len(list_IDs_val))\nprint('Number test samples:', len(list_IDs_test))","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:53:26.872325Z","iopub.execute_input":"2021-08-14T07:53:26.872754Z","iopub.status.idle":"2021-08-14T07:53:26.879382Z","shell.execute_reply.started":"2021-08-14T07:53:26.872726Z","shell.execute_reply":"2021-08-14T07:53:26.878178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Generator\n\nWe use a data generator to load the data on demand.","metadata":{}},{"cell_type":"code","source":"img_size = 32\nimg_channel = 3\nbatch_size = 64\n\nnum_classes = len(train_data['landmark_id'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:53:33.208634Z","iopub.execute_input":"2021-08-14T07:53:33.208976Z","iopub.status.idle":"2021-08-14T07:53:33.264934Z","shell.execute_reply.started":"2021-08-14T07:53:33.208928Z","shell.execute_reply":"2021-08-14T07:53:33.263624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataGenerator(Sequence):\n    def __init__(self, path, list_IDs, data, img_size, img_channel, batch_size):\n        self.path = path\n        self.list_IDs = list_IDs\n        self.data = data\n        self.img_size = img_size\n        self.img_channel = img_channel\n        self.batch_size = batch_size\n        self.indexes = np.arange(len(self.list_IDs))\n        \n    def __len__(self):\n        len_ = int(len(self.list_IDs)/self.batch_size)\n        if len_*self.batch_size < len(self.list_IDs):\n            len_ += 1\n        return len_\n    \n    def __getitem__(self, index):\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n        X, y = self.__data_generation(list_IDs_temp)\n        return X, y\n            \n    \n    def __data_generation(self, list_IDs_temp):\n        X = np.zeros((self.batch_size, self.img_size, self.img_size, self.img_channel))\n        y = np.zeros((self.batch_size, 1), dtype=int)\n        for i, ID in enumerate(list_IDs_temp):\n            \n            image_id = self.data.loc[ID, 'id']\n            file = image_id+'.jpg'\n            subpath = '/'.join([char for char in image_id[0:3]]) \n            \n            img = cv2.imread(self.path+subpath+'/'+file)\n            \n            img = cv2.resize(img, (self.img_size, self.img_size))\n            X[i, ] = img/255\n            if self.path.find('train')>=0:\n                y[i, ] = self.data.loc[ID, 'landmark_id']\n            else:\n                y[i, ] = 0\n        return X, y","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:53:37.106017Z","iopub.execute_input":"2021-08-14T07:53:37.106387Z","iopub.status.idle":"2021-08-14T07:53:37.130295Z","shell.execute_reply.started":"2021-08-14T07:53:37.106357Z","shell.execute_reply":"2021-08-14T07:53:37.129422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Use the DataGenerator class to define the data generators for train, validation and test data:","metadata":{}},{"cell_type":"code","source":"train_generator = DataGenerator(path+'train/', list_IDs_train, train_data, img_size, img_channel, batch_size)\nval_generator = DataGenerator(path+'train/', list_IDs_val, train_data, img_size, img_channel, batch_size)\ntest_generator = DataGenerator(path+'test/', list_IDs_test, samp_subm, img_size, img_channel, batch_size)","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:53:41.060539Z","iopub.execute_input":"2021-08-14T07:53:41.060868Z","iopub.status.idle":"2021-08-14T07:53:41.066178Z","shell.execute_reply.started":"2021-08-14T07:53:41.060839Z","shell.execute_reply":"2021-08-14T07:53:41.065095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"markdown","source":"Load pretrained model:","metadata":{}},{"cell_type":"code","source":"weights='../input/models/resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5'\nconv_base = ResNet50(weights=weights,\n                     include_top=False,\n                     input_shape=(img_size, img_size, img_channel))\nconv_base.trainable = True","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:53:43.412408Z","iopub.execute_input":"2021-08-14T07:53:43.412728Z","iopub.status.idle":"2021-08-14T07:53:49.49946Z","shell.execute_reply.started":"2021-08-14T07:53:43.4127Z","shell.execute_reply":"2021-08-14T07:53:49.498591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define Model","metadata":{}},{"cell_type":"code","source":"model = Sequential()\nmodel.add(conv_base)\nmodel.add(Flatten())\n#model.add(Dense(64, activation='relu'))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(num_classes, activation='softmax'))\n\nmodel.compile(optimizer = Adam(lr=1e-4),\n              loss=\"sparse_categorical_crossentropy\",\n              metrics=['sparse_categorical_accuracy'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:53:51.249692Z","iopub.execute_input":"2021-08-14T07:53:51.250039Z","iopub.status.idle":"2021-08-14T07:53:51.662645Z","shell.execute_reply.started":"2021-08-14T07:53:51.250007Z","shell.execute_reply":"2021-08-14T07:53:51.661685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 1","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:53:57.102649Z","iopub.execute_input":"2021-08-14T07:53:57.102983Z","iopub.status.idle":"2021-08-14T07:53:57.108687Z","shell.execute_reply.started":"2021-08-14T07:53:57.102931Z","shell.execute_reply":"2021-08-14T07:53:57.10791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(generator=train_generator,\n                              validation_data=val_generator,\n                              epochs = epochs, workers=4)","metadata":{"execution":{"iopub.status.busy":"2021-08-14T07:54:00.18311Z","iopub.execute_input":"2021-08-14T07:54:00.18343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict Test Data","metadata":{}},{"cell_type":"code","source":"y_pred = model.predict_generator(test_generator, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-08-13T08:19:05.262193Z","iopub.status.idle":"2021-08-13T08:19:05.264469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(samp_subm.index)):\n    category = np.argmax(y_pred[i])\n    score = y_pred[i][np.argmax(y_pred[i])].round(2)\n    samp_subm.loc[i, 'landmarks'] = str(category)+' '+str(score)","metadata":{"execution":{"iopub.status.busy":"2021-08-13T08:19:05.268872Z","iopub.status.idle":"2021-08-13T08:19:05.273659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samp_subm.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-13T08:19:05.275491Z","iopub.status.idle":"2021-08-13T08:19:05.276954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Export","metadata":{}},{"cell_type":"code","source":"samp_subm.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-08-13T08:19:05.279327Z","iopub.status.idle":"2021-08-13T08:19:05.282432Z"},"trusted":true},"execution_count":null,"outputs":[]}]}