{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"****","metadata":{}},{"cell_type":"markdown","source":"# **Google Landmark Recognition🗽 🗼**\n**This notebook contains the analyzing and cleaning process of the dataset and at the end the training**","metadata":{}},{"cell_type":"markdown","source":"#### Import and download libraries","metadata":{"execution":{"iopub.status.busy":"2022-04-06T10:38:32.966568Z","iopub.execute_input":"2022-04-06T10:38:32.96733Z","iopub.status.idle":"2022-04-06T10:38:32.971048Z","shell.execute_reply.started":"2022-04-06T10:38:32.967295Z","shell.execute_reply":"2022-04-06T10:38:32.970391Z"}}},{"cell_type":"code","source":"import os\nimport random\nimport seaborn as sns\nimport cv2\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport PIL\nimport IPython.display as ipd\nimport glob\nimport h5py\nimport plotly.graph_objs as go\nimport plotly.express as px\nfrom PIL import Image\nfrom tempfile import mktemp\nfrom bokeh.plotting import figure, output_notebook, show\nfrom math import pi\n\nfrom tqdm import tqdm\nfrom tqdm.notebook import tqdm_notebook\ntqdm_notebook.pandas()\n\noutput_notebook()\n\nfrom IPython.display import Image, display\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-15T14:15:58.834033Z","iopub.execute_input":"2022-04-15T14:15:58.835866Z","iopub.status.idle":"2022-04-15T14:16:02.379764Z","shell.execute_reply.started":"2022-04-15T14:15:58.835787Z","shell.execute_reply":"2022-04-15T14:16:02.378832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Load dataset","metadata":{}},{"cell_type":"code","source":"DATASET_DIR = '../input/landmark-recognition-2021'\n\nTRAIN_IMAGE_DIR = f'{DATASET_DIR}/train'\nTEST_IMAGE_DIR = f'{DATASET_DIR}/test'\n\ntrain = pd.read_csv(f'{DATASET_DIR}/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-04-15T14:16:03.701361Z","iopub.execute_input":"2022-04-15T14:16:03.701710Z","iopub.status.idle":"2022-04-15T14:16:05.589264Z","shell.execute_reply.started":"2022-04-15T14:16:03.701677Z","shell.execute_reply":"2022-04-15T14:16:05.588311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Explore training dataset","metadata":{}},{"cell_type":"code","source":"print(train.head())\nprint(\"Training data shape :\", train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T14:16:06.963435Z","iopub.execute_input":"2022-04-15T14:16:06.963780Z","iopub.status.idle":"2022-04-15T14:16:06.978803Z","shell.execute_reply.started":"2022-04-15T14:16:06.963748Z","shell.execute_reply":"2022-04-15T14:16:06.977844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-04-15T14:16:09.880930Z","iopub.execute_input":"2022-04-15T14:16:09.881265Z","iopub.status.idle":"2022-04-15T14:16:10.069791Z","shell.execute_reply.started":"2022-04-15T14:16:09.881227Z","shell.execute_reply":"2022-04-15T14:16:10.068573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"value_counts = train['landmark_id'].value_counts() # normalize=True returns relative frequency\n\nfreq_df = pd.DataFrame(value_counts)\nfreq_df.reset_index(inplace=True)\nfreq_df.columns = ['landmark_id','frequency']\nfreq_df","metadata":{"execution":{"iopub.status.busy":"2022-04-15T14:16:13.382396Z","iopub.execute_input":"2022-04-15T14:16:13.382777Z","iopub.status.idle":"2022-04-15T14:16:13.449986Z","shell.execute_reply.started":"2022-04-15T14:16:13.382736Z","shell.execute_reply":"2022-04-15T14:16:13.449359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Prepare dataset\nThere is a total of **81313** different classes for landmarks. Because this is a great amount we were planning to only take an **x percentage** of each class and delete classes with less than x frequency. This plan wont work out because more than 41k classes have less than 10 images and there are only 7 classes with more than 1000 images. For a good model you need at least 1000 images per class. ","metadata":{}},{"cell_type":"code","source":"freq_df[freq_df['frequency'] < 10]","metadata":{"execution":{"iopub.status.busy":"2022-04-15T14:16:56.543350Z","iopub.execute_input":"2022-04-15T14:16:56.544303Z","iopub.status.idle":"2022-04-15T14:16:56.563712Z","shell.execute_reply.started":"2022-04-15T14:16:56.544244Z","shell.execute_reply":"2022-04-15T14:16:56.562981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# freq_df = train[~train.isin(freq_df)].dropna(how ='all')","metadata":{"execution":{"iopub.status.busy":"2022-04-08T12:26:38.364607Z","iopub.execute_input":"2022-04-08T12:26:38.365337Z","iopub.status.idle":"2022-04-08T12:26:38.369768Z","shell.execute_reply.started":"2022-04-08T12:26:38.3653Z","shell.execute_reply":"2022-04-08T12:26:38.368799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"value_counts.index[:10].tolist()","metadata":{"execution":{"iopub.status.busy":"2022-04-15T14:17:02.688969Z","iopub.execute_input":"2022-04-15T14:17:02.690238Z","iopub.status.idle":"2022-04-15T14:17:02.697281Z","shell.execute_reply.started":"2022-04-15T14:17:02.690174Z","shell.execute_reply":"2022-04-15T14:17:02.696351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Create a new column with jpg url","metadata":{}},{"cell_type":"code","source":"def jpgurl(df, dir_name='../input/landmark-recognition-2021/train/'):\n    \"\"\"This function will create a url based on the first 3 values of ID\"\"\"\n    for row in range(len(df.index)):\n        df.at[row, 'url'] = os.path.join(dir_name, df['id'][row][0], df['id'][row][1], df['id'][row][2], df['id'][row] + \".\" + 'jpg')\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-04-15T14:17:11.274119Z","iopub.execute_input":"2022-04-15T14:17:11.274829Z","iopub.status.idle":"2022-04-15T14:17:11.282648Z","shell.execute_reply.started":"2022-04-15T14:17:11.274790Z","shell.execute_reply":"2022-04-15T14:17:11.281720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = jpgurl(train)\nsample.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-15T14:17:12.957166Z","iopub.execute_input":"2022-04-15T14:17:12.957515Z","iopub.status.idle":"2022-04-15T14:18:45.223630Z","shell.execute_reply.started":"2022-04-15T14:17:12.957480Z","shell.execute_reply":"2022-04-15T14:18:45.222659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Non-landmarks should be deleted from the dataset but that's not within our scope","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(30,30))\ncolumns = 10\nrows = 10\nfor i in range(1, columns*rows +1):\n    img = PIL.Image.open(sample['url'][i], mode='r')\n    fig.add_subplot(rows, columns, i)\n    plt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-15T14:18:45.225187Z","iopub.execute_input":"2022-04-15T14:18:45.225432Z","iopub.status.idle":"2022-04-15T14:19:02.265578Z","shell.execute_reply.started":"2022-04-15T14:18:45.225404Z","shell.execute_reply":"2022-04-15T14:19:02.264544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The dataframe we use to train:** sample","metadata":{}},{"cell_type":"markdown","source":"## **Training** ","metadata":{}},{"cell_type":"code","source":"!pip install ../input/keras-efficientnet-whl/Keras_Applications-1.0.8-py3-none-any.whl\n!pip install ../input/keras-efficientnet-whl/efficientnet-1.1.1-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2022-04-15T14:49:02.725660Z","iopub.execute_input":"2022-04-15T14:49:02.727119Z","iopub.status.idle":"2022-04-15T14:49:21.869444Z","shell.execute_reply.started":"2022-04-15T14:49:02.727040Z","shell.execute_reply":"2022-04-15T14:49:21.868510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport efficientnet.keras as efn\nimport tensorflow.keras.layers as L\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.preprocessing import image\nfrom random import shuffle\nfrom sklearn.model_selection import train_test_split\nimport math","metadata":{"execution":{"iopub.status.busy":"2022-04-15T14:55:35.235681Z","iopub.execute_input":"2022-04-15T14:55:35.236098Z","iopub.status.idle":"2022-04-15T14:55:38.364749Z","shell.execute_reply.started":"2022-04-15T14:55:35.236061Z","shell.execute_reply":"2022-04-15T14:55:38.363836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataGenerator(Sequence):\n    def __init__(self, path, list_IDs, data, img_size, img_channel, batch_size):\n        self.path = path\n        self.list_IDs = list_IDs\n        self.data = data\n        self.img_size = img_size\n        self.img_channel = img_channel\n        self.batch_size = batch_size\n        self.indexes = np.arange(len(self.list_IDs))\n        \n    def __len__(self):\n        len_ = int(len(self.list_IDs)/self.batch_size)\n        if len_*self.batch_size < len(self.list_IDs):\n            len_ += 1\n        return len_\n    \n    def __getitem__(self, index):\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n        X, y = self.__data_generation(list_IDs_temp)\n        return X, y\n            \n    \n    def __data_generation(self, list_IDs_temp):\n        X = np.zeros((self.batch_size, self.img_size, self.img_size, self.img_channel))\n        y = np.zeros((self.batch_size, 1), dtype=int)\n        for i, ID in enumerate(list_IDs_temp):\n            \n            image_id = self.data.loc[ID, 'id']\n            file = image_id+'.jpg'\n            subpath = '/'.join([char for char in image_id[0:3]]) \n            img = cv2.imread(self.path+subpath+'/'+file)\n            img = img/255\n            img = cv2.resize(img, (self.img_size, self.img_size))\n            X[i, ] = img\n            if self.path.find('train')>=0:\n                y[i, ] = self.data.loc[ID, 'landmark_id']\n            else:\n                y[i, ] = 0\n        return X, y\n    \nimg_size = 256\nimg_channel = 3\n\nbatch_size = 1\nsub = pd.read_csv('../input/landmark-recognition-2021/sample_submission.csv')\nlist_IDs_test = list(sub.index)\n\ntest_generator = DataGenerator('../input/landmark-recognition-2021/'+'test/', list_IDs_test, sub, img_size, img_channel, batch_size)","metadata":{"execution":{"iopub.status.busy":"2022-04-15T14:56:20.652366Z","iopub.execute_input":"2022-04-15T14:56:20.652683Z","iopub.status.idle":"2022-04-15T14:56:20.685120Z","shell.execute_reply.started":"2022-04-15T14:56:20.652654Z","shell.execute_reply":"2022-04-15T14:56:20.684144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.models.load_model('../input/effnetb0/efficientnetb0_notop.h5')","metadata":{"execution":{"iopub.status.busy":"2022-04-15T15:19:37.359037Z","iopub.execute_input":"2022-04-15T15:19:37.361138Z","iopub.status.idle":"2022-04-15T15:19:37.432676Z","shell.execute_reply.started":"2022-04-15T15:19:37.361068Z","shell.execute_reply":"2022-04-15T15:19:37.431126Z"},"trusted":true},"execution_count":null,"outputs":[]}]}