{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport cv2\nimport matplotlib.pyplot as plt #for plotting the graphs or images\nimport seaborn as sns\nimport plotly.offline as py\nimport plotly.express as px\nimport plotly.graph_objs as go\nimport plotly.tools as tls#visualization\nimport plotly.figure_factory as ff#visualization\nimport matplotlib.image as mpimg\nfrom PIL import Image\nfrom keras.models import Sequential\nfrom keras.preprocessing.image import ImageDataGenerator\nimport keras.layers as L\nfrom keras import regularizers, optimizers\nfrom collections import Counter\nimport keras\nfrom keras import Model\nimport tensorflow as tf\nfrom tensorflow.keras.applications.xception import Xception\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nfrom keras.models import load_model\nfrom random import randrange","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:37:17.458647Z","iopub.execute_input":"2022-08-06T13:37:17.460115Z","iopub.status.idle":"2022-08-06T13:37:17.471324Z","shell.execute_reply.started":"2022-08-06T13:37:17.460048Z","shell.execute_reply":"2022-08-06T13:37:17.46918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data= pd.read_csv(\"../input/landmark-recognition-2020/train.csv\")\ntrain_data.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:32:32.769093Z","iopub.execute_input":"2022-08-06T13:32:32.769685Z","iopub.status.idle":"2022-08-06T13:32:34.796959Z","shell.execute_reply.started":"2022-08-06T13:32:32.76965Z","shell.execute_reply":"2022-08-06T13:32:34.795388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:32:53.32381Z","iopub.execute_input":"2022-08-06T13:32:53.32485Z","iopub.status.idle":"2022-08-06T13:32:53.434048Z","shell.execute_reply.started":"2022-08-06T13:32:53.324784Z","shell.execute_reply":"2022-08-06T13:32:53.432898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:32:58.647251Z","iopub.execute_input":"2022-08-06T13:32:58.647705Z","iopub.status.idle":"2022-08-06T13:32:58.733073Z","shell.execute_reply.started":"2022-08-06T13:32:58.647669Z","shell.execute_reply":"2022-08-06T13:32:58.732092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['landmark_id'].value_counts()\n#There are 81,313 unique landmarks","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:33:00.973261Z","iopub.execute_input":"2022-08-06T13:33:00.974635Z","iopub.status.idle":"2022-08-06T13:33:01.044884Z","shell.execute_reply.started":"2022-08-06T13:33:00.974574Z","shell.execute_reply":"2022-08-06T13:33:01.043158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_data)\n#There are 1,580,470 pictures in the dataset","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:33:07.269525Z","iopub.execute_input":"2022-08-06T13:33:07.269961Z","iopub.status.idle":"2022-08-06T13:33:07.278922Z","shell.execute_reply.started":"2022-08-06T13:33:07.269924Z","shell.execute_reply":"2022-08-06T13:33:07.277465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize = (12,8))\n\ncount = train_data.landmark_id.value_counts().sort_values(ascending=False)[:10]\n\nsns.countplot(x=train_data.landmark_id,\n             order = train_data.landmark_id.value_counts().sort_values(ascending=False).iloc[:10].index)\n\nplt.xlabel(\"LandMark Id\")\nplt.ylabel(\"Frequency\")\nplt.title(\"Top 10 Classes in the Dataset\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:33:58.214898Z","iopub.execute_input":"2022-08-06T13:33:58.215328Z","iopub.status.idle":"2022-08-06T13:33:58.852539Z","shell.execute_reply.started":"2022-08-06T13:33:58.215287Z","shell.execute_reply":"2022-08-06T13:33:58.851162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig= plt.figure(figsize=(20,10))\nindex= '../input/landmark-recognition-2020/train/0/f/f/0fffe5f3c73bb4d0.jpg'\na= fig.add_subplot(2,3,1)\na.set_title(index.split(\"/\")[-1])\nplt.imshow(plt.imread(index))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:34:24.556212Z","iopub.execute_input":"2022-08-06T13:34:24.556783Z","iopub.status.idle":"2022-08-06T13:34:24.880303Z","shell.execute_reply.started":"2022-08-06T13:34:24.556726Z","shell.execute_reply":"2022-08-06T13:34:24.878354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counts = train_data.landmark_id.value_counts().sort_values(ascending=False)\nbelow = counts[counts < 15].index.shape[0] #Subject to change\nbelow","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:34:32.24529Z","iopub.execute_input":"2022-08-06T13:34:32.245822Z","iopub.status.idle":"2022-08-06T13:34:32.296268Z","shell.execute_reply.started":"2022-08-06T13:34:32.245776Z","shell.execute_reply":"2022-08-06T13:34:32.295134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Removing landmark_ids that have lower than 15 counts\n#No noise\nselected_classes = counts[counts >= 15].index #Subject to change\nlabel = train_data.loc[train_data.landmark_id.isin(selected_classes)]\nprint(label.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:34:39.551346Z","iopub.execute_input":"2022-08-06T13:34:39.55175Z","iopub.status.idle":"2022-08-06T13:34:39.62633Z","shell.execute_reply.started":"2022-08-06T13:34:39.55172Z","shell.execute_reply":"2022-08-06T13:34:39.625107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['landmark_id'] = train_data.landmark_id.astype(str)\ntrain_data['id'] = train_data.id.str[0] + '/' + train_data.id.str[1] + '/' + train_data.id.str[2]+'/' + train_data.id + '.jpg'","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:34:44.28799Z","iopub.execute_input":"2022-08-06T13:34:44.288533Z","iopub.status.idle":"2022-08-06T13:34:49.223823Z","shell.execute_reply.started":"2022-08-06T13:34:44.288381Z","shell.execute_reply":"2022-08-06T13:34:49.222781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def my_model(input_shape, num_classes, dropout, learning_rate = 0.0002):\n\n    base_model = Xception(input_shape=input_shape,weights='imagenet', include_top=False)\n    #Imagenet dataset is used for labels\n    \n    x = base_model.output\n    x = L.Dropout(dropout)(x)\n    x = L.SeparableConv2D(256, kernel_size=(3, 3), activation='relu',kernel_initializer = tf.keras.initializers.he_uniform(seed=1))(x)\n    x = L.BatchNormalization()(x)\n    x = L.SeparableConv2D(128, kernel_size=(3, 3), activation='relu',kernel_initializer = tf.keras.initializers.he_uniform(seed=3))(x)\n    x = L.BatchNormalization()(x)\n    x = L.SeparableConv2D(num_classes,kernel_size = (1,1), depth_multiplier=1, activation = 'relu',\n                kernel_initializer = tf.keras.initializers.he_uniform(seed=0),\n                kernel_regularizer=tf.keras.regularizers.l1_l2(l1=0.1, l2=0.01)\n                )(x)\n    x = L.GlobalMaxPooling2D()(x)\n    x = L.BatchNormalization()(x)\n    x = L.Flatten()(x)\n\n    pred = L.Dense(num_classes, activation = 'softmax')(x)\n    \n    for layer in base_model.layers:\n        layer.trainable = False\n\n    model = Model(inputs = base_model.input,outputs = pred,name='model')\n\n    model.compile(loss='categorical_crossentropy',experimental_steps_per_execution=8, optimizer = tf.keras.optimizers.Adagrad(learning_rate=0.01), metrics='categorical_accuracy')\n\n    model.summary()\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:35:07.142554Z","iopub.execute_input":"2022-08-06T13:35:07.142977Z","iopub.status.idle":"2022-08-06T13:35:07.156103Z","shell.execute_reply.started":"2022-08-06T13:35:07.142944Z","shell.execute_reply":"2022-08-06T13:35:07.154965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications.xception import Xception\nmodel = my_model(input_shape = (img_width, img_height, 3), num_classes = num_classes, dropout = 0.5)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:38:04.409707Z","iopub.execute_input":"2022-08-06T13:38:04.410266Z","iopub.status.idle":"2022-08-06T13:38:11.94881Z","shell.execute_reply.started":"2022-08-06T13:38:04.410202Z","shell.execute_reply":"2022-08-06T13:38:11.947235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_split = 0.25\nbatch_size = 32\nimg_width = img_height = 192","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:37:31.295989Z","iopub.execute_input":"2022-08-06T13:37:31.296399Z","iopub.status.idle":"2022-08-06T13:37:31.303425Z","shell.execute_reply.started":"2022-08-06T13:37:31.296367Z","shell.execute_reply":"2022-08-06T13:37:31.301663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\ncount = Counter(label.landmark_id.values)\nnum_classes = len(count)\nnum_classes","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:37:35.222395Z","iopub.execute_input":"2022-08-06T13:37:35.222846Z","iopub.status.idle":"2022-08-06T13:37:35.436137Z","shell.execute_reply.started":"2022-08-06T13:37:35.222813Z","shell.execute_reply":"2022-08-06T13:37:35.434547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 30 # Defining epochs for the model\ntrain_samples  = int(len(label)*(1-val_split))//batch_size\nvalidation_samples  = int(len(label)*val_split)//batch_size\n\nprint(train_samples)\nprint(validation_samples)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:38:34.97184Z","iopub.execute_input":"2022-08-06T13:38:34.972421Z","iopub.status.idle":"2022-08-06T13:38:34.981562Z","shell.execute_reply.started":"2022-08-06T13:38:34.972381Z","shell.execute_reply":"2022-08-06T13:38:34.980147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define call backs and metrics:\n\ncheckpointer = ModelCheckpoint('basic_cnn.h5', monitor='val_loss', verbose=1, save_best_only=True)\n\n# Early stopping\nearly_stopping = EarlyStopping(monitor='val_loss', verbose=1, patience=10)\n\nMETRICS = [\n    keras.metrics.Accuracy(name= \"accuracy\"),\n    keras.metrics.Precision(name = \"precision\"),\n    keras.metrics.Recall(name = 'recall'),\n    keras.metrics.AUC(name = 'auc'),\n]","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:39:40.524689Z","iopub.execute_input":"2022-08-06T13:39:40.525115Z","iopub.status.idle":"2022-08-06T13:39:40.552911Z","shell.execute_reply.started":"2022-08-06T13:39:40.525081Z","shell.execute_reply":"2022-08-06T13:39:40.55101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in model.layers:\n    layer.trainable = True\n    \nmodel.compile(loss='categorical_crossentropy', experimental_steps_per_execution=8, optimizer = tf.keras.optimizers.Adam(learning_rate=0.0001), metrics='categorical_accuracy')\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:41:54.483858Z","iopub.execute_input":"2022-08-06T13:41:54.484559Z","iopub.status.idle":"2022-08-06T13:41:54.554078Z","shell.execute_reply.started":"2022-08-06T13:41:54.484481Z","shell.execute_reply":"2022-08-06T13:41:54.552817Z"},"trusted":true},"execution_count":null,"outputs":[]}]}