{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#imported usefull libraries\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport PIL\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Sequential","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T04:17:37.955675Z","iopub.execute_input":"2022-08-11T04:17:37.956489Z","iopub.status.idle":"2022-08-11T04:17:46.058587Z","shell.execute_reply.started":"2022-08-11T04:17:37.956447Z","shell.execute_reply":"2022-08-11T04:17:46.057105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **EDA**","metadata":{}},{"cell_type":"code","source":"import pandas as pd \ntrain_csv = pd.read_csv('/kaggle/input/landmark-recognition-2020/train.csv')\ntrain_csv.head()\n#read the main file train and displayed top 5 rows","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:17:46.061287Z","iopub.execute_input":"2022-08-11T04:17:46.062463Z","iopub.status.idle":"2022-08-11T04:17:48.184018Z","shell.execute_reply.started":"2022-08-11T04:17:46.062403Z","shell.execute_reply":"2022-08-11T04:17:48.182621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Number of unique landmarks\ntrain_csv['landmark_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:17:48.189846Z","iopub.execute_input":"2022-08-11T04:17:48.193592Z","iopub.status.idle":"2022-08-11T04:17:48.251441Z","shell.execute_reply.started":"2022-08-11T04:17:48.193546Z","shell.execute_reply":"2022-08-11T04:17:48.249746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv['landmark_id'].value_counts()\n#count of each landmark","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:17:48.259500Z","iopub.execute_input":"2022-08-11T04:17:48.262602Z","iopub.status.idle":"2022-08-11T04:17:48.344753Z","shell.execute_reply.started":"2022-08-11T04:17:48.262554Z","shell.execute_reply":"2022-08-11T04:17:48.343330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#top 10 landmarks which were most visited\ntop10 = pd.DataFrame(train_csv['landmark_id'].value_counts().head(10))\ntop10.reset_index(inplace=True)\ntop10.columns = ['landmark_id', 'value_count']\ntop10.set_index('landmark_id',inplace=True)\ntop10","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:17:48.350066Z","iopub.execute_input":"2022-08-11T04:17:48.353032Z","iopub.status.idle":"2022-08-11T04:17:48.442872Z","shell.execute_reply.started":"2022-08-11T04:17:48.352987Z","shell.execute_reply":"2022-08-11T04:17:48.441637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualization of top-10 landmarks\n!pip install pandas-bokeh\nimport pandas_bokeh\npandas_bokeh.output_notebook()\ntop10.plot_bokeh.bar(\n    ylabel=\"Count\", \n    title=\"Top-10 landmark id\", \n    alpha=0.6)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:17:53.841031Z","iopub.execute_input":"2022-08-11T04:17:53.841435Z","iopub.status.idle":"2022-08-11T04:18:08.934601Z","shell.execute_reply.started":"2022-08-11T04:17:53.841404Z","shell.execute_reply":"2022-08-11T04:18:08.933006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#below 20 landmarks -> least visited | or have least images\nleast10 = pd.DataFrame(train_csv['landmark_id'].value_counts().tail(20))\nleast10.reset_index(inplace=True)\nleast10.columns = ['landmark_id', 'value_count']\nleast10.set_index('landmark_id',inplace=True)\nleast10","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:18:18.944647Z","iopub.execute_input":"2022-08-11T04:18:18.945059Z","iopub.status.idle":"2022-08-11T04:18:19.014270Z","shell.execute_reply.started":"2022-08-11T04:18:18.945029Z","shell.execute_reply":"2022-08-11T04:18:19.011749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#least 20 landmarks\nimport pandas_bokeh\npandas_bokeh.output_notebook()\nplt.figure(figsize=(50,50))\nleast10.plot_bokeh.bar(\n    ylabel=\"Count\", \n    title=\"Top-20 landmark id\", \n    alpha=0.6)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:18:29.690035Z","iopub.execute_input":"2022-08-11T04:18:29.690662Z","iopub.status.idle":"2022-08-11T04:18:29.798556Z","shell.execute_reply.started":"2022-08-11T04:18:29.690618Z","shell.execute_reply":"2022-08-11T04:18:29.797209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv('/kaggle/input/landmark-recognition-2020/train.csv')\ntrain_csv.head(10)\n\n# put .jpg into the file name\ndef add_txt(fn):\n    return fn+'.jpg'\n\ntrain_csv['id'] = train_csv['id'].apply(add_txt)\n\n\n\n# choose those labels with more than 200 images, and choose the first 200 images of each label\n# move every training files to the same folder\n%cd /kaggle/working\nif not os.path.exists('training'):\n    os.mkdir('training')\nif not os.path.exists('validation'):\n    os.mkdir('validation')\nif not os.path.exists('testing'):\n    os.mkdir('testing')    \n\nimport shutil\nimport random\n\nlabel_list = train_csv['landmark_id'].unique()\ncnt = 0\nfinal_label_list = []\n\nfor label in list(label_list): # label order by random\n    file_list = list(train_csv['id'][train_csv['landmark_id']==label])\n    if len(file_list) >= 200:\n        final_label_list.append(label)\n        if not os.path.exists('/kaggle/working/training/'+str(label)):\n            os.mkdir('/kaggle/working/training/'+str(label))\n        if not os.path.exists('/kaggle/working/validation/'+str(label)):\n            os.mkdir('/kaggle/working/validation/'+str(label))\n        if not os.path.exists('/kaggle/working/testing/'+str(label)):\n            os.mkdir('/kaggle/working/testing/'+str(label))\n        for file in file_list[:120]:  # 120 files for training\n            src = '/kaggle/input/landmark-recognition-2020/train/'+file[0]+'/'+file[1]+'/'+file[2]+'/'+file\n            dst = '/kaggle/working/training/'+str(label)+'/'+file\n            if not os.path.exists(dst):\n                shutil.copyfile(src, dst)\n        for file in file_list[120:160]: # 40 files for validation\n            src = '/kaggle/input/landmark-recognition-2020/train/'+file[0]+'/'+file[1]+'/'+file[2]+'/'+file\n            dst = '/kaggle/working/validation/'+str(label)+'/'+file\n            if not os.path.exists(dst):\n                shutil.copyfile(src, dst)\n        for file in file_list[160:200]: # 40 files for testing\n            src = '/kaggle/input/landmark-recognition-2020/train/'+file[0]+'/'+file[1]+'/'+file[2]+'/'+file\n            dst = '/kaggle/working/testing/'+str(label)+'/'+file\n            if not os.path.exists(dst):\n                shutil.copyfile(src, dst)\n        cnt += 1\n    if cnt == 100: # only need 100 labels\n        break","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:18:45.650586Z","iopub.execute_input":"2022-08-11T04:18:45.651011Z","iopub.status.idle":"2022-08-11T04:22:46.299449Z","shell.execute_reply.started":"2022-08-11T04:18:45.650982Z","shell.execute_reply":"2022-08-11T04:22:46.297979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n#Our original imagess are in RGB coefficients ranging from 0 to 255,\n#but such values would be too high for our model to process, \n#therefore we scale with a 1/255 to target values between 0 and 1\n\ntrain_datagen = ImageDataGenerator(rescale=1./255)\n\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_dir = '/kaggle/working/training'\nvalidation_dir = '/kaggle/working/validation'\ntest_dir = '/kaggle/working/testing'\n\n# The flow_from_directory() method allows to read the images directly from the directory train and valid files and augment them \ntrain_generator = train_datagen.flow_from_directory(\n    train_dir,\n    target_size=(200,200),\n    batch_size = 32,\n    class_mode='categorical',\n    seed=42)\n\nvalidation_generator = test_datagen.flow_from_directory(\n    validation_dir,\n    target_size=(200,200),\n    batch_size = 32,\n    class_mode='categorical',\n    seed=123)\n\ntest_generator = test_datagen.flow_from_directory(\n    test_dir,\n    target_size=(200,200),\n    batch_size = 1,\n    class_mode='categorical',\n    seed=123)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:22:46.303948Z","iopub.execute_input":"2022-08-11T04:22:46.304304Z","iopub.status.idle":"2022-08-11T04:22:47.430563Z","shell.execute_reply.started":"2022-08-11T04:22:46.304273Z","shell.execute_reply":"2022-08-11T04:22:47.429235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **EfficientNet**","metadata":{}},{"cell_type":"code","source":"#This neural network uses  compound scaling that addsmore layers, more nodes per layer, or better resolution for filters \n# whichcan improve a model. \n# Compound scaling uniformly scales all three of these dimensions with a compound coefficient\n\na = tf.keras.applications.EfficientNetB2(include_top=False,\n                    weights=\"imagenet\",\n                    input_shape=(200,200, 3))\na.trainable = True\na.summary()\n#getting an error while using input_shape same as target size","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:22:47.434105Z","iopub.execute_input":"2022-08-11T04:22:47.435028Z","iopub.status.idle":"2022-08-11T04:22:55.859847Z","shell.execute_reply.started":"2022-08-11T04:22:47.434982Z","shell.execute_reply":"2022-08-11T04:22:55.857910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import Dense, Dropout, MaxPooling2D, GlobalAveragePooling2D, Flatten, Conv2D, Input\nfrom tensorflow.keras import optimizers\nimport tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:22:55.863167Z","iopub.execute_input":"2022-08-11T04:22:55.863947Z","iopub.status.idle":"2022-08-11T04:22:55.873668Z","shell.execute_reply.started":"2022-08-11T04:22:55.863900Z","shell.execute_reply":"2022-08-11T04:22:55.871778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential() # sequence of layer of neural network\nmodel.add(a)\nmodel.add(GlobalAveragePooling2D())\nmodel.add(Dense(100, activation='relu')) #adding a layer of neurons #rectified linear unit it oly pass value 0 or greater than 0 to the next layer in the network","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:22:55.876051Z","iopub.execute_input":"2022-08-11T04:22:55.876688Z","iopub.status.idle":"2022-08-11T04:22:57.310246Z","shell.execute_reply.started":"2022-08-11T04:22:55.876646Z","shell.execute_reply":"2022-08-11T04:22:57.308725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='Adam',\n              loss = 'categorical_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:22:57.312766Z","iopub.execute_input":"2022-08-11T04:22:57.313612Z","iopub.status.idle":"2022-08-11T04:22:57.337529Z","shell.execute_reply.started":"2022-08-11T04:22:57.313569Z","shell.execute_reply":"2022-08-11T04:22:57.336284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:22:57.342569Z","iopub.execute_input":"2022-08-11T04:22:57.342940Z","iopub.status.idle":"2022-08-11T04:22:57.445248Z","shell.execute_reply.started":"2022-08-11T04:22:57.342912Z","shell.execute_reply":"2022-08-11T04:22:57.441314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    epochs=10,\n    validation_data=validation_generator, verbose =1\n)\n#it tells us about accuracy of the model for epochs 10. this tell how much neural network is accurate","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:22:57.447221Z","iopub.execute_input":"2022-08-11T04:22:57.448143Z","iopub.status.idle":"2022-08-11T04:51:07.315190Z","shell.execute_reply.started":"2022-08-11T04:22:57.448093Z","shell.execute_reply":"2022-08-11T04:51:07.313882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.evaluate(test_generator)\nprint(scores)\n#it test the accuracy on the test data it make prediction on the test data accuracy","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:51:07.317296Z","iopub.execute_input":"2022-08-11T04:51:07.318142Z","iopub.status.idle":"2022-08-11T04:52:46.931422Z","shell.execute_reply.started":"2022-08-11T04:51:07.318100Z","shell.execute_reply":"2022-08-11T04:52:46.929862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = range(1, len(history.history['accuracy'])+1)\n\nplt.plot(epochs, history.history['accuracy'], '#21466C', label='Train accuracy')\nplt.plot(epochs, history.history['val_accuracy'], '#cc1123', label='Validation accuracy')\nplt.title('Training and validation accuracy')\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:52:46.936320Z","iopub.execute_input":"2022-08-11T04:52:46.937471Z","iopub.status.idle":"2022-08-11T04:52:47.202355Z","shell.execute_reply.started":"2022-08-11T04:52:46.937442Z","shell.execute_reply":"2022-08-11T04:52:47.201086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **RESNET**","metadata":{}},{"cell_type":"code","source":"#ResNet-50  reduced the vanishing gradient problem while also not adding additional parameters to the model.\n\nb = tf.keras.applications.ResNet50V2(include_top=False,\n                    weights=\"imagenet\",\n                    input_shape=(200,200, 3))\nb.trainable = True\nb.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:52:56.815631Z","iopub.execute_input":"2022-08-11T04:52:56.816093Z","iopub.status.idle":"2022-08-11T04:52:59.602462Z","shell.execute_reply.started":"2022-08-11T04:52:56.816062Z","shell.execute_reply":"2022-08-11T04:52:59.600995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model2 = Sequential()\nmodel2.add(b)\nmodel2.add(GlobalAveragePooling2D())\n#model2.add(Flatten())\nmodel2.add(Dense(100,activation='sigmoid'))\n#model2.add(Dense(10))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:52:59.605377Z","iopub.execute_input":"2022-08-11T04:52:59.606381Z","iopub.status.idle":"2022-08-11T04:53:00.093729Z","shell.execute_reply.started":"2022-08-11T04:52:59.606338Z","shell.execute_reply":"2022-08-11T04:53:00.092378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model2.compile(optimizer='adam',loss = 'categorical_crossentropy',metrics=['accuracy'])\n\nmodel2.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:53:11.668458Z","iopub.execute_input":"2022-08-11T04:53:11.669026Z","iopub.status.idle":"2022-08-11T04:53:11.700125Z","shell.execute_reply.started":"2022-08-11T04:53:11.668996Z","shell.execute_reply":"2022-08-11T04:53:11.698560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history2 = model2.fit(train_generator,epochs=10,validation_data=validation_generator,verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:53:13.199015Z","iopub.execute_input":"2022-08-11T04:53:13.199421Z","iopub.status.idle":"2022-08-11T05:16:27.939902Z","shell.execute_reply.started":"2022-08-11T04:53:13.199391Z","shell.execute_reply":"2022-08-11T05:16:27.938493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores2 = model2.evaluate(test_generator)\nprint(scores2)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:17:27.874894Z","iopub.execute_input":"2022-08-11T05:17:27.875317Z","iopub.status.idle":"2022-08-11T05:18:49.854911Z","shell.execute_reply.started":"2022-08-11T05:17:27.875284Z","shell.execute_reply":"2022-08-11T05:18:49.853420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = range(1, len(history2.history['accuracy'])+1)\n\nplt.plot(epochs, history2.history['accuracy'], '#21466C', label='Train accuracy')\nplt.plot(epochs, history2.history['val_accuracy'], '#cc1123', label='Validation accuracy')\nplt.title('Training and validation accuracy')\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:18:49.857440Z","iopub.execute_input":"2022-08-11T05:18:49.857764Z","iopub.status.idle":"2022-08-11T05:18:50.113387Z","shell.execute_reply.started":"2022-08-11T05:18:49.857733Z","shell.execute_reply":"2022-08-11T05:18:50.111895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **CONCLUSION**","metadata":{}},{"cell_type":"markdown","source":"**ResNet was clearly the best performing model with a test accuracy of 61% on the 4000 test images from 100 classes using the epochs 10 where training dataset shows the accuracy of 97%**","metadata":{}},{"cell_type":"markdown","source":"# **RESNET:**\n* Train Accuracy: 96.60%\n* Validation Accuracy: 59.38%\n* Test Accuracy: 61%\n\n# **EfficientNet**\n* Train Accuracy: 3.09%\n* Validation Accuracy: 1.08%\n* Test Accuracy: 1%","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}