{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n'''for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))'''\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-10T06:16:45.679648Z","iopub.execute_input":"2021-11-10T06:16:45.680267Z","iopub.status.idle":"2021-11-10T06:16:45.706555Z","shell.execute_reply.started":"2021-11-10T06:16:45.680174Z","shell.execute_reply":"2021-11-10T06:16:45.70563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport PIL\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Sequential","metadata":{"execution":{"iopub.status.busy":"2021-11-10T06:16:45.710255Z","iopub.execute_input":"2021-11-10T06:16:45.711183Z","iopub.status.idle":"2021-11-10T06:16:52.039969Z","shell.execute_reply.started":"2021-11-10T06:16:45.711138Z","shell.execute_reply":"2021-11-10T06:16:52.038885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pathlib\nimport pandas as pd \ntrain_csv = pd.read_csv('/kaggle/input/landmark-recognition-2020/train.csv')\ntrain_csv.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-10T06:16:52.042478Z","iopub.execute_input":"2021-11-10T06:16:52.042806Z","iopub.status.idle":"2021-11-10T06:16:53.899783Z","shell.execute_reply.started":"2021-11-10T06:16:52.042772Z","shell.execute_reply":"2021-11-10T06:16:53.898705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_classes = len(train_csv['landmark_id'].unique())\nprint(num_classes)","metadata":{"execution":{"iopub.status.busy":"2021-11-10T06:16:53.901353Z","iopub.execute_input":"2021-11-10T06:16:53.902275Z","iopub.status.idle":"2021-11-10T06:16:53.936918Z","shell.execute_reply.started":"2021-11-10T06:16:53.902227Z","shell.execute_reply":"2021-11-10T06:16:53.935985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv('/kaggle/input/landmark-recognition-2020/train.csv')\ntrain_csv.head(10)\n\n# put .jpg into the file name\ndef add_txt(fn):\n    return fn+'.jpg'\n\ntrain_csv['id'] = train_csv['id'].apply(add_txt)\n\n\n\n# choose those labels with more than 200 images, and choose the first 200 images of each label\n# move every training files to the same folder\n%cd /kaggle/working\nif not os.path.exists('training'):\n    os.mkdir('training')\nif not os.path.exists('validation'):\n    os.mkdir('validation')\nif not os.path.exists('testing'):\n    os.mkdir('testing')    \n\nimport shutil\nimport random\n\nlabel_list = train_csv['landmark_id'].unique()\ncnt = 0\nfinal_label_list = []\n\nfor label in list(label_list): # label order by random\n    file_list = list(train_csv['id'][train_csv['landmark_id']==label])\n    if len(file_list) >= 200:\n        final_label_list.append(label)\n        if not os.path.exists('/kaggle/working/training/'+str(label)):\n            os.mkdir('/kaggle/working/training/'+str(label))\n        if not os.path.exists('/kaggle/working/validation/'+str(label)):\n            os.mkdir('/kaggle/working/validation/'+str(label))\n        if not os.path.exists('/kaggle/working/testing/'+str(label)):\n            os.mkdir('/kaggle/working/testing/'+str(label))\n        for file in file_list[:120]:  # 120 files for training\n            src = '/kaggle/input/landmark-recognition-2020/train/'+file[0]+'/'+file[1]+'/'+file[2]+'/'+file\n            dst = '/kaggle/working/training/'+str(label)+'/'+file\n            if not os.path.exists(dst):\n                shutil.copyfile(src, dst)\n        for file in file_list[120:160]: # 40 files for validation\n            src = '/kaggle/input/landmark-recognition-2020/train/'+file[0]+'/'+file[1]+'/'+file[2]+'/'+file\n            dst = '/kaggle/working/validation/'+str(label)+'/'+file\n            if not os.path.exists(dst):\n                shutil.copyfile(src, dst)\n        for file in file_list[160:200]: # 40 files for testing\n            src = '/kaggle/input/landmark-recognition-2020/train/'+file[0]+'/'+file[1]+'/'+file[2]+'/'+file\n            dst = '/kaggle/working/testing/'+str(label)+'/'+file\n            if not os.path.exists(dst):\n                shutil.copyfile(src, dst)\n        cnt += 1\n    if cnt == 100: # only need 100 labels\n        break","metadata":{"execution":{"iopub.status.busy":"2021-11-10T06:16:53.940139Z","iopub.execute_input":"2021-11-10T06:16:53.940608Z","iopub.status.idle":"2021-11-10T06:21:54.245014Z","shell.execute_reply.started":"2021-11-10T06:16:53.940573Z","shell.execute_reply":"2021-11-10T06:21:54.24393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True)\n\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_dir = '/kaggle/working/training'\nvalidation_dir = '/kaggle/working/validation'\ntest_dir = '/kaggle/working/testing'\n\ntrain_generator = train_datagen.flow_from_directory(\n    train_dir,\n    target_size=(256, 256),\n    batch_size = 32,\n    class_mode='categorical',\n    seed=42)\n\nvalidation_generator = test_datagen.flow_from_directory(\n    validation_dir,\n    target_size=(256, 256),\n    batch_size = 32,\n    class_mode='categorical',\n    seed=42)\n\ntest_generator = test_datagen.flow_from_directory(\n    test_dir,\n    target_size=(256, 256),\n    batch_size = 1,\n    class_mode='categorical',\n    seed=42)","metadata":{"execution":{"iopub.status.busy":"2021-11-10T06:21:54.2481Z","iopub.execute_input":"2021-11-10T06:21:54.248648Z","iopub.status.idle":"2021-11-10T06:21:55.133002Z","shell.execute_reply.started":"2021-11-10T06:21:54.248603Z","shell.execute_reply":"2021-11-10T06:21:55.131951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import ResNet50V2\nconv_base = ResNet50V2(include_top=False,\n                    weights=\"imagenet\",\n                    input_shape=(256, 256, 3))\nconv_base.trainable = True\nconv_base.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-10T06:21:55.134841Z","iopub.execute_input":"2021-11-10T06:21:55.135171Z","iopub.status.idle":"2021-11-10T06:22:03.456469Z","shell.execute_reply.started":"2021-11-10T06:21:55.135127Z","shell.execute_reply":"2021-11-10T06:22:03.455398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import Dense, Dropout, MaxPooling2D, GlobalAveragePooling2D, Flatten, Conv2D, Input\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras import optimizers\nimport tensorflow as tf\n\nmodel = Sequential()\nmodel.add(conv_base)\nmodel.add(GlobalAveragePooling2D())\nmodel.add(Dense(100, activation='softmax'))\nmodel.compile(optimizer='adam',\n              loss = 'categorical_crossentropy',\n              metrics=['accuracy'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-10T06:22:03.458978Z","iopub.execute_input":"2021-11-10T06:22:03.459323Z","iopub.status.idle":"2021-11-10T06:22:03.978216Z","shell.execute_reply.started":"2021-11-10T06:22:03.459278Z","shell.execute_reply":"2021-11-10T06:22:03.977245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    epochs=50,\n    validation_data=validation_generator,\n    verbose=2\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-10T06:22:03.980132Z","iopub.execute_input":"2021-11-10T06:22:03.980452Z","iopub.status.idle":"2021-11-10T06:25:37.721936Z","shell.execute_reply.started":"2021-11-10T06:22:03.980408Z","shell.execute_reply":"2021-11-10T06:25:37.720263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(1, len(acc)+1)\n\nplt.plot(epochs, acc, '#21466C', label='Training acc')\nplt.plot(epochs, val_acc, '#ff0051', label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.legend()\n\nplt.figure()\n\nplt.plot(epochs, loss, '#21466C', label='Training loss')\nplt.plot(epochs, val_loss, '#ff0051', label='Validation loss')\nplt.title('Training and validation loss')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-10T06:25:37.72337Z","iopub.status.idle":"2021-11-10T06:25:37.729033Z","shell.execute_reply.started":"2021-11-10T06:25:37.728751Z","shell.execute_reply":"2021-11-10T06:25:37.728781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.evaluate(test_generator)\nprint('loss:', scores[0])\nprint('accuracy:', scores[1])","metadata":{"execution":{"iopub.status.busy":"2021-11-10T06:25:37.732546Z","iopub.status.idle":"2021-11-10T06:25:37.733374Z","shell.execute_reply.started":"2021-11-10T06:25:37.733102Z","shell.execute_reply":"2021-11-10T06:25:37.733131Z"},"trusted":true},"execution_count":null,"outputs":[]}]}