{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n'''for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))'''\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-20T02:00:24.783119Z","iopub.execute_input":"2021-06-20T02:00:24.783481Z","iopub.status.idle":"2021-06-20T02:00:24.790401Z","shell.execute_reply.started":"2021-06-20T02:00:24.783449Z","shell.execute_reply":"2021-06-20T02:00:24.789436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ntrain_csv = pd.read_csv('/kaggle/input/landmark-recognition-2020/train.csv')\ntrain_csv.head(10)","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:00:25.266547Z","iopub.execute_input":"2021-06-20T02:00:25.26682Z","iopub.status.idle":"2021-06-20T02:00:26.64588Z","shell.execute_reply.started":"2021-06-20T02:00:25.266792Z","shell.execute_reply":"2021-06-20T02:00:26.6451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_csv)","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:00:26.647246Z","iopub.execute_input":"2021-06-20T02:00:26.647554Z","iopub.status.idle":"2021-06-20T02:00:26.652747Z","shell.execute_reply.started":"2021-06-20T02:00:26.647525Z","shell.execute_reply":"2021-06-20T02:00:26.651943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_csv['landmark_id'].unique())","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:00:26.65463Z","iopub.execute_input":"2021-06-20T02:00:26.655217Z","iopub.status.idle":"2021-06-20T02:00:26.678234Z","shell.execute_reply.started":"2021-06-20T02:00:26.655156Z","shell.execute_reply":"2021-06-20T02:00:26.677338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.image as img\nimport matplotlib.pyplot as plt\ndef show_img(file_name):\n    image = img.imread('/kaggle/input/landmark-recognition-2020/train/'+file_name[0]+'/'+file_name[1]+'/'+file_name[2]+'/'+file_name)\n    plt.imshow(image)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:00:26.679736Z","iopub.execute_input":"2021-06-20T02:00:26.680071Z","iopub.status.idle":"2021-06-20T02:00:26.686082Z","shell.execute_reply.started":"2021-06-20T02:00:26.680036Z","shell.execute_reply":"2021-06-20T02:00:26.685291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_img('0000059611c7d079.jpg')","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:00:26.687436Z","iopub.execute_input":"2021-06-20T02:00:26.687858Z","iopub.status.idle":"2021-06-20T02:00:26.8925Z","shell.execute_reply.started":"2021-06-20T02:00:26.687821Z","shell.execute_reply":"2021-06-20T02:00:26.891675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_csv = pd.read_csv('/kaggle/input/landmark-recognition-2020/train.csv')\ntrain_csv.head(10)\n\n# put .jpg into the file name\ndef add_txt(fn):\n    return fn+'.jpg'\n\ntrain_csv['id'] = train_csv['id'].apply(add_txt)\n\n\n\n# choose those labels with more than 200 images, and choose the first 200 images of each label\n# move every training files to the same folder\n%cd /kaggle/working\nif not os.path.exists('training'):\n    os.mkdir('training')\nif not os.path.exists('validation'):\n    os.mkdir('validation')\nif not os.path.exists('testing'):\n    os.mkdir('testing')    \n\nimport shutil\nimport random\n\nlabel_list = train_csv['landmark_id'].unique()\ncnt = 0\nfinal_label_list = []\n\nfor label in list(label_list): # label order by random\n    file_list = list(train_csv['id'][train_csv['landmark_id']==label])\n    if len(file_list) >= 200:\n        final_label_list.append(label)\n        if not os.path.exists('/kaggle/working/training/'+str(label)):\n            os.mkdir('/kaggle/working/training/'+str(label))\n        if not os.path.exists('/kaggle/working/validation/'+str(label)):\n            os.mkdir('/kaggle/working/validation/'+str(label))\n        if not os.path.exists('/kaggle/working/testing/'+str(label)):\n            os.mkdir('/kaggle/working/testing/'+str(label))\n        for file in file_list[:120]:  # 120 files for training\n            src = '/kaggle/input/landmark-recognition-2020/train/'+file[0]+'/'+file[1]+'/'+file[2]+'/'+file\n            dst = '/kaggle/working/training/'+str(label)+'/'+file\n            if not os.path.exists(dst):\n                shutil.copyfile(src, dst)\n        for file in file_list[120:160]: # 40 files for validation\n            src = '/kaggle/input/landmark-recognition-2020/train/'+file[0]+'/'+file[1]+'/'+file[2]+'/'+file\n            dst = '/kaggle/working/validation/'+str(label)+'/'+file\n            if not os.path.exists(dst):\n                shutil.copyfile(src, dst)\n        for file in file_list[160:200]: # 40 files for testing\n            src = '/kaggle/input/landmark-recognition-2020/train/'+file[0]+'/'+file[1]+'/'+file[2]+'/'+file\n            dst = '/kaggle/working/testing/'+str(label)+'/'+file\n            if not os.path.exists(dst):\n                shutil.copyfile(src, dst)\n        cnt += 1\n    if cnt == 100: # only need 100 labels\n        break\n# 20,000 files in total","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:00:26.893537Z","iopub.execute_input":"2021-06-20T02:00:26.893851Z","iopub.status.idle":"2021-06-20T02:03:11.758401Z","shell.execute_reply.started":"2021-06-20T02:00:26.893818Z","shell.execute_reply":"2021-06-20T02:03:11.757527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(final_label_list)","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:03:11.75972Z","iopub.execute_input":"2021-06-20T02:03:11.760071Z","iopub.status.idle":"2021-06-20T02:03:11.767497Z","shell.execute_reply.started":"2021-06-20T02:03:11.760024Z","shell.execute_reply":"2021-06-20T02:03:11.766557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(os.listdir('./training')), len(os.listdir('./validation')), len(os.listdir('./testing')))","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:03:11.770352Z","iopub.execute_input":"2021-06-20T02:03:11.770732Z","iopub.status.idle":"2021-06-20T02:03:11.778088Z","shell.execute_reply.started":"2021-06-20T02:03:11.770696Z","shell.execute_reply":"2021-06-20T02:03:11.777227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from shutil import make_archive\nmake_archive('train_data', 'zip', '/kaggle/working')","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:03:11.779869Z","iopub.execute_input":"2021-06-20T02:03:11.780279Z","iopub.status.idle":"2021-06-20T02:03:50.545217Z","shell.execute_reply.started":"2021-06-20T02:03:11.780245Z","shell.execute_reply":"2021-06-20T02:03:50.544429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/working\nfrom IPython.display import FileLink\nFileLink(r'train_data.zip')","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:03:50.54648Z","iopub.execute_input":"2021-06-20T02:03:50.546816Z","iopub.status.idle":"2021-06-20T02:03:50.554418Z","shell.execute_reply.started":"2021-06-20T02:03:50.546778Z","shell.execute_reply":"2021-06-20T02:03:50.553596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\n\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True)\n\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_dir = '/kaggle/working/training'\nvalidation_dir = '/kaggle/working/validation'\ntest_dir = '/kaggle/working/testing'\n\ntrain_generator = train_datagen.flow_from_directory(\n    train_dir,\n    target_size=(256, 256),\n    batch_size = 32,\n    class_mode='categorical',\n    seed=42)\n\nvalidation_generator = test_datagen.flow_from_directory(\n    validation_dir,\n    target_size=(256, 256),\n    batch_size = 32,\n    class_mode='categorical',\n    seed=42)\n\ntest_generator = test_datagen.flow_from_directory(\n    test_dir,\n    target_size=(256, 256),\n    batch_size = 1,\n    class_mode='categorical',\n    seed=42)","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:03:50.555572Z","iopub.execute_input":"2021-06-20T02:03:50.556041Z","iopub.status.idle":"2021-06-20T02:03:51.485824Z","shell.execute_reply.started":"2021-06-20T02:03:50.556006Z","shell.execute_reply":"2021-06-20T02:03:51.485003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import MobileNetV2\n\nconv_base = MobileNetV2(include_top=False,\n                        weights=\"imagenet\",\n                        input_shape=(256, 256, 3)\n)\n\nconv_base.trainable = True\n\nconv_base.summary()","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:03:51.487096Z","iopub.execute_input":"2021-06-20T02:03:51.487543Z","iopub.status.idle":"2021-06-20T02:03:54.894667Z","shell.execute_reply.started":"2021-06-20T02:03:51.487507Z","shell.execute_reply":"2021-06-20T02:03:54.89373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.layers import Dense, Dropout, MaxPooling2D, GlobalAveragePooling2D, Flatten, Conv2D, Input\nfrom keras.models import Sequential\nfrom keras import optimizers\nimport tensorflow as tf\n\nmodel = Sequential()\nmodel.add(conv_base)\nmodel.add(GlobalAveragePooling2D())\nmodel.add(Dense(100, activation='softmax'))\nmodel.compile(optimizer=optimizers.RMSprop(lr=2e-5),\n              loss = 'categorical_crossentropy',\n              metrics=['accuracy'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:03:54.896207Z","iopub.execute_input":"2021-06-20T02:03:54.896706Z","iopub.status.idle":"2021-06-20T02:03:55.289738Z","shell.execute_reply.started":"2021-06-20T02:03:54.896663Z","shell.execute_reply":"2021-06-20T02:03:55.288873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    epochs=100, \n    validation_data=validation_generator,\n    verbose=2\n)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:03:55.290991Z","iopub.execute_input":"2021-06-20T02:03:55.291391Z","iopub.status.idle":"2021-06-20T02:24:49.361057Z","shell.execute_reply.started":"2021-06-20T02:03:55.291349Z","shell.execute_reply":"2021-06-20T02:24:49.360157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot the training results\nimport matplotlib.pyplot as plt\n\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(1, len(acc)+1)\n\nplt.plot(epochs, acc, '#21466C', label='Training acc')\nplt.plot(epochs, val_acc, '#ff0051', label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.legend()\n\nplt.figure()\n\nplt.plot(epochs, loss, '#21466C', label='Training loss')\nplt.plot(epochs, val_loss, '#ff0051', label='Validation loss')\nplt.title('Training and validation loss')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-20T02:36:04.108649Z","iopub.execute_input":"2021-06-20T02:36:04.108983Z","iopub.status.idle":"2021-06-20T02:36:04.396455Z","shell.execute_reply.started":"2021-06-20T02:36:04.108951Z","shell.execute_reply":"2021-06-20T02:36:04.395652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.evaluate(test_generator)\nprint('loss:', scores[0])\nprint('accuracy:', scores[1])","metadata":{"execution":{"iopub.status.busy":"2021-06-20T03:23:16.397876Z","iopub.execute_input":"2021-06-20T03:23:16.398226Z","iopub.status.idle":"2021-06-20T03:23:54.447371Z","shell.execute_reply.started":"2021-06-20T03:23:16.398191Z","shell.execute_reply":"2021-06-20T03:23:54.446578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Test by eyes","metadata":{}},{"cell_type":"code","source":"image_index = 0 # choose a image (0-3999)\nimage = test_generator[image_index][0] # [0][0]: the first 0 stands for the 0th image, the second 0 stands for the image array\nimage = image.reshape((256,256,3))\nplt.imshow(image)\nplt.show()\nprint(test_generator[0][1][0]) # array of the image's label","metadata":{"execution":{"iopub.status.busy":"2021-06-20T03:55:32.154785Z","iopub.execute_input":"2021-06-20T03:55:32.155124Z","iopub.status.idle":"2021-06-20T03:55:32.318739Z","shell.execute_reply.started":"2021-06-20T03:55:32.155088Z","shell.execute_reply":"2021-06-20T03:55:32.317874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_label(arr):\n    for i in range(len(arr)):\n        if arr[i] == max(arr):\n            return i","metadata":{"execution":{"iopub.status.busy":"2021-06-20T03:55:33.009895Z","iopub.execute_input":"2021-06-20T03:55:33.010301Z","iopub.status.idle":"2021-06-20T03:55:33.01688Z","shell.execute_reply.started":"2021-06-20T03:55:33.010258Z","shell.execute_reply":"2021-06-20T03:55:33.016058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label from original data\nget_label(test_generator[image_index][1][0])","metadata":{"execution":{"iopub.status.busy":"2021-06-20T03:55:33.355981Z","iopub.execute_input":"2021-06-20T03:55:33.356338Z","iopub.status.idle":"2021-06-20T03:55:33.370647Z","shell.execute_reply.started":"2021-06-20T03:55:33.356302Z","shell.execute_reply":"2021-06-20T03:55:33.369718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label by prediction\nget_label(model.predict(test_generator[image_index][0])[0])","metadata":{"execution":{"iopub.status.busy":"2021-06-20T03:55:33.671505Z","iopub.execute_input":"2021-06-20T03:55:33.671755Z","iopub.status.idle":"2021-06-20T03:55:33.72645Z","shell.execute_reply.started":"2021-06-20T03:55:33.67173Z","shell.execute_reply":"2021-06-20T03:55:33.725535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label = get_label(model.predict(test_generator[image_index][0])[0])\nimage_list = [] # image with the same label\nfor i in range(4000):\n    if get_label(test_generator[i][1][0]) == label:\n        image_list.append(i)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-20T03:55:33.861626Z","iopub.execute_input":"2021-06-20T03:55:33.861916Z","iopub.status.idle":"2021-06-20T03:56:00.346386Z","shell.execute_reply.started":"2021-06-20T03:55:33.861889Z","shell.execute_reply":"2021-06-20T03:56:00.345531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if len(image_list) > 10:\n    for i in range(10):\n        if i != image_index:\n            image = test_generator[image_list[i]][0]\n            image = image.reshape((256,256,3))\n            plt.imshow(image)\n            plt.show()\n        \nelse:\n    for i in range(len(image_list)):\n        if i != image_index:\n            image = test_generator[image_list[i]][0]\n            image = image.reshape((256,256,3))\n            plt.imshow(image)\n            plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2021-06-20T03:56:51.992004Z","iopub.execute_input":"2021-06-20T03:56:51.992358Z","iopub.status.idle":"2021-06-20T03:56:53.205555Z","shell.execute_reply.started":"2021-06-20T03:56:51.992324Z","shell.execute_reply":"2021-06-20T03:56:53.204524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Save model and variable","metadata":{}},{"cell_type":"code","source":"%cd /kaggle/working\nos.mkdir('model')\n%cd /kaggle/working/model\nmodel.save('my_model.h5')\n","metadata":{"execution":{"iopub.status.busy":"2021-06-20T03:06:20.22734Z","iopub.execute_input":"2021-06-20T03:06:20.22767Z","iopub.status.idle":"2021-06-20T03:06:20.565855Z","shell.execute_reply.started":"2021-06-20T03:06:20.227638Z","shell.execute_reply":"2021-06-20T03:06:20.565101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/working/model\nfrom IPython.display import FileLink\nFileLink('my_model.h5')","metadata":{"execution":{"iopub.status.busy":"2021-06-20T03:07:39.410603Z","iopub.execute_input":"2021-06-20T03:07:39.410928Z","iopub.status.idle":"2021-06-20T03:07:39.421577Z","shell.execute_reply.started":"2021-06-20T03:07:39.410894Z","shell.execute_reply":"2021-06-20T03:07:39.42052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n  \n# Create a variable\n\n  \n# Open a file and use dump()\nwith open('file_3.pkl', 'wb') as file:\n      \n    # A new file will be created\n    pickle.dump(history.history, file)","metadata":{"execution":{"iopub.status.busy":"2021-06-20T03:17:48.501857Z","iopub.execute_input":"2021-06-20T03:17:48.502201Z","iopub.status.idle":"2021-06-20T03:17:48.508387Z","shell.execute_reply.started":"2021-06-20T03:17:48.502152Z","shell.execute_reply":"2021-06-20T03:17:48.507261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n  \n# Open the file in binary mode\nwith open('file_3.pkl', 'rb') as file:\n      \n    # Call load method to deserialze\n    myvar = pickle.load(file)\n  \n","metadata":{"execution":{"iopub.status.busy":"2021-06-20T03:18:10.789921Z","iopub.execute_input":"2021-06-20T03:18:10.790255Z","iopub.status.idle":"2021-06-20T03:18:10.796891Z","shell.execute_reply.started":"2021-06-20T03:18:10.790222Z","shell.execute_reply":"2021-06-20T03:18:10.79605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"myvar['accuracy']","metadata":{"execution":{"iopub.status.busy":"2021-06-20T03:18:31.138354Z","iopub.execute_input":"2021-06-20T03:18:31.13877Z","iopub.status.idle":"2021-06-20T03:18:31.143526Z","shell.execute_reply.started":"2021-06-20T03:18:31.138735Z","shell.execute_reply":"2021-06-20T03:18:31.142714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}