{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom pathlib import Path\nimport shutil\nimport math\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\ninput_dir = '/kaggle/input/landmark-recognition-2021/'\noutput_dir = '/kaggle/working/landmark-recognition-2021/'\n\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-16T07:52:14.949579Z","iopub.execute_input":"2021-11-16T07:52:14.949895Z","iopub.status.idle":"2021-11-16T07:52:14.955944Z","shell.execute_reply.started":"2021-11-16T07:52:14.949857Z","shell.execute_reply":"2021-11-16T07:52:14.955094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\ndef file_index_split(length):\n    train, test, validation = 0,0,0\n    if length <10:\n        test = 1\n        validation = 1\n    else:\n        test = math.floor(length*.2)\n        validation = test\n    return length-(test+validation), length-validation\n\nfile_index_split(40)","metadata":{"execution":{"iopub.status.busy":"2021-11-16T07:52:14.957625Z","iopub.execute_input":"2021-11-16T07:52:14.958363Z","iopub.status.idle":"2021-11-16T07:52:14.976205Z","shell.execute_reply.started":"2021-11-16T07:52:14.95832Z","shell.execute_reply":"2021-11-16T07:52:14.975041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_df = pd.read_csv(input_dir + 'train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-11-16T07:52:14.977768Z","iopub.execute_input":"2021-11-16T07:52:14.978162Z","iopub.status.idle":"2021-11-16T07:52:16.09141Z","shell.execute_reply.started":"2021-11-16T07:52:14.978122Z","shell.execute_reply":"2021-11-16T07:52:16.090515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_df","metadata":{"execution":{"iopub.status.busy":"2021-11-16T07:52:16.092599Z","iopub.execute_input":"2021-11-16T07:52:16.092974Z","iopub.status.idle":"2021-11-16T07:52:16.105464Z","shell.execute_reply.started":"2021-11-16T07:52:16.092939Z","shell.execute_reply":"2021-11-16T07:52:16.104733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nlabel_list = data_df['landmark_id'].unique()\ncnt = 0\n#final_label_list = []\ntrain = {'id':[],'landmark_id':[]}\ntest = {'id':[],'landmark_id':[]}\nvalidation = {'id':[], 'landmark_id':[]}\n\nfor label in list(label_list): # label order by random\n    file_list = list(data_df['id'][data_df['landmark_id']==label])\n    if len(file_list) >= 4:\n        train_length, test_length = file_index_split(len(file_list))\n        for file in file_list[:train_length]:  # 120 files for training\n            train['id'].append(file)\n            train['landmark_id'].append(label)\n        for file in file_list[train_length:test_length]: # 40 files for validation\n            test['id'].append(file)\n            test['landmark_id'].append(label)\n        for file in file_list[test_length:len(file_list)]: # 40 files for testing\n            validation['id'].append(file)\n            validation['landmark_id'].append(label)\n        cnt += 1\n    if cnt == 1000: # only need 1000 labels\n        break\n        ","metadata":{"execution":{"iopub.status.busy":"2021-11-16T07:52:37.450188Z","iopub.execute_input":"2021-11-16T07:52:37.45048Z","iopub.status.idle":"2021-11-16T07:52:39.443658Z","shell.execute_reply.started":"2021-11-16T07:52:37.450452Z","shell.execute_reply":"2021-11-16T07:52:39.442952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.DataFrame(train,columns=['id','landmark_id'])\ntest_df = pd.DataFrame(test,columns=['id','landmark_id'])\nvalidation_df = pd.DataFrame(validation,columns=['id','landmark_id'])","metadata":{"execution":{"iopub.status.busy":"2021-11-16T07:52:41.447633Z","iopub.execute_input":"2021-11-16T07:52:41.447907Z","iopub.status.idle":"2021-11-16T07:52:41.476191Z","shell.execute_reply.started":"2021-11-16T07:52:41.44788Z","shell.execute_reply":"2021-11-16T07:52:41.475477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_df.to_csv(output_dir + 'train.csv', index = False)\ntest_df.to_csv(output_dir + 'test.csv', index = False)\nvalidation_df.to_csv(output_dir + 'validation.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2021-11-16T07:59:23.882938Z","iopub.execute_input":"2021-11-16T07:59:23.883771Z","iopub.status.idle":"2021-11-16T07:59:23.94878Z","shell.execute_reply.started":"2021-11-16T07:59:23.883731Z","shell.execute_reply":"2021-11-16T07:59:23.948105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_df = pd.concat([pd.concat([train_df,test_df], ignore_index = True, axis= 0),validation_df], ignore_index = True, axis= 0)","metadata":{"execution":{"iopub.status.busy":"2021-11-15T16:33:13.768783Z","iopub.execute_input":"2021-11-15T16:33:13.769392Z","iopub.status.idle":"2021-11-15T16:33:13.778015Z","shell.execute_reply.started":"2021-11-15T16:33:13.769353Z","shell.execute_reply":"2021-11-15T16:33:13.777279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_df","metadata":{"execution":{"iopub.status.busy":"2021-11-15T16:33:37.594816Z","iopub.execute_input":"2021-11-15T16:33:37.595096Z","iopub.status.idle":"2021-11-15T16:33:37.609334Z","shell.execute_reply.started":"2021-11-15T16:33:37.595067Z","shell.execute_reply":"2021-11-15T16:33:37.608495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# copy train file\nfor id in total_df['id']:\n    subdir = id[0] + '/' + id[1] + '/' + id[2] + '/'\n    src_file = input_dir + 'train/' + subdir + id + '.jpg'\n    dist_dir = output_dir + 'train/' + subdir\n    dist_file = dist_dir + id + '.jpg'\n    Path(dist_dir).mkdir(parents=True, exist_ok=True)\n    shutil.copyfile(src_file, dist_file)","metadata":{"execution":{"iopub.status.busy":"2021-11-15T16:34:40.907772Z","iopub.execute_input":"2021-11-15T16:34:40.908086Z","iopub.status.idle":"2021-11-15T16:37:44.778983Z","shell.execute_reply.started":"2021-11-15T16:34:40.908054Z","shell.execute_reply":"2021-11-15T16:37:44.778296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shutil.make_archive('/kaggle/working/data', 'zip', output_dir)","metadata":{"execution":{"iopub.status.busy":"2021-11-15T16:37:44.780173Z","iopub.execute_input":"2021-11-15T16:37:44.780643Z","iopub.status.idle":"2021-11-15T16:38:40.128381Z","shell.execute_reply.started":"2021-11-15T16:37:44.780614Z","shell.execute_reply":"2021-11-15T16:38:40.127603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/working\nfrom IPython.display import FileLink\nFileLink('data.zip')","metadata":{"execution":{"iopub.status.busy":"2021-11-15T16:38:40.129575Z","iopub.execute_input":"2021-11-15T16:38:40.12982Z","iopub.status.idle":"2021-11-15T16:38:40.137828Z","shell.execute_reply.started":"2021-11-15T16:38:40.129792Z","shell.execute_reply":"2021-11-15T16:38:40.137081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}