{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, filenames in os.listdir('../input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline \nimport time, os, glob\nimport cv2\n\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers,models ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/landmark-retrieval-2020/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"t = train.groupby('landmark_id')['id'].count().to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"t = t[t['id']>100]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train[train['landmark_id'].isin(list(t.index))]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from glob import glob\ntrain_files = glob('../input/landmark-retrieval-2020/train/*/*/*/*')\nindex_files = glob('../input/landmark-retrieval-2020/index/*/*/*/*')\ntest_files = glob('../input/landmark-retrieval-2020/test/*/*/*/*')\nlen(train_files),len(index_files),len(test_files)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"t = pd.DataFrame({'train_path' : train_files})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"t = pd.DataFrame({'train_path' : train_files})\nt['id'] = t['train_path'].apply(lambda x: x.split('/')[-1][:-4])\ntrain = train.merge(t,how='inner',on=['id'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"### cosine 유사도","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics.pairwise import cosine_similarity","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"inf_image = []\nfor i,row in train[0:20].iterrows():\n    path = row['train_path']\n    example = cv2.imread(path)\n    example = cv2.resize(example, dsize=(20,20))\n    example = cv2.cvtColor(example, cv2.COLOR_BGR2GRAY)\n#     print(example)\n    plt.imshow(example)\n#     gray_ex = [cv2.cvtColor(example[i], cv2.COLOR_BGR2GRAY) for i in range(len(inf_image))]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def make_df_cosine(idx):\n    inf_image = []\n    dfsample = train[train['landmark_id']==idx]\n    for i,row in dfsample.reset_index().iterrows():\n        path = row['train_path']\n        example = cv2.imread(path)\n        example = cv2.resize(example, dsize=(3,3))\n        inf_image.append(example)\n    \n    gray_ex = [cv2.cvtColor(inf_image[i], cv2.COLOR_BGR2GRAY) for i in range(len(inf_image))]\n#     print(gray_ex)\n    inf_avg_cos = []\n    for i in range(len(inf_image)):\n        cosine = 0\n        for y in range(len(inf_image)):\n            cos = cosine_similarity(gray_ex[i],gray_ex[y]).mean()\n            cosine += cos\n        avg_cos = cosine/len(range(len(inf_image)))\n        inf_avg_cos.append(avg_cos)\n        \n    avg_cos_df = pd.DataFrame({'train_path' : dfsample['train_path'],'avg_cosine' : inf_avg_cos})\n    avg_cos_df = avg_cos_df.sort_values('avg_cosine', ascending = False)\n    avg_cos_df = avg_cos_df.reset_index()\n    for i in range(len(gray_ex)):\n#         print(i)\n        cos = cosine_similarity(gray_ex[avg_cos_df.index[0]],gray_ex[i]).mean()\n        if cos < 0.9:\n            avg_cos_df = avg_cos_df.drop(i)\n            \n    return avg_cos_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tqdm import tqdm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"selected_df = pd.DataFrame()\nfor idx in tqdm(train['landmark_id'].unique()):\n    print(idx)\n    inf_df = make_df_cosine(idx)\n    selected_df = selected_df.append(inf_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"selected_df","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}