{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\n\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"BASE_DIR = '../input/landmark-recognition-2020'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls {BASE_DIR}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(os.path.join(BASE_DIR, 'train.csv'))\ntrain_df.head(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"print(f'Total number of training images: {len(train_df)}')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There are ~1.58 million training images. That's huge!","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f'Total number of landmarks in training dataset: {train_df[\"landmark_id\"].nunique()}')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There are 81313 landmarks and 1580470 images, therefore only ~19 images per landmark (average) for training data. Let's see actual distribution of landmarks.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## Distribution of Landmarks","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"target_dist = train_df.groupby('landmark_id', as_index=False)['id'].count().sort_values('id', ascending=False).reset_index(drop=True)\ntarget_dist = target_dist.rename(columns={'id':'count'})\ntarget_dist","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"sns.set(rc={'figure.figsize':(11,8)})\nsns.set(style=\"whitegrid\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The distribution of labels is highly imbalanced. Let's first see the distribution of top 50 landmarks (which occurs the most)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"ax = sns.distplot(train_df['landmark_id'].value_counts()[:50])\nax.set(xlabel='Landmark Counts', ylabel='Probability Density', title='Distribution of top 50 landmarks')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Now let's plot the PDF of rest of the landmarks","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"ax = sns.distplot(train_df['landmark_id'].value_counts()[51:])\nax.set(xlabel='Landmark Counts', ylabel='Probability Density')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There are some landmarks having only ~2 training images!","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## Top 6 Landmarks (by count)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_image(image_id):\n    img = cv2.imread(os.path.join(os.path.join(BASE_DIR, 'train'), image_id[0], image_id[1], image_id[2], image_id + '.jpg'))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    return img\n\ndef get_image_id(landmark_id):\n    return train_df[train_df['landmark_id'] == landmark_id]['id'][:1].values[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(nrows=2, ncols=3, figsize=(30, 15))\nax = ax.flatten()\nlandmark_ids = target_dist['landmark_id'][:6].values\n\nfor i in range(6):\n    ax[i].imshow(get_image(get_image_id(landmark_ids[i])))\n    ax[i].grid(False)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Bottom 6 Landmarks (by count)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(nrows=2, ncols=3, figsize=(30, 15))\nax = ax.flatten()\nlandmark_ids = target_dist['landmark_id'][-6:].values\n\nfor i in range(6):\n    ax[i].imshow(get_image(get_image_id(landmark_ids[i])))\n    ax[i].grid(False)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}