{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"from pathlib import Path\nfrom functools import reduce\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport urllib.parse","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.random.seed(25)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"BASE_DIR = Path(\"/kaggle/input/landmark-retrieval-2020\")\nTRAIN_CSV = BASE_DIR / \"train.csv\"\nTRAIN_DATA = BASE_DIR / \"train\"\nTEST_DATA = BASE_DIR / \"test\"\nIDX_DATA = BASE_DIR / \"index\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv(TRAIN_CSV)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check if any data is missing\ndf.isna().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Get the number of images in each folders\n\ntrain_imgs = TRAIN_DATA.rglob(\"*.jpg\")\nreduce(lambda acc, e: acc + 1, train_imgs, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"idx_imgs = IDX_DATA.rglob(\"*.jpg\")\nreduce(lambda acc, e: acc + 1, idx_imgs, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_imgs = TEST_DATA.rglob(\"*.jpg\")\nreduce(lambda acc, e: acc + 1, test_imgs, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check if id i.e image id column is unique, as mentioned in the data description\ndf['id'].is_unique","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Get the number of images per landmark\nlandmark_dist = df['landmark_id'].value_counts().rename_axis('landmark_id')\\\n.reset_index(name=\"count\")\\\n.sort_values(by=['count'], ascending=[False])\n\n(\"Max : {} | Min : {}\".format(landmark_dist['count'].max(), landmark_dist['count'].min()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"landmark_dist['count'].describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"As seen in the above block, about 75% landmarks have less than or equal to 20 images. Looks like data is highly imbalanced","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check the distribution of landmark's images\nsns.set(style=\"darkgrid\")\nfig, (ax1, ax2) = plt.subplots(2, 1)\nfig.set_size_inches(20, 12)\nsns.countplot(\n    df['landmark_id'],\n    order=df['landmark_id'].value_counts().index[:25],\n    ax=ax1\n)\nsns.countplot(\n    df['landmark_id'],\n    order=df['landmark_id'].value_counts().index[-25:],\n    ax=ax2\n)\nax1.set_title('distribution of landmark images | top 25')\nax2.set_title('distribution of landmark images | last 25')\nax2.set(ylim=(0, 50))\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The landmark id `138982` weirdly has too many images, so it could be the class label to show non-labeled images or it could be a valid landmark but just have two many images. One way to check is too see few images randomly that belongs to that landmark id.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Courtesy :: https://www.kaggle.com/sudeepshouche/identify-landmark-name-from-landmark-id\n\nurl = 'https://s3.amazonaws.com/google-landmark/metadata/train_label_to_category.csv'\ndf_classes = pd.read_csv(url, index_col = 'landmark_id', encoding='latin', engine='python')\n\nget_landmark_name = lambda x: urllib.parse.unquote(x['category'].replace('http://commons.wikimedia.org/wiki/Category:', ''))\ndf_classes['landmark_name'] = df_classes.apply(get_landmark_name, axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_classes.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Randomly chooses at max 12 images for any landmark based on its id\ndef render(img_path, nrow, col, ax, row):\n    print(\"Loading : {}\".format(img_path))\n    img = cv2.imread(img_path)\n    ax[nrow, col].imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n    \n    if row is not None:\n        ax[nrow, col].set_title(row['landmark_name'])\n\n\ndef get_images(landmark_id):    \n    _df = df.loc[df['landmark_id'] == landmark_id, :]\n    \n    plt.rcParams[\"axes.grid\"] = False\n    \n    _df = _df.sample(n=12).reset_index()\n    \n    _df = pd.merge(_df, df_classes, on=['landmark_id'], how=\"inner\")\n    \n    no_row = np.math.ceil(min(len(_df), 12)/3)\n    f, ax = plt.subplots(no_row, 3, figsize=(24, 20))\n    print(\"Fig Shape : {0}\".format((no_row, 3)))\n        \n    \n    nrow = 0\n    for idx, row in _df.iterrows():\n        image_id = row['id'] + \".jpg\"\n        \n        img_path = TRAIN_DATA / image_id[0] / image_id[1] / image_id[2] / image_id\n        \n        col = int(idx % 3)\n        render(str(img_path), nrow, col, ax, row)\n                \n        # when all columns are filled in a row\n        if col == 2:\n            nrow += 1\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"get_images(138982)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"After couple of runs, it looks like the images belong to `138982` are unlabeled ones","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# 2nd Landmark Imgs\nget_images(126637)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Get random images from folders\n\ndef get_images_from_dir(img_dir):\n    imgs = np.random.choice(list(TEST_DATA.rglob(\"*.jpg\")), 12)\n    \n    fig, ax = plt.subplots(4, 3, figsize=(24, 20))\n    \n    row = 0\n    for idx, img_path in enumerate(imgs):\n        \n        col = idx % 3\n        \n        render(str(img_path), row, col, ax, row=None)\n        \n        if col == 2:\n            row += 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Random images from test dir\nget_images_from_dir(TEST_DATA)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Random images from index dir\nget_images_from_dir(IDX_DATA)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}