{"cells":[{"metadata":{},"cell_type":"markdown","source":"<div align = \"center\">\n    <h1>Google Landmark Recognition Challenge</h1>\n    <img src = \"https://miro.medium.com/max/1280/1*OVP48VCImepxkHl7AVzkug.png\">\n</div>","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"<div align = \"center\">\n    <h3>What is this challenge all about?</h3>\n    <br>\n</div>\n    \n<div align = \"center\">Did you ever think about a place you visited earlier and forgot its name or location? Landmark recognition can help! <b>Google</b> aims to predict landmark labels directly from image pixels, to help people better understand and organize their photo collections.</div>\n  \n","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"Let's start by importing the required libraries:","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom collections import Counter\nimport plotly.express as px\nimport matplotlib.pyplot as plt\nimport cv2","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## About the dataset:","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"The `train.csv` contains two columns, `id` and `landmark_id`:","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/landmark-recognition-2020/train.csv')\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Woah! The training dataset has ~1.5 million images!","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of training images:\", len(df))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"But we got only ~81k landmarks:","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of landmarks:\" ,df['landmark_id'].nunique())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Most occuring landmarks:\n\nLet's see the most occuring landmarks:","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"landmark_counts = dict(Counter(df['landmark_id']))\nlandmark_dict = {'landmark_id': list(landmark_counts.keys()), 'count': list(landmark_counts.values())}\n\nlandmark_count_df = pd.DataFrame.from_dict(landmark_dict)\nlandmark_count_sorted = landmark_count_df.sort_values('count', ascending = False)\nlandmark_count_sorted.head(30)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Distribution of Landmarks with their counts:","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig_count = px.histogram(landmark_count_df, x = 'landmark_id', y = 'count')\nfig_count.update_layout(\n    title_text='Distribution of Landmarks',\n    xaxis_title_text='Landmark ID',\n    yaxis_title_text='Count'\n)\n\nfig_count.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Common Image Sizes:\n\nLet's see which image sizes are common in the dataset:\n\n> NOTE: I'm using the first 1000 images, since I'm reading the image to calculate image sizes.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"BASE_DIR = '../input/landmark-recognition-2020'\nTRAIN_DIR = BASE_DIR + '/train'\n\nimport os\n\nfilelist = []\nfor root, dirs, files in os.walk(TRAIN_DIR):\n    for file in files:\n        filelist.append(os.path.join(root,file))\nlen(filelist)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_sizes = []\n\nfor img_path in filelist[:1000]:\n    img = cv2.imread(img_path)\n    img_sizes.append(\"{}x{}\".format(img.shape[0], img.shape[1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"size_counts = dict(Counter(img_sizes))\nsize_dict = {'size': list(size_counts.keys()), 'count': list(size_counts.values())}\n\nsize_df = pd.DataFrame.from_dict(size_dict)\nsize_sorted = size_df.sort_values('count', ascending = False)\nsize_sorted = size_sorted[:10]\n\nfig_image_sizes = px.bar(size_sorted, x = 'size', y = 'count')\nfig_image_sizes.update_layout(title = 'Image Sizes')\nfig_image_sizes.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Let's see the top 10 landmarks!","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def retrieve_image(image_id):\n    img = cv2.imread(os.path.join(os.path.join(BASE_DIR, 'train'), image_id[0], image_id[1], image_id[2], image_id + '.jpg'))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    return img\n\ndef get_image_id(image_id):\n    return df[df['landmark_id'] == image_id]['id'][:1].values[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(5, 2, figsize = (30, 30), dpi = 250)\nax = ax.flatten()\n\ntop_10_landmarks = landmark_count_sorted['landmark_id'][:10].values\n\nfor i in range(10):    \n    ax[i].set_title(get_image_id(top_10_landmarks[i]))\n    ax[i].set_xticks([])\n    ax[i].set_yticks([])\n    ax[i].imshow(retrieve_image(get_image_id(top_10_landmarks[i])))\nfig.tight_layout()    \n# plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Bottom 10 landmarks:","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(5, 2, figsize = (30, 30), dpi = 250)\nax = ax.flatten()\nbottom_10_landmarks = landmark_count_sorted['landmark_id'][-10:].values\n\nfor i in range(10):\n    ax[i].set_xticks([])\n    ax[i].set_yticks([])    \n    ax[i].imshow(retrieve_image(get_image_id(bottom_10_landmarks[i])))\n    ax[i].set_title(get_image_id(bottom_10_landmarks[i]))\nfig.tight_layout()\n# plt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}