{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Google Landmark Recognition 2020\n\n**Let's perform Exploratory Data Analysis to understand the data better**\n\n","attachments":{}},{"metadata":{},"cell_type":"markdown","source":"# 1. Let's begin by importing libraries and packages"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os\n\nimport random\nimport seaborn as sns\nimport cv2\n\n\nimport pandas as pd\npd.set_option('display.max_colwidth', 1000)\nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport PIL\nimport IPython.display as ipd\nimport glob\nimport h5py\nimport plotly.graph_objs as go\nimport plotly.express as px\nfrom PIL import Image, ImageDraw\nfrom tempfile import mktemp\n\n\nfrom bokeh.layouts import column, row\nfrom bokeh.models import ColumnDataSource, LinearAxis, Range1d\nfrom bokeh.models.tools import HoverTool\nfrom bokeh.palettes import BuGn4\nfrom bokeh.plotting import figure, output_notebook, show\nfrom bokeh.transform import cumsum\nfrom math import pi\n\noutput_notebook()\n\nfrom IPython.display import Image, display\nimport warnings\nwarnings.filterwarnings(\"ignore\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 2. **Loading Data**"},{"metadata":{"trusted":true},"cell_type":"code","source":"image_samples = os.listdir('../input/landmark-recognition-2020/')\n\nBASE_PATH = '../input/landmark-recognition-2020'\n\nTRAIN_DIR = f'{BASE_PATH}/train'\nTRST_DIR = f'{BASE_PATH}/test'\n\nprint('Reading Data ...')\ntrain = pd.read_csv(f'{BASE_PATH}/train.csv')\nsubmission = pd.read_csv(f'{BASE_PATH}/sample_submission.csv')\nprint('Reading data is completed')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The dataset comprises of following important files:\n\n**train.csv**: This file contains, ids and targets\n* id: image id\n* landmark_id: target landmark id"},{"metadata":{"trusted":true},"cell_type":"code","source":"display(train.head(10))\nprint(\"Shape of train_data : \", train.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"display(submission.head())\nprint(\"Shape of Sample Submission\", submission.shape)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 3. Performing Exploratory Data Analysis"},{"metadata":{},"cell_type":"markdown","source":"# Target Distribution (Number of images per landmark_id)"},{"metadata":{"trusted":true},"cell_type":"code","source":"# display top 10 landmarks\n\nlandmark = train.landmark_id.value_counts()\nlandmark_df = pd.DataFrame({'landmark_id': landmark.index, 'frequency': landmark.values}).head(10)\n\nlandmark_df['landmark_id'] = landmark_df.landmark_id.apply(lambda x: f'landmark_id_{x}')\nprint(landmark_df.head())\n\nfig = px.bar(landmark_df, x=\"frequency\", y = \"landmark_id\", color='landmark_id', hover_data = [\"landmark_id\", \"frequency\"],\n            height = 500, title = 'Number of Images per landmark_id (Top 10 landmark_ids)'\n            )\n\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**The most frequent landmark_id is 138982 and the frequency is 6272**"},{"metadata":{},"cell_type":"markdown","source":"# Let's see least frequent landmarks"},{"metadata":{"trusted":true},"cell_type":"code","source":"# display bottom 10 landmarks\n\nlandmark = train.landmark_id.value_counts()\nlandmark_df = pd.DataFrame({'landmark_id': landmark.index, 'frequency': landmark.values}).tail(10)\n\nlandmark_df['landmark_id'] = landmark_df.landmark_id.apply(lambda x: f'landmark_id_{x}')\n\n\nfig = px.bar(landmark_df, x=\"frequency\", y = \"landmark_id\", color='landmark_id', hover_data = [\"landmark_id\", \"frequency\"],\n            height = 500, title = 'Number of Images per landmark_id (Top 10 landmark_ids)'\n            )\n\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**There are many least frequency landmarks with frequency as 2**"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Missing Data in the training set\ntotal = train.isnull().sum().sort_values(ascending= False)\npercent = (train.isnull().sum()/train.isnull().count()).sort_values(ascending = False)\nmissing_train_data = pd.concat([total, percent], axis = 1, keys = ['Total', 'Percent'])\nmissing_train_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Class distribution\n\nplt.figure(figsize = (10, 8))\nplt.title('Category Distribuition')\nsns.distplot(train['landmark_id'])\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of classes under 20 occurences\",\n      (train['landmark_id'].value_counts() <= 20).sum(),\n      'out of total number of categories',len(train['landmark_id'].unique()))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 4 Visulaization of Images"},{"metadata":{"trusted":true},"cell_type":"code","source":"import PIL\nfrom PIL import Image, ImageDraw\n\ndef display_images(images, title=None): \n    f, ax = plt.subplots(5,5, figsize=(18,22))\n    if title:\n        f.suptitle(title, fontsize = 30)\n\n    for i, image_id in enumerate(images):\n        image_path = os.path.join(TRAIN_DIR, f'{image_id[0]}/{image_id[1]}/{image_id[2]}/{image_id}.jpg')\n        image = Image.open(image_path)\n        \n        ax[i//5, i%5].imshow(image) \n        image.close()       \n        ax[i//5, i%5].axis('off')\n\n        landmark_id = train[train.id==image_id.split('.')[0]].landmark_id.values[0]\n        ax[i//5, i%5].set_title(f\"ID: {image_id.split('.')[0]}\\nLandmark_id: {landmark_id}\", fontsize=\"12\")\n\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"samples = train.sample(25).id.values\ndisplay_images(samples)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Visualizing landmarks with most number of images**"},{"metadata":{"trusted":true},"cell_type":"code","source":"samples = train[train.landmark_id == 138982].sample(25).id.values\n\n\ndisplay_images(samples)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lands = pd.DataFrame(train.landmark_id.value_counts())\nlands.reset_index(inplace=True)\nlands.columns = ['landmark_id','count']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of classes {}\".format(lands.shape[0]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Total of examples in train set = \",lands['count'].sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"NUM_THRESHOLD = 50\ntop_lands = set(lands[lands['count'] >= NUM_THRESHOLD]['landmark_id'])\nprint(\"Number of TOP classes {}\".format(len(top_lands)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_train = train[train['landmark_id'].isin(top_lands)]\nprint(\"Total of examples in subset of train: {}\".format(new_train.shape[0]))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Graphical Visualization of Landmarks Vs. Counts"},{"metadata":{"trusted":true},"cell_type":"code","source":"ax = lands['count'].plot(loglog=True, grid=True)\nax.set(xlabel=\"Landmarks\", ylabel=\"Count\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"**#References -**\nhttps://www.kaggle.com/rohitsingh9990/glr-eda-all-you-need-to-know/data\n\nhttps://www.kaggle.com/rsmits/keras-landmark-or-non-landmark-identification\n\nhttps://www.kaggle.com/codename007/a-very-extensive-landmark-exploratory-analysis"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}