{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Import libraries","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true,"_kg_hide-output":true,"_kg_hide-input":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nfrom os import listdir\nimport cv2\nimport matplotlib.pyplot as plt\nimport glob\n%matplotlib inline  \n# To store resultimg plots/graphs in the notebook document below the respective code cells\n\n!pip install chart_studio\nimport plotly.express as px\nimport chart_studio.plotly as py\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nimport cufflinks\n#Required to apply plotly\ncufflinks.go_offline()\ncufflinks.set_config_file(world_readable=True, theme='pearl')\n\nimport seaborn as sns\nsns.set(style='whitegrid')\n\nimport pydicom\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nplt.style.use('fivethirtyeight')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(os.listdir('../input/landmark-recognition-2020/'))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Training Data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"BASE_DIR = '../input/landmark-recognition-2020/'\n\ntrain_df = pd.read_csv(f'{BASE_DIR}train.csv')\nsample_df = pd.read_csv(f'{BASE_DIR}sample_submission.csv')\n\nprint('Number of training examples {}'.format(train_df.shape[0]))\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('sample_submission shape {}'.format(sample_df.shape))\nsample_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Train data info","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print('## train_info ##')\nprint(train_df.info())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Clearly we don't have any missing values","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## Number of unique landmarks","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"\nlandmarks = len(train_df['landmark_id'].unique())\nprint('Number of unique landmarks in train {}'.format(landmarks))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Top few landmark_ids by count')\n\nz = train_df.landmark_id.value_counts().head(10).to_frame()\nz.reset_index(inplace=True)\nz.columns=['landmark_id','count']\nz.landmark_id = z.landmark_id.apply(lambda x: f'id_{x}')\n\nz.style.background_gradient(cmap='Oranges')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Let's visualize some distributions","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# distribution of landmark_ids\n\ntrain_df['landmark_id'].value_counts().sort_values(ascending=False)\\\n.iplot(kind='barh',\n      xTitle='Count',\n      yTitle='landmark_id',\n      linecolor='black',\n      opacity=0.7,\n      color='orange',\n      theme='pearl',\n      bargap=0,\n      gridcolor='white',\n      title='[Interactive] Distribution of landmark_ids from training set')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# distribution of top few landmark_ids based on count\n\nplt.figure(figsize=(14,5))\nplt.title('Top few landmark_id(s) based on count')\n\nsns.set_color_codes(\"pastel\")\nsns.barplot(x='landmark_id', y='count', data=z,)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Bottom few landmark_ids based on count')\n\nz_ = train_df.landmark_id.value_counts().tail(10).to_frame()\nz_.reset_index(inplace=True)\nz_.columns=['landmark_id','count']\nz_.landmark_id = z_.landmark_id.apply(lambda x: f'id_{x}')\n\nz_.style.background_gradient(cmap='Oranges')\n#few landmark_ids with least count","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# distribution of bottom few landmark_ids based on count\n\nplt.figure(figsize=(14,5))\nplt.title('Bottom few landmark_id(s) based on count')\n\nsns.set_color_codes(\"pastel\")\nsns.barplot(x='landmark_id', y='count', data=z_,)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#density plot\n\nplt.figure(figsize=(9,5))\nplt.title('landmark_id distribution')\nplt.ylabel('Density')\nsns.distplot(train_df.landmark_id, label='Train landmark_ids',color='#fdc029')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Scatter plot for Number of images for each landmark_id","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#scatter plot\ntemp = train_df.landmark_id.value_counts().to_frame()\ntemp.reset_index(inplace=True)\ntemp.columns=['landmark_id','count']\n\nplt.figure(figsize=(14,8))\nsns.scatterplot(x='landmark_id', y='count', data=temp)\nplt.ylabel('# of images')\nplt.xlabel('landmark id')\nplt.title('Number of images for each landmark category')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Count of landmark_ids whick are in less than 100 images {}'.format(len(temp[temp['count']<100])))\npercentage = len(temp[temp['count']<100])/landmarks * 100\nprint('{0:.2f}% of landmark_ids with less than 100 reference images'.format(percentage))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Lets plot some random images from train","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_list = glob.glob('../input/landmark-recognition-2020/train/*/*/*/*')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, axes = plt.subplots(3, 4, figsize=(48, 20))\nrnd = np.random.choice(100)\n\ncurr_row = 0\nfor i in range(12):\n    image = cv2.imread(train_list[i+rnd])\n    \n    col = i%4\n    axes[curr_row, col].imshow(image)\n    if col == 3:\n        curr_row += 1","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### More is coming...","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"### References:\n[https://www.kaggle.com/huangxiaoquan/google-landmarks-v2-exploratory-data-analysis-eda/notebook](https://www.kaggle.com/huangxiaoquan/google-landmarks-v2-exploratory-data-analysis-eda/notebook)\n[https://www.kaggle.com/codename007/a-very-extensive-landmark-exploratory-analysis](https://www.kaggle.com/codename007/a-very-extensive-landmark-exploratory-analysis)","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# If you like my kernel, do upvote :)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}