{"cells":[{"metadata":{},"cell_type":"markdown","source":"We use plotly to perform some basic EDA on the classes \n\nThen we check some sample images from the top 3 classes and start to wonder what are the media contributed by ETH (Zurich) ! \n\n\nInspiration:\n\nhttps://www.kaggle.com/seriousran/google-landmark-retrieval-2020-eda/notebook\n\nhttps://www.kaggle.com/sudeepshouche/identify-landmark-name-from-landmark-id\n","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport glob\nimport cv2\nimport numpy as np \nimport pandas as pd \n\nimport matplotlib.pyplot as plt\n\nfrom scipy import stats\n\n# Load plotly related packages\nfrom plotly.offline import init_notebook_mode, iplot, plot\nimport plotly as py\ninit_notebook_mode(connected=True)\nimport plotly.graph_objs as go\n\n%matplotlib inline\n%config InlineBackend.figure_format = 'retina'","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# Read train csv and human-radable categories\ntrain_df = pd.read_csv('../input/landmark-retrieval-2020/train.csv')\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Download csn and get classes\nurl = 'https://s3.amazonaws.com/google-landmark/metadata/train_label_to_category.csv'\nclasses = pd.read_csv(url, index_col = 'landmark_id', encoding='latin', engine='python') #['category'].to_dict()\nclasses['classes'] = classes['category'].apply(lambda x : x.replace('http://commons.wikimedia.org/wiki/Category:', ''))\nclasses.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_cnt = pd.DataFrame(train_df['landmark_id'].value_counts(False))\nclass_cnt.rename(columns={'landmark_id':'count'}, inplace=True)\nclass_cnt.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"classNameCnt = class_cnt.merge(classes, left_index=True, right_index=True)\nclassNameCnt.head(10)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"# Select top 20 classes\n\ntop20 = classNameCnt.head(20)\n\ntrace = go.Bar(\n    x=top20['classes'],\n    y=top20['count'],\n    marker=dict(color = 'rgba(255, 17, 25, 0.8)')\n)\n\ndata = [trace]\nlayout = go.Layout(title='Top 20 class names', \n                   yaxis = dict(title = '# of images in train set')\n                  )\n\nfig = go.Figure(data=data, layout=layout)\nfig['layout']['xaxis'].update(dict(title = 'Classes', \n                                   tickfont = dict(size = 12)))\nfig = go.Figure(data = data, layout = layout)\niplot(fig)\n\n#write image to file\n#fig.write_image('top20classes.jpeg')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_list = glob.glob('../input/landmark-retrieval-2020/train/*/*/*/*')\ntest_list = glob.glob('../input/landmark-retrieval-2020/test/*/*/*/*')\nindex_list = glob.glob('../input/landmark-retrieval-2020/index/*/*/*/*')\n\nprint(\"Train images: \", len(train_list) )\nprint(\"Test images: \", len(test_list))\nprint(\"Index images: \", len(index_list))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Lets check some sample images from the different classes. We will check for the top 3 classes, viz:\n\n1. Media_contributed_by_the_ETH, Bibliothek\n2. Corktown, Toronto\n3. Noraduz Cemetery","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"# Function to display 12 images \n\ndef display_sample(sample_df):\n    plt.rcParams[\"axes.grid\"] = False\n    f, axarr = plt.subplots(4, 3, figsize=(24, 22))\n\n    curr_row = 0\n    for i in range(12):\n        imageFile = sample.iloc[i]['id']\n        path = \"../input/landmark-retrieval-2020/train/\"+imageFile[0]+\"/\"+imageFile[1]+\"/\"+imageFile[2]+\"/\"+imageFile+'.jpg'\n        example = cv2.imread(path)\n        example = example[:,:,::-1]\n\n        col = i%4\n        axarr[col, curr_row].imshow(example)\n        #cv2.imwrite(imageFile + '.jpg', example)\n        if col == 3:\n            curr_row += 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Lets check some random images from the largest class\n\nsample = train_df[train_df['landmark_id'] == 138982].sample(12) #.reset_index(inplace=True)\ndisplay_sample(sample)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Lets check some random images from the largest class (Corktown Toronto)\n\nsample = train_df[train_df['landmark_id'] == 126637].sample(12) #.reset_index(inplace=True)\ndisplay_sample(sample)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Lets check some random images from the largest class (Noraduz Cemetry)\n\nsample = train_df[train_df['landmark_id'] == 20409].sample(12) #.reset_index(inplace=True)\ndisplay_sample(sample)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}