{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":29762,"databundleVersionId":2541532,"sourceType":"competition"},{"sourceId":11270196,"sourceType":"datasetVersion","datasetId":7038549}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport glob\nimport cv2\n\nfrom tqdm import tqdm_notebook as tqdm\nimport matplotlib.image as mpimg\n\nimport plotly.graph_objects as go","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T11:29:40.300514Z","iopub.execute_input":"2025-04-05T11:29:40.300867Z","iopub.status.idle":"2025-04-05T11:29:43.684044Z","shell.execute_reply.started":"2025-04-05T11:29:40.300829Z","shell.execute_reply":"2025-04-05T11:29:43.682584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/landmark-recognition-2021/train.csv')\n\ntest_list = glob.glob('/kaggle/input/landmark-recognition-2021/test/*/*/*/*')\ntrain_list= glob.glob('/kaggle/input/landmark-recognition-2021/train/*/*/*/*')\n\nlandmark_info = pd.read_csv('/kaggle/input/mappings-gldv2/train_label_to_category.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-05T07:42:47.985402Z","iopub.execute_input":"2025-04-05T07:42:47.985744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(landmark_info[['landmark_id', 'category']].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T10:10:05.693859Z","iopub.status.idle":"2025-04-04T10:10:05.694306Z","shell.execute_reply":"2025-04-04T10:10:05.694146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"landmark_info.to_csv('train_label_to_hierarchical_modified.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T16:23:33.374413Z","iopub.execute_input":"2025-04-03T16:23:33.374806Z","iopub.status.idle":"2025-04-03T16:23:33.817349Z","shell.execute_reply.started":"2025-04-03T16:23:33.374778Z","shell.execute_reply":"2025-04-03T16:23:33.816296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"landmark_info['landmark_id'] = landmark_info['landmark_id'] + 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T16:22:38.491293Z","iopub.execute_input":"2025-04-03T16:22:38.491666Z","iopub.status.idle":"2025-04-03T16:22:38.497427Z","shell.execute_reply.started":"2025-04-03T16:22:38.491638Z","shell.execute_reply":"2025-04-03T16:22:38.496237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"landmark_info_modified = pd.read_csv('/kaggle/working/train_label_to_hierarchical_modified.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T16:24:24.766490Z","iopub.execute_input":"2025-04-03T16:24:24.766913Z","iopub.status.idle":"2025-04-03T16:24:25.089355Z","shell.execute_reply.started":"2025-04-03T16:24:24.766870Z","shell.execute_reply":"2025-04-03T16:24:25.088250Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ***Training Set*** #","metadata":{}},{"cell_type":"code","source":"print(\"Train data shape -  rows:\",train_df.shape[0],\" columns:\", train_df.shape[1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T15:33:41.729207Z","iopub.execute_input":"2025-04-03T15:33:41.729535Z","iopub.status.idle":"2025-04-03T15:33:41.735279Z","shell.execute_reply.started":"2025-04-03T15:33:41.729510Z","shell.execute_reply":"2025-04-03T15:33:41.734086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T14:24:41.176018Z","iopub.execute_input":"2025-04-03T14:24:41.176323Z","iopub.status.idle":"2025-04-03T14:24:41.202238Z","shell.execute_reply.started":"2025-04-03T14:24:41.176298Z","shell.execute_reply":"2025-04-03T14:24:41.201240Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"landmark_counts = train_df['landmark_id'].value_counts().head(50)\nlandmark_counts_least = train_df['landmark_id'].value_counts().tail(50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T14:24:51.318313Z","iopub.execute_input":"2025-04-03T14:24:51.318638Z","iopub.status.idle":"2025-04-03T14:24:51.358916Z","shell.execute_reply.started":"2025-04-03T14:24:51.318611Z","shell.execute_reply":"2025-04-03T14:24:51.357862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.histplot(train_df['landmark_id'].value_counts(), bins=50, kde=True)\nplt.title('Distribution of Landmark Appearances')\nplt.xlabel('Number of Appearances')\nplt.ylabel('Count')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T14:25:36.182441Z","iopub.execute_input":"2025-04-03T14:25:36.182967Z","iopub.status.idle":"2025-04-03T14:25:36.975285Z","shell.execute_reply.started":"2025-04-03T14:25:36.182930Z","shell.execute_reply":"2025-04-03T14:25:36.974130Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the chart above, it can be seen that most landmarks fall on to the left sided tail, which means\n=> Most landmarks only have a few images.","metadata":{}},{"cell_type":"code","source":"sns.set()\nplt.title('Training set: number of images per class(line plot)')\nlandmarks_fold = pd.DataFrame(train_df['landmark_id'].value_counts())\nlandmarks_fold.reset_index(inplace=True)\nlandmarks_fold.columns = ['landmark_id','count']\nax = landmarks_fold['count'].plot(logy=True, grid=True)\nlocs, labels = plt.xticks()\nplt.setp(labels, rotation=30)\nax.set(xlabel=\"Landmarks\", ylabel=\"Number of images\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T14:27:26.326509Z","iopub.execute_input":"2025-04-03T14:27:26.326912Z","iopub.status.idle":"2025-04-03T14:27:27.256751Z","shell.execute_reply.started":"2025-04-03T14:27:26.326878Z","shell.execute_reply":"2025-04-03T14:27:27.255627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (8, 2))\nplt.title('Landmark id density plot')\nsns.kdeplot(train_df['landmark_id'], color=\"tomato\", shade=True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T14:27:52.359787Z","iopub.execute_input":"2025-04-03T14:27:52.360205Z","iopub.status.idle":"2025-04-03T14:27:59.439504Z","shell.execute_reply.started":"2025-04-03T14:27:52.360172Z","shell.execute_reply":"2025-04-03T14:27:59.438361Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Show some random images in the train set ##","metadata":{}},{"cell_type":"code","source":"num_images = min(12, len(test_list))\n\nplt.rcParams[\"axes.grid\"] = False\nf, axarr = plt.subplots(4, 3, figsize=(24, 22))\n\ncurr_row = 0\nfor i in range(num_images):  # Loop over the number of images, up to 12 or length of the list\n    example = cv2.imread(test_list[i])\n    example = example[:,:,::-1]  \n    \n    col = i % 3 \n    axarr[curr_row, col].imshow(example)\n    axarr[curr_row, col].axis('off')  # Hide axes for cleaner images\n    \n    if col == 2:  # Move to next row after 3 columns\n        curr_row += 1\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T14:36:24.016770Z","iopub.execute_input":"2025-04-03T14:36:24.017179Z","iopub.status.idle":"2025-04-03T14:36:27.465247Z","shell.execute_reply.started":"2025-04-03T14:36:24.017146Z","shell.execute_reply":"2025-04-03T14:36:27.464085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.rcParams[\"axes.grid\"] = True\nf, axarr = plt.subplots(6, 5, figsize=(24, 22))\n\ncurr_row = 0\nfor i in range(30):\n    example = cv2.imread(train_list[i])\n    example = example[:,:,::-1]\n    \n    col = i%6\n    axarr[col, curr_row].imshow(example)\n    if col == 5:\n        curr_row += 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T14:36:41.189296Z","iopub.execute_input":"2025-04-03T14:36:41.189656Z","iopub.status.idle":"2025-04-03T14:36:51.072040Z","shell.execute_reply.started":"2025-04-03T14:36:41.189622Z","shell.execute_reply":"2025-04-03T14:36:51.070260Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_landmark_ids = train_df[~train_df['landmark_id'].isin(landmark_info['landmark_id'])]\n\n# Count how many unique landmark_id values are missing\nmissing_landmark_count = missing_landmark_ids['landmark_id'].nunique()\n\nprint(\"Total number of missing landmark IDs:\", missing_landmark_count)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T16:25:11.464970Z","iopub.execute_input":"2025-04-03T16:25:11.465305Z","iopub.status.idle":"2025-04-03T16:25:11.489809Z","shell.execute_reply.started":"2025-04-03T16:25:11.465276Z","shell.execute_reply":"2025-04-03T16:25:11.488726Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Top 10, top 50, top 100 landmarks with the highest appearances count","metadata":{}},{"cell_type":"code","source":"temp = pd.DataFrame(train_df.landmark_id.value_counts().head(10))\ntemp.reset_index(inplace=True)\ntemp.columns = ['landmark_id', 'count']\ntemp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T14:28:58.049160Z","iopub.execute_input":"2025-04-03T14:28:58.049491Z","iopub.status.idle":"2025-04-03T14:28:58.087458Z","shell.execute_reply.started":"2025-04-03T14:28:58.049465Z","shell.execute_reply":"2025-04-03T14:28:58.086410Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"temp = pd.DataFrame(train_df.landmark_id.value_counts().head(50))\ntemp.reset_index(inplace=True)\ntemp.columns = ['landmark_id','count']\ntemp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T15:57:29.792900Z","iopub.execute_input":"2025-04-03T15:57:29.793222Z","iopub.status.idle":"2025-04-03T15:57:29.829664Z","shell.execute_reply.started":"2025-04-03T15:57:29.793199Z","shell.execute_reply":"2025-04-03T15:57:29.828633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"temp = pd.DataFrame(train_df.landmark_id.value_counts().head(100))\ntemp.reset_index(inplace=True)\ntemp.columns = ['landmark_id', 'count']\n\n# Define the chunk size\nchunk_size = 50\n\n# Print the data in chunks\nfor start in range(50, len(temp), chunk_size):\n    end = min(start + chunk_size, len(temp))  # To ensure we don't go out of bounds\n    print(f\"Showing rows {start} to {end - 1}:\")\n    print(temp.iloc[start:end])\n    print(\"\\n\" + \"=\"*50 + \"\\n\")  # Se","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T18:16:03.454177Z","iopub.execute_input":"2025-04-03T18:16:03.454568Z","iopub.status.idle":"2025-04-03T18:16:03.487252Z","shell.execute_reply.started":"2025-04-03T18:16:03.454533Z","shell.execute_reply":"2025-04-03T18:16:03.485988Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The corresponding top 50 landmarks with their url ","metadata":{}},{"cell_type":"code","source":"sns.set()\n# plt.figure(figsize=(9, 8))\nplt.title('Most frequent landmarks')\nsns.set_color_codes(\"pastel\")\nsns.barplot(x=\"landmark_id\", y=\"count\", data=temp,\n            label=\"Count\")\nlocs, labels = plt.xticks()\nplt.setp(labels, rotation=45)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T14:29:09.914419Z","iopub.execute_input":"2025-04-03T14:29:09.914784Z","iopub.status.idle":"2025-04-03T14:29:10.228121Z","shell.execute_reply.started":"2025-04-03T14:29:09.914755Z","shell.execute_reply":"2025-04-03T14:29:10.227091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.max_colwidth', None) \nmerged_data = pd.merge(temp, landmark_info_modified, on='landmark_id', how='left')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T16:24:53.533085Z","iopub.execute_input":"2025-04-03T16:24:53.533419Z","iopub.status.idle":"2025-04-03T16:24:53.560867Z","shell.execute_reply.started":"2025-04-03T16:24:53.533384Z","shell.execute_reply":"2025-04-03T16:24:53.559748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(merged_data[['landmark_id', 'category']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T16:24:55.635522Z","iopub.execute_input":"2025-04-03T16:24:55.635937Z","iopub.status.idle":"2025-04-03T16:24:55.644312Z","shell.execute_reply.started":"2025-04-03T16:24:55.635903Z","shell.execute_reply":"2025-04-03T16:24:55.643140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(landmark_info['landmark_id'].nunique())\nprint(train_df['landmark_id'].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T16:25:05.663460Z","iopub.execute_input":"2025-04-03T16:25:05.663855Z","iopub.status.idle":"2025-04-03T16:25:05.689641Z","shell.execute_reply.started":"2025-04-03T16:25:05.663786Z","shell.execute_reply":"2025-04-03T16:25:05.688422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_landmark_ids = train_df[~train_df['landmark_id'].isin(landmark_info_1['landmark_id'])]\n\nprint(\"Missing Landmark IDs in Metadata:\")\nprint(missing_landmark_ids[['id', 'landmark_id']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T16:25:08.946427Z","iopub.execute_input":"2025-04-03T16:25:08.946794Z","iopub.status.idle":"2025-04-03T16:25:08.974362Z","shell.execute_reply.started":"2025-04-03T16:25:08.946763Z","shell.execute_reply":"2025-04-03T16:25:08.973179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"landmark_id_to_search = 138982\nresult = landmark_info[landmark_info['landmark_id'] == landmark_id_to_search]\n\n# Display the result\nprint(result)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T16:26:48.311722Z","iopub.execute_input":"2025-04-03T16:26:48.312129Z","iopub.status.idle":"2025-04-03T16:26:48.321160Z","shell.execute_reply.started":"2025-04-03T16:26:48.312097Z","shell.execute_reply":"2025-04-03T16:26:48.319850Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"landmark_id_to_search = 1924\ncount_images = train_df[train_df['landmark_id'] == landmark_id_to_search].shape[0]\n\n# Display the count\nprint(f\"Number of images with landmark_id {landmark_id_to_search}: {count_images}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T16:14:41.462064Z","iopub.execute_input":"2025-04-03T16:14:41.462398Z","iopub.status.idle":"2025-04-03T16:14:41.470849Z","shell.execute_reply.started":"2025-04-03T16:14:41.462373Z","shell.execute_reply":"2025-04-03T16:14:41.469750Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Display the images of the landmark_id ##","metadata":{}},{"cell_type":"code","source":"landmark_id_to_search = 138982\nimage_ids_with_landmark = train_df[train_df['landmark_id'] == landmark_id_to_search]['id'].tolist()\nresult = landmark_info[landmark_info['landmark_id'] == landmark_id_to_search]\n\n# Display the result\nprint(result)\n\nimage_paths_to_display = [train_list[i] for i in range(len(train_list)) if train_df['id'][i] in image_ids_with_landmark]\n\n# Set up the plot to display images\nplt.rcParams[\"axes.grid\"] = True\nf, axarr = plt.subplots(6, 5, figsize=(24, 22))\n\ncurr_row = 0\nnum_images_to_display = min(30, len(image_paths_to_display))  # Display at most 30 images, if available\n\nfor i in range(num_images_to_display):\n    # Read the image (make sure to read the correct path)\n    example = cv2.imread(image_paths_to_display[i])\n    example = cv2.cvtColor(example, cv2.COLOR_BGR2RGB)  # Convert BGR to RGB for matplotlib\n    \n    # Calculate row and column position in the subplot grid\n    col = i % 5  # 5 columns\n    axarr[curr_row, col].imshow(example)\n    axarr[curr_row, col].axis('off')  # Turn off axis for better visualization\n    \n    # Move to the next row after 5 images\n    if col == 4:\n        curr_row += 1\n\n# Show the plot with the images\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T16:26:56.770734Z","iopub.execute_input":"2025-04-03T16:26:56.771113Z","iopub.status.idle":"2025-04-03T16:30:31.624346Z","shell.execute_reply.started":"2025-04-03T16:26:56.771082Z","shell.execute_reply":"2025-04-03T16:30:31.622708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"landmark_id_0_rows = train_df[train_df['landmark_id'] == '0']\n\n# Display the rows with landmark_id = 0\nprint(landmark_id_0_rows)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T16:20:51.967439Z","iopub.execute_input":"2025-04-03T16:20:51.968073Z","iopub.status.idle":"2025-04-03T16:20:51.975703Z","shell.execute_reply.started":"2025-04-03T16:20:51.968032Z","shell.execute_reply":"2025-04-03T16:20:51.974484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"temp = pd.DataFrame(train_df.landmark_id.value_counts().tail(10))\ntemp.reset_index(inplace=True)\ntemp.columns = ['landmark_id', 'count']\ntemp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T14:29:23.177113Z","iopub.execute_input":"2025-04-03T14:29:23.177465Z","iopub.status.idle":"2025-04-03T14:29:23.210863Z","shell.execute_reply.started":"2025-04-03T14:29:23.177434Z","shell.execute_reply":"2025-04-03T14:29:23.209834Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Rewritten a bit :D ##","metadata":{}},{"cell_type":"code","source":"INPUT_PATH = os.path.join('..', 'input')\nDATASET_PATH = os.path.join(INPUT_PATH, 'landmark-recognition-2021')\nTRAIN_IMAGE_PATH = os.path.join(DATASET_PATH, 'train')\nTEST_IMAGE_PATH = os.path.join(DATASET_PATH, 'test')\nTRAIN_CSV_PATH = os.path.join(DATASET_PATH, 'train.csv')\nSUBMISSION_CSV_PATH = os.path.join(DATASET_PATH, 'sample_submission.csv')\n\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\nprint(f\"{'--'*20} \\n SNIPPET OF TRAINING DATA: \\n {train_df.head()} \\n {'--'*20} \\n Number of rows in train data: {train_df.shape[0]} \\n {'--'*20}\")\n\nsubmission_df = pd.read_csv(SUBMISSION_CSV_PATH)\nprint(f\"{'--'*20} \\n SNIPPET OF TEST DATA: \\n {submission_df.head()} \\n {'--'*20} \\n Number of rows in test data: {submission_df.shape[0]} \\n {'--'*20}\")\n\nprint(f\"EXAMPLE FOR LANDMARK-LABEL MAPPING FOR [17660ef415d37059.jpg] \\n FOLDER STRUCTURE: \\n |---1 \\n \\t |---7 \\n \\t \\t |---6 \\n \\t \\t \\t|---<17660ef415d37059.jpg>\")\ni=0\nprint(f\"Image name: {train_df['id'].iloc[i]}\")\nprint(f\"First folder to look inside: {train_df['id'][i][0]}\")\nprint(f\"Second folder to look inside: {train_df['id'][i][1]}\")\nprint(f\"Second folder to look inside: {train_df['id'][i][2]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T17:54:08.360290Z","iopub.execute_input":"2025-04-03T17:54:08.362874Z","iopub.status.idle":"2025-04-03T17:54:09.678183Z","shell.execute_reply.started":"2025-04-03T17:54:08.362787Z","shell.execute_reply":"2025-04-03T17:54:09.676835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"{'---'*20} \\n Creating training data mapping \\n {'---'*20}\")\ndata_label_dict = {'image': [], 'target': []}\nfor i in tqdm(range(train_df.shape[0])):\n    data_label_dict['image'].append(\n        TRAIN_IMAGE_PATH + '/' +\n        train_df['id'][i][0] + '/' + \n        train_df['id'][i][1]+ '/' +\n        train_df['id'][i][2]+ '/' +\n        train_df['id'][i] + \".jpg\")\n    data_label_dict['target'].append(\n        train_df['landmark_id'][i])\n#Convert to dataframe\ntrain_pathlabel_df = pd.DataFrame(data_label_dict)\nprint(train_pathlabel_df.head())\n    \nprint(f\"{'---'*20} \\n Creating test data mapping \\n {'---'*20}\")\ndata_label_dict = {'image': []}\nfor i in tqdm(range(submission_df.shape[0])):\n    data_label_dict['image'].append(\n        TEST_IMAGE_PATH + '/' +\n        submission_df['id'][i][0] + '/' + \n        submission_df['id'][i][1]+ '/' +\n        submission_df['id'][i][2]+ '/' +\n        submission_df['id'][i] + \".jpg\")\n\ntest_pathlabel_df = pd.DataFrame(data_label_dict)\nprint(test_pathlabel_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T17:54:45.191620Z","iopub.execute_input":"2025-04-03T17:54:45.192008Z","iopub.status.idle":"2025-04-03T17:55:37.597914Z","shell.execute_reply.started":"2025-04-03T17:54:45.191977Z","shell.execute_reply":"2025-04-03T17:55:37.596517Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Print images of some landmark_id ##","metadata":{}},{"cell_type":"code","source":"print(f\"The data has {train_pathlabel_df['target'].nunique()} unique classes\")\n\nfor tar in train_pathlabel_df['target'].unique()[:4]: \n    #Subset to just that target \n    label_df = train_pathlabel_df[train_pathlabel_df['target']==tar].reset_index()\n    cols = 2\n    rows = 2\n    fig = plt.figure(figsize = (4*cols - 1, 4.5*rows - 1))\n    for c in range(cols):\n        for r in range(rows):\n            ax = fig.add_subplot(rows, cols, c*rows + r + 1)\n            img = mpimg.imread(label_df['image'][c+r])\n            ax.imshow(img)#label_df[][c+r])\n    fig.suptitle(f\"Images corresponding to label [{tar}] with a total of {label_df.shape[0]} images available\")\n    plt.show()\n    plt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T17:57:22.387396Z","iopub.execute_input":"2025-04-03T17:57:22.387794Z","iopub.status.idle":"2025-04-03T17:57:26.588547Z","shell.execute_reply.started":"2025-04-03T17:57:22.387763Z","shell.execute_reply":"2025-04-03T17:57:26.587360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_landmark_id = 138982\n\n# Filter the DataFrame for the specific target landmark ID\nlabel_df = train_pathlabel_df[train_pathlabel_df['target'] == target_landmark_id].reset_index()\n\n# Set up grid for plotting images\ncols = 2\nrows = 2\nfig = plt.figure(figsize = (4*cols - 1, 4.5*rows - 1))\n\n# Loop through the images and plot\nfor c in range(cols):\n    for r in range(rows):\n        ax = fig.add_subplot(rows, cols, c*rows + r + 1)\n        if c + r < len(label_df):  # Ensure we don't go out of bounds\n            img = mpimg.imread(label_df['image'][c+r])\n            ax.imshow(img)\n            ax.axis('off')  # Hide axes for better visualization\n        else:\n            ax.axis('off')  # If there are fewer images, hide the empty subplot\n\n# Title for the figure\nfig.suptitle(f\"Images corresponding to landmark_id [{target_landmark_id}] with a total of {label_df.shape[0]} images available\")\n\n# Show the images\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T17:58:47.675969Z","iopub.execute_input":"2025-04-03T17:58:47.676303Z","iopub.status.idle":"2025-04-03T17:58:48.449534Z","shell.execute_reply.started":"2025-04-03T17:58:47.676276Z","shell.execute_reply":"2025-04-03T17:58:48.448226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ntarget_landmark_id = 138982\n\n# Filter the DataFrame for the specific target landmark_id\nlabel_df = train_df[train_df['landmark_id'] == target_landmark_id].reset_index()\n\n# Set up grid for plotting images\ncols = 2\nrows = 2\nfig = plt.figure(figsize=(4 * cols - 1, 4.5 * rows - 1))\n\n# Loop through the images and plot\nfor c in range(cols):\n    for r in range(rows):\n        ax = fig.add_subplot(rows, cols, c * rows + r + 1)\n        if c + r < len(label_df):  # Ensure we don't go out of bounds\n            img_id = label_df['id'][c + r]  # Image ID from the DataFrame\n            \n            # Construct the image file path dynamically using the first three characters of the image ID\n            first_char, second_char, third_char = img_id[:3]\n            img_full_path = f'/kaggle/input/landmark-recognition-2021/train/{first_char}/{second_char}/{third_char}/{img_id}.jpg'\n            \n            # Get the image size in bytes and convert it to KB/MB\n            try:\n                img_size_bytes = os.path.getsize(img_full_path)\n                img_size_kb = img_size_bytes / 1024  # Convert to KB\n                img_size_mb = img_size_kb / 1024  # Convert to MB\n\n                # Print the image size in the console\n                print(f\"Image {img_id} size: {img_size_kb:.2f} KB ({img_size_mb:.2f} MB)\")\n\n                # Read and display the image\n                img = mpimg.imread(img_full_path)\n                ax.imshow(img)\n                ax.set_title(f\"Size: {img_size_kb:.2f} KB\")  # Show the size on the image\n                ax.axis('off')  # Hide axes for better visualization\n            except FileNotFoundError:\n                ax.axis('off')  # If the image is not found, hide the subplot\n                print(f\"Image {img_id} not found at {img_full_path}\")\n\n        else:\n            ax.axis('off')  # If there are fewer images, hide the empty subplot\n\n# Title for the figure\nfig.suptitle(f\"Images corresponding to landmark_id [{target_landmark_id}] with a total of {label_df.shape[0]} images available\")\n\n# Show the images\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:01:28.281527Z","iopub.execute_input":"2025-04-04T11:01:28.281863Z","iopub.status.idle":"2025-04-04T11:01:29.013846Z","shell.execute_reply.started":"2025-04-04T11:01:28.281838Z","shell.execute_reply":"2025-04-04T11:01:29.012382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_landmark_id = 1924\n\n# Filter the DataFrame for the specific target landmark_id\nlabel_df = train_df[train_df['landmark_id'] == target_landmark_id].reset_index()\n\n# Set up grid for plotting images\ncols = 2\nrows = 2\nfig = plt.figure(figsize=(4 * cols - 1, 4.5 * rows - 1))\n\n# Loop through the images and plot\nfor c in range(cols):\n    for r in range(rows):\n        ax = fig.add_subplot(rows, cols, c * rows + r + 1)\n        if c + r < len(label_df):  # Ensure we don't go out of bounds\n            img_id = label_df['id'][c + r]  # Image ID from the DataFrame\n            \n            # Construct the image file path dynamically using the first three characters of the image ID\n            first_char, second_char, third_char = img_id[:3]\n            img_full_path = f'/kaggle/input/landmark-recognition-2021/train/{first_char}/{second_char}/{third_char}/{img_id}.jpg'\n            \n            # Get the image size in bytes and convert it to KB/MB\n            try:\n                img_size_bytes = os.path.getsize(img_full_path)\n                img_size_kb = img_size_bytes / 1024  # Convert to KB\n                img_size_mb = img_size_kb / 1024  # Convert to MB\n\n                # Print the image size in the console\n                print(f\"Image {img_id} size: {img_size_kb:.2f} KB ({img_size_mb:.2f} MB)\")\n\n                # Read and display the image\n                img = mpimg.imread(img_full_path)\n                ax.imshow(img)\n                ax.set_title(f\"Size: {img_size_kb:.2f} KB\")  # Show the size on the image\n                ax.axis('off')  # Hide axes for better visualization\n            except FileNotFoundError:\n                ax.axis('off')  # If the image is not found, hide the subplot\n                print(f\"Image {img_id} not found at {img_full_path}\")\n\n        else:\n            ax.axis('off')  # If there are fewer images, hide the empty subplot\n\n# Title for the figure\nfig.suptitle(f\"Images corresponding to landmark_id [{target_landmark_id}] with a total of {label_df.shape[0]} images available\")\n\n# Show the images\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:01:41.758546Z","iopub.execute_input":"2025-04-04T11:01:41.758909Z","iopub.status.idle":"2025-04-04T11:01:42.482542Z","shell.execute_reply.started":"2025-04-04T11:01:41.758873Z","shell.execute_reply":"2025-04-04T11:01:42.481158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_sizes = []\n\n# Loop through the image IDs in the DataFrame and calculate sizes\nfor img_id in train_df['id']:\n    # Construct the image file path dynamically using the first three characters of the image ID\n    first_char, second_char, third_char = img_id[:3]\n    img_full_path = f'/kaggle/input/landmark-recognition-2021/train/{first_char}/{second_char}/{third_char}/{img_id}.jpg'\n\n    try:\n        # Get the image size in bytes\n        img_size_bytes = os.path.getsize(img_full_path)\n        image_sizes.append(img_size_bytes / 1024)  # Convert size to KB\n    except FileNotFoundError:\n        print(f\"Image {img_id} not found at {img_full_path}\")\n        image_sizes.append(0)  # If image not found, append 0 (you can handle this case as needed)\n\n# Plot the histogram of image sizes\nplt.figure(figsize=(10, 6))\nplt.hist(image_sizes, bins=50, color='skyblue', edgecolor='black')\nplt.title('Distribution of Image Sizes in the Train Set')\nplt.xlabel('Image Size (KB)')\nplt.ylabel('Frequency')\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:02:45.273931Z","iopub.execute_input":"2025-04-04T11:02:45.274357Z","iopub.status.idle":"2025-04-04T12:39:18.952461Z","shell.execute_reply.started":"2025-04-04T11:02:45.274319Z","shell.execute_reply":"2025-04-04T12:39:18.951046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"counts_df = pd.DataFrame(train_df[['landmark_id']].value_counts().reset_index())\ncounts_df.columns = ['Landmark', 'Count']\ncounts_df.sort_values('Count', ascending=False, inplace=True)\nprint(counts_df.head())\n\nfig = go.Figure(data = [go.Bar(x = counts_df[:20].index,\n                              y=counts_df[:20]['Count'],\n                              text=counts_df[:20]['Count'],\n                              textposition = 'outside')])\nfig.update_layout(title=\"Image counts across top 20 landmarks\",\n                 xaxis_title = \"Landmark id\",\n                 yaxis_title = \"Count of images\")\nfig.update_xaxes(ticktext= counts_df[:20]['Landmark'],\n                tickvals = counts_df[:20].index)\nfig.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T10:58:13.197878Z","iopub.execute_input":"2025-04-04T10:58:13.198216Z","iopub.status.idle":"2025-04-04T10:58:14.086606Z","shell.execute_reply.started":"2025-04-04T10:58:13.198188Z","shell.execute_reply":"2025-04-04T10:58:14.085186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"counts_df['Bin'] = np.where((counts_df['Count']<=10), \"(0, 10]\", \"Rest\")\ncounts_df['Bin'] = np.where((counts_df['Count']>10) & (counts_df['Count']<=20), \"(10, 20]\", counts_df['Bin'])\ncounts_df['Bin'] = np.where((counts_df['Count']>20) & (counts_df['Count']<=30), \"(20, 30]\", counts_df['Bin'])\ncounts_df['Bin'] = np.where((counts_df['Count']>30) & (counts_df['Count']<=50), \"(30, 50]\", counts_df['Bin'])\ncounts_df['Bin'] = np.where((counts_df['Count']>50) & (counts_df['Count']<=70), \"(50, 70]\", counts_df['Bin'])\ncounts_df['Bin'] = np.where((counts_df['Count']>70) & (counts_df['Count']<=100), \"(70, 100]\", counts_df['Bin'])\ncounts_df['Bin'] = np.where((counts_df['Count']>100) & (counts_df['Count']<=150), \"(100, 150]\", counts_df['Bin'])\n# counts_df['Bin'] = np.where((counts_df['Count']>=20) & (counts_df['Count']<30), \"Bin 3: 20-30\", counts_df['Bin'])\nbin_df = counts_df.groupby('Bin')['Count'].count().reset_index()\nbin_df['Bin'] = bin_df['Bin'].astype('str')\nprint(bin_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T17:59:57.392103Z","iopub.execute_input":"2025-04-03T17:59:57.392467Z","iopub.status.idle":"2025-04-03T17:59:57.441527Z","shell.execute_reply.started":"2025-04-03T17:59:57.392437Z","shell.execute_reply":"2025-04-03T17:59:57.440464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = go.Figure(data = [go.Bar(x=bin_df['Bin'],\n                               y=bin_df['Count'],\n                               text = bin_df['Count'],\n                              textposition = 'outside')])\nfig.update_layout(title='Ímage counts across bins',\n                  xaxis_title = \"Bin/interval of image counts per landmark\",\n                 yaxis_title = \"Count of images in bin\")\n# fig.update_xaxes('Bins/intervals of image counts')\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-03T18:06:22.583115Z","iopub.execute_input":"2025-04-03T18:06:22.583494Z","iopub.status.idle":"2025-04-03T18:06:22.596904Z","shell.execute_reply.started":"2025-04-03T18:06:22.583458Z","shell.execute_reply":"2025-04-03T18:06:22.595786Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"44,646 landmarks have at least 10 images.\n\nOnly 879 landmarks have more than 100 images\n","metadata":{}},{"cell_type":"markdown","source":"# Test Set #","metadata":{}},{"cell_type":"markdown","source":"There are 117,577 images in the test set.\n\nAnd some of the images are not landmark-related, portrait of a man,etc","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/mappings-gldv2/test.csv')\n\n# Set up grid for plotting images\ncols = 2\nrows = 2\nfig = plt.figure(figsize=(4 * cols - 1, 4.5 * rows - 1))\n\n# Loop through the images and plot\nfor c in range(cols):\n    for r in range(rows):\n        ax = fig.add_subplot(rows, cols, c * rows + r + 1)\n        if c + r < len(test_df):  # Ensure we don't go out of bounds\n            img_id = test_df['id'][c + r]  # Image ID from the DataFrame\n            \n            # Construct the image file path dynamically using the first three characters of the image ID\n            first_char, second_char, third_char = img_id[:3]\n            img_full_path = f'/kaggle/input/landmark-recognition-2021/test/{first_char}/{second_char}/{third_char}/{img_id}.jpg'\n            \n            # Read and display the image\n            try:\n                img = mpimg.imread(img_full_path)\n                ax.imshow(img)\n                ax.set_title(f\"ID: {img_id}\")  # Show the image ID on the title\n                ax.axis('off')  # Hide axes for better visualization\n            except FileNotFoundError:\n                ax.axis('off')  # If the image is not found, hide the subplot\n                print(f\"Image {img_id} not found at {img_full_path}\")\n\n        else:\n            ax.axis('off')  # If there are fewer images, hide the empty subplot\n\n# Title for the figure\nfig.suptitle(f\"Some images from the test set\")\n\n# Show the images\nplt.show()\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T12:39:18.954108Z","iopub.execute_input":"2025-04-04T12:39:18.954473Z","iopub.status.idle":"2025-04-04T12:39:19.214821Z","shell.execute_reply.started":"2025-04-04T12:39:18.954442Z","shell.execute_reply":"2025-04-04T12:39:19.213424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/mappings-gldv2/test.csv')\n\n# Set up grid for plotting images\ncols = 2\nrows = 2\nfig = plt.figure(figsize=(4 * cols - 1, 4.5 * rows - 1))\n\n# Loop through the images and plot\nfor c in range(cols):\n    for r in range(rows):\n        ax = fig.add_subplot(rows, cols, c * rows + r + 1)\n        if c + r < len(test_df):  # Ensure we don't go out of bounds\n            img_id = test_df['id'][c + r]  # Image ID from the DataFrame\n            \n            # Construct the image file path dynamically using the first three characters of the image ID\n            first_char, second_char, third_char = img_id[:3]\n            img_full_path = f'/kaggle/input/landmark-recognition-2021/test/{first_char}/{second_char}/{third_char}/{img_id}.jpg'\n            \n            # Print the path to debug the issue\n            print(f\"Trying to load image: {img_full_path}\")\n            \n            # Read and display the image\n            try:\n                img = mpimg.imread(img_full_path)\n                ax.imshow(img)\n                ax.set_title(f\"ID: {img_id}\")  # Show the image ID on the title\n                ax.axis('off')  # Hide axes for better visualization\n            except FileNotFoundError:\n                ax.axis('off')  # If the image is not found, hide the subplot\n                print(f\"Image {img_id} not found at {img_full_path}\")\n\n        else:\n            ax.axis('off')  # If there are fewer images, hide the empty subplot\n\n# Title for the figure\nfig.suptitle(f\"Some images from the test set\")\n\n# Show the images\nplt.show()\nplt.close()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}