{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport os\n\nBASE_PATH = \"../input/landmark-recognition-2021\"\n\ntrain_df = pd.read_csv(BASE_PATH + \"/train.csv\")\nsubmission_df = pd.read_csv(BASE_PATH + \"/sample_submission.csv\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-16T15:26:43.831564Z","iopub.execute_input":"2021-08-16T15:26:43.831948Z","iopub.status.idle":"2021-08-16T15:26:45.062886Z","shell.execute_reply.started":"2021-08-16T15:26:43.831900Z","shell.execute_reply":"2021-08-16T15:26:45.062052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.axes_grid1 import ImageGrid\nfrom PIL import Image\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2021-08-16T16:05:52.918557Z","iopub.execute_input":"2021-08-16T16:05:52.918900Z","iopub.status.idle":"2021-08-16T16:05:53.963151Z","shell.execute_reply.started":"2021-08-16T16:05:52.918870Z","shell.execute_reply":"2021-08-16T16:05:53.962153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.shape)\nprint(submission_df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-08-16T15:26:54.226058Z","iopub.execute_input":"2021-08-16T15:26:54.226511Z","iopub.status.idle":"2021-08-16T15:26:54.231689Z","shell.execute_reply.started":"2021-08-16T15:26:54.226483Z","shell.execute_reply":"2021-08-16T15:26:54.230548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.head())","metadata":{"execution":{"iopub.status.busy":"2021-08-16T14:51:32.463660Z","iopub.execute_input":"2021-08-16T14:51:32.464167Z","iopub.status.idle":"2021-08-16T14:51:32.492657Z","shell.execute_reply.started":"2021-08-16T14:51:32.464118Z","shell.execute_reply":"2021-08-16T14:51:32.491463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(submission_df.head())","metadata":{"execution":{"iopub.status.busy":"2021-08-16T15:27:12.992021Z","iopub.execute_input":"2021-08-16T15:27:12.992380Z","iopub.status.idle":"2021-08-16T15:27:13.010302Z","shell.execute_reply.started":"2021-08-16T15:27:12.992349Z","shell.execute_reply":"2021-08-16T15:27:13.009269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plot some samples and see the resolutions","metadata":{}},{"cell_type":"code","source":"def image_grid3x3(image_array, landmarks):\n    fig = plt.figure(figsize=(12., 12.))\n    grid = ImageGrid(fig, 111,\n                     nrows_ncols=(3, 3),\n                     axes_pad=1)\n    \n    for idx, (ax, im) in enumerate(zip(grid, image_array)):\n        ax.imshow(im)\n        ax.set_title(landmarks[idx])\n        ax.set_xlabel(f'{im.shape}')\n        \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-16T16:01:36.064422Z","iopub.execute_input":"2021-08-16T16:01:36.064945Z","iopub.status.idle":"2021-08-16T16:01:36.070921Z","shell.execute_reply.started":"2021-08-16T16:01:36.064899Z","shell.execute_reply":"2021-08-16T16:01:36.070111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_img_path(img_id):\n    return \"/\".join([char for char in img_id[:3]]) + \"/\" + img_id + \".jpg\"","metadata":{"execution":{"iopub.status.busy":"2021-08-16T16:01:36.280409Z","iopub.execute_input":"2021-08-16T16:01:36.280890Z","iopub.status.idle":"2021-08-16T16:01:36.284853Z","shell.execute_reply.started":"2021-08-16T16:01:36.280858Z","shell.execute_reply":"2021-08-16T16:01:36.284172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_img_numpy(img_id, base=\"/train\"):\n    img_path = make_img_path(img_id)\n    img = Image.open(base + \"/\" + img_path)\n    return np.asarray(img)","metadata":{"execution":{"iopub.status.busy":"2021-08-16T16:01:36.548952Z","iopub.execute_input":"2021-08-16T16:01:36.549321Z","iopub.status.idle":"2021-08-16T16:01:36.554970Z","shell.execute_reply.started":"2021-08-16T16:01:36.549286Z","shell.execute_reply":"2021-08-16T16:01:36.553886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_array = [get_img_numpy(img, BASE_PATH + \"/train\") for img in train_df['id'][1000:1009]]\n\nimage_grid3x3(img_array, [landmark for landmark in train_df['landmark_id'][1000:1009]])","metadata":{"execution":{"iopub.status.busy":"2021-08-16T16:02:17.348944Z","iopub.execute_input":"2021-08-16T16:02:17.349329Z","iopub.status.idle":"2021-08-16T16:02:19.323802Z","shell.execute_reply.started":"2021-08-16T16:02:17.349293Z","shell.execute_reply":"2021-08-16T16:02:19.322785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_array = [get_img_numpy(img, BASE_PATH + \"/train\") for img in train_df['id'][10000:10009]]\n\nimage_grid3x3(img_array, [landmark for landmark in train_df['landmark_id'][10000:10009]])","metadata":{"execution":{"iopub.status.busy":"2021-08-16T16:04:13.776760Z","iopub.execute_input":"2021-08-16T16:04:13.777240Z","iopub.status.idle":"2021-08-16T16:04:15.610880Z","shell.execute_reply.started":"2021-08-16T16:04:13.777203Z","shell.execute_reply":"2021-08-16T16:04:15.610137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get some graphs going about class distributions","metadata":{}},{"cell_type":"code","source":"sns.histplot(data=train_df, x=\"landmark_id\", bins=1000)","metadata":{"execution":{"iopub.status.busy":"2021-08-16T16:15:11.908788Z","iopub.execute_input":"2021-08-16T16:15:11.909166Z","iopub.status.idle":"2021-08-16T16:15:36.729445Z","shell.execute_reply.started":"2021-08-16T16:15:11.909135Z","shell.execute_reply":"2021-08-16T16:15:36.728406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Conclusions:\n\n- We need to resize images to some reasonable threshold.\n- We also need augmentation techniques to solve class imbalance.\n  - HIGH_CLASS_SAMPLES=6272\n  - MIN_CLASS_SAMPLES=2","metadata":{}}]}