{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --upgrade pip\n!pip install datasets>=2.6.1\n!pip install git+https://github.com/huggingface/transformers\n!pip install datasets[vision]  \n!pip install torch torchvision","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-04-09T11:20:04.236012Z","iopub.execute_input":"2023-04-09T11:20:04.236487Z","iopub.status.idle":"2023-04-09T11:22:03.753003Z","shell.execute_reply.started":"2023-04-09T11:20:04.236444Z","shell.execute_reply":"2023-04-09T11:22:03.751269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport torch\n\nfrom torch.utils.data import DataLoader\nfrom datasets import load_dataset, Dataset, Image\nfrom huggingface_hub import HfApi, HfFolder, notebook_login","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-09T11:22:03.755513Z","iopub.execute_input":"2023-04-09T11:22:03.756738Z","iopub.status.idle":"2023-04-09T11:22:07.734738Z","shell.execute_reply.started":"2023-04-09T11:22:03.756685Z","shell.execute_reply":"2023-04-09T11:22:07.733305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:22:07.736338Z","iopub.execute_input":"2023-04-09T11:22:07.737070Z","iopub.status.idle":"2023-04-09T11:22:07.744393Z","shell.execute_reply.started":"2023-04-09T11:22:07.737028Z","shell.execute_reply":"2023-04-09T11:22:07.742754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv('/kaggle/input/ibiohash-2023-fgvc10/train.csv')\nprint(len(train_labels))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:22:07.747664Z","iopub.execute_input":"2023-04-09T11:22:07.748035Z","iopub.status.idle":"2023-04-09T11:22:08.381713Z","shell.execute_reply.started":"2023-04-09T11:22:07.748001Z","shell.execute_reply":"2023-04-09T11:22:08.380330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Query Dataset","metadata":{}},{"cell_type":"markdown","source":"## Load the dataset","metadata":{}},{"cell_type":"code","source":"query_dir = '/kaggle/input/ibiohash-2023-fgvc10/iBioHash_Query/Query'\nfilenames = os.listdir(query_dir)\nfilepaths = [os.path.join(query_dir, filename) for filename in filenames]\nds_train = Dataset.from_dict({'image': filepaths}).cast_column('image', Image())\n\n# ds_train = ds_train['train']\nprint(ds_train)\nds_train[0]['image']","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-04-09T11:23:30.664711Z","iopub.execute_input":"2023-04-09T11:23:30.666078Z","iopub.status.idle":"2023-04-09T11:23:30.701565Z","shell.execute_reply.started":"2023-04-09T11:23:30.666007Z","shell.execute_reply":"2023-04-09T11:23:30.700050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA","metadata":{}},{"cell_type":"code","source":"print(\"Number of rows in dataset:\")\nprint(len(ds_train))\nprint(\"Number of categories in dataset: 1000 (given)\")","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:24:06.768042Z","iopub.execute_input":"2023-04-09T11:24:06.768629Z","iopub.status.idle":"2023-04-09T11:24:06.776684Z","shell.execute_reply.started":"2023-04-09T11:24:06.768574Z","shell.execute_reply":"2023-04-09T11:24:06.775042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Display some sample images","metadata":{}},{"cell_type":"code","source":"# def display_images(labels, batch):\n#     num_labels = len(labels)\n\n#     fig = plt.figure(figsize=(10,10))\n#     fig.suptitle(\"Some examples of images of the dataset\", fontsize=16)\n#     for i in range(num_labels):\n#         df = df_train_labels[df_train_labels['label'] == labels[i]]\n#         df_samples = df.sample(n=batch, random_state=42)\n#         img_indices = df_samples['idx'].values\n#         for j in range(batch):\n#             plt.subplot(batch, num_labels, i+num_labels*j+1)\n#             plt.xticks([])\n#             plt.yticks([])\n#             plt.grid(False)\n#             plt.imshow(ds_train[int(img_indices[j])]['image'], cmap=plt.cm.binary)\n#             plt.xlabel(labels[i])\n#     plt.show()\n\ndef display_images():\n    '''Display randomly 25 images from the dataset'''\n    img_indices = np.random.randint(0, len(ds_train), 25)\n    \n    fig = plt.figure(figsize=(10,10))\n    fig.suptitle(\"Some examples of images of the dataset\", fontsize=16)\n    for i in range(25):\n        plt.subplot(5, 5, i+1)\n        plt.xticks([])\n        plt.yticks([])\n        plt.grid(False)\n        plt.imshow(ds_train[int(img_indices[i])]['image'], cmap=plt.cm.binary)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:29:44.683458Z","iopub.execute_input":"2023-04-09T11:29:44.684811Z","iopub.status.idle":"2023-04-09T11:29:44.694048Z","shell.execute_reply.started":"2023-04-09T11:29:44.684762Z","shell.execute_reply":"2023-04-09T11:29:44.692748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_images()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:29:45.378575Z","iopub.execute_input":"2023-04-09T11:29:45.379013Z","iopub.status.idle":"2023-04-09T11:29:48.440673Z","shell.execute_reply.started":"2023-04-09T11:29:45.378974Z","shell.execute_reply":"2023-04-09T11:29:48.439370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Explore image shapes","metadata":{}},{"cell_type":"code","source":"# ds_train = ds_train.with_format('torch', device=device)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:34:57.47758Z","iopub.execute_input":"2023-04-09T10:34:57.478062Z","iopub.status.idle":"2023-04-09T10:34:57.484126Z","shell.execute_reply.started":"2023-04-09T10:34:57.478018Z","shell.execute_reply":"2023-04-09T10:34:57.48277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_train[0]['image'].size","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:30:20.947716Z","iopub.execute_input":"2023-04-09T11:30:20.948518Z","iopub.status.idle":"2023-04-09T11:30:20.970880Z","shell.execute_reply.started":"2023-04-09T11:30:20.948476Z","shell.execute_reply":"2023-04-09T11:30:20.969554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"############################\n# SKIP THIS CELL IF YOU HAVE THE EXPORTED IMG SHAPES CSV FILE\n#############################\n\n# Define function to get image shape\n## When use batched=True, each argument passed in map func is an entire batch\ndef get_img_shape(samples):\n    img_shapes = [img.size for img in samples['image']]\n    return {'img_shapes': img_shapes}\n\n# By default, the map function will return the entire dataset mapped with the new columns, \n# so we should remove them \n%time ds_img_shapes = ds_train.map(get_img_shape, batched=True, num_proc=4, remove_columns=ds_train.column_names)\n\nnp_img_shapes = np.array(ds_img_shapes['img_shapes'])\nprint(np_img_shapes.shape)\ndf_img_shapes = pd.DataFrame(np_img_shapes, columns=['w', 'h'])\ndf_img_shapes['aspect_ratio_w:h'] = df_img_shapes['w']/df_img_shapes['h']","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:35:10.585794Z","iopub.execute_input":"2023-04-09T11:35:10.587146Z","iopub.status.idle":"2023-04-09T11:35:51.246990Z","shell.execute_reply.started":"2023-04-09T11:35:10.587085Z","shell.execute_reply":"2023-04-09T11:35:51.245200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"############################\n# SKIP THIS CELL IF YOU DON't HAVE THE EXPORTED IMG SHAPES CSV FILE\n#############################\n\n# Load img shapes from csv files\ndf_img_shapes = pd.read_csv('/kaggle/working/ds_img_shapes.csv', index_col=0)\ndf_img_shapes.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:41:38.958355Z","iopub.execute_input":"2023-04-09T10:41:38.959041Z","iopub.status.idle":"2023-04-09T10:41:39.03037Z","shell.execute_reply.started":"2023-04-09T10:41:38.959Z","shell.execute_reply":"2023-04-09T10:41:39.029084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img_shapes.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:36:06.394302Z","iopub.execute_input":"2023-04-09T11:36:06.395196Z","iopub.status.idle":"2023-04-09T11:36:06.443837Z","shell.execute_reply.started":"2023-04-09T11:36:06.395142Z","shell.execute_reply":"2023-04-09T11:36:06.442429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img_shapes['aspect_ratio_w:h'].hist(grid=True, bins=50, range=(0, 3.5))\nplt.xlabel('aspect ratio (w/h)')\nplt.ylabel('count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:36:13.655014Z","iopub.execute_input":"2023-04-09T11:36:13.655471Z","iopub.status.idle":"2023-04-09T11:36:14.031671Z","shell.execute_reply.started":"2023-04-09T11:36:13.655429Z","shell.execute_reply":"2023-04-09T11:36:14.029901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Comments\nSlightly bimodal distribution but most of the images are in the aspect ratio  range of (0.7  …  1.7)\n\n**Suggestion**:  going with a non-destructive resize -> Pad approach","metadata":{}},{"cell_type":"code","source":"df_img_shapes.to_csv('ds_img_shapes.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-09T08:58:15.246348Z","iopub.execute_input":"2023-04-09T08:58:15.246767Z","iopub.status.idle":"2023-04-09T08:58:15.566747Z","shell.execute_reply.started":"2023-04-09T08:58:15.24673Z","shell.execute_reply":"2023-04-09T08:58:15.565701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Gallery Dataset","metadata":{}},{"cell_type":"code","source":"gallery_dir = '/kaggle/input/ibiohash-2023-fgvc10/iBioHash_Gallery/Gallery'\nfilenames = os.listdir(gallery_dir)\nfilepaths = [os.path.join(gallery_dir, filename) for filename in filenames]\nds_train = Dataset.from_dict({'image': filepaths}).cast_column('image', Image())\n\n# ds_train = ds_train['train']\nprint(ds_train)\nds_train[0]['image']","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:39:41.919298Z","iopub.execute_input":"2023-04-09T11:39:41.920174Z","iopub.status.idle":"2023-04-09T11:39:42.338255Z","shell.execute_reply.started":"2023-04-09T11:39:41.920116Z","shell.execute_reply":"2023-04-09T11:39:42.336965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### EDA","metadata":{}},{"cell_type":"code","source":"print(\"Number of rows in dataset:\")\nprint(len(ds_train))\nprint(\"Number of categories in dataset: 1000 (given)\")","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:40:08.580181Z","iopub.execute_input":"2023-04-09T11:40:08.581390Z","iopub.status.idle":"2023-04-09T11:40:08.586489Z","shell.execute_reply.started":"2023-04-09T11:40:08.581329Z","shell.execute_reply":"2023-04-09T11:40:08.585498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_images():\n    '''Display randomly 25 images from the dataset'''\n    img_indices = np.random.randint(0, len(ds_train), 25)\n    \n    fig = plt.figure(figsize=(10,10))\n    fig.suptitle(\"Some examples of images of the dataset\", fontsize=16)\n    for i in range(25):\n        plt.subplot(5, 5, i+1)\n        plt.xticks([])\n        plt.yticks([])\n        plt.grid(False)\n        plt.imshow(ds_train[int(img_indices[i])]['image'], cmap=plt.cm.binary)\n    plt.show()\ndisplay_images()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:40:47.732026Z","iopub.execute_input":"2023-04-09T11:40:47.732998Z","iopub.status.idle":"2023-04-09T11:40:50.836163Z","shell.execute_reply.started":"2023-04-09T11:40:47.732930Z","shell.execute_reply":"2023-04-09T11:40:50.834670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"############################\n# SKIP THIS CELL IF YOU HAVE THE EXPORTED IMG SHAPES CSV FILE\n#############################\n\n# Define function to get image shape\n## When use batched=True, each argument passed in map func is an entire batch\ndef get_img_shape(samples):\n    img_shapes = [img.size for img in samples['image']]\n    return {'img_shapes': img_shapes}\n\n# By default, the map function will return the entire dataset mapped with the new columns, \n# so we should remove them \n%time ds_img_shapes = ds_train.map(get_img_shape, batched=True, num_proc=4, remove_columns=ds_train.column_names)\n\nnp_img_shapes = np.array(ds_img_shapes['img_shapes'])\nprint(np_img_shapes.shape)\ndf_img_shapes = pd.DataFrame(np_img_shapes, columns=['w', 'h'])\ndf_img_shapes['aspect_ratio_w:h'] = df_img_shapes['w']/df_img_shapes['h']","metadata":{"execution":{"iopub.status.busy":"2023-04-09T11:41:23.264198Z","iopub.execute_input":"2023-04-09T11:41:23.265570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img_shapes.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img_shapes['aspect_ratio_w:h'].hist(grid=True, bins=50, range=(0, 3.5))\nplt.xlabel('aspect ratio (w/h)')\nplt.ylabel('count')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}