{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --upgrade pip\n!pip install datasets>=2.6.1\n!pip install git+https://github.com/huggingface/transformers\n!pip install datasets[vision]  \n!pip install torch torchvision","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-04-09T05:36:24.944961Z","iopub.execute_input":"2023-04-09T05:36:24.945242Z","iopub.status.idle":"2023-04-09T05:37:54.304842Z","shell.execute_reply.started":"2023-04-09T05:36:24.945213Z","shell.execute_reply":"2023-04-09T05:37:54.303578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport torch\n\nfrom torch.utils.data import DataLoader\nfrom datasets import load_dataset, Dataset, Image\nfrom huggingface_hub import HfApi, HfFolder, notebook_login","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-09T10:33:55.850775Z","iopub.execute_input":"2023-04-09T10:33:55.851752Z","iopub.status.idle":"2023-04-09T10:33:55.858783Z","shell.execute_reply.started":"2023-04-09T10:33:55.851672Z","shell.execute_reply":"2023-04-09T10:33:55.857387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:33:58.054798Z","iopub.execute_input":"2023-04-09T10:33:58.055475Z","iopub.status.idle":"2023-04-09T10:33:58.063116Z","shell.execute_reply.started":"2023-04-09T10:33:58.055420Z","shell.execute_reply":"2023-04-09T10:33:58.061772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Training Dataset","metadata":{}},{"cell_type":"markdown","source":"## Load the dataset","metadata":{}},{"cell_type":"code","source":"ds_train = load_dataset('pphuc25/iBioHash50k_labels_150k')","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-04-09T10:34:09.538550Z","iopub.execute_input":"2023-04-09T10:34:09.539032Z","iopub.status.idle":"2023-04-09T10:34:11.179130Z","shell.execute_reply.started":"2023-04-09T10:34:09.538991Z","shell.execute_reply":"2023-04-09T10:34:11.177958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_train = ds_train['train']\nprint(ds_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:34:16.289347Z","iopub.execute_input":"2023-04-09T10:34:16.290586Z","iopub.status.idle":"2023-04-09T10:34:16.297043Z","shell.execute_reply.started":"2023-04-09T10:34:16.290547Z","shell.execute_reply":"2023-04-09T10:34:16.295531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA the train dataset","metadata":{}},{"cell_type":"markdown","source":"### a. labels count","metadata":{}},{"cell_type":"code","source":"print(\"Number of rows in dataset:\")\nprint(len(ds_train))\nprint(\"Number of categories in dataset:\")\nprint(len(ds_train.unique('labels')))","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:34:18.215280Z","iopub.execute_input":"2023-04-09T10:34:18.216507Z","iopub.status.idle":"2023-04-09T10:34:18.228010Z","shell.execute_reply.started":"2023-04-09T10:34:18.216454Z","shell.execute_reply":"2023-04-09T10:34:18.226739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_train_labels = pd.Series(ds_train['labels'])\nlabels_count = ds_train_labels.value_counts()\n# Sort the count based on the labels\nlabels_count = labels_count.sort_index() \nprint(labels_count)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:34:18.879826Z","iopub.execute_input":"2023-04-09T10:34:18.882017Z","iopub.status.idle":"2023-04-09T10:34:18.974515Z","shell.execute_reply.started":"2023-04-09T10:34:18.881969Z","shell.execute_reply":"2023-04-09T10:34:18.973298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_count_df = labels_count.reset_index()\nlabels_count_df.columns = ['label', 'count']\nlabels_count_df['per total (%)'] = labels_count_df['count']/len(ds_train)*100\nlabels_count_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:34:19.781781Z","iopub.execute_input":"2023-04-09T10:34:19.782856Z","iopub.status.idle":"2023-04-09T10:34:19.797373Z","shell.execute_reply.started":"2023-04-09T10:34:19.782807Z","shell.execute_reply":"2023-04-09T10:34:19.796237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Top 5 most AND least freqeuent labels\nprint('Top 5 most counts')\nprint(labels_count_df.sort_values('count', ascending=False).head(5))\nprint('\\n')\nprint('Top 5 least counts')\nprint(labels_count_df.sort_values('count', ascending=True).head(5))\n\n# labels_count_top10.plot(x='label', y='count', kind='bar', legend=False)\n# plt.title('Top 10 label counts in training dataset')\n# plt.xlabel('Label')\n# plt.ylabel('Count')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:34:20.431573Z","iopub.execute_input":"2023-04-09T10:34:20.433026Z","iopub.status.idle":"2023-04-09T10:34:20.445911Z","shell.execute_reply.started":"2023-04-09T10:34:20.432978Z","shell.execute_reply":"2023-04-09T10:34:20.444463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### b. Display some sample images from each labels","metadata":{}},{"cell_type":"code","source":"df_train_labels = ds_train_labels.reset_index()\ndf_train_labels.columns = ['idx', 'label']\ndf_label_0 = df_train_labels[df_train_labels['label'] == 0]\ndf_label_0.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:34:21.767674Z","iopub.execute_input":"2023-04-09T10:34:21.768485Z","iopub.status.idle":"2023-04-09T10:34:21.785098Z","shell.execute_reply.started":"2023-04-09T10:34:21.768443Z","shell.execute_reply":"2023-04-09T10:34:21.783510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_images(labels, batch):\n    num_labels = len(labels)\n\n    fig = plt.figure(figsize=(10,10))\n    fig.suptitle(\"Some examples of images of the dataset\", fontsize=16)\n    for i in range(num_labels):\n        df = df_train_labels[df_train_labels['label'] == labels[i]]\n        df_samples = df.sample(n=batch, random_state=42)\n        img_indices = df_samples['idx'].values\n        for j in range(batch):\n            plt.subplot(batch, num_labels, i+num_labels*j+1)\n            plt.xticks([])\n            plt.yticks([])\n            plt.grid(False)\n            plt.imshow(ds_train[int(img_indices[j])]['image'], cmap=plt.cm.binary)\n            plt.xlabel(labels[i])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:34:52.137268Z","iopub.execute_input":"2023-04-09T10:34:52.137941Z","iopub.status.idle":"2023-04-09T10:34:52.147107Z","shell.execute_reply.started":"2023-04-09T10:34:52.137903Z","shell.execute_reply":"2023-04-09T10:34:52.145854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = 4\nlabels_random = np.random.randint(0, 1000, n)\nprint(labels_random)\ndisplay_images(labels_random, batch=4)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:34:52.688017Z","iopub.execute_input":"2023-04-09T10:34:52.688792Z","iopub.status.idle":"2023-04-09T10:34:55.502813Z","shell.execute_reply.started":"2023-04-09T10:34:52.688747Z","shell.execute_reply":"2023-04-09T10:34:55.501761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### c. image shapes","metadata":{}},{"cell_type":"code","source":"# ds_train = ds_train.with_format('torch', device=device)","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:34:57.477580Z","iopub.execute_input":"2023-04-09T10:34:57.478062Z","iopub.status.idle":"2023-04-09T10:34:57.484126Z","shell.execute_reply.started":"2023-04-09T10:34:57.478018Z","shell.execute_reply":"2023-04-09T10:34:57.482770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_train[0]['image'].size","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:34:58.340756Z","iopub.execute_input":"2023-04-09T10:34:58.341238Z","iopub.status.idle":"2023-04-09T10:34:58.370987Z","shell.execute_reply.started":"2023-04-09T10:34:58.341195Z","shell.execute_reply":"2023-04-09T10:34:58.369919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"############################\n# SKIP THIS CELL IF YOU HAVE THE EXPORTED IMG SHAPES CSV FILE\n#############################\n# Define function to get image shape\n## When use batched=True, each argument passed in map func is an entire batch\ndef get_img_shape(samples):\n    img_shapes = [img.size for img in samples['image']]\n    return {'img_shapes': img_shapes}\n\n# By default, the map function will return the entire dataset mapped with the new columns, \n# so we should remove them \nds_img_shapes = ds_train.map(get_img_shape, batched=True, num_proc=4, remove_columns=ds_train.column_names)\n\nnp_img_shapes = np.array(ds_img_shapes['img_shapes'])\nprint(np_img_shapes.shape)\ndf_img_shapes = pd.DataFrame(np_img_shapes, columns=['w', 'h'])\ndf_img_shapes['aspect_ratio_w:h'] = df_img_shapes['w']/df_img_shapes['h']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load img shapes from csv files\ndf_img_shapes = pd.read_csv('/kaggle/working/ds_img_shapes.csv', index_col=0)\ndf_img_shapes.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:41:38.958355Z","iopub.execute_input":"2023-04-09T10:41:38.959041Z","iopub.status.idle":"2023-04-09T10:41:39.030370Z","shell.execute_reply.started":"2023-04-09T10:41:38.959000Z","shell.execute_reply":"2023-04-09T10:41:39.029084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img_shapes.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:41:46.318715Z","iopub.execute_input":"2023-04-09T10:41:46.319781Z","iopub.status.idle":"2023-04-09T10:41:46.353628Z","shell.execute_reply.started":"2023-04-09T10:41:46.319724Z","shell.execute_reply":"2023-04-09T10:41:46.352249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img_shapes['aspect_ratio_w:h'].hist(grid=True, bins=50, range=(0, 3.5))\nplt.xlabel('aspect ratio (w/h)')\nplt.ylabel('count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T10:41:48.986470Z","iopub.execute_input":"2023-04-09T10:41:48.986869Z","iopub.status.idle":"2023-04-09T10:41:49.547861Z","shell.execute_reply.started":"2023-04-09T10:41:48.986837Z","shell.execute_reply":"2023-04-09T10:41:49.546662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Comments\nSlightly bimodal distribution but most of the images are in the aspect ratio  range of (0.7  …  1.7)\n\n**Suggestion**:  going with a non-destructive resize -> Pad approach","metadata":{}},{"cell_type":"code","source":"df_img_shapes.to_csv('ds_img_shapes.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-09T08:58:15.246348Z","iopub.execute_input":"2023-04-09T08:58:15.246767Z","iopub.status.idle":"2023-04-09T08:58:15.566747Z","shell.execute_reply.started":"2023-04-09T08:58:15.246730Z","shell.execute_reply":"2023-04-09T08:58:15.565701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}