{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\nThis Kernel is used to extract width & height for the train and test images.","metadata":{}},{"cell_type":"markdown","source":"# Let's do it","metadata":{}},{"cell_type":"markdown","source":"Install needed resources, import packages, read the train data.","metadata":{}},{"cell_type":"code","source":"!pip install imagesize","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-05T21:00:26.839279Z","iopub.execute_input":"2022-02-05T21:00:26.840216Z","iopub.status.idle":"2022-02-05T21:00:34.888042Z","shell.execute_reply.started":"2022-02-05T21:00:26.840171Z","shell.execute_reply":"2022-02-05T21:00:34.886919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport imagesize\nimport time","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-05T21:00:34.890072Z","iopub.execute_input":"2022-02-05T21:00:34.890345Z","iopub.status.idle":"2022-02-05T21:00:34.894479Z","shell.execute_reply.started":"2022-02-05T21:00:34.890315Z","shell.execute_reply":"2022-02-05T21:00:34.893856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/train.csv')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-05T21:00:34.895728Z","iopub.execute_input":"2022-02-05T21:00:34.89593Z","iopub.status.idle":"2022-02-05T21:00:34.981816Z","shell.execute_reply.started":"2022-02-05T21:00:34.895906Z","shell.execute_reply":"2022-02-05T21:00:34.981102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fix species names, add family information.","metadata":{}},{"cell_type":"code","source":"train_df.loc[train_df.species.str.contains('beluga'), 'species'] = 'beluga_whale'\ntrain_df.loc[train_df.species.str.contains('globis'), 'species'] = 'globis_whale'\ntrain_df['family'] = train_df.species.map(lambda x: 'whale' if 'whale' in x else 'dolphin')\ntrain_df['species'] = train_df['species'].str.replace('bottlenose_dolpin','bottlenose_dolphin')\ntrain_df['species'] = train_df['species'].str.replace('kiler_whale','killer_whale')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-05T21:00:34.983224Z","iopub.execute_input":"2022-02-05T21:00:34.983697Z","iopub.status.idle":"2022-02-05T21:00:35.095816Z","shell.execute_reply.started":"2022-02-05T21:00:34.983665Z","shell.execute_reply":"2022-02-05T21:00:35.094938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define a function that read the image sizes.","metadata":{}},{"cell_type":"code","source":"def get_image_sizes_imagesize(file_name):\n    width, height = imagesize.get('/kaggle/input/happy-whale-and-dolphin/train_images/' + file_name)\n    return [width, height]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-05T21:00:35.098189Z","iopub.execute_input":"2022-02-05T21:00:35.099586Z","iopub.status.idle":"2022-02-05T21:00:35.104475Z","shell.execute_reply.started":"2022-02-05T21:00:35.099536Z","shell.execute_reply":"2022-02-05T21:00:35.103588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Apply it for train images.","metadata":{}},{"cell_type":"code","source":"start_time = time.time()\nsample_size = train_df.shape[0]\nm = np.stack(train_df['image'].apply(get_image_sizes_imagesize))\ndf = pd.DataFrame(m,columns=['w','h'])\nprint(f\"Total processing time for {sample_size} images (using imagesize): {round(time.time()-start_time, 2)} sec.\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-05T21:00:35.105719Z","iopub.execute_input":"2022-02-05T21:00:35.105939Z","iopub.status.idle":"2022-02-05T21:00:51.582234Z","shell.execute_reply.started":"2022-02-05T21:00:35.105913Z","shell.execute_reply":"2022-02-05T21:00:51.58044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Merge the image info to the train data.","metadata":{}},{"cell_type":"code","source":"train_img_df = pd.concat([train_df, df], axis=1, sort=False)\nprint(f\"Number of different image size ( images samples): {train_img_df.groupby(['w','h']).count().shape[0]}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-05T21:00:51.583265Z","iopub.status.idle":"2022-02-05T21:00:51.583632Z","shell.execute_reply.started":"2022-02-05T21:00:51.583449Z","shell.execute_reply":"2022-02-05T21:00:51.583471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Prepare a dataset with image names for test set.","metadata":{}},{"cell_type":"code","source":"test_image_list = list(os.listdir('/kaggle/input/happy-whale-and-dolphin/test_images'))\ntest_df = pd.DataFrame(test_image_list, columns=[\"image\"])\nprint(test_df.shape)\ntest_df.head(2)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-05T21:00:51.585922Z","iopub.status.idle":"2022-02-05T21:00:51.586467Z","shell.execute_reply.started":"2022-02-05T21:00:51.586252Z","shell.execute_reply":"2022-02-05T21:00:51.586286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Extract image size for test data.","metadata":{}},{"cell_type":"code","source":"def get_image_sizes_imagesize_test(file_name):\n    width, height = imagesize.get('/kaggle/input/happy-whale-and-dolphin/test_images/' + file_name)\n    return [width, height]","metadata":{"execution":{"iopub.status.busy":"2022-02-05T21:04:17.584939Z","iopub.execute_input":"2022-02-05T21:04:17.585246Z","iopub.status.idle":"2022-02-05T21:04:17.59022Z","shell.execute_reply.started":"2022-02-05T21:04:17.585215Z","shell.execute_reply":"2022-02-05T21:04:17.589269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start_time = time.time()\nsample_size = test_df.shape[0]\nm = np.stack(test_df['image'].apply(get_image_sizes_imagesize_test))\ndf = pd.DataFrame(m,columns=['w','h'])\nprint(f\"Total processing time for {sample_size} images from test data (using imagesize): {round(time.time()-start_time, 2)} sec.\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-05T21:04:19.261098Z","iopub.execute_input":"2022-02-05T21:04:19.261432Z","iopub.status.idle":"2022-02-05T21:04:23.624526Z","shell.execute_reply.started":"2022-02-05T21:04:19.261382Z","shell.execute_reply":"2022-02-05T21:04:23.623375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_img_df = pd.concat([test_df, df], axis=1, sort=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T21:00:51.589591Z","iopub.status.idle":"2022-02-05T21:00:51.590234Z","shell.execute_reply.started":"2022-02-05T21:00:51.590055Z","shell.execute_reply":"2022-02-05T21:00:51.590075Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Save the train and test data including image width.","metadata":{}},{"cell_type":"code","source":"train_img_df.to_csv(\"train_img_size.csv\", index=False)\ntest_img_df.to_csv(\"test_img_size.csv\", index=False)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-05T21:00:51.591123Z","iopub.status.idle":"2022-02-05T21:00:51.591877Z","shell.execute_reply.started":"2022-02-05T21:00:51.591692Z","shell.execute_reply":"2022-02-05T21:00:51.591713Z"},"trusted":true},"execution_count":null,"outputs":[]}]}