{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\n\nThis Kernel explores the **Happywhale - Whale and Dolphin Identification competition** dataset.\n\n<img src=\"https://images.unsplash.com/photo-1570913179118-f3d24be1d1f7?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=2143&q=80\" width=500></img>","metadata":{}},{"cell_type":"markdown","source":"# Analysis preparation\n\nLet's load the data and explore it preliminarly.\n\n<img src=\"https://images.unsplash.com/photo-1568430328012-21ed450453ea?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=1174&q=80\" width=500></img>","metadata":{}},{"cell_type":"code","source":"!pip install imagesize","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:02.473410Z","iopub.execute_input":"2022-02-20T11:51:02.473746Z","iopub.status.idle":"2022-02-20T11:51:13.442786Z","shell.execute_reply.started":"2022-02-20T11:51:02.473655Z","shell.execute_reply":"2022-02-20T11:51:13.441895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport imagesize","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:13.445269Z","iopub.execute_input":"2022-02-20T11:51:13.445616Z","iopub.status.idle":"2022-02-20T11:51:14.821758Z","shell.execute_reply.started":"2022-02-20T11:51:13.445572Z","shell.execute_reply":"2022-02-20T11:51:14.820530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Files and folders: {os.listdir('/kaggle/input/happy-whale-and-dolphin')}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:14.822993Z","iopub.execute_input":"2022-02-20T11:51:14.823225Z","iopub.status.idle":"2022-02-20T11:51:14.832017Z","shell.execute_reply.started":"2022-02-20T11:51:14.823196Z","shell.execute_reply":"2022-02-20T11:51:14.831206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's load `train.csv` and `sample_submission.csv` first","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/train.csv')\nsubmission_df = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/sample_submission.csv')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:14.833896Z","iopub.execute_input":"2022-02-20T11:51:14.834334Z","iopub.status.idle":"2022-02-20T11:51:14.993689Z","shell.execute_reply.started":"2022-02-20T11:51:14.834302Z","shell.execute_reply":"2022-02-20T11:51:14.992891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:14.995101Z","iopub.execute_input":"2022-02-20T11:51:14.995561Z","iopub.status.idle":"2022-02-20T11:51:15.017307Z","shell.execute_reply.started":"2022-02-20T11:51:14.995529Z","shell.execute_reply":"2022-02-20T11:51:15.016547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:15.018313Z","iopub.execute_input":"2022-02-20T11:51:15.018519Z","iopub.status.idle":"2022-02-20T11:51:15.031225Z","shell.execute_reply.started":"2022-02-20T11:51:15.018493Z","shell.execute_reply":"2022-02-20T11:51:15.030150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data exploration\n\n<img src=\"https://images.unsplash.com/photo-1611890129309-31e797820019?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=1170&q=80\" width=500></img>","metadata":{}},{"cell_type":"markdown","source":"Let's get some more insight into the train data and train and test images.","metadata":{}},{"cell_type":"code","source":"print(f\"Images in train index file: {train_df.image.nunique()}\")\nprint(f\"Species in train index file: {train_df.species.nunique()}\")\nprint(f\"Individual IDs in train index file: {train_df.individual_id.nunique()}\")\n\nprint(f\"Images in train images folder: {len(os.listdir('/kaggle/input/happy-whale-and-dolphin/train_images'))}\")\nprint(f\"Images in test images folder: {len(os.listdir('/kaggle/input/happy-whale-and-dolphin/test_images'))}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:15.032568Z","iopub.execute_input":"2022-02-20T11:51:15.033255Z","iopub.status.idle":"2022-02-20T11:51:16.228159Z","shell.execute_reply.started":"2022-02-20T11:51:15.033216Z","shell.execute_reply":"2022-02-20T11:51:16.227306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's look to the complete list of species.","metadata":{}},{"cell_type":"code","source":"print(f\"Species: {train_df.species.unique()}\")","metadata":{"execution":{"iopub.status.busy":"2022-02-20T11:51:16.229766Z","iopub.execute_input":"2022-02-20T11:51:16.230105Z","iopub.status.idle":"2022-02-20T11:51:16.239521Z","shell.execute_reply.started":"2022-02-20T11:51:16.230061Z","shell.execute_reply":"2022-02-20T11:51:16.238857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From discussions on the discussion board for this competition and other Notebooks (ex: [Happywhale: Data Distribution](https://www.kaggle.com/awsaf49/happywhale-data-distribution/notebook)) we learn that:\n* beluga and globis are whales;  \n* we can identify some of the species as being dolphin and other as whales (using their suffix); therefore, we will also rename beluga and globis;  \nWe also observe that\n* `bottlenose dolphin` has sometime a typo (typed `dolpin`)\n* `killer whale` is typed incorrectly as `kiler`.\n","metadata":{}},{"cell_type":"code","source":"train_df.loc[train_df.species.str.contains('beluga'), 'species'] = 'beluga_whale'\ntrain_df.loc[train_df.species.str.contains('globis'), 'species'] = 'globis_whale'","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:16.241175Z","iopub.execute_input":"2022-02-20T11:51:16.241723Z","iopub.status.idle":"2022-02-20T11:51:16.316264Z","shell.execute_reply.started":"2022-02-20T11:51:16.241676Z","shell.execute_reply":"2022-02-20T11:51:16.315545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['class'] = train_df.species.map(lambda x: 'whale' if 'whale' in x else 'dolphin')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:16.318908Z","iopub.execute_input":"2022-02-20T11:51:16.319352Z","iopub.status.idle":"2022-02-20T11:51:16.338637Z","shell.execute_reply.started":"2022-02-20T11:51:16.319319Z","shell.execute_reply":"2022-02-20T11:51:16.337982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['species'] = train_df['species'].str.replace('bottlenose_dolpin','bottlenose_dolphin')\ntrain_df['species'] = train_df['species'].str.replace('kiler_whale','killer_whale')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:16.339801Z","iopub.execute_input":"2022-02-20T11:51:16.340209Z","iopub.status.idle":"2022-02-20T11:51:16.429671Z","shell.execute_reply.started":"2022-02-20T11:51:16.340179Z","shell.execute_reply":"2022-02-20T11:51:16.428757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check how many species of dolphin vs. whale are.","metadata":{}},{"cell_type":"code","source":"temp = train_df.groupby([\"class\"])[\"species\"].nunique()\ndf = pd.DataFrame({'Classes': temp.index,\n                   'Species': temp.values\n                  })\ndf = df.sort_values(['Species'], ascending=False)\nplt.figure(figsize = (6,6))\nplt.title('Species distribution - grouped on Dolphins and Whales - train dataset')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Classes', y=\"Species\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:16.431079Z","iopub.execute_input":"2022-02-20T11:51:16.431926Z","iopub.status.idle":"2022-02-20T11:51:16.668637Z","shell.execute_reply.started":"2022-02-20T11:51:16.431879Z","shell.execute_reply":"2022-02-20T11:51:16.667837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check more details about the column `individual_id` from `train_df` values distribution.","metadata":{}},{"cell_type":"code","source":"print(\"Top 10 individual_id\")\ntrain_df.individual_id.value_counts().head(10)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:16.670344Z","iopub.execute_input":"2022-02-20T11:51:16.670661Z","iopub.status.idle":"2022-02-20T11:51:16.691582Z","shell.execute_reply.started":"2022-02-20T11:51:16.670618Z","shell.execute_reply":"2022-02-20T11:51:16.690952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 1, figsize=(7, 7))\nsns.kdeplot(np.log(train_df.individual_id.value_counts()))\nplt.title(\"Logaritmic distribution of individual_id frequency in images\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:16.692483Z","iopub.execute_input":"2022-02-20T11:51:16.692801Z","iopub.status.idle":"2022-02-20T11:51:17.056708Z","shell.execute_reply.started":"2022-02-20T11:51:16.692773Z","shell.execute_reply":"2022-02-20T11:51:17.055889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's also look separatelly on dolphins and whales.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 1, figsize=(7, 7))\nsns.kdeplot(np.log(train_df.loc[train_df[\"class\"]=='whale'].individual_id.value_counts()))\nsns.kdeplot(np.log(train_df.loc[train_df[\"class\"]=='dolphin'].individual_id.value_counts()))\nax.legend(labels=['whale', 'dolphin'])\nplt.title(\"Logaritmic distribution of individual_id frequency in images\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:17.058208Z","iopub.execute_input":"2022-02-20T11:51:17.059101Z","iopub.status.idle":"2022-02-20T11:51:17.424743Z","shell.execute_reply.started":"2022-02-20T11:51:17.059052Z","shell.execute_reply":"2022-02-20T11:51:17.423906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check as well frequency of species in train dataset.","metadata":{}},{"cell_type":"code","source":"df = train_df.groupby([\"class\", \"species\"])[\"image\"].count().reset_index()\ndf.columns = [\"Class\", \"Species\", \"Images\"]\ndf = df.sort_values(['Images'], ascending=False)\nplt.figure(figsize = (12,6))\nplt.title('Species distribution - images per each species - train dataset')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Species', y=\"Images\", hue='Class', data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:17.426459Z","iopub.execute_input":"2022-02-20T11:51:17.427022Z","iopub.status.idle":"2022-02-20T11:51:18.161230Z","shell.execute_reply.started":"2022-02-20T11:51:17.426971Z","shell.execute_reply":"2022-02-20T11:51:18.160313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's see now how many individual ids are per each species.","metadata":{}},{"cell_type":"code","source":"df = train_df.groupby([\"class\", \"species\"])[\"individual_id\"].nunique().reset_index()\ndf.columns = [\"Class\", \"Species\", \"Unique ID Count\"]\ndf = df.sort_values([\"Unique ID Count\"], ascending=False)\n\ndf = df.sort_values(['Unique ID Count'], ascending=False)\nplt.figure(figsize = (12,6))\nplt.title('Species distribution - Individual IDs per each species - train dataset')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Species', y=\"Unique ID Count\", hue='Class', data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:18.162301Z","iopub.execute_input":"2022-02-20T11:51:18.162522Z","iopub.status.idle":"2022-02-20T11:51:19.114911Z","shell.execute_reply.started":"2022-02-20T11:51:18.162494Z","shell.execute_reply":"2022-02-20T11:51:19.114184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check now the image sizes in train and test images datasets.","metadata":{}},{"cell_type":"markdown","source":"Let's check if set of images listed in `train_df` is identical with set of images in folder `train_images` ","metadata":{}},{"cell_type":"code","source":"train_df_list = list(train_df.image.unique())\ntrain_images_list = list(os.listdir('/kaggle/input/happy-whale-and-dolphin/train_images'))\ndelta = set(train_df_list) & set(train_images_list)\nminus = set(train_df_list) - set(train_images_list)\nprint(f\"Images in train dataset: {len(train_df_list)}\\nImages in train folder: {len(train_images_list)}\\nIntersection: {len(delta)}\\nDifference: {len(minus)}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:19.116122Z","iopub.execute_input":"2022-02-20T11:51:19.116562Z","iopub.status.idle":"2022-02-20T11:51:19.189319Z","shell.execute_reply.started":"2022-02-20T11:51:19.116508Z","shell.execute_reply":"2022-02-20T11:51:19.188155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All images indexed in `train_df` are present in the images folder and viceversa.","metadata":{}},{"cell_type":"markdown","source":"# Images data exploration","metadata":{}},{"cell_type":"markdown","source":"First we test which function (based on cv2 or based on imagesize) will run faster.","metadata":{}},{"cell_type":"code","source":"# image size using cv2 imread shape\ndef read_image_sizes_cv2(file_name):\n    image = cv2.imread('/kaggle/input/happy-whale-and-dolphin/train_images/' + file_name)\n    return list(image.shape)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:19.190745Z","iopub.execute_input":"2022-02-20T11:51:19.191161Z","iopub.status.idle":"2022-02-20T11:51:19.196447Z","shell.execute_reply.started":"2022-02-20T11:51:19.191115Z","shell.execute_reply":"2022-02-20T11:51:19.195505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image size using imagesize\ndef get_image_sizes_imagesize(file_name):\n    width, height = imagesize.get('/kaggle/input/happy-whale-and-dolphin/train_images/' + file_name)\n    return [width, height]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:19.197665Z","iopub.execute_input":"2022-02-20T11:51:19.197964Z","iopub.status.idle":"2022-02-20T11:51:19.213112Z","shell.execute_reply.started":"2022-02-20T11:51:19.197926Z","shell.execute_reply":"2022-02-20T11:51:19.212183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nsample_size = 100\nstart_time = time.time()\ntrain_sample_df = train_df.sample(sample_size)\nm = np.stack(train_sample_df['image'].apply(read_image_sizes_cv2))\ndf = pd.DataFrame(m,columns=['w','h','c'])\nprint(f\"Total processing time for {sample_size} images (using cv2): {round(time.time()-start_time, 2)} sec.\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:19.214537Z","iopub.execute_input":"2022-02-20T11:51:19.215244Z","iopub.status.idle":"2022-02-20T11:51:27.313707Z","shell.execute_reply.started":"2022-02-20T11:51:19.215198Z","shell.execute_reply":"2022-02-20T11:51:27.312689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nsample_size = 100\nstart_time = time.time()\ntrain_sample_df = train_df.sample(sample_size)\nm = np.stack(train_sample_df['image'].apply(get_image_sizes_imagesize))\ndf = pd.DataFrame(m,columns=['w','h'])\nprint(f\"Total processing time for {sample_size} images (using imagesize): {round(time.time()-start_time, 2)} sec.\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:27.315220Z","iopub.execute_input":"2022-02-20T11:51:27.315537Z","iopub.status.idle":"2022-02-20T11:51:29.000936Z","shell.execute_reply.started":"2022-02-20T11:51:27.315494Z","shell.execute_reply":"2022-02-20T11:51:28.999932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We decide to use imagesize based function, since this one is more effective.\nWe will run it for 2500 samples.","metadata":{}},{"cell_type":"code","source":"import time\nsample_size = 2500\nstart_time = time.time()\ntrain_sample_df = train_df.sample(sample_size)\nm = np.stack(train_sample_df['image'].apply(get_image_sizes_imagesize))\ndf = pd.DataFrame(m,columns=['w','h'])\nprint(f\"Total processing time for {sample_size} images (using imagesize): {round(time.time()-start_time, 2)} sec.\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:29.002746Z","iopub.execute_input":"2022-02-20T11:51:29.003499Z","iopub.status.idle":"2022-02-20T11:51:54.114733Z","shell.execute_reply.started":"2022-02-20T11:51:29.003453Z","shell.execute_reply":"2022-02-20T11:51:54.114070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All image sizes are extracted in a different Kernel, [Images Sizes Makes Whales (and Dolphins) Happy](https://www.kaggle.com/gpreda/images-sizes-makes-whales-and-dolphins-happy).","metadata":{}},{"cell_type":"code","source":"train_img_df = pd.concat([train_sample_df, df], axis=1, sort=False)\nprint(f\"Number of different image size ( images samples): {train_img_df.groupby(['w','h']).count().shape[0]}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:54.115694Z","iopub.execute_input":"2022-02-20T11:51:54.115954Z","iopub.status.idle":"2022-02-20T11:51:54.133742Z","shell.execute_reply.started":"2022-02-20T11:51:54.115923Z","shell.execute_reply":"2022-02-20T11:51:54.132932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It appears that there are many images sizes (we only sampled less than 5% of the total number of images).\n\nLet's visualize the distribution of width/height and colors per species.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (12,6))\nplt.title('Species distribution - width per each species - train dataset (5% random data sample)')\nsns.set_color_codes(\"pastel\")\ns = sns.boxplot(x = 'species', y=\"w\", data=train_img_df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:54.134795Z","iopub.execute_input":"2022-02-20T11:51:54.135019Z","iopub.status.idle":"2022-02-20T11:51:54.663609Z","shell.execute_reply.started":"2022-02-20T11:51:54.134992Z","shell.execute_reply":"2022-02-20T11:51:54.662800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (12,6))\nplt.title('Species distribution - height per each species - train dataset (5% random data sample)')\nsns.set_color_codes(\"pastel\")\ns = sns.boxplot(x = 'species', y=\"h\", data=train_img_df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:54.664647Z","iopub.execute_input":"2022-02-20T11:51:54.664868Z","iopub.status.idle":"2022-02-20T11:51:55.178520Z","shell.execute_reply.started":"2022-02-20T11:51:54.664841Z","shell.execute_reply":"2022-02-20T11:51:55.177900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's show the distribution of width and height per species using a scatterplot.","metadata":{}},{"cell_type":"code","source":"def plot_species_scatter(train_img_df):\n    i = 0\n    sns.set_style('whitegrid')\n    plt.figure()\n    species = list(train_img_df.species.unique())\n    fig, ax = plt.subplots(5, 5,figsize=(15, 12))\n\n    for spec in species:\n        i += 1\n        plt.subplot(5, 5,i)\n        df = train_img_df.loc[train_img_df.species==spec]\n        plt.scatter(df['w'], df['h'], marker='+')\n        plt.xlabel(spec, fontsize=9)\n    plt.show();\nplot_species_scatter(train_img_df.dropna())","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:55.179492Z","iopub.execute_input":"2022-02-20T11:51:55.180126Z","iopub.status.idle":"2022-02-20T11:51:58.527592Z","shell.execute_reply.started":"2022-02-20T11:51:55.180094Z","shell.execute_reply":"2022-02-20T11:51:58.526906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The number of colors seems to be allways 3 for the 5% random sample used.","metadata":{}},{"cell_type":"markdown","source":"\n\nLet's sample few of the train images, grouped on species.  \n\nWe create first a plotting function.","metadata":{}},{"cell_type":"code","source":"def plot_image_samples(species):\n    root_path = \"/kaggle/input/happy-whale-and-dolphin/\"\n    fig.subplots_adjust(hspace = .1, wspace=.1)\n    images_folder=\"train_images/\"\n    df = train_df[train_df['species']==species].copy()\n    df.index = range(len(df.index))\n\n    f, ax = plt.subplots(4, 4, figsize=(16,16))\n\n    for i in range(16):\n        file = df.loc[i, 'image']\n        species = df.loc[i, 'species']\n        identifier = df.loc[i, 'individual_id']\n        img = cv2.imread(root_path+images_folder+file)\n        ax[i//4, i%4].imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n        ax[i//4, i%4].set_title(identifier+\" (\"+species+\")\")\n        ax[i//4, i%4].axis('off')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:58.530986Z","iopub.execute_input":"2022-02-20T11:51:58.531746Z","iopub.status.idle":"2022-02-20T11:51:58.540059Z","shell.execute_reply.started":"2022-02-20T11:51:58.531691Z","shell.execute_reply":"2022-02-20T11:51:58.539252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"bottlenose_dolphin\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:51:58.541171Z","iopub.execute_input":"2022-02-20T11:51:58.541533Z","iopub.status.idle":"2022-02-20T11:52:16.053888Z","shell.execute_reply.started":"2022-02-20T11:51:58.541501Z","shell.execute_reply":"2022-02-20T11:52:16.052682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"beluga_whale\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:52:16.055235Z","iopub.execute_input":"2022-02-20T11:52:16.055478Z","iopub.status.idle":"2022-02-20T11:52:21.039968Z","shell.execute_reply.started":"2022-02-20T11:52:16.055447Z","shell.execute_reply":"2022-02-20T11:52:21.038985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"humpback_whale\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:52:21.041166Z","iopub.execute_input":"2022-02-20T11:52:21.041393Z","iopub.status.idle":"2022-02-20T11:52:35.364806Z","shell.execute_reply.started":"2022-02-20T11:52:21.041366Z","shell.execute_reply":"2022-02-20T11:52:35.363941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"blue_whale\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:52:35.366139Z","iopub.execute_input":"2022-02-20T11:52:35.366365Z","iopub.status.idle":"2022-02-20T11:52:41.655671Z","shell.execute_reply.started":"2022-02-20T11:52:35.366337Z","shell.execute_reply":"2022-02-20T11:52:41.654913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"killer_whale\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:52:41.657203Z","iopub.execute_input":"2022-02-20T11:52:41.657629Z","iopub.status.idle":"2022-02-20T11:52:59.703578Z","shell.execute_reply.started":"2022-02-20T11:52:41.657590Z","shell.execute_reply":"2022-02-20T11:52:59.702584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"spotted_dolphin\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:52:59.704784Z","iopub.execute_input":"2022-02-20T11:52:59.705054Z","iopub.status.idle":"2022-02-20T11:53:07.528259Z","shell.execute_reply.started":"2022-02-20T11:52:59.705020Z","shell.execute_reply":"2022-02-20T11:53:07.527557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's also look to a sample of test images.","metadata":{}},{"cell_type":"code","source":"def plot_image_samples_test():\n    root_path = \"/kaggle/input/happy-whale-and-dolphin/\"\n    fig.subplots_adjust(hspace = .1, wspace=.1)\n    images_folder=\"test_images/\"\n\n    f, ax = plt.subplots(4, 4, figsize=(16,16))\n    file_list = list(os.listdir(root_path+images_folder))\n    for i in range(16):\n        file = file_list[i]\n        img = cv2.imread(root_path+images_folder+file)\n        ax[i//4, i%4].imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n        ax[i//4, i%4].set_title(\"Test image: \"+file)\n        ax[i//4, i%4].axis('off')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:53:07.529238Z","iopub.execute_input":"2022-02-20T11:53:07.529880Z","iopub.status.idle":"2022-02-20T11:53:07.537802Z","shell.execute_reply.started":"2022-02-20T11:53:07.529832Z","shell.execute_reply":"2022-02-20T11:53:07.536839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples_test()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:53:07.539082Z","iopub.execute_input":"2022-02-20T11:53:07.539355Z","iopub.status.idle":"2022-02-20T11:53:20.323677Z","shell.execute_reply.started":"2022-02-20T11:53:07.539325Z","shell.execute_reply":"2022-02-20T11:53:20.322716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preliminary submission\n\n<img src=\"https://images.unsplash.com/photo-1602264985195-52b338cb937b?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=1170&q=80\" width=500></img>\n\nLet's rotate the identifiers so that`new_individual` became the first option.","metadata":{}},{"cell_type":"code","source":"def rotate_values(x):\n    xcopy = x.split()\n    temp = xcopy[4]\n    xcopy[4] = xcopy[0]\n    xcopy[0] = temp\n    xcopy = \" \".join(xcopy)\n    return xcopy","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:53:20.324957Z","iopub.execute_input":"2022-02-20T11:53:20.325197Z","iopub.status.idle":"2022-02-20T11:53:20.330576Z","shell.execute_reply.started":"2022-02-20T11:53:20.325167Z","shell.execute_reply":"2022-02-20T11:53:20.329534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df[\"predictions\"] = submission_df[\"predictions\"].apply(lambda x: rotate_values(x))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:53:20.331813Z","iopub.execute_input":"2022-02-20T11:53:20.332088Z","iopub.status.idle":"2022-02-20T11:53:20.380311Z","shell.execute_reply.started":"2022-02-20T11:53:20.332057Z","shell.execute_reply":"2022-02-20T11:53:20.379399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:53:20.381408Z","iopub.execute_input":"2022-02-20T11:53:20.381718Z","iopub.status.idle":"2022-02-20T11:53:20.391623Z","shell.execute_reply.started":"2022-02-20T11:53:20.381675Z","shell.execute_reply":"2022-02-20T11:53:20.391058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We output the prepared submission file.","metadata":{}},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=False)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-20T11:53:20.392613Z","iopub.execute_input":"2022-02-20T11:53:20.393222Z","iopub.status.idle":"2022-02-20T11:53:20.533781Z","shell.execute_reply.started":"2022-02-20T11:53:20.393188Z","shell.execute_reply":"2022-02-20T11:53:20.532887Z"},"trusted":true},"execution_count":null,"outputs":[]}]}