{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\n\nThis Kernel explores the **Happywhale - Whale and Dolphin Identification competition** dataset.\n\n<img src=\"https://images.unsplash.com/photo-1570913179118-f3d24be1d1f7?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=2143&q=80\" width=500></img>","metadata":{}},{"cell_type":"markdown","source":"# Analysis preparation\n\nLet's load the data and explore it preliminarly.\n\n<img src=\"https://images.unsplash.com/photo-1568430328012-21ed450453ea?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=1174&q=80\" width=500></img>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:30:16.663245Z","iopub.execute_input":"2022-02-02T12:30:16.663722Z","iopub.status.idle":"2022-02-02T12:30:17.928597Z","shell.execute_reply.started":"2022-02-02T12:30:16.663615Z","shell.execute_reply":"2022-02-02T12:30:17.927834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Files and folders: {os.listdir('/kaggle/input/happy-whale-and-dolphin')}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:30:17.929934Z","iopub.execute_input":"2022-02-02T12:30:17.930112Z","iopub.status.idle":"2022-02-02T12:30:17.935056Z","shell.execute_reply.started":"2022-02-02T12:30:17.930090Z","shell.execute_reply":"2022-02-02T12:30:17.934231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's load `train.csv` and `sample_submission.csv` first","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/train.csv')\nsubmission_df = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/sample_submission.csv')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:30:17.936271Z","iopub.execute_input":"2022-02-02T12:30:17.936508Z","iopub.status.idle":"2022-02-02T12:30:18.104498Z","shell.execute_reply.started":"2022-02-02T12:30:17.936480Z","shell.execute_reply":"2022-02-02T12:30:18.103591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:30:18.106272Z","iopub.execute_input":"2022-02-02T12:30:18.106508Z","iopub.status.idle":"2022-02-02T12:30:18.124679Z","shell.execute_reply.started":"2022-02-02T12:30:18.106471Z","shell.execute_reply":"2022-02-02T12:30:18.123923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:30:18.125978Z","iopub.execute_input":"2022-02-02T12:30:18.126203Z","iopub.status.idle":"2022-02-02T12:30:18.134374Z","shell.execute_reply.started":"2022-02-02T12:30:18.126166Z","shell.execute_reply":"2022-02-02T12:30:18.133574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data exploration\n\n<img src=\"https://images.unsplash.com/photo-1611890129309-31e797820019?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=1170&q=80\" width=500></img>","metadata":{}},{"cell_type":"markdown","source":"Let's get some more insight into the train data and train and test images.","metadata":{}},{"cell_type":"code","source":"print(f\"Images in train index file: {train_df.image.nunique()}\")\nprint(f\"Species in train index file: {train_df.species.nunique()}\")\nprint(f\"Individual IDs in train index file: {train_df.individual_id.nunique()}\")\n\nprint(f\"Images in train images folder: {len(os.listdir('/kaggle/input/happy-whale-and-dolphin/train_images'))}\")\nprint(f\"Images in test images folder: {len(os.listdir('/kaggle/input/happy-whale-and-dolphin/test_images'))}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:30:18.135509Z","iopub.execute_input":"2022-02-02T12:30:18.136045Z","iopub.status.idle":"2022-02-02T12:30:19.229635Z","shell.execute_reply.started":"2022-02-02T12:30:18.136001Z","shell.execute_reply":"2022-02-02T12:30:19.228902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check more details about the column `individual_id` from `train_df` values distribution.","metadata":{}},{"cell_type":"code","source":"print(\"Top 10 individual_id\")\ntrain_df.individual_id.value_counts().head(10)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:30:19.230632Z","iopub.execute_input":"2022-02-02T12:30:19.231022Z","iopub.status.idle":"2022-02-02T12:30:19.253299Z","shell.execute_reply.started":"2022-02-02T12:30:19.230983Z","shell.execute_reply":"2022-02-02T12:30:19.252734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 1, figsize=(7, 7))\nsns.kdeplot(np.log(train_df.individual_id.value_counts()))\nplt.title(\"Logaritmic distribution of individual_id frequency in images\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:30:19.254178Z","iopub.execute_input":"2022-02-02T12:30:19.254831Z","iopub.status.idle":"2022-02-02T12:30:19.667595Z","shell.execute_reply.started":"2022-02-02T12:30:19.254778Z","shell.execute_reply":"2022-02-02T12:30:19.666827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check as well frequency of species in train dataset.","metadata":{}},{"cell_type":"code","source":"temp = train_df[\"species\"].value_counts()\ndf = pd.DataFrame({'Species': temp.index,\n                   'Images': temp.values\n                  })\nplt.figure(figsize = (12,6))\nplt.title('Species distribution - images per each species - train dataset')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Species', y=\"Images\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:30:19.668612Z","iopub.execute_input":"2022-02-02T12:30:19.668792Z","iopub.status.idle":"2022-02-02T12:30:20.123071Z","shell.execute_reply.started":"2022-02-02T12:30:19.668769Z","shell.execute_reply":"2022-02-02T12:30:20.122348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's see now how many individual ids are per each species.","metadata":{}},{"cell_type":"code","source":"temp = train_df.groupby([\"species\"])[\"individual_id\"].nunique()\ndf = pd.DataFrame({'Species': temp.index,\n                   'Unique ID Count': temp.values\n                  })\ndf = df.sort_values(['Unique ID Count'], ascending=False)\nplt.figure(figsize = (12,6))\nplt.title('Species distribution - Individual IDs per each species - train dataset')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Species', y=\"Unique ID Count\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:30:20.125878Z","iopub.execute_input":"2022-02-02T12:30:20.126100Z","iopub.status.idle":"2022-02-02T12:30:20.519277Z","shell.execute_reply.started":"2022-02-02T12:30:20.126074Z","shell.execute_reply":"2022-02-02T12:30:20.518507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check now the image sizes in train and test images datasets.","metadata":{}},{"cell_type":"markdown","source":"Let's check if set of images listed in `train_df` is identical with set of images in folder `train_images` ","metadata":{}},{"cell_type":"code","source":"train_df_list = list(train_df.image.unique())\ntrain_images_list = list(os.listdir('/kaggle/input/happy-whale-and-dolphin/train_images'))\ndelta = set(train_df_list) & set(train_images_list)\nminus = set(train_df_list) - set(train_images_list)\nprint(f\"Images in train dataset: {len(train_df_list)}\\nImages in train folder: {len(train_images_list)}\\nIntersection: {len(delta)}\\nDifference: {len(minus)}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:30:20.520520Z","iopub.execute_input":"2022-02-02T12:30:20.520731Z","iopub.status.idle":"2022-02-02T12:30:20.586687Z","shell.execute_reply.started":"2022-02-02T12:30:20.520705Z","shell.execute_reply":"2022-02-02T12:30:20.585848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All images indexed in `train_df` are present in the images folder and viceversa.","metadata":{}},{"cell_type":"code","source":"def read_image_sizes(file_name):\n    image = cv2.imread('/kaggle/input/happy-whale-and-dolphin/train_images/' + file_name)\n    return list(image.shape)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:30:20.588729Z","iopub.execute_input":"2022-02-02T12:30:20.589272Z","iopub.status.idle":"2022-02-02T12:30:20.593944Z","shell.execute_reply.started":"2022-02-02T12:30:20.589226Z","shell.execute_reply":"2022-02-02T12:30:20.593031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Images data exploration","metadata":{}},{"cell_type":"markdown","source":"Because the processing of images to get images dimmension, we will only process a sample of 2500 images.","metadata":{}},{"cell_type":"code","source":"import time\nsample_size = 2500\nstart_time = time.time()\ntrain_sample_df = train_df.sample(sample_size)\nm = np.stack(train_sample_df['image'].apply(read_image_sizes))\ndf = pd.DataFrame(m,columns=['w','h','c'])\nprint(f\"Total processing time for {sample_size} images: {round(time.time()-start_time, 2)} sec.\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:30:20.595217Z","iopub.execute_input":"2022-02-02T12:30:20.595518Z","iopub.status.idle":"2022-02-02T12:33:14.744536Z","shell.execute_reply.started":"2022-02-02T12:30:20.595480Z","shell.execute_reply":"2022-02-02T12:33:14.743900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_img_df = pd.concat([train_sample_df, df], axis=1, sort=False)\nprint(f\"Number of different image size ( images samples): {train_img_df.groupby(['w','h', 'c']).count().shape[0]}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:33:14.745542Z","iopub.execute_input":"2022-02-02T12:33:14.745906Z","iopub.status.idle":"2022-02-02T12:33:14.761758Z","shell.execute_reply.started":"2022-02-02T12:33:14.745871Z","shell.execute_reply":"2022-02-02T12:33:14.761040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It appears that there are many images sizes (we only sampled less than 5% of the total number of images).\n\nLet's visualize the distribution of width/height and colors per species.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (12,6))\nplt.title('Species distribution - width per each species - train dataset (5% random data sample)')\nsns.set_color_codes(\"pastel\")\ns = sns.boxplot(x = 'species', y=\"w\", data=train_img_df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:33:14.763554Z","iopub.execute_input":"2022-02-02T12:33:14.763884Z","iopub.status.idle":"2022-02-02T12:33:15.372110Z","shell.execute_reply.started":"2022-02-02T12:33:14.763843Z","shell.execute_reply":"2022-02-02T12:33:15.371357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (12,6))\nplt.title('Species distribution - height per each species - train dataset (5% random data sample)')\nsns.set_color_codes(\"pastel\")\ns = sns.boxplot(x = 'species', y=\"h\", data=train_img_df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:33:15.373647Z","iopub.execute_input":"2022-02-02T12:33:15.373937Z","iopub.status.idle":"2022-02-02T12:33:15.861886Z","shell.execute_reply.started":"2022-02-02T12:33:15.373902Z","shell.execute_reply":"2022-02-02T12:33:15.860818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The number of colors seems to be allways 3 for the 5% random sample used.","metadata":{}},{"cell_type":"markdown","source":"\n\nLet's sample few of the train images, grouped on species.  \n\nWe create first a plotting function.","metadata":{}},{"cell_type":"code","source":"def plot_image_samples(species):\n    root_path = \"/kaggle/input/happy-whale-and-dolphin/\"\n    fig.subplots_adjust(hspace = .1, wspace=.1)\n    images_folder=\"train_images/\"\n    df = train_df[train_df['species']==species].copy()\n    df.index = range(len(df.index))\n\n    f, ax = plt.subplots(4, 4, figsize=(16,16))\n\n    for i in range(16):\n        file = df.loc[i, 'image']\n        species = df.loc[i, 'species']\n        identifier = df.loc[i, 'individual_id']\n        img = cv2.imread(root_path+images_folder+file)\n        ax[i//4, i%4].imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n        ax[i//4, i%4].set_title(identifier+\" (\"+species+\")\")\n        ax[i//4, i%4].axis('off')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:33:15.863056Z","iopub.execute_input":"2022-02-02T12:33:15.863246Z","iopub.status.idle":"2022-02-02T12:33:15.871954Z","shell.execute_reply.started":"2022-02-02T12:33:15.863224Z","shell.execute_reply":"2022-02-02T12:33:15.871169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"bottlenose_dolphin\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:33:15.872963Z","iopub.execute_input":"2022-02-02T12:33:15.873174Z","iopub.status.idle":"2022-02-02T12:33:32.618147Z","shell.execute_reply.started":"2022-02-02T12:33:15.873149Z","shell.execute_reply":"2022-02-02T12:33:32.617584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"beluga\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:33:32.618958Z","iopub.execute_input":"2022-02-02T12:33:32.619147Z","iopub.status.idle":"2022-02-02T12:33:36.569761Z","shell.execute_reply.started":"2022-02-02T12:33:32.619123Z","shell.execute_reply":"2022-02-02T12:33:36.568682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"humpback_whale\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:33:36.570822Z","iopub.execute_input":"2022-02-02T12:33:36.571043Z","iopub.status.idle":"2022-02-02T12:33:49.266533Z","shell.execute_reply.started":"2022-02-02T12:33:36.571016Z","shell.execute_reply":"2022-02-02T12:33:49.265932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"blue_whale\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:33:49.267686Z","iopub.execute_input":"2022-02-02T12:33:49.268982Z","iopub.status.idle":"2022-02-02T12:33:54.595238Z","shell.execute_reply.started":"2022-02-02T12:33:49.268930Z","shell.execute_reply":"2022-02-02T12:33:54.594686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's also look to a sample of test images.","metadata":{}},{"cell_type":"code","source":"def plot_image_samples_test():\n    root_path = \"/kaggle/input/happy-whale-and-dolphin/\"\n    fig.subplots_adjust(hspace = .1, wspace=.1)\n    images_folder=\"test_images/\"\n\n    f, ax = plt.subplots(4, 4, figsize=(16,16))\n    file_list = list(os.listdir(root_path+images_folder))\n    for i in range(16):\n        file = file_list[i]\n        img = cv2.imread(root_path+images_folder+file)\n        ax[i//4, i%4].imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n        ax[i//4, i%4].set_title(\"Test image: \"+file)\n        ax[i//4, i%4].axis('off')","metadata":{"execution":{"iopub.status.busy":"2022-02-02T12:33:54.596281Z","iopub.execute_input":"2022-02-02T12:33:54.596624Z","iopub.status.idle":"2022-02-02T12:33:54.603275Z","shell.execute_reply.started":"2022-02-02T12:33:54.596570Z","shell.execute_reply":"2022-02-02T12:33:54.602359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples_test()","metadata":{"execution":{"iopub.status.busy":"2022-02-02T12:33:54.604482Z","iopub.execute_input":"2022-02-02T12:33:54.604695Z","iopub.status.idle":"2022-02-02T12:34:05.887775Z","shell.execute_reply.started":"2022-02-02T12:33:54.604671Z","shell.execute_reply":"2022-02-02T12:34:05.883846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preliminary submission\n\n<img src=\"https://images.unsplash.com/photo-1602264985195-52b338cb937b?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=1170&q=80\" width=500></img>\n\nLet's rotate the identifiers so that`new_individual` became the first option.","metadata":{}},{"cell_type":"code","source":"def rotate_values(x):\n    xcopy = x.split()\n    temp = xcopy[4]\n    xcopy[4] = xcopy[0]\n    xcopy[0] = temp\n    xcopy = \" \".join(xcopy)\n    return xcopy","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:34:05.888914Z","iopub.execute_input":"2022-02-02T12:34:05.889131Z","iopub.status.idle":"2022-02-02T12:34:05.894055Z","shell.execute_reply.started":"2022-02-02T12:34:05.889105Z","shell.execute_reply":"2022-02-02T12:34:05.893228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df[\"predictions\"] = submission_df[\"predictions\"].apply(lambda x: rotate_values(x))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:34:05.895401Z","iopub.execute_input":"2022-02-02T12:34:05.895951Z","iopub.status.idle":"2022-02-02T12:34:05.939467Z","shell.execute_reply.started":"2022-02-02T12:34:05.895909Z","shell.execute_reply":"2022-02-02T12:34:05.938552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:34:05.940634Z","iopub.execute_input":"2022-02-02T12:34:05.940889Z","iopub.status.idle":"2022-02-02T12:34:05.950477Z","shell.execute_reply.started":"2022-02-02T12:34:05.940860Z","shell.execute_reply":"2022-02-02T12:34:05.949929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We output the prepared submission file.","metadata":{}},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=False)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-02T12:34:05.951584Z","iopub.execute_input":"2022-02-02T12:34:05.952169Z","iopub.status.idle":"2022-02-02T12:34:06.107248Z","shell.execute_reply.started":"2022-02-02T12:34:05.952138Z","shell.execute_reply":"2022-02-02T12:34:06.106480Z"},"trusted":true},"execution_count":null,"outputs":[]}]}