{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction","metadata":{}},{"cell_type":"markdown","source":"This notebook explore the dataset of Happywhale - whale and dolphin and identify it.","metadata":{}},{"cell_type":"markdown","source":"# Analysis","metadata":{}},{"cell_type":"markdown","source":"load the data and explore it preliminarly.\n\n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-05T05:34:18.631964Z","iopub.execute_input":"2022-02-05T05:34:18.632290Z","iopub.status.idle":"2022-02-05T05:34:18.637715Z","shell.execute_reply.started":"2022-02-05T05:34:18.632257Z","shell.execute_reply":"2022-02-05T05:34:18.636934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Files and folders: {os.listdir('/kaggle/input/happy-whale-and-dolphin')}\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:34:32.353088Z","iopub.execute_input":"2022-02-05T05:34:32.353385Z","iopub.status.idle":"2022-02-05T05:34:32.362263Z","shell.execute_reply.started":"2022-02-05T05:34:32.353355Z","shell.execute_reply":"2022-02-05T05:34:32.361201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's read and explore train.csv and sample_submission.csv first","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/train.csv')\nsubmission_df = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:35:26.592784Z","iopub.execute_input":"2022-02-05T05:35:26.593136Z","iopub.status.idle":"2022-02-05T05:35:26.769092Z","shell.execute_reply.started":"2022-02-05T05:35:26.593104Z","shell.execute_reply":"2022-02-05T05:35:26.768007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:35:36.272260Z","iopub.execute_input":"2022-02-05T05:35:36.272542Z","iopub.status.idle":"2022-02-05T05:35:36.292450Z","shell.execute_reply.started":"2022-02-05T05:35:36.272512Z","shell.execute_reply":"2022-02-05T05:35:36.291774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:35:46.991370Z","iopub.execute_input":"2022-02-05T05:35:46.992025Z","iopub.status.idle":"2022-02-05T05:35:47.001518Z","shell.execute_reply.started":"2022-02-05T05:35:46.991984Z","shell.execute_reply":"2022-02-05T05:35:47.000693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Exploration","metadata":{}},{"cell_type":"markdown","source":"Let's get some more insight into the train data and train and test images.","metadata":{}},{"cell_type":"code","source":"print(f\"Images in train index file: {train_df.image.nunique()}\")\nprint(f\"Species in train index file: {train_df.species.nunique()}\")\nprint(f\"Individual IDs in train index file: {train_df.individual_id.nunique()}\")\n\nprint(f\"Images in train images folder: {len(os.listdir('/kaggle/input/happy-whale-and-dolphin/train_images'))}\")\nprint(f\"Images in test images folder: {len(os.listdir('/kaggle/input/happy-whale-and-dolphin/test_images'))}\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:37:00.272534Z","iopub.execute_input":"2022-02-05T05:37:00.273113Z","iopub.status.idle":"2022-02-05T05:37:01.493136Z","shell.execute_reply.started":"2022-02-05T05:37:00.273074Z","shell.execute_reply":"2022-02-05T05:37:01.492350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check more details about the column individual_id from train_df values distribution.\n\n","metadata":{}},{"cell_type":"code","source":"print(\"Top 10 individual_id\")\ntrain_df.individual_id.value_counts().head(10)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:37:29.612110Z","iopub.execute_input":"2022-02-05T05:37:29.612722Z","iopub.status.idle":"2022-02-05T05:37:29.637209Z","shell.execute_reply.started":"2022-02-05T05:37:29.612681Z","shell.execute_reply":"2022-02-05T05:37:29.636361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 1, figsize=(7, 7))\nsns.kdeplot(np.log(train_df.individual_id.value_counts()))\nplt.title(\"Logaritmic distribution of individual_id frequency in images\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:37:56.562647Z","iopub.execute_input":"2022-02-05T05:37:56.563110Z","iopub.status.idle":"2022-02-05T05:37:56.975303Z","shell.execute_reply.started":"2022-02-05T05:37:56.563066Z","shell.execute_reply":"2022-02-05T05:37:56.974477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check as well frequency of species in train dataset.","metadata":{}},{"cell_type":"code","source":"temp = train_df[\"species\"].value_counts()\ndf = pd.DataFrame({'Species': temp.index,\n                   'Images': temp.values\n                  })\nplt.figure(figsize = (12,6))\nplt.title('Species distribution - images per each species - train dataset')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Species', y=\"Images\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:38:32.732552Z","iopub.execute_input":"2022-02-05T05:38:32.732857Z","iopub.status.idle":"2022-02-05T05:38:33.179052Z","shell.execute_reply.started":"2022-02-05T05:38:32.732825Z","shell.execute_reply":"2022-02-05T05:38:33.177948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's see now how many individual ids are per each species.","metadata":{}},{"cell_type":"code","source":"temp = train_df.groupby([\"species\"])[\"individual_id\"].nunique()\ndf = pd.DataFrame({'Species': temp.index,\n                   'Unique ID Count': temp.values\n                  })\ndf = df.sort_values(['Unique ID Count'], ascending=False)\nplt.figure(figsize = (12,6))\nplt.title('Species distribution - Individual IDs per each species - train dataset')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Species', y=\"Unique ID Count\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:39:00.127481Z","iopub.execute_input":"2022-02-05T05:39:00.128143Z","iopub.status.idle":"2022-02-05T05:39:00.595766Z","shell.execute_reply.started":"2022-02-05T05:39:00.128098Z","shell.execute_reply":"2022-02-05T05:39:00.594839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check now the image sizes in train and test images datasets.\n\nLet's check if set of images listed in train_df is identical with set of images in folder train_images","metadata":{}},{"cell_type":"code","source":"train_df_list = list(train_df.image.unique())\ntrain_images_list = list(os.listdir('/kaggle/input/happy-whale-and-dolphin/train_images'))\ndelta = set(train_df_list) & set(train_images_list)\nminus = set(train_df_list) - set(train_images_list)\nprint(f\"Images in train dataset: {len(train_df_list)}\\nImages in train folder: {len(train_images_list)}\\nIntersection: {len(delta)}\\nDifference: {len(minus)}\")\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:39:39.032282Z","iopub.execute_input":"2022-02-05T05:39:39.032593Z","iopub.status.idle":"2022-02-05T05:39:39.107443Z","shell.execute_reply.started":"2022-02-05T05:39:39.032555Z","shell.execute_reply":"2022-02-05T05:39:39.106563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All images indexed in train_df are present in the images folder and viceversa.","metadata":{}},{"cell_type":"code","source":"def read_image_sizes(file_name):\n    image = cv2.imread('/kaggle/input/happy-whale-and-dolphin/train_images/' + file_name)\n    return list(image.shape)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:39:59.952168Z","iopub.execute_input":"2022-02-05T05:39:59.952879Z","iopub.status.idle":"2022-02-05T05:39:59.961532Z","shell.execute_reply.started":"2022-02-05T05:39:59.952834Z","shell.execute_reply":"2022-02-05T05:39:59.960351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Images data exploration","metadata":{}},{"cell_type":"markdown","source":"Because the processing of images to get images dimmension, we will only process a sample of 2500 images.","metadata":{}},{"cell_type":"code","source":"import time\nsample_size = 2500\nstart_time = time.time()\ntrain_sample_df = train_df.sample(sample_size)\nm = np.stack(train_sample_df['image'].apply(read_image_sizes))\ndf = pd.DataFrame(m,columns=['w','h','c'])\nprint(f\"Total processing time for {sample_size} images: {round(time.time()-start_time, 2)} sec.\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:40:52.917435Z","iopub.execute_input":"2022-02-05T05:40:52.918121Z","iopub.status.idle":"2022-02-05T05:44:00.047729Z","shell.execute_reply.started":"2022-02-05T05:40:52.918087Z","shell.execute_reply":"2022-02-05T05:44:00.046729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_img_df = pd.concat([train_sample_df, df], axis=1, sort=False)\nprint(f\"Number of different image size ( images samples): {train_img_df.groupby(['w','h', 'c']).count().shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:44:00.049051Z","iopub.execute_input":"2022-02-05T05:44:00.049259Z","iopub.status.idle":"2022-02-05T05:44:00.068396Z","shell.execute_reply.started":"2022-02-05T05:44:00.049233Z","shell.execute_reply":"2022-02-05T05:44:00.067517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It appears that there are many images sizes (we only sampled less than 5% of the total number of images).\n\nLet's visualize the distribution of width/height and colors per species.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (12,6))\nplt.title('Species distribution - width per each species - train dataset (5% random data sample)')\nsns.set_color_codes(\"pastel\")\ns = sns.boxplot(x = 'species', y=\"w\", data=train_img_df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:44:00.070120Z","iopub.execute_input":"2022-02-05T05:44:00.070617Z","iopub.status.idle":"2022-02-05T05:44:00.802287Z","shell.execute_reply.started":"2022-02-05T05:44:00.070569Z","shell.execute_reply":"2022-02-05T05:44:00.801440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (12,6))\nplt.title('Species distribution - height per each species - train dataset (5% random data sample)')\nsns.set_color_codes(\"pastel\")\ns = sns.boxplot(x = 'species', y=\"h\", data=train_img_df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:44:00.804540Z","iopub.execute_input":"2022-02-05T05:44:00.804850Z","iopub.status.idle":"2022-02-05T05:44:01.423593Z","shell.execute_reply.started":"2022-02-05T05:44:00.804808Z","shell.execute_reply":"2022-02-05T05:44:01.422601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's show the distribution of width and height per species using a scatterplot.","metadata":{}},{"cell_type":"code","source":"def plot_species_scatter(train_img_df):\n    i = 0\n    sns.set_style('whitegrid')\n    plt.figure()\n    species = list(train_img_df.species.unique())\n    fig, ax = plt.subplots(5, 5,figsize=(15, 12))\n\n    for spec in species:\n        i += 1\n        plt.subplot(5, 5,i)\n        df = train_img_df.loc[train_img_df.species==spec]\n        plt.scatter(df['w'], df['h'], marker='+')\n        plt.xlabel(spec, fontsize=9)\n    plt.show();\nplot_species_scatter(train_img_df.dropna())","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:44:01.425083Z","iopub.execute_input":"2022-02-05T05:44:01.425339Z","iopub.status.idle":"2022-02-05T05:44:04.852955Z","shell.execute_reply.started":"2022-02-05T05:44:01.425309Z","shell.execute_reply":"2022-02-05T05:44:04.852276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The number of colors seems to be allways 3 for the 5% random sample used.\n\nLet's sample few of the train images, grouped on species.\n\nWe create first a plotting function.","metadata":{}},{"cell_type":"code","source":"def plot_image_samples(species):\n    root_path = \"/kaggle/input/happy-whale-and-dolphin/\"\n    fig.subplots_adjust(hspace = .1, wspace=.1)\n    images_folder=\"train_images/\"\n    df = train_df[train_df['species']==species].copy()\n    df.index = range(len(df.index))\n\n    f, ax = plt.subplots(4, 4, figsize=(16,16))\n\n    for i in range(16):\n        file = df.loc[i, 'image']\n        species = df.loc[i, 'species']\n        identifier = df.loc[i, 'individual_id']\n        img = cv2.imread(root_path+images_folder+file)\n        ax[i//4, i%4].imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n        ax[i//4, i%4].set_title(identifier+\" (\"+species+\")\")\n        ax[i//4, i%4].axis('off')","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:44:23.613117Z","iopub.execute_input":"2022-02-05T05:44:23.613564Z","iopub.status.idle":"2022-02-05T05:44:23.621726Z","shell.execute_reply.started":"2022-02-05T05:44:23.613521Z","shell.execute_reply":"2022-02-05T05:44:23.620790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"bottlenose_dolphin\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:44:33.233031Z","iopub.execute_input":"2022-02-05T05:44:33.233433Z","iopub.status.idle":"2022-02-05T05:44:57.068957Z","shell.execute_reply.started":"2022-02-05T05:44:33.233307Z","shell.execute_reply":"2022-02-05T05:44:57.066424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"beluga\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:44:57.073397Z","iopub.execute_input":"2022-02-05T05:44:57.074509Z","iopub.status.idle":"2022-02-05T05:45:02.079123Z","shell.execute_reply.started":"2022-02-05T05:44:57.074449Z","shell.execute_reply":"2022-02-05T05:45:02.078433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"humpback_whale\")\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:45:02.080461Z","iopub.execute_input":"2022-02-05T05:45:02.080861Z","iopub.status.idle":"2022-02-05T05:45:16.379954Z","shell.execute_reply.started":"2022-02-05T05:45:02.080814Z","shell.execute_reply":"2022-02-05T05:45:16.378908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"blue_whale\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:45:16.381686Z","iopub.execute_input":"2022-02-05T05:45:16.381973Z","iopub.status.idle":"2022-02-05T05:45:22.701698Z","shell.execute_reply.started":"2022-02-05T05:45:16.381937Z","shell.execute_reply":"2022-02-05T05:45:22.700722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"killer_whale\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:45:22.703407Z","iopub.execute_input":"2022-02-05T05:45:22.703910Z","iopub.status.idle":"2022-02-05T05:45:41.122246Z","shell.execute_reply.started":"2022-02-05T05:45:22.703850Z","shell.execute_reply":"2022-02-05T05:45:41.121198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(\"spotted_dolphin\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:45:41.123949Z","iopub.execute_input":"2022-02-05T05:45:41.124202Z","iopub.status.idle":"2022-02-05T05:45:49.098646Z","shell.execute_reply.started":"2022-02-05T05:45:41.124163Z","shell.execute_reply":"2022-02-05T05:45:49.097842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's also look to a sample of test images.","metadata":{}},{"cell_type":"code","source":"def plot_image_samples_test():\n    root_path = \"/kaggle/input/happy-whale-and-dolphin/\"\n    fig.subplots_adjust(hspace = .1, wspace=.1)\n    images_folder=\"test_images/\"\n\n    f, ax = plt.subplots(4, 4, figsize=(16,16))\n    file_list = list(os.listdir(root_path+images_folder))\n    for i in range(16):\n        file = file_list[i]\n        img = cv2.imread(root_path+images_folder+file)\n        ax[i//4, i%4].imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n        ax[i//4, i%4].set_title(\"Test image: \"+file)\n        ax[i//4, i%4].axis('off')","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:45:55.491576Z","iopub.execute_input":"2022-02-05T05:45:55.492222Z","iopub.status.idle":"2022-02-05T05:45:55.500196Z","shell.execute_reply.started":"2022-02-05T05:45:55.492182Z","shell.execute_reply":"2022-02-05T05:45:55.498943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples_test()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:46:05.352637Z","iopub.execute_input":"2022-02-05T05:46:05.352933Z","iopub.status.idle":"2022-02-05T05:46:18.058856Z","shell.execute_reply.started":"2022-02-05T05:46:05.352902Z","shell.execute_reply":"2022-02-05T05:46:18.058029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"markdown","source":"Let's rotate the identifiers so thatnew_individual became the first option.","metadata":{}},{"cell_type":"code","source":"def rotate_values(x):\n    xcopy = x.split()\n    temp = xcopy[4]\n    xcopy[4] = xcopy[0]\n    xcopy[0] = temp\n    xcopy = \" \".join(xcopy)\n    return xcopy","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:46:41.793327Z","iopub.execute_input":"2022-02-05T05:46:41.793706Z","iopub.status.idle":"2022-02-05T05:46:41.799169Z","shell.execute_reply.started":"2022-02-05T05:46:41.793669Z","shell.execute_reply":"2022-02-05T05:46:41.798282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df[\"predictions\"] = submission_df[\"predictions\"].apply(lambda x: rotate_values(x))","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:46:49.791550Z","iopub.execute_input":"2022-02-05T05:46:49.791859Z","iopub.status.idle":"2022-02-05T05:46:49.831629Z","shell.execute_reply.started":"2022-02-05T05:46:49.791823Z","shell.execute_reply":"2022-02-05T05:46:49.830963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:46:56.912143Z","iopub.execute_input":"2022-02-05T05:46:56.912433Z","iopub.status.idle":"2022-02-05T05:46:56.921879Z","shell.execute_reply.started":"2022-02-05T05:46:56.912404Z","shell.execute_reply":"2022-02-05T05:46:56.921305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We output the prepared submission file.","metadata":{}},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T05:47:22.252458Z","iopub.execute_input":"2022-02-05T05:47:22.252984Z","iopub.status.idle":"2022-02-05T05:47:22.399504Z","shell.execute_reply.started":"2022-02-05T05:47:22.252934Z","shell.execute_reply":"2022-02-05T05:47:22.398481Z"},"trusted":true},"execution_count":null,"outputs":[]}]}