{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Happy Whaling\n\n The main purpose of this notebook and comptetion is to create an accurate model for tracking marine life by the shape and markings on their tails, dorsal fins, heads and other body parts.","metadata":{}},{"cell_type":"markdown","source":"## UPVOTE If you like the work","metadata":{}},{"cell_type":"markdown","source":"## Importing Libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2 as cv\nimport plotly.express as px\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:00.129344Z","iopub.execute_input":"2022-02-06T17:14:00.130183Z","iopub.status.idle":"2022-02-06T17:14:00.135079Z","shell.execute_reply.started":"2022-02-06T17:14:00.130131Z","shell.execute_reply":"2022-02-06T17:14:00.134345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reading Data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\")\ntrain_path = '../input/happy-whale-and-dolphin/train_images'\ntest_path= '../input/happy-whale-and-dolphin/test_images'","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:00.136765Z","iopub.execute_input":"2022-02-06T17:14:00.137174Z","iopub.status.idle":"2022-02-06T17:14:00.211050Z","shell.execute_reply.started":"2022-02-06T17:14:00.137139Z","shell.execute_reply":"2022-02-06T17:14:00.210262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Basic Exploration","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:00.212404Z","iopub.execute_input":"2022-02-06T17:14:00.212710Z","iopub.status.idle":"2022-02-06T17:14:00.223528Z","shell.execute_reply.started":"2022-02-06T17:14:00.212669Z","shell.execute_reply":"2022-02-06T17:14:00.222836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Images - This is the name of the image that we can find in train_images directory\n* Species - This contains name of the species of the whales and dolphines.\n* Individual_ID - This contains ID of the individual whale/dolphin. This is the target variable in this competition.","metadata":{}},{"cell_type":"code","source":"train_df['path'] = train_path+\"/\"+train_df['image']\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:00.224490Z","iopub.execute_input":"2022-02-06T17:14:00.224704Z","iopub.status.idle":"2022-02-06T17:14:00.252361Z","shell.execute_reply.started":"2022-02-06T17:14:00.224677Z","shell.execute_reply":"2022-02-06T17:14:00.251473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape\nprint(\"In the train_df, we have 51033 rows and 4 columns\")","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:00.254682Z","iopub.execute_input":"2022-02-06T17:14:00.254916Z","iopub.status.idle":"2022-02-06T17:14:00.259548Z","shell.execute_reply.started":"2022-02-06T17:14:00.254888Z","shell.execute_reply":"2022-02-06T17:14:00.258831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying an image from the dataset\nplt.imshow(plt.imread(train_df['path'][3]))","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:00.260725Z","iopub.execute_input":"2022-02-06T17:14:00.261152Z","iopub.status.idle":"2022-02-06T17:14:01.905024Z","shell.execute_reply.started":"2022-02-06T17:14:00.261122Z","shell.execute_reply":"2022-02-06T17:14:01.904074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Species Distribution","metadata":{}},{"cell_type":"code","source":"# Checking the unique species\nprint(f\"No. of unique species in dataset is {train_df.species.nunique()}\")","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:01.906476Z","iopub.execute_input":"2022-02-06T17:14:01.906820Z","iopub.status.idle":"2022-02-06T17:14:01.918262Z","shell.execute_reply.started":"2022-02-06T17:14:01.906774Z","shell.execute_reply":"2022-02-06T17:14:01.917223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Species in the dataset\ntrain_df.species.unique()","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:01.919721Z","iopub.execute_input":"2022-02-06T17:14:01.920495Z","iopub.status.idle":"2022-02-06T17:14:01.934346Z","shell.execute_reply.started":"2022-02-06T17:14:01.920451Z","shell.execute_reply":"2022-02-06T17:14:01.933784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"se = train_df.species.value_counts()\nse","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:01.935898Z","iopub.execute_input":"2022-02-06T17:14:01.936370Z","iopub.status.idle":"2022-02-06T17:14:01.952605Z","shell.execute_reply.started":"2022-02-06T17:14:01.936326Z","shell.execute_reply":"2022-02-06T17:14:01.951693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Following are the First 5 Species based on count :-')\ndisplay(se.head(5))","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:01.955535Z","iopub.execute_input":"2022-02-06T17:14:01.955754Z","iopub.status.idle":"2022-02-06T17:14:01.966009Z","shell.execute_reply.started":"2022-02-06T17:14:01.955727Z","shell.execute_reply":"2022-02-06T17:14:01.965126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Following are the last 5 Species based on count :-')\ndisplay(se.tail(5))","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:01.967497Z","iopub.execute_input":"2022-02-06T17:14:01.967785Z","iopub.status.idle":"2022-02-06T17:14:01.977375Z","shell.execute_reply.started":"2022-02-06T17:14:01.967749Z","shell.execute_reply":"2022-02-06T17:14:01.976233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting the unique species\nplt.figure(figsize=(15,20))\nplt.title(\"Species Distribution\")\nsns.countplot(y='species',data=train_df)","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:01.979117Z","iopub.execute_input":"2022-02-06T17:14:01.979437Z","iopub.status.idle":"2022-02-06T17:14:02.517076Z","shell.execute_reply.started":"2022-02-06T17:14:01.979396Z","shell.execute_reply":"2022-02-06T17:14:02.516274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Two species killer_whale and kiler_whale,bottlenose_dolphin and botlenose_dolphin have been mislabeled so we can merge them into one.","metadata":{}},{"cell_type":"code","source":"for i in range(train_df.shape[0]):\n    if train_df.species[i] == 'kiler_whale':\n        train_df.species[i] = 'killer_whale'\n    elif train_df.species[i] == 'bottlenose_dolpin':\n        train_df.species[i] = 'bottlenose_dolphin'","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:02.520318Z","iopub.execute_input":"2022-02-06T17:14:02.520909Z","iopub.status.idle":"2022-02-06T17:14:04.568539Z","shell.execute_reply.started":"2022-02-06T17:14:02.520853Z","shell.execute_reply":"2022-02-06T17:14:04.567669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Again checking the unique species\nprint(f\"No. of unique species is {train_df.species.nunique()}\")\ntrain_df.species.unique()","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:04.569911Z","iopub.execute_input":"2022-02-06T17:14:04.570737Z","iopub.status.idle":"2022-02-06T17:14:04.587763Z","shell.execute_reply.started":"2022-02-06T17:14:04.570692Z","shell.execute_reply":"2022-02-06T17:14:04.586904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of animals are \",len(train_df.individual_id.unique()))\nprint(\"Here individual id corresponds to a single animal so that means we have many photos for a single animal\")","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:04.589227Z","iopub.execute_input":"2022-02-06T17:14:04.589866Z","iopub.status.idle":"2022-02-06T17:14:04.603522Z","shell.execute_reply.started":"2022-02-06T17:14:04.589824Z","shell.execute_reply":"2022-02-06T17:14:04.602418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**How many types of whales and how many types of dolphins do we have ?**","metadata":{}},{"cell_type":"code","source":"whale_species = []\ndolphin_species = []\nfor i in range(train_df.shape[0]):\n    if ((\"whale\" in train_df['species'][i]) or ('beluga' in train_df['species'][i]) or ('globis' in train_df['species'][i])):\n        if train_df.species[i] not in whale_species: \n            whale_species.append(train_df['species'][i])\n        \n    elif 'dolphin' in train_df.species[i]:\n        if train_df.species[i] not in dolphin_species:\n            dolphin_species.append(train_df['species'][i])    \n    else:\n        continue\n        \nclass_ = ['whale' , 'dolphin']\nfig = px.bar(x=[len(whale_species) , len(dolphin_species)] , y=class_ , color = class_,  title = 'type')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:04.606897Z","iopub.execute_input":"2022-02-06T17:14:04.608559Z","iopub.status.idle":"2022-02-06T17:14:06.210313Z","shell.execute_reply.started":"2022-02-06T17:14:04.608521Z","shell.execute_reply":"2022-02-06T17:14:06.209479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Displaying Images","metadata":{}},{"cell_type":"code","source":"train_jpg= '../input/happy-whale-and-dolphin/train_images'\ntest_jpg = '../input/happy-whale-and-dolphin/test_images'\n\n#function to retrieve image paths from directories\n\ndef getImgPaths(path):\n    image_names = []\n    for dirname, _, filenames in os.walk(path):\n        for filename in filenames:\n            fullpath = os.path.join(dirname, filename)\n            image_names.append(fullpath)\n    return image_names\n\n\ntrain_images_path = getImgPaths(train_jpg)\ntest_images_path = getImgPaths(test_jpg)","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:06.211773Z","iopub.execute_input":"2022-02-06T17:14:06.211992Z","iopub.status.idle":"2022-02-06T17:14:16.718463Z","shell.execute_reply.started":"2022-02-06T17:14:06.211965Z","shell.execute_reply":"2022-02-06T17:14:16.717846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Number of train images: {len(train_images_path)}\\n\")\nprint(f\"Number of test images: {len(test_images_path)}\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:14:16.719603Z","iopub.execute_input":"2022-02-06T17:14:16.720003Z","iopub.status.idle":"2022-02-06T17:14:16.725042Z","shell.execute_reply.started":"2022-02-06T17:14:16.719961Z","shell.execute_reply":"2022-02-06T17:14:16.724169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creating function for displaying images","metadata":{}},{"cell_type":"code","source":"def display_(images_paths, rows, cols,title):\n    \n    figure, ax = plt.subplots(nrows=rows,ncols=cols,figsize=(16,8))  \n    plt.suptitle(title, fontsize=20)                                 \n    for ind,image_path in enumerate(images_paths):                   \n        image = cv.imread(image_path)                               \n        image = cv.cvtColor(image, cv.COLOR_BGR2RGB)               \n        try:                                                         \n            ax.ravel()[ind].imshow(image)                           \n            ax.ravel()[ind].set_axis_off()\n        except:\n            continue\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:15:22.360086Z","iopub.execute_input":"2022-02-06T17:15:22.360713Z","iopub.status.idle":"2022-02-06T17:15:22.367808Z","shell.execute_reply.started":"2022-02-06T17:15:22.360663Z","shell.execute_reply":"2022-02-06T17:15:22.366921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_(train_images_path[100:125], 5, 5,\"Train images\")","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:15:43.412962Z","iopub.execute_input":"2022-02-06T17:15:43.413637Z","iopub.status.idle":"2022-02-06T17:15:53.529593Z","shell.execute_reply.started":"2022-02-06T17:15:43.413595Z","shell.execute_reply":"2022-02-06T17:15:53.528713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_(test_images_path[0:25], 5, 5,\"Test images\")","metadata":{"execution":{"iopub.status.busy":"2022-02-06T17:16:41.556533Z","iopub.execute_input":"2022-02-06T17:16:41.556972Z","iopub.status.idle":"2022-02-06T17:16:52.821001Z","shell.execute_reply.started":"2022-02-06T17:16:41.556935Z","shell.execute_reply":"2022-02-06T17:16:52.820228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### More to Come...","metadata":{}}]}