{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pickle\n\nfrom keras import layers\nfrom keras.models import Sequential\nfrom keras.preprocessing import image\nfrom keras.layers import Input, Dense, Activation, Dropout\nfrom keras.layers import Flatten, BatchNormalization, Conv2D\nfrom keras.layers import MaxPooling2D, AveragePooling2D\nfrom keras.applications.imagenet_utils import preprocess_input\n\nfrom PIL import Image\nfrom tqdm import tqdm\nimport random as rnd\nimport cv2\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom numpy import expand_dims\n\n!pip install livelossplot\nfrom livelossplot import PlotLossesKeras\n\n%matplotlib inline","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading Dataset \nWe'll use here the Pandas to load the dataset into memory","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/happy-whale-and-dolphin/train.csv')\ntrain_df['path'] = '../input/happy-whale-and-dolphin/train_images/' + train_df['image']\n\npred_df = pd.read_csv('../input/happy-whale-and-dolphin/sample_submission.csv')\npred_df['path'] = '../input/happy-whale-and-dolphin/test_images/' + pred_df['image']","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.272574Z","iopub.status.idle":"2022-03-08T13:56:40.272889Z","shell.execute_reply.started":"2022-03-08T13:56:40.272725Z","shell.execute_reply":"2022-03-08T13:56:40.272741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Having two csv files\n* train.csv - contain image name,species and individual_id\n* sample_submission.csv - contain image name, dummy label for the images in the test folde\n## And two folders contain the images\n* train - having 51033 images of different type of whales and dolphins. There Labels have provided in the train.csv file\n* test - having 27956 images of different type of whales and dolphins. We need to predict their labels","metadata":{}},{"cell_type":"code","source":"train_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.273951Z","iopub.status.idle":"2022-03-08T13:56:40.274263Z","shell.execute_reply.started":"2022-03-08T13:56:40.274094Z","shell.execute_reply":"2022-03-08T13:56:40.274110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Train samples count: ', len(train_df))\ntrain_df.columns","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.275184Z","iopub.status.idle":"2022-03-08T13:56:40.275472Z","shell.execute_reply.started":"2022-03-08T13:56:40.275313Z","shell.execute_reply":"2022-03-08T13:56:40.275336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Species Count: ',len(train_df['species'].value_counts()))\ntrain_df['species'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.276531Z","iopub.status.idle":"2022-03-08T13:56:40.276810Z","shell.execute_reply.started":"2022-03-08T13:56:40.276661Z","shell.execute_reply":"2022-03-08T13:56:40.276676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Cleaning\n## Fixing Duplicate Labels\n* bottlenose_dolpin -> bottlenose_dolphin\n* kiler_whale -> killer_whale\n* beluga -> beluga_whale\n## Changing Label due to extreme similarities\n* globis & pilot_whale -> short_finned_pilot_whale","metadata":{}},{"cell_type":"code","source":"print('Before fixing duplicate labels : ')\nprint(\"Number of unique species : \", train_df['species'].nunique())\n\ntrain_df['species'].replace({\n    'bottlenose_dolpin' : 'bottlenose_dolphin',\n    'kiler_whale' : 'killer_whale',\n    'beluga' : 'beluga_whale',\n    'globis' : 'short_finned_pilot_whale',\n    'pilot_whale' : 'short_finned_pilot_whale'\n},inplace =True)\n\nprint('\\nAfter fixing duplicate labels : ')\nprint(\"Number of unique species : \", train_df['species'].nunique())\n\n\ntrain_df['class'] = train_df['species'].apply(lambda x: x.split('_')[-1])\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.277763Z","iopub.status.idle":"2022-03-08T13:56:40.278054Z","shell.execute_reply.started":"2022-03-08T13:56:40.277907Z","shell.execute_reply":"2022-03-08T13:56:40.277922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Checking missing data\nLets check if there is any missing values in our dataset","metadata":{}},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.279231Z","iopub.status.idle":"2022-03-08T13:56:40.279534Z","shell.execute_reply.started":"2022-03-08T13:56:40.279373Z","shell.execute_reply":"2022-03-08T13:56:40.279389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(os.listdir('../input/happy-whale-and-dolphin/train_images'))","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.280522Z","iopub.status.idle":"2022-03-08T13:56:40.280806Z","shell.execute_reply.started":"2022-03-08T13:56:40.280656Z","shell.execute_reply":"2022-03-08T13:56:40.280671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualization\n### Looking at some random beauties \nIt's a great deal of fun to explore the data and play around with matplotlib","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (15,12))\nfor idx,i in enumerate(train_df.species.unique()):\n    plt.subplot(4,7,idx+1)\n    df = train_df[train_df['species'] ==i].reset_index(drop = True)\n    image_path = df.loc[rnd.randint(0, len(df))-1,'path']\n    img = Image.open(image_path)\n    img = img.resize((224,224))\n    plt.imshow(img)\n    plt.axis('off')\n    plt.title(i)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.281714Z","iopub.status.idle":"2022-03-08T13:56:40.281992Z","shell.execute_reply.started":"2022-03-08T13:56:40.281844Z","shell.execute_reply":"2022-03-08T13:56:40.281859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_species(df,species_name):\n    plt.figure(figsize = (12,12))\n    species_df = df[df['species'] ==species_name].reset_index(drop = True)\n    plt.suptitle(species_name)\n    for idx,i in enumerate(np.random.choice(species_df['path'],32)):\n        plt.subplot(8,8,idx+1)\n        image_path = i\n        img = Image.open(image_path)\n        img = img.resize((224,224))\n        plt.imshow(img)\n        plt.axis('off')\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.282864Z","iopub.status.idle":"2022-03-08T13:56:40.283171Z","shell.execute_reply.started":"2022-03-08T13:56:40.282999Z","shell.execute_reply":"2022-03-08T13:56:40.283014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Plotting more images from each species","metadata":{}},{"cell_type":"code","source":"for species in train_df['species'].unique():\n    #print('\\n\\n')\n    plot_species(train_df , species)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.284414Z","iopub.status.idle":"2022-03-08T13:56:40.284719Z","shell.execute_reply.started":"2022-03-08T13:56:40.284567Z","shell.execute_reply":"2022-03-08T13:56:40.284583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lets see some image by individual_id\nWe have to predict individual_id from image. So lets see how each individual looks like.","metadata":{}},{"cell_type":"code","source":"def plot_individual(df,individual_id):\n    plt.figure(figsize = (12,12))\n    species_df = df[df['individual_id'] ==individual_id].reset_index(drop = True)\n    plt.suptitle(individual_id)\n    for idx,i in enumerate(np.random.choice(species_df['path'],24)):\n        plt.subplot(8,8,idx+1)\n        image_path = i\n        img = Image.open(image_path)\n        img = img.resize((224,224))\n        plt.imshow(img)\n        plt.axis('off')\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.286085Z","iopub.status.idle":"2022-03-08T13:56:40.286416Z","shell.execute_reply.started":"2022-03-08T13:56:40.286249Z","shell.execute_reply":"2022-03-08T13:56:40.286265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Top 5 most frequent individual","metadata":{}},{"cell_type":"code","source":"top_5_ids = train_df.individual_id.value_counts().head(5)\nfor i in top_5_ids.index:\n    #print('\\n\\n')\n    plot_individual(train_df , i)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.287371Z","iopub.status.idle":"2022-03-08T13:56:40.287651Z","shell.execute_reply.started":"2022-03-08T13:56:40.287501Z","shell.execute_reply":"2022-03-08T13:56:40.287516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Top 5 least frequent individual\nWe will get duplicate images because many individual has only one training image.","metadata":{}},{"cell_type":"code","source":"last_5_ids = train_df.individual_id.value_counts().tail(5)\nfor i in last_5_ids.index:\n    #print('\\n\\n')\n    plot_individual(train_df , i)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.288437Z","iopub.status.idle":"2022-03-08T13:56:40.288714Z","shell.execute_reply.started":"2022-03-08T13:56:40.288565Z","shell.execute_reply":"2022-03-08T13:56:40.288580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets see some test images","metadata":{}},{"cell_type":"code","source":"t_df = pd.read_csv('../input/happy-whale-and-dolphin/sample_submission.csv')\nt_df['path'] = '../input/happy-whale-and-dolphin/test_images/' + t_df['image']\n\ndef plot_testimages(df):\n    plt.figure(figsize = (12,12))\n    plt.suptitle('Test Images')\n    for idx,i in enumerate(np.random.choice(df['path'],48)):\n        plt.subplot(8,8,idx+1)\n        image_path = i\n        img = Image.open(image_path)\n        img = img.resize((224,224))\n        plt.imshow(img)\n        plt.axis('off')\n    plt.tight_layout()\n    plt.show()\n\nplot_testimages(t_df)\ndel t_df","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.289514Z","iopub.status.idle":"2022-03-08T13:56:40.289794Z","shell.execute_reply.started":"2022-03-08T13:56:40.289645Z","shell.execute_reply":"2022-03-08T13:56:40.289660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations regarding handpicked images\n* There are some abnormal images in both train and test dataset\n* Some training images contains people, boats, birds, penguins etc\n* Many training images are cropped but some are not.\n* The uncropped images must be taken care of.\n* There are some images take from under water","metadata":{}},{"cell_type":"markdown","source":"## Class Distribution Analysis\nIn this section we will be analyzing the number of training and test samples in each class. It will give us a better understanding of our dataset and provide us the necessary information to preprocess our dataset before the training phase.","metadata":{}},{"cell_type":"code","source":"plot = sns.countplot(x = train_df['class'], color = '#2596be')\nsns.despine()\nplot.set_title('Class Distribution\\n', font = 'serif', x = 0.1, y=1, fontsize = 16);\nplot.set_ylabel(\"Count\", x = 0.02, font = 'serif', fontsize = 12)\nplot.set_xlabel(\"Specie\", fontsize = 12, font = 'serif')\n\nfor p in plot.patches:\n    plot.annotate(format(p.get_height(), '.0f'), (p.get_x() + p.get_width() / 2, p.get_height()), \n       ha = 'center', va = 'center', xytext = (0, -20),font = 'serif', textcoords = 'offset points', size = 15)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.290722Z","iopub.status.idle":"2022-03-08T13:56:40.291000Z","shell.execute_reply.started":"2022-03-08T13:56:40.290851Z","shell.execute_reply":"2022-03-08T13:56:40.290867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Percentage of images of whale and dolphin in the dataset¶**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(5,5))\nclass_cnt = train_df.groupby(['class']).size().reset_index(name = 'counts')\ncolors = sns.color_palette('Paired')[0:9]\nplt.pie(class_cnt['counts'], labels=class_cnt['class'], colors=colors, autopct='%1.1f%%')\nplt.legend(loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.291898Z","iopub.status.idle":"2022-03-08T13:56:40.292223Z","shell.execute_reply.started":"2022-03-08T13:56:40.292040Z","shell.execute_reply":"2022-03-08T13:56:40.292056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Number of training images of each species**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8,8))\nsns.countplot(data=train_df, y = 'species',  palette='crest', dodge=False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.293333Z","iopub.status.idle":"2022-03-08T13:56:40.293621Z","shell.execute_reply.started":"2022-03-08T13:56:40.293469Z","shell.execute_reply":"2022-03-08T13:56:40.293485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Number of training images of each species of whale and dolphin**","metadata":{}},{"cell_type":"code","source":"fig,ax = plt.subplots(1,2,figsize=(10,5))\n\nwhales = train_df[train_df['class']=='whale']\ndolphins = train_df[train_df['class']!='whale']\n\nsns.countplot(y=\"species\", data=whales, order=whales.iloc[0:][\"species\"].value_counts().index, ax=ax[0], color = \"#0077b6\")\nax[0].set_title('Most frequent whales')\nax[0].set_ylabel(None)\n    \nsns.countplot(y=\"species\", data=dolphins,order=dolphins.iloc[0:][\"species\"].value_counts().index, ax=ax[1], color = \"#90e0ef\")\nax[1].set_title('Most frequent dolphins')\nax[1].set_ylabel(None)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.294708Z","iopub.status.idle":"2022-03-08T13:56:40.294993Z","shell.execute_reply.started":"2022-03-08T13:56:40.294843Z","shell.execute_reply":"2022-03-08T13:56:40.294858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Number of training images of top 10 individuals**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12,4))\ntop_ten_ids = train_df.individual_id.value_counts().head(24)\ntop_ten_ids = pd.DataFrame({'individual_id':top_ten_ids.index, 'frequency':top_ten_ids.values})\n\nplt.bar(top_ten_ids['individual_id'],top_ten_ids['frequency'],width = 0.8,color='c',zorder=4)\nplt.xticks(rotation=90)\nplt.ylabel(\"frequency\")\nplt.xlabel(\"Individual Ids\")\nplt.title(\"Top 10 Individual Ids used by frequency\")\nplt.grid(visible = True, color ='grey',linestyle ='-', linewidth = 0.9,alpha = 0.2, zorder=0)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.296964Z","iopub.status.idle":"2022-03-08T13:56:40.297510Z","shell.execute_reply.started":"2022-03-08T13:56:40.297316Z","shell.execute_reply":"2022-03-08T13:56:40.297346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Plot the value count graph of each individual**","metadata":{}},{"cell_type":"code","source":"train_df['individual_id'].value_counts().plot()\nplt.xticks(rotation=90)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.298790Z","iopub.status.idle":"2022-03-08T13:56:40.299123Z","shell.execute_reply.started":"2022-03-08T13:56:40.298947Z","shell.execute_reply":"2022-03-08T13:56:40.298972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Number of unique individuals in the dataset**","metadata":{}},{"cell_type":"code","source":"len(train_df.individual_id.unique())","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.306618Z","iopub.status.idle":"2022-03-08T13:56:40.306950Z","shell.execute_reply.started":"2022-03-08T13:56:40.306770Z","shell.execute_reply":"2022-03-08T13:56:40.306793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Image count of individuals**","metadata":{}},{"cell_type":"code","source":"train_df['count'] = train_df.groupby('individual_id',as_index=False)['individual_id'].transform(lambda x: x.count())\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.308327Z","iopub.status.idle":"2022-03-08T13:56:40.308644Z","shell.execute_reply.started":"2022-03-08T13:56:40.308473Z","shell.execute_reply":"2022-03-08T13:56:40.308496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Individuals with only one training image**","metadata":{}},{"cell_type":"code","source":"train_df[train_df['count']==1]","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.309719Z","iopub.status.idle":"2022-03-08T13:56:40.310073Z","shell.execute_reply.started":"2022-03-08T13:56:40.309898Z","shell.execute_reply":"2022-03-08T13:56:40.309922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Percentage of Individuals with less then 5 images**","metadata":{}},{"cell_type":"code","source":"tmp = train_df[train_df['count']<=4]\nlen(tmp)/len(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.310829Z","iopub.status.idle":"2022-03-08T13:56:40.311115Z","shell.execute_reply.started":"2022-03-08T13:56:40.310961Z","shell.execute_reply":"2022-03-08T13:56:40.310977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Percentage of Individuals with more then 20 images**","metadata":{}},{"cell_type":"code","source":"count = 0\nfor i in train_df['count']:\n    if(i > 21):\n        count += 1\nprint(count/len(train_df))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observation Regarding Class Distribution\nThere is a huge disbalance in the data. There are many classes with only one or several samples:\n\n* Total Number of individuals are 15587\n* 9258 individuals have just one image\n* Single whale with most images have 400 of them\n* Images dsitribution:\n    * almost 40% comes from whales with 4 or less images.\n    * almost 23% comes from whales with 5-20 images.\n    * rest 37% comes from individual with >20 images.","metadata":{}},{"cell_type":"markdown","source":"## Image Resolutions","metadata":{}},{"cell_type":"code","source":"widths, heights = [], []\n\nfor path in tqdm(train_df[\"path\"]):\n    width, height = Image.open(path).size\n    widths.append(width)\n    heights.append(height)\n    \ntrain_df[\"width\"] = widths\ntrain_df[\"height\"] = heights\ntrain_df[\"dimension\"] = train_df[\"width\"] * train_df[\"height\"]","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.313828Z","iopub.status.idle":"2022-03-08T13:56:40.314124Z","shell.execute_reply.started":"2022-03-08T13:56:40.313969Z","shell.execute_reply":"2022-03-08T13:56:40.313984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Lets see some small images**","metadata":{}},{"cell_type":"code","source":"train_df.sort_values('width').head(84)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.315471Z","iopub.status.idle":"2022-03-08T13:56:40.315962Z","shell.execute_reply.started":"2022-03-08T13:56:40.315785Z","shell.execute_reply":"2022-03-08T13:56:40.315805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Color Analysis\nWe need to do some color analysis to get an ida about the augmentation technique needed for this problem","metadata":{}},{"cell_type":"code","source":"def is_grey_scale(givenImage):\n    w,h = givenImage.size\n    for i in range(w):\n        for j in range(h):\n            r,g,b = givenImage.getpixel((i,j))\n            if r != g != b: return False\n    return True","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.316742Z","iopub.status.idle":"2022-03-08T13:56:40.317343Z","shell.execute_reply.started":"2022-03-08T13:56:40.317142Z","shell.execute_reply":"2022-03-08T13:56:40.317182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Check color scale of Train images**","metadata":{}},{"cell_type":"code","source":"sampleFrac = 0.1\n#get our sampled images\nisGreyList = []\nfor imageName in train_df['path'].sample(frac=sampleFrac):\n    val = Image.open(imageName).convert('RGB')\n    isGreyList.append(is_grey_scale(val))\nprint(np.sum(isGreyList) / len(isGreyList))\ndel isGreyList","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.318297Z","iopub.status.idle":"2022-03-08T13:56:40.318869Z","shell.execute_reply.started":"2022-03-08T13:56:40.318681Z","shell.execute_reply":"2022-03-08T13:56:40.318700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Check color scale of Test images**","metadata":{}},{"cell_type":"code","source":"sampleFrac = 0.1\n#get our sampled images\nisGreyList_test = []\nfor imageName in pred_df['path'].sample(frac=sampleFrac):\n    val = Image.open(imageName).convert('RGB')\n    isGreyList_test.append(is_grey_scale(val))\nprint(np.sum(isGreyList_test) / len(isGreyList_test))\ndel isGreyList_test","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.319932Z","iopub.status.idle":"2022-03-08T13:56:40.320384Z","shell.execute_reply.started":"2022-03-08T13:56:40.320217Z","shell.execute_reply":"2022-03-08T13:56:40.320236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Get mean intensity for each channel RGB**","metadata":{}},{"cell_type":"code","source":"def get_rgb_men(row):\n    img = cv2.imread(row['path'])\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    return np.sum(img[:,:,0]), np.sum(img[:,:,1]), np.sum(img[:,:,2])\n\ntqdm.pandas()\ntrain_df['R'], train_df['G'], train_df['B'] = zip(*train_df.progress_apply(lambda row: get_rgb_men(row), axis=1) )","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.321305Z","iopub.status.idle":"2022-03-08T13:56:40.321716Z","shell.execute_reply.started":"2022-03-08T13:56:40.321559Z","shell.execute_reply":"2022-03-08T13:56:40.321576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_color_dist(df, count):\n    fig, axr = plt.subplots(count,2,figsize=(15,15))\n    for idx, i in enumerate(np.random.choice(df['path'], count)):\n        img = cv2.imread(i)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        axr[idx,0].imshow(img)\n        axr[idx,0].axis('off')\n        axr[idx,1].set_title('R={:.0f}, G={:.0f}, B={:.0f} '.format(np.mean(img[:,:,0]), np.mean(img[:,:,1]), np.mean(img[:,:,2]))) \n        x, y = np.histogram(img[:,:,0], bins=255)\n        axr[idx,1].bar(y[:-1], x, label='R', alpha=0.8, color='red')\n        x, y = np.histogram(img[:,:,1], bins=255)\n        axr[idx,1].bar(y[:-1], x, label='G', alpha=0.8, color='green')\n        x, y = np.histogram(img[:,:,2], bins=255)\n        axr[idx,1].bar(y[:-1], x, label='B', alpha=0.8, color='blue')\n        axr[idx,1].legend()\n        axr[idx,1].axis('off')","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.322639Z","iopub.status.idle":"2022-03-08T13:56:40.323224Z","shell.execute_reply.started":"2022-03-08T13:56:40.323009Z","shell.execute_reply":"2022-03-08T13:56:40.323031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Red images and their color distribution**<br>\nSince we are picking random images, some image may appear multiple times","metadata":{}},{"cell_type":"code","source":"df = train_df[((train_df['B']*1.05) < train_df['R']) & ((train_df['G']*1.05) < train_df['R'])]\nshow_color_dist(df, 8)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.323946Z","iopub.status.idle":"2022-03-08T13:56:40.324266Z","shell.execute_reply.started":"2022-03-08T13:56:40.324076Z","shell.execute_reply":"2022-03-08T13:56:40.324091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Blue images and their color distribution**","metadata":{}},{"cell_type":"code","source":"df = train_df[(train_df['B'] > 1.3*train_df['R']) & (train_df['B'] > 1.3*train_df['G'])]\nshow_color_dist(df, 8)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.325761Z","iopub.status.idle":"2022-03-08T13:56:40.326311Z","shell.execute_reply.started":"2022-03-08T13:56:40.326102Z","shell.execute_reply":"2022-03-08T13:56:40.326122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Green images and their color distribution**","metadata":{}},{"cell_type":"code","source":"df = train_df[(train_df['G'] > 1.05*train_df['R']) & (train_df['G'] > 1.05*train_df['B'])]\nshow_color_dist(df, 8)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.327076Z","iopub.status.idle":"2022-03-08T13:56:40.327680Z","shell.execute_reply.started":"2022-03-08T13:56:40.327487Z","shell.execute_reply":"2022-03-08T13:56:40.327508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Observation Regarding Color Distribution\n1. We see that around 3% of the images in the training set are greyscale. While 1% in the Test set are greyscale.\n2. Some whales have yellow spots and some images are reddish.This can happened due to sunset.\n3. This suggests that we need to create image transformations that are very agnostic to the RGB spectrum (i.e. bump up the number of greyscaled images in the smaller classes).","metadata":{}},{"cell_type":"code","source":"def plot_augimages(paths, datagen):\n    plt.figure(figsize = (14,28))\n    plt.suptitle('Augmented Images')\n    \n    midx = 0\n    for path in paths:\n        data = Image.open(path)\n        data = data.resize((224,224))\n        samples = expand_dims(data, 0)\n        it = datagen.flow(samples, batch_size=1)\n    \n        # Show Original Image\n        plt.subplot(10,5, midx+1)\n        plt.imshow(data)\n        plt.axis('off')\n    \n        # Show Augmented Images\n        for idx, i in enumerate(range(4)):\n            midx += 1\n            plt.subplot(10,5, midx+1)\n            \n            batch = it.next()\n            image = batch[0].astype('uint8')\n            plt.imshow(image)\n            plt.axis('off')\n        midx += 1\n    \n    plt.tight_layout()\n    plt.show()\n\n    \ndatagen = ImageDataGenerator(\n    rotation_range=20,\n    zoom_range=0.10,\n    brightness_range=[0.6,1.4],\n    channel_shift_range=0.7,\n    width_shift_range=0.15,\n    height_shift_range=0.15,\n    shear_range=0.15,\n    horizontal_flip=True,\n    fill_mode='nearest'\n) \nplot_augimages(np.random.choice(train_df['path'],10), datagen)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.328712Z","iopub.status.idle":"2022-03-08T13:56:40.329254Z","shell.execute_reply.started":"2022-03-08T13:56:40.329060Z","shell.execute_reply":"2022-03-08T13:56:40.329080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing\nEncoding Labels","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\n\nX = train_df.iloc[:, 3].values\ny = train_df.iloc[:, 2].values\n\nlabel_encoder = LabelEncoder()\ny = label_encoder.fit_transform(y)\nonehot_encoder = OneHotEncoder(sparse=False)\ny = y.reshape(len(y), 1)\ny = onehot_encoder.fit_transform(y)","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.330013Z","iopub.status.idle":"2022-03-08T13:56:40.330711Z","shell.execute_reply.started":"2022-03-08T13:56:40.330399Z","shell.execute_reply":"2022-03-08T13:56:40.330433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-08T13:56:40.332342Z","iopub.status.idle":"2022-03-08T13:56:40.333118Z","shell.execute_reply.started":"2022-03-08T13:56:40.332933Z","shell.execute_reply":"2022-03-08T13:56:40.332956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}