{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![](https://blogger.googleusercontent.com/img/a/AVvXsEjVBFVJg4z5YeCrBaFoTfJ7wcZxlFt9WrepKwOADEBtZe96pJI1KKryhjQupntMedLQgwOTdMTEHkyZm-LAHCrU7JD_1UNjoQdTGzvVe1-XVfA1rocSbCCmLfLqS7sm-_wKIsYwm5hW35RDF0wAhHYkL7MoWLM4QsowyQsNRAzb8xMaKhvgDVoMwH-b=s782)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nfrom scipy import stats\nimport glob\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2\nfrom glob import glob\nimport gc\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n%matplotlib inline\nimport matplotlib.pyplot as plt\nfrom IPython.display import Image,display\nimport seaborn as sns\nimport matplotlib.image as mpimg\nimport scipy.spatial.distance as dist\nfrom sklearn.model_selection import train_test_split\nimport os\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-04T08:15:23.494366Z","iopub.execute_input":"2022-02-04T08:15:23.495005Z","iopub.status.idle":"2022-02-04T08:15:25.067133Z","shell.execute_reply.started":"2022-02-04T08:15:23.494910Z","shell.execute_reply":"2022-02-04T08:15:25.066006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-04T08:15:25.069155Z","iopub.execute_input":"2022-02-04T08:15:25.069442Z","iopub.status.idle":"2022-02-04T08:15:25.204517Z","shell.execute_reply.started":"2022-02-04T08:15:25.069411Z","shell.execute_reply":"2022-02-04T08:15:25.203409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for Duplicates\ntrain.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2022-02-04T08:15:25.206297Z","iopub.execute_input":"2022-02-04T08:15:25.206547Z","iopub.status.idle":"2022-02-04T08:15:25.252564Z","shell.execute_reply.started":"2022-02-04T08:15:25.206514Z","shell.execute_reply":"2022-02-04T08:15:25.251313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Training data size\",train.shape)\nsubmission = pd.read_csv(\"../input/happy-whale-and-dolphin/sample_submission.csv\")\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-04T08:15:25.255642Z","iopub.execute_input":"2022-02-04T08:15:25.256008Z","iopub.status.idle":"2022-02-04T08:15:25.329723Z","shell.execute_reply.started":"2022-02-04T08:15:25.255960Z","shell.execute_reply":"2022-02-04T08:15:25.328694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['individual_id'].value_counts().hist()","metadata":{"execution":{"iopub.status.busy":"2022-02-04T08:15:25.331067Z","iopub.execute_input":"2022-02-04T08:15:25.331398Z","iopub.status.idle":"2022-02-04T08:15:25.653202Z","shell.execute_reply.started":"2022-02-04T08:15:25.331357Z","shell.execute_reply":"2022-02-04T08:15:25.652163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# missing data in training data \ntotal = train.isnull().sum().sort_values(ascending = False)\npercent = (train.isnull().sum()/train.isnull().count()).sort_values(ascending = False)\nmissing_train_data = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\nmissing_train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-04T08:15:25.654399Z","iopub.execute_input":"2022-02-04T08:15:25.654622Z","iopub.status.idle":"2022-02-04T08:15:25.730453Z","shell.execute_reply.started":"2022-02-04T08:15:25.654586Z","shell.execute_reply":"2022-02-04T08:15:25.729843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Occurance of individual_id in decreasing order(Top categories)\ntemp = pd.DataFrame(train.individual_id.value_counts().head(8))\ntemp.reset_index(inplace=True)\ntemp.columns = ['individual_id','count']\ntemp","metadata":{"execution":{"iopub.status.busy":"2022-02-04T08:15:25.731547Z","iopub.execute_input":"2022-02-04T08:15:25.732198Z","iopub.status.idle":"2022-02-04T08:15:25.758913Z","shell.execute_reply.started":"2022-02-04T08:15:25.732160Z","shell.execute_reply":"2022-02-04T08:15:25.758043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wd = pd.DataFrame(train.groupby(['individual_id'])['individual_id'].count())\nwd.rename(columns={'individual_id': 'Count_Images'}, inplace=True)\nwd.reset_index(inplace=True)\nwd.sort_values(by=['Count_Images'],ascending=False, inplace=True)\nwd['Cummulative_Count'] = wd['Count_Images'].cumsum()\nwd['Cummulative_Pctg']= wd['Cummulative_Count']/wd['Count_Images'].sum()\nwd['Row_id'] = np.arange(len(wd))\nfig = plt.figure()\nax = plt.axes()\nax.plot(wd['Row_id'], wd['Cummulative_Pctg']);\nax.set(xlabel='Count of Classes', ylabel='Cummulative %',\n       title='Cummulative distribution of images by class count');","metadata":{"execution":{"iopub.status.busy":"2022-02-04T08:15:25.760034Z","iopub.execute_input":"2022-02-04T08:15:25.760236Z","iopub.status.idle":"2022-02-04T08:15:26.034148Z","shell.execute_reply.started":"2022-02-04T08:15:25.760212Z","shell.execute_reply":"2022-02-04T08:15:26.033301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of classes that contribute 95% of total images is: \"+ str(len(wd[wd['Cummulative_Pctg']<=0.95])))\nprint(\"Number of classes that contribute 99% of total images is: \"+ str(len(wd[wd['Cummulative_Pctg']<=0.99])))\nprint(str(train['individual_id'].nunique()- len(wd[wd['Cummulative_Pctg']<=0.99]))+ \" Classes contribute just about remaining 1% of images\")\nprint(\"Number of classes with just two images is:\"+ str(len(wd[wd['Count_Images']<=2])))","metadata":{"execution":{"iopub.status.busy":"2022-02-04T08:15:26.035263Z","iopub.execute_input":"2022-02-04T08:15:26.035504Z","iopub.status.idle":"2022-02-04T08:15:26.058943Z","shell.execute_reply.started":"2022-02-04T08:15:26.035474Z","shell.execute_reply":"2022-02-04T08:15:26.058113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the most frequent landmark_ids\nplt.figure(figsize = (9, 8))\nplt.title('Most frequent individual_id')\nsns.set_color_codes(\"pastel\")\nsns.barplot(x=\"individual_id\", y=\"count\", data=temp,\n            label=\"Count\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-04T08:15:26.060355Z","iopub.execute_input":"2022-02-04T08:15:26.060601Z","iopub.status.idle":"2022-02-04T08:15:26.300819Z","shell.execute_reply.started":"2022-02-04T08:15:26.060554Z","shell.execute_reply":"2022-02-04T08:15:26.300150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Occurance of landmark_id in increasing order\ntemp = pd.DataFrame(train.individual_id.value_counts().tail(8))\ntemp.reset_index(inplace=True)\ntemp.columns = ['individual_id','count']\ntemp","metadata":{"execution":{"iopub.status.busy":"2022-02-04T08:15:26.304269Z","iopub.execute_input":"2022-02-04T08:15:26.304638Z","iopub.status.idle":"2022-02-04T08:15:26.329545Z","shell.execute_reply.started":"2022-02-04T08:15:26.304604Z","shell.execute_reply":"2022-02-04T08:15:26.328641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the least frequent landmark_ids\nplt.figure(figsize = (9, 8))\nplt.title('Least frequent individual_id')\nsns.set_color_codes(\"pastel\")\nsns.barplot(x=\"individual_id\", y=\"count\", data=temp,\n            label=\"Count\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-04T08:15:26.331115Z","iopub.execute_input":"2022-02-04T08:15:26.331346Z","iopub.status.idle":"2022-02-04T08:15:26.556621Z","shell.execute_reply.started":"2022-02-04T08:15:26.331318Z","shell.execute_reply":"2022-02-04T08:15:26.555580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-02-04T08:15:26.558145Z","iopub.execute_input":"2022-02-04T08:15:26.559054Z","iopub.status.idle":"2022-02-04T08:15:26.595564Z","shell.execute_reply.started":"2022-02-04T08:15:26.559007Z","shell.execute_reply":"2022-02-04T08:15:26.594517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\na = train['individual_id'].tolist()\nletter_counts = Counter(a)\ndf = pd.DataFrame.from_dict(letter_counts, orient='index')\ndf.plot(kind='bar')","metadata":{"execution":{"iopub.status.busy":"2022-02-04T08:15:26.596865Z","iopub.execute_input":"2022-02-04T08:15:26.597164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of classes under 20 occurences\",(train['individual_id'].value_counts() <= 20).sum(),'out of total number of categories',len(train['individual_id'].unique()))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set()\nplt.title('Training set: number of images per class(line plot)')\nsns.set_color_codes(\"pastel\")\nindividual_id = pd.DataFrame(train['individual_id'].value_counts())\nindividual_id.reset_index(inplace=True)\nindividual_id.columns = ['individual_id','count']\nax = individual_id['count'].plot(logy=True, grid=True)\nlocs, labels = plt.xticks()\nplt.setp(labels, rotation=30)\nax.set(xlabel=\"individual_id\", ylabel=\"Number of images\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize outliers, min/max or quantiles of the landmarks count\nsns.set()\nax = individual_id.boxplot(column='count')\nax.set_yscale('log')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mainPath = '../input/happy-whale-and-dolphin/train_images/'\nall_img_paths = [y for x in os.walk(mainPath) for y in glob(os.path.join(x[0], '*.jpg'))]\nall_filenames = []\nfor filepath in all_img_paths:\n    FileName = os.path.basename(filepath)\n    all_filenames.append(FileName)\npath_dict = dict(zip(all_filenames,all_img_paths))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wd","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##Getting the list of ids where top 5 and bottom 5 categories\ntop5_cats = wd[wd['Row_id']<=4].individual_id.tolist()\nbottom5_cats = wd[wd['Row_id']>=(wd['Row_id'].max()-4)].individual_id.tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Top 5 image categories are : \" + str(top5_cats))\nprint(\"Bottom 5 image categories are : \" + str(bottom5_cats))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['image']=train['individual_id']+str(\".jpg\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img0 = cv2.imread('../input/happy-whale-and-dolphin/train_images/00021adfb725ed.jpg')\n\ntrain_hist = plt.hist(img0.ravel(), bins = 256, color = 'orange', )\ntrain_hist = plt.hist(img0[:, :, 0].ravel(), bins = 256, color = 'red', alpha = 0.5)\ntrain_hist = plt.hist(img0[:, :, 1].ravel(), bins = 256, color = 'Green', alpha = 0.5)\ntrain_hist = plt.hist(img0[:, :, 2].ravel(), bins = 256, color = 'Blue', alpha = 0.5)\ntrain_hist = plt.xlabel('Intensity Value')\ntrain_hist = plt.ylabel('Count')\ntrain_hist = plt.legend(['Total', 'Red Channel', 'Green Channel', 'Blue Channel'])\nprint('Intensity Histogram of Train Image')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img1 = cv2.imread('../input/happy-whale-and-dolphin/test_images/000110707af0ba.jpg')\n\ntest_hist = plt.hist(img1.ravel(), bins = 256, color = 'orange', )\ntest_hist = plt.hist(img1[:, :, 0].ravel(), bins = 256, color = 'red', alpha = 0.5)\ntest_hist = plt.hist(img1[:, :, 1].ravel(), bins = 256, color = 'Green', alpha = 0.5)\ntest_hist = plt.hist(img1[:, :, 2].ravel(), bins = 256, color = 'Blue', alpha = 0.5)\ntest_hist = plt.xlabel('Intensity Value')\ntest_hist = plt.ylabel('Count')\ntest_hist = plt.legend(['Total', 'Red Channel', 'Green Channel', 'Blue Channel'])\nprint('Intensity Histogram of Test Image')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_list = glob('../input/happy-whale-and-dolphin/train_images/*')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.rcParams[\"axes.grid\"] = False\nf, axarr = plt.subplots(4, 3, figsize=(24, 22))\n\ncurr_row = 0\nfor i in range(12):\n    example = cv2.imread(train_list[i])\n    example = example[:,:,::-1]\n    \n    col = i%4\n    axarr[col, curr_row].imshow(example)\n    if col == 3:\n        curr_row += 1","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}