{"nbformat_minor": 1, "nbformat": 4, "cells": [{"source": "**Humpback Whale Identification - CNN with Keras**", "cell_type": "markdown", "metadata": {"_uuid": "6eaa7f762391b0666eb9df4f1324a54f8b34b05d", "trusted": true}}, {"source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mplimg\nfrom matplotlib.pyplot import imshow\nfrom IPython.display import  Image\n\nfrom sklearn.preprocessing import LabelEncoder,OneHotEncoder\n\nfrom keras import layers\nfrom keras.preprocessing import image\nfrom keras.applications.imagenet_utils import preprocess_input\nfrom keras.layers import Input,Dense,Activation,BatchNormalization,Conv2D,Flatten\nfrom keras.layers import AveragePooling2D,MaxPooling2D,Dropout\nfrom keras.models import Model\n\nimport keras.backend as k\nfrom keras.models import Sequential\n\nimport warnings\nwarnings.simplefilter(\"ignore\", category=DeprecationWarning)\n\n", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "8f2839f25d086af736a60e9eeb907d3b93b6e0e5", "trusted": true, "_cell_guid": "b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}}, {"source": "os.listdir('../input/')", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "79ee00b5d581c38847e6cf1c57895ce747acbeb2", "trusted": true}}, {"source": "train_df=pd.read_csv('../input/train.csv')\n\ntrain_df.head()", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "d629ff2d2480ee46fbb7e2d37f6b5fab8052498a", "trusted": true, "_cell_guid": "79c7e3d0-c299-4dcb-8224-4455121ee9b0"}}, {"source": "train_df.columns", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "e6d0af8fe0d303e473bf2ec1eddceeb7743cdaab", "trusted": true}}, {"source": "# Random image\n\nfrom IPython.display import  Image\nimport random\n\nImage(filename='../input/train/'+random.choice(train_df.Image))\n", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "5c20160f233a3b81ff93bfbd4db8985e6ba5206a", "trusted": true}}, {"source": "**Preprocessing the images in training then converting into Array**", "cell_type": "markdown", "metadata": {"_uuid": "716e082e30bfd8dd15299271f8dcc25503e7d037"}}, {"source": "def\nX_train=np.zeros((train_df.shape[0],100,100,3))\ncount=0\n\nfor fig in train_df['Image']:\n    img=image.load_img('../input/train/'+fig,target_size=(100,100,3))\n    x=image.img_to_array(img)\n#     x=preprocess_input(img)\n    \n    X_train[count]=x\n    \nreturn X_train\n\n", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "974ac880103c97cff7b4364fc47ccc1d1f4a75a2", "trusted": true}}, {"source": "def prepareImage(data,n,datset):\n    print('Printing Images')\n    X_train=np.zeros((n,100,100,3))\n    count=0\n    \n    for fig in data['Image']:\n        img=image.load_img('../input/'+datset+'/'+ fig,target_size=(100,100,3))\n        x=image.img_to_array(img)\n        x=preprocess_input(x)\n        \n        X_train[count]=x\n        \n        if(count%500==0):\n            print(\"ProcessingImage : \" , count+1,\", \",fig)\n        count +=1\n        \n    return X_train     \n   ", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "ff72e60d4747392981cc530a034b2e7e6f408985", "trusted": true}}, {"source": "X = prepareImage(train_df, train_df.shape[0], \"train\")\nX /= 255", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "70c618dc64266d69972d04344925af1f68ade122", "trusted": true}}, {"source": "", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "12394d1fcfe16806a079a402df22470dc86acb53", "trusted": true}}, {"source": "def prepare_labels(y):\n    values=np.array(y)\n    label_encoder=LabelEncoder()\n    integer_encoded=label_encoder.fit_transform(values)\n    print(integer_encoded)\n    \n    onehot_encoder=OneHotEncoder(sparse=False)\n    integer_encoded=integer_encoded.reshape(len(integer_encoded),1)\n    onehot_encoded=onehot_encoder.fit_transform(integer_encoded)\n    print(onehot_encoded)\n    y=onehot_encoded\n    return y,label_encoder", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "e857487bfdff18eee4f607be08c0b9819905959b", "trusted": true}}, {"source": "y,label_encoder=prepare_labels(train_df['Id'])", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "62245f37dfb48d1ddca7e24f6e0099f2ae6c1cff", "trusted": true}}, {"source": "y.shape\n", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "d60387c80f3db93858184032feb88c971991c358", "trusted": true}}, {"source": "# CNN architecture \n\nmodel=Sequential()\n\n# convolution\nmodel.add(Conv2D(32,(7,7),strides=(1,1),name = 'conv0', input_shape=(100,100,3)))\n\n#Batch Normalization\n\nmodel.add(BatchNormalization(axis=3,name='bn0'))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D((2,2),name='max_pool'))\n\nmodel.add(Conv2D(64,(3,3),strides=(1,1),name='conv1'))\nmodel.add(Activation('relu'))\nmodel.add(AveragePooling2D((3,3),name='avg_pool'))\n\nmodel.add(Flatten())\nmodel.add(Dense(500,activation='relu',name='r1'))\nmodel.add(Dropout(0.8))\nmodel.add(Dense(y.shape[1],activation='softmax',name='sm'))\n\nmodel.compile(loss='categorical_crossentropy',optimizer='adam',metrics=['accuracy'])\nmodel.summary()", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "71a3ff44f89c4f27137a459e3946876bc3417062", "trusted": true}}, {"source": "history=model.fit(X,y,epochs=15,batch_size=100,verbose=1)\ngc.collect()", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "4b7e12cf9d1a070b00fb6fcef5131c3d7475bd0a", "trusted": true}}, {"source": "import gc\ngc.collect()", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "87520715a803739ed2906deb68a5558656cb17e2", "trusted": true}}, {"source": "plt.plot(history.history['acc'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.show()", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "9573fad2bd7562e02bcfa40c6cea0149544d19ef", "trusted": true}}, {"source": "test = os.listdir(\"../input/test/\")\nprint(len(test))", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "124113f29073021fa0005c71d84efd40e976986f", "trusted": true}}, {"source": "col = ['Image']\ntest_df = pd.DataFrame(test, columns=col)\ntest_df['Id'] = ''", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "197ae48f48ff0f2e3d6891eda7143798e054a992", "trusted": true}}, {"source": "X = prepareImage(test_df, test_df.shape[0], \"test\")\nX /= 255", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "26746a35b2bc5b256b6d5f79debbf806200477f0", "trusted": true}}, {"source": "predictions = model.predict(np.array(X), verbose=1)", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "e7181f23f7c2880b93c9eaa72a426da678c6932d", "trusted": true}}, {"source": "for i, pred in enumerate(predictions):\n    test_df.loc[i, 'Id'] = ' '.join(label_encoder.inverse_transform(pred.argsort()[-5:][::-1]))", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "2c009d5319e1ac14b6b0de1417f599721c690765", "trusted": true}}, {"source": "test_df.head(10)\ntest_df.to_csv('submission.csv', index=False)", "cell_type": "code", "execution_count": null, "outputs": [], "metadata": {"_uuid": "fc48e869ef317f51dbde5be72a852f6abe8f6e00", "trusted": true}}], "metadata": {"kernelspec": {"display_name": "Python 3", "name": "python3", "language": "python"}, "language_info": {"mimetype": "text/x-python", "nbconvert_exporter": "python", "name": "python", "pygments_lexer": "ipython3", "version": "3.6.6", "file_extension": ".py", "codemirror_mode": {"version": 3, "name": "ipython"}}}}