{"cells":[{"metadata":{"_uuid":"0e01ec8f6fe648b36fd52a058ba8684467601f01"},"cell_type":"markdown","source":"## Humpback Whale prediction using Keras"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b2424d801240d9c454b7bc5684956626a0ffde0"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"46cbe06dad9c2c6c479929daa2a1ff98c2de447d"},"cell_type":"code","source":"train['Id'].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4a329470bf6e44eca8561c67ea4196e46c929cb2"},"cell_type":"code","source":"y_train = train['Id']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c7548d5c5d13fafb72722bc7a6ddd3e0e9fbd195"},"cell_type":"code","source":"from keras.preprocessing import image\nfrom keras.applications.imagenet_utils import preprocess_input\n\ndef prepareImages(train, shape, path):\n    \n    x_train = np.zeros((shape, 100, 100, 3))\n    count = 0\n    \n    for fig in train['Image']:\n        \n        #load images into images of size 100x100x3\n        img = image.load_img(\"../input/\"+path+\"/\"+fig, target_size=(100, 100, 3))\n        x = image.img_to_array(img)\n        x = preprocess_input(x)\n\n        x_train[count] = x\n        if (count%500 == 0):\n            print(\"Processing image: \", count+1, \", \", fig)\n        count += 1\n    \n    return x_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e4053ceccc01d179ba72df51c9da6ebbcd605d7e"},"cell_type":"code","source":"X_train = prepareImages(train, train.shape[0], 'train')\nX_train/=255","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b9f46779e6f03fee79e61dcfe76983a4c23927f"},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom keras.utils.np_utils import to_categorical","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aacf272b6cd29ea3b90c92ce85784f1997aa3b88"},"cell_type":"code","source":"label_encoder = LabelEncoder()\ny_train = label_encoder.fit_transform(y_train)\ny_train = to_categorical(y_train, num_classes = 5005)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3d078e65a704d744f59c552f192beb0570439b39"},"cell_type":"code","source":"y_train.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"825bce00af2cce0aedb6198c943e146d5d59b13c"},"cell_type":"markdown","source":"### Preparing the model"},{"metadata":{"trusted":true,"_uuid":"bb121886fa068c3b7c950d9a627f7e3b2db66a09"},"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.layers import Dropout, Flatten, MaxPooling2D, Conv2D, Dense\nfrom keras.layers.normalization import BatchNormalization","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8f62db3eefe72d357fa44f673be01926b5f0b00e"},"cell_type":"code","source":"model = Sequential()\n\nmodel.add(Conv2D(32, (5,5), strides = (1,1), padding='same', activation = 'relu', input_shape = (100, 100, 3)))\nmodel.add(Conv2D(32, (5,5), strides = (1,1), padding = 'same', activation='relu'))\nmodel.add(MaxPooling2D((2,2)))\n\nmodel.add(Conv2D(32, (3,3), strides = (2,2), padding='same', activation='relu'))\nmodel.add(Conv2D(32, (3,3), strides = (2,2), padding='same', activation='relu'))\nmodel.add(MaxPooling2D((2,2), strides = (2,2)))\n\nmodel.add(Conv2D(64, (3,3), strides = (1,1), padding='same', activation='relu'))\nmodel.add(Conv2D(64, (3,3), strides=(1,1), padding='same', activation='relu'))\nmodel.add(MaxPooling2D((2,2), strides = (2,2)))\n\nmodel.add(Dropout(0.2))\nmodel.add(Flatten())\n\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(256, activation='relu'))\nmodel.add(Dense(y_train.shape[1], activation = 'softmax'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"35b935ba70de30cd274d5bcd32eed0d859dbedfb"},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2ec5b81bd2a72e8359ae954d1cc7569f87e39948"},"cell_type":"code","source":"model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f247a2af653cd6dd5e2b9d47cad73305ed8071cb"},"cell_type":"code","source":"epochs = 100\nbatchsize = 1024","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7406a7ec2dafa38dd5bd78907c47ab36245cf1e2"},"cell_type":"code","source":"history = model.fit(X_train, y_train, epochs = epochs, batch_size = batchsize, verbose=2)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cd86692e9f0e1774ae235065b81a223472668432"},"cell_type":"markdown","source":"### Checking out the loss and accuracy of the model through the training process"},{"metadata":{"trusted":true,"_uuid":"422fe15bd754ee11d26ce788a7d6043f1c2d3bd5"},"cell_type":"code","source":"plt.plot(history.history['loss'], color='r', label=\"Train Loss\")\nplt.title(\"Train Loss\")\nplt.xlabel(\"Number of Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"322a73e6fefaf5f3aeb6eae3f839cd5414954496"},"cell_type":"code","source":"plt.plot(history.history['acc'], color='g', label=\"Train Accuracy\")\nplt.title(\"Train Accuracy\")\nplt.xlabel(\"Number of Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2836a0bdcc8675b074d9fb311fa4f84df456d719"},"cell_type":"code","source":"print('Train accuracy of the model: ',history.history['acc'][-1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1aa3752657db289e378d6242b0b613d8dca80c73"},"cell_type":"code","source":"test = os.listdir(\"../input/test/\")\nprint(len(test))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"92d3a8e4c118be0f353a0b8882b9919a6ec1c302"},"cell_type":"code","source":"test_data = pd.DataFrame(test, columns=['Image'])\ntest_data['Id'] = ''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ff871a87d92f22e85df4849a3a1943b80c50d440"},"cell_type":"code","source":"X_test = prepareImages(test_data, test_data.shape[0], \"test\")\nX_test /= 255","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9036dc44a53869e39ab7c76c21db053aee772a85"},"cell_type":"code","source":"predictions = model.predict(np.array(X_test), verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ddca8422a3ad4a7b88bb04dac5d7d142ab8ce1a8"},"cell_type":"code","source":"for i, pred in enumerate(predictions):\n    test_data.loc[i, 'Id'] = ' '.join(label_encoder.inverse_transform(pred.argsort()[-5:][::-1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"13d7f96b6a49cff477f28ac4e9c9707c0621af99"},"cell_type":"code","source":"test_data.to_csv('submission_1.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}