{"metadata": {"kernelspec": {"language": "python", "display_name": "Python 3", "name": "python3"}, "language_info": {"mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "file_extension": ".py", "version": "3.6.1", "codemirror_mode": {"name": "ipython", "version": 3}}}, "nbformat": 4, "nbformat_minor": 1, "cells": [{"cell_type": "code", "metadata": {"_uuid": "6edcc629573d3ef52b0a758f798e5aa923b50226", "_cell_guid": "c57e30dc-fe6e-48b1-9e85-2487a7b2a84b"}, "execution_count": null, "outputs": [], "source": ["# This Python 3 environment comes with many helpful analytics libraries installed\n", "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n", "# For example, here's several helpful packages to load in \n", "\n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "import matplotlib.pyplot as plt\n", "from skimage.data import imread \n", "import json\n", "import bson\n", "import io\n", "\n", "f = open(\"../input/train_example.bson\",'rb')\n", "bs = f.read()\n", "docs = bson.decode_all(bs)\n", "data = pd.DataFrame.from_dict(docs)\n", "data.head()\n", "\n", "# Any results you write to the current directory are saved as output."]}, {"cell_type": "code", "metadata": {}, "execution_count": null, "outputs": [], "source": ["#Printing sample images for each category\n", "for i in range(5):\n", "    picture = imread(io.BytesIO(data.imgs[i][0]['picture']))\n", "    plt.figure()\n", "    plt.imshow(picture)"]}, {"cell_type": "code", "metadata": {}, "execution_count": null, "outputs": [], "source": ["X = []\n", "for i in range(data.shape[0]):\n", "    X.append(imread(io.BytesIO(data.imgs[i][0]['picture'])))\n", "    \n", "X = np.array(X,dtype=np.float32)/255. \n", "X.shape"]}, {"cell_type": "code", "metadata": {}, "execution_count": null, "outputs": [], "source": ["y = data.category_id.values\n", "y.shape"]}, {"cell_type": "code", "metadata": {}, "execution_count": null, "outputs": [], "source": ["from keras.utils import np_utils\n", "from sklearn.preprocessing import LabelEncoder\n", "\n", "encoder = LabelEncoder()\n", "encoder.fit(y)\n", "encoded_y = encoder.transform(y)\n", "dummy_y = np_utils.to_categorical(encoded_y)\n", "dummy_y.shape"]}, {"cell_type": "code", "metadata": {}, "execution_count": null, "outputs": [], "source": ["from sklearn.cross_validation import train_test_split\n", "X_train,X_test,Y_train,Y_test = train_test_split(X,dummy_y,test_size=0.3)\n", "print(X_train.shape,X_test.shape,Y_train.shape,Y_test.shape)\n"]}, {"cell_type": "code", "metadata": {}, "execution_count": null, "outputs": [], "source": ["import keras\n", "from keras.datasets import mnist\n", "from keras.models import Sequential, Model\n", "from keras.layers import Dense, Dropout, Flatten,Input\n", "from keras.layers import Conv2D, MaxPooling2D\n", "from keras import backend as K\n", "\n", "Inp=Input(shape=(180,180,3))\n", "x = Conv2D(32, kernel_size=(3, 3), activation='relu',name = 'Conv_01')(Inp)\n", "x = Conv2D(64, (3, 3), activation='relu',name = 'Conv_02')(x)\n", "x = MaxPooling2D(pool_size=(2, 2),name = 'MaxPool_01')(x)\n", "x = Dropout(0.25,name = 'Dropout_01')(x)\n", "x = Flatten(name = 'Flatten_01')(x)\n", "x = Dense(128, activation='relu',name = 'Dense_01')(x)\n", "x = Dropout(0.5,name = 'Dropout_02')(x)\n", "output = Dense(36, activation='softmax',name = 'Dense_02')(x)\n", "model = Model(Inp,output)\n", "model.summary()"]}, {"cell_type": "code", "metadata": {"collapsed": true}, "execution_count": null, "outputs": [], "source": ["model.compile(loss=keras.losses.categorical_crossentropy,\n", "              optimizer=keras.optimizers.Adadelta(),\n", "              metrics=['accuracy'])"]}, {"cell_type": "code", "metadata": {}, "execution_count": null, "outputs": [], "source": ["batch_size = 32\n", "epochs = 10\n", "hist = model.fit(X_train, Y_train,\n", "          batch_size=batch_size,\n", "          epochs=epochs,\n", "          verbose=1,\n", "          callbacks = None,\n", "          validation_data=(X_test, Y_test))"]}, {"cell_type": "code", "metadata": {}, "execution_count": null, "outputs": [], "source": ["def plot_train(hist):\n", "    h = hist.history\n", "    if 'acc' in h:\n", "        meas='acc'\n", "        loc='lower right'\n", "    else:\n", "        meas='loss'\n", "        loc='upper right'\n", "    plt.plot(hist.history[meas])\n", "    plt.plot(hist.history['val_'+meas])\n", "    plt.title('model '+meas)\n", "    plt.ylabel(meas)\n", "    plt.xlabel('epoch')\n", "    plt.legend(['train', 'validation'], loc=loc)\n", "plot_train(hist)"]}, {"cell_type": "markdown", "metadata": {}, "source": ["First kernel try in keras \n", "Lot more to come!!\n", "Stay Advance :)"]}]}