{"metadata": {"language_info": {"codemirror_mode": {"version": 3, "name": "ipython"}, "nbconvert_exporter": "python", "file_extension": ".py", "version": "3.6.1", "name": "python", "pygments_lexer": "ipython3", "mimetype": "text/x-python"}, "kernelspec": {"language": "python", "name": "python3", "display_name": "Python 3"}}, "nbformat_minor": 1, "cells": [{"metadata": {"_uuid": "0bc0d2d2c13d2f47f813f1511ff4363a2249aa6c", "collapsed": true, "_cell_guid": "42dd3d2c-2092-4ed0-832d-951ff650b33e"}, "outputs": [], "cell_type": "code", "source": ["import io\n", "import bson                       # this is installed with the pymongo package\n", "import matplotlib.pyplot as plt\n", "from skimage.data import imread   # or, whatever image library you prefer\n", "import multiprocessing as mp      # will come in handy due to the size of the data"], "execution_count": 10}, {"metadata": {"_uuid": "f49af7b723616654fbbbc2faf6521b3393f68012", "collapsed": true, "_cell_guid": "ca82d248-bd14-429f-aa15-fc2b37458454"}, "outputs": [], "cell_type": "code", "source": ["import pandas as pd\n", "import numpy as np"], "execution_count": 11}, {"metadata": {"_uuid": "594e1a526ed69f7dc54e942125a4def172e54b60", "_cell_guid": "75eec940-28b3-4333-ac6c-2730d772afff"}, "outputs": [], "cell_type": "code", "source": ["from keras.models import Sequential\n", "from keras.layers import Dense\n", "from keras.layers import Dropout\n", "from keras.layers import Flatten\n", "from keras.constraints import maxnorm\n", "from keras.optimizers import SGD\n", "from keras.layers.convolutional import Conv2D\n", "from keras.layers.convolutional import MaxPooling2D\n", "from keras.utils import np_utils\n", "from keras import backend as K\n", "import keras\n", "K.set_image_dim_ordering('th')"], "execution_count": 12}, {"metadata": {"_uuid": "8a112b0e34c4fd436caae782dd177d923484d599", "_cell_guid": "490ba55d-df72-4046-90a7-984c1a3308e6"}, "outputs": [], "cell_type": "code", "source": ["data = bson.decode_file_iter(open('../input/train_example.bson', 'rb'))\n", "\n", "lst_prod = []\n", "for c, d in enumerate(data):\n", "    lst_prod.append(d['category_id'])"], "execution_count": 13}, {"metadata": {"_uuid": "a9ec8835c3f964eb63fb996284cd2d913ba1a7b5", "_cell_guid": "e9b3f711-37f7-4ef4-9162-45edb0beab85"}, "outputs": [], "cell_type": "code", "source": ["from keras.utils import np_utils\n", "from sklearn.preprocessing import LabelEncoder\n", "\n", "y = lst_prod\n", "\n", "encoder = LabelEncoder()\n", "encoder.fit(y)\n", "encoded_y = encoder.transform(y)\n", "dummy_y = np_utils.to_categorical(encoded_y)\n", "dummy_y.shape"], "execution_count": 14}, {"metadata": {"_uuid": "5fea5073dc1f99f14ef32f556691af09d927e436", "collapsed": true, "_cell_guid": "8eb96322-53ce-4237-a6ab-b3966a812889"}, "outputs": [], "cell_type": "code", "source": ["num_classes=len(dummy_y[81])\n", "epochs = 5"], "execution_count": 15}, {"metadata": {"_uuid": "9ba5d4012414b33d10ef7045a29b268295ea74d1", "_cell_guid": "463a2a8b-4c73-4a76-8cb7-e66bd07bd577"}, "cell_type": "markdown", "source": ["## Change model layer to improve acc"]}, {"metadata": {"_uuid": "4d2c99b587264a371a0617b87fe25a4488bd5b43", "collapsed": true, "_cell_guid": "2682f3ea-7e37-45e9-930b-05bd50fb8a5b"}, "outputs": [], "cell_type": "code", "source": ["model = Sequential()\n", "# Convolutional Layer\n", "model = Sequential()\n", "# Convolutional Layer\n", "model.add(Conv2D(180, (3,3), input_shape = (180,180,3), activation='relu'))\n", "\n", "# Pooling Layer\n", "model.add(MaxPooling2D(pool_size=(1, 1)))\n", "\n", "# Fully conected Layer\n", "model.add(Flatten())\n", "model.add(Dense(num_classes, activation='softmax'))\n", "\n", "\n", "lrate = 0.01\n", "decay = lrate/epochs\n", "sgd = SGD(lr=lrate, momentum=0.9, decay=decay, nesterov=False)\n", "\n", "model.compile(loss='categorical_crossentropy', optimizer=sgd, metrics=['accuracy'])"], "execution_count": 16}, {"metadata": {"_uuid": "ca664cc6176eeb4b81194402f59752ae0ca5aefb", "_cell_guid": "d4c7bc13-d7fe-4381-ab17-36818468599b"}, "cell_type": "markdown", "source": ["---"]}, {"metadata": {"_uuid": "71f3d163e2b773d6d8958d770e2d9cf3a6bf3942", "scrolled": true, "_cell_guid": "9a6b18b7-048c-455c-bb04-5258ac9d4322"}, "outputs": [], "cell_type": "code", "source": ["data = bson.decode_file_iter(open('../input/train_example.bson', 'rb'))\n", "\n", "prod_to_category = dict()\n", "i=0\n", "for c, d in enumerate(data):\n", "    \n", "    lst_pic = []\n", "    for e, pic in enumerate(d['imgs']):\n", "        picture = imread(io.BytesIO(pic['picture']))\n", "        \n", "        picture = picture.reshape(1,180,180,3)\n", "        # do something with the picture, etc\n", "#         print(picture.shape)\n", "        lst_pic.append(picture)\n", "\n", "    # train on single row\n", "    for j in lst_pic:\n", "        X_batch = j\n", "        Y_batch = dummy_y[i]\n", "        Y_batch = Y_batch.reshape(1,num_classes)\n", "        model.fit(X_batch, Y_batch, batch_size=32, epochs=epochs)\n", "    i = i+1"], "execution_count": 17}, {"metadata": {"_uuid": "3010c7614636377d4bdf60141366b7c1a5244c4a", "collapsed": true, "_cell_guid": "5738e780-5ce9-44c5-bb1b-db343d68b8db"}, "outputs": [], "cell_type": "code", "source": ["# picture.reshape(1,180,180,3)"], "execution_count": 18}, {"metadata": {"_uuid": "fd88f85de171a174612e905e8fb3845933a12a73", "collapsed": true, "_cell_guid": "bc9845a5-c3e4-453c-81e8-4553f5da6cff"}, "outputs": [], "cell_type": "code", "source": [], "execution_count": null}], "nbformat": 4}