{"cells": [{"outputs": [], "metadata": {"_uuid": "907bfcef5a9532a4ac6a07da03a8b3bf0ea0e673", "_cell_guid": "a30d0276-3fb2-4a6f-a103-182f4c6b361f"}, "execution_count": null, "cell_type": "code", "source": ["import numpy as np \n", "import pandas as pd\n", "from sklearn.model_selection import train_test_split\n", "\n", "import io\n", "import bson # this is installed with the pymongo package\n", "import matplotlib\n", "import matplotlib.pyplot as plt\n", "from skimage.data import imread   # or, whatever image library you prefer\n", "import multiprocessing as mp      # will come in handy due to the size of the data\n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "from collections import Counter\n", "from keras.utils import np_utils\n", "from tflearn.data_utils import to_categorical\n", "\n", "from tflearn.layers.core import input_data, dropout, fully_connected\n", "from tflearn.layers.conv import conv_2d, max_pool_2d\n", "from tflearn.layers.estimator import regression\n", "\n", "import tensorflow as tf\n", "import tflearn"]}, {"outputs": [], "metadata": {"_uuid": "edb04253f1082f20e9d9b0db38bb9a79f89cac28", "collapsed": true, "_cell_guid": "3d59df96-b98b-4b06-9c59-3b1657696960"}, "execution_count": null, "cell_type": "code", "source": ["def getRawFeatures(picture):\n", "    red = []\n", "    green = []\n", "    blue = []\n", "    for row in range(picture.shape[0]):\n", "        for col in range(picture.shape[1]):\n", "            red.append(picture[row][col][0])\n", "            green.append(picture[row][col][1])\n", "            blue.append(picture[row][col][2])\n", "    feature = red\n", "    feature.extend(green)\n", "    feature.extend(blue)\n", "    return feature"]}, {"outputs": [], "metadata": {"_uuid": "d3b8c6606452c91b9144ba081edfee5631929b93", "collapsed": true, "_cell_guid": "df32cdb2-96d1-4b7e-9166-480c7c900833"}, "execution_count": null, "cell_type": "code", "source": ["# Simple data processing\n", "count_images = 0\n", "image_names_array = []\n", "category_id_array = []\n", "data = bson.decode_file_iter(open('../input/train_example.bson', 'rb'))\n", "pictures = []\n", "count = 0\n", "prod_to_category = dict()\n", "\n", "for c, d in enumerate(data):\n", "    #for each product_id\n", "    product_id = d['_id']\n", "    category_id = d['category_id'] # This won't be in Test data\n", "    prod_to_category[product_id] = category_id\n", "    \n", "    for e, pic in enumerate(d['imgs']):\n", "        #for each image\n", "        picture = imread(io.BytesIO(pic['picture']))\n", "        pictures.append(picture)\n", "        count = count + 1\n", "        # do something with the picture, etc\n", "#         image_name = \"prod_id-\" + str(product_id) + \"-\" + \"image-\" + str(e)\n", "#         print(\"PRODUCT ID:\", product_id, \"NUMBER\", e)\n", "#         plt.imshow(picture)\n", "#         fig1 = plt.gcf()\n", "#         plt.show()\n", "#         plt.draw()\n", "        count_images = count_images + 1\n", "#         image_names_array.append(image_name)\n", "        category_id_array.append(str(category_id))\n", "        #fig1.savefig(\"img/\" + str(image_name), dpi=100)\n", "#     break\n", "\n", "prod_to_category = pd.DataFrame.from_dict(prod_to_category, orient='index')\n", "prod_to_category.index.name = '_id'\n", "prod_to_category.rename(columns={0: 'category_id'}, inplace=True)"]}, {"outputs": [], "metadata": {"_uuid": "a0de41754653f149c486d638f4b3e60a2b4bb046", "collapsed": true, "_cell_guid": "8c6c5612-5ff8-46e0-892e-6f02e33afd8c"}, "execution_count": null, "cell_type": "code", "source": ["X = np.asarray(pictures)\n", "y = np.asarray(category_id_array)"]}, {"outputs": [], "metadata": {"_uuid": "92a0a9fc52c9afcfb20697b66a14d02e90f7e8bf", "_cell_guid": "66fc29d6-ffb5-4799-9145-63ceac01128d"}, "execution_count": null, "cell_type": "code", "source": ["X.shape"]}, {"outputs": [], "metadata": {"_uuid": "392a64c8813d18d5ac801ab72245a248d7814fd5", "_cell_guid": "a1d24696-cfe9-49f2-aa61-1462d8336db1"}, "execution_count": null, "cell_type": "code", "source": ["y.shape"]}, {"outputs": [], "metadata": {"_uuid": "af197c03541dda67907f080ce2438d0ddbba1195", "collapsed": true, "_cell_guid": "c8ccdac7-53fd-4e23-a45e-60723dfeb875"}, "execution_count": null, "cell_type": "code", "source": ["X = X.reshape(X.shape[0], 3, 180, 180).astype('float32')\n", "X = X - np.mean(X) / X.std()"]}, {"outputs": [], "metadata": {"_uuid": "3f30d84b017aa24c555e6eb90e4f3445ab621ba2", "_cell_guid": "ac2c5191-150c-4287-8053-33904bd61b52"}, "execution_count": null, "cell_type": "code", "source": ["y"]}, {"outputs": [], "metadata": {"_uuid": "d8937b1b0ba6659e538e429de6c52994154b0333", "collapsed": true, "_cell_guid": "ae08e877-d96d-4811-a0f3-d76a373d9a37"}, "execution_count": null, "cell_type": "code", "source": ["b,c = np.unique(y, return_inverse=True)"]}, {"outputs": [], "metadata": {"_uuid": "b77fe6b77415c7c108ba34bd37b1ab2036f902f6", "_cell_guid": "c5a33057-856d-478c-abe2-051c36004570"}, "execution_count": null, "cell_type": "code", "source": ["Counter(c)\n", "y = c\n", "y"]}, {"outputs": [], "metadata": {"_uuid": "3be702215e6fe89641c0679e8e0bec67c1c770d3", "collapsed": true, "_cell_guid": "a10c1c2b-c4a1-4e17-8702-75ebffd26895"}, "execution_count": null, "cell_type": "code", "source": ["y = np_utils.to_categorical(y)"]}, {"outputs": [], "metadata": {"_uuid": "25a793bf49164839f61c73103aaa84055b913bd2", "_cell_guid": "0f160ce2-e6e0-4012-880c-f0c1a7982024"}, "execution_count": null, "cell_type": "code", "source": ["print(X.shape)\n", "print(y.shape)"]}, {"outputs": [], "metadata": {"_uuid": "fee9bf004b5faa8c9a9017d3b4e3f536c7e40a76", "_cell_guid": "c30bf63f-2976-49e3-a034-993c7a742f17"}, "execution_count": null, "cell_type": "code", "source": ["#lets split\n", "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=1234)\n", "X_train = X\n", "y_train = y\n", "\n", "print(X_train.shape, y_train.shape, X_test.shape, y_test.shape)"]}, {"outputs": [], "metadata": {"_uuid": "2f8a510cb73a6e00b9a4714ad93501cafc99dc2b", "_cell_guid": "d36632c3-5f3d-4f33-96cd-c2944c3c427f"}, "execution_count": null, "cell_type": "code", "source": ["model = input_data(shape=[None,3,180,180])\n", "model = conv_2d(model,180,10,activation='elu')\n", "model = max_pool_2d(model,2)\n", "model = conv_2d(model,360,5,activation='relu')\n", "model = max_pool_2d(model,2)\n", "model = dropout(model,0.3)\n", "model = conv_2d(model,360,5, activation='elu')\n", "model = max_pool_2d(model,2)\n", "model = fully_connected(model,1024,activation='sigmoid')\n", "model = dropout(model,0.3)\n", "model = fully_connected(model,512,activation='sigmoid')\n", "model = dropout(model,0.3)\n", "model = fully_connected(model,36,activation='softmax')\n", "model = regression(model,optimizer='adagrad',loss='categorical_crossentropy',learning_rate=0.05)"]}, {"outputs": [], "metadata": {"_uuid": "570512a51e0db69eb81d78bb993020d44efa00ef", "_cell_guid": "71944e57-c4ea-4a9a-96d1-d9c44c5fc582"}, "execution_count": null, "cell_type": "code", "source": ["with tf.device('cpu:0'):\n", "   model = tflearn.DNN(model)\n", "   model.fit(X_train , y_train, n_epoch=5, validation_set = (X_test, y_test), batch_size = 10)"]}, {"outputs": [], "metadata": {"_uuid": "7223b4fc5be2d015339adf84be607d3306009711", "_cell_guid": "832f9b30-5a76-4b24-9f9f-333f790d16db"}, "execution_count": null, "cell_type": "code", "source": ["model.evaluate(X_test, y_test)"]}, {"outputs": [], "metadata": {"_uuid": "3bfae01a1190e27061783b584d261670b06ef24c", "collapsed": true, "_cell_guid": "da4270bf-932b-49c1-ba53-e743bf975c17"}, "execution_count": null, "cell_type": "code", "source": []}], "nbformat_minor": 1, "nbformat": 4, "metadata": {"kernelspec": {"display_name": "Python 3", "name": "python3", "language": "python"}, "language_info": {"name": "python", "pygments_lexer": "ipython3", "codemirror_mode": {"name": "ipython", "version": 3}, "nbconvert_exporter": "python", "version": "3.6.3", "file_extension": ".py", "mimetype": "text/x-python"}}}