{"cells": [{"source": ["# This Python 3 environment comes with many helpful analytics libraries installed\n", "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n", "# For example, here's several helpful packages to load in \n", "\n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "from sklearn.model_selection import train_test_split"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "419948f1de60686ace56c21f07734edfab62d973", "_cell_guid": "4f7ca8e0-6af7-40a5-ae36-c5fdccd6ec5b", "collapsed": true}, "execution_count": 1}, {"source": ["def getRawFeatures(picture):\n", "    red = []\n", "    green = []\n", "    blue = []\n", "    for row in range(picture.shape[0]):\n", "        for col in range(picture.shape[1]):\n", "            red.append(picture[row][col][0])\n", "            green.append(picture[row][col][1])\n", "            blue.append(picture[row][col][2])\n", "    feature = red\n", "    feature.extend(green)\n", "    feature.extend(blue)\n", "    return feature"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "7fc1b38e38da8ebee2d499b4cc627a723a07addf", "_cell_guid": "cd799485-93e0-4d76-9624-0b628504bf88", "collapsed": true}, "execution_count": 2}, {"source": ["import io\n", "import bson # this is installed with the pymongo package\n", "import matplotlib\n", "import matplotlib.pyplot as plt\n", "from skimage.data import imread   # or, whatever image library you prefer\n", "import multiprocessing as mp      # will come in handy due to the size of the data\n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "89d8ce2e6149ea1a8bf7db009e7db5f8b10adcec", "_cell_guid": "f88cfc36-a17c-4eda-b5f2-fe567201441a", "collapsed": true}, "execution_count": 3}, {"source": ["# Simple data processing\n", "count_images = 0\n", "image_names_array = []\n", "category_id_array = []\n", "data = bson.decode_file_iter(open('../input/train_example.bson', 'rb'))\n", "pictures = []\n", "count = 0\n", "prod_to_category = dict()\n", "\n", "for c, d in enumerate(data):\n", "    #for each product_id\n", "    product_id = d['_id']\n", "    category_id = d['category_id'] # This won't be in Test data\n", "    prod_to_category[product_id] = category_id\n", "    \n", "    for e, pic in enumerate(d['imgs']):\n", "        #for each image\n", "        picture = imread(io.BytesIO(pic['picture']))\n", "        pictures.append(picture)\n", "        count = count + 1\n", "        # do something with the picture, etc\n", "#         image_name = \"prod_id-\" + str(product_id) + \"-\" + \"image-\" + str(e)\n", "#         print(\"PRODUCT ID:\", product_id, \"NUMBER\", e)\n", "#         plt.imshow(picture)\n", "#         fig1 = plt.gcf()\n", "#         plt.show()\n", "#         plt.draw()\n", "        count_images = count_images + 1\n", "#         image_names_array.append(image_name)\n", "        category_id_array.append(str(category_id))\n", "        #fig1.savefig(\"img/\" + str(image_name), dpi=100)\n", "#     break\n", "\n", "prod_to_category = pd.DataFrame.from_dict(prod_to_category, orient='index')\n", "prod_to_category.index.name = '_id'\n", "prod_to_category.rename(columns={0: 'category_id'}, inplace=True)"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "bdfffde30a8d59257d649a0379e3a33a44744d8b", "_cell_guid": "7a754b19-e70c-4f08-98cc-66b549a4a280", "collapsed": true}, "execution_count": 4}, {"source": ["X_train = np.asarray(pictures)\n", "y_train = np.asarray(category_id_array)\n", "y_train.shape"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "dd9c5d9b3078672abc57612f7302cd40c207b2d4", "_cell_guid": "5c9ff9fb-0aa2-4646-b270-cfb503e59088"}, "execution_count": 5}, {"source": ["X_train = X_train.reshape(X_train.shape[0], 3, 180, 180).astype('float32')\n", "X_train = X_train - np.mean(X_train) / X_train.std()"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "a598bc27fa63b16b02cf53df434c8c4f295e0f4d", "_cell_guid": "4e70c847-e53e-464a-970e-204502ecf584", "collapsed": true}, "execution_count": 6}, {"source": ["y_train"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "d39d93483789299c7b0682b6fc6ad9304a791160", "_cell_guid": "c264809c-7c93-4710-adc6-20fd5e451218"}, "execution_count": 7}, {"source": ["b,c = np.unique(y_train, return_inverse=True)"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "6074f841a1e47c2799c0b86555d0aaea8ca186eb", "_cell_guid": "8bbe7c31-295f-437f-9736-b041afbf9179", "collapsed": true}, "execution_count": 8}, {"source": ["from collections import Counter\n", "d = Counter(c)"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "218cfbf405c95753628ee70780d513c87dabb1a7", "_cell_guid": "03ff71db-9254-47c8-80a3-73ed1b00e34e", "collapsed": true}, "execution_count": 9}, {"source": ["y_train = c"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "c849c91309c0900788639c1af03b75e7a72a7a89", "_cell_guid": "37ed0c0f-e378-4e98-9906-331b5f1be3a3", "collapsed": true}, "execution_count": 10}, {"source": ["# y_train"], "outputs": [], "cell_type": "code", "metadata": {"scrolled": true, "_uuid": "1fdfd5277811f1db662c3c7384026d683eba2a41", "_cell_guid": "1dd12042-468d-4ce2-80ef-53f6eaddee5e", "collapsed": true}, "execution_count": 11}, {"source": ["X = X_train\n", "y = y_train\n", "print(X.shape)\n", "print(y.shape)\n", "# X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=20)\n", "\n", "# print(X_train.shape)\n", "# print(y_train.shape)\n", "# print(X_test.shape)\n", "# print(y_test.shape)"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "93bb0c61c1a5aee2fbc288032c37799d16bc7222", "_cell_guid": "239ebfaf-9777-4625-aaf8-2f143a1becd9"}, "execution_count": 12}, {"source": ["from keras.utils import np_utils\n", "from tflearn.data_utils import to_categorical\n", "y_train = np_utils.to_categorical(y_train)\n", "# y_test = np_utils.to_categorical(y_test)"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "56077036cbc3da6fcd24e653370547de30054cf3", "_cell_guid": "4fb8235c-c68d-48c0-8903-a09920738418"}, "execution_count": 13}, {"source": ["# y_train"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "9e45d340b3690317fcc7a19123fda9af3d16d206", "_cell_guid": "e6b4be89-e7dd-4287-932f-1333404b34ac", "collapsed": true}, "execution_count": 14}, {"source": ["from tflearn.layers.core import input_data, dropout, fully_connected\n", "from tflearn.layers.conv import conv_2d, max_pool_2d\n", "from tflearn.layers.estimator import regression\n", "model = input_data(shape=[None,3,180,180])\n", "model = conv_2d(model,32,5,activation='elu')\n", "model = max_pool_2d(model,2)\n", "model = conv_2d(model,64,3,activation='relu')\n", "model = max_pool_2d(model,2)\n", "model = dropout(model,0.3)\n", "model = conv_2d(model,64,3, activation='elu')\n", "model = max_pool_2d(model,2)\n", "model = fully_connected(model,512,activation='sigmoid')\n", "model = dropout(model,0.3)\n", "model = fully_connected(model,36,activation='softmax')\n", "model = regression(model,optimizer='adagrad',loss='categorical_crossentropy',learning_rate=0.05)"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "38d44e5ae3ba1742fbf0267e5509caad9373522c", "_cell_guid": "29f10004-3395-4b54-9d13-dc00e4aab544"}, "execution_count": 15}, {"source": ["import tensorflow as tf\n", "import tflearn\n", "with tf.device('cpu:0'):\n", "   model = tflearn.DNN(model)\n", "   model.fit(X_train , y_train, n_epoch=5, validation_set = (X_train, y_train), batch_size = 10)"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "3a0d1fcd894e46a1568b52fbeeaa0c3a5752e0e2", "_cell_guid": "88391713-fdc6-423e-939d-3b43f55c23c5"}, "execution_count": 16}, {"source": ["#get a few test examples\n", "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=20)\n", "\n", "print(X_train.shape)\n", "print(y_train.shape)\n", "print(X_test.shape)\n", "print(y_test.shape)\n"], "outputs": [], "cell_type": "code", "metadata": {}, "execution_count": 17}, {"source": ["pred = model.predict(X_test)\n", "pred"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "cd73241895ac81c33eae59d884a9e458507f8d50", "_cell_guid": "1fa0da61-cde3-4d62-86f3-f5ea33e12127"}, "execution_count": 18}, {"source": ["\n", "# y_test = np_utils.to_categorical(y_test)\n", "# model.evaluate(X_test, y_test)\n", "y_test.shape\n"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "950778f25669ca60caacb4b85cc07b1f44172289", "_cell_guid": "13b3154d-f451-45fa-8329-f570359af365"}, "execution_count": 25}, {"source": ["# from keras.preprocessing import image\n", "# img = image.load_img('',target_size=(180,180))\n", "# img = image.img_to_array(img)\n", "# img = np.expand_dims(img, axis=0)\n", "img = X_train[0].reshape(1,3,180,180)\n", "# img.shape\n", "preds = model.predict(img)\n", "preds"], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "8187177c9b8f3e34395e0062d8bd99dcf9149a50", "_cell_guid": "5dcdd45d-a158-47a2-ab88-001e9ef477b6"}, "execution_count": 20}, {"source": [], "outputs": [], "cell_type": "code", "metadata": {"_uuid": "87057e6c109ca94a90524ebb7346ae2bb8566f5f", "_cell_guid": "2335b788-ae45-46ae-beb3-8f7981507589", "collapsed": true}, "execution_count": null}, {"source": [], "outputs": [], "cell_type": "code", "metadata": {"collapsed": true}, "execution_count": null}], "nbformat_minor": 1, "nbformat": 4, "metadata": {"language_info": {"file_extension": ".py", "codemirror_mode": {"name": "ipython", "version": 3}, "pygments_lexer": "ipython3", "name": "python", "nbconvert_exporter": "python", "version": "3.6.3", "mimetype": "text/x-python"}, "kernelspec": {"name": "python3", "language": "python", "display_name": "Python 3"}}}