{"cells": [{"source": ["# This Python 3 environment comes with many helpful analytics libraries installed\n", "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n", "# For example, here's several helpful packages to load in \n", "\n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "\n", "print(\"Hello\")"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"_uuid": "419948f1de60686ace56c21f07734edfab62d973", "_cell_guid": "4f7ca8e0-6af7-40a5-ae36-c5fdccd6ec5b"}}, {"source": ["def getRawFeatures(picture):\n", "    red = []\n", "    green = []\n", "    blue = []\n", "    for row in range(picture.shape[0]):\n", "        for col in range(picture.shape[1]):\n", "            red.append(picture[row][col][0])\n", "            green.append(picture[row][col][1])\n", "            blue.append(picture[row][col][2])\n", "    feature = red\n", "    feature.extend(green)\n", "    feature.extend(blue)\n", "    return feature"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"collapsed": true, "_uuid": "7fc1b38e38da8ebee2d499b4cc627a723a07addf", "_cell_guid": "cd799485-93e0-4d76-9624-0b628504bf88"}}, {"source": ["np.fromstring(b'\\x00\\x00\\x80?\\x00\\x00\\x00@\\x00\\x00@@\\x00\\x00\\x80@', dtype='<f4')"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"_uuid": "b566880034948ec5e5b3592b7dc183d037a4a3ac", "_cell_guid": "545cb449-2582-4cda-a2ab-b23cfc8c32aa"}}, {"source": ["import io\n", "import bson # this is installed with the pymongo package\n", "import matplotlib\n", "import matplotlib.pyplot as plt\n", "from skimage.data import imread   # or, whatever image library you prefer\n", "import multiprocessing as mp      # will come in handy due to the size of the data\n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"collapsed": true, "_uuid": "89d8ce2e6149ea1a8bf7db009e7db5f8b10adcec", "_cell_guid": "f88cfc36-a17c-4eda-b5f2-fe567201441a"}}, {"source": ["# Simple data processing\n", "count_images = 0\n", "image_names_array = []\n", "category_id_array = []\n", "data = bson.decode_file_iter(open('../input/train_example.bson', 'rb'))\n", "pictures = []\n", "count = 0\n", "prod_to_category = dict()\n", "\n", "for c, d in enumerate(data):\n", "    #for each product_id\n", "    product_id = d['_id']\n", "    category_id = d['category_id'] # This won't be in Test data\n", "    prod_to_category[product_id] = category_id\n", "    \n", "    for e, pic in enumerate(d['imgs']):\n", "        #for each image\n", "        picture = imread(io.BytesIO(pic['picture']))\n", "        pictures.append(picture)\n", "        count = count + 1\n", "        # do something with the picture, etc\n", "#         image_name = \"prod_id-\" + str(product_id) + \"-\" + \"image-\" + str(e)\n", "#         print(\"PRODUCT ID:\", product_id, \"NUMBER\", e)\n", "#         plt.imshow(picture)\n", "#         fig1 = plt.gcf()\n", "#         plt.show()\n", "#         plt.draw()\n", "        count_images = count_images + 1\n", "        image_names_array.append(image_name)\n", "        category_id_array.append(str(category_id))\n", "        #fig1.savefig(\"img/\" + str(image_name), dpi=100)\n", "    print(\"done\")\n", "#     break\n", "\n", "prod_to_category = pd.DataFrame.from_dict(prod_to_category, orient='index')\n", "prod_to_category.index.name = '_id'\n", "prod_to_category.rename(columns={0: 'category_id'}, inplace=True)"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"_uuid": "bdfffde30a8d59257d649a0379e3a33a44744d8b", "_cell_guid": "7a754b19-e70c-4f08-98cc-66b549a4a280"}}, {"source": ["X_train = np.asarray(pictures)\n", "y_train = np.asarray(category_id_array)\n", "y_train.shape"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"_uuid": "dd9c5d9b3078672abc57612f7302cd40c207b2d4", "_cell_guid": "5c9ff9fb-0aa2-4646-b270-cfb503e59088"}}, {"source": ["X_train.shape"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {}}, {"source": ["X_train = X_train.reshape(X_train.shape[0], 3, 180, 180).astype('float32')\n", "X_train = X_train - np.mean(X_train) / X_train.std()"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"collapsed": true}}, {"source": ["# listFeatureVector = []\n", "\n", "# for picture in pictures:\n", "#     featureVector = getRawFeatures(picture)\n", "#     listFeatureVector.append(featureVector)"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"collapsed": true, "_uuid": "501608b6ad3a38cb5c0bffeb673f82c99d8fa227", "_cell_guid": "eef8decd-6eb1-486f-ac06-74a4d3724777"}}, {"source": ["# print(len(listFeatureVector[0]))\n", "\n", "# X = listFeatureVector\n", "# y = category_id_array"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"collapsed": true, "_uuid": "7b6b9552045767aaac42d1cd2ab08e84121349ee", "_cell_guid": "b53e1cd6-fa71-4cb2-8b56-b5c2cc6c075a"}}, {"source": ["# len(list(set(y)))"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"collapsed": true, "_uuid": "aa61f62f16119f9aad826f4eeab5c0186929fa93", "_cell_guid": "88dd3296-b648-48ee-b88d-36225659e2c3"}}, {"source": ["y_train"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {}}, {"source": ["b,c = np.unique(y_train, return_inverse=True)"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"collapsed": true}}, {"source": ["from collections import Counter\n", "d = Counter(c)"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"collapsed": true}}, {"source": ["y_train = c"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {}}, {"source": ["y_train"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"scrolled": true}}, {"source": ["from keras.utils import np_utils\n", "from tflearn.data_utils import to_categorical\n", "y_train = np_utils.to_categorical(y_train)"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"collapsed": true}}, {"source": ["y_train"], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {}}, {"source": [], "cell_type": "code", "outputs": [], "execution_count": null, "metadata": {"collapsed": true}}], "nbformat_minor": 1, "nbformat": 4, "metadata": {"language_info": {"version": "3.6.3", "file_extension": ".py", "nbconvert_exporter": "python", "mimetype": "text/x-python", "codemirror_mode": {"version": 3, "name": "ipython"}, "pygments_lexer": "ipython3", "name": "python"}, "kernelspec": {"language": "python", "display_name": "Python 3", "name": "python3"}}}