{"metadata":{"language_info":{"mimetype":"text/x-python","version":"3.6.1","file_extension":".py","name":"python","nbconvert_exporter":"python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3"},"kernelspec":{"name":"python3","language":"python","display_name":"Python 3"}},"cells":[{"source":"Thank you to inversion, whose [code](https://www.kaggle.com/inversion/processing-bson-files) have been used for importing BSON files.\n\nStarting with train_example.bson. Later I will split  the tran.bson in train/dev/test sets.\n\n\n","cell_type":"markdown","metadata":{"_cell_guid":"0343b341-2d4d-4334-bf2d-f09c753caca6","_uuid":"8983b06eb00239d178aea794566c048899011da3"}},{"outputs":[],"source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport io, bson\nimport matplotlib.pyplot as plt\nfrom skimage.data import imread   # or, whatever image library you prefer\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.","cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"89be1c9d-c6bf-4c23-923d-94a6e7db1e39","_uuid":"80c1a485637bc5e11c5e1441ae6f0d41e5f8e4f2"}},{"source":"The code below will read the BSON files into a pandas dataframe, and then read the category_id that we are trying to predict into a numpy matrix Y, the images that we are using to predict into a matrix X, and the unique ID's into a matrix X_ids.","cell_type":"markdown","metadata":{"_cell_guid":"fefd81db-b3a7-4f1f-8668-979736a9b49a","_uuid":"a697825d6867fa28681cf5219b3b760081d36e49"}},{"outputs":[],"source":"# Simple data processing\n#change here to decode train instead of train_example\ndata = bson.decode_file_iter(open('../input/train.bson', 'rb'))\n# read bson file into pandas DataFrame\nwith open('../input/train_example.bson','rb') as b:\n    df = pd.DataFrame(bson.decode_all(b.read()))\n\n#Get shape of first image \nfor e, pic in enumerate(df['imgs'][0]):\n        picture = imread(io.BytesIO(pic['picture']))\n        pix_x,pix_y,rgb = picture.shape\nn = 200  #cols of data in train set - number of examples\nX_ids = np.zeros((n,1)).astype(int)\nY = np.zeros((n,1)).astype(int) #category_id for each row\nX_images = np.zeros((n,pix_x,pix_y,rgb)) #m images are 180 by 180 by 3\n\nprint(\"Examples:\", n)\nprint(\"Dimensions of Y: \",Y.shape)\nprint(\"Dimensions of X_images: \",X_images.shape)\n\n# prod_to_category = dict()\ni = 0\nfor c, d in enumerate(data):\n    X_ids[i] = d['_id'] \n    Y[i] = d['category_id'] \n    for e, pic in enumerate(d['imgs']):\n        picture = imread(io.BytesIO(pic['picture']))\n    X_images[i] = picture #add only the last image \n    i+=1\n    if (i==n):\n        break\nprint (\"loaded\" + str(i)+ \"images\")\n","cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"79b17760-ecb2-4b6a-8204-61d2149d3f50","_uuid":"52e0404eab02285be162ea3e1765af29cfc4bd2d"}},{"outputs":[],"source":"#Lets take a look at the category names supplied to us:\ndf_categories = pd.read_csv('../input/category_names.csv', index_col='category_id')\n\ncount_unique_cats = len(df_categories.index)\n\nprint(\"There are \", count_unique_cats, \" unique categories to predict. E.g.\")\nprint(\"\")\nprint(df_categories.head())","cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"8ce694b6-afb9-4a0d-8e0b-4725c74df85b","_uuid":"f35fc13789dd8540efc8eca9f1acfc1ad91ba511"}},{"source":"There are 5270 unique categories. You can use the code below to take a look at some of the images and their respective categories.","cell_type":"markdown","metadata":{"_cell_guid":"3b9983f4-1dae-4872-864e-5874c099f28f","_uuid":"3f1139746cfb260093875a301c4b8aeb74bfca76"}},{"outputs":[],"source":"#Function to return the category description from df_categories\ndef get_category(category_id,level):\n    if(level in range(1,4)):\n        try:\n            return df_categories.iloc[df_categories.index == category_id[0],level-1].values[0]\n        except:\n            print(\"Error - category_id does not exist\")\n    else:\n        print(\"Error - level must be between 1 - 3\")\n\n#Play around with the index and cat levels to explore the images in the test data set\nindex = 3\ncat_desc_level = 1 # level 1 - 3\nprint(\"ID: \",X_ids[index][0], \"category_id: \",Y[index][0], \"category_description: \",get_category(Y[index],cat_desc_level))\nplt.imshow(X_images[index])","cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"50fd725e-690f-43a6-b257-7b93100be22f","_uuid":"308dfcec913b1da759201eace06f4e425cdcc591"}},{"source":"","cell_type":"markdown","metadata":{"_cell_guid":"db8e5ad1-3257-4f83-b2a6-67312c27d496","_uuid":"3e4d25488efe809404cce03ddf891f912789d783"}},{"outputs":[],"source":"","cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f7c11817-0513-4d07-95d5-2edf6fc58d96","collapsed":true,"_uuid":"e05208833e9cfa3301c4831d950c99c882c592be"}},{"outputs":[],"source":"from sklearn import preprocessing\nimport warnings\n\nwarnings.filterwarnings(\"ignore\") \n\n#full list of classes\ncategory_classes = df_categories.index.values\ncategory_classes = category_classes.reshape(category_classes.shape[0],1)\n\n#using a label encoder, and binarizer to convert all unique category_ids to have a column for each class \nle = preprocessing.LabelEncoder() \nlb = preprocessing.LabelBinarizer()\n\nle.fit(df_categories.index.values)\ny_encoded = le.transform(Y)\nY_flat = y_encoded\n#lb.fit(y_encoded)\n#Y_flat = lb.transform(y_encoded)\n\n#redimension X for our model\nX_flat = X_images.reshape(X_images.shape[0], -1)\nY_flat = Y_flat\nm = X_flat.shape[1]\nn = Y_flat.shape\n#Scale RGB data for learning\nX_flat = X_flat/255\n#print results\nprint(\"X Shape =\", X_flat.shape, \"Y Shape =\",Y_flat.shape, \"m = \",m, \"n classes found in test data=\", n)\n","cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"2b0563d1-672d-45e0-8976-120e4a3ddded","_uuid":"eceef4cb1ebb3768d7b4c4fd8e742ad1b0ef1cb7"}},{"source":"See the encoding of the classes","cell_type":"markdown","metadata":{"_cell_guid":"cb842471-3c96-453d-a88a-591a7ff160ed","_uuid":"1099f86782b2c56d82b8a432d577ddafd83b558a"}},{"outputs":[],"source":"print(Y_flat)","cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"6dd0bd88-acfa-4352-8730-e157cfd9b2fb","_uuid":"53e5b27b1acafe0103ead61ec63dbfd9f51d975e"}},{"outputs":[],"source":"import tensorflow as tf\n\n\n\n# Specify that all features have real-value data\nfeature_columns = [tf.contrib.layers.real_valued_column(\"\", dimension=m)]\n\n# Build 2 layer DNN with 10, 20, 10 units respectively.\nclassifier = tf.contrib.learn.DNNClassifier(feature_columns=feature_columns,\n                                            hidden_units=[1000, 200],\n                                            n_classes=count_unique_cats,\n                                            model_dir=\"tmp/adrian1000p200/\")\n\n# Fit model.\nclassifier.fit(x=X_flat,\n               y=Y_flat,\n               steps=100)\n\n# Evaluate accuracy.\naccuracy_score = classifier.evaluate(x=X_flat,\n                                     y=Y_flat)[\"accuracy\"]\nprint('Accuracy: {0:f}'.format(accuracy_score))\n\n\n#y = list(classifier.predict(new_samples, as_iterable=True))\n#print('Predictions: {}'.format(str(y)))\n\n\n\n#","cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"d6e0dfd0-e1fa-4d17-9804-1b017991dc31","_kg_hide-output":false,"scrolled":false,"_uuid":"21217786d7734b0962f374fd48906595cb54bf59"}},{"source":"Now we have to tune the NN architecture\n","cell_type":"markdown","metadata":{"_cell_guid":"0987f89c-a156-4974-b9fd-84b7eb2e2c54","_uuid":"88b58d94ceb8de4d8cbf2b5950c88b1b0c2349d7"}}],"nbformat":4,"nbformat_minor":1}