{"nbformat": 4, "cells": [{"cell_type": "code", "outputs": [], "metadata": {"_uuid": "80c1a485637bc5e11c5e1441ae6f0d41e5f8e4f2", "_cell_guid": "89be1c9d-c6bf-4c23-923d-94a6e7db1e39"}, "execution_count": null, "source": ["import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "\n", "import io, bson\n", "import matplotlib.pyplot as plt\n", "from skimage.data import imread   # or, whatever image library you prefer\n", "\n", "from subprocess import check_output\n", "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n", "\n", "# Any results you write to the current directory are saved as output."]}, {"cell_type": "markdown", "metadata": {"_uuid": "a697825d6867fa28681cf5219b3b760081d36e49", "_cell_guid": "fefd81db-b3a7-4f1f-8668-979736a9b49a"}, "source": ["The code below will read the BSON files into a pandas dataframe, and then read the category_id that we are trying to predict into a numpy matrix Y, the images that we are using to predict into a matrix X, and the unique ID's into a matrix X_ids."]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "3cc7ef50247d5eb9814cfd15af40bafc9305a038", "collapsed": true, "_cell_guid": "61f5aebe-a476-4d6f-96c6-9ffdec7d1c55"}, "execution_count": null, "source": ["from scipy.misc import imshow\n", "%matplotlib inline"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "8c6c522f1b493689071c4c0ad958579667528dca", "_cell_guid": "a4b4e589-cad4-4f1a-a810-60cade667856"}, "execution_count": null, "source": ["n = 5000\n", "pix_x =180\n", "pix_y =180\n", "rgb = 3\n", "X_ids = np.zeros((n,1)).astype(int)\n", "Y = np.zeros((n,1)).astype(int) #category_id for each row\n", "X_images = np.zeros((n,pix_x,pix_y,rgb)) #m images are 180 by 180 by 3\n", "\n", "\n", "with open('../input/train.bson', 'rb') as f:\n", "    data = bson.decode_file_iter(f)\n", "    counter = 0\n", "    for c, d in enumerate(data):\n", "\n", "\n", "        X_ids[counter] = d['_id'] \n", "        Y[counter] = d['category_id'] \n", "        for e, pic in enumerate(d['imgs']):\n", "            picture = imread(io.BytesIO(pic['picture']))\n", "\n", "\n", "        print(counter)\n", "        X_images[counter] = picture #add only the last image \n", "\n", "        counter+=1\n", "        if counter >= n:\n", "            break\n"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "26b86c5ff6193f9c93fe56085c1104874fb8c922", "collapsed": true, "_cell_guid": "706a732b-cab5-47ec-8aa9-c30cd48e3862"}, "execution_count": null, "source": ["# from matplotlib import pyplot as plt\n", "\n", "# for i in range(10):\n", "#     plt.imshow(X_images[i], interpolation='nearest')\n", "#     print(Y[i])\n", "#     plt.show()\n"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "f35fc13789dd8540efc8eca9f1acfc1ad91ba511", "_cell_guid": "8ce694b6-afb9-4a0d-8e0b-4725c74df85b"}, "execution_count": null, "source": ["#Lets take a look at the category names supplied to us:\n", "df_categories = pd.read_csv('../input/category_names.csv', index_col='category_id')\n", "\n", "count_unique_cats = len(df_categories.index)\n", "\n", "print(\"There are \", count_unique_cats, \" unique categories to predict. E.g.\")\n", "print(\"\")\n", "print(df_categories.head())"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "308dfcec913b1da759201eace06f4e425cdcc591", "_cell_guid": "50fd725e-690f-43a6-b257-7b93100be22f"}, "execution_count": null, "source": ["#Function to return the category description from df_categories\n", "def get_category(data, category_id,level):\n", "    if(level in range(1,4)):\n", "        try:\n", "            return data.iloc[data.index == category_id[0],level-1].values[0]\n", "        except:\n", "            print(\"Error - category_id does not exist\")\n", "    else:\n", "        print(\"Error - level must be between 1 - 3\")\n", "\n", "#Play around with the index and cat levels to explore the images in the test data set\n", "index = 14\n", "cat_desc_level = 1 # level 1 - 3\n", "print(\"ID: \",X_ids[index][0], \"category_id: \",Y[index][0], \"category_description: \",get_category(df_categories,Y[index],cat_desc_level))\n", "plt.imshow(X_images[index])"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "c45cb187729f3957b3319e4d90fb591cbcb83e71", "collapsed": true, "_cell_guid": "6282d027-95d7-422d-a6e6-ff6279554ba8"}, "execution_count": null, "source": ["Y_new = []\n", "for i in range(len(Y)):\n", "    Y_new.append( get_category(df_categories,Y[i],1))"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "a0daa8927b47fb654905205f19d06d5a528f63d4", "collapsed": true, "_cell_guid": "d7ccbae0-19c8-4c3b-b787-c52ac234799a"}, "execution_count": null, "source": ["from sklearn import preprocessing\n", "import warnings\n", "warnings.filterwarnings(\"ignore\") \n", "\n", "#full list of classes\n", "# category_classes = df_categories.index.values\n", "# category_classes = category_classes.reshape(category_classes.shape[0],1)\n"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "adc86baed0ebc5d0d3ba353739326e8db9cf196b", "_cell_guid": "0c3ae054-0229-4b60-9440-bdde2da9a69b"}, "execution_count": null, "source": ["df_categories['category_level1'].values"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "2917fcdf657372c0b95b267c1c0f0fed9b93fa57", "collapsed": true, "_cell_guid": "88b4764e-98b5-46b2-8b9c-ae7711cf928c"}, "execution_count": null, "source": ["le = preprocessing.LabelEncoder() \n", "lb = preprocessing.LabelBinarizer()\n", "le.fit(df_categories['category_level1'].values)\n", "y_encoded = le.transform(Y_new) # Label Encoding"]}, {"cell_type": "code", "outputs": [], "metadata": {}, "execution_count": null, "source": []}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "bdf8d0be5365d8f9522fbe7d3b88f0368929dbb1", "collapsed": true, "_cell_guid": "9fadab60-1ac3-4333-8879-f426383c340d"}, "execution_count": null, "source": ["# #binarizer to convert all unique category_ids to have a column for each class \n", "# lb.fit(y_encoded)\n", "# Y_flat = lb.transform(y_encoded)"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "546eac20b71b50fec596fad2e9f3d8ca25602ba8", "collapsed": true, "_cell_guid": "b37a8686-6397-4937-b5f8-c2de85e6975c"}, "execution_count": null, "source": ["# Y_flat.shape"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "f0a8a269620868093fcc92120d8720bf07c76b69", "collapsed": true, "_cell_guid": "4763ffce-8bf3-433d-9868-909048bbef99"}, "execution_count": null, "source": ["X_flat = X_images.reshape((n,-1))"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "eceef4cb1ebb3768d7b4c4fd8e742ad1b0ef1cb7", "_cell_guid": "2b0563d1-672d-45e0-8976-120e4a3ddded"}, "execution_count": null, "source": ["\n", "# m = X_flat.shape[1]\n", "# n = Y_flat.shape[1]\n", "\n", "#Scale RGB data for learning\n", "X_flat = X_flat/255\n", "#print results\n", "print(\"X Shape =\", X_flat.shape, \"Y Shape =\",y_encoded.shape,\"N_unique_label=\",len(set(y_encoded)) )\n"]}, {"cell_type": "markdown", "metadata": {}, "source": ["### Try PCA for dimension red:"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "1099f86782b2c56d82b8a432d577ddafd83b558a", "collapsed": true, "_cell_guid": "cb842471-3c96-453d-a88a-591a7ff160ed"}, "execution_count": null, "source": ["from sklearn.decomposition import PCA"]}, {"cell_type": "code", "outputs": [], "metadata": {}, "execution_count": null, "source": ["pca = PCA(n_components=100,random_state=0)\n", "pca.fit(X_flat)"]}, {"cell_type": "code", "outputs": [], "metadata": {}, "execution_count": null, "source": ["pd.DataFrame(pca.explained_variance_).plot()"]}, {"cell_type": "code", "outputs": [], "metadata": {"collapsed": true}, "execution_count": null, "source": ["pca = PCA(n_components=20,random_state=0)\n", "new_X = pca.fit_transform(X_flat)"]}, {"cell_type": "code", "outputs": [], "metadata": {}, "execution_count": null, "source": ["new_X.shape"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "21217786d7734b0962f374fd48906595cb54bf59", "collapsed": true, "_cell_guid": "d6e0dfd0-e1fa-4d17-9804-1b017991dc31"}, "execution_count": null, "source": ["from sklearn.linear_model import LogisticRegression \n", "model = LogisticRegression(random_state=0,solver='sag',n_jobs=-1) "]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "373ab33fb2af60f301bb09f5e262007171bcefd7", "_cell_guid": "6d206b1c-5279-4781-a271-41ddc685f659", "scrolled": true}, "execution_count": null, "source": ["model.fit(X=new_X,y=y_encoded)"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "6ff09f2a896de8d4f45f0718477deafd6dfe7f4f", "_cell_guid": "cd77bd83-6bb6-40a6-9914-ee62dee66f7f"}, "execution_count": null, "source": ["pred = model.predict(new_X)"]}, {"cell_type": "code", "outputs": [], "metadata": {}, "execution_count": null, "source": ["model.predict_proba(new_X)"]}, {"cell_type": "code", "outputs": [], "metadata": {}, "execution_count": null, "source": ["pred = pred.tolist()\n", "y_encoded = y_encoded.tolist()"]}, {"cell_type": "code", "outputs": [], "metadata": {"scrolled": true}, "execution_count": null, "source": ["sum([pred[i]==y_encoded[i] for i in range(len(pred))])/float(len(y_encoded))"]}, {"cell_type": "markdown", "metadata": {}, "source": ["### Submission"]}, {"cell_type": "code", "outputs": [], "metadata": {"collapsed": true}, "execution_count": null, "source": ["sub = pd.read_csv('../input/sample_submission.csv')"]}, {"cell_type": "code", "outputs": [], "metadata": {"collapsed": true}, "execution_count": null, "source": ["from sklearn.utils import shuffle"]}, {"cell_type": "code", "outputs": [], "metadata": {}, "execution_count": null, "source": ["with open('../input/test.bson', 'rb') as f:\n", "    data = bson.decode_file_iter(f)\n", "\n", "    counter = 0\n", "    for c, d in enumerate(data):\n", "\n", "\n", "        \n", "        pred = []\n", "        for e, pic in enumerate(d['imgs']):\n", "            picture = imread(io.BytesIO(pic['picture']))\n", "            picture = picture.reshape(1,-1)\n", "            picture = picture/255.0\n", "            picture = pca.transform(picture)\n", "            pred1 = model.predict_proba(picture)\n", "            pred.append(pred1)\n", "        \n", "        \n", "        pred = np.array(pred).mean(axis=0) # average all single predictions\n", "        pred = model.classes_[np.argmax(pred)]\n", "        pred2 = le.classes_[pred]\n", "\n", "        res_dat = df_categories[df_categories['category_level1']==pred2]\n", "        shuffle(res_dat)\n", "        sub.iloc[counter]['category_id'] = res_dat.index.values[0]\n", "        if counter % 2000 == 0:\n", "            print(counter)\n", "\n", "# #         X_images[counter] = picture #add only the last image \n", "\n", "        counter+=1\n"]}, {"cell_type": "code", "outputs": [], "metadata": {"collapsed": true}, "execution_count": null, "source": []}, {"cell_type": "markdown", "metadata": {}, "source": []}, {"cell_type": "markdown", "metadata": {"_uuid": "88b58d94ceb8de4d8cbf2b5950c88b1b0c2349d7", "_cell_guid": "0987f89c-a156-4974-b9fd-84b7eb2e2c54"}, "source": []}], "metadata": {"kernelspec": {"language": "python", "name": "python3", "display_name": "Python 3"}, "language_info": {"file_extension": ".py", "codemirror_mode": {"name": "ipython", "version": 3}, "pygments_lexer": "ipython3", "version": "3.6.1", "name": "python", "nbconvert_exporter": "python", "mimetype": "text/x-python"}}, "nbformat_minor": 1}