{"metadata": {"kernelspec": {"display_name": "Python 3", "name": "python3", "language": "python"}, "language_info": {"codemirror_mode": {"version": 3, "name": "ipython"}, "mimetype": "text/x-python", "file_extension": ".py", "version": "3.6.3", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "name": "python"}}, "nbformat": 4, "cells": [{"execution_count": null, "source": ["# This Python 3 environment comes with many helpful analytics libraries installed\n", "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n", "# For example, here's several helpful packages to load in \n", "\n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "\n", "# Input data files are available in the \"../input/\" directory.\n", "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n", "\n", "from subprocess import check_output\n", "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n", "\n", "# Any results you write to the current directory are saved as output."], "cell_type": "code", "metadata": {"_cell_guid": "f924f8ab-3ce1-42f6-b69d-b17d381fc255", "_uuid": "11edab33c25546d747c5adac82bbf9e70153d016"}, "outputs": []}, {"execution_count": null, "source": ["import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "\n", "import bson\n", "import io\n", "import matplotlib.pyplot as plt\n", "from skimage.data import imread   # or, whatever image library you prefer\n", "from sklearn.neighbors import KNeighborsClassifier\n", "from sklearn.cross_validation import train_test_split\n", "from sklearn.naive_bayes import GaussianNB"], "cell_type": "code", "metadata": {"_cell_guid": "b346cf1e-ace2-4b03-a63f-d6c9001b1f53", "_uuid": "c5f9aba55a00b8c71ee0d2a98665c1362e1b5669"}, "outputs": []}, {"execution_count": null, "source": ["# read bson file into pandas DataFrame\n", "\n", "data = bson.decode_file_iter(open('../input/train_example.bson', 'rb'))\n", "\n", "n = 82 #cols of data in train_example set\n", "X_ids = np.zeros((n,1)).astype(int)\n", "Y = np.zeros((n,1)).astype(int) #category_id for each row\n", "X_images = np.zeros((n,180,180,3)) #m images are 180 by 180 by 3\n", "\n", "print(\"Examples:\", n)\n", "print(\"Dimensions of Y: \",Y.shape)\n", "print(\"Dimensions of X_images: \",X_images.shape)"], "cell_type": "code", "metadata": {"_cell_guid": "a5c48eb8-7934-4ede-b9af-1a34867ebe15", "_uuid": "aeb49c9236ec57cc461abea4473555af9d80d397"}, "outputs": []}, {"execution_count": null, "source": ["# prod_to_category = dict()\n", "i = 0\n", "for c, d in enumerate(data):\n", "    X_ids[i] = d['_id'] \n", "    Y[i] = d['category_id'] \n", "    for e, pic in enumerate(d['imgs']):\n", "        picture = imread(io.BytesIO(pic['picture']))\n", "    X_images[i] = picture #add only the last image \n", "    i+=1\n", "    \n", "    #show update every 10 images\n", "    if c > 0 and c % 10 == 0:\n", "        print(\"[INFO] processed {}/{}\".format(c, 82))"], "cell_type": "code", "metadata": {"_cell_guid": "e35f81f3-4301-4493-ab5c-68575d380b4a", "_uuid": "d7e038d5a16981f7a10b6c452a65918cd451a234"}, "outputs": []}, {"execution_count": null, "source": ["# flatten images\n", "X_flat = X_images.reshape(X_images.shape[0], -1)\n", "X_flat = X_flat/255"], "cell_type": "code", "metadata": {"_cell_guid": "4e8b8b60-ace3-4c66-a69a-8b2cc57dcfac", "scrolled": false, "_uuid": "47aa1fbcfa6a1393483ba2a5412fe85df7a22cc6", "collapsed": true}, "outputs": []}, {"execution_count": null, "source": ["# partition the data into training and testing splits, using 75%\n", "# of the data for training and the remaining 25% for testing\n", "(trainRI, testRI, trainRL, testRL) = train_test_split(\n", "    X_flat, Y, test_size=0.25, random_state=42)"], "cell_type": "code", "metadata": {"_cell_guid": "fb517a02-4974-416f-b484-bbb39544cf14", "_uuid": "92168cfe8782460b6e1b01bcd3f65e6208860772", "collapsed": true}, "outputs": []}, {"execution_count": null, "source": ["# train and evaluate a k-NN classifer on the raw pixel intensities\n", "print(\"[INFO] evaluating raw pixel accuracy...\")\n", "model = KNeighborsClassifier(n_jobs=-1)\n", "model.fit(trainRI, trainRL)\n", "acc = model.score(testRI, testRL)\n", "print(\"[INFO] raw pixel accuracy: {:.2f}%\".format(acc * 100))"], "cell_type": "code", "metadata": {"_cell_guid": "b5737ce0-3bdc-4ed6-86e8-9edd0a85f44c", "_uuid": "8a2f7039e2a74e00ee1838d8f43bd6fc2576ff40"}, "outputs": []}, {"execution_count": null, "source": ["model.predict(testRI)"], "cell_type": "code", "metadata": {"_cell_guid": "5caa122d-39f9-4c35-a9a1-6a816d94fdc6", "_uuid": "bde307b0f7c2f0e1dee24b8619248fb35b882060"}, "outputs": []}, {"execution_count": null, "source": ["testRL"], "cell_type": "code", "metadata": {"_cell_guid": "8e25aa0b-4aa0-4ad6-b1da-13dc33ebcb2c", "_uuid": "d6dbd34a5ea6df18b9f6818b790d7a28a1b1b591"}, "outputs": []}, {"execution_count": null, "source": ["# train and evaluate a Gaussian Naive Bayes classifer on the raw pixel intensities\n", "### Training a model\n", "gnb = GaussianNB()\n", "gnb = gnb.fit(trainRI,trainRL)\n", "\n", "### Prediction result\n", "acc = gnb.score(testRI, testRL)\n", "print(\"[INFO] raw pixel accuracy: {:.2f}%\".format(acc * 100))"], "cell_type": "code", "metadata": {"_cell_guid": "c40f2071-13bc-46ed-8b97-dfacdc0a9464", "_uuid": "081d95e86a716b6d0de0767b9c09c193758a8d71"}, "outputs": []}, {"execution_count": null, "source": ["gnb.predict(testRI)"], "cell_type": "code", "metadata": {"_cell_guid": "f390a87b-c690-4cd8-9fce-586aad3e0259", "_uuid": "3dc9362f94163b747de8b29b9e5c53fc4a0cf36d"}, "outputs": []}, {"execution_count": null, "source": ["testRL"], "cell_type": "code", "metadata": {"_cell_guid": "e1ef3bb7-a4a8-4b65-be5c-db6d50736140", "_uuid": "a3b432174c7e4d412ede14850715ce0aec92f8a7"}, "outputs": []}, {"execution_count": null, "source": ["# Now, your classifier is 'svm'\n", "from sklearn import svm\n", "# kernel: specifies the kernel type to be used in the algorithm (linear, poly, rbf, sgmoid, precomputed)\n", "# C: penalty parameter C of the error term\n", "print(\"Support Vector Machine(SVM)\")\n", "clf = svm.SVC(kernel='linear', C=1).fit(trainRI, trainRL)\n", "print(clf.score(testRI, testRL))"], "cell_type": "code", "metadata": {"_cell_guid": "eecec3ec-9aa9-4c7b-a368-19fd05e53d26", "_uuid": "123620992c15736e7184fca82dbdc864d6f56ea1"}, "outputs": []}, {"execution_count": null, "source": ["clf.predict(testRI)"], "cell_type": "code", "metadata": {"_cell_guid": "23cfda9e-7eef-4d3d-8b90-5e70bd80416c", "_uuid": "cf3b142a5b19307278075ac6528da9eb4dd3a9f9"}, "outputs": []}, {"execution_count": null, "source": ["testRL"], "cell_type": "code", "metadata": {"_cell_guid": "41dc27d4-9369-4555-866e-930f69885595", "_uuid": "98e4a2d07006c23b5c3daae332921ed6bfbf65cf"}, "outputs": []}, {"execution_count": null, "source": [], "cell_type": "code", "metadata": {"_cell_guid": "381e7498-9db6-4802-978d-44193f64b702", "_uuid": "671eb4937885c8ad739e450519059f7adce539ac", "collapsed": true}, "outputs": []}], "nbformat_minor": 1}