{"cells": [{"cell_type": "code", "execution_count": null, "metadata": {"_uuid": "11928f6144a4dab798706e60138b04ed4519bcfd", "_cell_guid": "e239ba35-4559-41d7-947d-19898a6c3e3f"}, "outputs": [], "source": ["# This Python 3 environment comes with many helpful analytics libraries installed\n", "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n", "# For example, here's several helpful packages to load in \n", "\n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "\n", "# Input data files are available in the \"../input/\" directory.\n", "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n", "\n", "from subprocess import check_output\n", "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n", "\n", "# Any results you write to the current directory are saved as output."]}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": ["import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "\n", "import bson\n", "import io\n", "import matplotlib.pyplot as plt\n", "from skimage.data import imread   # or, whatever image library you prefer\n", "from sklearn.neighbors import KNeighborsClassifier\n", "from sklearn.cross_validation import train_test_split\n", "from sklearn.naive_bayes import GaussianNB"]}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": ["# read bson file into pandas DataFrame\n", "\n", "data = bson.decode_file_iter(open('../input/train_example.bson', 'rb'))\n", "\n", "n = 82 #cols of data in train_example set\n", "X_ids = np.zeros((n,1)).astype(int)\n", "Y = np.zeros((n,1)).astype(int) #category_id for each row\n", "X_images = np.zeros((n,180,180,3)) #m images are 180 by 180 by 3\n", "\n", "print(\"Examples:\", n)\n", "print(\"Dimensions of Y: \",Y.shape)\n", "print(\"Dimensions of X_images: \",X_images.shape)"]}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": ["# prod_to_category = dict()\n", "i = 0\n", "for c, d in enumerate(data):\n", "    X_ids[i] = d['_id'] \n", "    Y[i] = d['category_id'] \n", "    for e, pic in enumerate(d['imgs']):\n", "        picture = imread(io.BytesIO(pic['picture']))\n", "    X_images[i] = picture #add only the last image \n", "    i+=1\n", "    \n", "    #show update every 10 images\n", "    if c > 0 and c % 10 == 0:\n", "        print(\"[INFO] processed {}/{}\".format(c, 82))"]}, {"cell_type": "code", "execution_count": null, "metadata": {"collapsed": true}, "outputs": [], "source": ["# flatten images\n", "X_flat = X_images.reshape(X_images.shape[0], -1)\n", "X_flat = X_flat/255"]}, {"cell_type": "code", "execution_count": null, "metadata": {"collapsed": true}, "outputs": [], "source": ["# partition the data into training and testing splits, using 75%\n", "# of the data for training and the remaining 25% for testing\n", "(trainRI, testRI, trainRL, testRL) = train_test_split(\n", "    X_flat, Y, test_size=0.25, random_state=42)"]}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": ["# train and evaluate a k-NN classifer on the raw pixel intensities\n", "print(\"[INFO] evaluating raw pixel accuracy...\")\n", "model = KNeighborsClassifier(n_jobs=-1)\n", "model.fit(trainRI, trainRL)\n", "acc = model.score(testRI, testRL)\n", "print(\"[INFO] raw pixel accuracy: {:.2f}%\".format(acc * 100))"]}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": ["model.predict(testRI)"]}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": ["testRL"]}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": ["# train and evaluate a Gaussian Naive Bayes classifer on the raw pixel intensities\n", "### Training a model\n", "gnb = GaussianNB()\n", "gnb = gnb.fit(trainRI,trainRL)\n", "\n", "### Prediction result\n", "acc = gnb.score(testRI, testRL)\n", "print(\"[INFO] raw pixel accuracy: {:.2f}%\".format(acc * 100))"]}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": ["gnb.predict(testRI)"]}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": ["testRL"]}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": ["# Now, your classifier is 'svm'\n", "from sklearn import svm\n", "# kernel: specifies the kernel type to be used in the algorithm (linear, poly, rbf, sgmoid, precomputed)\n", "# C: penalty parameter C of the error term\n", "print(\"Support Vector Machine(SVM)\")\n", "clf = svm.SVC(kernel='linear', C=1).fit(trainRI, trainRL)\n", "print(clf.score(testRI, testRL))"]}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": ["clf.predict(testRI)"]}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": ["testRL"]}], "nbformat_minor": 1, "metadata": {"language_info": {"file_extension": ".py", "name": "python", "pygments_lexer": "ipython3", "version": "3.6.3", "mimetype": "text/x-python", "nbconvert_exporter": "python", "codemirror_mode": {"name": "ipython", "version": 3}}, "kernelspec": {"display_name": "Python 3", "language": "python", "name": "python3"}}, "nbformat": 4}