{"nbformat": 4, "cells": [{"cell_type": "markdown", "source": ["This notebook is just a (very) small improvement over most common baseline.\n", "\n", "It loads a few images from train and resize it to 8x8 pixels to generate a 64 (8 x 8) feature vector.\n", "\n", "Then, it uses KNN to find the most similar image on test set.\n", "\n", "Unfortunatelly, due to limitations on Kernel, only a few test images are classified."], "metadata": {}}, {"cell_type": "code", "execution_count": null, "outputs": [], "source": ["import numpy as np\n", "import pandas as pd\n", "import io\n", "import bson\n", "import cv2\n", "import matplotlib.pyplot as plt\n", "from tqdm import tqdm_notebook\n", "import concurrent.futures\n", "from multiprocessing import cpu_count"], "metadata": {"_uuid": "99b9a425db0c128d80fda1ecc9b42f952a77c93c", "collapsed": true, "_cell_guid": "7d7afe89-0c7e-43cb-9ba6-0e93db30b4bd"}}, {"cell_type": "code", "execution_count": null, "outputs": [], "source": ["num_images = 200000\n", "\n", "def imread(buf):\n", "    return cv2.imdecode(np.frombuffer(buf, np.uint8), cv2.IMREAD_GRAYSCALE)\n", "\n", "def img2feat(im):\n", "    return cv2.resize(im, (8, 8), interpolation=cv2.INTER_AREA).ravel()\n", "\n", "X = []\n", "y = []\n", "\n", "bar = tqdm_notebook(total=num_images)\n", "with open('../input/train.bson', 'rb') as f:\n", "    data = bson.decode_file_iter(f)\n", "\n", "    i = 0\n", "    try:\n", "        for c, d in enumerate(data):\n", "            target = d['category_id']\n", "            for e, pic in enumerate(d['imgs']):\n", "                picture = imread(pic['picture'])\n", "                x = img2feat(picture)\n", "                \n", "                X.append(x)\n", "                y.append(target)\n", "                \n", "                i = i + 1\n", "                bar.update()\n", "\n", "                if i >= num_images:\n", "                    raise IndexError()\n", "\n", "    except IndexError:\n", "        pass;"], "metadata": {"_uuid": "4a5081a65e4d22316b7306cddd80332712b21b0d", "_cell_guid": "82f5d889-b14d-4cb7-86b0-5bd726304d8b"}}, {"cell_type": "code", "execution_count": null, "outputs": [], "source": ["X = np.array(X, dtype=np.float32)\n", "y = pd.Series(y)\n", "\n", "X.shape, y.shape"], "metadata": {"_uuid": "39c439b021cfe966165d78d29486f97ea9b84c5b", "_cell_guid": "5cfda60d-d8c8-481c-833b-85f170564cc6"}}, {"cell_type": "code", "execution_count": null, "outputs": [], "source": ["num_classes = 30  # This will reduce the max accuracy to just above 0.2\n", "\n", "# Now we must find the most `num_classes-1` frequent classes\n", "# (there will be an aditional 'other' class)\n", "valid_targets = set(y.value_counts().index[:num_classes-1].tolist())\n", "valid_y = y.isin(valid_targets)\n", "\n", "# Set other classes to -1\n", "y[~valid_y] = -1\n", "\n", "max_acc = valid_y.mean()\n", "print(max_acc)"], "metadata": {"_uuid": "56969c620e60a3e55f0684b412fe1f6b7590ba01", "_cell_guid": "34c9bf08-9f89-461f-b74f-1f24291a5829"}}, {"cell_type": "markdown", "source": ["Note that the max accuracy reported before is greater than ~0.2 reported [here](http://https://www.kaggle.com/bguberfain/naive-statistics) due to smaller train set."], "metadata": {}}, {"cell_type": "code", "execution_count": null, "outputs": [], "source": ["# Now we categorize the dataframe\n", "y, rev_labels = pd.factorize(y)"], "metadata": {"_uuid": "1975ac67714c59998a9c8e587c7fd5554fd7597b", "collapsed": true, "_cell_guid": "3c137f38-d10d-49e7-a9e5-b1628d2e0567"}}, {"cell_type": "code", "execution_count": null, "outputs": [], "source": ["# Now we have a X,y pair. Let's train a simple KNN Classifier\n", "# import xgboost as xgb  # This run out of time for this task\n", "\n", "from sklearn.neighbors import KNeighborsClassifier\n", "\n", "knn = KNeighborsClassifier(3)\n", "knn.fit(X, y)"], "metadata": {"_uuid": "96da9277c9c7fed8cb871a3c434be783c9eb10f1", "_cell_guid": "e6368858-f2ce-4426-85af-89219c651a06"}}, {"cell_type": "markdown", "source": ["Now we evaluate the test set using the previous trained KNN."], "metadata": {}}, {"cell_type": "code", "execution_count": null, "outputs": [], "source": ["submission = pd.read_csv('../input/sample_submission.csv', index_col='_id')\n", "\n", "most_frequent_guess = 1000018296\n", "submission['category_id'] = most_frequent_guess # Most frequent guess"], "metadata": {"_uuid": "5811926ec436c8820ccaa468fc2ac6a09dfbd3bc", "_cell_guid": "4425bdc4-dd25-40c8-a26e-fe2eb11d3398"}}, {"cell_type": "code", "execution_count": null, "outputs": [], "source": ["num_images_test = 100000  # We only have time for a few test images..\n", "num_cpus = cpu_count()\n", "\n", "def predict(d, bar):\n", "    picture = imread(d['imgs'][0]['picture'])\n", "    x = img2feat(picture)\n", "    y_cat = rev_labels[knn.predict(x[None])[0]]\n", "    if y_cat == -1:\n", "        y_cat = most_frequent_guess\n", "\n", "    bar.update()\n", "    \n", "    return d['_id'], y_cat\n", "\n", "bar = tqdm_notebook(total=num_images_test)\n", "with open('../input/test.bson', 'rb') as f, \\\n", "         concurrent.futures.ThreadPoolExecutor(num_cpus) as executor:\n", "\n", "    data = bson.decode_file_iter(f)\n", "\n", "    future_predict = []\n", "    \n", "    for i,d in enumerate(data):\n", "        if i >= num_images_test:\n", "            break\n", "        future_predict.append(executor.submit(predict, d, bar))\n", "\n", "    for future in concurrent.futures.as_completed(future_predict):\n", "        _id, y_cat = future.result()\n", "        submission.loc[_id, 'category_id'] = y_cat"], "metadata": {"_uuid": "bc6ce0ae1620a9ac5a481d0a7e4a0f0e25510fd1", "_cell_guid": "f7465e07-cc74-4371-9cc1-3639641f4110"}}, {"cell_type": "code", "execution_count": null, "outputs": [], "source": ["submission.to_csv('new_submission.csv.gz', compression='gzip')"], "metadata": {"_uuid": "f806114fc478e841392589d4f7e2f0b828b20311", "collapsed": true, "_cell_guid": "dd34386c-f28a-42d7-a3be-f09126fd61c7"}}], "metadata": {"kernelspec": {"display_name": "Python 3", "name": "python3", "language": "python"}, "language_info": {"name": "python", "version": "3.6.1", "mimetype": "text/x-python", "pygments_lexer": "ipython3", "codemirror_mode": {"version": 3, "name": "ipython"}, "nbconvert_exporter": "python", "file_extension": ".py"}}, "nbformat_minor": 1}