{"nbformat": 4, "cells": [{"cell_type": "code", "outputs": [], "metadata": {"_uuid": "b96f3cf36926a803aa8b8a1ade98a4a16b9e85ba", "collapsed": true, "_cell_guid": "7fa24cc0-983d-444c-aee4-aa63e28ac9ab"}, "execution_count": null, "source": ["import numpy as np\n", "import pandas as pd\n", "import io\n", "import bson\n", "import cv2\n", "import matplotlib.pyplot as plt\n", "from tqdm import tqdm_notebook\n", "import concurrent.futures\n", "from multiprocessing import cpu_count"]}, {"cell_type": "code", "outputs": [], "metadata": {"collapsed": true}, "execution_count": null, "source": ["num_images = 1000000\n", "im_size = 16\n", "num_cpus = cpu_count()"]}, {"cell_type": "code", "outputs": [], "metadata": {}, "execution_count": null, "source": ["def imread(buf):\n", "    return cv2.imdecode(np.frombuffer(buf, np.uint8), cv2.IMREAD_ANYCOLOR)\n", "\n", "def img2feat(im):\n", "    x = cv2.resize(im, (im_size, im_size), interpolation=cv2.INTER_AREA)\n", "    return np.float32(x) / 255\n", "\n", "X = np.empty((num_images, im_size, im_size, 3), dtype=np.float32)\n", "y = []\n", "\n", "def load_image(pic, target, bar):\n", "    picture = imread(pic)\n", "    x = img2feat(picture)\n", "    bar.update()\n", "    \n", "    return x, target\n", "\n", "bar = tqdm_notebook(total=num_images)\n", "with open('../input/train.bson', 'rb') as f, \\\n", "        concurrent.futures.ThreadPoolExecutor(num_cpus) as executor:\n", "\n", "    data = bson.decode_file_iter(f)\n", "    delayed_load = []\n", "\n", "    i = 0\n", "    try:\n", "        for c, d in enumerate(data):\n", "            target = d['category_id']\n", "            for e, pic in enumerate(d['imgs']):\n", "                delayed_load.append(executor.submit(load_image, pic['picture'], target, bar))\n", "                \n", "                i = i + 1\n", "\n", "                if i >= num_images:\n", "                    raise IndexError()\n", "\n", "    except IndexError:\n", "        pass;\n", "    \n", "    for i, future in enumerate(concurrent.futures.as_completed(delayed_load)):\n", "        x, target = future.result()\n", "        \n", "        X[i] = x\n", "        y.append(target)"]}, {"cell_type": "code", "outputs": [], "metadata": {}, "execution_count": null, "source": ["X.shape, len(y)"]}, {"cell_type": "code", "outputs": [], "metadata": {}, "execution_count": null, "source": ["y = pd.Series(y)\n", "\n", "num_classes =800 \n", "valid_targets = set(y.value_counts().index[:num_classes-1].tolist())\n", "valid_y = y.isin(valid_targets)\n", "\n", "y[~valid_y] = -1\n", "\n", "max_acc = valid_y.mean()\n", "print(max_acc)"]}, {"cell_type": "code", "outputs": [], "metadata": {"collapsed": true}, "execution_count": null, "source": ["y, rev_labels = pd.factorize(y)"]}, {"cell_type": "code", "outputs": [], "metadata": {}, "execution_count": null, "source": ["from keras.preprocessing import image\n", "from keras.applications import inception_v3\n", "# Load pre-trained image recognition model\n", "model = inception_v3.InceptionV3()\n", "\n", "model.summary()\n", "\n", "model.fit(X, y, validation_split=0.1, epochs=5)\n", "\n", "\n"]}, {"cell_type": "code", "outputs": [], "metadata": {"collapsed": true}, "execution_count": null, "source": ["submission = pd.read_csv('../input/sample_submission.csv', index_col='_id')\n", "\n", "#most_frequent_guess =1000018296\n", "#submission['category_id'] = most_frequent_guess \n", "\n", "num_images_test = 80000\n", "with open('../input/test.bson', 'rb') as f, \\\n", "         concurrent.futures.ThreadPoolExecutor(num_cpus) as executor:\n", "\n", "    data = bson.decode_file_iter(f)\n", "\n", "    future_load = []\n", "\n", "    for i,d in enumerate(data):\n", "        if i >= num_images_test:\n", "              break\n", "        future_load.append(executor.submit(load_image, d['imgs'][0]['picture'], d['_id'], bar))\n", "        \n", "        print(\"Starting future processing\")\n", "    for future in concurrent.futures.as_completed(future_load):\n", "        x, _id = future.result()\n", "        \n", "        y_cat = rev_labels[np.argmax(model.predict(x[None])[0])]\n", "        #if y_cat == -1:\n", "            #y_cat = most_frequent_guess\n", "\n", "        bar.update()\n", "        submission.loc[_id, 'category_id'] = y_cat\n", "print('Finished')"]}, {"cell_type": "code", "outputs": [], "metadata": {}, "execution_count": null, "source": ["submission.to_csv('new_submission.csv.gz', compression='gzip')"]}], "metadata": {"kernelspec": {"language": "python", "name": "python3", "display_name": "Python 3"}, "language_info": {"file_extension": ".py", "codemirror_mode": {"name": "ipython", "version": 3}, "pygments_lexer": "ipython3", "version": "3.6.1", "name": "python", "nbconvert_exporter": "python", "mimetype": "text/x-python"}}, "nbformat_minor": 1}