{"nbformat_minor": 1, "cells": [{"cell_type": "code", "outputs": [], "source": ["import numpy as np\n", "import pandas as pd\n", "import io\n", "import bson\n", "import cv2\n", "import matplotlib.pyplot as plt\n", "from tqdm import tqdm_notebook\n", "import concurrent.futures\n", "from multiprocessing import cpu_count"], "execution_count": null, "metadata": {"_uuid": "000ca54c5ae104339118737515a0c249e6a8b9e0", "collapsed": true, "_cell_guid": "a7c119aa-bd53-4335-8b4b-fb52f3e87649"}}, {"cell_type": "code", "outputs": [], "source": ["num_images = 440000\n", "im_size = 16\n", "num_cpus = cpu_count()"], "execution_count": null, "metadata": {"_uuid": "939b5df9891321a696b5b7d7ad77c6c10f5feff8", "collapsed": true, "_cell_guid": "be9ad405-6d7c-4c0c-a4d8-bdbedfe6b291"}}, {"cell_type": "code", "outputs": [], "source": ["def imread(buf):\n", "    return cv2.imdecode(np.frombuffer(buf, np.uint8), cv2.IMREAD_ANYCOLOR)\n", "\n", "def img2feat(im):\n", "    x = cv2.resize(im, (im_size, im_size), interpolation=cv2.INTER_AREA)\n", "    return np.float32(x) / 255\n", "\n", "X = np.empty((num_images, im_size, im_size, 3), dtype=np.float32)\n", "y = []\n", "\n", "def load_image(pic, target, bar):\n", "    picture = imread(pic)\n", "    x = img2feat(picture)\n", "    bar.update()\n", "    \n", "    return x, target\n", "\n", "bar = tqdm_notebook(total=num_images)\n", "with open('../input/train.bson', 'rb') as f, \\\n", "        concurrent.futures.ThreadPoolExecutor(num_cpus) as executor:\n", "\n", "    data = bson.decode_file_iter(f)\n", "    delayed_load = []\n", "\n", "    i = 0\n", "    try:\n", "        for c, d in enumerate(data):\n", "            target = d['category_id']\n", "            for e, pic in enumerate(d['imgs']):\n", "                delayed_load.append(executor.submit(load_image, pic['picture'], target, bar))\n", "                \n", "                i = i + 1\n", "\n", "                if i >= num_images:\n", "                    raise IndexError()\n", "\n", "    except IndexError:\n", "        pass;\n", "    \n", "    for i, future in enumerate(concurrent.futures.as_completed(delayed_load)):\n", "        x, target = future.result()\n", "        \n", "        X[i] = x\n", "        y.append(target)"], "execution_count": null, "metadata": {"_uuid": "f287029a9bc0bf8349295e4451d8b14212fc3a45", "_cell_guid": "adf7ff04-ce7a-4fcb-87e0-d71fb0fd4868"}}, {"cell_type": "code", "outputs": [], "source": ["X.shape, len(y)"], "execution_count": null, "metadata": {"_uuid": "d2a79afee31d2b8cf9940b407659ada232b6688d", "_cell_guid": "e50158ec-afa7-4c2e-a39b-3d2a98c468aa"}}, {"cell_type": "code", "outputs": [], "source": ["y = pd.Series(y)\n", "\n", "num_classes =500 \n", "valid_targets = set(y.value_counts().index[:num_classes-1].tolist())\n", "valid_y = y.isin(valid_targets)\n", "\n", "y[~valid_y] = -1\n", "\n", "max_acc = valid_y.mean()\n", "print(max_acc)"], "execution_count": null, "metadata": {"_uuid": "f0a550c5ff4ab3463b2ea15a439a02ab0b3da6b8", "_cell_guid": "f7ddaf41-1883-4f55-b58b-bf350229b3e8"}}, {"cell_type": "code", "outputs": [], "source": ["y, rev_labels = pd.factorize(y)"], "execution_count": null, "metadata": {"_uuid": "27ce33e9cb77eb653663985f0cfe38fa7d34d9a7", "collapsed": true, "_cell_guid": "63d0996b-fdbc-45c7-b102-7284ed51390f"}}, {"cell_type": "code", "outputs": [], "source": ["from keras.layers import Conv2D, MaxPooling2D, Dropout, Dense, Flatten\n", "#from keras.layers import Dense, Dropout, Activation, Flatten\n", "from keras.models import Sequential\n", "from keras.optimizers import Adam\n", "\n", "model = Sequential()\n", "model.add(Conv2D(16, 3 , activation='relu', padding='same', input_shape=X.shape[1:]))\n", "model.add(Conv2D(16, 3, activation='relu', padding='same'))\n", "model.add(MaxPooling2D(2))\n", "model.add(Dropout(0.25))\n", "\n", "model.add(Conv2D(32, 3, activation='relu', padding='same'))\n", "model.add(Conv2D(32, 3, activation='relu', padding='same'))\n", "model.add(MaxPooling2D(2))\n", "model.add(Dropout(0.25))\n", "\n", "model.add(Flatten())\n", "model.add(Dense(num_classes, activation='relu'))\n", "model.add(Dropout(0.5))\n", "model.add(Dense(num_classes, activation='softmax'))\n", "\n", "\n", "opt = Adam(lr=0.001)\n", "\n", "model.compile('adam', 'sparse_categorical_crossentropy', metrics=['accuracy'])\n", "\n", "model.summary()\n", "\n", "\n", "model.fit(X, y, validation_split=0.1, epochs=3)\n", "\n", "\n", "model.save_weights('model.h5')"], "execution_count": null, "metadata": {"_uuid": "35ddb3061e3528bea2f5e00e36741d8d4251f89d", "_cell_guid": "c7221e97-c757-4610-819e-000aa6d50793"}}, {"cell_type": "code", "outputs": [], "source": ["submission = pd.read_csv('../input/sample_submission.csv', index_col='_id')\n", "\n", "most_frequent_guess =1000018296\n", "submission['category_id'] = most_frequent_guess \n", "\n", "num_images_test = 440000\n", "with open('../input/test.bson', 'rb') as f, \\\n", "         concurrent.futures.ThreadPoolExecutor(num_cpus) as executor:\n", "\n", "    data = bson.decode_file_iter(f)\n", "\n", "    future_load = []\n", "\n", "    for i,d in enumerate(data):\n", "        if i >= num_images_test:\n", "              break\n", "        future_load.append(executor.submit(load_image, d['imgs'][0]['picture'], d['_id'], bar))\n", "        \n", "        print(\"Starting future processing\")\n", "    for future in concurrent.futures.as_completed(future_load):\n", "        x, _id = future.result()\n", "        \n", "        y_cat = rev_labels[np.argmax(model.predict(x[None])[0])]\n", "        if y_cat == -1:\n", "            y_cat = most_frequent_guess\n", "\n", "        bar.update()\n", "        submission.loc[_id, 'category_id'] = y_cat\n", "print('Finished')"], "execution_count": null, "metadata": {"_uuid": "c5f309ae6044d5014534dea9436373c42fa8e289", "_cell_guid": "ebb3d0a2-9ac3-439f-84b8-5c359f0e1692"}}, {"cell_type": "code", "outputs": [], "source": ["submission.to_csv('new_submission.csv.gz', compression='gzip')"], "execution_count": null, "metadata": {"_uuid": "0047a8d7b1b0f7667f299ac5c39d8e45862cfdbb", "collapsed": true, "_cell_guid": "c328b383-2bbd-495d-920c-f8a06544036d"}}], "nbformat": 4, "metadata": {"kernelspec": {"display_name": "Python 3", "name": "python3", "language": "python"}, "language_info": {"file_extension": ".py", "pygments_lexer": "ipython3", "version": "3.6.3", "name": "python", "nbconvert_exporter": "python", "mimetype": "text/x-python", "codemirror_mode": {"version": 3, "name": "ipython"}}}}