{"nbformat": 4, "nbformat_minor": 1, "metadata": {"kernelspec": {"language": "python", "name": "python3", "display_name": "Python 3"}, "language_info": {"file_extension": ".py", "nbconvert_exporter": "python", "mimetype": "text/x-python", "pygments_lexer": "ipython3", "codemirror_mode": {"name": "ipython", "version": 3}, "name": "python", "version": "3.6.3"}}, "cells": [{"execution_count": null, "source": ["# This Python 3 environment comes with many helpful analytics libraries installed\n", "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n", "# For example, here's several helpful packages to load in \n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "# Input data files are available in the \"../input/\" directory.\n", "# For example, running texithis (by clicking run or pressing Shift+Enter) will list the files in the input directory\n", "from subprocess import check_output\n", "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n", "# Any results you write to the current directory are saved as output.\n", "category_id = [1000017037,\n", "1000017053,\n", "1000017055,\n", "1000017057,\n", "1000017059,\n", "1000017181,\n", "1000017061,\n", "1000017183,\n", "1000017063,\n", "1000017065,\n", "1000017067,\n", "1000017069,\n", "1000017071,\n", "1000017075,\n", "1000017077]\n", "# 1000017081,1000017083,1000017093,1000017097,1000017101,1000017107,1000017111,1000017133,1000017143,1000017173"], "cell_type": "code", "metadata": {"_uuid": "b6c8ac12476d5331f2d156f77547fc9c06c24514", "_cell_guid": "7c6a027b-7d9c-4071-bfb7-c68de7d20557"}, "outputs": []}, {"execution_count": null, "source": ["import io\n", "import bson                       # this is installed with the pymongo package\n", "import matplotlib.pyplot as plt\n", "from skimage.data import imread   # or, whatever image library you prefer\n", "import multiprocessing as mp      # will come in handy due to the size of the data\n", "from multiprocessing import cpu_count\n", "import concurrent.futures as cf \n", "from tqdm import tqdm_notebook as ttnote\n", "import cv2\n", "num_processes     =    16\n", "im_adw            =    180\n", "im_adh            =    180\n", "#pool             =     mp.Pool(processes = num_processes)\n", "num_cpus          =     cpu_count()\n", "print(num_cpus)"], "cell_type": "code", "metadata": {"_uuid": "00594caa126a1a133a7d40c3c462141ed3a66b23", "_cell_guid": "d0ddf098-be4e-4b54-904f-758de23dd983"}, "outputs": []}, {"execution_count": null, "source": ["# multiprocess version to count the number of images in this 25 categories\n", "def parallel_count(filepath, num_images, category_id):\n", "    bar = ttnote(total = num_images)\n", "    with open(filepath, 'rb') as f, cf.ThreadPoolExecutor(num_cpus) as executor:\n", "        data = bson.decode_file_iter(f)\n", "        delayed_load = []\n", "        i = 0\n", "        try:\n", "            for c, d in enumerate(data):\n", "                target = d['category_id']\n", "                if target in category_id:\n", "                    for e, pic in enumerate(d['imgs']):\n", "                        i = i + 1\n", "                        if i >= num_images:\n", "                            print('too many images found')\n", "                            raise IndexError\n", "\n", "        except IndexError:\n", "            pass;\n", "        print('the number of images to train is %d'%i)\n", " \n", "    return i\n", "\n", "img_cnt = parallel_count('../input/train.bson', 100000, category_id)\n", "#X_test, Y_test = parallel_read('../input/test.bson', num_images_test, im_adw, im_adh)\n", "\n", "                "], "cell_type": "code", "metadata": {"_uuid": "d51bd410621de7e125b5bff3c45ebcac7e2bd618", "collapsed": true, "_cell_guid": "2518107d-8f49-4be6-82b0-d3375b847057"}, "outputs": []}, {"execution_count": null, "source": ["# multiprocess version\n", "# a little bug is it applies operation block by block, and have multiple variables\n", "# I guess we could accelerate it by vecterization\n", "def imread(buf):\n", "    return cv2.imdecode(np.frombuffer(buf, np.uint8), cv2.IMREAD_ANYCOLOR)\n", "\n", "def img2adjust(im):\n", "    x = cv2.resize(im, (im_adw, im_adh), interpolation = cv2.INTER_AREA)\n", "    return np.float32(x)/255\n", "\n", "def load_image(pic, target, bar):\n", "    picture = imread(pic)\n", "    x = img2adjust(picture)\n", "    bar.update() # seems that we are using the function from\n", "    return x, target\n", "    \n", "def parallel_read(filepath, num_images, im_adw, im_adh):\n", "    X = np.empty((num_images, im_adw, im_adh, 3), dtype = np.float32)\n", "    Y = []\n", "    bar = ttnote(total = num_images)\n", "    with open(filepath, 'rb') as f, cf.ThreadPoolExecutor(num_cpus) as executor:\n", "        data = bson.decode_file_iter(f)\n", "        delayed_load = []\n", "        i = 0\n", "        try:\n", "            for c, d in enumerate(data):\n", "                target = d['category_id']\n", "                if target in category_id:\n", "                    for e, pic in enumerate(d['imgs']):\n", "                        delayed_load.append(executor.submit(load_image, pic['picture'],target, bar))\n", "                        i = i + 1\n", "                        if i >= num_images:\n", "                            raise IndexError\n", "\n", "        except IndexError:\n", "            print('the maximium is the best we could do')\n", "            pass;\n", " \n", "        for i, item in enumerate(cf.as_completed(delayed_load)):\n", "            x, target = item.result()\n", "            X[i] = x\n", "            Y.append(target)\n", "    return X, Y\n", "\n", "num_images_train  =    40000\n", "num_images_test   =    round(num_images_train/10)\n", "X_train = np.empty((num_images_train, im_adw, im_adh, 3), dtype = np.float32)\n", "Y_train = []\n", "X_test = np.empty((num_images_test, im_adw, im_adh, 3), dtype = np.float32)\n", "Y_test = []\n", "X_train, Y_train =     parallel_read('../input/train.bson', 10000, im_adw, im_adh)\n", "#X_test, Y_test = parallel_read('../input/test.bson', num_images_test, im_adw, im_adh)\n", "        \n", "                "], "cell_type": "code", "metadata": {"_uuid": "8c031ef8f2b4b5e59dccd7c504011cd62b6bf17d", "_cell_guid": "97cb8b2c-4fec-46c7-82dd-37048f3c1da8"}, "outputs": []}, {"execution_count": null, "source": ["from IPython.display import HTML, Image\n", "print('X_train shape', X_train.shape)\n", "print(len(Y_train))"], "cell_type": "code", "metadata": {"_uuid": "a6c136176a412b34a278162190e52422ec8a9df1", "collapsed": true, "_cell_guid": "c12084c3-9d88-4802-819a-6382aef58093"}, "outputs": []}, {"execution_count": null, "source": ["rows = 16\n", "cols = 6\n", "fig, ax = plt.subplots(rows, cols, frameon=False, figsize=(15, 25))\n", "fig.suptitle('Image from Each Product', fontsize=20)\n", "for i in range(rows):\n", "    for j in range(cols):\n", "        product_image = X_train[i*cols + j,:,:,:]\n", "        product_cater = Y_train[i*cols + j]\n", "        ax[i][j].imshow(product_image)\n", "        ec = (0, .6, .1)\n", "        fc = (0, .7, .2)\n", "        ax[i][j].text(0, -20, product_cater, size=10, rotation=0,\n", "                ha=\"left\", va=\"top\", \n", "                bbox=dict(boxstyle=\"round\", ec=ec, fc=fc))\n", "plt.setp(ax, xticks=[], yticks=[])\n", "plt.tight_layout(rect=[0, 0.03, 1, 0.95])"], "cell_type": "code", "metadata": {"_uuid": "77a9733953e2258a84029fde9a39872c1f687e3e", "collapsed": true, "_cell_guid": "f00a7bf7-6cd9-4d65-89e6-f0d3f7bd165e"}, "outputs": []}, {"execution_count": null, "source": ["from ipywidgets import interact, interactive, fixed\n", "import ipywidgets as widgets\n", "\n", "@interact(n = (0, len(X_train)))\n", "def show_pic(n):\n", "    plt.imshow(X_train[n])\n", "    print('Category :', Y_train[n])"], "cell_type": "code", "metadata": {"_uuid": "737b4a2b907a03a1a9d56295fa10f047406ba0b2", "collapsed": true, "_cell_guid": "fa5673d2-b048-4dae-a565-1dbf245a58ec"}, "outputs": []}, {"source": ["**Image Augmentation with Keras**"], "cell_type": "markdown", "metadata": {"_uuid": "29bb601dcba9c36b302dc55edb50fb7440d1b5ef", "_cell_guid": "82b57c56-965f-44f6-a485-9263dc740c2e"}}, {"execution_count": null, "source": ["from keras.layers import Flatten, Dense, Conv2D, AveragePooling2D, Dropout, GlobalAveragePooling2D\n", "from keras.layers import MaxPooling2D, ZeroPadding2D\n", "from keras.layers.normalization import BatchNormalization\n", "from keras.models import Model, Sequential\n", "from keras.optimizers import SGD, Adam\n", "from keras.activations import softmax\n", "from keras.regularizers import l2\n", "from keras.callbacks import ModelCheckpoint, CSVLogger, LearningRateScheduler\n", "from keras.callbacks import EarlyStopping, ReduceLROnPlateau, TensorBoard\n", "from keras.preprocessing.image import ImageDataGenerator\n", "from keras.utils.np_utils import to_categorical\n", "from keras.preprocessing import image\n", "#keras,tensorflow, pytorch, \n", "\n", "#second part\n", "from keras.models import load_model\n", "import keras.backend as K\n", "from keras.metrics import categorical_crossentropy, top_k_categorical_accuracy\n", "import tensorflow as tf\n", "# from multiGPU import MultiGPUModel\n", "\n", "num_class = 500\n", "#Y_train_cat = to_categorical(str(Y_train[:]),num_classes = num_class)\n", "from keras.utils import np_utils\n", "from sklearn.preprocessing import LabelEncoder\n", "import math\n", "\n"], "cell_type": "code", "metadata": {"_uuid": "22d6a72b31d87a7b09f8a5c1e1aff236b71162a8", "collapsed": true, "_cell_guid": "901dfaaa-fc5e-472d-aed2-b2b0d25ee0d8"}, "outputs": []}, {"execution_count": null, "source": ["# encoder the label \n", "encoder = LabelEncoder()\n", "encoder.fit(Y_train)\n", "encoded_Y = encoder.transform(Y_train)\n", "# convert integers to dummy variables (i.e. one hot encoded)\n", "Y_train_cat = np_utils.to_categorical(encoded_Y, num_classes = 1000)\n", "print(Y_train_cat[:1])\n", "modelname = 'InceptionV5'\n", "savemodel = './models/' + modelname\n", "# Generate batches of tensor image data with real-time data augmentation, looped over\n", "TrainDatagen = ImageDataGenerator(\n", "    rescale           = 1,\n", "    zoom_range        = 1,\n", "    width_shift_range = 0.1,# range of random horizontal shift \n", "    height_shift_range= 0.1,\n", "    horizontal_flip   = True,\n", "    data_format=K.image_data_format() # channels last default\n", ")\n", "# batches of augmented/normalized data\n", "train_generator = TrainDatagen.flow(\n", "    X_train,\n", "    Y_train_cat,\n", "    batch_size = 32,\n", "    seed = 11\n", ")"], "cell_type": "code", "metadata": {"_uuid": "7c53e2d4c05e38b36a0c42eb94c51a2d8d9db026", "collapsed": true, "_cell_guid": "70bcc002-6c9f-4951-92d3-099b2a0e9f2a"}, "outputs": []}, {"source": ["****Model Preparation****"], "cell_type": "markdown", "metadata": {"_uuid": "3073f00b8a5a396b6bc87b206b4bb380d419a97b", "_cell_guid": "f321c100-9784-4bb5-a8d1-f8f4f1a905ca"}}, {"execution_count": null, "source": ["print(X_train.shape)\n", "print(len(Y_train))\n", "print(len(Y_train_cat))\n", "print(Y_train[1])\n", "print(Y_train_cat[100])"], "cell_type": "code", "metadata": {"_uuid": "d0bcc7320ecff464dc9b248dbf6d98e8db4b9fda", "collapsed": true, "_cell_guid": "ef064ff9-6802-42bd-b891-80554945a496"}, "outputs": []}, {"execution_count": null, "source": ["# Choice 1: DIY model;\n", "# Choice 2: reuse model;\n", "from keras.applications.xception import Xception\n", "from keras.applications.inception_v3 import InceptionV3\n", "from keras.preprocessing import image\n", "from keras.layers import Input\n", "model = Sequential()\n", "model.add(Conv2D(16, 3, activation='relu', padding = 'same', input_shape=(140, 140, 3)))\n", "model.add(Conv2D(16, 3, activation='relu', padding = 'same'))\n", "model.add(MaxPooling2D(2))\n", "model.add(Dropout(0.25))\n", "model.add(Conv2D(32, 3, activation='relu', padding = 'same'))\n", "model.add(Conv2D(32, 3, activation='relu', padding = 'same'))\n", "model.add(MaxPooling2D(2))\n", "model.add(Dropout(0.2))\n", "model.add(Flatten())\n", "# \n", "model.add(Dense(num_class*2, activation='relu'))\n", "model.add(Dropout(0.5))\n", "model.add(Dense(num_class*2, activation='relu'))\n", "model.add(Dropout(0.5))\n", "model.add(Dense(num_class*2, activation='softmax'))\n", "\n", "\n", "optim = Adam(lr=0.001, beta_1=0.9, beta_2=0.999, epsilon=1e-08, decay=0.0)\n", "model.compile(optimizer = optim, \n", "              loss  = 'categorical_crossentropy',\n", "              metrics = ['accuracy'])\n", "model.fit(X_train, Y_train_cat,\n", "          epochs=10,\n", "          batch_size=128)\n", "# base_model = InceptionV3(weights='imagenet', include_top=False, input_tensor=None, input_shape=(im_adw, im_adh, 3))\n", "# base_model = Xception(include_top=True, weights='imagenet', input_tensor=None, input_shape=None, pooling=None, classes=1000)\n", "model.save_weights('model.h5')\n"], "cell_type": "code", "metadata": {"_uuid": "02fe00d0af0c6b07f4263a7028634682de3d8124", "collapsed": true, "_cell_guid": "9d076bdf-1b59-42f9-a9a8-f01ab4065b8e"}, "outputs": []}, {"execution_count": null, "source": ["model.save_weights('model.h5')\n", "score = model.evaluate(x_test, y_test, batch_size=28)\n", "print(score)"], "cell_type": "code", "metadata": {"_uuid": "9b466a6505396be6650efbf996aa216d0172dc3e", "collapsed": true, "_cell_guid": "c1b98003-2bbf-4657-8835-2400a417f433"}, "outputs": []}, {"execution_count": null, "source": [], "cell_type": "code", "metadata": {"_uuid": "b0b9f0badecb820e158527c84bb3c92f8633cc97", "collapsed": true, "_cell_guid": "2a798f03-6316-461e-8291-0810c722c4a0"}, "outputs": []}]}