{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9797fe0f-827e-ae22-534b-1afd721f785d"
      },
      "outputs": [],
      "source": [
        "# This Python 3 environment comes with many helpful analytics libraries installed\n",
        "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "import numpy as np # linear algebra\n",
        "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "from subprocess import check_output\n",
        "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "363d1965-6189-036e-5c50-d71295efa4d2"
      },
      "outputs": [],
      "source": [
        "trainLabels = pd.read_csv(\"../input/trainLabels.csv\")\n",
        "trainLabels.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3fa9ce14-90c7-3aec-9420-07686a20872a"
      },
      "outputs": [],
      "source": [
        "import os\n",
        "\n",
        "listing = os.listdir(\"../input\") \n",
        "listing.remove(\"trainLabels.csv\")\n",
        "np.size(listing)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "998abb9a-d0e2-820d-c134-ade008d208ac"
      },
      "outputs": [],
      "source": [
        "from PIL import Image\n",
        "from keras.applications.vgg16 import preprocess_input\n",
        "from keras.preprocessing import image\n",
        "\n",
        "# input image dimensions\n",
        "img_rows, img_cols = 224, 224\n",
        "\n",
        "immatrix = []\n",
        "imlabel = []\n",
        "\n",
        "for file in listing:\n",
        "    base = os.path.basename(\"../input/\" + file)\n",
        "    fileName = os.path.splitext(base)[0]\n",
        "    imlabel.append(trainLabels.loc[trainLabels.image==fileName, 'level'].values[0])\n",
        "    im = Image.open(\"../input/\" + file)\n",
        "    img = im.resize((img_rows,img_cols))\n",
        "    #img4d = np.expand_dims(img, axis=0)\n",
        "    #img4d = preprocess_input(img4d)\n",
        "    immatrix.append(np.array(img))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "26eebab4-f834-85fb-377b-fc7b9c85c31c"
      },
      "outputs": [],
      "source": [
        "immatrix = np.asarray(immatrix)\n",
        "imlabel = np.asarray(imlabel)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "55658036-1457-3c42-5715-81655cd6576d"
      },
      "outputs": [],
      "source": [
        "from sklearn.utils import shuffle\n",
        "\n",
        "data,Label = shuffle(immatrix,imlabel, random_state=2)\n",
        "train_data = [data,Label]\n",
        "type(train_data)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "52f33ea7-2949-5837-2539-dde1e730c24c"
      },
      "outputs": [],
      "source": [
        "(X, y) = (train_data[0],train_data[1])"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "11bf8b85-bd8f-38cf-ffe6-d07ecde109c9"
      },
      "outputs": [],
      "source": [
        "from sklearn.cross_validation import train_test_split\n",
        "\n",
        "# STEP 1: split X and y into training and testing sets\n",
        "\n",
        "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=4)\n",
        "\n",
        "print(X_train.shape)\n",
        "print(X_test.shape)\n",
        "\n",
        "X_train = X_train.astype('float32')\n",
        "X_test = X_test.astype('float32')\n",
        "\n",
        "X_train /= 255\n",
        "X_test /= 255\n",
        "\n",
        "print('X_train shape:', X_train.shape)\n",
        "print(X_train.shape[0], 'train samples')\n",
        "print(X_test.shape[0], 'test samples')"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a05568ec-7228-f38f-a2d8-ca038b61f1c6"
      },
      "outputs": [],
      "source": [
        "from keras.utils import np_utils\n",
        "\n",
        "# number of output classes\n",
        "nb_classes = 5\n",
        "\n",
        "# convert class vectors to binary class matrices\n",
        "Y_train = np_utils.to_categorical(y_train, nb_classes)\n",
        "Y_test = np_utils.to_categorical(y_test, nb_classes)\n",
        "\n",
        "i = 100\n",
        "plt.imshow(X_train[i, 0], interpolation='nearest')\n",
        "print(\"label : \", Y_train[i,:])"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f4905271-902b-1169-598b-cabadbd63940"
      },
      "outputs": [],
      "source": [
        "from sklearn.cross_validation import train_test_split\n",
        "\n",
        "# STEP 1: split X and y into training and testing sets\n",
        "\n",
        "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=4)\n",
        "\n",
        "print(X_train.shape)\n",
        "print(X_test.shape)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c91d9103-422a-4849-60c2-27c0a8ae1112"
      },
      "outputs": [],
      "source": [
        "from keras.utils import np_utils\n",
        "\n",
        "# convert class vectors to binary class matrices\n",
        "Y_train = np_utils.to_categorical(y_train, nb_classes)\n",
        "Y_test = np_utils.to_categorical(y_test, nb_classes)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7c5c0b74-a945-11ad-c7e0-1e2997b54372"
      },
      "outputs": [],
      "source": [
        "from keras.applications.vgg16 import VGG16\n",
        "\n",
        "vgg16_model = VGG16(weights=\"imagenet\", include_top=True)\n",
        " \n",
        "    #visualize layers\n",
        "print(\"VGG16 model layers\")\n",
        "for i, layer in enumerate(vgg16_model.layers):\n",
        "    print(i, layer.name, layer.output_shape)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "21e3bad1-f501-b0a7-2138-577739518150"
      },
      "outputs": [],
      "source": [
        "from keras.models import Model, load_model\n",
        "\n",
        "# (2) remove the top layer\n",
        "base_model = Model(input=vgg16_model.input, \n",
        "                   output=vgg16_model.get_layer(\"block5_pool\").output)\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2d5b11d9-98b7-e742-e10d-200a21305cfa"
      },
      "outputs": [],
      "source": [
        "from keras.layers import Dense, Dropout, Reshape\n",
        "\n",
        "# (3) attach a new top layer\n",
        "base_out = base_model.output\n",
        "base_out = Reshape((25088,))(base_out)\n",
        "top_fc1 = Dense(256, activation=\"relu\")(base_out)\n",
        "top_fc1 = Dropout(0.5)(top_fc1)\n",
        "# output layer: (None, 5)\n",
        "top_preds = Dense(5, activation=\"softmax\")(top_fc1)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "da3bfadb-136b-a72b-ff38-8f7d7f484b3b"
      },
      "outputs": [],
      "source": [
        "# (4) freeze weights until the last but one convolution layer (block4_pool)\n",
        "for layer in base_model.layers[0:14]:\n",
        "    layer.trainable = False"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "39b9434d-6b88-ff92-fc66-7c909d21512b"
      },
      "outputs": [],
      "source": [
        "# (5) create new hybrid model\n",
        "model = Model(input=base_model.input, output=top_preds)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "445c2e87-edd1-ef79-6fd6-5887d4ba1d7d"
      },
      "outputs": [],
      "source": [
        "from keras.optimizers import SGD\n",
        "\n",
        "BATCH_SIZE = 32\n",
        "NUM_EPOCHS = 5\n",
        "\n",
        "# (6) compile and train the model\n",
        "sgd = SGD(lr=1e-4, momentum=0.9)\n",
        "model.compile(optimizer=sgd, loss=\"categorical_crossentropy\",\n",
        "              metrics=[\"accuracy\"])\n",
        "\n",
        "history = model.fit([X_train], [Y_train], nb_epoch=NUM_EPOCHS, \n",
        "                    batch_size=BATCH_SIZE, validation_split=0.1, \n",
        "                    callbacks=[checkpoint])"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2ea6e6a7-458d-452a-92ed-e8cd3fd46645"
      },
      "outputs": [],
      "source": [
        "# evaluate final model\n",
        "Ytest = model.predict(X_test)"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}