{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input/driver_imgs_list.csv\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "from __future__ import print_function\n\nimport os\nimport sys\nimport timeit\n\nimport numpy\n\nimport theano\nimport theano.tensor as T\nfrom theano.tensor.signal import pool\nfrom theano.tensor.nnet import conv2d\n\nfrom logistic_sgd import LogisticRegression, load_data\nfrom mlp import HiddenLayer\n\n\nclass LeNetConvPoolLayer(object):\n    \"\"\"Pool Layer of a convolutional network \"\"\"\n\n    def __init__(self, rng, input, filter_shape, image_shape, poolsize=(2, 2)):\n        \"\"\"\n        Allocate a LeNetConvPoolLayer with shared variable internal parameters.\n\n        :type rng: numpy.random.RandomState\n        :param rng: a random number generator used to initialize weights\n\n        :type input: theano.tensor.dtensor4\n        :param input: symbolic image tensor, of shape image_shape\n\n        :type filter_shape: tuple or list of length 4\n        :param filter_shape: (number of filters, num input feature maps,\n                              filter height, filter width)\n\n        :type image_shape: tuple or list of length 4\n        :param image_shape: (batch size, num input feature maps,\n                             image height, image width)\n\n        :type poolsize: tuple or list of length 2\n        :param poolsize: the downsampling (pooling) factor (#rows, #cols)\n        \"\"\"\n\n        assert image_shape[1] == filter_shape[1]\n        self.input = input\n\n        # there are \"num input feature maps * filter height * filter width\"\n        # inputs to each hidden unit\n        fan_in = numpy.prod(filter_shape[1:])\n        # each unit in the lower layer receives a gradient from:\n        # \"num output feature maps * filter height * filter width\" /\n        #   pooling size\n        fan_out = (filter_shape[0] * numpy.prod(filter_shape[2:]) //\n                   numpy.prod(poolsize))\n        # initialize weights with random weights\n        W_bound = numpy.sqrt(6. / (fan_in + fan_out))\n        self.W = theano.shared(\n            numpy.asarray(\n                rng.uniform(low=-W_bound, high=W_bound, size=filter_shape),\n                dtype=theano.config.floatX\n            ),\n            borrow=True\n        )\n\n        # the bias is a 1D tensor -- one bias per output feature map\n        b_values = numpy.zeros((filter_shape[0],), dtype=theano.config.floatX)\n        self.b = theano.shared(value=b_values, borrow=True)\n\n        # convolve input feature maps with filters\n        conv_out = conv2d(\n            input=input,\n            filters=self.W,\n            filter_shape=filter_shape,\n            input_shape=image_shape\n        )\n\n        # pool each feature map individually, using maxpooling\n        pooled_out = pool.pool_2d(\n            input=conv_out,\n            ds=poolsize,\n            ignore_border=True\n        )\n\n        # add the bias term. Since the bias is a vector (1D array), we first\n        # reshape it to a tensor of shape (1, n_filters, 1, 1). Each bias will\n        # thus be broadcasted across mini-batches and feature map\n        # width & height\n        self.output = T.tanh(pooled_out + self.b.dimshuffle('x', 0, 'x', 'x'))\n\n        # store parameters of this layer\n        self.params = [self.W, self.b]\n\n        # keep track of model input\n        self.input = input\n\n\ndef evaluate_lenet5(learning_rate=0.1, n_epochs=200,\n                    dataset='mnist.pkl.gz',\n                    nkerns=[20, 50], batch_size=500):\n    \"\"\" Demonstrates lenet on MNIST dataset\n\n    :type learning_rate: float\n    :param learning_rate: learning rate used (factor for the stochastic\n                          gradient)\n\n    :type n_epochs: int\n    :param n_epochs: maximal number of epochs to run the optimizer\n\n    :type dataset: string\n    :param dataset: path to the dataset used for training /testing (MNIST here)\n\n    :type nkerns: list of ints\n    :param nkerns: number of kernels on each layer\n    \"\"\"\n\n    rng = numpy.random.RandomState(23455)\n\n    datasets = load_data(dataset)\n\n    train_set_x, train_set_y = datasets[0]\n    valid_set_x, valid_set_y = datasets[1]\n    test_set_x, test_set_y = datasets[2]\n\n    # compute number of minibatches for training, validation and testing\n    n_train_batches = train_set_x.get_value(borrow=True).shape[0]\n    n_valid_batches = valid_set_x.get_value(borrow=True).shape[0]\n    n_test_batches = test_set_x.get_value(borrow=True).shape[0]\n    n_train_batches //= batch_size\n    n_valid_batches //= batch_size\n    n_test_batches //= batch_size\n\n    # allocate symbolic variables for the data\n    index = T.lscalar()  # index to a [mini]batch\n\n    # start-snippet-1\n    x = T.matrix('x')   # the data is presented as rasterized images\n    y = T.ivector('y')  # the labels are presented as 1D vector of\n                        # [int] labels\n\n    ######################\n    # BUILD ACTUAL MODEL #\n    ######################\n    print('... building the model')\n\n    # Reshape matrix of rasterized images of shape (batch_size, 28 * 28)\n    # to a 4D tensor, compatible with our LeNetConvPoolLayer\n    # (28, 28) is the size of MNIST images.\n    layer0_input = x.reshape((batch_size, 1, 28, 28))\n\n    # Construct the first convolutional pooling layer:\n    # filtering reduces the image size to (28-5+1 , 28-5+1) = (24, 24)\n    # maxpooling reduces this further to (24/2, 24/2) = (12, 12)\n    # 4D output tensor is thus of shape (batch_size, nkerns[0], 12, 12)\n    layer0 = LeNetConvPoolLayer(\n        rng,\n        input=layer0_input,\n        image_shape=(batch_size, 1, 28, 28),\n        filter_shape=(nkerns[0], 1, 5, 5),\n        poolsize=(2, 2)\n    )\n\n    # Construct the second convolutional pooling layer\n    # filtering reduces the image size to (12-5+1, 12-5+1) = (8, 8)\n    # maxpooling reduces this further to (8/2, 8/2) = (4, 4)\n    # 4D output tensor is thus of shape (batch_size, nkerns[1], 4, 4)\n    layer1 = LeNetConvPoolLayer(\n        rng,\n        input=layer0.output,\n        image_shape=(batch_size, nkerns[0], 12, 12),\n        filter_shape=(nkerns[1], nkerns[0], 5, 5),\n        poolsize=(2, 2)\n    )\n\n    # the HiddenLayer being fully-connected, it operates on 2D matrices of\n    # shape (batch_size, num_pixels) (i.e matrix of rasterized images).\n    # This will generate a matrix of shape (batch_size, nkerns[1] * 4 * 4),\n    # or (500, 50 * 4 * 4) = (500, 800) with the default values.\n    layer2_input = layer1.output.flatten(2)\n\n    # construct a fully-connected sigmoidal layer\n    layer2 = HiddenLayer(\n        rng,\n        input=layer2_input,\n        n_in=nkerns[1] * 4 * 4,\n        n_out=500,\n        activation=T.tanh\n    )\n\n    # classify the values of the fully-connected sigmoidal layer\n    layer3 = LogisticRegression(input=layer2.output, n_in=500, n_out=10)\n\n    # the cost we minimize during training is the NLL of the model\n    cost = layer3.negative_log_likelihood(y)\n\n    # create a function to compute the mistakes that are made by the model\n    test_model = theano.function(\n        [index],\n        layer3.errors(y),\n        givens={\n            x: test_set_x[index * batch_size: (index + 1) * batch_size],\n            y: test_set_y[index * batch_size: (index + 1) * batch_size]\n        }\n    )\n\n    validate_model = theano.function(\n        [index],\n        layer3.errors(y),\n        givens={\n            x: valid_set_x[index * batch_size: (index + 1) * batch_size],\n            y: valid_set_y[index * batch_size: (index + 1) * batch_size]\n        }\n    )\n\n    # create a list of all model parameters to be fit by gradient descent\n    params = layer3.params + layer2.params + layer1.params + layer0.params\n\n    # create a list of gradients for all model parameters\n    grads = T.grad(cost, params)\n\n    # train_model is a function that updates the model parameters by\n    # SGD Since this model has many parameters, it would be tedious to\n    # manually create an update rule for each model parameter. We thus\n    # create the updates list by automatically looping over all\n    # (params[i], grads[i]) pairs.\n    updates = [\n        (param_i, param_i - learning_rate * grad_i)\n        for param_i, grad_i in zip(params, grads)\n    ]\n\n    train_model = theano.function(\n        [index],\n        cost,\n        updates=updates,\n        givens={\n            x: train_set_x[index * batch_size: (index + 1) * batch_size],\n            y: train_set_y[index * batch_size: (index + 1) * batch_size]\n        }\n    )\n    # end-snippet-1\n\n    ###############\n    # TRAIN MODEL #\n    ###############\n    print('... training')\n    # early-stopping parameters\n    patience = 10000  # look as this many examples regardless\n    patience_increase = 2  # wait this much longer when a new best is\n                           # found\n    improvement_threshold = 0.995  # a relative improvement of this much is\n                                   # considered significant\n    validation_frequency = min(n_train_batches, patience // 2)\n                                  # go through this many\n                                  # minibatche before checking the network\n                                  # on the validation set; in this case we\n                                  # check every epoch\n\n    best_validation_loss = numpy.inf\n    best_iter = 0\n    test_score = 0.\n    start_time = timeit.default_timer()\n\n    epoch = 0\n    done_looping = False\n\n    while (epoch < n_epochs) and (not done_looping):\n        epoch = epoch + 1\n        for minibatch_index in range(n_train_batches):\n\n            iter = (epoch - 1) * n_train_batches + minibatch_index\n\n            if iter % 100 == 0:\n                print('training @ iter = ', iter)\n            cost_ij = train_model(minibatch_index)\n\n            if (iter + 1) % validation_frequency == 0:\n\n                # compute zero-one loss on validation set\n                validation_losses = [validate_model(i) for i\n                                     in range(n_valid_batches)]\n                this_validation_loss = numpy.mean(validation_losses)\n                print('epoch %i, minibatch %i/%i, validation error %f %%' %\n                      (epoch, minibatch_index + 1, n_train_batches,\n                       this_validation_loss * 100.))\n\n                # if we got the best validation score until now\n                if this_validation_loss < best_validation_loss:\n\n                    #improve patience if loss improvement is good enough\n                    if this_validation_loss < best_validation_loss *  \\\n                       improvement_threshold:\n                        patience = max(patience, iter * patience_increase)\n\n                    # save best validation score and iteration number\n                    best_validation_loss = this_validation_loss\n                    best_iter = iter\n\n                    # test it on the test set\n                    test_losses = [\n                        test_model(i)\n                        for i in range(n_test_batches)\n                    ]\n                    test_score = numpy.mean(test_losses)\n                    print(('     epoch %i, minibatch %i/%i, test error of '\n                           'best model %f %%') %\n                          (epoch, minibatch_index + 1, n_train_batches,\n                           test_score * 100.))\n\n            if patience <= iter:\n                done_looping = True\n                break\n\n    end_time = timeit.default_timer()\n    print('Optimization complete.')\n    print('Best validation score of %f %% obtained at iteration %i, '\n          'with test performance %f %%' %\n          (best_validation_loss * 100., best_iter + 1, test_score * 100.))\n    print(('The code for file ' +\n           os.path.split(__file__)[1] +\n           ' ran for %.2fm' % ((end_time - start_time) / 60.)), file=sys.stderr)\n\nif __name__ == '__main__':\n    evaluate_lenet5()\n\n\ndef experiment(state, channel):\n    evaluate_lenet5(state.learning_rate, dataset=state.dataset)"
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}