{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Importing libraries"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport keras\nimport tensorflow as tf\nimport keras.layers as L\nimport keras.models as M\nimport keras.initializers as I\nimport keras.backend as K\nfrom keras import optimizers\nfrom sklearn.model_selection import train_test_split\nfrom keras.utils import to_categorical\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Importing the data"},{"metadata":{"trusted":true},"cell_type":"code","source":"data=pd.read_csv('../input/digit-recognizer/train.csv')\ndata.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X=data.drop('label',axis=1).values\ny2=data['label'].values","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Making squash function"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Makign the squash function\ndef squash(vectors, axis=-1):\n    squared_norm=K.sum(K.square(vectors),axis,keepdims=True)\n    scale=squared_norm/(1+squared_norm)/(K.sqrt(squared_norm)+K.epsilon())\n    return scale*vectors\n    ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Adding Input layer,convulational layers and primary capsule"},{"metadata":{"trusted":true},"cell_type":"code","source":"img_shape=(28,28,1)\ninp=L.Input(img_shape,100)\n# Adding the first conv1 layer\nconv1=L.Conv2D(filters=256,kernel_size=(2,2),activation='relu',padding='valid')(inp)\n# Adding Maxpooling layer\nmaxpool1=L.MaxPooling2D(pool_size=(1,1))(conv1)\n# Adding second convulational layer\nconv2=L.Conv2D(filters=128,kernel_size=(9,9),activation='relu',padding='valid')(maxpool1)\n# Adding primary cap layer\nconv2=L.Conv2D(filters=8*16,kernel_size=(9,9),strides=2,padding='valid',activation=None)(conv2)\n# Adding the squash activation\nreshape2=L.Reshape([-1,8])(conv2)\nsquashed_output=L.Lambda(squash)(reshape2)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Making Capsule Layer"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Making capsule layer from scratch\nclass CapsuleLayer(L.Layer):\n    def __init__(self,num_capsule,dim_capsule,routing=3,kernel_initializer='glorot_uniform',**kwargs):\n        super(CapsuleLayer,self).__init__(**kwargs)\n        self.num_capsule=num_capsule\n        self.dim_capsule=dim_capsule\n        self.routing=routing\n        self.kernel_initializer=kernel_initializer\n    def build(self,input_shape):\n        assert len(input_shape) >= 3\n        self.input_num_capsule=input_shape[1]\n        self.input_dim_capsule=input_shape[2]\n        \n        #transforming the matrix\n        self.W= self.add_weight(shape=[self.num_capsule,self.input_num_capsule,self.dim_capsule,self.input_dim_capsule],initializer=self.kernel_initializer,name='w')\n        self.built=True\n    def call(self,inputs,training=None):\n        input_expand=tf.expand_dims(tf.expand_dims(inputs,1),-1)\n        inputs_tiled=K.tile(input_expand,[1,self.num_capsule,1,1,1])\n        input_hat=tf.squeeze(tf.map_fn(lambda x: tf.matmul(self.W,x),elems=inputs_tiled))\n        b=tf.zeros(shape=[inputs.shape[0],self.num_capsule,1,self.input_num_capsule])\n        assert self.routing > 0\n        for i in range(self.routing):\n            c=tf.nn.softmax(b,axis=1)\n            output=squash(tf.matmul(c,input_hat))\n            if i<self.routing-1:\n                b+=tf.matmul(output,input_hat,transpose_b=True)\n        return tf.squeeze(output)\n    def compute_output_shape(self,input_shape):\n        return tuple([None,self.num_capsule,self.dim_capsule])\n    def get_config(self):\n        config = {\n            'num_capsule': self.num_capsule,\n            'dim_capsule': self.dim_capsule,\n            'routings': self.routings\n        }\n        base_config = super(CapsuleLayer, self).get_config()\n        return dict(list(base_config.items()) + list(config.items()))\n        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"digitcaps = CapsuleLayer(num_capsule=10, dim_capsule=16, routing=3, name='digitcaps')(squashed_output)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Making length layer which will calculate the length of the vectors"},{"metadata":{"trusted":true},"cell_type":"code","source":"class Length(L.Layer):\n    def call(self,inputs,**kwargs):\n        return tf.sqrt(tf.reduce_sum(tf.square(inputs),-1))\n    def compute_output_shape(self,input_shape):\n        return input_shape[:-1]\n    def get_config(self):\n        config = super(Length, self).get_config()\n        return config","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Layer 4: This is an auxiliary layer to replace each capsule with its length. Just to match the true label's shape.\n# If using tensorflow, this will not be necessary. :)\nout_caps = Length(name='capsnet')(digitcaps)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Making the Masking layer"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Making the masking layer\nclass Mask(L.Layer):\n    def call(self,inputs,**kwargs):\n        if type(inputs) is list:\n            assert len(inputs)==2\n            inputs,mask=inputs\n        else:\n            x=tf.sqrt(tf.reduce_sum(tf.square(inputs),-1))\n            mask=tf.one_hot(indices=tf.argmax(x, 1), depth=x.shape[1])\n        masked=K.batch_flatten(inputs*tf.expand_dims(mask,-1))\n        return masked\n    def compute_output_shape(self, input_shape):\n        if type(input_shape[0]) is tuple:  # true label provided\n            return tuple([None, input_shape[0][1] * input_shape[0][2]])\n        else:  # no true label provided\n            return tuple([None, input_shape[1] * input_shape[2]])\n\n    def get_config(self):\n        config = super(Mask, self).get_config()\n        return config","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y = L.Input(shape=(10,))\nmasked_by_y = Mask()([digitcaps, y])  # The true label is used to mask the output of capsule layer. For training\nmasked = Mask()(digitcaps)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Making the decoder model"},{"metadata":{"trusted":true},"cell_type":"code","source":"decoder = M.Sequential(name='decoder')\ndecoder.add(L.Dense(512, activation='relu', input_dim=16 * 10))\ndecoder.add(L.Dense(1024, activation='relu'))\ndecoder.add(L.Dense(np.prod((28,28,1)), activation='sigmoid'))\ndecoder.add(L.Reshape(target_shape=(28,28,1), name='out_recon'))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Making models"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_model = M.Model([inp, y], [out_caps, decoder(masked_by_y)])\neval_model = M.Model(inp, [out_caps, decoder(masked)])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Making the loss function"},{"metadata":{"trusted":true},"cell_type":"code","source":"def margin_loss(y_true, y_pred):\n    \"\"\"\n    Margin loss for Eq.(4). When y_true[i, :] contains not just one `1`, this loss should work too. Not test it.\n    :param y_true: [None, n_classes]\n    :param y_pred: [None, num_capsule]\n    :return: a scalar loss value.\n    \"\"\"\n    # return tf.reduce_mean(tf.square(y_pred))\n    L = y_true * tf.square(tf.maximum(0., 0.9 - y_pred)) + \\\n        0.5 * (1 - y_true) * tf.square(tf.maximum(0., y_pred - 0.1))\n\n    return tf.reduce_mean(tf.reduce_sum(L, 1))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Final Training Model"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_model.summary()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Training the model"},{"metadata":{"trusted":true},"cell_type":"code","source":"X=np.array(X)\ny2=np.array(y2)\nX_train, X_test, y_train2, y_test2 = train_test_split(X, y2, test_size=0.1, random_state=42)\nx_train = X_train.astype('float32') / 255.\nx_train = x_train.reshape(-1,28,28,1)\ny_train = np.array(to_categorical(y_train2.astype('float32')))\n\nx_test = X_test.astype('float32') / 255.\nx_test = x_test.reshape(-1,28,28,1)\ny_test = np.array(to_categorical(y_test2.astype('float32')))\n\nx_output = x_train.reshape(-1,784)\nX_valid_output = x_test.reshape(-1,784)\n\nn_samples = 5\n\nplt.figure(figsize=(n_samples * 2, 3))\nfor index in range(n_samples):\n    plt.subplot(1, n_samples, index + 1)\n    sample_image = x_test[index].reshape(28, 28)\n    plt.imshow(sample_image, cmap=\"binary\")\n    plt.title(\"Label:\" + str(y_test2[index]))\n    plt.axis(\"off\")\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"m = 100\nepochs = 16\n# Using EarlyStopping, end training when val_accuracy is not improved for 10 consecutive times\nearly_stopping = keras.callbacks.EarlyStopping(monitor='val_capsnet_accuracy',mode='max',\n                                    patience=2,restore_best_weights=True)\n\n# Using ReduceLROnPlateau, the learning rate is reduced by half when val_accuracy is not improved for 5 consecutive times\nlr_scheduler = keras.callbacks.ReduceLROnPlateau(monitor='val_capsnet_accuracy',mode='max',factor=0.5,patience=4)\ntrain_model.compile(optimizer=keras.optimizers.Adam(lr=0.001),loss=[margin_loss,'mse'],loss_weights = [1. ,0.0005],metrics=['accuracy'])\ntrain_model.fit([x_train, y_train],[y_train,x_train], batch_size = m, epochs = epochs, validation_data = ([x_test, y_test],[y_test,x_test]),callbacks=[early_stopping,lr_scheduler])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"label_predicted, image_predicted = train_model.predict([x_test[:4000], y_test[:4000]])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n_samples = 5\n\nplt.figure(figsize=(n_samples * 2, 3))\nfor index in range(n_samples):\n    plt.subplot(1, n_samples, index + 1)\n    sample_image = x_test[index].reshape(28, 28)\n    plt.imshow(sample_image, cmap=\"binary\")\n    plt.title(\"Label:\" + str(y_test2[index]))\n    plt.axis(\"off\")\n\nplt.show()\n\nplt.figure(figsize=(n_samples * 2, 3))\nfor index in range(n_samples):\n    plt.subplot(1, n_samples, index + 1)\n    sample_image = image_predicted[index].reshape(28, 28)\n    plt.imshow(sample_image, cmap=\"binary\")\n    plt.title(\"Predicted:\" + str(np.argmax(label_predicted[index])))\n    plt.axis(\"off\")\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Making Submission File"},{"metadata":{"trusted":true},"cell_type":"code","source":"test_file=pd.read_csv('../input/digit-recognizer/test.csv')\ntest_file.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_x=test_file.values\ntest_x=np.array(test_x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_x = test_x.astype('float32') / 255.\ntest_x = test_x.reshape(-1,28,28,1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"op=eval_model.predict(test_x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions=[]\nfor i in op[0]:\n    predictions.append(np.argmax(i))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Making the submission file\nsubmission=pd.DataFrame()\nsubmission['ImageId']=[i+1 for i in range(len(predictions))]\nsubmission['Label']=predictions\nsubmission.to_csv('Submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Thank you"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}