{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-02T09:56:52.380353Z","iopub.execute_input":"2022-08-02T09:56:52.380854Z","iopub.status.idle":"2022-08-02T09:56:52.407527Z","shell.execute_reply.started":"2022-08-02T09:56:52.380738Z","shell.execute_reply":"2022-08-02T09:56:52.406580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/digit-recognizer/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:56:53.010250Z","iopub.execute_input":"2022-08-02T09:56:53.010943Z","iopub.status.idle":"2022-08-02T09:56:56.409765Z","shell.execute_reply.started":"2022-08-02T09:56:53.010904Z","shell.execute_reply":"2022-08-02T09:56:56.408447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Getting the data ready for use\ndf = np.array(df)\nnp.random.shuffle(df)\nm, n = df.shape\n\n#Splitting the data into development and test data\ndev_data = df[:1000].T\ny_dev = dev_data[0]\nX_dev = dev_data[1:n]/255\n\ntest_data = df[1000:m].T\ny_test = test_data[0]\nX_test = test_data[1:n]/255","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:56:59.077170Z","iopub.execute_input":"2022-08-02T09:56:59.077600Z","iopub.status.idle":"2022-08-02T09:56:59.897913Z","shell.execute_reply.started":"2022-08-02T09:56:59.077564Z","shell.execute_reply":"2022-08-02T09:56:59.896589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#The following code are a bunch of functions that will be used in gradient descent\n#in order to optimize the neural network.\n\n#Weight intialization using \"he\" method, because we will use relu activation functions.\ndef weight_init():\n    w_1 = np.random.randn(10, 784) *np.sqrt(2/784)\n    b_1 = np.random.randn(10, 1) *np.sqrt(2/784)\n    w_2 = np.random.randn(10, 10) *np.sqrt(2/784)\n    b_2 = np.random.randn(10, 1) *np.sqrt(2/784)\n    w_3 = np.random.randn(10, 10) *np.sqrt(2/784)\n    b_3 = np.random.randn(10, 1) *np.sqrt(2/784)\n    return w_1, b_1, w_2, b_2, w_3, b_3\n\n#Forward propagation\ndef forward_prop(X, w_1, b_1, w_2, b_2, w_3, b_3):\n    z_1 = w_1.dot(X) + b_1\n    a_1 = relu(z_1)\n    \n    z_2 = w_2.dot(a_1) + b_2\n    a_2 = relu(z_2)\n    \n    z_3 = w_3.dot(a_2) + b_3\n    a_3 = softmax(z_3)\n    return z_1, a_1, z_2, a_2, z_3, a_3\n\n#Defining the relu function\ndef relu(Z):\n    return np.maximum(0, Z)\n\n#Defining the softmax function\ndef softmax(Z):\n    return np.exp(Z)/sum(np.exp(Z))\n\n#Putting y in a NumPy array\ndef one_hot(y):\n    \"\"\"\n    This function main purpose is to put y in a vector, so it can be used to calculate the loss function.\n    \"\"\"\n    one_hot_y = np.zeros((y.size, y.max() + 1))\n    one_hot_y[np.arange(y.size), y] = 1\n    one_hot_y = one_hot_y.T\n    return one_hot_y\n\n#Backward propagation\ndef back_prop(X , y, a_3, w_3, a_2, w_2, a_1, z_3, z_2, z_1):\n    \"\"\"\n    Essentialy, backprop is an algorithm used to find the derivates of the neural network weights.\n    I derived them on paper first, and then coded them here, I'd suggest doing that too.\n    \n    dz_3 : It's the derivate of the loss function with respect to z_3\n    dw_3 : ...................................... with respect to w_3\n    db_3 : ...................................... with respect to b_3\n    dz_2 : ...................................... with respect to z_3\n    ...\n    ..\n    .\n    \"\"\"\n    one_hot_y = one_hot(y)\n    dz_3 = a_3 - one_hot_y\n    dw_3 = 1/m * dz_3.dot(a_2.T)\n    db_3 = 1/m * np.sum(dz_3)\n    \n    dz_2 = w_3.T.dot(dz_3) * relu_deriv(z_2)\n    dw_2 = 1/m * dz_2.dot(a_1.T)\n    db_2 = 1/m * np.sum(dz_2)\n    \n    dz_1 = w_2.T.dot(dz_2) * relu_deriv(z_1)\n    dw_1 = 1/m * dz_1.dot(X.T)\n    db_1 = 1/m * np.sum(dz_1)\n    return dw_1, db_1, dw_2, db_2, dw_3, db_3\n\n#The derivative of relu \ndef relu_deriv(Z):\n    return Z > 0\n\n#Parameters update\ndef update_params(alpha, dw_1, db_1, dw_2, db_2, dw_3, db_3, w_1, b_1, w_2, b_2, w_3, b_3):\n    \"\"\"\n    Here's where backprop's magic takes place. We update the parameters as if they are gradient\n    descent by their own, and by a large number of iterations, we find better values for our neural\n    network.\n    \n    alpha : It's the learning rate. too large and the gradient overshoots, too small and we take a lot \n    of examples and iterations.\n    \"\"\"\n    w_1 += -alpha*dw_1\n    b_1 += -alpha*db_1\n    \n    w_2 += -alpha*dw_2\n    b_2 += -alpha*db_2\n    \n    w_3 += -alpha*dw_3\n    b_3 += -alpha*db_3\n    return w_1, b_1, w_2, b_2, w_3, b_3\n\n#Gets Predictions\ndef get_predictions(A3):\n    return np.argmax(A3, 0)\n\n#Gets Accuracy\ndef get_accuracy(predictions, Y):\n    print(predictions, Y)\n    return np.sum(predictions == Y) / Y.size","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:18:28.871403Z","iopub.execute_input":"2022-08-02T10:18:28.871869Z","iopub.status.idle":"2022-08-02T10:18:28.894323Z","shell.execute_reply.started":"2022-08-02T10:18:28.871832Z","shell.execute_reply":"2022-08-02T10:18:28.892876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Gradient descent\ndef gradient_descent(X, y, iterations, alpha):\n    \n    w_1, b_1, w_2, b_2, w_3, b_3 = weight_init()\n    \n    for i in range(iterations):\n        \n        z_1, a_1, z_2, a_2, z_3, a_3 = forward_prop(X, w_1, b_1, w_2, b_2, w_3, b_3)\n        \n        dw_1, db_1, dw_2, db_2, dw_3, db_3 = back_prop(X , y, a_3, w_3, a_2, w_2, a_1, z_3, z_2, z_1)\n        \n        w_1, b_1, w_2, b_2, w_3, b_3 = update_params(alpha, dw_1, db_1, dw_2, db_2, dw_3, db_3, w_1, b_1, w_2, b_2, w_3, b_3)\n        \n        if i % 1000 == 0:\n            print(\"Iteration: \", i)                     \n            predictions = get_predictions(a_3)\n            print(get_accuracy(predictions, y))\n            \"\"\"\n            The if statement prints the prediction matrix, the y matrix, and the accuracy of our predictions every 1000 iterations.\n            \"\"\"\n    return w_1, b_1, w_2, b_2, w_3, b_3","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:27:16.712177Z","iopub.execute_input":"2022-08-02T10:27:16.713223Z","iopub.status.idle":"2022-08-02T10:27:16.726511Z","shell.execute_reply.started":"2022-08-02T10:27:16.713098Z","shell.execute_reply":"2022-08-02T10:27:16.724703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"w_1, b_1, w_2, b_2, w_3, b_3 = gradient_descent(X_test, y_test, 10000, 0.1)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:27:17.661894Z","iopub.execute_input":"2022-08-02T10:27:17.662618Z","iopub.status.idle":"2022-08-02T10:35:47.607918Z","shell.execute_reply.started":"2022-08-02T10:27:17.662580Z","shell.execute_reply":"2022-08-02T10:35:47.606200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_predictions(X, w_1, b_1, w_2, b_2, w_3, b_3):\n    _, _, _, _, _, a_3 = forward_prop(X, w_1, b_1, w_2, b_2, w_3, b_3)\n    predictions = get_predictions(a_3)\n    return predictions\n\ndef test_prediction(index, w_1, b_1, w_2, b_2, w_3, b_3):\n    current_image = X_test[:, index, None]\n    prediction = make_predictions(X_test[:, index, None], w_1, b_1, w_2, b_2, w_3, b_3)\n    label = y_test[index]\n    print(\"Prediction: \", prediction)\n    print(\"Label: \", label)\n    \n    current_image = current_image.reshape((28, 28)) * 255\n    plt.gray()\n    plt.imshow(current_image, interpolation='nearest')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:47:08.416363Z","iopub.execute_input":"2022-08-02T10:47:08.416827Z","iopub.status.idle":"2022-08-02T10:47:08.426689Z","shell.execute_reply.started":"2022-08-02T10:47:08.416790Z","shell.execute_reply":"2022-08-02T10:47:08.424972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(5):\n    test_prediction(i, w_1, b_1, w_2, b_2, w_3, b_3)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:51:46.605069Z","iopub.execute_input":"2022-08-02T10:51:46.605522Z","iopub.status.idle":"2022-08-02T10:51:47.570384Z","shell.execute_reply.started":"2022-08-02T10:51:46.605489Z","shell.execute_reply":"2022-08-02T10:51:47.569107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We will calculate how well our neural network will do on our dev set\ndev_predictions = make_predictions(X_dev, w_1, b_1, w_2, b_2, w_3, b_3)\nget_accuracy(dev_predictions, y_dev)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:48:48.045526Z","iopub.execute_input":"2022-08-02T10:48:48.045989Z","iopub.status.idle":"2022-08-02T10:48:48.075721Z","shell.execute_reply.started":"2022-08-02T10:48:48.045957Z","shell.execute_reply":"2022-08-02T10:48:48.073975Z"},"trusted":true},"execution_count":null,"outputs":[]}]}