{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# NN from scratch using only numpy, pandas and matplotlib\n## created with the tutorial of Samson Zhang","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom matplotlib import pyplot as plt\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T20:32:01.545299Z","iopub.execute_input":"2022-08-13T20:32:01.545859Z","iopub.status.idle":"2022-08-13T20:32:01.558739Z","shell.execute_reply.started":"2022-08-13T20:32:01.545808Z","shell.execute_reply":"2022-08-13T20:32:01.557512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('../input/digit-recognizer/train.csv')\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T20:32:01.562135Z","iopub.execute_input":"2022-08-13T20:32:01.563127Z","iopub.status.idle":"2022-08-13T20:32:04.203808Z","shell.execute_reply.started":"2022-08-13T20:32:01.563074Z","shell.execute_reply":"2022-08-13T20:32:04.201998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = np.array(data)\nm, n = data.shape\nnp.random.shuffle(data)\n\ndata_dev = data[0:1000].T\nY_dev = data_dev[0]\nX_dev = data_dev[1:n]\nX_dev = X_dev / 255.0\n\ndata_train = data[1000:m].T\nY_train = data_train[0]\nX_train = data_train[1:n]\nX_train = X_train / 255.0\n_,m_train = X_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-13T20:32:04.205719Z","iopub.execute_input":"2022-08-13T20:32:04.206300Z","iopub.status.idle":"2022-08-13T20:32:05.085895Z","shell.execute_reply.started":"2022-08-13T20:32:04.206263Z","shell.execute_reply":"2022-08-13T20:32:05.084753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Forward propagation**\n\n$$Z^{[1]} = W^{[1]} X + b^{[1]}$$\n$$A^{[1]} = g_{\\text{ReLU}}(Z^{[1]}))$$\n$$Z^{[2]} = W^{[2]} A^{[1]} + b^{[2]}$$\n$$A^{[2]} = g_{\\text{softmax}}(Z^{[2]})$$\n\n**Backward propagation**\n\n$$dZ^{[2]} = A^{[2]} - Y$$\n$$dW^{[2]} = \\frac{1}{m} dZ^{[2]} A^{[1]T}$$\n$$dB^{[2]} = \\frac{1}{m} \\Sigma {dZ^{[2]}}$$\n$$dZ^{[1]} = W^{[2]T} dZ^{[2]} .* g^{[1]\\prime} (z^{[1]})$$\n$$dW^{[1]} = \\frac{1}{m} dZ^{[1]} A^{[0]T}$$\n$$dB^{[1]} = \\frac{1}{m} \\Sigma {dZ^{[1]}}$$\n\n**Parameter updates**\n\n$$W^{[2]} := W^{[2]} - \\alpha dW^{[2]}$$\n$$b^{[2]} := b^{[2]} - \\alpha db^{[2]}$$\n$$W^{[1]} := W^{[1]} - \\alpha dW^{[1]}$$\n$$b^{[1]} := b^{[1]} - \\alpha db^{[1]}$$\n\n**Vars and shapes**\n\nForward prop\n\n- $A^{[0]} = X$: 784 x m\n- $Z^{[1]} \\sim A^{[1]}$: 10 x m\n- $W^{[1]}$: 10 x 784 (as $W^{[1]} A^{[0]} \\sim Z^{[1]}$)\n- $B^{[1]}$: 10 x 1\n- $Z^{[2]} \\sim A^{[2]}$: 10 x m\n- $W^{[1]}$: 10 x 10 (as $W^{[2]} A^{[1]} \\sim Z^{[2]}$)\n- $B^{[2]}$: 10 x 1\n\nBackprop\n\n- $dZ^{[2]}$: 10 x m ($~A^{[2]}$)\n- $dW^{[2]}$: 10 x 10\n- $dB^{[2]}$: 10 x 1\n- $dZ^{[1]}$: 10 x m ($~A^{[1]}$)\n- $dW^{[1]}$: 10 x 10\n- $dB^{[1]}$: 10 x 1","metadata":{}},{"cell_type":"code","source":"def init_params():\n    W1 = np.random.rand(10, 784) - 0.5\n    b1 = np.random.rand(10, 1) - 0.5\n    W2 = np.random.rand(10, 10) - 0.5\n    b2 = np.random.rand(10, 1) - 0.5\n    return W1, b1, W2, b2\n\ndef ReLU(Z):\n    return np.maximum(0, Z)\n\ndef dReLU(Z):\n    return Z > 0\n\ndef softmax(Z):\n    A = np.exp(Z) / sum(np.exp(Z))\n    return A\n\ndef forward_prop(W1, b1, W2, b2, X):\n    Z1 = W1.dot(X) + b1\n    A1 = ReLU(Z1)\n    Z2 = W2.dot(A1) + b2\n    A2 = softmax(Z2)\n    return Z1, A1, Z2, A2\n\ndef one_hot(Y):\n    one_hot_Y = np.zeros((Y.size, Y.max() + 1))\n    one_hot_Y[np.arange(Y.size), Y] = 1\n    one_hot_Y = one_hot_Y.T\n    return one_hot_Y\n\ndef back_prop(Z1, A1, Z2, A2, W2, X, Y):\n    m = Y.size\n    one_hot_Y = one_hot(Y)\n    dZ2 = A2 - one_hot_Y\n    dW2 = 1 / m * dZ2.dot(A1.T)\n    db2 = 1 / m * np.sum(dZ2)\n    dZ1 = W2.T.dot(dZ2) * dReLU(Z1)\n    dW1 = 1 / m * dZ1.dot(X.T)\n    db1 = 1 / m * np.sum(dZ1)\n    return dW1, db1, dW2, db2\n\ndef update_params(W1, b1, W2, b2, dW1, db1, dW2, db2, alpha):\n    W1 = W1 - alpha * dW1\n    b1 = b1 - alpha * db1\n    W2 = W2 - alpha * dW2\n    b2 = b2 - alpha * db2\n    return W1, b1, W2, b2","metadata":{"execution":{"iopub.status.busy":"2022-08-13T20:32:05.087342Z","iopub.execute_input":"2022-08-13T20:32:05.088377Z","iopub.status.idle":"2022-08-13T20:32:05.103118Z","shell.execute_reply.started":"2022-08-13T20:32:05.088338Z","shell.execute_reply":"2022-08-13T20:32:05.102173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Returns the index of the maximum value.\n# Finds the output with the highest probability to be correct. \ndef get_predictions(A2):\n    return np.argmax(A2, 0)\n\ndef get_accuracy(predictions, Y):\n    print(predictions, Y)\n    return np.sum(predictions == Y) / Y.size\n\ndef gradient_descent(X, Y, iterations, alpha):\n    W1, b1, W2, b2 = init_params()\n    \n    for i in range(iterations):\n        Z1, A1, Z2, A2 = forward_prop(W1, b1, W2, b2, X)\n        dW1, db1, dW2, db2 = back_prop(Z1, A1, Z2, A2, W2, X, Y)\n        W1, b1, W2, b2 = update_params(W1, b1, W2, b2, dW1, db1, dW2, db2, alpha)\n        if (i % 10 == 0):\n            print('Iteration: ', i)\n            print('Accuracy: ', get_accuracy(get_predictions(A2), Y))\n    return W1, b1, W2, b2","metadata":{"execution":{"iopub.status.busy":"2022-08-13T20:32:05.105450Z","iopub.execute_input":"2022-08-13T20:32:05.106194Z","iopub.status.idle":"2022-08-13T20:32:05.119432Z","shell.execute_reply.started":"2022-08-13T20:32:05.106149Z","shell.execute_reply":"2022-08-13T20:32:05.118315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"W1, b1, W2, b2 = gradient_descent(X_train, Y_train, 500, 0.17)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T20:32:05.121744Z","iopub.execute_input":"2022-08-13T20:32:05.122355Z","iopub.status.idle":"2022-08-13T20:33:20.094465Z","shell.execute_reply.started":"2022-08-13T20:32:05.122310Z","shell.execute_reply":"2022-08-13T20:33:20.092482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_predictions(W1, b1, W2, b2, X):\n    _, _, _, A2 = forward_prop(W1, b1, W2, b2, X)\n    predictions = get_predictions(A2)\n    probability = np.amax(A2) * 100\n    return predictions, probability\n\ndef test_prediction(index, W1, b1, W2, b2):\n    current_image = X_train[:, index, None]\n    prediction, prob = make_predictions(W1, b1, W2, b2, X_train[:, index, None])\n    label = Y_train[index]\n    \n    print('Prediction :', prediction)\n    print('Probability: ', prob, '%')\n    print('Label: ', label)\n    \n    current_image = current_image.reshape((28, 28)) * 255\n    plt.gray()\n    plt.imshow(current_image, interpolation='nearest')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T20:33:20.103177Z","iopub.execute_input":"2022-08-13T20:33:20.104153Z","iopub.status.idle":"2022-08-13T20:33:20.124292Z","shell.execute_reply.started":"2022-08-13T20:33:20.104111Z","shell.execute_reply":"2022-08-13T20:33:20.122144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_prediction(11, W1, b1, W2, b2)\ntest_prediction(91, W1, b1, W2, b2)\ntest_prediction(8, W1, b1, W2, b2)\ntest_prediction(4, W1, b1, W2, b2)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T20:38:32.229194Z","iopub.execute_input":"2022-08-13T20:38:32.229610Z","iopub.status.idle":"2022-08-13T20:38:32.900984Z","shell.execute_reply.started":"2022-08-13T20:38:32.229576Z","shell.execute_reply":"2022-08-13T20:38:32.899852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dev_predictions, _ = make_predictions(W1, b1, W2, b2, X_dev)\ndev_acc = get_accuracy(dev_predictions, Y_dev) * 100\n\nprint('\\nAccuracy on test data: ', dev_acc, '%')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T20:33:20.863271Z","iopub.execute_input":"2022-08-13T20:33:20.863644Z","iopub.status.idle":"2022-08-13T20:33:20.889895Z","shell.execute_reply.started":"2022-08-13T20:33:20.863611Z","shell.execute_reply":"2022-08-13T20:33:20.888260Z"},"trusted":true},"execution_count":null,"outputs":[]}]}