{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nfrom matplotlib import pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T11:41:27.141269Z","iopub.execute_input":"2022-07-27T11:41:27.141652Z","iopub.status.idle":"2022-07-27T11:41:27.146888Z","shell.execute_reply.started":"2022-07-27T11:41:27.141624Z","shell.execute_reply":"2022-07-27T11:41:27.145972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/digit-recognizer/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:41:27.183257Z","iopub.execute_input":"2022-07-27T11:41:27.183834Z","iopub.status.idle":"2022-07-27T11:41:29.931539Z","shell.execute_reply.started":"2022-07-27T11:41:27.183792Z","shell.execute_reply":"2022-07-27T11:41:29.930159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:41:29.933711Z","iopub.execute_input":"2022-07-27T11:41:29.934036Z","iopub.status.idle":"2022-07-27T11:41:29.952000Z","shell.execute_reply.started":"2022-07-27T11:41:29.934008Z","shell.execute_reply":"2022-07-27T11:41:29.951235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = np.array(data)\n#dimension of data\nm, n = data.shape\n#risk of overfitting, not generalized model. Set asside a chunk of data in cross-validation data.\n#shuffle data\nnp.random.shuffle(data)\ndata_dev = data[0:1000].T\nY_dev = data_dev[0]\nX_dev = data_dev[1:n]\nX_dev = X_dev / 255.\n\ndata_train = data[1000:m].T\nY_train = data_train[0]\nX_train = data_train[1:n]\nX_train = X_train / 255.\n_,m_train = X_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:41:29.953036Z","iopub.execute_input":"2022-07-27T11:41:29.953775Z","iopub.status.idle":"2022-07-27T11:41:30.734031Z","shell.execute_reply.started":"2022-07-27T11:41:29.953745Z","shell.execute_reply":"2022-07-27T11:41:30.733225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train[:,0].shape\n#first column has 784 pixels in it.\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:41:30.737036Z","iopub.execute_input":"2022-07-27T11:41:30.737788Z","iopub.status.idle":"2022-07-27T11:41:30.746164Z","shell.execute_reply.started":"2022-07-27T11:41:30.737746Z","shell.execute_reply":"2022-07-27T11:41:30.744522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#initialize all of the parameters.\ndef init_params():\n    #weight 1\n    W1 = np.random.rand(10, 784) - 0.5\n    b1 = np.random.rand(10, 1) - 0.5\n    W2 = np.random.rand(10, 10) - 0.5\n    b2 = np.random.rand(10, 1) - 0.5\n    return W1, b1, W2, b2\n\n# first activation function\ndef ReLU(Z):\n    #whichever is max between Z and 0\n    return np.maximum(Z, 0)\n    \n#second activation function\ndef softmax(Z):\n    A = np.exp(Z) / sum(np.exp(Z))\n    return A\n\n#forward propogation, maths starts from here\ndef for_prop(W1, b1, W2, b2, X):\n    Z1 = W1.dot(X) + b1\n    A1 = ReLU(Z1)\n    Z2 = W2.dot(A1) + b2\n    A2 = softmax(Z2)\n    return Z1, A1, Z2, A2\n    \ndef one_hot(Y):\n    #this creates a perfectly sized matrix\n    one_hot_Y = np.zeros((Y.size, Y.max() + 1))\n    #The code below is really interesting to understand.\n    #np.arange(Y.size) specify what column to access, and specify the value to 1\n    one_hot_Y[np.arange(Y.size), Y] = 1\n    #Now we transpose the value.\n    one_hot_Y = one_hot_Y.T\n    return one_hot_Y\n\ndef der_relu(Z):\n    #fancy but basically, when bool converts to num, true = 1, false = 0\n    # if one element is greater than zero, it returns 1, or else zero\n    return Z > 0\n\n# backpropogation now\ndef back_prop(Z1, A1, Z2, A2, W1, W2, X, Y):\n    one_hot_Y = one_hot(Y)\n    dZ2 = A2 - one_hot_Y\n    dW2 = 1/m * dZ2.dot(A1.T)\n    db2 = 1 / m * np.sum(dZ2)\n    # now for derivation or g prime or derivative of activation funciton, which is reLU\n    dZ1 = W2.T.dot(dZ2)*der_relu(Z1)\n    dW1 = 1/m * dZ1.dot(X.T)\n    db1 = 1 / m * np.sum(dZ1)\n    return dW1, db1, dW2, db2\n\ndef upd_params(W1, b1, W2, b2, dW1, db1, dW2, db2, alpha):\n    W1 = W1 - alpha * dW1\n    b1 = b1 - alpha * db1\n    W2 = W2 - alpha * dW2\n    b2 = b2 - alpha * db2\n    \n    return W1, b1, W2, b2\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:41:30.748492Z","iopub.execute_input":"2022-07-27T11:41:30.748953Z","iopub.status.idle":"2022-07-27T11:41:30.768088Z","shell.execute_reply.started":"2022-07-27T11:41:30.748913Z","shell.execute_reply":"2022-07-27T11:41:30.766796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#after we have all the functions we need from maths.\ndef get_predictions(A2):\n    return np.argmax(A2, 0)\n\ndef accuracy(predictions, Y):\n    print(predictions, Y)\n    return np.sum(predictions == Y) / Y.size\n\ndef grad_descent(X,Y, alpha, iterate):\n    W1, b1, W2, b2 = init_params()\n    for i in range(iterate):\n        Z1, A1, Z2, A2 = for_prop(W1, b1, W2, b2, X)\n        dW1, db1, dW2, db2 = back_prop(Z1, A1, Z2, A2, W1, W2, X, Y)\n        W1, b1, W2, b2 = upd_params(W1, b1, W2, b2, dW1, db1, dW2, db2, alpha)\n        if i % 50 ==0:\n            print(\"Iteration: \", i)\n            pred = get_predictions(A2)\n            print(accuracy(pred, Y))\n    return W1, b1, W2, b2","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:41:30.770638Z","iopub.execute_input":"2022-07-27T11:41:30.771399Z","iopub.status.idle":"2022-07-27T11:41:30.787898Z","shell.execute_reply.started":"2022-07-27T11:41:30.771357Z","shell.execute_reply":"2022-07-27T11:41:30.786739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"W1, b1, W2, b2 = grad_descent(X_train, Y_train, 0.10, 500)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:56:35.083486Z","iopub.execute_input":"2022-07-27T11:56:35.083884Z","iopub.status.idle":"2022-07-27T11:57:56.201529Z","shell.execute_reply.started":"2022-07-27T11:56:35.083856Z","shell.execute_reply":"2022-07-27T11:57:56.199891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now with 84% accuracy, we still have room for improvement and definetly after I get more comfortable with neural networks and more layers, I'd love to come back and try reach higher accuracy.","metadata":{}},{"cell_type":"code","source":"def predictionDisplay(X, W1, b1, W2, b2):\n    _, _, _, A2 = for_prop(W1, b1, W2, b2, X)\n    predictions = get_predictions(A2)\n    return predictions\n\ndef testPredictions(i, W1, b1,W2, b2):\n    image = X_train[:, i, None]\n    prediction = predictionDisplay(X_train[:, i, None], W1, b1, W2, b2)\n    label = Y_train[i]\n    print(\"Prediction: \", prediction)\n    print(\"Label: \", label)\n    \n    image = image.reshape((28,28))* 255\n    plt.gray()\n    plt.imshow(image, interpolation = 'nearest')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:44:41.792458Z","iopub.execute_input":"2022-07-27T11:44:41.793191Z","iopub.status.idle":"2022-07-27T11:44:41.802251Z","shell.execute_reply.started":"2022-07-27T11:44:41.793157Z","shell.execute_reply":"2022-07-27T11:44:41.801010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testPredictions(20,W1,b1,W2,b2)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:45:19.140854Z","iopub.execute_input":"2022-07-27T11:45:19.141240Z","iopub.status.idle":"2022-07-27T11:45:19.327753Z","shell.execute_reply.started":"2022-07-27T11:45:19.141210Z","shell.execute_reply":"2022-07-27T11:45:19.325882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So this model works really well for the effort it took and it took a decent sweat to find even a single bad apple. So here it seems to recognize\n4 as one, clearly this is because it does look alot like 9 but angular. Such problems could be fixed if we had alot more layers and units to train on.\nI'll try to fix this next time.\n","metadata":{}},{"cell_type":"code","source":"testPredictions(88,W1,b1,W2,b2)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:49:28.803291Z","iopub.execute_input":"2022-07-27T11:49:28.803717Z","iopub.status.idle":"2022-07-27T11:49:29.108601Z","shell.execute_reply.started":"2022-07-27T11:49:28.803686Z","shell.execute_reply":"2022-07-27T11:49:29.107687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's try this model on non training dataset, so inorder to also detect any overloading","metadata":{}},{"cell_type":"code","source":"dev_predictions = predictionDisplay(X_dev, W1, b1, W2, b2)\naccuracy(dev_predictions, Y_dev)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:53:08.473036Z","iopub.execute_input":"2022-07-27T11:53:08.473431Z","iopub.status.idle":"2022-07-27T11:53:08.499585Z","shell.execute_reply.started":"2022-07-27T11:53:08.473403Z","shell.execute_reply":"2022-07-27T11:53:08.498301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"84% is not bad at all for data set the model is not trained in. ","metadata":{}}]}