{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Introduction\n\nI created this notebook while exploring scikit-learn and tensorflow. Both libraries are used to solve the same toy problem using a few different algorithms. The algorithms aren't optimized, it's basically \"hello world\" so I could get familiar with the APIs.\n\nAlso, more importantly, a friend bet me a chocolate bar that I wouldn't do anything useful during quarantine. So even if the machine learning isn't novel, using Python to generate chocolate almost certainly is :-).","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport os, math\nfrom sklearn import neighbors, svm\nimport tensorflow as tf","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Load data / Define functions used for all algorithms","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Input data files are available in the read-only \"../input/\" directory\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n# https://www.kaggle.com/dansbecker/finding-your-files-in-kaggle-kernels\n\n#Import training set and separate labels from data\ntrain = pd.read_csv('/kaggle/input/digit-recognizer/train.csv')\ny = train['label']\nX = train.drop('label', axis='columns')\n\n#Import test set\ntest = pd.read_csv('/kaggle/input/digit-recognizer/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#This function formats predictions for kaggle submission\n#predictions must be an numpy array\ndef write_submission(predictions, fname):\n    if type(predictions) is np.ndarray:\n        output = pd.DataFrame()\n        output['ImageId'] = range(1,len(predictions) + 1)\n        output['Label'] = predictions\n        output.to_csv(os.path.join('/kaggle/working',fname), index_label=False, index=False)\n    else:\n        raise ValueError('predictions must be a numpy array. %s provided instead' % type(predictions))\n\n#This function converts digit images to a different format\n#input format - pandas DataFrame where each image is a 1d pixel vector\n#output format - np.array where each image is a square image matrix\ndef unflatten(kaggle_df):\n    ret = []\n    kaggle_array = np.array(kaggle_df)\n    dim = int(math.sqrt(kaggle_array.shape[1]))\n    for i in kaggle_array:\n        ret.append(i.reshape(dim, dim))\n    return np.array(ret)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# K Nearest Neighbors using Sklearn","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#Train a KNN classifier using sklearn\nclassifier = neighbors.KNeighborsClassifier(n_neighbors=10)\nclassifier.fit(X[:5000],y[:5000])","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"trusted":true},"cell_type":"code","source":"#Check classifier accuracy on a subset of test data.\n#Running .score on all test data takes a long time to finish\nclassifier.score(X.iloc[500:1500],y.iloc[500:1500])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Generate predictions for all rows in test set\n#Future work: figure out why this takes significantly longer than classifier.fit()\npredictions = classifier.predict(test)\nwrite_submission(predictions, 'knn_predictions.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Support Vector Classifier using Sklearn","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#This is also pretty slow so only training on a subset of the data.\nclassifier = svm.SVC()\nclassifier.fit(X[:5000],y[:5000])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Generate predictions for all rows in test set\npredictions = classifier.predict(test)\nwrite_submission(predictions, 'svc_predictions.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Neural Networks using TensorFlow.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#Normalize input for Neural Networks\n#Reshape input data from a collection of flat pixel vectors to a collection of square images\nX_nn = unflatten(X)\ntest_nn = unflatten(test)\n\n#Map pixel values to interval [0,1]\nX_nn = X_nn / 255.0\ntest_nn = test_nn / 255.0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Define Traditional Neural Network model layers using keras Functional API\ninputs = tf.keras.Input(shape=(28, 28, 1))\ntemp = tf.keras.layers.Flatten()(inputs)\ntemp = tf.keras.layers.Dense(128, activation=\"relu\")(temp)\noutputs = tf.keras.layers.Dense(10, activation=\"softmax\")(temp)\n\n# Instantiate model instance\nnn_model = tf.keras.Model(inputs, outputs)\n\n# Compile the model\nnn_model.compile(optimizer=\"adam\", loss=\"sparse_categorical_crossentropy\")\n\n# Train the model\nnn_model.fit(X_nn, y, batch_size=64, epochs=1)\n\n#Generate predictions for all rows in test set\nprediction_probs = nn_model.predict(test_nn)\npredictions = np.array([np.argmax(p) for p in prediction_probs])\nwrite_submission(predictions, 'nn_predictions.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Convolutional Neural Networks using TensorFlow.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Define Convolutional Neural Network model layers using keras Functional API\ninputs = tf.keras.Input(shape=(28, 28, 1))\ntemp = tf.keras.layers.Conv2D(64, (3,3), activation='relu')(inputs)\ntemp = tf.keras.layers.MaxPooling2D(2,2)(temp)\ntemp = tf.keras.layers.Conv2D(64, (3,3), activation='relu')(temp)\ntemp = tf.keras.layers.MaxPooling2D(2,2)(temp)\ntemp = tf.keras.layers.Flatten()(temp)\ntemp = tf.keras.layers.Dense(128, activation=\"relu\")(temp)\noutputs = tf.keras.layers.Dense(10, activation=\"softmax\")(temp)\n\n# Instantiate model instance\ncnn_model = tf.keras.Model(inputs, outputs)\n\n# Compile the model\ncnn_model.compile(optimizer=\"adam\", loss=\"sparse_categorical_crossentropy\")\n\n# Train the model\ncnn_model.fit(X_nn, y, batch_size=64, epochs=1)\n\n#Generate predictions for all rows in test set\nprediction_probs = cnn_model.predict(test_nn)\npredictions = np.array([np.argmax(p) for p in prediction_probs])\nwrite_submission(predictions, 'cnn_predictions.csv')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}