{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T08:00:29.930566Z","iopub.execute_input":"2022-07-28T08:00:29.931254Z","iopub.status.idle":"2022-07-28T08:00:29.942509Z","shell.execute_reply.started":"2022-07-28T08:00:29.931216Z","shell.execute_reply":"2022-07-28T08:00:29.941143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/digit-recognizer/train.csv\")\ntest_df = pd.read_csv(\"../input/digit-recognizer/test.csv\")\n\ny_train = train_df.label.values\n# print(type(y_train))\nx_train = train_df.loc[:,train_df.columns != \"label\"].values/255\n# print(type(x_train),x_train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:00:29.945172Z","iopub.execute_input":"2022-07-28T08:00:29.946184Z","iopub.status.idle":"2022-07-28T08:00:33.679823Z","shell.execute_reply.started":"2022-07-28T08:00:29.946116Z","shell.execute_reply":"2022-07-28T08:00:33.678823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train, x_valid, y_train, y_valid = train_test_split(x_train, y_train, test_size = 0.2)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:00:33.681632Z","iopub.execute_input":"2022-07-28T08:00:33.681976Z","iopub.status.idle":"2022-07-28T08:00:34.054588Z","shell.execute_reply.started":"2022-07-28T08:00:33.681939Z","shell.execute_reply":"2022-07-28T08:00:34.053361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import Libraries\nimport torch\nimport torch.nn as nn\nfrom torch.autograd import Variable\nfrom torch.utils.data import DataLoader\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:00:34.056149Z","iopub.execute_input":"2022-07-28T08:00:34.056526Z","iopub.status.idle":"2022-07-28T08:00:34.062282Z","shell.execute_reply.started":"2022-07-28T08:00:34.056489Z","shell.execute_reply":"2022-07-28T08:00:34.061063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = torch.from_numpy(x_train).type(torch.float32)\ny_train = torch.from_numpy(y_train).type(torch.int64) # data type is long\n\n# create feature and targets tensor for test set.\nx_valid = torch.from_numpy(x_valid).type(torch.float32)\ny_valid = torch.from_numpy(y_valid).type(torch.int64) # data type is long","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:00:34.065310Z","iopub.execute_input":"2022-07-28T08:00:34.065774Z","iopub.status.idle":"2022-07-28T08:00:34.162167Z","shell.execute_reply.started":"2022-07-28T08:00:34.065736Z","shell.execute_reply":"2022-07-28T08:00:34.161231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = torch.utils.data.TensorDataset(x_train,y_train)\nvalid_dataset = torch.utils.data.TensorDataset(x_valid,y_valid)\n\nbatch_size = 100\ntrain_loader = DataLoader(train_dataset,batch_size = batch_size,shuffle=True)\nvalid_loader = DataLoader(valid_dataset,batch_size = batch_size,shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:00:34.163564Z","iopub.execute_input":"2022-07-28T08:00:34.163903Z","iopub.status.idle":"2022-07-28T08:00:34.170433Z","shell.execute_reply.started":"2022-07-28T08:00:34.163869Z","shell.execute_reply":"2022-07-28T08:00:34.169515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Logistic Regression Model\nclass LogisticRegressionModel(nn.Module):\n    def __init__(self, input_dim, output_dim):\n        super(LogisticRegressionModel, self).__init__()\n        # Linear part\n        self.linear = nn.Linear(input_dim, output_dim)\n    \n    def forward(self, x):\n        out = self.linear(x)\n        return out","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:00:34.171865Z","iopub.execute_input":"2022-07-28T08:00:34.172704Z","iopub.status.idle":"2022-07-28T08:00:34.180570Z","shell.execute_reply.started":"2022-07-28T08:00:34.172669Z","shell.execute_reply":"2022-07-28T08:00:34.179373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:00:34.181804Z","iopub.execute_input":"2022-07-28T08:00:34.182327Z","iopub.status.idle":"2022-07-28T08:00:34.193119Z","shell.execute_reply.started":"2022-07-28T08:00:34.182290Z","shell.execute_reply":"2022-07-28T08:00:34.192323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport sys\nimport torch.optim as optim\nimport torch.nn.functional as F \n\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom torch.utils import data\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:00:34.194363Z","iopub.execute_input":"2022-07-28T08:00:34.195875Z","iopub.status.idle":"2022-07-28T08:00:34.202581Z","shell.execute_reply.started":"2022-07-28T08:00:34.195832Z","shell.execute_reply":"2022-07-28T08:00:34.201322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = test_df.values / 255\nx_test = torch.from_numpy(x_test).type(torch.float32)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:00:25.043335Z","iopub.execute_input":"2022-07-28T08:00:25.043726Z","iopub.status.idle":"2022-07-28T08:00:25.143480Z","shell.execute_reply.started":"2022-07-28T08:00:25.043696Z","shell.execute_reply":"2022-07-28T08:00:25.142515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import SGDClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:12:22.414972Z","iopub.execute_input":"2022-07-28T08:12:22.415777Z","iopub.status.idle":"2022-07-28T08:12:22.432963Z","shell.execute_reply.started":"2022-07-28T08:12:22.415725Z","shell.execute_reply":"2022-07-28T08:12:22.428459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# knn,\nknn = KNeighborsClassifier(n_neighbors = 10)\nknn.fit(x_train, y_train)\nacc_knn = round(knn.score(x_valid, y_valid) * 100, 2)\nacc_knn","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:06:04.394178Z","iopub.execute_input":"2022-07-28T08:06:04.394901Z","iopub.status.idle":"2022-07-28T08:06:17.783516Z","shell.execute_reply.started":"2022-07-28T08:06:04.394863Z","shell.execute_reply":"2022-07-28T08:06:17.782552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Random Forest\n\nrandom_forest = RandomForestClassifier(n_estimators=100)\nrandom_forest.fit(x_train, y_train)\nacc_random_forest = round(random_forest.score(x_valid, y_valid) * 100, 2)\nacc_random_forest","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:06:17.785207Z","iopub.execute_input":"2022-07-28T08:06:17.786861Z","iopub.status.idle":"2022-07-28T08:06:38.056533Z","shell.execute_reply.started":"2022-07-28T08:06:17.786815Z","shell.execute_reply":"2022-07-28T08:06:38.055536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Support Vector Machines\nsvc = SVC()\nsvc.fit(x_train, y_train)\nacc_svc = round(svc.score(x_valid, y_valid) * 100, 2)\nacc_svc","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:09:21.647431Z","iopub.execute_input":"2022-07-28T08:09:21.648101Z","iopub.status.idle":"2022-07-28T08:11:33.304464Z","shell.execute_reply.started":"2022-07-28T08:09:21.648056Z","shell.execute_reply":"2022-07-28T08:11:33.303512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Stochastic Gradient Descent\nsgd = SGDClassifier()\nsgd.fit(x_train, y_train)\nacc_sgd = round(sgd.score(x_valid, y_valid) * 100, 2)\nacc_sgd","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:12:25.480288Z","iopub.execute_input":"2022-07-28T08:12:25.480653Z","iopub.status.idle":"2022-07-28T08:12:39.629490Z","shell.execute_reply.started":"2022-07-28T08:12:25.480618Z","shell.execute_reply":"2022-07-28T08:12:39.628306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_scores = pd.DataFrame({\n    'model': [knn,random_forest,SVC,SGDClassifier],\n    'score': [acc_knn,acc_random_forest,acc_svc,acc_sgd]\n})\nmodel_scores.sort_values(by='score',ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:14:01.311651Z","iopub.execute_input":"2022-07-28T08:14:01.312384Z","iopub.status.idle":"2022-07-28T08:14:01.370442Z","shell.execute_reply.started":"2022-07-28T08:14:01.312349Z","shell.execute_reply":"2022-07-28T08:14:01.369383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = svc.predict(x_test)\nsubmission = pd.DataFrame({\n    'Image':test_df.index + 1,\n    'Label':preds\n})\n\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:15:41.720031Z","iopub.execute_input":"2022-07-28T08:15:41.720380Z","iopub.status.idle":"2022-07-28T08:18:42.498865Z","shell.execute_reply.started":"2022-07-28T08:15:41.720350Z","shell.execute_reply":"2022-07-28T08:18:42.497628Z"},"trusted":true},"execution_count":null,"outputs":[]}]}