{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T21:24:55.426585Z","iopub.execute_input":"2022-08-13T21:24:55.426963Z","iopub.status.idle":"2022-08-13T21:24:55.437839Z","shell.execute_reply.started":"2022-08-13T21:24:55.426936Z","shell.execute_reply":"2022-08-13T21:24:55.437219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing data","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/digit-recognizer/train.csv')\ndf_test = pd.read_csv('/kaggle/input/digit-recognizer/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:24:55.486898Z","iopub.execute_input":"2022-08-13T21:24:55.487450Z","iopub.status.idle":"2022-08-13T21:24:58.264064Z","shell.execute_reply.started":"2022-08-13T21:24:55.487424Z","shell.execute_reply":"2022-08-13T21:24:58.262590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample =  pd.read_csv('/kaggle/input/digit-recognizer/sample_submission.csv')\nsample","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:24:58.265771Z","iopub.execute_input":"2022-08-13T21:24:58.266148Z","iopub.status.idle":"2022-08-13T21:24:58.280531Z","shell.execute_reply.started":"2022-08-13T21:24:58.266123Z","shell.execute_reply":"2022-08-13T21:24:58.279622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analysing data","metadata":{}},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:24:58.281522Z","iopub.execute_input":"2022-08-13T21:24:58.281761Z","iopub.status.idle":"2022-08-13T21:24:58.301467Z","shell.execute_reply.started":"2022-08-13T21:24:58.281739Z","shell.execute_reply":"2022-08-13T21:24:58.300887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\ndigit = 301\nax = sns.heatmap(df_train.iloc[digit,1:].values.reshape(28,28),cbar=False)\nprint('label',df_train.loc[digit]['label'])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:24:58.303256Z","iopub.execute_input":"2022-08-13T21:24:58.303639Z","iopub.status.idle":"2022-08-13T21:24:58.529797Z","shell.execute_reply.started":"2022-08-13T21:24:58.303615Z","shell.execute_reply":"2022-08-13T21:24:58.528032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating more data from df_train","metadata":{}},{"cell_type":"code","source":"import scipy","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:24:58.531114Z","iopub.execute_input":"2022-08-13T21:24:58.531357Z","iopub.status.idle":"2022-08-13T21:24:58.535095Z","shell.execute_reply.started":"2022-08-13T21:24:58.531334Z","shell.execute_reply":"2022-08-13T21:24:58.534442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def left_shift(row):\n    label = row.label\n    image = row.iloc[1:].values.reshape(28,28)\n    new_image = scipy.ndimage.shift(image,[0,-1],cval=0)\n    return pd.Series(np.concatenate([[label],new_image.reshape(1,784)[0]],axis=0))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:24:58.536066Z","iopub.execute_input":"2022-08-13T21:24:58.536410Z","iopub.status.idle":"2022-08-13T21:24:58.546576Z","shell.execute_reply.started":"2022-08-13T21:24:58.536386Z","shell.execute_reply":"2022-08-13T21:24:58.545500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def right_shift(row):\n    label = row.label\n    image = row.iloc[1:].values.reshape(28,28)\n    new_image = scipy.ndimage.shift(image,[0,1],cval=0)\n    return pd.Series(np.concatenate([[label],new_image.reshape(1,784)[0]],axis=0))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:24:58.547726Z","iopub.execute_input":"2022-08-13T21:24:58.548040Z","iopub.status.idle":"2022-08-13T21:24:58.560321Z","shell.execute_reply.started":"2022-08-13T21:24:58.548006Z","shell.execute_reply":"2022-08-13T21:24:58.558557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def up_shift(row):\n    label = row.label\n    image = row.iloc[1:].values.reshape(28,28)\n    new_image = scipy.ndimage.shift(image,[-1,0],cval=0)\n    return pd.Series(np.concatenate([[label],new_image.reshape(1,784)[0]],axis=0))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:24:58.561785Z","iopub.execute_input":"2022-08-13T21:24:58.562209Z","iopub.status.idle":"2022-08-13T21:24:58.572174Z","shell.execute_reply.started":"2022-08-13T21:24:58.562099Z","shell.execute_reply":"2022-08-13T21:24:58.571021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def down_shift(row):\n    label = row.label\n    image = row.iloc[1:].values.reshape(28,28)\n    new_image = scipy.ndimage.shift(image,[1,0],cval=0)\n    return pd.Series(np.concatenate([[label],new_image.reshape(1,784)[0]],axis=0))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:24:58.573494Z","iopub.execute_input":"2022-08-13T21:24:58.573922Z","iopub.status.idle":"2022-08-13T21:24:58.583259Z","shell.execute_reply.started":"2022-08-13T21:24:58.573835Z","shell.execute_reply":"2022-08-13T21:24:58.582247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"left_shifted_digit = left_shift(df_train.iloc[digit,:])\nsns.heatmap(left_shifted_digit.loc[1:].values.reshape(28,28),cbar=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:24:58.588530Z","iopub.execute_input":"2022-08-13T21:24:58.588833Z","iopub.status.idle":"2022-08-13T21:24:58.802657Z","shell.execute_reply.started":"2022-08-13T21:24:58.588805Z","shell.execute_reply":"2022-08-13T21:24:58.801655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_leftshifted = df_train.apply(left_shift,axis=1)\ndf_train_rightshifted = df_train.apply(right_shift,axis=1)\ndf_train_upshifted = df_train.apply(up_shift,axis=1)\ndf_train_downshifted = df_train.apply(down_shift,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:24:58.803862Z","iopub.execute_input":"2022-08-13T21:24:58.804210Z","iopub.status.idle":"2022-08-13T21:25:56.213676Z","shell.execute_reply.started":"2022-08-13T21:24:58.804176Z","shell.execute_reply":"2022-08-13T21:25:56.212338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"column_dict = {}\nfor i in range(len(df_train.columns)):\n    column_dict[i] = df_train.columns[i]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:25:56.215034Z","iopub.execute_input":"2022-08-13T21:25:56.215277Z","iopub.status.idle":"2022-08-13T21:25:56.221149Z","shell.execute_reply.started":"2022-08-13T21:25:56.215255Z","shell.execute_reply":"2022-08-13T21:25:56.219578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_leftshifted = df_train_leftshifted.rename(columns = column_dict)\ndf_train_rightshifted = df_train_rightshifted.rename(columns = column_dict)\ndf_train_upshifted = df_train_upshifted.rename(columns = column_dict)\ndf_train_downshifted = df_train_downshifted.rename(columns = column_dict)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:25:56.222334Z","iopub.execute_input":"2022-08-13T21:25:56.222758Z","iopub.status.idle":"2022-08-13T21:25:56.387800Z","shell.execute_reply.started":"2022-08-13T21:25:56.222733Z","shell.execute_reply":"2022-08-13T21:25:56.386910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.concat([df_train,df_train_leftshifted,df_train_rightshifted,df_train_upshifted,df_train_downshifted], ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:25:56.388900Z","iopub.execute_input":"2022-08-13T21:25:56.389145Z","iopub.status.idle":"2022-08-13T21:25:56.646237Z","shell.execute_reply.started":"2022-08-13T21:25:56.389123Z","shell.execute_reply":"2022-08-13T21:25:56.645574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:25:56.647317Z","iopub.execute_input":"2022-08-13T21:25:56.647571Z","iopub.status.idle":"2022-08-13T21:25:56.673129Z","shell.execute_reply.started":"2022-08-13T21:25:56.647547Z","shell.execute_reply":"2022-08-13T21:25:56.672184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating stratified test set from training set","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\ndef train_test_indexes(df, test_ratio):\n    try:\n        test_indexes = pd.read_csv('../input/test-indexes/test_indexes.csv')\n        return test_indexes\n    except:\n        value_counts = df.label.value_counts().sort_values(ascending=False)\n        test_indexes = np.array([])\n        for digit in range(10):\n            digit_index = df.where(df.label == digit).dropna().index\n            new_indexes = np.random.choice(digit_index,int(value_counts[digit]*test_ratio),replace=False)\n            test_indexes = np.append(test_indexes,new_indexes)\n            pd.DataFrame(test_indexes).to_csv('./test_indexes.csv',index=False)\n        return test_indexes","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:25:56.674477Z","iopub.execute_input":"2022-08-13T21:25:56.674776Z","iopub.status.idle":"2022-08-13T21:25:56.681176Z","shell.execute_reply.started":"2022-08-13T21:25:56.674753Z","shell.execute_reply":"2022-08-13T21:25:56.680276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_indexes = train_test_indexes(df_train,0.2)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:25:56.682499Z","iopub.execute_input":"2022-08-13T21:25:56.683042Z","iopub.status.idle":"2022-08-13T21:25:56.703861Z","shell.execute_reply.started":"2022-08-13T21:25:56.682959Z","shell.execute_reply":"2022-08-13T21:25:56.703238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_indexes.values.T[0]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:25:56.706924Z","iopub.execute_input":"2022-08-13T21:25:56.707798Z","iopub.status.idle":"2022-08-13T21:25:56.714254Z","shell.execute_reply.started":"2022-08-13T21:25:56.707768Z","shell.execute_reply":"2022-08-13T21:25:56.713664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training models","metadata":{}},{"cell_type":"code","source":"df_train_test = df_train.loc[test_indexes.values.T[0]]\nX_train_test, y_train_test = df_train_test.iloc[:,1:]/255 ,df_train_test.label","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:25:56.715074Z","iopub.execute_input":"2022-08-13T21:25:56.715756Z","iopub.status.idle":"2022-08-13T21:25:57.129369Z","shell.execute_reply.started":"2022-08-13T21:25:56.715732Z","shell.execute_reply":"2022-08-13T21:25:57.128465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_indexes = df_train.index.drop(test_indexes,errors='ignore')\ndf_train_train = df_train.loc[train_indexes]\nX_train_train, y_train_train = df_train_train.iloc[:,1:]/255, df_train_train.label","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:25:57.130537Z","iopub.execute_input":"2022-08-13T21:25:57.130758Z","iopub.status.idle":"2022-08-13T21:25:57.978318Z","shell.execute_reply.started":"2022-08-13T21:25:57.130736Z","shell.execute_reply":"2022-08-13T21:25:57.977649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neural_network import MLPClassifier #1.1988246201344106 +/- 0.07096598682216043\nfrom sklearn.ensemble import RandomForestClassifier #0.9763725040745472 +/- 0.12476488900812599 - Promissor\nfrom sklearn.svm import SVC #0.8251053371295141 +/- 0.12485206341906953 - Promissor\n# SVC OVO - 0.8172937120641922 +/- 0.08368583647567356\nfrom sklearn.linear_model import LogisticRegression #1.45943823841368 +/- 0.07831091096670423\nfrom sklearn.linear_model import SGDClassifier #ruim\nfrom sklearn.neighbors import KNeighborsClassifier #0.9737559402321555 +/- 0.07849562462161529 - Promissor\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.svm import LinearSVC","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:25:57.980045Z","iopub.execute_input":"2022-08-13T21:25:57.980362Z","iopub.status.idle":"2022-08-13T21:25:57.985959Z","shell.execute_reply.started":"2022-08-13T21:25:57.980330Z","shell.execute_reply":"2022-08-13T21:25:57.985370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#clf = SVC(C=10, decision_function_shape='ovo').fit(X_train_train,y_train_train)\nclf = RandomForestClassifier().fit(X_train_train,y_train_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:25:57.986904Z","iopub.execute_input":"2022-08-13T21:25:57.987641Z","iopub.status.idle":"2022-08-13T21:29:12.104722Z","shell.execute_reply.started":"2022-08-13T21:25:57.987614Z","shell.execute_reply":"2022-08-13T21:29:12.103687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#scores = cross_val_score(clf, X_train_test, y_train_test, cv=10,scoring='neg_mean_squared_error')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:29:12.106119Z","iopub.execute_input":"2022-08-13T21:29:12.106385Z","iopub.status.idle":"2022-08-13T21:29:12.109535Z","shell.execute_reply.started":"2022-08-13T21:29:12.106359Z","shell.execute_reply":"2022-08-13T21:29:12.108920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''scores\nprint('scores:',np.sqrt(-scores))\nprint('mean:',np.sqrt(-scores).mean())\nprint('std:',np.sqrt(-scores).std())'''","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:29:12.110325Z","iopub.execute_input":"2022-08-13T21:29:12.111020Z","iopub.status.idle":"2022-08-13T21:29:12.124974Z","shell.execute_reply.started":"2022-08-13T21:29:12.110993Z","shell.execute_reply":"2022-08-13T21:29:12.123767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Optimizing hyperparams","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV, RandomizedSearchCV","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:29:12.126786Z","iopub.execute_input":"2022-08-13T21:29:12.127705Z","iopub.status.idle":"2022-08-13T21:29:12.134360Z","shell.execute_reply.started":"2022-08-13T21:29:12.127677Z","shell.execute_reply":"2022-08-13T21:29:12.133407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''parameters = {'C':[5,10,15],'decision_function_shape':('ovo', 'ovr')}\nsvc = SVC()\ncv = GridSearchCV(svc, parameters)\ncv.fit(X_train_train,y_train_train)'''","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:29:12.135622Z","iopub.execute_input":"2022-08-13T21:29:12.136429Z","iopub.status.idle":"2022-08-13T21:29:12.150230Z","shell.execute_reply.started":"2022-08-13T21:29:12.136394Z","shell.execute_reply":"2022-08-13T21:29:12.149129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#cv.best_estimator_","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:29:12.151421Z","iopub.execute_input":"2022-08-13T21:29:12.152479Z","iopub.status.idle":"2022-08-13T21:29:12.157475Z","shell.execute_reply.started":"2022-08-13T21:29:12.152451Z","shell.execute_reply":"2022-08-13T21:29:12.156913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Testing model","metadata":{}},{"cell_type":"code","source":"y_pred = clf.predict(df_test/255)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:29:12.160995Z","iopub.execute_input":"2022-08-13T21:29:12.161524Z","iopub.status.idle":"2022-08-13T21:29:13.579670Z","shell.execute_reply.started":"2022-08-13T21:29:12.161501Z","shell.execute_reply":"2022-08-13T21:29:13.578959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame()\nsubmission[\"Label\"] = y_pred\nsubmission.index.name = \"ImageId\"\nsubmission.index = submission.index + 1\nsubmission.to_csv(\"./submission.csv\")\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-08-13T21:29:13.580666Z","iopub.execute_input":"2022-08-13T21:29:13.580957Z","iopub.status.idle":"2022-08-13T21:29:13.627925Z","shell.execute_reply.started":"2022-08-13T21:29:13.580932Z","shell.execute_reply":"2022-08-13T21:29:13.626934Z"},"trusted":true},"execution_count":null,"outputs":[]}]}