{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## **AIM :- TO BUILD A MODEL USING KNN CLASSIFIER ON TITANIC DATASET** ##","metadata":{}},{"cell_type":"markdown","source":"### **IMPORTING NECESSARY LIBRARIES FOR EDA** ##","metadata":{}},{"cell_type":"code","source":"\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport csv\n\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport pandas as pd\nimport csv\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-16T23:39:10.220112Z","iopub.execute_input":"2022-07-16T23:39:10.221169Z","iopub.status.idle":"2022-07-16T23:39:10.231813Z","shell.execute_reply.started":"2022-07-16T23:39:10.221129Z","shell.execute_reply":"2022-07-16T23:39:10.230938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **importing training data** ###","metadata":{}},{"cell_type":"code","source":"tit = pd.read_csv('../input/titanic/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:39:13.295050Z","iopub.execute_input":"2022-07-16T23:39:13.295441Z","iopub.status.idle":"2022-07-16T23:39:13.307891Z","shell.execute_reply.started":"2022-07-16T23:39:13.295407Z","shell.execute_reply":"2022-07-16T23:39:13.306781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tit.info()\ntit.shape\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:39:14.942776Z","iopub.execute_input":"2022-07-16T23:39:14.943163Z","iopub.status.idle":"2022-07-16T23:39:14.963376Z","shell.execute_reply.started":"2022-07-16T23:39:14.943132Z","shell.execute_reply":"2022-07-16T23:39:14.962601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Finding out the nan values,present in the data** ##","metadata":{}},{"cell_type":"code","source":"feanan = list(tit.columns.values)\n\nfor featu in feanan:\n    print (featu,\": \",sum(pd.isnull(tit[featu])))","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:39:17.334565Z","iopub.execute_input":"2022-07-16T23:39:17.335203Z","iopub.status.idle":"2022-07-16T23:39:17.346114Z","shell.execute_reply.started":"2022-07-16T23:39:17.335165Z","shell.execute_reply":"2022-07-16T23:39:17.345033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(tit.isnull(),yticklabels=False,cbar=False,cmap='viridis')","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:39:21.237201Z","iopub.execute_input":"2022-07-16T23:39:21.237571Z","iopub.status.idle":"2022-07-16T23:39:21.427897Z","shell.execute_reply.started":"2022-07-16T23:39:21.237538Z","shell.execute_reply":"2022-07-16T23:39:21.426924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### replacing nan values \nfor numerical values  I have replaced it with mean \nand for Categorical data I have replaced it with most frequent occuring value\n","metadata":{}},{"cell_type":"code","source":"df=tit\nfeature_cols = [col for col in df.columns ]\n\n\ncat_cols = [col for col in feature_cols if df[col].dtype == 'O']\ncont_cols = [col for col in feature_cols if col not in cat_cols]\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.impute import SimpleImputer\n\nfrom sklearn.impute import KNNImputer\n\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\n\ndf_simple_imputer = df.copy()\nimputer = SimpleImputer(strategy='mean')\n\ndf_simple_imputer[cont_cols] = imputer.fit_transform(df_simple_imputer[cont_cols])\n\n\nimputer = SimpleImputer(strategy='most_frequent')\n\ndf_simple_imputer[cat_cols] = imputer.fit_transform(df_simple_imputer[cat_cols])","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:39:25.871551Z","iopub.execute_input":"2022-07-16T23:39:25.871945Z","iopub.status.idle":"2022-07-16T23:39:25.893912Z","shell.execute_reply.started":"2022-07-16T23:39:25.871913Z","shell.execute_reply":"2022-07-16T23:39:25.892391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### confirming if all nan values are handled\nsns.heatmap(df_simple_imputer.isnull(), cmap='Blues', cbar=False, yticklabels=False, xticklabels=df.columns);","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:39:27.948443Z","iopub.execute_input":"2022-07-16T23:39:27.948864Z","iopub.status.idle":"2022-07-16T23:39:28.111885Z","shell.execute_reply.started":"2022-07-16T23:39:27.948828Z","shell.execute_reply":"2022-07-16T23:39:28.110574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=df_simple_imputer\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:39:30.224114Z","iopub.execute_input":"2022-07-16T23:39:30.224524Z","iopub.status.idle":"2022-07-16T23:39:30.252298Z","shell.execute_reply.started":"2022-07-16T23:39:30.224474Z","shell.execute_reply":"2022-07-16T23:39:30.251508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Family'] = df['SibSp'] + df['Parch'] + 1\ndf = df.drop('SibSp', axis=1,)\ndf = df.drop('Parch', axis=1,)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:39:34.385952Z","iopub.execute_input":"2022-07-16T23:39:34.386322Z","iopub.status.idle":"2022-07-16T23:39:34.396081Z","shell.execute_reply.started":"2022-07-16T23:39:34.386291Z","shell.execute_reply":"2022-07-16T23:39:34.394963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:39:36.273373Z","iopub.execute_input":"2022-07-16T23:39:36.274373Z","iopub.status.idle":"2022-07-16T23:39:36.302774Z","shell.execute_reply.started":"2022-07-16T23:39:36.274319Z","shell.execute_reply":"2022-07-16T23:39:36.301648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **EDA is done, now we start making knn model for our dataset** ##","metadata":{}},{"cell_type":"code","source":"\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.neighbors import NearestNeighbors\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn import metrics\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import precision_recall_fscore_support\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import KFold\nfrom collections import Counter\nfrom sklearn.model_selection import cross_validate","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:39:40.572631Z","iopub.execute_input":"2022-07-16T23:39:40.573204Z","iopub.status.idle":"2022-07-16T23:39:40.578855Z","shell.execute_reply.started":"2022-07-16T23:39:40.573170Z","shell.execute_reply":"2022-07-16T23:39:40.577778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### replacing catogorical values with numerical\ndf.Sex.replace(['male', 'female'], [1,0], inplace=True)\ndf.Embarked.replace(['S', 'C', 'Q'], [1, 2, 3], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:39:42.718285Z","iopub.execute_input":"2022-07-16T23:39:42.718675Z","iopub.status.idle":"2022-07-16T23:39:42.728300Z","shell.execute_reply.started":"2022-07-16T23:39:42.718640Z","shell.execute_reply":"2022-07-16T23:39:42.727446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.array(df.filter(['Pclass','Sex','Embarked','Family','Age'], axis=1))\ny = np.array(df.filter(['Survived'], axis=1))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:39:44.773018Z","iopub.execute_input":"2022-07-16T23:39:44.773393Z","iopub.status.idle":"2022-07-16T23:39:44.782598Z","shell.execute_reply.started":"2022-07-16T23:39:44.773362Z","shell.execute_reply":"2022-07-16T23:39:44.781202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Now,splitting the train dataset , and doing cross validation ","metadata":{}},{"cell_type":"code","source":"X_1, X_test, y_1, y_test = train_test_split(X,y, test_size=0.3)\nX_tr, X_cv, y_tr, y_cv = train_test_split(X_1, y_1, test_size=0.3)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:39:47.372656Z","iopub.execute_input":"2022-07-16T23:39:47.373033Z","iopub.status.idle":"2022-07-16T23:39:47.379912Z","shell.execute_reply.started":"2022-07-16T23:39:47.373002Z","shell.execute_reply":"2022-07-16T23:39:47.379040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### calculating for which k does the model works best ###","metadata":{}},{"cell_type":"code","source":"ans = []\nfor i in range(1,30,2):\n    knn = KNeighborsClassifier(n_neighbors = i)\n    knn.fit(X_tr, y_tr.ravel())\n    pred = knn.predict(X_cv)\n    acc = accuracy_score(y_cv, pred, normalize=True) * float(100)\n    ans.append(acc)\n    print('\\n CV accuracy for k=%d is %d'%(i,acc))","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:39:48.703748Z","iopub.execute_input":"2022-07-16T23:39:48.704637Z","iopub.status.idle":"2022-07-16T23:39:48.841336Z","shell.execute_reply.started":"2022-07-16T23:39:48.704601Z","shell.execute_reply":"2022-07-16T23:39:48.840168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"opk = ans.index(max(ans))\nprint(opk)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:41:24.392122Z","iopub.execute_input":"2022-07-16T23:41:24.392488Z","iopub.status.idle":"2022-07-16T23:41:24.398924Z","shell.execute_reply.started":"2022-07-16T23:41:24.392456Z","shell.execute_reply":"2022-07-16T23:41:24.397675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Importing test data set and doing same step as performed on train dataset ###","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv('../input/titanic/test.csv')\ndf_test = df_test.drop('Name', axis=1,)\ndf_test = df_test.drop('Ticket', axis=1,)\ndf_test = df_test.drop('Fare', axis=1,)\ndf_test = df_test.drop('Cabin', axis=1,)\ndf_test['Family'] = df_test['SibSp'] + df_test['Parch'] + 1\ndf_test = df_test.drop('SibSp', axis=1,)\ndf_test = df_test.drop('Parch', axis=1,)\ndft=df_test","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:41:37.957398Z","iopub.execute_input":"2022-07-16T23:41:37.957773Z","iopub.status.idle":"2022-07-16T23:41:37.978120Z","shell.execute_reply.started":"2022-07-16T23:41:37.957741Z","shell.execute_reply":"2022-07-16T23:41:37.976918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_cols = [col for col in dft.columns ]\n\n\ncat_cols = [col for col in feature_cols if dft[col].dtype == 'O']\ncont_cols = [col for col in feature_cols if col not in cat_cols]\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.impute import SimpleImputer\n\nfrom sklearn.impute import KNNImputer\n\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\n\ndf_simple_imputer = dft.copy()\nimputer = SimpleImputer(strategy='mean')\n\ndf_simple_imputer[cont_cols] = imputer.fit_transform(df_simple_imputer[cont_cols])\n\n\nimputer = SimpleImputer(strategy='most_frequent')\n\ndf_simple_imputer[cat_cols] = imputer.fit_transform(df_simple_imputer[cat_cols])","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:41:41.672015Z","iopub.execute_input":"2022-07-16T23:41:41.672382Z","iopub.status.idle":"2022-07-16T23:41:41.692162Z","shell.execute_reply.started":"2022-07-16T23:41:41.672350Z","shell.execute_reply":"2022-07-16T23:41:41.690964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(df_simple_imputer.isnull(), cmap='Blues', cbar=False, yticklabels=False, xticklabels=dft.columns);","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:41:44.033407Z","iopub.execute_input":"2022-07-16T23:41:44.033808Z","iopub.status.idle":"2022-07-16T23:41:44.135809Z","shell.execute_reply.started":"2022-07-16T23:41:44.033774Z","shell.execute_reply":"2022-07-16T23:41:44.134514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dft=df_simple_imputer","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:41:46.844294Z","iopub.execute_input":"2022-07-16T23:41:46.844700Z","iopub.status.idle":"2022-07-16T23:41:46.849170Z","shell.execute_reply.started":"2022-07-16T23:41:46.844663Z","shell.execute_reply":"2022-07-16T23:41:46.848198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dft.Embarked.replace(['S', 'C', 'Q'], [1, 2, 3], inplace=True)\ndft.Sex.replace(['male', 'female'], [1,0], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:41:49.089179Z","iopub.execute_input":"2022-07-16T23:41:49.090247Z","iopub.status.idle":"2022-07-16T23:41:49.098932Z","shell.execute_reply.started":"2022-07-16T23:41:49.090209Z","shell.execute_reply":"2022-07-16T23:41:49.097789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### predicting value for test data set ###","metadata":{}},{"cell_type":"code","source":"X_test = np.array(dft.filter(['Pclass','Sex','Embarked','Family','Age'], axis=1))\nknn = KNeighborsClassifier(n_neighbors = opk)\nknn.fit(X_tr, y_tr.ravel())\npred = knn.predict(X_test)\nprint(pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:41:56.978830Z","iopub.execute_input":"2022-07-16T23:41:56.979248Z","iopub.status.idle":"2022-07-16T23:41:57.010446Z","shell.execute_reply.started":"2022-07-16T23:41:56.979214Z","shell.execute_reply":"2022-07-16T23:41:57.009259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dft['Survived'] = pd.Series(pred, index=dft.index)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:42:00.642268Z","iopub.execute_input":"2022-07-16T23:42:00.642702Z","iopub.status.idle":"2022-07-16T23:42:00.649026Z","shell.execute_reply.started":"2022-07-16T23:42:00.642667Z","shell.execute_reply":"2022-07-16T23:42:00.647885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"answer = dft.filter(['PassengerId','Survived'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:42:02.654839Z","iopub.execute_input":"2022-07-16T23:42:02.655238Z","iopub.status.idle":"2022-07-16T23:42:02.663014Z","shell.execute_reply.started":"2022-07-16T23:42:02.655204Z","shell.execute_reply":"2022-07-16T23:42:02.661861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Finally converting file for submission ###","metadata":{}},{"cell_type":"code","source":"answer.to_csv(\"jujutsu.csv\", encoding='utf-8')","metadata":{"execution":{"iopub.status.busy":"2022-07-16T23:42:04.855912Z","iopub.execute_input":"2022-07-16T23:42:04.856315Z","iopub.status.idle":"2022-07-16T23:42:04.863686Z","shell.execute_reply.started":"2022-07-16T23:42:04.856281Z","shell.execute_reply":"2022-07-16T23:42:04.862725Z"},"trusted":true},"execution_count":null,"outputs":[]}]}