{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T05:49:12.033790Z","iopub.execute_input":"2022-08-10T05:49:12.034239Z","iopub.status.idle":"2022-08-10T05:49:12.044519Z","shell.execute_reply.started":"2022-08-10T05:49:12.034207Z","shell.execute_reply":"2022-08-10T05:49:12.043036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.neighbors import KNeighborsClassifier as KNN\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.preprocessing import StandardScaler","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:12.081445Z","iopub.execute_input":"2022-08-10T05:49:12.082957Z","iopub.status.idle":"2022-08-10T05:49:12.089663Z","shell.execute_reply.started":"2022-08-10T05:49:12.082892Z","shell.execute_reply":"2022-08-10T05:49:12.088802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/titanic/train.csv')\ndf.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:12.101472Z","iopub.execute_input":"2022-08-10T05:49:12.102230Z","iopub.status.idle":"2022-08-10T05:49:12.120832Z","shell.execute_reply.started":"2022-08-10T05:49:12.102164Z","shell.execute_reply":"2022-08-10T05:49:12.119792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:12.127965Z","iopub.execute_input":"2022-08-10T05:49:12.128601Z","iopub.status.idle":"2022-08-10T05:49:12.140955Z","shell.execute_reply.started":"2022-08-10T05:49:12.128566Z","shell.execute_reply":"2022-08-10T05:49:12.139996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(columns=[ 'PassengerId','Name', 'Cabin', 'Ticket'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:12.153486Z","iopub.execute_input":"2022-08-10T05:49:12.154583Z","iopub.status.idle":"2022-08-10T05:49:12.161218Z","shell.execute_reply.started":"2022-08-10T05:49:12.154540Z","shell.execute_reply":"2022-08-10T05:49:12.160016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['Embarked'].isna() == True]","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:12.185156Z","iopub.execute_input":"2022-08-10T05:49:12.186436Z","iopub.status.idle":"2022-08-10T05:49:12.202004Z","shell.execute_reply.started":"2022-08-10T05:49:12.186382Z","shell.execute_reply":"2022-08-10T05:49:12.201064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[['Embarked']] = SimpleImputer(strategy='most_frequent').fit_transform(df[['Embarked']])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:12.204036Z","iopub.execute_input":"2022-08-10T05:49:12.204430Z","iopub.status.idle":"2022-08-10T05:49:12.218781Z","shell.execute_reply.started":"2022-08-10T05:49:12.204396Z","shell.execute_reply":"2022-08-10T05:49:12.217323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Sex'] = LabelEncoder().fit_transform(df['Sex'])\ndf['Embarked'] = LabelEncoder().fit_transform(df['Embarked'])\ndf_guide = {\n    'Sex':{'male':0,'female':1},\n    'Embarked':{'S':0,'C':1,'Q':2}\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:12.220854Z","iopub.execute_input":"2022-08-10T05:49:12.221827Z","iopub.status.idle":"2022-08-10T05:49:12.229675Z","shell.execute_reply.started":"2022-08-10T05:49:12.221790Z","shell.execute_reply":"2022-08-10T05:49:12.228576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8,8))\nsns.heatmap(df.corr(),cmap='coolwarm',annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:12.235630Z","iopub.execute_input":"2022-08-10T05:49:12.236033Z","iopub.status.idle":"2022-08-10T05:49:12.745485Z","shell.execute_reply.started":"2022-08-10T05:49:12.236000Z","shell.execute_reply":"2022-08-10T05:49:12.744289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The Age column hasen't significant impact on Survived columns I can drop it but I am going to fill Age column based on other column for more practise","metadata":{}},{"cell_type":"code","source":"df_age = df[df['Age'].isna() == False]\nplt.figure(figsize=(8,8))\nsns.heatmap(df_age.corr(),cmap='coolwarm',annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:12.747818Z","iopub.execute_input":"2022-08-10T05:49:12.748776Z","iopub.status.idle":"2022-08-10T05:49:13.490080Z","shell.execute_reply.started":"2022-08-10T05:49:12.748739Z","shell.execute_reply":"2022-08-10T05:49:13.488956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_age['Age'][df_age['Pclass'] == 1].mean(),df_age['Age'][df_age['Pclass'] == 2].mean(),df_age['Age'][df_age['Pclass'] == 3].mean()\n\n# there is a meaningfull diffrence between maen age of each group","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:13.491632Z","iopub.execute_input":"2022-08-10T05:49:13.491991Z","iopub.status.idle":"2022-08-10T05:49:13.504588Z","shell.execute_reply.started":"2022-08-10T05:49:13.491959Z","shell.execute_reply":"2022-08-10T05:49:13.503156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8,8))\nsns.stripplot(x='SibSp',y='Age',data=df_age)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:13.507100Z","iopub.execute_input":"2022-08-10T05:49:13.507894Z","iopub.status.idle":"2022-08-10T05:49:13.748014Z","shell.execute_reply.started":"2022-08-10T05:49:13.507857Z","shell.execute_reply":"2022-08-10T05:49:13.746989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_age['Age'][df_age['SibSp'] == 3].mean(),df_age['Age'][df_age['SibSp'] == 4].mean(),df_age['Age'][df_age['SibSp'] == 5].mean()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:13.749304Z","iopub.execute_input":"2022-08-10T05:49:13.749615Z","iopub.status.idle":"2022-08-10T05:49:13.760411Z","shell.execute_reply.started":"2022-08-10T05:49:13.749587Z","shell.execute_reply":"2022-08-10T05:49:13.759303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age_nan_list = df['Age'][df['Age'].isna() == True]\nfor i in age_nan_list.index:\n    if df.at[i,'SibSp'] == 3:\n        df.at[i,'Age'] = 13.9\n    elif df.at[i,'SibSp'] == 4:\n        df.at[i,'Age'] = 7\n    elif df.at[i,'SibSp'] == 5:\n        df.at[i,'Age'] = 10.2\n    elif df.at[i,'Pclass'] == 1:\n        df.at[i,'Age'] = 38.2\n    elif df.at[i,'Pclass'] == 2:\n        df.at[i,'Age'] = 29.8\n    else:\n        df.at[i,'Age'] = 25.1","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:13.761864Z","iopub.execute_input":"2022-08-10T05:49:13.762684Z","iopub.status.idle":"2022-08-10T05:49:13.780240Z","shell.execute_reply.started":"2022-08-10T05:49:13.762649Z","shell.execute_reply":"2022-08-10T05:49:13.779094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['SibSp','Age','Parch'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:13.781607Z","iopub.execute_input":"2022-08-10T05:49:13.781915Z","iopub.status.idle":"2022-08-10T05:49:13.787937Z","shell.execute_reply.started":"2022-08-10T05:49:13.781889Z","shell.execute_reply":"2022-08-10T05:49:13.786806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df.drop(columns='Survived')\nY = df['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:13.789593Z","iopub.execute_input":"2022-08-10T05:49:13.790295Z","iopub.status.idle":"2022-08-10T05:49:13.799862Z","shell.execute_reply.started":"2022-08-10T05:49:13.790250Z","shell.execute_reply":"2022-08-10T05:49:13.798995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X[['Fare']] = StandardScaler(with_mean=False).fit_transform(X[['Fare']])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:13.801344Z","iopub.execute_input":"2022-08-10T05:49:13.801827Z","iopub.status.idle":"2022-08-10T05:49:13.816744Z","shell.execute_reply.started":"2022-08-10T05:49:13.801797Z","shell.execute_reply":"2022-08-10T05:49:13.815265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_test, y_train, y_test = train_test_split(X,Y,test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:13.820023Z","iopub.execute_input":"2022-08-10T05:49:13.820420Z","iopub.status.idle":"2022-08-10T05:49:13.829440Z","shell.execute_reply.started":"2022-08-10T05:49:13.820387Z","shell.execute_reply":"2022-08-10T05:49:13.828269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = KNN(n_neighbors=1)\nclf.fit(x_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:13.830860Z","iopub.execute_input":"2022-08-10T05:49:13.831284Z","iopub.status.idle":"2022-08-10T05:49:13.847049Z","shell.execute_reply.started":"2022-08-10T05:49:13.831252Z","shell.execute_reply":"2022-08-10T05:49:13.845615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = clf.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:13.848167Z","iopub.execute_input":"2022-08-10T05:49:13.849489Z","iopub.status.idle":"2022-08-10T05:49:13.863487Z","shell.execute_reply.started":"2022-08-10T05:49:13.849436Z","shell.execute_reply":"2022-08-10T05:49:13.862456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"100 * accuracy_score(y_true=y_test, y_pred=y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:49:13.864841Z","iopub.execute_input":"2022-08-10T05:49:13.865177Z","iopub.status.idle":"2022-08-10T05:49:13.875394Z","shell.execute_reply.started":"2022-08-10T05:49:13.865147Z","shell.execute_reply":"2022-08-10T05:49:13.874067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/titanic/test.csv')\ndf_test['Sex'] = LabelEncoder().fit_transform(df_test['Sex'])\ndf_test['Embarked'] = LabelEncoder().fit_transform(df_test['Embarked'])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:50:09.291583Z","iopub.execute_input":"2022-08-10T05:50:09.292717Z","iopub.status.idle":"2022-08-10T05:50:09.309521Z","shell.execute_reply.started":"2022-08-10T05:50:09.292674Z","shell.execute_reply":"2022-08-10T05:50:09.308348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X2 = df_test.drop(columns=['PassengerId','SibSp','Age','Parch','Name', 'Cabin', 'Ticket'])\nX2[['Fare']] = SimpleImputer(strategy='mean').fit_transform(X2[['Fare']])\nX2.isna().sum()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:50:31.065772Z","iopub.execute_input":"2022-08-10T05:50:31.066694Z","iopub.status.idle":"2022-08-10T05:50:31.088058Z","shell.execute_reply.started":"2022-08-10T05:50:31.066654Z","shell.execute_reply":"2022-08-10T05:50:31.086446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X2[['Fare']] = StandardScaler(with_mean=False).fit_transform(X2[['Fare']])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:50:54.960740Z","iopub.execute_input":"2022-08-10T05:50:54.961158Z","iopub.status.idle":"2022-08-10T05:50:54.974906Z","shell.execute_reply.started":"2022-08-10T05:50:54.961127Z","shell.execute_reply":"2022-08-10T05:50:54.973416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred2 = clf.predict(X2)\ny_pred2","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:51:25.900610Z","iopub.execute_input":"2022-08-10T05:51:25.901047Z","iopub.status.idle":"2022-08-10T05:51:25.927325Z","shell.execute_reply.started":"2022-08-10T05:51:25.901015Z","shell.execute_reply":"2022-08-10T05:51:25.926000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'PassengerId':df_test.PassengerId,'Survived':y_pred2})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T05:51:47.403962Z","iopub.execute_input":"2022-08-10T05:51:47.404418Z","iopub.status.idle":"2022-08-10T05:51:47.414056Z","shell.execute_reply.started":"2022-08-10T05:51:47.404383Z","shell.execute_reply":"2022-08-10T05:51:47.413094Z"},"trusted":true},"execution_count":null,"outputs":[]}]}