{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T12:19:18.401002Z","iopub.execute_input":"2022-07-28T12:19:18.401858Z","iopub.status.idle":"2022-07-28T12:19:19.866039Z","shell.execute_reply.started":"2022-07-28T12:19:18.401740Z","shell.execute_reply":"2022-07-28T12:19:19.864910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(\"../input/titanic/train.csv\")\ntest = pd.read_csv('../input/titanic/test.csv')\ndata","metadata":{"execution":{"iopub.status.busy":"2022-07-28T12:54:32.618146Z","iopub.execute_input":"2022-07-28T12:54:32.618554Z","iopub.status.idle":"2022-07-28T12:54:32.650957Z","shell.execute_reply.started":"2022-07-28T12:54:32.618522Z","shell.execute_reply":"2022-07-28T12:54:32.650109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the columns with null to impute them later\ndata.isnull().sum(axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T12:54:33.483490Z","iopub.execute_input":"2022-07-28T12:54:33.484056Z","iopub.status.idle":"2022-07-28T12:54:33.492301Z","shell.execute_reply.started":"2022-07-28T12:54:33.484023Z","shell.execute_reply":"2022-07-28T12:54:33.491409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# handling NaN values\ndata['Age'].fillna(data['Age'].mean(), inplace=True)\n\n#there are values in Cabin column of the form (Cabin)(Number) where only the cabin alone gives information about the social status of the passenger.\ndata['Cabin'] = data['Cabin'].str.replace(pat=r'(?P<Cabin>\\D+)(?P<Number>\\d+)\\s*', repl=lambda x: x.group('Cabin'), regex=True)\ndata['Cabin'] = data['Cabin'].fillna('NA')\ndata['Embarked'] = data['Embarked'].fillna('NA')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T12:54:34.224370Z","iopub.execute_input":"2022-07-28T12:54:34.224930Z","iopub.status.idle":"2022-07-28T12:54:34.233657Z","shell.execute_reply.started":"2022-07-28T12:54:34.224900Z","shell.execute_reply":"2022-07-28T12:54:34.232521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#encoding columns with meaningful entries (like sex, cabin, etc.,)\ndata['Sex'] = data['Sex'].apply(lambda x: 0 if x=='female' else 1)\n\n#Pclass even though it is a number, it determines the class of the passenger, ie, people with Pclass 2 are second class. So they form a category by themselves.\n#Or rather, each of the value has a pattern of it's own.\noh_cols = ['Cabin', 'Embarked', 'Pclass'] #One hot encode the columns with decent information\n\nohc = OneHotEncoder(handle_unknown='ignore', sparse=False)\ndata_ohc = pd.DataFrame(ohc.fit_transform(data[oh_cols]))\ndata = pd.concat([data, data_ohc], axis=1)\n\ndata.drop(columns=['PassengerId', 'Name', 'Ticket', 'Cabin', 'Embarked', 'Pclass'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T12:54:35.908127Z","iopub.execute_input":"2022-07-28T12:54:35.909076Z","iopub.status.idle":"2022-07-28T12:54:35.925815Z","shell.execute_reply.started":"2022-07-28T12:54:35.909023Z","shell.execute_reply":"2022-07-28T12:54:35.924874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fitting a model to the data\nfrom sklearn.linear_model import LogisticRegression\nrf = LogisticRegression(max_iter=1000).fit(data.drop(columns=['Survived']), data['Survived'])","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:19:46.739911Z","iopub.execute_input":"2022-07-28T13:19:46.740316Z","iopub.status.idle":"2022-07-28T13:19:47.031327Z","shell.execute_reply.started":"2022-07-28T13:19:46.740256Z","shell.execute_reply":"2022-07-28T13:19:47.030034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#doing the same preprocessing for test set (could have used Pipelines, but I prefer it this way.)\ntest['Age'].fillna(test['Age'].mean(), inplace=True)\ntest['Cabin'] = test['Cabin'].str.replace(pat=r'(?P<Cabin>\\D+)(?P<Number>\\d+)\\s*', repl=lambda x: x.group('Cabin'), regex=True)\ntest['Cabin'] = test['Cabin'].fillna('NA')\ntest['Embarked'] = test['Embarked'].fillna('NA')\ntest['Fare'].fillna(test.Fare.mean(), inplace=True)\n\ntest['Sex'] = test['Sex'].apply(lambda x: 0 if x=='female' else 1)\n\ntest_ohc = pd.DataFrame(ohc.transform(test[oh_cols]))\ntest = pd.concat([test, test_ohc], axis=1)\n\nids = test['PassengerId']\ntest.drop(columns=['PassengerId', 'Name', 'Ticket', 'Cabin', 'Embarked', 'Pclass'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T12:55:07.594208Z","iopub.execute_input":"2022-07-28T12:55:07.594651Z","iopub.status.idle":"2022-07-28T12:55:07.615643Z","shell.execute_reply.started":"2022-07-28T12:55:07.594618Z","shell.execute_reply":"2022-07-28T12:55:07.614579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = pd.Series(rf.predict(test), index=ids, name='Survived')\npred.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:20:05.508097Z","iopub.execute_input":"2022-07-28T13:20:05.508511Z","iopub.status.idle":"2022-07-28T13:20:05.522228Z","shell.execute_reply.started":"2022-07-28T13:20:05.508473Z","shell.execute_reply":"2022-07-28T13:20:05.520903Z"},"trusted":true},"execution_count":null,"outputs":[]}]}