{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Kaggle Titanic Submission 6 - new Family_group feature\nversion2 with model trained on full train data\n\nAdding improvement one by one to see if the score gets improved \n\nfor Submission 6, I will combine SibSp and Parch features and make new Family_group feature and see if the prediction accuracy improves","metadata":{"id":"tL2KiFXpT8mh"}},{"cell_type":"code","source":"# Loading libraries\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"executionInfo":{"elapsed":270,"status":"ok","timestamp":1651975635254,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"1324dce0","execution":{"iopub.status.busy":"2022-05-30T08:58:57.062049Z","iopub.execute_input":"2022-05-30T08:58:57.062676Z","iopub.status.idle":"2022-05-30T08:58:57.080519Z","shell.execute_reply.started":"2022-05-30T08:58:57.062635Z","shell.execute_reply":"2022-05-30T08:58:57.079287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# delete submission.csv if it is already in Output folder\n\n# os.remove('submission.csv')","metadata":{"executionInfo":{"elapsed":269,"status":"ok","timestamp":1651975635258,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"0d12d5ef","execution":{"iopub.status.busy":"2022-05-30T08:58:57.101735Z","iopub.execute_input":"2022-05-30T08:58:57.102153Z","iopub.status.idle":"2022-05-30T08:58:57.105098Z","shell.execute_reply.started":"2022-05-30T08:58:57.102125Z","shell.execute_reply":"2022-05-30T08:58:57.104497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading the Train dataset (train.csv)\n\ntrain_data = pd.read_csv('/kaggle/input/titanic/train.csv')\ntrain_data.head()","metadata":{"executionInfo":{"elapsed":270,"status":"ok","timestamp":1651975635261,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"45da6b04","execution":{"iopub.status.busy":"2022-05-30T08:58:57.182004Z","iopub.execute_input":"2022-05-30T08:58:57.182295Z","iopub.status.idle":"2022-05-30T08:58:57.203807Z","shell.execute_reply.started":"2022-05-30T08:58:57.182264Z","shell.execute_reply":"2022-05-30T08:58:57.202778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading the Test dataset (test.csv)\n\ntest_data = pd.read_csv('/kaggle/input/titanic/test.csv')\ntest_data.head()","metadata":{"executionInfo":{"elapsed":265,"status":"ok","timestamp":1651975635264,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"3a2b4614","execution":{"iopub.status.busy":"2022-05-30T08:58:57.281041Z","iopub.execute_input":"2022-05-30T08:58:57.281363Z","iopub.status.idle":"2022-05-30T08:58:57.304061Z","shell.execute_reply.started":"2022-05-30T08:58:57.281324Z","shell.execute_reply":"2022-05-30T08:58:57.303162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concat train_data and test_data\n\n# before .concat() drop Survived feature from train dataset \n# as test dataset do not have this feature\n\n# join with train dataset later on\nSurvived_data = train_data['Survived']\ntrain_data = train_data.drop('Survived', axis=1)\n\ndf = pd.concat([train_data, test_data], sort=False, ignore_index=True)\n\ndf2 = df.copy()\n\n# check ignore_index is working and index is continuous in whole dataset\ndf.tail()","metadata":{"executionInfo":{"elapsed":264,"status":"ok","timestamp":1651975635266,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"db1cfba7","outputId":"34590255-5ffe-4bbe-b53d-e1d59c3d32fd","execution":{"iopub.status.busy":"2022-05-30T08:58:57.381067Z","iopub.execute_input":"2022-05-30T08:58:57.381374Z","iopub.status.idle":"2022-05-30T08:58:57.406321Z","shell.execute_reply.started":"2022-05-30T08:58:57.381343Z","shell.execute_reply":"2022-05-30T08:58:57.404292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Exploratory Data Analysis (EDA) to understand the dataset","metadata":{"id":"4556beef"}},{"cell_type":"code","source":"df.shape","metadata":{"executionInfo":{"elapsed":259,"status":"ok","timestamp":1651975635270,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"1c07998f","outputId":"049a410b-7ec7-4bac-c40a-86b90eeb77ca","execution":{"iopub.status.busy":"2022-05-30T08:58:57.420996Z","iopub.execute_input":"2022-05-30T08:58:57.422205Z","iopub.status.idle":"2022-05-30T08:58:57.427844Z","shell.execute_reply.started":"2022-05-30T08:58:57.422147Z","shell.execute_reply":"2022-05-30T08:58:57.427191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"executionInfo":{"elapsed":241,"status":"ok","timestamp":1651975635275,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"b9206213","outputId":"ff53ed91-801a-489d-85ab-d00a0ceb91bd","execution":{"iopub.status.busy":"2022-05-30T08:58:57.501258Z","iopub.execute_input":"2022-05-30T08:58:57.502199Z","iopub.status.idle":"2022-05-30T08:58:57.519998Z","shell.execute_reply.started":"2022-05-30T08:58:57.502143Z","shell.execute_reply":"2022-05-30T08:58:57.518806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"so train and test datasets combined, now 1309 rows","metadata":{"id":"6ffe3ab6"}},{"cell_type":"code","source":"df.describe().T","metadata":{"executionInfo":{"elapsed":234,"status":"ok","timestamp":1651975635278,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"7089a0c0","outputId":"7bb5b94c-0ef7-4e4d-c6cc-d1e345cdaf40","execution":{"iopub.status.busy":"2022-05-30T08:58:57.582082Z","iopub.execute_input":"2022-05-30T08:58:57.582906Z","iopub.status.idle":"2022-05-30T08:58:57.619808Z","shell.execute_reply.started":"2022-05-30T08:58:57.58286Z","shell.execute_reply":"2022-05-30T08:58:57.619118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Data Pre-processing - preparing the data for modelling","metadata":{"id":"8a435c79"}},{"cell_type":"markdown","source":"Omit features that are not important \n\nremoving columns which I think are not important as input features to predict the Survived feature based on EDA earlier, \n\nI may adjust this at later stage when I evaluate the model accuracy","metadata":{"id":"94f7d5d6"}},{"cell_type":"code","source":"columns_to_drop = ['PassengerId', 'Name','Ticket', 'Cabin', 'Embarked']\n\ndf = df.drop(columns_to_drop, axis=1)\n\ndf.head()","metadata":{"executionInfo":{"elapsed":230,"status":"ok","timestamp":1651975635281,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"84661855","outputId":"70fd4058-1fc0-48a6-b4db-9fbc1b82a1dc","execution":{"iopub.status.busy":"2022-05-30T08:58:57.665652Z","iopub.execute_input":"2022-05-30T08:58:57.666116Z","iopub.status.idle":"2022-05-30T08:58:57.68183Z","shell.execute_reply.started":"2022-05-30T08:58:57.666085Z","shell.execute_reply":"2022-05-30T08:58:57.681153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Non numeric feature to numeric feature","metadata":{"id":"21d339d9"}},{"cell_type":"code","source":"df.info()","metadata":{"executionInfo":{"elapsed":221,"status":"ok","timestamp":1651975635283,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"QHolZnywdaoM","outputId":"3865175c-7c93-4e34-acca-34ce4cf17f68","execution":{"iopub.status.busy":"2022-05-30T08:58:57.721656Z","iopub.execute_input":"2022-05-30T08:58:57.722203Z","iopub.status.idle":"2022-05-30T08:58:57.735917Z","shell.execute_reply.started":"2022-05-30T08:58:57.722167Z","shell.execute_reply":"2022-05-30T08:58:57.735281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Sex feature needs to be turned into a numerical feature","metadata":{"id":"36f3c8d5"}},{"cell_type":"code","source":"# There is no missing values in Sex column but checking if there are values other than female and male\n\ndf['Sex'].value_counts(dropna=False)","metadata":{"executionInfo":{"elapsed":210,"status":"ok","timestamp":1651975635285,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"298bb8ef","outputId":"e7e71d8a-55f3-4cc2-8323-bca28ed1438a","execution":{"iopub.status.busy":"2022-05-30T08:58:57.772739Z","iopub.execute_input":"2022-05-30T08:58:57.773089Z","iopub.status.idle":"2022-05-30T08:58:57.782211Z","shell.execute_reply.started":"2022-05-30T08:58:57.773052Z","shell.execute_reply":"2022-05-30T08:58:57.781347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Sex'] = df['Sex'].map(lambda x: 1 if x == 'male' else 0)\ndf['Sex'].value_counts()","metadata":{"executionInfo":{"elapsed":196,"status":"ok","timestamp":1651975635289,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"fa81db9e","outputId":"19199635-8b79-4842-d0f0-94c3c36a41fe","execution":{"iopub.status.busy":"2022-05-30T08:58:57.840823Z","iopub.execute_input":"2022-05-30T08:58:57.841838Z","iopub.status.idle":"2022-05-30T08:58:57.852493Z","shell.execute_reply.started":"2022-05-30T08:58:57.841795Z","shell.execute_reply":"2022-05-30T08:58:57.851783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Missing values","metadata":{"id":"e2e16f49"}},{"cell_type":"code","source":"df.info()","metadata":{"executionInfo":{"elapsed":188,"status":"ok","timestamp":1651975635295,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"c293bfdc","outputId":"8e4a4e2d-1a02-42f7-982a-61087a250902","execution":{"iopub.status.busy":"2022-05-30T08:58:57.880385Z","iopub.execute_input":"2022-05-30T08:58:57.881439Z","iopub.status.idle":"2022-05-30T08:58:57.893204Z","shell.execute_reply.started":"2022-05-30T08:58:57.881386Z","shell.execute_reply":"2022-05-30T08:58:57.892333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fare feature has missing value but only for 1 row,\nso this will be replaced by the mean value of Fare feature","metadata":{"id":"a4967a4a"}},{"cell_type":"code","source":"df['Fare'] = df['Fare'].fillna(df['Fare'].mean())","metadata":{"executionInfo":{"elapsed":176,"status":"ok","timestamp":1651975635298,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"ccbd87c9","execution":{"iopub.status.busy":"2022-05-30T08:58:57.921687Z","iopub.execute_input":"2022-05-30T08:58:57.922385Z","iopub.status.idle":"2022-05-30T08:58:57.927984Z","shell.execute_reply.started":"2022-05-30T08:58:57.922341Z","shell.execute_reply":"2022-05-30T08:58:57.927309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"executionInfo":{"elapsed":175,"status":"ok","timestamp":1651975635300,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"4kjhFy47Hp01","outputId":"3c8e548b-27f2-4840-b089-cf058df7a664","execution":{"iopub.status.busy":"2022-05-30T08:58:57.981598Z","iopub.execute_input":"2022-05-30T08:58:57.982096Z","iopub.status.idle":"2022-05-30T08:58:57.99518Z","shell.execute_reply.started":"2022-05-30T08:58:57.982061Z","shell.execute_reply":"2022-05-30T08:58:57.994536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Age'].isnull().sum()","metadata":{"executionInfo":{"elapsed":161,"status":"ok","timestamp":1651975635302,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"3777bd06","outputId":"ff136bab-7753-4862-eaa7-3a8a5635509a","execution":{"iopub.status.busy":"2022-05-30T08:58:58.04058Z","iopub.execute_input":"2022-05-30T08:58:58.041061Z","iopub.status.idle":"2022-05-30T08:58:58.047827Z","shell.execute_reply.started":"2022-05-30T08:58:58.041026Z","shell.execute_reply":"2022-05-30T08:58:58.046712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Age feature has 263 missing values\n\nAs tested in my previous notebook version, I will impute this with Iterative imputation with RandomForestRegressor","metadata":{"id":"430da705"}},{"cell_type":"code","source":"# before imputation\n\ndf['Age'].describe()","metadata":{"executionInfo":{"elapsed":145,"status":"ok","timestamp":1651975635305,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"37336de8","outputId":"c48991cf-ff1b-4b33-83d9-8a8f94e3742f","execution":{"iopub.status.busy":"2022-05-30T08:58:58.100529Z","iopub.execute_input":"2022-05-30T08:58:58.100955Z","iopub.status.idle":"2022-05-30T08:58:58.111531Z","shell.execute_reply.started":"2022-05-30T08:58:58.100924Z","shell.execute_reply":"2022-05-30T08:58:58.110562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr = df.copy()","metadata":{"executionInfo":{"elapsed":131,"status":"ok","timestamp":1651975635308,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"Dc6zHhAeVnO0","execution":{"iopub.status.busy":"2022-05-30T08:58:58.122164Z","iopub.execute_input":"2022-05-30T08:58:58.122611Z","iopub.status.idle":"2022-05-30T08:58:58.128598Z","shell.execute_reply.started":"2022-05-30T08:58:58.122562Z","shell.execute_reply":"2022-05-30T08:58:58.127718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Iterative imputation with RandomForestRegressor\n\nI will only use columns that did not have any missing values to start with (exclusing columns I already dropped)\n\nto predict missing Age values\n\n","metadata":{"id":"9u6TlR3mTrag"}},{"cell_type":"code","source":"df_rfr.info()","metadata":{"executionInfo":{"elapsed":132,"status":"ok","timestamp":1651975635311,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"6ANxJDIDVBGr","outputId":"a860eadd-6af7-44b6-b022-1ad0acc1c70d","execution":{"iopub.status.busy":"2022-05-30T08:58:58.181615Z","iopub.execute_input":"2022-05-30T08:58:58.181909Z","iopub.status.idle":"2022-05-30T08:58:58.194473Z","shell.execute_reply.started":"2022-05-30T08:58:58.181878Z","shell.execute_reply":"2022-05-30T08:58:58.193482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\n# define features to predict the age\nage_df = df_rfr[['Age', 'Pclass', 'Sex', 'SibSp', 'Parch']]\n\n# separate age_df into train (with Age) and test (Age is NaN) and to ndarray\nhave_age = age_df[age_df['Age'].notnull()].values\nno_age = age_df[age_df['Age'].isnull()].values\n\n# separate train data to X and y\n\nX = have_age[:, 1:]\ny = have_age[:, 0]","metadata":{"executionInfo":{"elapsed":120,"status":"ok","timestamp":1651975635313,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"ozX51mDIVBZL","execution":{"iopub.status.busy":"2022-05-30T08:58:58.216636Z","iopub.execute_input":"2022-05-30T08:58:58.21706Z","iopub.status.idle":"2022-05-30T08:58:58.224782Z","shell.execute_reply.started":"2022-05-30T08:58:58.21703Z","shell.execute_reply":"2022-05-30T08:58:58.22415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build age prediction model with RandomForest\n\nrfr = RandomForestRegressor(random_state = 0, n_estimators = 100, n_jobs = -1)\nrfr.fit(X, y)","metadata":{"executionInfo":{"elapsed":1152,"status":"ok","timestamp":1651975636348,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"2HtHGjPPVBkr","outputId":"2357c4ed-cbc8-48df-88fe-3bee6d7f3718","execution":{"iopub.status.busy":"2022-05-30T08:58:58.285332Z","iopub.execute_input":"2022-05-30T08:58:58.285742Z","iopub.status.idle":"2022-05-30T08:58:58.639Z","shell.execute_reply.started":"2022-05-30T08:58:58.285713Z","shell.execute_reply":"2022-05-30T08:58:58.637973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use the model to predict the Age for test data\n\n# fix df_copy later\nage_predicted = rfr.predict(no_age[:, 1:])\n\n# age_predicted = np.round_(age_predicted, decimals=1)\nprint(age_predicted)\n","metadata":{"executionInfo":{"elapsed":231,"status":"ok","timestamp":1651975636350,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"xJ4iZ6JvZiTu","outputId":"583ebf31-42de-4364-bb9d-f640e4fa3b47","execution":{"iopub.status.busy":"2022-05-30T08:58:58.640824Z","iopub.execute_input":"2022-05-30T08:58:58.64105Z","iopub.status.idle":"2022-05-30T08:58:58.754032Z","shell.execute_reply.started":"2022-05-30T08:58:58.641022Z","shell.execute_reply":"2022-05-30T08:58:58.752968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replace missing Age values with age_predicted\n\ndf_rfr.loc[(df_rfr['Age'].isnull()), 'Age'] = age_predicted","metadata":{"executionInfo":{"elapsed":211,"status":"ok","timestamp":1651975636353,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"e11FUi2Lb34m","execution":{"iopub.status.busy":"2022-05-30T08:58:58.755491Z","iopub.execute_input":"2022-05-30T08:58:58.755743Z","iopub.status.idle":"2022-05-30T08:58:58.762279Z","shell.execute_reply.started":"2022-05-30T08:58:58.755713Z","shell.execute_reply":"2022-05-30T08:58:58.761268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr['Age'] = df_rfr['Age'].map(lambda x: round(x, 2))","metadata":{"executionInfo":{"elapsed":212,"status":"ok","timestamp":1651975636357,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"kdPtxehOn9sF","execution":{"iopub.status.busy":"2022-05-30T08:58:58.764752Z","iopub.execute_input":"2022-05-30T08:58:58.765037Z","iopub.status.idle":"2022-05-30T08:58:58.780145Z","shell.execute_reply.started":"2022-05-30T08:58:58.765006Z","shell.execute_reply":"2022-05-30T08:58:58.779219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr['Age'].isnull().sum()","metadata":{"executionInfo":{"elapsed":213,"status":"ok","timestamp":1651975636361,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"0obvXCa8Z6Lg","outputId":"7ad1e96b-5aff-4c3f-aa0c-d9b3e8112e24","execution":{"iopub.status.busy":"2022-05-30T08:58:58.782095Z","iopub.execute_input":"2022-05-30T08:58:58.782385Z","iopub.status.idle":"2022-05-30T08:58:58.79515Z","shell.execute_reply.started":"2022-05-30T08:58:58.782352Z","shell.execute_reply":"2022-05-30T08:58:58.793919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compare Age feature before imputation and after\n\nprint('Before imputation')\nprint(df['Age'].describe())\nprint(df['Age'].value_counts().sort_index())\nprint(\" \")\n\nprint('Age feature for df_rfr')\nprint(df_rfr['Age'].describe())\nprint(df_rfr['Age'].value_counts().sort_index())","metadata":{"executionInfo":{"elapsed":198,"status":"ok","timestamp":1651975636365,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"zs7iAz3Vo4la","outputId":"b89487d8-480d-45bb-ea95-118284905477","execution":{"iopub.status.busy":"2022-05-30T08:58:58.797574Z","iopub.execute_input":"2022-05-30T08:58:58.798517Z","iopub.status.idle":"2022-05-30T08:58:58.82716Z","shell.execute_reply.started":"2022-05-30T08:58:58.798463Z","shell.execute_reply":"2022-05-30T08:58:58.826535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking if there are any outliers or unknown values","metadata":{"id":"2ad16a02"}},{"cell_type":"code","source":"df_rfr.describe().T","metadata":{"executionInfo":{"elapsed":185,"status":"ok","timestamp":1651975636369,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"8716a166","outputId":"c606ae36-b3aa-4acf-e88e-1c3bc4b7f384","execution":{"iopub.status.busy":"2022-05-30T08:58:58.828214Z","iopub.execute_input":"2022-05-30T08:58:58.829092Z","iopub.status.idle":"2022-05-30T08:58:58.859904Z","shell.execute_reply.started":"2022-05-30T08:58:58.829058Z","shell.execute_reply":"2022-05-30T08:58:58.859272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df_rfr.columns:\n  print(\"Unique values for column: \" + i)\n  print(df_rfr[i].value_counts(dropna=False))\n  print(\" \")","metadata":{"executionInfo":{"elapsed":180,"status":"ok","timestamp":1651975636372,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"fc4aafb9","outputId":"ea5aeaf1-4e38-40ec-aa77-2186e5388b83","execution":{"iopub.status.busy":"2022-05-30T08:58:58.860849Z","iopub.execute_input":"2022-05-30T08:58:58.861701Z","iopub.status.idle":"2022-05-30T08:58:58.876613Z","shell.execute_reply.started":"2022-05-30T08:58:58.861666Z","shell.execute_reply":"2022-05-30T08:58:58.875705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr['Fare'].value_counts().sort_index()","metadata":{"executionInfo":{"elapsed":167,"status":"ok","timestamp":1651975636375,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"aa78d161","outputId":"89d9fcf0-2d2d-443e-9701-0c848c7ccc9b","execution":{"iopub.status.busy":"2022-05-30T08:58:58.878143Z","iopub.execute_input":"2022-05-30T08:58:58.878666Z","iopub.status.idle":"2022-05-30T08:58:58.888346Z","shell.execute_reply.started":"2022-05-30T08:58:58.878626Z","shell.execute_reply":"2022-05-30T08:58:58.887726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The .min() value for Fare feature is 0 and it appears 17 times\n\nit is questionable whether this is because the passenger did not pay at all \n\nor 0 value because the fare paid by the 17 passengers are unknown","metadata":{"id":"35c00871"}},{"cell_type":"code","source":"fare_unknown = df_rfr[df_rfr['Fare'] == 0]\nfare_unknown","metadata":{"executionInfo":{"elapsed":159,"status":"ok","timestamp":1651975636384,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"5ac52b0c","outputId":"5e330672-6ba4-47e7-a4ee-bcad202d78a0","execution":{"iopub.status.busy":"2022-05-30T08:58:58.890252Z","iopub.execute_input":"2022-05-30T08:58:58.890649Z","iopub.status.idle":"2022-05-30T08:58:58.912804Z","shell.execute_reply.started":"2022-05-30T08:58:58.890619Z","shell.execute_reply":"2022-05-30T08:58:58.911463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some of the passengers with 'unknown' fare are Pclass 1 (i.e. first class) passengers\n\nand it is very unlikely that first passengers did not pay or pay little fare (unless they were invited passegners or so) \n\nso I assume that 'fare_unknown' passengers, how much fare they paid is unknown, rather than they did not pay any fare at all","metadata":{"id":"e01389c6"}},{"cell_type":"markdown","source":"In my earlier notebooks/versions I left the 0 fare values as they were, and replaced 1 NaN value with the mean Fare value \n\nbut in my previous notebook/version I tested and compared different ways to impute 0 Fare values\n\nAs the result, I impute 0 Fare values by median of Pclass and Embarked features\n\nI have to get the row which had NaN in the original dataset, so I can also treat it as a row with 0 Fare value and impute","metadata":{"id":"lytm8L7-QzxZ"}},{"cell_type":"code","source":"# df2 is the copied dataframe of the original dataframe\n\ndf2[df2['Fare'].isnull()].index","metadata":{"id":"QVrmpVPvIN1E","executionInfo":{"status":"ok","timestamp":1651975636387,"user_tz":-540,"elapsed":154,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"3a30280e-091f-4c73-af68-8769d46c118c","execution":{"iopub.status.busy":"2022-05-30T08:58:58.914625Z","iopub.execute_input":"2022-05-30T08:58:58.914947Z","iopub.status.idle":"2022-05-30T08:58:58.928626Z","shell.execute_reply.started":"2022-05-30T08:58:58.914913Z","shell.execute_reply":"2022-05-30T08:58:58.927775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Index of the row with missing Fare value is 1043 and Fare replaced by 0 value\n\nso I can transform all 0 Fare value rows together","metadata":{"id":"u7mtiZ0-JQGJ"}},{"cell_type":"code","source":"df_rfr3 = df_rfr.copy()","metadata":{"executionInfo":{"elapsed":146,"status":"ok","timestamp":1651975636391,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"IJ8817g1RPyC","execution":{"iopub.status.busy":"2022-05-30T08:58:58.932851Z","iopub.execute_input":"2022-05-30T08:58:58.933518Z","iopub.status.idle":"2022-05-30T08:58:58.939092Z","shell.execute_reply.started":"2022-05-30T08:58:58.933461Z","shell.execute_reply":"2022-05-30T08:58:58.937979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr3.loc[1043, :]\n\n# Fare value here is the mean value imputed","metadata":{"id":"tfgqx4ixLtVS","executionInfo":{"status":"ok","timestamp":1651975636394,"user_tz":-540,"elapsed":147,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"387b5314-31c8-44b5-d341-dc1a12329ce0","execution":{"iopub.status.busy":"2022-05-30T08:58:58.961326Z","iopub.execute_input":"2022-05-30T08:58:58.961768Z","iopub.status.idle":"2022-05-30T08:58:58.968092Z","shell.execute_reply.started":"2022-05-30T08:58:58.961737Z","shell.execute_reply":"2022-05-30T08:58:58.967527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr3.loc[1043, 'Fare'] = 0","metadata":{"id":"9c968JaWIl5Z","executionInfo":{"status":"ok","timestamp":1651975636398,"user_tz":-540,"elapsed":139,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"execution":{"iopub.status.busy":"2022-05-30T08:58:58.975835Z","iopub.execute_input":"2022-05-30T08:58:58.976254Z","iopub.status.idle":"2022-05-30T08:58:58.98548Z","shell.execute_reply.started":"2022-05-30T08:58:58.976204Z","shell.execute_reply":"2022-05-30T08:58:58.984686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df_rfr3[df_rfr3['Fare'] == 0])","metadata":{"id":"b7uMNhRyJz3i","executionInfo":{"status":"ok","timestamp":1651975636402,"user_tz":-540,"elapsed":141,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"af423083-d4cf-41fb-d997-f6d317890b53","execution":{"iopub.status.busy":"2022-05-30T08:58:59.04021Z","iopub.execute_input":"2022-05-30T08:58:59.040501Z","iopub.status.idle":"2022-05-30T08:58:59.048589Z","shell.execute_reply.started":"2022-05-30T08:58:59.04047Z","shell.execute_reply":"2022-05-30T08:58:59.047662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fare is determined by Pclass as well as by Embarked\n\nso I impute 0 Fare value with the median Fare of Pclass and Embarked\n\nI have dropped Embarked column earlier in the process so I put it back temporarily to get the median value with Pclass","metadata":{"id":"dkD82wtEsxu7"}},{"cell_type":"code","source":"df_rfr3.head()","metadata":{"executionInfo":{"elapsed":132,"status":"ok","timestamp":1651975636407,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"tbBiG2GtrjIH","outputId":"09bdaccf-a29c-489d-f08f-f97c42b31bb4","execution":{"iopub.status.busy":"2022-05-30T08:58:59.1012Z","iopub.execute_input":"2022-05-30T08:58:59.102008Z","iopub.status.idle":"2022-05-30T08:58:59.116926Z","shell.execute_reply.started":"2022-05-30T08:58:59.101968Z","shell.execute_reply":"2022-05-30T08:58:59.115991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr3['Embarked'] = df2['Embarked']\n\ndf_rfr3.head()","metadata":{"id":"FJCyp2ldPq05","executionInfo":{"status":"ok","timestamp":1651975636410,"user_tz":-540,"elapsed":131,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"61c49517-3d69-4054-c76d-2070cc3d0708","execution":{"iopub.status.busy":"2022-05-30T08:58:59.197943Z","iopub.execute_input":"2022-05-30T08:58:59.199075Z","iopub.status.idle":"2022-05-30T08:58:59.215694Z","shell.execute_reply.started":"2022-05-30T08:58:59.199003Z","shell.execute_reply":"2022-05-30T08:58:59.214889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check the mean and median of Fare values by Pclass and Embarked","metadata":{"id":"E1EaCWytRs55"}},{"cell_type":"code","source":"data2 = df_rfr3.loc[df_rfr3['Fare'] != 0,:].groupby(['Pclass', 'Embarked']).agg(['mean', 'median', 'count'])['Fare']\nprint(data2)","metadata":{"id":"ZDE175U0DTMG","executionInfo":{"status":"ok","timestamp":1651975636413,"user_tz":-540,"elapsed":124,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"96c614fa-4546-4913-cda2-6e23d1f4cc1a","execution":{"iopub.status.busy":"2022-05-30T08:58:59.244955Z","iopub.execute_input":"2022-05-30T08:58:59.245286Z","iopub.status.idle":"2022-05-30T08:58:59.268986Z","shell.execute_reply.started":"2022-05-30T08:58:59.245248Z","shell.execute_reply":"2022-05-30T08:58:59.268297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df_rfr3['Pclass'].unique():\n  for location in df_rfr3['Embarked'].unique():\n\n    df_rfr3.loc[(df_rfr3['Fare'] == 0) & (df_rfr3['Pclass'] == i) & (df_rfr3['Embarked'] == location), 'Fare'] = df_rfr3.loc[(df_rfr3['Fare'] == 0) & (df_rfr3['Pclass'] == i) & (df_rfr3['Embarked'] == location), 'Fare'].map(\n        lambda x: df_rfr3[(df_rfr3['Fare'] != 0) & (df_rfr3['Pclass'] == i) & (df_rfr3['Embarked'] == location)]['Fare'].median()\n    )\n    \n    # print(i,location) \n    # print(df_rfr3[(df_rfr3['Fare'] != 0) & (df_rfr3['Pclass'] == i) & (df_rfr3['Embarked'] == location)]['Fare'].median())","metadata":{"id":"vkaBgLnlalb_","executionInfo":{"status":"ok","timestamp":1651975637116,"user_tz":-540,"elapsed":812,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"execution":{"iopub.status.busy":"2022-05-30T08:58:59.280174Z","iopub.execute_input":"2022-05-30T08:58:59.280657Z","iopub.status.idle":"2022-05-30T08:58:59.352427Z","shell.execute_reply.started":"2022-05-30T08:58:59.280625Z","shell.execute_reply":"2022-05-30T08:58:59.351682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# All 0 Fare values have been imputed\n\ndf_rfr3[df_rfr3['Fare'] == 0]","metadata":{"executionInfo":{"elapsed":120,"status":"ok","timestamp":1651975637119,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"outputId":"ea4faac8-309f-42e2-d511-94b9ec7616e8","id":"9z9UMCqXRs6E","execution":{"iopub.status.busy":"2022-05-30T08:58:59.35976Z","iopub.execute_input":"2022-05-30T08:58:59.360026Z","iopub.status.idle":"2022-05-30T08:58:59.371492Z","shell.execute_reply.started":"2022-05-30T08:58:59.35999Z","shell.execute_reply":"2022-05-30T08:58:59.370331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Before imputation\")\nprint(data2)\nprint(\"\")\nprint(\"After imputation\")\nprint(df_rfr3.groupby(['Pclass', 'Embarked']).agg(['mean', 'median', 'count'])['Fare'])","metadata":{"executionInfo":{"elapsed":115,"status":"ok","timestamp":1651975637123,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"outputId":"3271a5c4-ab27-4a81-f1a4-09d82d8edfa6","id":"zWk7T5z-Rs6G","execution":{"iopub.status.busy":"2022-05-30T08:58:59.402391Z","iopub.execute_input":"2022-05-30T08:58:59.402814Z","iopub.status.idle":"2022-05-30T08:58:59.433201Z","shell.execute_reply.started":"2022-05-30T08:58:59.402783Z","shell.execute_reply":"2022-05-30T08:58:59.431972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr3 = df_rfr3.drop(columns='Embarked', axis=1)","metadata":{"id":"eRCNNEcWvAH7","executionInfo":{"status":"ok","timestamp":1651975637126,"user_tz":-540,"elapsed":60,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"execution":{"iopub.status.busy":"2022-05-30T08:58:59.440565Z","iopub.execute_input":"2022-05-30T08:58:59.441339Z","iopub.status.idle":"2022-05-30T08:58:59.447675Z","shell.execute_reply.started":"2022-05-30T08:58:59.44129Z","shell.execute_reply":"2022-05-30T08:58:59.446765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"New Family_group feature\n\nI will combine SibSp and Parch features and make Family_group feature","metadata":{}},{"cell_type":"code","source":"df_rfr4 = df_rfr3.copy()\ndf_rfr4.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-30T08:58:59.501534Z","iopub.execute_input":"2022-05-30T08:58:59.502028Z","iopub.status.idle":"2022-05-30T08:58:59.514459Z","shell.execute_reply.started":"2022-05-30T08:58:59.501981Z","shell.execute_reply":"2022-05-30T08:58:59.513535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the original train dataset to see how SibSp and Parch features are correlated to Survival rate\n\ntrain_data['Survived'] = Survived_data\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-30T08:58:59.588403Z","iopub.execute_input":"2022-05-30T08:58:59.588737Z","iopub.status.idle":"2022-05-30T08:58:59.607745Z","shell.execute_reply.started":"2022-05-30T08:58:59.588697Z","shell.execute_reply":"2022-05-30T08:58:59.606812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Survived_SibSp = train_data[['SibSp', 'Survived']].groupby('SibSp')['Survived'].agg(['count', 'sum'])\nSurvived_SibSp['Survived_rate'] = train_data[['SibSp', 'Survived']].groupby('SibSp')['Survived'].apply(lambda x: (x.sum()/x.count())*100)\n\nSurvived_SibSp","metadata":{"execution":{"iopub.status.busy":"2022-05-30T08:58:59.615944Z","iopub.execute_input":"2022-05-30T08:58:59.616685Z","iopub.status.idle":"2022-05-30T08:58:59.637082Z","shell.execute_reply.started":"2022-05-30T08:58:59.616642Z","shell.execute_reply":"2022-05-30T08:58:59.636475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Survived_Parch = train_data[['Parch', 'Survived']].groupby('Parch')['Survived'].agg(['count', 'sum'])\nSurvived_Parch['Survived_rate'] = train_data[['Parch', 'Survived']].groupby('Parch')['Survived'].apply(lambda x: (x.sum()/x.count())*100)\n\nSurvived_Parch","metadata":{"execution":{"iopub.status.busy":"2022-05-30T08:58:59.685281Z","iopub.execute_input":"2022-05-30T08:58:59.686078Z","iopub.status.idle":"2022-05-30T08:58:59.706838Z","shell.execute_reply.started":"2022-05-30T08:58:59.686041Z","shell.execute_reply":"2022-05-30T08:58:59.705789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I can see that there is a strong correlation between the number of SibSp/Parch and the Survival rate,\n\nso rather than keeping the two features independent, I will combine and assign as the Family_group feature with 3 different groups based on the respective survival rate","metadata":{}},{"cell_type":"code","source":"df_rfr4['Family']=df_rfr4['SibSp'] + df_rfr4['Parch'] + 1\n\ndf_rfr4.loc[(df_rfr4['Family']>=2) & (df_rfr4['Family']<=4), 'Family_group'] = 2\ndf_rfr4.loc[(df_rfr4['Family']>=5) & (df_rfr4['Family']<=7) | (df_rfr4['Family']==1), 'Family_group'] = 1  \ndf_rfr4.loc[(df_rfr4['Family']>=8), 'Family_group'] = 0\n\n# drop SibSp, Parch and Family columns \ndf_rfr4 = df_rfr4.drop(columns = ['SibSp', 'Parch', 'Family'])\n\ndf_rfr4.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-30T08:58:59.740511Z","iopub.execute_input":"2022-05-30T08:58:59.740842Z","iopub.status.idle":"2022-05-30T08:58:59.764063Z","shell.execute_reply.started":"2022-05-30T08:58:59.740803Z","shell.execute_reply":"2022-05-30T08:58:59.763439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Building the model","metadata":{"id":"0e337aba"}},{"cell_type":"markdown","source":"Now that full dataset (train and test combined) has been preprocessed \n\nI will now separate the full dataset back to train and test datasets then build the model","metadata":{"id":"3ef2d00d"}},{"cell_type":"markdown","source":"Build the model\n\nI will build a **random forest model** first (as recommended by Kaggle example notebook) \n\nand compare the accuracy with other algorithms at later stage","metadata":{"id":"3530168b"}},{"cell_type":"code","source":"# separate the full dataset back to train and test\n\ntrain = df_rfr4.iloc[:891, :]\ntest = df_rfr4.iloc[891:,:]","metadata":{"executionInfo":{"elapsed":61,"status":"ok","timestamp":1651975637130,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"w18IQb-Yufsi","execution":{"iopub.status.busy":"2022-05-30T08:58:59.770898Z","iopub.execute_input":"2022-05-30T08:58:59.771525Z","iopub.status.idle":"2022-05-30T08:58:59.77758Z","shell.execute_reply.started":"2022-05-30T08:58:59.771441Z","shell.execute_reply":"2022-05-30T08:58:59.776501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# adding back the Survived feature \n# that I separated from train_data earlier before datasets concat\ntrain['Survived'] = Survived_data\n\ntrain.info()","metadata":{"executionInfo":{"elapsed":62,"status":"ok","timestamp":1651975637133,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"outputId":"28a4622b-2be6-4856-b212-38ca3fe1344f","id":"ULJ-irtVufsk","execution":{"iopub.status.busy":"2022-05-30T08:58:59.824728Z","iopub.execute_input":"2022-05-30T08:58:59.825169Z","iopub.status.idle":"2022-05-30T08:58:59.843394Z","shell.execute_reply.started":"2022-05-30T08:58:59.82513Z","shell.execute_reply":"2022-05-30T08:58:59.841775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train.drop('Survived', axis=1)\ny = train['Survived']","metadata":{"executionInfo":{"elapsed":47,"status":"ok","timestamp":1651975637136,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"5g8_EsSKufsn","execution":{"iopub.status.busy":"2022-05-30T08:58:59.869889Z","iopub.execute_input":"2022-05-30T08:58:59.871168Z","iopub.status.idle":"2022-05-30T08:58:59.877886Z","shell.execute_reply.started":"2022-05-30T08:58:59.87111Z","shell.execute_reply":"2022-05-30T08:58:59.877156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# this submission version trains the model on full train data\n# from sklearn.model_selection import train_test_split  \n# X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state=42)\n\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import GridSearchCV\nclf = RandomForestClassifier() \n\n# cross validation of all combinations of the parameters\nparam_grid = [\n  {\"n_estimators\": [i for i in range(10, 100, 10)], \n   \"criterion\": [\"gini\", \"entropy\"],\n   \"max_depth\": [i for i in range(10, 15, 1)],\n   \"min_samples_split\": [2, 4, 10, 12, 16],\n   \"random_state\": [42]}\n]\n\n# cv=3 means cross validation with 3 folds\n# performance scoring metrics is accuracy\ngrid_search = GridSearchCV(clf, param_grid, cv=3, scoring=\"accuracy\", return_train_score=True)\ngrid_search.fit(X, y)","metadata":{"id":"R7tPKV5UDSIu","executionInfo":{"status":"ok","timestamp":1651975698614,"user_tz":-540,"elapsed":61522,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"ab6c023a-f7be-4b00-ce80-275121b04b6e","execution":{"iopub.status.busy":"2022-05-30T08:58:59.921447Z","iopub.execute_input":"2022-05-30T08:58:59.921905Z","iopub.status.idle":"2022-05-30T09:01:56.189026Z","shell.execute_reply.started":"2022-05-30T08:58:59.921873Z","shell.execute_reply":"2022-05-30T09:01:56.187989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get the combination with the best performing score\n\nfinal_clf = grid_search.best_estimator_\nfinal_clf","metadata":{"id":"aeaj52xAGvUk","executionInfo":{"status":"ok","timestamp":1651975698617,"user_tz":-540,"elapsed":100,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"1acfa1d0-0e25-44f7-d86c-0578a10dc000","execution":{"iopub.status.busy":"2022-05-30T09:01:56.190751Z","iopub.execute_input":"2022-05-30T09:01:56.191253Z","iopub.status.idle":"2022-05-30T09:01:56.197932Z","shell.execute_reply.started":"2022-05-30T09:01:56.191186Z","shell.execute_reply":"2022-05-30T09:01:56.197121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"and now I will use the model on test data to make submission prediction","metadata":{"id":"huLHi-OWsEu_"}},{"cell_type":"code","source":"# check feature importance\n\nfinal_clf.feature_importances_","metadata":{"id":"twBb_Rg6Xj2S","executionInfo":{"status":"ok","timestamp":1651975698621,"user_tz":-540,"elapsed":79,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"888ae015-1d07-49d8-86da-c1526de06bea","execution":{"iopub.status.busy":"2022-05-30T09:01:56.198859Z","iopub.execute_input":"2022-05-30T09:01:56.199644Z","iopub.status.idle":"2022-05-30T09:01:56.220444Z","shell.execute_reply.started":"2022-05-30T09:01:56.199603Z","shell.execute_reply":"2022-05-30T09:01:56.219364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index_list = X.columns\nprint(index_list)","metadata":{"id":"nbZVUsdeXkGq","executionInfo":{"status":"ok","timestamp":1651975699197,"user_tz":-540,"elapsed":633,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"ef6a1659-b586-4655-e47b-9958c370456d","execution":{"iopub.status.busy":"2022-05-30T09:01:56.222543Z","iopub.execute_input":"2022-05-30T09:01:56.222933Z","iopub.status.idle":"2022-05-30T09:01:56.230746Z","shell.execute_reply.started":"2022-05-30T09:01:56.222904Z","shell.execute_reply":"2022-05-30T09:01:56.229793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_importance = pd.DataFrame(\n    { \"score\": final_clf.feature_importances_},\n    index = index_list\n)\n\nfeature_importance.sort_values('score', ascending=False)","metadata":{"id":"zA4JPGnNZGdA","executionInfo":{"status":"ok","timestamp":1651975699202,"user_tz":-540,"elapsed":192,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"76fa94d6-9a9f-4fcd-f6ef-0147580d5633","execution":{"iopub.status.busy":"2022-05-30T09:01:56.232431Z","iopub.execute_input":"2022-05-30T09:01:56.232647Z","iopub.status.idle":"2022-05-30T09:01:56.252585Z","shell.execute_reply.started":"2022-05-30T09:01:56.232622Z","shell.execute_reply":"2022-05-30T09:01:56.251713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Prediction on test dataset for submission","metadata":{"id":"bb77cbe5"}},{"cell_type":"code","source":"test = df_rfr4.iloc[891:, :]\n\ntest.head()","metadata":{"executionInfo":{"elapsed":185,"status":"ok","timestamp":1651975699206,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"d9dde138","outputId":"4ba39679-7faf-49b8-d0e5-3fdd2e28a147","execution":{"iopub.status.busy":"2022-05-30T09:01:56.254112Z","iopub.execute_input":"2022-05-30T09:01:56.254629Z","iopub.status.idle":"2022-05-30T09:01:56.267603Z","shell.execute_reply.started":"2022-05-30T09:01:56.254597Z","shell.execute_reply":"2022-05-30T09:01:56.266631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"executionInfo":{"elapsed":179,"status":"ok","timestamp":1651975699209,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"t-ohC9lEeO9J","outputId":"70352d67-b044-4bb9-ce3d-134a92327cac","execution":{"iopub.status.busy":"2022-05-30T09:01:56.268799Z","iopub.execute_input":"2022-05-30T09:01:56.269035Z","iopub.status.idle":"2022-05-30T09:01:56.284902Z","shell.execute_reply.started":"2022-05-30T09:01:56.269007Z","shell.execute_reply":"2022-05-30T09:01:56.283872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I will need PassengerId column for submission later on \n\nid_column = test_data['PassengerId']\nid_column","metadata":{"executionInfo":{"elapsed":160,"status":"ok","timestamp":1651975699213,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"c8a260a7","outputId":"48247dcb-6369-41cc-ae68-e9c0b87a662b","execution":{"iopub.status.busy":"2022-05-30T09:01:56.287215Z","iopub.execute_input":"2022-05-30T09:01:56.287625Z","iopub.status.idle":"2022-05-30T09:01:56.302928Z","shell.execute_reply.started":"2022-05-30T09:01:56.287594Z","shell.execute_reply":"2022-05-30T09:01:56.302304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now the test data is ready to make prediction","metadata":{"id":"aaa0fe13"}},{"cell_type":"code","source":"submission_prediction = final_clf.predict(test)","metadata":{"executionInfo":{"elapsed":150,"status":"ok","timestamp":1651975699216,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"1448ec6e","execution":{"iopub.status.busy":"2022-05-30T09:01:56.303848Z","iopub.execute_input":"2022-05-30T09:01:56.304375Z","iopub.status.idle":"2022-05-30T09:01:56.322532Z","shell.execute_reply.started":"2022-05-30T09:01:56.304345Z","shell.execute_reply":"2022-05-30T09:01:56.321792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_prediction","metadata":{"executionInfo":{"elapsed":150,"status":"ok","timestamp":1651975699219,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"b164cad7","outputId":"a148a0d9-b9a6-4e38-96a4-b2f0310a4faa","execution":{"iopub.status.busy":"2022-05-30T09:01:56.324626Z","iopub.execute_input":"2022-05-30T09:01:56.32515Z","iopub.status.idle":"2022-05-30T09:01:56.330855Z","shell.execute_reply.started":"2022-05-30T09:01:56.32512Z","shell.execute_reply":"2022-05-30T09:01:56.33024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Prediction submission","metadata":{"id":"155796bb"}},{"cell_type":"code","source":"final_df = pd.DataFrame(id_column)\n\nfinal_df['Survived'] = submission_prediction\n\nfinal_df.head()","metadata":{"executionInfo":{"elapsed":141,"status":"ok","timestamp":1651975699222,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"6092764e","outputId":"38d0678d-ccb5-493a-c524-2dca6f99b019","execution":{"iopub.status.busy":"2022-05-30T09:01:56.331982Z","iopub.execute_input":"2022-05-30T09:01:56.332266Z","iopub.status.idle":"2022-05-30T09:01:56.349144Z","shell.execute_reply.started":"2022-05-30T09:01:56.332214Z","shell.execute_reply":"2022-05-30T09:01:56.348191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.info()","metadata":{"executionInfo":{"elapsed":137,"status":"ok","timestamp":1651975699225,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"9f9a990b","outputId":"53df2869-0883-430f-ea5c-3a4330bd82c8","execution":{"iopub.status.busy":"2022-05-30T09:01:56.350411Z","iopub.execute_input":"2022-05-30T09:01:56.350661Z","iopub.status.idle":"2022-05-30T09:01:56.368659Z","shell.execute_reply.started":"2022-05-30T09:01:56.350633Z","shell.execute_reply":"2022-05-30T09:01:56.367683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.to_csv('submission.csv', index=False)","metadata":{"executionInfo":{"elapsed":122,"status":"ok","timestamp":1651975699228,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"63d0e6b3","execution":{"iopub.status.busy":"2022-05-30T09:01:56.370312Z","iopub.execute_input":"2022-05-30T09:01:56.370661Z","iopub.status.idle":"2022-05-30T09:01:56.385078Z","shell.execute_reply.started":"2022-05-30T09:01:56.37063Z","shell.execute_reply":"2022-05-30T09:01:56.384305Z"},"trusted":true},"execution_count":null,"outputs":[]}]}