{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Kaggle Titanic Submission 4 - Light GBM with StratifiedKFold \n\nAdding improvement one by one to see if the score gets improved \n\nlast submission scores - 0.76555, 0.76076, 0.76315","metadata":{"id":"tL2KiFXpT8mh"}},{"cell_type":"code","source":"# Loading libraries\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"executionInfo":{"elapsed":270,"status":"ok","timestamp":1651975635254,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"1324dce0","execution":{"iopub.status.busy":"2022-07-21T02:21:09.950508Z","iopub.execute_input":"2022-07-21T02:21:09.950973Z","iopub.status.idle":"2022-07-21T02:21:10.624017Z","shell.execute_reply.started":"2022-07-21T02:21:09.950876Z","shell.execute_reply":"2022-07-21T02:21:10.622833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# delete submission.csv if it is already in Output folder\n\n# os.remove('submission.csv')","metadata":{"executionInfo":{"elapsed":269,"status":"ok","timestamp":1651975635258,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"0d12d5ef","execution":{"iopub.status.busy":"2022-07-21T02:21:10.625820Z","iopub.execute_input":"2022-07-21T02:21:10.626410Z","iopub.status.idle":"2022-07-21T02:21:10.630905Z","shell.execute_reply.started":"2022-07-21T02:21:10.626371Z","shell.execute_reply":"2022-07-21T02:21:10.629963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading the Train dataset (train.csv)\n\ntrain_data = pd.read_csv('/kaggle/input/titanic/train.csv')\ntrain_data.head()","metadata":{"executionInfo":{"elapsed":270,"status":"ok","timestamp":1651975635261,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"45da6b04","execution":{"iopub.status.busy":"2022-07-21T02:21:10.631961Z","iopub.execute_input":"2022-07-21T02:21:10.632926Z","iopub.status.idle":"2022-07-21T02:21:10.680720Z","shell.execute_reply.started":"2022-07-21T02:21:10.632892Z","shell.execute_reply":"2022-07-21T02:21:10.679796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading the Test dataset (test.csv)\n\ntest_data = pd.read_csv('/kaggle/input/titanic/test.csv')\ntest_data.head()","metadata":{"executionInfo":{"elapsed":265,"status":"ok","timestamp":1651975635264,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"3a2b4614","execution":{"iopub.status.busy":"2022-07-21T02:21:10.682699Z","iopub.execute_input":"2022-07-21T02:21:10.683395Z","iopub.status.idle":"2022-07-21T02:21:10.712272Z","shell.execute_reply.started":"2022-07-21T02:21:10.683357Z","shell.execute_reply":"2022-07-21T02:21:10.711138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concat train_data and test_data\n\n# before .concat() drop Survived feature from train dataset \n# as test dataset do not have this feature\n\n# join with train dataset later on\nSurvived_data = train_data['Survived']\ntrain_data = train_data.drop('Survived', axis=1)\n\ndf = pd.concat([train_data, test_data], sort=False, ignore_index=True)\n\ndf2 = df.copy()\n\n# check ignore_index is working and index is continuous in whole dataset\ndf.tail()","metadata":{"executionInfo":{"elapsed":264,"status":"ok","timestamp":1651975635266,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"db1cfba7","outputId":"34590255-5ffe-4bbe-b53d-e1d59c3d32fd","execution":{"iopub.status.busy":"2022-07-21T02:21:10.714021Z","iopub.execute_input":"2022-07-21T02:21:10.714638Z","iopub.status.idle":"2022-07-21T02:21:10.748127Z","shell.execute_reply.started":"2022-07-21T02:21:10.714596Z","shell.execute_reply":"2022-07-21T02:21:10.747295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Exploratory Data Analysis (EDA) to understand the dataset","metadata":{"id":"4556beef"}},{"cell_type":"code","source":"df.shape","metadata":{"executionInfo":{"elapsed":259,"status":"ok","timestamp":1651975635270,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"1c07998f","outputId":"049a410b-7ec7-4bac-c40a-86b90eeb77ca","execution":{"iopub.status.busy":"2022-07-21T02:21:10.749747Z","iopub.execute_input":"2022-07-21T02:21:10.750522Z","iopub.status.idle":"2022-07-21T02:21:10.757535Z","shell.execute_reply.started":"2022-07-21T02:21:10.750487Z","shell.execute_reply":"2022-07-21T02:21:10.756355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"executionInfo":{"elapsed":241,"status":"ok","timestamp":1651975635275,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"b9206213","outputId":"ff53ed91-801a-489d-85ab-d00a0ceb91bd","execution":{"iopub.status.busy":"2022-07-21T02:21:10.758864Z","iopub.execute_input":"2022-07-21T02:21:10.759822Z","iopub.status.idle":"2022-07-21T02:21:10.787661Z","shell.execute_reply.started":"2022-07-21T02:21:10.759788Z","shell.execute_reply":"2022-07-21T02:21:10.786307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"so train and test datasets combined, now 1309 rows","metadata":{"id":"6ffe3ab6"}},{"cell_type":"code","source":"df.describe().T","metadata":{"executionInfo":{"elapsed":234,"status":"ok","timestamp":1651975635278,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"7089a0c0","outputId":"7bb5b94c-0ef7-4e4d-c6cc-d1e345cdaf40","execution":{"iopub.status.busy":"2022-07-21T02:21:10.789594Z","iopub.execute_input":"2022-07-21T02:21:10.790085Z","iopub.status.idle":"2022-07-21T02:21:10.827961Z","shell.execute_reply.started":"2022-07-21T02:21:10.790038Z","shell.execute_reply":"2022-07-21T02:21:10.826842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Data Pre-processing - preparing the data for modelling","metadata":{"id":"8a435c79"}},{"cell_type":"markdown","source":"Omit features that are not important \n\nremoving columns which I think are not important as input features to predict the Survived feature based on EDA earlier, \n\nI may adjust this at later stage when I evaluate the model accuracy","metadata":{"id":"94f7d5d6"}},{"cell_type":"code","source":"columns_to_drop = ['PassengerId', 'Name','Ticket', 'Cabin', 'Embarked']\n\ndf = df.drop(columns_to_drop, axis=1)\n\ndf.head()","metadata":{"executionInfo":{"elapsed":230,"status":"ok","timestamp":1651975635281,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"84661855","outputId":"70fd4058-1fc0-48a6-b4db-9fbc1b82a1dc","execution":{"iopub.status.busy":"2022-07-21T02:21:10.829697Z","iopub.execute_input":"2022-07-21T02:21:10.830596Z","iopub.status.idle":"2022-07-21T02:21:10.846403Z","shell.execute_reply.started":"2022-07-21T02:21:10.830560Z","shell.execute_reply":"2022-07-21T02:21:10.845334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Non numeric feature to numeric feature","metadata":{"id":"21d339d9"}},{"cell_type":"code","source":"df.info()","metadata":{"executionInfo":{"elapsed":221,"status":"ok","timestamp":1651975635283,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"QHolZnywdaoM","outputId":"3865175c-7c93-4e34-acca-34ce4cf17f68","execution":{"iopub.status.busy":"2022-07-21T02:21:10.850748Z","iopub.execute_input":"2022-07-21T02:21:10.851286Z","iopub.status.idle":"2022-07-21T02:21:10.866144Z","shell.execute_reply.started":"2022-07-21T02:21:10.851187Z","shell.execute_reply":"2022-07-21T02:21:10.865184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Sex feature needs to be turned into a numerical feature","metadata":{"id":"36f3c8d5"}},{"cell_type":"code","source":"# There is no missing values in Sex column but checking if there are values other than female and male\n\ndf['Sex'].value_counts(dropna=False)","metadata":{"executionInfo":{"elapsed":210,"status":"ok","timestamp":1651975635285,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"298bb8ef","outputId":"e7e71d8a-55f3-4cc2-8323-bca28ed1438a","execution":{"iopub.status.busy":"2022-07-21T02:21:10.867581Z","iopub.execute_input":"2022-07-21T02:21:10.868564Z","iopub.status.idle":"2022-07-21T02:21:10.878891Z","shell.execute_reply.started":"2022-07-21T02:21:10.868479Z","shell.execute_reply":"2022-07-21T02:21:10.877803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Sex'] = df['Sex'].map(lambda x: 1 if x == 'male' else 0)\ndf['Sex'].value_counts()","metadata":{"executionInfo":{"elapsed":196,"status":"ok","timestamp":1651975635289,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"fa81db9e","outputId":"19199635-8b79-4842-d0f0-94c3c36a41fe","execution":{"iopub.status.busy":"2022-07-21T02:21:10.881553Z","iopub.execute_input":"2022-07-21T02:21:10.882443Z","iopub.status.idle":"2022-07-21T02:21:10.895920Z","shell.execute_reply.started":"2022-07-21T02:21:10.882407Z","shell.execute_reply":"2022-07-21T02:21:10.894307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Missing values","metadata":{"id":"e2e16f49"}},{"cell_type":"code","source":"df.info()","metadata":{"executionInfo":{"elapsed":188,"status":"ok","timestamp":1651975635295,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"c293bfdc","outputId":"8e4a4e2d-1a02-42f7-982a-61087a250902","execution":{"iopub.status.busy":"2022-07-21T02:21:10.897524Z","iopub.execute_input":"2022-07-21T02:21:10.898742Z","iopub.status.idle":"2022-07-21T02:21:10.914502Z","shell.execute_reply.started":"2022-07-21T02:21:10.898655Z","shell.execute_reply":"2022-07-21T02:21:10.913279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fare feature has missing value but only for 1 row,\nso this will be replaced by the mean value of Fare feature","metadata":{"id":"a4967a4a"}},{"cell_type":"code","source":"df['Fare'] = df['Fare'].fillna(df['Fare'].mean())","metadata":{"executionInfo":{"elapsed":176,"status":"ok","timestamp":1651975635298,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"ccbd87c9","execution":{"iopub.status.busy":"2022-07-21T02:21:10.916344Z","iopub.execute_input":"2022-07-21T02:21:10.916794Z","iopub.status.idle":"2022-07-21T02:21:10.923635Z","shell.execute_reply.started":"2022-07-21T02:21:10.916751Z","shell.execute_reply":"2022-07-21T02:21:10.922418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"executionInfo":{"elapsed":175,"status":"ok","timestamp":1651975635300,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"4kjhFy47Hp01","outputId":"3c8e548b-27f2-4840-b089-cf058df7a664","execution":{"iopub.status.busy":"2022-07-21T02:21:10.925177Z","iopub.execute_input":"2022-07-21T02:21:10.926401Z","iopub.status.idle":"2022-07-21T02:21:10.945111Z","shell.execute_reply.started":"2022-07-21T02:21:10.926352Z","shell.execute_reply":"2022-07-21T02:21:10.943747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Age'].isnull().sum()","metadata":{"executionInfo":{"elapsed":161,"status":"ok","timestamp":1651975635302,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"3777bd06","outputId":"ff136bab-7753-4862-eaa7-3a8a5635509a","execution":{"iopub.status.busy":"2022-07-21T02:21:10.947126Z","iopub.execute_input":"2022-07-21T02:21:10.947921Z","iopub.status.idle":"2022-07-21T02:21:10.956670Z","shell.execute_reply.started":"2022-07-21T02:21:10.947872Z","shell.execute_reply":"2022-07-21T02:21:10.955517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Age feature has 263 missing values\n\nAs tested in my previous notebook version, I will impute this with Iterative imputation with RandomForestRegressor","metadata":{"id":"430da705"}},{"cell_type":"code","source":"# before imputation\n\ndf['Age'].describe()","metadata":{"executionInfo":{"elapsed":145,"status":"ok","timestamp":1651975635305,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"37336de8","outputId":"c48991cf-ff1b-4b33-83d9-8a8f94e3742f","execution":{"iopub.status.busy":"2022-07-21T02:21:10.958264Z","iopub.execute_input":"2022-07-21T02:21:10.959317Z","iopub.status.idle":"2022-07-21T02:21:10.975089Z","shell.execute_reply.started":"2022-07-21T02:21:10.959270Z","shell.execute_reply":"2022-07-21T02:21:10.973716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr = df.copy()","metadata":{"executionInfo":{"elapsed":131,"status":"ok","timestamp":1651975635308,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"Dc6zHhAeVnO0","execution":{"iopub.status.busy":"2022-07-21T02:21:10.976694Z","iopub.execute_input":"2022-07-21T02:21:10.977158Z","iopub.status.idle":"2022-07-21T02:21:10.984260Z","shell.execute_reply.started":"2022-07-21T02:21:10.977123Z","shell.execute_reply":"2022-07-21T02:21:10.982809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Iterative imputation with RandomForestRegressor\n\nI will only use columns that did not have any missing values to start with (exclusing columns I already dropped)\n\nto predict missing Age values\n\n","metadata":{"id":"9u6TlR3mTrag"}},{"cell_type":"code","source":"df_rfr.info()","metadata":{"executionInfo":{"elapsed":132,"status":"ok","timestamp":1651975635311,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"6ANxJDIDVBGr","outputId":"a860eadd-6af7-44b6-b022-1ad0acc1c70d","execution":{"iopub.status.busy":"2022-07-21T02:21:10.986132Z","iopub.execute_input":"2022-07-21T02:21:10.986926Z","iopub.status.idle":"2022-07-21T02:21:11.004554Z","shell.execute_reply.started":"2022-07-21T02:21:10.986879Z","shell.execute_reply":"2022-07-21T02:21:11.003131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\n# define features to predict the age\nage_df = df_rfr[['Age', 'Pclass', 'Sex', 'SibSp', 'Parch']]\n\n# separate age_df into train (with Age) and test (Age is NaN) and to ndarray\nhave_age = age_df[age_df['Age'].notnull()].values\nno_age = age_df[age_df['Age'].isnull()].values\n\n# separate train data to X and y\n\nX = have_age[:, 1:]\ny = have_age[:, 0]","metadata":{"executionInfo":{"elapsed":120,"status":"ok","timestamp":1651975635313,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"ozX51mDIVBZL","execution":{"iopub.status.busy":"2022-07-21T02:21:11.006359Z","iopub.execute_input":"2022-07-21T02:21:11.007135Z","iopub.status.idle":"2022-07-21T02:21:11.377669Z","shell.execute_reply.started":"2022-07-21T02:21:11.007089Z","shell.execute_reply":"2022-07-21T02:21:11.376487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build age prediction model with RandomForest\n\nrfr = RandomForestRegressor(random_state = 0, n_estimators = 100, n_jobs = -1)\nrfr.fit(X, y)","metadata":{"executionInfo":{"elapsed":1152,"status":"ok","timestamp":1651975636348,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"2HtHGjPPVBkr","outputId":"2357c4ed-cbc8-48df-88fe-3bee6d7f3718","execution":{"iopub.status.busy":"2022-07-21T02:21:11.379367Z","iopub.execute_input":"2022-07-21T02:21:11.380090Z","iopub.status.idle":"2022-07-21T02:21:11.643049Z","shell.execute_reply.started":"2022-07-21T02:21:11.380044Z","shell.execute_reply":"2022-07-21T02:21:11.641693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use the model to predict the Age for test data\n\n# fix df_copy later\nage_predicted = rfr.predict(no_age[:, 1:])\n\n# age_predicted = np.round_(age_predicted, decimals=1)\nprint(age_predicted)\n","metadata":{"executionInfo":{"elapsed":231,"status":"ok","timestamp":1651975636350,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"xJ4iZ6JvZiTu","outputId":"583ebf31-42de-4364-bb9d-f640e4fa3b47","execution":{"iopub.status.busy":"2022-07-21T02:21:11.644307Z","iopub.execute_input":"2022-07-21T02:21:11.644616Z","iopub.status.idle":"2022-07-21T02:21:11.755545Z","shell.execute_reply.started":"2022-07-21T02:21:11.644589Z","shell.execute_reply":"2022-07-21T02:21:11.754600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replace missing Age values with age_predicted\n\ndf_rfr.loc[(df_rfr['Age'].isnull()), 'Age'] = age_predicted","metadata":{"executionInfo":{"elapsed":211,"status":"ok","timestamp":1651975636353,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"e11FUi2Lb34m","execution":{"iopub.status.busy":"2022-07-21T02:21:11.756795Z","iopub.execute_input":"2022-07-21T02:21:11.757291Z","iopub.status.idle":"2022-07-21T02:21:11.762577Z","shell.execute_reply.started":"2022-07-21T02:21:11.757254Z","shell.execute_reply":"2022-07-21T02:21:11.761592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr['Age'] = df_rfr['Age'].map(lambda x: round(x, 2))","metadata":{"executionInfo":{"elapsed":212,"status":"ok","timestamp":1651975636357,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"kdPtxehOn9sF","execution":{"iopub.status.busy":"2022-07-21T02:21:11.764086Z","iopub.execute_input":"2022-07-21T02:21:11.764737Z","iopub.status.idle":"2022-07-21T02:21:11.777907Z","shell.execute_reply.started":"2022-07-21T02:21:11.764703Z","shell.execute_reply":"2022-07-21T02:21:11.776767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr['Age'].isnull().sum()","metadata":{"executionInfo":{"elapsed":213,"status":"ok","timestamp":1651975636361,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"0obvXCa8Z6Lg","outputId":"7ad1e96b-5aff-4c3f-aa0c-d9b3e8112e24","execution":{"iopub.status.busy":"2022-07-21T02:21:11.779557Z","iopub.execute_input":"2022-07-21T02:21:11.780321Z","iopub.status.idle":"2022-07-21T02:21:11.790506Z","shell.execute_reply.started":"2022-07-21T02:21:11.780282Z","shell.execute_reply":"2022-07-21T02:21:11.788821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compare Age feature before imputation and after\n\nprint('Before imputation')\nprint(df['Age'].describe())\nprint(df['Age'].value_counts().sort_index())\nprint(\" \")\n\nprint('Age feature for df_rfr')\nprint(df_rfr['Age'].describe())\nprint(df_rfr['Age'].value_counts().sort_index())","metadata":{"executionInfo":{"elapsed":198,"status":"ok","timestamp":1651975636365,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"zs7iAz3Vo4la","outputId":"b89487d8-480d-45bb-ea95-118284905477","execution":{"iopub.status.busy":"2022-07-21T02:21:11.792158Z","iopub.execute_input":"2022-07-21T02:21:11.793093Z","iopub.status.idle":"2022-07-21T02:21:11.824022Z","shell.execute_reply.started":"2022-07-21T02:21:11.793045Z","shell.execute_reply":"2022-07-21T02:21:11.822595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking if there are any outliers or unknown values","metadata":{"id":"2ad16a02"}},{"cell_type":"code","source":"df_rfr.describe().T","metadata":{"executionInfo":{"elapsed":185,"status":"ok","timestamp":1651975636369,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"8716a166","outputId":"c606ae36-b3aa-4acf-e88e-1c3bc4b7f384","execution":{"iopub.status.busy":"2022-07-21T02:21:11.832864Z","iopub.execute_input":"2022-07-21T02:21:11.836740Z","iopub.status.idle":"2022-07-21T02:21:11.873573Z","shell.execute_reply.started":"2022-07-21T02:21:11.836647Z","shell.execute_reply":"2022-07-21T02:21:11.872189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df_rfr.columns:\n  print(\"Unique values for column: \" + i)\n  print(df_rfr[i].value_counts(dropna=False))\n  print(\" \")","metadata":{"executionInfo":{"elapsed":180,"status":"ok","timestamp":1651975636372,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"fc4aafb9","outputId":"ea5aeaf1-4e38-40ec-aa77-2186e5388b83","execution":{"iopub.status.busy":"2022-07-21T02:21:11.875213Z","iopub.execute_input":"2022-07-21T02:21:11.875889Z","iopub.status.idle":"2022-07-21T02:21:11.895343Z","shell.execute_reply.started":"2022-07-21T02:21:11.875833Z","shell.execute_reply":"2022-07-21T02:21:11.894183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr['Fare'].value_counts().sort_index()","metadata":{"executionInfo":{"elapsed":167,"status":"ok","timestamp":1651975636375,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"aa78d161","outputId":"89d9fcf0-2d2d-443e-9701-0c848c7ccc9b","execution":{"iopub.status.busy":"2022-07-21T02:21:11.896624Z","iopub.execute_input":"2022-07-21T02:21:11.897110Z","iopub.status.idle":"2022-07-21T02:21:11.907042Z","shell.execute_reply.started":"2022-07-21T02:21:11.897081Z","shell.execute_reply":"2022-07-21T02:21:11.906072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The .min() value for Fare feature is 0 and it appears 17 times\n\nit is questionable whether this is because the passenger did not pay at all \n\nor 0 value because the fare paid by the 17 passengers are unknown","metadata":{"id":"35c00871"}},{"cell_type":"code","source":"fare_unknown = df_rfr[df_rfr['Fare'] == 0]\nfare_unknown","metadata":{"executionInfo":{"elapsed":159,"status":"ok","timestamp":1651975636384,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"5ac52b0c","outputId":"5e330672-6ba4-47e7-a4ee-bcad202d78a0","execution":{"iopub.status.busy":"2022-07-21T02:21:11.908527Z","iopub.execute_input":"2022-07-21T02:21:11.909039Z","iopub.status.idle":"2022-07-21T02:21:11.926858Z","shell.execute_reply.started":"2022-07-21T02:21:11.909008Z","shell.execute_reply":"2022-07-21T02:21:11.925963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some of the passengers with 'unknown' fare are Pclass 1 (i.e. first class) passengers\n\nand it is very unlikely that first passengers did not pay or pay little fare (unless they were invited passegners or so) \n\nso I assume that 'fare_unknown' passengers, how much fare they paid is unknown, rather than they did not pay any fare at all","metadata":{"id":"e01389c6"}},{"cell_type":"markdown","source":"In my earlier notebooks/versions I left the 0 fare values as they were, and replaced 1 NaN value with the mean Fare value \n\nbut in my previous notebook/version I tested and compared different ways to impute 0 Fare values\n\nAs the result, I impute 0 Fare values by median of Pclass and Embarked features\n\nI have to get the row which had NaN in the original dataset, so I can also treat it as a row with 0 Fare value and impute","metadata":{"id":"lytm8L7-QzxZ"}},{"cell_type":"code","source":"# df2 is the copied dataframe of the original dataframe\n\ndf2[df2['Fare'].isnull()].index","metadata":{"id":"QVrmpVPvIN1E","executionInfo":{"status":"ok","timestamp":1651975636387,"user_tz":-540,"elapsed":154,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"3a30280e-091f-4c73-af68-8769d46c118c","execution":{"iopub.status.busy":"2022-07-21T02:21:11.928183Z","iopub.execute_input":"2022-07-21T02:21:11.928706Z","iopub.status.idle":"2022-07-21T02:21:11.936043Z","shell.execute_reply.started":"2022-07-21T02:21:11.928676Z","shell.execute_reply":"2022-07-21T02:21:11.935248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Index of the row with missing Fare value is 1043 and Fare replaced by 0 value\n\nso I can transform all 0 Fare value rows together","metadata":{"id":"u7mtiZ0-JQGJ"}},{"cell_type":"code","source":"df_rfr3 = df_rfr.copy()","metadata":{"executionInfo":{"elapsed":146,"status":"ok","timestamp":1651975636391,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"IJ8817g1RPyC","execution":{"iopub.status.busy":"2022-07-21T02:21:11.937341Z","iopub.execute_input":"2022-07-21T02:21:11.937842Z","iopub.status.idle":"2022-07-21T02:21:11.943755Z","shell.execute_reply.started":"2022-07-21T02:21:11.937805Z","shell.execute_reply":"2022-07-21T02:21:11.942894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr3.loc[1043, :]\n\n# Fare value here is the mean value imputed","metadata":{"id":"tfgqx4ixLtVS","executionInfo":{"status":"ok","timestamp":1651975636394,"user_tz":-540,"elapsed":147,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"387b5314-31c8-44b5-d341-dc1a12329ce0","execution":{"iopub.status.busy":"2022-07-21T02:21:11.945009Z","iopub.execute_input":"2022-07-21T02:21:11.945558Z","iopub.status.idle":"2022-07-21T02:21:11.963988Z","shell.execute_reply.started":"2022-07-21T02:21:11.945525Z","shell.execute_reply":"2022-07-21T02:21:11.959379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr3.loc[1043, 'Fare'] = 0","metadata":{"id":"9c968JaWIl5Z","executionInfo":{"status":"ok","timestamp":1651975636398,"user_tz":-540,"elapsed":139,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"execution":{"iopub.status.busy":"2022-07-21T02:21:11.965510Z","iopub.execute_input":"2022-07-21T02:21:11.966169Z","iopub.status.idle":"2022-07-21T02:21:11.975514Z","shell.execute_reply.started":"2022-07-21T02:21:11.966128Z","shell.execute_reply":"2022-07-21T02:21:11.973801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df_rfr3[df_rfr3['Fare'] == 0])","metadata":{"id":"b7uMNhRyJz3i","executionInfo":{"status":"ok","timestamp":1651975636402,"user_tz":-540,"elapsed":141,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"af423083-d4cf-41fb-d997-f6d317890b53","execution":{"iopub.status.busy":"2022-07-21T02:21:11.977092Z","iopub.execute_input":"2022-07-21T02:21:11.978082Z","iopub.status.idle":"2022-07-21T02:21:11.988971Z","shell.execute_reply.started":"2022-07-21T02:21:11.978037Z","shell.execute_reply":"2022-07-21T02:21:11.987614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fare is determined by Pclass as well as by Embarked\n\nso I impute 0 Fare value with the median Fare of Pclass and Embarked\n\nI have dropped Embarked column earlier in the process so I put it back temporarily to get the median value with Pclass","metadata":{"id":"dkD82wtEsxu7"}},{"cell_type":"code","source":"df_rfr3.head()","metadata":{"executionInfo":{"elapsed":132,"status":"ok","timestamp":1651975636407,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"tbBiG2GtrjIH","outputId":"09bdaccf-a29c-489d-f08f-f97c42b31bb4","execution":{"iopub.status.busy":"2022-07-21T02:21:11.990619Z","iopub.execute_input":"2022-07-21T02:21:11.991810Z","iopub.status.idle":"2022-07-21T02:21:12.013461Z","shell.execute_reply.started":"2022-07-21T02:21:11.991776Z","shell.execute_reply":"2022-07-21T02:21:12.012255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr3['Embarked'] = df2['Embarked']\n\ndf_rfr3.head()","metadata":{"id":"FJCyp2ldPq05","executionInfo":{"status":"ok","timestamp":1651975636410,"user_tz":-540,"elapsed":131,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"61c49517-3d69-4054-c76d-2070cc3d0708","execution":{"iopub.status.busy":"2022-07-21T02:21:12.015216Z","iopub.execute_input":"2022-07-21T02:21:12.016482Z","iopub.status.idle":"2022-07-21T02:21:12.039581Z","shell.execute_reply.started":"2022-07-21T02:21:12.016437Z","shell.execute_reply":"2022-07-21T02:21:12.038129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check the mean and median of Fare values by Pclass and Embarked","metadata":{"id":"E1EaCWytRs55"}},{"cell_type":"code","source":"data2 = df_rfr3.loc[df_rfr3['Fare'] != 0,:].groupby(['Pclass', 'Embarked']).agg(['mean', 'median', 'count'])['Fare']\nprint(data2)","metadata":{"id":"ZDE175U0DTMG","executionInfo":{"status":"ok","timestamp":1651975636413,"user_tz":-540,"elapsed":124,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"outputId":"96c614fa-4546-4913-cda2-6e23d1f4cc1a","execution":{"iopub.status.busy":"2022-07-21T02:21:12.041360Z","iopub.execute_input":"2022-07-21T02:21:12.041835Z","iopub.status.idle":"2022-07-21T02:21:12.072348Z","shell.execute_reply.started":"2022-07-21T02:21:12.041792Z","shell.execute_reply":"2022-07-21T02:21:12.071422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df_rfr3['Pclass'].unique():\n  for location in df_rfr3['Embarked'].unique():\n\n    df_rfr3.loc[(df_rfr3['Fare'] == 0) & (df_rfr3['Pclass'] == i) & (df_rfr3['Embarked'] == location), 'Fare'] = df_rfr3.loc[(df_rfr3['Fare'] == 0) & (df_rfr3['Pclass'] == i) & (df_rfr3['Embarked'] == location), 'Fare'].map(\n        lambda x: df_rfr3[(df_rfr3['Fare'] != 0) & (df_rfr3['Pclass'] == i) & (df_rfr3['Embarked'] == location)]['Fare'].median().astype(float)\n    )\n    \n    \n    # print(i,location) \n    # print(df_rfr3[(df_rfr3['Fare'] != 0) & (df_rfr3['Pclass'] == i) & (df_rfr3['Embarked'] == location)]['Fare'].median())","metadata":{"id":"vkaBgLnlalb_","executionInfo":{"status":"ok","timestamp":1651975637116,"user_tz":-540,"elapsed":812,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"execution":{"iopub.status.busy":"2022-07-21T02:21:12.073449Z","iopub.execute_input":"2022-07-21T02:21:12.073936Z","iopub.status.idle":"2022-07-21T02:21:12.164362Z","shell.execute_reply.started":"2022-07-21T02:21:12.073906Z","shell.execute_reply":"2022-07-21T02:21:12.163224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# All 0 Fare values have been imputed\n\ndf_rfr3[df_rfr3['Fare'] == 0]","metadata":{"executionInfo":{"elapsed":120,"status":"ok","timestamp":1651975637119,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"outputId":"ea4faac8-309f-42e2-d511-94b9ec7616e8","id":"9z9UMCqXRs6E","execution":{"iopub.status.busy":"2022-07-21T02:21:12.166077Z","iopub.execute_input":"2022-07-21T02:21:12.166424Z","iopub.status.idle":"2022-07-21T02:21:12.176809Z","shell.execute_reply.started":"2022-07-21T02:21:12.166393Z","shell.execute_reply":"2022-07-21T02:21:12.175960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Before imputation\")\nprint(data2)\nprint(\"\")\nprint(\"After imputation\")\nprint(df_rfr3.groupby(['Pclass', 'Embarked']).agg(['mean', 'median', 'count'])['Fare'])","metadata":{"executionInfo":{"elapsed":115,"status":"ok","timestamp":1651975637123,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"outputId":"3271a5c4-ab27-4a81-f1a4-09d82d8edfa6","id":"zWk7T5z-Rs6G","execution":{"iopub.status.busy":"2022-07-21T02:21:12.178371Z","iopub.execute_input":"2022-07-21T02:21:12.179486Z","iopub.status.idle":"2022-07-21T02:21:12.220759Z","shell.execute_reply.started":"2022-07-21T02:21:12.179446Z","shell.execute_reply":"2022-07-21T02:21:12.219360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr3['Fare'] = df_rfr3['Fare'].astype('float64')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T02:21:12.223516Z","iopub.execute_input":"2022-07-21T02:21:12.224410Z","iopub.status.idle":"2022-07-21T02:21:12.230629Z","shell.execute_reply.started":"2022-07-21T02:21:12.224361Z","shell.execute_reply":"2022-07-21T02:21:12.229745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rfr3 = df_rfr3.drop(columns='Embarked', axis=1)","metadata":{"id":"eRCNNEcWvAH7","executionInfo":{"status":"ok","timestamp":1651975637126,"user_tz":-540,"elapsed":60,"user":{"displayName":"Saya M","userId":"06841214524322532382"}},"execution":{"iopub.status.busy":"2022-07-21T02:21:12.232051Z","iopub.execute_input":"2022-07-21T02:21:12.232472Z","iopub.status.idle":"2022-07-21T02:21:12.245216Z","shell.execute_reply.started":"2022-07-21T02:21:12.232430Z","shell.execute_reply":"2022-07-21T02:21:12.243047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"New Family_group feature\n\nI will combine SibSp and Parch features and make Family_group feature","metadata":{}},{"cell_type":"code","source":"df_rfr4 = df_rfr3.copy()\ndf_rfr4.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T02:21:12.246837Z","iopub.execute_input":"2022-07-21T02:21:12.249754Z","iopub.status.idle":"2022-07-21T02:21:12.265999Z","shell.execute_reply.started":"2022-07-21T02:21:12.249700Z","shell.execute_reply":"2022-07-21T02:21:12.264842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the original train dataset to see how SibSp and Parch features are correlated to Survival rate\n\ntrain_data['Survived'] = Survived_data\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T02:21:12.268799Z","iopub.execute_input":"2022-07-21T02:21:12.269282Z","iopub.status.idle":"2022-07-21T02:21:12.291675Z","shell.execute_reply.started":"2022-07-21T02:21:12.269222Z","shell.execute_reply":"2022-07-21T02:21:12.290173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Survived_SibSp = train_data[['SibSp', 'Survived']].groupby('SibSp')['Survived'].agg(['count', 'sum'])\nSurvived_SibSp['Survived_rate'] = train_data[['SibSp', 'Survived']].groupby('SibSp')['Survived'].apply(lambda x: (x.sum()/x.count())*100)\n\nSurvived_SibSp","metadata":{"execution":{"iopub.status.busy":"2022-07-21T02:21:12.293561Z","iopub.execute_input":"2022-07-21T02:21:12.294738Z","iopub.status.idle":"2022-07-21T02:21:12.319475Z","shell.execute_reply.started":"2022-07-21T02:21:12.294682Z","shell.execute_reply":"2022-07-21T02:21:12.318192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Survived_Parch = train_data[['Parch', 'Survived']].groupby('Parch')['Survived'].agg(['count', 'sum'])\nSurvived_Parch['Survived_rate'] = train_data[['Parch', 'Survived']].groupby('Parch')['Survived'].apply(lambda x: (x.sum()/x.count())*100)\n\nSurvived_Parch","metadata":{"execution":{"iopub.status.busy":"2022-07-21T02:21:12.321381Z","iopub.execute_input":"2022-07-21T02:21:12.322289Z","iopub.status.idle":"2022-07-21T02:21:12.344214Z","shell.execute_reply.started":"2022-07-21T02:21:12.322223Z","shell.execute_reply":"2022-07-21T02:21:12.343005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I can see that there is a strong correlation between the number of SibSp/Parch and the Survival rate,\n\nso rather than keeping the two features independent, I will combine and assign as the Family_group feature with 3 different groups based on the respective survival rate","metadata":{}},{"cell_type":"code","source":"df_rfr4['Family']=df_rfr4['SibSp'] + df_rfr4['Parch'] + 1\n\ndf_rfr4.loc[(df_rfr4['Family']>=2) & (df_rfr4['Family']<=4), 'Family_group'] = 2\ndf_rfr4.loc[(df_rfr4['Family']>=5) & (df_rfr4['Family']<=7) | (df_rfr4['Family']==1), 'Family_group'] = 1  \ndf_rfr4.loc[(df_rfr4['Family']>=8), 'Family_group'] = 0\n\n# drop SibSp, Parch and Family columns \ndf_rfr4 = df_rfr4.drop(columns = ['SibSp', 'Parch', 'Family'])\n\ndf_rfr4.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T02:21:12.346127Z","iopub.execute_input":"2022-07-21T02:21:12.347022Z","iopub.status.idle":"2022-07-21T02:21:12.371522Z","shell.execute_reply.started":"2022-07-21T02:21:12.346837Z","shell.execute_reply":"2022-07-21T02:21:12.370327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Building the model","metadata":{"id":"0e337aba"}},{"cell_type":"markdown","source":"Now that full dataset (train and test combined) has been preprocessed \n\nI will now separate the full dataset back to train and test datasets then build the model","metadata":{"id":"3ef2d00d"}},{"cell_type":"markdown","source":"Build the model\n\nI will build a **Light GBM model** with StratifiedKFold this time","metadata":{"id":"3530168b"}},{"cell_type":"code","source":"# separate the full dataset back to train and test\n\ntrain = df_rfr4.iloc[:891, :]\ntest = df_rfr4.iloc[891:,:]","metadata":{"executionInfo":{"elapsed":61,"status":"ok","timestamp":1651975637130,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"w18IQb-Yufsi","execution":{"iopub.status.busy":"2022-07-21T02:21:12.374217Z","iopub.execute_input":"2022-07-21T02:21:12.377593Z","iopub.status.idle":"2022-07-21T02:21:12.383168Z","shell.execute_reply.started":"2022-07-21T02:21:12.377550Z","shell.execute_reply":"2022-07-21T02:21:12.381858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# adding back the Survived feature \n# that I separated from train_data earlier before datasets concat\ntrain['Survived'] = Survived_data\n\ntrain.info()","metadata":{"executionInfo":{"elapsed":62,"status":"ok","timestamp":1651975637133,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"outputId":"28a4622b-2be6-4856-b212-38ca3fe1344f","id":"ULJ-irtVufsk","execution":{"iopub.status.busy":"2022-07-21T02:21:12.384761Z","iopub.execute_input":"2022-07-21T02:21:12.385133Z","iopub.status.idle":"2022-07-21T02:21:12.404594Z","shell.execute_reply.started":"2022-07-21T02:21:12.385102Z","shell.execute_reply":"2022-07-21T02:21:12.403283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train.drop('Survived', axis=1)\ny = train['Survived']","metadata":{"executionInfo":{"elapsed":47,"status":"ok","timestamp":1651975637136,"user":{"displayName":"Saya M","userId":"06841214524322532382"},"user_tz":-540},"id":"5g8_EsSKufsn","execution":{"iopub.status.busy":"2022-07-21T02:21:12.406068Z","iopub.execute_input":"2022-07-21T02:21:12.407130Z","iopub.status.idle":"2022-07-21T02:21:12.413041Z","shell.execute_reply.started":"2022-07-21T02:21:12.407092Z","shell.execute_reply":"2022-07-21T02:21:12.412291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nimport lightgbm as lgb\n\ny_data_preds = []\nmodels = []\noof_data_train = np.zeros((len(X),))\ncv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\nparams= {\n    'objective':'binary'\n}\n\n# adding y to cv.split() as StratifiedKFold folds based on y distribution\nfor fold_id, (train_index, valid_index) in enumerate(cv.split(X, y)):\n    X_tr = X.loc[train_index, :]\n    X_val = X.loc[valid_index, :]\n    y_tr = y[train_index]\n    y_val = y[valid_index]\n    \n    lgb_data_train = lgb.Dataset(X_tr, y_tr)\n    lgb_data_eval = lgb.Dataset(X_val, y_val, reference=lgb_data_train)\n    \n    model = lgb.train(params, lgb_data_train, valid_sets=lgb_data_eval,\n                 verbose_eval=10, #print every 10 train\n                 num_boost_round=1000, #repeat train 1000 times\n                 early_stopping_rounds=10) \n    \n    # prediction for validation data/index\n    oof_data_train[valid_index] = model.predict(X_val, num_iteration=model.best_iteration)\n    # prediction for test dataset\n    y_data_pred = model.predict(test, num_iteration=model.best_iteration)\n    \n    y_data_preds.append(y_data_pred)\n    models.append(model)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T02:21:12.414352Z","iopub.execute_input":"2022-07-21T02:21:12.415317Z","iopub.status.idle":"2022-07-21T02:21:13.734935Z","shell.execute_reply.started":"2022-07-21T02:21:12.415276Z","shell.execute_reply":"2022-07-21T02:21:13.734017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Calculate the score (i.e. binary_logloss) and mean of each model","metadata":{}},{"cell_type":"code","source":"scores = [\n    m.best_score['valid_0']['binary_logloss'] for m in models\n]\nscore = sum(scores) / len(scores)\nprint(scores)\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T02:21:13.736807Z","iopub.execute_input":"2022-07-21T02:21:13.737651Z","iopub.status.idle":"2022-07-21T02:21:13.746413Z","shell.execute_reply.started":"2022-07-21T02:21:13.737606Z","shell.execute_reply":"2022-07-21T02:21:13.745450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Calculate the accuracy score for oof_data_train\n\n(accuracy of the model with train dataset prediction)","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\ny_data_pred_oof = (oof_data_train > 0.5).astype(int)\naccuracy_score(y, y_data_pred_oof)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T02:21:13.750385Z","iopub.execute_input":"2022-07-21T02:21:13.750793Z","iopub.status.idle":"2022-07-21T02:21:13.762816Z","shell.execute_reply.started":"2022-07-21T02:21:13.750758Z","shell.execute_reply":"2022-07-21T02:21:13.761624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As I have already made prediction on test dataset with each fold,\n\nI will divide the sum of the prediction by the length of prediction list\n\n(which is the number of folds made) to get the submission data","metadata":{}},{"cell_type":"markdown","source":"## 4. Prediction on test dataset for submission","metadata":{}},{"cell_type":"code","source":"submission_prediction = sum(y_data_preds) / len(y_data_preds)\nsubmission_prediction = (submission_prediction > 0.5).astype(int)\nsubmission_prediction[:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T02:21:13.764117Z","iopub.execute_input":"2022-07-21T02:21:13.765029Z","iopub.status.idle":"2022-07-21T02:21:13.774282Z","shell.execute_reply.started":"2022-07-21T02:21:13.764993Z","shell.execute_reply":"2022-07-21T02:21:13.773171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Prediction submission","metadata":{}},{"cell_type":"code","source":"# PassengerId column for submission file\n\nid_column = test_data['PassengerId']","metadata":{"execution":{"iopub.status.busy":"2022-07-21T02:21:13.782488Z","iopub.execute_input":"2022-07-21T02:21:13.783114Z","iopub.status.idle":"2022-07-21T02:21:13.789108Z","shell.execute_reply.started":"2022-07-21T02:21:13.783065Z","shell.execute_reply":"2022-07-21T02:21:13.787985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df = pd.DataFrame(id_column)\n\nfinal_df['Survived'] = submission_prediction\n\nfinal_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T02:21:13.790707Z","iopub.execute_input":"2022-07-21T02:21:13.791362Z","iopub.status.idle":"2022-07-21T02:21:13.808731Z","shell.execute_reply.started":"2022-07-21T02:21:13.791317Z","shell.execute_reply":"2022-07-21T02:21:13.807520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T02:21:13.810316Z","iopub.execute_input":"2022-07-21T02:21:13.810728Z","iopub.status.idle":"2022-07-21T02:21:13.819146Z","shell.execute_reply.started":"2022-07-21T02:21:13.810682Z","shell.execute_reply":"2022-07-21T02:21:13.818133Z"},"trusted":true},"execution_count":null,"outputs":[]}]}