{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"The attributes have the following meaning:\n* **PassengerId**: a unique identifier for each passenger\n* **Survived**: that's the target, 0 means the passenger did not survive, while 1 means he/she survived.\n* **Pclass**: passenger class.\n* **Name**, **Sex**, **Age**: self-explanatory\n* **SibSp**: how many siblings & spouses of the passenger aboard the Titanic.\n* **Parch**: how many children & parents of the passenger aboard the Titanic.\n* **Ticket**: ticket id\n* **Fare**: price paid (in pounds)\n* **Cabin**: passenger's cabin number\n* **Embarked**: where the passenger embarked the Titanic","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.model_selection import train_test_split\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-07T10:46:55.386531Z","iopub.execute_input":"2022-07-07T10:46:55.388142Z","iopub.status.idle":"2022-07-07T10:46:56.210732Z","shell.execute_reply.started":"2022-07-07T10:46:55.387959Z","shell.execute_reply":"2022-07-07T10:46:56.209262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/tabular-playground-series-apr-2021/train.csv')\ntest_df = pd.read_csv('../input/tabular-playground-series-apr-2021/test.csv')                       ","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:46:56.212793Z","iopub.execute_input":"2022-07-07T10:46:56.213239Z","iopub.status.idle":"2022-07-07T10:46:56.845375Z","shell.execute_reply.started":"2022-07-07T10:46:56.213206Z","shell.execute_reply":"2022-07-07T10:46:56.844126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()\nprint('-.'*40)\nprint(test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:46:56.846896Z","iopub.execute_input":"2022-07-07T10:46:56.847386Z","iopub.status.idle":"2022-07-07T10:46:56.930895Z","shell.execute_reply.started":"2022-07-07T10:46:56.847351Z","shell.execute_reply":"2022-07-07T10:46:56.929471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:46:56.933502Z","iopub.execute_input":"2022-07-07T10:46:56.934112Z","iopub.status.idle":"2022-07-07T10:46:56.962286Z","shell.execute_reply.started":"2022-07-07T10:46:56.934069Z","shell.execute_reply":"2022-07-07T10:46:56.961004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_to_drop = ['PassengerId', 'Name', 'Ticket']\nnum_features = ['Age', 'Fare'] #2\ncat_features = ['Pclass', 'Sex', 'SibSp', 'Parch', 'Embarked', 'IsCabin', 'AgeBucket'] #5\nfeatures = num_features + cat_features\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:46:56.966100Z","iopub.execute_input":"2022-07-07T10:46:56.966453Z","iopub.status.idle":"2022-07-07T10:46:56.972977Z","shell.execute_reply.started":"2022-07-07T10:46:56.966421Z","shell.execute_reply":"2022-07-07T10:46:56.971730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def proprocess_fn(df):\n    #drop unuse cols\n    df.drop(columns=cols_to_drop, inplace=True)\n    #create new col\n    df['IsCabin'] = df['Cabin'].notna()\n    df[\"AgeBucket\"] = df[\"Age\"] // 15 * 15\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:46:56.974452Z","iopub.execute_input":"2022-07-07T10:46:56.975157Z","iopub.status.idle":"2022-07-07T10:46:56.988327Z","shell.execute_reply.started":"2022-07-07T10:46:56.975121Z","shell.execute_reply":"2022-07-07T10:46:56.986816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = proprocess_fn(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:46:56.990311Z","iopub.execute_input":"2022-07-07T10:46:56.990779Z","iopub.status.idle":"2022-07-07T10:46:57.020739Z","shell.execute_reply.started":"2022-07-07T10:46:56.990733Z","shell.execute_reply":"2022-07-07T10:46:57.019604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = proprocess_fn(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:46:57.022862Z","iopub.execute_input":"2022-07-07T10:46:57.023331Z","iopub.status.idle":"2022-07-07T10:46:57.054075Z","shell.execute_reply.started":"2022-07-07T10:46:57.023283Z","shell.execute_reply":"2022-07-07T10:46:57.053055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(train_df[features].isna().sum()/len(train_df)*100).sort_values()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:46:57.056378Z","iopub.execute_input":"2022-07-07T10:46:57.056948Z","iopub.status.idle":"2022-07-07T10:46:57.097485Z","shell.execute_reply.started":"2022-07-07T10:46:57.056917Z","shell.execute_reply":"2022-07-07T10:46:57.096240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(16,4))\nplt.suptitle('Age Distribution', fontsize=14)\nsns.histplot(train_df.Age, ax=ax1)\nsns.boxplot(data=train_df, x='Age')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:46:57.099609Z","iopub.execute_input":"2022-07-07T10:46:57.101137Z","iopub.status.idle":"2022-07-07T10:46:57.641519Z","shell.execute_reply.started":"2022-07-07T10:46:57.101092Z","shell.execute_reply":"2022-07-07T10:46:57.640279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(16,4))\nplt.suptitle('Survival Age Distribution', fontsize=14)\nsns.histplot(data=train_df, x='Age',hue='Survived', ax=ax1)\nsns.boxplot(data=train_df, x='Age',hue='Survived', ax=ax2)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:46:57.644584Z","iopub.execute_input":"2022-07-07T10:46:57.645340Z","iopub.status.idle":"2022-07-07T10:46:58.312532Z","shell.execute_reply.started":"2022-07-07T10:46:57.645294Z","shell.execute_reply":"2022-07-07T10:46:58.311683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(16,4))\nplt.suptitle('Survival Fare Distribution', fontsize=14)\nsns.histplot(data=train_df, x='Fare',hue='Survived', ax=ax1)\nsns.boxplot(data=train_df, x='Fare',hue='Survived', ax=ax2)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:46:58.313874Z","iopub.execute_input":"2022-07-07T10:46:58.314382Z","iopub.status.idle":"2022-07-07T10:47:02.076760Z","shell.execute_reply.started":"2022-07-07T10:46:58.314352Z","shell.execute_reply":"2022-07-07T10:47:02.075399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer, SimpleImputer\nimport xgboost\n\nnum_pipeline = Pipeline([\n    ('imputer',IterativeImputer(\n                estimator=xgboost.XGBRegressor(\n                n_estimators=100,\n                random_state=1),\n#                 tree_method='gpu_hist'),\n                missing_values=np.nan,\n                max_iter=5,\n                initial_strategy='mean',\n                imputation_order='ascending',\n                verbose=0,\n                random_state=42)),\n    ('scaler',StandardScaler())\n])\n\ncat_pipeline = Pipeline([\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('cat_encoder', OneHotEncoder(sparse=False))\n])\n\n#combine pipeline\nfrom sklearn.compose import ColumnTransformer\n\npreprocess_pipeline = ColumnTransformer([\n    ('num', num_pipeline, num_features),\n    ('cat', cat_pipeline, cat_features)\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:47:02.078824Z","iopub.execute_input":"2022-07-07T10:47:02.079183Z","iopub.status.idle":"2022-07-07T10:47:02.372310Z","shell.execute_reply.started":"2022-07-07T10:47:02.079147Z","shell.execute_reply":"2022-07-07T10:47:02.371066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = preprocess_pipeline.fit_transform(train_df[features])\ny_train = train_df.Survived","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:47:02.373878Z","iopub.execute_input":"2022-07-07T10:47:02.374299Z","iopub.status.idle":"2022-07-07T10:47:27.321604Z","shell.execute_reply.started":"2022-07-07T10:47:02.374257Z","shell.execute_reply":"2022-07-07T10:47:27.319359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\nxgb = xgboost.XGBClassifier(n_estimators=300, learning_rate=0.03)\nxgb_scores = cross_val_score(xgb, X_train, y_train, scoring='accuracy', cv=3, n_jobs=-1)\nxgb_score = np.mean(xgb_scores)\nxgb_score","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:47:27.326844Z","iopub.execute_input":"2022-07-07T10:47:27.327532Z","iopub.status.idle":"2022-07-07T10:48:18.320928Z","shell.execute_reply.started":"2022-07-07T10:47:27.327469Z","shell.execute_reply":"2022-07-07T10:48:18.319655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:48:18.323127Z","iopub.execute_input":"2022-07-07T10:48:18.323492Z","iopub.status.idle":"2022-07-07T10:48:43.443701Z","shell.execute_reply.started":"2022-07-07T10:48:18.323458Z","shell.execute_reply":"2022-07-07T10:48:43.442806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = preprocess_pipeline.transform(test_df[features])\ny_pred = xgb.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:48:43.445228Z","iopub.execute_input":"2022-07-07T10:48:43.445872Z","iopub.status.idle":"2022-07-07T10:48:44.233410Z","shell.execute_reply.started":"2022-07-07T10:48:43.445833Z","shell.execute_reply":"2022-07-07T10:48:44.232452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.read_csv('../input/tabular-playground-series-apr-2021/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:51:19.411184Z","iopub.execute_input":"2022-07-07T10:51:19.411707Z","iopub.status.idle":"2022-07-07T10:51:19.454041Z","shell.execute_reply.started":"2022-07-07T10:51:19.411660Z","shell.execute_reply":"2022-07-07T10:51:19.452665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'PassengerId':submission_df.PassengerId, 'Survived':y_pred})\noutput.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:52:11.709757Z","iopub.execute_input":"2022-07-07T10:52:11.710129Z","iopub.status.idle":"2022-07-07T10:52:11.725580Z","shell.execute_reply.started":"2022-07-07T10:52:11.710099Z","shell.execute_reply":"2022-07-07T10:52:11.724540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T10:52:35.050700Z","iopub.execute_input":"2022-07-07T10:52:35.051101Z","iopub.status.idle":"2022-07-07T10:52:35.226099Z","shell.execute_reply.started":"2022-07-07T10:52:35.051066Z","shell.execute_reply":"2022-07-07T10:52:35.224579Z"},"trusted":true},"execution_count":null,"outputs":[]}]}