{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-31T21:45:26.027356Z","iopub.execute_input":"2022-07-31T21:45:26.027817Z","iopub.status.idle":"2022-07-31T21:45:26.036692Z","shell.execute_reply.started":"2022-07-31T21:45:26.027781Z","shell.execute_reply":"2022-07-31T21:45:26.035793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"![](https://www.encyclopedia-titanica.org/files/1/figure-one-side-view.gif)","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/tabular-playground-series-apr-2021/train.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:45:29.432730Z","iopub.execute_input":"2022-07-31T21:45:29.433150Z","iopub.status.idle":"2022-07-31T21:45:29.757804Z","shell.execute_reply.started":"2022-07-31T21:45:29.433113Z","shell.execute_reply":"2022-07-31T21:45:29.756507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Analyst","metadata":{}},{"cell_type":"code","source":"df.sample(10, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:45:32.332336Z","iopub.execute_input":"2022-07-31T21:45:32.333028Z","iopub.status.idle":"2022-07-31T21:45:32.378447Z","shell.execute_reply.started":"2022-07-31T21:45:32.332990Z","shell.execute_reply":"2022-07-31T21:45:32.375795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"Survived\"].mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:45:34.752929Z","iopub.execute_input":"2022-07-31T21:45:34.753351Z","iopub.status.idle":"2022-07-31T21:45:34.766182Z","shell.execute_reply.started":"2022-07-31T21:45:34.753315Z","shell.execute_reply":"2022-07-31T21:45:34.764809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('Sex').Survived.mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:45:36.904783Z","iopub.execute_input":"2022-07-31T21:45:36.905763Z","iopub.status.idle":"2022-07-31T21:45:36.937206Z","shell.execute_reply.started":"2022-07-31T21:45:36.905713Z","shell.execute_reply":"2022-07-31T21:45:36.936335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('Age').Survived.mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:45:38.968897Z","iopub.execute_input":"2022-07-31T21:45:38.969433Z","iopub.status.idle":"2022-07-31T21:45:38.988386Z","shell.execute_reply.started":"2022-07-31T21:45:38.969372Z","shell.execute_reply":"2022-07-31T21:45:38.987165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nplt.rcParams['figure.figsize'] = (12, 9)\n\nsns.violinplot(x='Survived', y='Age', data=df)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:45:41.091837Z","iopub.execute_input":"2022-07-31T21:45:41.092270Z","iopub.status.idle":"2022-07-31T21:45:42.424925Z","shell.execute_reply.started":"2022-07-31T21:45:41.092234Z","shell.execute_reply":"2022-07-31T21:45:42.423579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('Pclass').Survived.mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:45:45.725537Z","iopub.execute_input":"2022-07-31T21:45:45.727958Z","iopub.status.idle":"2022-07-31T21:45:45.739863Z","shell.execute_reply.started":"2022-07-31T21:45:45.727906Z","shell.execute_reply":"2022-07-31T21:45:45.738647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df.Parch > 0].Survived.mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:45:48.147759Z","iopub.execute_input":"2022-07-31T21:45:48.149929Z","iopub.status.idle":"2022-07-31T21:45:48.164983Z","shell.execute_reply.started":"2022-07-31T21:45:48.149887Z","shell.execute_reply":"2022-07-31T21:45:48.163804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df.SibSp > 0].Survived.mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:45:50.785048Z","iopub.execute_input":"2022-07-31T21:45:50.785486Z","iopub.status.idle":"2022-07-31T21:45:50.801041Z","shell.execute_reply.started":"2022-07-31T21:45:50.785454Z","shell.execute_reply":"2022-07-31T21:45:50.799989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df.SibSp == 0].Survived.mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:45:52.979375Z","iopub.execute_input":"2022-07-31T21:45:52.979838Z","iopub.status.idle":"2022-07-31T21:45:53.005064Z","shell.execute_reply.started":"2022-07-31T21:45:52.979799Z","shell.execute_reply":"2022-07-31T21:45:53.003842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('Embarked').Survived.mean().to_frame()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:45:55.430834Z","iopub.execute_input":"2022-07-31T21:45:55.432146Z","iopub.status.idle":"2022-07-31T21:45:55.457876Z","shell.execute_reply.started":"2022-07-31T21:45:55.432071Z","shell.execute_reply":"2022-07-31T21:45:55.456867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.Cabin.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:45:57.904320Z","iopub.execute_input":"2022-07-31T21:45:57.904821Z","iopub.status.idle":"2022-07-31T21:45:57.939056Z","shell.execute_reply.started":"2022-07-31T21:45:57.904781Z","shell.execute_reply":"2022-07-31T21:45:57.937919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['CabinLettre'] = df.Cabin.apply(lambda x: x[0] if not pd.isna(x) else x)\ndf.groupby('CabinLettre').Survived.agg(['count', 'mean'])","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:46:00.610304Z","iopub.execute_input":"2022-07-31T21:46:00.610992Z","iopub.status.idle":"2022-07-31T21:46:00.742189Z","shell.execute_reply.started":"2022-07-31T21:46:00.610944Z","shell.execute_reply":"2022-07-31T21:46:00.741045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['FamilyName'] = df.Name.apply(lambda x: x.split(',')[0])\nt_df = df.groupby('FamilyName').Survived.agg(['count', 'mean']).sort_values('count', ascending=False)\nt_df[(t_df['count'] > 1) & (t_df['count'] < 20)]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:46:03.650227Z","iopub.execute_input":"2022-07-31T21:46:03.650794Z","iopub.status.idle":"2022-07-31T21:46:03.800469Z","shell.execute_reply.started":"2022-07-31T21:46:03.650736Z","shell.execute_reply":"2022-07-31T21:46:03.798157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.pop('PassengerId')\ndf.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:46:06.369103Z","iopub.execute_input":"2022-07-31T21:46:06.369508Z","iopub.status.idle":"2022-07-31T21:46:06.414981Z","shell.execute_reply.started":"2022-07-31T21:46:06.369474Z","shell.execute_reply":"2022-07-31T21:46:06.413790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.corr()['Survived'].T","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:46:08.584266Z","iopub.execute_input":"2022-07-31T21:46:08.584717Z","iopub.status.idle":"2022-07-31T21:46:08.619411Z","shell.execute_reply.started":"2022-07-31T21:46:08.584678Z","shell.execute_reply":"2022-07-31T21:46:08.618215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(df.corr()[['Survived']].T)\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:46:10.652076Z","iopub.execute_input":"2022-07-31T21:46:10.652469Z","iopub.status.idle":"2022-07-31T21:46:10.978853Z","shell.execute_reply.started":"2022-07-31T21:46:10.652437Z","shell.execute_reply":"2022-07-31T21:46:10.977686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pre-processing.","metadata":{}},{"cell_type":"code","source":"import re\ndef extract_ticket_prefix(ticket):\n    if pd.isna(ticket):\n        return ticket\n    first_digit_search = re.search(r'\\d', ticket)\n    if first_digit_search:\n        return ticket[:first_digit_search.span()[0]].strip()\n    return None\n\ndef preprocess_dataframe(df):\n    df = df.copy()\n    df['CabinLetter'] = df.Cabin.apply(lambda v : v if pd.isna(v) else v[0])\n    df['FamilyId'] = df.Name.str.lower().str.split(', ').str[0] + df.CabinLetter + df.Embarked + df.Pclass.astype(str)\n    df['TicketPrefix'] = df.Ticket.apply(extract_ticket_prefix)\n    df['FirstName'] = df.Name.str.split(' ').apply(lambda vs : vs[1].lower())\n    df['Pclass'] = df['Pclass'].astype(str)\n    return df\n\ndf = pd.read_csv('/kaggle/input/tabular-playground-series-apr-2021/train.csv')\ndf = preprocess_dataframe(df)\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:49:58.710780Z","iopub.execute_input":"2022-07-31T21:49:58.711196Z","iopub.status.idle":"2022-07-31T21:50:00.386834Z","shell.execute_reply.started":"2022-07-31T21:49:58.711163Z","shell.execute_reply":"2022-07-31T21:50:00.385680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train the Survival Prediction Model.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ndf_train, df_test = train_test_split(df, test_size = 0.25, random_state = 0)\ndf_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-31T21:55:03.636667Z","iopub.execute_input":"2022-07-31T21:55:03.637099Z","iopub.status.idle":"2022-07-31T21:55:03.713578Z","shell.execute_reply.started":"2022-07-31T21:55:03.637057Z","shell.execute_reply":"2022-07-31T21:55:03.712417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUMERIC_FEATURES = ['Age', 'SibSp', 'Parch', 'Fare']\nnum_df = df_train[NUMERIC_FEATURES].copy()\nnum_df.loc[num_df['Age'].isna(), 'Age'] = num_df['Age'].mean()\nnum_df","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:01:33.685889Z","iopub.execute_input":"2022-07-31T22:01:33.686359Z","iopub.status.idle":"2022-07-31T22:01:33.709636Z","shell.execute_reply.started":"2022-07-31T22:01:33.686316Z","shell.execute_reply":"2022-07-31T22:01:33.708799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\nCATEGORICAL_FEATURES = ['Pclass', 'Sex', 'Embarked', 'CabinLetter']\ncat_df = df_train[CATEGORICAL_FEATURES].copy()\ncat_df","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:02:20.090152Z","iopub.execute_input":"2022-07-31T22:02:20.090552Z","iopub.status.idle":"2022-07-31T22:02:20.114436Z","shell.execute_reply.started":"2022-07-31T22:02:20.090519Z","shell.execute_reply":"2022-07-31T22:02:20.113488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"one_hot_encoder = OneHotEncoder(drop='first')\ncat_features = one_hot_encoder.fit_transform(cat_df).todense()\npd.DataFrame(cat_features, columns=one_hot_encoder.get_feature_names())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:10:07.003742Z","iopub.execute_input":"2022-07-31T22:10:07.004166Z","iopub.status.idle":"2022-07-31T22:10:07.261886Z","shell.execute_reply.started":"2022-07-31T22:10:07.004131Z","shell.execute_reply":"2022-07-31T22:10:07.260963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\n\n\ndf_train, df_test = train_test_split(df, test_size=0.25, random_state=0)\n\n\ndef fit_transformers(df_train):\n    one_hot_encoder = OneHotEncoder(drop='first')\n    one_hot_encoder.fit(df_train[CATEGORICAL_FEATURES])\n    scaler = StandardScaler()\n    scaler.fit(compute_features(df_train, one_hot_encoder, scaler=None))\n    return scaler, one_hot_encoder\n\ndef compute_features(df, one_hot_encoder, scaler):\n    df = preprocess_dataframe(df)\n    cat_features = one_hot_encoder.transform(df[CATEGORICAL_FEATURES])\n    cat_df = pd.DataFrame(cat_features.todense(), columns=one_hot_encoder.get_feature_names()).reset_index(drop=True)\n    num_df = df[NUMERIC_FEATURES].reset_index(drop=True)\n    num_df['Age'] = num_df['Age'].fillna(38.0)\n    num_df['Fare'] = num_df['Fare'].fillna(44)\n    features = pd.concat([num_df, cat_df], axis=1)\n    if scaler:\n        features = pd.DataFrame(scaler.transform(features), columns=features.columns)\n    return features\n\nscaler, one_hot_encoder = fit_transformers(df_train)\n\nX_train = compute_features(df_train, one_hot_encoder, scaler)\ny_train = df_train.Survived\n\nX_test = compute_features(df_test, one_hot_encoder, scaler)\ny_test = df_test.Survived","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:27:09.839511Z","iopub.execute_input":"2022-07-31T22:27:09.840543Z","iopub.status.idle":"2022-07-31T22:27:13.084134Z","shell.execute_reply.started":"2022-07-31T22:27:09.840497Z","shell.execute_reply":"2022-07-31T22:27:13.083074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom xgboost import XGBClassifier\nfrom sklearn.dummy import DummyClassifier\n\nmodel = XGBClassifier()\nbaseline = DummyClassifier(strategy='most_frequent')\n\nmodel.fit(X_train, y_train)\nbaseline.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:40:55.004952Z","iopub.execute_input":"2022-07-31T22:40:55.005378Z","iopub.status.idle":"2022-07-31T22:40:59.375039Z","shell.execute_reply.started":"2022-07-31T22:40:55.005344Z","shell.execute_reply":"2022-07-31T22:40:59.373955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report, roc_auc_score\n\nprint(classification_report(y_test, model.predict(X_test)))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:41:07.348807Z","iopub.execute_input":"2022-07-31T22:41:07.349997Z","iopub.status.idle":"2022-07-31T22:41:07.444453Z","shell.execute_reply.started":"2022-07-31T22:41:07.349945Z","shell.execute_reply":"2022-07-31T22:41:07.443304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"# Train the model on all training data.\nsub_train_df = pd.read_csv('../input/tabular-playground-series-apr-2021/train.csv')\nsub_train_df = preprocess_dataframe(sub_train_df)\nscaler, one_hot_encoder = fit_transformers(sub_train_df)\n\nX_train = compute_features(sub_train_df, one_hot_encoder, scaler)\ny_train = sub_train_df.Survived\n\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:52:58.468228Z","iopub.execute_input":"2022-07-31T22:52:58.468663Z","iopub.status.idle":"2022-07-31T22:53:09.098326Z","shell.execute_reply.started":"2022-07-31T22:52:58.468625Z","shell.execute_reply":"2022-07-31T22:53:09.097364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Run inference on the test dataset and create a submission.\nsub_test_df = pd.read_csv('../input/tabular-playground-series-apr-2021/test.csv')\nX_test = compute_features(sub_test_df, one_hot_encoder, scaler)\n\nsubmission_df = sub_test_df[['PassengerId']].copy()\nsubmission_df['Survived'] = model.predict(X_test)\nsubmission_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:53:36.708422Z","iopub.execute_input":"2022-07-31T22:53:36.709279Z","iopub.status.idle":"2022-07-31T22:53:38.923450Z","shell.execute_reply.started":"2022-07-31T22:53:36.709243Z","shell.execute_reply":"2022-07-31T22:53:38.922430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:53:48.326883Z","iopub.execute_input":"2022-07-31T22:53:48.327258Z","iopub.status.idle":"2022-07-31T22:53:48.509279Z","shell.execute_reply.started":"2022-07-31T22:53:48.327228Z","shell.execute_reply":"2022-07-31T22:53:48.508153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:53:56.725623Z","iopub.execute_input":"2022-07-31T22:53:56.726010Z","iopub.status.idle":"2022-07-31T22:53:56.738828Z","shell.execute_reply.started":"2022-07-31T22:53:56.725977Z","shell.execute_reply":"2022-07-31T22:53:56.737866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2022-07-31T22:54:07.661488Z","iopub.execute_input":"2022-07-31T22:54:07.661925Z","iopub.status.idle":"2022-07-31T22:54:08.435231Z","shell.execute_reply.started":"2022-07-31T22:54:07.661888Z","shell.execute_reply":"2022-07-31T22:54:08.433472Z"},"trusted":true},"execution_count":null,"outputs":[]}]}