{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Hello there (or as we say in the north of germany: moin)! :)\n\nI started with Python in February 2022 and started with kaggle more or less in July 2022. Now I'd like to improve my coding and data science skills. Therefore I share my notebooks and I am glad for your feedback. Let's learn together :)\n\n# Please give me some feedback with regard to the following points:\n\nHow could we improve this notebook?\n\nDid this notebook help you in some regard?\n\nIn case you liked it I'd be happy for an upvote.\n\nThanks, have a nice day and happy coding :)","metadata":{}},{"cell_type":"markdown","source":"# Importing libraries and data","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-26T18:38:17.878660Z","iopub.execute_input":"2022-07-26T18:38:17.879074Z","iopub.status.idle":"2022-07-26T18:38:17.909525Z","shell.execute_reply.started":"2022-07-26T18:38:17.878982Z","shell.execute_reply":"2022-07-26T18:38:17.908575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport missingno as msno\nsns.set()\npd.set_option(\"display.max_rows\", 200)\n\nseed = 19","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:38:17.911434Z","iopub.execute_input":"2022-07-26T18:38:17.911788Z","iopub.status.idle":"2022-07-26T18:38:19.016308Z","shell.execute_reply.started":"2022-07-26T18:38:17.911756Z","shell.execute_reply":"2022-07-26T18:38:19.015408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/titanic/train.csv\"\ndf = pd.read_csv(path)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:38:19.017885Z","iopub.execute_input":"2022-07-26T18:38:19.018516Z","iopub.status.idle":"2022-07-26T18:38:19.062119Z","shell.execute_reply.started":"2022-07-26T18:38:19.018454Z","shell.execute_reply":"2022-07-26T18:38:19.060729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:38:19.065075Z","iopub.execute_input":"2022-07-26T18:38:19.065577Z","iopub.status.idle":"2022-07-26T18:38:19.093406Z","shell.execute_reply.started":"2022-07-26T18:38:19.065519Z","shell.execute_reply":"2022-07-26T18:38:19.092031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum().sort_values(ascending = False)\n\n# Drop Cabin? Impute Age & Embarked?","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:38:32.407838Z","iopub.execute_input":"2022-07-26T18:38:32.408212Z","iopub.status.idle":"2022-07-26T18:38:32.421684Z","shell.execute_reply.started":"2022-07-26T18:38:32.408182Z","shell.execute_reply":"2022-07-26T18:38:32.420421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:38:36.909914Z","iopub.execute_input":"2022-07-26T18:38:36.910302Z","iopub.status.idle":"2022-07-26T18:38:36.928087Z","shell.execute_reply.started":"2022-07-26T18:38:36.910268Z","shell.execute_reply":"2022-07-26T18:38:36.927180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(df.Pclass)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:38:37.750036Z","iopub.execute_input":"2022-07-26T18:38:37.750706Z","iopub.status.idle":"2022-07-26T18:38:37.989957Z","shell.execute_reply.started":"2022-07-26T18:38:37.750669Z","shell.execute_reply":"2022-07-26T18:38:37.988466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#comma_index = df.Name.str.index(\",\")\n#names_titel = []\n#for i, comma in enumerate(comma_index):\n#    if df.Name[i][comma + 2:].split()[0] in names_titel:\n#        continue\n#    else:\n#        names_titel.append(df.Name[i][comma + 2:].split()[0])\n#    \n##    df.Name[i][comma + 2:].split()[0]\n#names_titel","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:55:44.750715Z","iopub.execute_input":"2022-07-24T17:55:44.751188Z","iopub.status.idle":"2022-07-24T17:55:44.757431Z","shell.execute_reply.started":"2022-07-24T17:55:44.751150Z","shell.execute_reply":"2022-07-24T17:55:44.756097Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#names = df.Name.str.split()\n#names_titel = []\n#\n#for ind in df.index:\n#    if names[ind][1] in (names_titel):\n#        continue\n#    else:\n#        names_titel.append(names[ind][1])\n#\n#names_titel","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:55:54.379153Z","iopub.execute_input":"2022-07-24T17:55:54.380417Z","iopub.status.idle":"2022-07-24T17:55:54.384818Z","shell.execute_reply.started":"2022-07-24T17:55:54.380376Z","shell.execute_reply":"2022-07-24T17:55:54.383634Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df.drop(\"Name\", axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:56:03.961130Z","iopub.execute_input":"2022-07-24T17:56:03.961688Z","iopub.status.idle":"2022-07-24T17:56:03.967717Z","shell.execute_reply.started":"2022-07-24T17:56:03.961628Z","shell.execute_reply":"2022-07-24T17:56:03.966182Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.Cabin.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:38:55.901999Z","iopub.execute_input":"2022-07-26T18:38:55.902534Z","iopub.status.idle":"2022-07-26T18:38:55.916286Z","shell.execute_reply.started":"2022-07-26T18:38:55.902472Z","shell.execute_reply":"2022-07-26T18:38:55.915237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Trying to split the Ticket column and see, whether it could be useful.\ndf.Ticket.apply(lambda x: x.split())","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:40:02.124008Z","iopub.execute_input":"2022-07-26T18:40:02.124444Z","iopub.status.idle":"2022-07-26T18:40:02.138509Z","shell.execute_reply.started":"2022-07-26T18:40:02.124410Z","shell.execute_reply":"2022-07-26T18:40:02.137516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df[\"Ticket_prefix\"] = df.Ticket.apply(lambda x: x.split()[0])\n#df[\"Ticket_suffix\"] = df.Ticket.apply(lambda x: x.split()[1] if len(x.split()) == 2 else \"NA\")\n#df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:57:40.709856Z","iopub.execute_input":"2022-07-24T17:57:40.710341Z","iopub.status.idle":"2022-07-24T17:57:40.716097Z","shell.execute_reply.started":"2022-07-24T17:57:40.710304Z","shell.execute_reply":"2022-07-24T17:57:40.714711Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(df[\"Ticket_prefix\"].value_counts())\n#print(df[\"Ticket_suffix\"].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:57:49.872996Z","iopub.execute_input":"2022-07-24T17:57:49.873461Z","iopub.status.idle":"2022-07-24T17:57:49.878877Z","shell.execute_reply.started":"2022-07-24T17:57:49.873423Z","shell.execute_reply":"2022-07-24T17:57:49.877886Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## doesn't really work because of the missing values...\n##df_cabin = df[~df.Cabin.isna()]\n##cabin_letter = []\n##for i in df_cabin.index:\n##    cabin_letter.append(df_cabin.Cabin[i][0])\n\n#df[\"Cabin_na\"] = df.Cabin.fillna(\"N999\")\n#df[\"Cabin_na\"] = df.Cabin_na.apply(lambda x: x.split()[0])\n#df[\"Cabin_letter\"] = df.Cabin_na.apply(lambda x: x[0])\n#df[\"Cabin_num\"] = df.Cabin_na.apply(lambda x: x[1:])\n#df.Cabin_num.unique()\n\n#df.head()\n##df = df.loc[:, [\n##            #'PassengerId', \n##            'Survived', \n##            'Pclass', \n##            'Sex', \n##            'Age', \n##            'SibSp', \n##            'Parch',\n##            #'Ticket', \n##            'Fare', \n##            #'Cabin', \n##            'Embarked', \n##            #'Cabin_na', \n##           'Cabin_letter',\n##           'Cabin_num',\n##           'Ticket_prefix',\n##           'Ticket_suffix']]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:58:01.339809Z","iopub.execute_input":"2022-07-24T17:58:01.340334Z","iopub.status.idle":"2022-07-24T17:58:01.346409Z","shell.execute_reply.started":"2022-07-24T17:58:01.340290Z","shell.execute_reply":"2022-07-24T17:58:01.345230Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # cat_col = df.columns[(df.dtypes == \"object\").values]\n# cat_col = ['Pclass', 'Sex', 'Embarked', 'Cabin_letter', 'Cabin_num', 'Ticket_prefix', 'Ticket_suffix']\n\n# for col in cat_col:\n#     sns.catplot(\n#             x=col, \n#             col = \"Survived\", \n#             data = df, \n#             kind = \"count\"\n#             )\n\n# # Maybe embarked is correlated with something like the fare?\n# # Cabin Letter N -> NA\n# # Cabin Number 999 -> NA","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:58:10.502101Z","iopub.execute_input":"2022-07-24T17:58:10.502579Z","iopub.status.idle":"2022-07-24T17:58:10.508417Z","shell.execute_reply.started":"2022-07-24T17:58:10.502538Z","shell.execute_reply":"2022-07-24T17:58:10.506737Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# searchfor = [\n#     #'Mr.',\n#     #'Mrs.',\n#     #'Miss.',\n#     'Master.',\n#     'Don.',\n#     'Rev.',\n#     'Dr.',\n#     #'Mme.',\n#     #'Ms.',\n#     'Major.',\n#     'Lady.',\n#     'Sir.',\n#     'Mlle.',\n#     'Col.',\n#     'Capt.',\n#     #'the',\n#     'Jonkheer.'\n# ]\n# df[\"special_titel\"] = df.Name.str.contains('|'.join(searchfor))\n\n# # df.drop(\"special_titel\", inplace = True, axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:58:18.572525Z","iopub.execute_input":"2022-07-24T17:58:18.572988Z","iopub.status.idle":"2022-07-24T17:58:18.578310Z","shell.execute_reply.started":"2022-07-24T17:58:18.572950Z","shell.execute_reply":"2022-07-24T17:58:18.576904Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.catplot(x=\"Survived\", \n#             data = df, \n#             #row = \"special_titel\", \n#             col = \"Pclass\", \n#             kind = \"count\")\n# df.groupby([\"special_titel\", \"Pclass\"]).Survived.mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:58:25.580163Z","iopub.execute_input":"2022-07-24T17:58:25.580631Z","iopub.status.idle":"2022-07-24T17:58:25.586880Z","shell.execute_reply.started":"2022-07-24T17:58:25.580581Z","shell.execute_reply":"2022-07-24T17:58:25.585540Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # univariat eda\n# def eda_uni(series):\n#     if (series.dtypes != \"object\"):\n#         sns.kdeplot(series)\n#     elif (series.dtypes == \"object\") & (len(series.unique()) < np.sqrt(len(series)) * 2):\n#         sns.countplot(x = series)\n#     else:\n#         print(series.value_counts().sort_values(ascending = False).head(10))\n\n# for col in df.columns:\n#     eda_uni(df[col])\n#     plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:58:33.793343Z","iopub.execute_input":"2022-07-24T17:58:33.793812Z","iopub.status.idle":"2022-07-24T17:58:33.799532Z","shell.execute_reply.started":"2022-07-24T17:58:33.793772Z","shell.execute_reply":"2022-07-24T17:58:33.798272Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# msno.matrix(df.sort_values(\"Pclass\")) # Pclass and Fare seem to correlate with missing Cabins\nmsno.matrix(df.sort_values(\"Fare\"))\n# msno.heatmap(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:40:53.689413Z","iopub.execute_input":"2022-07-26T18:40:53.689865Z","iopub.status.idle":"2022-07-26T18:40:54.198739Z","shell.execute_reply.started":"2022-07-26T18:40:53.689833Z","shell.execute_reply":"2022-07-26T18:40:54.197531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:40:58.251198Z","iopub.execute_input":"2022-07-26T18:40:58.252604Z","iopub.status.idle":"2022-07-26T18:40:58.271683Z","shell.execute_reply.started":"2022-07-26T18:40:58.252554Z","shell.execute_reply":"2022-07-26T18:40:58.270270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = df.loc[:, [\"Survived\", \"Pclass\", \"Age\", \"Fare\"]]\n\nsns.pairplot(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:40:59.214747Z","iopub.execute_input":"2022-07-26T18:40:59.215620Z","iopub.status.idle":"2022-07-26T18:41:02.728932Z","shell.execute_reply.started":"2022-07-26T18:40:59.215558Z","shell.execute_reply":"2022-07-26T18:41:02.727534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(df.corr(), annot = True)\n# Correlation   between Pclass and Fare = -.55\n#               between SibSp and Parch = .41","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:41:10.770916Z","iopub.execute_input":"2022-07-26T18:41:10.771468Z","iopub.status.idle":"2022-07-26T18:41:11.294452Z","shell.execute_reply.started":"2022-07-26T18:41:10.771417Z","shell.execute_reply":"2022-07-26T18:41:11.292949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# scikit-learn","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import RandomizedSearchCV\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder\n\nfrom sklearn.preprocessing import FunctionTransformer\nfrom sklearn.linear_model import LogisticRegression\n\nfrom xgboost import XGBClassifier\n\nseed = 19","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:41:19.834431Z","iopub.execute_input":"2022-07-26T18:41:19.834880Z","iopub.status.idle":"2022-07-26T18:41:20.330519Z","shell.execute_reply.started":"2022-07-26T18:41:19.834844Z","shell.execute_reply":"2022-07-26T18:41:20.329347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:41:20.332788Z","iopub.execute_input":"2022-07-26T18:41:20.333365Z","iopub.status.idle":"2022-07-26T18:41:20.342308Z","shell.execute_reply.started":"2022-07-26T18:41:20.333315Z","shell.execute_reply":"2022-07-26T18:41:20.341002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"na_mask = df.isna().sum(axis = 1)\n\nnum_mask = (df.dtypes != \"object\").values\nnum_mask","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:41:21.125464Z","iopub.execute_input":"2022-07-26T18:41:21.125895Z","iopub.status.idle":"2022-07-26T18:41:21.136387Z","shell.execute_reply.started":"2022-07-26T18:41:21.125861Z","shell.execute_reply":"2022-07-26T18:41:21.135446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df[\"Pclass\"] = df.Pclass.astype(\"object\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:03:30.811588Z","iopub.execute_input":"2022-07-24T18:03:30.812088Z","iopub.status.idle":"2022-07-24T18:03:30.817449Z","shell.execute_reply.started":"2022-07-24T18:03:30.812051Z","shell.execute_reply":"2022-07-24T18:03:30.816408Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X = df.loc[na_mask, num_mask].drop(\"PassengerId\")\nX = df.loc[:\n    #, num_mask\n    ].drop([\"Survived\"], axis = 1\n)\n\nX.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:41:27.543378Z","iopub.execute_input":"2022-07-26T18:41:27.543838Z","iopub.status.idle":"2022-07-26T18:41:27.564972Z","shell.execute_reply.started":"2022-07-26T18:41:27.543802Z","shell.execute_reply":"2022-07-26T18:41:27.564036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y = df.loc[na_mask, \"Survived\"]\ny = df.Survived\ny","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:41:37.805354Z","iopub.execute_input":"2022-07-26T18:41:37.805757Z","iopub.status.idle":"2022-07-26T18:41:37.815259Z","shell.execute_reply.started":"2022-07-26T18:41:37.805724Z","shell.execute_reply":"2022-07-26T18:41:37.814272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(X.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:41:38.013055Z","iopub.execute_input":"2022-07-26T18:41:38.013708Z","iopub.status.idle":"2022-07-26T18:41:38.020978Z","shell.execute_reply.started":"2022-07-26T18:41:38.013659Z","shell.execute_reply":"2022-07-26T18:41:38.019972Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:41:38.373631Z","iopub.execute_input":"2022-07-26T18:41:38.374069Z","iopub.status.idle":"2022-07-26T18:41:38.383694Z","shell.execute_reply.started":"2022-07-26T18:41:38.374031Z","shell.execute_reply":"2022-07-26T18:41:38.382250Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:41:40.827423Z","iopub.execute_input":"2022-07-26T18:41:40.828001Z","iopub.status.idle":"2022-07-26T18:41:40.849511Z","shell.execute_reply.started":"2022-07-26T18:41:40.827951Z","shell.execute_reply":"2022-07-26T18:41:40.847934Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prefix_split(df):\n    df = df.copy()\n    return df.loc[:,\"Ticket\"].apply(lambda x: x.split()[0])\n    \ndef suffix_split(df):\n    df = df.copy()\n    return df.loc[:,\"Ticket\"].apply(lambda x: x.split()[1] if len(x.split()) == 2 else \"NA\")\n\ndef cabin_letter_split(df):\n    df = df.copy()\n    return df.loc[:,\"Cabin\"].fillna(\"N999\").apply(lambda x: x.split()[0]).apply(lambda x: x[0])\n\ndef cabin_num_split(df):\n    df = df.copy()\n    return df.loc[:,\"Cabin\"].fillna(\"N999\").apply(lambda x: x.split()[0])\n\ndef type_object(df):\n    df = df.copy()\n    return df.loc[:,\"Pclass\"].astype(\"object\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:42:30.700631Z","iopub.execute_input":"2022-07-26T18:42:30.701116Z","iopub.status.idle":"2022-07-26T18:42:30.713319Z","shell.execute_reply.started":"2022-07-26T18:42:30.701079Z","shell.execute_reply":"2022-07-26T18:42:30.712442Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Combining all the preprocessing steps in one function to transform it in a scikit-learn usable object.\n\ndef preproc(df):\n    df = df.copy()\n    df[\"Ticket_prefix\"] = df[\"Ticket\"].apply(lambda x: x.split()[0])\n    df[\"Ticket_suffix\"] = df[\"Ticket\"].apply(lambda x: x.split()[1] if len(x.split()) == 2 else \"NA\")\n    df[\"Cabin_na\"] = df[\"Cabin\"].fillna(\"N999\")\n    df[\"Cabin_na\"] = df[\"Cabin_na\"].apply(lambda x: x.split()[0])\n    df[\"Cabin_letter\"] = df[\"Cabin_na\"].apply(lambda x: x[0])\n    df[\"Cabin_num\"] = df[\"Cabin_na\"].apply(lambda x: x[1:])\n    df[\"Pclass\"] = df[\"Pclass\"].astype(\"object\")\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:42:31.317820Z","iopub.execute_input":"2022-07-26T18:42:31.318732Z","iopub.status.idle":"2022-07-26T18:42:31.328809Z","shell.execute_reply.started":"2022-07-26T18:42:31.318672Z","shell.execute_reply":"2022-07-26T18:42:31.327581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preproc_pipe = FunctionTransformer(preproc)\n\n#prefix_split_pipe = FunctionTransformer(prefix_split)\n    \n#suffix_split_pipe = FunctionTransformer(suffix_split)\n\n#cabin_letter_split_pipe = FunctionTransformer(cabin_letter_split)\n\n#cabin_num_split_pipe = FunctionTransformer(cabin_num_split)\n\n#type_object_pipe = FunctionTransformer(type_object)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:06:05.385369Z","iopub.execute_input":"2022-07-24T18:06:05.385848Z","iopub.status.idle":"2022-07-24T18:06:05.392384Z","shell.execute_reply.started":"2022-07-24T18:06:05.385808Z","shell.execute_reply":"2022-07-24T18:06:05.391398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# knnimputer = KNNImputer()\n# ohe = OneHotEncoder()\n\nnum_pipe = Pipeline(steps = [\n    (\"imputer\", KNNImputer()),\n    (\"scaler\", StandardScaler())\n])\n\ncat_pipe = Pipeline(steps = [\n    (\"imputer\", SimpleImputer(strategy=\"constant\", fill_value=\"NA\")),\n    (\"ohe\", OneHotEncoder(handle_unknown='ignore', sparse = False))\n])\n\npreprocessor = ColumnTransformer([                                                      # When df is filtered on the relevant columns:\n    (\"num_preproc\", num_pipe, ['Age', 'SibSp', 'Parch', 'Fare']),                          # X.select_dtypes(exclude = \"object\").columns\n    (\"cat_preproc\", cat_pipe, ['Pclass',\n                            'Sex',\n                            'Embarked',\n                            'Ticket_prefix',\n                            'Ticket_suffix',\n                            'Cabin_letter',\n                            'Cabin_num']),                                                     # X.select_dtypes(include = \"object\").columns\n])\n\npipe = Pipeline(steps = [\n    (\"preproc_pipe\", preproc_pipe),\n    #(\"prefix_split_pipe\", prefix_split_pipe),   \n    #(\"suffix_split_pipe\", suffix_split_pipe),\n    #(\"cabin_letter_split_pipe\", cabin_letter_split_pipe),\n    #(\"cabin_num_split_pipe\", cabin_num_split_pipe),\n    #(\"type_object_pipe\", type_object_pipe),\n    (\"preprocessor\", preprocessor),\n    (\"estimator\", XGBClassifier())\n])\n\nparams_grid = {\n    \"preprocessor__num_preproc__imputer\": [KNNImputer()],\n    \"preprocessor__num_preproc__imputer__n_neighbors\": [3, 4, 5],\n    \"estimator__max_depth\": [7, 8, 9],\n    \"estimator__colsample_bytree\": [.8, .9],\n    \"estimator__colsample_bylevel\": [8, .9],\n    \"estimator__subsample\": [.6, .7],\n    \"estimator__eta\": [.03, .05, .07],\n    \"estimator__gamma\": [1, 2],\n    \"estimator__n_estimators\": [40, 50, 60],\n    \"estimator__min_child_weight\": [1, 2, 3]\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:14:25.752201Z","iopub.execute_input":"2022-07-24T18:14:25.752695Z","iopub.status.idle":"2022-07-24T18:14:25.765773Z","shell.execute_reply.started":"2022-07-24T18:14:25.752625Z","shell.execute_reply":"2022-07-24T18:14:25.764142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    X, y, \n    test_size = .2, \n    random_state = seed,\n    stratify = y\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:14:39.539744Z","iopub.execute_input":"2022-07-24T18:14:39.540179Z","iopub.status.idle":"2022-07-24T18:14:39.553169Z","shell.execute_reply.started":"2022-07-24T18:14:39.540145Z","shell.execute_reply":"2022-07-24T18:14:39.551536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gridcv = GridSearchCV(\n    estimator = pipe, \n    param_grid = params_grid, \n    cv = 5, \n    n_jobs = -1,\n    #scoring=\"recall\"\n)\n\n#gridcv = RandomizedSearchCV(\n#    estimator = pipe,\n#    param_distributions = params_grid,\n#    cv = 5,\n#    n_jobs = -1,\n#    n_iter = 300,\n#    random_state = seed\n#)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:14:52.190985Z","iopub.execute_input":"2022-07-24T18:14:52.191429Z","iopub.status.idle":"2022-07-24T18:14:52.199281Z","shell.execute_reply.started":"2022-07-24T18:14:52.191391Z","shell.execute_reply":"2022-07-24T18:14:52.197571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gridcv.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:15:55.491568Z","iopub.execute_input":"2022-07-24T18:15:55.492081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Best params: \", gridcv.best_params_)\nprint(\"Best score: \", gridcv.best_score_)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gridcv.score(X_train, y_train)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gridcv.score(X_test, y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparing for submission","metadata":{}},{"cell_type":"code","source":"# test.csv\n\npath_upload = \"/kaggle/input/titanic/test.csv\"\ndf_upload = pd.read_csv(path_upload)\ndf_upload.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PassengerId = df_upload.PassengerId\n\nupload_num_mask = (df_upload.dtypes != \"object\").values\n\n\n# X_upload = df_upload.loc[:, upload_num_mask].drop(\"PassengerId\", axis = 1)\n# Survived = gridcv.predict(X_upload)\nSurvived = gridcv.predict(df_upload)\n\nSurvived","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"upload_dict = {\n    \"PassengerId\": PassengerId,\n    \"Survived\": Survived\n}\n\ndf_upload_ready = pd.DataFrame(upload_dict)\ndf_upload_ready.to_csv(\"/kaggle/working/submission.csv\", index = False)\ndf_upload_ready.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}