{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n# import seaborn as sns\n# import matplotlib.pyplot as plt\nimport polars as pl\nimport time\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\ndir_path = '/kaggle/input/home-credit-credit-risk-model-stability/'\n\n# import os\n# for dirname, _, filenames in os.walk(dir_path + 'csv_files/train/'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-28T17:52:12.885240Z","iopub.execute_input":"2024-02-28T17:52:12.886334Z","iopub.status.idle":"2024-02-28T17:52:14.483650Z","shell.execute_reply.started":"2024-02-28T17:52:12.886274Z","shell.execute_reply":"2024-02-28T17:52:14.482419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data as the starter notebook\n### From: DANIEL HERMAN\nhttps://www.kaggle.com/code/jetakow/home-credit-2024-starter-notebook","metadata":{}},{"cell_type":"code","source":"def set_table_dtypes(df: pl.DataFrame) -> pl.DataFrame:\n    # implement here all desired dtypes for tables\n    # the following is just an example\n    for col in df.columns:\n        # last letter of column name will help you determine the type\n        if col[-1] in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n\n    return df\n\ndef convert_strings(df: pd.DataFrame) -> pd.DataFrame:\n    for col in df.columns:  \n        if df[col].dtype.name in ['object', 'string']:\n            df[col] = df[col].astype(\"string\").astype('category')\n            current_categories = df[col].cat.categories\n            new_categories = current_categories.to_list() + [\"Unknown\"]\n            new_dtype = pd.CategoricalDtype(categories=new_categories, ordered=True)\n            df[col] = df[col].astype(new_dtype)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:14.486013Z","iopub.execute_input":"2024-02-28T17:52:14.487095Z","iopub.status.idle":"2024-02-28T17:52:14.498531Z","shell.execute_reply.started":"2024-02-28T17:52:14.487038Z","shell.execute_reply":"2024-02-28T17:52:14.497039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_basetable = pl.read_csv(dir_path + \"csv_files/train/train_base.csv\")\ntrain_static = pl.concat(\n    [\n        pl.read_csv(dir_path + \"csv_files/train/train_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dir_path + \"csv_files/train/train_static_0_1.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntrain_static_cb = pl.read_csv(dir_path + \"csv_files/train/train_static_cb_0.csv\").pipe(set_table_dtypes)\ntrain_person_1 = pl.read_csv(dir_path + \"csv_files/train/train_person_1.csv\").pipe(set_table_dtypes) \ntrain_credit_bureau_b_2 = pl.read_csv(dir_path + \"csv_files/train/train_credit_bureau_b_2.csv\").pipe(set_table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:14.500840Z","iopub.execute_input":"2024-02-28T17:52:14.501489Z","iopub.status.idle":"2024-02-28T17:52:33.160462Z","shell.execute_reply.started":"2024-02-28T17:52:14.501440Z","shell.execute_reply":"2024-02-28T17:52:33.159332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_basetable = pl.read_csv(dir_path + \"csv_files/test/test_base.csv\")\ntest_static = pl.concat(\n    [\n        pl.read_csv(dir_path + \"csv_files/test/test_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dir_path + \"csv_files/test/test_static_0_1.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dir_path + \"csv_files/test/test_static_0_2.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntest_static_cb = pl.read_csv(dir_path + \"csv_files/test/test_static_cb_0.csv\").pipe(set_table_dtypes)\ntest_person_1 = pl.read_csv(dir_path + \"csv_files/test/test_person_1.csv\").pipe(set_table_dtypes) \ntest_credit_bureau_b_2 = pl.read_csv(dir_path + \"csv_files/test/test_credit_bureau_b_2.csv\").pipe(set_table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:33.163276Z","iopub.execute_input":"2024-02-28T17:52:33.163681Z","iopub.status.idle":"2024-02-28T17:52:33.246228Z","shell.execute_reply.started":"2024-02-28T17:52:33.163647Z","shell.execute_reply":"2024-02-28T17:52:33.244717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineer as the starter notebook\n### From: DANIEL HERMAN\nhttps://www.kaggle.com/code/jetakow/home-credit-2024-starter-notebook","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split ","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:33.248157Z","iopub.execute_input":"2024-02-28T17:52:33.248559Z","iopub.status.idle":"2024-02-28T17:52:34.025236Z","shell.execute_reply.started":"2024-02-28T17:52:33.248526Z","shell.execute_reply":"2024-02-28T17:52:34.024135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We need to use aggregation functions in tables with depth > 1, so tables that contain num_group1 column or \n# also num_group2 column.\ntrain_person_1_feats_1 = train_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\n# Here num_group1=0 has special meaning, it is the person who applied for the loan.\ntrain_person_1_feats_2 = train_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\n# Here we have num_goup1 and num_group2, so we need to aggregate again.\ntrain_credit_bureau_b_2_feats = train_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\n# We will process in this examples only A-type and M-type columns, so we need to select them.\nselected_static_cols = []\nfor col in train_static.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cols.append(col)\nprint(selected_static_cols)\n\nselected_static_cb_cols = []\nfor col in train_static_cb.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cb_cols.append(col)\nprint(selected_static_cb_cols)\n\n# Join all tables together.\ndata = train_basetable.join(\n    train_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    train_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:34.026948Z","iopub.execute_input":"2024-02-28T17:52:34.027333Z","iopub.status.idle":"2024-02-28T17:52:37.198533Z","shell.execute_reply.started":"2024-02-28T17:52:34.027300Z","shell.execute_reply":"2024-02-28T17:52:37.197357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_person_1_feats_1 = test_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\ntest_person_1_feats_2 = test_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\ntest_credit_bureau_b_2_feats = test_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\ndata_submission = test_basetable.join(\n    test_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    test_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:37.199606Z","iopub.execute_input":"2024-02-28T17:52:37.200056Z","iopub.status.idle":"2024-02-28T17:52:37.215544Z","shell.execute_reply.started":"2024-02-28T17:52:37.200022Z","shell.execute_reply":"2024-02-28T17:52:37.214577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# case_ids = data[\"case_id\"].unique().shuffle(seed=814)\n# case_ids_train, case_ids_test = train_test_split(case_ids, train_size=0.6, random_state=814)\n# case_ids_valid, case_ids_test = train_test_split(case_ids_test, train_size=0.5, random_state=814)\n\n# cols_pred = []\n# for col in data.columns:\n#     if col[-1].isupper() and col[:-1].islower():\n#         cols_pred.append(col)\n\n# print(cols_pred)\n\n# def from_polars_to_pandas(case_ids: pl.DataFrame) -> pl.DataFrame:\n#     return (\n#         data.filter(pl.col(\"case_id\").is_in(case_ids))[[\"case_id\", \"WEEK_NUM\", \"target\"]].to_pandas(),\n#         data.filter(pl.col(\"case_id\").is_in(case_ids))[cols_pred].to_pandas(),\n#         data.filter(pl.col(\"case_id\").is_in(case_ids))[\"target\"].to_pandas()\n#     )\n\n# base_train, X_train, y_train = from_polars_to_pandas(case_ids_train)\n# base_valid, X_valid, y_valid = from_polars_to_pandas(case_ids_valid)\n# base_test, X_test, y_test = from_polars_to_pandas(case_ids_test)\n\n# for df in [X_train, X_valid, X_test]:\n#     df = convert_strings(df)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:37.217238Z","iopub.execute_input":"2024-02-28T17:52:37.217906Z","iopub.status.idle":"2024-02-28T17:52:37.226126Z","shell.execute_reply.started":"2024-02-28T17:52:37.217870Z","shell.execute_reply":"2024-02-28T17:52:37.225129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(f\"Train: {X_train.shape}\")\n# print(f\"Valid: {X_valid.shape}\")\n# print(f\"Test: {X_test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:37.227510Z","iopub.execute_input":"2024-02-28T17:52:37.228535Z","iopub.status.idle":"2024-02-28T17:52:37.242688Z","shell.execute_reply.started":"2024-02-28T17:52:37.228501Z","shell.execute_reply":"2024-02-28T17:52:37.241558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizations\n### ...and Extra Processing","metadata":{}},{"cell_type":"code","source":"train_data_df = data.to_pandas()\nfinal_test_df = data_submission.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:37.247249Z","iopub.execute_input":"2024-02-28T17:52:37.248036Z","iopub.status.idle":"2024-02-28T17:52:38.273097Z","shell.execute_reply.started":"2024-02-28T17:52:37.247992Z","shell.execute_reply":"2024-02-28T17:52:38.272125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:38.275294Z","iopub.execute_input":"2024-02-28T17:52:38.275793Z","iopub.status.idle":"2024-02-28T17:52:38.504433Z","shell.execute_reply.started":"2024-02-28T17:52:38.275748Z","shell.execute_reply":"2024-02-28T17:52:38.503088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_test_df","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:38.506007Z","iopub.execute_input":"2024-02-28T17:52:38.506398Z","iopub.status.idle":"2024-02-28T17:52:38.546013Z","shell.execute_reply.started":"2024-02-28T17:52:38.506363Z","shell.execute_reply":"2024-02-28T17:52:38.545104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_df","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:38.547342Z","iopub.execute_input":"2024-02-28T17:52:38.548426Z","iopub.status.idle":"2024-02-28T17:52:39.534313Z","shell.execute_reply.started":"2024-02-28T17:52:38.548390Z","shell.execute_reply":"2024-02-28T17:52:39.533071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_data_df, x = 'target');","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:39.535718Z","iopub.execute_input":"2024-02-28T17:52:39.536071Z","iopub.status.idle":"2024-02-28T17:52:39.894355Z","shell.execute_reply.started":"2024-02-28T17:52:39.536045Z","shell.execute_reply":"2024-02-28T17:52:39.893103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pretty unbalanced data set from above plot","metadata":{}},{"cell_type":"code","source":"train_data_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:39.896007Z","iopub.execute_input":"2024-02-28T17:52:39.897362Z","iopub.status.idle":"2024-02-28T17:52:41.162493Z","shell.execute_reply.started":"2024-02-28T17:52:39.897314Z","shell.execute_reply":"2024-02-28T17:52:41.161368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:41.164538Z","iopub.execute_input":"2024-02-28T17:52:41.165003Z","iopub.status.idle":"2024-02-28T17:52:42.362723Z","shell.execute_reply.started":"2024-02-28T17:52:41.164965Z","shell.execute_reply":"2024-02-28T17:52:42.361516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.heatmap(data=all_data_df.corr(numeric_only=True))","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:42.365412Z","iopub.execute_input":"2024-02-28T17:52:42.366286Z","iopub.status.idle":"2024-02-28T17:52:42.371657Z","shell.execute_reply.started":"2024-02-28T17:52:42.366244Z","shell.execute_reply":"2024-02-28T17:52:42.370545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"No field with a strong correlation (numerics)","metadata":{}},{"cell_type":"code","source":"train_data_df.corr(numeric_only=True)['target'].sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:42.373706Z","iopub.execute_input":"2024-02-28T17:52:42.374208Z","iopub.status.idle":"2024-02-28T17:52:48.009755Z","shell.execute_reply.started":"2024-02-28T17:52:42.374168Z","shell.execute_reply":"2024-02-28T17:52:48.008541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_df[train_data_df.columns[train_data_df.dtypes=='object']].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:48.011195Z","iopub.execute_input":"2024-02-28T17:52:48.011637Z","iopub.status.idle":"2024-02-28T17:52:49.558557Z","shell.execute_reply.started":"2024-02-28T17:52:48.011594Z","shell.execute_reply":"2024-02-28T17:52:49.557249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_df['lastapprcommoditytypec_5251766M']","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:49.560145Z","iopub.execute_input":"2024-02-28T17:52:49.560797Z","iopub.status.idle":"2024-02-28T17:52:49.569613Z","shell.execute_reply.started":"2024-02-28T17:52:49.560763Z","shell.execute_reply":"2024-02-28T17:52:49.568391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show the catagorical data (non-numeric)","metadata":{}},{"cell_type":"code","source":"for col in train_data_df.columns[train_data_df.dtypes=='object']:\n    plt.figure(figsize=(10, 5)) \n    plt.title(col)\n    sns.countplot(data=train_data_df, x = col);\n    plt.xticks(rotation=90);","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:52:49.570840Z","iopub.execute_input":"2024-02-28T17:52:49.571190Z","iopub.status.idle":"2024-02-28T17:53:21.617971Z","shell.execute_reply.started":"2024-02-28T17:52:49.571164Z","shell.execute_reply":"2024-02-28T17:53:21.616558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Drop the catagorical fields with tons of options for now","metadata":{}},{"cell_type":"code","source":"train_data_df.columns[train_data_df.dtypes=='object']","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:21.619828Z","iopub.execute_input":"2024-02-28T17:53:21.620696Z","iopub.status.idle":"2024-02-28T17:53:21.629223Z","shell.execute_reply.started":"2024-02-28T17:53:21.620655Z","shell.execute_reply":"2024-02-28T17:53:21.627680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = ['description_5085714M','education_1103M','education_88M','maritalst_385M','maritalst_893M','person_housetype','pmts_dpdvalue_108P_over31','mainoccupationinc_384A_any_selfemployed']","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:21.630719Z","iopub.execute_input":"2024-02-28T17:53:21.631098Z","iopub.status.idle":"2024-02-28T17:53:21.641568Z","shell.execute_reply.started":"2024-02-28T17:53:21.631067Z","shell.execute_reply":"2024-02-28T17:53:21.640532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Only use the float columns (drop the date, target and id)","metadata":{}},{"cell_type":"code","source":"cont_cols = list(train_data_df.columns[train_data_df.dtypes=='float'])","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:21.642889Z","iopub.execute_input":"2024-02-28T17:53:21.643646Z","iopub.status.idle":"2024-02-28T17:53:21.660495Z","shell.execute_reply.started":"2024-02-28T17:53:21.643616Z","shell.execute_reply":"2024-02-28T17:53:21.658906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_col = ['target']\nids_col = ['case_id']","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:21.662483Z","iopub.execute_input":"2024-02-28T17:53:21.663348Z","iopub.status.idle":"2024-02-28T17:53:21.672623Z","shell.execute_reply.started":"2024-02-28T17:53:21.663305Z","shell.execute_reply":"2024-02-28T17:53:21.671317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_cols = ids_col+cont_cols + cat_cols +y_col\nall_cols","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:21.674088Z","iopub.execute_input":"2024-02-28T17:53:21.674554Z","iopub.status.idle":"2024-02-28T17:53:21.687350Z","shell.execute_reply.started":"2024-02-28T17:53:21.674524Z","shell.execute_reply":"2024-02-28T17:53:21.686224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_df[all_cols].info()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:21.689213Z","iopub.execute_input":"2024-02-28T17:53:21.689651Z","iopub.status.idle":"2024-02-28T17:53:22.842174Z","shell.execute_reply.started":"2024-02-28T17:53:21.689613Z","shell.execute_reply":"2024-02-28T17:53:22.840602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## One hot encode the catagorical data (that I kept)","metadata":{}},{"cell_type":"code","source":"all_cols","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:22.848456Z","iopub.execute_input":"2024-02-28T17:53:22.849027Z","iopub.status.idle":"2024-02-28T17:53:22.863974Z","shell.execute_reply.started":"2024-02-28T17:53:22.848976Z","shell.execute_reply":"2024-02-28T17:53:22.862289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_no_targ = all_cols[:-1]","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:22.866455Z","iopub.execute_input":"2024-02-28T17:53:22.867005Z","iopub.status.idle":"2024-02-28T17:53:22.872715Z","shell.execute_reply.started":"2024-02-28T17:53:22.866958Z","shell.execute_reply":"2024-02-28T17:53:22.871528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_no_targ","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:22.873792Z","iopub.execute_input":"2024-02-28T17:53:22.874223Z","iopub.status.idle":"2024-02-28T17:53:22.888081Z","shell.execute_reply.started":"2024-02-28T17:53:22.874193Z","shell.execute_reply":"2024-02-28T17:53:22.886876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_test_one_hot_df = pd.get_dummies(final_test_df[cols_no_targ],drop_first=True).fillna(0)\ntrain_one_hot_df = pd.get_dummies(train_data_df[all_cols],drop_first=True).fillna(0)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:22.889577Z","iopub.execute_input":"2024-02-28T17:53:22.890367Z","iopub.status.idle":"2024-02-28T17:53:25.659092Z","shell.execute_reply.started":"2024-02-28T17:53:22.890329Z","shell.execute_reply":"2024-02-28T17:53:25.658091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_one_hot_df = train_one_hot_df[list(final_test_one_hot_df.columns) + ['target']]","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:25.662221Z","iopub.execute_input":"2024-02-28T17:53:25.662766Z","iopub.status.idle":"2024-02-28T17:53:25.973171Z","shell.execute_reply.started":"2024-02-28T17:53:25.662721Z","shell.execute_reply":"2024-02-28T17:53:25.971896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_one_hot_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:25.975120Z","iopub.execute_input":"2024-02-28T17:53:25.975606Z","iopub.status.idle":"2024-02-28T17:53:26.145480Z","shell.execute_reply.started":"2024-02-28T17:53:25.975550Z","shell.execute_reply":"2024-02-28T17:53:26.144334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train, Test, and Valid split","metadata":{}},{"cell_type":"code","source":"train_one_hot_df","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:26.146954Z","iopub.execute_input":"2024-02-28T17:53:26.147731Z","iopub.status.idle":"2024-02-28T17:53:26.708882Z","shell.execute_reply.started":"2024-02-28T17:53:26.147698Z","shell.execute_reply":"2024-02-28T17:53:26.707345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shuffled = train_one_hot_df.sample(frac=1)\n# ids = shuffled['case_id']\ny = shuffled['target']\nX = shuffled.drop(columns=['target'])\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=814)\n# X_test, X_valid, y_test, y_valid = train_test_split(X, y, test_size=0.5, random_state=814)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:26.710682Z","iopub.execute_input":"2024-02-28T17:53:26.711136Z","iopub.status.idle":"2024-02-28T17:53:29.750405Z","shell.execute_reply.started":"2024-02-28T17:53:26.711100Z","shell.execute_reply":"2024-02-28T17:53:29.748968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:29.752170Z","iopub.execute_input":"2024-02-28T17:53:29.753218Z","iopub.status.idle":"2024-02-28T17:53:30.212302Z","shell.execute_reply.started":"2024-02-28T17:53:29.753173Z","shell.execute_reply":"2024-02-28T17:53:30.211054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_train = X_train['case_id']\nX_train = X_train.drop(columns=['case_id'])\nid_test = X_test['case_id']\nX_test = X_test.drop(columns=['case_id'])\n# id_valid = X_valid['case_id']\n# X_valid = X_valid.drop(columns=['case_id'])\nid_final_test = final_test_one_hot_df['case_id']\nX_final_test = final_test_one_hot_df.drop(columns=['case_id'])","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:30.213744Z","iopub.execute_input":"2024-02-28T17:53:30.214119Z","iopub.status.idle":"2024-02-28T17:53:30.502273Z","shell.execute_reply.started":"2024-02-28T17:53:30.214089Z","shell.execute_reply":"2024-02-28T17:53:30.501055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Train: {X_train.shape}\")\nprint(f\"Test: {X_test.shape}\")\nprint(f\"Final Test: {X_final_test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-28T17:53:30.503880Z","iopub.execute_input":"2024-02-28T17:53:30.504367Z","iopub.status.idle":"2024-02-28T17:53:30.511665Z","shell.execute_reply.started":"2024-02-28T17:53:30.504325Z","shell.execute_reply":"2024-02-28T17:53:30.510296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier,HistGradientBoostingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import roc_auc_score\nscaler = StandardScaler()","metadata":{"execution":{"iopub.status.busy":"2024-02-28T18:06:51.367662Z","iopub.execute_input":"2024-02-28T18:06:51.368165Z","iopub.status.idle":"2024-02-28T18:06:51.374659Z","shell.execute_reply.started":"2024-02-28T18:06:51.368123Z","shell.execute_reply":"2024-02-28T18:06:51.373479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler.fit(X_train)\nX_train_scaled = scaler.transform(X_train)\nX_test_scaled = scaler.transform(X_test)\nX_final_test_scaled = scaler.transform(X_final_test)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T18:06:52.389268Z","iopub.execute_input":"2024-02-28T18:06:52.389750Z","iopub.status.idle":"2024-02-28T18:06:57.512773Z","shell.execute_reply.started":"2024-02-28T18:06:52.389713Z","shell.execute_reply":"2024-02-28T18:06:57.511410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log = LogisticRegression(class_weight='balanced')\nt1 = time.time()\nlog.fit(X_train_scaled, y_train)\ntotal = time.time() - t1\nprint(f'Logistic Regression took {total} s')\nlog_pred = log.predict_proba(X_test_scaled)\nroc_score = roc_auc_score(y_test, log_pred[:,1])\nprint(f'Logistic Regression roc {round(roc_score,8)}')\nprint(f'Logistic Regression gini {round(2*roc_score-1,8)}')","metadata":{"execution":{"iopub.status.busy":"2024-02-28T18:28:54.755095Z","iopub.execute_input":"2024-02-28T18:28:54.755561Z","iopub.status.idle":"2024-02-28T18:29:09.537080Z","shell.execute_reply.started":"2024-02-28T18:28:54.755527Z","shell.execute_reply":"2024-02-28T18:29:09.535399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rfc = RandomForestClassifier(n_estimators=100, class_weight=\"balanced\")\n# t1 = time.time()\n# rfc.fit(X_train_scaled, y_train)\n# total = time.time() - t1\n# print(f'RFC took {total} s')\n# rfc_pred = rfc.predict(X_test_scaled)\n# roc_score = roc_auc_score(y_test, rfc_pred)\n# print(f'RFC roc {round(roc_score,8)}')\n# print(f'RFC gini {round(2*roc_score-1,8)}')","metadata":{"execution":{"iopub.status.busy":"2024-02-28T18:06:25.311677Z","iopub.execute_input":"2024-02-28T18:06:25.312210Z","iopub.status.idle":"2024-02-28T18:06:25.319661Z","shell.execute_reply.started":"2024-02-28T18:06:25.312169Z","shell.execute_reply":"2024-02-28T18:06:25.318130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hgbc = HistGradientBoostingClassifier(class_weight=\"balanced\")\n# t1 = time.time()\n# hgbc.fit(X_train_scaled, y_train)\n# total = time.time() - t1\n# print(f'HGBC took {total} s')\n# hgbc_pred = hgbc.predict(X_test_scaled)\n# roc_score = roc_auc_score(y_test, hgbc_pred)\n# print(f'HGBC roc {round(roc_score,8)}')\n# print(f'HGBC gini {round(2*roc_score-1,8)}')","metadata":{"execution":{"iopub.status.busy":"2024-02-28T18:06:03.221043Z","iopub.status.idle":"2024-02-28T18:06:03.221428Z","shell.execute_reply.started":"2024-02-28T18:06:03.221244Z","shell.execute_reply":"2024-02-28T18:06:03.221261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_preds = log.predict_proba(X_final_test_scaled)\npreds_df = pd.DataFrame(final_preds[:,1],columns=['score'])\noutput_df = pd.concat([id_final_test,preds_df.round(1)],axis=1)\noutput_df","metadata":{"execution":{"iopub.status.busy":"2024-02-28T18:31:16.511226Z","iopub.execute_input":"2024-02-28T18:31:16.511748Z","iopub.status.idle":"2024-02-28T18:31:16.531482Z","shell.execute_reply.started":"2024-02-28T18:31:16.511708Z","shell.execute_reply":"2024-02-28T18:31:16.529896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_df.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T18:31:49.341029Z","iopub.execute_input":"2024-02-28T18:31:49.341542Z","iopub.status.idle":"2024-02-28T18:31:49.354629Z","shell.execute_reply.started":"2024-02-28T18:31:49.341506Z","shell.execute_reply":"2024-02-28T18:31:49.352972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}