{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":7584174,"sourceType":"datasetVersion","datasetId":4414761}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook, a relatively simple NN is developed with features extracted from the public notebook (link: https://www.kaggle.com/code/greysky/home-credit-baseline). Several considerations of the model:\n\nNN itself cannot deal with NaN values.\n\nI use Hold-One-Out Cross Validation (the WEEK_NUM threshold I set is 70) at the moment just in order to save time. \n\nFor NN models, I use Dropout, Batch Normalisation, Swish activation function instead of ReLU, and label smoothing to prevent overfitting. \n\nFine-tuning the model on its own fold-validation set for a few epochs with small lr.","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import TimeSeriesSplit, GroupKFold, StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\n\nimport joblib\n\nimport lightgbm as lgb\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:18:44.450048Z","iopub.execute_input":"2024-04-02T08:18:44.450865Z","iopub.status.idle":"2024-04-02T08:18:51.625854Z","shell.execute_reply.started":"2024-04-02T08:18:44.450788Z","shell.execute_reply":"2024-04-02T08:18:51.623775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.7:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:18:51.628430Z","iopub.execute_input":"2024-04-02T08:18:51.633548Z","iopub.status.idle":"2024-04-02T08:18:51.651905Z","shell.execute_reply.started":"2024-04-02T08:18:51.633473Z","shell.execute_reply":"2024-04-02T08:18:51.650182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs\n","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:18:51.653989Z","iopub.execute_input":"2024-04-02T08:18:51.654713Z","iopub.status.idle":"2024-04-02T08:18:51.672929Z","shell.execute_reply.started":"2024-04-02T08:18:51.654655Z","shell.execute_reply":"2024-04-02T08:18:51.670905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    \n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        chunks.append(df)\n    \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:18:51.676539Z","iopub.execute_input":"2024-04-02T08:18:51.677056Z","iopub.status.idle":"2024-04-02T08:18:51.691213Z","shell.execute_reply.started":"2024-04-02T08:18:51.677018Z","shell.execute_reply":"2024-04-02T08:18:51.689779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n    df_base = df_base.pipe(Pipeline.handle_dates)\n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:18:51.693661Z","iopub.execute_input":"2024-04-02T08:18:51.694447Z","iopub.status.idle":"2024-04-02T08:18:51.707079Z","shell.execute_reply.started":"2024-04-02T08:18:51.694405Z","shell.execute_reply":"2024-04-02T08:18:51.705248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:18:51.709419Z","iopub.execute_input":"2024-04-02T08:18:51.710348Z","iopub.status.idle":"2024-04-02T08:18:51.724642Z","shell.execute_reply.started":"2024-04-02T08:18:51.710301Z","shell.execute_reply":"2024-04-02T08:18:51.723469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Configuration","metadata":{}},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:18:51.726398Z","iopub.execute_input":"2024-04-02T08:18:51.727114Z","iopub.status.idle":"2024-04-02T08:18:51.735911Z","shell.execute_reply.started":"2024-04-02T08:18:51.727074Z","shell.execute_reply":"2024-04-02T08:18:51.734678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Train Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_applprev_2.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_person_2.parquet\", 2)\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:18:51.737789Z","iopub.execute_input":"2024-04-02T08:18:51.738659Z","iopub.status.idle":"2024-04-02T08:21:44.864234Z","shell.execute_reply.started":"2024-04-02T08:18:51.738615Z","shell.execute_reply":"2024-04-02T08:21:44.862665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_eng(**data_store)\n\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:21:44.865508Z","iopub.execute_input":"2024-04-02T08:21:44.865922Z","iopub.status.idle":"2024-04-02T08:22:04.087416Z","shell.execute_reply.started":"2024-04-02T08:21:44.865866Z","shell.execute_reply":"2024-04-02T08:22:04.085743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Test Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TEST_DIR / \"test_applprev_2.parquet\", 2),\n        read_file(TEST_DIR / \"test_person_2.parquet\", 2)\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:22:04.093240Z","iopub.execute_input":"2024-04-02T08:22:04.093675Z","iopub.status.idle":"2024-04-02T08:22:04.982718Z","shell.execute_reply.started":"2024-04-02T08:22:04.093644Z","shell.execute_reply":"2024-04-02T08:22:04.981416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(**data_store)\n\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:22:04.984813Z","iopub.execute_input":"2024-04-02T08:22:04.985449Z","iopub.status.idle":"2024-04-02T08:22:05.042050Z","shell.execute_reply.started":"2024-04-02T08:22:04.985400Z","shell.execute_reply":"2024-04-02T08:22:05.040944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.pipe(Pipeline.filter_cols)\ndf_test = df_test.select([col for col in df_train.columns if col != \"target\"])\n\nprint(\"train data shape:\\t\", df_train.shape)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:22:05.043588Z","iopub.execute_input":"2024-04-02T08:22:05.044289Z","iopub.status.idle":"2024-04-02T08:22:09.020171Z","shell.execute_reply.started":"2024-04-02T08:22:05.044250Z","shell.execute_reply":"2024-04-02T08:22:09.017984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, cat_cols = to_pandas(df_train)\ndf_test, cat_cols = to_pandas(df_test, cat_cols)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:22:09.022526Z","iopub.execute_input":"2024-04-02T08:22:09.023691Z","iopub.status.idle":"2024-04-02T08:22:33.262103Z","shell.execute_reply.started":"2024-04-02T08:22:09.023623Z","shell.execute_reply":"2024-04-02T08:22:33.260576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:22:33.263936Z","iopub.execute_input":"2024-04-02T08:22:33.264443Z","iopub.status.idle":"2024-04-02T08:22:33.439624Z","shell.execute_reply.started":"2024-04-02T08:22:33.264405Z","shell.execute_reply":"2024-04-02T08:22:33.438390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Training","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n\nX_train = df_train.drop(columns=[\"target\", \"case_id\",\"WEEK_NUM\"])\ny_train = df_train[\"target\"]\nweeks = df_train[\"WEEK_NUM\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:22:33.441005Z","iopub.execute_input":"2024-04-02T08:22:33.441850Z","iopub.status.idle":"2024-04-02T08:22:34.943576Z","shell.execute_reply.started":"2024-04-02T08:22:33.441792Z","shell.execute_reply":"2024-04-02T08:22:34.941754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = df_test.drop(columns=[\"WEEK_NUM\"])\nX_test = X_test.set_index(\"case_id\")","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:22:34.945245Z","iopub.execute_input":"2024-04-02T08:22:34.945643Z","iopub.status.idle":"2024-04-02T08:22:34.962400Z","shell.execute_reply.started":"2024-04-02T08:22:34.945607Z","shell.execute_reply":"2024-04-02T08:22:34.960490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **NN**","metadata":{}},{"cell_type":"code","source":"def fill_na(df, fill_value_numeric=0, fill_value_categorical='Missing', fill_value_default='Unknown'):\n    \"\"\"\n    Fills missing values in a DataFrame.\n\n    Parameters:\n    df (pd.DataFrame): The DataFrame to fill missing values in.\n    fill_value_numeric (int or float): The value to fill missing values with in numeric columns.\n    fill_value_categorical (str): The value to fill missing values with in categorical columns.\n    fill_value_default (str): The value to fill missing values with in other types of columns.\n\n    Returns:\n    pd.DataFrame: DataFrame with missing values filled.\n    \"\"\"\n    for col in df.columns:\n        if df[col].dtype.name == 'category':\n            # Add a new category for missing values and fill with it\n            df[col] = df[col].cat.add_categories([fill_value_categorical]).fillna(fill_value_categorical)\n        elif pd.api.types.is_numeric_dtype(df[col]):\n            # Fill numeric columns with the specified numeric value\n            df[col] = df[col].fillna(fill_value_numeric)\n        else:\n            # Fill other types of columns with the specified default value\n            df[col] = df[col].fillna(fill_value_default)\n    return df\n\nX_train = fill_na(X_train)\nX_test = fill_na(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:22:34.965188Z","iopub.execute_input":"2024-04-02T08:22:34.965683Z","iopub.status.idle":"2024-04-02T08:22:39.541101Z","shell.execute_reply.started":"2024-04-02T08:22:34.965648Z","shell.execute_reply":"2024-04-02T08:22:39.539131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = [col for col in X_train.columns if X_train[col].dtype == 'category']\nnum_cols = [col for col in X_train.columns if col not in cat_cols]","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:22:39.543349Z","iopub.execute_input":"2024-04-02T08:22:39.543828Z","iopub.status.idle":"2024-04-02T08:22:39.570784Z","shell.execute_reply.started":"2024-04-02T08:22:39.543782Z","shell.execute_reply":"2024-04-02T08:22:39.569322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.description_5085714M.dtype","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:22:39.572847Z","iopub.execute_input":"2024-04-02T08:22:39.574327Z","iopub.status.idle":"2024-04-02T08:22:39.593065Z","shell.execute_reply.started":"2024-04-02T08:22:39.574243Z","shell.execute_reply":"2024-04-02T08:22:39.591044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"There are {len(cat_cols)} categorical variables\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\n\nlabel_encoders = {}\nfor col in cat_cols:\n    le = LabelEncoder()\n    all_values = pd.concat([X_train[col], X_test[col]], axis=0).astype(str)\n    le.fit(all_values)\n    X_train[col] = le.transform(X_train[col].astype(str))\n    X_test[col] = le.transform(X_test[col].astype(str))\n    label_encoders[col] = le\n    gc.collect()\n\nX_train = np.hstack([X_train[cat_cols].values, X_train[num_cols].values]).astype('float32')\nX_test = np.hstack([X_test[cat_cols].values, X_test[num_cols].values]).astype('float32')\n","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:22:39.595370Z","iopub.execute_input":"2024-04-02T08:22:39.595920Z","iopub.status.idle":"2024-04-02T08:25:29.081372Z","shell.execute_reply.started":"2024-04-02T08:22:39.595867Z","shell.execute_reply":"2024-04-02T08:25:29.079849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:25:29.083634Z","iopub.execute_input":"2024-04-02T08:25:29.084088Z","iopub.status.idle":"2024-04-02T08:25:29.238585Z","shell.execute_reply.started":"2024-04-02T08:25:29.084054Z","shell.execute_reply":"2024-04-02T08:25:29.236704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"week_threshold = 70\n\n# train_indices = df_train[df_train.WEEK_NUM < week_threshold].index\n# valid_indices = df_train[df_train.WEEK_NUM >= week_threshold].index\n\nX_valid = X_train[df_train[df_train.WEEK_NUM >= week_threshold].index]\ny_valid = df_train.loc[df_train[df_train.WEEK_NUM >= week_threshold].index, 'target']\n\nX_train = X_train[df_train[df_train.WEEK_NUM < week_threshold].index]\ny_train = df_train.loc[df_train[df_train.WEEK_NUM < week_threshold].index, 'target'] ","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:25:29.240526Z","iopub.execute_input":"2024-04-02T08:25:29.241277Z","iopub.status.idle":"2024-04-02T08:25:34.425286Z","shell.execute_reply.started":"2024-04-02T08:25:29.241234Z","shell.execute_reply":"2024-04-02T08:25:34.423702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping,ModelCheckpoint\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Dense, BatchNormalization, Dropout, Activation\nimport os\nfrom tensorflow.keras import regularizers\n\nimport tensorflow as tf\n\ndef create_deep_mlp(num_columns, num_labels, hidden_units, dropout_rates, label_smoothing, learning_rate):\n    \n    inp = tf.keras.layers.Input(shape=(num_columns,))\n    x = tf.keras.layers.BatchNormalization()(inp)\n    x = tf.keras.layers.Dropout(dropout_rates[0])(x)\n\n    # Create multiple dense layers as specified by the hidden_units list\n    for i in range(len(hidden_units)): \n        x = tf.keras.layers.Dense(hidden_units[i], activation=None,kernel_regularizer=regularizers.l2(0.01))(x)  # Use None for activation here since we're applying it after batch norm\n        x = tf.keras.layers.BatchNormalization()(x)\n        x = tf.keras.layers.Activation(tf.keras.activations.swish)(x)\n        x = tf.keras.layers.Dropout(dropout_rates[i + 1])(x)  # Notice dropout_rates[i + 1] because of initial dropout after input layer\n\n    # Output layer\n    x = tf.keras.layers.Dense(num_labels, activation=None,kernel_regularizer=regularizers.l2(0.01))(x)  # Use None for activation here since we're applying it after\n    out = tf.keras.layers.Activation('sigmoid')(x)\n    \n    # Create model\n    model = tf.keras.models.Model(inputs=inp, outputs=out)\n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=learning_rate),\n                  loss=tf.keras.losses.BinaryCrossentropy(label_smoothing=label_smoothing), \n                  metrics=[tf.keras.metrics.AUC(name='AUC')])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:25:34.427437Z","iopub.execute_input":"2024-04-02T08:25:34.427872Z","iopub.status.idle":"2024-04-02T08:25:53.763873Z","shell.execute_reply.started":"2024-04-02T08:25:34.427835Z","shell.execute_reply":"2024-04-02T08:25:53.760102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\n\nrandom_seed = 12345\nrandom.seed(random_seed)\nnp.random.seed(random_seed)\ntf.random.set_seed(random_seed)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:25:53.771119Z","iopub.execute_input":"2024-04-02T08:25:53.775578Z","iopub.status.idle":"2024-04-02T08:25:53.789737Z","shell.execute_reply.started":"2024-04-02T08:25:53.775495Z","shell.execute_reply":"2024-04-02T08:25:53.788019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, ModelCheckpoint, EarlyStopping\nfrom sklearn.metrics import roc_auc_score\nimport gc\n\nckp_path = 'NNModel.hdf5'\nlearning_rate = 1e-4\nlabel_smoothing = 0.01\nbatch_size = 128\nhidden_units = [64, 128, 64, 32]\ndropout_rates = [0.1, 0.1, 0.1,0.1, 0.1]\n\n# Model creation\nmodel = create_deep_mlp(X_train.shape[1], 1, hidden_units, dropout_rates, label_smoothing, learning_rate)\n\n# Callbacks\nrlr = ReduceLROnPlateau(monitor='val_AUC', factor=0.1, patience=3, verbose=1, min_delta=1e-4, mode='max')\nckp = ModelCheckpoint(ckp_path, monitor='val_AUC', verbose=1, save_best_only=True, save_weights_only=True, mode='max')\nes = EarlyStopping(monitor='val_AUC', min_delta=1e-4, patience=3, mode='max', restore_best_weights=True, verbose=1)\n\n# Model fitting\nmodel.fit(X_train, y_train, validation_data=(X_valid, y_valid), epochs=100, batch_size=batch_size, callbacks=[rlr, ckp, es], verbose=1)\n\n# Load the best weights\nmodel.load_weights(ckp_path)\n\n# Fine-tuning on validation data\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=learning_rate / 100),\n              loss=tf.keras.losses.BinaryCrossentropy(label_smoothing=label_smoothing), \n              metrics=[tf.keras.metrics.AUC(name='AUC')])\nmodel.fit(X_valid, y_valid, epochs=5, batch_size=batch_size, verbose=1)\n\n# Save the fine-tuned weights\nmodel.save_weights(ckp_path)\n\n# Evaluation\n# Here you should use your validation set or a separate test set\npredicted_valid = model.predict(X_valid, batch_size=batch_size * 4).ravel()\nauc_score = roc_auc_score(y_valid, predicted_valid)\nprint(f'ROC AUC Score after fine-tuning: {auc_score}')\n\n# Cleanup\ntf.keras.backend.clear_session()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:37:49.536758Z","iopub.execute_input":"2024-04-02T08:37:49.537316Z","iopub.status.idle":"2024-04-02T08:56:32.937840Z","shell.execute_reply.started":"2024-04-02T08:37:49.537278Z","shell.execute_reply":"2024-04-02T08:56:32.936070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:37:27.632958Z","iopub.execute_input":"2024-04-02T08:37:27.633542Z","iopub.status.idle":"2024-04-02T08:37:27.641534Z","shell.execute_reply.started":"2024-04-02T08:37:27.633499Z","shell.execute_reply":"2024-04-02T08:37:27.640478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn_pred = model.predict(X_test).ravel()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:36:06.912069Z","iopub.execute_input":"2024-04-02T08:36:06.913710Z","iopub.status.idle":"2024-04-02T08:36:07.458450Z","shell.execute_reply.started":"2024-04-02T08:36:06.913566Z","shell.execute_reply":"2024-04-02T08:36:07.454333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = nn_pred","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:36:07.464626Z","iopub.execute_input":"2024-04-02T08:36:07.469324Z","iopub.status.idle":"2024-04-02T08:36:07.539039Z","shell.execute_reply.started":"2024-04-02T08:36:07.469105Z","shell.execute_reply":"2024-04-02T08:36:07.535563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Check null: \", df_subm[\"score\"].isnull().any())","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:36:07.558971Z","iopub.execute_input":"2024-04-02T08:36:07.560052Z","iopub.status.idle":"2024-04-02T08:36:07.586063Z","shell.execute_reply.started":"2024-04-02T08:36:07.559967Z","shell.execute_reply":"2024-04-02T08:36:07.581081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:36:07.591460Z","iopub.execute_input":"2024-04-02T08:36:07.593792Z","iopub.status.idle":"2024-04-02T08:36:07.622935Z","shell.execute_reply.started":"2024-04-02T08:36:07.593677Z","shell.execute_reply":"2024-04-02T08:36:07.616742Z"},"trusted":true},"execution_count":null,"outputs":[]}]}