{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport polars as pl\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score \nimport warnings\n\nwarnings.filterwarnings('ignore')\n\ndataPath = \"/kaggle/input/home-credit-credit-risk-model-stability/\"\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-02-11T00:22:09.296214Z","iopub.execute_input":"2024-02-11T00:22:09.296960Z","iopub.status.idle":"2024-02-11T00:22:10.828616Z","shell.execute_reply.started":"2024-02-11T00:22:09.296909Z","shell.execute_reply":"2024-02-11T00:22:10.827666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Goal of the Competition**\n\nThe goal of this competition is to predict which clients are more likely to default on their loans. The evaluation will favor solutions that are stable over time.\n\nYour participation may offer consumer finance providers a more reliable and longer-lasting way to assess a potential client’s default risk.\n\n### I will continue to work and update this notebook. Please upvote it if you find it useful in this interesting challenge!\n\n## **Loading the Data**\n\nWe will now load the data and get it in the right format. We will conduct exploratory data analysis and create a baseline submission that we can iterate and improve on.","metadata":{}},{"cell_type":"code","source":"def set_table_dtypes(df: pl.DataFrame) -> pl.DataFrame:\n    # implement here all desired dtypes for tables\n    # the following is just an example\n    for col in df.columns:\n        # last letter of column name will help you determine the type\n        if col[-1] in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n\n    return df\n\ndef convert_strings(df: pd.DataFrame) -> pd.DataFrame:\n    for col in df.columns:  \n        if df[col].dtype.name in ['object', 'string']:\n            df[col] = df[col].astype(\"string\").astype('category')\n            current_categories = df[col].cat.categories\n            new_categories = current_categories.to_list() + [\"Unknown\"]\n            new_dtype = pd.CategoricalDtype(categories=new_categories, ordered=True)\n            df[col] = df[col].astype(new_dtype)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-02-11T00:22:10.830011Z","iopub.execute_input":"2024-02-11T00:22:10.831060Z","iopub.status.idle":"2024-02-11T00:22:10.841352Z","shell.execute_reply.started":"2024-02-11T00:22:10.831014Z","shell.execute_reply":"2024-02-11T00:22:10.840120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_basetable = pl.read_csv(dataPath + \"csv_files/train/train_base.csv\")\ntrain_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_1.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntrain_static_cb = pl.read_csv(dataPath + \"csv_files/train/train_static_cb_0.csv\").pipe(set_table_dtypes)\ntrain_person_1 = pl.read_csv(dataPath + \"csv_files/train/train_person_1.csv\").pipe(set_table_dtypes) \ntrain_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/train/train_credit_bureau_b_2.csv\").pipe(set_table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-02-11T00:22:10.843068Z","iopub.execute_input":"2024-02-11T00:22:10.843479Z","iopub.status.idle":"2024-02-11T00:22:27.477283Z","shell.execute_reply.started":"2024-02-11T00:22:10.843438Z","shell.execute_reply":"2024-02-11T00:22:27.475013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_basetable = pl.read_csv(dataPath + \"csv_files/test/test_base.csv\")\ntest_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_1.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_2.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntest_static_cb = pl.read_csv(dataPath + \"csv_files/test/test_static_cb_0.csv\").pipe(set_table_dtypes)\ntest_person_1 = pl.read_csv(dataPath + \"csv_files/test/test_person_1.csv\").pipe(set_table_dtypes) \ntest_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/test/test_credit_bureau_b_2.csv\").pipe(set_table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-02-11T00:22:27.482407Z","iopub.execute_input":"2024-02-11T00:22:27.483273Z","iopub.status.idle":"2024-02-11T00:22:27.554344Z","shell.execute_reply.started":"2024-02-11T00:22:27.483234Z","shell.execute_reply":"2024-02-11T00:22:27.552842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Feature Engineering**\n\nWe will join the tables on case_id. There are additional ways we can work with the data, but we will leave this as is for now.","metadata":{}},{"cell_type":"code","source":"# We need to use aggregation functions in tables with depth > 1, so tables that contain num_group1 column or \n# also num_group2 column.\ntrain_person_1_feats_1 = train_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\n# Here num_group1=0 has special meaning, it is the person who applied for the loan.\ntrain_person_1_feats_2 = train_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\n# Here we have num_goup1 and num_group2, so we need to aggregate again.\ntrain_credit_bureau_b_2_feats = train_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\n# We will process in this examples only A-type and M-type columns, so we need to select them.\nselected_static_cols = []\nfor col in train_static.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cols.append(col)\nprint(selected_static_cols)\n\nselected_static_cb_cols = []\nfor col in train_static_cb.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cb_cols.append(col)\nprint(selected_static_cb_cols)\n\n# Join all tables together.\ndata = train_basetable.join(\n    train_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    train_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T00:22:27.556148Z","iopub.execute_input":"2024-02-11T00:22:27.556665Z","iopub.status.idle":"2024-02-11T00:22:31.900239Z","shell.execute_reply.started":"2024-02-11T00:22:27.556619Z","shell.execute_reply":"2024-02-11T00:22:31.898015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_person_1_feats_1 = test_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\ntest_person_1_feats_2 = test_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\ntest_credit_bureau_b_2_feats = test_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\ndata_submission = test_basetable.join(\n    test_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    test_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T00:22:31.903921Z","iopub.execute_input":"2024-02-11T00:22:31.906482Z","iopub.status.idle":"2024-02-11T00:22:31.935335Z","shell.execute_reply.started":"2024-02-11T00:22:31.906421Z","shell.execute_reply":"2024-02-11T00:22:31.933352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"case_ids = data[\"case_id\"].unique().shuffle(seed=1)\ncase_ids_train, case_ids_test = train_test_split(case_ids, train_size=0.6, random_state=1)\ncase_ids_valid, case_ids_test = train_test_split(case_ids_test, train_size=0.5, random_state=1)\n\ncols_pred = []\nfor col in data.columns:\n    if col[-1].isupper() and col[:-1].islower():\n        cols_pred.append(col)\n\nprint(cols_pred)\n\ndef from_polars_to_pandas(case_ids: pl.DataFrame) -> pl.DataFrame:\n    return (\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[[\"case_id\", \"WEEK_NUM\", \"target\"]].to_pandas(),\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[cols_pred].to_pandas(),\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[\"target\"].to_pandas()\n    )\n\nbase_train, X_train, y_train = from_polars_to_pandas(case_ids_train)\nbase_valid, X_valid, y_valid = from_polars_to_pandas(case_ids_valid)\nbase_test, X_test, y_test = from_polars_to_pandas(case_ids_test)\n\nfor df in [X_train, X_valid, X_test]:\n    df = convert_strings(df)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T00:22:31.937777Z","iopub.execute_input":"2024-02-11T00:22:31.939533Z","iopub.status.idle":"2024-02-11T00:22:49.047179Z","shell.execute_reply.started":"2024-02-11T00:22:31.939475Z","shell.execute_reply":"2024-02-11T00:22:49.042494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_submission = data_submission[cols_pred].to_pandas()\nX_submission = convert_strings(X_submission)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T00:22:49.052509Z","iopub.execute_input":"2024-02-11T00:22:49.054607Z","iopub.status.idle":"2024-02-11T00:22:49.137372Z","shell.execute_reply.started":"2024-02-11T00:22:49.054431Z","shell.execute_reply":"2024-02-11T00:22:49.135880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Train: {X_train.shape}\")\nprint(f\"Valid: {X_valid.shape}\")\nprint(f\"Test: {X_test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-11T00:22:49.139218Z","iopub.execute_input":"2024-02-11T00:22:49.139696Z","iopub.status.idle":"2024-02-11T00:22:49.150377Z","shell.execute_reply.started":"2024-02-11T00:22:49.139654Z","shell.execute_reply":"2024-02-11T00:22:49.148643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fill_na(df, fill_value_numeric=0, fill_value_categorical='Missing', fill_value_default='Unknown'):\n    \"\"\"\n    Fills missing values in a DataFrame.\n\n    Parameters:\n    df (pd.DataFrame): The DataFrame to fill missing values in.\n    fill_value_numeric (int or float): The value to fill missing values with in numeric columns.\n    fill_value_categorical (str): The value to fill missing values with in categorical columns.\n    fill_value_default (str): The value to fill missing values with in other types of columns.\n\n    Returns:\n    pd.DataFrame: DataFrame with missing values filled.\n    \"\"\"\n    for col in df.columns:\n        if df[col].dtype.name == 'category':\n            # Add a new category for missing values and fill with it\n            df[col] = df[col].cat.add_categories([fill_value_categorical]).fillna(fill_value_categorical)\n        elif pd.api.types.is_numeric_dtype(df[col]):\n            # Fill numeric columns with the specified numeric value\n            df[col] = df[col].fillna(fill_value_numeric)\n        else:\n            # Fill other types of columns with the specified default value\n            df[col] = df[col].fillna(fill_value_default)\n    return df\n\n\nX_train= fill_na(X_train)\nX_valid = fill_na(X_valid)\nX_test = fill_na(X_test)\nX_submission = fill_na(X_submission)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T00:22:49.152713Z","iopub.execute_input":"2024-02-11T00:22:49.153224Z","iopub.status.idle":"2024-02-11T00:22:50.702132Z","shell.execute_reply.started":"2024-02-11T00:22:49.153183Z","shell.execute_reply":"2024-02-11T00:22:50.697343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = [col for col in X_train.columns if X_train[col].dtype == 'category']\nnum_cols = [col for col in X_train.columns if col not in cat_cols]","metadata":{"execution":{"iopub.status.busy":"2024-02-11T00:22:50.708028Z","iopub.execute_input":"2024-02-11T00:22:50.710138Z","iopub.status.idle":"2024-02-11T00:22:50.737286Z","shell.execute_reply.started":"2024-02-11T00:22:50.710040Z","shell.execute_reply":"2024-02-11T00:22:50.733129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nlabel_encoders = {}\nfor col in cat_cols:\n    le = LabelEncoder()\n    all_values = pd.concat([X_train[col], X_valid[col],X_test[col],X_submission[col]], axis=0).astype(str)\n    le.fit(all_values)\n    X_train[col] = le.transform(X_train[col].astype(str))\n    X_valid[col] = le.transform(X_valid[col].astype(str))\n    X_test[col] = le.transform(X_test[col].astype(str))\n    X_submission[col] = le.transform(X_submission[col].astype(str))\n    label_encoders[col] = le\n\nX_test_combined = np.hstack([X_test[cat_cols].values, X_test[num_cols].values])\nX_train_combined = np.hstack([X_train[cat_cols].values, X_train[num_cols].values])\nX_valid_combined = np.hstack([X_valid[cat_cols].values, X_valid[num_cols].values])\nX_submission_combined = np.hstack([X_submission[cat_cols].values, X_submission[num_cols].values])\n\nX_test = X_test_combined.astype('float32')\nX_train = X_train_combined.astype('float32')\nX_valid = X_valid_combined.astype('float32')\nX_submission = X_submission_combined.astype('float32')","metadata":{"execution":{"iopub.status.busy":"2024-02-11T00:22:50.742199Z","iopub.execute_input":"2024-02-11T00:22:50.743670Z","iopub.status.idle":"2024-02-11T00:23:09.728052Z","shell.execute_reply.started":"2024-02-11T00:22:50.743603Z","shell.execute_reply":"2024-02-11T00:23:09.726569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Deep Learning**\n\nDeep learning is a type of machine learning that is used to teach artificial neural networks how to learn from vast amounts of data. These neural networks are designed to mimic the structure of the human brain, and they are composed of layers of interconnected nodes called neurons that work in concert to process and analyze information.\n\nIn the process of deep learning, the neural network is fed large amounts of data and then adjusts its parameters through a process called backpropagation to minimize the difference between its predicted output and the actual output. This allows the neural network to recognize patterns and make accurate predictions.\n\nDeep learning has many practical applications, such as image and speech recognition, natural language processing, and self-driving cars. It has revolutionized the field of artificial intelligence and has contributed to significant advancements in research and development.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.callbacks import TensorBoard\n\nnum_features = X_train.shape[1]  # Number of features\n\n# Define the input shape based on the dataset\ninput_shape = (num_features,)\n\n# Define the model architecture\ninputs = tf.keras.layers.Input(shape=input_shape)\nx = tf.keras.layers.Dense(128, activation='relu')(inputs)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Dropout(0.3)(x)\nx = tf.keras.layers.Dense(64, activation='relu')(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Dropout(0.3)(x)\nx = tf.keras.layers.Dense(32, activation='relu')(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Dropout(0.3)(x)\noutputs = tf.keras.layers.Dense(1, activation='sigmoid')(x)\n\n# Create the model\nmodel = tf.keras.models.Model(inputs=inputs, outputs=outputs)\n\n# Define TensorBoard callback\ntensorboard_callback = TensorBoard(log_dir='logs')\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['AUC'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-02-11T00:23:09.732421Z","iopub.execute_input":"2024-02-11T00:23:09.732804Z","iopub.status.idle":"2024-02-11T00:23:14.416104Z","shell.execute_reply.started":"2024-02-11T00:23:09.732773Z","shell.execute_reply":"2024-02-11T00:23:14.414021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\nhistory = model.fit(X_train, y_train, validation_data=(X_valid, y_valid), batch_size=128, epochs=10,\n                   callbacks=tensorboard_callback)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T00:23:14.419198Z","iopub.execute_input":"2024-02-11T00:23:14.419734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot training and validation loss with standard deviation shading\nplt.figure(figsize=(10, 5))\n\n# Training loss and its standard deviation\nplt.plot(history.history['loss'], label='Training Loss')\nplt.fill_between(range(len(history.history['loss'])),\n                 np.array(history.history['loss']) - np.std(history.history['loss']),\n                 np.array(history.history['loss']) + np.std(history.history['loss']),\n                 color='gray', alpha=0.2)\n\n# Validation loss and its standard deviation\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.fill_between(range(len(history.history['val_loss'])),\n                 np.array(history.history['val_loss']) - np.std(history.history['val_loss']),\n                 np.array(history.history['val_loss']) + np.std(history.history['val_loss']),\n                 color='gray', alpha=0.2)\n\nplt.title('Training and Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\n\nplt.figure(figsize=(10, 5))\n\n# Training AUC and its standard deviation\nplt.plot(history.history['auc'], label='Training AUC') \nplt.fill_between(range(len(history.history['auc'])),\n                 np.array(history.history['auc']) - np.std(history.history['auc']),\n                 np.array(history.history['auc']) + np.std(history.history['auc']),\n                 color='gray', alpha=0.2)\n\n# Validation AUC and its standard deviation\nplt.plot(history.history['val_auc'], label='Validation AUC')  \nplt.fill_between(range(len(history.history['val_auc'])),\n                 np.array(history.history['val_auc']) - np.std(history.history['val_auc']),\n                 np.array(history.history['val_auc']) + np.std(history.history['val_auc']),\n                 color='gray', alpha=0.2)\n\nplt.title('Training and Validation AUC')\nplt.xlabel('Epochs')\nplt.ylabel('AUC')\nplt.legend()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After running the code, TensorBoard logs can be visualized using the following command. This will open a web interface where you can view the loss and other metrics.","metadata":{}},{"cell_type":"code","source":"%load_ext tensorboard\n%tensorboard --logdir logs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will now evaluate the neural network performance with AUC and gini stability below.","metadata":{}},{"cell_type":"code","source":"for base, X in [(base_train, X_train), (base_valid, X_valid), (base_test, X_test)]:\n    # Get predictions from the neural network model\n    y_pred = model.predict(X).ravel() \n    base[\"score\"] = y_pred\n\nprint(f'The AUC score on the train set is: {roc_auc_score(base_train[\"target\"], base_train[\"score\"])}') \nprint(f'The AUC score on the valid set is: {roc_auc_score(base_valid[\"target\"], base_valid[\"score\"])}') \nprint(f'The AUC score on the test set is: {roc_auc_score(base_test[\"target\"], base_test[\"score\"])}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\n\nstability_score_train = gini_stability(base_train)\nstability_score_valid = gini_stability(base_valid)\nstability_score_test = gini_stability(base_test)\n\nprint(f'The stability score on the train set is: {stability_score_train}') \nprint(f'The stability score on the valid set is: {stability_score_valid}') \nprint(f'The stability score on the test set is: {stability_score_test}')  ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Submission**\n\nWe will now make predictions with the neural network for the submission dataset.","metadata":{}},{"cell_type":"code","source":"y_submission_pred = model.predict(X_submission).ravel()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n    \"case_id\": data_submission[\"case_id\"].to_numpy(),\n    \"score\": y_submission_pred\n}).set_index('case_id')\nsubmission.to_csv(\"./submission.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"View the submission dataframe.","metadata":{}},{"cell_type":"code","source":"submission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}