{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom glob import glob\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-04T18:28:38.976877Z","iopub.execute_input":"2024-05-04T18:28:38.977665Z","iopub.status.idle":"2024-05-04T18:28:40.336868Z","shell.execute_reply.started":"2024-05-04T18:28:38.977615Z","shell.execute_reply":"2024-05-04T18:28:40.335490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" **Data Wrangling**","metadata":{}},{"cell_type":"code","source":"import polars as pl\n\ndataPath = \"/kaggle/input/home-credit-credit-risk-model-stability/\"\ntrainPath = dataPath+\"parquet_files/train/\"\ntestPath = dataPath+\"parquet_files/test/\"","metadata":{"execution":{"iopub.status.busy":"2024-05-04T18:28:49.986926Z","iopub.execute_input":"2024-05-04T18:28:49.988713Z","iopub.status.idle":"2024-05-04T18:28:50.411400Z","shell.execute_reply.started":"2024-05-04T18:28:49.988634Z","shell.execute_reply":"2024-05-04T18:28:50.409680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Load Data**","metadata":{}},{"cell_type":"code","source":"def get_features(df):\n    cols=[col for col in df.columns if col[-1] in (\"P\",\"M\",\"A\",\"D\")]\n    expr_max=[pl.max(col).alias(f\"max_{col}\") for col in cols]\n    expr_mean=[pl.mean(col).alias(f\"mean{col}\") for col in cols]\n    return expr_max+expr_mean\n\ndef convert_dtypes(df):\n    for col in df.columns:\n        if col in [\"case_id\",\"WEEK_NUM\",\"num_group1\",\"num_group2\"] or col[-1] in [\"P\",]:\n            df=df.with_columns(pl.col(col).cast(pl.Int64))\n        if col[-1] in [\"A\",]:\n            df=df.with_columns(pl.col(col).cast(pl.Float32))\n    return df\n    \ndef read_file(path,depth=0):\n    df=pl.read_parquet(path)\n    df=convert_dtypes(df)\n    if depth==1 or depth==2:\n        df=df.group_by(\"case_id\").agg(get_features(df))\n    return df\n\ndef read_files(path,depth=0):\n    files=[]\n    for path_ in glob(str(path)):\n        df=pl.read_parquet(path_)\n        df=convert_dtypes(df)\n        if depth==1 or depth==2:\n            df=df.group_by(\"case_id\").agg(get_features(df))\n        files.append(df)\n    df=pl.concat(files, how=\"vertical_relaxed\")\n    df=df.unique(subset=\"case_id\")\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-04T18:28:54.754891Z","iopub.execute_input":"2024-05-04T18:28:54.755288Z","iopub.status.idle":"2024-05-04T18:28:54.769121Z","shell.execute_reply.started":"2024-05-04T18:28:54.755259Z","shell.execute_reply":"2024-05-04T18:28:54.767955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store={\n    #this idea comes from https://www.kaggle.com/code/hytopiaskywalker/home-credit-lgb-cat-ensemble/edit\n    #also from https://www.kaggle.com/code/greysky/home-credit-baseline/notebook?fbclid=IwAR3Skv2WH44fPgLcgLjG45m9bJMd0Ss4df8KsfSXG7POK8VzNadJkUVKoZ4\n    \"df_base\": read_file(trainPath + \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(trainPath + \"train_static_cb_0.parquet\"),\n        read_files(trainPath + \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(trainPath + \"train_applprev_1_*.parquet\", 1),\n        read_file(trainPath + \"train_tax_registry_a_1.parquet\", 1),\n        read_file(trainPath + \"train_tax_registry_b_1.parquet\", 1),\n        read_file(trainPath + \"train_tax_registry_c_1.parquet\", 1),\n        read_files(trainPath + \"train_credit_bureau_a_1_*.parquet\", 1),\n        read_file(trainPath + \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(trainPath + \"train_other_1.parquet\", 1),\n        read_file(trainPath + \"train_person_1.parquet\", 1),\n        read_file(trainPath + \"train_deposit_1.parquet\", 1),\n        read_file(trainPath + \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(trainPath + \"train_credit_bureau_b_2.parquet\", 2),\n        read_files(trainPath + \"train_credit_bureau_a_2_*.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-04T18:29:01.267591Z","iopub.execute_input":"2024-05-04T18:29:01.268059Z","iopub.status.idle":"2024-05-04T18:31:09.526579Z","shell.execute_reply.started":"2024-05-04T18:29:01.268021Z","shell.execute_reply":"2024-05-04T18:31:09.524918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_base=data_store[\"df_base\"]\nfor i, df in enumerate(data_store[\"depth_0\"] + data_store[\"depth_1\"] + data_store[\"depth_2\"]):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-04T18:28:32.555712Z","iopub.execute_input":"2024-05-04T18:28:32.556504Z","iopub.status.idle":"2024-05-04T18:28:32.894265Z","shell.execute_reply.started":"2024-05-04T18:28:32.556466Z","shell.execute_reply":"2024-05-04T18:28:32.892556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train=df_base","metadata":{"execution":{"iopub.status.busy":"2024-05-04T18:26:29.579853Z","iopub.execute_input":"2024-05-04T18:26:29.581660Z","iopub.status.idle":"2024-05-04T18:26:29.593600Z","shell.execute_reply.started":"2024-05-04T18:26:29.581427Z","shell.execute_reply":"2024-05-04T18:26:29.591235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store={\n    #this idea comes from https://www.kaggle.com/code/hytopiaskywalker/home-credit-lgb-cat-ensemble/edit\n    #also from https://www.kaggle.com/code/greysky/home-credit-baseline/notebook?fbclid=IwAR3Skv2WH44fPgLcgLjG45m9bJMd0Ss4df8KsfSXG7POK8VzNadJkUVKoZ4\n    \"df_base\": read_file(testPath + \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(testPath + \"test_static_cb_0.parquet\"),\n        read_files(testPath + \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(testPath + \"test_applprev_1_*.parquet\", 1),\n        read_file(testPath + \"test_tax_registry_a_1.parquet\", 1),\n        read_file(testPath + \"test_tax_registry_b_1.parquet\", 1),\n        read_file(testPath + \"test_tax_registry_c_1.parquet\", 1),\n        read_files(testPath + \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(testPath + \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(testPath + \"test_other_1.parquet\", 1),\n        read_file(testPath + \"test_person_1.parquet\", 1),\n        read_file(testPath + \"test_deposit_1.parquet\", 1),\n        read_file(testPath + \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(testPath + \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(testPath + \"test_credit_bureau_a_2_*.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-04T18:26:29.595956Z","iopub.execute_input":"2024-05-04T18:26:29.596874Z","iopub.status.idle":"2024-05-04T18:26:30.988225Z","shell.execute_reply.started":"2024-05-04T18:26:29.596808Z","shell.execute_reply":"2024-05-04T18:26:30.983883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test=data_store[\"df_base\"]\nfor i, df in enumerate(data_store[\"depth_0\"] + data_store[\"depth_1\"] + data_store[\"depth_2\"]):\n        df_test = df_test.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-04T18:26:30.994635Z","iopub.execute_input":"2024-05-04T18:26:30.998163Z","iopub.status.idle":"2024-05-04T18:26:31.098501Z","shell.execute_reply.started":"2024-05-04T18:26:30.997844Z","shell.execute_reply":"2024-05-04T18:26:31.095445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Pipeline**","metadata":{}},{"cell_type":"markdown","source":"**Modelling**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score \nfrom sklearn.metrics import accuracy_score\nfrom xgboost import XGBClassifier","metadata":{"execution":{"iopub.status.busy":"2024-05-04T18:26:31.102011Z","iopub.execute_input":"2024-05-04T18:26:31.102960Z","iopub.status.idle":"2024-05-04T18:26:38.440663Z","shell.execute_reply.started":"2024-05-04T18:26:31.102870Z","shell.execute_reply":"2024-05-04T18:26:38.438181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train=df_train.drop([\"case_id\",\"MONTH\", \"WEEK_NUM\", \"date_decision\"])\ndf_test=df_test.drop([\"case_id\",\"MONTH\", \"WEEK_NUM\", \"date_decision\"])\nX=df_train.drop([\"target\"])\nY=df_train[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2024-05-04T18:26:38.453036Z","iopub.execute_input":"2024-05-04T18:26:38.456145Z","iopub.status.idle":"2024-05-04T18:26:38.478055Z","shell.execute_reply.started":"2024-05-04T18:26:38.456062Z","shell.execute_reply":"2024-05-04T18:26:38.475163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, Y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T18:26:38.481339Z","iopub.execute_input":"2024-05-04T18:26:38.482225Z","iopub.status.idle":"2024-05-04T18:27:03.622068Z","shell.execute_reply.started":"2024-05-04T18:26:38.482185Z","shell.execute_reply":"2024-05-04T18:27:03.620328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_classifier = XGBClassifier()\nxgb_classifier.fit(X_train, y_train)\n\ny_pred = xgb_classifier.predict(X_test)\nprint(f'Accuracy = {accuracy_score(y_test, y_pred).round(3)}')","metadata":{"execution":{"iopub.status.busy":"2024-05-04T18:27:03.624226Z","iopub.execute_input":"2024-05-04T18:27:03.624836Z"},"trusted":true},"execution_count":null,"outputs":[]}]}