{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-06T06:39:38.459001Z","iopub.execute_input":"2024-03-06T06:39:38.459352Z","iopub.status.idle":"2024-03-06T06:39:39.542430Z","shell.execute_reply.started":"2024-03-06T06:39:38.459324Z","shell.execute_reply":"2024-03-06T06:39:39.541507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:39.545279Z","iopub.execute_input":"2024-03-06T06:39:39.545891Z","iopub.status.idle":"2024-03-06T06:39:40.462999Z","shell.execute_reply.started":"2024-03-06T06:39:39.545849Z","shell.execute_reply":"2024-03-06T06:39:40.461886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## What I want?\n\nI am pretty new to data science and also to Kaggle. I have participated in few competitons before but mostly they are playground series with toy datasets. I wish to delve deeper into serious and more real world challenges. This competiton might be too much for me but I wish to give it a try anyway.\nMy primary focus is to learn new things and thats it!!\nHere is my wish list\n- Improving on their large data manipulation skill\n- Use the `polar` library written in `Rust` for data manipulation and get used to it\n- Do basic and simple EDA and get insight about data\n- Once I have a good understanding of the data, start compiling this into more concise formats\n- Do some feature engineering\n- Make simple model\n- Analyze feature importance and repeat","metadata":{}},{"cell_type":"markdown","source":"## About this challenge\n\nThe goal of this competition is to predict which clients are more likely to default on their loans. The evaluation will favor solutions that are stable over time.","metadata":{}},{"cell_type":"markdown","source":"## About the datasets","metadata":{}},{"cell_type":"code","source":"base_path = \"/kaggle/input/home-credit-credit-risk-model-stability/\"\ntrain_path = os.path.join(base_path, \"parquet_files/train\")\ntest_path = os.path.join(base_path, \"parquet_files/test\")","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:40.464137Z","iopub.execute_input":"2024-03-06T06:39:40.464429Z","iopub.status.idle":"2024-03-06T06:39:40.471474Z","shell.execute_reply.started":"2024-03-06T06:39:40.464405Z","shell.execute_reply":"2024-03-06T06:39:40.470224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_desc_df = pl.read_csv(os.path.join(base_path, \"feature_definitions.csv\"))\n# simple function for column description\ndef show_feature_def(feature_name: list[str]) -> pl.DataFrame:\n    if not isinstance(feature_name, list):\n        feature_name = [feature_name]\n    feature_def = feature_desc_df.filter(pl.col(\"Variable\").is_in(feature_name))\n    table_def_list = [(row[0], row[1]) for row in feature_def.iter_rows()]\n    table_def_list = sorted(table_def_list, key = lambda x: x[0][-1])\n    transform_type = ''\n    for row in table_def_list:\n        if row[0][-1] != transform_type:\n            transform_type = row[0][-1]\n            print(f\"\\n\\n---- Column transform type {transform_type} ---- \\n\\n\")\n        print(f\"{row[0]} --> {row[1]}\")","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:40.474663Z","iopub.execute_input":"2024-03-06T06:39:40.476048Z","iopub.status.idle":"2024-03-06T06:39:40.622316Z","shell.execute_reply.started":"2024-03-06T06:39:40.475964Z","shell.execute_reply":"2024-03-06T06:39:40.621339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# only interseted on training files as of now\ntrain_file_names = []\nfor dirname, _, filenames in os.walk(train_path):\n    for filename in filenames:\n        if filename.endswith(\".parquet\"):\n            train_file_names.append(filename)\nprint(*sorted(train_file_names), sep='\\n')\nprint(f\"\\n\\nTotal number of training files is {len(train_file_names)}\")","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:40.623291Z","iopub.execute_input":"2024-03-06T06:39:40.623694Z","iopub.status.idle":"2024-03-06T06:39:40.635012Z","shell.execute_reply.started":"2024-03-06T06:39:40.623657Z","shell.execute_reply":"2024-03-06T06:39:40.633587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Base table\ntrain_base_df = pl.read_parquet(os.path.join(train_path, \"train_base.parquet\"))\nprint(train_base_df.shape)\nprint(train_base_df.head(10))","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:40.636792Z","iopub.execute_input":"2024-03-06T06:39:40.637290Z","iopub.status.idle":"2024-03-06T06:39:40.852524Z","shell.execute_reply.started":"2024-03-06T06:39:40.637249Z","shell.execute_reply":"2024-03-06T06:39:40.851388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Base tables store the basic information about the observation and case_id. This is a unique identification of every observation and you need to use it to join the other tables to base tables.\n\nDepth values:\n\n- depth=0 - These are static features directly tied to a specific case_id.\n- depth=1 - Each case_id has an associated historical record, indexed by num_group1.\n- depth=2 - Each case_id has an associated historical record, indexed by both num_group1 and num_group2.","metadata":{}},{"cell_type":"code","source":"# depth-0 tables\ntrain_static_0_0_df = pl.read_parquet(os.path.join(train_path, \"train_static_0_0.parquet\"))\ntrain_static_0_1_df = pl.read_parquet(os.path.join(train_path, \"train_static_0_1.parquet\"))\nprint(train_static_0_0_df.shape)\nprint(train_static_0_0_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:40.853684Z","iopub.execute_input":"2024-03-06T06:39:40.854083Z","iopub.status.idle":"2024-03-06T06:39:43.976600Z","shell.execute_reply.started":"2024-03-06T06:39:40.854045Z","shell.execute_reply":"2024-03-06T06:39:43.975706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Depth-1 and Depth-2 tables require some kind of aggregation across num_groupN before use. These tables have multiple records corresponding to single case-id. Mostly past loan applications and their status","metadata":{}},{"cell_type":"code","source":"# depth-1 tables\ntrain_applprev_1_0_df = pl.read_parquet(os.path.join(train_path, \"train_applprev_1_0.parquet\"))\ntrain_applprev_1_1_df = pl.read_parquet(os.path.join(train_path, \"train_applprev_1_1.parquet\"))\nprint(train_applprev_1_0_df.shape)\nprint(train_applprev_1_0_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:43.977505Z","iopub.execute_input":"2024-03-06T06:39:43.977797Z","iopub.status.idle":"2024-03-06T06:39:47.697033Z","shell.execute_reply.started":"2024-03-06T06:39:43.977774Z","shell.execute_reply":"2024-03-06T06:39:47.695836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_applprev_1_0_df.group_by(pl.col(\"case_id\"), \n                                     maintain_order=True).len())","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:47.698416Z","iopub.execute_input":"2024-03-06T06:39:47.698863Z","iopub.status.idle":"2024-03-06T06:39:47.984107Z","shell.execute_reply.started":"2024-03-06T06:39:47.698824Z","shell.execute_reply":"2024-03-06T06:39:47.983155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show a single case_id in depth-1 table\nprint(train_applprev_1_0_df.filter(pl.col(\"case_id\") == 2651089)\n     .select(pl.col([\"case_id\", \"num_group1\",\"status_219L\",\n                     'creationdate_885D', \"credamount_590A\", \n                    \"approvaldate_319D\", \"credacc_actualbalance_314A\",\n                    \"downpmt_134A\"])))","metadata":{"execution":{"iopub.status.busy":"2024-03-06T08:03:58.695554Z","iopub.execute_input":"2024-03-06T08:03:58.696023Z","iopub.status.idle":"2024-03-06T08:03:58.739330Z","shell.execute_reply.started":"2024-03-06T08:03:58.695990Z","shell.execute_reply":"2024-03-06T08:03:58.738122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# depth-1 tables\ntrain_applprev_2_df = pl.read_parquet(os.path.join(train_path, \"train_applprev_2.parquet\"))\nprint(train_applprev_2_df.shape)\nprint(train_applprev_2_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:48.027340Z","iopub.execute_input":"2024-03-06T06:39:48.028457Z","iopub.status.idle":"2024-03-06T06:39:49.345078Z","shell.execute_reply.started":"2024-03-06T06:39:48.028416Z","shell.execute_reply":"2024-03-06T06:39:49.344136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_applprev_2_df.group_by([\"case_id\", \"num_group1\"], maintain_order=True).agg(pl.col(\"num_group2\").len()))","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:49.345958Z","iopub.execute_input":"2024-03-06T06:39:49.346252Z","iopub.status.idle":"2024-03-06T06:39:51.579392Z","shell.execute_reply.started":"2024-03-06T06:39:49.346228Z","shell.execute_reply":"2024-03-06T06:39:51.578517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Find out table contents","metadata":{}},{"cell_type":"markdown","source":"Various predictors were transformed, therefore we have the following notation for similar groups of transformations\n\n- P - Transform DPD (Days past due)\n- M - Masking categories\n- A - Transform amount\n- D - Transform date\n- T - Unspecified Transform\n- L - Unspecified Transform\n\nPlease note that transformations within a group are denoted by a capital letter at the end of the predictor name (e.g., **maxdbddpdtollast6m_4187119P**). ","metadata":{}},{"cell_type":"code","source":"# previous applications_1\nprint(train_applprev_1_0_df.shape)\nshow_feature_def(train_applprev_1_0_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:51.606349Z","iopub.execute_input":"2024-03-06T06:39:51.606798Z","iopub.status.idle":"2024-03-06T06:39:51.628468Z","shell.execute_reply.started":"2024-03-06T06:39:51.606763Z","shell.execute_reply":"2024-03-06T06:39:51.627397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# applicaton previous_2\nprint(train_applprev_2_df.shape)\nshow_feature_def(train_applprev_2_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:51.629952Z","iopub.execute_input":"2024-03-06T06:39:51.630590Z","iopub.status.idle":"2024-03-06T06:39:51.637701Z","shell.execute_reply.started":"2024-03-06T06:39:51.630553Z","shell.execute_reply":"2024-03-06T06:39:51.636454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# static columns\nprint(train_static_0_0_df.shape)\nshow_feature_def(train_static_0_0_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:51.639345Z","iopub.execute_input":"2024-03-06T06:39:51.640036Z","iopub.status.idle":"2024-03-06T06:39:51.652947Z","shell.execute_reply.started":"2024-03-06T06:39:51.639941Z","shell.execute_reply":"2024-03-06T06:39:51.651664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_static_cb_0_df = pl.read_parquet(os.path.join(train_path, \"train_static_cb_0.parquet\"))\nprint(train_static_cb_0_df.shape)\nshow_feature_def(train_static_cb_0_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:51.654491Z","iopub.execute_input":"2024-03-06T06:39:51.655122Z","iopub.status.idle":"2024-03-06T06:39:52.337084Z","shell.execute_reply.started":"2024-03-06T06:39:51.655082Z","shell.execute_reply":"2024-03-06T06:39:52.336079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_credit_bureau_a_1_0_df = pl.read_parquet(os.path.join(train_path, \"train_credit_bureau_a_1_0.parquet\"))\n# train_credit_bureau_a_1_1_df = pl.read_parquet(os.path.join(train_path, \"train_credit_bureau_a_1_1.parquet\"))\n# train_credit_bureau_a_1_2_df = pl.read_parquet(os.path.join(train_path, \"train_credit_bureau_a_1_2.parquet\"))\n# train_credit_bureau_a_1_3_df = pl.read_parquet(os.path.join(train_path, \"train_credit_bureau_a_1_3.parquet\"))\nprint(train_credit_bureau_a_1_0_df.shape)\nshow_feature_def(train_credit_bureau_a_1_0_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:52.338098Z","iopub.execute_input":"2024-03-06T06:39:52.338424Z","iopub.status.idle":"2024-03-06T06:39:54.843847Z","shell.execute_reply.started":"2024-03-06T06:39:52.338397Z","shell.execute_reply":"2024-03-06T06:39:54.842697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_credit_bureau_a_2_0_df = pl.read_parquet(os.path.join(train_path, \"train_credit_bureau_a_2_0.parquet\"))\nprint(train_credit_bureau_a_2_0_df.shape)\nshow_feature_def(train_credit_bureau_a_2_0_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:54.844971Z","iopub.execute_input":"2024-03-06T06:39:54.845382Z","iopub.status.idle":"2024-03-06T06:39:55.729197Z","shell.execute_reply.started":"2024-03-06T06:39:54.845345Z","shell.execute_reply":"2024-03-06T06:39:55.728009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_credit_bureau_b_1_df = pl.read_parquet(os.path.join(train_path, \"train_credit_bureau_b_1.parquet\"))\nprint(train_credit_bureau_b_1_df.shape)\nshow_feature_def(train_credit_bureau_b_1_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:55.730606Z","iopub.execute_input":"2024-03-06T06:39:55.731045Z","iopub.status.idle":"2024-03-06T06:39:55.808803Z","shell.execute_reply.started":"2024-03-06T06:39:55.731006Z","shell.execute_reply":"2024-03-06T06:39:55.807911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_credit_bureau_b_2_df = pl.read_parquet(os.path.join(train_path, \"train_credit_bureau_b_2.parquet\"))\nprint(train_credit_bureau_b_2_df.shape)\nshow_feature_def(train_credit_bureau_b_2_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:55.809838Z","iopub.execute_input":"2024-03-06T06:39:55.810238Z","iopub.status.idle":"2024-03-06T06:39:55.918866Z","shell.execute_reply.started":"2024-03-06T06:39:55.810201Z","shell.execute_reply":"2024-03-06T06:39:55.917724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_debitcard_1_df = pl.read_parquet(os.path.join(train_path, \"train_debitcard_1.parquet\"))\nprint(train_debitcard_1_df.shape)\nshow_feature_def(train_debitcard_1_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:55.920025Z","iopub.execute_input":"2024-03-06T06:39:55.920434Z","iopub.status.idle":"2024-03-06T06:39:55.952328Z","shell.execute_reply.started":"2024-03-06T06:39:55.920394Z","shell.execute_reply":"2024-03-06T06:39:55.951494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_deposit_1_df = pl.read_parquet(os.path.join(train_path, \"train_deposit_1.parquet\"))\nprint(train_deposit_1_df.shape)\nshow_feature_def(train_deposit_1_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:55.953427Z","iopub.execute_input":"2024-03-06T06:39:55.954200Z","iopub.status.idle":"2024-03-06T06:39:55.987324Z","shell.execute_reply.started":"2024-03-06T06:39:55.954168Z","shell.execute_reply":"2024-03-06T06:39:55.986516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_person_1_df = pl.read_parquet(os.path.join(train_path, \"train_person_1.parquet\"))\nprint(train_person_1_df.shape)\nshow_feature_def(train_person_1_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:55.988734Z","iopub.execute_input":"2024-03-06T06:39:55.989806Z","iopub.status.idle":"2024-03-06T06:39:57.310921Z","shell.execute_reply.started":"2024-03-06T06:39:55.989765Z","shell.execute_reply":"2024-03-06T06:39:57.310027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_person_2_df = pl.read_parquet(os.path.join(train_path, \"train_person_2.parquet\"))\nprint(train_person_2_df.shape)\nshow_feature_def(train_person_2_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:57.312204Z","iopub.execute_input":"2024-03-06T06:39:57.312626Z","iopub.status.idle":"2024-03-06T06:39:57.538383Z","shell.execute_reply.started":"2024-03-06T06:39:57.312585Z","shell.execute_reply":"2024-03-06T06:39:57.537613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_tax_registry_a_1_df = pl.read_parquet(os.path.join(train_path, \"train_tax_registry_a_1.parquet\"))\nprint(train_tax_registry_a_1_df.shape)\nshow_feature_def(train_tax_registry_a_1_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:57.539284Z","iopub.execute_input":"2024-03-06T06:39:57.539572Z","iopub.status.idle":"2024-03-06T06:39:58.557610Z","shell.execute_reply.started":"2024-03-06T06:39:57.539548Z","shell.execute_reply":"2024-03-06T06:39:58.556716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_other_1_df = pl.read_parquet(os.path.join(train_path, \"train_other_1.parquet\"))\nprint(train_other_1_df.shape)\nshow_feature_def(train_other_1_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:58.558473Z","iopub.execute_input":"2024-03-06T06:39:58.558831Z","iopub.status.idle":"2024-03-06T06:39:58.586857Z","shell.execute_reply.started":"2024-03-06T06:39:58.558795Z","shell.execute_reply":"2024-03-06T06:39:58.585736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Concat similar tables","metadata":{}},{"cell_type":"code","source":"train_static_df = pl.concat([\n    train_static_0_0_df, train_static_0_1_df\n])\nprint(train_static_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:39:58.592288Z","iopub.execute_input":"2024-03-06T06:39:58.593240Z","iopub.status.idle":"2024-03-06T06:40:00.802162Z","shell.execute_reply.started":"2024-03-06T06:39:58.593201Z","shell.execute_reply":"2024-03-06T06:40:00.801135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_applprev_1_df = pl.concat([\n    train_applprev_1_0_df, train_applprev_1_1_df\n])\nprint(train_applprev_1_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:40:00.803241Z","iopub.execute_input":"2024-03-06T06:40:00.803681Z","iopub.status.idle":"2024-03-06T06:40:03.698685Z","shell.execute_reply.started":"2024-03-06T06:40:00.803622Z","shell.execute_reply":"2024-03-06T06:40:03.697867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Simple EDA","metadata":{}},{"cell_type":"code","source":"px.pie(train_base_df.select(pl.col(\"target\")).to_series().value_counts(),\n       values=\"count\", names=\"target\",\n       title=\"Target distribution in training data\")","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:40:03.699907Z","iopub.execute_input":"2024-03-06T06:40:03.700234Z","iopub.status.idle":"2024-03-06T06:40:05.647735Z","shell.execute_reply.started":"2024-03-06T06:40:03.700206Z","shell.execute_reply":"2024-03-06T06:40:05.646708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Only a small fraction is 1, heavily imbalance dataset","metadata":{}},{"cell_type":"markdown","source":"We need to have prediction stability across `WEEK_NUM`. So it is worth inspecting data distribution across this.","metadata":{}},{"cell_type":"code","source":"week_dist = train_base_df.group_by(\"WEEK_NUM\", maintain_order=True).len().cast(pl.Int32)\nfig = px.bar(week_dist,\n       x=\"WEEK_NUM\", y=\"len\", title=\"Number of data points accross WEEK_NUM\")\nfig.update_traces(marker_color='green')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:44:25.834854Z","iopub.execute_input":"2024-03-06T06:44:25.835226Z","iopub.status.idle":"2024-03-06T06:44:25.914877Z","shell.execute_reply.started":"2024-03-06T06:44:25.835196Z","shell.execute_reply":"2024-03-06T06:44:25.913690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"week_wise_target_dist = train_base_df.group_by(\"WEEK_NUM\", maintain_order=True)\\\n.agg(pl.col(\"target\").mean().alias(\"avg_target_count\"))\npx.bar(week_wise_target_dist,\n       x=\"WEEK_NUM\", y=\"avg_target_count\",\n       title = \"Average defaults-rate(1) accross WEEK_NUM\")","metadata":{"execution":{"iopub.status.busy":"2024-03-06T08:07:44.789374Z","iopub.execute_input":"2024-03-06T08:07:44.789851Z","iopub.status.idle":"2024-03-06T08:07:44.872083Z","shell.execute_reply.started":"2024-03-06T08:07:44.789816Z","shell.execute_reply":"2024-03-06T08:07:44.870940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Null values","metadata":{}},{"cell_type":"code","source":"def visualize_null(df: pl.DataFrame)-> None:\n    null_df = df.select(pl.all().is_null().mean())\\\n                .transpose(include_header=True, column_names=[\"null_count\"])\\\n                .filter(pl.col(\"null_count\") > 0)\n    fig = px.bar(null_df, y=\"null_count\", x=\"column\", \n          title=f\"Fraction of null values across {null_df.shape[0]} columns\")\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-06T07:52:41.167577Z","iopub.execute_input":"2024-03-06T07:52:41.167973Z","iopub.status.idle":"2024-03-06T07:52:41.174674Z","shell.execute_reply.started":"2024-03-06T07:52:41.167943Z","shell.execute_reply":"2024-03-06T07:52:41.173492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_null(train_applprev_1_df)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T07:52:43.021944Z","iopub.execute_input":"2024-03-06T07:52:43.022385Z","iopub.status.idle":"2024-03-06T07:52:43.088010Z","shell.execute_reply.started":"2024-03-06T07:52:43.022354Z","shell.execute_reply":"2024-03-06T07:52:43.087014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_null(train_applprev_2_df)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T07:58:29.039081Z","iopub.execute_input":"2024-03-06T07:58:29.039457Z","iopub.status.idle":"2024-03-06T07:58:29.103198Z","shell.execute_reply.started":"2024-03-06T07:58:29.039429Z","shell.execute_reply":"2024-03-06T07:58:29.101949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_null(train_static_df)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T07:52:54.269250Z","iopub.execute_input":"2024-03-06T07:52:54.269986Z","iopub.status.idle":"2024-03-06T07:52:54.335448Z","shell.execute_reply.started":"2024-03-06T07:52:54.269941Z","shell.execute_reply":"2024-03-06T07:52:54.334396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_null(train_static_cb_0_df)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T07:59:22.246539Z","iopub.execute_input":"2024-03-06T07:59:22.247006Z","iopub.status.idle":"2024-03-06T07:59:22.313489Z","shell.execute_reply.started":"2024-03-06T07:59:22.246972Z","shell.execute_reply":"2024-03-06T07:59:22.312351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_null(train_credit_bureau_a_1_0_df)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T07:54:47.961015Z","iopub.execute_input":"2024-03-06T07:54:47.961391Z","iopub.status.idle":"2024-03-06T07:54:48.029263Z","shell.execute_reply.started":"2024-03-06T07:54:47.961363Z","shell.execute_reply":"2024-03-06T07:54:48.028226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Most of the column in datasets has large number of null values\n- We need to consider if a feature can be used in our model if it has large number of null values","metadata":{}}]}