{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":7832975,"sourceType":"datasetVersion","datasetId":4590919}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import polars as pl\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.metrics import roc_auc_score \nimport pickle\nimport matplotlib.pyplot as plt\n\ndataPath = \"/kaggle/input/home-credit-credit-risk-model-stability/\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-14T01:25:36.839404Z","iopub.execute_input":"2024-03-14T01:25:36.839811Z","iopub.status.idle":"2024-03-14T01:25:38.351635Z","shell.execute_reply.started":"2024-03-14T01:25:36.839779Z","shell.execute_reply":"2024-03-14T01:25:38.350152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:25:38.353756Z","iopub.execute_input":"2024-03-14T01:25:38.354330Z","iopub.status.idle":"2024-03-14T01:25:39.825129Z","shell.execute_reply.started":"2024-03-14T01:25:38.354281Z","shell.execute_reply":"2024-03-14T01:25:39.823816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:25:39.826487Z","iopub.execute_input":"2024-03-14T01:25:39.826887Z","iopub.status.idle":"2024-03-14T01:25:39.855421Z","shell.execute_reply.started":"2024-03-14T01:25:39.826855Z","shell.execute_reply":"2024-03-14T01:25:39.854225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation metric","metadata":{}},{"cell_type":"markdown","source":"Submissions are evaluated using a gini stability metric. A gini score is calculated for predictions corresponding to each WEEK_NUM.\n\n$gini=2∗AUC−1$\n\nA linear regression, $𝑎⋅𝑥+𝑏$\n, is fit through the weekly gini scores, and a falling_rate is calculated as $min(0,𝑎)$\n. This is used to penalize models that drop off in predictive ability.\n\nFinally, the variability of the predictions are calculated by taking the standard deviation of the residuals from the above linear regression, applying a penalty to model variablity.\n\nThe final metric is calculated as\n\n$stability metric=𝑚𝑒𝑎𝑛(𝑔𝑖𝑛𝑖)+88.0⋅𝑚𝑖𝑛(0,𝑎)−0.5⋅𝑠𝑡𝑑(residuals)$","metadata":{}},{"cell_type":"code","source":"def gini_stability(prediction_df, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = prediction_df.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:25:39.860688Z","iopub.execute_input":"2024-03-14T01:25:39.861060Z","iopub.status.idle":"2024-03-14T01:25:39.870421Z","shell.execute_reply.started":"2024-03-14T01:25:39.861032Z","shell.execute_reply":"2024-03-14T01:25:39.868994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_df = pd.read_csv('/kaggle/input/hc-train-pred/train_prediction.csv')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:27:16.786203Z","iopub.execute_input":"2024-03-14T01:27:16.786654Z","iopub.status.idle":"2024-03-14T01:27:18.084862Z","shell.execute_reply.started":"2024-03-14T01:27:16.786618Z","shell.execute_reply":"2024-03-14T01:27:18.083521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_df","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:27:18.740988Z","iopub.execute_input":"2024-03-14T01:27:18.741415Z","iopub.status.idle":"2024-03-14T01:27:18.759272Z","shell.execute_reply.started":"2024-03-14T01:27:18.741379Z","shell.execute_reply":"2024-03-14T01:27:18.757855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_by_week = prediction_df.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n    .sort_values(\"WEEK_NUM\")\\\n    .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n    .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).rename('gini_score').reset_index()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:27:19.254774Z","iopub.execute_input":"2024-03-14T01:27:19.257856Z","iopub.status.idle":"2024-03-14T01:27:20.063027Z","shell.execute_reply.started":"2024-03-14T01:27:19.257803Z","shell.execute_reply":"2024-03-14T01:27:20.062124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_by_week","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:27:20.064613Z","iopub.execute_input":"2024-03-14T01:27:20.065151Z","iopub.status.idle":"2024-03-14T01:27:20.077711Z","shell.execute_reply.started":"2024-03-14T01:27:20.065119Z","shell.execute_reply":"2024-03-14T01:27:20.076581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_gini = gini_by_week['gini_score'].mean()\nprint(f\"mean gini score: {mean_gini}\")","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:27:21.662837Z","iopub.execute_input":"2024-03-14T01:27:21.663214Z","iopub.status.idle":"2024-03-14T01:27:21.670012Z","shell.execute_reply.started":"2024-03-14T01:27:21.663185Z","shell.execute_reply":"2024-03-14T01:27:21.668603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nax.plot(gini_by_week['WEEK_NUM'], gini_by_week['gini_score'], marker='.', linestyle=':')\nplt.xlabel('Week Number')\nplt.ylabel('Gini Score')\nplt.title('Gini Score vs Week')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:27:22.157646Z","iopub.execute_input":"2024-03-14T01:27:22.158067Z","iopub.status.idle":"2024-03-14T01:27:22.455058Z","shell.execute_reply.started":"2024-03-14T01:27:22.158034Z","shell.execute_reply":"2024-03-14T01:27:22.453706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = gini_by_week['WEEK_NUM'].values\ny = gini_by_week['gini_score'].values\na, b = np.polyfit(x, y, 1)\ny_hat = a*x + b\nresiduals = y - y_hat\nstd_residuals = np.std(residuals)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:27:24.013569Z","iopub.execute_input":"2024-03-14T01:27:24.013952Z","iopub.status.idle":"2024-03-14T01:27:24.021477Z","shell.execute_reply.started":"2024-03-14T01:27:24.013922Z","shell.execute_reply":"2024-03-14T01:27:24.020057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nax.plot(gini_by_week['WEEK_NUM'], gini_by_week['gini_score'], marker='.', linestyle=':', label='data')\nax.plot(x, y_hat, color='red', label=f'y ={round(a,4)}*x + {round(b, 3)}')\nplt.xlabel('Week Number')\nplt.ylabel('Gini Score')\nplt.title('Gini Score vs Week')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:27:24.766911Z","iopub.execute_input":"2024-03-14T01:27:24.767325Z","iopub.status.idle":"2024-03-14T01:27:25.049327Z","shell.execute_reply.started":"2024-03-14T01:27:24.767295Z","shell.execute_reply":"2024-03-14T01:27:25.048081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_by_week['mean_gini'] = mean_gini\ngini_by_week['residuals'] = residuals\ngini_by_week['std_residuals'] = std_residuals","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:25:43.590392Z","iopub.execute_input":"2024-03-14T01:25:43.590776Z","iopub.status.idle":"2024-03-14T01:25:43.599438Z","shell.execute_reply.started":"2024-03-14T01:25:43.590744Z","shell.execute_reply":"2024-03-14T01:25:43.598085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stability_score = mean_gini + 88*min(0,a) - 0.5 * std_residuals","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:25:43.601594Z","iopub.execute_input":"2024-03-14T01:25:43.602685Z","iopub.status.idle":"2024-03-14T01:25:43.609132Z","shell.execute_reply.started":"2024-03-14T01:25:43.602607Z","shell.execute_reply":"2024-03-14T01:25:43.607625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"mean gini: {round(mean_gini, 4)}\")\nprint(f\"falling rate: {round(a, 4)}\")\nprint(f\"standard deviation of residuals: {round(std_residuals, 4)}\")\nprint(f\"stability_score: {round(stability_score, 4)}\")","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:25:43.611159Z","iopub.execute_input":"2024-03-14T01:25:43.611992Z","iopub.status.idle":"2024-03-14T01:25:43.623951Z","shell.execute_reply.started":"2024-03-14T01:25:43.611944Z","shell.execute_reply":"2024-03-14T01:25:43.622673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Problem","metadata":{}},{"cell_type":"code","source":"# When week num is less than mean week num, make your prediction worse\ncondition = prediction_df['WEEK_NUM'] < (prediction_df['WEEK_NUM'].max()-prediction_df['WEEK_NUM'].min())/2+prediction_df['WEEK_NUM'].min() \nprediction_df.loc[condition, 'score'] = (prediction_df.loc[condition, 'score'] - 0.02).clip(0) ","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:26:03.577898Z","iopub.execute_input":"2024-03-14T01:26:03.578372Z","iopub.status.idle":"2024-03-14T01:26:03.630858Z","shell.execute_reply.started":"2024-03-14T01:26:03.578336Z","shell.execute_reply":"2024-03-14T01:26:03.629707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_by_week = prediction_df.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n    .sort_values(\"WEEK_NUM\")\\\n    .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n    .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).rename('gini_score').reset_index()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:26:16.986433Z","iopub.execute_input":"2024-03-14T01:26:16.987692Z","iopub.status.idle":"2024-03-14T01:26:17.730978Z","shell.execute_reply.started":"2024-03-14T01:26:16.987651Z","shell.execute_reply":"2024-03-14T01:26:17.730125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = gini_by_week['WEEK_NUM'].values\ny = gini_by_week['gini_score'].values\na, b = np.polyfit(x, y, 1)\ny_hat = a*x + b\nresiduals = y - y_hat\nstd_residuals = np.std(residuals)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:26:32.264930Z","iopub.execute_input":"2024-03-14T01:26:32.265360Z","iopub.status.idle":"2024-03-14T01:26:32.273247Z","shell.execute_reply.started":"2024-03-14T01:26:32.265326Z","shell.execute_reply":"2024-03-14T01:26:32.271879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nax.plot(gini_by_week['WEEK_NUM'], gini_by_week['gini_score'], marker='.', linestyle=':', label='data')\nax.plot(x, y_hat, color='red', label=f'y ={round(a,4)}*x + {round(b, 3)}')\nplt.xlabel('Week Number')\nplt.ylabel('Gini Score')\nplt.title('Gini Score vs Week')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:26:56.185243Z","iopub.execute_input":"2024-03-14T01:26:56.185705Z","iopub.status.idle":"2024-03-14T01:26:56.529566Z","shell.execute_reply.started":"2024-03-14T01:26:56.185670Z","shell.execute_reply":"2024-03-14T01:26:56.527629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Solution","metadata":{}},{"cell_type":"markdown","source":"* As we indicated previously, we will be changing date columns in test_base table, namely date_decision, MONTH and WEEK_NUM. The MONTH and WEEK_NUM columns will now only have one constant value\n* Training data is not affacted. \n* They will hidden values for WEEK_NUM to evaluate models on LB.","metadata":{}},{"cell_type":"markdown","source":"# Some EDA","metadata":{}},{"cell_type":"markdown","source":"## Base table","metadata":{}},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in ['date_decision']:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            # A - transfer amount, P - days past due \n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            # M - masked category\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            # D - date\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df \n    \ndef read_file(path):\n    df = pl.read_parquet(path).pipe(Pipeline.set_table_dtypes)\n    return df ","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:35:11.776533Z","iopub.execute_input":"2024-03-14T01:35:11.777329Z","iopub.status.idle":"2024-03-14T01:35:11.790342Z","shell.execute_reply.started":"2024-03-14T01:35:11.777284Z","shell.execute_reply":"2024-03-14T01:35:11.788606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_base = read_file(\"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_base.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:52:25.086456Z","iopub.execute_input":"2024-03-14T01:52:25.086891Z","iopub.status.idle":"2024-03-14T01:52:25.859722Z","shell.execute_reply.started":"2024-03-14T01:52:25.086857Z","shell.execute_reply":"2024-03-14T01:52:25.858218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"agg_by_week = df_base.group_by('WEEK_NUM').len().sort('WEEK_NUM')\nagg_by_week_target = df_base.filter(pl.col('target') == 1).group_by('WEEK_NUM').len().sort('WEEK_NUM').rename({'len':'target'})\nagg_by_week = agg_by_week.join(agg_by_week_target, on=['WEEK_NUM']).rename({'len':'total_case', 'target':'target_case'})\nagg_by_week = agg_by_week.with_columns((pl.col('target_case') / pl.col('total_case')).alias(\"target_case_%\"))\n\n# Create figure and axes\nfig, ax1 = plt.subplots()\n\n# Plot total with bar plot\nax1.bar(agg_by_week['WEEK_NUM'], agg_by_week['total_case'])\nax1.set_xlabel('Week Number')\nax1.set_ylabel('Total')\n\n# Create a second y-axis for percentage\nax2 = ax1.twinx()\nax2.plot(agg_by_week['WEEK_NUM'], agg_by_week['target_case_%'], color='r', marker='o')\nax2.set_ylabel('Target case %', color='r')\n\n# Title\nplt.title('Total case and target case % by week')\n\n# Show plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:53:40.102643Z","iopub.execute_input":"2024-03-14T01:53:40.103107Z","iopub.status.idle":"2024-03-14T01:53:40.848833Z","shell.execute_reply.started":"2024-03-14T01:53:40.103069Z","shell.execute_reply":"2024-03-14T01:53:40.847924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"agg_by_month = df_base.group_by('MONTH').len().sort('MONTH')\nagg_by_month_target = df_base.filter(pl.col('target') == 1).group_by('MONTH').len().sort('MONTH').rename({'len':'target'})\nagg_by_month = agg_by_month.join(agg_by_month_target, on=['MONTH']).rename({'len':'total_case', 'target':'target_case'})\nagg_by_month = agg_by_month.with_columns((pl.col('target_case') / pl.col('total_case')).alias(\"target_case_%\"))","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:54:04.546870Z","iopub.execute_input":"2024-03-14T01:54:04.547270Z","iopub.status.idle":"2024-03-14T01:54:04.591438Z","shell.execute_reply.started":"2024-03-14T01:54:04.547238Z","shell.execute_reply":"2024-03-14T01:54:04.590488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"agg_by_month = agg_by_month.cast({'MONTH': pl.String})","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:54:10.081006Z","iopub.execute_input":"2024-03-14T01:54:10.081431Z","iopub.status.idle":"2024-03-14T01:54:10.089919Z","shell.execute_reply.started":"2024-03-14T01:54:10.081396Z","shell.execute_reply":"2024-03-14T01:54:10.088324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create figure and axes\nfig, ax1 = plt.subplots()\n\n# Plot total with bar plot\nax1.bar(agg_by_month['MONTH'], agg_by_month['total_case'])\nax1.set_xlabel('Month')\nax1.set_ylabel('Total')\n\n# Create a second y-axis for percentage\nax2 = ax1.twinx()\nax2.plot(agg_by_month['MONTH'], agg_by_month['target_case_%'], color='r', marker='o')\nax2.set_ylabel('Target case %', color='r')\n\nax1.set_xticklabels(agg_by_month['MONTH'], rotation=45)\n\n# Title\nplt.title('Total case and target case % by month')\n\n# Show plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:54:20.814348Z","iopub.execute_input":"2024-03-14T01:54:20.814994Z","iopub.status.idle":"2024-03-14T01:54:21.346375Z","shell.execute_reply.started":"2024-03-14T01:54:20.814955Z","shell.execute_reply":"2024-03-14T01:54:21.344755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"credit_bureau = read_file(\"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_static_cb_0.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:56:33.770718Z","iopub.execute_input":"2024-03-14T01:56:33.771162Z","iopub.status.idle":"2024-03-14T01:56:36.267187Z","shell.execute_reply.started":"2024-03-14T01:56:33.771129Z","shell.execute_reply":"2024-03-14T01:56:36.265431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"credit_bureau","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:56:36.270138Z","iopub.execute_input":"2024-03-14T01:56:36.270650Z","iopub.status.idle":"2024-03-14T01:56:36.316840Z","shell.execute_reply.started":"2024-03-14T01:56:36.270612Z","shell.execute_reply":"2024-03-14T01:56:36.315615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cb_scoring = credit_bureau['riskassesment_302T'].value_counts()\ncredit_bureau = credit_bureau.join(df_base[['case_id', 'target']], on='case_id')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:56:36.318790Z","iopub.execute_input":"2024-03-14T01:56:36.319403Z","iopub.status.idle":"2024-03-14T01:56:36.864572Z","shell.execute_reply.started":"2024-03-14T01:56:36.319354Z","shell.execute_reply":"2024-03-14T01:56:36.862929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cb_scoring = cb_scoring.with_columns(pl.col('riskassesment_302T').fill_null(''))","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:56:43.400413Z","iopub.execute_input":"2024-03-14T01:56:43.401040Z","iopub.status.idle":"2024-03-14T01:56:43.411463Z","shell.execute_reply.started":"2024-03-14T01:56:43.400990Z","shell.execute_reply":"2024-03-14T01:56:43.409585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cb_scoring = cb_scoring.join(credit_bureau.filter(pl.col('target') == 1).\n                group_by('riskassesment_302T').len().\n                rename({'len':'target_case'}).\n                with_columns(pl.col(\"riskassesment_302T\").fill_null('')), on='riskassesment_302T', how='left')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:57:08.477811Z","iopub.execute_input":"2024-03-14T01:57:08.478254Z","iopub.status.idle":"2024-03-14T01:57:08.535210Z","shell.execute_reply.started":"2024-03-14T01:57:08.478220Z","shell.execute_reply":"2024-03-14T01:57:08.533603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cb_scoring = cb_scoring.with_columns((pl.col('target_case') / pl.col('count')).alias('target_case_%')).sort('target_case_%')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:57:24.979934Z","iopub.execute_input":"2024-03-14T01:57:24.980615Z","iopub.status.idle":"2024-03-14T01:57:24.987173Z","shell.execute_reply.started":"2024-03-14T01:57:24.980572Z","shell.execute_reply":"2024-03-14T01:57:24.985900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cb_scoring","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:57:30.773489Z","iopub.execute_input":"2024-03-14T01:57:30.774014Z","iopub.status.idle":"2024-03-14T01:57:30.784569Z","shell.execute_reply.started":"2024-03-14T01:57:30.773977Z","shell.execute_reply":"2024-03-14T01:57:30.783114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scoring_cases = credit_bureau.filter((~pl.col('riskassesment_940T').is_null()))","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:57:53.227087Z","iopub.execute_input":"2024-03-14T01:57:53.227561Z","iopub.status.idle":"2024-03-14T01:57:53.267810Z","shell.execute_reply.started":"2024-03-14T01:57:53.227526Z","shell.execute_reply":"2024-03-14T01:57:53.266100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scoring_cases[['target', 'riskassesment_940T']]","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:58:31.181789Z","iopub.execute_input":"2024-03-14T01:58:31.182217Z","iopub.status.idle":"2024-03-14T01:58:31.192610Z","shell.execute_reply.started":"2024-03-14T01:58:31.182186Z","shell.execute_reply":"2024-03-14T01:58:31.191252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nax.hist(scoring_cases.filter(pl.col('target') == 1)['riskassesment_940T'], bins=20, color='blue', label='Target cases', alpha=0.5, density=True)\nax.hist(scoring_cases.filter(pl.col('target') == 0)['riskassesment_940T'], bins=20, color='red', label='Non target cases', alpha=0.5, density=True)\nax.set_xlabel('score')\nax.set_ylabel('density')\nax.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:58:03.677695Z","iopub.execute_input":"2024-03-14T01:58:03.678125Z","iopub.status.idle":"2024-03-14T01:58:04.020609Z","shell.execute_reply.started":"2024-03-14T01:58:03.678091Z","shell.execute_reply":"2024-03-14T01:58:04.019362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scoring_cases.group_by('riskassesment_302T').agg(pl.col('riskassesment_940T').mean()).sort('riskassesment_940T')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:59:03.317580Z","iopub.execute_input":"2024-03-14T01:59:03.318026Z","iopub.status.idle":"2024-03-14T01:59:03.330826Z","shell.execute_reply.started":"2024-03-14T01:59:03.317993Z","shell.execute_reply":"2024-03-14T01:59:03.329606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nax.hist(scoring_cases.filter(pl.col('riskassesment_302T')==\"67% - 100%\")['riskassesment_940T'], bins=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T01:59:24.151645Z","iopub.execute_input":"2024-03-14T01:59:24.152156Z","iopub.status.idle":"2024-03-14T01:59:24.409304Z","shell.execute_reply.started":"2024-03-14T01:59:24.152122Z","shell.execute_reply":"2024-03-14T01:59:24.408057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}