{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import polars as pl\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score \nimport pickle\nimport matplotlib.pyplot as plt\n\ndataPath = \"/kaggle/input/home-credit-credit-risk-model-stability/\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-13T06:08:39.400280Z","iopub.execute_input":"2024-03-13T06:08:39.400706Z","iopub.status.idle":"2024-03-13T06:08:39.738277Z","shell.execute_reply.started":"2024-03-13T06:08:39.400676Z","shell.execute_reply":"2024-03-13T06:08:39.737370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(dataPath + \"csv_files/train/train_base.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:09:32.963090Z","iopub.execute_input":"2024-03-13T06:09:32.963459Z","iopub.status.idle":"2024-03-13T06:09:33.436718Z","shell.execute_reply.started":"2024-03-13T06:09:32.963430Z","shell.execute_reply":"2024-03-13T06:09:33.435339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_folds = 5","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:10:06.769733Z","iopub.execute_input":"2024-03-13T06:10:06.770078Z","iopub.status.idle":"2024-03-13T06:10:06.774646Z","shell.execute_reply.started":"2024-03-13T06:10:06.770052Z","shell.execute_reply":"2024-03-13T06:10:06.773618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(num_folds):\n    col_name = f'fold_{i}'\n    # -1 = unassigned, 1 = train, 0 = test\n    df[col_name] = -1","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:10:33.577377Z","iopub.execute_input":"2024-03-13T06:10:33.577738Z","iopub.status.idle":"2024-03-13T06:10:33.593601Z","shell.execute_reply.started":"2024-03-13T06:10:33.577713Z","shell.execute_reply":"2024-03-13T06:10:33.592457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_weeks = df['WEEK_NUM'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:11:02.513005Z","iopub.execute_input":"2024-03-13T06:11:02.513340Z","iopub.status.idle":"2024-03-13T06:11:02.530912Z","shell.execute_reply.started":"2024-03-13T06:11:02.513311Z","shell.execute_reply":"2024-03-13T06:11:02.529691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for w_num in num_weeks:\n    week_df = df[df['WEEK_NUM'] == w_num]\n    week_df.reset_index(inplace=True)\n    skf = StratifiedKFold(n_splits=num_folds, shuffle=True, random_state=123)\n    iterator = skf.split(week_df[['case_id']], week_df['target'])\n    for i in range(num_folds):\n        col_name = f'fold_{i}'\n        train_idx, valid_idx = next(iterator)\n        train_idx = week_df.loc[train_idx, 'index']\n        valid_idx = week_df.loc[valid_idx, 'index']\n        df.loc[train_idx, col_name] = 1 \n        df.loc[valid_idx, col_name] = 0 ","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:11:27.184996Z","iopub.execute_input":"2024-03-13T06:11:27.185350Z","iopub.status.idle":"2024-03-13T06:11:28.740504Z","shell.execute_reply.started":"2024-03-13T06:11:27.185321Z","shell.execute_reply":"2024-03-13T06:11:28.739389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('WEEK_NUM')['case_id'].count().reset_index().rename(columns={'case_id':'target_case'})","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:20:33.780066Z","iopub.execute_input":"2024-03-13T06:20:33.780638Z","iopub.status.idle":"2024-03-13T06:20:33.809333Z","shell.execute_reply.started":"2024-03-13T06:20:33.780612Z","shell.execute_reply":"2024-03-13T06:20:33.808479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_by_week(df, fold_num=0, mode='all'):\n    col_name = f'fold_{fold_num}'\n    if mode == 'valid':\n        filtered = df[df[col_name] == 0]\n    elif mode == 'train':\n        filtered = df[df[col_name] == 1]\n    else:\n        filtered = df\n    agg_by_week = filtered.groupby('WEEK_NUM')['case_id'].count().reset_index().rename(columns={'case_id':'total_case'})\n    agg_by_week_target = filtered[filtered['target'] == 1].groupby('WEEK_NUM')['case_id'].count().reset_index().rename(columns={'case_id':'target_case'})\n    agg_by_week = agg_by_week.merge(agg_by_week_target, on='WEEK_NUM', how='left').fillna(0)\n    agg_by_week['target_case_%'] = agg_by_week['target_case'] / agg_by_week['total_case']\n    # Create figure and axes\n    fig, ax1 = plt.subplots()\n\n    # Plot total with bar plot\n    ax1.bar(agg_by_week['WEEK_NUM'], agg_by_week['total_case'])\n    ax1.set_xlabel('Week Number')\n    ax1.set_ylabel('Total')\n\n    # Create a second y-axis for percentage\n    ax2 = ax1.twinx()\n    ax2.plot(agg_by_week['WEEK_NUM'], agg_by_week['target_case_%'], color='r', marker='o')\n    ax2.set_ylabel('Target case %', color='r')\n\n    # Title\n    plt.title(f'Total case and target case % by week ({mode}, {col_name})')\n\n    # Show plot\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:28:31.081335Z","iopub.execute_input":"2024-03-13T06:28:31.081690Z","iopub.status.idle":"2024-03-13T06:28:31.093010Z","shell.execute_reply.started":"2024-03-13T06:28:31.081663Z","shell.execute_reply":"2024-03-13T06:28:31.091765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_by_week(df, mode='valid')","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:28:52.918498Z","iopub.execute_input":"2024-03-13T06:28:52.918858Z","iopub.status.idle":"2024-03-13T06:28:53.882945Z","shell.execute_reply.started":"2024-03-13T06:28:52.918829Z","shell.execute_reply":"2024-03-13T06:28:53.882298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_by_week(df, mode='valid', fold_num=1)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:29:10.501896Z","iopub.execute_input":"2024-03-13T06:29:10.502277Z","iopub.status.idle":"2024-03-13T06:29:10.903748Z","shell.execute_reply.started":"2024-03-13T06:29:10.502248Z","shell.execute_reply":"2024-03-13T06:29:10.902902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_by_week(df, mode='train')","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:26:03.681086Z","iopub.execute_input":"2024-03-13T06:26:03.681505Z","iopub.status.idle":"2024-03-13T06:26:04.183842Z","shell.execute_reply.started":"2024-03-13T06:26:03.681476Z","shell.execute_reply":"2024-03-13T06:26:04.182393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_by_week(df, fold_num=4)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:27:07.470251Z","iopub.execute_input":"2024-03-13T06:27:07.470635Z","iopub.status.idle":"2024-03-13T06:27:07.882295Z","shell.execute_reply.started":"2024-03-13T06:27:07.470607Z","shell.execute_reply":"2024-03-13T06:27:07.880554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('./train_base_week_folds.csv', index='False')","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:31:11.023528Z","iopub.execute_input":"2024-03-13T06:31:11.023897Z","iopub.status.idle":"2024-03-13T06:31:13.370289Z","shell.execute_reply.started":"2024-03-13T06:31:11.023866Z","shell.execute_reply":"2024-03-13T06:31:13.368265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(dataPath + \"csv_files/train/train_base.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:31:36.348493Z","iopub.execute_input":"2024-03-13T06:31:36.348842Z","iopub.status.idle":"2024-03-13T06:31:36.779083Z","shell.execute_reply.started":"2024-03-13T06:31:36.348816Z","shell.execute_reply":"2024-03-13T06:31:36.777959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(num_folds):\n    col_name = f'fold_{i}'\n    # -1 = unassigned, 1 = train, 0 = test\n    df[col_name] = -1","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:32:03.719065Z","iopub.execute_input":"2024-03-13T06:32:03.719474Z","iopub.status.idle":"2024-03-13T06:32:03.731041Z","shell.execute_reply.started":"2024-03-13T06:32:03.719442Z","shell.execute_reply":"2024-03-13T06:32:03.729969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=num_folds, shuffle=True, random_state=123)\niterator = skf.split(df[['case_id']], df['target'])\nfor i in range(num_folds):\n    col_name = f'fold_{i}'\n    train_idx, valid_idx = next(iterator)\n    df.loc[train_idx, col_name] = 1\n    df.loc[valid_idx, col_name] = 0\n    ","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:35:53.496158Z","iopub.execute_input":"2024-03-13T06:35:53.496521Z","iopub.status.idle":"2024-03-13T06:35:53.929917Z","shell.execute_reply.started":"2024-03-13T06:35:53.496489Z","shell.execute_reply":"2024-03-13T06:35:53.928533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_by_week(df)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:36:31.195540Z","iopub.execute_input":"2024-03-13T06:36:31.195906Z","iopub.status.idle":"2024-03-13T06:36:31.588048Z","shell.execute_reply.started":"2024-03-13T06:36:31.195878Z","shell.execute_reply":"2024-03-13T06:36:31.586572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_by_week(df, mode='train')","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:36:41.937695Z","iopub.execute_input":"2024-03-13T06:36:41.938016Z","iopub.status.idle":"2024-03-13T06:36:42.406279Z","shell.execute_reply.started":"2024-03-13T06:36:41.937995Z","shell.execute_reply":"2024-03-13T06:36:42.404955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_by_week(df, mode='valid')","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:36:52.332835Z","iopub.execute_input":"2024-03-13T06:36:52.333217Z","iopub.status.idle":"2024-03-13T06:36:52.716936Z","shell.execute_reply.started":"2024-03-13T06:36:52.333185Z","shell.execute_reply":"2024-03-13T06:36:52.715683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('./train_base_folds.csv', index='False')","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:38:15.348075Z","iopub.execute_input":"2024-03-13T06:38:15.348423Z","iopub.status.idle":"2024-03-13T06:38:19.070174Z","shell.execute_reply.started":"2024-03-13T06:38:15.348381Z","shell.execute_reply":"2024-03-13T06:38:19.069469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}