{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Environment settings","metadata":{}},{"cell_type":"code","source":"import sys\nprint(sys.version)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:18.636922Z","iopub.execute_input":"2024-05-02T13:09:18.637606Z","iopub.status.idle":"2024-05-02T13:09:18.675813Z","shell.execute_reply.started":"2024-05-02T13:09:18.637569Z","shell.execute_reply":"2024-05-02T13:09:18.674178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings('ignore')\n\nprint(pd.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:18.678251Z","iopub.execute_input":"2024-05-02T13:09:18.678727Z","iopub.status.idle":"2024-05-02T13:09:20.364702Z","shell.execute_reply.started":"2024-05-02T13:09:18.678693Z","shell.execute_reply":"2024-05-02T13:09:20.363327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA - dive into data","metadata":{}},{"cell_type":"markdown","source":"## Files structure","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\n\n!tree '/kaggle/input/home-credit-credit-risk-model-stability' -d \n\nROOT = '/kaggle/input/home-credit-credit-risk-model-stability'\nfolders = ['csv_files/train', 'parquet_files/train', 'csv_files/test', 'parquet_files/test']\nextensions = ['.csv',  '.parquet'] * 2\n\nfor dir_, ext in zip(folders, extensions):\n    folder_path = Path(ROOT) / dir_\n    num_files = len(list(folder_path.glob(f'*{ext}')))\n    print(f\"Number of files in: {'./' + dir_:>21}: {num_files}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:20.366632Z","iopub.execute_input":"2024-05-02T13:09:20.367176Z","iopub.status.idle":"2024-05-02T13:09:21.539895Z","shell.execute_reply.started":"2024-05-02T13:09:20.367142Z","shell.execute_reply":"2024-05-02T13:09:21.538318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature data schema","metadata":{}},{"cell_type":"markdown","source":"[Data schema reference](https://www.kaggle.com/code/sergiosaharovskiy/home-credit-crms-2024-eda-and-submission?scriptVersionId=161988552&cellId=8)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T13:58:25.950019Z","iopub.execute_input":"2024-04-17T13:58:25.951316Z","iopub.status.idle":"2024-04-17T13:58:25.961325Z","shell.execute_reply.started":"2024-04-17T13:58:25.951257Z","shell.execute_reply":"2024-04-17T13:58:25.958930Z"}}},{"cell_type":"code","source":"base_info_train = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_base.parquet')\nbase_info_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:21.543396Z","iopub.execute_input":"2024-05-02T13:09:21.543808Z","iopub.status.idle":"2024-05-02T13:09:21.992333Z","shell.execute_reply.started":"2024-05-02T13:09:21.543770Z","shell.execute_reply":"2024-05-02T13:09:21.991118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(base_info_train.shape[0])\nprint(base_info_train['date_decision'].unique().shape[0])\nprint(base_info_train['MONTH'].unique().shape[0])\nprint(base_info_train['case_id'].duplicated().sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:21.994100Z","iopub.execute_input":"2024-05-02T13:09:21.994582Z","iopub.status.idle":"2024-05-02T13:09:22.197424Z","shell.execute_reply.started":"2024-05-02T13:09:21.994540Z","shell.execute_reply":"2024-05-02T13:09:22.196022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_info_test = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/test/test_base.parquet')\nprint(base_info_test.shape[0])\nprint(np.intersect1d(base_info_train['case_id'].values, base_info_test['case_id'].values).shape[0])","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:22.198955Z","iopub.execute_input":"2024-05-02T13:09:22.199307Z","iopub.status.idle":"2024-05-02T13:09:22.291686Z","shell.execute_reply.started":"2024-05-02T13:09:22.199279Z","shell.execute_reply":"2024-05-02T13:09:22.290388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# depth==0\nsample_df_0 = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_static_0_0.parquet')\nsample_df_0.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:22.293048Z","iopub.execute_input":"2024-05-02T13:09:22.293575Z","iopub.status.idle":"2024-05-02T13:09:26.558663Z","shell.execute_reply.started":"2024-05-02T13:09:22.293542Z","shell.execute_reply":"2024-05-02T13:09:26.557149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del sample_df_0","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:26.560731Z","iopub.execute_input":"2024-05-02T13:09:26.561294Z","iopub.status.idle":"2024-05-02T13:09:26.567656Z","shell.execute_reply.started":"2024-05-02T13:09:26.561248Z","shell.execute_reply":"2024-05-02T13:09:26.566166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# depth==1\nsample_df_1 = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_applprev_1_0.parquet')\nsample_df_1.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:26.569261Z","iopub.execute_input":"2024-05-02T13:09:26.569918Z","iopub.status.idle":"2024-05-02T13:09:31.558252Z","shell.execute_reply.started":"2024-05-02T13:09:26.569873Z","shell.execute_reply":"2024-05-02T13:09:31.557205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df_1['num_group1'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:31.562994Z","iopub.execute_input":"2024-05-02T13:09:31.563616Z","iopub.status.idle":"2024-05-02T13:09:31.615530Z","shell.execute_reply.started":"2024-05-02T13:09:31.563583Z","shell.execute_reply":"2024-05-02T13:09:31.614086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sample_df_1.duplicated(subset=['case_id']).sum())\nprint(sample_df_1.duplicated(subset=['case_id', 'num_group1']).sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:31.617263Z","iopub.execute_input":"2024-05-02T13:09:31.617725Z","iopub.status.idle":"2024-05-02T13:09:32.592181Z","shell.execute_reply.started":"2024-05-02T13:09:31.617689Z","shell.execute_reply":"2024-05-02T13:09:32.590711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del sample_df_1","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:32.593757Z","iopub.execute_input":"2024-05-02T13:09:32.594261Z","iopub.status.idle":"2024-05-02T13:09:32.601382Z","shell.execute_reply.started":"2024-05-02T13:09:32.594221Z","shell.execute_reply":"2024-05-02T13:09:32.599779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# depth==2\nsample_df_2 = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_credit_bureau_a_2_0.parquet')\nsample_df_2.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:32.603450Z","iopub.execute_input":"2024-05-02T13:09:32.603995Z","iopub.status.idle":"2024-05-02T13:09:34.219192Z","shell.execute_reply.started":"2024-05-02T13:09:32.603947Z","shell.execute_reply":"2024-05-02T13:09:34.217949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sample_df_2.duplicated(subset=['case_id', 'num_group1']).sum())\nprint(sample_df_2.duplicated(subset=['case_id', 'num_group2']).sum())\nprint(sample_df_2.duplicated(subset=['case_id', 'num_group1', 'num_group2']).sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:34.220544Z","iopub.execute_input":"2024-05-02T13:09:34.220911Z","iopub.status.idle":"2024-05-02T13:09:35.898078Z","shell.execute_reply.started":"2024-05-02T13:09:34.220882Z","shell.execute_reply":"2024-05-02T13:09:35.896058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del sample_df_2","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:35.899495Z","iopub.execute_input":"2024-05-02T13:09:35.899834Z","iopub.status.idle":"2024-05-02T13:09:35.906124Z","shell.execute_reply.started":"2024-05-02T13:09:35.899806Z","shell.execute_reply":"2024-05-02T13:09:35.904369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Label distribution","metadata":{}},{"cell_type":"code","source":"def get_date_interval_info(df):\n    df['date_decision'] = pd.to_datetime(df['date_decision'])\n    date_delta = df['date_decision'].drop_duplicates().sort_values().diff()\n    len_uniq_dates = len(df.date_decision.unique())\n    print(\n        f'\\n[INFO] Actual date range:  {date_delta.sum().days + 1} day(s).',\n        f'\\n[INFO] Total unique dates: {len_uniq_dates} day(s).'\n    )\n\n    print(f'\\n[INFO] Min date: {df.date_decision.dt.date.min()}',\n          f'\\n[INFO] Max date: {df.date_decision.dt.date.max()}')\n    \nprint(\"\\n----------Train data----------\\n\")\nget_date_interval_info(base_info_train)\nprint(\"\\n----------Test data----------\\n\")\nget_date_interval_info(base_info_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:35.908090Z","iopub.execute_input":"2024-05-02T13:09:35.908435Z","iopub.status.idle":"2024-05-02T13:09:38.042266Z","shell.execute_reply.started":"2024-05-02T13:09:35.908406Z","shell.execute_reply":"2024-05-02T13:09:38.040944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(base_info_train['target'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:38.043826Z","iopub.execute_input":"2024-05-02T13:09:38.044168Z","iopub.status.idle":"2024-05-02T13:09:38.067148Z","shell.execute_reply.started":"2024-05-02T13:09:38.044140Z","shell.execute_reply":"2024-05-02T13:09:38.065692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_date_components_histogram_seaborn(df, component):\n    fig = plt.figure(figsize=(15, 5))\n    sns.histplot(data=df, x=component, bins=30, multiple=\"stack\", kde=False, hue='target')\n    plt.title(f'{component.capitalize()} Distribution')\n    plt.xlabel(component.capitalize())\n    plt.ylabel('Frequency')\n    plt.show()\n\n# base_info_train['year'] = base_info_train.date_decision.dt.year-2019\nbase_info_train['month'] = base_info_train.date_decision.dt.month\nbase_info_train['day_of_year'] = base_info_train.date_decision.dt.day_of_year\nbase_info_train['day_of_week'] = base_info_train.date_decision.dt.day_of_week\n\n\nplot_date_components_histogram_seaborn(base_info_train, 'date_decision')\nplot_date_components_histogram_seaborn(base_info_train, 'month')\nplot_date_components_histogram_seaborn(base_info_train, 'day_of_year')\nplot_date_components_histogram_seaborn(base_info_train, 'day_of_week')","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:38.069278Z","iopub.execute_input":"2024-05-02T13:09:38.069765Z","iopub.status.idle":"2024-05-02T13:09:44.560795Z","shell.execute_reply.started":"2024-05-02T13:09:38.069722Z","shell.execute_reply":"2024-05-02T13:09:44.559203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Missing values","metadata":{}},{"cell_type":"code","source":"import subprocess\n\ndef get_disk_usage(directory):\n    cmd = f'du {directory}/* -h | sort -rh'\n    result = subprocess.run(cmd, shell=True, stdout=subprocess.PIPE, text=True)\n    output_lines = result.stdout.split('\\n')\n\n    # Extract file/directory names and sizes\n    data = [line.split('\\t') for line in output_lines if line]\n    df = pd.DataFrame(data, columns=['size', 'path'])\n    df['file_name'] = df.path.str.replace('train_|test_', '', regex=True).\\\n    apply(lambda x: Path(x).stem)\n    return df\n\ntrain_disk_usage = get_disk_usage(f'{ROOT}/parquet_files/train').reset_index()\ntest_disk_usage = get_disk_usage(f'{ROOT}/parquet_files/test')\n\ntrain_disk_usage.reset_index().merge(test_disk_usage, on=['file_name'],\n                                     how='outer', suffixes=['_train', '_test'])\\\n                                     .sort_values(by='index').drop(columns=['index'])","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:44.562448Z","iopub.execute_input":"2024-05-02T13:09:44.562814Z","iopub.status.idle":"2024-05-02T13:09:44.677807Z","shell.execute_reply.started":"2024-05-02T13:09:44.562783Z","shell.execute_reply":"2024-05-02T13:09:44.676644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nimport polars as pl\n\nshapes, nan_total_count = [], []\nfor fp in tqdm(train_disk_usage.path):\n    df = pl.read_parquet(fp) \n    shapes.append(df.shape)\n    nan_total_count.append(df.null_count().to_pandas().sum().sum())\n    del df\n\ntrain_disk_usage[['height', 'width']] = shapes\ntrain_disk_usage['null_count'] = nan_total_count\ntrain_disk_usage['isna_%'] = train_disk_usage.null_count / np.prod(shapes, 1) ","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:09:44.679295Z","iopub.execute_input":"2024-05-02T13:09:44.679943Z","iopub.status.idle":"2024-05-02T13:11:47.968399Z","shell.execute_reply.started":"2024-05-02T13:09:44.679904Z","shell.execute_reply":"2024-05-02T13:11:47.966882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = train_disk_usage['isna_%'].values.tolist()\ntotal_area = train_disk_usage['height'] * train_disk_usage['width']\ntotal_area_scaled = total_area / total_area.max()\n\nrows = 4\ncols = 8\nfig, axs = plt.subplots(rows, cols, figsize=(18, 11))\n\nfor i, ax in enumerate(axs.flat):\n    \n    outer_square_side = np.sqrt(total_area_scaled[i])\n    inner_square_side = np.sqrt(total_area_scaled[i]*values[i])\n\n    # Add the small square inside the 1x1 image\n    ax.add_patch(plt.Rectangle((0.5 - outer_square_side / 2, 0.5 - outer_square_side / 2),\n                               outer_square_side, outer_square_side,\n                               color='#F03F47', label='Total Records'))\n    \n    ax.add_patch(plt.Rectangle((0.4 - inner_square_side / 2, 0.6 - inner_square_side / 2),\n                               inner_square_side, inner_square_side,\n                               color='#645F64', label='Null Values'))\n\n    ax.set_xticks([])\n    ax.set_yticks([])\n    ax.set_aspect('equal')\n    ax.set_title(f'{train_disk_usage.file_name.iloc[i]}\\n'\n                 f'{train_disk_usage[\"size\"].iloc[i]:}\\nNull_%: {values[i]*100:.2f}')\n\nplt.legend(bbox_to_anchor=(-4, -.4), loc='lower center', ncol=2)\nplt.suptitle('\\nNull values% in Train files scaled and shaped as Squares')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:11:47.970959Z","iopub.execute_input":"2024-05-02T13:11:47.971491Z","iopub.status.idle":"2024-05-02T13:11:50.377469Z","shell.execute_reply.started":"2024-05-02T13:11:47.971448Z","shell.execute_reply":"2024-05-02T13:11:50.376080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dealing with massive features","metadata":{}},{"cell_type":"code","source":"feature_definitions = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/feature_definitions.csv')\nfeature_definitions['transformation_group'] = feature_definitions['Variable'].apply(lambda x: x[-1 ])","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:11:50.379037Z","iopub.execute_input":"2024-05-02T13:11:50.379536Z","iopub.status.idle":"2024-05-02T13:11:50.409656Z","shell.execute_reply.started":"2024-05-02T13:11:50.379491Z","shell.execute_reply":"2024-05-02T13:11:50.407510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_definitions.groupby('transformation_group').head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:11:50.412189Z","iopub.execute_input":"2024-05-02T13:11:50.412907Z","iopub.status.idle":"2024-05-02T13:11:50.457324Z","shell.execute_reply.started":"2024-05-02T13:11:50.412870Z","shell.execute_reply":"2024-05-02T13:11:50.455315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int32))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                df = df.with_columns(pl.col(col).cast(pl.Float32))\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:11:50.461172Z","iopub.execute_input":"2024-05-02T13:11:50.461748Z","iopub.status.idle":"2024-05-02T13:11:50.484665Z","shell.execute_reply.started":"2024-05-02T13:11:50.461701Z","shell.execute_reply":"2024-05-02T13:11:50.483100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_def_map = dict(zip(feature_definitions['Variable'], feature_definitions['Description']))","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:11:50.486560Z","iopub.execute_input":"2024-05-02T13:11:50.487021Z","iopub.status.idle":"2024-05-02T13:11:50.501417Z","shell.execute_reply.started":"2024-05-02T13:11:50.486991Z","shell.execute_reply":"2024-05-02T13:11:50.499484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_feature_table(s):\n    import re\n    \n    pattern = r'_\\d+'\n    matches = re.findall(pattern, s)\n    \n    if len(matches) == 1:\n        return s\n    else:\n        return s.rsplit('_', 1)[0]\n\ntrain_disk_usage['feature_table'] = train_disk_usage['file_name'].apply(extract_feature_table)\ntest_disk_usage['feature_table'] = test_disk_usage['file_name'].apply(extract_feature_table)\ntrain_disk_usage.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:11:50.502743Z","iopub.execute_input":"2024-05-02T13:11:50.503192Z","iopub.status.idle":"2024-05-02T13:11:50.530203Z","shell.execute_reply.started":"2024-05-02T13:11:50.503159Z","shell.execute_reply":"2024-05-02T13:11:50.529149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"table_feature_reference = []\nfor i, row in train_disk_usage.drop_duplicates(subset=['feature_table']).iterrows():\n    if row['feature_table'] == 'base':\n        continue\n    df = pl.read_parquet(row['path'])\n    this_table_features = df.columns\n    table_feature_reference.extend([{'feature_name': feat, 'definition': feature_def_map.get(feat), \\\n                                    'table': row['feature_table']} for feat in this_table_features])\n    del df","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:11:50.531462Z","iopub.execute_input":"2024-05-02T13:11:50.532646Z","iopub.status.idle":"2024-05-02T13:12:16.400043Z","shell.execute_reply.started":"2024-05-02T13:11:50.532598Z","shell.execute_reply":"2024-05-02T13:12:16.399005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(table_feature_reference).to_csv('/kaggle/working/table_feature_def.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:12:16.409038Z","iopub.execute_input":"2024-05-02T13:12:16.410389Z","iopub.status.idle":"2024-05-02T13:12:16.422985Z","shell.execute_reply.started":"2024-05-02T13:12:16.410347Z","shell.execute_reply":"2024-05-02T13:12:16.421789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# start from here","metadata":{}},{"cell_type":"markdown","source":"# load feature tables and preprocess","metadata":{}},{"cell_type":"code","source":"def set_table_dtypes(df: pl.DataFrame) -> pl.DataFrame:\n    # implement here all desired dtypes for tables\n    # the following is just an example\n    for col in df.columns:\n        # last letter of column name will help you determine the type\n        if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int32))\n        elif col in [\"date_decision\"]:\n            df = df.with_columns(pl.col(col).cast(pl.Date))\n        elif col[-1] in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n        elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n        elif col[-1] in (\"D\",):\n            df = df.with_columns(pl.col(col).cast(pl.Date))\n\n    return df\n\ndef convert_strings(df: pd.DataFrame) -> pd.DataFrame:\n    for col in df.columns:  \n        if df[col].dtype.name in ['object', 'string']:\n            df[col] = df[col].astype(\"string\").astype('category')\n            current_categories = df[col].cat.categories\n            new_categories = current_categories.to_list() + [\"Unknown\"]\n            new_dtype = pd.CategoricalDtype(categories=new_categories, ordered=True)\n            df[col] = df[col].astype(new_dtype)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:12:16.424531Z","iopub.execute_input":"2024-05-02T13:12:16.425616Z","iopub.status.idle":"2024-05-02T13:12:16.439304Z","shell.execute_reply.started":"2024-05-02T13:12:16.425578Z","shell.execute_reply":"2024-05-02T13:12:16.438100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_feature_table_files(table_name, convert_pd=True):\n    \n    train_dfs, test_dfs = [], []\n    for i, row in train_disk_usage.loc[train_disk_usage['feature_table'] == table_name].iterrows():\n        train_dfs.append(pl.read_parquet(row['path']).pipe(set_table_dtypes))\n    for i, row in test_disk_usage.loc[test_disk_usage['feature_table'] == table_name].iterrows():\n        test_dfs.append(pl.read_parquet(row['path']).pipe(set_table_dtypes))\n    \n    train_feature_df = pl.concat(train_dfs, how='vertical_relaxed')\n    test_feature_df = pl.concat(test_dfs, how='vertical_relaxed')\n    \n    if convert_pd:\n        return train_feature_df.to_pandas(), test_feature_df.to_pandas()\n    else:\n        return train_feature_df, test_feature_df","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:12:16.441104Z","iopub.execute_input":"2024-05-02T13:12:16.441936Z","iopub.status.idle":"2024-05-02T13:12:16.460787Z","shell.execute_reply.started":"2024-05-02T13:12:16.441891Z","shell.execute_reply":"2024-05-02T13:12:16.459503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def filter_nan_cols(df, group_keys=['case_id'], tolerance=0.5):\n    \n    check_nan_df = pd.concat([df[group_keys], df.drop(columns=group_keys).isna()], axis=1)\n    grp_nan_stats = (check_nan_df.groupby(group_keys).all().sum(axis=0) / check_nan_df.shape[0])\n    cols_to_drop = grp_nan_stats.loc[grp_nan_stats > tolerance].index.tolist()\n    print(f\"With tolerance {tolerance}, drop {len(cols_to_drop)} out of {df.shape[1]} columns\")\n    \n    return df.drop(columns=cols_to_drop)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:12:16.462956Z","iopub.execute_input":"2024-05-02T13:12:16.463461Z","iopub.status.idle":"2024-05-02T13:12:16.475006Z","shell.execute_reply.started":"2024-05-02T13:12:16.463416Z","shell.execute_reply":"2024-05-02T13:12:16.473691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base, test_base = load_feature_table_files('base')\ntrain_static_0, test_static_0 = load_feature_table_files('static_0')\ntrain_static_cb_0, test_static_cb_0 = load_feature_table_files('static_cb_0')\ntrain_person_1, test_person_1 = load_feature_table_files('person_1')\ntrain_credit_bureau_b_2, test_credit_bureau_b_2 = load_feature_table_files('credit_bureau_b_2')","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:12:16.476528Z","iopub.execute_input":"2024-05-02T13:12:16.477791Z","iopub.status.idle":"2024-05-02T13:12:35.978991Z","shell.execute_reply.started":"2024-05-02T13:12:16.477744Z","shell.execute_reply":"2024-05-02T13:12:35.977598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_person_1.shape)\ntrain_person_1 = filter_nan_cols(train_person_1, tolerance=0.5)\nprint(train_person_1.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:12:35.980243Z","iopub.execute_input":"2024-05-02T13:12:35.980608Z","iopub.status.idle":"2024-05-02T13:12:44.411062Z","shell.execute_reply.started":"2024-05-02T13:12:35.980579Z","shell.execute_reply":"2024-05-02T13:12:44.409875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_static_0 = filter_nan_cols(train_static_0, tolerance=0.5)\ntrain_static_cb_0 = filter_nan_cols(train_static_cb_0, tolerance=0.5)\ntrain_credit_bureau_b_2 = filter_nan_cols(train_credit_bureau_b_2, tolerance=0.5)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:12:44.412391Z","iopub.execute_input":"2024-05-02T13:12:44.412734Z","iopub.status.idle":"2024-05-02T13:12:54.303531Z","shell.execute_reply.started":"2024-05-02T13:12:44.412705Z","shell.execute_reply":"2024-05-02T13:12:54.302172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## aggregate and merge features","metadata":{}},{"cell_type":"code","source":"import copy\n\ntrain_joint_df = copy.deepcopy(train_base)\nprint(train_joint_df.shape)\ntrain_joint_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:12:54.309517Z","iopub.execute_input":"2024-05-02T13:12:54.310075Z","iopub.status.idle":"2024-05-02T13:12:54.391415Z","shell.execute_reply.started":"2024-05-02T13:12:54.310029Z","shell.execute_reply":"2024-05-02T13:12:54.390241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_definitions.loc[feature_definitions['Variable'].isin(train_person_1.columns)]","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:12:54.392709Z","iopub.execute_input":"2024-05-02T13:12:54.394321Z","iopub.status.idle":"2024-05-02T13:12:54.413667Z","shell.execute_reply.started":"2024-05-02T13:12:54.394272Z","shell.execute_reply":"2024-05-02T13:12:54.412361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# digit features first\ntrain_joint_df = pd.merge(train_joint_df, \\\n                            train_person_1.groupby('case_id').agg(\n                                mainoccupationinc_384A_max=pd.NamedAgg(column=\"mainoccupationinc_384A\", aggfunc=\"max\"),\n                            ).reset_index(),\n                         on='case_id', how='left')","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:12:54.415515Z","iopub.execute_input":"2024-05-02T13:12:54.415951Z","iopub.status.idle":"2024-05-02T13:12:54.717977Z","shell.execute_reply.started":"2024-05-02T13:12:54.415917Z","shell.execute_reply":"2024-05-02T13:12:54.716518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# categorical features\n# here num_group1 == 0 indicates that it is the person who applied for the loan\nselected_cols = ['case_id', 'incometype_1044T', 'housetype_905L', 'familystate_447L', 'empl_employedtotal_800L']\ntrain_joint_df = pd.merge(train_joint_df, train_person_1.loc[train_person_1['num_group1'] == 0, selected_cols], \\\n                          on='case_id', how='left')","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:12:54.719359Z","iopub.execute_input":"2024-05-02T13:12:54.719722Z","iopub.status.idle":"2024-05-02T13:12:55.275908Z","shell.execute_reply.started":"2024-05-02T13:12:54.719692Z","shell.execute_reply":"2024-05-02T13:12:55.274711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_digit_columns(df):\n    digit_cols = [col for col, dtype in df.dtypes.items() if dtype in ['float64', 'int32']]\n    return digit_cols, df[digit_cols]","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:12:55.277588Z","iopub.execute_input":"2024-05-02T13:12:55.278312Z","iopub.status.idle":"2024-05-02T13:12:55.284431Z","shell.execute_reply.started":"2024-05-02T13:12:55.278278Z","shell.execute_reply":"2024-05-02T13:12:55.283111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_joint_df = pd.merge(train_joint_df, extract_digit_columns(train_static_0)[1], on='case_id', how='left')\ntrain_joint_df = pd.merge(train_joint_df, extract_digit_columns(train_static_cb_0)[1], on='case_id', how='left')","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:12:55.286576Z","iopub.execute_input":"2024-05-02T13:12:55.287015Z","iopub.status.idle":"2024-05-02T13:13:00.472487Z","shell.execute_reply.started":"2024-05-02T13:12:55.286978Z","shell.execute_reply":"2024-05-02T13:13:00.471592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# digit features from bureau data\ntrain_joint_df = pd.merge(train_joint_df, \\\n                        train_credit_bureau_b_2.groupby('case_id').agg(\n                            pmts_dpdvalue_108P_max=pd.NamedAgg(column=\"pmts_dpdvalue_108P\", aggfunc=\"max\"),\n                            pmts_pmtsoverdue_635A_max=pd.NamedAgg(column=\"pmts_pmtsoverdue_635A\", aggfunc=\"max\"),\n                        ).reset_index(),\n                    on='case_id', how='left')","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:13:00.473927Z","iopub.execute_input":"2024-05-02T13:13:00.474522Z","iopub.status.idle":"2024-05-02T13:13:02.358375Z","shell.execute_reply.started":"2024-05-02T13:13:00.474489Z","shell.execute_reply":"2024-05-02T13:13:02.357172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_joint_df.fillna(np.nan, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:13:02.360221Z","iopub.execute_input":"2024-05-02T13:13:02.360613Z","iopub.status.idle":"2024-05-02T13:13:07.870891Z","shell.execute_reply.started":"2024-05-02T13:13:02.360555Z","shell.execute_reply":"2024-05-02T13:13:07.869512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_joint_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:13:07.872791Z","iopub.execute_input":"2024-05-02T13:13:07.873753Z","iopub.status.idle":"2024-05-02T13:13:07.902974Z","shell.execute_reply.started":"2024-05-02T13:13:07.873707Z","shell.execute_reply":"2024-05-02T13:13:07.901785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## normalization","metadata":{}},{"cell_type":"code","source":"feature_names = np.setdiff1d(train_joint_df.columns, train_base.columns)\nprint(len(feature_names))","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:13:07.904499Z","iopub.execute_input":"2024-05-02T13:13:07.904901Z","iopub.status.idle":"2024-05-02T13:13:07.911405Z","shell.execute_reply.started":"2024-05-02T13:13:07.904840Z","shell.execute_reply":"2024-05-02T13:13:07.910251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"digit_features, _ = extract_digit_columns(train_joint_df)\nnon_digit_features = np.setdiff1d(feature_names, digit_features)\nprint(train_joint_df[digit_features].dtypes)\nprint(train_joint_df[non_digit_features].dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:38:29.650698Z","iopub.execute_input":"2024-05-02T13:38:29.654817Z","iopub.status.idle":"2024-05-02T13:38:33.558740Z","shell.execute_reply.started":"2024-05-02T13:38:29.654660Z","shell.execute_reply":"2024-05-02T13:38:33.557182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import RobustScaler\n\nsc = RobustScaler()\ntrain_joint_df[digit_features] = sc.fit_transform(train_joint_df[digit_features])","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:38:39.376978Z","iopub.execute_input":"2024-05-02T13:38:39.378316Z","iopub.status.idle":"2024-05-02T13:38:49.539536Z","shell.execute_reply.started":"2024-05-02T13:38:39.378258Z","shell.execute_reply":"2024-05-02T13:38:49.538420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nle = LabelEncoder()\nfor feat in non_digit_features:\n    train_joint_df[feat] = le.fit_transform(train_joint_df[feat])","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:39:13.195081Z","iopub.execute_input":"2024-05-02T13:39:13.197618Z","iopub.status.idle":"2024-05-02T13:39:14.998616Z","shell.execute_reply.started":"2024-05-02T13:39:13.197529Z","shell.execute_reply":"2024-05-02T13:39:14.996469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_joint_df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:39:15.001114Z","iopub.execute_input":"2024-05-02T13:39:15.001480Z","iopub.status.idle":"2024-05-02T13:39:29.258273Z","shell.execute_reply.started":"2024-05-02T13:39:15.001452Z","shell.execute_reply":"2024-05-02T13:39:29.256967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# feature selection","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_eval, y_train, y_eval = train_test_split(train_joint_df[feature_names], train_joint_df['target'], \\\n                                                   test_size=0.4, random_state=27)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:39:50.504577Z","iopub.execute_input":"2024-05-02T13:39:50.505292Z","iopub.status.idle":"2024-05-02T13:39:56.670405Z","shell.execute_reply.started":"2024-05-02T13:39:50.505258Z","shell.execute_reply":"2024-05-02T13:39:56.669068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.fillna(0.0, inplace=True)\nX_eval.fillna(0.0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:39:56.672762Z","iopub.execute_input":"2024-05-02T13:39:56.673538Z","iopub.status.idle":"2024-05-02T13:39:57.384354Z","shell.execute_reply.started":"2024-05-02T13:39:56.673483Z","shell.execute_reply":"2024-05-02T13:39:57.383036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lasso\n\n[LassoCV](https://scikit-learn.org/stable/modules/generated/sklearn.linear_model.LassoCV.html#sklearn.linear_model.LassoCV)","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LassoCV","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:40:01.499492Z","iopub.execute_input":"2024-05-02T13:40:01.500201Z","iopub.status.idle":"2024-05-02T13:40:01.640629Z","shell.execute_reply.started":"2024-05-02T13:40:01.500160Z","shell.execute_reply":"2024-05-02T13:40:01.638969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lasso_cv = LassoCV(cv=5, n_alphas=5, random_state=0)\nlasso_cv.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:40:02.761577Z","iopub.execute_input":"2024-05-02T13:40:02.762104Z","iopub.status.idle":"2024-05-02T13:40:24.977370Z","shell.execute_reply.started":"2024-05-02T13:40:02.762063Z","shell.execute_reply":"2024-05-02T13:40:24.975488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lasso_selected_features = [feature_names[ix] for ix, coef in enumerate(lasso_cv.coef_) if coef != 0]","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:40:24.987164Z","iopub.execute_input":"2024-05-02T13:40:24.988225Z","iopub.status.idle":"2024-05-02T13:40:24.998216Z","shell.execute_reply.started":"2024-05-02T13:40:24.988155Z","shell.execute_reply":"2024-05-02T13:40:24.996312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(lasso_selected_features))\nprint(lasso_selected_features)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:40:25.000955Z","iopub.execute_input":"2024-05-02T13:40:25.002387Z","iopub.status.idle":"2024-05-02T13:40:25.015133Z","shell.execute_reply.started":"2024-05-02T13:40:25.002057Z","shell.execute_reply":"2024-05-02T13:40:25.013559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ElasticNetCV\n\n[ElasticNetCV](https://scikit-learn.org/stable/modules/generated/sklearn.linear_model.ElasticNetCV.html#sklearn.linear_model.ElasticNetCV)","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import ElasticNetCV\n\nen_cv = ElasticNetCV(cv=5, n_alphas=5, random_state=0)\nen_cv.fit(X_train, y_train)\nen_selected_features = [feature_names[ix] for ix, coef in enumerate(en_cv.coef_) if coef != 0]\nprint(len(en_selected_features))\nprint(en_selected_features)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:40:42.999376Z","iopub.execute_input":"2024-05-02T13:40:42.999786Z","iopub.status.idle":"2024-05-02T13:41:03.959108Z","shell.execute_reply.started":"2024-05-02T13:40:42.999755Z","shell.execute_reply":"2024-05-02T13:41:03.957317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LarsCV\n\n[LarsCV](https://scikit-learn.org/stable/modules/generated/sklearn.linear_model.LarsCV.html#sklearn.linear_model.LarsCV)","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LarsCV\n\nlars_cv = LarsCV(cv=5, max_n_alphas=5, max_iter=100)\nlars_cv.fit(X_train, y_train)\nlars_selected_features = [feature_names[ix] for ix, coef in enumerate(lars_cv.coef_) if coef != 0]\nprint(len(lars_selected_features))\nprint(lars_selected_features)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:41:13.369096Z","iopub.execute_input":"2024-05-02T13:41:13.369519Z","iopub.status.idle":"2024-05-02T13:41:27.736938Z","shell.execute_reply.started":"2024-05-02T13:41:13.369490Z","shell.execute_reply":"2024-05-02T13:41:27.735725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Recursive feature elimination\n\n[RFE](https://scikit-learn.org/stable/modules/generated/sklearn.feature_selection.RFECV.html#sklearn.feature_selection.RFECV)","metadata":{}},{"cell_type":"code","source":"# # warning: high computation cost!\n# from sklearn.feature_selection import RFECV\n# from sklearn.tree import DecisionTreeClassifier\n\n# estimator = DecisionTreeClassifier()\n# rfe_selector = RFECV(estimator, step=2, cv=5)\n# rfe_selector.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:47:04.914343Z","iopub.execute_input":"2024-05-02T13:47:04.914786Z","iopub.status.idle":"2024-05-02T13:47:04.921165Z","shell.execute_reply.started":"2024-05-02T13:47:04.914752Z","shell.execute_reply":"2024-05-02T13:47:04.919478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(rfe_selector.support_)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:47:06.723348Z","iopub.execute_input":"2024-05-02T13:47:06.723879Z","iopub.status.idle":"2024-05-02T13:47:06.730462Z","shell.execute_reply.started":"2024-05-02T13:47:06.723820Z","shell.execute_reply":"2024-05-02T13:47:06.728612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rfe_selected_features = feature_names[rfe_selector.support_]\n# print(len(rfe_selected_features))\n# print(rfe_selected_features)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:47:08.631587Z","iopub.execute_input":"2024-05-02T13:47:08.632124Z","iopub.status.idle":"2024-05-02T13:47:08.638720Z","shell.execute_reply.started":"2024-05-02T13:47:08.632086Z","shell.execute_reply":"2024-05-02T13:47:08.636937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## lightgbm\n\n[lightgbm documentation](https://lightgbm.readthedocs.io/en/stable/Python-API.html)","metadata":{}},{"cell_type":"code","source":"from lightgbm import LGBMClassifier\n\nlgb_clf = LGBMClassifier()\nlgb_clf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:47:11.835670Z","iopub.execute_input":"2024-05-02T13:47:11.836638Z","iopub.status.idle":"2024-05-02T13:47:50.586059Z","shell.execute_reply.started":"2024-05-02T13:47:11.836596Z","shell.execute_reply":"2024-05-02T13:47:50.584753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_clf.feature_importances_","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:47:50.588175Z","iopub.execute_input":"2024-05-02T13:47:50.588864Z","iopub.status.idle":"2024-05-02T13:47:50.602187Z","shell.execute_reply.started":"2024-05-02T13:47:50.588814Z","shell.execute_reply":"2024-05-02T13:47:50.600782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_clf.score(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:48:31.574689Z","iopub.execute_input":"2024-05-02T13:48:31.575279Z","iopub.status.idle":"2024-05-02T13:48:41.005314Z","shell.execute_reply.started":"2024-05-02T13:48:31.575235Z","shell.execute_reply":"2024-05-02T13:48:41.003964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}