{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"},{"sourceId":7624293,"sourceType":"datasetVersion","datasetId":4441488}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Setup\n## Dask Setup","metadata":{}},{"cell_type":"code","source":"!pip install dask_ml -q --no-index --find-links=file:/kaggle/input/dask-install/dask_ml","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:17:20.820864Z","iopub.execute_input":"2024-02-15T10:17:20.821764Z","iopub.status.idle":"2024-02-15T10:17:35.728319Z","shell.execute_reply.started":"2024-02-15T10:17:20.821700Z","shell.execute_reply":"2024-02-15T10:17:35.727138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from dask.distributed import Client\n\nclient = Client(n_workers=4)\nclient","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-15T10:17:35.730359Z","iopub.execute_input":"2024-02-15T10:17:35.730749Z","iopub.status.idle":"2024-02-15T10:17:39.827122Z","shell.execute_reply.started":"2024-02-15T10:17:35.730700Z","shell.execute_reply":"2024-02-15T10:17:39.825984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## General Library Imports","metadata":{}},{"cell_type":"code","source":"import os\n\nfrom collections.abc import Iterator\n\nimport dask\nimport dask.dataframe as dd\nimport dask.array as da\nfrom dask.dataframe import to_datetime\nfrom dask_ml.preprocessing import OneHotEncoder\nfrom dask_ml.model_selection import train_test_split\nfrom dask_ml.metrics import accuracy_score\n\nimport xgboost\n\nimport pandas as pd\nimport numpy as np\n\nfile_path = \"/kaggle/input/home-credit-credit-risk-model-stability/\"\n\nhandled_features = []\n\ndesired_partition_size = \"256MB\"\n\n# Number of folds in each trial. This also determines the train/test split\n# (e.g. N_FOLDS=5 -> train=4/5 of the total data, test=1/5)\nN_FOLDS = 5","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:17:39.828840Z","iopub.execute_input":"2024-02-15T10:17:39.829550Z","iopub.status.idle":"2024-02-15T10:17:43.575794Z","shell.execute_reply.started":"2024-02-15T10:17:39.829507Z","shell.execute_reply":"2024-02-15T10:17:43.574729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Processing Config","metadata":{}},{"cell_type":"code","source":"date_variable_names = [\n    \"datelastunpaid_3546854D\", \n    \"lastrepayingdate_696D\", \n    \"lastrejectdate_50D\", \n    \"datefirstoffer_1144D\", \n    \"firstclxcampaign_1125D\",\n    \"maxdpdinstldate_3546855D\", \n    \"firstdatedue_489D\", \n    \"validfrom_1069D\", \n    \"lastdelinqdate_224D\", \n    \"dtlastpmtallstes_4499206D\", \n    \"lastapprdate_640D\", \n    \"payvacationpostpone_4187118D\", \n    \"lastactivateddate_801D\", \n    'lastapplicationdate_877D', \n    \"datelastinstal40dpd_247D\"\n]\n\ncategorical_features = [\n    'disbursementtype_67L', \n    'cardtype_51L', \n    'typesuite_864L', \n    'inittransactioncode_186L', \n    'lastrejectreason_759M', \n    'lastrejectreasonclient_4145040M', \n    'bankacctype_710L', \n    'lastst_736L', \n    'paytype_783L', \n    'credtype_322L', \n    'twobodfilling_608L', \n    'previouscontdistrict_112M', \n    'paytype1st_925L', \n    'lastapprcommoditycat_1041M', \n    'lastrejectcommodtypec_5251769M', \n    'lastcancelreason_561M', \n    'lastrejectcommoditycat_161M', \n    'lastapprcommoditytypec_5251766M', \n    'equalitydataagreement_891L',\n    'equalityempfrom_62L', \n    'isbidproductrequest_292L', \n    'isdebitcard_729L', \n    'opencred_647L',\n    'isbidproduct_1095L'\n]","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-15T10:17:43.578406Z","iopub.execute_input":"2024-02-15T10:17:43.578830Z","iopub.status.idle":"2024-02-15T10:17:43.589413Z","shell.execute_reply.started":"2024-02-15T10:17:43.578788Z","shell.execute_reply":"2024-02-15T10:17:43.587718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"float64_null_value_method = \"none\" # fill_median , fill_mean , none\n\nstring_null_value_method = \"none\" # fill_placeholder , fill_mode , none\n\ndate_null_value_method = \"ignore\"","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:17:43.591027Z","iopub.execute_input":"2024-02-15T10:17:43.591430Z","iopub.status.idle":"2024-02-15T10:17:43.614255Z","shell.execute_reply.started":"2024-02-15T10:17:43.591399Z","shell.execute_reply":"2024-02-15T10:17:43.613325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Importing Data","metadata":{}},{"cell_type":"code","source":"train_base = dd.read_csv(file_path + \"/csv_files/train/train_base.csv\")\ntest_base = dd.read_csv(file_path + \"/csv_files/test/test_base.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:17:43.617136Z","iopub.execute_input":"2024-02-15T10:17:43.617508Z","iopub.status.idle":"2024-02-15T10:17:43.727959Z","shell.execute_reply.started":"2024-02-15T10:17:43.617475Z","shell.execute_reply":"2024-02-15T10:17:43.725458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_static = dd.concat(\n    [\n        dd.read_parquet(file_path + \"/parquet_files/train/train_static_0_*.parquet\")\n    ]\n)\n\ntest_static = dd.concat(\n    [\n        dd.read_parquet(file_path + \"/parquet_files/test/test_static_0_*.parquet\")\n    ]\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:17:43.732247Z","iopub.execute_input":"2024-02-15T10:17:43.732610Z","iopub.status.idle":"2024-02-15T10:17:44.082414Z","shell.execute_reply.started":"2024-02-15T10:17:43.732575Z","shell.execute_reply":"2024-02-15T10:17:44.081086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Merging Data","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_unprocessed = train_base.set_index('case_id').join(train_static.set_index('case_id'))\ntest_unprocessed = test_base.set_index('case_id').join(test_static.set_index('case_id')) \n\n# Keep only a fraction of the data, just for testing \ntrain_unprocessed = train_unprocessed.sample(frac=0.1, random_state=42)\n\ntrain_unprocessed = train_unprocessed.repartition(partition_size=desired_partition_size)\ntest_unprocessed = test_unprocessed.repartition(partition_size=desired_partition_size)\n\ntrain_unprocessed = train_unprocessed.persist()\ntest_unprocessed = test_unprocessed.persist()","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:17:44.083706Z","iopub.execute_input":"2024-02-15T10:17:44.084017Z","iopub.status.idle":"2024-02-15T10:18:26.260713Z","shell.execute_reply.started":"2024-02-15T10:17:44.083992Z","shell.execute_reply":"2024-02-15T10:18:26.258732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_unprocessed","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:18:26.266918Z","iopub.execute_input":"2024-02-15T10:18:26.267968Z","iopub.status.idle":"2024-02-15T10:18:26.681238Z","shell.execute_reply.started":"2024-02-15T10:18:26.267924Z","shell.execute_reply":"2024-02-15T10:18:26.679785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Processing","metadata":{}},{"cell_type":"code","source":"def handle_null_values_for_float64(df: dd.DataFrame, method: str) -> dd.DataFrame:\n    \"\"\"\n    Finds all features with null values of type float64 in a Dask DataFrame and applies the specified method to deal with these null values.\n\n    Parameters:\n    df (dd.DataFrame): The input Dask DataFrame.\n    method (str): The method to use for handling null values. Options are 'fill_mean', 'drop', 'interpolate', 'fill_median'.\n\n    Returns:\n    dd.DataFrame: The DataFrame after handling null values.\n    \"\"\"\n    # Ensure the method is one of the allowed options\n    if method not in ['fill_mean', 'drop', 'interpolate', 'fill_median', 'none']:\n        raise ValueError(\"Invalid method specified. Choose from 'fill_mean', 'drop', 'interpolate', 'fill_median', 'none'.\")\n\n    # Iterate over each column in the DataFrame\n    for col_name, dtype in df.dtypes.items():\n        # Check if the column is of float64 type\n        if dtype == 'float64':\n            handled_features.append(col_name)\n            \n            if method == 'none':\n                continue  # Skip further processing if method is 'none'\n            \n            # Apply specified method to handle null values\n            if method == 'fill_mean':\n                # Fill null values with the mean of the column\n                mean_value = df[col_name].mean().compute()\n                df[col_name] = df[col_name].fillna(mean_value)\n            elif method == 'fill_median':\n                # Fill null values with the median of the column\n                median_value = df[col_name].quantile(0.5).compute()\n                df[col_name] = df[col_name].fillna(median_value)\n            elif method == 'drop':\n                # Drop rows with null values in the column\n                df = df.dropna(subset=[col_name])\n            elif method == 'interpolate':\n                # Interpolate missing values\n                df[col_name] = df[col_name].interpolate()\n\n    return df","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-15T10:18:26.692918Z","iopub.execute_input":"2024-02-15T10:18:26.698702Z","iopub.status.idle":"2024-02-15T10:18:26.735695Z","shell.execute_reply.started":"2024-02-15T10:18:26.698627Z","shell.execute_reply":"2024-02-15T10:18:26.730678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def handle_null_values_for_strings(df: dd.DataFrame, method: str, placeholder=\"NaN\") -> dd.DataFrame:\n    \"\"\"\n    Handles null values in string columns of a Dask DataFrame according to the specified method.\n\n    Parameters:\n    df (dd.DataFrame): The input Dask DataFrame.\n    method (str): Method for handling null values. Options: 'fill_placeholder', 'fill_mode', 'none'.\n    placeholder (str, optional): Placeholder value to use when method is 'fill_placeholder'. Defaults to \"NaN\".\n\n    Returns:\n    dd.DataFrame: DataFrame after handling null values in string columns.\n    \"\"\"\n    if method not in ['fill_placeholder', 'fill_mode', 'none']:\n        raise ValueError(\"Invalid method specified. Choose from 'fill_placeholder', 'fill_mode', 'none'.\")\n\n    if method == 'none': \n        return df \n    \n    # Dask DataFrame doesn't directly support `.mode()`. Consider alternatives for 'fill_mode'.\n    for col_name, dtype in df.dtypes.items():\n        if dtype == 'object':  # Assuming string columns are of dtype 'object'\n            if method == 'fill_placeholder':\n                # Fill null values with the placeholder\n                df[col_name] = df[col_name].fillna(placeholder)\n            elif method == 'fill_mode':\n                # Compute mode value, if computation is heavy consider a different strategy\n                mode_value = df[col_name].value_counts().idxmax().compute()\n                df[col_name] = df[col_name].fillna(mode_value)\n            else:\n                raise ValueError(\"Invalid method specified. Choose from 'fill_placeholder', 'fill_mode', 'none'.\")\n\n    return df","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-15T10:18:26.737835Z","iopub.execute_input":"2024-02-15T10:18:26.738494Z","iopub.status.idle":"2024-02-15T10:18:26.764195Z","shell.execute_reply.started":"2024-02-15T10:18:26.738451Z","shell.execute_reply":"2024-02-15T10:18:26.760866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def transform_dates_to_days_since(df, current_date_col, date_features, missing_value_handling):\n    \"\"\"\n    Adapted version of the function to preprocess date columns in a Dask DataFrame by calculating the \n    number of days since each date feature occurred, relative to a current date column.\n\n    Parameters:\n    - df (dd.DataFrame): The Dask DataFrame to process.\n    - current_date_col (str): The column name that holds the current date.\n    - date_features (list): A list of column names that contain dates to process.\n    - missing_value_handling (str): Strategy for handling missing values in date columns.\n\n    Returns:\n    - dd.DataFrame: The processed Dask DataFrame with new features indicating the days since each date occurred.\n    \"\"\"\n    # Handling missing values based on the specified strategy\n    if missing_value_handling == \"fill_with_specific_date\":\n        fill_value = pd.Timestamp(\"1900-01-01\")  # Example fill value, adjust as needed\n        df[current_date_col] = df[current_date_col].fillna(fill_value)\n        for feature in date_features:\n            df[feature] = df[feature].fillna(fill_value)\n    elif missing_value_handling == \"ignore\":\n        pass  # Do nothing, simply proceed with the existing null values\n    else:\n        raise ValueError(\"Invalid missing value handling method specified.\")\n    \n    # Convert the current date column to datetime\n    df[current_date_col] = dd.to_datetime(df[current_date_col], errors='coerce')\n\n    # Process each date feature individually\n    for feature in date_features:\n        df[feature] = dd.to_datetime(df[feature], errors='coerce')\n        days_since_feature = (df[current_date_col] - df[feature]).dt.days\n        df[f\"days_since_{feature}\"] = days_since_feature\n        # Optionally drop the original date feature column\n        df = df.drop(columns=[feature])\n        \n    handled_features.append(date_features)\n\n    return df","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-15T10:18:26.766381Z","iopub.execute_input":"2024-02-15T10:18:26.767205Z","iopub.status.idle":"2024-02-15T10:18:26.800287Z","shell.execute_reply.started":"2024-02-15T10:18:26.767159Z","shell.execute_reply":"2024-02-15T10:18:26.798868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fit_onehot_encoder(dask_df, column_names):\n    \"\"\"\n    Fits a OneHotEncoder from Dask ML to specified columns of a Dask DataFrame.\n\n    Parameters:\n    - dask_df (dd.DataFrame): The Dask DataFrame.\n    - column_names (list): List of column names to encode.\n\n    Returns:\n    - encoder (OneHotEncoder): The fitted Dask ML OneHotEncoder.\n    \"\"\"\n    dask_df = dask_df.categorize(columns=column_names)\n\n    encoder = OneHotEncoder()\n    encoder.fit(dask_df[column_names])\n    \n    return encoder","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-15T10:18:26.801515Z","iopub.execute_input":"2024-02-15T10:18:26.801937Z","iopub.status.idle":"2024-02-15T10:18:26.818264Z","shell.execute_reply.started":"2024-02-15T10:18:26.801900Z","shell.execute_reply":"2024-02-15T10:18:26.817151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apply_onehot_encoder(encoder, dask_df, column_names):\n    \"\"\"\n    Applies a fitted Dask ML OneHotEncoder to specified columns of a Dask DataFrame,\n    and returns a new Dask DataFrame with encoded columns.\n\n    Parameters:\n    - encoder (OneHotEncoder): The fitted Dask ML OneHotEncoder.\n    - dask_df (dd.DataFrame): The Dask DataFrame to encode.\n    - column_names (list): List of column names to encode.\n\n    Returns:\n    - encoded_dask_df (dd.DataFrame): Dask DataFrame with encoded columns.\n    \"\"\"\n    # Create a copy of the DataFrame to avoid modifying the original\n    dask_df_copy = dask_df.copy()\n    \n    # Apply the encoder to the specified columns\n    # dask_df.categorize(columns=column_names)\n    encoded_data = encoder.transform(dask_df_copy[column_names])\n    \n    # Drop the original columns from the DataFrame\n    dask_df_encoded = dask_df_copy.drop(columns=column_names)\n    \n    # Concatenate the encoded DataFrame with the rest of the columns\n    encoded_dask_df = dd.concat([dask_df_encoded, encoded_data], axis=1)\n    \n    return encoded_dask_df","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-15T10:18:26.819515Z","iopub.execute_input":"2024-02-15T10:18:26.819952Z","iopub.status.idle":"2024-02-15T10:18:26.834486Z","shell.execute_reply.started":"2024-02-15T10:18:26.819914Z","shell.execute_reply":"2024-02-15T10:18:26.833248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_unprocessed = handle_null_values_for_float64(train_unprocessed, float64_null_value_method)\ntest_unprocessed = handle_null_values_for_float64(test_unprocessed, float64_null_value_method)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:18:26.835770Z","iopub.execute_input":"2024-02-15T10:18:26.836227Z","iopub.status.idle":"2024-02-15T10:18:26.851232Z","shell.execute_reply.started":"2024-02-15T10:18:26.836189Z","shell.execute_reply":"2024-02-15T10:18:26.848466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_unprocessed = handle_null_values_for_strings(train_unprocessed, string_null_value_method)\ntest_unprocessed = handle_null_values_for_strings(test_unprocessed, string_null_value_method)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:18:26.853210Z","iopub.execute_input":"2024-02-15T10:18:26.856391Z","iopub.status.idle":"2024-02-15T10:18:26.869033Z","shell.execute_reply.started":"2024-02-15T10:18:26.856321Z","shell.execute_reply":"2024-02-15T10:18:26.867217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_unprocessed = transform_dates_to_days_since(train_unprocessed, \n                                                  \"date_decision\", \n                                                  date_variable_names, \n                                                  date_null_value_method)\n\ntest_unprocessed = transform_dates_to_days_since(test_unprocessed, \n                                                  \"date_decision\", \n                                                  date_variable_names, \n                                                  date_null_value_method)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:18:26.871016Z","iopub.execute_input":"2024-02-15T10:18:26.871476Z","iopub.status.idle":"2024-02-15T10:18:35.138060Z","shell.execute_reply.started":"2024-02-15T10:18:26.871429Z","shell.execute_reply":"2024-02-15T10:18:35.136125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nencoder = fit_onehot_encoder(train_unprocessed, categorical_features)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:18:35.140014Z","iopub.execute_input":"2024-02-15T10:18:35.140458Z","iopub.status.idle":"2024-02-15T10:19:06.902702Z","shell.execute_reply.started":"2024-02-15T10:18:35.140418Z","shell.execute_reply":"2024-02-15T10:19:06.901405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nprocessed_train = apply_onehot_encoder(encoder, train_unprocessed, categorical_features)\nprocessed_test = apply_onehot_encoder(encoder, test_unprocessed, categorical_features)\n\nprocessed_train = processed_train.repartition(partition_size=desired_partition_size)\nprocessed_test = processed_test.repartition(partition_size=desired_partition_size)\n\nprocessed_train = processed_train.persist()\nprocessed_test = processed_test.persist()","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:19:06.904530Z","iopub.execute_input":"2024-02-15T10:19:06.904899Z","iopub.status.idle":"2024-02-15T10:19:22.326533Z","shell.execute_reply.started":"2024-02-15T10:19:06.904869Z","shell.execute_reply":"2024-02-15T10:19:22.322814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Analysis","metadata":{}},{"cell_type":"code","source":"df_descriptions = pd.read_csv(file_path + \"feature_definitions.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:19:22.328317Z","iopub.execute_input":"2024-02-15T10:19:22.329609Z","iopub.status.idle":"2024-02-15T10:19:22.607944Z","shell.execute_reply.started":"2024-02-15T10:19:22.329547Z","shell.execute_reply":"2024-02-15T10:19:22.588299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def describe_unhandled_features(data_frame: dd.DataFrame, descriptions_frame: pd.DataFrame, handled_features: list) -> pd.DataFrame:\n    \"\"\"\n    Creates a summary of features in a Dask DataFrame that are not listed in handled_features, including\n    their type, minimum, maximum, median values, null count, and description from a Pandas DataFrame.\n\n    Parameters:\n    - data_frame (dd.DataFrame): The Dask DataFrame containing the data.\n    - descriptions_frame (pd.DataFrame): The Pandas DataFrame containing descriptions and variables.\n    - handled_features (list): A list of feature names that have already been handled.\n\n    Returns:\n    - pd.DataFrame: A Pandas DataFrame summarizing unhandled features.\n    \"\"\"\n    # Step 1: Extract feature names\n    desc_feature_names = descriptions_frame['Variable'].tolist()\n    data_feature_names = data_frame.columns\n\n    # Step 2: Identify unhandled features by finding the intersection and then excluding handled features\n    unhandled_features = set(desc_feature_names).intersection(set(data_feature_names))\n    unhandled_features = unhandled_features.difference(set(handled_features))\n\n    # Step 3: Compile information for unhandled features\n    features_info = []\n\n    for feature in unhandled_features:\n        feature_info = {\n            'Variable': feature,\n            'Type': str(data_frame[feature].dtype),\n            'Description': descriptions_frame[descriptions_frame['Variable'] == feature]['Description'].iloc[0],\n        }\n\n        features_info.append(feature_info)\n\n    return pd.DataFrame(features_info)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-15T10:19:22.611047Z","iopub.execute_input":"2024-02-15T10:19:22.611560Z","iopub.status.idle":"2024-02-15T10:19:22.639280Z","shell.execute_reply.started":"2024-02-15T10:19:22.611521Z","shell.execute_reply":"2024-02-15T10:19:22.637493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# By switching `handled_features` for an empty list, you can easily print out all features. \nhandled_features = [str(feature) for feature in handled_features]  # Ensure all elements are strings\ndf_features_info = describe_unhandled_features(processed_train, df_descriptions, handled_features)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:19:22.646521Z","iopub.execute_input":"2024-02-15T10:19:22.646943Z","iopub.status.idle":"2024-02-15T10:19:22.661767Z","shell.execute_reply.started":"2024-02-15T10:19:22.646908Z","shell.execute_reply":"2024-02-15T10:19:22.657152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_rows, _ = df_features_info.shape\n\npd.set_option('display.max_rows', num_rows)\npd.set_option('display.max_colwidth', None)\n\ndf_features_info.head(num_rows)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:19:22.664224Z","iopub.execute_input":"2024-02-15T10:19:22.665049Z","iopub.status.idle":"2024-02-15T10:19:22.686557Z","shell.execute_reply.started":"2024-02-15T10:19:22.665005Z","shell.execute_reply":"2024-02-15T10:19:22.684768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Setup","metadata":{}},{"cell_type":"code","source":"# Here we subset data for cross-validation\ndef make_cv_splits(\n    n_folds: int = N_FOLDS,\n) -> Iterator[tuple[dd.DataFrame, dd.DataFrame]]:\n    frac = [1 / n_folds] * n_folds\n    splits = processed_train.random_split(frac, shuffle=True)\n    for i in range(n_folds):\n        train = [splits[j] for j in range(n_folds) if j != i]\n        val = splits[i]\n        yield dd.concat(train), val","metadata":{"execution":{"iopub.status.busy":"2024-02-15T10:19:22.690531Z","iopub.execute_input":"2024-02-15T10:19:22.691170Z","iopub.status.idle":"2024-02-15T10:19:22.702243Z","shell.execute_reply.started":"2024-02-15T10:19:22.691077Z","shell.execute_reply":"2024-02-15T10:19:22.700894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nfrom sklearn.metrics import roc_auc_score\n\nscores = []\n\nfor i, (train, val) in enumerate(make_cv_splits()):\n    print(f\"Training/Val split #{i}\")\n    y_train = train['target']\n    X_train = train.drop(['target',\"date_decision\", \"MONTH\", \"WEEK_NUM\"], axis=1)\n    y_val = val[\"target\"]\n    X_val = val.drop(['target',\"date_decision\", \"MONTH\", \"WEEK_NUM\"], axis=1)\n\n    print(\"Building DMatrix...\")\n    d_train = xgboost.dask.DaskDMatrix(\n        None, X_train, y_train, enable_categorical=True\n    )\n\n    print(\"Training model...\")\n    model = xgboost.dask.train(\n        None,\n        {\"tree_method\": \"hist\", \"eval_metric\": \"auc\", \"objective\": \"binary:logistic\"},\n        d_train,\n        num_boost_round=4,\n        evals=[(d_train, \"train\")],\n    )\n\n    print(\"Running model on test data...\")\n    predictions = xgboost.dask.predict(None, model, X_val)\n    \n    print(\"Measuring accuracy of model vs. ground truth...\")\n    y_val, predictions = dask.compute(y_val, predictions)\n    score = roc_auc_score(y_val, predictions)\n    \n    print(f\"AUC ROC Score: {score}\")\n\n    # Compute predictions and mean squared error for this iteration\n    # while we start the next one\n    scores.append(score.reshape(1))\n    print(\"-\" * 80)\n\nscores = da.concatenate(scores).compute()\nprint(f\"AUC ROC={scores.mean()} +/- {scores.std()}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-15T11:09:09.990768Z","iopub.execute_input":"2024-02-15T11:09:09.991150Z","iopub.status.idle":"2024-02-15T11:12:49.737789Z","shell.execute_reply.started":"2024-02-15T11:09:09.991121Z","shell.execute_reply":"2024-02-15T11:12:49.736131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nX_test = processed_test.drop([\"date_decision\", \"MONTH\", \"WEEK_NUM\"], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T11:13:36.955753Z","iopub.execute_input":"2024-02-15T11:13:36.956167Z","iopub.status.idle":"2024-02-15T11:13:37.243784Z","shell.execute_reply.started":"2024-02-15T11:13:36.956134Z","shell.execute_reply":"2024-02-15T11:13:37.239632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nprint(\"Running model on test data...\")\nfinal_predictions = xgboost.dask.predict(None, model, X_test)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T11:13:44.317320Z","iopub.execute_input":"2024-02-15T11:13:44.317764Z","iopub.status.idle":"2024-02-15T11:13:44.386402Z","shell.execute_reply.started":"2024-02-15T11:13:44.317727Z","shell.execute_reply":"2024-02-15T11:13:44.383411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm = pd.read_csv(file_path + \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = final_predictions","metadata":{"execution":{"iopub.status.busy":"2024-02-15T11:13:58.834245Z","iopub.execute_input":"2024-02-15T11:13:58.834761Z","iopub.status.idle":"2024-02-15T11:13:59.244869Z","shell.execute_reply.started":"2024-02-15T11:13:58.834710Z","shell.execute_reply":"2024-02-15T11:13:59.243019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Check null: \", df_subm[\"score\"].isnull().any())","metadata":{"execution":{"iopub.status.busy":"2024-02-15T11:14:01.558320Z","iopub.execute_input":"2024-02-15T11:14:01.558747Z","iopub.status.idle":"2024-02-15T11:14:01.565917Z","shell.execute_reply.started":"2024-02-15T11:14:01.558713Z","shell.execute_reply":"2024-02-15T11:14:01.564687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-15T11:14:02.529233Z","iopub.execute_input":"2024-02-15T11:14:02.530462Z","iopub.status.idle":"2024-02-15T11:14:02.542860Z","shell.execute_reply.started":"2024-02-15T11:14:02.530394Z","shell.execute_reply":"2024-02-15T11:14:02.541497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-15T11:14:05.262850Z","iopub.execute_input":"2024-02-15T11:14:05.263274Z","iopub.status.idle":"2024-02-15T11:14:05.278638Z","shell.execute_reply.started":"2024-02-15T11:14:05.263243Z","shell.execute_reply":"2024-02-15T11:14:05.276769Z"},"trusted":true},"execution_count":null,"outputs":[]}]}