{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":10170605,"sourceType":"datasetVersion","datasetId":6281294}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Outline\n\nThis note is mainly to verify the effectiveness of the memory and missing value processing in the previous note: [https://www.kaggle.com/code/nicolesy/data-pre-processing-memory-and-missing-value](http://)\n\n* Experiment 1----lgb baseline+ raw data (Delete partition_id=0)\n* Experiment 2----lgb baseline+ compressed data (Delete earlier data)\n* Experiment 3----lgb baseline+ compressed data (Delete earlier data) + data after imputation of systematically missing values\n* Experiment 4----lgb baseline+ compressed data (Delete earlier data) + data after imputation of systematic missing values + data after imputation of random missing values\n\n\n\nLGBM baseline Reference notes for LGBM:  [http://www.kaggle.com/code/nicolesy/lgbm-baseline-simple-submission](http://)\n\nIn order to reduce the memory usage during the training process, we **divide the existing parquet files into five new parquet files** in sequence according to date_id and save them, so that we only need to read the data of the corresponding fold when using the 50% fold cross validation.","metadata":{}},{"cell_type":"code","source":"import os\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport lightgbm as lgb\nimport joblib\n\nfrom tqdm import tqdm\nimport pyarrow.parquet as pq\nimport shutil\nimport time\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-11T16:41:52.470359Z","iopub.execute_input":"2024-12-11T16:41:52.471476Z","iopub.status.idle":"2024-12-11T16:41:52.481283Z","shell.execute_reply.started":"2024-12-11T16:41:52.471433Z","shell.execute_reply":"2024-12-11T16:41:52.479774Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reslice the data set","metadata":{}},{"cell_type":"code","source":"def re_slice_parquet (folder_path):\n\n    # Traverse through all parquet files in the folder and subfolders\n    folder_path = Path(folder_path)\n    parquet_files = list(folder_path.rglob(\"*.parquet\"))\n\n    if not parquet_files:\n        print(\"No parquet files found in the specified folder.\")\n    else:\n        date_counts = pd.Series(dtype=int)\n\n        for parquet_file in parquet_files:\n            try:\n                df = pd.read_parquet(parquet_file)\n\n                # Check if 'date_id' column exists\n                if 'date_id' not in df.columns:\n                    print(f\"Skipping {parquet_file}: 'date_id' column not found.\")\n                    continue\n\n                date_counts = date_counts.add(df['date_id'].value_counts(), fill_value=0)\n\n            except Exception as e:\n                print(f\"Error reading {parquet_file}: {e}\")\n\n        if not date_counts.empty:\n            # Sort date_counts by date_id and determine date ranges for 5 parts\n            date_counts = date_counts.sort_index()\n            total_days = len(date_counts)\n            split_indices = np.array_split(np.arange(total_days), 5)\n\n            date_ranges = [(date_counts.index[indices[0]], date_counts.index[indices[-1]]) for indices in split_indices]\n\n            print(\"Determined date ranges for splitting:\", date_ranges)\n\n    return date_ranges","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T16:41:52.486590Z","iopub.execute_input":"2024-12-11T16:41:52.487000Z","iopub.status.idle":"2024-12-11T16:41:52.531862Z","shell.execute_reply.started":"2024-12-11T16:41:52.486965Z","shell.execute_reply":"2024-12-11T16:41:52.530538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_and_split_parquet_files(input_folder, output_folder, date_ranges):\n    \"\"\"\n    Process all parquet files in the input folder and split them into multiple parts based on date ranges.\n    If only start_date exists in one file, keep all data after start_date. Then continue to look for end_date\n    in subsequent files, merging the data until the range is complete. Include data from start_date and end_date.\n\n    Parameters:\n        input_folder (str): Path to the folder containing input parquet files.\n        output_folder (str): Path to the folder where the output parquet files will be saved.\n        date_ranges (list): List of (start_date, end_date) tuples for splitting.\n\n    Outputs:\n        Saves split parquet files in the output folder without deleting the original files.\n    \"\"\"\n    input_folder = Path(input_folder)\n    output_folder = Path(output_folder)\n    parquet_files = sorted(list(input_folder.rglob(\"*.parquet\")), key=lambda x: int(x.stem.split('=')[1]))\n\n    if not parquet_files:\n        print(\"No parquet files found in the specified input folder.\")\n        return\n\n    output_folder.mkdir(parents=True, exist_ok=True)  # Ensure the output folder exists\n\n    for i, (start_date, end_date) in tqdm(enumerate(date_ranges), desc=\"Processing date ranges\"):\n        split_df = pd.DataFrame()\n        collect_data = False  # Flag to indicate whether to start collecting data\n\n        for parquet_file in parquet_files:\n            try:\n                df = pd.read_parquet(parquet_file)\n\n                # Check if 'date_id' column exists\n                if 'date_id' not in df.columns:\n                    continue\n\n                # If start_date has not been found, continue looking for it\n                if not collect_data:\n                    if start_date in df['date_id'].values:\n                        collect_data = True\n                        split_df = pd.concat([split_df, df[df['date_id'] >= start_date]], ignore_index=True)\n\n                # If start_date is already found, start collecting data\n                else:\n                    # If end_date is found in the current file, save up to end_date and stop collecting\n                    if end_date in df['date_id'].values:\n                        split_chunk = df[(df['date_id'] >= start_date) & (df['date_id'] <= end_date)]\n                        split_df = pd.concat([split_df, split_chunk], ignore_index=True)\n                        break\n                    else:\n                        # If end_date is not in the current file, save all data and continue searching\n                        split_df = pd.concat([split_df, df], ignore_index=True)\n\n            except Exception as e:\n                print(f\"Error processing file {parquet_file}: {e}\")\n\n        # Save the split DataFrame\n        output_file = output_folder / f\"combined_part{i}.parquet\"\n        split_df.to_parquet(output_file, index=False)\n        print(f\"Saved split file: {output_file}\")\n\n    print(\"All files have been processed and split into parts.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T16:44:53.835115Z","iopub.execute_input":"2024-12-11T16:44:53.835667Z","iopub.status.idle":"2024-12-11T16:44:53.849341Z","shell.execute_reply.started":"2024-12-11T16:44:53.835532Z","shell.execute_reply":"2024-12-11T16:44:53.847906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lower_path = \"/kaggle/input/js-dpp-mmv/lower\"\ndeal_sys_miss_path = \"/kaggle/input/js-dpp-mmv/deal_sys_miss\"\ndeal_random_miss_path = \"/kaggle/input/js-dpp-mmv/deal_random_miss\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nlower_ranges = re_slice_parquet (lower_path)\nprocess_and_split_parquet_files(lower_path, \"/kaggle/working/lower\", lower_ranges)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T16:44:53.851316Z","iopub.execute_input":"2024-12-11T16:44:53.851920Z","iopub.status.idle":"2024-12-11T16:51:40.034719Z","shell.execute_reply.started":"2024-12-11T16:44:53.851870Z","shell.execute_reply":"2024-12-11T16:51:40.033251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ndeal_sys_miss_ranges = re_slice_parquet (deal_sys_miss_path)\nprocess_and_split_parquet_files(deal_sys_miss_path, \"/kaggle/working/deal_sys_miss\", deal_sys_miss_ranges)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ndeal_random_miss_ranges = re_slice_parquet (deal_random_miss_path)\nprocess_and_split_parquet_files(deal_random_miss_path, \"/kaggle/working/deal_random_miss\", deal_random_miss_ranges)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LGBM Baseline","metadata":{}},{"cell_type":"code","source":"TARGET = 'responder_6'\nFEAT_COLS = [f\"feature_{i:02d}\" for i in range(79)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T16:51:40.036349Z","iopub.execute_input":"2024-12-11T16:51:40.036817Z","iopub.status.idle":"2024-12-11T16:51:40.043664Z","shell.execute_reply.started":"2024-12-11T16:51:40.036781Z","shell.execute_reply":"2024-12-11T16:51:40.042414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_r2(y_true, y_pred, weights):\n    \"\"\"Calculate the R2 score, weighted by the provided weights.\"\"\"\n    numerator = np.sum(weights * (y_true - y_pred) ** 2)\n    denominator = np.sum(weights * (y_true ** 2))\n    r2_score = 1 - (numerator / denominator)\n    return r2_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T16:51:40.045445Z","iopub.execute_input":"2024-12-11T16:51:40.045967Z","iopub.status.idle":"2024-12-11T16:51:40.060865Z","shell.execute_reply.started":"2024-12-11T16:51:40.045930Z","shell.execute_reply":"2024-12-11T16:51:40.059492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_single_fold(fold_idx, data_dir, n_splits, save_path, FEAT_COLS, TARGET, save_model=True):\n    \"\"\"\n    Train a single fold for cross-validation.\n\n    Parameters:\n    - fold_idx: the index of the fold (0 to n_splits-1)\n    - data_dir: the directory containing the .parquet files\n    - n_splits: total number of splits\n    - save_path: path to save the model\n    - FEAT_COLS: list of feature columns\n    - TARGET: target column\n    - save_model: whether to save the model after training\n    \"\"\"\n    # For each fold, load the current fold file\n    valid_file = os.path.join(data_dir, f\"combined_part{fold_idx}.parquet\")\n    print(f\"Fold {fold_idx}: validation file {valid_file}\")\n\n    # Load the current fold's data\n    fold_data = pd.read_parquet(valid_file)\n\n    # Split the data: the `i`-th part is validation, the rest is training\n    valid_data = fold_data.iloc[int(len(fold_data) * (fold_idx / n_splits)): int(len(fold_data) * ((fold_idx + 1) / n_splits))]\n    train_data = pd.concat([fold_data.iloc[:int(len(fold_data) * (fold_idx / n_splits))], \n                            fold_data.iloc[int(len(fold_data) * ((fold_idx + 1) / n_splits)):]],\n                           ignore_index=True)\n    \n\n    # Print date_id range for each of the 5 parts (4 training and 1 validation)\n    print(f\"Fold {fold_idx}:\")\n    for i in range(n_splits):\n        # For each fold, get the date_id range for each part\n        if i == fold_idx:\n            part_data = valid_data\n            part_type = 'Validation'\n        else:\n            part_data = train_data.iloc[int(len(train_data) * (i / n_splits)): int(len(train_data) * ((i + 1) / n_splits))]\n            part_type = f\"Training part {i}\"\n        \n        print(f\"  {part_type} date_id range: ({part_data['date_id'].min()}, {part_data['date_id'].max()})\")\n        \n\n    train_weight = train_data['weight']\n    valid_weight = valid_data['weight']\n\n    # Default LightGBM parameters\n    LGB_PARAMS = {\n        'objective': 'regression_l2',\n        'metric': 'rmse',\n        'learning_rate': 0.05,\n        'num_leaves': 31,\n        'max_depth': -1,\n        'random_state': 42,\n        'device': 'gpu',\n        'verbosity': -1,  # Suppress LightGBM logs\n    }\n\n    # Build the LightGBM dataset\n    train_ds = lgb.Dataset(train_data[FEAT_COLS + ['weight']], label=train_data[TARGET], weight=train_weight)\n    valid_ds = lgb.Dataset(valid_data[FEAT_COLS + ['weight']], label=valid_data[TARGET], weight=valid_weight, reference=train_ds)\n\n    # Callback functions\n    early_stopping_callback = lgb.early_stopping(100)\n    verbose_eval_callback = lgb.log_evaluation(period=50)\n\n    # Train the model\n    model = lgb.train(\n        LGB_PARAMS,\n        train_ds,\n        num_boost_round=1000,\n        valid_sets=[train_ds, valid_ds],  \n        valid_names=['train', 'valid'],\n        callbacks=[early_stopping_callback, verbose_eval_callback],\n    )\n\n    # Save the model for this fold\n    if save_model:\n        model_file = os.path.join(save_path, f\"lgb_model_fold_{fold_idx}.pkl\")\n        joblib.dump(model, model_file)\n        print(f\"Saved model for fold {fold_idx} to {model_file}\")\n    \n    # Evaluate the model\n    y_valid_pred = model.predict(valid_data[FEAT_COLS + ['weight']])\n    r2_score = calculate_r2(valid_data[TARGET], y_valid_pred, valid_weight)\n    print(f\"Fold {fold_idx} validation R2 score: {r2_score}\")\n\n    return model, r2_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T16:56:55.674962Z","iopub.execute_input":"2024-12-11T16:56:55.675512Z","iopub.status.idle":"2024-12-11T16:56:55.692204Z","shell.execute_reply.started":"2024-12-11T16:56:55.675473Z","shell.execute_reply":"2024-12-11T16:56:55.690665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the base directory and subdirectories\nbase_dir = '/kaggle/working/model'\nsubdirs = ['lower', 'deal_sys_miss', 'deal_random_miss']\n\n# Create the directories\nfor subdir in subdirs:\n    dir_path = os.path.join(base_dir, subdir)\n    os.makedirs(dir_path, exist_ok=True)\n\nprint(\"Directories created successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T16:54:07.660033Z","iopub.execute_input":"2024-12-11T16:54:07.660489Z","iopub.status.idle":"2024-12-11T16:54:07.668596Z","shell.execute_reply.started":"2024-12-11T16:54:07.660444Z","shell.execute_reply":"2024-12-11T16:54:07.667332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\ncv_scores = []\n\nfor fold_idx in range(5):\n    model, r2_score = train_single_fold(fold_idx, data_dir=\"/kaggle/working/lower\", n_splits=5, \n                                    save_path=\"/kaggle/working/model/lower\", \n                                    FEAT_COLS=FEAT_COLS, TARGET=TARGET, save_model=True)\n    cv_scores.append(r2_score)\n\nprint(f\"Mean R2 score: {np.mean(cv_scores)}, Std: {np.std(cv_scores)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T16:57:13.278015Z","iopub.execute_input":"2024-12-11T16:57:13.278477Z","iopub.status.idle":"2024-12-11T17:18:34.796840Z","shell.execute_reply.started":"2024-12-11T16:57:13.278441Z","shell.execute_reply":"2024-12-11T17:18:34.795377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\ncv_scores = []\n\nfor fold_idx in range(5):\n    model, r2_score = train_single_fold(fold_idx, data_dir=\"/kaggle/working/deal_sys_miss\", n_splits=5, \n                                    save_path=\"/kaggle/working/model/deal_sys_miss\", \n                                    FEAT_COLS=FEAT_COLS, TARGET=TARGET, save_model=True)\n    cv_scores.append(r2_score)\n\nprint(f\"Mean R2 score: {np.mean(cv_scores)}, Std: {np.std(cv_scores)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\ncv_scores = []\n\nfor fold_idx in range(5):\n    model, r2_score = train_single_fold(fold_idx, data_dir=\"/kaggle/working/deal_random_miss\", n_splits=5, \n                                    save_path=\"/kaggle/working/model/deal_random_miss\", \n                                    FEAT_COLS=FEAT_COLS, TARGET=TARGET, save_model=True)\n    cv_scores.append(r2_score)\n\nprint(f\"Mean R2 score: {np.mean(cv_scores)}, Std: {np.std(cv_scores)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}