{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":10175682,"sourceType":"datasetVersion","datasetId":6285101},{"sourceId":10274299,"sourceType":"datasetVersion","datasetId":6357257}],"dockerImageVersionId":30805,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom pathlib import Path\nimport numpy as np\nimport polars as pl\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport lightgbm as lgb\n\nfrom tqdm import tqdm\nimport pyarrow.parquet as pq\nimport shutil\nimport time\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nimport kaggle_evaluation.jane_street_inference_server\nimport joblib","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-16T04:05:22.266316Z","iopub.execute_input":"2024-12-16T04:05:22.266739Z","iopub.status.idle":"2024-12-16T04:05:27.589574Z","shell.execute_reply.started":"2024-12-16T04:05:22.266698Z","shell.execute_reply":"2024-12-16T04:05:27.588311Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"|Methods|Delete feature_08 in 8 completely missing date|Ffill & Bfill partially missing|Groupby(symbol_id) + ffill & bfill partially missing|\n|---|---|---|---|\n|Experiment 1|--------------------------√--------------------------|   |   |\n|Experiment 2|   |--------------√--------------|   |\n|Experiment 3|   |   |----------------------------√----------------------------|\n\nThe results are:\n|Pre-Processing| | |CV| | | |LB|\n|---|---|---|---|---|---|---|---|\n|***methods***|***fold_0***|***fold_1***|***fold_2***|***fold_3***|***fold_4***|***mean***|***fold_2***|\n|compress memory|0.01478|0.00801|0.00865|0.00844|0.00588|0.00915|0.0043|\n|Experiment 1|0.01478|0.00802|0.00865|0.00760|0.00571|0.00895|0.0043|\n|Experiment 2|0.01450|0.00812|0.00805|0.00770|0.00572|0.00882|0.0029|\n|Experiment 3|0.01474|0.00805|0.00859|0.00822|0.00563|0.00905|0.0036|\n|This notebook|0.01474|0.00800|0.00866|0.00777|0.00578|0.00899|LB|","metadata":{}},{"cell_type":"code","source":"folder_path = '/kaggle/input/js-dpp-cvp/lower'\nfile_names = [f'combined_part{i}.parquet' for i in range(5)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T04:05:43.462484Z","iopub.execute_input":"2024-12-16T04:05:43.463008Z","iopub.status.idle":"2024-12-16T04:05:43.469191Z","shell.execute_reply.started":"2024-12-16T04:05:43.462963Z","shell.execute_reply":"2024-12-16T04:05:43.467789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# data = pd.DataFrame()\n\n# for file_name in tqdm(file_names):\n#     file_path = os.path.join(folder_path, file_name)\n#     temp_data = pd.read_parquet(file_path)\n#     data = pd.concat([data, temp_data], ignore_index=True)\n\n# missing_columns_by_date = {}\n\n# grouped = data.groupby('date_id')\n\n# for date_id, group in grouped:\n    \n#     missing_cols = group.columns[group.isnull().all()].tolist()\n#     if missing_cols:\n#         missing_columns_by_date[date_id] = missing_cols\n\n\n# for date_id, missing_cols in missing_columns_by_date.items():\n#     print(f\"date_id: {date_id}, missing columns: {missing_cols}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T03:55:35.299390Z","iopub.execute_input":"2024-12-16T03:55:35.301362Z","iopub.status.idle":"2024-12-16T03:55:35.312616Z","shell.execute_reply.started":"2024-12-16T03:55:35.301217Z","shell.execute_reply":"2024-12-16T03:55:35.311424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Create the target directory if it does not exist\n# output_dir = '/kaggle/working/fill_1_Ffour'\n# os.makedirs(output_dir, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T03:55:47.874504Z","iopub.execute_input":"2024-12-16T03:55:47.875002Z","iopub.status.idle":"2024-12-16T03:55:47.881092Z","shell.execute_reply.started":"2024-12-16T03:55:47.874965Z","shell.execute_reply":"2024-12-16T03:55:47.879798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# data = pd.DataFrame()\n\n# # Define the columns to fill with 1 for the specified date_id range\n# columns_to_fill = ['feature_21', 'feature_26', 'feature_27', 'feature_31']\n# date_id_range_to_modify = (248, 538)\n\n# # Process each file\n# for idx, file_name in tqdm(enumerate(file_names)):\n#     file_path = os.path.join(folder_path, file_name)\n#     temp_data = pd.read_parquet(file_path)\n    \n#     # Only modify the first file for the specific date_id range\n#     mask = (temp_data['date_id'] >= date_id_range_to_modify[0]) & (temp_data['date_id'] <= date_id_range_to_modify[1])\n#     temp_data.loc[mask, columns_to_fill] = temp_data.loc[mask, columns_to_fill].fillna(1)\n    \n#     # Save the data (modified or copied) to the output directory\n#     output_file = os.path.join(output_dir, file_name)\n#     temp_data.to_parquet(output_file, index=False)\n\n#     print(f\"Saved {file_name} to {output_file}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T03:56:23.039882Z","iopub.execute_input":"2024-12-16T03:56:23.041119Z","iopub.status.idle":"2024-12-16T04:01:31.117617Z","shell.execute_reply.started":"2024-12-16T03:56:23.041075Z","shell.execute_reply":"2024-12-16T04:01:31.116247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# folder_path = '/kaggle/working/fill_1_Ffour'\n# data = pd.DataFrame()\n\n# for file_name in tqdm(file_names):\n#     file_path = os.path.join(folder_path, file_name)\n#     temp_data = pd.read_parquet(file_path)\n#     data = pd.concat([data, temp_data], ignore_index=True)\n\n# missing_columns_by_date = {}\n\n# grouped = data.groupby('date_id')\n\n# for date_id, group in grouped:\n    \n#     missing_cols = group.columns[group.isnull().all()].tolist()\n#     if missing_cols:\n#         missing_columns_by_date[date_id] = missing_cols\n\n\n# for date_id, missing_cols in missing_columns_by_date.items():\n#     print(f\"date_id: {date_id}, missing columns: {missing_cols}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T04:05:46.271704Z","iopub.execute_input":"2024-12-16T04:05:46.272159Z","iopub.status.idle":"2024-12-16T04:07:35.527124Z","shell.execute_reply.started":"2024-12-16T04:05:46.272120Z","shell.execute_reply":"2024-12-16T04:07:35.525876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TARGET = 'responder_6'\nFEAT_COLS = [f\"feature_{i:02d}\" for i in range(79)]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_r2(y_true, y_pred, weights):\n    \"\"\"Calculate the R2 score, weighted by the provided weights.\"\"\"\n    numerator = np.sum(weights * (y_true - y_pred) ** 2)\n    denominator = np.sum(weights * (y_true ** 2))\n    r2_score = 1 - (numerator / denominator)\n    return r2_score","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_single_fold(fold_idx, data_dir, n_splits, save_path, FEAT_COLS, TARGET, save_model=True):\n    \"\"\"\n    Train a single fold for cross-validation.\n\n    Parameters:\n    - fold_idx: the index of the fold (0 to n_splits-1)\n    - data_dir: the directory containing the .parquet files\n    - n_splits: total number of splits\n    - save_path: path to save the model\n    - FEAT_COLS: list of feature columns\n    - TARGET: target column\n    - save_model: whether to save the model after training\n    \"\"\"\n    # For each fold, load the current fold file\n    valid_file = os.path.join(data_dir, f\"combined_part{fold_idx}.parquet\")\n    print(f\"Fold {fold_idx}: validation file {valid_file}\")\n\n    # Load the current fold's data\n    fold_data = pd.read_parquet(valid_file)\n\n    # Split the data: the `i`-th part is validation, the rest is training\n    valid_data = fold_data.iloc[int(len(fold_data) * (fold_idx / n_splits)): int(len(fold_data) * ((fold_idx + 1) / n_splits))]\n    train_data = pd.concat([fold_data.iloc[:int(len(fold_data) * (fold_idx / n_splits))], \n                            fold_data.iloc[int(len(fold_data) * ((fold_idx + 1) / n_splits)):]],\n                           ignore_index=True)\n    \n\n    # Print date_id range for each of the 5 parts (4 training and 1 validation)\n    print(f\"Fold {fold_idx}:\")\n    for i in range(n_splits):\n        # For each fold, get the date_id range for each part\n        if i == fold_idx:\n            part_data = valid_data\n            part_type = 'Validation'\n        else:\n            part_data = train_data.iloc[int(len(train_data) * (i / n_splits)): int(len(train_data) * ((i + 1) / n_splits))]\n            part_type = f\"Training part {i}\"\n        \n        print(f\"  {part_type} date_id range: ({part_data['date_id'].min()}, {part_data['date_id'].max()})\")\n        \n\n    train_weight = train_data['weight']\n    valid_weight = valid_data['weight']\n\n    # Default LightGBM parameters\n    LGB_PARAMS = {\n        'objective': 'regression_l2',\n        'metric': 'rmse',\n        'learning_rate': 0.05,\n        'num_leaves': 31,\n        'max_depth': -1,\n        'random_state': 42,\n        'device': 'gpu',\n        'verbosity': -1,  # Suppress LightGBM logs\n    }\n\n    # Build the LightGBM dataset\n    train_ds = lgb.Dataset(train_data[FEAT_COLS + ['weight']], label=train_data[TARGET], weight=train_weight)\n    valid_ds = lgb.Dataset(valid_data[FEAT_COLS + ['weight']], label=valid_data[TARGET], weight=valid_weight, reference=train_ds)\n\n    # Callback functions\n    early_stopping_callback = lgb.early_stopping(100)\n    verbose_eval_callback = lgb.log_evaluation(period=50)\n\n    # Train the model\n    model = lgb.train(\n        LGB_PARAMS,\n        train_ds,\n        num_boost_round=1000,\n        valid_sets=[train_ds, valid_ds],  \n        valid_names=['train', 'valid'],\n        callbacks=[early_stopping_callback, verbose_eval_callback],\n    )\n\n    # Save the model for this fold\n    if save_model:\n        model_file = os.path.join(save_path, f\"lgb_model_fold_{fold_idx}.pkl\")\n        joblib.dump(model, model_file)\n        print(f\"Saved model for fold {fold_idx} to {model_file}\")\n    \n    # Evaluate the model\n    y_valid_pred = model.predict(valid_data[FEAT_COLS + ['weight']])\n    r2_score = calculate_r2(valid_data[TARGET], y_valid_pred, valid_weight)\n    print(f\"Fold {fold_idx} validation R2 score: {r2_score}\")\n\n    return model, r2_score","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_model(file_path):\n    model = joblib.load(file_path)\n    print(f\"Loaded the model group from {file_path}\")\n    return model","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Create the target directory if it does not exist\n# output_dir = '/kaggle/working/fill_1_Ffour_models'\n# os.makedirs(output_dir, exist_ok=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# %%time\n\n# cv_scores = []\n\n# for fold_idx in range(5):\n#     model, r2_score = train_single_fold(fold_idx, data_dir=\"/kaggle/working/fill_1_Ffour\", n_splits=5, \n#                                     save_path=\"/kaggle/working/fill_1_Ffour_models\", \n#                                     FEAT_COLS=FEAT_COLS, TARGET=TARGET, save_model=True)\n#     cv_scores.append(r2_score)\n\n# print(f\"Mean R2 score: {np.mean(cv_scores)}, Std: {np.std(cv_scores)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_model = load_model(file_path=\"/kaggle/input/js-pp-feafour/fill_1_Ffour_models/lgb_model_fold_2.pkl\")\nprint(f\"Loaded model from the saved file.\")\nprint(f\"Type of lgb_model: {type(lgb_model)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lags_ : pl.DataFrame | None = None\n\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    \n    feat = test[FEAT_COLS+['weight']].to_pandas()\n\n    pred = lgb_model.predict(feat)\n\n    \n    predictions = predictions.with_columns(pl.Series('responder_6', pred.ravel()))\n    print(predictions)\n\n    assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n\n    assert list(predictions.columns) == ['row_id', 'responder_6']\n\n    assert len(predictions) == len(test)\n\n    return predictions","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}