{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9849268,"sourceType":"competition"},{"sourceId":9640394,"sourceType":"datasetVersion","datasetId":5882430},{"sourceId":201255000,"sourceType":"kernelVersion"},{"sourceId":201377683,"sourceType":"kernelVersion"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **FOREWORD**\n\nThis is the last kernel in the series - I use the provided API and submit to the leaderboard. This is currently a very simple function, but will get complex as we move along. <br>\n\nPlease note the sequence of kernels as below- <br>\n1. Imports     - https://www.kaggle.com/code/ravi20076/janestreet2024-imports-v1 <br>\n2. Data load   - https://www.kaggle.com/code/ravi20076/janestreet2024-dataload-v1 <br>\n3. Model save  - https://www.kaggle.com/datasets/ravi20076/janestreetpublicv1 <br>\n4. Model train - https://www.kaggle.com/code/ravi20076/janestreet2024-baseline-train-v1 <br>\n4. Inference and submision - this one <br>\n\nI refer to the below kernel for the API usage- <br>\nhttps://www.kaggle.com/code/ryanholbrook/jane-street-rmf-demo-submission\n","metadata":{}},{"cell_type":"markdown","source":"# **IMPORTS**","metadata":{}},{"cell_type":"code","source":"%%time \n\n!pip install polars[gpu]==1.9.0 -q --no-index --find-links=/kaggle/input/janestreet2024-imports-v1/polars\n!pip install lightgbm==4.5.0 -q --no-index --find-links=/kaggle/input/janestreet2024-imports-v1/packages\n!pip install scikit-learn==1.5.2 -q --no-index --find-links=/kaggle/input/janestreet2024-imports-v1/packages\n\nexec(\n    open(\"/kaggle/input/janestreet2024-imports-v1/myimports.py\", \"r\"\n        ).read()\n)\n\nprint()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:03:46.779434Z","iopub.execute_input":"2024-10-17T19:03:46.779722Z","iopub.status.idle":"2024-10-17T19:04:47.888267Z","shell.execute_reply.started":"2024-10-17T19:03:46.779688Z","shell.execute_reply":"2024-10-17T19:04:47.887217Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **CONFIGURATION**","metadata":{}},{"cell_type":"code","source":"%%time\n\ntarget     = \"responder_6\"\nop_path    = f\"/kaggle/working\"\nip_path    = f\"/kaggle/input/janestreet2024-dataload-v1\"\nstate      = 42\nmethod     = \"CB1R\"\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:04:47.890209Z","iopub.execute_input":"2024-10-17T19:04:47.890783Z","iopub.status.idle":"2024-10-17T19:04:47.896545Z","shell.execute_reply.started":"2024-10-17T19:04:47.890745Z","shell.execute_reply":"2024-10-17T19:04:47.895644Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **VERSION DETAILS**","metadata":{}},{"cell_type":"markdown","source":"|Submission Number|Date|Description|CV score | LB score|\n|:-:|:-:|-----------------------|:-:| :-: |\n|LGBMV1_1|15Oct2024|* Single LGBM trained from day 500|0.007555|0.0043|\n|LGBMV1_2|15Oct2024|* Single LGBM trained from day 250|0.007569|0.0044|\n|LGBMV1_4|15Oct2024|* Single LGBM trained from day 100|0.007882|0.0043|\n|LGBMV1_5|15Oct2024|* Single LGBM trained from day 1|0.007667|0.0041|\n|LGBM    |15Oct2024|* Simple average of V1_1 - V1_5 ||0.0044|\n|CBV1_1  |16Oct2024|* Single Catboost trained from day 500|0.007171 ||\n|CBV1_2  |16Oct2024|* Single Catboost trained from day 100|0.007734|0.0039|\n|CBV1_3  |16Oct2024|* Single Catboost trained from day 1|0.007381||\n|CBV1_4  |16Oct2024|* Single Catboost trained from day 200|0.007171||\n|CBV1_5  |16Oct2024|* Single Catboost trained from day 750|0.007171||\n|CBV1_6  |16Oct2024|* Single Catboost trained from day 100|0.007696||\n|LGBMCB  |16Oct2024|* Simple average of selected LGBM and Catboost models|||","metadata":{}},{"cell_type":"markdown","source":"# **PREPROCESSING**\n\nWe load the models and feature lists here and prepare for inference in the next stsp <br>","metadata":{}},{"cell_type":"code","source":"%%time \n\nsel_cols = \\\n[\n'symbol_id', \n'feature_00', 'feature_01', 'feature_02', 'feature_03', 'feature_04',\n'feature_05', 'feature_06', 'feature_07', 'feature_08', 'feature_09', 'feature_10',\n'feature_11', 'feature_12', 'feature_13', 'feature_14', 'feature_15', 'feature_16',\n'feature_17', 'feature_18', 'feature_19', 'feature_20', 'feature_21', 'feature_22',\n'feature_23', 'feature_24', 'feature_25', 'feature_26', 'feature_27', 'feature_28',\n'feature_29', 'feature_30', 'feature_31', 'feature_32', 'feature_33', 'feature_34',\n'feature_35', 'feature_36', 'feature_37', 'feature_38', 'feature_39', 'feature_40',\n'feature_41', 'feature_42', 'feature_43', 'feature_44', 'feature_45', 'feature_46',\n'feature_47', 'feature_48', 'feature_49', 'feature_50', 'feature_51', 'feature_52',\n'feature_53', 'feature_54', 'feature_55', 'feature_56', 'feature_57', 'feature_58',\n'feature_59', 'feature_60', 'feature_61', 'feature_62', 'feature_63', 'feature_64',\n'feature_65', 'feature_66', 'feature_67', 'feature_68', 'feature_69', 'feature_70',\n'feature_71', 'feature_72', 'feature_73', 'feature_74', 'feature_75', 'feature_76',\n'feature_77', 'feature_78'\n]\n\nall_files  = sorted(os.listdir(f\"/kaggle/input/janestreetpublicv1\"))\nsel_models = [\"CBV1_2.joblib\", \"CBV1_3.joblib\", \"CBV1_5.joblib\", \n              \"LGBMV1_1.joblib\", \"LGBMV1_2.joblib\", \"LGBMV1_4.joblib\", \"LGBMV1_5.joblib\",\n             ]\nall_files  = list(set(all_files).intersection(set(sel_models)))\n\nmodels = []\nfor file in sorted(all_files):\n    PrintColor(f\"---> Current model file - {file}\", color = Fore.CYAN)\n    fitted_model = \\\n    joblib.load(\n        os.path.join(f\"/kaggle/input/janestreetpublicv1\", file)\n    )[\"Online\"]\n        \n    models.append(fitted_model)\n    del fitted_model\n     \nPrintColor(f\"\\n---> Models for inference\\n\")\npprint(models)\n\nprint()\ncollect();","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:04:47.897784Z","iopub.execute_input":"2024-10-17T19:04:47.898077Z","iopub.status.idle":"2024-10-17T19:05:15.441860Z","shell.execute_reply.started":"2024-10-17T19:04:47.898045Z","shell.execute_reply":"2024-10-17T19:05:15.441090Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **INFERENCE AND SUBMISSION**","metadata":{}},{"cell_type":"code","source":"%%time \n\nimport kaggle_evaluation.jane_street_inference_server\n\nlags_ : pl.DataFrame | None = None\n\ndef predict(\n    test: pl.DataFrame, \n    lags: pl.DataFrame | None\n) -> pl.DataFrame | pd.DataFrame:\n    \"This is the inference and submission function used to predict the test set for the competition\"\n\n    global lags_, models, sel_cols, target\n    \n    if lags is not None:\n        lags_ = lags\n        \n    test_preds = []\n    for model in tqdm(models):\n        test_preds.append(\n            model.predict(\n                test.select(pl.col(sel_cols)).to_pandas()\n            )\n        )\n        \n    test_preds = \\\n    np.average(\n        np.stack(test_preds, axis=1), \n        axis    = 1,\n        weights = [0.10, 0.10, 0.10, 0.10, 0.25, 0.25, 0.10]\n    )\n    \n    predictions = \\\n    test.select('row_id').\\\n    with_columns(\n        pl.Series(\n            name   = 'responder_6', \n            values = np.clip(test_preds, a_min = -5, a_max = 5),\n            dtype  = pl.Float64,\n        )\n    )  \n            \n    assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n    assert predictions.columns == ['row_id', 'responder_6']\n    assert len(predictions) == len(test)\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:05:15.446079Z","iopub.execute_input":"2024-10-17T19:05:15.448333Z","iopub.status.idle":"2024-10-17T19:05:15.678288Z","shell.execute_reply.started":"2024-10-17T19:05:15.448279Z","shell.execute_reply":"2024-10-17T19:05:15.677336Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = \\\nkaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:05:15.679335Z","iopub.execute_input":"2024-10-17T19:05:15.679812Z","iopub.status.idle":"2024-10-17T19:05:16.055194Z","shell.execute_reply.started":"2024-10-17T19:05:15.679780Z","shell.execute_reply":"2024-10-17T19:05:16.054103Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}