{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":9883535,"sourceType":"datasetVersion","datasetId":6068987}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip install -q optuna\n# !pip install -q dask[dataframe]","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:20:53.516941Z","iopub.execute_input":"2024-11-14T10:20:53.517463Z","iopub.status.idle":"2024-11-14T10:20:53.549201Z","shell.execute_reply.started":"2024-11-14T10:20:53.517399Z","shell.execute_reply":"2024-11-14T10:20:53.547976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:20:53.551984Z","iopub.execute_input":"2024-11-14T10:20:53.553104Z","iopub.status.idle":"2024-11-14T10:20:55.317370Z","shell.execute_reply.started":"2024-11-14T10:20:53.553041Z","shell.execute_reply":"2024-11-14T10:20:55.316094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pl.read_parquet('/kaggle/input/js-2024-train/train_df_half.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:20:55.319085Z","iopub.execute_input":"2024-11-14T10:20:55.319645Z","iopub.status.idle":"2024-11-14T10:21:20.990075Z","shell.execute_reply.started":"2024-11-14T10:20:55.319601Z","shell.execute_reply":"2024-11-14T10:21:20.988763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = [col for col in df.columns if ('feature' in col) or ('_lag_' in col)]\nresponders = [col for col in df.columns if ('_lag_' in col)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:21:20.993008Z","iopub.execute_input":"2024-11-14T10:21:20.993492Z","iopub.status.idle":"2024-11-14T10:21:21.006156Z","shell.execute_reply.started":"2024-11-14T10:21:20.993444Z","shell.execute_reply":"2024-11-14T10:21:21.004430Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Again taking 50% but we can comment it out if we got more RAM\ndf = df.slice(int(df.shape[0] * 0.5), None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:21:21.007988Z","iopub.execute_input":"2024-11-14T10:21:21.009044Z","iopub.status.idle":"2024-11-14T10:21:21.035893Z","shell.execute_reply.started":"2024-11-14T10:21:21.008979Z","shell.execute_reply":"2024-11-14T10:21:21.034255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.fill_null(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:21:21.037627Z","iopub.execute_input":"2024-11-14T10:21:21.038085Z","iopub.status.idle":"2024-11-14T10:21:21.302644Z","shell.execute_reply.started":"2024-11-14T10:21:21.038039Z","shell.execute_reply":"2024-11-14T10:21:21.301285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df[features].slice(1, None)\n# X_r = df[responders].slice(1, None)\ny = df['responder_6'].slice(1, None)\nweights = df['weight'].slice(1, None)\ndf = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:21:21.304074Z","iopub.execute_input":"2024-11-14T10:21:21.305270Z","iopub.status.idle":"2024-11-14T10:21:21.326378Z","shell.execute_reply.started":"2024-11-14T10:21:21.305223Z","shell.execute_reply":"2024-11-14T10:21:21.324914Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:21:21.328177Z","iopub.execute_input":"2024-11-14T10:21:21.328704Z","iopub.status.idle":"2024-11-14T10:21:21.448221Z","shell.execute_reply.started":"2024-11-14T10:21:21.328655Z","shell.execute_reply":"2024-11-14T10:21:21.446760Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nimport xgboost as xgb\nfrom sklearn.linear_model import Ridge, LinearRegression, Lasso\nfrom sklearn.model_selection import TimeSeriesSplit","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:21:21.450391Z","iopub.execute_input":"2024-11-14T10:21:21.451311Z","iopub.status.idle":"2024-11-14T10:21:25.948214Z","shell.execute_reply.started":"2024-11-14T10:21:21.451249Z","shell.execute_reply":"2024-11-14T10:21:25.946843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import r2_score\nfrom sklearn.model_selection import TimeSeriesSplit\n\ndef weighted_r2_score(y_true, y_pred, sample_weight=None):\n    return r2_score(y_true, y_pred, sample_weight=sample_weight)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:21:25.953149Z","iopub.execute_input":"2024-11-14T10:21:25.953905Z","iopub.status.idle":"2024-11-14T10:21:25.961113Z","shell.execute_reply.started":"2024-11-14T10:21:25.953849Z","shell.execute_reply":"2024-11-14T10:21:25.959434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:21:25.962895Z","iopub.execute_input":"2024-11-14T10:21:25.963448Z","iopub.status.idle":"2024-11-14T10:21:25.990329Z","shell.execute_reply.started":"2024-11-14T10:21:25.963384Z","shell.execute_reply":"2024-11-14T10:21:25.989141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# tscv = TimeSeriesSplit(n_splits=5)\n# scores = []\n\n# for train_index, test_index in tqdm(tscv.split(X)):\n#         X_train_fold = X.to_numpy()[train_index]\n#         X_test_fold = X.to_numpy()[test_index]\n#         y_train_fold = y.to_numpy()[train_index]\n#         y_test_fold = y.to_numpy()[test_index]\n#         weights_train_fold = weights.to_numpy()[train_index]\n#         weights_test_fold = weights.to_numpy()[test_index]\n\n#         # Create and train LightGBM model\n#         model = Ridge()\n\n#         model.fit(X_train_fold, y_train_fold)\n\n#         # Predict and calculate weighted R2\n#         y_pred_fold = model.predict(X_test_fold)\n#         score = weighted_r2_score(y_test_fold, y_pred_fold, sample_weight=weights_test_fold)\n#         scores.append(score)\n#         print(score)\n# np.mean(scores)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:21:25.991958Z","iopub.execute_input":"2024-11-14T10:21:25.992466Z","iopub.status.idle":"2024-11-14T10:21:25.998677Z","shell.execute_reply.started":"2024-11-14T10:21:25.992408Z","shell.execute_reply":"2024-11-14T10:21:25.997424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = Ridge()\n\nmodel.fit(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:21:26.000288Z","iopub.execute_input":"2024-11-14T10:21:26.000790Z","iopub.status.idle":"2024-11-14T10:22:01.807531Z","shell.execute_reply.started":"2024-11-14T10:21:26.000735Z","shell.execute_reply":"2024-11-14T10:22:01.806204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:22:01.809045Z","iopub.execute_input":"2024-11-14T10:22:01.809455Z","iopub.status.idle":"2024-11-14T10:22:01.819754Z","shell.execute_reply.started":"2024-11-14T10:22:01.809413Z","shell.execute_reply":"2024-11-14T10:22:01.818299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pd.DataFrame:\n    global lags_\n    \n    test_pd = test.to_pandas()\n    \n    if lags is not None:\n        lags_ = lags\n\n        # Convert lags to Pandas DataFrame if it's Polars\n        lags_pd = lags.to_pandas()\n        lags_pd = lags_pd.groupby(['date_id', 'symbol_id']).last().reset_index() \n        test_pd = test_pd.merge(lags_pd, on=['date_id', 'symbol_id'], how='left')\n    else:\n        \n        for idx in range(9):\n            test_pd[f'responder_{idx}_lag_1'] = 0.0\n\n    features = [col for col in test_pd.columns if ('feature' in col) or ('responder' in col)]\n    print(features)  \n    # Prepare the features for prediction\n    test_pd = test_pd.fillna(0)\n    print(test_pd.isna().sum().sum())\n    predictions =  model.predict(test_pd[features])\n\n    # Create a DataFrame for the output\n    output = pd.DataFrame({\n        'row_id': test_pd['row_id'],\n        'responder_6': predictions\n    })\n\n    # Ensure the output DataFrame has the correct format\n    assert output.columns.tolist() == ['row_id', 'responder_6']\n    assert len(output) == len(test)\n\n    return output\n\nimport kaggle_evaluation.jane_street_inference_server\nimport os\n\n# Set up the inference server\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T10:22:01.821953Z","iopub.execute_input":"2024-11-14T10:22:01.822485Z","iopub.status.idle":"2024-11-14T10:22:02.400042Z","shell.execute_reply.started":"2024-11-14T10:22:01.822414Z","shell.execute_reply":"2024-11-14T10:22:02.398573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}