{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-15T02:16:55.884474Z","iopub.execute_input":"2024-12-15T02:16:55.88481Z","iopub.status.idle":"2024-12-15T02:16:55.934054Z","shell.execute_reply.started":"2024-12-15T02:16:55.884779Z","shell.execute_reply":"2024-12-15T02:16:55.9332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport polars as pl\nimport numpy as np\nimport os, gc\nfrom tqdm.auto import tqdm\nfrom matplotlib import pyplot as plt\nimport pickle\n\nfrom sklearn.metrics import r2_score\nfrom lightgbm import LGBMRegressor\nimport lightgbm as lgb\nfrom xgboost import XGBRegressor\nimport xgboost as xgb\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\n# import kaggle_evaluation.jane_street_inference_server\n\npl.Config.set_tbl_rows(100)\npl.Config.set_tbl_cols(400)\npl.Config.set_fmt_table_cell_list_len(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T02:16:58.563462Z","iopub.execute_input":"2024-12-15T02:16:58.56381Z","iopub.status.idle":"2024-12-15T02:16:59.924552Z","shell.execute_reply.started":"2024-12-15T02:16:58.563778Z","shell.execute_reply":"2024-12-15T02:16:59.923613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CONFIG:\n    debug = False\n    seed = 42\n    target_col = \"responder_6\"\n    lag_cols_rename = { f\"responder_{idx}_lag_1\" : f\"responder_{idx}\" for idx in range(9)}\n    lag_target_cols_name = [f\"responder_{idx}\" for idx in range(9)]\n    lag_cols_original = [\"date_id\", \"time_id\", \"symbol_id\"] + [f\"responder_{idx}\" for idx in range(9)]\n    feature_cols = [\"date_id\", \"time_id\", \"symbol_id\"] + [f\"feature_{idx:02d}\" for idx in range(79)]\n    model_path = \"/kaggle/input/janestreet-public-model/xgb_001.pkl\"\n    lag_ndays = 4","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T02:16:59.926076Z","iopub.execute_input":"2024-12-15T02:16:59.926873Z","iopub.status.idle":"2024-12-15T02:16:59.932626Z","shell.execute_reply.started":"2024-12-15T02:16:59.926827Z","shell.execute_reply":"2024-12-15T02:16:59.931756Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Prepare data","metadata":{"execution":{"iopub.status.busy":"2024-11-24T04:14:03.772494Z","iopub.execute_input":"2024-11-24T04:14:03.773218Z","iopub.status.idle":"2024-11-24T04:14:04.213222Z","shell.execute_reply.started":"2024-11-24T04:14:03.773183Z","shell.execute_reply":"2024-11-24T04:14:04.212039Z"}}},{"cell_type":"code","source":"history = pl.scan_parquet(\n    \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet\"\n).select(['date_id','time_id','symbol_id'] + [f\"responder_{idx}\" for idx in range(9)] + [f\"feature_{idx:02d}\" for idx in range(79)]).filter(\n    pl.col(\"time_id\")==967\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T02:17:02.050433Z","iopub.execute_input":"2024-12-15T02:17:02.050761Z","iopub.status.idle":"2024-12-15T02:17:02.103822Z","shell.execute_reply.started":"2024-12-15T02:17:02.050732Z","shell.execute_reply":"2024-12-15T02:17:02.102911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = history.collect()\n\nhistory_column_types = {\n    'date_id': pl.Int16,\n    'time_id': pl.Int16,\n    'symbol_id': pl.Int16\n}\n\nfeature_column_types = {}\nfor f in [f\"feature_{idx:02d}\" for idx in range(79)]:\n    feature_column_types[f] = pl.Float32\n\nresponder_column_types = {}\nfor f in [f\"responder_{idx}\" for idx in range(9)]:\n    responder_column_types[f] = pl.Float32\n\nhistory = history.cast(history_column_types)\nhistory = history.cast(responder_column_types)\nhistory.tail()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T02:17:03.20553Z","iopub.execute_input":"2024-12-15T02:17:03.205912Z","iopub.status.idle":"2024-12-15T02:17:33.561724Z","shell.execute_reply.started":"2024-12-15T02:17:03.205866Z","shell.execute_reply":"2024-12-15T02:17:33.560692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import pandas as pd\n# import numpy as np\n# import tensorflow as tf\n# from tensorflow.keras import layers, models\n# # Separate features and responders\n# import pandas as pd\n# import gc\n# # Initialize a list to hold samples from each file\n# samples = []\n# # Load a sample from each file\n# #for i in range(10):\n# for i in [8]:\n#     file_path = f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet\"\n#     chunk = pd.read_parquet(file_path)\n    \n#     # Take a sample of the data (adjust sample size as needed)\n#     #sample_chunk = chunk.sample(n=500000, random_state=42)  # For example, 100 rows\n#     sample_chunk = chunk\n#     samples.append(sample_chunk)\n# # Concatenate all samples into one DataFrame if needed\n# del chunk\n# gc.collect()  # Forces garbage collection\n# sample_df = pd.concat(samples, ignore_index=True)\n# del samples\n# gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T03:39:13.482501Z","iopub.execute_input":"2024-12-08T03:39:13.483213Z","iopub.status.idle":"2024-12-08T03:39:13.487186Z","shell.execute_reply.started":"2024-12-08T03:39:13.483176Z","shell.execute_reply":"2024-12-08T03:39:13.486311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = history.select(['symbol_id', 'date_id']+[f\"feature_{idx:02d}\" for idx in range(79)]+[CONFIG.target_col]).filter(\n    pl.col(\"date_id\")<=1500\n).to_numpy()\ny_train = history.select(['symbol_id', 'date_id', CONFIG.target_col]).filter(\n    pl.col(\"date_id\") <= 1500\n).to_numpy()\n\nX_test = history.select(['symbol_id', 'date_id']+[f\"feature_{idx:02d}\" for idx in range(79)]).filter(\n    (pl.col(\"date_id\")>1500) & (pl.col(\"date_id\")<=1628)\n).to_numpy()\n\ny_test = history.select(['symbol_id', 'date_id', CONFIG.target_col]).filter(\n    (pl.col(\"date_id\")>1500) & (pl.col(\"date_id\")<=1628)\n).to_numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T02:24:59.719765Z","iopub.execute_input":"2024-12-15T02:24:59.720644Z","iopub.status.idle":"2024-12-15T02:24:59.767716Z","shell.execute_reply.started":"2024-12-15T02:24:59.720611Z","shell.execute_reply":"2024-12-15T02:24:59.767058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = X_train.filter(regex='^feature_')\nresponders = sample_df.filter(regex='^responder_')[['responder_6']]\nweights = sample_df['weight']\n# Convert to numpy arrays for TensorFlow\nX = features.values  # Features for input\n#y = responders.values  # Responders for output\n# Assuming you have a DataFrame `y_train` with all responders\ny = responders.values.squeeze()  # Keep only responder_6\nX = np.nan_to_num(X, nan=0.0, posinf=0.0, neginf=0.0)\ny = np.nan_to_num(y, nan=0.0, posinf=0.0, neginf=0.0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T04:13:37.453779Z","iopub.execute_input":"2024-12-08T04:13:37.454135Z","iopub.status.idle":"2024-12-08T04:13:37.48537Z","shell.execute_reply.started":"2024-12-08T04:13:37.454107Z","shell.execute_reply":"2024-12-08T04:13:37.484225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:14:12.332548Z","iopub.execute_input":"2024-12-15T03:14:12.333511Z","iopub.status.idle":"2024-12-15T03:14:12.347047Z","shell.execute_reply.started":"2024-12-15T03:14:12.333473Z","shell.execute_reply":"2024-12-15T03:14:12.346076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_input.sort_values(['unique_id','ds'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:20:27.005212Z","iopub.execute_input":"2024-12-15T03:20:27.00605Z","iopub.status.idle":"2024-12-15T03:20:27.02005Z","shell.execute_reply.started":"2024-12-15T03:20:27.006018Z","shell.execute_reply":"2024-12-15T03:20:27.019216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_input = pd.DataFrame(data=y_train, columns=['unique_id', 'ds', 'y'])\ny_trans = y_input.sort_values(['ds']).groupby('unique_id')['y'].apply(lambda x: np.array(x)).reset_index()\ny_final = y_trans['y'].values.tolist()\ny_freq = [0]*len(y_final)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:28:10.169387Z","iopub.execute_input":"2024-12-15T03:28:10.17025Z","iopub.status.idle":"2024-12-15T03:28:10.180682Z","shell.execute_reply.started":"2024-12-15T03:28:10.170212Z","shell.execute_reply":"2024-12-15T03:28:10.179924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:41:53.538754Z","iopub.execute_input":"2024-12-15T03:41:53.53913Z","iopub.status.idle":"2024-12-15T03:41:53.543775Z","shell.execute_reply.started":"2024-12-15T03:41:53.539096Z","shell.execute_reply":"2024-12-15T03:41:53.542802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:47:44.799371Z","iopub.execute_input":"2024-12-15T03:47:44.800005Z","iopub.status.idle":"2024-12-15T03:47:44.804167Z","shell.execute_reply.started":"2024-12-15T03:47:44.799968Z","shell.execute_reply":"2024-12-15T03:47:44.803065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_freq","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:47:49.361081Z","iopub.execute_input":"2024-12-15T03:47:49.361743Z","iopub.status.idle":"2024-12-15T03:47:49.367825Z","shell.execute_reply.started":"2024-12-15T03:47:49.361708Z","shell.execute_reply":"2024-12-15T03:47:49.366621Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%pip install timesfm[pax]\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T01:39:33.10051Z","iopub.execute_input":"2024-12-15T01:39:33.10084Z","iopub.status.idle":"2024-12-15T01:42:35.310071Z","shell.execute_reply.started":"2024-12-15T01:39:33.10081Z","shell.execute_reply":"2024-12-15T01:42:35.308952Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load the model","metadata":{}},{"cell_type":"code","source":"import timesfm\n\ntfm = timesfm.TimesFm(\n    hparams=timesfm.TimesFmHparams(\n        backend=\"gpu\",  # or \"cpu\" based on your setup\n        per_core_batch_size=32,\n        horizon_len=128,\n    ),\n    checkpoint=timesfm.TimesFmCheckpoint(\n        huggingface_repo_id=\"google/timesfm-1.0-200m\"\n    ),\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T01:59:54.462579Z","iopub.execute_input":"2024-12-15T01:59:54.463405Z","iopub.status.idle":"2024-12-15T02:00:57.839757Z","shell.execute_reply.started":"2024-12-15T01:59:54.463368Z","shell.execute_reply":"2024-12-15T02:00:57.838872Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Forecast","metadata":{}},{"cell_type":"code","source":"forecast_input = [\n    np.sin(np.linspace(0, 20, 100)),\n    np.sin(np.linspace(0, 20, 200)),\n    np.sin(np.linspace(0, 20, 400)),\n]\nforecast_input","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"forecast_input = [\n    y_final[0],\n    y_final[1],\n    y_final[2]\n]\nforecast_input","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:45:58.128288Z","iopub.execute_input":"2024-12-15T03:45:58.12862Z","iopub.status.idle":"2024-12-15T03:45:58.151768Z","shell.execute_reply.started":"2024-12-15T03:45:58.128591Z","shell.execute_reply":"2024-12-15T03:45:58.150828Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_input = pd.DataFrame(data=y_train, columns=['unique_id', 'ds', 'y'])\ny_trans = y_input.sort_values(['ds']).groupby('unique_id')['y'].apply(lambda x: np.array(x)).reset_index()\ny_final = y_trans['y'].values.tolist()\ny_freq = [0]*len(y_final)\npoint_forecast, experimental_quantile_forecast = tfm.forecast(\n    y_final,\n    freq=y_freq,\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:48:13.869149Z","iopub.execute_input":"2024-12-15T03:48:13.869925Z","iopub.status.idle":"2024-12-15T03:48:13.981951Z","shell.execute_reply.started":"2024-12-15T03:48:13.869889Z","shell.execute_reply":"2024-12-15T03:48:13.981045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(point_forecast)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:49:37.601377Z","iopub.execute_input":"2024-12-15T03:49:37.602013Z","iopub.status.idle":"2024-12-15T03:49:37.607436Z","shell.execute_reply.started":"2024-12-15T03:49:37.601979Z","shell.execute_reply":"2024-12-15T03:49:37.606586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:51:39.67236Z","iopub.execute_input":"2024-12-15T03:51:39.672746Z","iopub.status.idle":"2024-12-15T03:51:39.679098Z","shell.execute_reply.started":"2024-12-15T03:51:39.672712Z","shell.execute_reply":"2024-12-15T03:51:39.678165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"point_forecast[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T03:51:30.042605Z","iopub.execute_input":"2024-12-15T03:51:30.043008Z","iopub.status.idle":"2024-12-15T03:51:30.050744Z","shell.execute_reply.started":"2024-12-15T03:51:30.042963Z","shell.execute_reply":"2024-12-15T03:51:30.049729Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluation","metadata":{}},{"cell_type":"code","source":"# Initialize a list to hold samples from each file\nsamples = []\n# Load a sample from each file\n#for i in range(10):\nfor i in [9]:\n    file_path = f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet\"\n    chunk = pd.read_parquet(file_path)\n    \n    # Take a sample of the data (adjust sample size as needed)\n    #sample_chunk = chunk.sample(n=500000, random_state=42)  # For example, 100 rows\n    sample_chunk = chunk\n    samples.append(sample_chunk)\n# Concatenate all samples into one DataFrame if needed\ndel chunk\ngc.collect()  # Forces garbage collection\ntest_df = pd.concat(samples, ignore_index=True)\ndel samples\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T05:41:36.569814Z","iopub.execute_input":"2024-11-24T05:41:36.570803Z","iopub.status.idle":"2024-11-24T05:41:40.920962Z","shell.execute_reply.started":"2024-11-24T05:41:36.570746Z","shell.execute_reply":"2024-11-24T05:41:40.920063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T05:46:27.920215Z","iopub.execute_input":"2024-11-24T05:46:27.921134Z","iopub.status.idle":"2024-11-24T05:46:27.949904Z","shell.execute_reply.started":"2024-11-24T05:46:27.921079Z","shell.execute_reply":"2024-11-24T05:46:27.948903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.shape, point_forecast.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-24T05:42:19.018415Z","iopub.execute_input":"2024-11-24T05:42:19.01979Z","iopub.status.idle":"2024-11-24T05:42:19.026287Z","shell.execute_reply.started":"2024-11-24T05:42:19.019708Z","shell.execute_reply":"2024-11-24T05:42:19.02545Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import r2_score\nimport pandas as pd\n\n# Extract the first 128 values from test_df to align dimensions\ny_true_aligned = test_df['responder_6'][:128].values.squeeze()  # Select corresponding segment\n\n# Reshape point_forecast to (128,)\npoint_forecast = point_forecast.squeeze()\n\n# Ensure shapes match\nassert y_true_aligned.shape == point_forecast.shape, \"Shapes must align for R² calculation.\"\n\n# Compute R²\ny_mean = np.mean(y_true_aligned)\nss_res = np.sum((y_true_aligned - point_forecast) ** 2)\nss_tot = np.sum((y_true_aligned - y_mean) ** 2)\nr2 = 1 - (ss_res / ss_tot)\n\nprint(\"R²:\", r2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T05:42:30.226754Z","iopub.execute_input":"2024-11-24T05:42:30.227546Z","iopub.status.idle":"2024-11-24T05:42:30.234366Z","shell.execute_reply.started":"2024-11-24T05:42:30.227509Z","shell.execute_reply":"2024-11-24T05:42:30.233438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Suspect the evaluation is not correct, coz the Id is not aligned. but lets see next week","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T05:47:42.371828Z","iopub.execute_input":"2024-11-24T05:47:42.372249Z","iopub.status.idle":"2024-11-24T05:47:42.376292Z","shell.execute_reply.started":"2024-11-24T05:47:42.372201Z","shell.execute_reply":"2024-11-24T05:47:42.37539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}