{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":203627527,"sourceType":"kernelVersion"},{"sourceId":147672,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":125307,"modelId":148289},{"sourceId":147696,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":125330,"modelId":148289}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Basic LSTM with a single sequence\n","metadata":{}},{"cell_type":"markdown","source":"I've prepared a very basic training loop in torch using an LSTM, this notebook only uses one sequence and doesn't use the available lags. So far my experimentations have resulted in worse overfitting when using additional time steps. The basic LSTM overfits quite intensely as well. I'd love some suggestions on how to fix this.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-27T03:40:13.806804Z","iopub.execute_input":"2024-10-27T03:40:13.807212Z","iopub.status.idle":"2024-10-27T03:40:14.267627Z","shell.execute_reply.started":"2024-10-27T03:40:13.807168Z","shell.execute_reply":"2024-10-27T03:40:14.266600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\n\nimport torch.nn as nn\n\nimport torch.nn.functional as F\n\nimport pandas as pd\n\n\n\nimport numpy as np\n\nimport statistics as stat\n\nfrom sklearn.preprocessing import StandardScaler\n\nfrom sklearn.metrics import roc_auc_score\n\nimport copy\n\nfrom torch.utils.data import TensorDataset, DataLoader\n\nfrom sklearn.model_selection import train_test_split\n\nimport torch.optim as optim\n\nimport copy\n\nimport polars as pl\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:14.269355Z","iopub.execute_input":"2024-10-27T03:40:14.269796Z","iopub.status.idle":"2024-10-27T03:40:18.382177Z","shell.execute_reply.started":"2024-10-27T03:40:14.269754Z","shell.execute_reply":"2024-10-27T03:40:18.381318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"jane_street_real_time_market_data_forecasting_path = '/kaggle/input/jane-street-real-time-market-data-forecasting'","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:18.383416Z","iopub.execute_input":"2024-10-27T03:40:18.383880Z","iopub.status.idle":"2024-10-27T03:40:18.388377Z","shell.execute_reply.started":"2024-10-27T03:40:18.383844Z","shell.execute_reply":"2024-10-27T03:40:18.387525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Basic Model architecture","metadata":{}},{"cell_type":"markdown","source":"### Gaussian Nose\nI added some gaussian noise to the model to attempt to reduce overfitting, it hepls a bit but not perfectly.","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\n\nclass GaussianNoise(nn.Module):\n    def __init__(self, std=0.1):\n        super().__init__()\n        self.std = std\n\n    def forward(self, x):\n        if self.training:  # Only add noise during training\n            noise = torch.randn_like(x) * self.std\n            return x + noise\n        return x","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:18.389512Z","iopub.execute_input":"2024-10-27T03:40:18.389807Z","iopub.status.idle":"2024-10-27T03:40:18.400585Z","shell.execute_reply.started":"2024-10-27T03:40:18.389776Z","shell.execute_reply":"2024-10-27T03:40:18.399633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Basic LSTM","metadata":{}},{"cell_type":"code","source":"class LSTM(nn.Module):\n    def __init__(self, input_size, hidden_dim, output_size, num_layers):\n        super(LSTM, self).__init__()\n        self.hidden_dim = hidden_dim\n        self.num_layers = num_layers\n        self.noise = GaussianNoise(std = .1)\n        self.lstm = nn.LSTM(input_size, hidden_dim, num_layers, batch_first=True)\n        self.fc = nn.Linear(hidden_dim, output_size)\n    def forward(self, x):\n        x = x.unsqueeze(1)\n        x = self.noise(x)\n        h0 = torch.zeros(self.num_layers, x.size(0), self.hidden_dim).to(x.device)\n        c0 = torch.zeros(self.num_layers, x.size(0), self.hidden_dim).to(x.device)\n        out, _ = self.lstm(x, (h0, c0))\n        out = self.fc(out[:, -1, :])\n        return out.squeeze()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:18.403902Z","iopub.execute_input":"2024-10-27T03:40:18.404275Z","iopub.status.idle":"2024-10-27T03:40:18.412957Z","shell.execute_reply.started":"2024-10-27T03:40:18.404238Z","shell.execute_reply":"2024-10-27T03:40:18.412006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training Data Normalization","metadata":{}},{"cell_type":"markdown","source":"I've found this to be quite necessary for the data, I've collected the means and standard deviations for all of the feature columns in the data and placed them in dicts here. It's what allows this model to have a postive R^2 but it also increases the overfitting. ","metadata":{}},{"cell_type":"code","source":"means = {'feature_00': 0.640198826789856, 'feature_01': 0.03755598142743111, 'feature_02': 0.6368075609207153, 'feature_03': 0.6365063786506653, 'feature_04': 0.013741530478000641, 'feature_05': -0.02173694409430027, 'feature_06': -0.006415014620870352, 'feature_07': -0.010971736162900925, 'feature_08': -0.04653771221637726, 'feature_09': 32.596106194690265, 'feature_10': 4.95929203539823, 'feature_11': 167.6541592920354, 'feature_12': -0.13415881991386414, 'feature_13': -0.07573335617780685, 'feature_14': -0.12015637010335922, 'feature_15': -0.7470195889472961, 'feature_16': -0.6257441639900208, 'feature_17': -0.7294047474861145, 'feature_18': -0.042215555906295776, 'feature_19': -0.08798160403966904, 'feature_20': -0.15741558372974396, 'feature_21': 0.10528526455163956, 'feature_22': 0.018054703250527382, 'feature_23': 0.03165541961789131, 'feature_24': 2.733017921447754, 'feature_25': 0.39958420395851135, 'feature_26': -0.11045943945646286, 'feature_27': -0.5332594513893127, 'feature_28': -0.4522790312767029, 'feature_29': -0.5739678144454956, 'feature_30': -0.7905704975128174, 'feature_31': 0.10600688308477402, 'feature_32': 0.40044134855270386, 'feature_33': -0.021725023165345192, 'feature_34': 0.4226262867450714, 'feature_35': 0.42143046855926514, 'feature_36': -0.00023802756913937628, 'feature_37': 0.027961043640971184, 'feature_38': 0.010258913040161133, 'feature_39': 0.005768273025751114, 'feature_40': 0.017485467717051506, 'feature_41': 0.038347117602825165, 'feature_42': -0.06123563274741173, 'feature_43': -0.11644423753023148, 'feature_44': -0.12342483550310135, 'feature_45': -0.028769943863153458, 'feature_46': -0.015200662426650524, 'feature_47': 0.015717582777142525, 'feature_48': -0.0033910537604242563, 'feature_49': -0.0052393232472240925, 'feature_50': -0.2285808026790619, 'feature_51': -0.3548349440097809, 'feature_52': -0.358092725276947, 'feature_53': 0.2607136368751526, 'feature_54': 0.18796788156032562, 'feature_55': 0.3154229521751404, 'feature_56': -0.1471923440694809, 'feature_57': 0.15730056166648865, 'feature_58': -0.021774644032120705, 'feature_59': -0.0037768862675875425, 'feature_60': -0.010220836848020554, 'feature_61': -0.03178725391626358, 'feature_62': -0.3769100308418274, 'feature_63': -0.3229374587535858, 'feature_64': -0.3718394339084625, 'feature_65': -0.10233989357948303, 'feature_66': -0.13688170909881592, 'feature_67': -0.14402112364768982, 'feature_68': -0.06875362992286682, 'feature_69': -0.11862917989492416, 'feature_70': -0.11789549142122269, 'feature_71': -0.06013699993491173, 'feature_72': -0.10766122490167618, 'feature_73': -0.09921672940254211, 'feature_74': -0.10233042389154434, 'feature_75': -0.05991339311003685, 'feature_76': -0.06349952518939972, 'feature_77': -0.07424316555261612, 'feature_78': -0.07759837061166763}\nstds = {'feature_00': 1.027751088142395, 'feature_01': 1.0967519283294678, 'feature_02': 1.0156300067901611, 'feature_03': 1.0170334577560425, 'feature_04': 1.0726385116577148, 'feature_05': 0.9639211297035217, 'feature_06': 1.0963259935379028, 'feature_07': 1.0789952278137207, 'feature_08': 0.7962697148323059, 'feature_09': 23.72976726545254, 'feature_10': 3.1867162933797224, 'feature_11': 163.44513161352285, 'feature_12': 0.6700984835624695, 'feature_13': 0.5805172920227051, 'feature_14': 0.664044201374054, 'feature_15': 0.37517768144607544, 'feature_16': 0.3393096327781677, 'feature_17': 0.3603287935256958, 'feature_18': 0.9911752939224243, 'feature_19': 1.0550744533538818, 'feature_20': 0.6643751263618469, 'feature_21': 0.38239365816116333, 'feature_22': 0.950261116027832, 'feature_23': 0.8119344711303711, 'feature_24': 1.4362775087356567, 'feature_25': 1.0947270393371582, 'feature_26': 1.077124834060669, 'feature_27': 1.0645726919174194, 'feature_28': 1.0676648616790771, 'feature_29': 0.2640742361545563, 'feature_30': 0.19689509272575378, 'feature_31': 0.3815343976020813, 'feature_32': 1.2996565103530884, 'feature_33': 0.9989405870437622, 'feature_34': 1.3409572839736938, 'feature_35': 1.3365675210952759, 'feature_36': 0.8695492148399353, 'feature_37': 0.7334080934524536, 'feature_38': 0.698810338973999, 'feature_39': 0.7965824604034424, 'feature_40': 0.518515944480896, 'feature_41': 0.6384949088096619, 'feature_42': 0.8168442249298096, 'feature_43': 0.5228385925292969, 'feature_44': 0.6521403193473816, 'feature_45': 0.8666537404060364, 'feature_46': 0.9039222002029419, 'feature_47': 3.2711963653564453, 'feature_48': 0.6570901274681091, 'feature_49': 0.7083076238632202, 'feature_50': 1.0132617950439453, 'feature_51': 0.6081287860870361, 'feature_52': 0.9250587224960327, 'feature_53': 1.0421689748764038, 'feature_54': 0.5859629511833191, 'feature_55': 0.9191848039627075, 'feature_56': 0.9549097418785095, 'feature_57': 1.0204777717590332, 'feature_58': 0.8327276110649109, 'feature_59': 0.8309783339500427, 'feature_60': 0.8389413356781006, 'feature_61': 1.192766547203064, 'feature_62': 1.388945460319519, 'feature_63': 0.09957146644592285, 'feature_64': 0.3396177291870117, 'feature_65': 1.01683509349823, 'feature_66': 1.0824761390686035, 'feature_67': 0.642227828502655, 'feature_68': 0.5312599539756775, 'feature_69': 0.6208390593528748, 'feature_70': 0.6724499464035034, 'feature_71': 0.5356909036636353, 'feature_72': 0.6534596681594849, 'feature_73': 1.0855497121810913, 'feature_74': 1.0880277156829834, 'feature_75': 1.2321789264678955, 'feature_76': 1.2345560789108276, 'feature_77': 1.0921478271484375, 'feature_78': 1.0924347639083862}\ndef normalize_dataframe(df: pl.DataFrame, means: dict, stds: dict) -> pl.DataFrame:\n    \"\"\"\n    Normalize a Polars DataFrame using the provided means and standard deviations.\n\n    Args:\n    df (pl.DataFrame): The input DataFrame to normalize\n    means (dict): A dictionary of column means\n    stds (dict): A dictionary of column standard deviations\n\n    Returns:\n    pl.DataFrame: The normalized DataFrame\n    \"\"\"\n\n    # Create a list to store our normalization expressions\n    normalize_exprs = []\n\n    for col in df.columns:\n        if col in means and col in stds:\n            # Ensure we don't divide by zero\n            if stds[col] != 0:\n                normalize_exprs.append(\n                    ((pl.col(col) - means[col]) / stds[col]).alias(col)\n                )\n            else:\n                # If std is 0, just subtract the mean\n                normalize_exprs.append(\n                    (pl.col(col) - means[col]).alias(col)\n                )\n        else:\n            # If we don't have mean/std for this column, leave it as is\n            normalize_exprs.append(pl.col(col))\n\n    # Apply the normalization to the dataframe\n    normalized_df = df.select(normalize_exprs)\n\n    return normalized_df","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:18.414096Z","iopub.execute_input":"2024-10-27T03:40:18.414432Z","iopub.status.idle":"2024-10-27T03:40:18.441541Z","shell.execute_reply.started":"2024-10-27T03:40:18.414399Z","shell.execute_reply":"2024-10-27T03:40:18.440511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparing the Training Data","metadata":{}},{"cell_type":"markdown","source":"Date to to start the training data from, data collection gotten from https://www.kaggle.com/code/chumajin/janestreet-updated-simulator-for-time-series-api","metadata":{}},{"cell_type":"code","source":"valid_from = 1455 # for private you should change to 1455 (1 year)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:18.442821Z","iopub.execute_input":"2024-10-27T03:40:18.443200Z","iopub.status.idle":"2024-10-27T03:40:18.456710Z","shell.execute_reply.started":"2024-10-27T03:40:18.443148Z","shell.execute_reply":"2024-10-27T03:40:18.455781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"alltraindata = pl.scan_parquet(f\"{jane_street_real_time_market_data_forecasting_path}/train.parquet\")\ntrain = alltraindata.filter(pl.col(\"date_id\")>=valid_from).collect()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:18.457733Z","iopub.execute_input":"2024-10-27T03:40:18.458051Z","iopub.status.idle":"2024-10-27T03:40:26.883011Z","shell.execute_reply.started":"2024-10-27T03:40:18.458017Z","shell.execute_reply":"2024-10-27T03:40:26.882171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_names = [f\"feature_{i:02d}\" for i in range(79)]","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:26.884127Z","iopub.execute_input":"2024-10-27T03:40:26.884427Z","iopub.status.idle":"2024-10-27T03:40:26.889407Z","shell.execute_reply.started":"2024-10-27T03:40:26.884395Z","shell.execute_reply":"2024-10-27T03:40:26.888420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features = train.select(feature_names)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:26.890854Z","iopub.execute_input":"2024-10-27T03:40:26.891144Z","iopub.status.idle":"2024-10-27T03:40:26.919715Z","shell.execute_reply.started":"2024-10-27T03:40:26.891112Z","shell.execute_reply":"2024-10-27T03:40:26.918675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features = train_features.fill_null(strategy='forward').fill_null(0)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:26.921029Z","iopub.execute_input":"2024-10-27T03:40:26.921357Z","iopub.status.idle":"2024-10-27T03:40:29.152403Z","shell.execute_reply.started":"2024-10-27T03:40:26.921302Z","shell.execute_reply":"2024-10-27T03:40:29.151544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features = normalize_dataframe(train_features,means,stds)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:29.153588Z","iopub.execute_input":"2024-10-27T03:40:29.153962Z","iopub.status.idle":"2024-10-27T03:40:30.892132Z","shell.execute_reply.started":"2024-10-27T03:40:29.153926Z","shell.execute_reply":"2024-10-27T03:40:30.891117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X  = train_features.to_numpy()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:30.893383Z","iopub.execute_input":"2024-10-27T03:40:30.893794Z","iopub.status.idle":"2024-10-27T03:40:34.118852Z","shell.execute_reply.started":"2024-10-27T03:40:30.893760Z","shell.execute_reply":"2024-10-27T03:40:34.117846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train.select('responder_6').to_numpy().reshape(-1)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:34.123696Z","iopub.execute_input":"2024-10-27T03:40:34.124085Z","iopub.status.idle":"2024-10-27T03:40:34.161204Z","shell.execute_reply.started":"2024-10-27T03:40:34.124042Z","shell.execute_reply":"2024-10-27T03:40:34.160163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights = train.select('weight').to_numpy().reshape(-1)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:34.162759Z","iopub.execute_input":"2024-10-27T03:40:34.163206Z","iopub.status.idle":"2024-10-27T03:40:34.197413Z","shell.execute_reply.started":"2024-10-27T03:40:34.163157Z","shell.execute_reply":"2024-10-27T03:40:34.196331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Validation Test Split","metadata":{}},{"cell_type":"markdown","source":"Using 80% of the data for training 10% for Validation and early stopping, 10% for final Testing","metadata":{}},{"cell_type":"code","source":"n_test = int(len(X) * .2)\ntrain_X,t_X = X[:-n_test], X[-n_test:]\ntrain_y ,t_y = y[:-n_test], y[-n_test:]\ntrain_weights, t_weights = weights[:-n_test], weights[-n_test:]\nval_n = int(len(t_y) * .5)\nval_X,test_X = t_X[:-val_n], t_X[-val_n:]\nval_y, test_y = t_y[:-val_n], t_y[-val_n:]\nval_weights, test_weights = t_weights[:-val_n], t_weights[-val_n:]","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:34.198862Z","iopub.execute_input":"2024-10-27T03:40:34.199583Z","iopub.status.idle":"2024-10-27T03:40:34.208026Z","shell.execute_reply.started":"2024-10-27T03:40:34.199532Z","shell.execute_reply":"2024-10-27T03:40:34.207241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:34.209691Z","iopub.execute_input":"2024-10-27T03:40:34.210073Z","iopub.status.idle":"2024-10-27T03:40:34.260556Z","shell.execute_reply.started":"2024-10-27T03:40:34.210029Z","shell.execute_reply":"2024-10-27T03:40:34.259429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X = torch.tensor(train_X,dtype = torch.float32).to(device)\n\n\ntrain_y = torch.tensor(train_y, dtype = torch.float32).to(device)\nval_X = torch.tensor(val_X,dtype = torch.float32).to(device)\n\nval_y = torch.tensor(val_y,dtype = torch.float32).to(device)\ntest_X = torch.tensor(test_X,dtype = torch.float32).to(device)\n\ntest_y = torch.tensor(test_y,dtype = torch.float32).to(device)\ntrain_weights = torch.tensor(train_weights, dtype=torch.float32).to(device)\nval_weights = torch.tensor(val_weights, dtype=torch.float32).to(device)\ntest_weights = torch.tensor(test_weights, dtype=torch.float32).to(device)\ntrain_X.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:34.262158Z","iopub.execute_input":"2024-10-27T03:40:34.262627Z","iopub.status.idle":"2024-10-27T03:40:36.669782Z","shell.execute_reply.started":"2024-10-27T03:40:34.262578Z","shell.execute_reply":"2024-10-27T03:40:36.668785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\n\ndef print_model_size(model):\n    \"\"\"Prints the size of a PyTorch model in memory.\"\"\"\n    param_size = 0\n    for param in model.parameters():\n        param_size += param.nelement() * param.element_size()\n    buffer_size = 0\n    for buffer in model.buffers():\n        buffer_size += buffer.nelement() * buffer.element_size()\n    size_all_mb = (param_size + buffer_size) / 1024**2\n    print('Model size: {:.3f} MB'.format(size_all_mb))\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:36.671609Z","iopub.execute_input":"2024-10-27T03:40:36.672215Z","iopub.status.idle":"2024-10-27T03:40:36.678995Z","shell.execute_reply.started":"2024-10-27T03:40:36.672166Z","shell.execute_reply":"2024-10-27T03:40:36.678038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import TensorDataset, DataLoader\n\ntrain_dataset = TensorDataset(train_X, train_y, train_weights)\nval_dataset = TensorDataset(val_X, val_y, val_weights)\ntest_dataset = TensorDataset(test_X, test_y, test_weights)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:36.680213Z","iopub.execute_input":"2024-10-27T03:40:36.680682Z","iopub.status.idle":"2024-10-27T03:40:36.688621Z","shell.execute_reply.started":"2024-10-27T03:40:36.680638Z","shell.execute_reply":"2024-10-27T03:40:36.687780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 4096\ntrain_loader  = DataLoader(train_dataset, batch_size=batch_size, shuffle=False)\nval_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)\ntest_loader = DataLoader(test_dataset, batch_size = batch_size,shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:36.689788Z","iopub.execute_input":"2024-10-27T03:40:36.690097Z","iopub.status.idle":"2024-10-27T03:40:36.699708Z","shell.execute_reply.started":"2024-10-27T03:40:36.690064Z","shell.execute_reply":"2024-10-27T03:40:36.698786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"run_type = os.environ.get('KAGGLE_KERNEL_RUN_TYPE', 'Local')","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:36.700851Z","iopub.execute_input":"2024-10-27T03:40:36.701162Z","iopub.status.idle":"2024-10-27T03:40:36.709437Z","shell.execute_reply.started":"2024-10-27T03:40:36.701129Z","shell.execute_reply":"2024-10-27T03:40:36.708634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LSTM(input_size=79, hidden_dim=512, output_size = 1, num_layers = 1).to(device)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:36.710850Z","iopub.execute_input":"2024-10-27T03:40:36.711146Z","iopub.status.idle":"2024-10-27T03:40:36.824301Z","shell.execute_reply.started":"2024-10-27T03:40:36.711114Z","shell.execute_reply":"2024-10-27T03:40:36.823435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# weighted r2 gotten from https://www.kaggle.com/code/chumajin/janestreet-updated-simulator-for-time-series-api\ndef r2_score(y_true, y_pred, weights):\n    \"\"\"\n    Calculate the sample weighted zero-mean R-squared score.\n\n    Parameters:\n    y_true (numpy.ndarray): Ground-truth values for responder_6.\n    y_pred (numpy.ndarray): Predicted values for responder_6.\n    weights (numpy.ndarray): Sample weight vector.\n\n    Returns:\n    float: The weighted zero-mean R-squared score.\n    \"\"\"\n    numerator = np.sum(weights * (y_true - y_pred)**2)\n    denominator = np.sum(weights * y_true**2)\n\n    r2_score = 1 - numerator / denominator\n    return r2_score","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:36.825483Z","iopub.execute_input":"2024-10-27T03:40:36.825883Z","iopub.status.idle":"2024-10-27T03:40:36.831562Z","shell.execute_reply.started":"2024-10-27T03:40:36.825840Z","shell.execute_reply":"2024-10-27T03:40:36.830534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Loop","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\ndef train_model(model, loader, optimizer, loss_function, device):\n    model.train()\n    total_loss = 0\n    all_probs = []\n    all_targets = []\n    all_weights = []\n    for X_batch, y_batch, weights_batch in loader:\n        X_batch, y_batch = X_batch.to(device), y_batch.to(device)\n        weights_batch = weights_batch.to(device)\n        optimizer.zero_grad()\n\n\n        outputs = model(X_batch)\n\n        loss_per_sample = loss_function(outputs, y_batch)\n        weighted_loss = loss_per_sample * weights_batch\n        loss = weighted_loss.mean()\n\n        loss.backward()\n\n        optimizer.step()\n\n        total_loss += loss.item()\n\n\n        all_probs.append(outputs.detach().cpu())\n        all_targets.append(y_batch.cpu())\n\n        all_weights.append(weights_batch.cpu())\n\n    all_probs = torch.cat(all_probs).numpy()\n    all_targets = torch.cat(all_targets).numpy()\n    all_weights = torch.cat(all_weights).numpy()\n    mse = mean_squared_error(all_targets, all_probs,sample_weight = all_weights)\n    r2 = r2_score(all_targets, all_probs,all_weights)  # average parameter depends on your task\n\n    avg_loss = total_loss / len(loader)\n    return avg_loss, mse, r2","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:36.832724Z","iopub.execute_input":"2024-10-27T03:40:36.833023Z","iopub.status.idle":"2024-10-27T03:40:36.844993Z","shell.execute_reply.started":"2024-10-27T03:40:36.832990Z","shell.execute_reply":"2024-10-27T03:40:36.844144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate_model(model, loader):\n    model.eval()\n    all_probs = []\n    all_targets = []\n    all_weights = []\n    with torch.no_grad():\n        for X_batch, y_batch, weights_batch  in loader:\n\n            outputs = model(X_batch)\n\n            \n            all_probs.append(outputs.cpu())\n            all_targets.append(y_batch.cpu())\n            all_weights.append(weights_batch.cpu())\n    all_probs = torch.cat(all_probs).numpy()\n    all_targets = torch.cat(all_targets).numpy()\n    all_weights = torch.cat(all_weights).numpy()\n    mse = mean_squared_error(all_targets, all_probs,sample_weight = all_weights)\n    r2 = r2_score(all_targets, all_probs,all_weights)\n\n    return mse, r2\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:36.846100Z","iopub.execute_input":"2024-10-27T03:40:36.846499Z","iopub.status.idle":"2024-10-27T03:40:36.854669Z","shell.execute_reply.started":"2024-10-27T03:40:36.846453Z","shell.execute_reply":"2024-10-27T03:40:36.853767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Very Basic Training on 20 epochs, with early stopping over 10","metadata":{}},{"cell_type":"markdown","source":"Im using very few epochs due to the overfitting, plus the dataset is large enough more epochs seem to be detremintal with this type of model. The learning rate is set to .001, I am also using a weighted MSE Loss","metadata":{}},{"cell_type":"code","source":"if run_type == 'Interactive':\n    optimizer = optim.Adam(model.parameters(), lr=1e-4)\n    loss_function = nn.MSELoss(reduction='none')\n    epochs = 20\n    best = float('-inf')\n    degraded = 0\n    best_model = model\n    for epoch in range(epochs):\n        train_loss, train_mse, train_r2 = train_model(model, train_loader, optimizer, loss_function, device)\n        val_mse,val_r2 = evaluate_model(model, val_loader)\n        print(f'epoch {epoch } train loss {train_loss:.4f}, train_r2 {train_r2:.4f}, train_mse {train_mse:.4f},val_mse {val_mse:.4f}, val_r2 { val_r2:.4f}')\n        if val_r2 > best:\n            best = val_r2\n            best_model = copy.deepcopy(model)\n            torch.save(best_model.state_dict(), 'torchlstm.pth')\n            degraded = 0\n        else:\n            degraded += 1\n        if degraded > 10:\n            break\n    model = best_model\n\n    test_mse, test_r2 = evaluate_model(model, test_loader)\n    print(f'test R2 score is {test_r2}')","metadata":{"execution":{"iopub.status.busy":"2024-10-27T03:40:36.855863Z","iopub.execute_input":"2024-10-27T03:40:36.856174Z","iopub.status.idle":"2024-10-27T04:34:20.020020Z","shell.execute_reply.started":"2024-10-27T03:40:36.856143Z","shell.execute_reply":"2024-10-27T04:34:20.018861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"markdown","source":"To use this notebook for a submission, train the model download it and upload it, replace your path to load the state dict with your uploaded input.","metadata":{}},{"cell_type":"code","source":"def predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    global model\n    global device\n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    # Load the model if it's not already loaded\n   \n    model.load_state_dict(torch.load('/kaggle/input/torchlstm/pytorch/2/1/torchlstm(1).pth', map_location=device,weights_only= True))\n    model.eval()  # Set the model to evaluation mode\n    sel_cols  = [f\"feature_{i:02d}\" for i in range(79)]\n\n    # Process the test data using Polars\n    # Select the same features used during training\n\n\n    # Ensure all required columns are present\n    missing_cols = set(sel_cols) - set(test.columns)\n    if missing_cols:\n        raise ValueError(f\"Missing columns in test data: {missing_cols}\")\n\n    # Select the features\n    test_features = test.select(sel_cols)\n\n    # **Apply forward fill and then fill remaining missing values with zero**\n    test_features = test_features.fill_null(strategy='forward').fill_null(0)\n    test_features = normalize_dataframe(test_features,means,stds)\n\n    # Convert Polars DataFrame to NumPy array\n    X_test = test_features.to_numpy()\n\n    # Convert to Torch tensor\n    X_test_tensor = torch.tensor(X_test, dtype=torch.float32).to(device)\n\n    # Make predictions\n    with torch.no_grad():\n        \n        outputs = model(X_test_tensor)\n        # Assuming the model outputs a tensor of shape (batch_size, 1)\n        predictions = outputs.squeeze().cpu().numpy()\n\n    # Prepare the predictions DataFrame\n    predictions_df = pl.DataFrame({\n        'row_id': test['row_id'],\n        'responder_6': predictions\n    })\n\n    # The predict function must return a DataFrame\n    assert isinstance(predictions_df, (pl.DataFrame, pd.DataFrame))\n    # with columns 'row_id', 'responder_6'\n    assert predictions_df.columns == ['row_id', 'responder_6']\n    # and as many rows as the test data.\n    assert len(predictions_df) == len(test)\n\n    return predictions_df","metadata":{"execution":{"iopub.status.busy":"2024-10-27T04:34:20.020781Z","iopub.status.idle":"2024-10-27T04:34:20.021141Z","shell.execute_reply.started":"2024-10-27T04:34:20.020961Z","shell.execute_reply":"2024-10-27T04:34:20.020980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport polars as pl\n\nimport kaggle_evaluation.jane_street_inference_server\n\n\ninference_server = \\\nkaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T04:34:20.022127Z","iopub.status.idle":"2024-10-27T04:34:20.022528Z","shell.execute_reply.started":"2024-10-27T04:34:20.022288Z","shell.execute_reply":"2024-10-27T04:34:20.022306Z"},"trusted":true},"execution_count":null,"outputs":[]}]}