{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport polars as pl\nfrom sklearn.ensemble import GradientBoostingRegressor\nimport kaggle_evaluation.jane_street_inference_server\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport numpy as np\nimport statistics as stat\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import roc_auc_score\nimport copy\nfrom torch.utils.data import TensorDataset, DataLoader\nfrom sklearn.model_selection import train_test_split\nimport torch.optim as optim","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:03.411109Z","iopub.execute_input":"2024-11-01T15:07:03.412115Z","iopub.status.idle":"2024-11-01T15:07:09.917671Z","shell.execute_reply.started":"2024-11-01T15:07:03.412067Z","shell.execute_reply":"2024-11-01T15:07:09.916499Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"jane_street_real_time_market_data_forecasting_path = '/kaggle/input/jane-street-real-time-market-data-forecasting'","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:09.920140Z","iopub.execute_input":"2024-11-01T15:07:09.920660Z","iopub.status.idle":"2024-11-01T15:07:09.926024Z","shell.execute_reply.started":"2024-11-01T15:07:09.920618Z","shell.execute_reply":"2024-11-01T15:07:09.924774Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class GaussianNoise(nn.Module):\n    def __init__(self, std=0.1):\n        super().__init__()\n        self.std = std\n\n    def forward(self, x):\n        if self.training:  # Only add noise during training\n            noise = torch.randn_like(x) * self.std\n            return x + noise\n        return x","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:09.927651Z","iopub.execute_input":"2024-11-01T15:07:09.928033Z","iopub.status.idle":"2024-11-01T15:07:09.939689Z","shell.execute_reply.started":"2024-11-01T15:07:09.927991Z","shell.execute_reply":"2024-11-01T15:07:09.938406Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class LSTM(nn.Module):\n    def __init__(self, input_size, hidden_dim, output_size, num_layers):\n        super(LSTM, self).__init__()\n        self.hidden_dim = hidden_dim\n        self.num_layers = num_layers\n        self.noise = GaussianNoise(std = .1)\n        self.lstm = nn.LSTM(input_size, hidden_dim, num_layers, batch_first=True)\n        self.fc = nn.Linear(hidden_dim, output_size)\n    def forward(self, x):\n        x = x.unsqueeze(1)\n        x = self.noise(x)\n        h0 = torch.zeros(self.num_layers, x.size(0), self.hidden_dim).to(x.device)\n        c0 = torch.zeros(self.num_layers, x.size(0), self.hidden_dim).to(x.device)\n        out, _ = self.lstm(x, (h0, c0))\n        out = self.fc(out[:, -1, :])\n        return out.squeeze()","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:09.941583Z","iopub.execute_input":"2024-11-01T15:07:09.942044Z","iopub.status.idle":"2024-11-01T15:07:09.954785Z","shell.execute_reply.started":"2024-11-01T15:07:09.941990Z","shell.execute_reply":"2024-11-01T15:07:09.953472Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"valid_from = 1455","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:09.958692Z","iopub.execute_input":"2024-11-01T15:07:09.959597Z","iopub.status.idle":"2024-11-01T15:07:09.965306Z","shell.execute_reply.started":"2024-11-01T15:07:09.959540Z","shell.execute_reply":"2024-11-01T15:07:09.964147Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"alltraindata = pl.scan_parquet(f\"{jane_street_real_time_market_data_forecasting_path}/train.parquet\")\ntrain = alltraindata.filter(pl.col(\"date_id\")>=valid_from).collect()","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:09.966761Z","iopub.execute_input":"2024-11-01T15:07:09.967220Z","iopub.status.idle":"2024-11-01T15:07:20.063904Z","shell.execute_reply.started":"2024-11-01T15:07:09.967169Z","shell.execute_reply":"2024-11-01T15:07:20.062872Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_names = [f\"feature_{i:02d}\" for i in range(79)]","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:20.065261Z","iopub.execute_input":"2024-11-01T15:07:20.065636Z","iopub.status.idle":"2024-11-01T15:07:20.070972Z","shell.execute_reply.started":"2024-11-01T15:07:20.065596Z","shell.execute_reply":"2024-11-01T15:07:20.069786Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_features = train.select(feature_names)","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:20.072457Z","iopub.execute_input":"2024-11-01T15:07:20.072840Z","iopub.status.idle":"2024-11-01T15:07:20.089625Z","shell.execute_reply.started":"2024-11-01T15:07:20.072800Z","shell.execute_reply":"2024-11-01T15:07:20.088500Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_features = train_features.fill_null(strategy='forward').fill_null(0)","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:20.091385Z","iopub.execute_input":"2024-11-01T15:07:20.091787Z","iopub.status.idle":"2024-11-01T15:07:23.052940Z","shell.execute_reply.started":"2024-11-01T15:07:20.091747Z","shell.execute_reply":"2024-11-01T15:07:23.051908Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X  = train_features.to_numpy()","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:23.054315Z","iopub.execute_input":"2024-11-01T15:07:23.054693Z","iopub.status.idle":"2024-11-01T15:07:24.507091Z","shell.execute_reply.started":"2024-11-01T15:07:23.054652Z","shell.execute_reply":"2024-11-01T15:07:24.505754Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = train.select('responder_6').to_numpy().reshape(-1)","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:24.508576Z","iopub.execute_input":"2024-11-01T15:07:24.508908Z","iopub.status.idle":"2024-11-01T15:07:24.551862Z","shell.execute_reply.started":"2024-11-01T15:07:24.508870Z","shell.execute_reply":"2024-11-01T15:07:24.550630Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"weights = train.select('weight').to_numpy().reshape(-1)","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:24.553383Z","iopub.execute_input":"2024-11-01T15:07:24.553738Z","iopub.status.idle":"2024-11-01T15:07:24.595428Z","shell.execute_reply.started":"2024-11-01T15:07:24.553700Z","shell.execute_reply":"2024-11-01T15:07:24.594297Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Training","metadata":{}},{"cell_type":"code","source":"n_test = int(len(X) * .2)\ntrain_X,t_X = X[:-n_test], X[-n_test:]\ntrain_y ,t_y = y[:-n_test], y[-n_test:]\ntrain_weights, t_weights = weights[:-n_test], weights[-n_test:]\nval_n = int(len(t_y) * .5)\nval_X,test_X = t_X[:-val_n], t_X[-val_n:]\nval_y, test_y = t_y[:-val_n], t_y[-val_n:]\nval_weights, test_weights = t_weights[:-val_n], t_weights[-val_n:]","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:24.596846Z","iopub.execute_input":"2024-11-01T15:07:24.597269Z","iopub.status.idle":"2024-11-01T15:07:24.619939Z","shell.execute_reply.started":"2024-11-01T15:07:24.597226Z","shell.execute_reply":"2024-11-01T15:07:24.618788Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:24.624499Z","iopub.execute_input":"2024-11-01T15:07:24.625033Z","iopub.status.idle":"2024-11-01T15:07:24.630501Z","shell.execute_reply.started":"2024-11-01T15:07:24.624983Z","shell.execute_reply":"2024-11-01T15:07:24.629295Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_X = torch.tensor(train_X,dtype = torch.float32).to(device)\n\n\ntrain_y = torch.tensor(train_y, dtype = torch.float32).to(device)\nval_X = torch.tensor(val_X,dtype = torch.float32).to(device)\n\nval_y = torch.tensor(val_y,dtype = torch.float32).to(device)\ntest_X = torch.tensor(test_X,dtype = torch.float32).to(device)\n\ntest_y = torch.tensor(test_y,dtype = torch.float32).to(device)\ntrain_weights = torch.tensor(train_weights, dtype=torch.float32).to(device)\nval_weights = torch.tensor(val_weights, dtype=torch.float32).to(device)\ntest_weights = torch.tensor(test_weights, dtype=torch.float32).to(device)\ntrain_X.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-01T15:07:24.631872Z","iopub.execute_input":"2024-11-01T15:07:24.632301Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def print_model_size(model):\n    \"\"\"Prints the size of a PyTorch model in memory.\"\"\"\n    param_size = 0\n    for param in model.parameters():\n        param_size += param.nelement() * param.element_size()\n    buffer_size = 0\n    for buffer in model.buffers():\n        buffer_size += buffer.nelement() * buffer.element_size()\n    size_all_mb = (param_size + buffer_size) / 1024**2\n    print('Model size: {:.3f} MB'.format(size_all_mb))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = TensorDataset(train_X, train_y, train_weights)\nval_dataset = TensorDataset(val_X, val_y, val_weights)\ntest_dataset = TensorDataset(test_X, test_y, test_weights)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size = 4096\ntrain_loader  = DataLoader(train_dataset, batch_size=batch_size, shuffle=False)\nval_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)\ntest_loader = DataLoader(test_dataset, batch_size = batch_size,shuffle=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"run_type = os.environ.get('KAGGLE_KERNEL_RUN_TYPE', 'Local')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = LSTM(input_size=79, hidden_dim=512, output_size = 1, num_layers = 1).to(device)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# weighted r2 gotten from https://www.kaggle.com/code/chumajin/janestreet-updated-simulator-for-time-series-api\ndef r2_score(y_true, y_pred, weights):\n    \"\"\"\n    Calculate the sample weighted zero-mean R-squared score.\n\n    Parameters:\n    y_true (numpy.ndarray): Ground-truth values for responder_6.\n    y_pred (numpy.ndarray): Predicted values for responder_6.\n    weights (numpy.ndarray): Sample weight vector.\n\n    Returns:\n    float: The weighted zero-mean R-squared score.\n    \"\"\"\n    numerator = np.sum(weights * (y_true - y_pred)**2)\n    denominator = np.sum(weights * y_true**2)\n\n    r2_score = 1 - numerator / denominator\n    return r2_score","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Training Loop","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\ndef train_model(model, loader, optimizer, loss_function, device):\n    model.train()\n    total_loss = 0\n    all_probs = []\n    all_targets = []\n    all_weights = []\n    for X_batch, y_batch, weights_batch in loader:\n        X_batch, y_batch = X_batch.to(device), y_batch.to(device)\n        weights_batch = weights_batch.to(device)\n        optimizer.zero_grad()\n\n\n        outputs = model(X_batch)\n\n        loss_per_sample = loss_function(outputs, y_batch)\n        weighted_loss = loss_per_sample * weights_batch\n        loss = weighted_loss.mean()\n\n        loss.backward()\n\n        optimizer.step()\n\n        total_loss += loss.item()\n\n\n        all_probs.append(outputs.detach().cpu())\n        all_targets.append(y_batch.cpu())\n\n        all_weights.append(weights_batch.cpu())\n\n    all_probs = torch.cat(all_probs).numpy()\n    all_targets = torch.cat(all_targets).numpy()\n    all_weights = torch.cat(all_weights).numpy()\n    mse = mean_squared_error(all_targets, all_probs,sample_weight = all_weights)\n    r2 = r2_score(all_targets, all_probs,all_weights)  # average parameter depends on your task\n\n    avg_loss = total_loss / len(loader)\n    return avg_loss, mse, r2","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate_model(model, loader):\n    model.eval()\n    all_probs = []\n    all_targets = []\n    all_weights = []\n    with torch.no_grad():\n        for X_batch, y_batch, weights_batch  in loader:\n\n            outputs = model(X_batch)\n\n            \n            all_probs.append(outputs.cpu())\n            all_targets.append(y_batch.cpu())\n            all_weights.append(weights_batch.cpu())\n    all_probs = torch.cat(all_probs).numpy()\n    all_targets = torch.cat(all_targets).numpy()\n    all_weights = torch.cat(all_weights).numpy()\n    mse = mean_squared_error(all_targets, all_probs,sample_weight = all_weights)\n    r2 = r2_score(all_targets, all_probs,all_weights)\n\n    return mse, r2","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if run_type == 'Interactive':\n    optimizer = optim.Adam(model.parameters(), lr=1e-4)\n    loss_function = nn.MSELoss(reduction='none')\n    epochs = 20\n    best = float('-inf')\n    degraded = 0\n    best_model = model\n    for epoch in range(epochs):\n        train_loss, train_mse, train_r2 = train_model(model, train_loader, optimizer, loss_function, device)\n        val_mse,val_r2 = evaluate_model(model, val_loader)\n        print(f'epoch {epoch } train loss {train_loss:.4f}, train_r2 {train_r2:.4f}, train_mse {train_mse:.4f},val_mse {val_mse:.4f}, val_r2 { val_r2:.4f}')\n        if val_r2 > best:\n            best = val_r2\n            best_model = copy.deepcopy(model)\n            torch.save(best_model.state_dict(), 'torchlstm.pth')\n            degraded = 0\n        else:\n            degraded += 1\n        if degraded > 10:\n            break\n    model = best_model\n\n    test_mse, test_r2 = evaluate_model(model, test_loader)\n    print(f'test R2 score is {test_r2}')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}