{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport polars as pl\nfrom typing import Dict, List\n\nimport torch\nfrom torch import nn\nfrom torch.utils.data import TensorDataset, DataLoader\nimport torch.nn.functional as F\nfrom torchmetrics.regression import R2Score\n\nfrom timeit import default_timer as timer\nfrom tqdm.auto import tqdm\n\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\n\nimport os\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:29.741938Z","iopub.execute_input":"2024-11-17T10:09:29.742804Z","iopub.status.idle":"2024-11-17T10:09:29.748923Z","shell.execute_reply.started":"2024-11-17T10:09:29.742761Z","shell.execute_reply":"2024-11-17T10:09:29.748029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PATH = '/kaggle/input/jane-street-real-time-market-data-forecasting'\n\n# Set mode\nTRAIN_MODE = False\nNUM_SAMPLE = 300\nTIME_STEP = 5\n\n# Set Model`s hyperparameters\nHIDDEN = 512\nLAYER = 1\nDROPOUT = 0.2\n\n# Set training loop configurations \nBATCH_SIZE = 32\nNUM_EPOCHS = 30\n\n# device agnostic code\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\nprint(f\"current device: {device }\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:29.753940Z","iopub.execute_input":"2024-11-17T10:09:29.754340Z","iopub.status.idle":"2024-11-17T10:09:29.772366Z","shell.execute_reply.started":"2024-11-17T10:09:29.754297Z","shell.execute_reply":"2024-11-17T10:09:29.771519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Paths and constants\ninput_path = '/kaggle/input/jane-street-real-time-market-data-forecasting'\ndef select_train_set(path, num_data_set):\n    # choose the number of training set \n    selected_set = [f\"partition_id={i}/part-0.parquet\" for i in range(num_data_set)]\n    # Load and filter the data from only the selected Parquet files\n    list_df = []\n    for file_name in selected_set:\n        data_path = f'{path}/train.parquet/{file_name}'\n        lazy_df = pl.scan_parquet(data_path)\n        df = lazy_df.collect()\n        list_df.append(df)\n\n    # Concatenate all dataframes into a single dataframe\n    if len(list_df) >= 2:\n        full_df = pl.concat(list_df)\n    else:\n        full_df = list_df[0]\n\n    return full_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:29.774273Z","iopub.execute_input":"2024-11-17T10:09:29.774859Z","iopub.status.idle":"2024-11-17T10:09:29.784721Z","shell.execute_reply.started":"2024-11-17T10:09:29.774816Z","shell.execute_reply":"2024-11-17T10:09:29.783977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_df = select_train_set(PATH, 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:29.785801Z","iopub.execute_input":"2024-11-17T10:09:29.786106Z","iopub.status.idle":"2024-11-17T10:09:30.231614Z","shell.execute_reply.started":"2024-11-17T10:09:29.786025Z","shell.execute_reply":"2024-11-17T10:09:30.230813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def Mode_selection(data, train_mode = None, num_sample = None):\n    if train_mode == True:\n        data = data.fill_null(strategy='forward').fill_null(0)\n        df = data.to_pandas()\n        df = df.sample(n=num_sample, random_state=42)\n    else:\n        data = data.fill_null(strategy='forward').fill_null(0)\n        df = data.to_pandas()\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:30.234189Z","iopub.execute_input":"2024-11-17T10:09:30.234502Z","iopub.status.idle":"2024-11-17T10:09:30.240223Z","shell.execute_reply.started":"2024-11-17T10:09:30.234469Z","shell.execute_reply":"2024-11-17T10:09:30.239352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_df = Mode_selection(selected_df, train_mode=TRAIN_MODE, num_sample=NUM_SAMPLE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:30.241420Z","iopub.execute_input":"2024-11-17T10:09:30.241808Z","iopub.status.idle":"2024-11-17T10:09:31.020336Z","shell.execute_reply.started":"2024-11-17T10:09:30.241766Z","shell.execute_reply":"2024-11-17T10:09:31.019518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = selected_df.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.021441Z","iopub.execute_input":"2024-11-17T10:09:31.021757Z","iopub.status.idle":"2024-11-17T10:09:31.026413Z","shell.execute_reply.started":"2024-11-17T10:09:31.021724Z","shell.execute_reply":"2024-11-17T10:09:31.025414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"drop_features = ['time_id',\t'date_id','symbol_id','responder_0', 'responder_1', 'responder_2', 'responder_3',\n       'responder_4', 'responder_5', 'responder_7','responder_8']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.027596Z","iopub.execute_input":"2024-11-17T10:09:31.027940Z","iopub.status.idle":"2024-11-17T10:09:31.035569Z","shell.execute_reply.started":"2024-11-17T10:09:31.027898Z","shell.execute_reply":"2024-11-17T10:09:31.034804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.drop(drop_features, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.036583Z","iopub.execute_input":"2024-11-17T10:09:31.036944Z","iopub.status.idle":"2024-11-17T10:09:31.045074Z","shell.execute_reply.started":"2024-11-17T10:09:31.036912Z","shell.execute_reply":"2024-11-17T10:09:31.044345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(df.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.046383Z","iopub.execute_input":"2024-11-17T10:09:31.046918Z","iopub.status.idle":"2024-11-17T10:09:31.055600Z","shell.execute_reply.started":"2024-11-17T10:09:31.046875Z","shell.execute_reply":"2024-11-17T10:09:31.054708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"null_cols =[]\n\nfor col in df.columns:\n    if df[col].isnull().sum() == 1944210:\n        null_cols.append(col)\n        df.drop(col, axis =1, inplace=True)\n\nprint(f\"Dropped columns: {null_cols}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.059289Z","iopub.execute_input":"2024-11-17T10:09:31.059580Z","iopub.status.idle":"2024-11-17T10:09:31.075401Z","shell.execute_reply.started":"2024-11-17T10:09:31.059549Z","shell.execute_reply":"2024-11-17T10:09:31.074505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.076421Z","iopub.execute_input":"2024-11-17T10:09:31.076704Z","iopub.status.idle":"2024-11-17T10:09:31.105733Z","shell.execute_reply.started":"2024-11-17T10:09:31.076673Z","shell.execute_reply":"2024-11-17T10:09:31.104785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_list = list(df.columns)  \nfeature_list.remove('responder_6') ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.106841Z","iopub.execute_input":"2024-11-17T10:09:31.107154Z","iopub.status.idle":"2024-11-17T10:09:31.113583Z","shell.execute_reply.started":"2024-11-17T10:09:31.107117Z","shell.execute_reply":"2024-11-17T10:09:31.112779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = df[feature_list].to_numpy()\ntarget = df['responder_6'].to_numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.114846Z","iopub.execute_input":"2024-11-17T10:09:31.115217Z","iopub.status.idle":"2024-11-17T10:09:31.124088Z","shell.execute_reply.started":"2024-11-17T10:09:31.115185Z","shell.execute_reply":"2024-11-17T10:09:31.123292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_sequences(data, target, time_steps: int):\n    \"\"\"\n    Create sequences from data and targets for LSTM training.\n\n    Args:\n        data (numpy.ndarray): Input features of shape (num_samples, num_features).\n        target (numpy.ndarray): Target values of shape (num_samples,).\n        time_steps (int): The sequence length (number of time steps for each input).\n\n    Returns:\n        tuple: A tuple containing:\n            - sequences (numpy.ndarray): Shape (num_sequences, time_steps, num_features).\n            - targets (numpy.ndarray): Shape (num_sequences,).\n    \"\"\"\n    if time_steps < 1:\n        raise ValueError(\"time_steps must be >= 1.\")\n    if len(data) <= time_steps:\n        raise ValueError(\"Data length must be greater than time_steps.\")\n\n    X, y = [], []\n    for i in range(len(data) - time_steps):\n        X.append(data[i:i + time_steps])  # Create sequences of length `time_steps`\n        y.append(target[i + time_steps])  # Align target with the end of the sequence\n\n    return np.array(X), np.array(y)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.125409Z","iopub.execute_input":"2024-11-17T10:09:31.125684Z","iopub.status.idle":"2024-11-17T10:09:31.134544Z","shell.execute_reply.started":"2024-11-17T10:09:31.125654Z","shell.execute_reply":"2024-11-17T10:09:31.133812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X, y = create_sequences(features, target, TIME_STEP)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.145330Z","iopub.execute_input":"2024-11-17T10:09:31.145646Z","iopub.status.idle":"2024-11-17T10:09:31.154592Z","shell.execute_reply.started":"2024-11-17T10:09:31.145575Z","shell.execute_reply":"2024-11-17T10:09:31.153755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split training and test set\ntrain_X, test_X = X[:int(len(X)*0.8)], X[int(len(X)*0.8):]\ntrain_y, test_y = y[:int(len(y)*0.8)], y[int(len(y)*0.8):]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.155661Z","iopub.execute_input":"2024-11-17T10:09:31.155937Z","iopub.status.idle":"2024-11-17T10:09:31.163219Z","shell.execute_reply.started":"2024-11-17T10:09:31.155901Z","shell.execute_reply":"2024-11-17T10:09:31.162402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert data to PyTorch tensors\ntrain_X, train_y = torch.tensor(train_X, dtype=torch.float32, device=device), torch.tensor(train_y, dtype=torch.float32, device=device)\ntest_X, test_y = torch.tensor(test_X, dtype=torch.float32, device=device), torch.tensor(test_y, dtype=torch.float32, device=device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.164354Z","iopub.execute_input":"2024-11-17T10:09:31.164628Z","iopub.status.idle":"2024-11-17T10:09:31.173520Z","shell.execute_reply.started":"2024-11-17T10:09:31.164598Z","shell.execute_reply":"2024-11-17T10:09:31.172770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_X.shape, test_X.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.174557Z","iopub.execute_input":"2024-11-17T10:09:31.174823Z","iopub.status.idle":"2024-11-17T10:09:31.183120Z","shell.execute_reply.started":"2024-11-17T10:09:31.174793Z","shell.execute_reply":"2024-11-17T10:09:31.182387Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Prepare DataLoader","metadata":{}},{"cell_type":"code","source":"# Turn datasets into iterable(batches)\ntrain_dataset = TensorDataset(train_X, train_y)\ntest_dataset = TensorDataset(test_X, test_y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.184172Z","iopub.execute_input":"2024-11-17T10:09:31.184541Z","iopub.status.idle":"2024-11-17T10:09:31.191246Z","shell.execute_reply.started":"2024-11-17T10:09:31.184489Z","shell.execute_reply":"2024-11-17T10:09:31.190373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataloader = DataLoader(train_dataset,\n                             batch_size=BATCH_SIZE,\n                             shuffle=False)\n\ntest_dataloader = DataLoader(test_dataset,\n                             batch_size=BATCH_SIZE,\n                             shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.192553Z","iopub.execute_input":"2024-11-17T10:09:31.192955Z","iopub.status.idle":"2024-11-17T10:09:31.203256Z","shell.execute_reply.started":"2024-11-17T10:09:31.192914Z","shell.execute_reply":"2024-11-17T10:09:31.202452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check out information of dataloaders\nprint(f\"Length of train_dataloader: {len(train_dataloader)} Batch Size: {BATCH_SIZE}\")\nprint(f\"Length of test_dataloader: {len(test_dataloader)} Batch Size: {BATCH_SIZE}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.204270Z","iopub.execute_input":"2024-11-17T10:09:31.204538Z","iopub.status.idle":"2024-11-17T10:09:31.213909Z","shell.execute_reply.started":"2024-11-17T10:09:31.204502Z","shell.execute_reply":"2024-11-17T10:09:31.212994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check out what`s inside the training dataloader\ntrain_feature_batch, train_target_batch = next(iter(train_dataloader))\n\ntrain_feature_batch.shape, train_target_batch.shape  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.214951Z","iopub.execute_input":"2024-11-17T10:09:31.215233Z","iopub.status.idle":"2024-11-17T10:09:31.226872Z","shell.execute_reply.started":"2024-11-17T10:09:31.215202Z","shell.execute_reply":"2024-11-17T10:09:31.225997Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Build a baseline LSTM Model","metadata":{}},{"cell_type":"code","source":"class LSTM_BASELINE(nn.Module):\n\n    def __init__(self, input_shape: int, hidden_units: int, output_shape: int, num_layer: int, dropout_rate : float):\n        super().__init__()\n        self.lstm = nn.LSTM(input_size=input_shape, \n                            hidden_size=hidden_units, \n                            num_layers=num_layer, \n                            batch_first=True\n                            )\n\n        \n        self.layer_norm = nn.LayerNorm(hidden_units)\n        \n        self.linear = nn.Linear(hidden_units, 1)\n\n    def forward(self, x): \n        x, _ = self.lstm(x)# Using '_' when the subsequent code doesn't require the hidden or cell states, focusing only on the output of the LSTM layer.\n        x = x[:, -1, :]\n        x = self.layer_norm(x)\n        x = F.relu(x)  # Apply ReLU activation function in an out-of-place manner\n        x = self.linear(x)\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.227952Z","iopub.execute_input":"2024-11-17T10:09:31.228619Z","iopub.status.idle":"2024-11-17T10:09:31.235805Z","shell.execute_reply.started":"2024-11-17T10:09:31.228585Z","shell.execute_reply":"2024-11-17T10:09:31.234956Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Functionizing training, testing loops, evaluation and prediction\n* training loop - train_step()\n* testing loop - test_step()\n* eval function - eval_model()\n* prediction function - make_predictions()s()","metadata":{}},{"cell_type":"code","source":"# Training loop\n\ndef train_step(model: torch.nn.Module, \n               dataloader,\n               loss_fn: torch.nn.Module,\n               optimizer: torch.optim.Optimizer,\n               score_fn,\n               device: torch.device = device):\n    train_loss = 0\n    score_fn.reset()  # Reset metric for a new epoch\n    \n    # Put model into training mode\n    model.train()\n\n    # Add a Loop to loop through the training batches\n    for batch, (X, y) in enumerate(dataloader):\n        #Put data on target device\n        X, y = X.to(device), y.to(device)\n\n        # 1. Forward pass \n        y_pred = model(X).squeeze(-1)\n\n        # 2. Calculate MSE loss and R2 score (per batch)\n        loss = loss_fn(y_pred, y)\n        score = score_fn(y_pred, y)\n        \n        train_loss += loss.item() # accumulate train loss\n        score_fn.update(y_pred, y)  # Update R2 score metric\n        \n        # 3. Optimizer zero grad\n        optimizer.zero_grad()\n\n        # 4. Loss backward\n        loss.backward()\n\n        # 5. Optimizer step(update the model`s parameters once per batch)\n        optimizer.step()\n    \n  \n    # Divide total train loss by length of train dataloader\n    train_loss /= len(dataloader)\n    # Compute the final R2 score for the epoch\n    train_score = score_fn.compute().item()\n\n\n    return train_loss, train_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.236731Z","iopub.execute_input":"2024-11-17T10:09:31.238428Z","iopub.status.idle":"2024-11-17T10:09:31.246332Z","shell.execute_reply.started":"2024-11-17T10:09:31.238385Z","shell.execute_reply":"2024-11-17T10:09:31.245468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def test_step(model: torch.nn.Module,\n             dataloader,\n             loss_fn: torch.nn.Module,\n             score_fn,\n             device: torch.device = device):\n    test_loss = 0\n    score_fn.reset()  # Reset metric for a new evaluation phase\n    # Put the model in eval mode\n    model.eval()\n   \n    # Turn on inference mode context manager\n    with torch.inference_mode():\n        for X, y in dataloader:\n            #Send the data to the target device\n            X, y = X.to(device), y.to(device)\n\n            # 1. Forward pass\n            test_pred = model(X).squeeze(-1)\n\n            # 2. Calculate MSE loss and R2 score\n        \n            loss = loss_fn(test_pred, y)\n            test_loss  += loss.item() # accumulate train loss\n            \n            score_fn.update(test_pred, y)  # Update metric for this batch\n\n    # Adjust metrics\n    test_loss /= len(dataloader)\n    test_score = score_fn.compute().item()\n\n    return test_loss, test_score\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.247463Z","iopub.execute_input":"2024-11-17T10:09:31.247736Z","iopub.status.idle":"2024-11-17T10:09:31.259747Z","shell.execute_reply.started":"2024-11-17T10:09:31.247706Z","shell.execute_reply":"2024-11-17T10:09:31.258898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def eval_model(model: torch.nn.Module,\n              dataloader,\n              loss_fn: torch.nn.Module,\n              score_fn,\n              device: torch.device = device):\n    eval_loss = 0\n    score_fn.reset()  # Reset metric for evaluation\n\n    model.eval()\n    with torch.inference_mode():\n        for X, y in tqdm(dataloader):\n            X, y = X.to(device), y.to(device)\n\n            y_pred = model(X).squeeze(-1)\n\n            # 2. Calculate MSE loss and R2 score\n        \n            loss = loss_fn(y_pred, y)\n        \n            eval_loss  += loss.item() # accumulate train loss\n            \n            # Update score function\n            score_fn.update(y_pred, y)\n\n\n        eval_loss /= len(dataloader)\n        eval_score = score_fn.compute().item()  # Compute accumulated metric\n\n    return {\"model_name\" : model.__class__.__name__,\n            \"model_loss\": eval_loss,\n            \"model_score\": eval_score}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.260775Z","iopub.execute_input":"2024-11-17T10:09:31.261094Z","iopub.status.idle":"2024-11-17T10:09:31.272401Z","shell.execute_reply.started":"2024-11-17T10:09:31.261054Z","shell.execute_reply":"2024-11-17T10:09:31.271581Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Creating a train() function to combine train_step() and test_step()","metadata":{}},{"cell_type":"code","source":"def train(model: torch.nn.Module,\n          train_dataloader: torch.utils.data.DataLoader,\n          test_dataloader: torch.utils.data.DataLoader,\n          optimizer: torch.optim.Optimizer,\n          loss_fn: torch.nn.Module,\n          score_fn,\n          epochs: int,\n          device: torch.device):\n\n    # Initialize results dictionary\n    results = {\n        \"train_loss\": [],\n        \"train_score\": [],\n        \"test_loss\": [],\n        \"test_score\": [],\n    }\n\n    # Loop through epochs\n    for epoch in tqdm(range(epochs), desc=\"Training Epochs\", leave=True):\n        \n        # Reset score metric\n        score_fn.reset()\n        # Perform training step\n        train_loss, train_score = train_step(\n            model=model,\n            dataloader=train_dataloader,\n            loss_fn=loss_fn,\n            score_fn=score_fn,\n            optimizer=optimizer,\n            device=device\n        )\n        \n        # Perform testing step\n        test_loss, test_score = test_step(\n            model=model,\n            dataloader=test_dataloader,\n            loss_fn=loss_fn,\n            score_fn=score_fn,\n            device=device\n        )\n\n        # Log epoch results\n        # Print out what's happening every 10 epochs\n        \n        if epoch % 10 == 0:\n            \n            print(\n                  f\"Epoch {epoch+1}/{epochs}: \"\n                  f\"Train Loss: {train_loss:.4f} | \"\n                  f\"Train Score: {train_score:.4f} | \"\n                  f\"Test Loss: {test_loss:.4f} | \"\n                  f\"Test Score: {test_score:.4f}\\n\"\n                  f\"{'-'*25*4}\"\n                  )\n\n\n        # Append metrics to results dictionary\n        results[\"train_loss\"].append(\n            train_loss.item() if isinstance(train_loss, torch.Tensor) else train_loss\n        )\n        results[\"train_score\"].append(\n            train_score.item() if isinstance(train_score, torch.Tensor) else train_score\n        )\n        results[\"test_loss\"].append(\n            test_loss.item() if isinstance(test_loss, torch.Tensor) else test_loss\n        )\n        results[\"test_score\"].append(\n            test_score.item() if isinstance(test_score, torch.Tensor) else test_score\n        )\n\n    # Return results dictionary\n    return results","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.278111Z","iopub.execute_input":"2024-11-17T10:09:31.278396Z","iopub.status.idle":"2024-11-17T10:09:31.288965Z","shell.execute_reply.started":"2024-11-17T10:09:31.278365Z","shell.execute_reply":"2024-11-17T10:09:31.288088Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training and Evaluating model","metadata":{}},{"cell_type":"code","source":"# Set random seeds\ntorch.manual_seed(42) \ntorch.cuda.manual_seed(42)\n\n# Recreate an instance of TinyVGG\nlstm_base_model = LSTM_BASELINE(input_shape=len(df.columns) -1, \n                                hidden_units=HIDDEN, \n                                output_shape=1,\n                                num_layer=LAYER,\n                                dropout_rate=DROPOUT\n                                ).to(device)\n# Setup loss function and optimizer\nloss_fn = nn.MSELoss()\nr2score = R2Score().to(device)\noptimizer = torch.optim.SGD(params=lstm_base_model.parameters(),lr=0.01)\n\n# Start the timer\nstart_time = timer()\n\n# Train baseline model\nlstm_base_results = train(model=lstm_base_model, \n                        train_dataloader=train_dataloader,\n                        test_dataloader=test_dataloader,\n                        optimizer=optimizer,\n                        loss_fn=loss_fn,\n                        score_fn=r2score,\n                        epochs=NUM_EPOCHS,\n                        device=device)\n\n# End the timer and print out how long it took\nend_time = timer()\nprint(f\"Total training time: {end_time-start_time:.3f} seconds\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:31.301582Z","iopub.execute_input":"2024-11-17T10:09:31.302176Z","iopub.status.idle":"2024-11-17T10:09:32.525825Z","shell.execute_reply.started":"2024-11-17T10:09:31.302135Z","shell.execute_reply":"2024-11-17T10:09:32.524878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_loss_curves(results: Dict[str, List[float]]):\n    \"\"\"\n    Plots training curves of a results dictionary.\n    \"\"\"\n\n    # Get the loss values of the results dictionary (training and test)\n    loss = results['train_loss']\n    test_loss = results['test_loss']\n\n    # Get the score values of the results dictionary (training and test)\n    score = results['train_score']\n    test_score = results['test_score']\n\n    # Figure out how many epochs there were\n    epochs = range(len(results['train_loss']))\n\n    # Setup a plot\n    plt.figure(figsize=(15, 7))\n\n    # Plot loss\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs, loss, label='train_loss')\n    plt.plot(epochs, test_loss, label='test_loss')\n    plt.title('MSE Loss')\n    plt.xlabel('Epochs')\n    plt.legend()\n\n    # Plot score\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs, score, label='train_score')\n    plt.plot(epochs, test_score, label='test_score')\n    plt.title('R2 score')\n    plt.xlabel('Epochs')\n    plt.legend();","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:32.527024Z","iopub.execute_input":"2024-11-17T10:09:32.527360Z","iopub.status.idle":"2024-11-17T10:09:32.535664Z","shell.execute_reply.started":"2024-11-17T10:09:32.527325Z","shell.execute_reply":"2024-11-17T10:09:32.534667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_loss_curves(lstm_base_results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:32.536860Z","iopub.execute_input":"2024-11-17T10:09:32.537165Z","iopub.status.idle":"2024-11-17T10:09:33.116748Z","shell.execute_reply.started":"2024-11-17T10:09:32.537134Z","shell.execute_reply":"2024-11-17T10:09:33.115782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get model`s result of evaluation\nmodel_result = eval_model(model=lstm_base_model,\n                          dataloader=test_dataloader,\n                          loss_fn=loss_fn,\n                          score_fn=r2score,\n                          device=device)\nmodel_result","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:33.118226Z","iopub.execute_input":"2024-11-17T10:09:33.119157Z","iopub.status.idle":"2024-11-17T10:09:33.149125Z","shell.execute_reply.started":"2024-11-17T10:09:33.119109Z","shell.execute_reply":"2024-11-17T10:09:33.148220Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Make predictions","metadata":{}},{"cell_type":"code","source":"#test = pl.scan_parquet(f'{PATH}/test.parquet/date_id=0/part-0.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:33.150290Z","iopub.execute_input":"2024-11-17T10:09:33.150584Z","iopub.status.idle":"2024-11-17T10:09:33.154776Z","shell.execute_reply.started":"2024-11-17T10:09:33.150551Z","shell.execute_reply":"2024-11-17T10:09:33.153780Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def ready_test_set(data):\n    data = data.collect()\n    data = data.fill_null(strategy='forward').fill_null(0)\n    data = data.to_pandas()\n    row_id = data['row_id']\n    not_cols = []\n    for i in list(data.columns):\n        if i not in feature_list:\n            not_cols.append(i)\n    \n    test_df = data.drop(not_cols, axis=1)\n    test_np = test_df.to_numpy()\n    \n    test_tensor =torch.tensor(test_np, dtype=torch.float32, device=device)\n\n    return test_tensor, row_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:33.155896Z","iopub.execute_input":"2024-11-17T10:09:33.156219Z","iopub.status.idle":"2024-11-17T10:09:33.166242Z","shell.execute_reply.started":"2024-11-17T10:09:33.156169Z","shell.execute_reply":"2024-11-17T10:09:33.165381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def make_predictions(model: torch.nn.Module,\n                     data,\n                     row_id,\n                     device: torch.device = device):\n    model.to(device)\n    model.eval()\n\n    with torch.inference_mode():\n        preds = model(data).squeeze().cpu().numpy()\n        pred_df = pd.DataFrame({\"row_id\": row_id, \"responder_6\": preds})\n\n    return pred_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:33.167385Z","iopub.execute_input":"2024-11-17T10:09:33.167706Z","iopub.status.idle":"2024-11-17T10:09:33.175501Z","shell.execute_reply.started":"2024-11-17T10:09:33.167657Z","shell.execute_reply":"2024-11-17T10:09:33.174733Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"# Global lags storage\nlags_: pl.DataFrame | None = None\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame:\n    global lags_, model_loaded # Declare models as global\n    \n    # Logic for saving or loading lags\n    if lags is not None:\n        lags_ = lags\n\n    \n    \n\n    test = pl.scan_parquet(f'{PATH}/test.parquet/date_id=0/part-0.parquet')\n    \n    test_tensor, row_id = ready_test_set(test)\n\n    test_tensor = test_tensor.unsqueeze(1) \n    \n    pred_df = make_predictions(model=lstm_base_model,\n                               data=test_tensor,\n                               row_id=row_id)\n\n        \n    return pl.from_pandas(pred_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:33.176611Z","iopub.execute_input":"2024-11-17T10:09:33.177226Z","iopub.status.idle":"2024-11-17T10:09:33.188236Z","shell.execute_reply.started":"2024-11-17T10:09:33.177179Z","shell.execute_reply":"2024-11-17T10:09:33.187422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\n# Running the inference server\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway((\n        '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n        '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n    ))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-17T10:09:33.189275Z","iopub.execute_input":"2024-11-17T10:09:33.189631Z","iopub.status.idle":"2024-11-17T10:09:33.231148Z","shell.execute_reply.started":"2024-11-17T10:09:33.189589Z","shell.execute_reply":"2024-11-17T10:09:33.230250Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}