{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":10363594,"sourceType":"datasetVersion","datasetId":6418722},{"sourceId":225506,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":186696,"modelId":208801}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Prep","metadata":{}},{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport pandas as pd\nimport numpy as np\nimport pickle\nimport polars as pl\nfrom torch.utils.data import DataLoader, Dataset\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import mean_squared_error, r2_score\n\n\nCONFIG = {\n    \"seq_length\": 1,  # Length of input sequences for the model\n    \"batch_size\": 256,  # Number of samples per gradient update\n    \"learning_rate\": 0.0001,  # Learning rate for the optimizer\n    \"num_epochs\": 7,  # Number of training epochs\n    \"num_workers\": 4,  # Number of subprocesses to use for data loading\n    \"num_heads\": 8, \n    \"num_hash_buckets\": 64,\n    \"pin_memory\": True,  # Whether to pin memory for faster data transfer to GPU\n    \"prefetch_factor\": 16,  # Number of batches to prefetch\n    \"dropout_rate\": 0.2,  # Dropout rate for regularization\n    \"input_channels\": 80,  # Number of input features\n    \"output_features\": 1,  # Number of output features after CNN\n    \"lstm_hidden_size\": 128,  # Hidden size for LSTM\n    \"embedding_dim\": 16,  # Dimension of the embedding for symbol_id\n    \"weight_decay\": 0.0001, \n    \"num_symbols\": 39,  # Number of unique symbol_ids (0 to 38)\n    \"feature_columns\": [f'feature_{i:02d}' for i in range(79)] + ['responder_6_lag_1'],  # List of feature column names\n    \"target_column\": \"responder_6\",  # Name of the target variable\n    \"data_path\": \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/\",  # Path to the dataset\n    \"dataset_save_path\": \"/kaggle/working/processed_dataset.parquet\",  # Path to save the processed dataset\n    \"model_save_path\": \"/kaggle/working/att_cnn_bi_lstm.pth\",  # Path to save the trained model\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T18:40:05.688208Z","iopub.execute_input":"2025-01-11T18:40:05.688590Z","iopub.status.idle":"2025-01-11T18:40:09.789103Z","shell.execute_reply.started":"2025-01-11T18:40:05.688546Z","shell.execute_reply":"2025-01-11T18:40:09.788417Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"class AddNorm(nn.Module):\n    def __init__(self, size, dropout=0.1):\n        super(AddNorm, self).__init__()\n        self.norm = nn.LayerNorm(size)\n        self.dropout = nn.Dropout(dropout)\n\n    def forward(self, x, residual):\n        return self.norm(x + self.dropout(residual))\n\nclass GatedResidualNetwork(nn.Module):\n    def __init__(self, input_size):\n        super(GatedResidualNetwork, self).__init__()\n        self.linear1 = nn.Linear(input_size, input_size)\n        self.linear2 = nn.Linear(input_size, input_size)\n        self.activation = nn.ReLU()\n\n    def forward(self, x):\n        gate = torch.sigmoid(self.linear2(x))  # Gating mechanism\n        residual = self.linear1(x)  # Linear transformation\n        return gate * residual + x  # Residual connection\n\nclass AttCNNBiLSTM(nn.Module):\n    def __init__(self, input_channels, lstm_hidden_size, output_features, num_hash_buckets, embedding_dim, num_heads, dropout_rate=0.2):\n        super(AttCNNBiLSTM, self).__init__()\n        \n        # Embedding layer for symbol_id using hashing trick\n        self.num_hash_buckets = num_hash_buckets\n        self.embedding = nn.Embedding(num_hash_buckets, embedding_dim)  \n        \n        # Convolutional layers\n        self.conv1 = nn.Conv1d(in_channels=input_channels + embedding_dim, out_channels=128, kernel_size=3, padding=1)\n        self.bn1 = nn.BatchNorm1d(128)\n        #self.pool1 = nn.MaxPool1d(kernel_size=2)\n        self.dropout1 = nn.Dropout(dropout_rate)\n\n        self.conv2 = nn.Conv1d(in_channels=128, out_channels=64, kernel_size=3, padding=1)\n        self.bn2 = nn.BatchNorm1d(64)\n        #self.pool2 = nn.MaxPool1d(kernel_size=2)\n        self.dropout2 = nn.Dropout(dropout_rate)\n        \n        self.conv3 = nn.Conv1d(in_channels=64, out_channels=32, kernel_size=3, padding=1)\n        self.bn3 = nn.BatchNorm1d(32)\n        #self.pool3 = nn.MaxPool1d(kernel_size=2)\n        self.dropout3 = nn.Dropout(dropout_rate)\n        \n        # Additional GRN after CNN layers\n        #self.grn_cnn_to_lstm = GatedResidualNetwork(32)  # Input size matches the output of the last CNN layer\n        #self.add_norm_cnn_to_lstm = AddNorm(32)  # Add & Norm after GRN        \n\n        # LSTM layers\n        self.lstm = nn.LSTM(input_size=32, hidden_size=lstm_hidden_size, num_layers=2, batch_first=True, bidirectional=True)\n        self.dropout_lstm = nn.Dropout(dropout_rate)\n\n        # Multi-Head Attention\n        self.multihead_attn = nn.MultiheadAttention(embed_dim=lstm_hidden_size * 2, num_heads=num_heads, dropout=0.2, add_zero_attn=False)\n\n        # Gated Residual Network after Attention\n        self.grn1 = GatedResidualNetwork(lstm_hidden_size * 2)  # Input size matches the output of the attention layer\n        self.add_norm_attn = AddNorm(lstm_hidden_size * 2)  # Add & Norm after GRN\n\n        # Fully connected layer for prediction\n        #self.fc1 = nn.Linear(lstm_hidden_size * 2, 16)  # First fully connected layer\n        #self.fc2 = nn.Linear(64, 16)  # Second fully connected layer\n        self.fc_output = nn.Linear(lstm_hidden_size * 2, output_features)  # Final output layer\n\n    def forward(self, x, symbol_ids, return_features=False):\n        # Hashing trick to map symbol_ids to indices\n        hashed_indices = torch.fmod(symbol_ids, self.num_hash_buckets)  # Hashing trick\n        \n        # Embedding for hashed symbol_ids\n        embedded_symbols = self.embedding(hashed_indices)  # Shape: (batch_size, seq_length, embedding_dim)\n        embedded_symbols = embedded_symbols.permute(0, 2, 1)  # Shape: (batch_size, embedding_dim, seq_length)\n\n        # Concatenate embeddings with features\n        x = torch.cat((x, embedded_symbols), dim=1)  # Shape: (batch_size, input_channels + embedding_dim, seq_length)\n\n        # Apply Convolutional Layers\n        x = F.relu(self.bn1(self.conv1(x)))\n        #x = self.pool1(x)\n        x = self.dropout1(x)\n\n        x = F.relu(self.bn2(self.conv2(x)))\n        #x = self.pool2(x)\n        x = self.dropout2(x)\n        \n        x = F.relu(self.bn3(self.conv3(x)))\n        #x = self.pool3(x)\n        x = self.dropout3(x)\n        \n        # Apply Gated Residual Network\n        #x = self.grn_cnn_to_lstm(x.permute(0, 2, 1))  # Shape: (batch_size, seq_length, num_channels)\n        #x = x.permute(0, 2, 1)  # Shape: (batch_size, num_channels, seq_length)\n        \n        # Prepare for LSTM\n        x = x.permute(0, 2, 1)  # Shape: (batch_size, seq_length, num_channels)\n        \n        # Apply Add & Norm\n        #x = self.add_norm_cnn_to_lstm(x, x)  # Use the same x for residual connection\n\n        # Apply LSTM\n        lstm_out, _ = self.lstm(x)\n        lstm_out = self.dropout_lstm(lstm_out)\n\n        # Multi-Head Attention\n        attn_out, _ = self.multihead_attn(lstm_out, lstm_out, lstm_out)\n\n        # Apply Gated Residual Network and Add & Norm\n        grn_out = self.grn1(attn_out.mean(dim=1))  # Shape: (batch_size, 128)\n        grn_out = self.add_norm_attn(grn_out, attn_out.mean(dim=1))  # Apply Add & Norm\n\n        if return_features:\n            return grn_out\n\n        # Fully connected layers\n        #fc_out = F.relu(self.fc1(grn_out))  # First fully connected layer\n        #fc_out = F.relu(self.fc2(fc_out))  # Second fully connected layer\n\n        # Fully connected layer for prediction\n        output = self.fc_output(grn_out)  # Final output layer\n        return output","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T20:08:33.552565Z","iopub.execute_input":"2025-01-10T20:08:33.553159Z","iopub.status.idle":"2025-01-10T20:08:33.571023Z","shell.execute_reply.started":"2025-01-10T20:08:33.553091Z","shell.execute_reply":"2025-01-10T20:08:33.569678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Helper f-tions\ndef load_min_max_and_every_nth(base_dir, partition_ids=None, step=10):\n    # (Function implementation remains unchanged)\n    # Load only the relevant rows from parquet files: min and max time_id for each symbol_id and date_id,\n    # and every nth time_id in between.                                           \n       \n    if partition_ids is None:\n        partition_ids = ['0', '1', '2', '3', '4', '5', '6', '7', '8', '9']\n    \n    lazy_frames = []\n\n    for partition_id in partition_ids:\n        partition_path = os.path.join(base_dir, f'partition_id={partition_id}')\n        if os.path.exists(partition_path):\n            for dirname, _, filenames in os.walk(partition_path):\n                for filename in filenames:\n                    if filename.endswith('.parquet'):\n                        file_path = os.path.join(dirname, filename)\n                        lazy_df = pl.scan_parquet(file_path)\n\n                        # Group by date_id and symbol_id to find min and max time_id\n                        min_max_time = (\n                            lazy_df.group_by([\"date_id\", \"symbol_id\"])\n                                  \n                            .agg([pl.min(\"time_id\").alias(\"min_time_id\"),\n                                  pl.max(\"time_id\").alias(\"max_time_id\")])\n                              \n                        )\n\n                        # Join back to the original LazyFrame to filter the relevant rows\n                        filtered_rows = (\n                            lazy_df.join(min_max_time, on=[\"date_id\", \"symbol_id\"], how=\"inner\")\n                            .filter((pl.col(\"time_id\") % step == 0))\n                        )\n\n                        # Append the processed LazyFrame to the list\n                        lazy_frames.append(filtered_rows)\n\n            print(f\"{partition_path} is done\")\n\n    # Concatenate all LazyFrames into a single LazyFrame\n    full_lazy_df = pl.concat(lazy_frames)\n\n    # Collect the final DataFrame\n    full_df = full_lazy_df.collect()\n\n    # Remove rows with NaN values\n    print(\"Removing rows with missing values...\")\n    #full_df = full_df.drop_nulls()  # Polars equivalent of dropna\n    full_df = full_df.fill_null(0)\n    if full_df.is_empty():\n        raise ValueError(\"Dataset is empty after removing missing values!\")\n\n    # Sort by symbol_id, date_id, and time_id\n    print(\"Sorting data by symbol_id, date_id, and time_id...\")\n    full_df = full_df.sort([\"date_id\", \"time_id\", \"symbol_id\"])\n    \n\n    # Drop unnecessary columns\n    full_df = full_df.drop(['min_time_id', 'max_time_id'])  # Drop the min and max time_id columns if needed\n    # Create the lagged column for responder_6\n    full_df = full_df.with_columns(\n        pl.col(\"responder_6\").shift(1).over([\"symbol_id\"]).alias(\"responder_6_lag_1\")\n    )\n    full_df = full_df.fill_null(0)\n    \n    return full_df.to_pandas()  # Convert to Pandas DataFrame before returning\n\n\nclass TimeSeriesDataset(Dataset):\n    def __init__(self, data, seq_length, feature_columns, target_column):\n        #unique_symbol_ids = data['symbol_id'].unique()  # Get unique symbol_ids\n        #sorted_symbol_ids = np.sort(unique_symbol_ids)\n        self.data = torch.tensor(data[feature_columns].values, dtype=torch.float32)\n        self.targets = torch.tensor(data[target_column].values, dtype=torch.float32)\n        self.symbol_ids = torch.tensor(data['symbol_id'].values, dtype=torch.long)  # Keep symbol_id as integers\n\n        # Ensure that the sequence length does not exceed the available data length\n        if seq_length > len(self.data):\n            raise ValueError(\"Sequence length must be less than or equal to the length of the data.\")\n\n        self.seq_length = seq_length\n\n    def __len__(self):\n        return len(self.data) - self.seq_length + 1\n\n    def __getitem__(self, idx):\n        seq = self.data[idx:idx + self.seq_length]  # Shape: (seq_length, num_features)\n        target = self.targets[idx + self.seq_length - 1]  # Shape: (1,)\n        symbol_ids = self.symbol_ids[idx:idx + self.seq_length]  # Get corresponding symbol_ids\n        # Ensure that seq and symbol_ids are of the same length\n        if len(seq) != self.seq_length or len(symbol_ids) != self.seq_length:\n            raise ValueError(f\"Sequence length mismatch: seq length {len(seq)}, symbol_ids length {len(symbol_ids)}\")\n        return seq.permute(1, 0), target.unsqueeze(0), symbol_ids  # Return (num_features, seq_length), (1,), (seq_length,)\n\ndef weighted_r2_score(y_true, y_pred, weights):\n    # Convert to numpy arrays if they are PyTorch tensors\n    if isinstance(y_true, torch.Tensor):\n        y_true = y_true.cpu().detach().numpy()\n    if isinstance(y_pred, torch.Tensor):\n        y_pred = y_pred.cpu().detach().numpy()\n    if isinstance(weights, torch.Tensor):\n        weights = weights.cpu().detach().numpy()\n\n    # Ensure all arrays are 1D\n    y_true = y_true.flatten()\n    y_pred = y_pred.flatten()\n    weights = weights.flatten()\n\n    # Calculate the weighted mean of y_true\n    weighted_mean = np.sum(weights * y_true) / np.sum(weights) if np.sum(weights) != 0 else 0\n\n    # Calculate the weighted residuals\n    residuals = y_true - y_pred\n    weighted_residual_sum = np.sum(weights * (residuals ** 2))\n    weighted_total_sum = np.sum(weights * ((y_true - weighted_mean) ** 2))\n\n    # Calculate R²\n    r2 = 1 - (weighted_residual_sum / weighted_total_sum) if weighted_total_sum != 0 else 0\n    return r2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T18:40:27.636210Z","iopub.execute_input":"2025-01-11T18:40:27.636970Z","iopub.status.idle":"2025-01-11T18:40:27.652987Z","shell.execute_reply.started":"2025-01-11T18:40:27.636936Z","shell.execute_reply":"2025-01-11T18:40:27.652181Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Workflow","metadata":{}},{"cell_type":"code","source":"# Step 1: Load Data\ndata = load_min_max_and_every_nth(CONFIG['data_path'], partition_ids =['0'], step=1)\n\n\n# Save as Parquet without the index\n#data.to_parquet(CONFIG['dataset_save_path'], index=False)  \n#print(f\"Data saved successfully to {CONFIG['dataset_save_path']}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T20:27:37.742569Z","iopub.execute_input":"2025-01-10T20:27:37.743454Z","iopub.status.idle":"2025-01-10T20:27:42.344551Z","shell.execute_reply.started":"2025-01-10T20:27:37.743407Z","shell.execute_reply":"2025-01-10T20:27:42.343521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = pd.read_parquet(\"/kaggle/input/janermf-data5/processed_dataset.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T18:40:44.014757Z","iopub.execute_input":"2025-01-11T18:40:44.015079Z","iopub.status.idle":"2025-01-11T18:41:01.865543Z","shell.execute_reply.started":"2025-01-11T18:40:44.015050Z","shell.execute_reply":"2025-01-11T18:41:01.864557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sorted_df = data.sort_values(by=[\"date_id\", \"time_id\", \"symbol_id\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T18:42:23.065764Z","iopub.execute_input":"2025-01-11T18:42:23.066570Z","iopub.status.idle":"2025-01-11T18:42:25.864013Z","shell.execute_reply.started":"2025-01-11T18:42:23.066538Z","shell.execute_reply":"2025-01-11T18:42:25.863058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_columns = CONFIG['feature_columns']\n# Select only the feature columns\ndata_features = data[feature_columns]\n# Compute min and max for each feature\nmin_values = data_features.min()\nmax_values = data_features.max()\n\n# Create a new DataFrame with feature_name, min, and max\nresult_df = pd.DataFrame({\n    'feature_name': min_values.index,  # Feature names\n    'min': min_values.values,         # Min values\n    'max': max_values.values          # Max values\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T15:13:51.756086Z","iopub.execute_input":"2025-01-08T15:13:51.756420Z","iopub.status.idle":"2025-01-08T15:13:54.241418Z","shell.execute_reply.started":"2025-01-08T15:13:51.756395Z","shell.execute_reply":"2025-01-08T15:13:54.240337Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Iterate over each row in result_df\nfor index, row in result_df.iterrows():\n    feature_name = row['feature_name']\n    min_value = row['min']\n    max_value = row['max']\n    \n    # Print or process the row data\n    print(f\"Feature: {feature_name}, Min: {min_value}, Max: {max_value}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-08T15:15:47.917392Z","iopub.execute_input":"2025-01-08T15:15:47.918359Z","iopub.status.idle":"2025-01-08T15:15:47.929186Z","shell.execute_reply.started":"2025-01-08T15:15:47.918322Z","shell.execute_reply":"2025-01-08T15:15:47.928062Z"},"collapsed":true,"jupyter":{"source_hidden":true,"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2: Prepare Dataset\ndata.fillna(data.mean(), inplace=True)\nfeature_columns = CONFIG['feature_columns']\ntarget_column = CONFIG['target_column']\nseq_length = CONFIG['seq_length']\n\n# Create dataset and dataloader\ndataset = TimeSeriesDataset(data, seq_length, feature_columns, target_column)\n#dataset = TimeSeriesDataset(sorted_df, seq_length, feature_columns, target_column)\ntrain_loader = DataLoader(dataset, batch_size=CONFIG['batch_size'], pin_memory=True, num_workers=4, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T20:27:50.614878Z","iopub.execute_input":"2025-01-10T20:27:50.616231Z","iopub.status.idle":"2025-01-10T20:27:53.253163Z","shell.execute_reply.started":"2025-01-10T20:27:50.616175Z","shell.execute_reply":"2025-01-10T20:27:53.249804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 3: Train the AttCNNBiLSTM Model\ninput_channels = CONFIG['input_channels']\noutput_features = CONFIG['output_features']\nlstm_hidden_size = CONFIG['lstm_hidden_size']\nnum_hash_buckets = CONFIG['num_hash_buckets']\nembedding_dim = CONFIG['embedding_dim']\nnum_symbols = CONFIG['num_symbols']\nnum_heads = CONFIG['num_heads']\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")  # Check if multiple GPUs are available\n\nmodel = AttCNNBiLSTM(input_channels, lstm_hidden_size, output_features, num_hash_buckets, embedding_dim, num_heads).to(device)\nif torch.cuda.device_count() > 1:\n    model = torch.nn.DataParallel(model)\n\ncriterion = torch.nn.SmoothL1Loss()\n#criterion = torch.nn.MSELoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=CONFIG['learning_rate'])\n#optimizer = torch.optim.AdamW(model.parameters(), lr=CONFIG['learning_rate'], weight_decay=CONFIG['weight_decay'])\n#optimizer = torch.optim.SGD(model.parameters(), lr=CONFIG['learning_rate'], momentum=0.9, weight_decay=CONFIG['weight_decay'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T16:54:14.695376Z","iopub.execute_input":"2025-01-10T16:54:14.696216Z","iopub.status.idle":"2025-01-10T16:54:16.096002Z","shell.execute_reply.started":"2025-01-10T16:54:14.696179Z","shell.execute_reply":"2025-01-10T16:54:16.094822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm  # Import tqdm for progress bar\n\n# Step 4: Training Loop\nnum_epochs = CONFIG['num_epochs']\n\nfor epoch in range(num_epochs):\n    model.train()\n    epoch_loss = 0\n    \n    # Get the total number of batches\n    total_batches = len(train_loader)\n    print(f'Epoch [{epoch + 1}/{num_epochs}], Total Batches: {total_batches}')\n    \n    # Use tqdm to create a progress bar for batches\n    for batch_X, batch_y, symbol_ids in tqdm(train_loader, desc=\"Processing Batches\", total=total_batches):\n        batch_X, batch_y, symbol_ids = batch_X.to(device), batch_y.to(device), symbol_ids.to(device)  # Move data to GPU\n        optimizer.zero_grad()\n        \n        outputs = model(batch_X, symbol_ids)  # Pass symbol_ids to the model\n        loss = criterion(outputs, batch_y)  # Compute loss\n        \n        # Check for NaN loss\n        if torch.isnan(loss).any():\n            print(\"Loss is NaN. Stopping training.\")\n            break\n        \n        loss.backward()  # Backward pass\n        torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)  # Gradient clipping\n        optimizer.step()  # Update weights\n        \n        epoch_loss += loss.item()\n    \n    print(f'Epoch [{epoch + 1}/{num_epochs}], Loss: {epoch_loss / len(train_loader):.4f}')\n    torch.save(model.state_dict(), CONFIG['model_save_path']) \n    #torch.save(model.state_dict(), f\"{CONFIG['model_save_path']}_epoch_{epoch + 1}.pth\")\n\n    print('Checkpoint created')\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T16:54:25.096334Z","iopub.execute_input":"2025-01-10T16:54:25.097366Z","iopub.status.idle":"2025-01-10T18:19:26.321604Z","shell.execute_reply.started":"2025-01-10T16:54:25.097329Z","shell.execute_reply":"2025-01-10T18:19:26.320337Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.save(model.state_dict(), CONFIG['model_save_path']) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T20:49:57.434107Z","iopub.execute_input":"2024-12-13T20:49:57.434492Z","iopub.status.idle":"2024-12-13T20:49:57.445470Z","shell.execute_reply.started":"2024-12-13T20:49:57.434463Z","shell.execute_reply":"2024-12-13T20:49:57.444584Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"input_channels = CONFIG['input_channels']\noutput_features = CONFIG['output_features']\nlstm_hidden_size = CONFIG['lstm_hidden_size']\nnum_hash_buckets = CONFIG['num_hash_buckets']\nembedding_dim = CONFIG['embedding_dim']\nnum_symbols = CONFIG['num_symbols']\nnum_heads = CONFIG['num_heads']\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")  # Check if multiple GPUs are available\nmodel = AttCNNBiLSTM(input_channels, lstm_hidden_size, output_features, num_hash_buckets, embedding_dim, num_heads)\n# Step 2: Load the model state from disk\nmodel_save_path = '/kaggle/input/janermf/pytorch/default/7/att_cnn_bi_lstm (20).pth'\n#model_save_path = '/kaggle/input/janermf/pytorch/default/2/att_cnn_bi_lstm (2).pth'\nstate_dict = torch.load(model_save_path, weights_only=True, map_location=torch.device('cpu'))\n\n# Step 3: Remove the \"module.\" prefix from the keys\nnew_state_dict = {k.replace(\"module.\", \"\"): v for k, v in state_dict.items()}\n\n# Step 4: Load the modified state dictionary into the model\nmodel.load_state_dict(new_state_dict)\n\n# Step 5: Set the model to evaluation mode\nmodel.eval()\nprint(model)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T20:13:17.178768Z","iopub.execute_input":"2025-01-10T20:13:17.179218Z","iopub.status.idle":"2025-01-10T20:13:17.247307Z","shell.execute_reply.started":"2025-01-10T20:13:17.179181Z","shell.execute_reply":"2025-01-10T20:13:17.245976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to evaluate the model\ndef evaluate_model(model, data_loader, device):\n    model.eval()\n    model.to(device)\n    all_predictions = []\n    all_targets = []\n    \n    with torch.no_grad():\n        for batch_X, batch_y, symbol_ids in data_loader:\n            batch_X = batch_X.to(device)\n            batch_y = batch_y.to(device)\n            symbol_ids = symbol_ids.to(device)\n            outputs = model(batch_X, symbol_ids)  # Get model predictions\n            all_predictions.append(outputs.cpu().numpy())\n            all_targets.append(batch_y.cpu().numpy())\n    \n    # Concatenate all predictions and targets\n    all_predictions = np.concatenate(all_predictions)\n    all_targets = np.concatenate(all_targets)\n\n    return all_predictions, all_targets\n\n# Evaluate the model\npredictions, targets = evaluate_model(model, train_loader, device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T20:28:02.974056Z","iopub.execute_input":"2025-01-10T20:28:02.975656Z","iopub.status.idle":"2025-01-10T20:30:50.435769Z","shell.execute_reply.started":"2025-01-10T20:28:02.975566Z","shell.execute_reply":"2025-01-10T20:30:50.433901Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate metrics\nmse = mean_squared_error(targets, predictions)\nr2 = r2_score(targets, predictions)\n\nprint(f'Mean Squared Error: {mse:.4f}')\nprint(f'R-squared: {r2:.4f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T20:31:19.977164Z","iopub.execute_input":"2025-01-10T20:31:19.977673Z","iopub.status.idle":"2025-01-10T20:31:20.020237Z","shell.execute_reply.started":"2025-01-10T20:31:19.977629Z","shell.execute_reply":"2025-01-10T20:31:20.018855Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plotting the results\nplt.figure(figsize=(20, 6))\nplt.plot(targets, label='Actual', color='blue')\nplt.plot(predictions, label='Predicted', color='red')\nplt.title('Model Predictions vs Actual Values')\nplt.xlabel('Sample Index')\nplt.ylabel('Value')\nplt.legend(loc='upper right')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T20:31:23.991605Z","iopub.execute_input":"2025-01-10T20:31:23.992002Z","iopub.status.idle":"2025-01-10T20:31:29.096677Z","shell.execute_reply.started":"2025-01-10T20:31:23.991969Z","shell.execute_reply":"2025-01-10T20:31:29.095349Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select a segment of the data starting from a specific index\nstart_index = 1055000  # Start from the middle of the data\nsegment_size = 100  # Number of samples to take from the start_index\n\n# Ensure we don't exceed the bounds of the array\nend_index = min(start_index + segment_size, len(targets))\n\n# Slice the predictions and targets for the selected segment\nsegment_predictions = predictions[start_index:end_index]\nsegment_targets = targets[start_index:end_index]\n\n# Plotting the results for the selected segment\nplt.figure(figsize=(20, 6))\nplt.plot(segment_targets, label='Actual', color='blue')\nplt.plot(segment_predictions, label='Predicted', color='red')\nplt.title('Model Predictions vs Actual Values (Segment from Middle)')\nplt.xlabel('Sample Index')\nplt.ylabel('Value')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T20:31:45.977570Z","iopub.execute_input":"2025-01-10T20:31:45.978048Z","iopub.status.idle":"2025-01-10T20:31:46.308602Z","shell.execute_reply.started":"2025-01-10T20:31:45.978008Z","shell.execute_reply":"2025-01-10T20:31:46.307408Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error, r2_score\n\n# Assuming 'data' is your DataFrame and you have added necessary lag features\nfeature_columns = CONFIG['feature_columns'] + ['symbol_id']\ntarget_column = CONFIG['target_column']\n\n#sorted_df\ndata['symbol_id'] = data['symbol_id'].astype('category')\nX = data[feature_columns]\ny = data[target_column]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T21:58:42.596869Z","iopub.execute_input":"2025-01-11T21:58:42.597499Z","iopub.status.idle":"2025-01-11T21:58:43.482170Z","shell.execute_reply.started":"2025-01-11T21:58:42.597458Z","shell.execute_reply":"2025-01-11T21:58:43.481204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the data into training and validation sets\n#X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n# Split the data into training and validation sets while maintaining order\ntrain_size = int(len(data) * 0.8)  # 80% for training\nX_train, X_val = X[:train_size], X[train_size:]\ny_train, y_val = y[:train_size], y[train_size:]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T21:58:48.300055Z","iopub.execute_input":"2025-01-11T21:58:48.300412Z","iopub.status.idle":"2025-01-11T21:58:48.312415Z","shell.execute_reply.started":"2025-01-11T21:58:48.300380Z","shell.execute_reply":"2025-01-11T21:58:48.311483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize the XGBoost regressor\nmodel_xgb = xgb.XGBRegressor(\n    objective='reg:squarederror',\n    learning_rate=0.05,\n    max_depth=11,\n    n_estimators=250,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    gamma=0.3,\n    reg_alpha=1.0,\n    reg_lambda=1.0,\n    random_state=42,\n    early_stopping_rounds=50,\n    tree_method='hist',\n    device='cuda',\n    enable_categorical=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T21:58:51.526792Z","iopub.execute_input":"2025-01-11T21:58:51.527119Z","iopub.status.idle":"2025-01-11T21:58:51.532207Z","shell.execute_reply.started":"2025-01-11T21:58:51.527090Z","shell.execute_reply":"2025-01-11T21:58:51.531297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model\nmodel_xgb.fit(\n    X_train, y_train,\n    eval_set=[(X_val, y_val)],\n    verbose=50\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T21:58:54.602114Z","iopub.execute_input":"2025-01-11T21:58:54.602978Z","iopub.status.idle":"2025-01-11T22:00:51.726961Z","shell.execute_reply.started":"2025-01-11T21:58:54.602945Z","shell.execute_reply":"2025-01-11T22:00:51.726214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on the validation set\ny_pred = model_xgb.predict(X_val)\n\n# Calculate metrics\nmse = mean_squared_error(y_val, y_pred)\nr2 = r2_score(y_val, y_pred)\n\nprint(f'Mean Squared Error: {mse:.4f}')\nprint(f'R-squared: {r2:.4f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T22:00:55.653673Z","iopub.execute_input":"2025-01-11T22:00:55.654285Z","iopub.status.idle":"2025-01-11T22:00:58.442343Z","shell.execute_reply.started":"2025-01-11T22:00:55.654232Z","shell.execute_reply":"2025-01-11T22:00:58.441513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plotting Actual vs Predicted\nplt.figure(figsize=(10, 6))\n\n# Scatter plot of actual vs predicted values\nplt.scatter(y_val, y_pred, color='blue', alpha=0.6, label='Predicted vs Actual')\n\n# Plotting the line for perfect predictions\nplt.plot([y_val.min(), y_val.max()], [y_val.min(), y_val.max()], color='red', linestyle='--', label='Perfect Prediction')\n\n# Adding labels and title\nplt.xlabel('Actual Values')\nplt.ylabel('Predicted Values')\nplt.title('Actual vs Predicted Values')\nplt.legend()\nplt.grid()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T22:01:01.191906Z","iopub.execute_input":"2025-01-11T22:01:01.192242Z","iopub.status.idle":"2025-01-11T22:01:13.639948Z","shell.execute_reply.started":"2025-01-11T22:01:01.192210Z","shell.execute_reply":"2025-01-11T22:01:13.639144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select a segment from the middle for visualization\nsegment_start = 100000  # Adjust as needed\nsegment_end = 101100   # Adjust as needed\ny_val_segment = y_val.iloc[segment_start:segment_end]\ny_pred_segment = y_pred[segment_start:segment_end]\n\n# Plotting\nplt.figure(figsize=(12, 6))\nplt.plot(y_val_segment.index, y_val_segment, color='blue', label='Actual Values', linewidth=2)\nplt.plot(y_val_segment.index, y_pred_segment, color='red', label='Predicted Values', linewidth=2)\nplt.title('Model Predictions vs Actual Values (Segment from Middle)')\nplt.xlabel('Sample Index')\nplt.ylabel('Value')\nplt.legend()\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T22:01:42.783891Z","iopub.execute_input":"2025-01-11T22:01:42.784582Z","iopub.status.idle":"2025-01-11T22:01:43.079198Z","shell.execute_reply.started":"2025-01-11T22:01:42.784548Z","shell.execute_reply":"2025-01-11T22:01:43.078325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_xgb.save_model('xgb_cmodel.json')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T21:56:38.622000Z","iopub.execute_input":"2025-01-11T21:56:38.622419Z","iopub.status.idle":"2025-01-11T21:56:39.252510Z","shell.execute_reply.started":"2025-01-11T21:56:38.622381Z","shell.execute_reply":"2025-01-11T21:56:39.251491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-11T21:58:25.483248Z","iopub.execute_input":"2025-01-11T21:58:25.483932Z","iopub.status.idle":"2025-01-11T21:58:25.631725Z","shell.execute_reply.started":"2025-01-11T21:58:25.483898Z","shell.execute_reply":"2025-01-11T21:58:25.630865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate_model(model, data_loader, device):\n    model.eval()\n    model.to(device)\n    all_predictions = {}\n    all_targets = {}\n    \n    with torch.no_grad():\n        for batch_X, batch_y, symbol_ids in data_loader:\n            batch_X = batch_X.to(device)\n            batch_y = batch_y.to(device)\n            symbol_ids = symbol_ids.to(device)\n            outputs = model(batch_X, symbol_ids)  # Get model predictions\n            \n            # Convert outputs and targets to numpy\n            outputs_np = outputs.cpu().numpy()\n            targets_np = batch_y.cpu().numpy()\n            symbol_ids_np = symbol_ids.cpu().numpy()\n            \n            # Store predictions and targets in the dictionary\n            for i in range(len(symbol_ids_np)):\n                symbol_id = symbol_ids_np[i]\n                if symbol_id not in all_predictions:\n                    all_predictions[symbol_id] = []\n                    all_targets[symbol_id] = []\n                all_predictions[symbol_id].append(outputs_np[i])\n                all_targets[symbol_id].append(targets_np[i])\n    \n    # Convert lists to numpy arrays\n    for symbol_id in all_predictions:\n        all_predictions[symbol_id] = np.array(all_predictions[symbol_id])\n        all_targets[symbol_id] = np.array(all_targets[symbol_id])\n\n    return all_predictions, all_targets\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:25:29.368150Z","iopub.execute_input":"2024-12-20T10:25:29.368573Z","iopub.status.idle":"2024-12-20T10:25:29.377875Z","shell.execute_reply.started":"2024-12-20T10:25:29.368540Z","shell.execute_reply":"2024-12-20T10:25:29.375037Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate the model\npredictions_by_symbol, targets_by_symbol = evaluate_model(model, train_loader, device)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T10:26:13.474765Z","iopub.execute_input":"2024-12-20T10:26:13.475496Z","iopub.status.idle":"2024-12-20T10:26:14.088121Z","shell.execute_reply.started":"2024-12-20T10:26:13.475461Z","shell.execute_reply":"2024-12-20T10:26:14.086785Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null}]}