{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11228175,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, Dataset\nfrom torch.nn.utils.rnn import pad_sequence, pack_padded_sequence, pad_packed_sequence\nimport csv\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-11T08:58:03.549930Z","iopub.execute_input":"2025-03-11T08:58:03.550380Z","iopub.status.idle":"2025-03-11T08:58:08.719381Z","shell.execute_reply.started":"2025-03-11T08:58:03.550347Z","shell.execute_reply":"2025-03-11T08:58:08.718607Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the dataset path\ndata_path = \"/kaggle/input/stanford-rna-3d-folding\"\n\n# Load datasets\ntrain_sequences = pd.read_csv(data_path + \"/train_sequences.csv\")\ntrain_labels = pd.read_csv(data_path + \"/train_labels.csv\")\nvalidation_sequences = pd.read_csv(data_path + \"/validation_sequences.csv\")\nvalidation_labels = pd.read_csv(data_path + \"/validation_labels.csv\")\ntest_sequences = pd.read_csv(data_path + \"/test_sequences.csv\")\n\nprint(\"Train Sequences:\")\nprint(train_sequences.head())\n\nprint(\"\\nTrain Labels:\")\nprint(train_labels.head())\n\nprint(\"\\nValidation Sequences:\")\nprint(validation_sequences.head())\n\nprint(\"\\nValidation Labels:\")\nprint(validation_labels.head())\n\nprint(\"\\nTest Sequences:\")\nprint(test_sequences.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-11T08:59:23.810825Z","iopub.execute_input":"2025-03-11T08:59:23.811154Z","iopub.status.idle":"2025-03-11T08:59:24.266981Z","shell.execute_reply.started":"2025-03-11T08:59:23.811131Z","shell.execute_reply":"2025-03-11T08:59:24.266195Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Map nucleotides to integers (A=0, C=1, G=2, U=3)\nnucleotide_map = {'A': 0, 'C': 1, 'G': 2, 'U': 3}\n\ndef encode_sequence(sequence):\n    # Map valid nucleotides; assign -1 for invalid characters (if any)\n    return [nucleotide_map.get(n, -1) for n in sequence]\n\n# Encode RNA sequences for training, validation, and test datasets\ntrain_sequences['encoded_seq'] = train_sequences['sequence'].apply(encode_sequence)\nvalidation_sequences['encoded_seq'] = validation_sequences['sequence'].apply(encode_sequence)\ntest_sequences['encoded_seq'] = test_sequences['sequence'].apply(encode_sequence)\n\nprint(\"\\nEncoded Train Sequences:\")\nprint(train_sequences.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-11T08:59:46.905302Z","iopub.execute_input":"2025-03-11T08:59:46.905605Z","iopub.status.idle":"2025-03-11T08:59:46.932344Z","shell.execute_reply.started":"2025-03-11T08:59:46.905581Z","shell.execute_reply":"2025-03-11T08:59:46.931163Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop non-numerical columns from train_labels and validation_labels\ntrain_labels_numerical = train_labels.drop(columns=['ID', 'resname', 'resid'])\nvalidation_labels_numerical = validation_labels.drop(columns=['ID', 'resname', 'resid'])\n\n# Ensure all remaining columns are numeric\ntrain_labels_numerical = train_labels_numerical.apply(pd.to_numeric, errors='coerce')\nvalidation_labels_numerical = validation_labels_numerical.apply(pd.to_numeric, errors='coerce')\n\n# Fill missing values (if any) with a default value (e.g., 0.0)\ntrain_labels_numerical = train_labels_numerical.fillna(0.0)\nvalidation_labels_numerical = validation_labels_numerical.fillna(0.0)\n\nprint(\"\\nTrain Labels (Numerical):\")\nprint(train_labels_numerical.head())\n\nprint(\"\\nValidation Labels (Numerical):\")\nprint(validation_labels_numerical.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-11T09:00:00.386035Z","iopub.execute_input":"2025-03-11T09:00:00.386357Z","iopub.status.idle":"2025-03-11T09:00:00.441604Z","shell.execute_reply.started":"2025-03-11T09:00:00.386330Z","shell.execute_reply":"2025-03-11T09:00:00.440632Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RNADataset(Dataset):\n    def __init__(self, sequences, labels=None):\n        self.sequences = sequences\n        self.labels = labels\n\n    def __len__(self):\n        return len(self.sequences)\n\n    def __getitem__(self, idx):\n        # Get encoded sequence and its length\n        seq = torch.tensor(self.sequences.iloc[idx]['encoded_seq'], dtype=torch.long)\n        seq_len = len(seq)\n        \n        if self.labels is not None:\n            # Use only numerical columns for labels\n            coords = torch.tensor(self.labels.iloc[idx].values.astype(float), dtype=torch.float32)\n            return seq, coords, seq_len\n        else:\n            return seq, seq_len\n\n# Create datasets for training and validation using numerical labels\ntrain_dataset = RNADataset(train_sequences, train_labels_numerical)\nval_dataset = RNADataset(validation_sequences, validation_labels_numerical)\n\n# Create data loaders with custom collate function for padding sequences\ndef collate_fn(batch):\n    sequences, coords, lengths = zip(*batch)\n    padded_sequences = pad_sequence(sequences, batch_first=True, padding_value=0)  # Padding with 0\n    coords = torch.stack(coords)  # Stack coordinates (no need for padding)\n    lengths = torch.tensor(lengths)  # Tensor of sequence lengths\n    return padded_sequences, coords, lengths\n\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True, collate_fn=collate_fn)\nval_loader = DataLoader(val_dataset, batch_size=32, collate_fn=collate_fn)\n\nprint(\"Data loaders created successfully!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-11T09:00:11.920288Z","iopub.execute_input":"2025-03-11T09:00:11.920638Z","iopub.status.idle":"2025-03-11T09:00:11.928820Z","shell.execute_reply.started":"2025-03-11T09:00:11.920610Z","shell.execute_reply":"2025-03-11T09:00:11.927837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RNA3DModel(nn.Module):\n    def __init__(self, input_dim=4, hidden_dim=128):\n        super(RNA3DModel, self).__init__()\n        self.embedding = nn.Embedding(input_dim, hidden_dim)  # Embed nucleotide sequences\n        self.lstm = nn.LSTM(hidden_dim, hidden_dim, batch_first=True)\n        self.fc = nn.Linear(hidden_dim, 3)  # Predict x, y, z coordinates\n\n    def forward(self, x):\n        x = self.embedding(x)\n        x, _ = self.lstm(x)\n        x = self.fc(x)\n        return x\n\n# Initialize the model\nmodel = RNA3DModel()\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel.to(device)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-11T09:00:23.500762Z","iopub.execute_input":"2025-03-11T09:00:23.501102Z","iopub.status.idle":"2025-03-11T09:00:23.951217Z","shell.execute_reply.started":"2025-03-11T09:00:23.501077Z","shell.execute_reply":"2025-03-11T09:00:23.950501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"criterion = nn.MSELoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-11T09:00:33.225126Z","iopub.execute_input":"2025-03-11T09:00:33.225467Z","iopub.status.idle":"2025-03-11T09:00:35.285861Z","shell.execute_reply.started":"2025-03-11T09:00:33.225441Z","shell.execute_reply":"2025-03-11T09:00:35.285090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_model(model, train_loader, val_loader, criterion, optimizer, epochs=10):\n    for epoch in range(epochs):\n        model.train()\n        train_loss = 0.0\n\n        for sequences, coords, lengths in train_loader:\n            sequences, coords = sequences.to(device), coords.to(device)\n\n            optimizer.zero_grad()\n            outputs = model(sequences)\n\n            # Mask out padded positions using sequence lengths\n            max_len = outputs.size(1)  # Maximum sequence length in batch\n            mask = torch.arange(max_len).unsqueeze(0).to(device) < lengths.unsqueeze(1)\n\n            outputs_masked = outputs[mask]\n            coords_masked = coords.view(-1, 3)[mask.view(-1)]\n\n            loss = criterion(outputs_masked.view(-1, 3), coords_masked.view(-1))\n            loss.backward()\n            optimizer.step()\n\n            train_loss += loss.item()\n\n        val_loss = validate_model(model, val_loader)\n\n        print(f\"Epoch {epoch+1}/{epochs}, Train Loss: {train_loss/len(train_loader):.4f}, Val Loss: {val_loss:.4f}\")\n\ndef validate_model(model_, val_loader_):\n    model_.eval()\n    val_loss_agg_=[]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-11T09:00:44.240446Z","iopub.execute_input":"2025-03-11T09:00:44.240899Z","iopub.status.idle":"2025-03-11T09:00:44.247798Z","shell.execute_reply.started":"2025-03-11T09:00:44.240873Z","shell.execute_reply":"2025-03-11T09:00:44.246719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def generate_predictions(model, test_sequences):\n    model.eval()\n    predictions = []\n\n    with torch.no_grad():\n        for _, row in test_sequences.iterrows():\n            seq_id = row['target_id']\n            sequence_encoded = torch.tensor(row['encoded_seq'], dtype=torch.long).unsqueeze(0).to(device)\n\n            pred_coords_list = []\n            for _ in range(5):  # Generate five predictions per sequence\n                pred_coords_list.append(model(sequence_encoded).cpu().numpy().flatten())\n\n            predictions.append([seq_id] + np.concatenate(pred_coords_list).tolist())\n\n    return predictions\n\npredictions = generate_predictions(model, test_sequences)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-11T09:00:55.375238Z","iopub.execute_input":"2025-03-11T09:00:55.375553Z","iopub.status.idle":"2025-03-11T09:00:55.731875Z","shell.execute_reply.started":"2025-03-11T09:00:55.375528Z","shell.execute_reply":"2025-03-11T09:00:55.731162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_file = \"submission.csv\"\n\n# Define header fields based on competition requirements\nheader = [\"ID\", \"resname\", \"resid\"] + [f\"x_{i},y_{i},z_{i}\" for i in range(1, 6)]\n\n# Generate predictions and save them to submission.csv\nwith open(submission_file, mode=\"w\", newline=\"\") as file:\n    writer = csv.writer(file)\n    \n    # Write header\n    writer.writerow(header)\n    \n    # Write predictions (ensure each row matches header format)\n    for pred in predictions:\n        # Flatten predictions into a single row (ensure correct number of fields)\n        writer.writerow(pred)\n\nprint(f\"Submission file saved to {submission_file}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-11T09:01:08.450027Z","iopub.execute_input":"2025-03-11T09:01:08.450364Z","iopub.status.idle":"2025-03-11T09:01:08.500160Z","shell.execute_reply.started":"2025-03-11T09:01:08.450332Z","shell.execute_reply":"2025-03-11T09:01:08.499213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import csv\n\n# Define expected header format\nexpected_header = [\"ID\", \"resname\", \"resid\"] + [f\"x_{i},y_{i},z_{i}\" for i in range(1, 6)]\n\n# Read and validate submission file\nwith open(\"submission.csv\", \"r\") as file:\n    reader = csv.reader(file)\n    header = next(reader)  # Read header\n    rows = list(reader)  # Read all rows\n\n# Validate header\nheader_matches = header == expected_header\n\n# Validate row lengths\ncorrect_row_lengths = all(len(row) == len(expected_header) for row in rows)\n\nprint(\"Header Matches Expected Format:\", header_matches)\nprint(\"All Rows Have Correct Number of Fields:\", correct_row_lengths)\nprint(\"Number of Rows in Submission File:\", len(rows))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-11T09:14:17.628562Z","iopub.execute_input":"2025-03-11T09:14:17.628874Z","iopub.status.idle":"2025-03-11T09:14:17.650034Z","shell.execute_reply.started":"2025-03-11T09:14:17.628847Z","shell.execute_reply":"2025-03-11T09:14:17.649248Z"}},"outputs":[],"execution_count":null}]}