{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":87793,"databundleVersionId":11403143,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-18T19:20:26.480045Z","iopub.execute_input":"2025-03-18T19:20:26.480358Z","iopub.status.idle":"2025-03-18T19:20:28.507988Z","shell.execute_reply.started":"2025-03-18T19:20:26.480317Z","shell.execute_reply":"2025-03-18T19:20:28.507121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install biopython","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T19:20:28.509022Z","iopub.execute_input":"2025-03-18T19:20:28.509457Z","iopub.status.idle":"2025-03-18T19:20:34.316766Z","shell.execute_reply.started":"2025-03-18T19:20:28.509422Z","shell.execute_reply":"2025-03-18T19:20:34.315674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport torch\nimport os\nfrom torch.utils.data import Dataset, DataLoader\nfrom Bio import SeqIO\nimport numpy as np\n\n# File Paths\ntrain_seq_path = \"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\"\ntrain_labels_path = \"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\"\nval_seq_path = \"/kaggle/input/stanford-rna-3d-folding/validation_sequences.csv\"\nval_labels_path = \"/kaggle/input/stanford-rna-3d-folding/validation_labels.csv\"\nmsa_dir = \"/kaggle/input/stanford-rna-3d-folding/MSA\"\n\n# Load sequences\ntrain_seqs = pd.read_csv(train_seq_path)\ntrain_labels = pd.read_csv(train_labels_path)\nval_seqs = pd.read_csv(val_seq_path)\nval_labels = pd.read_csv(val_labels_path)\n\n# Map RNA bases to numerical values\nrna_vocab = {'A': 0, 'C': 1, 'G': 2, 'U': 3, '-': 4}  # Adding '-' as a special case\n\ndef encode_sequence(seq):\n    return [rna_vocab.get(nt, 4) for nt in seq]  # Use `.get()` to avoid KeyError\n\n# Convert sequences to numerical format\ntrain_seqs['encoded_seq'] = train_seqs['sequence'].apply(encode_sequence)\nval_seqs['encoded_seq'] = val_seqs['sequence'].apply(encode_sequence)\n\nprint(\"Loaded Training & Validation Sequences!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T19:20:34.318790Z","iopub.execute_input":"2025-03-18T19:20:34.319126Z","iopub.status.idle":"2025-03-18T19:20:38.096515Z","shell.execute_reply.started":"2025-03-18T19:20:34.319099Z","shell.execute_reply":"2025-03-18T19:20:38.095601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob\n\ndef load_msa(msa_path):\n    sequences = []\n    with open(msa_path, \"r\") as file:\n        for record in SeqIO.parse(file, \"fasta\"):\n            sequences.append(str(record.seq))\n    return sequences\n\n# Load MSA features\nmsa_files = glob.glob(os.path.join(msa_dir, \"*.fasta\"))\nmsa_dict = {os.path.basename(f): load_msa(f) for f in msa_files}\n\nprint(f\"Loaded {len(msa_files)} MSA files!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T19:20:38.097445Z","iopub.execute_input":"2025-03-18T19:20:38.097683Z","iopub.status.idle":"2025-03-18T19:20:47.459270Z","shell.execute_reply.started":"2025-03-18T19:20:38.097662Z","shell.execute_reply":"2025-03-18T19:20:47.458282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RNADataset(Dataset):\n    def __init__(self, sequences, labels):\n        self.sequences = sequences\n        self.labels = labels\n\n    def __len__(self):\n        return len(self.sequences)\n\n    def __getitem__(self, idx):\n        seq = torch.tensor(self.sequences.iloc[idx]['encoded_seq'], dtype=torch.long)\n        label = torch.tensor(self.labels.iloc[idx, 1:].values, dtype=torch.float)  # Exclude ID\n        return seq, label\n\n# Create dataset objects\ntrain_dataset = RNADataset(train_seqs, train_labels)\nval_dataset = RNADataset(val_seqs, val_labels)\n\n# Create DataLoader\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=32, shuffle=False)\n\nprint(\"Dataloaders Ready!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T19:20:47.460204Z","iopub.execute_input":"2025-03-18T19:20:47.460416Z","iopub.status.idle":"2025-03-18T19:20:47.467257Z","shell.execute_reply.started":"2025-03-18T19:20:47.460397Z","shell.execute_reply":"2025-03-18T19:20:47.466295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nclass RNA3DModel(nn.Module):\n    def __init__(self, vocab_size=4, embed_dim=256, num_heads=8, num_layers=6, dropout=0.3, hidden_dim=512):\n        \"\"\"\n        Initialize the RNA3DModel.\n\n        Parameters:\n            vocab_size (int): The size of the vocabulary (RNA bases, typically 4: A, C, G, U)\n            embed_dim (int): The dimension of the embedding for each RNA base\n            num_heads (int): Number of heads in the multi-head attention mechanism\n            num_layers (int): Number of transformer encoder layers\n            dropout (float): Dropout probability to avoid overfitting\n            hidden_dim (int): The dimension of the hidden layer in the feed-forward networks of the transformer\n        \"\"\"\n        super(RNA3DModel, self).__init__()\n\n        # Embedding Layer for RNA sequence (transforming nucleotides to vectors)\n        self.embedding = nn.Embedding(vocab_size, embed_dim)\n\n        # Transformer Encoder\n        self.encoder = nn.TransformerEncoder(\n            nn.TransformerEncoderLayer(\n                d_model=embed_dim,  # Dimension of model (embedding size)\n                nhead=num_heads,  # Number of attention heads\n                dim_feedforward=hidden_dim,  # Feed-forward hidden dimension\n                dropout=dropout,  # Dropout probability\n                batch_first=True  # Set batch_first=True for easier batch handling\n            ),\n            num_layers=num_layers  # Number of transformer layers\n        )\n\n        # Graph Convolution-like Fully Connected Layers\n        self.gcn1 = nn.Linear(embed_dim, 128)  # First fully connected layer\n        self.gcn2 = nn.Linear(128, 64)  # Second fully connected layer\n\n        # Output Layer (predicts 3D coordinates)\n        self.output_layer = nn.Linear(64, 3)  # Output dimension is 3 for x, y, z coordinates\n\n    def forward(self, src):\n        \"\"\"\n        The forward pass of the model.\n\n        Parameters:\n            src (Tensor): The input RNA sequence tensor with shape (batch_size, sequence_length)\n\n        Returns:\n            Tensor: Predicted 3D coordinates of shape (batch_size, sequence_length, 3)\n        \"\"\"\n        # Convert sequence into embeddings\n        x = self.embedding(src)\n\n        # Pass through transformer layers\n        x = self.encoder(x)\n\n        # Apply graph convolution layers (fully connected layers)\n        x = F.relu(self.gcn1(x))  # First graph layer with ReLU activation\n        x = F.relu(self.gcn2(x))  # Second graph layer with ReLU activation\n\n        # Predict 3D coordinates (x, y, z) for each sequence\n        xyz = self.output_layer(x)\n\n        return xyz\n\n# Instantiate model and move to GPU (if available)\nmodel = RNA3DModel().cuda() if torch.cuda.is_available() else RNA3DModel()\nprint(model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-18T19:22:10.898197Z","iopub.execute_input":"2025-03-18T19:22:10.898535Z","iopub.status.idle":"2025-03-18T19:22:10.939076Z","shell.execute_reply.started":"2025-03-18T19:22:10.898504Z","shell.execute_reply":"2025-03-18T19:22:10.938276Z"}},"outputs":[],"execution_count":null}]}