{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":12024591,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-30T08:46:37.61703Z","iopub.execute_input":"2025-04-30T08:46:37.617365Z","iopub.status.idle":"2025-04-30T08:46:38.067458Z","shell.execute_reply.started":"2025-04-30T08:46:37.617342Z","shell.execute_reply":"2025-04-30T08:46:38.066288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n!pip install numpy pandas scikit-learn tqdm\n!pip install torch torchvision torchaudio\n!pip install torch-scatter -f https://data.pyg.org/whl/torch-2.1.0+cpu.html\n!pip install torch-sparse -f https://data.pyg.org/whl/torch-2.1.0+cpu.html\n!pip install torch-cluster -f https://data.pyg.org/whl/torch-2.1.0+cpu.html\n!pip install torch-spline-conv -f https://data.pyg.org/whl/torch-2.1.0+cpu.html\n!pip install torch-geometric","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T08:46:38.069303Z","iopub.execute_input":"2025-04-30T08:46:38.069602Z","iopub.status.idle":"2025-04-30T08:47:04.235342Z","shell.execute_reply.started":"2025-04-30T08:46:38.069579Z","shell.execute_reply":"2025-04-30T08:47:04.23405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom torch_geometric.data import Data, Dataset, DataLoader\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T09:18:49.832392Z","iopub.execute_input":"2025-04-30T09:18:49.83276Z","iopub.status.idle":"2025-04-30T09:18:49.839281Z","shell.execute_reply.started":"2025-04-30T09:18:49.832732Z","shell.execute_reply":"2025-04-30T09:18:49.838035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sequences_df = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\nlabels_df = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\n\ndef load_sequence_data():\n    sequence_data = {}\n    for _, row in sequences_df.iterrows():\n        sequence_data[row['target_id']] = row['sequence']\n    return sequence_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T09:18:52.78186Z","iopub.execute_input":"2025-04-30T09:18:52.782277Z","iopub.status.idle":"2025-04-30T09:18:53.053353Z","shell.execute_reply.started":"2025-04-30T09:18:52.782247Z","shell.execute_reply":"2025-04-30T09:18:53.052183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_label_data():\n    label_data = {}\n    for _, row in labels_df.iterrows():\n        target_id, residue = row['ID'].rsplit('_', 1)\n        coord = [row[f'x_1'], row[f'y_1'], row[f'z_1']]\n        if target_id not in label_data:\n            label_data[target_id] = []\n        label_data[target_id].append(coord)\n    return label_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T09:18:57.674658Z","iopub.execute_input":"2025-04-30T09:18:57.675171Z","iopub.status.idle":"2025-04-30T09:18:57.681327Z","shell.execute_reply.started":"2025-04-30T09:18:57.67514Z","shell.execute_reply":"2025-04-30T09:18:57.680295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RNADataset(Dataset):\n    def __init__(self, sequence_dict, label_dict):\n        self.sequence_dict = sequence_dict\n        self.label_dict = label_dict\n        self.keys = list(sequence_dict.keys())\n\n    def __len__(self):\n        return len(self.keys)\n\n    def __getitem__(self, idx):\n        target_id = self.keys[idx]\n        sequence = self.sequence_dict[target_id]\n        coords = torch.tensor(self.label_dict[target_id], dtype=torch.float)\n\n        # Encode nucleotides A, C, G, U as one-hot\n        nt_map = {'A': [1,0,0,0], 'C':[0,1,0,0], 'G':[0,0,1,0], 'U':[0,0,0,1]}\n        node_features = torch.tensor([nt_map.get(nt, [0,0,0,0]) for nt in sequence], dtype=torch.float)\n        \n        # Fully connected edges\n        edge_index = torch.combinations(torch.arange(len(sequence)), r=2).T\n        edge_index = torch.cat([edge_index, edge_index.flip(0)], dim=1)\n\n        return Data(x=node_features, edge_index=edge_index, y=coords)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T09:19:01.553118Z","iopub.execute_input":"2025-04-30T09:19:01.553441Z","iopub.status.idle":"2025-04-30T09:19:01.561816Z","shell.execute_reply.started":"2025-04-30T09:19:01.553417Z","shell.execute_reply":"2025-04-30T09:19:01.560689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch_geometric.nn import GCNConv\n\nclass RNAFoldingModel(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.conv1 = GCNConv(4, 64)\n        self.conv2 = GCNConv(64, 128)\n        self.lin = nn.Linear(128, 3)  # Predict x, y, z\n\n    def forward(self, data):\n        x, edge_index = data.x, data.edge_index\n        x = F.relu(self.conv1(x, edge_index))\n        x = F.relu(self.conv2(x, edge_index))\n        return self.lin(x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T09:19:06.056143Z","iopub.execute_input":"2025-04-30T09:19:06.056565Z","iopub.status.idle":"2025-04-30T09:19:06.064568Z","shell.execute_reply.started":"2025-04-30T09:19:06.056532Z","shell.execute_reply":"2025-04-30T09:19:06.063002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train(model, loader, optimizer, criterion, epochs=10):\n    model.train()\n    for epoch in range(epochs):\n        total_loss = 0\n        for data in loader:\n            data = data.to(device)\n            optimizer.zero_grad()\n            output = model(data)\n            loss = criterion(output, data.y)\n            loss.backward()\n            optimizer.step()\n            total_loss += loss.item()\n        print(f\"Epoch {epoch+1}, Loss: {total_loss/len(loader)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T09:19:11.773657Z","iopub.execute_input":"2025-04-30T09:19:11.774737Z","iopub.status.idle":"2025-04-30T09:19:11.780979Z","shell.execute_reply.started":"2025-04-30T09:19:11.774703Z","shell.execute_reply":"2025-04-30T09:19:11.779993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_and_format(model, test_loader, ids):\n    model.eval()\n    results = []\n    for i, data in enumerate(test_loader):\n        data = data.to(device)\n        pred = model(data).detach().cpu().numpy().flatten()\n        row = [f\"{ids[i]}_{j+1}\" for j in range(len(pred)//3)]\n        coords = pred.reshape(-1, 3).flatten().tolist()\n        results.extend(list(zip(row, coords)))\n    return pd.DataFrame(results, columns=['ID', 'Predicted'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T09:19:26.709066Z","iopub.execute_input":"2025-04-30T09:19:26.709386Z","iopub.status.idle":"2025-04-30T09:19:26.71664Z","shell.execute_reply.started":"2025-04-30T09:19:26.70936Z","shell.execute_reply":"2025-04-30T09:19:26.715629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nseqs = load_sequence_data()\nlabels = load_label_data()\n\ndataset = RNADataset(seqs, labels)\nloader = DataLoader(dataset, batch_size=1, shuffle=True)\n\nmodel = RNAFoldingModel().to(device)\n\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-3)\ncriterion = nn.MSELoss()\n\ntrain(model, loader, optimizer, criterion, epochs=20)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Format your predictions for submission\nsample_submission = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/test_sequences.csv')\npredictions = model(test_data)\n\n# Prepare submission with 5 predictions for each RNA sequence\nsubmission = sample_submission.copy()\nsubmission[['x_1', 'y_1', 'z_1', 'x_2', 'y_2', 'z_2', 'x_3', 'y_3', 'z_3', 'x_4', 'y_4', 'z_4', 'x_5', 'y_5', 'z_5']] = predictions\nsubmission.to_csv('submission.csv', index=False)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}