{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11228175,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":7395079,"sourceType":"datasetVersion","datasetId":4299455},{"sourceId":7639698,"sourceType":"datasetVersion","datasetId":4299272},{"sourceId":7687668,"sourceType":"datasetVersion","datasetId":4391856},{"sourceId":8318191,"sourceType":"datasetVersion","datasetId":4459124},{"sourceId":224703571,"sourceType":"kernelVersion"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import Dataset","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"pwd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T16:40:15.372015Z","iopub.execute_input":"2025-02-28T16:40:15.372416Z","iopub.status.idle":"2025-02-28T16:40:15.381417Z","shell.execute_reply.started":"2025-02-28T16:40:15.372387Z","shell.execute_reply":"2025-02-28T16:40:15.380131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# add following data\n!cp /kaggle/input/ribonanzanet-weights/RibonanzaNet.pt /kaggle/working/\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T16:40:19.329969Z","iopub.execute_input":"2025-02-28T16:40:19.33033Z","iopub.status.idle":"2025-02-28T16:40:20.135139Z","shell.execute_reply.started":"2025-02-28T16:40:19.3303Z","shell.execute_reply":"2025-02-28T16:40:20.133505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\n\nsys.path.append(\"/kaggle/input/ribonanzanet2d-final\")\n\n\nfrom Network import *","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T17:23:02.664417Z","iopub.execute_input":"2025-02-28T17:23:02.664802Z","iopub.status.idle":"2025-02-28T17:23:02.669943Z","shell.execute_reply.started":"2025-02-28T17:23:02.664769Z","shell.execute_reply":"2025-02-28T17:23:02.668934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport torch\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport torch\nimport random\nimport pickle\nfrom tqdm import tqdm\nfrom torch.utils.data import Dataset, DataLoader\nfrom ast import literal_eval\n\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T16:56:09.913376Z","iopub.execute_input":"2025-02-28T16:56:09.913705Z","iopub.status.idle":"2025-02-28T16:56:10.289591Z","shell.execute_reply.started":"2025-02-28T16:56:09.913679Z","shell.execute_reply":"2025-02-28T16:56:10.28843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.manual_seed(0)\nnp.random.seed(0)\nrandom.seed(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T16:56:10.359344Z","iopub.execute_input":"2025-02-28T16:56:10.359844Z","iopub.status.idle":"2025-02-28T16:56:10.367205Z","shell.execute_reply.started":"2025-02-28T16:56:10.359817Z","shell.execute_reply":"2025-02-28T16:56:10.365878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T16:56:17.883108Z","iopub.execute_input":"2025-02-28T16:56:17.883441Z","iopub.status.idle":"2025-02-28T16:56:17.889397Z","shell.execute_reply.started":"2025-02-28T16:56:17.883415Z","shell.execute_reply":"2025-02-28T16:56:17.888031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data=pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\ntrain_labels=pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data=pd.read_csv(\"/kaggle/input/stanford-ribonanza-2-rna-folding-in-3-d/test_sequences.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class finetuned_RibonanzaNet(RibonanzaNet):\n    def __init__(self, config):\n        config.dropout=0.2\n        super(finetuned_RibonanzaNet, self).__init__(config)\n        self.dropout=nn.Dropout(0.0)\n        self.xyz_predictor=nn.Linear(256,3)\n\n    def forward(self,src):\n        \n        #with torch.no_grad():\n        sequence_features, pairwise_features=self.get_embeddings(src, torch.ones_like(src).long().to(src.device))\n\n        xyz=self.xyz_predictor(sequence_features)\n\n        return xyz","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def load_config_from_yaml(file_path):\n#     with open(file_path, 'r') as file:\n#         config = yaml.safe_load(file)\n#     return Config(**config)\n\n# yaml_file=load_config_from_yaml(\"/kaggle/input/ribonanzanet2d-final/configs/pairwise.yaml\")\n\nwith open(file_path, 'r') as file:\n    config = yaml.safe_load(file)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model=finetuned_RibonanzaNet(config).cuda()\n\nmodel.load_state_dict(torch.load(\"/kaggle/input/ribonanzanet-3d-finetune/RibonanzaNet-3D.pt\"))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Using tranining dataset to fine tune\n","metadata":{}},{"cell_type":"code","source":"\nnucleotide_to_int = {'A': 0, 'C': 1, 'G': 2, 'U': 3}\n\ndef encode_sequence(seq):\n    return [nucleotide_to_int[nt] for nt in seq]\n\n# Convert all sequences\ntrain_sequences[\"encoded_sequence\"] = train_sequences[\"sequence\"].apply(encode_sequence)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ntrain_labels = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\ntrain_data = train_sequences.merge(train_labels, left_on=\"target_id\", right_on=\"ID\")\n\ntrain_data = train_data[[\"encoded_sequence\", \"x_1\", \"y_1\", \"z_1\"]]\n\nclass RNA3D_Dataset(Dataset):\n    def __init__(self, sequences_df, labels_df):\n        # Merge sequences and labels\n        self.data = sequences_df.merge(labels_df, left_on=\"target_id\", right_on=\"ID\")\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        row = self.data.iloc[idx]\n\n        # Convert sequence to numerical representation\n        sequence = torch.tensor(encode_sequence(row[\"sequence\"]), dtype=torch.long)\n\n        # Extract 3D coordinates\n        target_xyz = torch.tensor([row[\"x_1\"], row[\"y_1\"], row[\"z_1\"]], dtype=torch.float32)\n\n        return sequence, target_xyz\n\n\n# Create dataset\ntrain_dataset = RNA3D_Dataset(train_data)\n\n# DataLoader for batching\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import Dataset, DataLoader\n\nclass RibonanzaDataset(Dataset):\n    def __init__(self, sequences, labels=None, max_length=100):\n\n        \n        self.sequences = sequences\n        self.labels = labels\n    def __len__(self):\n        return len(self.sequences)\n\n    def __getitem__(self, idx):\n        if self.labels is not None:\n            return self.sequences[idx], torch.tensor(self.labels[idx], dtype=torch.float)\n        else:\n            return self.sequences[idx]  # 预测时不需要返回标签\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = RibonanzaNet(config).to(device)  # 初始化模型\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-3)\nloss_fn = nn.MSELoss() \n# 训练循环\nfor epoch in range(10):  # 训练 10 轮\n    for batch in dataloader:\n        src, label = batch\n        src, label = src.to(device), label.to(device)  # 迁移到 GPU/CPU\n\n        optimizer.zero_grad()\n        output = model(src)  # 前向传播\n        loss = loss_fn(output, label)  # 计算损失\n        loss.backward()  # 反向传播\n        optimizer.step()  # 更新参数\n\n    print(f\"Epoch {epoch+1}, Loss: {loss.item()}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Testing","metadata":{}},{"cell_type":"code","source":"\n\nclass RNADataset(Dataset):\n    def __init__(self,data):\n        self.data=data\n        self.tokens={nt:i for i,nt in enumerate('ACGU')}\n\n    def __len__(self):\n        return len(self.data)\n    \n    def __getitem__(self, idx):\n        sequence=[self.tokens[nt] for nt in (self.data.loc[idx,'sequence'])]\n        sequence=np.array(sequence)\n        sequence=torch.tensor(sequence)\n\n\n\n\n        return {'sequence':sequence}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_dataset=RNADataset(test_data)\ntest_dataset[0]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.eval()\npreds=[]\nfor i in range(len(test_dataset)):\n    src=test_dataset[i]['sequence'].long()\n    src=src.unsqueeze(0).cuda()\n\n    model.train()\n\n    tmp=[]\n    for i in range(4):\n        with torch.no_grad():\n            xyz=model(src).squeeze()\n        tmp.append(xyz.cpu().numpy())\n\n    model.eval()\n    with torch.no_grad():\n        xyz=model(src).squeeze()\n    tmp.append(xyz.cpu().numpy())\n\n    tmp=np.stack(tmp,0)\n    #exit()\n    preds.append(tmp)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ID=[]\nresname=[]\nresid=[]\nx=[]\ny=[]\nz=[]\n\ndata=[]\n\nfor i in range(len(test_data)):\n    #print(test_data.loc[i])\n\n    \n    for j in range(len(test_data.loc[i,'sequence'])):\n        # ID.append(test_data.loc[i,'sequence_id']+f\"_{j+1}\")\n        # resname.append(test_data.loc[i,'sequence'][j])\n        # resid.append(j+1) # 1 indexed\n        row=[test_data.loc[i,'target_id']+f\"_{j+1}\",\n             test_data.loc[i,'sequence'][j],\n             j+1]\n\n        for k in range(5):\n            for kk in range(3):\n                row.append(preds[i][k][j][kk])\n        data.append(row)\n\ncolumns=['ID','resname','resid']\nfor i in range(1,6):\n    columns+=[f\"x_{i}\"]\n    columns+=[f\"y_{i}\"]\n    columns+=[f\"z_{i}\"]\n\n\nsubmission=pd.DataFrame(data,columns=columns)\n\n\nsubmission\nsubmission.to_csv('submission.csv',index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}