{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:51:46.326301Z","iopub.execute_input":"2025-03-31T08:51:46.326773Z","iopub.status.idle":"2025-03-31T08:51:47.868115Z","shell.execute_reply.started":"2025-03-31T08:51:46.326735Z","shell.execute_reply":"2025-03-31T08:51:47.866991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:51:53.881067Z","iopub.execute_input":"2025-03-31T08:51:53.881381Z","iopub.status.idle":"2025-03-31T08:51:53.885832Z","shell.execute_reply.started":"2025-03-31T08:51:53.881356Z","shell.execute_reply":"2025-03-31T08:51:53.884800Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/stanford-rna-3d-folding\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:51:56.986580Z","iopub.execute_input":"2025-03-31T08:51:56.987043Z","iopub.status.idle":"2025-03-31T08:51:56.991260Z","shell.execute_reply.started":"2025-03-31T08:51:56.987012Z","shell.execute_reply":"2025-03-31T08:51:56.990022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences = pd.read_csv(f\"{DATA_DIR}/train_sequences.csv\")\ntrain_labels = pd.read_csv(f\"{DATA_DIR}/train_labels.csv\")\ntest_sequences = pd.read_csv(f\"{DATA_DIR}/test_sequences.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:51:59.613369Z","iopub.execute_input":"2025-03-31T08:51:59.613717Z","iopub.status.idle":"2025-03-31T08:51:59.878980Z","shell.execute_reply.started":"2025-03-31T08:51:59.613681Z","shell.execute_reply":"2025-03-31T08:51:59.878105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Train Sequences Columns:\", train_sequences.columns)\nprint(\"Train Labels Columns:\", train_labels.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:52:02.112252Z","iopub.execute_input":"2025-03-31T08:52:02.112568Z","iopub.status.idle":"2025-03-31T08:52:02.119707Z","shell.execute_reply.started":"2025-03-31T08:52:02.112543Z","shell.execute_reply":"2025-03-31T08:52:02.118592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_sequences[\"target_id\"].head())\nprint(train_labels[\"ID\"].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:52:05.625080Z","iopub.execute_input":"2025-03-31T08:52:05.625400Z","iopub.status.idle":"2025-03-31T08:52:05.633856Z","shell.execute_reply.started":"2025-03-31T08:52:05.625374Z","shell.execute_reply":"2025-03-31T08:52:05.632595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences[\"base_target_id\"] = train_sequences[\"target_id\"].apply(lambda x: x.split(\"_\")[0])\ntrain_labels[\"base_target_id\"] = train_labels[\"ID\"].apply(lambda x: x.split(\"_\")[0])\n\ntrain_df = train_sequences.merge(train_labels, left_on=\"base_target_id\", right_on=\"base_target_id\")\n\nprint(\"Merge successful! Train dataset shape:\", train_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:53:53.296630Z","iopub.execute_input":"2025-03-31T08:53:53.297026Z","iopub.status.idle":"2025-03-31T08:53:53.501311Z","shell.execute_reply.started":"2025-03-31T08:53:53.296992Z","shell.execute_reply":"2025-03-31T08:53:53.500480Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def encode_sequence(seq, max_length=100):\n    mapping = {\"A\": 0, \"U\": 1, \"G\": 2, \"C\": 3}  \n    encoded = [mapping.get(base, 4) for base in seq]  \n    encoded = encoded[:max_length] + [4] * (max_length - len(encoded))  \n    return np.array(encoded, dtype=np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:55:13.199994Z","iopub.execute_input":"2025-03-31T08:55:13.200643Z","iopub.status.idle":"2025-03-31T08:55:13.206348Z","shell.execute_reply.started":"2025-03-31T08:55:13.200602Z","shell.execute_reply":"2025-03-31T08:55:13.205019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RNADataset(Dataset):\n    def __init__(self, dataframe, is_test=False):\n        self.data = dataframe\n        self.is_test = is_test\n    \n    def __len__(self):\n        return len(self.data)\n    \n    def __getitem__(self, idx):\n        row = self.data.iloc[idx]\n        sequence = encode_sequence(row[\"sequence\"])\n        if self.is_test:\n            return torch.tensor(sequence, dtype=torch.float32)\n        else:\n            label = row[[\"x_1\", \"y_1\", \"z_1\"]].astype(float).values  \n            return torch.tensor(sequence, dtype=torch.float32), torch.tensor(label, dtype=torch.float32).squeeze()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:55:15.336784Z","iopub.execute_input":"2025-03-31T08:55:15.337354Z","iopub.status.idle":"2025-03-31T08:55:15.344975Z","shell.execute_reply.started":"2025-03-31T08:55:15.337213Z","shell.execute_reply":"2025-03-31T08:55:15.343575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size = 32\ntrain_dataset = RNADataset(train_df)\ntest_dataset = RNADataset(test_sequences, is_test=True)\n\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\ntest_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:55:18.093789Z","iopub.execute_input":"2025-03-31T08:55:18.094278Z","iopub.status.idle":"2025-03-31T08:55:18.101904Z","shell.execute_reply.started":"2025-03-31T08:55:18.094231Z","shell.execute_reply":"2025-03-31T08:55:18.099934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RNAFoldModel(nn.Module):\n    def __init__(self, input_size=100, hidden_size=64):\n        super(RNAFoldModel, self).__init__()\n        self.fc1 = nn.Linear(input_size, hidden_size)\n        self.fc2 = nn.Linear(hidden_size, 3)  \n\n    def forward(self, x):\n        x = torch.relu(self.fc1(x))\n        return self.fc2(x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:55:20.820281Z","iopub.execute_input":"2025-03-31T08:55:20.820759Z","iopub.status.idle":"2025-03-31T08:55:20.827472Z","shell.execute_reply.started":"2025-03-31T08:55:20.820716Z","shell.execute_reply":"2025-03-31T08:55:20.826113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Columns in dataset:\", train_labels.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:55:23.571891Z","iopub.execute_input":"2025-03-31T08:55:23.572370Z","iopub.status.idle":"2025-03-31T08:55:23.578664Z","shell.execute_reply.started":"2025-03-31T08:55:23.572331Z","shell.execute_reply":"2025-03-31T08:55:23.577302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.isna().sum())  \nprint(train_df.describe()) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:55:26.140707Z","iopub.execute_input":"2025-03-31T08:55:26.141157Z","iopub.status.idle":"2025-03-31T08:55:26.341449Z","shell.execute_reply.started":"2025-03-31T08:55:26.141114Z","shell.execute_reply":"2025-03-31T08:55:26.339977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[[\"x_1\", \"y_1\", \"z_1\"]] = train_df[[\"x_1\", \"y_1\", \"z_1\"]].fillna(train_df[[\"x_1\", \"y_1\", \"z_1\"]].mean())\nmost_common_seq = train_df[\"all_sequences\"].mode()[0]\ntrain_df[\"all_sequences\"] = train_df[\"all_sequences\"].fillna(most_common_seq)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:55:32.664140Z","iopub.execute_input":"2025-03-31T08:55:32.664617Z","iopub.status.idle":"2025-03-31T08:55:32.747543Z","shell.execute_reply.started":"2025-03-31T08:55:32.664573Z","shell.execute_reply":"2025-03-31T08:55:32.745929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.isna().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T08:55:37.195460Z","iopub.execute_input":"2025-03-31T08:55:37.195951Z","iopub.status.idle":"2025-03-31T08:55:37.318726Z","shell.execute_reply.started":"2025-03-31T08:55:37.195913Z","shell.execute_reply":"2025-03-31T08:55:37.317634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\ntrain_df[[\"x_1\", \"y_1\", \"z_1\"]] = scaler.fit_transform(train_df[[\"x_1\", \"y_1\", \"z_1\"]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T07:48:42.188924Z","iopub.execute_input":"2025-03-31T07:48:42.189294Z","iopub.status.idle":"2025-03-31T07:48:42.221300Z","shell.execute_reply.started":"2025-03-31T07:48:42.189263Z","shell.execute_reply":"2025-03-31T07:48:42.220188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = RNAFoldModel().to(device)\ncriterion = nn.MSELoss()  \noptimizer = optim.Adam(model.parameters(), lr=0.001)\n\nepochs = 10\nprint(\"Training model...\")\nfor epoch in range(epochs):\n    model.train()\n    total_loss = 0\n    for inputs, targets in train_loader:\n        inputs, targets = inputs.to(device), targets.to(device)\n        optimizer.zero_grad()\n        outputs = model(inputs)\n        loss = criterion(outputs, targets)\n        loss.backward()\n        optimizer.step()\n        total_loss += loss.item()\n    \n    print(f\"Epoch {epoch+1}/{epochs}, Loss: {total_loss/len(train_loader):.4f}\")\n\nprint(\"Generating predictions...\")\nmodel.eval()\npredictions = []\nwith torch.no_grad():\n    for inputs in test_loader:\n        inputs = inputs.to(device)\n        outputs = model(inputs)\n        predictions.extend(outputs.cpu().numpy())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T07:48:45.564528Z","iopub.execute_input":"2025-03-31T07:48:45.564887Z","iopub.status.idle":"2025-03-31T08:27:31.527160Z","shell.execute_reply.started":"2025-03-31T07:48:45.564857Z","shell.execute_reply":"2025-03-31T08:27:31.526160Z"}},"outputs":[],"execution_count":null}]}