{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom tqdm import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-24T13:46:10.753628Z","iopub.execute_input":"2025-03-24T13:46:10.753993Z","iopub.status.idle":"2025-03-24T13:46:16.272780Z","shell.execute_reply.started":"2025-03-24T13:46:10.753961Z","shell.execute_reply":"2025-03-24T13:46:16.272068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_seq = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_sequences.csv')\n# train_labels = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_labels.csv')\n# train_seq.head(1)\n# print(train_labels.head(1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T13:34:45.422097Z","iopub.execute_input":"2025-03-24T13:34:45.422685Z","iopub.status.idle":"2025-03-24T13:34:45.694618Z","shell.execute_reply.started":"2025-03-24T13:34:45.422643Z","shell.execute_reply":"2025-03-24T13:34:45.693171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(f'Shape: {train_labels.shape}')\n# train_labels.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T13:28:48.334277Z","iopub.execute_input":"2025-03-24T13:28:48.334834Z","iopub.status.idle":"2025-03-24T13:28:48.370383Z","shell.execute_reply.started":"2025-03-24T13:28:48.334796Z","shell.execute_reply":"2025-03-24T13:28:48.368946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_seq[train_seq['all_sequences'].isna()]\n# print(train_labels[train_labels['x_1'].isna()])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T13:26:46.827435Z","iopub.status.idle":"2025-03-24T13:26:46.827831Z","shell.execute_reply":"2025-03-24T13:26:46.827696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T13:46:16.276410Z","iopub.execute_input":"2025-03-24T13:46:16.276741Z","iopub.status.idle":"2025-03-24T13:46:16.358524Z","shell.execute_reply.started":"2025-03-24T13:46:16.276706Z","shell.execute_reply":"2025-03-24T13:46:16.357764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# 1. Data Preprocessing\n# =====================\ndef load_and_preprocess_data(train_seq_file, train_labels_file):\n    # Load train sequences\n    train_seq = pd.read_csv(train_seq_file)\n    train_labels = pd.read_csv(train_labels_file)\n\n    # Extract target_id from ID column\n    train_labels[\"target_id\"] = train_labels[\"ID\"].apply(lambda x: \"_\".join(x.split(\"_\")[:2]))\n\n    # Create sequence features\n    sequence_features = create_sequence_features(train_seq)\n\n    # Merge sequence features with labels\n    df_merged = pd.merge(train_labels, sequence_features, on=\"target_id\", how=\"left\")\n\n    # Normalize coordinate labels\n    scaler = StandardScaler()\n    coord_cols = [col for col in df_merged.columns if col.startswith((\"x_\", \"y_\", \"z_\"))]\n    df_merged[coord_cols] = scaler.fit_transform(df_merged[coord_cols])\n\n    # Split into train and validation sets\n    train_df, val_df = train_test_split(df_merged, test_size=0.2, random_state=42)\n\n    return train_df, val_df, scaler\n\ndef create_sequence_features(train_seq):\n    \"\"\"Convert RNA sequences into numerical features\"\"\"\n    sequence_features = {}\n\n    nucleotide_map = {'A': [1, 0, 0, 0], 'G': [0, 1, 0, 0], \n                      'C': [0, 0, 1, 0], 'U': [0, 0, 0, 1],\n                      'T': [0, 0, 0, 1]}  # Treat T as U\n\n    for _, row in train_seq.iterrows():\n        target_id = row['target_id']\n        sequence = row['sequence']\n        \n        seq_length = len(sequence)\n        nt_counts = {nt: sequence.count(nt) for nt in \"AGCU\"}\n        \n        features = {\n            'target_id': target_id,\n            'seq_length': seq_length,\n            'A_count': nt_counts['A'],\n            'G_count': nt_counts['G'],\n            'C_count': nt_counts['C'],\n            'U_count': nt_counts['U'],\n        }\n\n        # One-hot encoding per position (optional)\n        for i, nt in enumerate(sequence[:100]):  # Limit sequence length\n            for j, val in enumerate(nucleotide_map.get(nt, [0, 0, 0, 0])):\n                features[f'pos_{i}_{j}'] = val\n        \n        sequence_features[target_id] = features\n    \n    return pd.DataFrame.from_dict(sequence_features, orient=\"index\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T13:51:42.557056Z","iopub.execute_input":"2025-03-24T13:51:42.557410Z","iopub.status.idle":"2025-03-24T13:51:42.566762Z","shell.execute_reply.started":"2025-03-24T13:51:42.557379Z","shell.execute_reply":"2025-03-24T13:51:42.565463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# 2. Custom Dataset\n# =====================\nclass RNADataset(Dataset):\n    def __init__(self, dataframe):\n        self.data = dataframe\n        self.features = [col for col in dataframe.columns if col.startswith((\"seq_length\", \"A_count\", \"G_count\", \"C_count\", \"U_count\", \"pos_\"))]\n        self.targets = [col for col in dataframe.columns if col.startswith((\"x_\", \"y_\", \"z_\"))]\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n    # Ensure features are numeric and fill NaN values\n        features = self.data.iloc[idx][self.features].astype(float).fillna(0).values\n        targets = self.data.iloc[idx][self.targets].astype(float).fillna(0).values\n        return {\"features\": torch.tensor(features, dtype=torch.float32),\n                \"targets\": torch.tensor(targets, dtype=torch.float32)}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T13:54:28.507128Z","iopub.execute_input":"2025-03-24T13:54:28.507431Z","iopub.status.idle":"2025-03-24T13:54:28.513133Z","shell.execute_reply.started":"2025-03-24T13:54:28.507408Z","shell.execute_reply":"2025-03-24T13:54:28.512397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# 3. Model Definition\n# =====================\nclass RNAFoldingModel(nn.Module):\n    def __init__(self, input_dim, hidden_dim=128):\n        super(RNAFoldingModel, self).__init__()\n\n        self.model = nn.Sequential(\n            nn.Linear(input_dim, hidden_dim),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(hidden_dim, hidden_dim),\n            nn.ReLU(),\n            nn.Linear(hidden_dim, 3)  # Predict (x_1, y_1, z_1)\n        )\n\n    def forward(self, x):\n        return self.model(x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T13:49:02.533099Z","iopub.execute_input":"2025-03-24T13:49:02.533424Z","iopub.status.idle":"2025-03-24T13:49:02.538154Z","shell.execute_reply.started":"2025-03-24T13:49:02.533396Z","shell.execute_reply":"2025-03-24T13:49:02.537334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# 4. Training Function\n# =====================\ndef train_model(train_df, val_df, epochs=20, batch_size=64, lr=0.001):\n    # Create datasets\n    train_dataset = RNADataset(train_df)\n    val_dataset = RNADataset(val_df)\n\n    # Input dimension\n    input_dim = len(train_dataset.features)\n\n    # Create dataloaders\n    train_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\n    val_loader = DataLoader(val_dataset, batch_size=batch_size)\n\n    # Initialize model\n    model = RNAFoldingModel(input_dim).to(device)\n\n    # Loss function and optimizer\n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(model.parameters(), lr=lr)\n\n    # Training loop\n    for epoch in range(epochs):\n        model.train()\n        train_loss = 0\n\n        for batch in tqdm(train_loader, desc=f\"Epoch {epoch+1}/{epochs} (Training)\"):\n            features = batch[\"features\"].to(device)\n            targets = batch[\"targets\"].to(device)\n\n            optimizer.zero_grad()\n            outputs = model(features)\n            loss = criterion(outputs, targets)\n            loss.backward()\n            optimizer.step()\n\n            train_loss += loss.item()\n\n        train_loss /= len(train_loader)\n\n        # Validation\n        model.eval()\n        val_loss = 0\n        with torch.no_grad():\n            for batch in val_loader:\n                features = batch[\"features\"].to(device)\n                targets = batch[\"targets\"].to(device)\n\n                outputs = model(features)\n                loss = criterion(outputs, targets)\n                val_loss += loss.item()\n\n        val_loss /= len(val_loader)\n\n        print(f\"Epoch {epoch+1}/{epochs}, Train Loss: {train_loss:.4f}, Val Loss: {val_loss:.4f}\")\n\n    # Save model\n    torch.save(model.state_dict(), \"rna_model.pth\")\n    print(\"Model saved successfully!\")\n\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T13:49:46.815859Z","iopub.execute_input":"2025-03-24T13:49:46.816183Z","iopub.status.idle":"2025-03-24T13:49:46.824053Z","shell.execute_reply.started":"2025-03-24T13:49:46.816153Z","shell.execute_reply":"2025-03-24T13:49:46.823128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# 5. Generate Submission\n# =====================\ndef generate_submission(model, scaler, test_seq_file, sample_submission_file):\n    test_seq = pd.read_csv(test_seq_file)\n    sample_submission = pd.read_csv(sample_submission_file)\n\n    test_features = create_sequence_features(test_seq)\n\n    submission_df = sample_submission.copy()\n    submission_df[\"target_id\"] = submission_df[\"ID\"].apply(lambda x: \"_\".join(x.split(\"_\")[:2]))\n    submission_df = pd.merge(submission_df, test_features, on=\"target_id\", how=\"left\")\n\n    test_dataset = RNADataset(submission_df)\n    test_loader = DataLoader(test_dataset, batch_size=64, shuffle=False)\n\n    model.eval()\n    all_predictions = []\n\n    with torch.no_grad():\n        for batch in tqdm(test_loader, desc=\"Generating predictions\"):\n            features = batch[\"features\"].to(device)\n            outputs = model(features)\n            all_predictions.append(outputs.cpu().numpy())\n\n    all_predictions = np.vstack(all_predictions)\n    all_predictions = scaler.inverse_transform(all_predictions)\n\n    submission_df[[\"x_1\", \"y_1\", \"z_1\"]] = all_predictions\n    submission_df.to_csv(\"submission.csv\", index=False)\n    print(\"Submission generated successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T13:49:59.377272Z","iopub.execute_input":"2025-03-24T13:49:59.377729Z","iopub.status.idle":"2025-03-24T13:49:59.384085Z","shell.execute_reply.started":"2025-03-24T13:49:59.377694Z","shell.execute_reply":"2025-03-24T13:49:59.382984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# 6. Main Execution\n# =====================\nif __name__ == \"__main__\":\n    train_df, val_df, scaler = load_and_preprocess_data(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\", \"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\n    model = train_model(train_df, val_df)\n    generate_submission(model, scaler, \"/kaggle/input/stanford-rna-3d-folding/test_sequences.csv\", \"/kaggle/input/stanford-rna-3d-folding/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T13:54:32.598344Z","iopub.execute_input":"2025-03-24T13:54:32.598663Z","iopub.status.idle":"2025-03-24T15:18:36.614763Z","shell.execute_reply.started":"2025-03-24T13:54:32.598638Z","shell.execute_reply":"2025-03-24T15:18:36.613904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D\n\n\n# Extract residue IDs and coordinates\nresidues = 3\nx = -21\ny = 5\nz = 11\n\n# Create 3D Plot\nfig = plt.figure(figsize=(10, 6))\nax = fig.add_subplot(111, projection='3d')\n\n# Plot points in 3D\nax.scatter(x, y, z, c=residues, cmap='viridis', s=50, label=\"RNA Residues\")\nax.plot(x, y, z, linestyle='-', linewidth=2, color='blue', label=\"RNA Backbone\")\n\n# Labels\nax.set_xlabel(\"X Coordinate\")\nax.set_ylabel(\"Y Coordinate\")\nax.set_zlabel(\"Z Coordinate\")\nax.set_title(\"Predicted RNA 3D Folding Structure\")\n\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T13:26:32.773870Z","iopub.execute_input":"2025-03-25T13:26:32.774192Z","iopub.status.idle":"2025-03-25T13:26:33.074764Z","shell.execute_reply.started":"2025-03-25T13:26:32.774167Z","shell.execute_reply":"2025-03-25T13:26:33.073835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}