{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":87793,"databundleVersionId":11228175,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":" # Load and Preview Data","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Load datasets\ntrain_sequences = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\ntrain_labels = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:18.619066Z","iopub.execute_input":"2025-02-28T15:50:18.619359Z","iopub.status.idle":"2025-02-28T15:50:18.851903Z","shell.execute_reply.started":"2025-02-28T15:50:18.619338Z","shell.execute_reply":"2025-02-28T15:50:18.851104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display first few rows\ntrain_sequences.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:18.853322Z","iopub.execute_input":"2025-02-28T15:50:18.853693Z","iopub.status.idle":"2025-02-28T15:50:18.865233Z","shell.execute_reply.started":"2025-02-28T15:50:18.853656Z","shell.execute_reply":"2025-02-28T15:50:18.864259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:18.866889Z","iopub.execute_input":"2025-02-28T15:50:18.867228Z","iopub.status.idle":"2025-02-28T15:50:18.889682Z","shell.execute_reply.started":"2025-02-28T15:50:18.867204Z","shell.execute_reply":"2025-02-28T15:50:18.88874Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Merge Datasets on target_id","metadata":{}},{"cell_type":"code","source":"# Merge datasets on target_id\ndf_merged = train_sequences.merge(train_labels, left_on=\"target_id\", right_on=\"ID\", how=\"inner\")\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:18.8911Z","iopub.execute_input":"2025-02-28T15:50:18.891414Z","iopub.status.idle":"2025-02-28T15:50:18.940086Z","shell.execute_reply.started":"2025-02-28T15:50:18.891384Z","shell.execute_reply":"2025-02-28T15:50:18.939343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save as CSV (first attempt)\ndf_merged.to_csv(\"/kaggle/working/rna_3d_dataset.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:18.940991Z","iopub.execute_input":"2025-02-28T15:50:18.941301Z","iopub.status.idle":"2025-02-28T15:50:18.961873Z","shell.execute_reply.started":"2025-02-28T15:50:18.941276Z","shell.execute_reply":"2025-02-28T15:50:18.961101Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" # If CSV Is Too Large, Save in Chunks","metadata":{}},{"cell_type":"code","source":"chunk_size = 50000  # Adjust chunk size as needed\nfor i, start in enumerate(range(0, len(df_merged), chunk_size)):\n    chunk_file = f\"/kaggle/working/rna_3d_dataset_part{i+1}.csv\"\n    df_merged.iloc[start:start + chunk_size].to_csv(chunk_file, index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:18.962702Z","iopub.execute_input":"2025-02-28T15:50:18.962977Z","iopub.status.idle":"2025-02-28T15:50:18.980635Z","shell.execute_reply.started":"2025-02-28T15:50:18.962955Z","shell.execute_reply":"2025-02-28T15:50:18.980009Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  Create Metadata File for Kaggle Dataset","metadata":{}},{"cell_type":"code","source":"import json\n\nmetadata = {\n    \"title\": \"Stanford RNA 3D Folding Dataset\",\n    \"id\": \"stanford-rna-3d-folding\",\n    \"licenses\": [{\"name\": \"CC BY 4.0\"}],\n    \"isPrivate\": False\n}\n\nwith open(\"/kaggle/working/dataset-metadata.json\", \"w\") as f:\n    json.dump(metadata, f, indent=4)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:18.981356Z","iopub.execute_input":"2025-02-28T15:50:18.981614Z","iopub.status.idle":"2025-02-28T15:50:19.00458Z","shell.execute_reply.started":"2025-02-28T15:50:18.981592Z","shell.execute_reply":"2025-02-28T15:50:19.003664Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploratory Data Analysis (EDA)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:19.007399Z","iopub.execute_input":"2025-02-28T15:50:19.007711Z","iopub.status.idle":"2025-02-28T15:50:19.02601Z","shell.execute_reply.started":"2025-02-28T15:50:19.007677Z","shell.execute_reply":"2025-02-28T15:50:19.025058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load dataset\ntrain_sequences_path = '/kaggle/input/stanford-rna-3d-folding/train_sequences.csv'\ntrain_labels_path = '/kaggle/input/stanford-rna-3d-folding/train_labels.csv'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:19.027354Z","iopub.execute_input":"2025-02-28T15:50:19.027583Z","iopub.status.idle":"2025-02-28T15:50:19.045212Z","shell.execute_reply.started":"2025-02-28T15:50:19.027564Z","shell.execute_reply":"2025-02-28T15:50:19.044292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndf_sequences = pd.read_csv(train_sequences_path)\ndf_labels = pd.read_csv(train_labels_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:19.046126Z","iopub.execute_input":"2025-02-28T15:50:19.04646Z","iopub.status.idle":"2025-02-28T15:50:19.265446Z","shell.execute_reply.started":"2025-02-28T15:50:19.04643Z","shell.execute_reply":"2025-02-28T15:50:19.264359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display first few rows\nprint(\"Train Sequences:\")\ndf_sequences.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:19.266459Z","iopub.execute_input":"2025-02-28T15:50:19.266731Z","iopub.status.idle":"2025-02-28T15:50:19.277735Z","shell.execute_reply.started":"2025-02-28T15:50:19.266709Z","shell.execute_reply":"2025-02-28T15:50:19.276896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nTrain Labels:\")\ndf_labels.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:19.278669Z","iopub.execute_input":"2025-02-28T15:50:19.279015Z","iopub.status.idle":"2025-02-28T15:50:19.306841Z","shell.execute_reply.started":"2025-02-28T15:50:19.27899Z","shell.execute_reply":"2025-02-28T15:50:19.305683Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# RNA sequence length distribution","metadata":{}},{"cell_type":"code","source":"\ndf_sequences['sequence_length'] = df_sequences['sequence'].apply(len)\nsns.histplot(df_sequences['sequence_length'], bins=30, kde=True)\nplt.xlabel('Sequence Length')\nplt.ylabel('Frequency')\nplt.title('RNA Sequence Length Distribution')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:19.307779Z","iopub.execute_input":"2025-02-28T15:50:19.308189Z","iopub.status.idle":"2025-02-28T15:50:19.939315Z","shell.execute_reply.started":"2025-02-28T15:50:19.308142Z","shell.execute_reply":"2025-02-28T15:50:19.938347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Nucleotide composition\ndef nucleotide_counts(seq):\n    return pd.Series({\n        'A': seq.count('A'),\n        'C': seq.count('C'),\n        'G': seq.count('G'),\n        'U': seq.count('U')\n    })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:19.940184Z","iopub.execute_input":"2025-02-28T15:50:19.940483Z","iopub.status.idle":"2025-02-28T15:50:19.944555Z","shell.execute_reply.started":"2025-02-28T15:50:19.940445Z","shell.execute_reply":"2025-02-28T15:50:19.943921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nucleotide_df = df_sequences['sequence'].apply(nucleotide_counts)\nnucleotide_df.mean().plot(kind='bar', color=['red', 'blue', 'green', 'orange'])\nplt.xlabel('Nucleotide')\nplt.ylabel('Average Count per Sequence')\nplt.title('Average Nucleotide Composition in RNA Sequences')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:19.94544Z","iopub.execute_input":"2025-02-28T15:50:19.945656Z","iopub.status.idle":"2025-02-28T15:50:20.339501Z","shell.execute_reply.started":"2025-02-28T15:50:19.945636Z","shell.execute_reply":"2025-02-28T15:50:20.338476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Checking missing values\nprint(\"\\nMissing Values in Train Sequences:\")\nprint(df_sequences.isnull().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:20.340633Z","iopub.execute_input":"2025-02-28T15:50:20.340987Z","iopub.status.idle":"2025-02-28T15:50:20.347824Z","shell.execute_reply.started":"2025-02-28T15:50:20.340953Z","shell.execute_reply":"2025-02-28T15:50:20.347026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nMissing Values in Train Labels:\")\nprint(df_labels.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:20.348562Z","iopub.execute_input":"2025-02-28T15:50:20.348764Z","iopub.status.idle":"2025-02-28T15:50:20.391129Z","shell.execute_reply.started":"2025-02-28T15:50:20.348745Z","shell.execute_reply":"2025-02-28T15:50:20.390302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dense, Embedding, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom sklearn.model_selection import train_test_split\n\n# ✅ Load dataset\ntrain_sequences_path = '/kaggle/input/stanford-rna-3d-folding/train_sequences.csv'\ntrain_labels_path = '/kaggle/input/stanford-rna-3d-folding/train_labels.csv'\n\ndf_sequences = pd.read_csv(train_sequences_path)\ndf_labels = pd.read_csv(train_labels_path)\n\n# ✅ Check column names\nprint(\"Columns in df_sequences:\", df_sequences.columns)\nprint(\"Columns in df_labels:\", df_labels.columns)\n\n# ✅ Ensure 'sequence' column exists\nif 'sequence' not in df_sequences.columns:\n    raise KeyError(\"🚨 Error: 'sequence' column not found in df_sequences!\")\n\n# ✅ Handle missing values\ndf_sequences.fillna('', inplace=True)  # Replace NaN with empty strings\ndf_labels.fillna(df_labels.mean(), inplace=True)  # Fill numeric NaN with mean\n\n# ✅ Convert RNA sequences into numerical representation (A=0, C=1, G=2, U=3)\nnucleotide_map = {'A': 0, 'C': 1, 'G': 2, 'U': 3}\n\ndef encode_sequence(seq):\n    if not isinstance(seq, str) or seq.strip() == '':\n        return []  # Handle empty or invalid sequences\n    seq = seq.replace('-', '')  # Remove invalid characters\n    return [nucleotide_map.get(n, 0) for n in seq if n in nucleotide_map]\n\ndf_sequences['encoded_sequence'] = df_sequences['sequence'].apply(encode_sequence)\n\n# ✅ Remove empty sequences\ndf_sequences = df_sequences[df_sequences['encoded_sequence'].apply(len) > 0]\n\n# ✅ Check if valid sequences exist\nif df_sequences.empty:\n    raise ValueError(\"🚨 Error: No valid RNA sequences found after preprocessing!\")\n\n# ✅ Pad sequences to the same length\nmax_seq_length = max(df_sequences['encoded_sequence'].apply(len))\nX = tf.keras.preprocessing.sequence.pad_sequences(df_sequences['encoded_sequence'], maxlen=max_seq_length, padding='post')\n\n# ✅ Extract 3D coordinate labels (first 3D point)\ncoordinate_columns = ['x_1', 'y_1', 'z_1']\nif not all(col in df_labels.columns for col in coordinate_columns):\n    raise KeyError(f\"🚨 Error: Missing coordinate columns {coordinate_columns} in df_labels!\")\n\ny = df_labels[coordinate_columns].values\n\n# ✅ Ensure matching samples\nif X.shape[0] != y.shape[0]:\n    raise ValueError(f\"🚨 Shape mismatch: X.shape[0]={X.shape[0]}, y.shape[0]={y.shape[0]}\")\n\n# ✅ Train-test split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# ✅ Build LSTM Model\nmodel = Sequential([\n    Embedding(input_dim=4, output_dim=128, input_length=max_seq_length),\n    LSTM(256, return_sequences=True, dropout=0.3),\n    LSTM(128, return_sequences=False, dropout=0.3),\n    Dense(64, activation='relu'),\n    Dropout(0.3),\n    Dense(3)  # Output: (x, y, z) coordinates\n])\n\n# ✅ Compile model\nmodel.compile(optimizer=Adam(learning_rate=0.0005), loss='mse', metrics=['mae'])\n\n# ✅ Train model\nhistory = model.fit(X_train, y_train, validation_data=(X_val, y_val), epochs=30, batch_size=32)\n\n# ✅ Save model & history\nmodel.save('/kaggle/working/rna_lstm_model.h5')\nnp.save('/kaggle/working/training_history.npy', history.history)\n\nprint(\"✅ Model training complete & saved!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-28T15:50:20.392105Z","iopub.execute_input":"2025-02-28T15:50:20.392394Z","iopub.status.idle":"2025-02-28T15:50:46.181805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_merged['encoded_sequence'].head())  # Check first few values\nprint(df_merged['encoded_sequence'].apply(len).describe())  # Check sequence lengths\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}