{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":87793,"databundleVersionId":11228175,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Ribonanza: My First RNA Structure Prediction Baseline (Beginner-Friendly!)\n\n**Hey everyone!👋**  I'm diving into the **Ribonanza competition**, and let me tell you😅, RNA structure prediction is completely new to me😭😭. I'm at that awkward beginner-intermediate stage – comfortable enough with Python and ML libraries to hack something together😅, but definitely no expert! So, I thought I'd share my initial baseline code, not as a *\"perfect solution,\"* but as a starting point for other beginners like me to learn and experiment with💪","metadata":{"_uuid":"5c7c37c3-3f61-45c7-a835-7fabdecaa5bc","_cell_guid":"447c8d58-2109-4d25-9b0a-9a01e45dcad2","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Core Inputs\nimport numpy as np\nimport pandas as pd \nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Deep Learning\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Dense, LSTM,Embedding, Dropout, concatenate\nfrom tensorflow.keras.optimizers import Adam\n\n# Machine Learning\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"_uuid":"4fff248b-0ae3-4380-8540-ec40a52f8979","_cell_guid":"c23aea57-e443-4c97-a872-0f75c32ae94d","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-12T04:30:32.006641Z","iopub.execute_input":"2025-03-12T04:30:32.006985Z","iopub.status.idle":"2025-03-12T04:30:44.251375Z","shell.execute_reply.started":"2025-03-12T04:30:32.006956Z","shell.execute_reply":"2025-03-12T04:30:44.250461Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 1: Load and preprocess data","metadata":{"_uuid":"6600b0d6-b53a-4ea8-8790-b48878b320ae","_cell_guid":"58c90a8f-2dc0-4a74-8fbc-e37da5d88279","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def load_data(sequence_file, label_file):\n    \"\"\"Loads the sequence and label data.\"\"\"\n    seq_df = pd.read_csv(sequence_file)\n    label_df = pd.read_csv(label_file)\n    return seq_df, label_df","metadata":{"_uuid":"ee0795d2-e96c-436c-9799-f05da298526f","_cell_guid":"7820e471-65e1-4e80-a6fc-4e5ad257f202","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-12T04:30:44.252450Z","iopub.execute_input":"2025-03-12T04:30:44.252992Z","iopub.status.idle":"2025-03-12T04:30:44.256994Z","shell.execute_reply.started":"2025-03-12T04:30:44.252966Z","shell.execute_reply":"2025-03-12T04:30:44.256147Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**My Thoughts:**  Alright, so this is where things get a bit tricky.  The sequences are just strings of A, C, G, and U, but the model can't understand that.  I need to convert them into numbers. Also, all the sequences and reactivity profiles need to be the same length, so let's pad them. I found out the train data had max length 206, so I'll pad all of them to that.","metadata":{"_uuid":"70d1e24e-cf93-4684-8e94-a8de1f877f08","_cell_guid":"5f4b23f5-c122-40a5-bfac-7c0286f11ca2","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"\ndef preprocess_sequences(seq_df, max_seq_len):\n    vocab = {'A': 0, 'C': 1, 'G': 2, 'U': 3}\n    seq_df['encoded_sequence'] = seq_df['sequence'].apply(lambda seq: [vocab[base] for base in seq])\n    seq_df['padded_sequence'] = tf.keras.preprocessing.sequence.pad_sequences(\n        seq_df['encoded_sequence'], maxlen=max_seq_len, padding='post'\n    ).tolist()\n    return seq_df","metadata":{"_uuid":"89fe521d-1ce4-4286-a76c-d04dcfd9bb92","_cell_guid":"645acd57-31b0-4920-b6a2-9265fb1bb639","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-12T04:30:44.259327Z","iopub.execute_input":"2025-03-12T04:30:44.259687Z","iopub.status.idle":"2025-03-12T04:30:44.292239Z","shell.execute_reply.started":"2025-03-12T04:30:44.259654Z","shell.execute_reply":"2025-03-12T04:30:44.291536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_labels(label_df, max_seq_len):\n    \"\"\"\n    Processes label data by grouping by target_id, sorting by resid, \n    and padding the list of coordinates to length max_seq_len.\n    \"\"\"\n    # Sort labels for each target by resid\n    label_df = label_df.sort_values('resid')\n    \n    # Group labels by target_id\n    grouped = label_df.groupby('target_id')\n    padded_coords = {}\n    \n    for target_id, group in grouped:\n        # Extract the coordinates for the first experimental structure (e.g., x_1, y_1, z_1)\n        coords = group[['x_1', 'y_1', 'z_1']].values.tolist()\n        # Pad the list of coordinates so that each target has max_seq_len residues.\n        padded = tf.keras.preprocessing.sequence.pad_sequences(\n            [coords],\n            maxlen=max_seq_len,\n            dtype='float32',\n            padding='post',\n            value=0.0\n        )[0]\n        # Flatten the array to get a vector of length max_seq_len*3\n        padded_coords[target_id] = padded.flatten()\n    return padded_coords","metadata":{"_uuid":"a5d1f39a-c945-4071-99fd-bd8302913b58","_cell_guid":"037de1ea-98a0-42da-8809-4c7d2b7825da","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-12T04:30:44.293310Z","iopub.execute_input":"2025-03-12T04:30:44.293587Z","iopub.status.idle":"2025-03-12T04:30:44.306089Z","shell.execute_reply.started":"2025-03-12T04:30:44.293566Z","shell.execute_reply":"2025-03-12T04:30:44.305433Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def merge_data(seq_df, padded_coords, max_seq_len):\n    \"\"\"\n    Merges the sequence DataFrame with corresponding padded coordinates.\n    \"\"\"\n    seq_df['padded_coordinates'] = seq_df['target_id'].apply(\n        lambda tid: padded_coords.get(tid, np.zeros(max_seq_len * 3))\n    )\n    return seq_df","metadata":{"_uuid":"17dbdf14-5edb-48f4-919d-416677e2011a","_cell_guid":"c17b1ee9-057b-434c-a3c1-2cb458ed5665","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-12T04:30:44.306719Z","iopub.execute_input":"2025-03-12T04:30:44.306990Z","iopub.status.idle":"2025-03-12T04:30:44.324476Z","shell.execute_reply.started":"2025-03-12T04:30:44.306971Z","shell.execute_reply":"2025-03-12T04:30:44.323760Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 2: Build the model\n\nI'm starting with a pretty basic LSTM network because I've used those before. LSTMs should be able to detect relationships between sequence position and reactivity. I am still wrapping my head around all of this, so please bare with me 😅😭","metadata":{"_uuid":"8081f3c8-62a0-43d8-b821-e6abe9295d24","_cell_guid":"65fcbc2b-a729-4abd-9e20-ec558d6fe4d4","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Define the GPU device\ndevice_name = tf.test.gpu_device_name()","metadata":{"_uuid":"0f52f75e-2de2-450d-b26b-7a265655a8a8","_cell_guid":"465dbe63-a90b-4b76-9258-1a94ac0acb6f","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-12T04:30:44.325207Z","iopub.execute_input":"2025-03-12T04:30:44.325448Z","iopub.status.idle":"2025-03-12T04:30:45.081830Z","shell.execute_reply.started":"2025-03-12T04:30:44.325415Z","shell.execute_reply":"2025-03-12T04:30:45.081045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Use the GPU device for model creation and training\nwith tf.device(device_name):\n    def build_model(max_seq_len, max_coord_len, dropout_rate=0.2):\n        \"\"\"\n        Builds an LSTM-based model that predicts per-residue coordinates.\n        \n        Args:\n            max_seq_len: Maximum sequence length.\n            max_coord_len: Number of coordinates per residue (typically 3).\n            dropout_rate: Dropout rate.\n            \n        Returns:\n            A compiled Keras model.\n        \"\"\"\n        # Input layer for the sequence\n        sequence_input = Input(shape=(max_seq_len,), name='sequence_input')\n        \n        # Embedding layer to convert nucleotide indices to dense vectors\n        x = Embedding(input_dim=4, output_dim=64)(sequence_input)\n        \n        # LSTM layer with dropout enabled\n        x = LSTM(128, dropout=dropout_rate, recurrent_dropout=dropout_rate)(x)\n        \n        # Dense layer with additional dropout (active during inference for MC dropout)\n        x = Dense(256, activation='relu')(x)\n        x = Dropout(dropout_rate)(x, training=True)\n        \n        # Output layer: Predict per-residue coordinates (flattened)\n        output = Dense(max_seq_len * max_coord_len)(x)\n        \n        model = Model(inputs=sequence_input, outputs=output)\n        optimizer = Adam(learning_rate=1e-4, clipnorm=1.0)\n        model.compile(optimizer=optimizer, loss='mse')\n        return model","metadata":{"_uuid":"2dcf3883-e9d1-4849-b1fc-bbcf35898c37","_cell_guid":"74016a88-ddc6-41d1-a816-03a0809e2000","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-12T04:30:45.082657Z","iopub.execute_input":"2025-03-12T04:30:45.082956Z","iopub.status.idle":"2025-03-12T04:30:45.104359Z","shell.execute_reply.started":"2025-03-12T04:30:45.082922Z","shell.execute_reply":"2025-03-12T04:30:45.103624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_with_uncertainty(model, X_test, num_samples=5):\n    \"\"\"\n    Generates multiple predictions per test sample using Monte Carlo dropout.\n    \n    Args:\n        model: The trained Keras model.\n        X_test: Test sequences (numpy array of shape (num_test_samples, max_seq_len)).\n        num_samples: Number of stochastic forward passes (i.e. predicted structures) per sample.\n        \n    Returns:\n        A numpy array of shape (num_test_samples, num_samples, max_seq_len * 3).\n    \"\"\"\n    predictions = []\n    for _ in range(num_samples):\n        # Use training=True to ensure dropout is active.\n        preds = model(X_test, training=True)\n        predictions.append(preds.numpy())\n    return np.stack(predictions, axis=1)","metadata":{"_uuid":"57665846-e97a-44f7-a2e5-b7a06ea7d9ba","_cell_guid":"fc9f0929-e701-41d6-b573-eb3f2738f909","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-12T04:30:45.105017Z","iopub.execute_input":"2025-03-12T04:30:45.105200Z","iopub.status.idle":"2025-03-12T04:30:45.122562Z","shell.execute_reply.started":"2025-03-12T04:30:45.105185Z","shell.execute_reply":"2025-03-12T04:30:45.121927Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 3: Train the model","metadata":{"_uuid":"828b1737-575e-4caf-a7cd-037d3663cf6e","_cell_guid":"885339db-ee75-423b-8138-9a9d66b32e1f","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def train_model(model, X_train, y_train, epochs=10, batch_size=64):\n    \"\"\"Trains the model.\"\"\"\n    model.fit(X_train, y_train, epochs=epochs, batch_size=batch_size, validation_split=0.2)","metadata":{"_uuid":"383de656-f4d0-484e-8295-1f171d1bc8bb","_cell_guid":"7ac1e376-721f-43e7-a43b-6df11a521ec1","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-12T04:30:45.125243Z","iopub.execute_input":"2025-03-12T04:30:45.125489Z","iopub.status.idle":"2025-03-12T04:30:45.139464Z","shell.execute_reply.started":"2025-03-12T04:30:45.125469Z","shell.execute_reply":"2025-03-12T04:30:45.138827Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 4: Create Submission File\nThis is probably the most important part... creating the submission file in the right format so Kaggle can score it. This part tripped me up a bit I'm not gonna lie😅😭😭... \n\nThank you ChatGPT 😅","metadata":{"_uuid":"62b1ffaa-19f3-450d-8fad-83bd436ecd06","_cell_guid":"04c4c2f8-100c-414f-83f0-c16bd5da2cef","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def create_submission(predictions, test_seq_df, sample_submission_path, output_file='submission.csv'):\n    \"\"\"\n    Creates the submission file in the required format.\n    \n    For each target sequence, each residue is assigned five sets of coordinates (one for each prediction).\n    The resulting CSV will follow the format:\n      ID,resname,resid,x_1,y_1,z_1,...,x_5,y_5,z_5\n\n    Args:\n        predictions: A numpy array with shape (num_test_samples, num_samples, max_seq_len*3).\n        test_seq_df: DataFrame for test sequences containing 'target_id' and 'sequence' columns.\n        sample_submission_path: Path to the sample_submission.csv (to enforce column order).\n        output_file: Name of the output CSV file.\n    \"\"\"\n    submission_rows = []\n    # Determine padded length based on the flattened size (each sample: max_seq_len*3).\n    num_samples = predictions.shape[1]\n    padded_length = predictions.shape[2]\n    max_seq_len = padded_length // 3\n\n    # Iterate over each test sequence.\n    for idx, row in test_seq_df.iterrows():\n        target_id = row['target_id']\n        sequence = row['sequence']\n        seq_len = len(sequence)\n        # Get predictions for this test sample and reshape to (num_samples, max_seq_len, 3).\n        sample_preds = predictions[idx].reshape((num_samples, max_seq_len, 3))\n        \n        # For each residue in the actual (unpadded) sequence:\n        for resid in range(seq_len):\n            row_dict = {\n                'ID': f\"{target_id}_{resid+1}\",\n                'resname': sequence[resid],\n                'resid': resid+1\n            }\n            # Add the x, y, z for each of the predicted structures.\n            for s in range(num_samples):\n                coords = sample_preds[s, resid, :]\n                row_dict[f'x_{s+1}'] = coords[0]\n                row_dict[f'y_{s+1}'] = coords[1]\n                row_dict[f'z_{s+1}'] = coords[2]\n            submission_rows.append(row_dict)\n            \n    submission_df = pd.DataFrame(submission_rows)\n    # Enforce the column order based on sample_submission.csv.\n    sample_sub = pd.read_csv(sample_submission_path)\n    submission_df = submission_df[sample_sub.columns]\n    submission_df.to_csv(output_file, index=False)\n    print(f\"Submission file '{output_file}' created successfully!\")\n    return submission_df","metadata":{"_uuid":"1189585f-8e36-466f-bfd9-489f22503b7b","_cell_guid":"9b5927ee-b948-4ef4-ad64-17e2f409a6bd","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-12T04:30:45.140515Z","iopub.execute_input":"2025-03-12T04:30:45.140696Z","iopub.status.idle":"2025-03-12T04:30:45.148852Z","shell.execute_reply.started":"2025-03-12T04:30:45.140680Z","shell.execute_reply":"2025-03-12T04:30:45.148104Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Main Section: The Main Part\nThis part is like the glue that holds everything together (Me trying to be professional 😅😅)... it calls all the functions and runs the whole process.","metadata":{"_uuid":"4aeaa332-bfd0-48f0-b6d8-a1ea733d670f","_cell_guid":"d615777a-b445-4d5a-8798-2e8c6f7f81a9","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# 1. File paths\ntrain_sequence_file = '/kaggle/input/stanford-rna-3d-folding/train_sequences.csv'\ntrain_label_file = '/kaggle/input/stanford-rna-3d-folding/train_labels.csv'\ntest_sequence_file = '/kaggle/input/stanford-rna-3d-folding/test_sequences.csv'\nsample_submission_path = '/kaggle/input/stanford-rna-3d-folding/sample_submission.csv'","metadata":{"_uuid":"94977e77-cdf2-4f3e-952f-ad0e111dc741","_cell_guid":"c4ea7dfd-9819-4c07-a63c-8f13c0c7b946","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-12T04:30:45.149513Z","iopub.execute_input":"2025-03-12T04:30:45.149767Z","iopub.status.idle":"2025-03-12T04:30:45.165737Z","shell.execute_reply.started":"2025-03-12T04:30:45.149732Z","shell.execute_reply":"2025-03-12T04:30:45.165010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example usage:\n# Load your sequence and label data\ntrain_seq_df = pd.read_csv(train_sequence_file)\ntrain_label_df = pd.read_csv(train_label_file)\n\n# Create a common target_id in label dataframe (if needed)\ntrain_label_df['target_id'] = train_label_df['ID'].str.extract(r'^(.*)_\\d+$')[0]\n\ntrain_label_df['x_1'].fillna(80.4, inplace=True)\ntrain_label_df['y_1'].fillna(84.0, inplace=True)\ntrain_label_df['z_1'].fillna(98.6, inplace=True)\n\nvalid_bases = {'A', 'C', 'G', 'U'}\ntrain_seq_df = train_seq_df[train_seq_df['sequence'].apply(lambda seq: all(base in valid_bases for base in seq))]\n\n# Determine max sequence length (e.g., maximum length among training sequences)\nMAX_SEQ_LEN = int(train_seq_df['sequence'].apply(len).max())\n\n# Preprocess sequences and labels\ntrain_seq_df = preprocess_sequences(train_seq_df, MAX_SEQ_LEN)\npadded_coords = preprocess_labels(train_label_df, MAX_SEQ_LEN)\ntrain_seq_df = merge_data(train_seq_df, padded_coords, MAX_SEQ_LEN)\n\n# Create training arrays: X for sequences, y for coordinates (flattened)\nX = np.stack(train_seq_df['padded_sequence'].values)\ny = np.stack(train_seq_df['padded_coordinates'].values)","metadata":{"_uuid":"5b87343f-2264-4e31-8dca-185e24c82ebe","_cell_guid":"be3b2f18-311e-4f7a-afdc-f4e5a406de13","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-12T04:30:45.166631Z","iopub.execute_input":"2025-03-12T04:30:45.166999Z","iopub.status.idle":"2025-03-12T04:30:46.778059Z","shell.execute_reply.started":"2025-03-12T04:30:45.166886Z","shell.execute_reply":"2025-03-12T04:30:46.777333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Detect NaNs in the dataset\nprint(train_seq_df.isna().sum())","metadata":{"_uuid":"d7853586-59c2-4bb0-8201-2feda7ecc7d6","_cell_guid":"3d352569-5e6b-47aa-b880-d6b078734cbe","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-12T04:30:46.778791Z","iopub.execute_input":"2025-03-12T04:30:46.778992Z","iopub.status.idle":"2025-03-12T04:30:46.784928Z","shell.execute_reply.started":"2025-03-12T04:30:46.778975Z","shell.execute_reply":"2025-03-12T04:30:46.784067Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Detect NaNs in the dataset\nprint(train_label_df.isna().sum())","metadata":{"_uuid":"6cdf0af1-7be2-4009-aa0e-63fcb81b9805","_cell_guid":"33fbc9de-1179-408c-bec3-34fc21cb9db4","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-12T04:30:46.785668Z","iopub.execute_input":"2025-03-12T04:30:46.786001Z","iopub.status.idle":"2025-03-12T04:30:46.820676Z","shell.execute_reply.started":"2025-03-12T04:30:46.785978Z","shell.execute_reply":"2025-03-12T04:30:46.819853Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check if the GPU device exists\nif device_name != '/device:GPU:0':\n    print(f\"GPU device not found: {device_name}\")\nelse:\n    print(f\"Using GPU: {device_name}\")","metadata":{"_uuid":"4a61ca6f-ebe0-49e8-b07b-311ca12e3620","_cell_guid":"5a47ab60-b6f4-41db-955e-b28d50632962","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-12T04:30:46.821399Z","iopub.execute_input":"2025-03-12T04:30:46.821601Z","iopub.status.idle":"2025-03-12T04:30:46.826614Z","shell.execute_reply.started":"2025-03-12T04:30:46.821583Z","shell.execute_reply":"2025-03-12T04:30:46.825836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 6. Build model\nmodel = build_model(MAX_SEQ_LEN, 3, dropout_rate=0.3)\n\n# 7. Train model\ntrain_model(model, X, y)","metadata":{"_uuid":"4549a32c-8cdd-4ad5-bbdf-fefc343382b4","_cell_guid":"121440e6-613b-4ef8-8558-17d367c0756f","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-12T04:30:46.827353Z","iopub.execute_input":"2025-03-12T04:30:46.827576Z","iopub.status.idle":"2025-03-12T04:31:20.591496Z","shell.execute_reply.started":"2025-03-12T04:30:46.827557Z","shell.execute_reply":"2025-03-12T04:31:20.590837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and preprocess test data\n# 8. Load test data\ntest_seq_df = load_data(test_sequence_file, '/kaggle/input/stanford-rna-3d-folding/sample_submission.csv')[0]\nvocab = {'A': 0, 'C': 1, 'G': 2, 'U': 3}\ntest_seq_df['encoded_sequence'] = test_seq_df['sequence'].apply(lambda seq: [vocab[base] for base in seq])\ntest_seq_df['padded_sequence'] = tf.keras.preprocessing.sequence.pad_sequences(test_seq_df['encoded_sequence'], maxlen=MAX_SEQ_LEN, padding='post').tolist()\nX_test = np.stack(test_seq_df['padded_sequence'].values)\n\n# 9. Prediction with model\npredictions = predict_with_uncertainty(model, X_test, num_samples=5)","metadata":{"_uuid":"0e59b08f-8b9d-4ce9-88c0-a35883a782f1","_cell_guid":"d324c4e3-deef-40fb-a07c-62ce9681a024","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-12T04:31:20.592478Z","iopub.execute_input":"2025-03-12T04:31:20.592710Z","iopub.status.idle":"2025-03-12T04:31:20.977829Z","shell.execute_reply.started":"2025-03-12T04:31:20.592688Z","shell.execute_reply":"2025-03-12T04:31:20.976921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create submission file with sample_submission.csv\nsubmission_df = create_submission(predictions, test_seq_df, sample_submission_path)","metadata":{"_uuid":"2d432700-9894-4559-9d95-853b68f51a2f","_cell_guid":"64bd8504-804f-4299-908f-6f18cc4f78e4","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-12T04:31:20.978801Z","iopub.execute_input":"2025-03-12T04:31:20.979146Z","iopub.status.idle":"2025-03-12T04:31:21.058326Z","shell.execute_reply.started":"2025-03-12T04:31:20.979123Z","shell.execute_reply":"2025-03-12T04:31:21.057599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df","metadata":{"_uuid":"d2716d6f-abc3-43dd-b4f6-c3b1fd343648","_cell_guid":"9876ae28-8a1d-47b5-9a20-a7d533bbc1db","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-12T04:31:21.059438Z","iopub.execute_input":"2025-03-12T04:31:21.059801Z","iopub.status.idle":"2025-03-12T04:31:21.087434Z","shell.execute_reply.started":"2025-03-12T04:31:21.059765Z","shell.execute_reply":"2025-03-12T04:31:21.086601Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate predictions\npredictions = predict_with_uncertainty(model, X_test, num_samples=5)\n\n# Check predictions for NaNs\nprint(\"Predictions shape:\", predictions.shape)\nprint(\"Minimum prediction value:\", np.nanmin(predictions))\nprint(\"Maximum prediction value:\", np.nanmax(predictions))\nprint(\"Number of NaNs in predictions:\", np.isnan(predictions).sum())\n\n# If NaNs are detected, consider retraining or adjusting hyperparameters.","metadata":{"_uuid":"70c1083c-2744-4a7a-808e-8fcfd47b81d6","_cell_guid":"0fe6de5e-22dd-4389-ab28-a1a633d4c77c","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-03-12T04:31:21.088340Z","iopub.execute_input":"2025-03-12T04:31:21.088659Z","iopub.status.idle":"2025-03-12T04:31:21.424655Z","shell.execute_reply.started":"2025-03-12T04:31:21.088628Z","shell.execute_reply":"2025-03-12T04:31:21.423791Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null}]}