{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11228175,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\"\"\"\nThis script builds an ensemble of very small CNN models to predict 3D coordinates\nfor RNA nucleotide residues based on training data. It:\n  - Loads training labels (with per-residue coordinates) and sequences.\n  - For each target, uses the number of coordinate rows (residues) as the effective length and slices\n    the full sequence accordingly.\n  - Creates per-residue features: one-hot encoding of nucleotide and normalized position.\n  - Pads all sequences to a global maximum length.\n  - Defines a small CNN model using a Masking layer, a 1D convolution, and TimeDistributed dense layers.\n  - Trains five separate CNN models (an ensemble) on the training set.\n  - Loads test sequences, creates and pads features, and uses each model to predict coordinates.\n  - For each residue in each test sequence, collects the five coordinate predictions and writes a submission CSV.\n\"\"\"\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Masking, Conv1D, TimeDistributed, Dense\nimport tensorflow.keras.backend as K\nimport gc\n\n# --- Helper functions ---\n\ndef one_hot_encode(seq):\n    \"\"\"\n    Given an RNA sequence string, return a numpy array of shape (L, 4)\n    with one-hot encoding for nucleotides: A, C, G, U.\n    Unrecognized characters are encoded as zeros.\n    \"\"\"\n    mapping = {\n        'A': [1, 0, 0, 0],\n        'C': [0, 1, 0, 0],\n        'G': [0, 0, 1, 0],\n        'U': [0, 0, 0, 1]\n    }\n    seq = seq.upper().strip()\n    return np.array([mapping.get(nuc, [0, 0, 0, 0]) for nuc in seq], dtype=np.float32)\n\ndef prepare_train_data():\n    \"\"\"\n    Loads training labels and sequences.\n    Merges them using target_id (extracted from the ID column in train_labels.csv).\n    For each target, uses the number of label rows as the effective length,\n    and slices the full sequence to that length.\n    For each sequence, creates features for each residue:\n      - One-hot encoding (4 dims)\n      - Normalized residue index (1 dim)\n    Returns:\n      - X_list: list of (L, 5) arrays (features for each target)\n      - y_list: list of (L, 3) arrays (coordinates per residue)\n      - lengths: list of effective lengths L (number of residues)\n    \"\"\"\n    labels = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\n    seqs = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\n    \n    # Create target_id from the ID column (e.g. \"1SCL_A_1\" -> \"1SCL_A\")\n    labels['target_id'] = labels['ID'].apply(lambda x: \"_\".join(x.split('_')[:-1]))\n    \n    # Merge labels with sequences on target_id.\n    train = pd.merge(labels, seqs[['target_id', 'sequence']], on='target_id', how='left')\n    # Drop rows missing coordinates or sequence.\n    train = train.dropna(subset=['x_1', 'y_1', 'z_1', 'sequence'])\n    \n    X_list, y_list, lengths = [], [], []\n    # Group by target_id. Each group corresponds to one RNA target.\n    for target_id, group in train.groupby('target_id'):\n        group = group.sort_values('resid')\n        # Use the number of rows as the effective length.\n        L = group.shape[0]\n        # Slice the sequence to the effective length.\n        seq = group['sequence'].iloc[0].strip()[:L]\n        lengths.append(L)\n        # Create features: one-hot encoding (4 dims) and normalized position (1 dim).\n        onehot = one_hot_encode(seq)             # shape (L, 4)\n        norm_pos = np.array([[ (i+1)/L ] for i in range(L)], dtype=np.float32)  # shape (L, 1)\n        features = np.hstack([onehot, norm_pos])   # shape (L, 5)\n        X_list.append(features)\n        # Get coordinate targets as (L, 3) array (from columns x_1, y_1, z_1).\n        coords = group[['x_1','y_1','z_1']].to_numpy(dtype=np.float32)\n        y_list.append(coords)\n    return X_list, y_list, lengths\n\ndef prepare_test_data():\n    \"\"\"\n    Loads test sequences and builds per-residue features.\n    Returns:\n      - X_test_list: list of (L,5) feature arrays for each test sequence\n      - test_ids: list of target_id strings\n      - test_sequences: dict mapping target_id to the raw sequence string\n      - lengths: dict mapping target_id to sequence length\n    \"\"\"\n    test_df = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/test_sequences.csv\")\n    X_test_list = []\n    test_ids = []\n    test_sequences = {}\n    lengths = {}\n    for _, row in test_df.iterrows():\n        target_id = row['target_id']\n        seq = row['sequence'].strip()\n        L = len(seq)\n        lengths[target_id] = L\n        test_ids.append(target_id)\n        test_sequences[target_id] = seq\n        onehot = one_hot_encode(seq)  # shape (L,4)\n        norm_pos = np.array([[ (i+1)/L ] for i in range(L)], dtype=np.float32)  # shape (L,1)\n        features = np.hstack([onehot, norm_pos])  # shape (L,5)\n        X_test_list.append(features)\n    return X_test_list, test_ids, test_sequences, lengths\n\ndef pad_data(X_list, y_list, max_len):\n    \"\"\"\n    Pads each array in X_list and y_list (which are arrays of shape (L, d))\n    along the time (first) dimension to max_len.\n    For X, pad with zeros; for y, pad with zeros as well.\n    Returns numpy arrays of shape (n_samples, max_len, d).\n    \"\"\"\n    X_padded = []\n    y_padded = []\n    for i in range(len(X_list)):\n        X = X_list[i]\n        y = y_list[i]\n        L = X.shape[0]\n        pad_width = max_len - L\n        if pad_width > 0:\n            X_pad = np.pad(X, ((0, pad_width), (0, 0)), mode='constant', constant_values=0)\n            y_pad = np.pad(y, ((0, pad_width), (0, 0)), mode='constant', constant_values=0)\n        else:\n            X_pad = X\n            y_pad = y\n        X_padded.append(X_pad)\n        y_padded.append(y_pad)\n    return np.array(X_padded, dtype=np.float32), np.array(y_padded, dtype=np.float32)\n\ndef pad_test_data(X_list, max_len):\n    \"\"\"\n    Pads each array in X_list (for test data) to max_len.\n    Returns a numpy array of shape (n_samples, max_len, d).\n    \"\"\"\n    X_padded = []\n    for X in X_list:\n        L = X.shape[0]\n        pad_width = max_len - L\n        if pad_width > 0:\n            X_pad = np.pad(X, ((0, pad_width), (0, 0)), mode='constant', constant_values=0)\n        else:\n            X_pad = X\n        X_padded.append(X_pad)\n    return np.array(X_padded, dtype=np.float32)\n\ndef build_model(max_len, feature_dim=5):\n    \"\"\"\n    Builds a very small CNN model.\n    Input shape is (max_len, feature_dim) and output is (max_len, 3) for the coordinates.\n    A Masking layer is applied to ignore padded timesteps.\n    \"\"\"\n    model = Sequential([\n        Masking(mask_value=0.0, input_shape=(max_len, feature_dim)),\n        Conv1D(filters=16, kernel_size=3, padding='same', activation='relu'),\n        TimeDistributed(Dense(16, activation='relu')),\n        TimeDistributed(Dense(3))\n    ])\n    model.compile(optimizer='adam', loss='mse')\n    return model\n\n# --- Main training and prediction workflow ---\n\ndef main():\n    # Prepare training data.\n    X_train_list, y_train_list, train_lengths = prepare_train_data()\n    # Prepare test data.\n    X_test_list, test_ids, test_sequences, test_lengths_dict = prepare_test_data()\n    \n    # Determine global maximum sequence length from both training and test sets.\n    max_train = max(train_lengths) if train_lengths else 0\n    max_test = max(test_lengths_dict.values()) if test_lengths_dict else 0\n    global_max_len = max(max_train, max_test)\n    print(\"Global max sequence length:\", global_max_len)\n    \n    # Pad training and test data to global_max_len.\n    X_train, y_train = pad_data(X_train_list, y_train_list, global_max_len)\n    X_test = pad_test_data(X_test_list, global_max_len)\n    \n    # Ensemble: train 5 separate CNN models one at a time to save resources.\n    n_models = 5\n    predictions_ensemble = []  # to store predictions from each model on test data\n    epochs = 10  # adjust epochs as needed\n    batch_size = 2\n    \n    for m in range(n_models):\n        print(f\"\\nTraining CNN model {m+1}/{n_models}\")\n        # Build and train the model.\n        model = build_model(global_max_len, feature_dim=5)\n        model.fit(X_train, y_train, epochs=epochs, batch_size=batch_size, verbose=1)\n        \n        # Predict on test data; result shape: (n_test, global_max_len, 3)\n        pred = model.predict(X_test)\n        predictions_ensemble.append(pred)\n        \n        # Clear model from memory to free resources.\n        del model\n        K.clear_session()\n        gc.collect()\n    \n    # Build submission rows for each test sequence, using only non-padded positions.\n    submission_rows = []\n    n_test = len(X_test_list)\n    # The order of test_ids and test_sequences corresponds to X_test_list.\n    for i in range(n_test):\n        target_id = test_ids[i]\n        seq = test_sequences[target_id]\n        L = len(seq)\n        for j in range(L):\n            row = {}\n            row[\"ID\"] = f\"{target_id}_{j+1}\"\n            row[\"resname\"] = seq[j]\n            row[\"resid\"] = j+1\n            # For each model in the ensemble, record the prediction for residue j.\n            for m in range(n_models):\n                coord = predictions_ensemble[m][i, j]  # shape (3,)\n                row[f\"x_{m+1}\"] = coord[0]\n                row[f\"y_{m+1}\"] = coord[1]\n                row[f\"z_{m+1}\"] = coord[2]\n            submission_rows.append(row)\n    \n    submission = pd.DataFrame(submission_rows)\n    submission.to_csv(\"/kaggle/working/submission.csv\", index=False)\n    print(\"\\nSubmission file saved as submission.csv\")\n\nif __name__ == '__main__':\n    main()\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-09T14:33:40.877037Z","iopub.execute_input":"2025-03-09T14:33:40.877358Z"}},"outputs":[],"execution_count":null}]}