{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"environment":{"kernel":"conda-env-tensorflow-tensorflow","name":"workbench-notebooks.m113","type":"gcloud","uri":"gcr.io/deeplearning-platform-release/workbench-notebooks:m113"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":51294,"databundleVersionId":7331882,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<a href=\"https://www.kaggle.com/code/jeffreyesedo/1st-ribo-note?scriptVersionId=151222394\" target=\"_blank\"><img align=\"left\" alt=\"Kaggle\" title=\"Open in Kaggle\" src=\"https://kaggle.com/static/images/open-in-kaggle.svg\"></a>","metadata":{}},{"cell_type":"markdown","source":"# RNA Science Environment and Libraries","metadata":{}},{"cell_type":"code","source":"# Setting up an RNA Science Environment\n!pip install arnie\n!pip install draw_rna\n# !pip install viennarna\n!pip install swifter\n\n\n# Install EternaFold\n!conda config --set auto_update_conda false\n!conda install -c bioconda eternafold --yes\n\n# Install ViennaRNA with conda for arnie\n# The provided output shows you installed it with pip, but conda is generally\n# more reliable for setting up the binaries arnie needs.\n!conda install viennarna -c bioconda --yes\n\n# Manually setup EternaFold and Vienna for Kaggle notebook\n%env ETERNAFOLD_PATH=/opt/conda/bin/eternafold-bin\n%env ETERNAFOLD_PARAMETERS=/opt/conda/lib/eternafold-lib/parameters/EternaFoldParams.v1\n%env VIENNA_2_PATH = /opt/conda/bin\n%env TMP=/content/tmp","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:07:13.223467Z","iopub.execute_input":"2025-10-16T15:07:13.223889Z","iopub.status.idle":"2025-10-16T15:13:29.256329Z","shell.execute_reply.started":"2025-10-16T15:07:13.223840Z","shell.execute_reply":"2025-10-16T15:13:29.254904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!which RNAfold","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:13:29.259182Z","iopub.execute_input":"2025-10-16T15:13:29.259560Z","iopub.status.idle":"2025-10-16T15:13:30.452614Z","shell.execute_reply.started":"2025-10-16T15:13:29.259524Z","shell.execute_reply":"2025-10-16T15:13:30.450951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!which viennarna","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:13:30.454549Z","iopub.execute_input":"2025-10-16T15:13:30.455013Z","iopub.status.idle":"2025-10-16T15:13:31.626458Z","shell.execute_reply.started":"2025-10-16T15:13:30.454974Z","shell.execute_reply":"2025-10-16T15:13:31.624968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport re\nimport sys\nimport psutil\nimport gc\nimport random\nimport ast\nimport swifter\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport glob as glob\n\n\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm, trange\nfrom time import sleep","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T17:11:04.796949Z","iopub.execute_input":"2025-10-16T17:11:04.797412Z","iopub.status.idle":"2025-10-16T17:11:04.805158Z","shell.execute_reply.started":"2025-10-16T17:11:04.797378Z","shell.execute_reply":"2025-10-16T17:11:04.803485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import RNA\nimport arnie\nfrom arnie.mfe import mfe\nfrom arnie.bpps import bpps\nfrom arnie.pfunc import pfunc\nimport arnie.utils as utils\nfrom arnie.free_energy import free_energy\nfrom draw_rna.ipynb_draw import draw_struct","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:13:35.260262Z","iopub.execute_input":"2025-10-16T15:13:35.261002Z","iopub.status.idle":"2025-10-16T15:13:35.293319Z","shell.execute_reply.started":"2025-10-16T15:13:35.260962Z","shell.execute_reply":"2025-10-16T15:13:35.292351Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Import Datasets","metadata":{}},{"cell_type":"markdown","source":"## train data","metadata":{}},{"cell_type":"code","source":"train= pd.read_csv(\"/kaggle/input/stanford-ribonanza-rna-folding/train_data.csv\")\n\nprint(f\"Train dataset shape: {train.shape}\\n\")","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:13:35.294922Z","iopub.execute_input":"2025-10-16T15:13:35.295260Z","iopub.status.idle":"2025-10-16T15:15:30.923474Z","shell.execute_reply.started":"2025-10-16T15:13:35.295231Z","shell.execute_reply":"2025-10-16T15:15:30.922240Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## optimizing dataset for memory","metadata":{}},{"cell_type":"code","source":"def opt_num(df):\n    # optimize numerical data columns\n    df= df.copy()\n    \n    for col in df.columns:\n        df_col= df[col]\n        dn = df_col.dtype.name\n        \n        if dn == \"int64\":\n            df[col]= pd.to_numeric(df_col, downcast=\"integer\")\n        elif dn == \"float64\":\n            df[col]= pd.to_numeric(df_col, downcast=\"float\")\n        elif dn == \"object\":\n            num_unique_values = len(df_col.unique())\n            num_total_values = len(df_col)\n            if num_unique_values / num_total_values < 0.5:\n                df[col] = df_col.astype(\"category\")\n    return df","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:15:30.925171Z","iopub.execute_input":"2025-10-16T15:15:30.925520Z","iopub.status.idle":"2025-10-16T15:15:30.934136Z","shell.execute_reply.started":"2025-10-16T15:15:30.925489Z","shell.execute_reply":"2025-10-16T15:15:30.932608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"opt_train= opt_num(train)","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:15:30.935579Z","iopub.execute_input":"2025-10-16T15:15:30.935970Z","iopub.status.idle":"2025-10-16T15:15:59.750597Z","shell.execute_reply.started":"2025-10-16T15:15:30.935935Z","shell.execute_reply":"2025-10-16T15:15:59.749271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Train Dataset:{train.iloc[0:5, 0:10].info()}\")\nprint()\nprint()\nprint(f\"Optimized Dataset: {opt_train.iloc[0:5, 0:10].info()}\")","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:15:59.752865Z","iopub.execute_input":"2025-10-16T15:15:59.753345Z","iopub.status.idle":"2025-10-16T15:16:01.524920Z","shell.execute_reply.started":"2025-10-16T15:15:59.753300Z","shell.execute_reply":"2025-10-16T15:16:01.523718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del train\ngc.collect()","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:01.527216Z","iopub.execute_input":"2025-10-16T15:16:01.527646Z","iopub.status.idle":"2025-10-16T15:16:02.302313Z","shell.execute_reply.started":"2025-10-16T15:16:01.527601Z","shell.execute_reply":"2025-10-16T15:16:02.300775Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Exploration and Visualization","metadata":{}},{"cell_type":"code","source":"opt_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:02.303759Z","iopub.execute_input":"2025-10-16T15:16:02.304234Z","iopub.status.idle":"2025-10-16T15:16:02.350392Z","shell.execute_reply.started":"2025-10-16T15:16:02.304201Z","shell.execute_reply":"2025-10-16T15:16:02.349074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count columns based on their Dtype\n# dtype_counts = opt_train.dtypes.value_counts()\n# print(dtype_counts)","metadata":{"tags":[],"jupyter":{"source_hidden":true},"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:02.351950Z","iopub.execute_input":"2025-10-16T15:16:02.352291Z","iopub.status.idle":"2025-10-16T15:16:02.357753Z","shell.execute_reply.started":"2025-10-16T15:16:02.352262Z","shell.execute_reply":"2025-10-16T15:16:02.356271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"experiments_count= opt_train[\"experiment_type\"].value_counts()\nprint(experiments_count)","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:02.359286Z","iopub.execute_input":"2025-10-16T15:16:02.359659Z","iopub.status.idle":"2025-10-16T15:16:02.382058Z","shell.execute_reply.started":"2025-10-16T15:16:02.359628Z","shell.execute_reply":"2025-10-16T15:16:02.380713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rna_viz(data, experiment, package):\n    # Visualizing RNA  sequence base on experiment\n    exp_type= data[data.experiment_type == experiment]\n    seq_index= random.randint(0,len(exp_type.sequence))\n    \n    seq_exp = opt_train[data[\"experiment_type\"] == experiment].iloc[seq_index, 1:3]\n    print(seq_exp)\n    \n    structure = mfe(seq_exp.sequence,package=package)\n    print(structure)\n    \n    fig, axs = plt.subplots(1,1,  figsize=(8,7))\n    draw_struct(seq_exp.sequence, structure, ax=axs)\n    axs.set_title(seq_exp.experiment_type, loc='left', fontsize='medium')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:02.387140Z","iopub.execute_input":"2025-10-16T15:16:02.387490Z","iopub.status.idle":"2025-10-16T15:16:02.395148Z","shell.execute_reply.started":"2025-10-16T15:16:02.387461Z","shell.execute_reply":"2025-10-16T15:16:02.393905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualizing RNA  sequence for DMS MaP\nrna_viz(opt_train, \"DMS_MaP\",  \"eternafold\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:02.396551Z","iopub.execute_input":"2025-10-16T15:16:02.396883Z","iopub.status.idle":"2025-10-16T15:16:09.161042Z","shell.execute_reply.started":"2025-10-16T15:16:02.396855Z","shell.execute_reply":"2025-10-16T15:16:09.159597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualizing RNA  sequence for 2A3 MaP\nrna_viz(opt_train, \"2A3_MaP\",  \"eternafold\")","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:09.163641Z","iopub.execute_input":"2025-10-16T15:16:09.164031Z","iopub.status.idle":"2025-10-16T15:16:15.809488Z","shell.execute_reply.started":"2025-10-16T15:16:09.163987Z","shell.execute_reply":"2025-10-16T15:16:15.807955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seq_len= opt_train.sequence.apply(len)\nseq_len = seq_len.value_counts()\n# seq_len = pd.Series(seq_len)\n# seq_len\n\nseq_len.plot.bar()\nplt.xlabel(\"Sequences Lenght\")\nplt.title(\"Sequence lenght Distribution\")","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:15.811404Z","iopub.execute_input":"2025-10-16T15:16:15.811924Z","iopub.status.idle":"2025-10-16T15:16:16.557074Z","shell.execute_reply.started":"2025-10-16T15:16:15.811862Z","shell.execute_reply":"2025-10-16T15:16:16.555899Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"lengths of RNA sequence is between 115 to 206, while for the test the lengths are between 177 to 457.  Part of the challenge is to know whether the patterns recognized at length 115 to 206 will generalize to longer lengths [response found here.](https://www.kaggle.com/competitions/stanford-ribonanza-rna-folding/discussion/453147#2513582).","metadata":{}},{"cell_type":"code","source":"base= {\"A\": 0,\"C\":0,\"G\":0,\"U\":0}\n\nfor seq in opt_train.sequence:\n    for base_key in base.keys():\n        base[base_key] += seq.count(base_key)\n\n\nplt.bar(base.keys(), base.values())\nplt.xlabel('Base', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Count', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.title('Base Count', fontsize = 14, fontweight = 'bold', color = 'darkgreen')","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:16.558792Z","iopub.execute_input":"2025-10-16T15:16:16.559275Z","iopub.status.idle":"2025-10-16T15:16:23.435469Z","shell.execute_reply.started":"2025-10-16T15:16:16.559228Z","shell.execute_reply":"2025-10-16T15:16:23.434043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del seq_len\ngc.collect()","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:23.437507Z","iopub.execute_input":"2025-10-16T15:16:23.437894Z","iopub.status.idle":"2025-10-16T15:16:23.605887Z","shell.execute_reply.started":"2025-10-16T15:16:23.437848Z","shell.execute_reply":"2025-10-16T15:16:23.604507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"opt_train.reads.describe()","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:23.607499Z","iopub.execute_input":"2025-10-16T15:16:23.608144Z","iopub.status.idle":"2025-10-16T15:16:23.680567Z","shell.execute_reply.started":"2025-10-16T15:16:23.608094Z","shell.execute_reply":"2025-10-16T15:16:23.679364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"opt_train.signal_to_noise.describe()","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:23.682205Z","iopub.execute_input":"2025-10-16T15:16:23.682584Z","iopub.status.idle":"2025-10-16T15:16:23.762308Z","shell.execute_reply.started":"2025-10-16T15:16:23.682550Z","shell.execute_reply":"2025-10-16T15:16:23.761180Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"opt_train.SN_filter.describe()","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:23.764002Z","iopub.execute_input":"2025-10-16T15:16:23.764540Z","iopub.status.idle":"2025-10-16T15:16:23.806403Z","shell.execute_reply.started":"2025-10-16T15:16:23.764501Z","shell.execute_reply":"2025-10-16T15:16:23.804747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# checking number of columns NaN in reactivity and reactivity_error \nfloat_columns = opt_train.select_dtypes(include=['float'])\n\n# Columns that are NaN\nnum_empty_cols= 0\ncols_having_values=0\n\n\n# for col in float_columns.drop('signal_to_noise', axis=1):\nfor col in float_columns:\n    if float_columns[col].notna().sum() == 0:\n        num_empty_cols+=1\n    else:\n        cols_having_values+=1\n        \nprint(f\"Number of Columns with only NaN values: {num_empty_cols} of 412 columns\\n\")\nprint(f\"Number of Columns with values: {cols_having_values} of 412 columns\")","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:23.808056Z","iopub.execute_input":"2025-10-16T15:16:23.808394Z","iopub.status.idle":"2025-10-16T15:16:27.525459Z","shell.execute_reply.started":"2025-10-16T15:16:23.808365Z","shell.execute_reply":"2025-10-16T15:16:27.524079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del float_columns\ngc.collect()","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:27.527594Z","iopub.execute_input":"2025-10-16T15:16:27.528423Z","iopub.status.idle":"2025-10-16T15:16:27.673569Z","shell.execute_reply.started":"2025-10-16T15:16:27.528375Z","shell.execute_reply":"2025-10-16T15:16:27.671498Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Wrangling","metadata":{}},{"cell_type":"code","source":"def wrangle(df):\n    \n    # Drop rows based on SN Filter\n    df= df.loc[df.SN_filter == 1]\n    \n    # Drop duplicate\n    df= df.drop_duplicates(subset=[\"sequence_id\", \"experiment_type\"])\n    \n    \n    # Drop the columns \n    df= df.drop(columns=[\"reads\", \"signal_to_noise\",\"SN_filter\"], axis=1)\n    df= df.drop(columns=[col for col in df.columns if \"_error_\" in col], axis=1)\n  \n    # Set categories for categorical columns\n    for col in df.select_dtypes(include=\"category\"):\n        df[col] = df[col].cat.add_categories([0])\n    \n    \n    # Fill NaN value for reactivity & error\n    df= df[7:].fillna(0)\n    \n    # Reset index to start from 0\n    df.reset_index(drop=True, inplace= True)\n       \n      \n    return df","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:27.675952Z","iopub.execute_input":"2025-10-16T15:16:27.677199Z","iopub.status.idle":"2025-10-16T15:16:27.694858Z","shell.execute_reply.started":"2025-10-16T15:16:27.677152Z","shell.execute_reply":"2025-10-16T15:16:27.693299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get the clean dataset\ntrain_feat=  wrangle(opt_train)","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:27.696686Z","iopub.execute_input":"2025-10-16T15:16:27.697300Z","iopub.status.idle":"2025-10-16T15:16:33.626422Z","shell.execute_reply.started":"2025-10-16T15:16:27.697239Z","shell.execute_reply":"2025-10-16T15:16:33.625122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_feat.head()","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:33.628097Z","iopub.execute_input":"2025-10-16T15:16:33.628431Z","iopub.status.idle":"2025-10-16T15:16:33.664124Z","shell.execute_reply.started":"2025-10-16T15:16:33.628398Z","shell.execute_reply":"2025-10-16T15:16:33.662960Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_feat.info()","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:33.666543Z","iopub.execute_input":"2025-10-16T15:16:33.666921Z","iopub.status.idle":"2025-10-16T15:16:34.548337Z","shell.execute_reply.started":"2025-10-16T15:16:33.666888Z","shell.execute_reply":"2025-10-16T15:16:34.546958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del opt_train\ngc.collect()","metadata":{"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:34.549987Z","iopub.execute_input":"2025-10-16T15:16:34.550419Z","iopub.status.idle":"2025-10-16T15:16:34.680838Z","shell.execute_reply.started":"2025-10-16T15:16:34.550377Z","shell.execute_reply":"2025-10-16T15:16:34.679314Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Extraction and Engineering\n<!-- - Mean of Bpps -->\n<!-- - 3D Coords -->\n<!-- - Sequence lib -->\n<!-- - forming OpenKnots and the probabity using metadata -->\n<!-- - Probability codons  -->\n<!-- - Mean of probability of codons -->\n<!-- - propbability of forming 2D and 3D structures -->\n<!-- - sequence length -->\n<!-- - Mean reactivity -->\n<!-- - secondary structure and its' count [UMAR IGAN](https://www.kaggle.com/code/umar47/rna-folding-reduce-memory-add-features-seq2seq?scriptVersionId=147271807&cellId=31) -->\n<!-- - Adjacent Guanines count -->","metadata":{}},{"cell_type":"markdown","source":"### energetics and structure \nUsing the arnie package, to get the following features.\n- Bpps Thank to [JOCELYN DUMLAO](https://www.kaggle.com/jocelyndumlaohttps://www.kaggle.com/jocelyndumlao)\n- Dot-bracket notation\n- Free energy\n- pair and unpaired vector","metadata":{}},{"cell_type":"code","source":"# funcitons to get energitcis and structure data\n\ndef energtics_structure_parallel(sequence, package):\n    \"\"\"Get the secondary features for an RNA sequence, \n    derived using arnie and eternafold packages\n    \n    Parameters\n    ----------\n    sequence: str\n        sequence of bases for an RNA\n    \n    Returns\n    -------\n    features: DataFrame\n    - Minimum free energy (dot notation)\n    - free energy \n    \"\"\"\n    def process_seq(seq):\n        seq_info = {}\n\n        seq_info[\"mfe\"] = mfe(seq, package)\n        seq_info[\"free_energy\"] = free_energy(seq, package)\n\n        return seq_info\n\n    with ThreadPoolExecutor() as executor:\n        struc_energ = list(executor.map(process_seq, sequence))\n\n\n    df = pd.DataFrame(struc_energ)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:34.682428Z","iopub.execute_input":"2025-10-16T15:16:34.682780Z","iopub.status.idle":"2025-10-16T15:16:34.693985Z","shell.execute_reply.started":"2025-10-16T15:16:34.682748Z","shell.execute_reply":"2025-10-16T15:16:34.692551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get bpps and saving using np.save\n\ndef get_bpps(seqs, filename, package):\n    \"\"\"\n    Gets base pair probability matrices and saves them in .npy format.\n\n    Args:\n        seqs (list): A list of RNA sequences.\n        filename (str): The name of the file to save the data.\n    \"\"\"\n    bp_matrices = []\n    for seq in tqdm(seqs, desc=\"Processing\", unit=\"sequence\"):\n        # Assuming bpps() returns a NumPy array\n        bp_matrices.append(bpps(seq, package=package))\n    \n    # Save the list of arrays as a single .npy file\n    np.save(filename, bp_matrices)\n    print(f\"Successfully saved {len(bp_matrices)} matrices to {filename}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:34.695351Z","iopub.execute_input":"2025-10-16T15:16:34.695752Z","iopub.status.idle":"2025-10-16T15:16:34.713990Z","shell.execute_reply.started":"2025-10-16T15:16:34.695719Z","shell.execute_reply":"2025-10-16T15:16:34.712557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get the total probability that nucleotide i is paired for each sequence\n# This gives us a prediction of how accessible the nucleotide is, which is important in a lot of contexts -- \n# a prediction for structure probing data, a prediction for degradation rate, and possibly a prediction for protein binding.\n\ndef get_p_unp_vec(bp_matrix, filename):\n    \"\"\"\n    Calculates the probability that each nucleotide is unpaired (P_unp) for a list of BPP matrices, \n    and saves the resulting vectors to a .npy file.\n\n    Args:\n        bp_matrices (list): A list of 2D NumPy arrays (Base Pair Probability matrices).\n        filename (str): The name of the file to save the P_unp vectors (as a .npy file).\n    \"\"\"\n    p_unp_vec= []\n\n    for bp in tqdm(bp_matrix, desc=\"Processing\", unit=\"sequence\"):\n        # calculate paired\n        paired= np.sum(bp_matrix, axis=0)\n\n        # calculate unpaired\n        unpaired= 1 - paired\n        p_unp_vec.append(unpaired)\n\n    # Save the list of arrays as a single .npy file\n    np.save(filename, p_unp_vec)\n    print(f\"Successfully saved {len(p_unp_vec)} P(unpaired) vectors to {filename}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:34.715469Z","iopub.execute_input":"2025-10-16T15:16:34.715976Z","iopub.status.idle":"2025-10-16T15:16:34.733152Z","shell.execute_reply.started":"2025-10-16T15:16:34.715931Z","shell.execute_reply":"2025-10-16T15:16:34.731866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# batch processing for bp matrix\ndef batch_processing(data, batch_size, name, package):\n    \"\"\"Process large RNA sequence data in batches\n        \n    Parameters\n    ----------\n    data: str\n        List of RNA sequence or bpps data\n    \n    batch_size: int\n        size or number of rows for each batch\n    \n    filename: str\n        name for processed dataset\n\n    package: str\n        name of model\n    \n    Returns\n    -------\n        creates csv or npy files for each processed batch\n    \"\"\"\n    # Get batches\n    total_count = len(data)\n    chunks = (total_count - 1) // batch_size + 1 \n        \n    # Handling the batch data and filenaming \n    with tqdm(total=chunks, desc=\"Processing Batches\") as pbar: # progress bar for batch processing\n        sleep(0.1)\n        for i in range(chunks):\n            batch = data[i * batch_size: (i + 1) * batch_size]\n\n            # --- Get mfe and free energy data as csv ---\n            if name == \"train_mfe_free\":\n                filename = f'{name}_{i + 1}.csv'\n                feat_path= glob.glob(f\"/kaggle/input/features-data/{name}*.csv\")\n            \n                # Checkng the last batch, then processing the next batch\n                if os.path.exists(filename) or filename in feat_path: \n                    last_row = pd.read_csv(filename).iloc[-1].tolist()\n                else:\n                    energtics_structure_parallel(batch, package).to_csv(filename)\n            \n            # --- Get bpps data as npy ---    \n            elif name ==  \"bp_mat\":\n                filename = f'{name}_{i + 1}.npy'\n                bp_matrix_files= glob.glob(f\"/kaggle/input/{name}/{name}*.npy\")\n                \n                # Checkin the last batch, then processing the next batch\n                if os.path.exists(filename) or filename in bp_matrix_files:\n                    # 1. Load the entire list of arrays from the .npy file\n                    #    We use allow_pickle=True because it was saved as a Python list.\n                    loaded_matrices = np.load(filename, allow_pickle=True)\n                    \n                    # 2. Get the last array (the last matrix) from the loaded list\n                    if len(loaded_matrices) > 0:\n                        last_matrix_in_batch = loaded_matrices[-1]\n                        # last_matrix_in_batch is now the last array (e.g., a (170, 170) matrix)\n                        print(f\"Loaded existing batch {filename}. Shape of last array: {last_matrix_in_batch.shape}\")\n                    else:\n                        # Handle the case of an empty .npy file if necessary\n                        print(f\"File {filename} exists but is empty.\") \n                else:\n                    get_bpps(batch, filename, package)\n                    \n            # --- Get Probability for Pair and Unpaired nucleotide data as npy\n            elif name == \"p_unp\":\n                # bp_mat_filename = f'{name}_{i+1}.npy'\n                filename = f'{name}_{i + 1}.npy'\n                p_unp_files= glob.glob(f\"/kaggle/input/{name}/{name}*.npy\")\n\n                # Checkin the last batch, then processing the next batch\n                if os.path.exists(filename) or filename in p_unp_files:\n                    # Loading logic for p_unp files\n                    loaded_vecs = np.load(filename, allow_pickle=True)\n                    if len(loaded_vecs) > 0:\n                        print(f\"Skipping existing batch {filename}. Shape of last vector: {loaded_vecs[-1].shape}\")\n                    else:\n                        print(f\"File {filename} exists but is empty. Skipping.\")\n                \n                else:\n                    # 2. Dependency: Must load the BP matrices first\n                    if not os.path.exists(data):\n                        print(f\"Error: Dependency {data} not found. Cannot calculate P_unp.\")\n                        pbar.update(1)\n                        continue\n                        \n                    bp_matrices = np.load(filename, allow_pickle=True)\n                    \n                    # 3. Calculate and save P_unp\n                    get_p_unp_vec(data, p_unp_filename)\n            \n            else:\n                print(f\"Warning: Unknown 'name' parameter: {name}. Skipping batch {batch_num}.\")\n            pbar.update(1)  # Increment the progress \n\n\n        # Cleanup: remove all .ps files\n        for file in glob.glob(\"/kaggle/working/*.ps\"):\n            os.remove(file)\n            print(f\"Removed temporary file: {file}\")\n        \n        # Release unreferenced memory\n        del batch\n        gc.collect()\n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:26:43.776754Z","iopub.execute_input":"2025-10-16T15:26:43.778258Z","iopub.status.idle":"2025-10-16T15:26:43.795781Z","shell.execute_reply.started":"2025-10-16T15:26:43.778213Z","shell.execute_reply":"2025-10-16T15:26:43.794344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def batch_energy_struc(data, batch_size, name, package):\n#     \"\"\"Process large RNA sequence data in batches\n        \n#     Parameters\n#     ----------\n#     data: str\n#         List of RNA sequence data\n    \n#     batch_size: int\n#         size or number of rows for each batch\n    \n#     name: str\n#         name for processed dataset\n\n#     package: str\n#         name of model\n    \n#     Returns\n#     -------\n#         creates csv files for each processed batch\n#     \"\"\"\n#     # Get batches\n#     total_count = len(data)\n#     chunks = (total_count - 1) // batch_size + 1 \n        \n#     # Handling the batch data and filenaming \n#     with tqdm(total=chunks, desc=\"Processing Batches\") as pbar: # progress bar for batch processing\n#         sleep(0.1)\n#         for i in range(chunks):\n#             batch = data[i * batch_size: (i + 1) * batch_size]\n            \n#             filename = f'{name}_struc_{i + 1}.csv'\n#             feat_path= glob.glob(\"/kaggle/input/features-data/*.csv\")\n            \n#             # Checkng the last batch, then processing the next batch\n#             if os.path.exists(filename) or filename in feat_path: \n#                 last_row = pd.read_csv(filename).iloc[-1].tolist()\n#             else:\n#                 energtics_structure_parallel(batch, package).to_csv(filename)\n                \n#             pbar.update(1)  # Increment the progress \n\n#         # Cleanup: remove all .ps files\n#         for file in glob.glob(\"/kaggle/working/*.ps\"):\n#             os.remove(file)\n#             print(f\"Removed temporary file: {file}\")\n        \n#         # Release unreferenced memory\n#         del batch\n#         gc.collect()\n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:34.758058Z","iopub.execute_input":"2025-10-16T15:16:34.758427Z","iopub.status.idle":"2025-10-16T15:16:34.773717Z","shell.execute_reply.started":"2025-10-16T15:16:34.758396Z","shell.execute_reply":"2025-10-16T15:16:34.772234Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # batch processing for bp matrix\n# def batch_bp(data, batch_size, name, package):\n#     \"\"\"Process large RNA sequence data in batches\n        \n#     Parameters\n#     ----------\n#     data: str\n#         List of RNA sequence data\n    \n#     batch_size: int\n#         size or number of rows for each batch\n    \n#     filename: str\n#         name for processed dataset\n\n#     package: str\n#         name of model\n    \n#     Returns\n#     -------\n#         creates csv files for each processed batch\n#     \"\"\"\n#     # Get batches\n#     total_count = len(data)\n#     chunks = (total_count - 1) // batch_size + 1 \n        \n#     # Handling the batch data and filenaming \n#     with tqdm(total=chunks, desc=\"Processing Batches\") as pbar: # progress bar for batch processing\n#         sleep(0.1)\n#         for i in range(chunks):\n#             batch = data[i * batch_size: (i + 1) * batch_size]\n            \n#             filename = f'{name}_{i + 1}.npy'\n#             bp_matrix_files= glob.glob(\"/kaggle/input/bp_matrix/*.npy\")\n            \n#             # Checkin the last batch, then processing the next batch\n#             if os.path.exists(filename) or filename in bp_matrix_files:\n#                 # 1. Load the entire list of arrays from the .npy file\n#                 #    We use allow_pickle=True because it was saved as a Python list.\n#                 loaded_matrices = np.load(filename, allow_pickle=True)\n                \n#                 # 2. Get the last array (the last matrix) from the loaded list\n#                 if len(loaded_matrices) > 0:\n#                     last_matrix_in_batch = loaded_matrices[-1]\n#                     # last_matrix_in_batch is now the last array (e.g., a (170, 170) matrix)\n#                     print(f\"Loaded existing batch {filename}. Shape of last array: {last_matrix_in_batch.shape}\")\n#                 else:\n#                     # Handle the case of an empty .npy file if necessary\n#                     print(f\"File {filename} exists but is empty.\") \n#             else:\n#                 get_bpps(data, filename, package)\n                \n#             pbar.update(1)  # Increment the progress \n\n#         # Cleanup: remove all .ps files\n#         for file in glob.glob(\"/kaggle/working/*.ps\"):\n#             os.remove(file)\n#             print(f\"Removed temporary file: {file}\")\n        \n#         # Release unreferenced memory\n#         del batch\n#         gc.collect()\n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:16:34.775629Z","iopub.execute_input":"2025-10-16T15:16:34.775987Z","iopub.status.idle":"2025-10-16T15:16:34.793201Z","shell.execute_reply.started":"2025-10-16T15:16:34.775955Z","shell.execute_reply":"2025-10-16T15:16:34.791895Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get mfe and free energy\ndata= train_feat.sequence[:100]\nbatch_size= 10\nname= \"train_mfe_free\"\npackage= \"eternafold\"\nbatch_processing(data, batch_size, name, package)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:22:26.429328Z","iopub.execute_input":"2025-10-16T15:22:26.429791Z","iopub.status.idle":"2025-10-16T15:22:35.613608Z","shell.execute_reply.started":"2025-10-16T15:22:26.429757Z","shell.execute_reply":"2025-10-16T15:22:35.612346Z"},"jupyter":{"outputs_hidden":true},"collapsed":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# read the structure csv file into pandas\npd.read_csv(\"/kaggle/working/train_mfe_free_1.csv\").head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:23:35.925573Z","iopub.execute_input":"2025-10-16T15:23:35.926118Z","iopub.status.idle":"2025-10-16T15:23:35.944210Z","shell.execute_reply.started":"2025-10-16T15:23:35.926074Z","shell.execute_reply":"2025-10-16T15:23:35.941685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# gettign the bpps data\nbatch_processing(data, batch_size, \"bp_mat\", package)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:23:42.039975Z","iopub.execute_input":"2025-10-16T15:23:42.040555Z","iopub.status.idle":"2025-10-16T15:23:52.379203Z","shell.execute_reply.started":"2025-10-16T15:23:42.040505Z","shell.execute_reply":"2025-10-16T15:23:52.377959Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the data back into a list of NumPy arrays\nbp_matrix = np.load(\"/kaggle/working/bp_mat_1.npy\")\n\nprint(f\"Loaded a list of {len(bp_matrix)} matrices.\")\nprint(f\"Shape of array {len(bp_matrix.shape)}.\")\nprint(f\"Shape of the first matrix: {bp_matrix[0].shape}\")\nprint(bp_matrix[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:24:25.941419Z","iopub.execute_input":"2025-10-16T15:24:25.941851Z","iopub.status.idle":"2025-10-16T15:24:25.951101Z","shell.execute_reply.started":"2025-10-16T15:24:25.941780Z","shell.execute_reply":"2025-10-16T15:24:25.949760Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# read into pandas\nbp_matrix= pd.Series({\"bp_matrix\": bp_matrix})\n\nbp_matrix","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:24:28.754742Z","iopub.execute_input":"2025-10-16T15:24:28.756199Z","iopub.status.idle":"2025-10-16T15:24:29.116375Z","shell.execute_reply.started":"2025-10-16T15:24:28.756141Z","shell.execute_reply":"2025-10-16T15:24:29.115080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Checking the type\nprint(\"Data type\", type(bp_matrix.bp_matrix[0]))\nprint(\"\\n\")\nprint(\"\\n\")\nbp_matrix.bp_matrix[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:24:31.238520Z","iopub.execute_input":"2025-10-16T15:24:31.238984Z","iopub.status.idle":"2025-10-16T15:24:31.249637Z","shell.execute_reply.started":"2025-10-16T15:24:31.238946Z","shell.execute_reply":"2025-10-16T15:24:31.248116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize a sample matrix\nbp_m= bp_matrix.bp_matrix[0]\nplt.imshow(bp_m, origin='lower', cmap='gist_heat_r')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T15:24:33.999974Z","iopub.execute_input":"2025-10-16T15:24:34.000400Z","iopub.status.idle":"2025-10-16T15:24:34.337281Z","shell.execute_reply.started":"2025-10-16T15:24:34.000366Z","shell.execute_reply":"2025-10-16T15:24:34.335690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# getting p_unp vector\nbatch_processing(data, batch_size, \"p_unp\", \"eternafold\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T17:11:42.719914Z","iopub.execute_input":"2025-10-16T17:11:42.720355Z","iopub.status.idle":"2025-10-16T17:11:42.877257Z","shell.execute_reply.started":"2025-10-16T17:11:42.720323Z","shell.execute_reply":"2025-10-16T17:11:42.875701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dir= \"/kaggle/working/\"\n# bp_files= glob.glob(\"bp_matrix_*.npy\")\nfiles = os.listdir(dir)\nprint(files)\n# for files in dir:\n#     if file.prefi","metadata":{"trusted":true,"execution":{"execution_failed":"2025-10-09T15:47:13.555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_list = glob.glob(f\"/kaggle/working/*.npy\")\n# file_list\nextract_numeric_part = lambda x: int(x.split('_')[2].split('.')[0])\nsorted_file_list = sorted(file_list, key=extract_numeric_part)\n# # np.load(\"/kaggle/working/bp_mat_1.npy\")\nsorted_file_list\n\n# print(\"/kaggle/working/bp_mat_3.npy\".split('_')[2].split('.')[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T17:30:19.236145Z","iopub.execute_input":"2025-10-16T17:30:19.237599Z","iopub.status.idle":"2025-10-16T17:30:19.287702Z","shell.execute_reply.started":"2025-10-16T17:30:19.237546Z","shell.execute_reply":"2025-10-16T17:30:19.285833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# concatenating csv files into a csv file for test and training data\n\ndef concat_files(feature_prefix, output_file):\n    \"\"\" Combine all the csv files into one file\n    \n    Parameter\n    ---------\n    file_name: str\n        path to the csv files\n    \n    name: str\n        name of combined csv file\n    \n    \n    Returns\n    -------\n    combined (.csv or .npy) file\n    \n    \"\"\"\n\n    get_batch_number= re.search(r'(\\d+)\\.(csv|npy)$'\n                                \n    # Check which feature is been concatinated first\n    if feature_prefix == \"mfe_free\":\n        # Create a list of CSV files  to append\n        file_list = glob.glob(f\"/kaggle/working/{feature_prefix}*.csv\")\n        extract_numeric_part = lambda x: int(x.split('_')[2].split('.')[0])\n        sorted_list = sorted(file_list, key=extract_numeric_part)\n    \n        # Read each CSV file into a DataFrame\n        combined_csv = pd.concat([pd.read_csv(f) for f in sorted_list], ignore_index=True)\n    \n        # Save the concat DataFrame to a single CSV file\n        combined_csv.to_csv(f\"/kaggle/working/{output_file}.csv\", index=False)\n        print(f\"Combined CSV file saved to /kaggle/working/{output_file}.csv !!!\")\n        \n    elif feature_prefix == \"bp_mat\":\n        # Create a list of npy files  to append\n        file_list = glob.glob(f\"/kaggle/working/{feature_prefix}*.npy\")\n        extract_numeric_part = lambda x: int(x.split('_')[2].split('.')[0])\n        sorted_list = sorted(file_list, key=extract_numeric_part)\n\n        # Load each npy file into a single array\n        all_arrays= [np.load(f, allow_pickle= True) for f in sorted_list]\n        combined_npy = np.concatenate(all_arrays, axis=0)\n        \n        # Save the array as a single .npy file\n        np.save(f\"/kaggle/working/{output_file}.npy\", combined_npy)\n        print(f\"Combined NPY file saved to /kaggle/working/{output_file}.npy !!!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T17:06:10.518975Z","iopub.execute_input":"2025-10-16T17:06:10.519372Z","iopub.status.idle":"2025-10-16T17:06:10.529687Z","shell.execute_reply.started":"2025-10-16T17:06:10.519341Z","shell.execute_reply":"2025-10-16T17:06:10.528313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Concatenate the energetics and structure csv files\n\nconcat_files(\"train_mfe_free\", \"mfe_free\")\n# concat_files(\"bp_mat\", \"bp_mat\")\n# csv_concat(\"test_features\", \"combine_test\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T17:06:12.199513Z","iopub.execute_input":"2025-10-16T17:06:12.199955Z","iopub.status.idle":"2025-10-16T17:06:12.238257Z","shell.execute_reply.started":"2025-10-16T17:06:12.199920Z","shell.execute_reply":"2025-10-16T17:06:12.236882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# read concated file\nmfe_free= pd.read_csv(\"/kaggle/working/mfe_free_concat.csv\")\nprint(mfe_free.shape)\nprint(len(mfe_free))\nmfe_free.head()\n# combine_test_features= pd.read_csv(\"combine_test.csv\")\n# check.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T16:02:20.729515Z","iopub.execute_input":"2025-10-16T16:02:20.729961Z","iopub.status.idle":"2025-10-16T16:02:20.748567Z","shell.execute_reply.started":"2025-10-16T16:02:20.729927Z","shell.execute_reply":"2025-10-16T16:02:20.747259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mfe_free.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-16T16:03:43.540762Z","iopub.execute_input":"2025-10-16T16:03:43.541261Z","iopub.status.idle":"2025-10-16T16:03:43.549139Z","shell.execute_reply.started":"2025-10-16T16:03:43.541228Z","shell.execute_reply":"2025-10-16T16:03:43.548015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import structure and bpps to data\n\ntrain_struc_bpps= pd.read_csv(\"train_struc_bpps.csv\")\ntest_struc_bpps= pd.read_csv(\"test_struc_bpps.csv\")\n\nprint(f\"train extracted features shape: {test_struc_bpps.shape}\\ntest extracted features shape {\"test_struc_bpps\"}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### sequence lenght","metadata":{}},{"cell_type":"code","source":"# lenght of sequence to a column\ntrain_feat[\"sequnece_len\"]= train_feat.sequence.astype(str).apply(len)\nopt_test[\"sequnece_len\"]= opt_test.sequence.apply(len)\n\ntrain_feat","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"opt_test.info()","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### mean reactivity","metadata":{}},{"cell_type":"code","source":"# Get the mean of reactivity columns \ntrain_feat[\"react_mean\"]= train_feat[reactivity_cols].mean(axis=1)\ntrain_feat","metadata":{"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate mean BPPs\n\ntrain_struc_bpps[\"avg_bpps\"]= train_seq_bpp.mean(axis=1)\ntest_struc_bpps[\"avg_bpps\"] = test_seq_bpp.mean(axis=1)\n\n\n# print(f\"Train dataset shape: {train_struc_bpps}\")\n# print(f\"Test dataset shape: {test_struc_bpps}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to count parentheses\ndef count_parentheses(structure_string):\n    count = structure_string.count(\")\")\n    return count\n\n# Apply the function to the DataFrame column\n\ntq.pandas()\ntrain_struc_bpps['parentheses_counts'] = train_struc_bpps['sec_structure'].astype(str).apply(count_parentheses)\ntest_struc_bpps['parentheses_counts'] = test_struc_bpps['sec_structure'].astype(str).apply(count_parentheses)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### codon features\ncodon count, cps of sequence, codon probability","metadata":{}},{"cell_type":"code","source":"from collections import Counter\n\ndef codons_feats(seq):\n    codons = Counter(seq[i:i+3] for i in range(0, len(seq), 3))\n    pairs = Counter(seq[i:i+6] for i in range(0, len(seq)-1, 3))\n    cps = 0\n    for pair in pairs:\n        if codons[pair[:3]] == 0 or codons[pair[3:]] == 0:\n            continue\n        cps += pairs[pair]/(codons[pair[:3]]*codons[pair[3:]])\n    return {'codons': codons, 'pairs': pairs, 'cps': cps}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the codons, pairs, cps for train sequences\n# codon_feat= []\n\n# for seq in train_feat.sequence:\n#     codon_feat.append(codons_feats(seq))\n\n# # codon_feat= pd.DataFrame(train_feat.sequence.iloc[5:10].apply(codons_feats), columns= [\"codons\", \"pairs\", \"cps\"])\n# codon_feat= pd.DataFrame(codon_feat, columns=[\"codons\", \"pairs\", \"cps\"])\n# codon_feat.to_csv(\"train_codon_feat.csv\")\n\n# codon_feat.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the codons, pairs, cps for test sequences\n# codon_feat= []\n\n# for seq in opt_test.sequence:\n#     codon_feat.append(codons_feats(seq))\n\n# # codon_feat= pd.DataFrame(train_feat.sequence.iloc[5:10].apply(codons_feats), columns= [\"codons\", \"pairs\", \"cps\"])\n# codon_feat= pd.DataFrame(codon_feat, columns=[\"codons\", \"pairs\", \"cps\"])\n# codon_feat.to_csv(\"test_codon_feat.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define functionto calculate codon probabiltiy\ndef codon_probs_mean(seq):\n#     probs = seq.apply(RNA.codon_prob)\n    probs = [RNA.codon_prob(seq) for s in seq]\n    mean_probs= sum(probs.values()) / len(probs)\n    probs_mean= pd.DataFrame({\"probs\":probs, \"mean_probs\": mean_probs})\n    \n    return probs_mean\n\n\ntrain_condon_probs_mean= codon_probs_mean(train_feat.sequence[:10])\ntrain_condon_probs_mean\n# train_condon_probs_mean.to_csv(\"train_condon_probs_mean.csv\")\n\n# test_condon_probs_mean= codon_probs_mean(opt_test.sequence)\n# test_condon_probs_mean.to_csv(\"test_condon_probs_mean.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### count of adjecent guanines in sequence","metadata":{}},{"cell_type":"code","source":"# function to count adjacent guanines in a codon\n\ndef gg_count(seq):\n    \"\"\"\n    Returns:\n    list of adjcent gg or ggg counts for each sequence\n    \"\"\"\n    adj_guanine= []\n    # Count the number of adjacent guanines\n    for s in seq:\n    adj_guanine.append(gg_count = 0)\n    for i in range(len(s) - 1):\n        if s[i:i+2] == \"GG\" or \"GGG\":\n            gg_count += 1\n            \n    return gg_seq_num\n\n\ntrain_struc_bpps[\"adj_guanine\"]= gg_count(train_feat.sequence)\ntest_struc_bpps[\"adj_guanine\"]= gg_count(opt_test.sequence)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### concatenate features into datasets","metadata":{}},{"cell_type":"code","source":"# Concatenate All features to one Dataset called features and test_features\n\nfeatures= pd.concat([train_feat,train_struc_bpps, train_condon_probs_mean])\ntest_features= pd.concat([opt_test,test_struc_bpps,test_condon_probs_mean])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del train_feat\ndel opt_test\ndel train_struc_bpps\ndel test_struc_bpps\ndel train_condon_probs_mean\ndel test_condon_probs_mean\n\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## secondary structures data","metadata":{}},{"cell_type":"code","source":"# Import eterna openknot dataset\n# eterna_pos= pd.read_table(\"/kaggle/input/stanford-ribonanza-rna-folding/eterna_openknot_metadata/Positives240-2000.tsv\", sep=\"\\\\t\")\n# eterna_puz_132= pd.read_csv(\"/kaggle/input/stanford-ribonanza-rna-folding/eterna_openknot_metadata/puzzle 12378132.tsv\", sep= \"\\\\t\")\n# eterna_puz_RYOP50= pd.read_csv(\"/kaggle/input/stanford-ribonanza-rna-folding/eterna_openknot_metadata/puzzle_11318423_RYOP50_with_description.tsv\", sep= \"\\\\t\")\n# eterna_puz_RYOP90= pd.read_csv(\"/kaggle/input/stanford-ribonanza-rna-folding/eterna_openknot_metadata/puzzle_11387276_RYOP90_with_description.tsv\", sep= \"\\\\t\")\n# eterna_puz_RFAM= pd.read_csv(\"/kaggle/input/stanford-ribonanza-rna-folding/eterna_openknot_metadata/puzzle_11627601_with_descriptions_PLUS_RFAM.tsv\", sep= \"\\\\t\")\n# eterna_puz_118= pd.read_csv(\"/kaggle/input/stanford-ribonanza-rna-folding/eterna_openknot_metadata/puzzle_11836497_with_description.tsv\", sep= \"\\\\t\")","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Import Supplementary Silico prediction, that is, secondary structure predictions\n# gpn15k_preds= pd.read_csv(\"/kaggle/input/stanford-ribonanza-rna-folding/supplementary_silico_predictions/GPN15k_silico_predictions.csv\")\n# pk50_preds= pd.read_csv(\"/kaggle/input/stanford-ribonanza-rna-folding/supplementary_silico_predictions/PK50_silico_predictions.csv\")\n# pk90_preds= pd.read_csv(\"/kaggle/input/stanford-ribonanza-rna-folding/supplementary_silico_predictions/PK90_silico_predictions.csv\")\n# r1_preds= pd.read_csv(\"/kaggle/input/stanford-ribonanza-rna-folding/supplementary_silico_predictions/R1_silico_predictions.csv\")","metadata":{"jupyter":{"source_hidden":true},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# gpn15k_preds.shape\n# gpn15k_preds.head()\n# pk50_preds.shape\n# pk50_preds.head()\n# pk90_preds.shape\n# pk90_preds.head()\n# r1_preds.shape\n# r1_preds.head()","metadata":{"jupyter":{"source_hidden":true},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Building Model","metadata":{}},{"cell_type":"code","source":"from keras.models import Model\nfrom keras.layers import Dense, Conv1D, Flatten, Input, concatenate\n\n# sequence input (assuming one-hot encoded sequences of length 4)\nsequence_input = Input(shape=(None, 4))\nconv1 = Conv1D(64, kernel_size=3, activation='relu')(sequence_input)\nconv2 = Conv1D(32, kernel_size=3, activation='relu')(conv1)\nflat = Flatten()(conv2)\n\n# numerical/categorical input\nnumerical_input = Input(shape=(4,))\ndense1 = Dense(32, activation='relu')(numerical_input)\n\n# concatenate sequence and numerical inputs\nconcat = concatenate([flat, dense1])\n\n# output layer\noutput = Dense(1, activation='sigmoid')(concat)\n\n# create a model\nmodel = Model(inputs=[sequence_input, numerical_input], outputs=output)\n\n# compile model using MAE as a measure of model performance\nmodel.compile(optimizer='adam', loss='mean_absolute_error')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}