{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":51294,"databundleVersionId":6923401,"sourceType":"competition"},{"sourceId":151943406,"sourceType":"kernelVersion"}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# StructreGenerator\n\n- This note is to extract structres from sequences\n- You don't actually need to extract them because my dataset is public.\n    https://www.kaggle.com/datasets/crimson206/rna-smalldatawithstructure\n- However, if you want to extract structures even from train data with large errors, or from other external sequences, you can still use this code","metadata":{}},{"cell_type":"code","source":"import gzip\nimport shutil\nwith gzip.open('/kaggle/input/download-rnacentral/rnacentral_active.fasta.gz', 'rb') as f_in:\n    with open('/tmp/rnacentral_active.fasta', 'wb') as f_out:\n        shutil.copyfileobj(f_in, f_out)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T15:01:22.623108Z","iopub.execute_input":"2023-11-23T15:01:22.623648Z","iopub.status.idle":"2023-11-23T15:05:45.342839Z","shell.execute_reply.started":"2023-11-23T15:01:22.623603Z","shell.execute_reply":"2023-11-23T15:05:45.339909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\n\ndef get_seq_ids(input_fasta):\n    \"\"\"\n    Get a list of sequence ids from a fasta file.\n    \"\"\"\n    seq_ids = set()\n    counter = 0\n    squence_list = []\n    current_seq = None\n    with open(input_fasta) as f_in:\n        for line in f_in:\n            if line.startswith(\">\"):\n                if current_seq is not None :\n                    if len(current_seq)<512 :\n                        seq_ids.add(current_seq.replace('T','U'))\n                current_seq = \"\"\n            else :\n                current_seq = current_seq+line.strip(\"\\n\")\n    return seq_ids","metadata":{"execution":{"iopub.status.busy":"2023-11-23T15:31:13.749231Z","iopub.execute_input":"2023-11-23T15:31:13.749737Z","iopub.status.idle":"2023-11-23T15:31:13.759879Z","shell.execute_reply.started":"2023-11-23T15:31:13.749693Z","shell.execute_reply":"2023-11-23T15:31:13.758517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nseq_ids = get_seq_ids(\"/tmp/rnacentral_active.fasta\")","metadata":{"execution":{"iopub.status.busy":"2023-11-23T15:31:27.965515Z","iopub.execute_input":"2023-11-23T15:31:27.965980Z","iopub.status.idle":"2023-11-23T15:37:39.344900Z","shell.execute_reply.started":"2023-11-23T15:31:27.965946Z","shell.execute_reply":"2023-11-23T15:37:39.343354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl\ndf = pl.DataFrame(\n    {\n        \"sequence\": list(seq_ids)\n    }\n)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T15:41:46.112062Z","iopub.execute_input":"2023-11-23T15:41:46.112626Z","iopub.status.idle":"2023-11-23T15:42:37.626578Z","shell.execute_reply.started":"2023-11-23T15:41:46.112583Z","shell.execute_reply":"2023-11-23T15:42:37.624638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.write_parquet(\"rna_sequence_20m.parquet\")","metadata":{"execution":{"iopub.status.busy":"2023-11-23T15:43:45.246558Z","iopub.execute_input":"2023-11-23T15:43:45.247092Z","iopub.status.idle":"2023-11-23T15:44:41.786605Z","shell.execute_reply.started":"2023-11-23T15:43:45.247050Z","shell.execute_reply":"2023-11-23T15:44:41.784919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load or import necessary things","metadata":{}},{"cell_type":"code","source":"# %%capture\n# %pip install arnie\n# %pip install draw_rna\n# !conda config --set auto_update_conda false\n# !conda install -c bioconda eternafold --yes\n# %env ETERNAFOLD_PATH=/opt/conda/bin/eternafold-bin\n# %env ETERNAFOLD_PARAMETERS=/opt/conda/lib/eternafold-lib/parameters/EternaFoldParams.v1","metadata":{"execution":{"iopub.status.busy":"2023-10-28T21:53:02.074428Z","iopub.execute_input":"2023-10-28T21:53:02.074855Z","iopub.status.idle":"2023-10-28T21:56:16.770319Z","shell.execute_reply.started":"2023-10-28T21:53:02.074814Z","shell.execute_reply":"2023-10-28T21:56:16.768895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from tqdm import tqdm\n# from arnie.mfe import mfe\n\n# import numpy as np\n# import pandas as pd\n# import warnings\n# import math\n\n# warnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-10-28T21:57:43.653461Z","iopub.execute_input":"2023-10-28T21:57:43.65395Z","iopub.status.idle":"2023-10-28T21:57:43.6605Z","shell.execute_reply.started":"2023-10-28T21:57:43.65391Z","shell.execute_reply":"2023-10-28T21:57:43.659078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Functions\n\n### divide_dataframe\n\n- Extracting structures is time comsuming.\n- Therefore, we devide data into 9 chunks. Why 9 chunks?\n- You can run totally 10 cpus using \"Save Version\" => \"Save & Run All (Commit)\".\n- We extract structures parallelly from the 9 chunks.\n- You can use one quota left for other purposes.\n\n### compute_structure\n\n- I recommend you to read this code\n    https://www.kaggle.com/code/jocelyndumlao/rna-structure-prediction-performance-analysis","metadata":{}},{"cell_type":"code","source":"# def divide_dataframe(df, num_pieces):\n#     chunk_size = math.ceil(len(df) / num_pieces)\n#     return [df.iloc[i:i + chunk_size] for i in range(0, len(df), chunk_size)]\n\n# def compute_structure(sequence):\n#     return mfe(sequence, package=\"eternafold\")","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:16:31.694346Z","iopub.execute_input":"2023-10-28T22:16:31.694839Z","iopub.status.idle":"2023-10-28T22:16:31.702027Z","shell.execute_reply.started":"2023-10-28T22:16:31.694785Z","shell.execute_reply":"2023-10-28T22:16:31.700698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# example_sequence = 'GGGAACGACUCGAGUAGAGUCGAAAAUUUCCUUCCAAAUCCUGAGGGAGAGAUAGAGGCGGAGGGUCUGGGGGAGGAAUUAAAACACAAGGUCUCCUCCCCUCUCGCCUGUCCGAACUUGGGGGCACCCCGGCUCGUACUUCGGUACGAGCCGGGGAAAAGAAACAACAACAACAAC'","metadata":{"execution":{"iopub.status.busy":"2023-10-28T21:58:42.47694Z","iopub.execute_input":"2023-10-28T21:58:42.477378Z","iopub.status.idle":"2023-10-28T21:58:42.483481Z","shell.execute_reply.started":"2023-10-28T21:58:42.477325Z","shell.execute_reply":"2023-10-28T21:58:42.481915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# compute_structure(example_sequence)","metadata":{"execution":{"iopub.status.busy":"2023-10-28T21:59:04.226291Z","iopub.execute_input":"2023-10-28T21:59:04.22667Z","iopub.status.idle":"2023-10-28T21:59:04.337622Z","shell.execute_reply.started":"2023-10-28T21:59:04.226641Z","shell.execute_reply":"2023-10-28T21:59:04.336338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Usage\n\n- Set i = 0, and process \"Save Version\" => \"Save & Run All (Commit)\".\n- If your code is finished successfully, you can download the file with extracted structure from the output tap.\n- Repeat it for all i (0~9)","metadata":{}},{"cell_type":"code","source":"# # if you want to extract structures of test sequences, use => mode_type = \"test\"\n# i = 8\n# mode_type = \"train\"","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:17:00.392664Z","iopub.execute_input":"2023-10-28T22:17:00.393111Z","iopub.status.idle":"2023-10-28T22:17:00.399077Z","shell.execute_reply.started":"2023-10-28T22:17:00.393077Z","shell.execute_reply":"2023-10-28T22:17:00.397661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if mode_type is \"train\":\n#     df = pd.read_csv(\"/kaggle/input/stanford-ribonanza-rna-folding/train_data.csv\")\n# elif mode_type is \"test\":\n#     df = pd.read_csv(\"/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv\")","metadata":{"papermill":{"duration":7.74363,"end_time":"2023-10-18T12:51:39.565768","exception":false,"start_time":"2023-10-18T12:51:31.822138","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-10-28T21:59:49.070826Z","iopub.execute_input":"2023-10-28T21:59:49.071289Z","iopub.status.idle":"2023-10-28T22:01:44.098174Z","shell.execute_reply.started":"2023-10-28T21:59:49.071244Z","shell.execute_reply":"2023-10-28T22:01:44.0971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# chunks = divide_dataframe(df, 9)\n# sequences = chunks[i][\"sequence\"]","metadata":{"execution":{"iopub.status.busy":"2023-10-28T22:17:38.349414Z","iopub.execute_input":"2023-10-28T22:17:38.349899Z","iopub.status.idle":"2023-10-28T22:17:38.358068Z","shell.execute_reply.started":"2023-10-28T22:17:38.349851Z","shell.execute_reply":"2023-10-28T22:17:38.356797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from multiprocessing import Pool, cpu_count\n\n# num_processes = cpu_count()\n\n# with Pool(processes=num_processes) as pool:\n#     structures = list(tqdm(pool.imap(compute_structure, sequences), total=len(sequences)))\n\n# chunks[i]['structure'] = structures\n# chunks[i].to_parquet(f'{mode_type}_chunk{i}.parquet')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}