{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11228175,"sourceType":"competition"},{"sourceId":10922467,"sourceType":"datasetVersion","datasetId":6790676}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Predicting dot-bracket structure for all input data\n\nThanks to this [notebook]( ) for inspiration on installing Arnie and EteRNAfold.\n","metadata":{}},{"cell_type":"markdown","source":"## Installing libraries","metadata":{}},{"cell_type":"code","source":"!pip install arnie -q # <-- this should be on dependency script\n!pip install pandarallel -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T14:30:44.39871Z","iopub.execute_input":"2025-03-05T14:30:44.399123Z","iopub.status.idle":"2025-03-05T14:30:53.279303Z","shell.execute_reply.started":"2025-03-05T14:30:44.39909Z","shell.execute_reply":"2025-03-05T14:30:53.277631Z"}},"outputs":[],"execution_count":7},{"cell_type":"code","source":"! mkdir -p /kaggle/working/eternafold\n! cp /kaggle/input/eternafold/libstdc.so.6.0.32 /usr/lib/x86_64-linux-gnu/libstdc++.so.6.0.32\n! cp /kaggle/input/eternafold/libstdc.so.6 /usr/lib/x86_64-linux-gnu/libstdc++.so.6\n! cp /kaggle/input/eternafold/contrafold /kaggle/working/eternafold/contrafold\n! cp /kaggle/input/eternafold/EternaFoldParams.v1 /kaggle/working/eternafold/EternaFoldParams.v1\n! chmod +x /kaggle/working/eternafold/contrafold\n%env ETERNAFOLD_PATH=/kaggle/working/eternafold/\n%env ETERNAFOLD_PARAMETERS=/kaggle/working/eternafold/EternaFoldParams.v1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T14:30:53.281264Z","iopub.execute_input":"2025-03-05T14:30:53.281906Z","iopub.status.idle":"2025-03-05T14:30:54.125495Z","shell.execute_reply.started":"2025-03-05T14:30:53.281849Z","shell.execute_reply":"2025-03-05T14:30:54.124234Z"}},"outputs":[{"name":"stdout","text":"env: ETERNAFOLD_PATH=/kaggle/working/eternafold/\nenv: ETERNAFOLD_PARAMETERS=/kaggle/working/eternafold/EternaFoldParams.v1\n","output_type":"stream"}],"execution_count":8},{"cell_type":"markdown","source":"## Reading Sequence files\n\nSee [this notebook](https://www.kaggle.com/code/dalloliogm/eda-of-sequence-files) for more context.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport csv\n\n# Read the input sequences file, using multi-line separator\ndf = pd.read_csv(\n    \"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\",\n    engine=\"python\",\n    quoting=csv.QUOTE_MINIMAL,\n    usecols=[\"target_id\", \"sequence\", \"temporal_cutoff\", \"description\", \"all_sequences\"]\n)\n\n# Define a function that splits the multi-line text into a list\ndef split_all_sequences(text):\n    if pd.isna(text):\n        return []\n    # Split the text by newline and remove any empty strings\n    return [line.strip() for line in text.splitlines() if line.strip()]\n\n# Apply the function to the \"all_sequences\" column\ndf['all_sequences'] = df['all_sequences'].apply(split_all_sequences)\n\nprint(df.head(6))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T14:30:54.206208Z","iopub.execute_input":"2025-03-05T14:30:54.20652Z","iopub.status.idle":"2025-03-05T14:30:54.280985Z","shell.execute_reply.started":"2025-03-05T14:30:54.206493Z","shell.execute_reply":"2025-03-05T14:30:54.279756Z"}},"outputs":[{"name":"stdout","text":"  target_id                            sequence temporal_cutoff  \\\n0    1SCL_A       GGGUGCUCAGUACGAGAGGAACCGCACCC      1995-01-26   \n1    1RNK_A  GGCGCAGUGGGCUAGCGCCACUCAAAAGGCCCAU      1995-02-27   \n2    1RHT_A            GGGACUGACGAUCACGCAGUCUAU      1995-06-03   \n3    1HLX_A                GGGAUAACUUCGGUUGUCCC      1995-09-15   \n4    1HMH_E  GGCGACCCUGAUGAGGCCGAAAGGCCGAAACCGU      1995-12-07   \n5    1RNG_A                        GGCGCUUGCGUC      1995-12-07   \n\n                                         description  \\\n0               THE SARCIN-RICIN LOOP, A MODULAR RNA   \n1  THE STRUCTURE OF AN RNA PSEUDOKNOT THAT CAUSES...   \n2  24-MER RNA HAIRPIN COAT PROTEIN BINDING SITE F...   \n3  P1 HELIX NUCLEIC ACIDS (DNA/RNA) RIBONUCLEIC ACID   \n4  THREE-DIMENSIONAL STRUCTURE OF A HAMMERHEAD RI...   \n5  SOLUTION STRUCTURE OF THE CUUG HAIRPIN: A NOVE...   \n\n                                       all_sequences  \n0  [>1SCL_1|Chain A|RNA SARCIN-RICIN LOOP|Rattus ...  \n1  [>1RNK_1|Chain A|RNA PSEUDOKNOT|null, GGCGCAGU...  \n2  [>1RHT_1|Chain A|RNA (5'-R(P*GP*GP*GP*AP*CP*UP...  \n3  [>1HLX_1|Chain A|RNA (5'-R(*GP*GP*GP*AP*UP*AP*...  \n4  [>1HMH_1|Chains A, C, E|HAMMERHEAD RIBOZYME-RN...  \n5  [>1RNG_1|Chain A|RNA (5'-R(*GP*GP*CP*GP*CP*UP*...  \n","output_type":"stream"}],"execution_count":10},{"cell_type":"markdown","source":"### Predicting structures","metadata":{}},{"cell_type":"code","source":"from pandarallel import pandarallel\nfrom arnie.mfe import mfe\n\npandarallel.initialize(progress_bar=True)  # optional progress bar\n\ndf[\"structure\"] = df.sequence.parallel_apply(lambda x: mfe(x, package=\"eternafold\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T14:31:02.918546Z","iopub.execute_input":"2025-03-05T14:31:02.918922Z","execution_failed":"2025-03-05T15:12:03.507Z"}},"outputs":[{"name":"stdout","text":"INFO: Pandarallel will run on 2 workers.\nINFO: Pandarallel will use Memory file system to transfer data between the main process and workers.\n","output_type":"stream"},{"output_type":"display_data","data":{"text/plain":"VBox(children=(HBox(children=(IntProgress(value=0, description='0.00%', max=422), Label(value='0 / 422'))), HB…","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"594d339ca1f0494e95b24dc58ac56ef9"}},"metadata":{}}],"execution_count":null},{"cell_type":"code","source":"from arnie.mfe import mfe\ndf[\"structure\"] = df.sequence.apply(lambda x : mfe(x, package=\"eternafold\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T12:50:48.539435Z","iopub.execute_input":"2025-03-05T12:50:48.539766Z","execution_failed":"2025-03-05T14:27:16.071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-03-05T14:27:16.072Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Have fun!","metadata":{}},{"cell_type":"markdown","source":"## Saving file\n\nWe save the output to file, after removing the \"all_sequences\" column, which is not relevant for the prediction.","metadata":{}},{"cell_type":"code","source":"df_without_all_sequences = df.drop(\"all_sequences\", axis=1)\ndf.to_csv(\"kaggle/working/stanford_3d_rna_train_sequences_with_dotbrackets.csv\")","metadata":{"trusted":true,"execution":{"execution_failed":"2025-03-05T14:27:16.073Z"}},"outputs":[],"execution_count":null}]}