{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":12276181,"sourceType":"competition"},{"sourceId":11775065,"sourceType":"datasetVersion","datasetId":7392749},{"sourceId":11788894,"sourceType":"datasetVersion","datasetId":7402173},{"sourceId":12613207,"sourceType":"datasetVersion","datasetId":7967790}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Identify 3D templates for RNA targets\nThis notebook prepares a mock submission as an illustration of how to derive 3D templates for targets in the Stanford 3D RNA folding competition. See also https://www.kaggle.com/datasets/rhijudas/rna-3d-folding-templates/data for precomputed outputs useful for training models for RNA 3D folding.\n\n## Important options","metadata":{}},{"cell_type":"code","source":"# if False, don't check on temporal_cutoff -- during training on known structures, will get lots of leakage. \n# If going for Early Sharing Prize, set to True.\nCHECK_TEMPORAL_CUTOFF = True\n\n# Number of templates to use. \n# Here set to 5 to prepare a mock submission\n# Should make larger (e.g., 40) if using templates for modeling. \nMAX_TEMPLATES = 5\n\n# Better to use nan when preparing files for templates, to allow easy recognition of which coordinates are missing.\n# But for this example, using 0.0 to avoid errors in scoring the final submission.csv\nNULL_VALUE = 0.0  \n\n# directory with test_sequences.csv\nINPUT_DIR = '/kaggle/input/stanford-rna-3d-folding'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T18:13:16.005621Z","iopub.execute_input":"2025-07-29T18:13:16.006155Z","iopub.status.idle":"2025-07-29T18:13:16.019316Z","shell.execute_reply.started":"2025-07-29T18:13:16.006095Z","shell.execute_reply":"2025-07-29T18:13:16.017686Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Download MMseqs2","metadata":{}},{"cell_type":"code","source":"#!wget https://mmseqs.com/latest/mmseqs-linux-avx2.tar.gz\n#!tar xvfz /kaggle/working/mmseqs-linux-avx2.tar.gz\n!rsync -avL /kaggle/input/mmseqs2/mmseqs /kaggle/working/\n!chmod 755 /kaggle/working/mmseqs/bin/mmseqs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T18:13:16.022382Z","iopub.execute_input":"2025-07-29T18:13:16.023299Z","iopub.status.idle":"2025-07-29T18:13:17.078698Z","shell.execute_reply.started":"2025-07-29T18:13:16.023269Z","shell.execute_reply":"2025-07-29T18:13:17.076790Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create DB based on FASTA of all PDB nucleic acid sequences, which is part of PDB_RNA dataset","metadata":{}},{"cell_type":"code","source":"!/kaggle/working/mmseqs/bin/mmseqs createdb {INPUT_DIR}/PDB_RNA/pdb_seqres_NA.fasta pdb_seqres_NA --dbtype 2 # > MMseqs_createDB.log","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T18:13:17.080306Z","iopub.execute_input":"2025-07-29T18:13:17.080666Z","iopub.status.idle":"2025-07-29T18:13:17.770503Z","shell.execute_reply.started":"2025-07-29T18:13:17.080628Z","shell.execute_reply":"2025-07-29T18:13:17.768816Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Find templates for targets in test_sequences.csv by aligning to PDB with MMseqs2","metadata":{}},{"cell_type":"markdown","source":"### Need to  convert test_sequences.csv file to FASTA","metadata":{}},{"cell_type":"code","source":"import csv\ninput_file=f'{INPUT_DIR}/test_sequences.csv'\noutput_file='test_sequences.fasta'\nSEQ_COUNT = 0\nwith open(input_file, 'r', newline='') as csv_file, open(output_file, 'w') as fasta_file:\n    csv_reader = csv.reader(csv_file, quotechar='\"', delimiter=',', quoting=csv.QUOTE_ALL, skipinitialspace=True)\n    next(csv_reader)  # Skip the header row\n    for row in csv_reader:\n        if len(row) >= 2:\n            fasta_file.write(f\">{row[0]}\\n{row[1]}\\n\")\n            SEQ_COUNT += 1\nprint(f'Will run MMseqs2 on {SEQ_COUNT} sequences')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T18:13:17.772886Z","iopub.execute_input":"2025-07-29T18:13:17.773339Z","iopub.status.idle":"2025-07-29T18:13:17.786262Z","shell.execute_reply.started":"2025-07-29T18:13:17.773299Z","shell.execute_reply":"2025-07-29T18:13:17.785137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!/kaggle/working/mmseqs/bin/mmseqs easy-search /kaggle/working/test_sequences.fasta /kaggle/working/pdb_seqres_NA testResult.txt tmp --search-type 3 --format-output \"query,target,evalue,qstart,qend,tstart,tend,qaln,taln\" ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T18:13:17.787312Z","iopub.execute_input":"2025-07-29T18:13:17.787660Z","iopub.status.idle":"2025-07-29T18:13:38.693977Z","shell.execute_reply.started":"2025-07-29T18:13:17.787633Z","shell.execute_reply":"2025-07-29T18:13:38.692685Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Assemble file of template coordinates by going through .cif files found for each target by MMseqs2","metadata":{}},{"cell_type":"code","source":"!pip3 install /kaggle/input/biopython/biopython-1.85-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T18:13:38.695571Z","iopub.execute_input":"2025-07-29T18:13:38.696042Z","iopub.status.idle":"2025-07-29T18:13:44.084086Z","shell.execute_reply.started":"2025-07-29T18:13:38.695998Z","shell.execute_reply":"2025-07-29T18:13:44.081948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/input/create-templates-csv-py/')\n\nfrom create_templates_csv import get_template_labels\n\nsequence_file = input_file\nmmseqs_results_file = 'testResult.txt' \nskip_temporal_cutoff = not CHECK_TEMPORAL_CUTOFF\ncif_dir = f\"{INPUT_DIR}/PDB_RNA\"\noutput_labels,output_allatom_labels,targets = get_template_labels( input_file, mmseqs_results_file, skip_temporal_cutoff,\n                                                       MAX_TEMPLATES, cif_dir )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T18:13:44.087137Z","iopub.execute_input":"2025-07-29T18:13:44.087453Z","iopub.status.idle":"2025-07-29T18:15:12.363807Z","shell.execute_reply.started":"2025-07-29T18:13:44.087428Z","shell.execute_reply":"2025-07-29T18:15:12.362588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from create_templates_csv import output_template_labels_to_csv\n\n# Create a DataFrame and write to CSV\noutdir = './'\noutfile = 'submission.csv'\noutput_template_labels_to_csv( output_labels, output_allatom_labels, targets, outdir, outfile )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T18:15:12.364872Z","iopub.execute_input":"2025-07-29T18:15:12.365196Z","iopub.status.idle":"2025-07-29T18:15:13.091734Z","shell.execute_reply.started":"2025-07-29T18:15:12.365153Z","shell.execute_reply":"2025-07-29T18:15:13.090919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}