{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"},{"sourceId":11155363,"sourceType":"datasetVersion","datasetId":6929404}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"- Database: www.rcsb.org\n- Data Import: https://biopandas.github.io/biopandas/tutorials/Working_with_mmCIF_Structures_in_DataFrames/\n  \n### This code is designed to extract the following information: \n* target_id (list of strings): RNA ID\n* sequences (list of strings): RNA sequence\n* descriptions (list of strings): RNA information\n* temporal_cutoffs (list of strings): the first time the RNA is cut off\n* labels (list of numpy arrays(N,3) ): RNA coordinates","metadata":{}},{"cell_type":"markdown","source":"# IMPORT LIBRARY","metadata":{}},{"cell_type":"code","source":"!pip install /kaggle/input/srna3df-data-rna-glb/biopandas-0.5.1-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T01:33:38.394886Z","iopub.execute_input":"2025-03-31T01:33:38.395268Z","iopub.status.idle":"2025-03-31T01:33:45.132312Z","shell.execute_reply.started":"2025-03-31T01:33:38.395227Z","shell.execute_reply":"2025-03-31T01:33:45.131231Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from biopandas.mmcif import PandasMmcif\nfrom tqdm import tqdm\nimport numpy as np\nimport glob\nimport json\n\nfrom plotly.offline import init_notebook_mode, iplot\ninit_notebook_mode(connected=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T01:33:45.133440Z","iopub.execute_input":"2025-03-31T01:33:45.133846Z","iopub.status.idle":"2025-03-31T01:33:45.842763Z","shell.execute_reply.started":"2025-03-31T01:33:45.133795Z","shell.execute_reply":"2025-03-31T01:33:45.841866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pmmcif = PandasMmcif().fetch_mmcif('3eiy')\npmmcif.read_mmcif(\"/kaggle/input/srna3df-data-rna-glb/RCSB_RNA/2d19.cif\") \nprint('mmCIF Code: %s' % pmmcif.code)\nprint('mmCIF Header Line: %s' % pmmcif.header)\nprint('\\nRaw mmCIF file contents:\\n\\n%s\\n...' % pmmcif.pdb_text[:6000] + pmmcif.pdb_text[56000:60000])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T01:33:45.843681Z","iopub.execute_input":"2025-03-31T01:33:45.843974Z","iopub.status.idle":"2025-03-31T01:33:47.322926Z","shell.execute_reply.started":"2025-03-31T01:33:45.843938Z","shell.execute_reply":"2025-03-31T01:33:47.321824Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EXTRACT DATA","metadata":{}},{"cell_type":"code","source":"def check_section(text, sequence, asym_id):\n    condition = [text.startswith(letter+asym_id) for letter in sequence]\n    return True if sum(condition) else False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T01:33:47.325387Z","iopub.execute_input":"2025-03-31T01:33:47.325758Z","iopub.status.idle":"2025-03-31T01:33:47.330322Z","shell.execute_reply.started":"2025-03-31T01:33:47.325725Z","shell.execute_reply":"2025-03-31T01:33:47.329282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pmmcif = PandasMmcif().fetch_mmcif('3eiy')\ntarget_ids, sequences, temporal_cutoffs, descriptions, labels  = [], [], [], [], []\n\nfor file in tqdm(glob.glob(\"/kaggle/input/srna3df-data-rna-glb/RCSB_RNA/*.cif\")):\n    if file[-9] != '/': #Skip ..-sf.cif such as 100d-sf.cf, 157d-sf.cf\n        continue\n        \n    pmmcif.read_mmcif(file) \n\n    # XYZ Data\n    ATOM_df = pmmcif.df['ATOM']\n    ATOM_df['section'] = ATOM_df['label_comp_id'] + ATOM_df['label_asym_id'] + pmmcif.df['ATOM']['label_entity_id'].astype(str) + ATOM_df['label_seq_id'].astype(str)\n    ATOM_df = ATOM_df[['section', 'Cartn_x', 'Cartn_y', 'Cartn_z']]\n    ATOM_df.loc[:, 'section'] = ATOM_df['section'].str.replace(\"D\", \"\", regex=False)\n    ATOM_df = ATOM_df.groupby('section', sort=False, as_index=False).agg(lambda x: x.mean())\n\n    # Loop for extracting data\n    asym_set = list(dict.fromkeys(pmmcif.df['ATOM']['label_asym_id']))\n    for idx in range(len(asym_set)):\n        try:\n            #-- TARGET ID ----------\n            asym_id = asym_set[idx]\n            target_id = pmmcif.data[\"entry\"][\"id\"][0] + \"_\" + asym_id\n            \n            #-- SEQUENCE ----------\n            if idx > len(pmmcif.data[\"entity_poly\"][\"pdbx_seq_one_letter_code_can\"]) - 1:\n                sequence = pmmcif.data[\"entity_poly\"][\"pdbx_seq_one_letter_code_can\"][0].replace('\\n', '')\n            else:\n                sequence = pmmcif.data[\"entity_poly\"][\"pdbx_seq_one_letter_code_can\"][idx].replace('\\n', '')\n\n            if sequence == '': #If sequence is empty in one_letter_code part, extracting sequence from ATOM df\n                sequence_extraction_df = pmmcif.df['ATOM'][pmmcif.df['ATOM']['label_asym_id'] == asym_id][['label_comp_id', 'label_seq_id']]\n                sequence = sequence_extraction_df.groupby('label_seq_id', sort=False, as_index=False).agg(lambda x: list(set(x))[0])['label_comp_id']\n                sequence = ''.join(sequence)\n\n            if len(set(sequence)-{'A','C','G','U'}) > 0: #Skip the sequence including more than A,C,G,U\n                continue\n\n             #-- CUTOFF DATE ----------           \n            temporal_cutoff = pmmcif.data[\"pdbx_audit_revision_history\"][\"revision_date\"][0]\n\n             #-- DESCRIPTION ----------\n            description_temp = []\n            for title in list(pmmcif.data.keys()):\n                if 'pdbx_description' in list(pmmcif.data[title].keys()):\n                    if idx > len(pmmcif.data[title]['pdbx_description']) - 1:\n                        if pmmcif.data[title]['pdbx_description'][0] is not None:\n                            description_temp.append(pmmcif.data[title]['pdbx_description'][0])\n                    else:\n                        if pmmcif.data[title]['pdbx_description'][idx] is not None:\n                            description_temp.append(pmmcif.data[title]['pdbx_description'][idx])\n            if description_temp != []:\n                description = \"|\".join(description_temp)\n                \n             #-- XYZ COORDINATES ----------\n            label = ATOM_df.loc[ATOM_df['section'].apply(lambda col: check_section(col, sequence, asym_id))][['Cartn_x', 'Cartn_y', 'Cartn_z']].to_numpy(dtype='float32')\n            \n            assert len(sequence) == len(label) \n            #--  STORE DATA ----------------------------------------\n            target_ids.append(target_id)\n            sequences.append(sequence) \n            temporal_cutoffs.append(temporal_cutoff)\n            descriptions.append(description)\n            labels.append(label)\n            \n        except:\n            print(\"ERROR: {} {} - {}\".format(pmmcif.data[\"entry\"][\"id\"][0], asym_set[idx], \"Missing Coordinates\" if len(sequence) != len(label) else \"\"))\n            print(sequence)\n            continue","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T01:33:47.331897Z","iopub.execute_input":"2025-03-31T01:33:47.332297Z","iopub.status.idle":"2025-03-31T01:33:54.771916Z","shell.execute_reply.started":"2025-03-31T01:33:47.332256Z","shell.execute_reply":"2025-03-31T01:33:54.770510Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DATA VISUALIZATION","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T01:33:54.772926Z","iopub.execute_input":"2025-03-31T01:33:54.773245Z","iopub.status.idle":"2025-03-31T01:33:54.872003Z","shell.execute_reply.started":"2025-03-31T01:33:54.773219Z","shell.execute_reply":"2025-03-31T01:33:54.871049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_structure(df: pd.DataFrame, sequence_id: str) -> None:\n    sequence_df = df\n    sequence_points = sequence_df[[\"x_1\", \"y_1\", \"z_1\", \"resname\"]]\n    \n    colors = {\"A\": \"red\", \"G\": \"blue\", \"C\": \"green\", \"U\": \"orange\"}\n    fig = go.Figure()\n    \n    for resname, color in colors.items():\n        subset = sequence_df[sequence_df[\"resname\"] == resname]\n        fig.add_trace(go.Scatter3d(\n            x=subset[\"x_1\"], y=subset[\"y_1\"], z=subset[\"z_1\"],\n            mode='markers',\n            marker=dict(size=5, color=color),\n            name=resname,\n        ))\n    \n    fig.add_trace(go.Scatter3d(\n        x=sequence_df[\"x_1\"], y=sequence_df[\"y_1\"], z=sequence_df[\"z_1\"],\n        mode='lines',\n        line=dict(color='gray', width=2),\n        name='RNA Backbone'\n    ))\n    \n    fig.update_layout(\n            scene=dict(xaxis_title='X', yaxis_title='Y', zaxis_title='Z'),\n            title=f'3D RNA Structure of sequence {sequence_id}',\n        )\n            \n    return fig","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T01:33:54.873096Z","iopub.execute_input":"2025-03-31T01:33:54.873387Z","iopub.status.idle":"2025-03-31T01:33:54.880919Z","shell.execute_reply.started":"2025-03-31T01:33:54.873358Z","shell.execute_reply":"2025-03-31T01:33:54.879663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for index in range(5):\n    pmmcif.read_mmcif(\"/kaggle/input/srna3df-data-rna-glb/RCSB_RNA/{}.cif\".format(target_ids[index][:4].lower()))\n    asym_id = target_ids[index][-1]\n    xyz_df_raw = pd.DataFrame({\"x_1\": pmmcif.df['ATOM'][pmmcif.df['ATOM']['label_asym_id'] == asym_id]['Cartn_x'].tolist(), \n                           \"y_1\": pmmcif.df['ATOM'][pmmcif.df['ATOM']['label_asym_id'] == asym_id]['Cartn_y'].tolist(), \n                           \"z_1\": pmmcif.df['ATOM'][pmmcif.df['ATOM']['label_asym_id'] == asym_id]['Cartn_z'].tolist(), \n                           \"resname\": pmmcif.df['ATOM'][pmmcif.df['ATOM']['label_asym_id'] == asym_id]['label_comp_id'].tolist()}) \n    \n    xyz_df = pd.DataFrame({\"x_1\": labels[index][:, 0], \"y_1\": labels[index][:, 1], \"z_1\": labels[index][:, 2], \n                           \"resname\": list(sequences[index])})\n    \n    fig1 = plot_structure(xyz_df_raw, target_ids[index])\n    fig2 = plot_structure(xyz_df, target_ids[index])\n    \n    metric_figure = make_subplots(\n        rows=1, cols=2,\n        specs=[[{'type': 'surface'}, {'type': 'surface'}]],  # First row: 3D surfaces\n        subplot_titles=(\"{} - Original 3D Structure\".format(target_ids[index]), \"{} - Average 3D position of each nucleotide\".format(target_ids[index]))\n    )\n    for t in fig1.data:\n        metric_figure.append_trace(t, row=1, col=1)\n    for t in fig2.data:\n        metric_figure.append_trace(t, row=1, col=2)\n    metric_figure.show()\n    \n    print(\n        \"- target_id: {},\\n- sequences: {},\\n- temporal_cutoffs: {},\\n- descriptions: {},\\n- labels (x,y,z): \\n{}\".format(\n            target_ids[index], sequences[index], temporal_cutoffs[index], descriptions[index], labels[index][:10])\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T01:33:54.882150Z","iopub.execute_input":"2025-03-31T01:33:54.882547Z","iopub.status.idle":"2025-03-31T01:33:59.584920Z","shell.execute_reply.started":"2025-03-31T01:33:54.882493Z","shell.execute_reply":"2025-03-31T01:33:59.583715Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SAVE DATA","metadata":{}},{"cell_type":"code","source":"config_data = {\n    'target_ids': target_ids,\n    'sequences': sequences,\n    'temporal_cutoffs': temporal_cutoffs,\n    'descriptions': descriptions\n}\n\nconfig_filename = 'RNA_Data.json'\nwith open(config_filename, 'w') as config_file:\n    json.dump(config_data, config_file)\nprint(f\"Data successfully written to {config_filename}\")\n\n# Reading the data back\n# with open(config_filename, 'r') as config_file:\n#     data_loaded = json.load(config_file)\n\n# print(\"Data loaded from file:\")\n# print(data_loaded)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T01:33:59.586049Z","iopub.execute_input":"2025-03-31T01:33:59.586424Z","iopub.status.idle":"2025-03-31T01:33:59.594211Z","shell.execute_reply.started":"2025-03-31T01:33:59.586382Z","shell.execute_reply":"2025-03-31T01:33:59.592792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.save('RNA_Labels.npy', np.array(labels, dtype=object), allow_pickle=True)\n#np.load('RNA_Labels.npy', allow_pickle=True)\nlen(labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T01:33:59.595330Z","iopub.execute_input":"2025-03-31T01:33:59.595715Z","iopub.status.idle":"2025-03-31T01:33:59.625951Z","shell.execute_reply.started":"2025-03-31T01:33:59.595674Z","shell.execute_reply":"2025-03-31T01:33:59.624942Z"}},"outputs":[],"execution_count":null}]}