{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**RNA Sequence Folding Structures: Visualizing Secondary and Tertiary (3D) Structures**\n\nExploring RNA sequence folding through both secondary and tertiary (3D) visualizations. The secondary structure is created using vienna RNAfold, which generates the dot-bracket notation used for structural representation.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\npd.set_option('display.max_colwidth', None)\npd.set_option('display.max_rows', None)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        break;\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-22T00:30:01.765134Z","iopub.execute_input":"2025-04-22T00:30:01.765476Z","iopub.status.idle":"2025-04-22T00:30:04.050668Z","shell.execute_reply.started":"2025-04-22T00:30:01.765438Z","shell.execute_reply":"2025-04-22T00:30:04.049644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#read train_sequences.csv file\ntrain_sequences = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\nprint(train_sequences.shape)\ntrain_sequences.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T00:30:04.051661Z","iopub.execute_input":"2025-04-22T00:30:04.052138Z","iopub.status.idle":"2025-04-22T00:30:04.155410Z","shell.execute_reply.started":"2025-04-22T00:30:04.052111Z","shell.execute_reply":"2025-04-22T00:30:04.154362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#create squence_length column\ntrain_sequences['sequence_length'] = train_sequences['sequence'].str.len()\n#print summary of sequence_length and count\nprint(train_sequences['sequence_length'].describe(percentiles=[0.1, 0.2, 0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9, 0.95]))\n#group by sequence_length and count\nsequence_length_count = train_sequences.groupby('sequence_length').size().reset_index(name='count').sort_values(by='count', ascending=False).reset_index(drop=True)\nsequence_length_count.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T00:30:04.156427Z","iopub.execute_input":"2025-04-22T00:30:04.156732Z","iopub.status.idle":"2025-04-22T00:30:04.188051Z","shell.execute_reply.started":"2025-04-22T00:30:04.156706Z","shell.execute_reply":"2025-04-22T00:30:04.187066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#load train_labels.csv\ntrain_labels = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\nprint(train_labels.shape)\ntrain_labels.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T00:30:04.190061Z","iopub.execute_input":"2025-04-22T00:30:04.190424Z","iopub.status.idle":"2025-04-22T00:30:04.502227Z","shell.execute_reply.started":"2025-04-22T00:30:04.190389Z","shell.execute_reply":"2025-04-22T00:30:04.501287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nif os.path.exists(\"/kaggle\"):\n    !apt-get update && apt-get install -y vienna-rna","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T00:30:04.503514Z","iopub.execute_input":"2025-04-22T00:30:04.504033Z","iopub.status.idle":"2025-04-22T00:30:21.903337Z","shell.execute_reply.started":"2025-04-22T00:30:04.503966Z","shell.execute_reply":"2025-04-22T00:30:21.902258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom matplotlib.patches import Arc\nfrom mpl_toolkits.mplot3d import Axes3D\nimport subprocess\nimport re\n\ndef run_rnafold(sequence):\n    result = subprocess.run(['RNAfold'], input=sequence.encode(), stdout=subprocess.PIPE)\n    output = result.stdout.decode().split('\\n')\n    dot_bracket = output[1].split(' ')[0].strip()\n    return dot_bracket\n\n\ndef plot_dot_bracket(dot_bracket, ax, sequence):\n    stack = []\n    pairs = []\n    for i, symbol in enumerate(dot_bracket):\n        if symbol == '(':\n            stack.append(i)\n        elif symbol == ')':\n            j = stack.pop()\n            pairs.append((j, i))\n\n    # Draw backbone line\n    ax.plot(range(len(dot_bracket)), [0]*len(dot_bracket), color='gray', lw=1)\n\n    # Draw base pair arcs\n    for i, j in pairs:\n        arc = Arc(((i + j) / 2, 0), j - i, 1, theta1=0, theta2=180)\n        ax.add_patch(arc)\n\n    # Draw nucleotide letters below the line\n    for i, nt in enumerate(sequence):\n        ax.text(i, -0.2, nt, ha='center', va='center', fontsize=8, color='black')\n\n    ax.set_ylim(-1, 1.5)\n    ax.set_title(\"RNAfold Dot-Bracket Pairing\")\n    ax.axis('off')\n    \ndef sequence_2d_3d_visual(tgt_id):\n    # Filter data\n    seq_row = train_sequences[train_sequences[\"target_id\"] == tgt_id].iloc[0]\n    seq = seq_row[\"sequence\"]\n    labels = train_labels[train_labels[\"ID\"].str.startswith(tgt_id)]\n    # Extract sequence (nt letters) as list\n    nt_sequence = list(seq)\n\n    # 3D coordinates\n    x = labels[\"x_1\"]\n    y = labels[\"y_1\"]\n    z = labels[\"z_1\"]\n    \n    # Run RNAfold\n    dot_bracket = run_rnafold(seq)\n\n    # Plot\n    fig = plt.figure(figsize=(12, 6))\n    ax1 = fig.add_subplot(121)\n    plot_dot_bracket(dot_bracket, ax1, sequence=seq)\n    \n    # 3D panel\n    ax2 = fig.add_subplot(122, projection='3d')\n    ax2.plot(x, y, z, marker='o', color='blue')\n    ax2.set_title(f\"3D Backbone for {tgt_id}\")\n    ax2.set_xlabel(\"X\")\n    ax2.set_ylabel(\"Y\")\n    ax2.set_zlabel(\"Z\")\n\n    # Add base letters as annotations\n    for i, (xi, yi, zi) in enumerate(zip(x, y, z)):\n        if i < len(nt_sequence):\n            ax2.text(xi, yi, zi, nt_sequence[i], fontsize=8, color='red')\n    \n    plt.suptitle(f\"RNA Secondary vs Tertiary Structure\\nTarget ID: {tgt_id}\")\n    plt.tight_layout()\n    plt.show()\n\n\nlengths = [10, 20, 30, 40, 50]\n\nfor length in lengths:\n    match = train_sequences[train_sequences['sequence_length'] == length]\n    if not match.empty:\n        target_id = match['target_id'].iloc[0]\n        print(f\"\\nLength {length}: {target_id}, Sequence: {match['sequence'].iloc[0]}\")\n        sequence_2d_3d_visual(target_id)\n    else:\n        print(f\"Length {length}: No match found\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T00:56:02.296278Z","iopub.execute_input":"2025-04-22T00:56:02.296683Z","iopub.status.idle":"2025-04-22T00:56:04.606095Z","shell.execute_reply.started":"2025-04-22T00:56:02.296655Z","shell.execute_reply":"2025-04-22T00:56:04.605138Z"}},"outputs":[],"execution_count":null}]}