{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        os.path.join(dirname, filename)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-03T05:51:36.923950Z","iopub.execute_input":"2025-05-03T05:51:36.924595Z","iopub.status.idle":"2025-05-03T05:51:39.535530Z","shell.execute_reply.started":"2025-05-03T05:51:36.924572Z","shell.execute_reply":"2025-05-03T05:51:39.534554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\ntrain_labels_path = \"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\"\ntrain_labels = pd.read_csv(train_labels_path)\nprint(train_labels.head(10))\nprint(train_labels.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T07:19:58.271568Z","iopub.execute_input":"2025-05-03T07:19:58.271910Z","iopub.status.idle":"2025-05-03T07:19:58.450995Z","shell.execute_reply.started":"2025-05-03T07:19:58.271885Z","shell.execute_reply":"2025-05-03T07:19:58.450033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\ntrain_sequences_path = \"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\"\ntrain_sequences = pd.read_csv(train_sequences_path)\nprint(train_sequences.head(10))\nprint(train_sequences.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T07:20:11.898280Z","iopub.execute_input":"2025-05-03T07:20:11.898582Z","iopub.status.idle":"2025-05-03T07:20:11.934269Z","shell.execute_reply.started":"2025-05-03T07:20:11.898553Z","shell.execute_reply":"2025-05-03T07:20:11.933545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import plotly.graph_objects as go\nimport pandas as pd\nimport ipywidgets as widgets\nfrom IPython.display import display, clear_output\n\ntrain_labels = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\n\ntrain_labels['protein_id'] = train_labels['ID'].str.split('_').str[:2].str.join('_')\n\nunique_protein_ids = train_labels['protein_id'].unique()\n\noutput = widgets.Output()\n\nprotein_dropdown = widgets.Dropdown(\n    options=unique_protein_ids,\n    description='Protein ID:',\n    disabled=False,\n)\n\ndef show_plot(protein_id):\n    with output:\n        clear_output(wait=True)\n        df = train_labels[train_labels['protein_id'] == protein_id]\n        colors = {'A': 'red', 'C': 'green', 'G': 'blue', 'U': 'purple'}\n        fig = go.Figure()\n        fig.add_trace(\n            go.Scatter3d(\n                x=df['x_1'], y=df['y_1'], z=df['z_1'],\n                mode='lines',\n                line=dict(color='gray', width=4),\n                name='Backbone'\n            )\n        )\n        for base in ['A', 'C', 'G', 'U']:\n            subset = df[df['resname'] == base]\n            if not subset.empty:\n                fig.add_trace(\n                    go.Scatter3d(\n                        x=subset['x_1'], y=subset['y_1'], z=subset['z_1'],\n                        mode='markers',\n                        marker=dict(\n                            color=colors[base],\n                            size=6,\n                            opacity=0.8,\n                            symbol='circle'\n                        ),\n                        name=f'{base}'\n                    )\n                )\n                \n        fig.update_layout(\n            title=f'3D RNA Structure: {protein_id}',\n            scene=dict(\n                xaxis_title='X (Å)',\n                yaxis_title='Y (Å)',\n                zaxis_title='Z (Å)',\n            ),\n            legend_title='Nucleotides',\n            width=800,\n            height=600\n        )\n        \n        fig.show()\n\ndef on_protein_change(change):\n    show_plot(change.new)\n\nprotein_dropdown.observe(on_protein_change, names='value')\n\nshow_plot(protein_dropdown.value)\n\ndisplay(widgets.VBox([protein_dropdown, output]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T05:51:39.867297Z","iopub.execute_input":"2025-05-03T05:51:39.867542Z","iopub.status.idle":"2025-05-03T05:51:41.550034Z","shell.execute_reply.started":"2025-05-03T05:51:39.867522Z","shell.execute_reply":"2025-05-03T05:51:41.549208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\ntrain_labels[train_labels['ID'].str.startswith(\"1SCL_A\")]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T05:51:41.550870Z","iopub.execute_input":"2025-05-03T05:51:41.551153Z","iopub.status.idle":"2025-05-03T05:51:41.611457Z","shell.execute_reply.started":"2025-05-03T05:51:41.551130Z","shell.execute_reply":"2025-05-03T05:51:41.610487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install biopython","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T05:51:41.612574Z","iopub.execute_input":"2025-05-03T05:51:41.612893Z","iopub.status.idle":"2025-05-03T05:51:48.900324Z","shell.execute_reply.started":"2025-05-03T05:51:41.612865Z","shell.execute_reply":"2025-05-03T05:51:48.899231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from Bio import AlignIO\n\nfile_path = \"/kaggle/input/stanford-rna-3d-folding/MSA/1SCL_A.MSA.fasta\"\n\nalignment = AlignIO.read(file_path, \"fasta\")\n\nprint(f\"Number of sequences: {len(alignment)}\")\nprint(f\"Alignment length (columns): {alignment.get_alignment_length()}\\n\")\n\nfor rec in alignment[:10]:\n    print(f\">{rec.id}\")\n    print(rec.seq)\n    print()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T07:07:31.708603Z","iopub.execute_input":"2025-05-03T07:07:31.709559Z","iopub.status.idle":"2025-05-03T07:07:31.733868Z","shell.execute_reply.started":"2025-05-03T07:07:31.709526Z","shell.execute_reply":"2025-05-03T07:07:31.732828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom Bio import AlignIO\n\nseq_df = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\nlengths = seq_df.sequence.str.len()\nprint(\"Length: median\", lengths.median(), \"min\", lengths.min(), \"max\", lengths.max())\n\ndepths = {}\nfor tid in seq_df.target_id.iloc[:10]:\n    aln = AlignIO.read(f\"/kaggle/input/stanford-rna-3d-folding/MSA/{tid}.MSA.fasta\", \"fasta\")\n    depths[tid] = len(aln)\nprint(\"MSA depths:\", depths)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T07:29:10.873161Z","iopub.execute_input":"2025-05-03T07:29:10.873468Z","iopub.status.idle":"2025-05-03T07:29:10.952962Z","shell.execute_reply.started":"2025-05-03T07:29:10.873446Z","shell.execute_reply":"2025-05-03T07:29:10.952238Z"}},"outputs":[],"execution_count":null}]}