{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:25:20.008982Z","iopub.execute_input":"2025-03-31T23:25:20.009234Z","iopub.status.idle":"2025-03-31T23:25:22.641603Z","shell.execute_reply.started":"2025-03-31T23:25:20.009208Z","shell.execute_reply":"2025-03-31T23:25:22.640567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n# /kaggle/input/stanford-rna-3d-folding/sample_submission.csv\n# /kaggle/input/stanford-rna-3d-folding/validation_sequences.csv\n# /kaggle/input/stanford-rna-3d-folding/test_sequences.csv\n# /kaggle/input/stanford-rna-3d-folding/validation_labels.csv\n# /kaggle/input/stanford-rna-3d-folding/train_labels.csv\n# /kaggle/input/stanford-rna-3d-folding/train_sequences.csv\ntrain_sequences = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\n# print(train_sequences.head(10))\n\ntrain_labels = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\n# print(train_labels.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:25:22.642689Z","iopub.execute_input":"2025-03-31T23:25:22.643251Z","iopub.status.idle":"2025-03-31T23:25:23.013183Z","shell.execute_reply.started":"2025-03-31T23:25:22.643211Z","shell.execute_reply":"2025-03-31T23:25:23.012126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kaggle Cell 1 - Setup\n!apt-get install -y vienna-rna\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport subprocess\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:25:23.014170Z","iopub.execute_input":"2025-03-31T23:25:23.014503Z","iopub.status.idle":"2025-03-31T23:25:34.907437Z","shell.execute_reply.started":"2025-03-31T23:25:23.014453Z","shell.execute_reply":"2025-03-31T23:25:34.906121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Predict Structure (RNAfold CLI)\ndef rnafold(sequence):\n    process = subprocess.run(['RNAfold'], input=sequence.encode(), capture_output=True)\n    output = process.stdout.decode()\n    structure = output.strip().split('\\n')[1].split(' ')[0]\n    mfe = float(output.strip().split('\\n')[1].split('(')[-1].strip(')'))\n    return structure, mfe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:25:34.910064Z","iopub.execute_input":"2025-03-31T23:25:34.910385Z","iopub.status.idle":"2025-03-31T23:25:34.916122Z","shell.execute_reply.started":"2025-03-31T23:25:34.910347Z","shell.execute_reply":"2025-03-31T23:25:34.914995Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Simulate SHAPE Reactivity\ndef generate_mock_shape_data(sequence, seed=42):\n    np.random.seed(seed)\n    return np.random.uniform(0, 1, len(sequence))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:25:34.917685Z","iopub.execute_input":"2025-03-31T23:25:34.918006Z","iopub.status.idle":"2025-03-31T23:25:34.935390Z","shell.execute_reply.started":"2025-03-31T23:25:34.917977Z","shell.execute_reply":"2025-03-31T23:25:34.934343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot SHAPE Data\ndef plot_shape(shape_data):\n    plt.figure(figsize=(10,3))\n    plt.bar(range(1, len(shape_data)+1), shape_data)\n    plt.xlabel(\"Nucleotide Position\")\n    plt.ylabel(\"Simulated SHAPE Reactivity\")\n    plt.title(\"Mock SHAPE Reactivity Profile\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:25:34.936386Z","iopub.execute_input":"2025-03-31T23:25:34.936742Z","iopub.status.idle":"2025-03-31T23:25:34.954283Z","shell.execute_reply.started":"2025-03-31T23:25:34.936705Z","shell.execute_reply":"2025-03-31T23:25:34.953107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Full Example\nsequence = \"GCGGAUUUAGCUCAGUUGGGAGAGCGCCAGACUGAAA\"\n\n# Step 1: Fold without SHAPE\nstructure, mfe = rnafold(sequence)\n\n# Step 2: Generate SHAPE\nshape_data = generate_mock_shape_data(sequence)\n\n# Step 3: Display\nprint(f\"Sequence: {sequence}\")\nprint(f\"Predicted Structure: {structure}\")\nprint(f\"MFE: {mfe} kcal/mol\")\n\n# Step 4: Visualize SHAPE\nplot_shape(shape_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:25:34.955323Z","iopub.execute_input":"2025-03-31T23:25:34.955643Z","iopub.status.idle":"2025-03-31T23:25:35.311233Z","shell.execute_reply.started":"2025-03-31T23:25:34.955615Z","shell.execute_reply":"2025-03-31T23:25:35.310307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_sequences.index)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:25:35.312038Z","iopub.execute_input":"2025-03-31T23:25:35.312287Z","iopub.status.idle":"2025-03-31T23:25:35.317996Z","shell.execute_reply.started":"2025-03-31T23:25:35.312266Z","shell.execute_reply":"2025-03-31T23:25:35.317039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom mpl_toolkits.mplot3d import Axes3D\n\n# 1. Sequence Length Distribution\ntrain_sequences['sequence_length'] = train_sequences['sequence'].apply(len)\n\nplt.figure(figsize=(8, 5))\nsns.histplot(train_sequences['sequence_length'], bins=30, kde=True)\nplt.title(\"Sequence Length Distribution\")\nplt.xlabel(\"Sequence Length\")\nplt.ylabel(\"Frequency\")\nplt.show()\n\n# 2. Base Composition\nbases = ['A', 'C', 'G', 'U']\nbase_counts = {base: train_sequences['sequence'].str.count(base).sum() for base in bases}\n\nplt.figure(figsize=(6, 4))\nsns.barplot(x=list(base_counts.keys()), y=list(base_counts.values()))\nplt.title(\"Base Composition Across All Sequences\")\nplt.xlabel(\"Base\")\nplt.ylabel(\"Count\")\nplt.show()\n\n# 3. Residue Positioning (resid vs target_id)\nplt.figure(figsize=(10, 6))\nsns.histplot(train_labels['resid'], bins=50, kde=True)\nplt.title(\"Residue Index Distribution\")\nplt.xlabel(\"Residue Index (resid)\")\nplt.ylabel(\"Frequency\")\nplt.show()\n\n# 4. 3D Scatter Plot of Coordinates\nfig = plt.figure(figsize=(8, 6))\nax = fig.add_subplot(111, projection='3d')\nax.scatter(train_labels['x_1'], train_labels['y_1'], train_labels['z_1'], s=1, alpha=0.5)\nax.set_title(\"3D Scatter of Nucleotide Positions\")\nax.set_xlabel(\"x\")\nax.set_ylabel(\"y\")\nax.set_zlabel(\"z\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:25:35.318919Z","iopub.execute_input":"2025-03-31T23:25:35.319182Z","iopub.status.idle":"2025-03-31T23:25:40.303519Z","shell.execute_reply.started":"2025-03-31T23:25:35.319151Z","shell.execute_reply":"2025-03-31T23:25:40.302600Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install biopython\n!pip install torch_geometric\n# etc...\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:25:40.304676Z","iopub.execute_input":"2025-03-31T23:25:40.305402Z","iopub.status.idle":"2025-03-31T23:25:52.890278Z","shell.execute_reply.started":"2025-03-31T23:25:40.305361Z","shell.execute_reply":"2025-03-31T23:25:52.889068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# RNA Structure Predictor Notebook (BioEmu-inspired)\n\n# ================\n# Step 1 - Load CSVs \nfrom Bio import SeqIO\n\n# ================\n# Step 2 - Visualize a sample\n\ndef plot_sample_structure(target_id):\n    df = train_labels[train_labels['ID'].str.startswith(target_id)]\n    print(f\"has len of target_id df {len(df.index)}\")\n    plt.scatter(df['x_1'], df['y_1'], s=1)\n    plt.title(f\"{target_id} 2D projection\")\n    plt.xlabel(\"x_1\")\n    plt.ylabel(\"y_1\")\n    plt.axis('equal')\n    plt.show()\n\nfirst_train_seq = train_sequences['sequence'].iloc[0]\nprint(f\" length of the sequence is: {len(first_train_seq)}\")\nprint(f\" the 2D plot of the sequence: {train_sequences['sequence'].iloc[0]}\")\nplot_sample_structure(train_sequences['target_id'].iloc[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:25:52.891676Z","iopub.execute_input":"2025-03-31T23:25:52.892067Z","iopub.status.idle":"2025-03-31T23:25:53.333408Z","shell.execute_reply.started":"2025-03-31T23:25:52.892029Z","shell.execute_reply":"2025-03-31T23:25:53.332543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef plot_sample_structure(target_id, sequence):\n    df = train_labels[train_labels['ID'].str.startswith(target_id)]\n    print(f\"Target {target_id} has {len(df.index)} coordinates\")\n    \n    # 3D Scatter Plot\n    fig = plt.figure(figsize=(7, 6))\n    ax = fig.add_subplot(111, projection='3d')\n    ax.scatter(df['x_1'], df['y_1'], df['z_1'], s=10)\n    # Annotate each point with its nucleotide\n    for i, base in enumerate(sequence):\n        if i < len(df):\n            ax.text(df.loc[i, 'x_1'], df.loc[i, 'y_1'], df.loc[i, 'z_1'], base, fontsize=8)\n    ax.set_title(f\"{target_id} - 3D Structure\")\n    ax.set_xlabel('x_1')\n    ax.set_ylabel('y_1')\n    ax.set_zlabel('z_1')\n    plt.show()\n    \n    # 2D Projections\n    \n    # Projection 1: X vs Y\n    # figxy = plt.figure(figsize=(5,5))\n    fig, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize=(7, 3))\n    ax1.plot(df['x_1'], df['y_1'])\n    for i, base in enumerate(sequence):\n        if i < len(df):\n            ax1.text(df.loc[i, 'x_1'], df.loc[i, 'y_1'], base, fontsize=8)\n    ax1.set_title(f\"(X vs Y)\")\n    ax1.set_xlabel('x_1')\n    ax1.set_ylabel('y_1')\n    # plt.show()\n\n    # Projection 2: X vs Z\n    ax2.plot(df['x_1'], df['z_1'])\n    for i, base in enumerate(sequence):\n        if i < len(df):\n            ax2.text(df.loc[i, 'x_1'], df.loc[i, 'z_1'], base, fontsize=8)\n    ax2.set_title(f\"(X vs Z)\")\n    ax2.set_xlabel('x_1')\n    ax2.set_ylabel('z_1')\n    # plt.show()\n    \n    # # Projection 3: Y vs Z\n    ax3.plot(df['y_1'], df['z_1'])\n    for i, base in enumerate(sequence):\n        if i < len(df):\n            ax3.text(df.loc[i, 'y_1'], df.loc[i, 'z_1'], base, fontsize=8)\n    ax3.set_title(f\"(Y vs Z)\")\n    ax3.set_xlabel('y_1')\n    ax3.set_ylabel('z_1')\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:25:53.334356Z","iopub.execute_input":"2025-03-31T23:25:53.334645Z","iopub.status.idle":"2025-03-31T23:25:53.344873Z","shell.execute_reply.started":"2025-03-31T23:25:53.334621Z","shell.execute_reply":"2025-03-31T23:25:53.343770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"first_train_seq = train_sequences['sequence'].iloc[0]\nprint(f\" length of the sequence is: {len(first_train_seq)}\")\nprint(f\" the 2D plot of the sequence: {train_sequences['sequence'].iloc[0]}\")\nplot_sample_structure(train_sequences['target_id'].iloc[0], first_train_seq)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:25:53.348031Z","iopub.execute_input":"2025-03-31T23:25:53.348335Z","iopub.status.idle":"2025-03-31T23:25:54.164841Z","shell.execute_reply.started":"2025-03-31T23:25:53.348307Z","shell.execute_reply":"2025-03-31T23:25:54.163810Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load a sample FASTA file corresponding to the target_id if available\n# Let's assume the FASTA file is named as \"{target_id}.fasta\" in the data folder\ntarget_id = train_sequences['target_id'].iloc[0]\nfasta_file_path = f\"/kaggle/input/stanford-rna-3d-folding/MSA/{target_id}.MSA.fasta\"\n\n# Parse the FASTA file using Biopython\nfasta_sequences = list(SeqIO.parse(fasta_file_path, \"fasta\"))\n\n# Checking if we can infer motifs, ligands, or other information from description\nfasta_data = [{\n    \"id\": fasta.id,\n    \"description\": fasta.description,\n    \"sequence\": str(fasta.seq)\n} for fasta in fasta_sequences]\n\n# Display as dataframe for better readability\nfasta_df = pd.DataFrame(fasta_data)\nfrom IPython.display import display\ndisplay(fasta_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:29:17.256534Z","iopub.execute_input":"2025-03-31T23:29:17.256927Z","iopub.status.idle":"2025-03-31T23:29:17.296744Z","shell.execute_reply.started":"2025-03-31T23:29:17.256897Z","shell.execute_reply":"2025-03-31T23:29:17.295771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\n\ndef extract_fasta_features(fasta_df):\n    motif_keywords = ['hairpin', 'pseudoknot', 'quadruplex', 'stem-loop', 'kissing loop']\n    ligand_keywords = ['ligand', 'bound', 'binding', 'small molecule', 'antibiotic']\n    ion_keywords = ['Mg2+', 'K+', 'Na+', 'metal', 'ion']\n    partner_keywords = ['protein', 'DNA', 'RNA', 'partner', 'interacting']\n    experimental_keywords = ['in vitro', 'in vivo', 'crystal', 'NMR', 'EM']\n\n    features = []\n\n    for idx, row in fasta_df.iterrows():\n        desc = row.get('description', '').lower()\n\n        feature = {\n            'target_id': row.get('target_id', ''),\n            'has_motif': int(any(k in desc for k in motif_keywords)),\n            'has_ligand': int(any(k in desc for k in ligand_keywords)),\n            'has_ion_binding': int(any(k in desc for k in ion_keywords)),\n            'has_partner': int(any(k in desc for k in partner_keywords)),\n            'experimental_tag': int(any(k in desc for k in experimental_keywords)),\n            'num_motifs_mentioned': sum(k in desc for k in motif_keywords),\n            'description_length': len(desc.split())\n        }\n\n        features.append(feature)\n\n    features_df = pd.DataFrame(features)\n    return features_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:29:23.141001Z","iopub.execute_input":"2025-03-31T23:29:23.141387Z","iopub.status.idle":"2025-03-31T23:29:23.149581Z","shell.execute_reply.started":"2025-03-31T23:29:23.141346Z","shell.execute_reply":"2025-03-31T23:29:23.148334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_df = extract_fasta_features(train_sequences)\nprint(features_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:29:26.066569Z","iopub.execute_input":"2025-03-31T23:29:26.066931Z","iopub.status.idle":"2025-03-31T23:29:26.140173Z","shell.execute_reply.started":"2025-03-31T23:29:26.066905Z","shell.execute_reply":"2025-03-31T23:29:26.139328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\n\n# Feature extraction function\ndef extract_features_from_metadata(df):\n    features = []\n\n    for idx, row in df.iterrows():\n        desc = str(row.get('description', '')).lower()\n        all_seq = str(row.get('all_sequences', ''))\n\n        feature = {\n            'target_id': row['target_id'],\n            'has_partner_chain': int(all_seq.count('>') > 1),\n            'num_chains': all_seq.count('>'),\n            'ligand_present': int(bool(re.search(r'ligand|small molecule|antibiotic|drug|inhibitor', desc))),\n            'experiment_type_nmr': int('nmr' in desc),\n            'experiment_type_cryoem': int('cryo-em' in desc or 'cryo em' in desc),\n            'experiment_type_xray': int('x-ray' in desc or 'xray' in desc),\n            'has_protein_interaction': int('protein' in desc),\n            'has_viral_origin': int('viral' in desc or 'virus' in desc),\n            'has_pseudoknot': int('pseudoknot' in desc),\n            'has_gquadruplex': int('g-quadruplex' in desc or 'quadruplex' in desc),\n            'mentions_binding_site': int('binding' in desc or 'site' in desc or 'stem' in desc or 'loop' in desc or 'bulge' in desc)\n        }\n\n        features.append(feature)\n\n    feature_df = pd.DataFrame(features)\n    return feature_df\n\n# Apply feature extraction\nmetadata_features = extract_features_from_metadata(train_sequences)\n\n# Display extracted features\ndisplay(metadata_features)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T23:29:31.755192Z","iopub.execute_input":"2025-03-31T23:29:31.755569Z","iopub.status.idle":"2025-03-31T23:29:31.839860Z","shell.execute_reply.started":"2025-03-31T23:29:31.755541Z","shell.execute_reply":"2025-03-31T23:29:31.838901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}