{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":12024591,"sourceType":"competition"},{"sourceId":11676201,"sourceType":"datasetVersion","datasetId":7328231},{"sourceId":11690692,"sourceType":"datasetVersion","datasetId":7337695}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        os.path.join(dirname, filename)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-05T20:02:45.096067Z","iopub.execute_input":"2025-05-05T20:02:45.096360Z","iopub.status.idle":"2025-05-05T20:02:46.546247Z","shell.execute_reply.started":"2025-05-05T20:02:45.096303Z","shell.execute_reply":"2025-05-05T20:02:46.545696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install viennarna\n!pip install forgi\n!pip install LinearFold\n!pip install --no-deps pydca\n!pip install biopython\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T20:02:47.461517Z","iopub.execute_input":"2025-05-05T20:02:47.462297Z","iopub.status.idle":"2025-05-05T20:03:23.580001Z","shell.execute_reply.started":"2025-05-05T20:02:47.462271Z","shell.execute_reply":"2025-05-05T20:03:23.579031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nlab_df1 = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\n#lab_df2 = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.v2.csv\")\n#lab_df = pd.concat([lab_df1, lab_df2], ignore_index=True)\nlab_df = lab_df1\nprint(lab_df.shape)\nprint(lab_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T20:03:23.581607Z","iopub.execute_input":"2025-05-05T20:03:23.581883Z","iopub.status.idle":"2025-05-05T20:03:23.875780Z","shell.execute_reply.started":"2025-05-05T20:03:23.581861Z","shell.execute_reply":"2025-05-05T20:03:23.875066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nseq_df1 = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\n#seq_df2 = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.v2.csv\")\n#seq_df = pd.concat([seq_df1, seq_df2], ignore_index=True)\nseq_df = seq_df1\nprint(seq_df.shape)\nprint(seq_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T20:03:29.045468Z","iopub.execute_input":"2025-05-05T20:03:29.045738Z","iopub.status.idle":"2025-05-05T20:03:29.107766Z","shell.execute_reply.started":"2025-05-05T20:03:29.045718Z","shell.execute_reply":"2025-05-05T20:03:29.107098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seq_df = pd.read_csv('/kaggle/input/seq-df-2d/seq_df_2D.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T20:03:31.888852Z","iopub.execute_input":"2025-05-05T20:03:31.889114Z","iopub.status.idle":"2025-05-05T20:03:32.373911Z","shell.execute_reply.started":"2025-05-05T20:03:31.889095Z","shell.execute_reply":"2025-05-05T20:03:32.373347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# seq_df = seq_df.iloc[:200].reset_index(drop=True)\n\nprint(seq_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T20:03:33.220026Z","iopub.execute_input":"2025-05-05T20:03:33.220348Z","iopub.status.idle":"2025-05-05T20:03:33.226216Z","shell.execute_reply.started":"2025-05-05T20:03:33.220284Z","shell.execute_reply":"2025-05-05T20:03:33.225495Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### MSA Depth","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport os\nfrom tqdm import tqdm\nfrom concurrent.futures import ProcessPoolExecutor\n\ndef msa_depth(target_id: str) -> int:\n    path = f\"/kaggle/input/stanford-rna-3d-folding/MSA/{target_id}.MSA.fasta\"\n    try:\n        with open(path, \"r\") as f:\n            return sum(1 for line in f if line.startswith(\">\"))\n    except FileNotFoundError:\n        return 0\n\ntarget_ids = seq_df[\"target_id\"].tolist()\ndepths = []\nwith ProcessPoolExecutor() as exec:\n    for d in tqdm(exec.map(msa_depth, target_ids), total=len(target_ids), desc=\"MSA depth\"):\n        depths.append(d)\n\nseq_df[\"msa_depth\"] = depths\nprint(seq_df[[\"target_id\",\"msa_depth\"]].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T19:34:29.452800Z","iopub.execute_input":"2025-05-05T19:34:29.453079Z","iopub.status.idle":"2025-05-05T19:34:29.617010Z","shell.execute_reply.started":"2025-05-05T19:34:29.453060Z","shell.execute_reply":"2025-05-05T19:34:29.616027Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### MSA Diversity","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom Bio import AlignIO\nfrom multiprocessing import Pool, cpu_count\nfrom tqdm.auto import tqdm\n\ndef compute_msa_diversity(target_id: str):\n    \"\"\"\n    Read MSA/{target_id}.MSA.fasta, compute average pairwise identity.\n    Returns (target_id, diversity)\n    \"\"\"\n    path = f\"/kaggle/input/stanford-rna-3d-folding/MSA/{target_id}.MSA.fasta\"\n    try:\n        aln = AlignIO.read(path, \"fasta\")\n    except Exception:\n        return target_id, np.nan\n\n    seqs = [str(rec.seq) for rec in aln]\n    n = len(seqs)\n    if n < 2:\n        return target_id, 1.0\n\n    L = len(seqs[0])\n    arr = np.array([list(s) for s in seqs], dtype='<U1')\n\n    total_matches = 0\n    total_positions = 0\n    for i in range(n - 1):\n        eq = (arr[i] == arr[i+1:])\n        total_matches   += eq.sum()\n        total_positions += eq.size\n\n    diversity = total_matches / total_positions\n    return target_id, float(diversity)\n\nif __name__ == \"__main__\":\n    with Pool(processes=cpu_count()) as pool:\n        results = list(tqdm(\n            pool.imap_unordered(compute_msa_diversity, seq_df.target_id),\n            total=len(seq_df),\n            desc=\"Computing MSA diversity\"\n        ))\n\n    diversity_dict = dict(results)\n    seq_df[\"msa_diversity\"] = seq_df[\"target_id\"].map(diversity_dict)\n\n    print(seq_df[[\"target_id\",\"msa_diversity\"]].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T19:37:43.133501Z","iopub.execute_input":"2025-05-05T19:37:43.133857Z","iopub.status.idle":"2025-05-05T19:47:00.698722Z","shell.execute_reply.started":"2025-05-05T19:37:43.133828Z","shell.execute_reply":"2025-05-05T19:47:00.697543Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### GC Content Count","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom multiprocessing import Pool, cpu_count\nfrom tqdm import tqdm\n\ndef compute_gc(seq: str) -> float:\n    L = len(seq)\n    if L == 0:\n        return 0.0\n    return (seq.count('G') + seq.count('C')) / L\n\ndef parallel_gc(seqs, n_workers=None):\n    if n_workers is None:\n        n_workers = cpu_count()\n    with Pool(n_workers) as pool:\n        results = list(tqdm(pool.imap(compute_gc, seqs),\n                            total=len(seqs),\n                            desc=\"Computing GC\"))\n    return results\n\nseqs = seq_df['sequence'].tolist()\ngc_vals = parallel_gc(seqs)\nseq_df['gc_content'] = gc_vals\n\nprint(seq_df[['target_id','sequence','gc_content']].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T19:50:04.310452Z","iopub.execute_input":"2025-05-05T19:50:04.310835Z","iopub.status.idle":"2025-05-05T19:50:04.397658Z","shell.execute_reply.started":"2025-05-05T19:50:04.310805Z","shell.execute_reply":"2025-05-05T19:50:04.396757Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Predicted Number of Stems & Loops","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport RNA\nfrom concurrent.futures import ProcessPoolExecutor\nfrom tqdm.auto import tqdm\n\ndef predict_secstruct(seq: str):\n    \"\"\"Return dot-bracket, MFE, and trimmed pairing-probability matrix\"\"\"\n    fc = RNA.fold_compound(seq)\n    dotbracket, mfe = fc.mfe()\n    fc.pf()\n    pmat_full = fc.bpp()\n    pmat = np.array(pmat_full)\n    L = len(dotbracket)\n    pprob = pmat[1:L+1, 1:L+1]\n    return dotbracket, mfe, pprob\n\ndef count_stems_loops(dot: str):\n    \"\"\"\n    Count the number of stems (contiguous '(' runs) and loops (contiguous '.' runs)\n    in a dot-bracket string\n    \"\"\"\n    stems = 0\n    in_stem = False\n    loops = 0\n    in_loop = False\n    for c in dot:\n        if c == '(':\n            if not in_stem:\n                stems += 1\n            in_stem = True\n        else:\n            in_stem = False\n        if c == '.':\n            if not in_loop:\n                loops += 1\n            in_loop = True\n        else:\n            in_loop = False\n    return stems, loops\n\ndef process_sequence(seq: str):\n    dot, mfe, pprob = predict_secstruct(seq)\n    return count_stems_loops(dot)\n\nwith ProcessPoolExecutor() as executor:\n    results = list(\n        tqdm(\n            executor.map(process_sequence, seq_df['sequence']),\n            total=len(seq_df),\n            desc=\"Counting stems/loops\"\n        )\n    )\n\nstems, loops = zip(*results)\nseq_df['n_stems'] = stems\nseq_df['n_loops'] = loops\n\nprint(seq_df[['target_id', 'n_stems', 'n_loops']].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T19:51:19.769961Z","iopub.execute_input":"2025-05-05T19:51:19.770735Z","iopub.status.idle":"2025-05-05T19:51:32.720658Z","shell.execute_reply.started":"2025-05-05T19:51:19.770696Z","shell.execute_reply":"2025-05-05T19:51:32.719665Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Ligand/Metal Ion Annotation","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport re\nfrom multiprocessing import Pool, cpu_count\nfrom tqdm.auto import tqdm\n\ndef parse_description(desc: str):\n    \"\"\"\n    Returns (has_metal, has_ligand, category)\n    where category in {'none','metal','ligand','both'}.\n    \"\"\"\n    s = (desc or \"\").lower()\n\n    metals = [\n        r'\\bmg\\b', r'\\bmagnesium\\b',\n        r'\\bca\\b', r'\\bcalcium\\b',\n        r'\\bzn\\b', r'\\bzinc\\b',\n        r'\\bmn\\b', r'\\bmanganese\\b',\n        r'\\bco\\b', r'\\bcobalt\\b'\n    ]\n    has_metal = any(re.search(pat, s) for pat in metals)\n\n    has_ligand = ('ligand' in s) or ('bound to' in s) or ('binding to' in s)\n\n    if has_metal and has_ligand:\n        cat = 'both'\n    elif has_metal:\n        cat = 'metal'\n    elif has_ligand:\n        cat = 'ligand'\n    else:\n        cat = 'none'\n\n    return has_metal, has_ligand, cat\n\ndef _worker(desc):\n    return parse_description(desc)\n\nif __name__ == \"__main__\":\n    n_procs = max(1, cpu_count() - 1)\n\n    with Pool(n_procs) as pool:\n        results = list(\n            tqdm(\n                pool.imap(_worker, seq_df[\"description\"].fillna(\"\")),\n                total=len(seq_df),\n                desc=\"Parsing descriptions\"\n            )\n        )\n\n    metals, ligs, cats = zip(*results)\n    seq_df[\"has_metal\"]             = metals\n    seq_df[\"has_ligand\"]            = ligs\n    seq_df[\"ligand_metal_category\"] = cats\n\n    print(seq_df[[\"target_id\",\"has_metal\",\"has_ligand\",\"ligand_metal_category\"]].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T19:55:43.786020Z","iopub.execute_input":"2025-05-05T19:55:43.786387Z","iopub.status.idle":"2025-05-05T19:55:43.885576Z","shell.execute_reply.started":"2025-05-05T19:55:43.786359Z","shell.execute_reply":"2025-05-05T19:55:43.884542Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Experimental Method","metadata":{}},{"cell_type":"code","source":"import re\nimport pandas as pd\nfrom multiprocessing import Pool, cpu_count\nfrom tqdm import tqdm\n\ndef extract_experimental_method(all_seq_str: str) -> str:\n    \"\"\"\n    Pulls the 'method' out of a FASTA header, e.g.\n      >1SCL_A ... method: X-ray diffraction; resolution: 2.9 Å ...\n    Returns the method (e.g. \"X-ray diffraction\") or None if not found\n    \"\"\"\n    if not isinstance(all_seq_str, str):\n        return None\n    lines = all_seq_str.strip().splitlines()\n    if not lines:\n        return None\n    header = lines[0]\n    m = re.search(r\"method[:=]\\s*([^;|\\s]+(?:\\s+[^;|]+)*)\", header, flags=re.IGNORECASE)\n    if m:\n        return m.group(1).strip()\n    return None\n\nif __name__ == \"__main__\":\n    all_seqs = seq_df[\"all_sequences\"].fillna(\"\").tolist()\n\n    with Pool(processes=cpu_count()) as pool:\n        methods = list(\n            tqdm(pool.imap(extract_experimental_method, all_seqs),\n                 total=len(all_seqs),\n                 desc=\"Extracting expt methods\")\n        )\n\n    seq_df[\"experimental_method\"] = methods\n\n    print(seq_df[[\"target_id\", \"experimental_method\"]].head(10))\n\n    seq_df[\"experimental_method\"] = seq_df[\"experimental_method\"].astype(\"category\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T19:59:28.600114Z","iopub.execute_input":"2025-05-05T19:59:28.601002Z","iopub.status.idle":"2025-05-05T19:59:28.693169Z","shell.execute_reply.started":"2025-05-05T19:59:28.600963Z","shell.execute_reply":"2025-05-05T19:59:28.692126Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Partner Chains Count","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport re\nfrom tqdm.auto import tqdm\nfrom concurrent.futures import ProcessPoolExecutor\nimport os\n\ndef count_chains(fasta_str: str) -> int:\n    if pd.isna(fasta_str) or fasta_str.strip() == \"\":\n        return 0\n    # count lines that start with '>'\n    return len([_ for _ in fasta_str.splitlines() if _.startswith(\">\")])\n\nn = len(seq_df)\nwith ProcessPoolExecutor(max_workers=os.cpu_count()) as exe:\n    counts = list(tqdm(exe.map(count_chains, seq_df[\"all_sequences\"]),\n                       total=n,\n                       desc=\"Counting partner chains\"))\n\nseq_df[\"partner_chains_count\"] = counts\n\nprint(seq_df[[\"target_id\",\"partner_chains_count\"]].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T20:01:04.051957Z","iopub.execute_input":"2025-05-05T20:01:04.052720Z","iopub.status.idle":"2025-05-05T20:01:04.181230Z","shell.execute_reply.started":"2025-05-05T20:01:04.052688Z","shell.execute_reply":"2025-05-05T20:01:04.180448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seq_df.to_csv(\"seq_df_global.csv\", index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}