{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":12024591,"sourceType":"competition"},{"sourceId":11676201,"sourceType":"datasetVersion","datasetId":7328231}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        os.path.join(dirname, filename)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-05T12:52:28.600130Z","iopub.execute_input":"2025-05-05T12:52:28.600408Z","iopub.status.idle":"2025-05-05T12:52:30.216784Z","shell.execute_reply.started":"2025-05-05T12:52:28.600389Z","shell.execute_reply":"2025-05-05T12:52:30.215697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install viennarna\n!pip install forgi\n!pip install LinearFold\n!pip install --no-deps pydca\n!pip install biopython\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T12:52:30.218627Z","iopub.execute_input":"2025-05-05T12:52:30.218927Z","iopub.status.idle":"2025-05-05T12:53:18.452972Z","shell.execute_reply.started":"2025-05-05T12:52:30.218901Z","shell.execute_reply":"2025-05-05T12:53:18.451832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nlab_df1 = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\n#lab_df2 = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.v2.csv\")\n#lab_df = pd.concat([lab_df1, lab_df2], ignore_index=True)\nlab_df = lab_df1\nprint(lab_df.shape)\nprint(lab_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T12:53:18.454188Z","iopub.execute_input":"2025-05-05T12:53:18.454448Z","iopub.status.idle":"2025-05-05T12:53:18.814816Z","shell.execute_reply.started":"2025-05-05T12:53:18.454423Z","shell.execute_reply":"2025-05-05T12:53:18.813610Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nseq_df1 = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\n#seq_df2 = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.v2.csv\")\n#seq_df = pd.concat([seq_df1, seq_df2], ignore_index=True)\nseq_df = seq_df1\nprint(seq_df.shape)\nprint(seq_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T12:53:18.817109Z","iopub.execute_input":"2025-05-05T12:53:18.817438Z","iopub.status.idle":"2025-05-05T12:53:18.887139Z","shell.execute_reply.started":"2025-05-05T12:53:18.817412Z","shell.execute_reply":"2025-05-05T12:53:18.885992Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### One-Hot Encoding of Residues","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\n\nnt2idx = {\"A\": 0, \"C\": 1, \"G\": 2, \"U\": 3}\n\ndef one_hot_encode(seq: str) -> np.ndarray:\n    \"\"\"\n    Convert an RNA sequence into an (L,4) one-hot array\n    Unknown bases (e.g. N) are left as all-zero rows\n    \"\"\"\n    L = len(seq)\n    arr = np.zeros((L, 4), dtype=int)\n    for i, nt in enumerate(seq):\n        j = nt2idx.get(nt)\n        if j is not None:\n            arr[i, j] = 1\n    return arr\n\nexample_seq = seq_df.loc[0, \"sequence\"]\noh = one_hot_encode(example_seq)\nprint(f\"First sequence length: {len(example_seq)} → one-hot shape: {oh.shape}\")\nprint(oh[:5])\n\nseq_df[\"onehot\"] = seq_df[\"sequence\"].apply(one_hot_encode)\nprint(seq_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T09:58:37.745917Z","iopub.execute_input":"2025-05-04T09:58:37.746176Z","iopub.status.idle":"2025-05-04T09:58:37.816454Z","shell.execute_reply.started":"2025-05-04T09:58:37.746156Z","shell.execute_reply":"2025-05-04T09:58:37.815470Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Sequence Length","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\nseq_df[\"length\"] = seq_df[\"sequence\"].str.len()\n\nprint(seq_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T09:58:37.817413Z","iopub.execute_input":"2025-05-04T09:58:37.817663Z","iopub.status.idle":"2025-05-04T09:58:37.843001Z","shell.execute_reply.started":"2025-05-04T09:58:37.817643Z","shell.execute_reply":"2025-05-04T09:58:37.842043Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Mononucleotide Frequence Count","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\ndef compute_mononuc_freq(seq: str) -> np.ndarray:\n    \"\"\"\n    Compute mononucleotide frequencies for A, C, G, U\n    Returns a length-4 vector: [f_A, f_C, f_G, f_U]\n    \"\"\"\n    L = len(seq)\n    freqs = np.array([seq.count(nuc) / L for nuc in [\"A\", \"C\", \"G\", \"U\"]], dtype=float)\n    return freqs\n\nexample_seq = seq_df.loc[0, \"sequence\"]\nprint(\"Frequencies for\", seq_df.loc[0, \"target_id\"], \":\", compute_mononuc_freq(example_seq))\n\nall_freqs = np.stack(seq_df[\"sequence\"].map(compute_mononuc_freq).values)\nseq_df[[\"freq_A\", \"freq_C\", \"freq_G\", \"freq_U\"]] = all_freqs\n\nprint(seq_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T09:58:37.843999Z","iopub.execute_input":"2025-05-04T09:58:37.844330Z","iopub.status.idle":"2025-05-04T09:58:37.887800Z","shell.execute_reply.started":"2025-05-04T09:58:37.844288Z","shell.execute_reply":"2025-05-04T09:58:37.886968Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Di- and Tri-Nucleotide Frequence","metadata":{}},{"cell_type":"code","source":"import itertools\nimport numpy as np\nimport pandas as pd\n\ndef get_kmer_bias(seq: str, k: int):\n    alphabet = ['A','C','G','U']\n    kmers = [''.join(p) for p in itertools.product(alphabet, repeat=k)]\n    counts = dict.fromkeys(kmers, 0)\n    total = 0\n    for i in range(len(seq) - k + 1):\n        kmer = seq[i:i+k]\n        if all(nt in alphabet for nt in kmer):\n            counts[kmer] += 1\n            total += 1\n    if total > 0:\n        freqs = np.array([counts[k]/total for k in kmers], dtype=float)\n    else:\n        freqs = np.zeros(len(kmers), dtype=float)\n    return freqs, kmers\n\ndi_freqs, di_labels = get_kmer_bias(\"ACGU\", k=2)   # labels only\n_,        tri_labels = get_kmer_bias(\"ACGU\", k=3)\n\ndef compute_kmer_features(seq: str) -> pd.Series:\n    di_freqs, _  = get_kmer_bias(seq, 2)\n    tri_freqs, _ = get_kmer_bias(seq, 3)\n    data = {}\n    for label, freq in zip(di_labels, di_freqs):\n        data[f\"di_{label}\"] = freq\n    for label, freq in zip(tri_labels, tri_freqs):\n        data[f\"tri_{label}\"] = freq\n    return pd.Series(data)\n\nkmer_feats = seq_df['sequence'].apply(compute_kmer_features)\n\nseq_df = pd.concat([seq_df, kmer_feats], axis=1)\n\nprint(seq_df.columns.tolist())\nprint(seq_df[['target_id'] + [f\"di_{l}\" for l in di_labels] + [f\"tri_{l}\" for l in tri_labels]].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T09:58:37.888807Z","iopub.execute_input":"2025-05-04T09:58:37.889727Z","iopub.status.idle":"2025-05-04T09:58:38.387058Z","shell.execute_reply.started":"2025-05-04T09:58:37.889695Z","shell.execute_reply":"2025-05-04T09:58:38.386311Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Minimum Free Energy Dot-Brackets","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport RNA\nfrom multiprocessing import Pool, cpu_count\nfrom tqdm import tqdm\nimport LinearFold as lf\n\ndef mfe_and_dot(seq: str):\n    \"\"\"Return (dot-bracket, mfe) for a single RNA - takes much longer\"\"\"\n    fc = RNA.fold_compound(seq)\n    dot, mfe = fc.mfe()\n    return dot, mfe\n\ndef mfe_and_dot_linear(seq):\n    \"\"\"Return (dot-bracket, mfe) for a single RNA - returns an approximate MFE fold\"\"\"\n    dot, mfe = lf.fold(seq)\n    return dot, mfe\n\nseqs = seq_df['sequence'].tolist()\nn_workers = cpu_count()\n\ndots, mfes = [], []\nwith Pool(processes=n_workers) as pool:\n    for dot, mfe in tqdm(pool.imap(mfe_and_dot_linear, seqs),\n                         total=len(seqs),\n                         desc=\"Folding RNAs\"):\n        dots.append(dot)\n        mfes.append(mfe)\n\nseq_df['dotbracket'] = dots\nseq_df['mfe']        = mfes\nseq_df['pair_flag']  = seq_df['dotbracket'].apply(\n    lambda db: [1 if c in '()' else 0 for c in db]\n)\n\nprint(seq_df.columns.tolist())\nprint(seq_df[['target_id'] + ['sequence'] + ['dotbracket'] + ['mfe'] + ['pair_flag']].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T09:58:38.387863Z","iopub.execute_input":"2025-05-04T09:58:38.388117Z","iopub.status.idle":"2025-05-04T09:59:29.801657Z","shell.execute_reply.started":"2025-05-04T09:58:38.388074Z","shell.execute_reply":"2025-05-04T09:59:29.800425Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Loop Type Annotation","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport RNA\nfrom multiprocessing import Pool, cpu_count\nfrom tqdm import tqdm\nimport forgi.graph.bulge_graph as fgb\n\ndef loop_type_onehot(dot: str) -> np.ndarray:\n    bg = fgb.BulgeGraph.from_dotbracket(dot)\n    L = len(dot)\n    onehot = np.zeros((L, 3), int)\n    cat2idx = {\"stem\": 0, \"loop\": 1, \"bulge\": 2}\n    for pos in range(1, L+1):\n        elem = bg.get_elem(pos)\n        if elem.startswith(\"s\"):\n            cat = \"stem\"\n        elif elem.startswith(\"i\"):\n            dims = bg.get_bulge_dimensions(elem)\n            cat = \"bulge\" if (dims[0] == 0 or dims[1] == 0) else \"loop\"\n        else:\n            cat = \"loop\"\n        onehot[pos-1, cat2idx[cat]] = 1\n    return onehot\n\ndef compute_pp_and_onehot(args):\n    \"\"\"Given (sequence, dotbracket), return (pp_matrix, loop_onehot)\"\"\"\n    seq, dot = args\n    fc = RNA.fold_compound(seq)\n    fc.pf()\n    pmat = np.array(fc.bpp())\n    L = len(dot)\n    pp = pmat[1:L+1, 1:L+1]\n    onehot = loop_type_onehot(dot)\n    return pp, onehot\n\ninputs = list(zip(seq_df['sequence'], seq_df['dotbracket']))\n\npp_list, onehot_list = [], []\nwith Pool(processes=cpu_count()) as pool:\n    for pp, onehot in tqdm(pool.imap(compute_pp_and_onehot, inputs),\n                           total=len(inputs),\n                           desc=\"PP + one-hot\"):\n        pp_list.append(pp)\n        onehot_list.append(onehot)\n\nseq_df['pp']               = pp_list\nseq_df['loop_type_onehot'] = onehot_list\n\nprint(\"Shapes on row 0:\")\nprint(\" pp shape:\", seq_df.loc[0,'pp'].shape)\nprint(\" loop_onehot shape:\", seq_df.loc[0,'loop_type_onehot'].shape)\n\nprint(seq_df.columns.tolist())\nprint(seq_df[['target_id'] + ['sequence'] + ['pp'] + ['loop_type_onehot']].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T09:59:29.804375Z","iopub.execute_input":"2025-05-04T09:59:29.804645Z","iopub.status.idle":"2025-05-04T10:07:09.008212Z","shell.execute_reply.started":"2025-05-04T09:59:29.804622Z","shell.execute_reply":"2025-05-04T10:07:09.007110Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Per Residue Gibbs Free Energy Contribution","metadata":{}},{"cell_type":"code","source":"from multiprocessing import Pool, cpu_count\nfrom tqdm import tqdm\nimport numpy as np\nimport pandas as pd\nimport RNA\n\ndef compute_loop_dG_res(seq: str):\n    seq = seq.upper()\n    L = len(seq)\n    fc = RNA.fold_compound(seq)\n    dot, mfe = fc.mfe()\n    ptable = RNA.ptable(dot)\n    loop_contrib = np.zeros(L, dtype=float)\n    for i in range(1, L+1):\n        j = ptable[i]\n        if j > i:\n            E_loop = fc.eval_loop_pt(i, ptable) / 100.0\n            residues = list(range(i-1, j))\n            per_res = E_loop / len(residues)\n            for idx in residues:\n                loop_contrib[idx] += per_res\n    return loop_contrib\n\nsequences = seq_df['sequence'].tolist()\n\nwith Pool(processes=cpu_count()) as pool:\n    loop_results = list(tqdm(\n        pool.imap(compute_loop_dG_res, sequences),\n        total=len(sequences),\n        desc=\"Computing loop ΔG\"\n    ))\n\nseq_df['dG_loop_res'] = loop_results\n\nseq_df['total_dG']             = seq_df.dG_loop_res.map(np.sum).round(1)\n\nprint(seq_df[['target_id'] + ['sequence'] + ['total_dG']].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T10:07:09.009356Z","iopub.execute_input":"2025-05-04T10:07:09.010066Z","iopub.status.idle":"2025-05-04T10:11:31.983273Z","shell.execute_reply.started":"2025-05-04T10:07:09.010039Z","shell.execute_reply":"2025-05-04T10:11:31.982113Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Solvent Accessibility (Unpaired Probability) For Each Residue","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport RNA\nfrom multiprocessing import Pool, cpu_count\nfrom tqdm import tqdm\n\ndef predict_sa(seq: str) -> np.ndarray:\n    \"\"\"\n    Approximate per‐residue solvent accessibility as\n    the probability of being unpaired:\n      sa[i] = 1 − sum_j p_pair(i,j)\n    where p_pair is the ViennaRNA pairing‐probability matrix.\n    Returns an array of shape (L,)\n    \"\"\"\n    fc = RNA.fold_compound(seq)\n    fc.pf()\n    pmat_full = np.array(fc.bpp())\n    L = len(seq)\n    pprob = pmat_full[1 : L+1, 1 : L+1]\n    sa = 1.0 - pprob.sum(axis=1)\n    return sa\n\nsequences = seq_df['sequence'].tolist()\nn_workers = max(1, cpu_count() - 1)\n\nwith Pool(n_workers) as pool:\n    sa_arrays = list(\n        tqdm(pool.imap(predict_sa, sequences),\n             total=len(sequences),\n             desc=\"Computing SA\")\n    )\n\nseq_df['sa'] = sa_arrays\n\nprint(seq_df[['target_id', 'sequence', 'sa']].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T10:28:53.886596Z","iopub.execute_input":"2025-05-04T10:28:53.888036Z","iopub.status.idle":"2025-05-04T10:36:58.968273Z","shell.execute_reply.started":"2025-05-04T10:28:53.887995Z","shell.execute_reply":"2025-05-04T10:36:58.966530Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Flexibility Index","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport math\nfrom collections import Counter\nfrom Bio import AlignIO\nfrom multiprocessing import Pool, cpu_count\nfrom tqdm import tqdm\n\ndef compute_flexibility_index(msa_path: str) -> np.ndarray:\n    \"\"\"\n    Reads an MSA FASTA, computes Shannon entropy at each column,\n    and returns an (L,) array of entropies.\n    \"\"\"\n    aln = AlignIO.read(msa_path, \"fasta\")\n    seqs = [str(rec.seq) for rec in aln]\n    L = len(seqs[0])\n    arr = np.array([list(s) for s in seqs])\n    entropies = np.zeros(L, dtype=float)\n    for i in range(L):\n        col = arr[:, i]\n        counts = Counter(col)\n        total = sum(counts.values())\n        h = 0.0\n        for cnt in counts.values():\n            p = cnt / total\n            h -= p * math.log2(p)\n        entropies[i] = h\n    return entropies\n\nbase = \"/kaggle/input/stanford-rna-3d-folding/MSA\"\nmsa_paths = [f\"{base}/{tid}.MSA.fasta\" for tid in seq_df.target_id]\n\nn_procs = max(1, cpu_count() - 1)\nwith Pool(processes=n_procs) as pool:\n    flex_cols = list(tqdm(\n        pool.imap(compute_flexibility_index, msa_paths),\n        total=len(msa_paths),\n        desc=\"Computing flexibility\"\n    ))\n\nseq_df[\"flexibility_index\"] = flex_cols\n\nfor tid, seq, flex in zip(seq_df.target_id, seq_df.sequence, seq_df.flexibility_index):\n    assert len(flex) == len(seq), f\"Length mismatch for {tid}\"\n\nprint(seq_df[[\"target_id\",\"sequence\",\"flexibility_index\"]].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T10:42:07.913897Z","iopub.execute_input":"2025-05-04T10:42:07.914979Z","iopub.status.idle":"2025-05-04T10:44:15.246847Z","shell.execute_reply.started":"2025-05-04T10:42:07.914941Z","shell.execute_reply":"2025-05-04T10:44:15.245508Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Motif Match Matrices","metadata":{}},{"cell_type":"code","source":"!apt-get update -qq && apt-get install -qq -y infernal\n!wget -q ftp://ftp.ebi.ac.uk/pub/databases/Rfam/CURRENT/Rfam.cm.gz\n!gunzip Rfam.cm.gz\n!cmpress Rfam.cm\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T10:50:18.299232Z","iopub.execute_input":"2025-05-04T10:50:18.300175Z","iopub.status.idle":"2025-05-04T10:50:43.632972Z","shell.execute_reply.started":"2025-05-04T10:50:18.300135Z","shell.execute_reply":"2025-05-04T10:50:43.631588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import subprocess, tempfile\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\nimport multiprocessing as mp\n\n\nMOTIF_ACCESSIONS = {\n    \"GNRA_tetraloop\": \"RF00163\",\n    \"UNCG_tetraloop\": \"RF00162\",\n    \"CUUG_tetraloop\": \"RF00488\",\n    \"kink-turn\":       \"RF00173\",\n}\nM = len(MOTIF_ACCESSIONS)\n\ndef scan_rfam_motifs(seq: str,\n                     rfam_cm: str = \"Rfam.cm\",\n                     evalue: float = 0.01):\n    \"\"\"\n    Runs `cmscan` on a single-sequence FASTA and returns an (L×M) binary matrix.\n    Skips any lines that don't have enough columns\n    \"\"\"\n    L = len(seq)\n    mat = np.zeros((L, M), dtype=np.int8)\n\n    with tempfile.NamedTemporaryFile(\"w+\", suffix=\".fa\") as fa:\n        fa.write(f\">seq\\n{seq}\\n\")\n        fa.flush()\n        cmd = [\n            \"cmscan\", \"--noali\", \"--cut_ga\",\n            \"--tblout\", \"/dev/stdout\",\n            rfam_cm, fa.name\n        ]\n        p = subprocess.run(cmd, capture_output=True, text=True)\n    \n    for line in p.stdout.splitlines():\n        line = line.strip()\n        if not line or line.startswith(\"#\"):\n            continue\n        parts = line.split()\n        if len(parts) < 8:\n            continue\n        acc   = parts[1]                \n        try:\n            start = int(parts[6])       \n            stop  = int(parts[7])       \n        except ValueError:\n            continue\n        if acc in MOTIF_ACCESSIONS.values():\n            j = list(MOTIF_ACCESSIONS.values()).index(acc)\n            mat[start-1:stop, j] = 1\n\n    return mat\n\ndef _worker(record):\n    return scan_rfam_motifs(record[\"sequence\"])\n\nrecords = seq_df[[\"sequence\"]].to_dict(orient=\"records\")\n\nwith mp.Pool(mp.cpu_count()) as pool:\n    mats = list(tqdm(pool.imap(_worker, records),\n                     total=len(records),\n                     desc=\"Scanning Rfam motifs\"))\n\nseq_df[\"motif_match_matrix\"] = mats\n\nfor idx, row in seq_df.head(5).iterrows():\n    L = len(row.sequence)\n    print(row.target_id, row.motif_match_matrix.shape, \"expected\", (L, M))\n\nprint(seq_df[[\"target_id\",\"sequence\",\"motif_match_matrix\"]].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T11:01:34.253078Z","iopub.execute_input":"2025-05-04T11:01:34.254230Z","iopub.status.idle":"2025-05-04T11:36:16.498634Z","shell.execute_reply.started":"2025-05-04T11:01:34.254190Z","shell.execute_reply":"2025-05-04T11:36:16.497508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seq_df.to_csv(\"seq_df_1D.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T11:37:36.288689Z","iopub.execute_input":"2025-05-04T11:37:36.289174Z","iopub.status.idle":"2025-05-04T11:37:39.789906Z","shell.execute_reply.started":"2025-05-04T11:37:36.289139Z","shell.execute_reply":"2025-05-04T11:37:39.789133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}