{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":106809,"databundleVersionId":13056355}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\"\"\"\nCELL 1 — Imports & Environment\n-------------------------------\nWe import everything here so all downstream cells are clean.\nKey libraries:\n  - torch      : deep learning framework\n  - h5py       : read HDF5 neural recording files\n  - numpy      : array math\n  - tqdm       : progress bars\n  - pandas     : CSV submission\n\"\"\"\nimport os, re, math, string, glob\nfrom pathlib import Path\nfrom typing import List, Dict, Tuple, Optional, Any\nfrom dataclasses import dataclass\n\nimport numpy as np\nimport h5py\nimport pandas as pd\nfrom tqdm import tqdm\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.nn.utils.rnn import pad_sequence\n\n# ── Environment check ──────────────────────────────────────────────\nprint(f\"PyTorch : {torch.__version__}\")\nprint(f\"CUDA    : {torch.cuda.is_available()}\")\nif torch.cuda.is_available():\n    print(f\"GPU     : {torch.cuda.get_device_name(0)}\")\n    print(f\"VRAM    : {torch.cuda.get_device_properties(0).total_memory / 1e9:.1f} GB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T09:51:26.652371Z","iopub.execute_input":"2026-03-11T09:51:26.652628Z","iopub.status.idle":"2026-03-11T09:51:32.419931Z","shell.execute_reply.started":"2026-03-11T09:51:26.652599Z","shell.execute_reply":"2026-03-11T09:51:32.418910Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nCELL 2 — Global Configuration\n-------------------------------\nALL hyper-parameters live here. Change only this cell to run experiments.\n\nKey design decisions:\n  - d_model=512       : matches the 512 neural input channels, no mismatch\n  - conv_kernel=31    : Conformer works best with ~31 for neural data (~200Hz)\n  - ctc_weight=0.3    : 30% CTC + 70% CE — CTC guides encoder, CE refines decoder\n  - label_smoothing   : prevents overconfidence on noisy labels\n  - patience=8        : stop if val WER doesn't improve for 8 epochs\n\"\"\"\n@dataclass\nclass Config:\n    # ── Paths ──────────────────────────────────────────────────────\n    # Brain-to-Text '25 competition data path on Kaggle\n    data_dir        : str   = \"/kaggle/input/brain-to-text-25/t15_copyTask_neuralData/hdf5_data_final\"\n    checkpoint_path : str   = \"/kaggle/working/best_dsdnla_v2.pt\"\n    submission_path : str   = \"/kaggle/working/submission.csv\"\n\n    # ── Data ───────────────────────────────────────────────────────\n    n_channels      : int   = 512    # electrode channels (fixed by dataset)\n    max_neural_len  : int   = 1024   # clip trials longer than this (timesteps)\n    max_text_len    : int   = 100    # max decode length at inference\n\n    # ── Model ──────────────────────────────────────────────────────\n    d_model               : int   = 512   # hidden dimension throughout\n    n_encoder_layers      : int   = 8     # number of Conformer blocks\n    n_decoder_layers      : int   = 6     # number of Transformer decoder layers\n    n_heads               : int   = 8     # attention heads (d_model/n_heads = 64)\n    conv_kernel           : int   = 31    # Conformer depthwise conv kernel (must be odd)\n    ff_expansion          : int   = 4     # feedforward dim = d_model * ff_expansion\n    dropout               : float = 0.15  # dropout throughout\n    stochastic_depth_prob : float = 0.1   # max stochastic-depth drop-path rate\n    ctc_weight            : float = 0.3   # total = 0.3*CTC + 0.7*CE\n    label_smoothing       : float = 0.1   # smoothed CE loss\n\n    # ── Training ───────────────────────────────────────────────────\n    seed          : int   = 42\n    batch_size    : int   = 16\n    num_epochs    : int   = 60\n    learning_rate : float = 4e-4\n    weight_decay  : float = 1e-2\n    gradient_clip : float = 1.0\n    warmup_frac   : float = 0.05   # first 5% of steps = LR warmup\n    patience      : int   = 8      # early stopping\n    use_amp       : bool  = True   # automatic mixed precision (FP16 on GPU)\n\n    # ── Inference ──────────────────────────────────────────────────\n    ctc_decode : bool = True   # True=CTC greedy (fast), False=attention decoder\n\n    # ── Device (set at runtime below) ──────────────────────────────\n    device : str = \"cpu\"\n\n\nCFG = Config()\nCFG.device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\n# Reproducibility\ntorch.manual_seed(CFG.seed)\nnp.random.seed(CFG.seed)\nif torch.cuda.is_available():\n    torch.cuda.manual_seed_all(CFG.seed)\n\nprint(f\"Device : {CFG.device}\")\nprint(CFG)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T09:51:32.422262Z","iopub.execute_input":"2026-03-11T09:51:32.422844Z","iopub.status.idle":"2026-03-11T09:51:32.442172Z","shell.execute_reply.started":"2026-03-11T09:51:32.422819Z","shell.execute_reply":"2026-03-11T09:51:32.441297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nCELL 3 — Data Structure Explorer\n----------------------------------\nRun this once to understand what is inside the HDF5 files.\n\nWHAT IS INSIDE:\n  hdf5_data_final/\n    t15.2023.08.13/             ← one recording session per folder\n      data_train.hdf5\n      data_val.hdf5\n      data_test.hdf5\n        ├── trial_0000/\n        │     input_features    shape=(T, 512)  float32\n        │                       → T ≈ 150–400 timesteps (~200 Hz)\n        │                       → 512 channels = spike-band power features\n        │     attrs:\n        │       sentence_label  = \"Bring it closer.\"\n        │       n_time_steps    = 321\n        │       session         = \"t15.2023.08.13\"\n        └── trial_0001/ ...\n\nWHY HDF5?\n  HDF5 is a hierarchical file format like a filesystem inside a file.\n  h5py lets us read it lazily (without loading everything into RAM).\n  We keep files OPEN during training (self.files dict) to avoid repeated I/O.\n\"\"\"\ndef explore_data(data_dir: str) -> None:\n    \"\"\"Print structure and stats of the dataset.\"\"\"\n    data_path = Path(data_dir)\n    if not data_path.exists():\n        print(f\"[ERROR] data_dir does not exist: {data_dir}\")\n        print(\"  Expected: /kaggle/input/brain-to-text-25/...\")\n        return\n\n    sessions = sorted(data_path.glob(\"t15.*\"))\n    print(f\"Found {len(sessions)} session folder(s) under {data_dir}\")\n    for s in sessions[:5]:\n        files = sorted(s.glob(\"*.hdf5\"))\n        print(f\"  {s.name}/ → {[f.name for f in files]}\")\n    if len(sessions) > 5:\n        print(f\"  ... ({len(sessions) - 5} more sessions)\")\n\n    train_files = sorted(data_path.glob(\"t15.*/data_train.hdf5\"))\n    if not train_files:\n        print(\"[WARN] No data_train.hdf5 found — dataset may not be mounted.\")\n        return\n\n    print(f\"\\nInspecting first file: {train_files[0]}\")\n    with h5py.File(train_files[0], \"r\") as f:\n        keys = list(f.keys())\n        print(f\"  Trials (first 5): {keys[:5]}  ... ({len(keys)} total)\")\n\n        trial = f[keys[0]]\n        print(f\"  Datasets in trial: {list(trial.keys())}\")\n\n        if \"input_features\" in trial:\n            feat = trial[\"input_features\"][:]\n            print(f\"  input_features shape : {feat.shape}  dtype={feat.dtype}\")\n            print(f\"  input_features stats : mean={feat.mean():.4f}  std={feat.std():.4f}\")\n\n        print(f\"  Attributes: {dict(trial.attrs)}\")\n\n\nexplore_data(CFG.data_dir)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T09:51:32.443416Z","iopub.execute_input":"2026-03-11T09:51:32.443718Z","iopub.status.idle":"2026-03-11T09:51:32.466433Z","shell.execute_reply.started":"2026-03-11T09:51:32.443691Z","shell.execute_reply":"2026-03-11T09:51:32.465140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nCELL 4 — Character-level Tokenizer\n------------------------------------\nWHY CHARACTER-LEVEL?\n  - Small vocabulary (36 tokens) → fewer parameters, less overfitting\n  - No OOV (out-of-vocabulary) problem\n  - CTC works naturally at character level\n\nVOCABULARY (36 tokens):\n  ID  Token    Role\n  0   <blank>  CTC blank — used internally, NEVER appears in decoded output\n  1   <PAD>    Padding — positions the loss ignores\n  2   <BOS>    Begin of Sequence — first token fed to attention decoder\n  3   <EOS>    End of Sequence — decoder stops when this is predicted\n  4–29  a–z    26 lowercase letters\n  30    ' '    Space (word separator)\n  31–35 ' . , ! ?  Basic punctuation\n\nTWO ENCODING MODES:\n  encode()     → [BOS, chars..., EOS]  used for attention decoder (teacher forcing)\n  encode_ctc() → [chars...]            used for CTC loss (NO special tokens!)\n\nCTC GREEDY DECODE:\n  1. argmax at every time step  → raw token sequence\n  2. collapse consecutive duplicates (e.g., [a,a,b] → [a,b])\n  3. remove blank tokens (id=0)\n  4. convert IDs to characters\n\"\"\"\nclass CharTokenizer:\n    BLANK_ID    = 0\n    PAD_ID      = 1\n    BOS_ID      = 2\n    EOS_ID      = 3\n    SPECIAL_IDS = {0, 1, 2, 3}\n\n    def __init__(self):\n        self.chars: List[str] = [\"<blank>\", \"<PAD>\", \"<BOS>\", \"<EOS>\"]\n        self.chars += list(string.ascii_lowercase)       # ids 4..29\n        self.chars += [\" \", \"'\", \".\", \",\", \"!\", \"?\"]     # ids 30..35\n\n        self.char2id: Dict[str, int] = {c: i for i, c in enumerate(self.chars)}\n        self.id2char: Dict[int, str] = {i: c for i, c in enumerate(self.chars)}\n        self.vocab_size: int = len(self.chars)  # 36\n\n    def encode(self, text: str, add_bos_eos: bool = True) -> torch.Tensor:\n        \"\"\"\n        text → token-ID tensor.\n        \n        Example: \"hi\" → [2, 11, 12, 3]  (BOS, h, i, EOS)\n        \"\"\"\n        text = text.lower().strip()\n        ids = [self.BOS_ID] if add_bos_eos else []\n        for ch in text:\n            # unknown chars map to space\n            ids.append(self.char2id.get(ch, self.char2id[\" \"]))\n        if add_bos_eos:\n            ids.append(self.EOS_ID)\n        return torch.tensor(ids, dtype=torch.long)\n\n    def encode_ctc(self, text: str) -> torch.Tensor:\n        \"\"\"\n        Encode WITHOUT BOS/EOS — used as CTC target.\n        CTC loss requires plain character IDs with no special framing.\n        \n        Example: \"hi\" → [11, 12]  (h, i)\n        \"\"\"\n        return self.encode(text, add_bos_eos=False)\n\n    def decode(self, ids, remove_special: bool = True) -> str:\n        \"\"\"\n        token-IDs → string.\n        Stops at EOS. Skips special tokens (blank, PAD, BOS) if remove_special=True.\n        \"\"\"\n        if isinstance(ids, torch.Tensor):\n            ids = ids.cpu().tolist()\n        chars = []\n        for i in ids:\n            if i == self.EOS_ID:\n                break\n            if remove_special and i in self.SPECIAL_IDS:\n                continue\n            ch = self.id2char.get(i, \"\")\n            if remove_special and ch.startswith(\"<\"):\n                continue\n            chars.append(ch)\n        return \"\".join(chars)\n\n    def ctc_decode_greedy(self, log_probs: torch.Tensor) -> List[str]:\n        \"\"\"\n        Greedy CTC decode from log-softmax output.\n\n        Args:\n            log_probs: (B, T, V)  — output of F.log_softmax(ctc_head(z), dim=-1)\n        \n        Returns:\n            List[str] of length B  — one decoded sentence per batch item\n        \n        HOW CTC DECODE WORKS:\n          The encoder outputs a probability at every time step (every ~10ms of neural\n          data after subsampling). CTC allows the same character to \"repeat\" across\n          multiple frames, and uses blank tokens as separators. Greedy decode:\n            [h,h,blank,e,l,l,l,blank,l,o] → collapse → [h,blank,e,l,blank,l,o]\n                                           → remove blank → [h,e,l,l,o] = \"hello\"\n        \"\"\"\n        preds = log_probs.argmax(dim=-1)  # (B, T)\n        results = []\n        for seq in preds:\n            seq = seq.cpu().tolist()\n            # Step 1: collapse consecutive duplicates\n            collapsed, prev = [], -1\n            for t in seq:\n                if t != prev:\n                    collapsed.append(t)\n                prev = t\n            # Step 2: remove blank tokens\n            clean = [t for t in collapsed if t != self.BLANK_ID]\n            # Step 3: decode to string\n            results.append(self.decode(clean, remove_special=True))\n        return results\n\n\n# ── Tests ─────────────────────────────────────────────────────────\ntokenizer = CharTokenizer()\nprint(f\"Vocab size     : {tokenizer.vocab_size}\")\n\nsample = \"hello world\"\nenc    = tokenizer.encode(sample)\nprint(f\"encode         : {enc.tolist()}\")\nprint(f\"decode back    : '{tokenizer.decode(enc)}'\")\n\nctc_enc = tokenizer.encode_ctc(sample)\nprint(f\"encode_ctc     : {ctc_enc.tolist()}  (no BOS/EOS)\")\n\n# Round-trip test\nassert tokenizer.decode(enc) == sample, \"Round-trip failed!\"\n\n# CTC decode test: simulate [h,h,blank,e,l,l,l,o]\nimport torch\nfake_log = torch.zeros(1, 8, 36)   # batch=1, time=8, vocab=36\n# manually set argmax: h=11, h=11, blank=0, e=8, l=15, l=15, l=15, o=18\nfor t, tok in enumerate([11, 11, 0, 8, 15, 15, 15, 18]):\n    fake_log[0, t, tok] = 100.0\nctc_result = tokenizer.ctc_decode_greedy(F.log_softmax(fake_log, dim=-1))\nassert ctc_result[0] == \"helo\", f\"CTC decode test failed: {ctc_result}\"\nprint(f\"CTC decode test: 'helo' ✓\")\nprint(\"All tokenizer tests passed.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T09:51:32.467620Z","iopub.execute_input":"2026-03-11T09:51:32.467885Z","iopub.status.idle":"2026-03-11T09:51:32.587933Z","shell.execute_reply.started":"2026-03-11T09:51:32.467858Z","shell.execute_reply":"2026-03-11T09:51:32.586883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nCELL 5 — BrainToTextDataset + collate_fn\n------------------------------------------\nHOW THE DATASET WORKS:\n  1. On __init__: open all HDF5 files, collect (file_idx, trial_key) pairs\n     → keeps files open to avoid repeated open() overhead during training\n  2. On __getitem__: read neural features + sentence label from HDF5\n     → apply augmentation (train only)\n     → return dict with neural tensor + two target formats\n\nAUGMENTATION (train only, mimics SpecAugment for neural data):\n  1. Time masking      (40% prob): zero-out 1–2 windows of 5–20 frames\n                                    → forces encoder to not overfit to any 1 region\n  2. Channel dropout   (25% prob): randomly drop 15% of electrodes\n                                    → robust to broken/noisy electrodes\n  3. Gaussian noise    (30% prob): add σ=0.05 noise\n                                    → prevents memorization of exact amplitude patterns\n  4. Time sub-sampling (20% prob): resample by ±10%\n                                    → robust to minor speaking rate variation\n\nCOLLATE FUNCTION:\n  PyTorch's DataLoader calls collate_fn on each batch.\n  Since trials have different lengths, we need to:\n    - Pad all neural sequences to the max length in the batch\n    - Build a boolean \"padding mask\" (True = PAD frame, encoder ignores these)\n    - Pad token targets for the decoder\n  This is standard for seq2seq models.\n\"\"\"\nclass BrainToTextDataset(Dataset):\n    def __init__(\n        self,\n        hdf5_paths : List[str],\n        tokenizer  : CharTokenizer,\n        mode       : str  = \"train\",\n        augment    : bool = True,\n        max_neural : int  = 1024,\n    ):\n        self.tok     = tokenizer\n        self.mode    = mode\n        self.augment = augment and (mode == \"train\")\n        self.max_len = max_neural\n\n        self.trials: List[Tuple[int, str]] = []\n        self.files : Dict[int, h5py.File]  = {}\n\n        print(f\"Loading {mode} data...\")\n        for fi, path in enumerate(tqdm(hdf5_paths)):\n            f = h5py.File(path, \"r\")\n            self.files[fi] = f\n            for key in f.keys():\n                self.trials.append((fi, key))\n        print(f\"  → {len(self.trials)} trials loaded\")\n\n        if len(self.trials) == 0:\n            raise RuntimeError(\n                f\"No trials found in {len(hdf5_paths)} HDF5 files.\\n\"\n                \"  Check that the dataset path is correct and files contain 'input_features'.\"\n            )\n\n    def __len__(self) -> int:\n        return len(self.trials)\n\n    def __getitem__(self, idx: int) -> Dict[str, Any]:\n        fi, key = self.trials[idx]\n        trial   = self.files[fi][key]\n\n        # ── Neural features: (T, 512) ─────────────────────────────\n        neural = torch.tensor(trial[\"input_features\"][:], dtype=torch.float32)\n        if neural.size(0) > self.max_len:\n            neural = neural[: self.max_len]\n\n        # ── Text label ────────────────────────────────────────────\n        text = trial.attrs.get(\"sentence_label\", \"\") or \"\"\n        # h5py can return bytes on some systems\n        if isinstance(text, (bytes, np.bytes_)):\n            text = text.decode(\"utf-8\")\n        text = str(text).strip()\n\n        tokens  = self.tok.encode(text)      # [BOS, chars..., EOS] for decoder\n        ctc_tgt = self.tok.encode_ctc(text)  # [chars...] for CTC loss\n\n        # CTC requires non-empty targets\n        if ctc_tgt.numel() == 0:\n            ctc_tgt = torch.tensor([self.tok.char2id[\" \"]], dtype=torch.long)\n\n        # ── Augmentation ─────────────────────────────────────────\n        if self.augment:\n            neural = self._augment(neural)\n\n        return {\n            \"neural\"  : neural,    # (T, 512)\n            \"tokens\"  : tokens,    # (T_text,) — attention decoder target\n            \"ctc_tgt\" : ctc_tgt,   # (T_ctc,)  — CTC target\n            \"text\"    : text,\n        }\n\n    def _augment(self, x: torch.Tensor) -> torch.Tensor:\n        \"\"\"Apply stochastic data augmentation to one trial's neural features.\"\"\"\n        T, C = x.shape\n\n        # 1) Time masking — SpecAugment style\n        if torch.rand(1).item() < 0.4:\n            n_masks = torch.randint(1, 3, (1,)).item()\n            for _ in range(int(n_masks)):\n                mask_len = torch.randint(5, 20, (1,)).item()\n                start    = torch.randint(0, max(1, T - mask_len), (1,)).item()\n                x[start : start + mask_len] = 0.0\n\n        # 2) Channel (electrode) dropout\n        if torch.rand(1).item() < 0.25:\n            keep = torch.bernoulli(torch.full((C,), 0.85))\n            x    = x * keep.unsqueeze(0)\n\n        # 3) Gaussian noise\n        if torch.rand(1).item() < 0.3:\n            x = x + torch.randn_like(x) * 0.05\n\n        # 4) Time sub-sampling (mild speed perturbation)\n        if torch.rand(1).item() < 0.2 and T > 10:\n            fac   = 0.9 + 0.2 * torch.rand(1).item()  # 0.9–1.1\n            new_T = max(10, int(T * fac))\n            idx   = torch.linspace(0, T - 1, new_T).long().clamp(0, T - 1)\n            x     = x[idx]\n\n        return x\n\n\ndef collate_fn(batch: List[Dict[str, Any]]) -> Dict[str, Any]:\n    \"\"\"\n    Collate variable-length trials into a padded batch.\n\n    KEY CONCEPT — Padding mask:\n      Transformer attention must ignore padded positions.\n      We pass `key_padding_mask` (dtype=bool, True=ignore) to every attention layer.\n      Without this, the model would attend to zero-padded frames and produce garbage.\n    \"\"\"\n    neurals  = [b[\"neural\"]   for b in batch]\n    tokens_l = [b[\"tokens\"]   for b in batch]\n    ctc_l    = [b[\"ctc_tgt\"]  for b in batch]\n    texts    = [b[\"text\"]     for b in batch]\n\n    neural_lens = torch.tensor([n.size(0) for n in neurals], dtype=torch.long)\n    ctc_lens    = torch.tensor([c.size(0) for c in ctc_l],   dtype=torch.long)\n\n    B     = len(neurals)\n    T_max = int(neural_lens.max().item())\n\n    # Pad neural to (B, T_max, 512)\n    neural_pad = pad_sequence(neurals, batch_first=True, padding_value=0.0)\n\n    # Build padding mask: True = PAD (encoder must ignore)\n    neural_mask = torch.ones(B, T_max, dtype=torch.bool)   # start: all PAD\n    for i in range(B):\n        neural_mask[i, : int(neural_lens[i].item())] = False  # valid → False\n\n    tokens_pad = pad_sequence(tokens_l, batch_first=True, padding_value=CharTokenizer.PAD_ID)\n    ctc_pad    = pad_sequence(ctc_l,    batch_first=True, padding_value=CharTokenizer.PAD_ID)\n\n    return {\n        \"neural\"      : neural_pad,    # (B, T_max, 512)\n        \"neural_mask\" : neural_mask,   # (B, T_max)  True=PAD\n        \"neural_lens\" : neural_lens,   # (B,)  original lengths\n        \"tokens\"      : tokens_pad,    # (B, T_text)\n        \"ctc_tgt\"     : ctc_pad,       # (B, T_ctc)\n        \"ctc_lens\"    : ctc_lens,      # (B,)  CTC target lengths\n        \"texts\"       : texts,\n    }\n\n\nprint(\"Dataset helpers defined.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T09:51:32.589352Z","iopub.execute_input":"2026-03-11T09:51:32.589674Z","iopub.status.idle":"2026-03-11T09:51:32.612527Z","shell.execute_reply.started":"2026-03-11T09:51:32.589649Z","shell.execute_reply":"2026-03-11T09:51:32.611432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nCELL 6 — Conformer Encoder\n----------------------------\nWHY CONFORMER OVER PLAIN TRANSFORMER?\n  Neural spike-band power has TWO types of structure:\n    LOCAL:   Short bursts of activity (10–50ms) corresponding to phoneme boundaries\n    GLOBAL:  Sentence-level context (which word comes next?)\n\n  Plain Transformer: good at global (attention), weak at local (no inductive bias)\n  Plain CNN:         good at local (kernel captures nearby frames), weak at global\n  CONFORMER:         combines both — adds a depthwise conv module between attention layers\n\nCONFORMER BLOCK STRUCTURE:\n  x\n  │  ┌─────────────────────────┐\n  ├─►│  FF₁ × 0.5  (half-step) │  pre-LN feedforward\n  │  └─────────────────────────┘\n  │  ┌─────────────────────────┐\n  ├─►│  MHSA (multi-head attn) │  global context\n  │  └──────���──────────────────┘\n  │  ┌─────────────────────────┐\n  ├─►│  ConvModule              │  local temporal patterns\n  │  └─────────────────────────┘\n  │  ┌─────────────────────────┐\n  └─►│  FF₂ × 0.5  (half-step) │  pre-LN feedforward\n     └─────────────────────────┘\n     LayerNorm → output\n\nCONV MODULE DETAIL:\n  LayerNorm → Pointwise(×2) → GLU gate → Depthwise Conv (kernel=31) → BN → Swish → Pointwise\n\nSTOCHASTIC DEPTH:\n  During training, randomly skip entire residual branches with probability p.\n  Linearly increases from p=0 (first layer) to p=0.1 (last layer).\n  Acts as a form of ensemble regularization — the model learns to be robust\n  to any subset of layers being dropped.\n\nSUBSAMPLING:\n  A stride-2 Conv1d at the input reduces sequence length by 2×.\n  This halves the compute cost of all subsequent attention layers.\n  T=400 timesteps → L=200 encoder frames.\n  We also downsample the padding mask to match (take every 2nd position).\n\"\"\"\nclass Swish(nn.Module):\n    \"\"\"\n    Swish activation: x * sigmoid(x)\n    \n    Why Swish?\n      - Smooth (unlike ReLU)\n      - Slightly outperforms GELU/ReLU in Conformer architectures empirically\n      - Self-gated: the sigmoid acts as a smooth gate on x\n    \"\"\"\n    def forward(self, x: torch.Tensor) -> torch.Tensor:\n        return x * torch.sigmoid(x)\n\n\nclass PositionalEncoding(nn.Module):\n    \"\"\"\n    Sinusoidal positional encoding (Vaswani et al., 2017).\n\n    WHY POSITIONAL ENCODING?\n      Transformers are permutation-invariant by default — they treat position 1\n      and position 100 identically. Positional encoding injects position info\n      as a sin/cos signal so the model knows where in time each frame is.\n\n    FORMULA:\n      PE[pos, 2i]   = sin(pos / 10000^(2i/d_model))\n      PE[pos, 2i+1] = cos(pos / 10000^(2i/d_model))\n    \"\"\"\n    def __init__(self, d_model: int, dropout: float = 0.1, max_len: int = 8000):\n        super().__init__()\n        self.dropout = nn.Dropout(dropout)\n        pe  = torch.zeros(max_len, d_model)\n        pos = torch.arange(0, max_len, dtype=torch.float32).unsqueeze(1)\n        div = torch.exp(\n            torch.arange(0, d_model, 2, dtype=torch.float32)\n            * (-math.log(10000.0) / d_model)\n        )\n        pe[:, 0::2] = torch.sin(pos * div)\n        pe[:, 1::2] = torch.cos(pos * div)\n        self.register_buffer(\"pe\", pe.unsqueeze(0))  # (1, max_len, d_model)\n\n    def forward(self, x: torch.Tensor) -> torch.Tensor:\n        # x: (B, T, D)  →  add positional info  →  (B, T, D)\n        return self.dropout(x + self.pe[:, : x.size(1)])\n\n\nclass ConformerConvModule(nn.Module):\n    \"\"\"\n    Conformer Convolution Module.\n\n    WHY GLU (Gated Linear Unit)?\n      The pointwise conv doubles the channels, then GLU splits and gates:\n        output = first_half * sigmoid(second_half)\n      This gives the model a learnable gate to control how much each\n      channel's local temporal pattern passes through.\n\n    WHY DEPTHWISE CONV (groups=d_model)?\n      Each of the d_model channels gets its OWN conv filter.\n      This captures per-channel temporal patterns while staying efficient\n      (d_model params instead of d_model² params for a full conv).\n\n    WHY BATCHNORM AFTER DEPTHWISE?\n      Normalizes each channel independently. Works well because each channel\n      has its own filter and activation scale.\n    \"\"\"\n    def __init__(self, d_model: int, kernel_size: int = 31, dropout: float = 0.1):\n        super().__init__()\n        assert (kernel_size - 1) % 2 == 0, \"kernel_size must be odd for symmetric padding\"\n        self.ln   = nn.LayerNorm(d_model)\n        self.pw1  = nn.Conv1d(d_model, 2 * d_model, 1)    # pointwise: d → 2d\n        self.glu  = nn.GLU(dim=1)                          # GLU gate: 2d → d\n        self.dw   = nn.Conv1d(                             # depthwise: per-channel\n            d_model, d_model, kernel_size,\n            padding=(kernel_size - 1) // 2,\n            groups=d_model,\n        )\n        self.bn   = nn.BatchNorm1d(d_model)\n        self.act  = Swish()\n        self.pw2  = nn.Conv1d(d_model, d_model, 1)        # pointwise: d → d\n        self.drop = nn.Dropout(dropout)\n\n    def forward(self, x: torch.Tensor) -> torch.Tensor:\n        residual = x\n        x = self.ln(x).transpose(1, 2)          # (B, D, T) for Conv1d\n        x = self.glu(self.pw1(x))               # (B, D, T)\n        x = self.act(self.bn(self.dw(x)))       # (B, D, T)\n        x = self.drop(self.pw2(x)).transpose(1, 2)  # back to (B, T, D)\n        return x + residual  # residual connection\n\n\nclass ConformerBlock(nn.Module):\n    \"\"\"\n    One Conformer block.\n\n    BUG FIX (vs earlier versions):\n      MHSA LayerNorm must be computed ONCE and reused for Q, K, V.\n      If called 3× separately, each call has its own dropout mask → incorrect.\n    \"\"\"\n    def __init__(\n        self,\n        d_model      : int,\n        n_heads      : int,\n        ff_expansion : int   = 4,\n        conv_kernel  : int   = 31,\n        dropout      : float = 0.1,\n        drop_path    : float = 0.0,  # stochastic depth probability\n    ):\n        super().__init__()\n        d_ff = d_model * ff_expansion\n\n        # FF1: pre-LN feedforward (half-step weight 0.5)\n        self.ff1_ln = nn.LayerNorm(d_model)\n        self.ff1    = nn.Sequential(\n            nn.Linear(d_model, d_ff), Swish(), nn.Dropout(dropout),\n            nn.Linear(d_ff, d_model), nn.Dropout(dropout),\n        )\n\n        # MHSA: multi-head self-attention\n        self.mhsa_ln   = nn.LayerNorm(d_model)\n        self.mhsa      = nn.MultiheadAttention(\n            d_model, n_heads, dropout=dropout, batch_first=True\n        )\n        self.mhsa_drop = nn.Dropout(dropout)\n\n        # Conv module\n        self.conv = ConformerConvModule(d_model, conv_kernel, dropout)\n\n        # FF2: pre-LN feedforward (half-step weight 0.5)\n        self.ff2_ln = nn.LayerNorm(d_model)\n        self.ff2    = nn.Sequential(\n            nn.Linear(d_model, d_ff), Swish(), nn.Dropout(dropout),\n            nn.Linear(d_ff, d_model), nn.Dropout(dropout),\n        )\n\n        self.final_ln  = nn.LayerNorm(d_model)\n        self.drop_path = drop_path\n\n    def _stochastic_drop(self, residual: torch.Tensor, shortcut: torch.Tensor) -> torch.Tensor:\n        \"\"\"\n        Stochastic Depth: randomly zero out a residual branch during training.\n        At inference (eval mode), always use full residual.\n        \n        Why? Acts like an ensemble: the model learns robust representations\n        even when some layers are \"absent\". Reduces overfitting on small datasets.\n        \"\"\"\n        if self.training and self.drop_path > 0.0:\n            survive = (torch.rand(1, device=residual.device) > self.drop_path).float()\n            return shortcut + residual * survive\n        return shortcut + residual\n\n    def forward(\n        self,\n        x                : torch.Tensor,\n        key_padding_mask : Optional[torch.Tensor] = None,\n    ) -> torch.Tensor:\n        # FF1 (half-step: multiply output by 0.5 before adding residual)\n        x = self._stochastic_drop(0.5 * self.ff1(self.ff1_ln(x)), x)\n\n        # MHSA — compute normed ONCE (BUG FIX: reuse for Q, K, V)\n        normed = self.mhsa_ln(x)\n        attn_out, _ = self.mhsa(\n            normed, normed, normed,\n            key_padding_mask=key_padding_mask,\n            need_weights=False,  # don't waste memory on attention weights\n        )\n        x = self._stochastic_drop(self.mhsa_drop(attn_out), x)\n\n        # Conv module\n        x = self.conv(x)\n\n        # FF2 (half-step)\n        x = self._stochastic_drop(0.5 * self.ff2(self.ff2_ln(x)), x)\n\n        return self.final_ln(x)\n\n\nclass ConformerEncoder(nn.Module):\n    \"\"\"\n    Full Conformer encoder: Conv subsampling → Positional Encoding → N blocks.\n\n    Returns: (z, enc_mask)\n      z        : (B, T//2, D)  — encoded neural representations\n      enc_mask : (B, T//2)     — True=PAD, passed to decoder cross-attention\n    \"\"\"\n    def __init__(\n        self,\n        n_channels   : int   = 512,\n        d_model      : int   = 512,\n        n_layers     : int   = 8,\n        n_heads      : int   = 8,\n        ff_expansion : int   = 4,\n        conv_kernel  : int   = 31,\n        dropout      : float = 0.1,\n        drop_path    : float = 0.1,\n    ):\n        super().__init__()\n\n        # Conv subsampling: stride=2 halves sequence length\n        # Input: (B, n_channels, T) → Output: (B, d_model, T//2)\n        self.subsampler = nn.Sequential(\n            nn.Conv1d(n_channels, d_model, kernel_size=5, stride=2, padding=2),\n            nn.BatchNorm1d(d_model),\n            nn.GELU(),\n            nn.Dropout(dropout),\n        )\n        self.pos_enc = PositionalEncoding(d_model, dropout, max_len=8000)\n\n        # Linearly ramp stochastic depth from 0 → drop_path across layers\n        # (first layers are more critical → lower drop rate)\n        dpr = [drop_path * i / max(n_layers - 1, 1) for i in range(n_layers)]\n        self.blocks = nn.ModuleList([\n            ConformerBlock(d_model, n_heads, ff_expansion, conv_kernel, dropout, dp)\n            for dp in dpr\n        ])\n\n    def forward(\n        self,\n        x    : torch.Tensor,\n        mask : Optional[torch.Tensor] = None,\n    ) -> Tuple[torch.Tensor, Optional[torch.Tensor]]:\n        # x: (B, T, C)  mask: (B, T) True=PAD\n        x = x.transpose(1, 2)   # (B, C, T)\n        x = self.subsampler(x)  # (B, D, T//2)\n        x = x.transpose(1, 2)   # (B, T//2, D)\n        L = x.size(1)\n\n        # Downsample mask to match subsampled length\n        if mask is not None:\n            mask_ds = mask[:, ::2]   # stride-2 to match Conv stride\n            # Handle edge cases in length\n            if mask_ds.size(1) > L:\n                mask_ds = mask_ds[:, :L]\n            elif mask_ds.size(1) < L:\n                pad_cols = torch.ones(\n                    mask_ds.size(0), L - mask_ds.size(1),\n                    dtype=torch.bool, device=mask_ds.device,\n                )\n                mask_ds = torch.cat([mask_ds, pad_cols], dim=1)\n        else:\n            mask_ds = None\n\n        x = self.pos_enc(x)\n        for block in self.blocks:\n            x = block(x, key_padding_mask=mask_ds)\n\n        return x, mask_ds  # (B, L, D), (B, L)\n\n\nprint(\"ConformerEncoder defined.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T09:51:32.614648Z","iopub.execute_input":"2026-03-11T09:51:32.614902Z","iopub.status.idle":"2026-03-11T09:51:32.645018Z","shell.execute_reply.started":"2026-03-11T09:51:32.614878Z","shell.execute_reply":"2026-03-11T09:51:32.644095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nCELL 7 — Attention Decoder\n----------------------------\nROLE:\n  The decoder takes the encoder output z and generates text autoregressively.\n  At each step it:\n    1. Embeds the previously generated tokens\n    2. Self-attends (masked — can't look ahead)\n    3. Cross-attends to encoder output z (this is where neural info comes from)\n    4. Projects to vocabulary → probability over next character\n\nTEACHER FORCING (training):\n  Instead of feeding the model's own predictions (which may be wrong early on),\n  we feed the GROUND TRUTH prefix shifted by 1:\n    Input:  [BOS, h, e, l, l, o, ' ', w, o, r, l, d]\n    Target: [h,   e, l, l, o, ' ', w, o, r, l, d, EOS]\n  This is faster and more stable than using predicted tokens during training.\n\nCAUSAL MASK:\n  Self-attention in the decoder must be causal: position t can only attend to\n  positions 0..t, not t+1..T. We implement this with an upper-triangular bool mask:\n    [[F, T, T, T],   ← position 0 sees only itself\n     [F, F, T, T],   ← position 1 sees 0,1\n     [F, F, F, T],   ← position 2 sees 0,1,2\n     [F, F, F, F]]   ← position 3 sees all\n  T = attend, F = blocked.\n\n  BUG FIX: Use torch.bool masks throughout (not float -inf masks).\n  PyTorch ≥ 2.0 warns about mixing float causal + bool padding masks.\n\"\"\"\ndef make_causal_mask(T: int, device: torch.device) -> torch.Tensor:\n    \"\"\"\n    Upper-triangular boolean causal mask.\n    True = 'do NOT attend to this position' (masked/future tokens).\n    Shape: (T, T)\n    \"\"\"\n    return torch.triu(torch.ones(T, T, dtype=torch.bool, device=device), diagonal=1)\n\n\nclass AttentionDecoder(nn.Module):\n    \"\"\"\n    Transformer decoder that cross-attends to Conformer encoder output.\n\n    Uses pre-LN (norm_first=True) for more stable training gradients.\n    Pre-LN: normalize BEFORE the sublayer (more common in modern models).\n    Post-LN: normalize AFTER (original Transformer — less stable for deep nets).\n    \"\"\"\n    def __init__(\n        self,\n        vocab_size   : int,\n        d_model      : int   = 512,\n        n_layers     : int   = 6,\n        n_heads      : int   = 8,\n        ff_expansion : int   = 4,\n        dropout      : float = 0.1,\n    ):\n        super().__init__()\n        self.embed   = nn.Embedding(vocab_size, d_model, padding_idx=CharTokenizer.PAD_ID)\n        self.pos_enc = PositionalEncoding(d_model, dropout, max_len=4000)\n\n        dec_layer = nn.TransformerDecoderLayer(\n            d_model         = d_model,\n            nhead           = n_heads,\n            dim_feedforward = d_model * ff_expansion,\n            dropout         = dropout,\n            activation      = \"gelu\",\n            batch_first     = True,\n            norm_first      = True,  # pre-LN for stable training\n        )\n        self.decoder = nn.TransformerDecoder(dec_layer, num_layers=n_layers)\n        self.proj    = nn.Linear(d_model, vocab_size)\n\n    def forward(\n        self,\n        tgt         : torch.Tensor,                    # (B, T_tgt) token ids\n        memory      : torch.Tensor,                    # (B, T_enc, D) encoder output\n        memory_mask : Optional[torch.Tensor] = None,   # (B, T_enc) True=PAD\n    ) -> torch.Tensor:                                 # → (B, T_tgt, V)\n        T       = tgt.size(1)\n        causal  = make_causal_mask(T, tgt.device)     # (T, T) bool — future masking\n        tgt_pad = (tgt == CharTokenizer.PAD_ID)       # (B, T_tgt) bool — pad masking\n\n        emb = self.pos_enc(self.embed(tgt))           # (B, T, D)\n        out = self.decoder(\n            emb, memory,\n            tgt_mask                = causal,         # blocks future tokens in self-attn\n            tgt_key_padding_mask    = tgt_pad,        # blocks PAD in self-attn\n            memory_key_padding_mask = memory_mask,    # blocks PAD in cross-attn\n        )\n        return self.proj(out)                         # (B, T, V) logits\n\n    @torch.no_grad()\n    def generate(\n        self,\n        memory      : torch.Tensor,\n        memory_mask : Optional[torch.Tensor] = None,\n        max_len     : int = 120,\n        bos_id      : int = CharTokenizer.BOS_ID,\n        eos_id      : int = CharTokenizer.EOS_ID,\n    ) -> torch.Tensor:\n        \"\"\"\n        Greedy autoregressive decoding.\n        \n        At each step:\n          1. Feed current token sequence to decoder\n          2. Take argmax of last position's logits → next token\n          3. Append and repeat until EOS or max_len\n        \n        Returns: (B, ≤ max_len+1) token IDs including BOS\n        \"\"\"\n        B      = memory.size(0)\n        device = memory.device\n        tokens = torch.full((B, 1), bos_id, dtype=torch.long, device=device)\n        done   = torch.zeros(B, dtype=torch.bool, device=device)\n\n        for _ in range(max_len):\n            logits   = self.forward(tokens, memory, memory_mask)         # (B, T, V)\n            next_tok = logits[:, -1].argmax(dim=-1, keepdim=True)        # (B, 1)\n            tokens   = torch.cat([tokens, next_tok], dim=1)\n            done     = done | (next_tok.squeeze(-1) == eos_id)\n            if done.all():\n                break\n\n        return tokens\n\n\nprint(\"AttentionDecoder defined.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T09:51:32.646033Z","iopub.execute_input":"2026-03-11T09:51:32.646650Z","iopub.status.idle":"2026-03-11T09:51:32.682920Z","shell.execute_reply.started":"2026-03-11T09:51:32.646317Z","shell.execute_reply":"2026-03-11T09:51:32.681834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nCELL 8 — DSD-NLA v2 Full Model\n---------------------------------\nPUTS IT ALL TOGETHER:\n  1. ConformerEncoder:  (B, T, 512) → z (B, L, 512)  where L = T//2\n  2. CTC Head:          z → log_probs (B, L, V)\n  3. AttentionDecoder:  z + token prefix → logits (B, T_text, V)\n\nTRAINING (both losses):\n  model(neural, neural_mask, target_tokens) → {ctc_log_probs, dec_logits, ...}\n  Loss = 0.3 * CTC(ctc_log_probs, text) + 0.7 * CE(dec_logits, text)\n\nINFERENCE (CTC path, faster):\n  model.inference(neural) → CTC log-probs\n  tokenizer.ctc_decode_greedy(log_probs) → predicted text\n\nWHY TWO LOSSES?\n  - CTC alone: encoder learns good representations, but decoder never trained\n  - CE alone:  decoder gets good, but no alignment signal for encoder\n  - HYBRID: CTC regularizes encoder (forces it to be predictive at every frame),\n            CE trains decoder to produce fluent sequences using encoder context\n  \n  At inference we prefer CTC greedy (faster, often similar WER) but can optionally\n  use the attention decoder for potentially better quality.\n\"\"\"\nclass DSDNLA_v2(nn.Module):\n    def __init__(self, cfg: Config, vocab_size: int):\n        super().__init__()\n        self.cfg = cfg\n\n        # ── Conformer Encoder ────────────────────────────────────────\n        self.encoder = ConformerEncoder(\n            n_channels   = cfg.n_channels,\n            d_model      = cfg.d_model,\n            n_layers     = cfg.n_encoder_layers,\n            n_heads      = cfg.n_heads,\n            ff_expansion = cfg.ff_expansion,\n            conv_kernel  = cfg.conv_kernel,\n            dropout      = cfg.dropout,\n            drop_path    = cfg.stochastic_depth_prob,\n        )\n\n        # ── CTC Head ─────────────────────────────────────────────────\n        # Simple: LayerNorm → Linear(d_model, vocab_size)\n        # We apply log_softmax AFTER this in forward() to get log-probabilities\n        # (required by nn.CTCLoss)\n        self.ctc_head = nn.Sequential(\n            nn.LayerNorm(cfg.d_model),\n            nn.Linear(cfg.d_model, vocab_size),\n        )\n\n        # ── Attention Decoder ────────────────────────────────────────\n        self.decoder = AttentionDecoder(\n            vocab_size   = vocab_size,\n            d_model      = cfg.d_model,\n            n_layers     = cfg.n_decoder_layers,\n            n_heads      = cfg.n_heads,\n            ff_expansion = cfg.ff_expansion,\n            dropout      = cfg.dropout,\n        )\n\n    def forward(\n        self,\n        neural        : torch.Tensor,                    # (B, T, 512)\n        neural_mask   : Optional[torch.Tensor] = None,   # (B, T) True=PAD\n        target_tokens : Optional[torch.Tensor] = None,   # (B, T_text) for training\n    ) -> Dict[str, torch.Tensor]:\n        \"\"\"\n        Training forward pass.\n        \n        Returns dict with:\n          ctc_log_probs : (B, L, V)    log-softmax over vocab at each encoder frame\n          enc_mask      : (B, L)       downsampled padding mask\n          z             : (B, L, D)    raw encoder features\n          dec_logits    : (B, T_text, V) or None\n        \"\"\"\n        # Step 1: encode neural → latent representation\n        z, enc_mask = self.encoder(neural, neural_mask)           # (B, L, D)\n\n        # Step 2: CTC log-probabilities\n        # F.log_softmax ensures log-probs (required by CTCLoss)\n        ctc_log = F.log_softmax(self.ctc_head(z), dim=-1)        # (B, L, V)\n\n        # Step 3: attention decoder (only during training, when target_tokens provided)\n        dec_logits = None\n        if target_tokens is not None:\n            dec_logits = self.decoder(target_tokens, z, enc_mask)  # (B, T_text, V)\n\n        return {\n            \"ctc_log_probs\" : ctc_log,\n            \"enc_mask\"      : enc_mask,\n            \"z\"             : z,\n            \"dec_logits\"    : dec_logits,\n        }\n\n    @torch.no_grad()\n    def inference(\n        self,\n        neural      : torch.Tensor,\n        neural_mask : Optional[torch.Tensor] = None,\n        use_ctc     : bool = True,\n        max_len     : int  = 120,\n    ) -> Dict[str, Any]:\n        \"\"\"\n        Inference: neural features → text predictions.\n        \n        use_ctc=True:  CTC greedy (fast, ~same WER)\n        use_ctc=False: attention decoder greedy (slower, potentially better)\n        \"\"\"\n        z, enc_mask = self.encoder(neural, neural_mask)\n        ctc_log     = F.log_softmax(self.ctc_head(z), dim=-1)\n\n        tokens = None\n        if not use_ctc:\n            tokens = self.decoder.generate(z, enc_mask, max_len=max_len)\n\n        return {\"ctc_log_probs\": ctc_log, \"tokens\": tokens, \"enc_mask\": enc_mask}\n\n\n# ── Smoke test ────────────────────────────────────────────────────\nmodel    = DSDNLA_v2(CFG, tokenizer.vocab_size).to(CFG.device)\nn_params = sum(p.numel() for p in model.parameters())\nprint(f\"Model parameters : {n_params:,}\")\n\n# Trainable parameters breakdown\nfor name, module in [(\"encoder\", model.encoder),\n                     (\"ctc_head\", model.ctc_head),\n                     (\"decoder\", model.decoder)]:\n    n = sum(p.numel() for p in module.parameters())\n    print(f\"  {name:12s}: {n:>12,}\")\n\n# Quick forward pass test\n_B, _T = 2, 200\n_x   = torch.randn(_B, _T, 512, device=CFG.device)\n_tgt = torch.randint(4, 30, (_B, 20), device=CFG.device)\n_out = model(_x, target_tokens=_tgt)\nassert _out[\"ctc_log_probs\"].shape == (_B, _T // 2, tokenizer.vocab_size)\nassert _out[\"dec_logits\"].shape    == (_B, 20, tokenizer.vocab_size)\nprint(f\"\\nCTC log-probs shape : {_out['ctc_log_probs'].shape}  ✓\")\nprint(f\"Dec logits shape    : {_out['dec_logits'].shape}  ✓\")\nprint(\"Smoke test passed.\")\ndel _x, _tgt, _out","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T09:51:32.684375Z","iopub.execute_input":"2026-03-11T09:51:32.685348Z","iopub.status.idle":"2026-03-11T09:51:34.721035Z","shell.execute_reply.started":"2026-03-11T09:51:32.685302Z","shell.execute_reply":"2026-03-11T09:51:34.719854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}