{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import math\nimport copy\nimport random\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nimport timm\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nfrom PIL import Image\nfrom torchvision import transforms\nimport matplotlib.patches as patches\n\ntorch.manual_seed(42)\nnp.random.seed(42)\nrandom.seed(42)\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nprint(\"Torch :\", torch.__version__)\nprint(\"TIMM  :\", timm.__version__)\nprint(\"Device:\", device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:45:36.446756Z","iopub.execute_input":"2026-07-12T18:45:36.446992Z","iopub.status.idle":"2026-07-12T18:45:55.536776Z","shell.execute_reply.started":"2026-07-12T18:45:36.44697Z","shell.execute_reply":"2026-07-12T18:45:55.535712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load a pretrained DeiT-Small model.\n# We'll use its attention maps to implement the ATS scoring mechanism.\n\nmodel = timm.create_model(\n    \"deit_small_patch16_224\",\n    pretrained=True\n).to(device)\n\nmodel.eval()\n\nfor p in model.parameters():\n    p.requires_grad = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:46:32.171963Z","iopub.execute_input":"2026-07-12T18:46:32.172909Z","iopub.status.idle":"2026-07-12T18:46:36.19616Z","shell.execute_reply.started":"2026-07-12T18:46:32.172835Z","shell.execute_reply":"2026-07-12T18:46:36.195138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transform = transforms.Compose([\n    transforms.Resize((224,224)),\n    transforms.ToTensor(),\n    transforms.Normalize(\n        mean=[0.485,0.456,0.406],\n        std=[0.229,0.224,0.225]\n    )\n])\n\nimg_path=\"/kaggle/input/competitions/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/val/ILSVRC2012_val_00000001.JPEG\"\n\nimg=Image.open(img_path).convert(\"RGB\")\n\nimage=transform(img).unsqueeze(0).to(device)\n\nplt.figure(figsize=(6,6))\nplt.imshow(img)\nplt.axis(\"off\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:46:37.672237Z","iopub.execute_input":"2026-07-12T18:46:37.673059Z","iopub.status.idle":"2026-07-12T18:46:37.972438Z","shell.execute_reply.started":"2026-07-12T18:46:37.673028Z","shell.execute_reply":"2026-07-12T18:46:37.971461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert the image into patch tokens.\n\ntokens=model.patch_embed(image)\n\nprint(tokens.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:46:43.565216Z","iopub.execute_input":"2026-07-12T18:46:43.566047Z","iopub.status.idle":"2026-07-12T18:46:44.328504Z","shell.execute_reply.started":"2026-07-12T18:46:43.566014Z","shell.execute_reply":"2026-07-12T18:46:44.327391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cls=model.cls_token.expand(tokens.shape[0],-1,-1)\n\ntokens=torch.cat((cls,tokens),dim=1)\n\ntokens=tokens+model.pos_embed\n\nprint(tokens.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:46:49.935395Z","iopub.execute_input":"2026-07-12T18:46:49.936054Z","iopub.status.idle":"2026-07-12T18:46:49.989998Z","shell.execute_reply.started":"2026-07-12T18:46:49.936023Z","shell.execute_reply":"2026-07-12T18:46:49.989088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#first transformer block.\n\nblock=copy.deepcopy(model.blocks[0])\n\nblock.eval()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:47:10.612584Z","iopub.execute_input":"2026-07-12T18:47:10.613381Z","iopub.status.idle":"2026-07-12T18:47:10.825155Z","shell.execute_reply.started":"2026-07-12T18:47:10.613348Z","shell.execute_reply":"2026-07-12T18:47:10.824381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x=block.norm1(tokens)\n\nB,N,C=x.shape\n\nnum_heads=block.attn.num_heads\nhead_dim=block.attn.head_dim\n\nqkv=block.attn.qkv(x)\n\nqkv=qkv.reshape(\n    B,\n    N,\n    3,\n    num_heads,\n    head_dim\n)\n\nqkv=qkv.permute(2,0,3,1,4)\n\nq,k,v=qkv.unbind(0)\n\nq=block.attn.q_norm(q)\nk=block.attn.k_norm(k)\n\nq=q*block.attn.scale\n\nattn=q@k.transpose(-2,-1)\n\nattn=attn.softmax(dim=-1)\n\nprint(attn.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:47:16.509223Z","iopub.execute_input":"2026-07-12T18:47:16.509913Z","iopub.status.idle":"2026-07-12T18:47:16.847978Z","shell.execute_reply.started":"2026-07-12T18:47:16.509835Z","shell.execute_reply":"2026-07-12T18:47:16.847149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------------------------------\n# Compute token significance using the CLS attention and value vectors.\n# ------------------------------------------------------------------\n\n# CLS attends to all image patches (ignore CLS -> CLS)\ncls_attn = attn[:, :, 0, 1:]                     # (B, H, 196)\n\n# Magnitude of every value vector\nv_norm = torch.norm(\n    v[:, :, 1:, :],\n    dim=-1\n)                                                # (B, H, 196)\n\n# Importance score\nscore = cls_attn * v_norm\n\n# Aggregate across attention heads\nscore = score.sum(dim=1)                         # (B, 196)\n\n# Normalize into a probability distribution\nscore = score / score.sum(dim=1, keepdim=True)\n\nprint(score.shape)\nprint(\"Sum =\", score.sum().item())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:47:48.156954Z","iopub.execute_input":"2026-07-12T18:47:48.157256Z","iopub.status.idle":"2026-07-12T18:47:48.385763Z","shell.execute_reply.started":"2026-07-12T18:47:48.157234Z","shell.execute_reply":"2026-07-12T18:47:48.384825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score_map = score.reshape(14,14).detach().cpu()\n\nplt.figure(figsize=(6,6))\n\nplt.imshow(score_map, cmap=\"viridis\")\n\nplt.colorbar()\n\nplt.title(\"Token Significance Map\")\n\nplt.axis(\"off\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:47:59.713747Z","iopub.execute_input":"2026-07-12T18:47:59.71405Z","iopub.status.idle":"2026-07-12T18:47:59.894566Z","shell.execute_reply.started":"2026-07-12T18:47:59.714028Z","shell.execute_reply":"2026-07-12T18:47:59.893777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------------------------------\n# Paper-inspired adaptive sampling.\n#\n# Instead of selecting the Top-K tokens directly, tokens are sampled\n# according to their normalized significance scores.\n# ------------------------------------------------------------------\n\nkeep_ratio = 0.6\n\nnum_tokens = score.shape[1]\n\nnum_keep = int(num_tokens * keep_ratio)\n\nprob = score.squeeze(0)\n\nselected = torch.multinomial(\n    prob,\n    num_samples=num_keep,\n    replacement=False\n)\n\nselected = torch.sort(selected)[0]\n\nprint(\"Original tokens :\", num_tokens)\nprint(\"Retained tokens :\", len(selected))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:48:21.992218Z","iopub.execute_input":"2026-07-12T18:48:21.992493Z","iopub.status.idle":"2026-07-12T18:48:22.48902Z","shell.execute_reply.started":"2026-07-12T18:48:21.992472Z","shell.execute_reply":"2026-07-12T18:48:22.488331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mask = torch.zeros(14,14)\n\nfor idx in selected:\n    r = idx.item() // 14\n    c = idx.item() % 14\n    mask[r,c] = 1\n\nplt.figure(figsize=(6,6))\n\nplt.imshow(mask, cmap=\"gray\")\n\nplt.title(f\"Selected Tokens ({len(selected)})\")\n\nplt.axis(\"off\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:48:32.295687Z","iopub.execute_input":"2026-07-12T18:48:32.296052Z","iopub.status.idle":"2026-07-12T18:48:32.406742Z","shell.execute_reply.started":"2026-07-12T18:48:32.296023Z","shell.execute_reply":"2026-07-12T18:48:32.406104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Basic statistics of the significance scores.\n\nprint(f\"Total image patches      : {num_tokens}\")\nprint(f\"Selected patches         : {len(selected)}\")\nprint(f\"Selection ratio          : {len(selected)/num_tokens:.2f}\")\n\nprint()\n\nprint(f\"Maximum score            : {score.max().item():.6f}\")\nprint(f\"Minimum score            : {score.min().item():.6f}\")\nprint(f\"Mean score               : {score.mean().item():.6f}\")\nprint(f\"Standard deviation       : {score.std().item():.6f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:48:53.745474Z","iopub.execute_input":"2026-07-12T18:48:53.746218Z","iopub.status.idle":"2026-07-12T18:48:53.784274Z","shell.execute_reply.started":"2026-07-12T18:48:53.746185Z","shell.execute_reply":"2026-07-12T18:48:53.783517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"topk = 20\n\nvalues, indices = torch.topk(score.squeeze(), topk)\n\nprint(\"Top 20 Important Patches\\n\")\n\nfor rank, (idx, val) in enumerate(zip(indices, values), 1):\n\n    row = idx.item() // 14\n    col = idx.item() % 14\n\n    print(\n        f\"{rank:02d}. \"\n        f\"Patch {idx.item():3d} \"\n        f\"(row={row:2d}, col={col:2d}) \"\n        f\"Score={val.item():.6f}\"\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:49:21.529587Z","iopub.execute_input":"2026-07-12T18:49:21.530071Z","iopub.status.idle":"2026-07-12T18:49:21.543895Z","shell.execute_reply.started":"2026-07-12T18:49:21.530038Z","shell.execute_reply":"2026-07-12T18:49:21.542313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8,4))\n\nplt.hist(\n    score.cpu().numpy().flatten(),\n    bins=25,\n)\n\nplt.xlabel(\"Significance Score\")\nplt.ylabel(\"Number of Tokens\")\nplt.title(\"Distribution of Token Importance\")\n\nplt.grid(alpha=0.3)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:49:35.857768Z","iopub.execute_input":"2026-07-12T18:49:35.85811Z","iopub.status.idle":"2026-07-12T18:49:36.02362Z","shell.execute_reply.started":"2026-07-12T18:49:35.858086Z","shell.execute_reply":"2026-07-12T18:49:36.022909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sorted_scores = torch.sort(\n    score.squeeze(),\n    descending=True\n)[0]\n\ncum = torch.cumsum(sorted_scores, dim=0)\n\nplt.figure(figsize=(8,4))\n\nplt.plot(cum.cpu())\n\nplt.xlabel(\"Top Tokens\")\n\nplt.ylabel(\"Cumulative Importance\")\n\nplt.grid(alpha=0.3)\n\nplt.title(\"Cumulative Token Importance\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:49:43.868371Z","iopub.execute_input":"2026-07-12T18:49:43.869409Z","iopub.status.idle":"2026-07-12T18:49:44.070231Z","shell.execute_reply.started":"2026-07-12T18:49:43.869357Z","shell.execute_reply":"2026-07-12T18:49:44.069179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"keep_ratio = len(selected) / num_tokens\n\nprint(f\"Original Tokens : {num_tokens}\")\nprint(f\"Retained Tokens : {len(selected)}\")\n\nprint()\n\nprint(f\"Approximate Token Reduction : {(1-keep_ratio)*100:.2f}%\")\nprint(f\"Approximate Compute Kept    : {keep_ratio*100:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:49:50.063487Z","iopub.execute_input":"2026-07-12T18:49:50.064187Z","iopub.status.idle":"2026-07-12T18:49:50.069994Z","shell.execute_reply.started":"2026-07-12T18:49:50.064154Z","shell.execute_reply":"2026-07-12T18:49:50.068922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nsummary = pd.DataFrame({\n    \"Metric\":[\n        \"Original Tokens\",\n        \"Retained Tokens\",\n        \"Keep Ratio\",\n        \"Maximum Score\",\n        \"Minimum Score\",\n        \"Mean Score\",\n        \"Std Score\"\n    ],\n    \"Value\":[\n        num_tokens,\n        len(selected),\n        round(keep_ratio,3),\n        round(score.max().item(),6),\n        round(score.min().item(),6),\n        round(score.mean().item(),6),\n        round(score.std().item(),6)\n    ]\n})\n\nsummary","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-12T18:49:57.044534Z","iopub.execute_input":"2026-07-12T18:49:57.045349Z","iopub.status.idle":"2026-07-12T18:49:57.38067Z","shell.execute_reply.started":"2026-07-12T18:49:57.045319Z","shell.execute_reply":"2026-07-12T18:49:57.379848Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Notes\n\nInitially, I planned to reproduce the complete ATS implementation from the paper. However, after going through both the paper and the official implementation, I realised that ATS is much more than just computing token importance.\n\nThe paper mainly explains how token significance is computed, but the official implementation also modifies the transformer itself by introducing policy propagation, inverse transform sampling, attention reconstruction and progressive token pruning across multiple transformer blocks.\n\nFor this notebook, I focused on implementing the main idea behind ATS instead of reproducing the complete framework.\n\n### What I implemented\n\n- Used a pretrained DeiT-Small model as the backbone.\n- Computed the attention matrix manually.\n- Implemented the token significance score using the CLS attention and value vectors (Eq. 3 from the paper).\n- Normalized the significance scores.\n- Performed adaptive token selection based on these scores.\n- Visualized the selected tokens and analysed their distribution.\n\n### What is different from the official implementation\n\nThere are still a few things missing compared to the original ATS model.\n\n- I did not implement the policy tensor used to keep track of active tokens across transformer blocks.\n- The sampled tokens are not propagated to the next transformer layer.\n- The attention matrix is not reconstructed after sampling.\n- The implementation therefore does not actually reduce the computation inside the transformer.\n- Instead of the repository's inverse transform sampling pipeline, I used a simpler adaptive sampling strategy to demonstrate the token selection process.\n\n### Why I stopped here\n\nThe official ATS repository rewrites several parts of the Vision Transformer architecture. Reproducing it completely would require replacing the original attention blocks and modifying how tokens are passed through every transformer layer.\n\nSince my goal was to understand how ATS decides which tokens are important, I implemented the core scoring and sampling idea first. This gives a good intuition about the algorithm while keeping the implementation compatible with a standard pretrained DeiT model.\n\nThis notebook should therefore be considered a simplified implementation for understanding the paper rather than a full reproduction of the official ATS repository.","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}