{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":113204,"databundleVersionId":13751849,"sourceType":"competition"},{"sourceId":13471101,"sourceType":"datasetVersion","datasetId":8550389}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"9968af10-ba72-4e3a-b3a0-ea6a56110bd1","cell_type":"markdown","source":"### Fine-Tuned Models Overview\n\nThis directory (`/kaggle/input/finetuned-retrieval-pipeline-3-models`) contains **three fine-tuned models**, each initialized from pretrained *SentenceTransformer* backbones and trained locally on an **RTX 4080 GPU** as part of the **two-stage agentic retrieval pipeline**:\n\n- **`finetuned-document-ranking`** → Document-level bi-encoder retriever  \n- **`finetuned-chunk-retriever`** → Chunk-level bi-encoder retriever  \n- **`finetuned-chunk-reranker`** → Cross-encoder reranker for fine-grained relevance scoring  \n\nAll training scripts, ablation studies, and configuration files will be released publicly at:  \n[https://github.com/hutuhehe/icaif25-agentic-retrieval-pipeline](https://github.com/hutuhehe/icaif25-agentic-retrieval-pipeline)\n\n---\n\n### Inference for Kaggle Submission\n\nThe following code cells load these fine-tuned models and run inference to generate ranked document and chunk predictions for the official Kaggle competition submission.\n\n","metadata":{}},{"id":"ccbe2695-0c93-4d50-99ed-f2b54787f80b","cell_type":"code","source":"import os\n\n# List all  model folders in the input directory\ninput_path = '/kaggle/input/finetuned-retrieval-pipeline-3-models'\n\nfor folder in os.listdir(input_path):\n    folder_path = os.path.join(input_path, folder)\n    if os.path.isdir(folder_path):\n        print(folder)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T19:53:38.461733Z","iopub.execute_input":"2025-10-27T19:53:38.461985Z","iopub.status.idle":"2025-10-27T19:53:38.478294Z","shell.execute_reply.started":"2025-10-27T19:53:38.461959Z","shell.execute_reply":"2025-10-27T19:53:38.477366Z"}},"outputs":[],"execution_count":null},{"id":"c53585ad-6d6a-408e-af2e-5b4212428e5c","cell_type":"code","source":"# --- Config ---\nDOC_EVAL_PATH   = \"document_ranking_kaggle_eval.jsonl\"\nCHUNK_EVAL_PATH = \"chunk_ranking_kaggle_eval.jsonl\"\n#SUBMISSION_PATH = \"submission.csv\"\n\nCOMPETITION_NAME = \"agentic_retrieval_grand_challenge\"\n\n\n\n# Updated to use Kaggle dataset paths\nDOC_MODEL_DIR = \"/kaggle/input/finetuned-retrieval-pipeline-3-models/finetuned-document-ranking\"\nPRETRAINED_BIENC = \"/kaggle/input/finetuned-retrieval-pipeline-3-models/finetuned-chunk-retriever\"\nXENC_DIR = \"/kaggle/input/finetuned-retrieval-pipeline-3-models/finetuned-chunk-reranker\"\n\n\nMAX_SEQ_LEN = 256 # for document ranking biencoder\nTOP_K       = 5\n\n# Stage 1 retrieval (mirror evaluate_hybrid_retriever_rrf)\nCANDIDATE_CHUNK_DEPTH   = 140\nMAX_SUBCHUNKS_PER_CHUNK = 5\nRRF_K                   = 60\nCAND_PER_SYSTEM         = None\nMAX_LEN                 = 400\nOVERLAP                 = 50\n\nBATCH_SIZE_RETR         = 128   \nBATCH_SIZE_RERANK       = 32\n#BATCH_SIZE_RETR         = 4   # Reduced from 128 to run on cpu\n#BATCH_SIZE_RERANK       = 16   # Reduced from 32 to run on cpu\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T19:53:50.977794Z","iopub.execute_input":"2025-10-27T19:53:50.978277Z","iopub.status.idle":"2025-10-27T19:53:50.982770Z","shell.execute_reply.started":"2025-10-27T19:53:50.978254Z","shell.execute_reply":"2025-10-27T19:53:50.982040Z"}},"outputs":[],"execution_count":null},{"id":"e34cb8b2-7ddc-4cca-be42-3d59b31006b5","cell_type":"code","source":"# --- Config ---\nINPUT_PATH = \"/kaggle/input/acm-icaif-25-ai-agentic-retrieval-grand-challenge\"\n\nDOC_EVAL_PATH = f\"{INPUT_PATH}/document_ranking_kaggle_eval.jsonl\"\nCHUNK_EVAL_PATH = f\"{INPUT_PATH}/chunk_ranking_kaggle_eval.jsonl\"\n\n\n\n# Updated to use Kaggle dataset paths\nDOC_MODEL_DIR = \"/kaggle/input/finetuned-retrieval-pipeline-3-models/finetuned-document-ranking\"\nPRETRAINED_BIENC = \"/kaggle/input/finetuned-retrieval-pipeline-3-models/finetuned-chunk-retriever\"\nXENC_DIR = \"/kaggle/input/finetuned-retrieval-pipeline-3-models/finetuned-chunk-reranker\"\n\n\nMAX_SEQ_LEN = 256 # for document ranking biencoder\nTOP_K       = 5\n\n# Stage 1 retrieval (mirror evaluate_hybrid_retriever_rrf)\nCANDIDATE_CHUNK_DEPTH   = 140\nMAX_SUBCHUNKS_PER_CHUNK = 5\nRRF_K                   = 60\nCAND_PER_SYSTEM         = None\nMAX_LEN                 = 400\nOVERLAP                 = 50\nBATCH_SIZE_RETR         = 1\nBATCH_SIZE_RERANK       = 32\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T19:53:53.826485Z","iopub.execute_input":"2025-10-27T19:53:53.826734Z","iopub.status.idle":"2025-10-27T19:53:53.831474Z","shell.execute_reply.started":"2025-10-27T19:53:53.826715Z","shell.execute_reply":"2025-10-27T19:53:53.830590Z"}},"outputs":[],"execution_count":null},{"id":"06446853-8028-4993-a614-315f30479ca4","cell_type":"markdown","source":"### Imports and Utility Functions\n","metadata":{}},{"id":"9eb9fecc-48a6-4c6c-b36b-e9273dcbd3a6","cell_type":"code","source":"!pip install rank-bm25","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T19:54:10.251546Z","iopub.execute_input":"2025-10-27T19:54:10.251865Z","iopub.status.idle":"2025-10-27T19:54:16.117945Z","shell.execute_reply.started":"2025-10-27T19:54:10.251845Z","shell.execute_reply":"2025-10-27T19:54:16.117224Z"}},"outputs":[],"execution_count":null},{"id":"dc56e55a-0c4d-43e2-8a50-3984e76f6935","cell_type":"code","source":"import re\nimport json\nimport pandas as pd\nimport numpy as np\nfrom collections import defaultdict\nfrom typing import List, Dict, Any\n#from tqdm import tqdm\nfrom tqdm.auto import tqdm\n\nimport torch\nfrom sentence_transformers import SentenceTransformer, CrossEncoder, util\nfrom transformers import AutoTokenizer\nfrom rank_bm25 import BM25L\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T19:54:16.610078Z","iopub.execute_input":"2025-10-27T19:54:16.610729Z","iopub.status.idle":"2025-10-27T19:54:56.717418Z","shell.execute_reply.started":"2025-10-27T19:54:16.610702Z","shell.execute_reply":"2025-10-27T19:54:56.716817Z"}},"outputs":[],"execution_count":null},{"id":"bbf619ea-2deb-4ecf-909d-feab020dd2df","cell_type":"code","source":"def load_jsonl(path: str):\n    # Load a UTF-8 JSONL file into a list of dictionaries.\n    with open(path, \"r\", encoding=\"utf-8\") as f:\n        return [json.loads(line) for line in f if line.strip()]\n\n## extract questions from message content\ndef extract_question(content: str) -> str:\n    # Extract question text from message, preferring the section between \"Question:\" and \"Text chunks:\"\n    text = content.strip()\n    match = re.search(r\"Question:\\s*(.*?)\\s*Text chunks:\", text, re.DOTALL | re.IGNORECASE)\n    if match:\n        return match.group(1).strip()\n    all_matches = re.findall(r\"Question:\\s*(.*)\", text)\n    if all_matches:\n        return all_matches[-1].strip()\n    \n\n# extract chunks from message as list\ndef extract_chunks_as_strings(content: str):\n    # Extract all text chunks as list of strings, ignoring chunk indices\n    parts = content.split(\"Text chunks:\", 1)\n    if len(parts) < 2:\n        return []\n    chunks_text = parts[1].strip()\n    pattern = re.compile(r\"\\[Chunk Index (\\d+)\\]\\s*(.*?)(?=\\[Chunk Index \\d+\\]|\\Z)\", re.DOTALL)\n    chunks = []\n    for match in pattern.finditer(chunks_text):\n        chunks.append(match.group(2).strip())\n    return chunks\n\n\ndef finance_tokenizer(text: str):\n    raw = _FINANCE_PATTERN.findall(text)\n    tokens = []\n    for tok in raw:\n        # handle optional capture group (tickers)\n        if isinstance(tok, tuple):\n            tok = next(t for t in tok if t)\n        # preserve uppercase tickers; lowercase everything else\n        tokens.append(tok if _TICKER_FULL.fullmatch(tok) else tok.lower())\n    return tokens\n\n# split text into <= max_len tokens with overlap\ndef chunk_text(text, tokenizer, max_len=400, overlap=80):\n    tokens = tokenizer.encode(text, add_special_tokens=False)\n    chunks = []\n    start = 0\n    while start < len(tokens):\n        end = min(start + max_len, len(tokens))\n        chunk_tokens = tokens[start:end]\n        chunk_text = tokenizer.decode(chunk_tokens,skip_special_tokens=True)\n        chunks.append(chunk_text)\n        if end == len(tokens):\n            break\n        start = end - overlap\n    return chunks\n\n\n# Specific patterns first, generic last\n_FINANCE_PATTERN = re.compile(r\"\"\"\n    \\$\\d{1,3}(?:,\\d{3})*(?:\\.\\d+)?         # $1,000.50\n  | \\d+(?:\\.\\d+)?%                         # 12.4%\n  | (?<![A-Za-z0-9])([A-Z]{1,5})(?![A-Za-z0-9])  # Tickers: AAPL, TSLA (whole token)\n  | \\b\\d{1,2}-[A-Za-z]+\\b                  # SEC forms: 10-K, 8-K, 20-F, S-1, etc.\n  | [A-Za-z0-9_]+(?:-[A-Za-z0-9_]+)*       # Words (keep hyphenated compounds)\n\"\"\", re.VERBOSE)\n\n_TICKER_FULL = re.compile(r\"^[A-Z]{1,5}$\")\n\ndef scores_to_ranks_desc(scores: np.ndarray) -> np.ndarray:\n    \"\"\"\n    Convert scores to 1-based ranks in descending order (higher score = better rank).\n    Handles NaNs by treating them as -inf (worst).\n    Ties get the same rank as their first occurrence (stable ranking by order).\n    \"\"\"\n    if scores.ndim != 1:\n        scores = scores.ravel()\n\n    # Treat NaNs as -inf so they go to the bottom\n    scores = np.where(np.isnan(scores), -np.inf, scores)\n\n    order = np.argsort(-scores, kind=\"mergesort\")  # stable sort\n    ranks = np.empty_like(order)\n    ranks[order] = np.arange(1, scores.shape[0] + 1, dtype=np.int32)\n    return ranks\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T19:54:56.718648Z","iopub.execute_input":"2025-10-27T19:54:56.719420Z","iopub.status.idle":"2025-10-27T19:54:56.730163Z","shell.execute_reply.started":"2025-10-27T19:54:56.719370Z","shell.execute_reply":"2025-10-27T19:54:56.729306Z"}},"outputs":[],"execution_count":null},{"id":"70d411d7-8845-4026-afb3-212ecf88be7e","cell_type":"markdown","source":"##   1) Document ranking ","metadata":{}},{"id":"ac2d134a-d629-4454-9438-ee9c69c59a01","cell_type":"markdown","source":"## Ablation Study: Effect of Fine-Tuning and LLM-Generated Summaries\n\nWe initially hypothesized that a small pretrained bi-encoder (MiniLM-L6) would **struggle to understand the meaning of SEC filing types** such as *10-K*, *10-Q*, *8-K*, or *DEF14A*.  \nTo address this, the **submitted model for the ICAIF 2025 competition** was trained using **LLM-generated descriptive summaries** for each document type, under the assumption that richer textual context would improve semantic alignment.\n\n### Key Observation: Fine-Tuning Drives the Largest Gains\nAcross both configurations, **fine-tuning produces a major performance boost** over the zero-shot pretrained model — improving MAP@5 from roughly **0.84 → 0.94**, MRR@5 from **0.88 → 0.99**, and nDCG@5 from **0.69 → 0.91**.  \nThis confirms that **task-specific fine-tuning** is the primary driver of performance in document-type ranking.\n\n### Experimental Setup\n\nWe compared two configurations:\n\n1. **Without LLM Summary** – using only concise symbolic document-type labels (e.g., *10-K*, *10-Q*, *8-K*, *DEF14A*, *Earnings Report*).  \n2. **With LLM Summary (submitted version)** – replacing symbolic labels with descriptive sentences generated by a large language model.\n\n### Head-to-Head Comparison\n\n| Metric | Without LLM Summary | With LLM Summary (submitted) | Difference |\n|:--|:--:|:--:|:--:|\n| **Before fine-tuning** | **MAP@5: 0.8408**<br>**MRR@5: 0.8835**<br>**nDCG@5: 0.6955** | MAP@5: 0.8313<br>MRR@5: 0.8531<br>nDCG@5: 0.6760 | −0.95%<br>−3.44%<br>−2.80% |\n| **After fine-tuning** | **MAP@5: 0.9457**<br>**MRR@5: 0.9933**<br>**nDCG@5: 0.9142** | MAP@5: 0.9444<br>MRR@5: 0.9900<br>nDCG@5: 0.9138 | −0.14%<br>−0.33%<br>−0.04% |\n\n### Summary of Findings\n\nFine-tuning substantially improves ranking quality for both setups, confirming the effectiveness of the training pipeline.  \nHowever, **concise symbolic labels slightly outperform the LLM-descriptive version** after training.  \nThe largest gap appears in the **zero-shot (pre-fine-tuning)** case, where the LLM-summary variant underperforms by **3.4% MRR**.\n\nThese results show that the pretrained **MiniLM bi-encoder already encodes strong functional knowledge** of SEC filing types — understanding associations such as *“risk factors” → 10-K* or *“executive compensation” → DEF14A* even without additional descriptive text.  \nAdding LLM-generated summaries introduces redundant or overlapping financial terms (e.g., *“financials,” “management,” “earnings”*), which slightly reduce discriminative precision between types.\n\n**Conclusion:**  \nWhile the **submitted fine-tuned version** was trained with LLM-generated summaries for interpretability and semantic coverage, the ablation indicates that **concise symbolic labels** achieve slightly higher accuracy.  \nFine-tuning remains the dominant factor driving performance improvement.  \nAll released code and checkpoints replicate the **original submitted configuration** to ensure full consistency with the official competition results.\n","metadata":{}},{"id":"7dd23de2-ce93-4a56-bdbd-cfff29e6226b","cell_type":"markdown","source":"","metadata":{}},{"id":"3cbc01e7-a582-4f50-a65e-e26ef79916f2","cell_type":"code","source":"# Must match training order\nDOC_TYPE_TEXTS = [\n    \"DEF14A Proxy Statement: executive compensation, board governance, and shareholder proposals.\",\n    \"10-K Annual Report: audited yearly SEC filing with comprehensive financials and risk factors.\",\n    \"10-Q Quarterly Report: unaudited quarterly results, MD&A, and updated risk disclosures.\",\n    \"8-K Current Report: ad-hoc disclosure of material events and earnings releases.\",\n    \"Earnings Report: company-issued press release & earnings call highlights with KPIs and outlook.\"\n]\n\ndef generate_document_submission(doc_eval_path: str,\n                                 doc_model_dir: str,\n                                 max_seq_len: int = 256,\n                                 top_k: int = 5) -> pd.DataFrame:\n    model = SentenceTransformer(doc_model_dir)\n    model.max_seq_length = max_seq_len\n    device = model.device\n\n    type_embs = model.encode(\n        DOC_TYPE_TEXTS,\n        convert_to_tensor=True,\n        normalize_embeddings=True,\n        show_progress_bar=False,\n        device=device\n    )\n    \n    records = load_jsonl(doc_eval_path)\n    rows = []\n    for rec in tqdm(records, desc=\"Document ranking\"):\n        sample_id = rec[\"_id\"]\n        q = extract_question(rec[\"messages\"][0][\"content\"])\n        q_emb = model.encode(q, convert_to_tensor=True, normalize_embeddings=True,\n                             show_progress_bar=False, device=device)\n        sims = util.cos_sim(q_emb, type_embs).flatten()\n        ranked = torch.argsort(sims, descending=True).tolist()\n        for idx in ranked[:top_k]:\n            rows.append({\"sample_id\": sample_id, \"target_index\": idx})\n\n    return pd.DataFrame(rows)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T19:54:56.731026Z","iopub.execute_input":"2025-10-27T19:54:56.731642Z","iopub.status.idle":"2025-10-27T19:54:56.767209Z","shell.execute_reply.started":"2025-10-27T19:54:56.731603Z","shell.execute_reply":"2025-10-27T19:54:56.766499Z"}},"outputs":[],"execution_count":null},{"id":"098aa290-a65d-4f5d-b0be-5ce051ca8c3b","cell_type":"code","source":"    doc_df = generate_document_submission(\n        doc_eval_path=DOC_EVAL_PATH,\n        doc_model_dir=DOC_MODEL_DIR,\n        max_seq_len=MAX_SEQ_LEN,\n        top_k=TOP_K\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T19:54:56.768535Z","iopub.execute_input":"2025-10-27T19:54:56.768791Z","iopub.status.idle":"2025-10-27T19:55:10.349082Z","shell.execute_reply.started":"2025-10-27T19:54:56.768763Z","shell.execute_reply":"2025-10-27T19:55:10.348145Z"}},"outputs":[],"execution_count":null},{"id":"a805ca21-26d2-4569-b580-8d43a8425e3a","cell_type":"code","source":"doc_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T19:55:14.119988Z","iopub.execute_input":"2025-10-27T19:55:14.120826Z","iopub.status.idle":"2025-10-27T19:55:14.158065Z","shell.execute_reply.started":"2025-10-27T19:55:14.120790Z","shell.execute_reply":"2025-10-27T19:55:14.157428Z"}},"outputs":[],"execution_count":null},{"id":"0583b217-57cb-46de-9401-61c72fab2fd6","cell_type":"markdown","source":"## 2) chunck ranking\n###   Stage-1 candidates via evaluate_hybrid_retriever_rrf-style logic","metadata":{}},{"id":"68728459-1eb5-4733-8c94-2373ee41fc14","cell_type":"code","source":"def stage1_candidates_from_rrf(chunk_eval_path: str,\n                               pretrained_biencoder_name: str,\n                               rrf_k: int = 60,\n                               cand_per_system: int | None = None,\n                               candidate_chunk_depth: int = 80,\n                               max_subchunks_per_chunk: int = 5,\n                               max_len: int = 400,\n                               overlap: int = 80,\n                               batch_size: int = 128,\n                               max_seq_len: int = 256) -> List[Dict[str, Any]]:\n    dense_model = SentenceTransformer(pretrained_biencoder_name)\n    dense_model.max_seq_length = max_seq_len\n    device = dense_model.device\n    hf_model_name = dense_model[0].auto_model.config._name_or_path\n    tokenizer = AutoTokenizer.from_pretrained(hf_model_name)\n\n    data = load_jsonl(chunk_eval_path)\n    pool: List[Dict[str, Any]] = []\n\n    for ex in tqdm(data, desc=\"Stage 1: Hybrid RRF\"):\n        sample_id = ex[\"_id\"]\n        content   = ex[\"messages\"][0][\"content\"]\n        query     = extract_question(content)\n        chunks    = extract_chunks_as_strings(content)\n        if not query or not chunks:\n            continue\n\n        # Sub-chunking\n        subchunks, sub2parent = [], []\n        for cid, ch in enumerate(chunks):\n            pieces = chunk_text(ch, tokenizer, max_len=max_len, overlap=overlap)\n            subchunks.extend(pieces)\n            sub2parent.extend([cid] * len(pieces))\n        if not subchunks:\n            continue\n\n        # Dense scores\n        embs = dense_model.encode([query] + subchunks, convert_to_tensor=True,\n                                  show_progress_bar=False, device=device, batch_size=batch_size)\n        q_emb, sub_embs = embs[0], embs[1:]\n        dense_scores_all = util.cos_sim(q_emb, sub_embs)[0].cpu().numpy()\n\n        # BM25L scores (finance tokenizer)\n        tokenized_sub = [finance_tokenizer(sc) for sc in subchunks]\n        tokenized_q   = finance_tokenizer(query)\n        bm25 = BM25L(tokenized_sub, k1=1.2, b=0.9, delta=0.5)\n        bm25_scores_all = bm25.get_scores(tokenized_q)\n\n        # Candidate prefilter per system (optional)\n        if cand_per_system is not None:\n            dense_top = np.argsort(-dense_scores_all)[:cand_per_system]\n            bm25_top  = np.argsort(-bm25_scores_all)[:cand_per_system]\n            keep = np.unique(np.concatenate([dense_top, bm25_top]))\n        else:\n            keep = np.arange(len(subchunks))\n\n        # RRF fusion (mirror of evaluate_hybrid_retriever_rrf)\n        dense_rank = scores_to_ranks_desc(dense_scores_all[keep])\n        bm25_rank  = scores_to_ranks_desc(bm25_scores_all[keep])\n        rrf_scores_sub = 1.0/(rrf_k + dense_rank) + 1.0/(rrf_k + bm25_rank)\n        ranked_sub = sorted(zip(keep, rrf_scores_sub), key=lambda x: x[1], reverse=True)\n\n        # Collapse to parent chunk via max fused score\n        chunk_scores = {}\n        for sidx, s in ranked_sub:\n            pid = sub2parent[sidx]\n            if (pid not in chunk_scores) or (s > chunk_scores[pid]):\n                chunk_scores[pid] = s\n\n        # Keep top-N parent chunks\n        top_parents = sorted(chunk_scores.items(), key=lambda x: x[1], reverse=True)[:candidate_chunk_depth]\n        keep_parent_ids = {pid for pid, _ in top_parents}\n\n        # Up to M subchunks per kept parent\n        selected_sub, per_parent = [], defaultdict(int)\n        for sidx, s in ranked_sub:\n            pid = sub2parent[sidx]\n            if pid in keep_parent_ids and per_parent[pid] < max_subchunks_per_chunk:\n                selected_sub.append((sidx, pid))\n                per_parent[pid] += 1\n\n        pool.append({\n            \"sample_id\": sample_id,\n            \"query\": query,\n            \"candidates\": [subchunks[sid] for sid, _ in selected_sub],\n            \"parent_chunk_ids\": [pid for _, pid in selected_sub]  # aligned by position with candidates\n        })\n\n    return pool\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T19:55:17.710070Z","iopub.execute_input":"2025-10-27T19:55:17.710348Z","iopub.status.idle":"2025-10-27T19:55:17.722252Z","shell.execute_reply.started":"2025-10-27T19:55:17.710328Z","shell.execute_reply":"2025-10-27T19:55:17.721684Z"}},"outputs":[],"execution_count":null},{"id":"fefd5aa2-8981-47d1-8d49-971121fdc077","cell_type":"code","source":"    # Chunk Stage 1 (pretrained bi-encoder + BM25L + RRF; mirrors evaluate_hybrid_retriever_rrf)\n    pool = stage1_candidates_from_rrf(\n        chunk_eval_path=CHUNK_EVAL_PATH,\n        pretrained_biencoder_name=PRETRAINED_BIENC,\n        rrf_k=RRF_K,\n        cand_per_system=CAND_PER_SYSTEM,\n        candidate_chunk_depth=CANDIDATE_CHUNK_DEPTH,\n        max_subchunks_per_chunk=MAX_SUBCHUNKS_PER_CHUNK,\n        max_len=MAX_LEN,\n        overlap=OVERLAP,\n        batch_size=BATCH_SIZE_RETR,\n        max_seq_len=MAX_SEQ_LEN\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T19:55:20.519823Z","iopub.execute_input":"2025-10-27T19:55:20.520110Z","iopub.status.idle":"2025-10-27T20:02:37.483435Z","shell.execute_reply.started":"2025-10-27T19:55:20.520088Z","shell.execute_reply":"2025-10-27T20:02:37.482444Z"}},"outputs":[],"execution_count":null},{"id":"ff7117c7-dce0-4335-8d28-597275d4398d","cell_type":"markdown","source":"###  Stage-2 rerank with your fine-tuned cross-encoder","metadata":{}},{"id":"1a2f1e0f-de5b-4f37-8367-deedffc11936","cell_type":"code","source":"\n\ndef stage2_rerank_and_submission(candidate_pool: List[Dict[str, Any]],\n                                 cross_encoder_dir: str,\n                                 top_k: int = 5,\n                                 batch_size: int = 32) -> pd.DataFrame:\n    reranker = CrossEncoder(cross_encoder_dir)\n    rows = []\n    \n    for item in tqdm(candidate_pool, desc=\"Stage 2: Reranking\", position=0, leave=True, disable=False):\n        q           = item[\"query\"]\n        candidates  = item[\"candidates\"]\n        parent_cids = item[\"parent_chunk_ids\"]\n        sample_id   = item[\"sample_id\"]\n        \n        pairs  = [(q, c) for c in candidates]\n        scores = reranker.predict(pairs, batch_size=batch_size, show_progress_bar=False)\n        \n        ranked = sorted(zip(parent_cids, scores), key=lambda x: -float(x[1]))\n        \n        seen, final_parent = set(), []\n        for cid, sc in ranked:\n            if cid not in seen:\n                seen.add(cid)\n                final_parent.append(cid)\n            if len(final_parent) >= top_k:\n                break\n        \n        for cid in final_parent:\n            rows.append({\"sample_id\": sample_id, \"target_index\": cid})\n    \n    return pd.DataFrame(rows)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T20:02:45.423895Z","iopub.execute_input":"2025-10-27T20:02:45.424492Z","iopub.status.idle":"2025-10-27T20:02:45.430756Z","shell.execute_reply.started":"2025-10-27T20:02:45.424470Z","shell.execute_reply":"2025-10-27T20:02:45.430023Z"}},"outputs":[],"execution_count":null},{"id":"38573c11-0958-43fc-a8c9-e84a28131d97","cell_type":"code","source":"    # Chunk Stage 2 (fine-tuned cross-encoder)\n    chunk_df = stage2_rerank_and_submission(\n        candidate_pool=pool,\n        cross_encoder_dir=XENC_DIR,\n        top_k=TOP_K,\n        batch_size=BATCH_SIZE_RERANK\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T20:02:47.478378Z","iopub.execute_input":"2025-10-27T20:02:47.479127Z","iopub.status.idle":"2025-10-27T20:06:18.626237Z","shell.execute_reply.started":"2025-10-27T20:02:47.479099Z","shell.execute_reply":"2025-10-27T20:06:18.625352Z"}},"outputs":[],"execution_count":null},{"id":"e79e5438-d3f9-40e2-8303-93ae95615634","cell_type":"code","source":"chunk_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T20:06:44.856449Z","iopub.execute_input":"2025-10-27T20:06:44.856768Z","iopub.status.idle":"2025-10-27T20:06:44.866334Z","shell.execute_reply.started":"2025-10-27T20:06:44.856745Z","shell.execute_reply":"2025-10-27T20:06:44.865433Z"}},"outputs":[],"execution_count":null},{"id":"0f8d481e-fc93-4aca-9e7c-d56827df293c","cell_type":"code","source":"import os\n\n# Save to current directory\nSUBMISSION_PATH = \"kaggle_submission.csv\"\n\n# Merge and write\nfinal_df = pd.concat([chunk_df, doc_df], ignore_index=True)\nfinal_df.to_csv(SUBMISSION_PATH, index=False)\nprint(f\"Saved submission: {SUBMISSION_PATH}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T20:06:48.219508Z","iopub.execute_input":"2025-10-27T20:06:48.219918Z","iopub.status.idle":"2025-10-27T20:06:48.239722Z","shell.execute_reply.started":"2025-10-27T20:06:48.219885Z","shell.execute_reply":"2025-10-27T20:06:48.238894Z"}},"outputs":[],"execution_count":null}]}