{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":113204,"databundleVersionId":13751849,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"5e71c895","cell_type":"markdown","source":"# Baseline (fast_graphrag) – Agentic Retrieval Grand Challenge (Minimal demo)\n\nThis notebook demonstrates a minimal end-to-end flow using `fast_graphrag`:\n1) load `test.jsonl` (the small sample attached),\n2) create a `GraphRAG` instance with a lightweight, non-LLM information-extraction implementation (so the demo runs offline),\n3) insert a small synthetic document per document-type candidate,\n4) run queries and produce `submission.csv` with ranked candidate indices.\n\nNotes: This is a minimal demo meant to run inside Kaggle or locally without external LLM API calls. Replace the minimal information-extraction service with the default one when you have LLM access to get production-quality graphs.","metadata":{}},{"id":"61ad6a21-07f0-4c9e-9d11-d71419f78a68","cell_type":"code","source":"# !pip install openai tiktoken python-dotenv pydantic tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T09:07:43.55019Z","iopub.execute_input":"2025-11-16T09:07:43.550438Z","iopub.status.idle":"2025-11-16T09:07:43.554553Z","shell.execute_reply.started":"2025-11-16T09:07:43.550397Z","shell.execute_reply":"2025-11-16T09:07:43.553886Z"}},"outputs":[],"execution_count":null},{"id":"2138e9d2-049c-4660-b667-15716a375a57","cell_type":"code","source":"import asyncio\nimport csv\nimport json\nimport os\nimport traceback\nfrom typing import Dict, List\n\nimport tiktoken\nfrom dotenv import load_dotenv\nfrom openai import AsyncOpenAI\nfrom pydantic import BaseModel\nfrom tqdm.asyncio import tqdm\n\n\nfrom pathlib import Path\nimport re\nfrom collections import defaultdict\nload_dotenv()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T11:38:45.279457Z","iopub.execute_input":"2025-11-16T11:38:45.279741Z","iopub.status.idle":"2025-11-16T11:38:45.286449Z","shell.execute_reply.started":"2025-11-16T11:38:45.279722Z","shell.execute_reply":"2025-11-16T11:38:45.285829Z"}},"outputs":[],"execution_count":null},{"id":"f64d8d7d-5097-4f8c-8ea4-edb70052a5ee","cell_type":"code","source":"# import site\n# print(\"包安装位置:\")\n# for path in site.getsitepackages():\n#     print(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T11:31:02.627789Z","iopub.execute_input":"2025-11-16T11:31:02.628504Z","iopub.status.idle":"2025-11-16T11:31:02.631887Z","shell.execute_reply.started":"2025-11-16T11:31:02.628478Z","shell.execute_reply":"2025-11-16T11:31:02.631066Z"}},"outputs":[],"execution_count":null},{"id":"19429582-b6db-40c9-a579-92075ac20570","cell_type":"code","source":"# 在notebook开头添加这个单元格\nimport os\nif not os.path.exists('/usr/local/lib/python3.11/dist-packages/fast_graphrag'):\n    print(\"安装fast_graphrag...\")\n    !pip install fast_graphrag\nelse:\n    print(\"fast_graphrag已安装\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T11:31:03.025323Z","iopub.execute_input":"2025-11-16T11:31:03.025637Z","iopub.status.idle":"2025-11-16T11:32:14.769871Z","shell.execute_reply.started":"2025-11-16T11:31:03.025616Z","shell.execute_reply":"2025-11-16T11:32:14.769052Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"id":"7e37e230-f571-4688-8c44-0279e92fda9f","cell_type":"code","source":"# Import fast_graphrag and required internals to provide a minimal InformationExtraction service\nfrom fast_graphrag import GraphRAG, QueryParam\nfrom fast_graphrag._services._base import BaseInformationExtractionService\nfrom fast_graphrag._storage._gdb_igraph import IGraphStorage, IGraphStorageConfig\nfrom fast_graphrag._types import TEntity, TRelation, TDocument\nimport asyncio","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T11:38:50.473923Z","iopub.execute_input":"2025-11-16T11:38:50.474511Z","iopub.status.idle":"2025-11-16T11:38:50.47844Z","shell.execute_reply.started":"2025-11-16T11:38:50.474489Z","shell.execute_reply":"2025-11-16T11:38:50.477575Z"}},"outputs":[],"execution_count":null},{"id":"7feb32dc-703d-40a6-9ba0-b1b044bf1778","cell_type":"code","source":"BASE_DIR    = \"/kaggle/input/acm-icaif-25-ai-agentic-retrieval-grand-challenge\"\nDOC_DEV     = f\"{BASE_DIR}/document_ranking_kaggle_dev.jsonl\"\nCHUNK_DEV   = f\"{BASE_DIR}/chunk_ranking_kaggle_dev.jsonl\"\nDOC_EVAL    = f\"{BASE_DIR}/document_ranking_kaggle_eval.jsonl\"\nCHUNK_EVAL  = f\"{BASE_DIR}/chunk_ranking_kaggle_eval.jsonl\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T11:38:53.092231Z","iopub.execute_input":"2025-11-16T11:38:53.092865Z","iopub.status.idle":"2025-11-16T11:38:53.096845Z","shell.execute_reply.started":"2025-11-16T11:38:53.092832Z","shell.execute_reply":"2025-11-16T11:38:53.096105Z"}},"outputs":[],"execution_count":null},{"id":"e6c7a337-973b-4807-89ef-3987e9360b3b","cell_type":"code","source":"import json\nimport math\n\ndef sample_jsonl_streaming(input_path, output_path, sample_ratio=0.1):\n    \"\"\"\n    流式处理大JSONL文件，内存友好\n    \"\"\"\n    # 第一步：先计算有效行数\n    valid_count = 0\n    with open(input_path, 'r') as f_in:\n        for line in f_in:\n            line = line.strip()\n            if not line:\n                continue\n            try:\n                json.loads(line)\n                valid_count += 1\n            except json.JSONDecodeError:\n                continue\n    \n    # 第二步：计算抽样参数\n    sample_size = math.ceil(valid_count * sample_ratio)\n    step = max(1, valid_count // sample_size)\n    \n    # 第三步：流式抽样\n    current_index = 0\n    written_count = 0\n    \n    with open(input_path, 'r') as f_in, open(output_path, 'w') as f_out:\n        for line in f_in:\n            line = line.strip()\n            if not line:\n                continue\n            try:\n                # 验证JSON\n                json.loads(line)\n                \n                # 抽样逻辑\n                if current_index % step == 0 and written_count < sample_size:\n                    f_out.write(line + '\\n')\n                    written_count += 1\n                \n                current_index += 1\n            except json.JSONDecodeError:\n                continue\n    \n    print(f\"从 {input_path} 抽取了 {written_count} 条有效样本到 {output_path}\")\n    print(f\"原始文件有 {valid_count} 条有效JSON记录\")\n\nOUT_DIR    = \"/kaggle/working\"\nOUT_DOC= f\"{OUT_DIR}/document_ranking_kaggle_test.jsonl\"\nOUT_CHUNK=f\"{OUT_DIR}/chunk_ranking_kaggle_test.jsonl\"\n# 执行抽样\nsample_jsonl_streaming(DOC_DEV, OUT_DOC)\nsample_jsonl_streaming(CHUNK_DEV, OUT_CHUNK)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T11:38:53.882442Z","iopub.execute_input":"2025-11-16T11:38:53.883092Z","iopub.status.idle":"2025-11-16T11:39:43.823661Z","shell.execute_reply.started":"2025-11-16T11:38:53.883074Z","shell.execute_reply":"2025-11-16T11:39:43.822561Z"}},"outputs":[],"execution_count":null},{"id":"c6611fa5-175f-48a5-894b-ed97d45ca4b3","cell_type":"code","source":"import os\nfrom kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nsecret_value_0 = user_secrets.get_secret(\"DATABRICKS_TOKEN\")\nsecret_value_1 = user_secrets.get_secret(\"OPENAI_API_KEY\")\nos.environ[\"OPENAI_API_KEY\"] = secret_value_1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T11:39:43.825104Z","iopub.execute_input":"2025-11-16T11:39:43.825465Z","iopub.status.idle":"2025-11-16T11:39:43.933388Z","shell.execute_reply.started":"2025-11-16T11:39:43.825445Z","shell.execute_reply":"2025-11-16T11:39:43.932804Z"}},"outputs":[],"execution_count":null},{"id":"d84a586c-fde9-4c63-8f49-6a54452be6c2","cell_type":"markdown","source":"创建自定义的 Databricks LLM 服务类","metadata":{}},{"id":"4dee3e48-8864-46f2-8413-503232035263","cell_type":"code","source":"from fast_graphrag._llm._base import BaseLLMService\nfrom typing import Type, Any, Optional, List, Dict\nimport json\nfrom openai import AsyncOpenAI\n\n# 创建一个简单的类来模拟 Fast GraphRAG 期望的对象\nclass EntityResponse:\n    def __init__(self, named, generic):\n        self.named = named\n        self.generic = generic\n\n# 自定义 Databricks LLM 服务类\nclass DatabricksLLMService(BaseLLMService):\n    def __init__(self, databricks_token: str, base_url: str, model: str = \"databricks-model\"):\n        self.client = AsyncOpenAI(\n            api_key=databricks_token,\n            base_url=base_url\n        )\n        self.model = model\n    \n    async def send_message(\n        self,\n        prompt: str,\n        system_prompt: Optional[str] = None,\n        history_messages: Optional[List[Dict[str, str]]] = None,\n        response_model: Optional[Type[Any]] = None,\n        **kwargs: Any,\n    ) -> tuple:\n        messages = []\n        \n        if system_prompt:\n            messages.append({\"role\": \"system\", \"content\": system_prompt})\n        \n        if history_messages:\n            messages.extend(history_messages)\n        \n        messages.append({\"role\": \"user\", \"content\": prompt})\n        \n        # 使用 Databricks API\n        chat_completion = await self.client.chat.completions.create(\n            model=self.model,\n            messages=messages,\n            **kwargs\n        )\n        \n        content = chat_completion.choices[0].message.content\n\n        print(f\"\\n content = {content}\")\n        \n        # 如果指定了响应模型，尝试解析\n        if response_model:\n            try:\n                # 检查内容类型\n                if isinstance(content, (list, dict)):\n                    # 如果已经是列表或字典，直接使用\n                    parsed_response = content\n                else:\n                    # 如果是字符串，尝试 JSON 解析\n                    parsed_response = json.loads(content)\n                \n                # 如果解析后的响应是字典，转换为 EntityResponse 对象\n                if isinstance(parsed_response, dict) and \"named\" in parsed_response and \"generic\" in parsed_response:\n                    entity_response = EntityResponse(\n                        named=parsed_response[\"named\"],\n                        generic=parsed_response[\"generic\"]\n                    )\n                    return entity_response, chat_completion\n                else:\n                    # 如果不是期望的格式，直接返回\n                    return parsed_response, chat_completion\n                    \n            except (json.JSONDecodeError, TypeError) as e:\n                # 如果解析失败，返回原始内容\n                print(f\"⚠️ 响应解析失败，使用原始内容: {e}\")\n                return content, chat_completion\n        \n        # 如果没有指定响应模型，直接返回内容\n        return content, chat_completion","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T12:19:56.973736Z","iopub.execute_input":"2025-11-16T12:19:56.974475Z","iopub.status.idle":"2025-11-16T12:19:56.984567Z","shell.execute_reply.started":"2025-11-16T12:19:56.97445Z","shell.execute_reply":"2025-11-16T12:19:56.983816Z"}},"outputs":[],"execution_count":null},{"id":"8ab0cd16-bc6f-4193-811d-779e6bf9ae4b","cell_type":"code","source":"# 设置 Databricks 参数\nfrom kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nsecret_value_0 = user_secrets.get_secret(\"DATABRICKS_BASE_URL\")\nsecret_value_1 = user_secrets.get_secret(\"DATABRICKS_TOKEN\")\nsecret_value_2 = user_secrets.get_secret(\"OPENAI_API_KEY\")\n\n\nDATABRICKS_TOKEN = secret_value_1\nDATABRICKS_BASE_URL = secret_value_0\nMODEL_NAME = \"databricks-qwen3-next-80b-a3b-instruct\"  \n\n# 创建自定义 LLM 服务\nllm_service = DatabricksLLMService(\n    databricks_token=DATABRICKS_TOKEN,\n    base_url=DATABRICKS_BASE_URL,\n    model=MODEL_NAME\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T12:19:58.811427Z","iopub.execute_input":"2025-11-16T12:19:58.811724Z","iopub.status.idle":"2025-11-16T12:19:58.977888Z","shell.execute_reply.started":"2025-11-16T12:19:58.811704Z","shell.execute_reply":"2025-11-16T12:19:58.977299Z"}},"outputs":[],"execution_count":null},{"id":"4e3317ff-6019-45fa-8663-84b0d6986c43","cell_type":"code","source":"import time\n# 领域，它主要告诉你我们正在处理的是什么，影响整个最高层次的上下文；\nDOMAIN = \"Rank financial document types by relevance to the question. Provide your ranking as a list of indices from most relevant to least relevant.\"\n\n# 我们提供一些示例查询\nEXAMPLE_QUERIES = [\n          \n    \"How did analysts question the outlook for Agilent Technologies’ instrument sales performance?\",\n          \n    \"How do sustainability or ESG considerations influence customer demand in Agilent Technologies’ market?\",\n          \n    \"What exposure does Agilent Technologies, Inc. have to credit market conditions affecting customer financing for its scientific instruments?\",\n          \n    \"Can you elaborate on how new or potential tax reforms may affect the company¡¯s capital allocation strategy and long-term investment decisions?\",\n          \n    \"What major market changes has Apple recently highlighted, and how might these impact the company?\"\n          \n]\n\n# 设置我们希望处理的实体类型\n\nENTITY_TYPES = [\n    \"Company\",\n    \"Product\",\n    \"Financial Metric\",\n    \"Document Type\",\n    \"Event\",\n    \"Person\",\n    \"Date\",\n    \"Location\"\n]\n          \n# 创建唯一的工作目录（避免冲突）\ntimestamp = int(time.time())\n\n# 初始化 GraphRAG 时传入自定义服务\ngrag = GraphRAG(\n    working_dir=f\"/kaggle/working/graphrag_{timestamp}\",\n    domain=DOMAIN,\n    example_queries=\"\\n\".join(EXAMPLE_QUERIES),\n    entity_types=ENTITY_TYPES,\n    config=GraphRAG.Config(\n        llm_service=llm_service\n    )\n)\n         \n# grag = GraphRAG(\n          \n#     working_dir=f\"/kaggle/working/graphrag_{timestamp}\",\n          \n#     domain=DOMAIN,\n          \n#     example_queries=\"\\n\".join(EXAMPLE_QUERIES),\n          \n#     entity_types=ENTITY_TYPES\n          \n# )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T12:20:02.612736Z","iopub.execute_input":"2025-11-16T12:20:02.613425Z","iopub.status.idle":"2025-11-16T12:20:02.668948Z","shell.execute_reply.started":"2025-11-16T12:20:02.613401Z","shell.execute_reply":"2025-11-16T12:20:02.668324Z"}},"outputs":[],"execution_count":null},{"id":"be0fb738-4941-41e0-80e1-6e2211d8a2ca","cell_type":"code","source":"# 添加金融文档类型的知识\nfinancial_docs_knowledge = \"\"\"\nFinancial document types and their purposes for company analysis:\n\n1. DEF14A (Proxy Statement):\n   - Purpose: Contains information about director elections, executive compensation, and other corporate governance matters\n   - Relevance: Useful for understanding company leadership and governance structure\n   - Use cases: Analyzing executive compensation trends, board composition, shareholder proposals\n\n2. 10-K (Annual Report):\n   - Purpose: Provides a comprehensive overview of a company's financial performance, business risks, and management discussion\n   - Relevance: Most detailed financial document, essential for comprehensive company analysis\n   - Use cases: Analyzing annual financial performance, business risks, long-term strategy\n\n3. 10-Q (Quarterly Report):\n   - Purpose: Offers interim financial updates and disclosures about significant events during the quarter\n   - Relevance: Important for tracking recent performance and changes\n   - Use cases: Monitoring quarterly performance, identifying recent trends and events\n\n4. 8-K (Current Report):\n   - Purpose: Used to report major events that shareholders should know about\n   - Relevance: Critical for timely information about significant company events\n   - Use cases: Tracking mergers/acquisitions, leadership changes, financial restatements\n\n5. Earnings Reports:\n   - Purpose: Include earnings releases and earnings call transcripts\n   - Relevance: Provides management's discussion of recent performance and future outlook\n   - Use cases: Understanding quarterly results, management guidance, analyst Q&A\n\nRanking guidelines:\n- For questions about recent performance changes: Earnings > 10-Q > 10-K > 8-K > DEF14A\n- For questions about long-term trends: 10-K > 10-Q > Earnings > 8-K > DEF14A\n- For questions about specific events: 8-K > 10-Q > 10-K > Earnings > DEF14A\n- For questions about governance: DEF14A > 10-K > 10-Q > 8-K > Earnings\n\"\"\"\n# 使用异步方式插入数据\nasync def async_insert_data():\n    await grag.async_insert(financial_docs_knowledge)\n    print(\"数据插入成功！\")\n\n# 运行异步插入\ntry:\n    asyncio.run(async_insert_data())\nexcept RuntimeError as e:\n    # 如果事件循环已经在运行，使用当前事件循环\n    loop = asyncio.get_event_loop()\n    if loop.is_running():\n        loop.create_task(async_insert_data())\n    else:\n        loop.run_until_complete(async_insert_data())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T12:25:36.90001Z","iopub.execute_input":"2025-11-16T12:25:36.900769Z","iopub.status.idle":"2025-11-16T12:25:54.844198Z","shell.execute_reply.started":"2025-11-16T12:25:36.900745Z","shell.execute_reply":"2025-11-16T12:25:54.843477Z"}},"outputs":[],"execution_count":null},{"id":"6e593ba7-13e5-4fcf-82a2-d5017f498127","cell_type":"markdown","source":"","metadata":{}},{"id":"4c079579-1f0e-41ac-b3a2-a8ac4b0aed71","cell_type":"code","source":"def knowledge_based_ranking(question, knowledge_entities):\n    \"\"\"基于知识图谱结构的智能排名\"\"\"\n    question_lower = question.lower()\n    \n    # 从知识图谱中提取关键词映射\n    doc_keywords = {\n        \"Earnings Reports\": [\"earnings\", \"recent\", \"latest\", \"performance\", \"quarterly\", \"results\"],\n        \"10-Q\": [\"quarterly\", \"interim\", \"recent\", \"update\", \"short-term\"],\n        \"10-K\": [\"annual\", \"comprehensive\", \"long-term\", \"overview\", \"yearly\"],\n        \"8-K\": [\"event\", \"major\", \"current\", \"acquisition\", \"change\", \"announcement\"],\n        \"DEF14A\": [\"governance\", \"executive\", \"compensation\", \"board\", \"director\"]\n    }\n    \n    # 计算每个文档类型的相关性得分\n    scores = {doc: 0 for doc in doc_keywords.keys()}\n    \n    for doc, keywords in doc_keywords.items():\n        for keyword in keywords:\n            if keyword in question_lower:\n                scores[doc] += 1\n    \n    # 转换为索引排名\n    doc_to_index = {\n        \"DEF14A\": 0,\n        \"10-K\": 1, \n        \"10-Q\": 2,\n        \"8-K\": 3,\n        \"Earnings Reports\": 4\n    }\n    \n    # 按得分排序\n    ranked_docs = sorted(scores.items(), key=lambda x: x[1], reverse=True)\n    ranked_indices = [doc_to_index[doc] for doc, score in ranked_docs]\n    \n    return ranked_indices\n\n# # 使用基于知识图谱的排名\n# question = \"How has Apple's smartphone market share changed based on the latest earnings release?\"\n# ranking = knowledge_based_ranking(question, None)  # 这里可以传入具体的知识实体\n# print(f\"基于知识图谱的排名: {ranking}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T12:39:28.502302Z","iopub.execute_input":"2025-11-16T12:39:28.502617Z","iopub.status.idle":"2025-11-16T12:39:28.509013Z","shell.execute_reply.started":"2025-11-16T12:39:28.502596Z","shell.execute_reply":"2025-11-16T12:39:28.508102Z"}},"outputs":[],"execution_count":null},{"id":"14f0b07f-3ad3-4135-860f-598481b31b7a","cell_type":"code","source":"from openai import AsyncOpenAI\n\nasync def generate_answer_with_knowledge_graph(question):\n    \"\"\"利用已构建的知识图谱生成答案\"\"\"\n    client = AsyncOpenAI(\n        api_key=DATABRICKS_TOKEN,\n        base_url=DATABRICKS_BASE_URL\n    )\n    \n    # 构建利用知识图谱的提示词\n    prompt = f\"\"\"\n基于以下金融文档知识图谱，回答问题：\n\n金融文档类型知识：\n- DEF14A: 代理声明，包含董事选举、高管薪酬等公司治理信息\n- 10-K: 年度报告，提供公司财务绩效、业务风险和管理的全面概述\n- 10-Q: 季度报告，提供中期财务更新和重大事件披露\n- 8-K: 当前报告，用于报告股东应了解的重大事件\n- 收益报告: 包括收益发布和收益电话会议记录，提供管理层对近期绩效和未来展望的讨论\n\n问题: {question}\n\n请根据上述知识，对文档类型进行相关性排名，仅返回排名列表，格式为：[最相关, ..., 最不相关]\n\"\"\"\n    \n    try:\n        response = await client.chat.completions.create(\n            model=\"databricks-qwen3-next-80b-a3b-instruct\",\n            messages=[{\"role\": \"user\", \"content\": prompt}],\n            max_tokens=100,\n            temperature=0.1\n        )\n        return response.choices[0].message.content\n    except Exception as e:\n        return f\"Error: {e}\"\n\n# 测试\n# question = \"How has Apple's smartphone market share changed based on the latest earnings release?\"\n# result = await generate_answer_with_knowledge_graph(question)\n# print(f\"利用知识图谱的结果: {result}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T12:40:07.697817Z","iopub.execute_input":"2025-11-16T12:40:07.698584Z","iopub.status.idle":"2025-11-16T12:40:07.705764Z","shell.execute_reply.started":"2025-11-16T12:40:07.698551Z","shell.execute_reply":"2025-11-16T12:40:07.704648Z"}},"outputs":[],"execution_count":null},{"id":"6353300a-568f-45e1-af58-fe1ce856753a","cell_type":"code","source":"async def hybrid_ranking_solution(question):\n    \"\"\"混合解决方案：先尝试直接查询，失败时使用规则引擎\"\"\"\n    print(f\"问题: {question}\")\n    \n    # 1. 首先尝试直接查询（利用知识图谱）\n    try:\n        direct_result = await generate_answer_with_knowledge_graph(question)\n        if \"sorry\" not in direct_result.lower() and \"not able\" not in direct_result.lower():\n            print(\"✅ 直接查询成功\")\n            return direct_result\n        else:\n            raise Exception(\"模型拒绝回答\")\n    except Exception as e:\n        print(f\"❌ 直接查询失败: {e}\")\n    \n    # 2. 使用基于知识图谱的规则引擎\n    print(\"🔄 切换到规则引擎\")\n    ranking = knowledge_based_ranking(question, None)\n    return f\"基于知识图谱的规则排名: {ranking}\"\n\n# 使用混合方法\nresult = await hybrid_ranking_solution(question)\nprint(f\"最终结果: {result}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T12:40:10.340295Z","iopub.execute_input":"2025-11-16T12:40:10.340901Z","iopub.status.idle":"2025-11-16T12:40:11.283097Z","shell.execute_reply.started":"2025-11-16T12:40:10.340879Z","shell.execute_reply":"2025-11-16T12:40:11.282324Z"}},"outputs":[],"execution_count":null},{"id":"03ffd809-108f-4dc4-8f46-09c22003c692","cell_type":"code","source":"import nest_asyncio\nnest_asyncio.apply()\n\nimport asyncio\nfrom openai import AsyncOpenAI\n\nasync def test_databricks_endpoint():\n    client = AsyncOpenAI(\n        api_key=DATABRICKS_TOKEN,\n        base_url=DATABRICKS_BASE_URL\n    )\n    \n    try:\n        response = await client.chat.completions.create(\n            model=MODEL_NAME,\n            messages=[{\"role\": \"user\", \"content\": \"Hello\"}],\n            max_tokens=10\n        )\n        print(\"✅ 端点测试成功!\")\n        print(f\"响应内容: {response.choices[0].message.content}\")\n        return True\n    except Exception as e:\n        print(f\"❌ 端点测试失败: {e}\")\n        return False\n\n# 现在可以正常运行\nresult = asyncio.run(test_databricks_endpoint())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T12:20:11.525566Z","iopub.execute_input":"2025-11-16T12:20:11.526228Z","iopub.status.idle":"2025-11-16T12:20:12.434755Z","shell.execute_reply.started":"2025-11-16T12:20:11.526207Z","shell.execute_reply":"2025-11-16T12:20:12.434081Z"}},"outputs":[],"execution_count":null},{"id":"cb6b8a99-b20f-47e4-a076-63108800ec09","cell_type":"code","source":"import asyncio\n\n# 检查 GraphRAG 是否成功构建了知识图谱\nasync def check_knowledge_graph():\n    \"\"\"检查知识图谱状态\"\"\"\n    try:\n        # 尝试查询一些基本概念\n        test_queries = [\n            \"What is a 10-K document?\",\n            \"What is the difference between 10-K and 10-Q?\",\n            \"What information does DEF14A contain?\"\n        ]\n        \n        for query in test_queries:\n            result = grag.query(query)\n            print(f\"查询: {query}\")\n            print(f\"响应: {result.response[:100]}...\")\n            print(\"-\" * 50)\n            \n    except Exception as e:\n        print(f\"知识图谱检查失败: {e}\")\n\n# 正确运行异步函数\nawait check_knowledge_graph()  # 在 Jupyter Notebook 中使用 await","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T12:42:41.849882Z","iopub.execute_input":"2025-11-16T12:42:41.85016Z","iopub.status.idle":"2025-11-16T12:42:43.809547Z","shell.execute_reply.started":"2025-11-16T12:42:41.85014Z","shell.execute_reply":"2025-11-16T12:42:43.808764Z"}},"outputs":[],"execution_count":null},{"id":"e418a7d7-00f2-410d-a4b2-53980f1d3b74","cell_type":"code","source":"# 首先安装 nest_asyncio\n# !pip install nest_asyncio\n\nimport nest_asyncio\nnest_asyncio.apply()  # 允许嵌套事件循环\n\n# 提取问题\nquestion =  \"How has Apple's smartphone market share changed based on the latest earnings release?\"\n\n# 向GraphRAG提问\nquery_text = f\"For the question: '{question}', rank these document types by relevance: DEF14A, 10-K, 10-Q, 8-K, Earnings. Provide only the ranking as a list of indices in the format [most_relevant, ..., least_relevant].\"\nresult = grag.query(query_text)\nprint(result.response)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T12:26:26.503171Z","iopub.execute_input":"2025-11-16T12:26:26.503743Z","iopub.status.idle":"2025-11-16T12:26:27.423552Z","shell.execute_reply.started":"2025-11-16T12:26:26.50372Z","shell.execute_reply":"2025-11-16T12:26:27.422909Z"}},"outputs":[],"execution_count":null},{"id":"6b10ad37-6d8f-41cc-95e5-be46f749ebe3","cell_type":"code","source":"# 处理JSONL文件中的每个查询\ndef process_jsonl_file(jsonl_path):\n    results = []\n    \n    with open(jsonl_path, 'r') as f:\n        for line in f:\n            data = json.loads(line)\n            uuid = data['uuid']\n            user_content = data['messages'][0]['content']\n            \n            # 提取问题\n            question_start = user_content.find(\"Question:\") + len(\"Question:\")\n            question_end = user_content.find(\"Document Types to rank:\")\n            question = user_content[question_start:question_end].strip()\n            \n            # 向GraphRAG提问\n            query_text = f\"For the question: '{question}', rank these document types by relevance: DEF14A, 10-K, 10-Q, 8-K, Earnings. Provide only the ranking as a list of indices in the format [most_relevant, ..., least_relevant].\"\n            \n            try:\n                # 使用异步查询\n                async def async_query():\n                    result = await grag.async_query(query_text)\n                    return result\n                \n                # 运行异步查询\n                try:\n                    result = asyncio.run(async_query())\n                except RuntimeError as e:\n                    loop = asyncio.get_event_loop()\n                    if loop.is_running():\n                        # 如果事件循环已经在运行，我们需要使用不同的方法\n                        # 在这种情况下，我们使用同步查询作为备选\n                        result = grag.query(query_text)\n                    else:\n                        loop.run_until_complete(async_query())\n                \n                response = result.response\n                \n                # 从响应中提取排名列表\n                ranking = extract_ranking_from_response(response)\n                \n                results.append({\n                    \"uuid\": uuid,\n                    \"question\": question,\n                    \"ranking\": ranking,\n                    \"full_response\": response\n                })\n                \n                print(f\"Processed {uuid}: {ranking}\")\n                \n            except Exception as e:\n                print(f\"Error processing {uuid}: {e}\")\n                results.append({\n                    \"uuid\": uuid,\n                    \"error\": str(e)\n                })\n    \n    return results\n\n# 从响应中提取排名列表\ndef extract_ranking_from_response(response):\n    # 尝试找到列表格式的响应\n    import re\n    list_match = re.search(r'\\[[0-9,\\s]+\\]', response)\n    if list_match:\n        ranking_str = list_match.group()\n        try:\n            # 转换为列表\n            ranking = eval(ranking_str)\n            # 确保是有效的排名（包含0-4的所有数字）\n            if (isinstance(ranking, list) and \n                len(ranking) == 5 and \n                set(ranking) == {0, 1, 2, 3, 4}):\n                return ranking\n        except:\n            pass\n    \n    # 如果无法提取有效排名，使用默认排名\n    return [4, 2, 1, 0, 3]  # 默认排名\n\n# 保存结果\ndef save_results(results, output_path):\n    with open(output_path, 'w') as f:\n        for result in results:\n            f.write(json.dumps(result) + '\\n')\n\n# 主执行流程\nif __name__ == \"__main__\":\n    # 处理JSONL文件\n    jsonl_path = \"/kaggle/input/your-data/ranking_questions.jsonl\"  # 替换为您的文件路径\n    results = process_jsonl_file(jsonl_path)\n    \n    # 保存结果\n    output_path = f\"/kaggle/working/ranking_results_{timestamp}.jsonl\"\n    save_results(results, output_path)\n    \n    print(f\"Processing complete. Results saved to {output_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T09:02:46.414872Z","iopub.execute_input":"2025-11-16T09:02:46.415318Z","iopub.status.idle":"2025-11-16T09:02:46.441485Z","shell.execute_reply.started":"2025-11-16T09:02:46.415281Z","shell.execute_reply":"2025-11-16T09:02:46.439Z"}},"outputs":[],"execution_count":null}]}