feat: 完成 issue #133 对话助手增强版(引用溯源/路由分级/DLP 拦截,直连推理服务)
This commit is contained in:
@@ -0,0 +1,80 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""对话助手 RAG 引用语料编译脚本(issue #133 / PRD 5.4 引用溯源)。
|
||||
|
||||
扫描行业模板知识库文档(``templates/*/rag-kb/documents/*.md``),抽取
|
||||
「文档标题 + 章节 + 章节摘要」编译为 ``web/chat/citations.json``,
|
||||
供增强版对话页(assistant.html)做前端检索与**引用溯源展示**。
|
||||
|
||||
与 rag-kb 内核的关系:``core/rag-kb`` 是完整检索管线(暂无 HTTP 服务),
|
||||
本脚本只把文档资产转成前端可 fetch 的静态语料(Demo 级),
|
||||
引用条目带「文档名 + 章节」,对齐 PRD 5.4「强制引用溯源」的展示语义。
|
||||
|
||||
用法(仓库根目录下):
|
||||
python web/chat/scripts/build_citations.py
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import glob
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
_REPO_ROOT = os.path.dirname(
|
||||
os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
)
|
||||
_DOC_GLOB = os.path.join(_REPO_ROOT, "templates", "*", "rag-kb", "documents", "*.md")
|
||||
_OUT_PATH = os.path.join(_REPO_ROOT, "web", "chat", "citations.json")
|
||||
|
||||
#: 每个章节摘要的最大字符数(控制语料体积,前端检索只看要点)
|
||||
SNIPPET_MAX_CHARS = 160
|
||||
|
||||
|
||||
def extract_sections(path: str) -> dict:
|
||||
"""从一份 Markdown 文档抽取标题与章节摘要。"""
|
||||
with open(path, encoding="utf-8") as f:
|
||||
text = f.read()
|
||||
rel = os.path.relpath(path, _REPO_ROOT).replace(os.sep, "/")
|
||||
title_match = re.search(r"^#\s+(.+)$", text, flags=re.M)
|
||||
title = title_match.group(1).strip() if title_match else os.path.basename(path)
|
||||
|
||||
sections = []
|
||||
# 按 ## 章节切分;无章节时整篇作为一节
|
||||
parts = re.split(r"^##\s+(.+)$", text, flags=re.M)
|
||||
if len(parts) >= 3:
|
||||
for i in range(1, len(parts), 2):
|
||||
heading = parts[i].strip()
|
||||
body = parts[i + 1] if i + 1 < len(parts) else ""
|
||||
snippet = re.sub(r"\s+", " ", body).strip()[:SNIPPET_MAX_CHARS]
|
||||
if snippet:
|
||||
sections.append({"heading": heading, "snippet": snippet})
|
||||
else:
|
||||
snippet = re.sub(r"\s+", " ", text).strip()[:SNIPPET_MAX_CHARS]
|
||||
if snippet:
|
||||
sections.append({"heading": "全文", "snippet": snippet})
|
||||
|
||||
return {"title": title, "source": rel, "sections": sections}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
docs = []
|
||||
for path in sorted(glob.glob(_DOC_GLOB)):
|
||||
try:
|
||||
docs.append(extract_sections(path))
|
||||
print(f"[OK] {os.path.basename(path)} "
|
||||
f"({len(docs[-1]['sections'])} 章节)")
|
||||
except Exception as exc: # noqa: BLE001 - 聚合全部失败
|
||||
print(f"[FAIL] {path}: {exc}")
|
||||
return 1
|
||||
if not docs:
|
||||
print("未找到任何知识库文档")
|
||||
return 1
|
||||
with open(_OUT_PATH, "w", encoding="utf-8") as f:
|
||||
json.dump({"schema": "iAOP-chat-citations-v1", "documents": docs},
|
||||
f, ensure_ascii=False, indent=2)
|
||||
print(f"共 {len(docs)} 份文档 -> {os.path.relpath(_OUT_PATH, _REPO_ROOT)}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user