Files
iAOP/web/chat/scripts/build_citations.py
T

81 lines
3.0 KiB
Python
Raw Normal View History

# -*- coding: utf-8 -*-
"""对话助手 RAG 引用语料编译脚本(issue #133 / PRD 5.4 引用溯源)。
扫描行业模板知识库文档(``templates/*/rag-kb/documents/*.md``),抽取
「文档标题 + 章节 + 章节摘要」编译为 ``web/chat/citations.json``,
供增强版对话页(assistant.html)做前端检索与**引用溯源展示**。
与 rag-kb 内核的关系:``core/rag-kb`` 是完整检索管线(暂无 HTTP 服务),
本脚本只把文档资产转成前端可 fetch 的静态语料(Demo 级),
引用条目带「文档名 + 章节」,对齐 PRD 5.4「强制引用溯源」的展示语义。
用法(仓库根目录下):
python web/chat/scripts/build_citations.py
"""
from __future__ import annotations
import glob
import json
import os
import re
import sys
_REPO_ROOT = os.path.dirname(
os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
)
_DOC_GLOB = os.path.join(_REPO_ROOT, "templates", "*", "rag-kb", "documents", "*.md")
_OUT_PATH = os.path.join(_REPO_ROOT, "web", "chat", "citations.json")
#: 每个章节摘要的最大字符数(控制语料体积,前端检索只看要点)
SNIPPET_MAX_CHARS = 160
def extract_sections(path: str) -> dict:
"""从一份 Markdown 文档抽取标题与章节摘要。"""
with open(path, encoding="utf-8") as f:
text = f.read()
rel = os.path.relpath(path, _REPO_ROOT).replace(os.sep, "/")
title_match = re.search(r"^#\s+(.+)$", text, flags=re.M)
title = title_match.group(1).strip() if title_match else os.path.basename(path)
sections = []
# 按 ## 章节切分;无章节时整篇作为一节
parts = re.split(r"^##\s+(.+)$", text, flags=re.M)
if len(parts) >= 3:
for i in range(1, len(parts), 2):
heading = parts[i].strip()
body = parts[i + 1] if i + 1 < len(parts) else ""
snippet = re.sub(r"\s+", " ", body).strip()[:SNIPPET_MAX_CHARS]
if snippet:
sections.append({"heading": heading, "snippet": snippet})
else:
snippet = re.sub(r"\s+", " ", text).strip()[:SNIPPET_MAX_CHARS]
if snippet:
sections.append({"heading": "全文", "snippet": snippet})
return {"title": title, "source": rel, "sections": sections}
def main() -> int:
docs = []
for path in sorted(glob.glob(_DOC_GLOB)):
try:
docs.append(extract_sections(path))
print(f"[OK] {os.path.basename(path)} "
f"({len(docs[-1]['sections'])} 章节)")
except Exception as exc: # noqa: BLE001 - 聚合全部失败
print(f"[FAIL] {path}: {exc}")
return 1
if not docs:
print("未找到任何知识库文档")
return 1
with open(_OUT_PATH, "w", encoding="utf-8") as f:
json.dump({"schema": "iAOP-chat-citations-v1", "documents": docs},
f, ensure_ascii=False, indent=2)
print(f"共 {len(docs)} 份文档 -> {os.path.relpath(_OUT_PATH, _REPO_ROOT)}")
return 0
if __name__ == "__main__":
sys.exit(main())