Files
group_fqcd_jr/tools/build_knowledge_chunks.py
T
张胜宇 e239eb778b docs: 品牌全量口径统一为「南方基金」+ 作废文档清理
1) 客服 Agent 四份交付文档 + 构建脚手架:品牌由包装占位 XX科技 / 旧名 南方财富
   统一为南方基金(热线 400-889-8899 / 官网 nffund.com),系统名改为「智能服务系统」;
   同步追加 §0.4 修订记录行,工程记录行保留原占位字面以支撑硬编码扫描验收。
2) 开发文档:清理 28 份已作废/残留文档(14 份移出归档 + 14 份仓库副本),
   新增《文档规整方案与开发前待决事项-2026-09-17》。
3) 客服agent 四份交付文档首次纳入本分支。
2026-09-17 15:15:22 +08:00

326 lines
16 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""把 knowledge/ 下的文档切成可检索知识块,输出 JSONL(临时脚本,跑完即删)。
切片粒度决定检索质量。这里采用**叶子标题**策略:一个标题若其后没有更深的标题,
它就是一个切分点。相比"按固定级别切",这个策略同时满足了三种真实情况:
- 有的条款带 `#### X.Y` 子条款(如反洗钱第九条 9.1–9.4)→ 按子条款切,粒度更细;
- 有的条款没有子条款(如反洗钱第十一条)→ 自己就是叶子,单独成块,不会被并进上一块;
- 有的章没有小节(如企业信息「一、公司基本信息」)→ 章本身就是叶子,内容不会丢。
标题路径保留完整上级链;若切分点本身不是「第X条」(例如反洗钱第十三条下的
`### 第一类:资金流转异常`),会把最近的条款名补进路径,避免块失去归属。
纯「目录」块直接丢弃。
"""
import json
import re
from pathlib import Path
HEADING = re.compile(r"^(#{1,6})\s+(.+?)\s*$")
CLAUSE = re.compile(r"^第[一二三四五六七八九十百]+条")
# 每个文件的入库配置:集合、编号前缀、可见性、版本与生效日期(取自文档头部)
SOURCES: dict[str, dict[str, str]] = {
"policy/个人投资者适当性管理指南.md": {
"collection": "fin_policy_collection", "prefix": "POL-AST", "visibility": "public",
"version": "V3.2", "effective_date": "2024-01-15", "doc_no": "JR-AST-2024-001",
"tags": "适当性,C1-C5,R1-R5,双录,冷静期",
},
# 反洗钱合规操作手册**不入客服知识库**:本业务只做公募基金,不涉及银行转账与资金划付,
# 反洗钱属后续风控模块职责;且该手册标注「内部机密」、第十六条禁止向客户透露可疑交易
# 信息,混入面向客户的知识库存在制度性冲突。
"policy/理财产品销售管理办法.md": {
"collection": "fin_policy_collection", "prefix": "POL-SPM", "visibility": "public",
"version": "V3.0", "effective_date": "2024-02-01", "doc_no": "JR-SPM-2024-003",
"tags": "理财产品销售,双录,冷静期,费率,投诉",
},
"product/个人理财产品手册.md": {
"collection": "fin_product_collection", "prefix": "PROD", "visibility": "public",
"version": "V2.8", "effective_date": "", "doc_no": "",
"tags": "基金,银行理财,保险,费率,申赎",
},
# 只取「客户分层标准」与「各层级专属权益」两章:家族信托、资产配置流程、客户经理
# 考核指标、隐私应急预案属内部管理内容,客户咨询用不到,不入库。
"product/高净值客户服务规范.md": {
"collection": "fin_product_collection", "prefix": "HNW", "visibility": "public",
"version": "V2.1", "effective_date": "", "doc_no": "",
"tags": "高净值,VIP分级,层级权益,费率优惠",
"allow_chapters": ["一、", "二、"],
},
"company/企业信息.md": {
"collection": "fin_faq_collection", "prefix": "COMP", "visibility": "public",
"version": "", "effective_date": "", "doc_no": "",
"tags": "公司信息,金融牌照,资质",
},
}
def leaf_split_points(marks: list[tuple[int, int, str]]) -> set[int]:
"""叶子标题的行号集合:其后没有更深标题的标题。"""
points: set[int] = set()
for position, (index, level, _title) in enumerate(marks):
following = marks[position + 1] if position + 1 < len(marks) else None
if following is not None and following[1] > level:
continue # 有子标题,不是叶子
points.add(index)
return points
def chunk_markdown(text: str) -> list[dict[str, str]]:
lines = text.splitlines()
marks: list[tuple[int, int, str]] = []
for index, line in enumerate(lines):
match = HEADING.match(line)
if match:
marks.append((index, len(match.group(1)), match.group(2).strip()))
marks_by_line = {index: (level, title) for index, level, title in marks}
split_points = leaf_split_points(marks)
stack: dict[int, str] = {}
chunks: list[dict[str, str]] = []
buffer: list[str] = []
meta: dict[str, str] | None = None
last_clause = ""
def has_body() -> bool:
return any(not HEADING.match(line) and line.strip() for line in buffer)
def flush() -> None:
nonlocal buffer, meta
if meta is not None and has_body():
body = "\n".join(buffer).strip()
if body:
chunks.append({**meta, "content": body})
buffer = []
for index, line in enumerate(lines):
mark = marks_by_line.get(index)
if mark is not None:
level, title = mark
stack[level] = title
for deeper in [key for key in stack if key > level]:
del stack[deeper]
if CLAUSE.match(title):
last_clause = title
if index in split_points:
flush()
# 所属章取「最近的上级标题」,而不是最外层文档标题(sorted 后取首个会拿到 h1)
ancestors = [key for key in stack if key < level]
chapter = stack[max(ancestors)] if ancestors else ""
path = [stack[key] for key in sorted(stack) if key <= level]
if not CLAUSE.match(title) and last_clause and last_clause not in path:
path = [item for item in (chapter, last_clause) if item] + [title]
meta = {"title": " · ".join(path), "chapter": chapter, "section": title}
buffer.append(line)
flush()
return [chunk for chunk in chunks if chunk["section"].strip() != "目录"]
def chunk_qa(text: str) -> list[dict[str, str]]:
chunks: list[dict[str, str]] = []
for line in text.splitlines():
line = line.strip()
if not line or "\t" not in line:
continue
question, _, answer = line.partition("\t")
chunks.append({
"title": question.strip(),
"chapter": "高频问答",
"section": question.strip(),
"content": f"问:{question.strip()}\n答:{answer.strip()}",
})
return chunks
TABLE_ROW = re.compile(r"^\|(.+)\|\s*$")
LEADING_NUMBER = re.compile(r"^\d+(?:\.\d+)*\s*")
def expand_table_rows(chunk: dict[str, str], parent_id: str) -> list[dict[str, object]]:
"""把 Markdown 表格的每一行拆成自解释的小块(父块照旧保留)。
为什么需要:现在的粒度是"一个叶子标题 = 一块",产品手册里就是**整个产品小节**
(表格 + 说明)成一块。于是客户问「起投多少」和问「风险高吗」命中同一块、拿到
**完全相同**的整节内容——客户会觉得客服没听懂问题,只是把说明书重贴一遍。
顺带地,整节几百字的向量是"整节的混合语义",与"起投多少"这种具体小问题相似度
天然偏低(实测该问句向量 top1 只有 0.6291,够不到 0.75 门槛)。
小块必须**自解释**:只回「1万元」客户不知道说的是哪个产品,所以带上产品名与行标签。
父块保留,客户问「这个产品怎么样」时仍要能拿到完整一节。
## 2026-09-15 修掉的 bug:表头被当成数据行
原实现用「**第一个非分隔行**」当表头(`if not header: header = cells`),而 `header`
从不重置。于是**同一节里出现第二张表格时,它的表头行被当成数据行**,产出形如
「第九条 问卷内容及评分标准:选项 分值」的**零信息量碎片**:
《个人投资者适当性管理指南》第九条下有 16 张问卷表格 → 15 条碎片,且**正文逐字相同**。
检索时它们必然互相打平(实测把「风险评估问卷怎么评分」的 top1/次优差压到 **0.002**,
客服按"中置信需领先 ≥0.07"判并列 → 转人工),把真正有内容的块挤到第 5 名。
修法:markdown 表格的表头**只可能是紧邻分隔行 `|---|---|` 之前的那一行**,所以按分隔行
认表头,用 `prev` 延迟一行判断。修后块数 636 → 617,正文完全相同的组从 1 组 15 块降到 0。
"""
blocks: list[dict[str, object]] = []
name = LEADING_NUMBER.sub("", chunk["section"]).strip() or chunk["section"]
def emit(cells: list[str]) -> None:
if len(cells) < 2:
return
label, value = cells[0], cells[1]
if not label or not value:
return
blocks.append({
"title": f"{chunk['title']} · {label}",
"chapter": chunk["chapter"],
"section": f"{name} · {label}",
"content": f"{name}:{label} {value}",
"parent_id": parent_id,
})
prev: list[str] | None = None
for line in chunk["content"].splitlines():
match = TABLE_ROW.match(line.strip())
if not match:
continue
cells = [cell.strip() for cell in match.group(1).split("|")]
if all(set(cell) <= {"-", ":", " "} for cell in cells):
# 分隔行 |---|:紧邻它之前的那一行(`prev`)是**表头**,不能当数据行 → 丢弃。
prev = None
continue
if prev is not None:
emit(prev) # 没被分隔行认领为表头的行 = 数据行
prev = cells
if prev is not None:
emit(prev) # 收尾:最后一行也要处理(没有分隔行收尾的表格)
return blocks
def assert_no_duplicate_contents(records: list[dict[str, object]]) -> None:
"""自带守卫:**正文完全相同的块必须为 0**,否则直接失败退出、不生成 jsonl。
为什么用这一条当守卫:表头被误当数据行时的直接后果就是"**多张表格产出逐字相同的块**"
(第九条那 16 张问卷表 → 15 条一模一样的「…:选项 分值」)。这类块在检索里必然互相
打平,把 top1/次优差压到 0.07 门槛之下(实测 0.002)→ 客服判并列转人工。
本脚本是**一次性灌库脚本**、没有单测覆盖(这正是该 bug 活下来的原因),
所以把守卫放在脚本自己的执行路径上:**每次重灌都会跑一遍**。
为什么不是"块长度下限":短块本身是设计的一部分(「评审标准:管理人资质 15%」13 字,
但它是真实的数据行、是有效答案)。**内容逐字重复**才是缺陷特征,长度不是。
"""
seen: dict[str, list[str]] = {}
for record in records:
seen.setdefault(str(record["content"]), []).append(str(record["doc_id"]))
duplicated = {content: ids for content, ids in seen.items() if len(ids) > 1}
if duplicated:
detail = "\n".join(
f" {len(ids)} 份:{content[:60]!r} → {ids[:6]}"
for content, ids in list(duplicated.items())[:5]
)
raise SystemExit(
f"知识块自检失败:有 {len(duplicated)} 组正文完全相同的块。\n"
"这类块在检索里必然互相打平(把 top1/次优差压到 0.07 之下 → 客服转人工),"
"通常是**表格表头被当成了数据行**(见 `expand_table_rows` 的 docstring)。\n"
f"{detail}\n已中止,未写入 jsonl。"
)
records: list[dict[str, object]] = []
for relative, config in SOURCES.items():
text = (Path("knowledge") / relative).read_text(encoding="utf-8")
chunks = chunk_markdown(text)
# 章节白名单:只保留指定章下的块(用于剔除内部管理章节,如高净值规范只留分级与权益)
allowed = config.get("allow_chapters")
if isinstance(allowed, list):
chunks = [
chunk for chunk in chunks
if any(chunk["chapter"].startswith(prefix) for prefix in allowed)
]
for order, chunk in enumerate(chunks, 1):
parent_id = f"{config['prefix']}-{order:03d}"
records.append({
"doc_id": parent_id,
"collection": config["collection"],
"title": chunk["title"],
"content": chunk["content"],
"chapter": chunk["chapter"],
"section": chunk["section"],
"tags": config["tags"],
"doc_no": config["doc_no"],
"version": config["version"],
"effective_date": config["effective_date"],
"expire_date": "", "source_url": "", "reviewer": "",
"source_file": relative,
"visibility": config["visibility"],
"chars": len(chunk["content"]),
})
# 行级子块:挂在父块 doc_id 下(PROD-007-01 这种),父块编号不受新增子块影响,
# 因此反复重跑本脚本得到的 doc_id 是稳定的。
for row_order, block in enumerate(expand_table_rows(chunk, parent_id), 1):
records.append({
"doc_id": f"{parent_id}-{row_order:02d}",
"collection": config["collection"],
"title": block["title"],
"content": block["content"],
"chapter": block["chapter"],
"section": block["section"],
"tags": config["tags"],
"doc_no": config["doc_no"],
"version": config["version"],
"effective_date": config["effective_date"],
"expire_date": "", "source_url": "", "reviewer": "",
"source_file": relative,
"visibility": config["visibility"],
"chars": len(str(block["content"])),
})
for order, chunk in enumerate(
chunk_qa((Path("knowledge") / "faq/高频问答对.txt").read_text(encoding="utf-8")), 1
):
records.append({
"doc_id": f"FAQ-{order:04d}",
"collection": "fin_faq_collection",
"title": chunk["title"], "content": chunk["content"],
"chapter": chunk["chapter"], "section": chunk["section"],
"tags": "高频问答,FAQ",
"doc_no": "", "version": "", "effective_date": "", "expire_date": "",
"source_url": "", "reviewer": "",
"source_file": "faq/高频问答对.txt", "visibility": "public",
"chars": len(chunk["content"]),
})
assert_no_duplicate_contents(records)
(Path("knowledge") / "_chunks.jsonl").write_text(
"\n".join(json.dumps(record, ensure_ascii=False) for record in records), encoding="utf-8"
)
lines: list[str] = [f"总块数:{len(records)}\n"]
by_collection: dict[str, list[dict[str, object]]] = {}
for record in records:
by_collection.setdefault(str(record["collection"]), []).append(record)
for name, group in sorted(by_collection.items()):
sizes = sorted(int(record["chars"]) for record in group)
lines.append(
f"{name}: {len(group)} 块,字符数 最小 {sizes[0]} / 中位 {sizes[len(sizes)//2]} / 最大 {sizes[-1]}"
)
lines.append("\n各文件块数:")
by_file: dict[str, int] = {}
for record in records:
by_file[str(record["source_file"])] = by_file.get(str(record["source_file"]), 0) + 1
for name, count in sorted(by_file.items()):
lines.append(f" {name}: {count}")
over = [record for record in records if int(record["chars"]) > 1200]
lines.append(f"\n超过 1200 字符的块:{len(over)} 个")
for record in over[:10]:
lines.append(f" {record['doc_id']} {record['chars']} 字符 {str(record['title'])[:64]}")
lines.append("\n反洗钱手册的块标题(核对「第一类」归属是否带上了第十三条):")
for record in records:
if str(record["doc_id"]).startswith("POL-AML"):
lines.append(f" {record['doc_id']} {int(record['chars']):>5} 字符 {str(record['title'])[:70]}")
Path("_chunks_report.txt").write_text("\n".join(lines), encoding="utf-8")
print(f"已生成 {len(records)} 块 → knowledge/_chunks.jsonl;报告见 _chunks_report.txt")