Files
group_xinghuo_jinrong/tests/test_build_kb.py
T

135 lines
4.5 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""T21-3 build_kb 脚本单测:front-matter 解析 / 节切块 / Milvus 行构造(纯函数)。
脚本目录非包,sys.path 动态导入(同 test_demo_scripts 模式)。
不连 Ollama / Milvus / MySQL(真库联调归收尾验证步骤)。
"""
from __future__ import annotations
import sys
from pathlib import Path
import pytest
KB_DIR = Path(__file__).resolve().parents[1] / "scripts" / "kb"
sys.path.insert(0, str(KB_DIR))
from build_kb import ( # noqa: E402
DOC_TYPES,
build_chunk_rows,
parse_front_matter,
parse_kb_file,
parse_sections,
)
PRODUCT = {"product_id": "PROD-X", "product_name": "测试产品", "min_risk_code": "R2"}
SAMPLE = """---
product_id: PROD-X
version: 2026.09
effective_date: 2026-09-01
---
# 测试产品 · 产品手册
## 【prospectus】产品概况
定位与投资范围说明。
## 【fee】费率结构
管理费 0.20%/年,托管费 0.10%/年。
## 【rule】申赎与交易规则
申购 T+1 确认。
## 【risk】风险揭示
不保本不保收益。
"""
class TestFrontMatter:
def test_parse_meta_and_body_split(self):
meta, body = parse_front_matter(SAMPLE)
assert meta == {
"product_id": "PROD-X",
"version": "2026.09",
"effective_date": "2026-09-01",
}
assert body.startswith("# 测试产品")
def test_missing_front_matter_raises(self):
with pytest.raises(ValueError, match="front-matter"):
parse_front_matter("# 无 front-matter 正文")
class TestSections:
def test_four_sections_with_types(self):
sections = parse_sections(SAMPLE)
assert [s["doc_type"] for s in sections] == list(DOC_TYPES)
assert sections[0]["title"] == "产品概况"
assert "投资范围" in sections[0]["text"]
def test_section_text_excludes_heading_and_blank_lines(self):
sections = parse_sections(SAMPLE)
# 正文不含标题行、不含空行累计
assert "【fee】" not in sections[1]["text"]
assert sections[1]["text"].count("\n\n") == 0
def test_illegal_doc_type_raises(self):
bad = SAMPLE.replace("【fee】", "【rates】")
with pytest.raises(ValueError, match="doc_type 非法"):
parse_sections(bad)
def test_empty_body_sections_dropped(self):
body = "## 【fee】费率\n\n## 【risk】风险\n有内容"
sections = parse_sections(body)
assert len(sections) == 1
assert sections[0]["doc_type"] == "risk"
class TestChunkRows:
def test_row_fields_and_chunk_no(self):
meta = {"product_id": "PROD-X", "version": "2026.09", "effective_date": "2026-09-01"}
sections = parse_sections(SAMPLE)
rows = build_chunk_rows(meta, sections, PRODUCT)
assert [r["id"] for r in rows] == ["PROD-X_0", "PROD-X_1", "PROD-X_2", "PROD-X_3"]
first = rows[0]
# 溯源字段必填(03-milvus-collections.md §2.3)
assert first["source_doc_id"] == "KB-PROD-X"
assert first["source_version"] == "2026.09"
assert first["effective_date"] == "2026-09-01"
# product_name / risk_level 取 Core 权威(不入库文件自述值)
assert first["product_name"] == "测试产品"
assert first["risk_level"] == "R2"
assert first["chunk_no"] == 0
def test_missing_meta_field_raises(self):
meta = {"product_id": "PROD-X", "version": "2026.09"} # 缺 effective_date
with pytest.raises(ValueError, match="缺少字段"):
build_chunk_rows(meta, [{"doc_type": "fee", "title": "t", "text": "x"}], PRODUCT)
def test_chunk_text_truncated(self):
from build_kb import CHUNK_TEXT_MAX_CHARS
meta = {"product_id": "PROD-X", "version": "v", "effective_date": "2026-09-01"}
long_text = "长" * (CHUNK_TEXT_MAX_CHARS + 100)
rows = build_chunk_rows(meta, [{"doc_type": "fee", "title": "t", "text": long_text}], PRODUCT)
assert len(rows[0]["chunk_text"]) == CHUNK_TEXT_MAX_CHARS
class TestRealFiles:
"""真手册文件抽查:data/kb 全部文件可解析且 4 节齐。"""
def test_all_kb_files_parse(self):
kb_root = Path(__file__).resolve().parents[1] / "data" / "kb"
files = sorted(kb_root.glob("*.md"))
assert len(files) == 6
for f in files:
meta, sections = parse_kb_file(f)
assert meta["product_id"] == f.stem
assert [s["doc_type"] for s in sections] == ["prospectus", "fee", "rule", "risk"]
assert all(len(s["text"]) > 20 for s in sections)