wip: 客服Agent + RAG + 画像收尾(基于 6516ccb)

This commit is contained in:
2026-09-11 14:46:40 +08:00
parent 870fd0d44e
commit e4c4099aaa
82 changed files with 13901 additions and 2426 deletions
+104
View File
@@ -0,0 +1,104 @@
"""解析器契约测试。
除 brief 给定的 7 条契约外,另有一条**对真实源文件**的计数测试:`RAG-` 前缀
只覆盖 62/105 条,漏掉全部 43 条 `NF-*`(真实文件里共 12 种前缀)。真实源文件是
工作区外的交付物,缺失时 skip(不算失败),存在时按期望值强校验。
"""
from __future__ import annotations
import hashlib
from dataclasses import FrozenInstanceError
from pathlib import Path
import pytest
from tools.qa_source_parser import (
EXPECTED_RECORD_COUNT,
QaRecord,
normalize_phrase,
parse_qa_source,
phrase_hash,
)
SAMPLE = """
[RAG-PER-001]
问题:你是谁?
相似问法:你是人工吗?|你是真人客服吗?
回答:我是奶龙基金智能助手。
[NF-SVC-001]
问题:怎么联系你们?
相似问法:客服入口在哪?
回答:您可通过官网获取服务。
"""
#: 已审核源文件的实际位置(工作区外)。用于验证"11 种前缀"这一前提是否成立。
REAL_SOURCE = Path(
r"C:\Users\Windows\Desktop\DSH开发智能助手\胜宇前期资料"
r"\客服Agent知识库_QA问答对_v5_RAG发布候选版.txt"
)
def test_expected_count_is_105() -> None:
assert EXPECTED_RECORD_COUNT == 105
def test_parser_accepts_both_id_prefixes() -> None:
records = parse_qa_source(SAMPLE, expected_count=None)
assert [r.qa_id for r in records] == ["RAG-PER-001", "NF-SVC-001"]
def test_parser_splits_synonyms_on_pipe() -> None:
records = parse_qa_source(SAMPLE, expected_count=None)
assert records[0].synonyms == ("你是人工吗?", "你是真人客服吗?")
def test_parser_keeps_answer_verbatim() -> None:
records = parse_qa_source(SAMPLE, expected_count=None)
assert records[0].answer == "我是奶龙基金智能助手。"
def test_parser_rejects_record_without_synonyms() -> None:
with pytest.raises(ValueError, match="相似问法"):
parse_qa_source("[RAG-PER-001]\n问题:问?\n回答:答。\n", expected_count=None)
def test_parser_rejects_wrong_record_count_by_default() -> None:
with pytest.raises(ValueError, match="记录数不符"):
parse_qa_source(SAMPLE)
def test_phrase_hash_is_sha256_of_normalized_phrase() -> None:
value = "你是人工吗?"
assert phrase_hash(value) == hashlib.sha256(
normalize_phrase(value).encode("utf-8")).hexdigest()
assert len(phrase_hash(value)) == 64
def test_record_is_frozen_dataclass() -> None:
record = parse_qa_source(SAMPLE, expected_count=None)[0]
assert isinstance(record, QaRecord)
assert isinstance(record.synonyms, tuple)
with pytest.raises(FrozenInstanceError):
record.question = "改不了" # type: ignore[misc]
def test_parser_rejects_duplicate_synonym_within_record_after_normalization() -> None:
"""`uk_faq_synonym (knowledge_id, phrase_hash)` 只认归一化后的哈希。"""
text = "[RAG-PER-001]\n问题:你是谁?\n相似问法:你是谁?| 你是谁? |你 是 谁?\n回答:答。\n"
synonyms = parse_qa_source(text, expected_count=None)[0].synonyms
assert len({phrase_hash(item) for item in synonyms}) == len(synonyms)
@pytest.mark.skipif(not REAL_SOURCE.exists(), reason="已审核 QA 源文件不在本机")
def test_parser_counts_real_source_file() -> None:
records = parse_qa_source(REAL_SOURCE.read_text(encoding="utf-8"))
assert len(records) == EXPECTED_RECORD_COUNT
rag = [r for r in records if r.qa_id.startswith("RAG-")]
nf = [r for r in records if r.qa_id.startswith("NF-")]
assert (len(rag), len(nf)) == (62, 43)
assert all(r.question and r.answer and r.synonyms for r in records)
assert sum(len(r.synonyms) for r in records) == 421
# 前缀多于"RAG-"/"NF-"两类:只按 RAG- 解析会静默丢掉 NF-* 全部记录。
assert len([r for r in records if not r.qa_id.startswith(("RAG-", "NF-"))]) == 0