Files
group_fqcd_jr/tests/unit/service/test_customer_service_agent.py
T
张胜宇 35109af96f docs(W25): D2.9 手动测试用例按实测重建 + 展示层误删正文真缺陷修复
- D2.9 v1.3→v1.4:46 条金标追加「2026-09-21 实测(出口 · top1)」列与逐条答复原文,
  §3 边界 11 条 / §4 安全 4 条 / §1.3 / §5 指标表实测列刷新,新增 §2.10 场内基金演示线(5 条),
  §3.1 缺口由三个扩为四个(新增 ④ A-06),§3.2 整段重写,§8 新增 D-5 与 DEC-W20-8 细化
- D2.5 §4.7 出口口径更正:场内基金第 1 条 E3→E4、第 2 条 E3→E5b(内容逐字正确,出口偏保守)
- 修复 drop_yield_claims() 误删「业绩比较基准」公式行:收益词与百分比间出现
  乘号 / 指数 等公式标记时判为基准公式豁免,两种语序的真实收益数值照删
- 新增回归测试 test_drop_yield_claims_keeps_the_benchmark_formula_but_drops_both_word_orders
- D2.1 v6.39→v6.40 / D1.1 v1.15→v1.16(新增第二十八轮)/ D1.6 新增 §12 / D4.8 v1.2→v1.3(新增 §11)

实测:46 条金标全部 succeeded,HTTP 非 200 = 0;转人工 5 条(均白名单内);
M-1 46/46、M-4 46/46、M-6 5/46、M-2 28/31、M-2b 15/18、M-3 4/4、M-7/M-8/M-9/M-10 = 0;
pytest 1997 passed / 3 skipped;ruff 零新增告警。
2026-09-21 15:27:18 +08:00

2420 lines
115 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""客服 Agent 的分级回退(E5)与转人工白名单守卫。
为什么单独守这两个不变量:
- **白名单外不得转人工**:加档位、加身份之后,"答不上来就转人工"这句话很容易以各种
变体溜回来(新增一个兜底分支、把异常吞掉再转人工)。白名单越界即抛错,
让它在开发期就炸,而不是等验收时才发现转人工率又回去了。
- **E5b 不得置 transfer_required**:知识未命中 / 置信度不足 / 检索降级 / 画像查不到
都属"这次没查到",不是"必须人工办的事"。这一条正是「客服不智能」的修复点。
"""
import ast
from pathlib import Path
import pytest
from app.core.contracts import (
AgentRequest,
AgentRequestMetadata,
ConversationTurn,
CoreResult,
IntentResult,
RequestContext,
)
from app.core.customer_service_rules import ADVICE_BOUNDARY_REPLY
from app.service.agent.implementations import customer_service as customer_service_module
from app.service.agent.implementations.customer_service import CustomerServiceAgent
def build_agent() -> CustomerServiceAgent:
"""直接构造即可:本文件只测纯函数出口,不触发治理与工具调用。"""
return CustomerServiceAgent(CustomerServiceAgent.definition)
def test_agent_definition_stays_visitor_and_customer_only() -> None:
definition = CustomerServiceAgent.definition
assert definition.agent_type == "customer_service"
assert set(definition.allowed_roles) == {"visitor", "customer"}
# 客服不隐式召回长期画像;已登录用户的画像查询必须显式调用受控工具。
assert definition.recalls_customer_memory is False
def test_transfer_exit_accepts_every_whitelisted_reason() -> None:
from app.core.customer_service_rules import TRANSFER_REASONS
for reason in sorted(TRANSFER_REASONS):
result = build_agent()._exit_transfer(reason)
assert result.transfer_required is True
assert result.transfer_reason == reason
@pytest.mark.parametrize("reason", ["置信度不足", "知识库未命中", "agent_requested", ""])
def test_transfer_exit_rejects_reason_outside_whitelist(reason: str) -> None:
with pytest.raises(ValueError):
build_agent()._exit_transfer(reason)
@pytest.mark.parametrize(
("hits", "note"),
[
([], "知识库未命中"),
([], "知识检索降级:milvus_unavailable"),
([], "画像查询失败"),
([{"score": 0.2, "content": "某段弱相关内容"}], "置信度不足:score=0.200 gap=0.010"),
],
)
def test_partial_exit_never_requests_transfer(hits: list, note: str) -> None:
result = build_agent()._exit_partial(hits, note=note)
assert result.transfer_required is False
assert result.transfer_reason is None
assert "400-889-8899" in result.text
def test_partial_exit_shows_content_only_above_floor() -> None:
strong = build_agent()._exit_partial(
[{"score": 0.6, "content": "基金申购费率按金额分档。"}], note="置信度不足"
)
assert "基金申购费率按金额分档。" in strong.text
weak = build_agent()._exit_partial(
[{"score": 0.10, "content": "某段弱相关内容"}], note="置信度不足"
)
assert "某段弱相关内容" not in weak.text
def test_clarify_exit_returns_none_when_candidates_are_too_weak() -> None:
"""不拿噪声去问客户——低于澄清下限就交给 E5b。"""
assert build_agent()._exit_clarify([{"score": 0.10, "title": "某条"}]) is None
def test_clarify_exit_lists_candidates_and_asks_once() -> None:
hits = [
{"score": 0.55, "title": "基金申购费率"},
{"score": 0.50, "title": "基金赎回规则"},
]
result = build_agent()._exit_clarify(hits)
assert result is not None
assert result.clarification_required is True
assert result.transfer_required is False
assert "基金申购费率" in result.text
assert "基金赎回规则" in result.text
def test_clarify_exit_deduplicates_titles() -> None:
hits = [{"score": 0.55, "title": "同一标题"}, {"score": 0.50, "title": "同一标题"}]
result = build_agent()._exit_clarify(hits)
assert result is not None
assert result.text.count("同一标题") == 1
def test_profile_miss_exit_does_not_guess_a_level() -> None:
result = build_agent()._exit_profile_miss()
assert result.transfer_required is False
assert "不能靠猜" in result.text
# ---- 访客侧投资建议护栏(`C-09` 输出侧按主体分化) ----
VISITOR = RequestContext(user_id="visitor:test", trace_id="t", roles=("visitor",))
CUSTOMER = RequestContext(user_id="9001", trace_id="t", roles=("customer",))
#: 回归样本:该文本**原先真的在库**(public 档 `PROD-012`「按客户类型的推荐策略」),
#: 2026-09-18 语料修复已把它下线(证据 `docs/evidence/20260918-t2h-prod012-corpus-fix.json`)。
#: 保留为护栏的回归样本:即便将来又有语料/模型产出这类措辞,访客侧也必须拦得住。
ALLOCATION_ADVICE = "稳健型客户建议配置:货币基金 30% + 纯债基金 50% + 混合基金 20%。"
def test_visitor_answer_with_allocation_advice_is_replaced() -> None:
"""E3 是**原文直返**、不经模型改写,所以知识块自带的推介内容会原样到访客手里。"""
guarded = build_agent()._guard_visitor_advice(CoreResult(text=ALLOCATION_ADVICE), VISITOR)
assert guarded.text == ADVICE_BOUNDARY_REPLY
# 内容被换掉不是「必须人来办的事」,且访客没有工单承接方。
assert guarded.transfer_required is False
assert guarded.transfer_reason is None
def test_customer_answer_is_not_subject_to_the_visitor_rule() -> None:
"""客户侧不套用访客规则(两侧红线不同)—— 同一个答复对客户必须原样保留。"""
guarded = build_agent()._guard_visitor_advice(CoreResult(text=ALLOCATION_ADVICE), CUSTOMER)
assert guarded.text == ALLOCATION_ADVICE
def test_visitor_answer_without_advice_wording_is_untouched() -> None:
text = "货币基金风险等级 R1,1 元起投,赎回 T+1 到账。"
guarded = build_agent()._guard_visitor_advice(CoreResult(text=text), VISITOR)
assert guarded.text == text
def test_visitor_boundary_reply_does_not_trip_itself() -> None:
"""边界话术里含「是否适合您」——是**否认**给出建议,护栏不得再换一次(自绊)。"""
guarded = build_agent()._guard_visitor_advice(
CoreResult(text=ADVICE_BOUNDARY_REPLY), VISITOR
)
assert guarded.text == ADVICE_BOUNDARY_REPLY
# ---- `C-10`(乙·降级)护栏:知识出口不得启用未完成的来源引用链路 ----
def test_knowledge_exit_never_calls_the_disabled_reference_helper() -> None:
"""知识出口**不得**调用 `_references()`(`C-10` 乙 · `S-8`)。
该方法产出 `source_type="knowledge"` 的引用,而 `governance.review_output` 只认可
memory / tool 两类来源 —— 一旦启用,**整个 run 失败**(不是降级、不是少一个字段)。
所以它是**在位但不调用**的死代码:等基座支持 knowledge 引用(须会签)后再接。
用 AST 判定而不是文本搜索:`_references` 这个名字会合法地出现在注释与文档里,
文本搜索会把「提一句」误判成「调用」。
"""
tree = ast.parse(Path(customer_service_module.__file__).read_text(encoding="utf-8"))
calls = [
node for node in ast.walk(tree)
if isinstance(node, ast.Call)
and isinstance(node.func, ast.Attribute)
and node.func.attr == "_references"
]
assert calls == [], "知识出口调用了 _references():会让整个 run 失败(S-8)"
# 死代码本体要**保留在位**(供基座支持后启用),不要在清理时被连带删掉。
assert any(
isinstance(node, ast.FunctionDef) and node.name == "_references"
for node in ast.walk(tree)
), "`_references()` 是保留待用的死代码,不应被删除"
# ---- `H-01` 澄清出口 `E1`:触发判定与「同族不澄清」 ----
CLARIFY_REQUEST_KEY = "h01-clarify-key-0001"
RESOLVED_TOPIC_ANSWER = "南方季季盈90天:起投金额 1万元。"
#: 同族并列:同一个父块下的两个行级子块(`family_id` 相同)。
SAME_FAMILY_HITS = [
{"score": 0.62, "title": "南方季季盈90天:起投金额", "family_id": "PROD-004",
"content": "起投金额 1万元。"},
{"score": 0.60, "title": "南方季季盈90天:产品期限", "family_id": "PROD-004",
"content": "产品期限 90 天。"},
]
#: 跨族并列:两个不同族的块分数咬得很紧。
CROSS_FAMILY_HITS = [
{"score": 0.62, "title": "基金申购费率", "family_id": "FAQ-010",
"content": "申购费率按金额分档。"},
{"score": 0.60, "title": "基金赎回规则", "family_id": "FAQ-020",
"content": "赎回份额 T+1 确认。"},
]
#: 缺主语在**出口层**可达的形态:同族、领先够多、但分数没到能答的档。
#: (分数够就会直接作答 —— 检索有把握时不该反去问客户。)
MISSING_SUBJECT_HITS = [
{"score": 0.52, "title": "南方季季盈90天:起投金额", "family_id": "PROD-004",
"content": "起投金额 1万元。"},
{"score": 0.42, "title": "南方季季盈90天:产品期限", "family_id": "PROD-004",
"content": "产品期限 90 天。"},
]
#: 缺主语场景:候选跨族、分数也不低(不是"并列"问题,是"没说清问哪个")。
SUBJECTLESS_HITS = [
{"score": 0.62, "title": "南方季季盈90天:起投金额", "family_id": "PROD-004",
"content": "起投金额 1万元。"},
{"score": 0.42, "title": "南方稳健增利:起投金额", "family_id": "PROD-005",
"content": "起投金额 1000 元。"},
]
def build_request(
message: str,
*,
clarification_round: int = 0,
history: tuple[ConversationTurn, ...] = (),
) -> AgentRequest:
return AgentRequest(
agent_type="customer_service",
message=message,
session_id="s-h01",
idempotency_key=CLARIFY_REQUEST_KEY,
metadata=AgentRequestMetadata(clarification_round=clarification_round),
history=history,
)
def test_cross_family_tie_triggers_clarification() -> None:
"""DoD ②「跨族并列」:两个族的候选并驾齐驱 → 问一句,而不是推给人工。"""
agent = build_agent()
assert agent._clarify_reason(
build_request("基金费率怎么算"), CROSS_FAMILY_HITS, score=0.58, gap=0.02
) == "cross_family_tie"
def test_same_family_tie_does_not_clarify() -> None:
"""DoD ⑥:同族并列是「同一话题的不同细节」,该合并作答(`H-03`),问「你要哪个」是伪问题。"""
agent = build_agent()
assert agent._clarify_reason(
build_request("南方季季盈90天介绍一下"), SAME_FAMILY_HITS, score=0.58, gap=0.02
) is None
def test_weak_score_across_families_triggers_clarification() -> None:
"""DoD ②「分数不足且跨族」:分数没到能答的档、候选又散在多个族。"""
agent = build_agent()
assert agent._clarify_reason(
build_request("基金费率怎么算"), CROSS_FAMILY_HITS, score=0.45, gap=0.20
) == "weak_cross_family"
def test_missing_subject_triggers_clarification_when_history_cannot_resolve_it() -> None:
"""DoD ②「缺主语」:这一句自己说不清、上文也接不上。"""
agent = build_agent()
assert agent._clarify_reason(
build_request("起投多少"), SUBJECTLESS_HITS, score=0.62, gap=0.20
) == "missing_subject"
def test_missing_subject_is_not_clarified_when_history_resolves_it() -> None:
"""上文接得住指代时不问:客户只是用了指代,检索能补上主语,再问就是打扰。"""
agent = build_agent()
request = build_request(
"那它起投多少",
history=(ConversationTurn(role="assistant", content=RESOLVED_TOPIC_ANSWER),),
)
assert agent._clarify_reason(request, SUBJECTLESS_HITS, score=0.62, gap=0.20) is None
def test_classifier_needs_clarification_is_consumed() -> None:
"""DoD ①:`needs_clarification` 此前全仓无消费方,现在必须真的影响判定。
消息要**长到不构成缺主语**(否则先命中「缺主语」分支,测不到这个触发条件)。
"""
agent = build_agent()
agent._classified_intent = IntentResult(
intent="faq", confidence=0.4, needs_clarification=True
)
assert agent._clarify_reason(
build_request("基金的申购费率是怎么计算的"), SAME_FAMILY_HITS, score=0.62, gap=0.20
) == "low_intent_confidence"
def test_no_candidate_above_the_clarify_floor_means_no_clarification() -> None:
"""不拿噪声去问客户:一个够格的候选都没有时交给 E5b。"""
agent = build_agent()
weak = [{"score": 0.20, "title": "噪声", "family_id": "FAQ-999", "content": "…"}]
assert agent._clarify_reason(build_request("随便问问"), weak, score=0.20, gap=0.0) is None
def test_missing_family_label_is_treated_as_cross_family() -> None:
"""族标缺失时**不当作同族**:宁可多问一句,也不要把两个族的答案混着答。"""
agent = build_agent()
hits = [{"score": 0.62, "title": "甲"}, {"score": 0.60, "title": "乙"}]
assert agent._clarify_reason(
build_request("这个怎么算"), hits, score=0.58, gap=0.02
) == "cross_family_tie"
def stub_knowledge_tool(hits: list) -> object:
"""替换 `call_tool`:本文件不接工具执行器,只验证出口决策。"""
async def _call(name: str, arguments: dict, *, intent: str, context: RequestContext):
del name, arguments, intent, context
return {"hits": hits}
return _call
async def test_knowledge_exit_clarifies_on_cross_family_tie() -> None:
agent = build_agent()
agent.call_tool = stub_knowledge_tool(CROSS_FAMILY_HITS) # type: ignore[method-assign]
result = await agent._answer_from_knowledge(
build_request("基金费率怎么算"), CUSTOMER, customer_service_module.INTENT_FAQ
)
assert result.clarification_required is True
assert result.transfer_required is False
assert "基金申购费率" in result.text
assert "基金赎回规则" in result.text
async def test_knowledge_exit_does_not_clarify_on_same_family_tie() -> None:
"""DoD ⑥ 的出口级验证:同族并列走 E5b(本期),**不澄清、不转人工**。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(SAME_FAMILY_HITS) # type: ignore[method-assign]
result = await agent._answer_from_knowledge(
build_request("南方季季盈90天介绍一下"), CUSTOMER, customer_service_module.INTENT_FAQ
)
assert result.clarification_required is False
assert result.transfer_required is False
async def test_knowledge_exit_clarifies_when_subject_is_missing() -> None:
agent = build_agent()
agent.call_tool = stub_knowledge_tool(MISSING_SUBJECT_HITS) # type: ignore[method-assign]
result = await agent._answer_from_knowledge(
build_request("起投多少"), CUSTOMER, customer_service_module.INTENT_FAQ
)
assert result.clarification_required is True
assert result.transfer_required is False
async def test_clarification_round_cap_falls_back_to_partial_not_transfer() -> None:
"""DoD ⑤:同话题问满 2 轮后转 E5b(**不建单**),不能变成无限追问或推给人工。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(CROSS_FAMILY_HITS) # type: ignore[method-assign]
result = await agent._answer_from_knowledge(
build_request("基金费率怎么算", clarification_round=2),
CUSTOMER,
customer_service_module.INTENT_FAQ,
)
assert result.clarification_required is False
assert result.transfer_required is False
# ---- `H-04` 验收硬约束:白名单外发生转人工 = 验收不合格 ----
def _enclosing_function_names(tree: ast.Module) -> dict[int, str]:
"""行号 → 所属函数名(内层函数覆盖外层)。"""
owner: dict[int, str] = {}
def visit(node: ast.AST) -> None:
for child in ast.iter_child_nodes(node):
if isinstance(child, (ast.FunctionDef, ast.AsyncFunctionDef)):
for inner in ast.walk(child):
# `ast.arguments` 之类的节点没有 `lineno`,不能直接取。
lineno = getattr(inner, "lineno", None)
if lineno is not None:
owner[lineno] = child.name
visit(child)
else:
visit(child)
visit(tree)
return owner
def _transfer_sites(path: str) -> list[tuple[int, str, ast.AST | None]]:
"""所有 `transfer_required=True` 的落点:行号 / 所属函数 / `transfer_reason` 表达式。"""
tree = ast.parse(Path(path).read_text(encoding="utf-8"))
owners = _enclosing_function_names(tree)
sites: list[tuple[int, str, ast.AST | None]] = []
for node in ast.walk(tree):
if not isinstance(node, ast.Call):
continue
keywords = {item.arg: item.value for item in node.keywords if item.arg}
value = keywords.get("transfer_required")
if isinstance(value, ast.Constant) and value.value is True:
sites.append((node.lineno, owners.get(node.lineno, "<module>"),
keywords.get("transfer_reason")))
return sites
def test_transfer_requires_a_whitelisted_reason_code() -> None:
"""`H-04` DoD ⑤:**白名单外发生转人工 = 验收不合格**。
这条不能只靠"跑一遍看行为"守:转人工是最容易被新分支加回来的东西(新增一个兜底、
把异常吞掉再转人工)——跑一遍只看得到今天的行为,看不到明天新加的分支。所以直接扫
源码,要求每个 `transfer_required=True` 的落点满足其一:
- 落在客服 Agent 的 `_exit_transfer` 里(该函数内部有 `reason_code not in
TRANSFER_REASONS` 的运行时校验,越界即抛错);
- 在规则模块里带上一个**可解析的** `TRANSFER_REASON_*` 常量,且其值在白名单内。
"""
from app.core import customer_service_rules as rules_module
codes: list[str] = []
for module in (rules_module, customer_service_module):
for lineno, func_name, reason_node in _transfer_sites(module.__file__):
guarded = (
func_name == "_exit_transfer"
and isinstance(reason_node, ast.Name)
and reason_node.id == "reason_code"
)
if guarded:
codes.append("guarded_by_exit_transfer")
continue
assert isinstance(reason_node, ast.Name), (
f"{module.__name__}:{lineno} 请求了转人工却没带枚举原因码"
)
code = getattr(module, reason_node.id, None)
assert code in rules_module.TRANSFER_REASONS, (
f"{module.__name__}:{lineno} 的原因码 {code!r}({reason_node.id})不在白名单内"
)
codes.append(str(code))
# 白名单里**实际会发出**的三类必须在场;第 4 类 `account_data` 本期有意不发出
# (见规则模块 P1 分支与 `test_p1_account_data_never_opens_a_ticket_by_design`)。
assert {"safety_risk", "write_or_dispute", "explicit_request"} <= set(codes)
assert "guarded_by_exit_transfer" in codes, "`_exit_transfer` 的白名单守卫入口不见了"
def test_consecutive_fallback_transfer_rule_has_no_residue() -> None:
"""`H-04` DoD ②:「连续 2 轮兜底」已删除 —— `low_score_repeat` 必须零残留。
它曾是 v2.4 的转人工触发之一,v2.5 删除(理由:「兜底」是**能力不足的表征**,不是
风险)。这条守的是"别悄悄加回来"——那个理由码不在白名单内,一旦复活,上面那条结构性
守卫也会失败,但这条给出的信号更早、更直指原因。
"""
from app.core import customer_service_rules as rules_module
for module in (rules_module, customer_service_module):
source = Path(module.__file__).read_text(encoding="utf-8")
assert "low_score_repeat" not in source, (
f"{module.__name__} 复活了已删除的「连续 2 轮兜底」触发"
)
assert "low_score_repeat" not in rules_module.TRANSFER_REASONS
# ---- `H-02b` 计算型出口 `E2`:品类费率试算 / C—R 通用规则 ----
#
# 这一节守的是 `H-02` 的四条 DoD:① 参数只取当前档位可见的参数位;② 计算纯函数、不调模型;
# ③ 未指定产品时只给算法与区间;④ 访客分项开放;⑤ 取不到参数降 `E5b`、**不回退别的档位**。
#
# 参数一律取自 `knowledge/` 下的**真实语料原文**(截成"命中块"喂进来):手写几行假费率表
# 只能证明"代码按我写的样子跑",证明不了"语料里的表能被读出来"。
CORPUS = Path(__file__).resolve().parents[3] / "knowledge"
HANDBOOK = CORPUS / "product" / "个人理财产品手册.md"
GUIDE = CORPUS / "policy" / "个人投资者适当性管理指南.md"
def _section(text: str, start: str, end: str) -> str:
begin = text.index(start)
return text[begin:text.index(end, begin + len(start))]
def _fee_table_hits() -> list:
"""§6.1 费率总表块(`PROD-016` 的形状)。"""
return [{
"score": 0.72,
"doc_id": "PROD-016",
"title": "南方基金管理股份有限公司 公募基金与专户产品手册"
" · 六、费率说明 · 6.1 公募基金费率总表",
"visibility": "public",
"family_id": "PROD-016",
"param_class": "rate",
"intent": "product_inquiry",
"content": _section(
HANDBOOK.read_text(encoding="utf-8"), "### 6.1 公募基金费率总表", "### 6.2"
),
}]
def _matrix_hits() -> list:
"""第十二条匹配矩阵块(`POL-AST-012` 的形状)。"""
return [{
"score": 0.79,
"doc_id": "POL-AST-012-01",
"title": "个人投资者适当性管理指南"
" · 第四章 产品分类与风险等级对应 · 第十二条 投资者与产品匹配矩阵",
"visibility": "public",
"family_id": "POL-AST-012",
"param_class": "none",
"intent": "policy_explain",
"content": _section(
GUIDE.read_text(encoding="utf-8"), "### 第十二条 投资者与产品匹配矩阵", "### 第十三条"
),
}]
PRODUCT = "南方红利价值股票〔示例〕"
def _product_hits() -> list:
"""单只产品的赎回费行级子块(`PROD-006-16` 的形状,单元格照抄语料那一行)。"""
block = _section(
HANDBOOK.read_text(encoding="utf-8"),
f"### 1.6 {PRODUCT}",
"\n### ",
)
cell = next(
line.split("|")[2].strip() for line in block.splitlines()
if line.startswith("| 赎回费率 |")
)
return [{
"score": 0.68,
"doc_id": "PROD-006-16",
"title": "南方基金管理股份有限公司 公募基金与专户产品手册"
f" · 一、公募基金产品 · 1.6 {PRODUCT} · 赎回费",
"visibility": "public",
"family_id": "PROD-006",
"param_class": "rate",
"intent": "product_inquiry",
"content": f"{PRODUCT}:赎回费率 {cell}",
}]
async def _calculate(message: str, *, context: RequestContext = CUSTOMER, hits: list | None = None,
history: tuple[ConversationTurn, ...] = ()):
agent = build_agent()
agent.call_tool = stub_knowledge_tool(hits if hits is not None else []) # type: ignore[method-assign]
return await agent._answer_calculation(
build_request(message, history=history), context
)
async def test_category_purchase_fee_gives_algorithm_and_range_not_a_single_number() -> None:
"""`D-01`:未指定具体产品 → 给算法 + 费率区间,**不给**「最终只收 X 元」。"""
result = await _calculate("买 10 万股票基金,申购费大概多少?", hits=_fee_table_hits())
assert result is not None
assert "申购费 = 申购金额 × 申购费率" in result.text
assert "0.15%" in result.text and "1.5%" in result.text
assert "最终以产品说明书" in result.text
assert "最终只收" not in result.text
assert result.transfer_required is False
async def test_category_purchase_fee_never_states_one_absolute_amount() -> None:
"""金额出现在话术里时必须是**区间**:低 = 高时干脆不给钱数(DoD ③)。"""
result = await _calculate("买 10 万股票基金,申购费大概多少?", hits=_fee_table_hits())
assert result is not None
assert "150 元—1,500 元之间" in result.text
async def test_category_redemption_fee_hits_the_right_tier() -> None:
"""`D-02`:持有 20 天赎回混合基金 → 7—30 天档 = 0.75%。"""
result = await _calculate("我持有 20 天赎回混合基金,赎回费多少?", hits=_fee_table_hits())
assert result is not None
assert "0.75%" in result.text
assert result.transfer_required is False
async def test_redemption_follow_up_reads_the_category_from_the_previous_answer() -> None:
"""`D-03`:追问里没有类别,类别从**上一轮回答的主语**接上(8 个月 → 30—365 天档 0.5%)。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(_fee_table_hits()) # type: ignore[method-assign]
first = await agent._answer_calculation(
build_request("我持有 20 天赎回混合基金,赎回费多少?"), CUSTOMER
)
assert first is not None
# 这一条同时是结构性约束:`E2` 的回答**必须让下一轮接得住主语**——
# 首行不是「<类别>:…」的形状,下一句追问就会掉回知识出口、答不到档位。
assert CustomerServiceAgent._previous_topic(
build_request("持有 8 个月赎回要付费吗?",
history=(ConversationTurn(role="assistant", content=first.text),))
) == "混合基金"
follow_up = await agent._answer_calculation(
build_request(
"持有 8 个月赎回要付费吗?",
history=(ConversationTurn(role="assistant", content=first.text),),
),
CUSTOMER,
)
assert follow_up is not None
assert "0.5%" in follow_up.text
assert follow_up.transfer_required is False
async def test_visitor_gets_the_public_fee_rule_without_being_sent_to_login() -> None:
"""`D3.6` §9.1:公开产品的费用试算对访客开放 —— 不得回落成登录引导。"""
result = await _calculate(
"我持有 20 天赎回混合基金,赎回费多少?", context=VISITOR, hits=_fee_table_hits()
)
assert result is not None
assert "0.75%" in result.text
assert "请先登录" not in result.text
assert result.transfer_required is False
async def test_fee_exit_without_parameters_falls_back_to_partial_and_never_retries() -> None:
"""DoD ⑤:取不到参数 → `E5b`(不转人工),且**只有一次检索** —— 不回退别的档位再取一次。"""
calls: list[tuple[str, object]] = []
async def _call(name: str, arguments: dict, *, intent: str, context: RequestContext):
del intent, context
calls.append((name, arguments.get("query")))
return {"hits": []}
agent = build_agent()
agent.call_tool = _call # type: ignore[method-assign]
result = await agent._answer_calculation(
build_request("我持有 20 天赎回混合基金,赎回费多少?"), CUSTOMER
)
assert result is not None
assert result.transfer_required is False
assert "不能给您一个数字" in result.text
assert len(calls) == 1, "取不到参数时换了别的档位/集合再检索一次 —— 违反 INV-1/INV-5"
async def test_fee_exit_fails_closed_when_the_hit_is_not_a_fee_table() -> None:
result = await _calculate(
"我持有 20 天赎回混合基金,赎回费多少?", hits=[{"score": 0.5, "content": "无关内容"}]
)
assert result is not None
assert result.transfer_required is False
assert "不能给您一个数字" in result.text
async def test_general_suitability_rule_forbids_cross_level_purchase() -> None:
"""`D-04`:C1 能买 R3 吗 → **不能**(跨级禁止),且不能写成"可以,但需签署"。"""
result = await _calculate("我是 C1,能买 R3 的产品吗?", hits=_matrix_hits())
assert result is not None
assert "不可以购买" in result.text
assert "可以,但需签署" not in result.text
assert "本次自述不作为适当性判断依据" in result.text
assert result.transfer_required is False
async def test_general_suitability_rule_reports_the_disclosure_cell() -> None:
result = await _calculate("C3 客户能买 R4 的产品吗", hits=_matrix_hits())
assert result is not None
assert "需签署产品风险揭示书后可以购买" in result.text
async def test_general_suitability_rule_is_open_to_visitors() -> None:
"""`DEC-I8`:通用规则对访客开放(不读画像),但要把「按本人结果核对」指回登录。"""
result = await _calculate("我是 C1,能买 R3 的产品吗?", context=VISITOR, hits=_matrix_hits())
assert result is not None
assert "不可以购买" in result.text
assert "请先登录客户账户" in result.text
async def test_named_product_redemption_fee_uses_that_products_own_schedule() -> None:
"""`E2c`:主语是具体产品时,档位取自**该产品那一块**的赎回费单元格。"""
history = (
ConversationTurn(role="assistant", content=f"### 1.6 {PRODUCT}\n\n| 项目 | 详情 |"),
)
result = await _calculate(
"这只产品的赎回费是多少?持有 20 天。", hits=_product_hits(), history=history
)
assert result is not None
assert "0.75%" in result.text
assert "7—30 天" in result.text
assert result.transfer_required is False
async def test_named_product_fee_is_rejected_when_the_block_is_another_product() -> None:
"""跨块拼是这条路最容易犯的错:命中块不是这只产品时**必须判读不懂**,不能拿它的表算。"""
history = (
ConversationTurn(role="assistant", content=f"### 1.6 {PRODUCT}\n\n| 项目 | 详情 |"),
)
other = [dict(hit, content="别的产品〔示例〕:赎回费率 持有<7 天:1.5%;>7 天:0")
for hit in _product_hits()]
result = await _calculate(
"这只产品的赎回费是多少?持有 20 天。", hits=other, history=history
)
assert result is not None
assert "0.75%" not in result.text
assert result.transfer_required is False
async def test_calculation_answers_do_not_trip_the_visitor_advice_guard() -> None:
"""`E2` 的话术是**我们自己写的**,一旦含「更适合您」这类措辞就会被护栏整条换掉。"""
agent = build_agent()
texts = [
(await _calculate("买 10 万股票基金,申购费大概多少?", hits=_fee_table_hits())).text,
(await _calculate("我持有 20 天赎回混合基金,赎回费多少?", hits=_fee_table_hits())).text,
(await _calculate("我是 C1,能买 R3 的产品吗?", hits=_matrix_hits())).text,
]
for text in texts:
assert agent._guard_visitor_advice(CoreResult(text=text), VISITOR).text == text
async def test_calculation_does_not_intercept_plain_knowledge_questions() -> None:
"""宁可漏触发也不要抢答:类别/等级说不清的问题必须交回 `E3`。"""
for message in ("费用怎么收?", "介绍一下南方红利价值股票", "基金赎回几天到账"):
assert await _calculate(message, hits=_fee_table_hits()) is None
async def test_route_and_answer_routes_fee_questions_into_the_calculation_exit() -> None:
"""接线验证:出口必须真的挂到主分发上(单测只测方法本身会掩盖"没接线")。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(_fee_table_hits()) # type: ignore[method-assign]
result = await agent._route_and_answer(
build_request("我持有 20 天赎回混合基金,赎回费多少?"), CUSTOMER
)
assert "0.75%" in result.text
assert result.transfer_required is False
async def test_route_and_answer_lets_visitors_ask_general_suitability_rules() -> None:
"""接线验证:这条路径**必须早于访客意图白名单**,否则访客会被推去登录。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(_matrix_hits()) # type: ignore[method-assign]
result = await agent._route_and_answer(
build_request("我是 C1,能买 R3 的产品吗?"), VISITOR
)
assert "不可以购买" in result.text
assert "请先登录客户账户" in result.text
# ---- `H-03` 证据约束生成 `E4`:同章节多块合并作答 ----
#: `C-01` 的真实形状(2026-09-19 实测检索):四档权益同属一章、分数咬在 0.712—0.7612。
EVIDENCE_HITS = [
{
"doc_id": "HNW-006", "score": 0.7612, "family_id": "HNW-006",
"source_file": "product/高净值客户服务规范.md",
"title": "南方基金管理股份有限公司 高净值客户服务规范"
" · 二、各层级专属权益 · 2.3 钻石客户权益(600 万+)",
"content": "### 2.3 钻石客户权益(600 万+)\n- 资深客户经理 1 对 1 专属服务",
},
{
"doc_id": "HNW-005", "score": 0.7321, "family_id": "HNW-005",
"source_file": "product/高净值客户服务规范.md",
"title": "南方基金管理股份有限公司 高净值客户服务规范"
" · 二、各层级专属权益 · 2.2 白金客户权益(200 万+)",
"content": "### 2.2 白金客户权益(200 万+)\n- 1 对 1 高级客户经理服务",
},
{
"doc_id": "HNW-007", "score": 0.7124, "family_id": "HNW-007",
"source_file": "product/高净值客户服务规范.md",
"title": "南方基金管理股份有限公司 高净值客户服务规范"
" · 二、各层级专属权益 · 2.4 尊享客户权益(1,000 万+)",
"content": "### 2.4 尊享客户权益(1,000 万+)\n- 私人财富顾问 1 对 1 专属服务",
},
{
"doc_id": "HNW-004", "score": 0.712, "family_id": "HNW-004",
"source_file": "product/高净值客户服务规范.md",
"title": "南方基金管理股份有限公司 高净值客户服务规范"
" · 二、各层级专属权益 · 2.1 金卡客户权益(50 万+)",
"content": "### 2.1 金卡客户权益(50 万+)\n- 专属客户经理服务",
},
]
#: `C-01` 的四档权益全文(金标 `关键事实` 就是这四个门槛)。
EVIDENCE_ANSWER = (
"金卡(50 万+)含专属客户经理服务,白金(200 万+)含 1 对 1 高级客户经理服务,"
"钻石(600 万+)含资深客户经理 1 对 1 专属服务,尊享(1,000 万+)含私人财富顾问服务。"
)
class _FakeModelExecution:
def __init__(self, text: str) -> None:
self.text = text
class _FakeModelService:
"""只回一段预置文本的模型服务:本文件不连真端点,只验证出口决策。"""
def __init__(self, text: str) -> None:
self.text = text
self.prompts: list[str] = []
async def generate(self, endpoints: list, prompt: str, *, max_attempts: int = 2) -> object:
del endpoints, max_attempts
self.prompts.append(prompt)
return _FakeModelExecution(self.text)
class _FakeEndpointResolver:
async def resolve(self, *, agent_type: str, task_type: str) -> list[object]:
del agent_type, task_type
return [object()]
def stub_evidence_model(agent: CustomerServiceAgent, text: str, monkeypatch) -> _FakeModelService:
service = _FakeModelService(text)
agent._model_service = service # type: ignore[assignment]
monkeypatch.setattr(
customer_service_module, "DatabaseModelEndpointResolver",
lambda: _FakeEndpointResolver(),
)
return service
def evidence_payload(answer: str, used: list[str]) -> str:
import json
return json.dumps(
{
"answer": answer,
"used_chunk_ids": used,
"confidence": 0.9,
"unanswerable_reason": "",
},
ensure_ascii=False,
)
async def test_evidence_exit_merges_every_tier_of_the_same_chapter(monkeypatch) -> None:
"""`C-01` 金标:**只答其中一档 = 失败** —— 四档门槛必须全在答复里。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(EVIDENCE_HITS) # type: ignore[method-assign]
stub_evidence_model(
agent,
evidence_payload(EVIDENCE_ANSWER, ["HNW-004", "HNW-005", "HNW-006", "HNW-007"]),
monkeypatch,
)
result = await agent._answer_from_knowledge(
build_request("高净值客户有什么权益?"), CUSTOMER, customer_service_module.INTENT_PRODUCT
)
assert result.transfer_required is False
assert result.clarification_required is False
for tier in ("金卡", "白金", "钻石", "尊享"):
assert tier in result.text
assert "50 万+" in result.text and "1,000 万+" in result.text
async def test_evidence_exit_prompt_carries_doc_ids_and_family_labels(monkeypatch) -> None:
"""DoD ①:输入是**证据包**(块 + `doc_id` + 族标识),不是单块原文。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(EVIDENCE_HITS) # type: ignore[method-assign]
service = stub_evidence_model(
agent, evidence_payload(EVIDENCE_ANSWER, ["HNW-006"]), monkeypatch
)
await agent._answer_from_knowledge(
build_request("高净值客户有什么权益?"), CUSTOMER, customer_service_module.INTENT_PRODUCT
)
prompt = service.prompts[-1]
for doc_id in ("HNW-004", "HNW-005", "HNW-006", "HNW-007"):
assert f"doc_id={doc_id}" in prompt
assert "族标识=HNW-006" in prompt
assert "【用户问题】高净值客户有什么权益?" in prompt
async def test_evidence_exit_blocks_numbers_without_a_source(monkeypatch) -> None:
"""DoD ⑤:答复里出现**证据包外**的数字 → 拦回 `E5b`(不转人工、不建单)。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(EVIDENCE_HITS) # type: ignore[method-assign]
stub_evidence_model(
agent,
evidence_payload("尊享客户门槛为 5,000 万元。", ["HNW-007"]),
monkeypatch,
)
result = await agent._answer_from_knowledge(
build_request("高净值客户有什么权益?"), CUSTOMER, customer_service_module.INTENT_PRODUCT
)
assert "5,000 万元" not in result.text
assert result.transfer_required is False
assert result.clarification_required is False
async def test_evidence_exit_blocks_references_outside_the_pack(monkeypatch) -> None:
"""引用越界(用了包外 `doc_id`)同样拦回 `E5b`:来源可解析率 `M-5` 要 100%。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(EVIDENCE_HITS) # type: ignore[method-assign]
stub_evidence_model(agent, evidence_payload("四档权益。", ["HNW-999"]), monkeypatch)
result = await agent._answer_from_knowledge(
build_request("高净值客户有什么权益?"), CUSTOMER, customer_service_module.INTENT_PRODUCT
)
assert EVIDENCE_ANSWER not in result.text
assert result.transfer_required is False
async def test_evidence_exit_blocks_a_visitor_side_advice_sentence(monkeypatch) -> None:
"""访客侧红线在 `E4` 里**先拦一次**,且拦下来是 `E5b` 而不是边界话术替换。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(EVIDENCE_HITS) # type: ignore[method-assign]
stub_evidence_model(
agent,
evidence_payload("金卡权益更适合您,建议您尽快申购。", ["HNW-004"]),
monkeypatch,
)
result = await agent._answer_from_knowledge(
build_request("高净值客户有什么权益?"), VISITOR, customer_service_module.INTENT_PRODUCT
)
assert "更适合您" not in result.text
assert result.transfer_required is False
async def test_evidence_exit_reports_when_it_cannot_answer(monkeypatch) -> None:
"""模型自述"答不了" → **不接管**,原有的 `E3` 判定原样生效(不是降级成部分答)。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(EVIDENCE_HITS) # type: ignore[method-assign]
stub_evidence_model(
agent,
'{"answer": "", "used_chunk_ids": [], "confidence": 0,'
' "unanswerable_reason": "missing_subject"}',
monkeypatch,
)
result = await agent._answer_from_knowledge(
build_request("高净值客户有什么权益?"), CUSTOMER, customer_service_module.INTENT_PRODUCT
)
# `W20` 展示层净化后:`E3` 直返仍然以知识块原文为准,只是去掉 markdown 标记。
expected = customer_service_module.render_plain(EVIDENCE_HITS[0]["content"])
assert result.text == expected
assert "###" not in result.text
assert result.transfer_required is False
async def test_evidence_exit_steps_aside_when_the_model_is_unavailable() -> None:
"""模型端点不可用 → 不接管,`E3` 原文直返照旧(不建单、不转人工)。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(EVIDENCE_HITS) # type: ignore[method-assign]
result = await agent._answer_from_knowledge(
build_request("高净值客户有什么权益?"), CUSTOMER, customer_service_module.INTENT_PRODUCT
)
# `W20` 展示层净化后:`E3` 直返仍然以知识块原文为准,只是去掉 markdown 标记。
expected = customer_service_module.render_plain(EVIDENCE_HITS[0]["content"])
assert result.text == expected
assert "###" not in result.text
assert result.transfer_required is False
async def test_evidence_exit_is_reachable_from_the_router(monkeypatch) -> None:
"""接线验证:入口分支必须真的走到 `E4`(否则金标只会在单测里过)。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(EVIDENCE_HITS) # type: ignore[method-assign]
stub_evidence_model(
agent, evidence_payload(EVIDENCE_ANSWER, ["HNW-004", "HNW-005", "HNW-006", "HNW-007"]),
monkeypatch,
)
result = await agent._route_and_answer(build_request("高净值客户有什么权益?"), CUSTOMER)
assert "尊享" in result.text
# ---- `E4` 触发面:不该接管的一律不能接管 ----
def test_chapter_group_needs_three_title_segments() -> None:
agent = build_agent()
assert agent._chapter_group_of({
"title": "高频问答", "source_file": "faq/高频问答对.txt",
}) is None
assert agent._chapter_group_of({
"title": "南方基金 · 一、公司概况", "source_file": "company/企业信息.md",
}) is None
assert agent._chapter_group_of({
"title": "A · 二、各层级专属权益 · 2.1 金卡",
"source_file": "product/x.md",
}) == ("product/x.md", "二、各层级专属权益")
assert agent._chapter_group_of({"title": "A · 二、X · Y"}) is None # 缺 source_file
def test_evidence_pack_triggers_on_near_tie_in_two_shapes() -> None:
"""触发面(`W2` 后):先决是 `gap` 接近,再分「同章节多块」与「跨章节近分」两种形态。"""
agent = build_agent()
# 同章节多块 + 分数接近 → 接管
assert agent._evidence_pack(EVIDENCE_HITS, gap=0.029) is not None
# 🔴 领先明显 → **不接管**(原文直返没有幻觉面,能不用模型就不用)
assert agent._evidence_pack(EVIDENCE_HITS, gap=0.20) is None
# 🆕 跨章节近分 → **接管**。`W2`(2026-09-19 `H-06` 实测):首跑 46 条里 14 条落澄清,
# 其中 9 条 top1 已 ≥0.55 —— 旧实现只因"凑不出同章节组"就退澄清,而 `D2.4`
# 附录F.3 的原话是「TopK 内同族多块且分数接近 → 合并为一个答案(E4),**不澄清**」。
same_family = [
{"score": 0.62, "doc_id": "PROD-004-01", "title": "南方季季盈90天:起投金额",
"family_id": "PROD-004", "content": "起投金额 1万元。"},
{"score": 0.60, "doc_id": "PROD-004", "title": "南方季季盈90天:产品期限",
"family_id": "PROD-004", "content": "产品期限 90 天。"},
]
pack = agent._evidence_pack(same_family, gap=0.02)
assert pack is not None
assert [hit["doc_id"] for hit in pack] == ["PROD-004-01", "PROD-004"]
# 低分噪声仍然进不了包:全部低于 `E4_MIN_SCORE` 时**不接管**(噪声不能被拿去生成)
weak = [
{"score": 0.42, "doc_id": "X-1", "title": "A · 一、章 · 1", "source_file": "a.md",
"content": "x"},
{"score": 0.41, "doc_id": "X-2", "title": "A · 一、章 · 2", "source_file": "a.md",
"content": "y"},
]
assert agent._evidence_pack(weak, gap=0.01) is None
def test_evidence_pack_fallback_keeps_order_and_respects_the_floor() -> None:
"""跨章节证据包:只收 ≥ `E4_MIN_SCORE` 的块,且保留检索给出的分数序。"""
agent = build_agent()
hits = [
{"score": 0.70, "doc_id": "A", "title": "x", "content": "a"},
{"score": 0.69, "doc_id": "B", "title": "x", "content": "b"},
{"score": 0.30, "doc_id": "C", "title": "x", "content": "c"},
]
pack = agent._evidence_pack(hits, gap=0.01)
assert pack is not None
assert [hit["doc_id"] for hit in pack] == ["A", "B"]
def test_search_query_carries_the_subject_for_a_pronoun_followup() -> None:
"""`W3`/`H-01`:句首「那它…」的追问必须补上主语。
旧实现里「那它风险等级呢?」恰好 8 字,`len(message) < 8` **不成立**,而
`_REFERRING_WORDS` 又刻意不收单字「它」⇒ 主语整条丢,检索退化成泛问 → 落澄清。
"""
agent = build_agent()
history = (
ConversationTurn(role="user", content="南方稳健增利债券 A 的起投金额是多少?"),
ConversationTurn(
role="assistant", content="南方稳健增利债券 A〔示例〕:起投金额 1,000 元"
),
)
query = agent._search_query(build_request("那它风险等级呢?", history=history))
assert query.startswith("南方稳健增利债券 A")
def test_search_query_switches_the_category_and_keeps_the_predicate() -> None:
"""`W3`/`H-02` 反向考点:「混合基金呢?」自己没有谓词,谓词从上一位**客户问句**继承。
只看上一位**答复**取不到主语(FAQ 答复首行是「问:…」),而把「货币基金」当主语拼进来
会答错(客户问的是混合基金)—— 正确形态是"换主语、留谓词"。
"""
agent = build_agent()
history = (
ConversationTurn(role="user", content="货币基金赎回多久到账?"),
ConversationTurn(
role="assistant",
content="问:基金赎回到账需要多长时间?\n答:货币基金支持 T+0 快速赎回…",
),
)
assert agent._search_query(
build_request("混合基金呢?", history=history)
) == "混合基金赎回多久到账?"
def test_category_switch_rule_does_not_hijack_a_question_with_its_own_predicate() -> None:
"""判据必须窄:自己带谓词的问句不得被改写成上一位问句。"""
agent = build_agent()
history = (ConversationTurn(role="user", content="货币基金赎回多久到账?"),)
assert agent._search_query(
build_request("货币基金的费率是多少", history=history)
) == "货币基金的费率是多少"
def test_evidence_pack_keeps_the_best_matching_block() -> None:
"""`C-03` 的形状:top1 是一条 FAQ,答案却在本章节的父块里 —— 两者都要进包。"""
agent = build_agent()
guide = "个人投资者适当性管理指南 · 第二章 投资者分类标准"
hits = [
{"score": 0.8226, "doc_id": "FAQ-0042", "title": "什么是专业投资者?怎么申请认定?",
"family_id": "FAQ-0042", "source_file": "faq/高频问答对.txt", "content": "答:……"},
{"score": 0.8304, "doc_id": "POL-AST-005",
"title": f"{guide} · 第五条 专业投资者认定标准",
"family_id": "POL-AST-005", "source_file": "policy/适当性指南.md",
"content": "四项条件……"},
{"score": 0.7728, "doc_id": "POL-AST-005-03",
"title": f"{guide} · 第五条 专业投资者认定标准 · 投资经验",
"family_id": "POL-AST-005", "source_file": "policy/适当性指南.md",
"content": "投资经验……"},
{"score": 0.7449, "doc_id": "POL-AST-004-01",
"title": f"{guide} · 第四条 投资者分类框架 · 专业投资者",
"family_id": "POL-AST-004", "source_file": "policy/适当性指南.md",
"content": "分类框架……"},
]
pack = agent._evidence_pack(hits, gap=0.0498)
assert pack is not None
ids = [hit["doc_id"] for hit in pack]
assert ids[0] == "POL-AST-005"
assert "FAQ-0042" in ids
def test_number_tokens_normalise_units_and_skip_ordinals() -> None:
agent = build_agent()
tokens = agent._number_tokens(
"金卡 50 万+,白金 200 万元,申购费 1.50%,100,000 元,赎回 7—30 天,"
"R1 到 R5,共 4 档,7×24 服务"
)
assert "500000" in tokens # 50 万 → 元
assert "2000000" in tokens # 200 万元 → 元
assert "1.5%" in tokens # 1.50% 归一成 1.5%
assert "100000" in tokens # 千分位去掉
assert "4" not in tokens # 「共 4 档」是数量词,不是产品要素
assert "1" not in tokens and "5" not in tokens # R1/R5 是等级码
assert "7" not in tokens # 7×24 里的 7 不足 3 位且无单位
def test_ungrounded_numbers_accept_restating_the_question() -> None:
agent = build_agent()
evidence = [{"title": "t", "content": "金卡 50 万+;申购费 1.50%"}]
assert agent._ungrounded_numbers("金卡 50 万+,申购费 1.5%。", evidence) == []
assert agent._ungrounded_numbers("门槛 5,000 万元。", evidence) == ["50000000"]
# 用户自己说过的数字可以复述(不是幻觉)
assert agent._ungrounded_numbers("您说的 3 年…", evidence, "持有 3 年") == []
def test_clamp_answer_cuts_at_a_sentence_boundary() -> None:
"""`E4` 的合并答复会撞上限:宁可少说半句,也不要把句子劈成两半。
`E-06`:截断必须**说出来**。静默截断会让客户以为「资料就到这里」,于是既不
追问也不找人工 —— 这正是「客服不智能」观感的一部分。
"""
agent = build_agent()
long_text = "甲" * 1000 + "。" + "乙" * 800
clamped = agent._clamp_answer(long_text)
# 句子边界仍然守住(不劈句)
assert clamped.startswith("甲" * 1000 + "。")
assert "乙" not in clamped
# 并且**显式告知**这是摘要(`E-06`)
assert clamped.endswith(customer_service_module.TRUNCATION_NOTICE)
# 上限内的答复一个字都不动(既有行为不变)
assert agent._clamp_answer("很短的一句话。") == "很短的一句话。"
def test_parse_evidence_output_accepts_fenced_json() -> None:
agent = build_agent()
payload = agent._parse_evidence_output(
'```json\n{"answer": "a", "used_chunk_ids": ["X"], "confidence": 0.5,'
' "unanswerable_reason": ""}\n```'
)
assert payload is not None and payload["answer"] == "a"
assert agent._parse_evidence_output("抱歉,我不确定。") is None
assert agent._parse_evidence_output("[1, 2]") is None
async def test_visitor_suitability_intent_reaches_the_evidence_exit(monkeypatch) -> None:
"""`C-04` 的必要接线:实测意图分类器把「C1 客户能买什么?C5 呢?」判成
`suitability_check`(0.95),而该意图不在 `VISITOR_INTENTS` 里 —— 不加例外,
一条**公开规则题**会被直接推去登录,`E4` 永远够不着。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(EVIDENCE_HITS) # type: ignore[method-assign]
agent._classified_intent = IntentResult(intent="suitability_check", confidence=0.95)
stub_evidence_model(agent, evidence_payload(EVIDENCE_ANSWER, ["HNW-006"]), monkeypatch)
result = await agent._route_and_answer(build_request("C1 客户能买什么?C5 呢?"), VISITOR)
assert "请先登录客户账户" not in result.text
assert "尊享" in result.text
async def test_visitor_other_intents_still_go_to_login() -> None:
"""例外只开给 `suitability_check` 一个:其余未覆盖意图照旧引导登录,不得放宽。"""
agent = build_agent()
agent._classified_intent = IntentResult(intent="unknown_intent", confidence=0.5)
result = await agent._route_and_answer(build_request("随便问问这个"), VISITOR)
assert "请先登录客户账户" in result.text
# ---- `F-2`(2026-09-19 裁定):转人工只由「用户显式要求」触发 ----
#
# 这一节守的是一个**行为不变量**:意图标签 `transfer_human` 不再等于建单。
# 为什么值得单独立节:它是 `M-6` 转人工率的分子,也是「客服不智能」最直接的形态 ——
# 实测意图分类器把「南方基金客服现在方便联系吗?」判成 `transfer_human`(0.9),
# 而那句问的是**联系方式**(语料里有),旧实现据此直接建单(金标 `D-05` 期望 E3 作答)。
#
# 判据不是"再也不要转人工":显式要求人工由 `route_message()` 在**检索之前**确定性拦下,
# 那一条仍必须建单。两条分工不能互相侵蚀。
COMPANY_INFO = CORPUS / "company" / "企业信息.md"
def _contact_hits() -> list:
"""`COMP-022` 的形状:公司公开联系方式速查表(含热线与服务时段),取自真实语料。"""
text = COMPANY_INFO.read_text(encoding="utf-8")
return [{
"score": 0.71,
"doc_id": "COMP-022-01",
"title": "南方基金管理股份有限公司 企业信息 · 九、其他公开信息速查",
"family_id": "COMP-022",
"source_file": "company/企业信息.md",
"content": text[text.index("## 九、"):].strip(),
}]
def _recording_tool(hits: list, calls: list) -> object:
"""记录检索次数:这一节的两个方向都要证明「检索发生 / 未发生」。"""
async def _call(name: str, arguments: dict, *, intent: str, context: RequestContext):
del context
calls.append((name, arguments.get("query"), intent))
return {"hits": hits}
return _call
def test_knowledge_missed_counts_only_empty_answers() -> None:
"""`F-2` 的「答不上来」判据:只有 `E5b` **空答**算,澄清与已转人工都不算。"""
agent = build_agent()
assert agent._knowledge_missed(agent._exit_partial([], note="知识库未命中")) is True
assert agent._knowledge_missed(
agent._exit_partial([], note="知识检索降级:milvus_unavailable")
) is True
# 带内容的 E5b(部分命中)与澄清都不是"答不上来"
assert agent._knowledge_missed(
agent._exit_partial([{"score": 0.6, "content": "基金申购费率按金额分档。"}], note="n")
) is False
clarified = agent._exit_clarify(
[{"score": 0.55, "title": "甲"}, {"score": 0.50, "title": "乙"}]
)
assert clarified is not None and agent._knowledge_missed(clarified) is False
assert agent._knowledge_missed(agent._exit_transfer("explicit_request")) is False
async def test_transfer_intent_answers_from_knowledge_instead_of_transferring() -> None:
"""`D-05`:标签是 `transfer_human`、问的却是联系方式 —— 先检索、正常作答、**不建单**。"""
calls: list = []
agent = build_agent()
agent.call_tool = _recording_tool(_contact_hits(), calls) # type: ignore[method-assign]
agent._classified_intent = IntentResult(intent="transfer_human", confidence=0.9)
result = await agent._route_and_answer(build_request("南方基金客服现在方便联系吗?"), VISITOR)
assert len(calls) == 1
assert result.transfer_required is False
assert result.transfer_reason is None
assert "400-889-8899" in result.text
async def test_transfer_intent_transfers_only_after_knowledge_misses() -> None:
"""「先检索一次,答不上来再转」:知识点**空答**时才按用户想找人的原意建单。"""
calls: list = []
agent = build_agent()
agent.call_tool = _recording_tool([], calls) # type: ignore[method-assign]
agent._classified_intent = IntentResult(intent="transfer_human", confidence=0.9)
result = await agent._route_and_answer(build_request("我想找个人问问"), VISITOR)
assert len(calls) == 1
assert result.transfer_required is True
assert result.transfer_reason == "explicit_request"
assert "400-889-8899" in result.text
async def test_explicit_human_request_is_intercepted_before_any_retrieval() -> None:
"""显式要求人工仍走 `route_message()`(P2):**检索前**拦下,不得退化成"先查一次"。"""
calls: list = []
agent = build_agent()
agent.call_tool = _recording_tool(_contact_hits(), calls) # type: ignore[method-assign]
agent._classified_intent = IntentResult(intent="transfer_human", confidence=0.9)
result = await agent._route_and_answer(build_request("我要人工客服"), VISITOR)
assert calls == []
assert result.transfer_required is True
assert result.transfer_reason == "explicit_request"
# ---- `F-3`(2026-09-19 裁定):`E3`/`E4` 的**主体相关性闸门** ----
#
# 为什么需要:`B-04`「南方基金投顾服务起点是多少?」召回的是产品手册的产品参数块
# (**一块都没提到投顾**),`E4` 却按"同章节多块"接管,答成某只混合基金的整段参数 ——
# 题目问 A、答案是 B。
#
# 闸门**只在问句点名受控主题词时**启用:因此「资产到多少能升级?」这类没有主题词、
# 但检索正确的问句不受任何影响(下面第一条就用它把这条边界钉死)。
OFF_TOPIC_HITS = [
{
"doc_id": "PROD-003-17", "score": 0.7455, "family_id": "PROD-003",
"source_file": "product/个人理财产品手册.md",
"title": "南方基金管理股份有限公司 公募基金与专户产品手册"
" · 一、公募基金产品 · 1.3 南方平衡优选混合〔示例〕",
"content": "| 起投金额 | 5,000 元 |\n| 基金规模 | 约 92 亿元(截至 2026-06-30,虚构) |",
},
{
"doc_id": "PROD-001-05", "score": 0.7401, "family_id": "PROD-001",
"source_file": "product/个人理财产品手册.md",
"title": "南方基金管理股份有限公司 公募基金与专户产品手册"
" · 一、公募基金产品 · 1.1 南方现金添利货币市场基金〔示例〕 · 起投金额",
"content": "南方现金添利货币市场基金〔示例〕:起投金额 1 元",
},
]
def test_subject_terms_only_fire_when_the_question_names_one() -> None:
"""闸门**只在问句点名主题词时**启用:没有主题词的问句,判据必须原样放行。"""
# 「南方基金投顾服务」同时含三个同义主题词(基金投顾 / 投顾服务 / 投顾),都属同一主题
assert CustomerServiceAgent._subject_terms_in("南方基金投顾服务起点是多少?") == (
"投顾服务", "基金投顾", "投顾",
)
# 这两条正是"没有主题词但检索正确"的形态(`C-02` / `C-04`),必须什么都不触发
assert CustomerServiceAgent._subject_terms_in("资产到多少能升级?") == ()
assert CustomerServiceAgent._subject_terms_in("C1 客户能买什么?C5 呢?") == ()
def test_subject_gate_is_satisfied_by_title_or_content_of_any_hit() -> None:
hits = [{"title": "… · 六、费率说明 · 6.1 公募基金费率总表", "content": "赎回费率…"}]
assert CustomerServiceAgent._subject_covered_by(hits, ("费率",)) is True
assert CustomerServiceAgent._subject_covered_by(hits, ("投顾服务",)) is False
# 无主题词 = 闸门不启用;非 dict 命中不得当成证据
assert CustomerServiceAgent._subject_covered_by(hits, ()) is True
assert CustomerServiceAgent._subject_covered_by([None, "噪声"], ("费率",)) is False
async def test_off_topic_evidence_never_gets_generated(monkeypatch) -> None:
"""`B-04`:问的是投顾服务,证据里一块都没提到投顾 → **模型一次都不调**、交回 E5b。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(OFF_TOPIC_HITS) # type: ignore[method-assign]
service = stub_evidence_model(
agent,
evidence_payload("1.3 南方平衡优选混合:起投金额 5,000 元。", ["PROD-003-17"]),
monkeypatch,
)
result = await agent._answer_from_knowledge(
build_request("南方基金投顾服务起点是多少?"), CUSTOMER, customer_service_module.INTENT_FAQ
)
assert service.prompts == []
assert "5,000 元" not in result.text
assert "投顾服务" in result.text
assert result.transfer_required is False
async def test_subject_gate_leaves_matching_evidence_untouched(monkeypatch) -> None:
"""命中块含主题词时闸门不生效:`C-01` 的合并作答必须原样通过。"""
agent = build_agent()
agent.call_tool = stub_knowledge_tool(EVIDENCE_HITS) # type: ignore[method-assign]
service = stub_evidence_model(
agent,
evidence_payload(EVIDENCE_ANSWER, ["HNW-004", "HNW-005", "HNW-006", "HNW-007"]),
monkeypatch,
)
result = await agent._answer_from_knowledge(
build_request("高净值客户有什么权益?"), CUSTOMER, customer_service_module.INTENT_FAQ
)
assert len(service.prompts) == 1
assert "尊享" in result.text
assert result.transfer_required is False
def test_safety_exit_preserves_the_route_contract() -> None:
"""`W6-1`:安全出口抽成方法后**逐字等价** —— 文案 / 意图 / 建单三件套全部跟随 `route`。
为什么值得单测:抽方法是为了让探针打得到标(`_exit_safety` 是探针包装的终止型方法),
但重构本身不得改变任何行为;`P1` 与合规**不建单**这条也在这里钉死。
"""
from app.core.customer_service_rules import route_message
agent = build_agent()
messages = (
"我的验证码被人要走了怎么办?", # P0 反诈:建单
"我账户现在有多少钱?收益多少?", # P1 账户:不建单
"帮我把绑定银行卡换一下", # P2 代办:建单
"什么样的基金不会亏钱?", # 合规:不建单
)
for message in messages:
route = route_message(message)
assert route is not None, message
result = agent._exit_safety(route)
assert result.text == route.reply
assert result.transfer_required == route.transfer_required
assert result.transfer_reason == route.transfer_reason
assert result.clarification_required is False
def test_same_document_requires_a_single_known_source() -> None:
"""`W6`:同文档判据 —— 只有"全部同类源且都取得到"才算范围清楚。
判据宁严勿宽:澄清多问一句只是体验差,误判成"范围清楚"则可能答错主体。
"""
same = [{"source_file": "company/企业信息.md"}, {"source_file": "company/企业信息.md"}]
mixed = [{"source_file": "company/企业信息.md"}, {"source_file": "faq/高频问答对.txt"}]
assert CustomerServiceAgent._same_document(same) is True
assert CustomerServiceAgent._same_document(same[:1]) is True
assert CustomerServiceAgent._same_document(mixed) is False
assert CustomerServiceAgent._same_document([{"title": "取不到来源"}]) is False
assert CustomerServiceAgent._same_document([None, {"source_file": "a.md"}]) is False
# 空列表 = 没有任何信息 ⇒ 保守判 False(宁可多问一句,不可错答主体)
assert CustomerServiceAgent._same_document([]) is False
# ---- `W7`:三处实测缺口(本人档位问答 / 画像枚举码本地化 / E2c 查询串带参数) ----
@pytest.mark.parametrize(
"message",
["我够哪一档?", "我在哪一档", "我的档位是多少", "我属于什么分层", "本人是什么星级"],
)
def test_first_person_tier_questions_go_to_the_profile_exit(message: str) -> None:
"""`H-03` 实测漏网:这类问法问的是**本人档案里的分层**,知识库答不了。
旧实现在这里直接落到 `E5b-suitability`「给不出这个适当性结论」—— 属"能答而不答"。
"""
assert customer_service_module.is_profile_question(message) is True
@pytest.mark.parametrize(
"message",
[
"高净值客户有什么权益?", # 问他人/规则 → 知识库
"公司的客户分层标准是怎样的?", # 问规则 → 知识库
"我的等级能买 R5 吗", # "等级"指产品风险等级 → 走适当性
"C1 客户能买 R3 的产品吗?", # 通用规则题 → 走 E2c
],
)
def test_tier_detection_does_not_steal_rule_questions(message: str) -> None:
"""判据的**负向边界**:只认第一人称 + 档位词,不得把规则题抢进画像出口。"""
assert customer_service_module.is_profile_question(message) is False
async def test_tier_question_actually_calls_the_profile_tool() -> None:
"""端到端口径:该问句必须真的走 `query_customer_profile`,并把分层讲给客户。"""
agent = build_agent()
calls: list[str] = []
async def _call(name, arguments, *, intent, context): # noqa: ANN001, ANN202
del intent, context
calls.append(name)
return {"profile": {"customer_tier": "gold", "investor_type": "C1"}}
agent.call_tool = _call # type: ignore[method-assign]
result = await agent._route_and_answer(build_request("我够哪一档?"), CUSTOMER)
assert calls == ["query_customer_profile"]
assert "金卡" in result.text
assert result.transfer_required is False
@pytest.mark.parametrize(
"message",
["我的风险测评结果是什么", "我的测评结果", "我的风险测评"],
)
async def test_risk_assessment_result_goes_to_the_profile_exit(message: str) -> None:
"""`W15` 实测缺口:「风险测评结果」原先被 `P1_KEYWORDS` 的裸词拦成「无法读取本人账户数据」。
这是**能答而不答** —— `D2.2` §1.7 第 21 项要求「画像问答字段直返」,答案走受控工具
`query_customer_profile`(自我作用域 + 字段白名单 + 工具审计),**不是**账户数据。
同一诉求换个说法结论相反(「我的风险等级是多少」走画像作答)本身就说明分类错了。
本用例端到端守两件事:① 真的调了画像工具;② **没有**落到 `P1_REPLY`。
"""
agent = build_agent()
calls: list[str] = []
async def _call(name, arguments, *, intent, context): # noqa: ANN001, ANN202
del intent, context
calls.append(name)
assert arguments == {"customer_id": "9001"}, "必须只查 context.user_id"
return {"profile": {"investor_type": "C3", "customer_tier": "gold"}}
agent.call_tool = _call # type: ignore[method-assign]
result = await agent._route_and_answer(build_request(message), CUSTOMER)
assert calls == ["query_customer_profile"]
assert "C3" in result.text
assert "平衡型" in result.text
from app.core.customer_service_rules import P1_REPLY
assert P1_REPLY not in result.text
assert result.transfer_required is False
def test_mixed_profile_and_account_question_still_takes_p1() -> None:
"""反向守卫:画像词在前、账户词在后的**混问法**不得被画像豁免放行。
「…和持仓一起给我」的真实诉求包含**账户数据**(Agent 无权读取),必须照旧 `P1`;
否则客户会拿到一段只答画像的答复,而账户那半句被静默忽略。
"""
from app.core.customer_service_rules import P1_REPLY, route_message
route = route_message("我的风险测评结果和持仓一起给我")
assert route is not None
assert route.priority == "P1"
assert route.reply == P1_REPLY
assert route.transfer_required is False
def test_profile_render_localises_internal_codes() -> None:
"""快照里存的是内部码(`short_term` / `gold` / `money_fund`),不得原样吐给客户。"""
text = customer_service_module.render_profile({
"investor_type": "C1",
"investment_horizon": "short_term",
"trading_frequency": "low",
"preferred_asset_class": ["money_fund"],
"customer_tier": "gold",
})
assert "短期(1 年以内)" in text
assert "较低" in text
assert "货币基金" in text
assert "金卡" in text
for code in ("short_term", "money_fund", "gold", "low"):
assert code not in text
def test_profile_render_keeps_chinese_values_and_hides_unknown_codes() -> None:
"""另一条生产链写的是中文标签(`3至5年`)→ 原样保留;未知内部码 → 宁可不提。"""
text = customer_service_module.render_profile({
"investment_horizon": "3至5年",
"preferred_asset_class": ["固定收益类", "reits_v2"],
"customer_tier": "unknown_tier",
})
assert "3至5年" in text
assert "固定收益类" in text
assert "reits_v2" not in text
assert "unknown_tier" not in text
def test_profile_render_empty_fallback_does_not_push_to_human() -> None:
"""只含未知码的画像 → 走查不到口径,**不得**写成"建议转人工客服核实"。"""
text = customer_service_module.render_profile({"investment_horizon": "quarterly_v9"})
assert text == customer_service_module.PROFILE_MISS_TEMPLATE
assert "建议转人工" not in text
async def test_suitability_rule_query_carries_the_resolved_levels() -> None:
"""`D-04` 实测:矩阵按等级切块入库,泛问法只会撞上某一格(top1 是 C3 那格)。
所以查询串必须带上刚解析出的 `C1` / `R3` —— 这是**查询改写**,不是放宽判据。
"""
agent = build_agent()
queries: list[str] = []
async def _call(name, arguments, *, intent, context): # noqa: ANN001, ANN202
del name, intent, context
queries.append(str(arguments.get("query") or ""))
return {"hits": _matrix_hits()}
agent.call_tool = _call # type: ignore[method-assign]
result = await agent._answer_suitability_rule("我是 C1,能买 R3 的产品吗?", CUSTOMER)
assert queries and "C1" in queries[0] and "R3" in queries[0]
assert "不可以购买" in result.text
def test_partial_exit_shows_the_best_block_not_the_first_one() -> None:
"""`B-06` 实测:`E4` 降级交来的是**证据包**(章节组在前、高分补位块在尾)。
旧实现取 `hits[0]`,把"章节组里分最高的那块"当最佳证据展示 —— 客户拿到的是一段
答非所问的碎片,而真正贴题的那块在包尾。
"""
result = build_agent()._exit_partial(
[
{"score": 0.62, "doc_id": "POL-SPM-034-04",
"content": "其他费用:认购费 认购时一次性收取"},
{"score": 0.657, "doc_id": "PROD-018", "content": "### 6.3 费用计算示例"},
],
note="复现 B-06",
)
# 标记被去掉、字留下(`W20` `render_plain`):客户看到的是「6.3 费用计算示例」
assert "6.3 费用计算示例" in result.text
assert "###" not in result.text
assert "认购费 认购时一次性收取" not in result.text
assert result.transfer_required is False
def test_partial_exit_ignores_non_mapping_hits() -> None:
"""命中列表里混进非映射项时不得炸;全部低于 `PARTIAL_FLOOR` 时只说"没找到"。"""
agent = build_agent()
assert "没找到" in agent._exit_partial([None, "oops"]).text
assert "没找到" in agent._exit_partial([{"score": 0.1, "content": "噪声"}]).text
# ---------------------------------------------------------------------------
# `E-05` 出口主题声明:显式声明优先,存量行回落文本反解
# ---------------------------------------------------------------------------
def test_declared_topic_is_equivalent_to_legacy_parse() -> None:
"""`E3`/`E5b` 的声明值与旧的读侧反解**逐字相等** —— 证明这一步是重构而非改行为。
声明点从「读侧事后反解整段回答」挪到「出口对本次返回的块求一次」。只要两者
对同一段内容给出同一结果,多轮追问的检索词就不会因为这次改动而变。
"""
agent = build_agent()
samples = [
"南方季季盈90天:起投金额 1万元\n\n| 项目 | 详情 |",
"### 2.1 南方稳健增利债券A\n\n| 项目 | 详情 |",
"问:什么是基金定投?\n答:定期定额投资。",
"第十二条 投资者与产品匹配矩阵\n\n| C1 | R1 |",
"南方季季盈90天为 R2(中低风险),…",
"完全没有主语的兜底说明。",
]
for content in samples:
assert agent._declared_topic(content) == agent._topic_of(content)
def test_turn_topic_prefers_declared_subject() -> None:
"""`E-05` 读侧:有声明就用声明值,不再解析正文。"""
agent = build_agent()
declared = ConversationTurn(
role="assistant", content="(正文里没有任何产品名)", subject="南方季季盈90天"
)
assert agent._turn_topic(declared) == "南方季季盈90天"
# 声明值与正文不一致时,**以声明为准**(这正是"不再反解"的含义)
conflicting = ConversationTurn(
role="assistant", content="别的产品:起投金额 5万元", subject="南方稳健增利债券A"
)
assert agent._turn_topic(conflicting) == "南方稳健增利债券A"
def test_turn_topic_falls_back_for_legacy_rows() -> None:
"""存量消息(没有 `subject`)必须仍能取到主语 —— 否则多轮追问会突然断掉。"""
agent = build_agent()
legacy = ConversationTurn(
role="assistant", content="南方季季盈90天:起投金额 1万元"
)
assert legacy.subject == ""
assert agent._turn_topic(legacy) == "南方季季盈90天"
def test_search_query_uses_declared_subject_for_followups() -> None:
"""端到端效果:短追问的检索词带的是**声明的主语**。"""
agent = build_agent()
request = AgentRequest(
agent_type="customer_service",
message="那它风险高吗?",
session_id="s",
idempotency_key="k" * 16,
history=(
ConversationTurn(
role="assistant", content="(某段没有主语的行级说明)",
subject="南方季季盈90天",
),
),
)
assert agent._search_query(request).startswith("南方季季盈90天 ")
async def test_e2_exit_declares_the_category_it_computed_on(monkeypatch) -> None:
"""`E-05`:计算型出口**直接声明**它算的是哪个类目,不经任何文本反解。
用**真实语料**(`knowledge/` 的费率总表)喂 `_answer_category_fee`,断言声明的
主语就是类目本身 —— 这条同时钉住「`E2` 声明的是语义主语,不是文本前缀」。
"""
handbook = (
Path(__file__).resolve().parents[3]
/ "knowledge" / "product" / "个人理财产品手册.md"
)
content = handbook.read_text(encoding="utf-8")
agent = build_agent()
async def _stub_search_parameters(query, intent, context, **_kwargs):
del query, intent, context
return [{"title": "公募基金费率总表", "content": content, "score": 0.9}]
monkeypatch.setattr(agent, "_search_parameters", _stub_search_parameters)
result = await agent._answer_category_fee(
"货币基金申购和赎回费率是多少", CUSTOMER, "货币基金"
)
# 声明值就是类目本身
assert result.topic == "货币基金"
# 且与旧的文本反解**相同** —— `E2a` 同样是等价重构(实测答复首行就是「货币基金:…」)
assert result.topic == agent._topic_of(result.text)
# ---------------------------------------------------------------------------
# `E-04` 适当性不匹配:主动确认路径(红线 2:先揭示、后确认)
# ---------------------------------------------------------------------------
def test_suitability_mismatch_tells_customer_consequences_not_confirmation() -> None:
"""不匹配(`allowed=False`)时只如实告知 + 引导找客户经理,**不要求确认**(无可确认之事)。"""
text = customer_service_module.CustomerServiceAgent._suitability_text(
"南方季季盈90天", 3,
{"allowed": False, "customer_risk_level": 1, "reason_code": "RISK_NOT_MATCH"},
)
assert "不匹配" in text
assert "无法购买" in text
assert customer_service_module.SUITABILITY_CONFIRM_REQUEST not in text
def test_suitability_disclosure_asks_confirmation_after_disclosure() -> None:
"""披露情形:**揭示在前、确认要求在后**,且客服明确不出手代替确认。"""
text = customer_service_module.CustomerServiceAgent._suitability_text(
"南方季季盈90天", 4,
{
"allowed": True,
"customer_risk_level": 3,
"reason_code": "SUITABLE_WITH_DISCLOSURE",
"required_disclosure": True,
},
)
confirm = customer_service_module.SUITABILITY_CONFIRM_REQUEST
assert confirm in text
# 顺序红线:揭示必须先于确认要求(顺序反了揭示即失效)
assert text.index("揭示") < text.index(confirm)
# 客服**不产生**客户确认动作 —— 话术里必须说清这一点
assert "不会代替您做任何确认" in text
def test_suitability_text_avoids_zero_tolerance_words() -> None:
"""红线复验:这套话术不得命中零容忍字面(否则会被治理层整条替换成合规兜底)。"""
from app.core.customer_service_rules import ZERO_TOLERANCE_WORDS
texts = [
customer_service_module.CustomerServiceAgent._suitability_text(
"产品A", 4,
{"allowed": True, "customer_risk_level": 3,
"reason_code": "SUITABLE_WITH_DISCLOSURE", "required_disclosure": True},
),
customer_service_module.CustomerServiceAgent._suitability_text(
"产品A", 3,
{"allowed": False, "customer_risk_level": 1, "reason_code": "RISK_NOT_MATCH"},
),
customer_service_module.CustomerServiceAgent._suitability_rule_text(
1, 3, "FORBIDDEN", visitor=False
),
]
for text in texts:
for word in ZERO_TOLERANCE_WORDS:
# 「安全」在合规话术里也不该出现;这里一律要求不出现
assert word not in text, f"话术命中零容忍字面:{word}"
def test_suitability_never_claims_customer_confirmation() -> None:
"""红线 2 的机器判据:客服答复里不得出现「您已确认」这类措辞。"""
text = customer_service_module.CustomerServiceAgent._suitability_text(
"产品A", 4,
{"allowed": True, "customer_risk_level": 3,
"reason_code": "SUITABLE_WITH_DISCLOSURE", "required_disclosure": True},
)
for phrase in ("您已确认", "已为您确认", "视为您已同意", "我们已确认"):
assert phrase not in text
# ---- `W20`:`E2c-my`(按**本人**权威等级列可购买产品) ----
def _eligible_view(*, reason: str = "AUTHORITY_OK", level: int | None = 1) -> dict:
"""`query_eligible_products` 的返回形状(与 `EligibleProductView` 字段一致)。"""
return {
"customer_id": "9001",
"customer_risk_level": level,
"risk_score": None,
"risk_level_source": "fin_risk_assessment",
"authority_reason": reason,
"assessment_valid_until": "2027-01-01T00:00:00Z",
"allowed_levels": ["R1", "R2"],
"disclosure_levels": [],
"products": [
{"product_code": "159700", "product_name": "科创债ETF南方",
"product_category": "ETF", "risk_level": "R2"},
{"product_code": "511810", "product_name": "货币ETF南方",
"product_category": "ETF", "risk_level": "R1"},
],
"excluded_count": 16,
}
def stub_eligible_tool(payload: dict) -> object:
"""替换 `call_tool`:只验证出口决策,不接工具执行器。"""
async def _call(name: str, arguments: dict, *, intent: str, context: RequestContext):
assert name == customer_service_module.ELIGIBLE_TOOL_NAME
assert intent == customer_service_module.ELIGIBLE_WHITELIST_INTENT
del arguments, context
return payload
return _call
async def test_route_and_answer_lists_products_within_the_customers_own_level() -> None:
"""接线验证:问本人等级的匹配范围必须**列出产品**,而不是答成「不能推荐」。"""
agent = build_agent()
agent.call_tool = stub_eligible_tool(_eligible_view()) # type: ignore[method-assign]
result = await agent._route_and_answer(
build_request("我现在可以买什么等级的产品"), CUSTOMER
)
assert ADVICE_BOUNDARY_REPLY not in result.text
assert "保守型(C1)" in result.text
assert "可购买 R1、R2 等级的产品" in result.text
assert "159700 科创债ETF南方" in result.text
assert "另有 16 只在售产品超出该范围,未列入" in result.text
# 陈述而非引导:清单里不得出现指向性措辞。
assert customer_service_module.promotional_wording_violation(result.text) is None
assert result.transfer_required is False
async def test_eligible_exit_fails_closed_without_an_authoritative_level() -> None:
"""取不到权威等级 ⇒ `E5b` 如实告知,**不猜范围、不建单、不列清单**。"""
agent = build_agent()
agent.call_tool = stub_eligible_tool( # type: ignore[method-assign]
{**_eligible_view(reason="ASSESSMENT_MISSING", level=None),
"allowed_levels": [], "products": [], "excluded_count": 0}
)
result = await agent._route_and_answer(build_request("我能买什么产品"), CUSTOMER)
assert result.transfer_required is False
assert "159700" not in result.text
assert "400-889-8899" in result.text
async def test_eligible_exit_tells_the_customer_when_the_assessment_expired() -> None:
agent = build_agent()
agent.call_tool = stub_eligible_tool( # type: ignore[method-assign]
{**_eligible_view(reason="ASSESSMENT_EXPIRED", level=1),
"allowed_levels": [], "products": [], "excluded_count": 0}
)
result = await agent._route_and_answer(build_request("我能买什么产品"), CUSTOMER)
assert "风险测评已过有效期" in result.text
assert result.transfer_required is False
async def test_visitor_gets_the_public_matrix_instead_of_a_product_list() -> None:
"""访客问同一句:给**公开规则表**(`DEC-I8`),不读任何画像数据。"""
agent = build_agent()
async def _forbidden(*args: object, **kwargs: object) -> dict:
raise AssertionError("访客侧不得调用可买清单工具")
agent.call_tool = _forbidden # type: ignore[method-assign]
result = await agent._route_and_answer(build_request("我能买什么等级的产品"), VISITOR)
assert "C1 保守型 R1—R2" in result.text
assert "需要先登录" in result.text
assert result.transfer_required is False
async def test_eligible_exit_degrades_to_scope_only_if_the_template_ever_turns_promotional() -> None:
"""回归哨兵:模板若被人加了引导语,出口必须降级为**只讲范围**、不列清单。"""
agent = build_agent()
# 取 `__dict__` 里的 **staticmethod 对象**再还原:读 `Class.attr` 拿到的是"解绑后的
# 函数",把它赋回类属性会静默变成**普通方法**(此后每个测试都少传一个 self 而报
# `TypeError`,且只在后续测试里炸,极难定位)。实测 2026-09-20 踩到。
original_descriptor = customer_service_module.CustomerServiceAgent.__dict__["_eligible_text"]
original = customer_service_module.CustomerServiceAgent._eligible_text
def _promotional(
level: int, output: dict, *, with_products: bool = True,
asked_level: int | None = None,
) -> str:
text = original(
level, output, with_products=with_products, asked_level=asked_level
)
return text + "\n建议您购买第一只。" if with_products else text
customer_service_module.CustomerServiceAgent._eligible_text = staticmethod(_promotional)
try:
agent.call_tool = stub_eligible_tool(_eligible_view()) # type: ignore[method-assign]
result = await agent._route_and_answer(build_request("我能买什么产品"), CUSTOMER)
finally:
customer_service_module.CustomerServiceAgent._eligible_text = original_descriptor
assert "建议您购买第一只" not in result.text
assert "159700 科创债ETF南方" not in result.text
assert "可购买 R1、R2 等级的产品" in result.text
# ---------------------------------------------------------------------------
# `W20` 展示层净化:`render_plain` / `drop_yield_claims` / `prettify_title`
#
# 为什么单独立一组:这三件事不改答案内容,只改"客户看到的样子"。它们是**低风险高收益**
# 的改动,但也最容易在后续重构里被无声改坏 —— 一旦 `render_plain` 开始吞字,判分会从
# "答对"变成"没答"(金标 `key_facts` 是纯词语子串匹配);一旦 `drop_yield_claims` 误伤
# 费率表,`B-02` 的「申购」「赎回」就没了。这里把这两个方向都钉死。
# ---------------------------------------------------------------------------
def test_render_plain_strips_markdown_but_keeps_the_words() -> None:
src = (
"### 第二条 适用范围\n\n"
"本指南适用于下列销售活动:\n"
"- 公募基金\n"
"* 银行理财产品\n"
"> 注:含货币基金\n"
"**加粗**与`代码`\n"
"---\n"
"| 费用类型 | 货币基金 |\n"
"|----------|----------|\n"
"| 申购费率 | 0 |\n"
)
out = customer_service_module.render_plain(src)
for noise in ("###", "**", "`", "---", "|----------|"):
assert noise not in out
assert "第二条 适用范围" in out
assert "· 公募基金" in out
assert "· 银行理财产品" in out
assert "注:含货币基金" in out
assert "加粗" in out and "代码" in out
assert "费用类型 | 货币基金" in out
assert "申购费率 | 0" in out
def test_render_plain_keeps_the_b02_fee_keywords() -> None:
"""回归钉子:金标 `B-02` 的 `key_facts` 是「申购」「赎回」两个纯词。"""
seed = CORPUS / "product" / "个人理财产品手册.md"
if not seed.exists():
pytest.skip("知识语料不在位")
out = customer_service_module.render_plain(seed.read_text(encoding="utf-8"))
assert "申购费率(原费率) | 0 | 0.80%" in out
assert "赎回费率(<7 天)" in out
def test_drop_yield_claims_removes_yield_numbers_but_keeps_the_fee_table() -> None:
src = (
"近一年收益率 7.60%\n"
"自成立以来年化 5.2%\n"
"| 申购费率(原费率) | 0 | 0.80% |\n"
"本产品历史业绩不预示未来表现\n"
)
out = customer_service_module.drop_yield_claims(src)
assert "7.60%" not in out
assert "5.2%" not in out
assert "申购费率(原费率) | 0 | 0.80%" in out
assert "本产品历史业绩不预示未来表现" in out
def test_drop_yield_claims_keeps_the_benchmark_formula_but_drops_both_word_orders() -> None:
"""`A-06` 回归钉子:业绩比较基准里的 `×60%` 是**权重**,不是收益数值。
实测 2026-09-21(`W25` 全量复跑):`FAQ-0022` 的「答:…」整行同时含「业绩」与
「收益率×60%」,旧判据把整行删掉 —— 客户只收到光秃秃的「问:什么是业绩比较基准?」。
这里把两个方向同时钉死:**公式保留**、**两种语序的真实收益数值照删**。
"""
src = (
"问:什么是业绩比较基准?\n"
"答:业绩比较基准是产品设定的参考收益标准,不是对投资者的收益承诺。"
"例如「沪深300指数收益率×60%+中证全债指数收益率×40%」。实际收益可能高于或低于业绩比较基准。\n"
"近一年收益率 7.60%\n"
"7.60%的近一年收益率在同类中靠前\n"
)
out = customer_service_module.drop_yield_claims(src)
assert "沪深300指数收益率×60%+中证全债指数收益率×40%" in out
assert "不是对投资者的收益承诺" in out
assert "7.60%" not in out
def test_prettify_title_drops_the_doc_name_segment_and_dangling_number() -> None:
long_title = "南方基金管理股份有限公司 公募基金与专户产品手册 · 一、公募基金产品 · 1."
assert customer_service_module.prettify_title(long_title) == "一、公募基金产品"
def test_prettify_title_falls_back_to_chapter_when_nothing_else_is_left() -> None:
out = customer_service_module.prettify_title("南方基金服务协议 · 第三章 费用与税收")
assert "第三章 费用与税收" in out
assert len(customer_service_module.prettify_title("标" * 80)) <= 35
# ---------------------------------------------------------------------------
# `E2c-my` 点名档位时的**直接裁决**(`W20`)
#
# 实测缺陷:客户问「我可以买 R3 的产品吗」(第一人称 + 点名档位,问句里没有 C 等级、
# 于是 `E2c` 的 `is_general_suitability_question` 不触发),出口只回了一份
# 「您可购买 R1、R2」的清单 —— **那个"不"字始终没说出来**。金标 `D-04` 要的正是
# 「不能 / 不可以」。这里把"裁决 + 范围"的顺序钉死。
# ---------------------------------------------------------------------------
async def test_named_product_level_gets_a_direct_verdict_before_the_list() -> None:
agent = build_agent()
agent.call_tool = stub_eligible_tool(_eligible_view()) # type: ignore[method-assign]
result = await agent._route_and_answer(build_request("我可以买 R3 的产品吗"), CUSTOMER)
assert "不可以购买" in result.text
assert "R3(中风险)" in result.text
# 裁决在前、范围在后:客户先拿到答案,再拿到依据。
assert result.text.index("不可以购买") < result.text.index("可购买 R1、R2 等级的产品")
assert customer_service_module.promotional_wording_violation(result.text) is None
assert result.transfer_required is False
async def test_named_product_level_inside_the_scope_is_answered_yes() -> None:
agent = build_agent()
agent.call_tool = stub_eligible_tool(_eligible_view()) # type: ignore[method-assign]
result = await agent._route_and_answer(build_request("我能买 R2 的产品吗"), CUSTOMER)
assert "可以购买" in result.text
assert "不可以购买" not in result.text
async def test_named_disclosure_level_says_the_disclosure_requirement() -> None:
view = _eligible_view()
view["allowed_levels"] = ["R1", "R2"]
view["disclosure_levels"] = ["R3"]
agent = build_agent()
agent.call_tool = stub_eligible_tool(view) # type: ignore[method-assign]
result = await agent._route_and_answer(build_request("我可以买 R3 的产品吗"), CUSTOMER)
assert "需签署产品风险揭示书后可以购买" in result.text
async def test_scope_only_question_has_no_verdict_line() -> None:
"""没点名档位时**不得**多出一句裁决(问的是范围,答的就是范围)。"""
agent = build_agent()
agent.call_tool = stub_eligible_tool(_eligible_view()) # type: ignore[method-assign]
result = await agent._route_and_answer(build_request("我想买点理财产品"), CUSTOMER)
assert "您问的 R" not in result.text
assert "可购买 R1、R2 等级的产品" in result.text
# ---------------------------------------------------------------------------
# 展示层净化的**空结果护栏**(`W20`)
#
# `drop_yield_claims` 的判据是"整行含收益数值就不输出",而语料里有 15 个切片**整个块
# 只写了一个收益数字**(如「近三年收益率 11.85%(虚构)」)。净化后内容为空时若照常
# 返回,客户拿到的是一条**空气泡** —— 比答错更糟。这里把"清空即回退"钉死。
# ---------------------------------------------------------------------------
YIELD_ONLY_HIT = {
"doc_id": "PROD-001-08", "score": 0.91,
"title": "南方基金管理股份有限公司 公募基金与专户产品手册"
" · 一、公募基金产品 · 1.1 南方现金添利货币市场基金〔示例〕",
"content": "南方现金添利货币市场基金〔示例〕:七日年化收益率 约 1.92%(近 30 日均值,虚构)",
}
async def test_a_yield_only_chunk_never_becomes_an_empty_bubble() -> None:
agent = build_agent()
agent.call_tool = stub_knowledge_tool([YIELD_ONLY_HIT]) # type: ignore[method-assign]
result = await agent._answer_from_knowledge(
build_request("南方现金添利这只产品怎么样"), CUSTOMER, customer_service_module.INTENT_PRODUCT
)
assert result.text.strip()
assert "1.92%" not in result.text
assert result.transfer_required is False
def test_partial_exit_skips_a_chunk_that_the_sanitizer_emptied() -> None:
"""高分块被净化清空时,展示**下一个有内容的块**,而不是给一个空气泡。"""
result = build_agent()._exit_partial(
[
{"score": 0.80, "doc_id": "PROD-001-08",
"content": "南方现金添利货币市场基金〔示例〕:七日年化收益率 约 1.92%(虚构)"},
{"score": 0.62, "doc_id": "PROD-018", "content": "### 6.3 费用计算示例"},
]
)
assert "6.3 费用计算示例" in result.text
assert "1.92%" not in result.text
def test_partial_exit_falls_back_to_empty_template_when_everything_is_yield() -> None:
result = build_agent()._exit_partial([YIELD_ONLY_HIT])
assert "1.92%" not in result.text
assert "没找到" in result.text
assert result.transfer_required is False
# ---------------------------------------------------------------------------
# `W21`:智能度体检(55 条真实口语问法)实测出的四条短板的守卫
# ---------------------------------------------------------------------------
def test_topic_parser_rejects_a_narrative_lead() -> None:
"""`W21`:上一轮答复的首行「需要说明三点:」不是主语。
实测事故链:客户问「南方稳健增利债券 A 的风险等级是啥」→(旧实现被**改等级**红线
误拦)回答的首行是「您的风险等级(C1—C5)只能由…」、第二行是「需要说明三点:」——
旧 `_topic_in` 取到「需要说明三点」当主语,第二问「那费率呢」的检索词被污染成
「需要说明三点 那费率呢」,客户拿到的是「其他费用:专户业绩报酬」这种不相干的条款。
"""
agent = build_agent()
assert agent._topic_in("需要说明三点:") == ""
assert agent._topic_in("其他费用:专户业绩报酬") == ""
assert agent._topic_in("情况:如下") == ""
# 正常主语不受影响(整节块的标题行 / 行级子块的首段)
assert agent._topic_in("1.2 南方稳健增利债券 A〔示例〕") == "南方稳健增利债券 A〔示例〕"
assert agent._topic_in("南方稳健增利债券 A〔示例〕:风险等级 R2(中低风险)") == (
"南方稳健增利债券 A〔示例〕"
)
def test_search_query_falls_back_to_the_previous_customer_question() -> None:
"""`W21`:主语**取不到**时(上一轮是 `E4` 合并生成、不声明主语)退回上一位客户问句。
实测:「我想了解定投」(`E4`)→「最低多少钱」原先整句单独检索,候选是
「第九条 问卷内容及评分标准 / 第十六条 费率标准」—— 客户只是追问上一轮的金额门槛。
"""
agent = build_agent()
history = (
ConversationTurn(role="user", content="我想了解定投"),
ConversationTurn(role="assistant", content="关于定投,目前【证据】中只有…"),
)
assert agent._search_query(
build_request("最低多少钱", history=history)
) == "我想了解定投 最低多少钱"
def test_parameterized_followup_inherits_the_predicate_from_the_previous_question() -> None:
"""`W21`:「持有 8 个月呢」只有**参数**,谓词在上一轮客户问句里。"""
agent = build_agent()
history = (
ConversationTurn(role="user", content="赎回费怎么算"),
ConversationTurn(role="assistant", content="赎回费率按持有时间分档计算…"),
)
assert agent._search_query(
build_request("持有 8 个月呢", history=history)
) == "赎回费怎么算 持有 8 个月呢"
def test_parameterized_followup_does_not_hijack_a_self_contained_question() -> None:
"""判据必须窄:自己带谓词的问句(`D-03`)不得被改写成上一位问句。"""
agent = build_agent()
history = (ConversationTurn(role="user", content="持有 8 个月赎回要付费吗?"),)
assert agent._search_query(
build_request("持有 8 个月赎回要付费吗?", history=history)
) == "持有 8 个月赎回要付费吗?"
def test_general_knowledge_gate_only_opens_for_concept_and_timing_questions() -> None:
"""常识集合补位只对"概念 / 时效 / 品类风险"型问句开放(否则会抢答产品题)。"""
agent = build_agent()
for message in (
"基金和股票有啥区别", "基金分红是怎么回事", "什么是夏普比率", "最大回撤是啥",
"前端收费和后端收费的区别", "什么时候能卖", "周末能买吗", "货币基金会不会亏",
):
assert agent._is_general_knowledge_question(message) is True, message
for message in (
"南方稳健增利债券 A 的起投金额是多少?", "我是 C1,能买 R3 的产品吗?",
"买 10 万股票基金,申购费大概多少?", "南方稳健增利债券 A 的费率是多少?",
"基金申购和赎回有哪些费率?", "帮我挑一只收益最高的基金",
):
assert agent._is_general_knowledge_question(message) is False, message
def test_partial_exit_gate_drops_a_hit_with_no_shared_business_term() -> None:
"""`W21` E5b 相关性闸门:「什么时候能卖」不该把「第二十七条 生效日期」贴给客户。
判据看**标题 + 正文前 200 字**与问句的最长公共子串:连连续两个字都不重合,
说明这一块与问句没有任何共同业务词,拿它充数就是"答非所问"。
"""
agent = build_agent()
irrelevant = {
"title": "南方基金产品销售管理办法 · 第九章 附则 · 第二十七条 生效日期",
"content": "### 第二十七条 生效日期\n\n本办法自 **2026 年 9 月 1 日**起施行。",
}
relevant = {
"title": "南方基金产品销售管理办法 · 第三章 · 第三十四条 费用与费率管理",
"content": "其他费用:专户业绩报酬 按合同约定计提",
}
assert agent._hits_share_terms("什么时候能卖", irrelevant) is False
assert agent._hits_share_terms("费用怎么收", relevant) is True
# ---------------------------------------------------------------------------
# `W21` C-8 / C-9:账户盈亏问法的口径,以及多轮指代的**降级链**
# ---------------------------------------------------------------------------
def test_product_name_shape_extracts_the_product_from_a_previous_faq_answer() -> None:
"""`W21` C-9 第一级降级:上一轮是 FAQ 型答复(首行「问:…」)时仍要认出产品名。
实测:「我想买个债基」→「它适合我吗」——`_topic_of` 取不到主语(首行是「问:…」),
旧实现回一句「请告诉我具体的基金名称或代码」,而客户**刚刚**才被告知是那只产品。
把已经说过的话再问一遍,是"客服不智能"最直观的形态。
"""
agent = build_agent()
history = (
ConversationTurn(role="user", content="我想买个债基"),
ConversationTurn(
role="assistant",
content=(
"问:你们有没有债券型基金(债基)?\n"
"答:有。本公司旗下公募基金按类型覆盖货币市场基金、债券型基金…;"
"其中债券型基金的代表产品是南方稳健增利债券 A〔示例〕,风险等级 R2(中低风险)。"
),
),
)
assert agent._product_name_in_history(
build_request("它适合我吗", history=history)
) == "南方稳健增利债券 A"
def test_product_name_shape_does_not_mistake_the_company_for_a_product() -> None:
"""公司名不是产品名:判据要求「南方 + 名称 + **产品类型后缀**」。
没有后缀约束的话,「南方基金」「南方基金管理股份有限公司」都会被当产品,
而它们在语料里出现频率最高 —— 指代会被稳定地解析到错误对象上。
"""
agent = build_agent()
for answer in (
"南方基金管理股份有限公司成立于 1998 年,是经中国证监会批准设立的基金管理公司。",
"南方基金支持定投,最低每期 100 元起,可在 APP「定投管理」中设置。",
):
history = (
ConversationTurn(role="user", content="南方基金是什么"),
ConversationTurn(role="assistant", content=answer),
)
assert agent._product_name_in_history(
build_request("它适合我吗", history=history)
) == "", answer
async def test_suitability_exit_defers_a_pure_parameter_followup_to_knowledge() -> None:
"""`W21` C-9 前置闸门:纯参数追问**无条件**交回知识检索。
事故链:上一轮答复里提过「南方稳健增利债券 A」⇒ 形状法捞出产品名 ⇒
本出口给出一段"这只基适不适合你"的裁决,而客户问的是**持有 8 个月要交
多少赎回费**。从"答不上来"变成"答错题" —— 金融场景里后者更糟。
"""
agent = build_agent()
seen: list[str] = []
async def fake_knowledge(request, context, intent): # noqa: ANN001, ANN202
seen.append(intent)
return CoreResult(text="(知识检索答复)")
agent._answer_from_knowledge = fake_knowledge # type: ignore[method-assign]
history = (
ConversationTurn(role="user", content="赎回费怎么算"),
ConversationTurn(
role="assistant",
content="南方稳健增利债券 A〔示例〕为:持有<7 天 1.5%、7—30 天 0.75%。",
),
)
result = await agent._answer_suitability(
build_request("持有 8 个月呢", history=history), CUSTOMER
)
assert seen == [customer_service_module.INTENT_FAQ]
assert result.text == "(知识检索答复)"
async def test_suitability_exit_still_clarifies_on_a_first_turn_pronoun() -> None:
"""首轮「它费率多少?」没有指代对象 ⇒ 必须澄清,**不得**降级到知识检索。
金标 `E-01` 期望正是 `E1`:首轮指代是无解的,问清比乱猜好。
"""
agent = build_agent()
async def boom(*args, **kwargs): # noqa: ANN002, ANN003, ANN202
raise AssertionError("首轮指代没有指代对象,不该走知识检索")
agent._answer_from_knowledge = boom # type: ignore[method-assign]
result = await agent._answer_suitability(build_request("它费率多少?"), CUSTOMER)
assert result.clarification_required is True
assert result.transfer_required is False
# ---------------------------------------------------------------------------
# `W21-D1`~`D4`:甲方四项待决的落地守卫
# ---------------------------------------------------------------------------
def test_evidence_prompt_bans_yield_numbers_in_the_generated_answer() -> None:
"""`W21-D1`(`C-11`):根因在**生成侧**,所以修法是把禁令写进提示词。
三条实测问句(`南方现金添利怎么样` / `买基金要手续费吗` / `债基和货基哪个收益高`)
检索**完全正确**,生成稿里却出现收益数值,被 `hits_zero_tolerance` **整条**拦回 `E5b`。
改闸门(先净化再判合规)已被实测证伪(整行丢弃会把单行长段整条删空),
因此必须在源头说清楚"不要写数字"。
"""
template = customer_service_module.DEFAULT_EVIDENCE_TEMPLATE
assert "不要出现任何收益数值" in template
# 模板必须仍能被 `_answer_from_evidence` 正常 format:裸花括号会让运行期回落内置模板,
# 于是这段禁令**静默失效**——比不写更糟。
rendered = template.format(evidence="【证据】", message="测试问句")
assert "【证据】" in rendered
assert "测试问句" in rendered
def test_output_side_yield_check_is_not_relaxed_by_d1() -> None:
"""`W21-D1` 只改**生成侧话术**,输出侧红线一条没松(安全边界不退让)。"""
from app.core.customer_service_rules import hits_zero_tolerance
assert hits_zero_tolerance("该产品近一年收益率 11.85%。") is True
# 「问含义」照旧放行(`A-05`/`F-04` 要求「什么叫七日年化」必须能答)。
assert hits_zero_tolerance("什么叫七日年化?") is False
def test_partial_exit_uses_an_answering_tone() -> None:
"""`W21-D4`:`E5b` 开场白从"我把找到的资料放上来"改成"作答"语气。
内容一字未改,只换开场 —— 客户对"这个客服会不会答"的判断完全不同。
"""
result = build_agent()._exit_partial(
[{"score": 0.6, "content": "基金申购费率按金额分档。"}], note="置信度不足"
)
assert "关于这一点,公开资料里的口径是:" in result.text
assert "我先帮您把找到的公开资料放上来" not in result.text
assert result.transfer_required is False
def test_partial_exit_answers_a_faq_pair_directly() -> None:
"""`W21-D4`:命中块本身就是一条 FAQ 问答对时,**以答案正文开场**,不套前言。
实测「债基和货基哪个收益高」命中的 `FAQ-0068` 正文就是标准答案,
却因为兜底式开场显得"客服不会答"。
"""
content = (
"问:债券型基金和货币市场基金哪个收益高?\n"
"答:两者是不同风险收益特征的品类,本公司不提供这类比较结论。"
)
result = build_agent()._exit_partial(
[{"score": 0.7, "content": content}], note="置信度不足"
)
assert result.text.startswith("问:债券型基金和货币市场基金哪个收益高?")
assert "关于这一点,公开资料里的口径是" not in result.text
assert "400-889-8899" in result.text
def test_knowledge_miss_judgement_survives_the_template_change() -> None:
"""`W21-D4` 的连带风险:`F-2`/`G-05` 的转人工判据锚在 `KNOWLEDGE_MISS_TEXTS` 上。
改开场白必须让判据**同源移动**,否则会出现「模板改了、判据静默失效」——
该转人工的不转、或不该转的乱转。
"""
agent = build_agent()
for text in customer_service_module.KNOWLEDGE_MISS_TEXTS:
assert agent._knowledge_missed(CoreResult(text=text)) is True
assert agent._knowledge_missed(CoreResult(text="这是一段有实质内容的答复")) is False
def test_product_names_covers_both_corpus_shapes() -> None:
"""`W21-D3`:产品名在语料里有**两种真实写法**,只认一种会漏掉整类 ETF。"""
agent = build_agent()
assert agent._product_names("科创债ETF南方怎么样") == {"科创债ETF南方"}
assert agent._product_names("货币ETF南方和公司债ETF南方哪个好") == {
"货币ETF南方",
"公司债ETF南方",
}
# 归一化:语料里带不带空格都有,同一个名字必须判成同一只。
assert agent._product_names("南方稳健增利债券 A 的费率") == {"南方稳健增利债券A"}
# 类目词不是产品名;公司名也不是(后缀约束在起作用)。
assert agent._product_names("我想买个债基") == set()
assert agent._product_names("南方基金是什么公司") == set()
#: `docs/43` 场内基金手册的 20 只产品(13 ETF + 7 LOF),逐字取自产品清单。
#: 这份清单是**语料驱动的守卫**:手册新增一只名字形状不同的产品(例如又出现一种
#: 没有「南方」字样的写法),这条就会红,而不是等到客户问它时才在检索里暴露。
ONBOARD_MANUAL_PRODUCT_NAMES = (
"沙特ETF南方", "创业板人工智能ETF南方", "通信ETF南方", "恒生生物科技ETF南方",
"亚太精选ETF南方", "科创债ETF南方", "创业板ETF南方", "沪深300ETF",
"中证500ETF南方", "公司债ETF南方", "货币ETF南方", "红利低波50ETF南方",
"科创芯片ETF南方", "南方积极配置混合(LOF)", "南方新兴消费增长股票(LOF)A",
"南方金利定开债券A", "南方优势产业(LOF)", "南方创业板2年定期开放混合",
"南方原油A", "南方瑞合定开混合(LOF)",
)
def test_product_names_cover_every_onboard_manual_product() -> None:
"""`W23`:`docs/43` 入库后,手册里**每一只**产品都要能被认出来。
为什么单立一条:`W21-D3` 的一致性闸门(`_names_other_product`)以 `asked` 是否
为空为开关。名字认不出 ⇒ `asked` 是空集 ⇒ 闸门**静默失效**:不会造出假"没找到",
但客户问这只产品、系统却答另一只时,没有任何东西会拦住它。
LOF 的三种写法(`混合(LOF)` / `股票(LOF)A` / `定期开放混合`)与唯一没有厂商字样的
`沪深300ETF`,正是 `W21` 的后缀集收不到的四类。
"""
agent = build_agent()
missed = [name for name in ONBOARD_MANUAL_PRODUCT_NAMES
if name not in agent._product_names(name)]
assert missed == []
# 整句问法(含前缀噪声)同样要认出来,不能只在"名字单独出现"时成立。
assert agent._product_names("我想买南方积极配置混合(LOF)") == {"南方积极配置混合(LOF)"}
assert agent._product_names("南方原油A和南方金利定开债券A哪个好") == {
"南方原油A",
"南方金利定开债券A",
}
def test_the_brandless_product_does_not_open_the_gate_on_a_category_question() -> None:
"""`W23` 的**反向**守卫:把 `ETF` 放成通用后缀会造出假"没找到"。
`买ETF还是买LOF` 里被吞出来的「买ETF」不是产品名;一旦它进了 `asked`,
命中块里当然找不到这个名字 ⇒ 客户问一个**正常的品类比较题**却被告知"没有公开资料"。
所以品牌缺省的写法只能用枚举(`_BRANDLESS_PRODUCT_NAMES`),不能用规律。
"""
agent = build_agent()
assert agent._product_names("买ETF还是买LOF好") == set()
assert agent._product_names("沪深300ETF和科创债ETF南方哪个好") == {
"沪深300ETF",
"科创债ETF南方",
}
# 假"没找到"的真实形态:留声机式的"前一只 + 后一只"连问。
# 左起 `沪深300ETF` 的 `ETF` 会被当成第二只的前缀,切掉之后才剩「货币ETF南方」。
assert agent._product_names("沪深300ETF和货币ETF南方哪个好") == {
"沪深300ETF",
"货币ETF南方",
}
def test_product_names_survive_leading_noise_in_the_question() -> None:
"""`W21-D3`:正则没有左边界,**动词 / 连接词会被吞进产品名**,必须剥干净。
吞进来的后果不是"少判一次",而是**造出假「没找到」**:`asked` 变成了
「我想买科创债ETF南方」,命中块里当然找不到这个名字,于是客户明明问在库的产品
却被回一句"没有公开资料"。这条是有针对性的回归守卫。
"""
agent = build_agent()
assert agent._product_names("我想买科创债ETF南方怎么样") == {"科创债ETF南方"}
assert agent._product_names("货币ETF南方和公司债ETF南方哪个好") == {
"货币ETF南方",
"公司债ETF南方",
}
assert agent._product_names("申购南方平衡优选混合要多少钱") == {"南方平衡优选混合"}
# 最短的 ETF 名(2 字 + 后缀)不得被剥到失真。
assert agent._product_names("买货币ETF南方") == {"货币ETF南方"}
def test_product_mismatch_gate_catches_answering_with_another_product() -> None:
"""`W21-D3`:问 A 却拿 B 的产品卡 ⇒ 判"答非所问",`E5b` 宁可说没找到。"""
agent = build_agent()
hits = [
{
"title": "南方基金管理股份有限公司 公募基金与专户产品手册 · 一、公募基金产品 · 1.",
"content": "1.2 南方稳健增利债券 A〔示例〕\n风险等级 R2(中低风险)",
},
{"title": "… · 四、产品对比与适当性匹配", "content": "C1 可购买 R1、R2。"},
]
# 命中块讲的是**别的**产品 ⇒ 启用闸门。
assert agent._names_other_product("科创债ETF南方怎么样", hits) is True
# 命中块讲的就是问句里那只(哪怕带空格差异)⇒ 不判答非所问。
assert agent._names_other_product("南方稳健增利债券 A 怎么样", hits) is False
# 问句没点名具体产品(问的是类目)⇒ 闸门不启用。
assert agent._names_other_product("我想买个债基", hits) is False
# 命中块讲的是通用条款、一个产品名都没提 ⇒ 闸门不启用(通用条款本来就该照答)。
generic = [{"title": "第五章 费用与费率管理", "content": "费用按金额分档收取。"}]
assert agent._names_other_product("科创债ETF南方怎么样", generic) is False
async def test_suitability_exit_states_the_inferred_referent() -> None:
"""`W21-D2`:主语是**反解**出来的时必须摆出指代依据,客户认错要能立刻纠正。"""
agent = build_agent()
async def fake_risk_level(product: str, context: RequestContext) -> int:
return 2
async def fake_call_tool(name, arguments, *, intent, context): # noqa: ANN001, ANN202
return {
"allowed": True,
"reason_code": "SUITABLE",
"required_disclosure": False,
"requires_confirmation": False,
"requires_recording": False,
"customer_risk_level": 3,
"assessment_valid_until": None,
}
agent._product_risk_level = fake_risk_level # type: ignore[method-assign]
agent.call_tool = fake_call_tool # type: ignore[method-assign]
history = (
ConversationTurn(role="user", content="我想买个债基"),
ConversationTurn(
role="assistant",
content=(
"问:你们有没有债券型基金(债基)?\n"
"答:有。代表产品是南方稳健增利债券 A〔示例〕,风险等级 R2(中低风险)。"
),
),
)
result = await agent._answer_suitability(
build_request("它适合我吗", history=history), CUSTOMER
)
assert "您上一轮提到的是「南方稳健增利债券 A」" in result.text
assert "可以购买" in result.text