Files
group_fqcd_jr/app/core/conversation_privacy.py
T
张胜宇 e239eb778b docs: 品牌全量口径统一为「南方基金」+ 作废文档清理
1) 客服 Agent 四份交付文档 + 构建脚手架:品牌由包装占位 XX科技 / 旧名 南方财富
   统一为南方基金(热线 400-889-8899 / 官网 nffund.com),系统名改为「智能服务系统」;
   同步追加 §0.4 修订记录行,工程记录行保留原占位字面以支撑硬编码扫描验收。
2) 开发文档:清理 28 份已作废/残留文档(14 份移出归档 + 14 份仓库副本),
   新增《文档规整方案与开发前待决事项-2026-09-17》。
3) 客服agent 四份交付文档首次纳入本分支。
2026-09-17 15:15:22 +08:00

33 lines
1.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""客服会话落库前的敏感凭据最小化处理。
来源:同事 `ZSY_develop` 分支(`app/core/conversation_privacy.py`),整文件移植,未改语义。
纯正则、无外部依赖,供记忆投影(`MilvusProfileProjection`)在写入向量前做最后一道脱敏。
为什么要单独抽一层:记忆内容最终会进入 Milvus 与 Neo4j,一旦写入就脱离了会话事务的
管控范围。把脱敏放在**写入适配器内部**(而不是依赖调用方记得做),是为了让"未经脱敏的
文本不得落外部存储"成为代码保证,而不是流程约定。
"""
import re
# 替换顺序从带业务语义的凭据开始,避免通用数字规则先破坏上下文。
_SENSITIVE_PATTERNS: tuple[tuple[re.Pattern[str], str], ...] = (
(
re.compile(r"(?i)((?:登录|交易)?密码)\s*(?:[::=]|是)\s*[^\s,。;,;]{1,64}"),
r"\1[已隐藏]",
),
(re.compile(r"(?i)((?:登录|交易)?密码)\s*\d{4,32}"), r"\1[已隐藏]"),
(re.compile(r"(?i)(验证码|短信码|校验码)\s*(?:[::=]|是)?\s*\d{4,8}"), r"\1[已隐藏]"),
(re.compile(r"(?<!\d)\d{17}[\dXx](?!\d)"), "[证件号已隐藏]"),
(re.compile(r"(?<!\d)(?:\d[ -]?){15,18}\d(?!\d)"), "[银行卡号已隐藏]"),
(re.compile(r"(?<!\d)1[3-9]\d{9}(?!\d)"), "[手机号已隐藏]"),
)
def sanitize_customer_service_message(message: str) -> str:
"""保留风险关键词,移除不应进入会话、Outbox 或后续 Redis 的凭据值。"""
sanitized = message
for pattern, replacement in _SENSITIVE_PATTERNS:
sanitized = pattern.sub(replacement, sanitized)
return sanitized