Files
group_fqcd_jr/app/service/profile_assembly_service.py
T
lzf_0626 962a0a116f feat: 记忆→画像打通(事实提升 + 画像组装 + 版本快照)
补齐"记忆系统为画像服务"的断链,按 docs/23 的分层设计实现后三层。

1. 新增 app/model/profile.py:user_facts 与 profile_snapshots 的 ORM 映射。此前这两张表
   只有结构、没有 Model,实际没有任何代码在用。两处表结构特例在 docstring 里显式标注,
   避免后续有人按直觉写入踩坑:
   · user_facts.id 无 auto_increment,主键必须由应用提供(本实现用微秒时间戳,单调递增);
   · profile_snapshots.current_customer_id 是生成列(IF(is_current=1, customer_id, NULL)),
     故意不映射——映射了反而会在写入时与之冲突。

2. 新增 app/service/profile_assembly_service.py,三段职责:
   · 事实提升(中期→长期):evidence_count ≥ 2 或 confidence ≥ 0.90 才从 memory_unit
     提炼进 user_facts —— 这条门槛就是"客户随口一说不能变成画像结论"的落地方式;
   · 画像组装(长期→画像):按白名单映射进 fin_customer_profile,未列入白名单的事实
     (如 profile:family)只进 user_facts,保证画像的信噪比;
   · 版本留痕:每次重建写一条 profile_snapshots,generation_basis 逐字段记录来源,
     用于回答"当时凭什么这么判断"。

3. 新增 tools/rebuild_profile.py:手工触发入口(单客户或 --all)。画像暂无自动触发,
   这是目前唯一的重建方式,也便于排查"画像为什么没更新"。

红线由代码保证而非约定:investor_type 只从 fin_risk_assessment 最新一条读取,实现中
不存在任何记忆路径能写它。实测——客户 9001 问卷为 C2、对话自述"稳健型",重建后
investor_type 仍为 C2,自述信息进入 risk_tags 并标注"自述:"前缀。三方不一致保持可见,
但等级判定只认问卷,客户无法靠对话改变自己的可购范围。

另一处由实测修正的设计:fin_customer_profile 的 trade_account/real_name/total_asset/
behavior_score 均为 NOT NULL,说明画像行由开户流程创建(也印证了"注册时填问卷"是开户
前置条件)。原先"首次重建时创建画像行"的做法是错的——会写出一条假的开户记录,而画像
恰恰是风控要读的数据。已改为只更新已存在的画像,未开户时返回 reason=profile_row_not_opened
并如实报告,而不是静默成功。

同时新增 docs/23-记忆分层与画像设计.md:短期/中期/长期/画像四层各自存在哪里、谁写、
提升门槛、是否进画像,以及三条路径(问卷/行为/对话)在画像层汇合的设计。

验证:ruff 通过、mypy 109 文件无错;tools/rebuild_profile.py 对客户 9001 连续两次重建
产生 version=1/2 两条快照且 is_current 正确轮转(旧版本置 0)。
2026-09-10 21:36:23 +08:00

260 lines
11 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""画像组装:中期记忆 → 长期事实 → 画像 + 版本快照。
这是"记忆系统为画像服务"的落地环节。三段职责:
1. **事实提升(中期 → 长期)**:把 `memory_unit` 里证据足够的记忆提炼进 `user_facts`。
门槛是 `evidence_count >= 2` **或** `confidence >= 0.90` —— 这条门槛就是
"客户随口一说不能变成画像结论"的落地方式。
2. **画像组装(长期 → 画像)**:把 `user_facts` 按**白名单**映射进 `fin_customer_profile`。
未列入白名单的事实只进 `user_facts`,不进画像,避免画像被噪声撑大。
3. **版本留痕**:每次重建写一条 `profile_snapshots`,并用 `generation_basis` 记录
**每个字段分别来自哪里**——风控与合规复盘时要能回答"当时凭什么这么判断"。
## 两条必须由代码保证的红线
- **`investor_type` 只来自问卷测评**(`fin_risk_assessment` 最新一条)。下面的实现里
它只从问卷查询取数,任何记忆路径都碰不到它。客户在对话里说"我是激进型"不会改变它
——这是合规底线,不能只靠约定。
- **按字段所有权写入**:交易侧的客观字段(`total_asset`/`trading_frequency`/`behavior_score`)
本服务**不写**,留给交易模块,避免两个模块抢写同一列。
"""
import json as _json
from datetime import UTC, datetime
from hashlib import sha256
from typing import Any
from uuid import uuid4
from sqlalchemy import or_, select, text
from sqlalchemy.ext.asyncio import AsyncSession
from app.model.fund import FundCustomerProfile
from app.model.memory import MemoryUnit
from app.model.profile import ProfileSnapshot, UserFact
# 提升门槛
MIN_EVIDENCE = 2
HIGH_CONFIDENCE = 0.90
# 事实键 → 画像字段的**白名单**映射。没列在这里的事实(如 profile:family)只进 user_facts,
# 不进画像字段——画像要保持"能直接支撑决策"的信噪比。
FACT_TO_PROFILE_FIELD: dict[str, str] = {
"preference:asset_class": "preferred_asset_class",
"preference:horizon": "investment_horizon",
}
# 自述类事实(客户自己说的偏好)统一进 risk_tags,并标注来源为"自述"。
# 保留它们的价值在于:当出现「问卷 C4 / 自述稳健 / 行为买 R4」三方不一致时,
# 这种矛盾本身就是风控信号——但绝不能与问卷等级混进同一个字段。
SELF_REPORTED_PREFIXES = ("preference:risk_level", "preference:", "profile:")
# 关键事实:参与决策,标记出来便于下游优先读取
CRITICAL_FACTS = frozenset({
"preference:risk_level", "preference:horizon", "preference:asset_class",
})
# 画像中允许本服务写入的字段(其余字段归交易/注册侧所有)
PROFILE_OWNED_FIELDS = ("investor_type", "preferred_asset_class", "investment_horizon", "risk_tags")
def _now() -> datetime:
return datetime.now(UTC).replace(tzinfo=None)
def _fact_id() -> int:
"""`user_facts.id` 没有 auto_increment,主键由应用生成。
用微秒时间戳:单调递增、无需额外序列、同客户同微秒重复在单进程写入下不可能发生。
"""
return int(datetime.now(UTC).timestamp() * 1_000_000)
class ProfileAssemblyService:
def __init__(self, session: AsyncSession) -> None:
self.session = session
# ---------- 中期 → 长期 ----------
async def promote_facts(self, customer_id: int) -> list[str]:
"""把证据足够的记忆提炼为长期事实;返回本次提升的事实键。"""
now = _now()
rows = list(await self.session.scalars(
select(MemoryUnit).where(
MemoryUnit.customer_id == customer_id,
MemoryUnit.status == "active",
or_(
MemoryUnit.evidence_count >= MIN_EVIDENCE,
MemoryUnit.confidence >= HIGH_CONFIDENCE,
),
or_(MemoryUnit.valid_until.is_(None), MemoryUnit.valid_until > now),
)
))
promoted: list[str] = []
for memory in rows:
key = str(memory.memory_key)
value = self._fact_value(memory)
existing = await self.session.scalar(
select(UserFact).where(
UserFact.customer_id == customer_id, UserFact.fact_key == key
)
)
if existing is None:
self.session.add(UserFact(
# 主键显式赋值:该表无 auto_increment
id=_fact_id(),
customer_id=customer_id,
fact_key=key,
fact_value=value,
source_portal=str(memory.source_type or "conversation"),
source_episode_id=None,
confidence=float(memory.confidence or 0.0),
is_critical=key in CRITICAL_FACTS,
created_at=now,
))
else:
existing.fact_value = value
existing.confidence = float(memory.confidence or 0.0)
existing.is_critical = key in CRITICAL_FACTS
promoted.append(key)
await self.session.flush()
return promoted
@staticmethod
def _fact_value(memory: MemoryUnit) -> Any:
"""事实值优先取结构化值,回退到正文;始终以 JSON 可存的形式返回。"""
structured = memory.structured_value
if isinstance(structured, dict) and "value" in structured:
return structured["value"]
if structured is not None:
return structured
return memory.content or ""
# ---------- 长期 → 画像 ----------
async def rebuild_profile(self, customer_id: int) -> dict[str, Any]:
"""用长期事实 + 问卷重建画像,并写一条版本快照。"""
now = _now()
facts = list(await self.session.scalars(
select(UserFact).where(UserFact.customer_id == customer_id)
))
assessment = (await self.session.execute(text(
"""
SELECT investor_type, questionnaire_version, assessed_at, valid_until
FROM fin_risk_assessment
WHERE customer_id = :customer_id
ORDER BY assessed_at DESC, id DESC
LIMIT 1
"""
), {"customer_id": customer_id})).first()
values: dict[str, Any] = {}
basis: dict[str, Any] = {}
# 红线:风险等级只从问卷取;记忆里哪怕有 preference:risk_level 也不写这个字段
if assessment is not None and assessment[0]:
values["investor_type"] = str(assessment[0])
basis["investor_type"] = {
"source": "fin_risk_assessment",
"questionnaire_version": assessment[1],
"assessed_at": str(assessment[2]),
"valid_until": str(assessment[3]),
}
tags: list[str] = []
for fact in facts:
key = str(fact.fact_key)
field = FACT_TO_PROFILE_FIELD.get(key)
if field is not None:
values[field] = self._as_text(fact.fact_value)
basis[field] = {
"source": "user_facts", "fact_key": key,
"confidence": float(fact.confidence or 0.0),
}
elif key.startswith(SELF_REPORTED_PREFIXES):
# 自述信息进标签,并显式标注"自述",与问卷等级区分开
tags.append(f"自述:{key}={self._as_text(fact.fact_value)}")
basis.setdefault("risk_tags", {"source": "user_facts", "items": []})
basis["risk_tags"]["items"].append(key)
if tags:
values["risk_tags"] = ";".join(tags)
profile = await self.session.get(FundCustomerProfile, customer_id)
if profile is None:
# 画像行由开户流程创建:`trade_account` 等身份字段在库里是 NOT NULL,属注册/账户侧
# 所有。本服务不代替开户去造这些数据——否则会写出一条**假的**开户记录,
# 而画像恰恰是风控要读的东西,假数据比没有数据更危险。未开户时如实报告。
return {
"profile": None,
"reason": "profile_row_not_opened",
"generation_basis": basis,
"promoted": len(facts),
}
for field in PROFILE_OWNED_FIELDS:
if field in values:
setattr(profile, field, values[field])
profile.updated_at = now
await self.session.flush()
snapshot = {
**{field: getattr(profile, field, None) for field in PROFILE_OWNED_FIELDS},
"generated_at": now.isoformat(),
}
await self._write_snapshot(customer_id, snapshot, basis, now)
return {"profile": snapshot, "generation_basis": basis, "promoted": len(facts)}
@staticmethod
def _as_text(value: Any) -> str:
"""把 JSON 列里取出的值渲染成可读字符串。
字符串类型的值可能带着 JSON 序列化时的外层引号(取决于驱动如何回读 JSON 列),
这里去掉它们——`risk_tags` 是给风控与投顾看的,多一对引号会让人以为值本身包含引号。
"""
if isinstance(value, str):
return value.strip().strip('"')
return _json.dumps(value, ensure_ascii=False)
async def _write_snapshot(
self, customer_id: int, snapshot: dict[str, Any], basis: dict[str, Any], now: datetime
) -> None:
"""写入新版本快照并把旧版本置为非当前。
唯一键 `uk_profile_snapshot_current` 建立在生成列 `current_customer_id` 上,
保证「每个客户最多一条 current」;因此必须先清旧再写新,顺序不能反。
"""
previous = list(await self.session.scalars(
select(ProfileSnapshot).where(
ProfileSnapshot.customer_id == customer_id, ProfileSnapshot.is_current.is_(True)
)
))
for row in previous:
row.is_current = False
row.updated_at = now
await self.session.flush()
latest = await self.session.scalar(text(
"SELECT COALESCE(MAX(version), 0) FROM profile_snapshots WHERE customer_id = :cid"
), {"cid": customer_id})
version = int(latest or 0) + 1
payload = _json.dumps(snapshot, ensure_ascii=False, sort_keys=True)
self.session.add(ProfileSnapshot(
profile_uuid=str(uuid4()),
customer_id=customer_id,
version=version,
snapshot=snapshot,
generation_basis=basis,
snapshot_hash=sha256(payload.encode("utf-8")).hexdigest(),
is_current=True,
generated_at=now,
created_at=now,
updated_at=now,
))
await self.session.flush()
# ---------- 完整链路 ----------
async def rebuild(self, customer_id: int) -> dict[str, Any]:
promoted = await self.promote_facts(customer_id)
outcome = await self.rebuild_profile(customer_id)
outcome["promoted_keys"] = promoted
return outcome