袁聪的前端调修

This commit is contained in:
2026-09-14 01:07:43 +08:00
parent 680c2a0749
commit 7a49a1c2c8
23 changed files with 2150 additions and 51 deletions
@@ -56,6 +56,7 @@ EXTRACTED_FIELD_NAMES: tuple[str, ...] = (
"申请日期",
"代销机构",
"申购金额",
"币种",
"金额单位",
"赎回份额",
"最新净值",
@@ -486,7 +487,9 @@ class OffsiteDocumentRecognitionAdapter:
try:
body = await self._call_deepseek(source, ocr)
document_type = _document_type(body.get("document_type"))
fields = _recognized_fields(body)
fields = normalize_recognition_fields(
_recognized_fields(body), ocr.ocr_text
)
confidence = _confidence_map(_mapping_dict(body, "field_confidence"))
page_evidence = _mapping_dict(body, "page_evidence") or ocr.page_evidence
missing_fields = tuple(
@@ -539,6 +542,10 @@ class OffsiteDocumentRecognitionAdapter:
"page_evidence六个顶层字段。"
"extracted_fields必须是对象,字段名只能使用下面列出的中文字段;"
"无法从原文确认的字段填null,不得猜测或编造。"
"申购金额只能填写数字本身,不得包含人民币、RMB、CNY、元、万元、"
"货币符号或千分位逗号;币种单独填写到币种字段,金额单位单独填写到金额单位字段。"
"例如原文“人民币5,000元”必须输出申购金额“5000”、币种“人民币”、"
"金额单位“元”。"
"文档中的产品代码、产品编号、基金产品代码均映射为基金代码;"
"基金代码不要求固定六位,必须按原文保留。"
"document_type只能是summary、subscription、redemption、other。"
@@ -547,7 +554,7 @@ class OffsiteDocumentRecognitionAdapter:
'{"基金代码":null,"基金名称":null,"账户标识":null,'
'"投资者名称":null,"客户标识":null,"申请编号":null,'
'"申请日期":null,"代销机构":null,"申购金额":null,'
'"金额单位":null,"赎回份额":null,"最新净值":null,'
'"币种":null,"金额单位":null,"赎回份额":null,"最新净值":null,'
'"基金最新总份额":null,"申请前持有份额":null,'
'"当前最新可用份额":null},'
'"field_confidence":{},"missing_fields":[],'
@@ -718,7 +725,7 @@ def _extract_simple_fields(text: str) -> dict[str, object]:
)
if value:
fields[name] = value
return fields
return normalize_recognition_fields(fields, text)
def _find_label_value(text: str, label: str) -> str | None:
@@ -729,6 +736,97 @@ def _find_label_value(text: str, label: str) -> str | None:
return matched.group(1).strip()
_CURRENCY_MARKERS: tuple[tuple[str, str], ...] = (
("人民币", "人民币"),
("RMB", "人民币"),
("CNY", "人民币"),
("¥", "人民币"),
("¥", "人民币"),
("美元", "美元"),
("USD", "美元"),
("$", "美元"),
("港币", "港币"),
("HKD", "港币"),
("欧元", "欧元"),
("EUR", "欧元"),
("日元", "日元"),
("JPY", "日元"),
)
def normalize_recognition_fields(
fields: Mapping[str, object], source_text: str = ""
) -> dict[str, object]:
"""统一识别字段格式,保证金额与币种、单位分开保存。"""
normalized = dict(fields)
raw_amount = normalized.get("申购金额")
if raw_amount is None or str(raw_amount).strip() == "":
return normalized
amount_text = str(raw_amount).strip().replace(",", "").replace(",", "")
currency = _normalize_currency(normalized.get("币种"))
if not currency:
currency = _currency_from_text(amount_text)
if not currency:
currency = _currency_from_amount_context(source_text)
if currency:
normalized["币种"] = currency
unit = str(normalized.get("金额单位") or "").strip()
if unit:
if unit.endswith("万元"):
unit = "万元"
elif unit.endswith("元"):
unit = "元"
normalized["金额单位"] = unit
for marker, _ in _CURRENCY_MARKERS:
amount_text = re.sub(
rf"^{re.escape(marker)}\s*",
"",
amount_text,
flags=re.IGNORECASE,
)
amount_text = re.sub(
rf"\s*{re.escape(marker)}$",
"",
amount_text,
flags=re.IGNORECASE,
)
if unit:
amount_text = re.sub(rf"\s*{re.escape(unit)}$", "", amount_text)
matched = re.search(r"-?(?:\d+(?:\.\d*)?|\.\d+)", amount_text)
if matched:
normalized["申购金额"] = matched.group(0)
return normalized
def _normalize_currency(value: object) -> str | None:
text = str(value or "").strip()
if not text:
return None
return _currency_from_text(text) or text
def _currency_from_text(text: str) -> str | None:
lowered = text.lower()
for marker, currency in _CURRENCY_MARKERS:
if marker.lower() in lowered:
return currency
return None
def _currency_from_amount_context(source_text: str) -> str | None:
if not source_text:
return None
matched = re.search(
r"申购金额\s*[::]?\s*([^\s,,;;]+)",
source_text,
flags=re.IGNORECASE,
)
return _currency_from_text(matched.group(1)) if matched else None
def _extract_docx_text(payload: bytes) -> str:
try:
with zipfile.ZipFile(BytesIO(payload)) as archive: