袁聪的前端调修
This commit is contained in:
@@ -56,6 +56,7 @@ EXTRACTED_FIELD_NAMES: tuple[str, ...] = (
|
||||
"申请日期",
|
||||
"代销机构",
|
||||
"申购金额",
|
||||
"币种",
|
||||
"金额单位",
|
||||
"赎回份额",
|
||||
"最新净值",
|
||||
@@ -486,7 +487,9 @@ class OffsiteDocumentRecognitionAdapter:
|
||||
try:
|
||||
body = await self._call_deepseek(source, ocr)
|
||||
document_type = _document_type(body.get("document_type"))
|
||||
fields = _recognized_fields(body)
|
||||
fields = normalize_recognition_fields(
|
||||
_recognized_fields(body), ocr.ocr_text
|
||||
)
|
||||
confidence = _confidence_map(_mapping_dict(body, "field_confidence"))
|
||||
page_evidence = _mapping_dict(body, "page_evidence") or ocr.page_evidence
|
||||
missing_fields = tuple(
|
||||
@@ -539,6 +542,10 @@ class OffsiteDocumentRecognitionAdapter:
|
||||
"page_evidence六个顶层字段。"
|
||||
"extracted_fields必须是对象,字段名只能使用下面列出的中文字段;"
|
||||
"无法从原文确认的字段填null,不得猜测或编造。"
|
||||
"申购金额只能填写数字本身,不得包含人民币、RMB、CNY、元、万元、"
|
||||
"货币符号或千分位逗号;币种单独填写到币种字段,金额单位单独填写到金额单位字段。"
|
||||
"例如原文“人民币5,000元”必须输出申购金额“5000”、币种“人民币”、"
|
||||
"金额单位“元”。"
|
||||
"文档中的产品代码、产品编号、基金产品代码均映射为基金代码;"
|
||||
"基金代码不要求固定六位,必须按原文保留。"
|
||||
"document_type只能是summary、subscription、redemption、other。"
|
||||
@@ -547,7 +554,7 @@ class OffsiteDocumentRecognitionAdapter:
|
||||
'{"基金代码":null,"基金名称":null,"账户标识":null,'
|
||||
'"投资者名称":null,"客户标识":null,"申请编号":null,'
|
||||
'"申请日期":null,"代销机构":null,"申购金额":null,'
|
||||
'"金额单位":null,"赎回份额":null,"最新净值":null,'
|
||||
'"币种":null,"金额单位":null,"赎回份额":null,"最新净值":null,'
|
||||
'"基金最新总份额":null,"申请前持有份额":null,'
|
||||
'"当前最新可用份额":null},'
|
||||
'"field_confidence":{},"missing_fields":[],'
|
||||
@@ -718,7 +725,7 @@ def _extract_simple_fields(text: str) -> dict[str, object]:
|
||||
)
|
||||
if value:
|
||||
fields[name] = value
|
||||
return fields
|
||||
return normalize_recognition_fields(fields, text)
|
||||
|
||||
|
||||
def _find_label_value(text: str, label: str) -> str | None:
|
||||
@@ -729,6 +736,97 @@ def _find_label_value(text: str, label: str) -> str | None:
|
||||
return matched.group(1).strip()
|
||||
|
||||
|
||||
_CURRENCY_MARKERS: tuple[tuple[str, str], ...] = (
|
||||
("人民币", "人民币"),
|
||||
("RMB", "人民币"),
|
||||
("CNY", "人民币"),
|
||||
("¥", "人民币"),
|
||||
("¥", "人民币"),
|
||||
("美元", "美元"),
|
||||
("USD", "美元"),
|
||||
("$", "美元"),
|
||||
("港币", "港币"),
|
||||
("HKD", "港币"),
|
||||
("欧元", "欧元"),
|
||||
("EUR", "欧元"),
|
||||
("日元", "日元"),
|
||||
("JPY", "日元"),
|
||||
)
|
||||
|
||||
|
||||
def normalize_recognition_fields(
|
||||
fields: Mapping[str, object], source_text: str = ""
|
||||
) -> dict[str, object]:
|
||||
"""统一识别字段格式,保证金额与币种、单位分开保存。"""
|
||||
normalized = dict(fields)
|
||||
raw_amount = normalized.get("申购金额")
|
||||
if raw_amount is None or str(raw_amount).strip() == "":
|
||||
return normalized
|
||||
|
||||
amount_text = str(raw_amount).strip().replace(",", "").replace(",", "")
|
||||
currency = _normalize_currency(normalized.get("币种"))
|
||||
if not currency:
|
||||
currency = _currency_from_text(amount_text)
|
||||
if not currency:
|
||||
currency = _currency_from_amount_context(source_text)
|
||||
if currency:
|
||||
normalized["币种"] = currency
|
||||
|
||||
unit = str(normalized.get("金额单位") or "").strip()
|
||||
if unit:
|
||||
if unit.endswith("万元"):
|
||||
unit = "万元"
|
||||
elif unit.endswith("元"):
|
||||
unit = "元"
|
||||
normalized["金额单位"] = unit
|
||||
|
||||
for marker, _ in _CURRENCY_MARKERS:
|
||||
amount_text = re.sub(
|
||||
rf"^{re.escape(marker)}\s*",
|
||||
"",
|
||||
amount_text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
amount_text = re.sub(
|
||||
rf"\s*{re.escape(marker)}$",
|
||||
"",
|
||||
amount_text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
if unit:
|
||||
amount_text = re.sub(rf"\s*{re.escape(unit)}$", "", amount_text)
|
||||
matched = re.search(r"-?(?:\d+(?:\.\d*)?|\.\d+)", amount_text)
|
||||
if matched:
|
||||
normalized["申购金额"] = matched.group(0)
|
||||
return normalized
|
||||
|
||||
|
||||
def _normalize_currency(value: object) -> str | None:
|
||||
text = str(value or "").strip()
|
||||
if not text:
|
||||
return None
|
||||
return _currency_from_text(text) or text
|
||||
|
||||
|
||||
def _currency_from_text(text: str) -> str | None:
|
||||
lowered = text.lower()
|
||||
for marker, currency in _CURRENCY_MARKERS:
|
||||
if marker.lower() in lowered:
|
||||
return currency
|
||||
return None
|
||||
|
||||
|
||||
def _currency_from_amount_context(source_text: str) -> str | None:
|
||||
if not source_text:
|
||||
return None
|
||||
matched = re.search(
|
||||
r"申购金额\s*[::]?\s*([^\s,,;;]+)",
|
||||
source_text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
return _currency_from_text(matched.group(1)) if matched else None
|
||||
|
||||
|
||||
def _extract_docx_text(payload: bytes) -> str:
|
||||
try:
|
||||
with zipfile.ZipFile(BytesIO(payload)) as archive:
|
||||
|
||||
Reference in New Issue
Block a user