AIcoding 第 5 步 todo 开发(开发计划 v2.0)前半段: - T-1 3 张新表 DDL(core_convert_request 6 态 ENUM / core_trade_calendar / core_share_rule)+ 种子 + sqlite 单一事实源同步 + 列清单断言 - T-2/T-2b calc 扩展(product_round/redeem_amount/partial_qty)+ 真实净值实算回填 - T-3 convert_request_repository(6 态 + 条件 UPDATE 守卫)+ core_ro 三读方法 + share_lot_repository.available_qty_with_inflight(R-3 在途占用推导) - T-4 convert_repository.sync_mirror 成为 risk_convert_detail 唯一进度镜像写入口 (旧三方法标 Deprecated,T-7 后删) - T-5 locks.py 锁键构造器 convert_req_lock_key / convert_confirm_lock_key - T-6 convert_service.accept_convert 受理事务(八步:锁→幂等→校验→受理日顺延 →在途占用校验→落单→镜像+审计→202;不扣份额/不折算/不写流水) + tests/test_convert_accept.py(16 用例) + scripts/dev/verify_convert_accept.py(真库 36/36 一致) + trading_calendar.py 纯函数包(R-5)+ 21 用例 T-6 真库实测暴露并修复:confirm_eta 在日历数据边界抛 ValueError,会让已落库 的受理单在调用方眼里变 500;改为展示性字段容错 + 单测守护。 基线:798 passed / 10 skipped,零回归。
214 lines
11 KiB
Python
214 lines
11 KiB
Python
# -*- coding: utf-8 -*-
|
||
"""联网拉取**真实**基金历史净值,生成 `core_product_nav` 种子 SQL。
|
||
|
||
背景(为什么要有这个脚本)
|
||
------------------------------------------------------------------
|
||
旧种子 `06-seed-nav.sql` 只写了**一天**、且是人造数值(`nav_date` 写死 2026-09-04,
|
||
`PROD-110022` 净值 1.0300)。T+1 受理/确认分离模型要求「T 日受理、T+1 按 **T 日净值**确认」,
|
||
净值只有一天 ⇒ 只要跑批日期不是那一天,T 日净值必然缺失,确认流程直接卡死。
|
||
|
||
因此本脚本做两件事:
|
||
1. **联网**从天天基金(东方财富)公开接口拉取真实历史净值,生成多日种子;
|
||
2. 提供**增量补拉**能力:真实净值只到「今天」,明后天再跑演示仍会缺 T 日净值,
|
||
重跑本脚本即可把最新净值补进种子(不必再人造数据)。
|
||
|
||
数据来源(可复核)
|
||
------------------------------------------------------------------
|
||
- 接口:`https://fund.eastmoney.com/pingzhongdata/{基金代码}.js`
|
||
- 内容:该基金自成立以来的 `Data_netWorthTrend`(每日单位净值 + 日涨幅)。
|
||
- 用法:纯标准库 `urllib`,无第三方依赖(本项目红线:不新增第三方依赖)。
|
||
|
||
产品 ↔ 真实净值来源的映射(**按类型 + 风险等级配对**,不是按产品 ID 字面代码)
|
||
------------------------------------------------------------------
|
||
项目里 `PROD-110022` 这类 ID 只是长得像基金代码,实际名字与类型是**虚构**的
|
||
(如 `PROD-110022` 叫「稳健债基 A」/ bond / R1,但真实 110022 是「易方达消费行业股票」,
|
||
日波动 −1.63%,拿它当债基净值会让风控演示结论失真)。
|
||
故本脚本为每个产品**单独指定一只类型匹配的真实基金**作为净值数值来源,
|
||
只借用其数值与波动特征,基金名称照写在生成 SQL 的注释里,便于复核。
|
||
|
||
货币基金(`PROD-000001` 现金宝货币)
|
||
------------------------------------------------------------------
|
||
真实业务中货币基金**单位净值恒为 1.0000 元**,收益以份额结转,不存在净值序列
|
||
(实测:天弘余额宝 000198、广发钱袋子 000509 在该接口均无 `Data_netWorthTrend`)。
|
||
故货基不联网,直接按真实业务生成恒定 1.0000、日涨幅 0。
|
||
|
||
用法
|
||
------------------------------------------------------------------
|
||
python scripts/core/fetch_nav.py # 拉最近 60 个交易日,写 06-seed-nav.sql
|
||
python scripts/core/fetch_nav.py --days 120 # 拉更长区间
|
||
python scripts/core/fetch_nav.py --stdout # 只打印、不落盘
|
||
python scripts/core/fetch_nav.py --cache-dir .navcache # 复用/保存原始 js,离线可重放
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import datetime as dt
|
||
import json
|
||
import re
|
||
import sys
|
||
import urllib.request
|
||
from pathlib import Path
|
||
|
||
# ── 配置 ────────────────────────────────────────────────────────────────
|
||
PING_URL = "https://fund.eastmoney.com/pingzhongdata/{code}.js"
|
||
HEADERS = {
|
||
"User-Agent": "Mozilla/5.0",
|
||
"Referer": "https://fund.eastmoney.com/",
|
||
}
|
||
|
||
# product_id -> (真实基金代码, 真实基金名称, 配对理由)
|
||
NAV_SOURCE: dict[str, tuple[str, str, str]] = {
|
||
"PROD-110022": ("000191", "富国信用债债券A/B", "bond/R1:纯债,日波动 0.02%"),
|
||
"PROD-110023": ("000032", "易方达信用债债券A", "bond/R2:信用债,日波动 0.09%"),
|
||
"PROD-005827": ("005827", "易方达蓝筹精选混合", "mixed/R3:代码即真身,股债平衡"),
|
||
"PROD-005828": ("001182", "易方达安心回馈混合A", "mixed/R2:低波混合,日波动 0.73%"),
|
||
"PROD-161725": ("161631", "融通人工智能指数(LOF)A", "stock/R4:科技主题,日波动 2.57%"),
|
||
"PROD-161726": ("110022", "易方达消费行业股票", "stock/R4:消费主题,代码即真身"),
|
||
"PROD-003095": ("003095", "中欧医疗健康混合A", "stock/R4:医药主题,代码即真身"),
|
||
"PROD-510300": ("510300", "沪深300ETF华泰柏瑞", "index/R3:宽基指数,代码即真身"),
|
||
"PROD-510500": ("510500", "中证500ETF南方", "index/R4:宽基指数,代码即真身"),
|
||
"PROD-XYZ999": ("320007", "诺安成长混合A", "stock/R5:高波动成长,日波动 1.76%"),
|
||
"PROD-000002": ("014437", "鹏华中证同业存单AAA指数7天持有", "bond/R1:同业存单,日波动 0.01%"),
|
||
"PROD-WMG001": ("004155", "中信保诚至泰中短债A", "wealth_mgmt/R2:中短债,日波动 0.02%"),
|
||
"PROD-PRIV01": ("519062", "海富通阿尔法对冲混合A", "private_fund/R4:量化对冲,日波动 0.33%"),
|
||
}
|
||
|
||
# 货币基金:净值恒为 1.0000(真实业务,不联网)
|
||
MONEY_FUNDS: dict[str, str] = {
|
||
"PROD-000001": "现金宝货币(money/R1:货基单位净值恒为 1 元,收益以份额结转)",
|
||
}
|
||
|
||
DEFAULT_OUT = Path(__file__).with_name("06-seed-nav.sql")
|
||
DEFAULT_DAYS = 60
|
||
TIMEOUT = 30
|
||
|
||
|
||
# ── 抓取与解析 ───────────────────────────────────────────────────────────
|
||
def fetch_js(code: str, cache_dir: Path | None) -> str:
|
||
"""拉取(或读缓存)某只基金的 pingzhongdata js 文本。"""
|
||
dest = cache_dir / f"{code}.js" if cache_dir else None
|
||
if dest and dest.exists():
|
||
return dest.read_text(encoding="utf-8", errors="ignore")
|
||
req = urllib.request.Request(PING_URL.format(code=code), headers=HEADERS)
|
||
text = urllib.request.urlopen(req, timeout=TIMEOUT).read().decode("utf-8", errors="ignore")
|
||
if dest:
|
||
cache_dir.mkdir(parents=True, exist_ok=True)
|
||
dest.write_text(text, encoding="utf-8")
|
||
return text
|
||
|
||
|
||
def parse_series(text: str) -> list[tuple[dt.date, float, float]]:
|
||
"""从 js 里解析 `(nav_date, nav, 日涨幅%)` 序列,按日期升序。"""
|
||
match = re.search(r"var\s+Data_netWorthTrend\s*=\s*(\[.*?\]);", text, re.S)
|
||
if not match:
|
||
return []
|
||
raw = json.loads(match.group(1))
|
||
out: list[tuple[dt.date, float, float]] = []
|
||
for item in raw:
|
||
day = dt.datetime.fromtimestamp(item["x"] / 1000).date()
|
||
out.append((day, float(item["y"]), float(item.get("equityReturn") or 0)))
|
||
out.sort(key=lambda r: r[0])
|
||
return out
|
||
|
||
|
||
# ── 生成 SQL ─────────────────────────────────────────────────────────────
|
||
def build_sql(rows: list[tuple[str, dt.date, float, float]], fetched_at: str) -> str:
|
||
"""rows = (product_id, nav_date, nav, chg_pct)"""
|
||
sources = "\n".join(
|
||
f"-- {pid:<13} <- {code} {name}({why})"
|
||
for pid, (code, name, why) in sorted(NAV_SOURCE.items())
|
||
)
|
||
money = "\n".join(f"-- {pid:<13} <- 恒定 1.0000({why})" for pid, why in sorted(MONEY_FUNDS.items()))
|
||
dates = [r[1] for r in rows]
|
||
head = (
|
||
"USE jinrong_core;\n\n"
|
||
"-- 基金净值种子:**真实历史净值,非人造数值**\n"
|
||
"-- 生成方式:python scripts/core/fetch_nav.py --days {n}\n"
|
||
"-- 数据来源:天天基金(东方财富)公开接口 https://fund.eastmoney.com/pingzhongdata/{{代码}}.js\n"
|
||
"-- 抓取时间:{fetched}\n"
|
||
"-- 区间:{start} ~ {end}(共 {days} 个净值发布日)\n"
|
||
"--\n"
|
||
"-- 产品 ↔ 净值来源映射(产品 ID 为虚构,数值借用下列真实基金,按类型/风险等级配对):\n"
|
||
"{sources}\n"
|
||
"{money}\n"
|
||
"--\n"
|
||
"-- ⚠️ 真实净值只到抓取当日。之后运行需重跑 fetch_nav.py 增量补拉,\n"
|
||
"-- 否则 T+1 确认会取不到 T 日净值(真实业务同样是「净值 T+1 才公告」)。\n\n"
|
||
).format(
|
||
n=DEFAULT_DAYS,
|
||
fetched=fetched_at,
|
||
start=min(dates),
|
||
end=max(dates),
|
||
days=len(set(dates)),
|
||
sources=sources,
|
||
money=money,
|
||
)
|
||
|
||
lines: list[str] = []
|
||
cur_pid = None
|
||
for pid, day, nav, chg in rows:
|
||
if pid != cur_pid:
|
||
if cur_pid is not None:
|
||
lines[-1] = lines[-1].rstrip(",") + ";" # 上一组收尾
|
||
lines.append("INSERT INTO core_product_nav (product_id, nav, daily_chg_pct, nav_date) VALUES")
|
||
cur_pid = pid
|
||
lines.append(f"('{pid}', {nav:.4f}, {chg:.4f}, '{day}'),")
|
||
if lines:
|
||
lines[-1] = lines[-1].rstrip(",") + ";"
|
||
return head + "\n".join(lines) + "\n"
|
||
|
||
|
||
# ── 主流程 ───────────────────────────────────────────────────────────────
|
||
def main(argv: list[str] | None = None) -> int:
|
||
ap = argparse.ArgumentParser(description="联网拉取真实基金净值并生成 core_product_nav 种子 SQL")
|
||
ap.add_argument("--days", type=int, default=DEFAULT_DAYS, help="取最近多少个净值发布日(默认 60)")
|
||
ap.add_argument("--out", type=Path, default=DEFAULT_OUT, help="输出 SQL 路径")
|
||
ap.add_argument("--stdout", action="store_true", help="只打印到标准输出,不写文件")
|
||
ap.add_argument("--cache-dir", type=Path, default=None, help="原始 js 缓存目录(可离线重放)")
|
||
args = ap.parse_args(argv)
|
||
|
||
fetched_at = dt.datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||
rows: list[tuple[str, dt.date, float, float]] = []
|
||
failed: list[str] = []
|
||
|
||
for pid, (code, name, _why) in sorted(NAV_SOURCE.items()):
|
||
try:
|
||
series = parse_series(fetch_js(code, args.cache_dir))
|
||
except Exception as exc: # 网络/解析失败不静默:记下来,最后汇总报错
|
||
failed.append(f"{pid}({code} {name}): {exc}")
|
||
continue
|
||
if not series:
|
||
failed.append(f"{pid}({code} {name}): 接口无 Data_netWorthTrend")
|
||
continue
|
||
for day, nav, chg in series[-args.days:]:
|
||
rows.append((pid, day, nav, chg))
|
||
print(f"[OK] {pid} <- {code} {name}:{len(series[-args.days:])} 条,最新 {series[-1][0]} {series[-1][1]}")
|
||
|
||
# 货币基金:净值恒定 1.0000
|
||
if rows:
|
||
money_days = sorted({r[1] for r in rows})
|
||
for pid in sorted(MONEY_FUNDS):
|
||
for day in money_days:
|
||
rows.append((pid, day, 1.0000, 0.0))
|
||
print(f"[OK] {pid} <- 货基恒定净值 1.0000:{len(money_days)} 条")
|
||
|
||
if failed:
|
||
print("\n[FAIL] 以下产品拉取失败:", file=sys.stderr)
|
||
for item in failed:
|
||
print(" - " + item, file=sys.stderr)
|
||
return 1
|
||
|
||
rows.sort(key=lambda r: (r[0], r[1]))
|
||
sql = build_sql(rows, fetched_at)
|
||
if args.stdout:
|
||
print(sql)
|
||
else:
|
||
args.out.write_text(sql, encoding="utf-8")
|
||
print(f"\n已写入 {args.out}({len(rows)} 行)")
|
||
return 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
raise SystemExit(main())
|