167 lines
7.0 KiB
Python
167 lines
7.0 KiB
Python
"""校验 `docs/05-接口文档.md` §19「接口总目录」的端点编号唯一性(只读,不连数据库)。
|
||||
|
|
|
|||
|
|
## 为什么需要它
|
|||
|
|
|
|||
|
|
`tools/check_authoritative_docs.py` 只校验 `docs/` 的**文件名编号**与权威性声明,
|
|||
|
|
**不校验 §19 里每个端点的编号**。2026-09-12 出过一次真实事故:两条线各自新增端点时
|
|||
|
|
都占用了 `A034`/`A035`,合并后 §19 同时存在两个 `A034` 与两个 `A035`,
|
|||
|
|
而文档守卫照样通过 —— 编号复用比"编号不够"更麻烦,而且**静默遗留**。
|
|||
|
|
|
|||
|
|
## 口径(含覆盖自检)
|
|||
|
|
|
|||
|
|
- 扫描范围:§19 里**端点总目录**那张表的数据行首列 —— 以**表头首列是不是 `编号`**
|
|||
|
|
判定。§19 还有一张「段 / 前缀 / 挂载来源 / 权限口径」的说明表,它的首列是分组名
|
|||
|
|
(`**场外基金**` 等)、表头是 `段`,**会被整体跳过并计数**(跳过了几张表会打印出来,
|
|||
|
|
所以"忽略了什么"是可见的)。早期实现把整章的表都当端点表扫,把那 6 个分组名 +
|
|||
|
|
1 个表头报成"首列无法识别" —— 那是误报,而**误报比漏报更伤**:真出问题时会被
|
|||
|
|
当成噪声忽略掉;
|
|||
|
|
- 端点表**内部**形状不对的首列仍会被显式列出来(而不是静默跳过);
|
|||
|
|
- **不枚举前缀白名单**,前缀从数据里归纳后打印出来。同一件事上还踩过一次
|
|||
|
|
"扫描正则写成 `[AMKCS]` 就漏掉 `O`/`R` 两段,把 62 个端点报成 55 个" ——
|
|||
|
|
一旦被漏掉的号段将来被复用,脚本仍会报"重复 0"。所以这里把
|
|||
|
|
**"扫到哪几个号段、各多少条、共计多少"** 一并输出,让覆盖范围本身可核对;
|
|||
|
|
- 只读:不修改任何文件。
|
|||
|
|
|
|||
|
|
用法:
|
|||
|
|
|
|||
|
|
python tools/check_docs_endpoint_ids.py
|
|||
|
|
|
|||
|
|
退出码:0 = 无重复;1 = 有重复或文档结构异常。
|
|||
|
|
"""
|
|||
|
|
|
|||
|
|
from __future__ import annotations
|
|||
|
|
|
|||
|
|
import re
|
|||
|
|
from collections import Counter
|
|||
|
|
from pathlib import Path
|
|||
|
|
|
|||
|
|
PROJECT_ROOT = Path(__file__).resolve().parents[1]
|
|||
|
|
INTERFACE_DOC = PROJECT_ROOT / "docs" / "05-接口文档.md"
|
|||
|
|
|
|||
|
|
#: §19 的起止标题(`: \\s*` 之间允许一个空格,避免依赖标题里的具体文字)。
|
|||
|
|
SECTION_START = re.compile(r"^##\s*19\.")
|
|||
|
|
SECTION_END = re.compile(r"^##\s*20\.")
|
|||
|
|
|
|||
|
|
#: 表格行:只取首列,容忍单元格两侧空白。
|
|||
|
|
TABLE_ROW = re.compile(r"^\|(.+)\|\s*$")
|
|||
|
|
#: Markdown 分隔行:| --- | :---: | 之类。
|
|||
|
|
SEPARATOR = re.compile(r"^[\s\-:|]+$")
|
|||
|
|
#: 端点编号形状:1~4 个大写字母 + 3~4 位数字(A034 / M003 / R001)。
|
|||
|
|
ENDPOINT_ID = re.compile(r"^[A-Z]{1,4}\d{3,4}$")
|
|||
|
|
|
|||
|
|
|
|||
|
|
class DocumentStructureError(RuntimeError):
|
|||
|
|
"""`docs/05` 的 §19 结构不符合预期,无法安全校验。"""
|
|||
|
|
|
|||
|
|
|
|||
|
|
def read_section(path: Path = INTERFACE_DOC) -> list[str]:
|
|||
|
|
"""返回 §19 的正文行(不含起止标题)。"""
|
|||
|
|
if not path.exists():
|
|||
|
|
raise DocumentStructureError(f"缺少接口权威文档:{path}")
|
|||
|
|
|
|||
|
|
lines = path.read_text(encoding="utf-8").splitlines()
|
|||
|
|
start = next((i for i, line in enumerate(lines) if SECTION_START.match(line)), None)
|
|||
|
|
if start is None:
|
|||
|
|
raise DocumentStructureError("接口文档里找不到 §19 章节标题")
|
|||
|
|
end = next(
|
|||
|
|
(i for i in range(start + 1, len(lines)) if SECTION_END.match(lines[i])),
|
|||
|
|
len(lines),
|
|||
|
|
)
|
|||
|
|
return lines[start + 1 : end]
|
|||
|
|
|
|||
|
|
|
|||
|
|
def collect_rows(lines: list[str]) -> tuple[list[str], list[str], int]:
|
|||
|
|
"""返回 `(首列编号列表, 无法识别的首列列表, 跳过的非端点表数量)`。
|
|||
|
|
|
|||
|
|
⚠️ 按**表格块**识别,而不是"§19 里所有表格行"。
|
|||
|
|
§19 除端点总目录外还有一张说明表(`段 / 前缀 / 挂载来源 / 权限口径`),
|
|||
|
|
它的首列是分组名(`**场外基金**`、`账户与交易`…),表头是 `段` 而不是 `编号`。
|
|||
|
|
早期实现把整章的表都当端点表扫,于是那张说明表的 6 个分组名 + 1 个表头被
|
|||
|
|
报成"首列无法识别" —— **属于误报**,而误报会让这个检查失去可信度
|
|||
|
|
(真出问题时会被当成噪声忽略)。
|
|||
|
|
|
|||
|
|
所以这里只在**表头首列是 `编号`** 的表格里收集数据行;其它表格整体跳过并计数,
|
|||
|
|
跳过数量会打印出来,保证"忽略了什么"是可见的而不是静默的。
|
|||
|
|
端点表**内部**形状不对的首列仍然会被报出来 —— 那才是真正的结构异常。
|
|||
|
|
"""
|
|||
|
|
ids: list[str] = []
|
|||
|
|
unrecognized: list[str] = []
|
|||
|
|
skipped_tables = 0
|
|||
|
|
in_endpoint_table = False
|
|||
|
|
previous_line_was_row = False
|
|||
|
|
|
|||
|
|
for line in lines:
|
|||
|
|
match = TABLE_ROW.match(line.strip())
|
|||
|
|
if match is None:
|
|||
|
|
# 非表格行 = 表格块结束,下一行表格行将是新表的表头
|
|||
|
|
previous_line_was_row = False
|
|||
|
|
continue
|
|||
|
|
|
|||
|
|
first_cell = match.group(1).split("|", 1)[0].strip()
|
|||
|
|
|
|||
|
|
if not previous_line_was_row:
|
|||
|
|
# 新表格的表头行:只有首列是「编号」的才是端点总目录
|
|||
|
|
in_endpoint_table = first_cell == "编号"
|
|||
|
|
if not in_endpoint_table:
|
|||
|
|
skipped_tables += 1
|
|||
|
|
previous_line_was_row = True
|
|||
|
|
continue
|
|||
|
|
|
|||
|
|
previous_line_was_row = True
|
|||
|
|
if not in_endpoint_table:
|
|||
|
|
continue
|
|||
|
|
if not first_cell or SEPARATOR.match(first_cell):
|
|||
|
|
continue
|
|||
|
|
if ENDPOINT_ID.match(first_cell):
|
|||
|
|
ids.append(first_cell)
|
|||
|
|
else:
|
|||
|
|
unrecognized.append(first_cell)
|
|||
|
|
|
|||
|
|
return ids, unrecognized, skipped_tables
|
|||
|
|
|
|||
|
|
|
|||
|
|
def collect_findings(path: Path = INTERFACE_DOC) -> list[str]:
|
|||
|
|
"""返回问题列表;空列表表示一致。"""
|
|||
|
|
lines = read_section(path)
|
|||
|
|
ids, unrecognized, _ = collect_rows(lines)
|
|||
|
|
|
|||
|
|
problems: list[str] = []
|
|||
|
|
if not ids:
|
|||
|
|
problems.append("§19 没有扫描到任何端点编号,文档结构可能已变化,请人工确认")
|
|||
|
|
|
|||
|
|
counter = Counter(ids)
|
|||
|
|
for endpoint_id in sorted(code for code, count in counter.items() if count > 1):
|
|||
|
|
problems.append(f"端点编号 {endpoint_id} 重复 {counter[endpoint_id]} 次")
|
|||
|
|
|
|||
|
|
for cell in sorted(set(unrecognized)):
|
|||
|
|
problems.append(f"§19 端点表首列无法识别为端点编号:{cell!r}")
|
|||
|
|
|
|||
|
|
return problems
|
|||
|
|
|
|||
|
|
|
|||
|
|
def describe_coverage(path: Path = INTERFACE_DOC) -> str:
|
|||
|
|
"""返回覆盖范围报告 —— 让"扫到了什么"可见,而不是只报"没问题"。"""
|
|||
|
|
ids, _, skipped = collect_rows(read_section(path))
|
|||
|
|
prefixes = Counter(re.match(r"[A-Z]+", code).group() for code in ids)
|
|||
|
|
detail = "、".join(f"{prefix}×{count}" for prefix, count in sorted(prefixes.items()))
|
|||
|
|
return (
|
|||
|
|
f"§19 覆盖:{len(ids)} 个端点编号 / {len(prefixes)} 个号段({detail})"
|
|||
|
|
f";跳过 {skipped} 张非端点表"
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def main() -> int:
|
|||
|
|
problems = collect_findings()
|
|||
|
|
print(describe_coverage())
|
|||
|
|
if problems:
|
|||
|
|
for problem in problems:
|
|||
|
|
print(f" ✗ {problem}")
|
|||
|
|
print(f"docs/05 §19 端点编号校验失败:{len(problems)} 处问题")
|
|||
|
|
return 1
|
|||
|
|
print("docs/05 §19 端点编号无重复")
|
|||
|
|
return 0
|
|||
|
|
|
|||
|
|
|
|||
|
|
if __name__ == "__main__":
|
|||
|
|
raise SystemExit(main())
|