specification.md 按新需求重写: - 扫描模型反转:水位/ID 区间 → 处理标记为谓词(C-30 取代 C-1/C-2/C-13;INV-2b 替代 INV-2/4/5) - Redis 回归为查询投影:处理完成门(INV-23)、投影治理与同源读取(INV-24),INV-11b 扩充非权威清单 - 日计划快照语义反转:缺席航班删除、未携带字段清除(INV-15b),增量报文语义另立(INV-14b) - 终态记录归档 → 到期删除(INV-25),G-PROC-HST/G-HST-RETENTION/G-REPLAY-CHANNEL/G-FLOP-DIRECTION 关闭并清扫全仓引用 - 重放移出交付范围:R_keep 公式收窄、CLM-3 重定义为重处理幂等、Q6 删除 - 新增 INV-23~28:Redis 完成门、投影治理、清理谓词、参考数据逐类保存/门控、历史先行红线 - C-25/C-26 定案(原子级联不回发 EROR;快照未携带字段清除),Q13/Q14 关闭,Q6/Q12 删除,新增 C-30/C-31 联动:implementation.md 收报/回填/快照/生命周期/FLOP 方向各章按新口径重写;architecture.md D1/D4 改删除语义;reference.md 退役 archive-after;requirements.md OPS 表改为注册表定义 语法;AGENTS.md 状态边界随新口径更新;check-docs.py OPS 注册表节名同步。 scripts/check-docs.sh 全部通过。
328 lines
13 KiB
Python
Executable File
328 lines
13 KiB
Python
Executable File
#!/usr/bin/env python3
|
||
"""文档体系机械校验(docs/README.md「维护清单」)。
|
||
|
||
校验五件事:
|
||
① docs 顶层 Markdown 固定为六个文件;
|
||
② 稳定 ID 定义唯一性与语法:每个 ID 恰好定义一次,且全仓引用均有定义;
|
||
③ 旧文件名、章节号引用与已闭合 G 标记零命中(docs/legacy/ 外);
|
||
④ 仓库内 Markdown 链接有效。
|
||
⑤ 给出 Git 基线时,持久 ID 集合与活跃 G 集合保持不变。
|
||
|
||
用法:scripts/check-docs.py [仓库根目录] [Git 基线]
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import re
|
||
import subprocess
|
||
import sys
|
||
from pathlib import Path
|
||
|
||
ROOT = Path(sys.argv[1] if len(sys.argv) > 1 else ".").resolve()
|
||
BASELINE = sys.argv[2] if len(sys.argv) > 2 else None
|
||
|
||
DOCS = ROOT / "docs"
|
||
TOP_FILES = ["README.md", "requirements.md", "architecture.md", "specification.md",
|
||
"implementation.md", "reference.md"]
|
||
TEXT_SUFFIXES = {".md", ".kt", ".kts", ".yml", ".yaml", ".sql", ".py", ".sh"}
|
||
SKIP_DIRS = {
|
||
".git", ".gradle", ".gradle-home", ".gradletmp", ".idea", ".kotlin",
|
||
".opencode", ".qoder", ".zcode", "build", "node_modules", "out",
|
||
}
|
||
OWNER = {
|
||
"US": "requirements.md", "OPS": "requirements.md", "D": "architecture.md",
|
||
"C": "specification.md", "PRE": "specification.md", "INV": "specification.md",
|
||
"CLM": "specification.md", "Q": "specification.md", "G": "specification.md",
|
||
"PARAM": "reference.md",
|
||
}
|
||
# 文档中的 ID 族示例不是真实定义。
|
||
PLACEHOLDER_WORDS = {"G-NAME"}
|
||
|
||
PATTERNS = {
|
||
"US": re.compile(r"\bUS-\d+\b"),
|
||
"OPS": re.compile(r"\bOPS-\d+\b"),
|
||
"D": re.compile(r"(?<![A-Za-z0-9-])D[1-9]\d*(?![0-9])"),
|
||
"C": re.compile(r"\bC-\d+\b"),
|
||
"PRE": re.compile(r"\bPRE-\d+\b"),
|
||
"INV": re.compile(r"\bINV-\d+\b"),
|
||
"CLM": re.compile(r"\bCLM-\d+\b"),
|
||
"Q": re.compile(r"\bQ\d+\b"),
|
||
"G": re.compile(r"\bG-[A-Z][A-Z0-9-]*"),
|
||
"PARAM": re.compile(r"\b(?:msgx|mailbox|datasources|kafka)\.[A-Za-z0-9._-]+"),
|
||
}
|
||
USE_PATTERNS = {
|
||
**PATTERNS,
|
||
"PARAM": re.compile(r"(?<=PARAM:)(?:msgx|mailbox|datasources|kafka)\.[A-Za-z0-9._-]+"),
|
||
}
|
||
|
||
FIRST_COL = re.compile(r"^\|\s*`?([^|`]+?)`?\s*\|")
|
||
|
||
|
||
def is_placeholder(kind: str, ident: str) -> bool:
|
||
if ident in PLACEHOLDER_WORDS:
|
||
return True
|
||
# 通配/残缺键(如 `msgx.pipeline.*` 里被截出的 `msgx.pipeline.`)不算 ID
|
||
return ident.endswith(".") or "*" in ident
|
||
|
||
|
||
HEADING = re.compile(r"^(#{2,3})\s+(.+)$")
|
||
|
||
|
||
def section_of(lines: list[str]) -> list[str]:
|
||
"""为每一行给出其所属小节的标题文本(二级取全名,三级取 `二级 / 三级`)。"""
|
||
out, h2, h3 = [], "", ""
|
||
for line in lines:
|
||
m = HEADING.match(line)
|
||
if m:
|
||
if len(m.group(1)) == 2:
|
||
h2, h3 = m.group(2).strip(), ""
|
||
else:
|
||
h3 = m.group(2).strip()
|
||
out.append(f"{h2} / {h3}" if h3 else h2)
|
||
return out
|
||
|
||
|
||
def first_col(line: str) -> str | None:
|
||
"""取表格首列原始内容(不含两侧管道与空白);非表格行返回 None。"""
|
||
stripped = line.strip()
|
||
if not stripped.startswith("|"):
|
||
return None
|
||
cell = stripped[1:].split("|", 1)[0]
|
||
return cell.strip()
|
||
|
||
|
||
def repo_text_files(*, include_legacy: bool = False) -> list[Path]:
|
||
"""返回仓库内需受文档引用纪律约束的文本文件,排除生成物。"""
|
||
files: list[Path] = []
|
||
for path in ROOT.rglob("*"):
|
||
if not path.is_file() or any(part in SKIP_DIRS for part in path.parts):
|
||
continue
|
||
if (not include_legacy and "legacy" in path.parts) or path.suffix not in TEXT_SUFFIXES:
|
||
continue
|
||
files.append(path)
|
||
return sorted(files)
|
||
|
||
|
||
def git_top_docs(ref: str) -> dict[str, str]:
|
||
"""读取某 Git 基线的 docs 顶层 Markdown,不读取工作树或 legacy。"""
|
||
listed = subprocess.run(
|
||
["git", "ls-tree", "-r", "--name-only", ref, "--", "docs"],
|
||
cwd=ROOT,
|
||
check=True,
|
||
capture_output=True,
|
||
text=True,
|
||
).stdout.splitlines()
|
||
docs: dict[str, str] = {}
|
||
for name in listed:
|
||
path = Path(name)
|
||
if path.parent != Path("docs") or path.suffix != ".md":
|
||
continue
|
||
docs[name] = subprocess.run(
|
||
["git", "show", f"{ref}:{name}"],
|
||
cwd=ROOT,
|
||
check=True,
|
||
capture_output=True,
|
||
text=True,
|
||
).stdout
|
||
return docs
|
||
|
||
|
||
def ids_in(texts: list[str], kind: str) -> set[str]:
|
||
found: set[str] = set()
|
||
for text in texts:
|
||
for ident in PATTERNS[kind].findall(text):
|
||
ident = ident.rstrip("`")
|
||
if not is_placeholder(kind, ident):
|
||
found.add(ident)
|
||
return found
|
||
|
||
|
||
def active_g_in(texts: list[str]) -> set[str]:
|
||
"""从基线注册表首列提取未划销、未标已闭合的 G。"""
|
||
found: set[str] = set()
|
||
for text in texts:
|
||
for line in text.splitlines():
|
||
if "~~" in line or re.search(r"已(?:闭合|关闭)", line):
|
||
continue
|
||
for ident in PATTERNS["G"].findall(line):
|
||
if is_definition("G", ident, line):
|
||
found.add(ident)
|
||
return found
|
||
|
||
|
||
def is_definition(kind: str, ident: str, line: str) -> bool:
|
||
"""定义语法见 docs/README.md「ID 定义语法与引用纪律」。
|
||
|
||
注册表的表格首列在「单个 ID」时构成定义;成组登记与同行多 ID 均视为引用。
|
||
"""
|
||
if kind == "US":
|
||
return re.match(r"^###\s+" + re.escape(ident) + r"(\D|$)", line) is not None
|
||
if kind in ("C", "INV"):
|
||
return re.match(r"^-\s+\*\*" + re.escape(ident) + r"\*\*", line) is not None
|
||
cell = first_col(line)
|
||
if cell is None:
|
||
return False
|
||
# 严格匹配:`ID` 引用行(如 `INV-20` / `CLM-3`)不算定义
|
||
return cell in (ident, f"`{ident}`")
|
||
|
||
|
||
def main() -> int:
|
||
failures: list[str] = []
|
||
|
||
# ① docs 顶层固定为六个 Markdown 文件
|
||
actual_top = {path.name for path in DOCS.glob("*.md")}
|
||
expected_top = set(TOP_FILES)
|
||
if actual_top != expected_top:
|
||
failures.append(
|
||
"docs 顶层 Markdown 不等于固定六文件:"
|
||
f"缺少={sorted(expected_top - actual_top)},多出={sorted(actual_top - expected_top)}"
|
||
)
|
||
else:
|
||
print("OK docs 顶层固定为六个 Markdown 文件")
|
||
|
||
repo_files = repo_text_files()
|
||
all_repo_files = repo_text_files(include_legacy=True)
|
||
|
||
# ② ID 注册表:每个被引用的 ID 必须恰好有一处定义,且位于自己的注册表
|
||
# 触发检查的范围是「定义行」(加粗定义行 / US 标题 / 注册表首列),
|
||
# 与 docs/README.md「ID 定义语法与引用纪律」一致。
|
||
defs: dict[tuple[str, str], list[str]] = {}
|
||
for doc in sorted(DOCS.glob("*.md")):
|
||
lines = doc.read_text(encoding="utf-8").splitlines()
|
||
sections = section_of(lines)
|
||
for i, line in enumerate(lines, 1):
|
||
for kind, pattern in PATTERNS.items():
|
||
for ident in pattern.findall(line):
|
||
ident = ident.rstrip("`")
|
||
if is_placeholder(kind, ident) or not is_definition(kind, ident, line):
|
||
continue
|
||
defs.setdefault((kind, ident), []).append(f"{doc.name}:{i}:{sections[i - 1]}")
|
||
|
||
# 参数与指标同表登记、语义不同,二者都算已登记
|
||
REGISTRY = { # ID → (所属文件, 允许的定义小节)
|
||
"US": ("requirements.md", ["用户故事"]),
|
||
"OPS": ("requirements.md", ["运行验收", "运行与切流验收", "需求覆盖与依赖"]),
|
||
"D": ("architecture.md", ["关键决策"]),
|
||
"C": ("specification.md", ["契约"]),
|
||
"PRE": ("specification.md", ["前提"]),
|
||
"INV": ("specification.md", ["不变量"]),
|
||
"CLM": ("specification.md", ["声明边界"]),
|
||
"Q": ("specification.md", ["待确认事项台账"]),
|
||
"G": ("specification.md", ["当前已知偏差"]),
|
||
"PARAM": ("reference.md", ["参数注册表", "指标与健康"]),
|
||
}
|
||
totals = 0
|
||
for (kind, ident), where in sorted(defs.items()):
|
||
totals += 1
|
||
owner, sections = REGISTRY[kind]
|
||
ok = any(w.startswith(f"{owner}:") and any(sec in w for sec in sections) for w in where)
|
||
if len(where) > 1 and len(set(where)) > 1:
|
||
failures.append(f"{kind} {ident} 定义 {len(where)} 次({', '.join(where)})")
|
||
elif not ok:
|
||
failures.append(f"{kind} {ident} 定义不在 docs/{owner}{sections}:{where[0]}")
|
||
# 被引用但无定义的 ID(`defs` 已由上面的注册表检查确保位置与语法正确)
|
||
for kind, pattern in USE_PATTERNS.items():
|
||
used: set[str] = set()
|
||
for doc in repo_files:
|
||
for line in doc.read_text(encoding="utf-8").splitlines():
|
||
for ident in pattern.findall(line):
|
||
ident = ident.rstrip("`")
|
||
if not is_placeholder(kind, ident):
|
||
used.add(ident)
|
||
for ident in sorted(used):
|
||
if (kind, ident) not in defs:
|
||
failures.append(f"{kind} {ident} 已引用但无定义行")
|
||
if not failures:
|
||
print(f"OK ID 注册表:{totals} 个定义各一处且位于所属注册表")
|
||
|
||
# ③ 旧文件名、章节号引用与已闭合 G 标记零命中(legacy 外)
|
||
stale_name = re.compile(
|
||
r"(design|invariants|contracts|flight-state|user-stories|spec-boundary-closure"
|
||
r"|message-lifecycle|runbooks)\.md")
|
||
section_ref = re.compile(r"(?:§\s*\d|(?:第\s*)?\d+(?:\.\d+)*\s*节)")
|
||
closed_g = re.compile(r"G-[A-Z][A-Z0-9-]*.*(?:✓|已闭合|已关闭)")
|
||
stale_hits, section_hits, closed_g_hits = [], [], []
|
||
for f in all_repo_files:
|
||
try:
|
||
text = f.read_text(encoding="utf-8")
|
||
except (UnicodeDecodeError, OSError):
|
||
continue
|
||
for i, line in enumerate(text.splitlines(), 1):
|
||
if stale_name.search(line):
|
||
stale_hits.append(f"{f.relative_to(ROOT)}:{i} -> {stale_name.search(line).group(0)}")
|
||
if "legacy" not in f.parts and section_ref.search(line):
|
||
section_hits.append(f"{f.relative_to(ROOT)}:{i}")
|
||
if "legacy" not in f.parts and closed_g.search(line):
|
||
closed_g_hits.append(f"{f.relative_to(ROOT)}:{i}")
|
||
if stale_hits:
|
||
failures.append("存在指向已删除文档的文件名引用")
|
||
failures += stale_hits
|
||
else:
|
||
print("OK 旧文件名零命中")
|
||
if section_hits:
|
||
failures.append("存在章节号引用(应改为稳定 ID 或「文件名 + 小节名」)")
|
||
failures += section_hits
|
||
else:
|
||
print("OK 章节号引用零命中")
|
||
if closed_g_hits:
|
||
failures.append("存在已闭合 G 标记(应删除定义与全仓引用)")
|
||
failures += closed_g_hits
|
||
else:
|
||
print("OK 已闭合 G 标记零命中")
|
||
|
||
# ④ 仓库内 Markdown 链接
|
||
link_files = list(DOCS.rglob("*.md")) + [ROOT / "README.md", ROOT / "AGENTS.md"]
|
||
count, broken = 0, []
|
||
for f in link_files:
|
||
if not f.exists():
|
||
continue
|
||
for i, line in enumerate(f.read_text(encoding="utf-8").splitlines(), 1):
|
||
for m in re.finditer(r"\[[^\]]*\]\(([^)]+)\)", line):
|
||
target = m.group(1).split("#")[0].strip()
|
||
if not target or target.startswith(("http://", "https://", "mailto:")):
|
||
continue
|
||
count += 1
|
||
if not (f.parent / target).resolve().exists():
|
||
broken.append(f"{f.relative_to(ROOT)}:{i} -> {target}")
|
||
if broken:
|
||
failures.append("存在失效链接")
|
||
failures += broken
|
||
else:
|
||
print(f"OK 仓库内 Markdown 链接({count} 个)")
|
||
|
||
# ⑤ 文档合并/重命名时,与给定基线比较持久 ID 与活跃 G 集合
|
||
if BASELINE:
|
||
try:
|
||
baseline_texts = list(git_top_docs(BASELINE).values())
|
||
except subprocess.CalledProcessError as exc:
|
||
failures.append(f"无法读取 Git 基线 {BASELINE}: {exc.stderr.strip()}")
|
||
else:
|
||
for kind in ("US", "OPS", "D", "C", "PRE", "INV", "CLM", "Q", "PARAM"):
|
||
before = ids_in(baseline_texts, kind)
|
||
after = {ident for defined_kind, ident in defs if defined_kind == kind}
|
||
if before != after:
|
||
failures.append(
|
||
f"{kind} 集合相对 {BASELINE} 变化:"
|
||
f"删除={sorted(before - after)},新增={sorted(after - before)}"
|
||
)
|
||
before_g = active_g_in(baseline_texts)
|
||
after_g = {ident for defined_kind, ident in defs if defined_kind == "G"}
|
||
if before_g != after_g:
|
||
failures.append(
|
||
f"活跃 G 集合相对 {BASELINE} 变化:"
|
||
f"删除={sorted(before_g - after_g)},新增={sorted(after_g - before_g)}"
|
||
)
|
||
if not any("相对" in failure for failure in failures):
|
||
print(f"OK 持久 ID 与活跃 G 集合相对 {BASELINE} 无变化")
|
||
|
||
if failures:
|
||
for item in failures:
|
||
print(f"FAIL {item}")
|
||
print("存在失败项")
|
||
return 1
|
||
print("全部通过")
|
||
return 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
sys.exit(main())
|