2026-09-13 20:31:13 +08:00
|
|
|
|
#!/usr/bin/env python3
|
|
|
|
|
|
"""文档体系机械校验(docs/README.md「维护清单」)。
|
|
|
|
|
|
|
|
|
|
|
|
校验五件事:
|
|
|
|
|
|
① docs 顶层 Markdown 固定为六个文件;
|
|
|
|
|
|
② 稳定 ID 定义唯一性与语法:每个 ID 恰好定义一次,且全仓引用均有定义;
|
|
|
|
|
|
③ 旧文件名、章节号引用与已闭合 G 标记零命中(docs/legacy/ 外);
|
|
|
|
|
|
④ 仓库内 Markdown 链接有效。
|
|
|
|
|
|
⑤ 给出 Git 基线时,持久 ID 集合与活跃 G 集合保持不变。
|
|
|
|
|
|
|
|
|
|
|
|
用法:scripts/check-docs.py [仓库根目录] [Git 基线]
|
|
|
|
|
|
"""
|
|
|
|
|
|
from __future__ import annotations
|
|
|
|
|
|
|
|
|
|
|
|
import re
|
|
|
|
|
|
import subprocess
|
|
|
|
|
|
import sys
|
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
|
|
|
|
|
|
|
ROOT = Path(sys.argv[1] if len(sys.argv) > 1 else ".").resolve()
|
|
|
|
|
|
BASELINE = sys.argv[2] if len(sys.argv) > 2 else None
|
|
|
|
|
|
|
|
|
|
|
|
DOCS = ROOT / "docs"
|
|
|
|
|
|
TOP_FILES = ["README.md", "requirements.md", "architecture.md", "specification.md",
|
|
|
|
|
|
"implementation.md", "reference.md"]
|
|
|
|
|
|
TEXT_SUFFIXES = {".md", ".kt", ".kts", ".yml", ".yaml", ".sql", ".py", ".sh"}
|
|
|
|
|
|
SKIP_DIRS = {
|
|
|
|
|
|
".git", ".gradle", ".gradle-home", ".gradletmp", ".idea", ".kotlin",
|
|
|
|
|
|
".opencode", ".qoder", ".zcode", "build", "node_modules", "out",
|
|
|
|
|
|
}
|
|
|
|
|
|
OWNER = {
|
|
|
|
|
|
"US": "requirements.md", "OPS": "requirements.md", "D": "architecture.md",
|
|
|
|
|
|
"C": "specification.md", "PRE": "specification.md", "INV": "specification.md",
|
|
|
|
|
|
"CLM": "specification.md", "Q": "specification.md", "G": "specification.md",
|
|
|
|
|
|
"PARAM": "reference.md",
|
|
|
|
|
|
}
|
|
|
|
|
|
# 文档中的 ID 族示例不是真实定义。
|
|
|
|
|
|
PLACEHOLDER_WORDS = {"G-NAME"}
|
|
|
|
|
|
|
|
|
|
|
|
PATTERNS = {
|
|
|
|
|
|
"US": re.compile(r"\bUS-\d+\b"),
|
|
|
|
|
|
"OPS": re.compile(r"\bOPS-\d+\b"),
|
|
|
|
|
|
"D": re.compile(r"(?<![A-Za-z0-9-])D[1-9]\d*(?![0-9])"),
|
|
|
|
|
|
"C": re.compile(r"\bC-\d+\b"),
|
|
|
|
|
|
"PRE": re.compile(r"\bPRE-\d+\b"),
|
|
|
|
|
|
"INV": re.compile(r"\bINV-\d+\b"),
|
|
|
|
|
|
"CLM": re.compile(r"\bCLM-\d+\b"),
|
|
|
|
|
|
"Q": re.compile(r"\bQ\d+\b"),
|
|
|
|
|
|
"G": re.compile(r"\bG-[A-Z][A-Z0-9-]*"),
|
|
|
|
|
|
"PARAM": re.compile(r"\b(?:msgx|mailbox|datasources|kafka)\.[A-Za-z0-9._-]+"),
|
|
|
|
|
|
}
|
|
|
|
|
|
USE_PATTERNS = {
|
|
|
|
|
|
**PATTERNS,
|
|
|
|
|
|
"PARAM": re.compile(r"(?<=PARAM:)(?:msgx|mailbox|datasources|kafka)\.[A-Za-z0-9._-]+"),
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
FIRST_COL = re.compile(r"^\|\s*`?([^|`]+?)`?\s*\|")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def is_placeholder(kind: str, ident: str) -> bool:
|
|
|
|
|
|
if ident in PLACEHOLDER_WORDS:
|
|
|
|
|
|
return True
|
|
|
|
|
|
# 通配/残缺键(如 `msgx.pipeline.*` 里被截出的 `msgx.pipeline.`)不算 ID
|
|
|
|
|
|
return ident.endswith(".") or "*" in ident
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
HEADING = re.compile(r"^(#{2,3})\s+(.+)$")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def section_of(lines: list[str]) -> list[str]:
|
|
|
|
|
|
"""为每一行给出其所属小节的标题文本(二级取全名,三级取 `二级 / 三级`)。"""
|
|
|
|
|
|
out, h2, h3 = [], "", ""
|
|
|
|
|
|
for line in lines:
|
|
|
|
|
|
m = HEADING.match(line)
|
|
|
|
|
|
if m:
|
|
|
|
|
|
if len(m.group(1)) == 2:
|
|
|
|
|
|
h2, h3 = m.group(2).strip(), ""
|
|
|
|
|
|
else:
|
|
|
|
|
|
h3 = m.group(2).strip()
|
|
|
|
|
|
out.append(f"{h2} / {h3}" if h3 else h2)
|
|
|
|
|
|
return out
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def first_col(line: str) -> str | None:
|
|
|
|
|
|
"""取表格首列原始内容(不含两侧管道与空白);非表格行返回 None。"""
|
|
|
|
|
|
stripped = line.strip()
|
|
|
|
|
|
if not stripped.startswith("|"):
|
|
|
|
|
|
return None
|
|
|
|
|
|
cell = stripped[1:].split("|", 1)[0]
|
|
|
|
|
|
return cell.strip()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def repo_text_files(*, include_legacy: bool = False) -> list[Path]:
|
|
|
|
|
|
"""返回仓库内需受文档引用纪律约束的文本文件,排除生成物。"""
|
|
|
|
|
|
files: list[Path] = []
|
|
|
|
|
|
for path in ROOT.rglob("*"):
|
|
|
|
|
|
if not path.is_file() or any(part in SKIP_DIRS for part in path.parts):
|
|
|
|
|
|
continue
|
|
|
|
|
|
if (not include_legacy and "legacy" in path.parts) or path.suffix not in TEXT_SUFFIXES:
|
|
|
|
|
|
continue
|
|
|
|
|
|
files.append(path)
|
|
|
|
|
|
return sorted(files)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def git_top_docs(ref: str) -> dict[str, str]:
|
|
|
|
|
|
"""读取某 Git 基线的 docs 顶层 Markdown,不读取工作树或 legacy。"""
|
|
|
|
|
|
listed = subprocess.run(
|
|
|
|
|
|
["git", "ls-tree", "-r", "--name-only", ref, "--", "docs"],
|
|
|
|
|
|
cwd=ROOT,
|
|
|
|
|
|
check=True,
|
|
|
|
|
|
capture_output=True,
|
|
|
|
|
|
text=True,
|
|
|
|
|
|
).stdout.splitlines()
|
|
|
|
|
|
docs: dict[str, str] = {}
|
|
|
|
|
|
for name in listed:
|
|
|
|
|
|
path = Path(name)
|
|
|
|
|
|
if path.parent != Path("docs") or path.suffix != ".md":
|
|
|
|
|
|
continue
|
|
|
|
|
|
docs[name] = subprocess.run(
|
|
|
|
|
|
["git", "show", f"{ref}:{name}"],
|
|
|
|
|
|
cwd=ROOT,
|
|
|
|
|
|
check=True,
|
|
|
|
|
|
capture_output=True,
|
|
|
|
|
|
text=True,
|
|
|
|
|
|
).stdout
|
|
|
|
|
|
return docs
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def ids_in(texts: list[str], kind: str) -> set[str]:
|
|
|
|
|
|
found: set[str] = set()
|
|
|
|
|
|
for text in texts:
|
|
|
|
|
|
for ident in PATTERNS[kind].findall(text):
|
|
|
|
|
|
ident = ident.rstrip("`")
|
|
|
|
|
|
if not is_placeholder(kind, ident):
|
|
|
|
|
|
found.add(ident)
|
|
|
|
|
|
return found
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def active_g_in(texts: list[str]) -> set[str]:
|
|
|
|
|
|
"""从基线注册表首列提取未划销、未标已闭合的 G。"""
|
|
|
|
|
|
found: set[str] = set()
|
|
|
|
|
|
for text in texts:
|
|
|
|
|
|
for line in text.splitlines():
|
|
|
|
|
|
if "~~" in line or re.search(r"已(?:闭合|关闭)", line):
|
|
|
|
|
|
continue
|
|
|
|
|
|
for ident in PATTERNS["G"].findall(line):
|
|
|
|
|
|
if is_definition("G", ident, line):
|
|
|
|
|
|
found.add(ident)
|
|
|
|
|
|
return found
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def is_definition(kind: str, ident: str, line: str) -> bool:
|
|
|
|
|
|
"""定义语法见 docs/README.md「ID 定义语法与引用纪律」。
|
|
|
|
|
|
|
|
|
|
|
|
注册表的表格首列在「单个 ID」时构成定义;成组登记与同行多 ID 均视为引用。
|
|
|
|
|
|
"""
|
|
|
|
|
|
if kind == "US":
|
|
|
|
|
|
return re.match(r"^###\s+" + re.escape(ident) + r"(\D|$)", line) is not None
|
|
|
|
|
|
if kind in ("C", "INV"):
|
|
|
|
|
|
return re.match(r"^-\s+\*\*" + re.escape(ident) + r"\*\*", line) is not None
|
|
|
|
|
|
cell = first_col(line)
|
|
|
|
|
|
if cell is None:
|
|
|
|
|
|
return False
|
|
|
|
|
|
# 严格匹配:`ID` 引用行(如 `INV-20` / `CLM-3`)不算定义
|
|
|
|
|
|
return cell in (ident, f"`{ident}`")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def main() -> int:
|
|
|
|
|
|
failures: list[str] = []
|
|
|
|
|
|
|
|
|
|
|
|
# ① docs 顶层固定为六个 Markdown 文件
|
|
|
|
|
|
actual_top = {path.name for path in DOCS.glob("*.md")}
|
|
|
|
|
|
expected_top = set(TOP_FILES)
|
|
|
|
|
|
if actual_top != expected_top:
|
|
|
|
|
|
failures.append(
|
|
|
|
|
|
"docs 顶层 Markdown 不等于固定六文件:"
|
|
|
|
|
|
f"缺少={sorted(expected_top - actual_top)},多出={sorted(actual_top - expected_top)}"
|
|
|
|
|
|
)
|
|
|
|
|
|
else:
|
|
|
|
|
|
print("OK docs 顶层固定为六个 Markdown 文件")
|
|
|
|
|
|
|
|
|
|
|
|
repo_files = repo_text_files()
|
|
|
|
|
|
all_repo_files = repo_text_files(include_legacy=True)
|
|
|
|
|
|
|
|
|
|
|
|
# ② ID 注册表:每个被引用的 ID 必须恰好有一处定义,且位于自己的注册表
|
|
|
|
|
|
# 触发检查的范围是「定义行」(加粗定义行 / US 标题 / 注册表首列),
|
|
|
|
|
|
# 与 docs/README.md「ID 定义语法与引用纪律」一致。
|
|
|
|
|
|
defs: dict[tuple[str, str], list[str]] = {}
|
|
|
|
|
|
for doc in sorted(DOCS.glob("*.md")):
|
|
|
|
|
|
lines = doc.read_text(encoding="utf-8").splitlines()
|
|
|
|
|
|
sections = section_of(lines)
|
|
|
|
|
|
for i, line in enumerate(lines, 1):
|
|
|
|
|
|
for kind, pattern in PATTERNS.items():
|
|
|
|
|
|
for ident in pattern.findall(line):
|
|
|
|
|
|
ident = ident.rstrip("`")
|
|
|
|
|
|
if is_placeholder(kind, ident) or not is_definition(kind, ident, line):
|
|
|
|
|
|
continue
|
|
|
|
|
|
defs.setdefault((kind, ident), []).append(f"{doc.name}:{i}:{sections[i - 1]}")
|
|
|
|
|
|
|
|
|
|
|
|
# 参数与指标同表登记、语义不同,二者都算已登记
|
|
|
|
|
|
REGISTRY = { # ID → (所属文件, 允许的定义小节)
|
|
|
|
|
|
"US": ("requirements.md", ["用户故事"]),
|
2026-09-14 16:40:00 +08:00
|
|
|
|
"OPS": ("requirements.md", ["运行验收", "运行与切流验收", "需求覆盖与依赖"]),
|
2026-09-13 20:31:13 +08:00
|
|
|
|
"D": ("architecture.md", ["关键决策"]),
|
2026-09-16 11:34:54 +08:00
|
|
|
|
"C": ("specification.md", ["契约", "约定"]),
|
2026-09-13 20:31:13 +08:00
|
|
|
|
"PRE": ("specification.md", ["前提"]),
|
|
|
|
|
|
"INV": ("specification.md", ["不变量"]),
|
|
|
|
|
|
"CLM": ("specification.md", ["声明边界"]),
|
|
|
|
|
|
"Q": ("specification.md", ["待确认事项台账"]),
|
|
|
|
|
|
"G": ("specification.md", ["当前已知偏差"]),
|
|
|
|
|
|
"PARAM": ("reference.md", ["参数注册表", "指标与健康"]),
|
|
|
|
|
|
}
|
|
|
|
|
|
totals = 0
|
|
|
|
|
|
for (kind, ident), where in sorted(defs.items()):
|
|
|
|
|
|
totals += 1
|
|
|
|
|
|
owner, sections = REGISTRY[kind]
|
|
|
|
|
|
ok = any(w.startswith(f"{owner}:") and any(sec in w for sec in sections) for w in where)
|
|
|
|
|
|
if len(where) > 1 and len(set(where)) > 1:
|
|
|
|
|
|
failures.append(f"{kind} {ident} 定义 {len(where)} 次({', '.join(where)})")
|
|
|
|
|
|
elif not ok:
|
|
|
|
|
|
failures.append(f"{kind} {ident} 定义不在 docs/{owner}{sections}:{where[0]}")
|
|
|
|
|
|
# 被引用但无定义的 ID(`defs` 已由上面的注册表检查确保位置与语法正确)
|
|
|
|
|
|
for kind, pattern in USE_PATTERNS.items():
|
|
|
|
|
|
used: set[str] = set()
|
|
|
|
|
|
for doc in repo_files:
|
|
|
|
|
|
for line in doc.read_text(encoding="utf-8").splitlines():
|
|
|
|
|
|
for ident in pattern.findall(line):
|
|
|
|
|
|
ident = ident.rstrip("`")
|
|
|
|
|
|
if not is_placeholder(kind, ident):
|
|
|
|
|
|
used.add(ident)
|
|
|
|
|
|
for ident in sorted(used):
|
|
|
|
|
|
if (kind, ident) not in defs:
|
|
|
|
|
|
failures.append(f"{kind} {ident} 已引用但无定义行")
|
|
|
|
|
|
if not failures:
|
|
|
|
|
|
print(f"OK ID 注册表:{totals} 个定义各一处且位于所属注册表")
|
|
|
|
|
|
|
|
|
|
|
|
# ③ 旧文件名、章节号引用与已闭合 G 标记零命中(legacy 外)
|
|
|
|
|
|
stale_name = re.compile(
|
|
|
|
|
|
r"(design|invariants|contracts|flight-state|user-stories|spec-boundary-closure"
|
|
|
|
|
|
r"|message-lifecycle|runbooks)\.md")
|
|
|
|
|
|
section_ref = re.compile(r"(?:§\s*\d|(?:第\s*)?\d+(?:\.\d+)*\s*节)")
|
|
|
|
|
|
closed_g = re.compile(r"G-[A-Z][A-Z0-9-]*.*(?:✓|已闭合|已关闭)")
|
|
|
|
|
|
stale_hits, section_hits, closed_g_hits = [], [], []
|
|
|
|
|
|
for f in all_repo_files:
|
|
|
|
|
|
try:
|
|
|
|
|
|
text = f.read_text(encoding="utf-8")
|
|
|
|
|
|
except (UnicodeDecodeError, OSError):
|
|
|
|
|
|
continue
|
|
|
|
|
|
for i, line in enumerate(text.splitlines(), 1):
|
|
|
|
|
|
if stale_name.search(line):
|
|
|
|
|
|
stale_hits.append(f"{f.relative_to(ROOT)}:{i} -> {stale_name.search(line).group(0)}")
|
|
|
|
|
|
if "legacy" not in f.parts and section_ref.search(line):
|
|
|
|
|
|
section_hits.append(f"{f.relative_to(ROOT)}:{i}")
|
|
|
|
|
|
if "legacy" not in f.parts and closed_g.search(line):
|
|
|
|
|
|
closed_g_hits.append(f"{f.relative_to(ROOT)}:{i}")
|
|
|
|
|
|
if stale_hits:
|
|
|
|
|
|
failures.append("存在指向已删除文档的文件名引用")
|
|
|
|
|
|
failures += stale_hits
|
|
|
|
|
|
else:
|
|
|
|
|
|
print("OK 旧文件名零命中")
|
|
|
|
|
|
if section_hits:
|
|
|
|
|
|
failures.append("存在章节号引用(应改为稳定 ID 或「文件名 + 小节名」)")
|
|
|
|
|
|
failures += section_hits
|
|
|
|
|
|
else:
|
|
|
|
|
|
print("OK 章节号引用零命中")
|
|
|
|
|
|
if closed_g_hits:
|
|
|
|
|
|
failures.append("存在已闭合 G 标记(应删除定义与全仓引用)")
|
|
|
|
|
|
failures += closed_g_hits
|
|
|
|
|
|
else:
|
|
|
|
|
|
print("OK 已闭合 G 标记零命中")
|
|
|
|
|
|
|
|
|
|
|
|
# ④ 仓库内 Markdown 链接
|
|
|
|
|
|
link_files = list(DOCS.rglob("*.md")) + [ROOT / "README.md", ROOT / "AGENTS.md"]
|
|
|
|
|
|
count, broken = 0, []
|
|
|
|
|
|
for f in link_files:
|
|
|
|
|
|
if not f.exists():
|
|
|
|
|
|
continue
|
|
|
|
|
|
for i, line in enumerate(f.read_text(encoding="utf-8").splitlines(), 1):
|
|
|
|
|
|
for m in re.finditer(r"\[[^\]]*\]\(([^)]+)\)", line):
|
|
|
|
|
|
target = m.group(1).split("#")[0].strip()
|
|
|
|
|
|
if not target or target.startswith(("http://", "https://", "mailto:")):
|
|
|
|
|
|
continue
|
|
|
|
|
|
count += 1
|
|
|
|
|
|
if not (f.parent / target).resolve().exists():
|
|
|
|
|
|
broken.append(f"{f.relative_to(ROOT)}:{i} -> {target}")
|
|
|
|
|
|
if broken:
|
|
|
|
|
|
failures.append("存在失效链接")
|
|
|
|
|
|
failures += broken
|
|
|
|
|
|
else:
|
|
|
|
|
|
print(f"OK 仓库内 Markdown 链接({count} 个)")
|
|
|
|
|
|
|
|
|
|
|
|
# ⑤ 文档合并/重命名时,与给定基线比较持久 ID 与活跃 G 集合
|
|
|
|
|
|
if BASELINE:
|
|
|
|
|
|
try:
|
|
|
|
|
|
baseline_texts = list(git_top_docs(BASELINE).values())
|
|
|
|
|
|
except subprocess.CalledProcessError as exc:
|
|
|
|
|
|
failures.append(f"无法读取 Git 基线 {BASELINE}: {exc.stderr.strip()}")
|
|
|
|
|
|
else:
|
|
|
|
|
|
for kind in ("US", "OPS", "D", "C", "PRE", "INV", "CLM", "Q", "PARAM"):
|
|
|
|
|
|
before = ids_in(baseline_texts, kind)
|
|
|
|
|
|
after = {ident for defined_kind, ident in defs if defined_kind == kind}
|
|
|
|
|
|
if before != after:
|
|
|
|
|
|
failures.append(
|
|
|
|
|
|
f"{kind} 集合相对 {BASELINE} 变化:"
|
|
|
|
|
|
f"删除={sorted(before - after)},新增={sorted(after - before)}"
|
|
|
|
|
|
)
|
|
|
|
|
|
before_g = active_g_in(baseline_texts)
|
|
|
|
|
|
after_g = {ident for defined_kind, ident in defs if defined_kind == "G"}
|
|
|
|
|
|
if before_g != after_g:
|
|
|
|
|
|
failures.append(
|
|
|
|
|
|
f"活跃 G 集合相对 {BASELINE} 变化:"
|
|
|
|
|
|
f"删除={sorted(before_g - after_g)},新增={sorted(after_g - before_g)}"
|
|
|
|
|
|
)
|
|
|
|
|
|
if not any("相对" in failure for failure in failures):
|
|
|
|
|
|
print(f"OK 持久 ID 与活跃 G 集合相对 {BASELINE} 无变化")
|
|
|
|
|
|
|
|
|
|
|
|
if failures:
|
|
|
|
|
|
for item in failures:
|
|
|
|
|
|
print(f"FAIL {item}")
|
|
|
|
|
|
print("存在失败项")
|
|
|
|
|
|
return 1
|
|
|
|
|
|
print("全部通过")
|
|
|
|
|
|
return 0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
if __name__ == "__main__":
|
|
|
|
|
|
sys.exit(main())
|