Files
dsh_shenxian/dsh-server-docs/scripts/docs-consistency.py
T
admin 5ad755116e chore(docs): 文档库并入代码仓(R4 选 a)+ 索引/台账跟进
1) dsh-server-docs/ 从工作区(原 E:\...\aliyun-dsh-server\dsh-server-docs)**整体并入本仓**,
   保留目录名 ⇒ 仓库内 dsh-server-docs/... 的相对引用天然继续有效;旧目录(含其 .git)已归档到
   工作区 _中间产物_待清理/,未随本提交带入。
2) .gitattributes:新增 `dsh-server-docs/** -text` —— 原文档库是 `* -text` + autocrlf=false,
   必须保持纯 LF,否则会被本仓的 CRLF 规则翻掉。
3) 活引用里的绝对路径已全部改到新位置(docs 的 INDEX / README / scripts / skills + 用户级 skills
   + ~/.workbuddy/settings.json 的 hooks);历史档案(04-调整方案/、archive/)按「只增不改」未动。
   ⚠️ hooks 路径改动需「完全重启会话」才生效(配置是会话启动快照)。
4) 交接单/T08:新增 §16「生产整体切换执行记录」(形态 / 落地动作 / **4 个只有真上线才暴露的真 bug** /
   验收证据 / 回滚命令 / 残留项);台账 T08 行 → 已完成并归档;03-路线图 §二 登记 T08 收尾项。
5) 统一称谓:**「本机」只指跑 WorkBuddy 的开发机**,47 / 106 一律写「远程服务器」。
2026-09-15 18:47:13 +08:00

162 lines
6.1 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
docs-consistency.py —— 文档库「事实一致性」校验(只读,可复跑)
为什么需要它:`docs-audit.py` 查的是**结构问题**(编号冲突 / 悬空引用 / 重复子集),
**不查事实是否与现状一致** → 过时值会一直躺在库里没人发现。
两类检查:
【1】写死的取值 —— 会随并行改动过期,应改为"复跑取号"
· "下一号 = NN"(档案编号)
· 旧本机身份 `maidou` / `/c/Users/<user>/.workbuddy`(已迁 `E:\\ProgramData\\.workbuddy`)
【2】跨页取值不一致 —— 同一个事实键在多个「承诺现行」文件里取值不同
(首跑实证:`INDEX.md` 写下一号=72、`README.md` 写 69 → 同类事实两处打架)
判据:**「承诺现行」的文件不许出现已废止/写死/互相矛盾的取值。**
- 承诺现行(必查):`BRIEF.md` / `INDEX.md` / `README.md` / `CODEBUDDY.md` / `DEPLOY-本部署.md`
/ `03-路线图与待办.md` / `06-工作台UI规范.md` / `交接单/**` / `skills/**`
- 豁免(历史事实,只增不改):`04-调整方案/**`、`archive/**`、`01-规划与架构.md`、`02-运维手册.md`
—— 它们写的时候那个值是对的,回改反而破坏历史。
⚠️ **刻意不查**「旧域名 dsh.alotbuy.com」「旧配额 512M」:这两者在库里几乎都是
"旧域已 301" / "512M→384M" 的**合法历史表述**,正则无法可靠区分,误报率过高。
改由人工在改域名/改配额时顺手核(档案 22 / 58 是权威)。
用法:
python3 scripts/docs-consistency.py
退出码:0 = 无问题;1 = 有违背;2 = 环境错误
"""
import argparse
import io
import os
import re
import sys
from collections import defaultdict
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
CURRENT_PREFIXES = (
'BRIEF.md', 'INDEX.md', 'README.md', 'CODEBUDDY.md', 'DEPLOY-本部署.md',
'03-路线图与待办.md', '06-工作台UI规范.md', '交接单/', 'skills/',
)
# 【1】写死的取值:正则 → (说明, 修复建议)
HARDCODED = [
('档案「下一号」写死', r'下一号\s*[==]\s*\*{0,2}\d+', '改为"复跑取号,勿写死"(并行改动会打穿)'),
('旧本机用户名 maidou', r'maidou', '现行 Administrator'),
('旧技能/配置目录', r'[/\\][cC][/\\]Users[/\\][^/\\\s]+[/\\]\.workbuddy',
'现行 E:\\ProgramData\\.workbuddy(已迁 E 盘)'),
]
# 【2】跨页一致性:事实键 → 抽取正则(捕获组即取值)
CROSS_FACTS = {
'档案下一号': r'下一号\s*[==]\s*\*{0,2}(\d+)',
'代码 HEAD': r'代码 HEAD\s*[` ]?\s*\*{0,2}([0-9a-f]{7,10})',
'实例 MemoryMax': r'MemoryMax\s*[==]?\s*\*{0,2}(\d{3,4})\s*(?:MiB|M\b)?',
}
def read(p):
try:
return io.open(p, encoding='utf-8', errors='replace').read()
except Exception:
return ''
def collect():
out = []
for root, dirs, files in os.walk(ROOT):
dirs[:] = [d for d in dirs if d not in ('.git', 'node_modules', '__pycache__')]
for f in files:
if f.endswith(('.md', '.py', '.sh', '.cjs', '.mjs', '.js')):
out.append(os.path.relpath(os.path.join(root, f), ROOT).replace('\\', '/'))
return sorted(out)
def is_current(rel):
return any(rel == p or rel.startswith(p) for p in CURRENT_PREFIXES)
def is_quoted(line, start, end):
"""匹配是否被包住 —— 包住的内容视为**引用/举例**(如 T02 记的"下一号 = 20"已归零、
文档里把 `maidou` 当反例引用),不是当前断言,不算违规。
三类包裹:`" "` / `“ ”`(引文)与 `` ` ` ``(代码字面量)。"""
pre = line[:start].rstrip()
post = line[end:].lstrip()
return bool(pre and pre[-1] in '"“”`\'') and bool(post and post[0] in '"“”`\'')
def scan(rx, text):
"""返回一个文件里「不在引号内」的匹配数。"""
n = 0
for line in text.splitlines():
for m in rx.finditer(line):
if is_quoted(line, m.start(), m.end()):
continue
n += 1
return n
def main():
ap = argparse.ArgumentParser()
ap.parse_args()
files = collect()
cur = [f for f in files if is_current(f)]
print('文档库根:%s' % ROOT)
print('文件总数 %d | 承诺现行 %d 个(历史豁免 %d 个)\n' % (len(files), len(cur), len(files) - len(cur)))
bad = 0
print('──【1】写死的取值 ──')
for label, pat, fix in HARDCODED:
rx = re.compile(pat)
hits = [(r, scan(rx, read(os.path.join(ROOT, r)))) for r in cur]
hits = [h for h in hits if h[1]]
print(' 【%s】→ %s' % (label, fix))
if not hits:
print(' ✓ 无')
else:
bad += len(hits)
for r, n in sorted(hits, key=lambda x: -x[1])[:8]:
print(' ⚠ %-52s %d 处' % (r[:52], n))
if len(hits) > 8:
print(' …还有 %d 个文件' % (len(hits) - 8))
print()
print('──【2】跨页取值一致性 ──')
for name, pat in CROSS_FACTS.items():
rx = re.compile(pat)
vals = defaultdict(list)
for r in cur:
txt = read(os.path.join(ROOT, r))
for line in txt.splitlines():
for m in rx.finditer(line):
if is_quoted(line, m.start(), m.end()):
continue
vals[m.group(1)].append(r)
print(' 【%s】' % name)
if len(vals) <= 1:
print(' ✓ 取值唯一:%s' % (list(vals)[0] if vals else '(未出现)'))
else:
bad += 1
for v, rs in sorted(vals.items()):
print(' ⚠ 取值 %s ← %s' % (v, ', '.join(sorted(set(rs))[:4])))
print(' → 同一事实多处取值不一致,请校正为同一个权威值(或改为"复跑取号")')
print()
print('=' * 64)
if bad:
print('结论:**%d 项需处理**' % bad)
return 1
print('结论:承诺现行的文件与现行值一致 ✓')
return 0
if __name__ == '__main__':
sys.exit(main())