build / build-and-scan (push) Waiting to run
文档库:目录改为编号制(01-规范/02-架构设计/03-数据库/04-调整方案/
05-交接单/06-ops/07-scripts/08-skills/09-archive),顶层散文件归入 01-规范/;
INDEX.md 与 docs-manifest.json 重刷(档案 146 篇);旧目录名引用全量对齐。
IM 线:src/im/**(SDK / hub / store / presence / ws / gateway-token)、
src/web/routes/im.ts、src/db/plugin-data/**、src/supervisor/plugin-assembly.ts
及对应 test/**。
插件线:poc/{im-agent-bridge,im-connection-gateway,im-conversation-tabs,
business-plugins-im,carbon-mcp-probe}、src/web/routes/{sessions,overlay-device}.ts、
src/net/relay/{device-grant,instance-credential}.ts。
仓库卫生:清出 40 个历史误入库 / 已改名文件(34 个交接单归档 + 6 个旧结构,
本地均有副本);dsh-server-docs/.gitignore 补 tmp/;交接单不入库(政策)。
162 lines
6.3 KiB
Python
162 lines
6.3 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
docs-consistency.py —— 文档库「事实一致性」校验(只读,可复跑)
|
||
|
||
为什么需要它:`docs-audit.py` 查的是**结构问题**(编号冲突 / 悬空引用 / 重复子集),
|
||
**不查事实是否与现状一致** → 过时值会一直躺在库里没人发现。
|
||
|
||
两类检查:
|
||
|
||
【1】写死的取值 —— 会随并行改动过期,应改为"复跑取号"
|
||
· "下一号 = NN"(档案编号)
|
||
· 旧本机身份 `maidou` / `/c/Users/<user>/.workbuddy`(已迁 `E:\\ProgramData\\.workbuddy`)
|
||
|
||
【2】跨页取值不一致 —— 同一个事实键在多个「承诺现行」文件里取值不同
|
||
(首跑实证:`INDEX.md` 写下一号=72、`README.md` 写 69 → 同类事实两处打架)
|
||
|
||
判据:**「承诺现行」的文件不许出现已废止/写死/互相矛盾的取值。**
|
||
- 承诺现行(必查):`BRIEF.md` / `INDEX.md` / `README.md` / `CODEBUDDY.md` / `DEPLOY-本部署.md`
|
||
/ `01-规范/03-路线图与待办.md` / `01-规范/06-工作台UI规范.md` / `05-交接单/**` / `08-skills/**`
|
||
- 豁免(历史事实,只增不改):`04-调整方案/**`、`09-archive/**`、`01-规范/01-规划与架构.md`、`01-规范/02-运维手册.md`
|
||
—— 它们写的时候那个值是对的,回改反而破坏历史。
|
||
|
||
⚠️ **刻意不查**「旧域名 `dsh.alotbuy.com` / `alotbuy.com`」「旧配额 512M」:这些在库里几乎都是
|
||
"旧域已 301" / "512M→384M" 的**合法历史表述**,正则无法可靠区分,误报率过高。
|
||
改由人工在改域名/改配额时顺手核(档案 22 / 58 是权威;2026-09-19 域迁 `ai1net.com` 后旧域仅作过渡装置)。
|
||
|
||
用法:
|
||
python3 07-scripts/docs-consistency.py
|
||
退出码:0 = 无问题;1 = 有违背;2 = 环境错误
|
||
"""
|
||
import argparse
|
||
import io
|
||
import os
|
||
import re
|
||
import sys
|
||
from collections import defaultdict
|
||
|
||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
|
||
CURRENT_PREFIXES = (
|
||
'BRIEF.md', 'INDEX.md', 'README.md', 'CODEBUDDY.md', 'DEPLOY-本部署.md',
|
||
'01-规范/03-路线图与待办.md', '01-规范/06-工作台UI规范.md', '05-交接单/', '08-skills/',
|
||
)
|
||
|
||
# 【1】写死的取值:正则 → (说明, 修复建议)
|
||
HARDCODED = [
|
||
('档案「下一号」写死', r'下一号\s*[==]\s*\*{0,2}\d+', '改为"复跑取号,勿写死"(并行改动会打穿)'),
|
||
('旧本机用户名 maidou', r'maidou', '现行 Administrator'),
|
||
('旧技能/配置目录', r'[/\\][cC][/\\]Users[/\\][^/\\\s]+[/\\]\.workbuddy',
|
||
'现行 E:\\ProgramData\\.workbuddy(已迁 E 盘)'),
|
||
]
|
||
|
||
# 【2】跨页一致性:事实键 → 抽取正则(捕获组即取值)
|
||
CROSS_FACTS = {
|
||
'档案下一号': r'下一号\s*[==]\s*\*{0,2}(\d+)',
|
||
'代码 HEAD': r'代码 HEAD\s*[` ]?\s*\*{0,2}([0-9a-f]{7,10})',
|
||
'实例 MemoryMax': r'MemoryMax\s*[==]?\s*\*{0,2}(\d{3,4})\s*(?:MiB|M\b)?',
|
||
}
|
||
|
||
|
||
def read(p):
|
||
try:
|
||
return io.open(p, encoding='utf-8', errors='replace').read()
|
||
except Exception:
|
||
return ''
|
||
|
||
|
||
def collect():
|
||
out = []
|
||
for root, dirs, files in os.walk(ROOT):
|
||
dirs[:] = [d for d in dirs if d not in ('.git', 'node_modules', '__pycache__')]
|
||
for f in files:
|
||
if f.endswith(('.md', '.py', '.sh', '.cjs', '.mjs', '.js')):
|
||
out.append(os.path.relpath(os.path.join(root, f), ROOT).replace('\\', '/'))
|
||
return sorted(out)
|
||
|
||
|
||
def is_current(rel):
|
||
return any(rel == p or rel.startswith(p) for p in CURRENT_PREFIXES)
|
||
|
||
|
||
def is_quoted(line, start, end):
|
||
"""匹配是否被包住 —— 包住的内容视为**引用/举例**(如 T02 记的"下一号 = 20"已归零、
|
||
文档里把 `maidou` 当反例引用),不是当前断言,不算违规。
|
||
|
||
三类包裹:`" "` / `“ ”`(引文)与 `` ` ` ``(代码字面量)。"""
|
||
pre = line[:start].rstrip()
|
||
post = line[end:].lstrip()
|
||
return bool(pre and pre[-1] in '"“”`\'') and bool(post and post[0] in '"“”`\'')
|
||
|
||
|
||
def scan(rx, text):
|
||
"""返回一个文件里「不在引号内」的匹配数。"""
|
||
n = 0
|
||
for line in text.splitlines():
|
||
for m in rx.finditer(line):
|
||
if is_quoted(line, m.start(), m.end()):
|
||
continue
|
||
n += 1
|
||
return n
|
||
|
||
|
||
def main():
|
||
ap = argparse.ArgumentParser()
|
||
ap.parse_args()
|
||
|
||
files = collect()
|
||
cur = [f for f in files if is_current(f)]
|
||
print('文档库根:%s' % ROOT)
|
||
print('文件总数 %d | 承诺现行 %d 个(历史豁免 %d 个)\n' % (len(files), len(cur), len(files) - len(cur)))
|
||
|
||
bad = 0
|
||
|
||
print('──【1】写死的取值 ──')
|
||
for label, pat, fix in HARDCODED:
|
||
rx = re.compile(pat)
|
||
hits = [(r, scan(rx, read(os.path.join(ROOT, r)))) for r in cur]
|
||
hits = [h for h in hits if h[1]]
|
||
print(' 【%s】→ %s' % (label, fix))
|
||
if not hits:
|
||
print(' ✓ 无')
|
||
else:
|
||
bad += len(hits)
|
||
for r, n in sorted(hits, key=lambda x: -x[1])[:8]:
|
||
print(' ⚠ %-52s %d 处' % (r[:52], n))
|
||
if len(hits) > 8:
|
||
print(' …还有 %d 个文件' % (len(hits) - 8))
|
||
print()
|
||
|
||
print('──【2】跨页取值一致性 ──')
|
||
for name, pat in CROSS_FACTS.items():
|
||
rx = re.compile(pat)
|
||
vals = defaultdict(list)
|
||
for r in cur:
|
||
txt = read(os.path.join(ROOT, r))
|
||
for line in txt.splitlines():
|
||
for m in rx.finditer(line):
|
||
if is_quoted(line, m.start(), m.end()):
|
||
continue
|
||
vals[m.group(1)].append(r)
|
||
print(' 【%s】' % name)
|
||
if len(vals) <= 1:
|
||
print(' ✓ 取值唯一:%s' % (list(vals)[0] if vals else '(未出现)'))
|
||
else:
|
||
bad += 1
|
||
for v, rs in sorted(vals.items()):
|
||
print(' ⚠ 取值 %s ← %s' % (v, ', '.join(sorted(set(rs))[:4])))
|
||
print(' → 同一事实多处取值不一致,请校正为同一个权威值(或改为"复跑取号")')
|
||
print()
|
||
|
||
print('=' * 64)
|
||
if bad:
|
||
print('结论:**%d 项需处理**' % bad)
|
||
return 1
|
||
print('结论:承诺现行的文件与现行值一致 ✓')
|
||
return 0
|
||
|
||
|
||
if __name__ == '__main__':
|
||
sys.exit(main())
|