build / build-and-scan (push) Waiting to run
文档库:目录改为编号制(01-规范/02-架构设计/03-数据库/04-调整方案/
05-交接单/06-ops/07-scripts/08-skills/09-archive),顶层散文件归入 01-规范/;
INDEX.md 与 docs-manifest.json 重刷(档案 146 篇);旧目录名引用全量对齐。
IM 线:src/im/**(SDK / hub / store / presence / ws / gateway-token)、
src/web/routes/im.ts、src/db/plugin-data/**、src/supervisor/plugin-assembly.ts
及对应 test/**。
插件线:poc/{im-agent-bridge,im-connection-gateway,im-conversation-tabs,
business-plugins-im,carbon-mcp-probe}、src/web/routes/{sessions,overlay-device}.ts、
src/net/relay/{device-grant,instance-credential}.ts。
仓库卫生:清出 40 个历史误入库 / 已改名文件(34 个交接单归档 + 6 个旧结构,
本地均有副本);dsh-server-docs/.gitignore 补 tmp/;交接单不入库(政策)。
182 lines
8.4 KiB
Python
182 lines
8.4 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
docs-audit.py — 文档库质量扫描(只读,零副作用)
|
||
|
||
用途:一次性回答「文档是否清晰、无歧义、需要精简」这类问题,输出**可判定**的问题清单,
|
||
避免靠人逐篇读。用于文档库根目录,也可用于任何 md 目录。
|
||
|
||
检查项(均可机器判定):
|
||
1 档案编号冲突(同一编号被多份档案占用)
|
||
2 标题号 ≠ 文件名号
|
||
3 备份/临时残留(.bak / .orig / ~ 等,会污染对账与阅读)
|
||
4 术语与事实漂移(旧术语、旧域名、已废弃表名)——历史档案保留原文属正常,需在入口说明
|
||
5 体量分布(超长文件 → 考虑拆分;过短文件 → 考虑合并/指针)
|
||
6 交叉引用有效性(引用「档案 NN」/「04-调整方案/NN-」是否存在)
|
||
7 疑似重复/近重复(同一主题两处维护;含"子集包含"识别)
|
||
8 元信息规范(档案头部是否含 日期 / 状态)
|
||
9 非 md 文件的入库合理性提示
|
||
|
||
用法:
|
||
python3 07-scripts/docs-audit.py [文档库根目录,默认取本脚本的上一级]
|
||
|
||
退出码:0 = 无 P0 级问题;1 = 发现编号冲突或失效引用(便于接 CI)。
|
||
"""
|
||
import io, os, re, sys, collections
|
||
|
||
ROOT = sys.argv[1] if len(sys.argv) > 1 else os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
P0 = 0
|
||
|
||
def rd(rel):
|
||
return io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
|
||
|
||
def all_files():
|
||
out = []
|
||
for base, dirs, names in os.walk(ROOT):
|
||
if '.git' in dirs:
|
||
dirs.remove('.git')
|
||
for n in names:
|
||
out.append(os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/'))
|
||
return sorted(out)
|
||
|
||
FILES = all_files()
|
||
MD = [f for f in FILES if f.endswith('.md')]
|
||
OTHER = [f for f in FILES if not f.endswith('.md')]
|
||
|
||
print('=' * 72)
|
||
print('文档库质量扫描 root=%s' % ROOT)
|
||
print('总文件 %d(md %d / 其他 %d)' % (len(FILES), len(MD), len(OTHER)))
|
||
print('=' * 72)
|
||
|
||
# ── 1&2 编号 ──────────────────────────────────────────────
|
||
num_map = collections.defaultdict(list)
|
||
mismatch = []
|
||
titles = {}
|
||
for f in MD:
|
||
s = rd(f)
|
||
title = next((l.strip() for l in s.split('\n') if l.strip().startswith('#')), '')
|
||
titles[f] = title
|
||
mf = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
|
||
mt = re.match(r'^#\s*(?:调整方案\s*)?(\d+[a-z]?)\s*[·、..\-]', title)
|
||
if mf and ("/" not in f or f.startswith("04-调整方案/")):
|
||
# ⛔ 只认「根级 NN-x.md」与「04-调整方案/NN-x.md」两套体系;
|
||
# 08-skills/**、01-规范/** 等目录下的编号文件(如 07-并行调度详解.md)不是档案
|
||
num_map[mf.group(1)].append(f)
|
||
if mf and mt and mf.group(1) != mt.group(1):
|
||
mismatch.append((f, title[:70]))
|
||
|
||
print('\n【1】档案编号冲突')
|
||
conf = {n: fs for n, fs in num_map.items() if len(fs) > 1}
|
||
if not conf:
|
||
print(' ✓ 无冲突')
|
||
for n, fs in sorted(conf.items()):
|
||
# 顶层 NN-x.md 与 04-调整方案/NN-x.md 属两套体系,不算冲突
|
||
top = [x for x in fs if '/' not in x]
|
||
arch = [x for x in fs if x.startswith('04-调整方案/')]
|
||
if top and arch and len(fs) == 2:
|
||
print(' · 编号 %s:根级 vs 调整方案(**两套体系,非冲突**,但入口须写明)' % n)
|
||
continue
|
||
P0 = 1
|
||
print(' ⚠ 编号 %s 被 %d 份档案占用:' % (n, len(fs)))
|
||
for x in fs:
|
||
print(' %-56s 标题「%s」' % (x, titles[x][:44]))
|
||
|
||
print('\n【2】标题号 ≠ 文件名号')
|
||
if not mismatch:
|
||
print(' ✓ 无')
|
||
for f, t in mismatch:
|
||
P0 = 1
|
||
print(' ⚠ %-52s → 「%s」' % (f, t))
|
||
|
||
# ── 3 备份残留 ───────────────────────────────────────────
|
||
print('\n【3】备份/临时残留')
|
||
junk = [f for f in FILES if re.search(r'\.bak|~$|\.orig$|\.tmp$|\.swp$|\.new$', f)]
|
||
print(' 数量 %d %s' % (len(junk), '(建议清理或纳入 .gitignore)' if junk else ''))
|
||
for f in junk[:15]:
|
||
print(' -', f)
|
||
|
||
# ── 4 术语/事实漂移 ─────────────────────────────────────
|
||
print('\n【4】术语与事实漂移(历史档案保留原文=正常,需入口说明)')
|
||
TERMS = {'业务插件(旧术语)': '业务插件', 'dsh.alotbuy.com(旧域名)': 'dsh.alotbuy.com',
|
||
'folder_plugins(已废弃)': 'folder_plugins'}
|
||
for label, k in TERMS.items():
|
||
fs = [f for f in MD if k in rd(f)]
|
||
print(' %-26s %d 个文件' % (label, len(fs)))
|
||
if fs and len(fs) <= 6:
|
||
print(' %s' % ', '.join(fs))
|
||
|
||
# ── 5 体量 ──────────────────────────────────────────────
|
||
print('\n【5】体量分布(>300 行考虑拆分;<15 行考虑合并或指针化)')
|
||
rows = sorted(((rd(f).count('\n') + 1, len(rd(f).encode()), f) for f in MD), reverse=True)
|
||
print(' 最长 8:')
|
||
for ln, by, f in rows[:8]:
|
||
print(' %5d 行 %7.1f KB %s' % (ln, by / 1024, f))
|
||
print(' 最短 5:')
|
||
for ln, by, f in rows[-5:]:
|
||
print(' %5d 行 %7.1f KB %s' % (ln, by / 1024, f))
|
||
|
||
# ── 6 引用有效性 ────────────────────────────────────────
|
||
print('\n【6】交叉引用有效性')
|
||
existing = set(num_map.keys())
|
||
bad = collections.defaultdict(list)
|
||
for f in MD:
|
||
s = rd(f)
|
||
for m in re.finditer(r'档案\s*(\d{1,2}[a-z]?)', s):
|
||
n = m.group(1)
|
||
# 键已改为字符串(支持 '37a');原先的数值范围判断 1<=n<=99 换成形状校验
|
||
if re.fullmatch(r'\d{1,2}[a-z]?', n) and n not in existing:
|
||
bad[f].append(n)
|
||
if not bad:
|
||
print(' ✓ 无悬空档案号引用')
|
||
else:
|
||
P0 = 1
|
||
for f, ns in list(bad.items())[:12]:
|
||
print(' %-48s 引用了不存在的档案号 %s' % (f, sorted(set(ns))))
|
||
|
||
# ── 7 近重复 ────────────────────────────────────────────
|
||
print('\n【7】疑似重复 / 近重复')
|
||
sig = {f: re.sub(r'\s+', '', rd(f)) for f in MD}
|
||
found = False
|
||
keys = list(sig)
|
||
for i in range(len(keys)):
|
||
for j in range(i + 1, len(keys)):
|
||
a, b = sig[keys[i]], sig[keys[j]]
|
||
if len(a) < 200 or len(b) < 200:
|
||
continue
|
||
if a == b:
|
||
print(' %-44s = %-44s 完全相同' % (keys[i], keys[j])); found = True
|
||
elif a in b or b in a:
|
||
small, big = (keys[i], keys[j]) if len(a) < len(b) else (keys[j], keys[i])
|
||
print(' %-44s ⊂ %-44s 子集(%d/%d 字符)' % (small, big, min(len(a), len(b)), max(len(a), len(b)))); found = True
|
||
else:
|
||
ga = set(a[k:k + 3] for k in range(0, len(a) - 2, 3))
|
||
gb = set(b[k:k + 3] for k in range(0, len(b) - 2, 3))
|
||
if ga and gb:
|
||
r = len(ga & gb) / min(len(ga), len(gb))
|
||
if r > 0.55:
|
||
print(' %-44s ≈ %-44s 重合 %.0f%%' % (keys[i], keys[j], r * 100)); found = True
|
||
if not found:
|
||
print(' ✓ 未发现')
|
||
|
||
# ── 8 元信息 ────────────────────────────────────────────
|
||
print('\n【8】档案头部元信息(04-调整方案/ 内)')
|
||
no_date, no_status = [], []
|
||
for f in MD:
|
||
if not f.startswith('04-调整方案/'):
|
||
continue
|
||
head = '\n'.join(rd(f).split('\n')[:14])
|
||
if not re.search(r'日期|20\d\d-\d\d-\d\d', head):
|
||
no_date.append(f)
|
||
if not re.search(r'状态', head):
|
||
no_status.append(f)
|
||
print(' 缺「日期」%d 个 %s' % (len(no_date), [os.path.basename(x) for x in no_date[:6]]))
|
||
print(' 缺「状态」%d 个 %s' % (len(no_status), [os.path.basename(x) for x in no_status[:6]]))
|
||
|
||
# ── 9 非 md ─────────────────────────────────────────────
|
||
print('\n【9】非 md 文件 %d 个(确认均属应入库的资产/脚本)' % len(OTHER))
|
||
print(' ' + (', '.join(OTHER[:12]) + (' …' if len(OTHER) > 12 else '') if OTHER else '无'))
|
||
|
||
print('\n' + '=' * 72)
|
||
print('结论:%s' % ('发现问题(见上 ⚠)' if P0 else '无 P0 级问题'))
|
||
sys.exit(1 if P0 else 0)
|