Files
admin e6207aa691
build / build-and-scan (push) Waiting to run
chore(仓库对齐): 文档库结构治理 + IM/插件线落地
文档库:目录改为编号制(01-规范/02-架构设计/03-数据库/04-调整方案/
05-交接单/06-ops/07-scripts/08-skills/09-archive),顶层散文件归入 01-规范/;
INDEX.md 与 docs-manifest.json 重刷(档案 146 篇);旧目录名引用全量对齐。

IM 线:src/im/**(SDK / hub / store / presence / ws / gateway-token)、
src/web/routes/im.ts、src/db/plugin-data/**、src/supervisor/plugin-assembly.ts
及对应 test/**。

插件线:poc/{im-agent-bridge,im-connection-gateway,im-conversation-tabs,
business-plugins-im,carbon-mcp-probe}、src/web/routes/{sessions,overlay-device}.ts、
src/net/relay/{device-grant,instance-credential}.ts。

仓库卫生:清出 40 个历史误入库 / 已改名文件(34 个交接单归档 + 6 个旧结构,
本地均有副本);dsh-server-docs/.gitignore 补 tmp/;交接单不入库(政策)。
2026-09-24 07:25:16 +08:00

182 lines
8.4 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
docs-audit.py — 文档库质量扫描(只读,零副作用)
用途:一次性回答「文档是否清晰、无歧义、需要精简」这类问题,输出**可判定**的问题清单,
避免靠人逐篇读。用于文档库根目录,也可用于任何 md 目录。
检查项(均可机器判定):
1 档案编号冲突(同一编号被多份档案占用)
2 标题号 ≠ 文件名号
3 备份/临时残留(.bak / .orig / ~ 等,会污染对账与阅读)
4 术语与事实漂移(旧术语、旧域名、已废弃表名)——历史档案保留原文属正常,需在入口说明
5 体量分布(超长文件 → 考虑拆分;过短文件 → 考虑合并/指针)
6 交叉引用有效性(引用「档案 NN」/「04-调整方案/NN-」是否存在)
7 疑似重复/近重复(同一主题两处维护;含"子集包含"识别)
8 元信息规范(档案头部是否含 日期 / 状态)
9 非 md 文件的入库合理性提示
用法:
python3 07-scripts/docs-audit.py [文档库根目录,默认取本脚本的上一级]
退出码:0 = 无 P0 级问题;1 = 发现编号冲突或失效引用(便于接 CI)。
"""
import io, os, re, sys, collections
ROOT = sys.argv[1] if len(sys.argv) > 1 else os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
P0 = 0
def rd(rel):
return io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
def all_files():
out = []
for base, dirs, names in os.walk(ROOT):
if '.git' in dirs:
dirs.remove('.git')
for n in names:
out.append(os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/'))
return sorted(out)
FILES = all_files()
MD = [f for f in FILES if f.endswith('.md')]
OTHER = [f for f in FILES if not f.endswith('.md')]
print('=' * 72)
print('文档库质量扫描 root=%s' % ROOT)
print('总文件 %d(md %d / 其他 %d)' % (len(FILES), len(MD), len(OTHER)))
print('=' * 72)
# ── 1&2 编号 ──────────────────────────────────────────────
num_map = collections.defaultdict(list)
mismatch = []
titles = {}
for f in MD:
s = rd(f)
title = next((l.strip() for l in s.split('\n') if l.strip().startswith('#')), '')
titles[f] = title
mf = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
mt = re.match(r'^#\s*(?:调整方案\s*)?(\d+[a-z]?)\s*[·、..\-]', title)
if mf and ("/" not in f or f.startswith("04-调整方案/")):
# ⛔ 只认「根级 NN-x.md」与「04-调整方案/NN-x.md」两套体系;
# 08-skills/**、01-规范/** 等目录下的编号文件(如 07-并行调度详解.md)不是档案
num_map[mf.group(1)].append(f)
if mf and mt and mf.group(1) != mt.group(1):
mismatch.append((f, title[:70]))
print('\n【1】档案编号冲突')
conf = {n: fs for n, fs in num_map.items() if len(fs) > 1}
if not conf:
print(' ✓ 无冲突')
for n, fs in sorted(conf.items()):
# 顶层 NN-x.md 与 04-调整方案/NN-x.md 属两套体系,不算冲突
top = [x for x in fs if '/' not in x]
arch = [x for x in fs if x.startswith('04-调整方案/')]
if top and arch and len(fs) == 2:
print(' · 编号 %s:根级 vs 调整方案(**两套体系,非冲突**,但入口须写明)' % n)
continue
P0 = 1
print(' ⚠ 编号 %s 被 %d 份档案占用:' % (n, len(fs)))
for x in fs:
print(' %-56s 标题「%s」' % (x, titles[x][:44]))
print('\n【2】标题号 ≠ 文件名号')
if not mismatch:
print(' ✓ 无')
for f, t in mismatch:
P0 = 1
print(' ⚠ %-52s → 「%s」' % (f, t))
# ── 3 备份残留 ───────────────────────────────────────────
print('\n【3】备份/临时残留')
junk = [f for f in FILES if re.search(r'\.bak|~$|\.orig$|\.tmp$|\.swp$|\.new$', f)]
print(' 数量 %d %s' % (len(junk), '(建议清理或纳入 .gitignore)' if junk else ''))
for f in junk[:15]:
print(' -', f)
# ── 4 术语/事实漂移 ─────────────────────────────────────
print('\n【4】术语与事实漂移(历史档案保留原文=正常,需入口说明)')
TERMS = {'业务插件(旧术语)': '业务插件', 'dsh.alotbuy.com(旧域名)': 'dsh.alotbuy.com',
'folder_plugins(已废弃)': 'folder_plugins'}
for label, k in TERMS.items():
fs = [f for f in MD if k in rd(f)]
print(' %-26s %d 个文件' % (label, len(fs)))
if fs and len(fs) <= 6:
print(' %s' % ', '.join(fs))
# ── 5 体量 ──────────────────────────────────────────────
print('\n【5】体量分布(>300 行考虑拆分;<15 行考虑合并或指针化)')
rows = sorted(((rd(f).count('\n') + 1, len(rd(f).encode()), f) for f in MD), reverse=True)
print(' 最长 8:')
for ln, by, f in rows[:8]:
print(' %5d 行 %7.1f KB %s' % (ln, by / 1024, f))
print(' 最短 5:')
for ln, by, f in rows[-5:]:
print(' %5d 行 %7.1f KB %s' % (ln, by / 1024, f))
# ── 6 引用有效性 ────────────────────────────────────────
print('\n【6】交叉引用有效性')
existing = set(num_map.keys())
bad = collections.defaultdict(list)
for f in MD:
s = rd(f)
for m in re.finditer(r'档案\s*(\d{1,2}[a-z]?)', s):
n = m.group(1)
# 键已改为字符串(支持 '37a');原先的数值范围判断 1<=n<=99 换成形状校验
if re.fullmatch(r'\d{1,2}[a-z]?', n) and n not in existing:
bad[f].append(n)
if not bad:
print(' ✓ 无悬空档案号引用')
else:
P0 = 1
for f, ns in list(bad.items())[:12]:
print(' %-48s 引用了不存在的档案号 %s' % (f, sorted(set(ns))))
# ── 7 近重复 ────────────────────────────────────────────
print('\n【7】疑似重复 / 近重复')
sig = {f: re.sub(r'\s+', '', rd(f)) for f in MD}
found = False
keys = list(sig)
for i in range(len(keys)):
for j in range(i + 1, len(keys)):
a, b = sig[keys[i]], sig[keys[j]]
if len(a) < 200 or len(b) < 200:
continue
if a == b:
print(' %-44s = %-44s 完全相同' % (keys[i], keys[j])); found = True
elif a in b or b in a:
small, big = (keys[i], keys[j]) if len(a) < len(b) else (keys[j], keys[i])
print(' %-44s ⊂ %-44s 子集(%d/%d 字符)' % (small, big, min(len(a), len(b)), max(len(a), len(b)))); found = True
else:
ga = set(a[k:k + 3] for k in range(0, len(a) - 2, 3))
gb = set(b[k:k + 3] for k in range(0, len(b) - 2, 3))
if ga and gb:
r = len(ga & gb) / min(len(ga), len(gb))
if r > 0.55:
print(' %-44s ≈ %-44s 重合 %.0f%%' % (keys[i], keys[j], r * 100)); found = True
if not found:
print(' ✓ 未发现')
# ── 8 元信息 ────────────────────────────────────────────
print('\n【8】档案头部元信息(04-调整方案/ 内)')
no_date, no_status = [], []
for f in MD:
if not f.startswith('04-调整方案/'):
continue
head = '\n'.join(rd(f).split('\n')[:14])
if not re.search(r'日期|20\d\d-\d\d-\d\d', head):
no_date.append(f)
if not re.search(r'状态', head):
no_status.append(f)
print(' 缺「日期」%d 个 %s' % (len(no_date), [os.path.basename(x) for x in no_date[:6]]))
print(' 缺「状态」%d 个 %s' % (len(no_status), [os.path.basename(x) for x in no_status[:6]]))
# ── 9 非 md ─────────────────────────────────────────────
print('\n【9】非 md 文件 %d 个(确认均属应入库的资产/脚本)' % len(OTHER))
print(' ' + (', '.join(OTHER[:12]) + (' …' if len(OTHER) > 12 else '') if OTHER else '无'))
print('\n' + '=' * 72)
print('结论:%s' % ('发现问题(见上 ⚠)' if P0 else '无 P0 级问题'))
sys.exit(1 if P0 else 0)