2026-09-15 18:47:13 +08:00
|
|
|
|
#!/usr/bin/env python3
|
|
|
|
|
|
# -*- coding: utf-8 -*-
|
|
|
|
|
|
"""
|
|
|
|
|
|
docs-audit.py — 文档库质量扫描(只读,零副作用)
|
|
|
|
|
|
|
|
|
|
|
|
用途:一次性回答「文档是否清晰、无歧义、需要精简」这类问题,输出**可判定**的问题清单,
|
|
|
|
|
|
避免靠人逐篇读。用于文档库根目录,也可用于任何 md 目录。
|
|
|
|
|
|
|
|
|
|
|
|
检查项(均可机器判定):
|
|
|
|
|
|
1 档案编号冲突(同一编号被多份档案占用)
|
|
|
|
|
|
2 标题号 ≠ 文件名号
|
|
|
|
|
|
3 备份/临时残留(.bak / .orig / ~ 等,会污染对账与阅读)
|
|
|
|
|
|
4 术语与事实漂移(旧术语、旧域名、已废弃表名)——历史档案保留原文属正常,需在入口说明
|
|
|
|
|
|
5 体量分布(超长文件 → 考虑拆分;过短文件 → 考虑合并/指针)
|
|
|
|
|
|
6 交叉引用有效性(引用「档案 NN」/「04-调整方案/NN-」是否存在)
|
|
|
|
|
|
7 疑似重复/近重复(同一主题两处维护;含"子集包含"识别)
|
|
|
|
|
|
8 元信息规范(档案头部是否含 日期 / 状态)
|
|
|
|
|
|
9 非 md 文件的入库合理性提示
|
|
|
|
|
|
|
|
|
|
|
|
用法:
|
2026-09-24 07:25:16 +08:00
|
|
|
|
python3 07-scripts/docs-audit.py [文档库根目录,默认取本脚本的上一级]
|
2026-09-15 18:47:13 +08:00
|
|
|
|
|
|
|
|
|
|
退出码:0 = 无 P0 级问题;1 = 发现编号冲突或失效引用(便于接 CI)。
|
|
|
|
|
|
"""
|
|
|
|
|
|
import io, os, re, sys, collections
|
|
|
|
|
|
|
|
|
|
|
|
ROOT = sys.argv[1] if len(sys.argv) > 1 else os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
|
|
|
|
P0 = 0
|
|
|
|
|
|
|
|
|
|
|
|
def rd(rel):
|
|
|
|
|
|
return io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
|
|
|
|
|
|
|
|
|
|
|
|
def all_files():
|
|
|
|
|
|
out = []
|
|
|
|
|
|
for base, dirs, names in os.walk(ROOT):
|
|
|
|
|
|
if '.git' in dirs:
|
|
|
|
|
|
dirs.remove('.git')
|
|
|
|
|
|
for n in names:
|
|
|
|
|
|
out.append(os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/'))
|
|
|
|
|
|
return sorted(out)
|
|
|
|
|
|
|
|
|
|
|
|
FILES = all_files()
|
|
|
|
|
|
MD = [f for f in FILES if f.endswith('.md')]
|
|
|
|
|
|
OTHER = [f for f in FILES if not f.endswith('.md')]
|
|
|
|
|
|
|
|
|
|
|
|
print('=' * 72)
|
|
|
|
|
|
print('文档库质量扫描 root=%s' % ROOT)
|
|
|
|
|
|
print('总文件 %d(md %d / 其他 %d)' % (len(FILES), len(MD), len(OTHER)))
|
|
|
|
|
|
print('=' * 72)
|
|
|
|
|
|
|
|
|
|
|
|
# ── 1&2 编号 ──────────────────────────────────────────────
|
|
|
|
|
|
num_map = collections.defaultdict(list)
|
|
|
|
|
|
mismatch = []
|
|
|
|
|
|
titles = {}
|
|
|
|
|
|
for f in MD:
|
|
|
|
|
|
s = rd(f)
|
|
|
|
|
|
title = next((l.strip() for l in s.split('\n') if l.strip().startswith('#')), '')
|
|
|
|
|
|
titles[f] = title
|
|
|
|
|
|
mf = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
|
|
|
|
|
|
mt = re.match(r'^#\s*(?:调整方案\s*)?(\d+[a-z]?)\s*[·、..\-]', title)
|
2026-09-24 07:25:16 +08:00
|
|
|
|
if mf and ("/" not in f or f.startswith("04-调整方案/")):
|
|
|
|
|
|
# ⛔ 只认「根级 NN-x.md」与「04-调整方案/NN-x.md」两套体系;
|
|
|
|
|
|
# 08-skills/**、01-规范/** 等目录下的编号文件(如 07-并行调度详解.md)不是档案
|
2026-09-15 18:47:13 +08:00
|
|
|
|
num_map[mf.group(1)].append(f)
|
|
|
|
|
|
if mf and mt and mf.group(1) != mt.group(1):
|
|
|
|
|
|
mismatch.append((f, title[:70]))
|
|
|
|
|
|
|
|
|
|
|
|
print('\n【1】档案编号冲突')
|
|
|
|
|
|
conf = {n: fs for n, fs in num_map.items() if len(fs) > 1}
|
|
|
|
|
|
if not conf:
|
|
|
|
|
|
print(' ✓ 无冲突')
|
|
|
|
|
|
for n, fs in sorted(conf.items()):
|
|
|
|
|
|
# 顶层 NN-x.md 与 04-调整方案/NN-x.md 属两套体系,不算冲突
|
|
|
|
|
|
top = [x for x in fs if '/' not in x]
|
|
|
|
|
|
arch = [x for x in fs if x.startswith('04-调整方案/')]
|
|
|
|
|
|
if top and arch and len(fs) == 2:
|
2026-09-24 07:25:16 +08:00
|
|
|
|
print(' · 编号 %s:根级 vs 调整方案(**两套体系,非冲突**,但入口须写明)' % n)
|
2026-09-15 18:47:13 +08:00
|
|
|
|
continue
|
|
|
|
|
|
P0 = 1
|
|
|
|
|
|
print(' ⚠ 编号 %s 被 %d 份档案占用:' % (n, len(fs)))
|
|
|
|
|
|
for x in fs:
|
|
|
|
|
|
print(' %-56s 标题「%s」' % (x, titles[x][:44]))
|
|
|
|
|
|
|
|
|
|
|
|
print('\n【2】标题号 ≠ 文件名号')
|
|
|
|
|
|
if not mismatch:
|
|
|
|
|
|
print(' ✓ 无')
|
|
|
|
|
|
for f, t in mismatch:
|
|
|
|
|
|
P0 = 1
|
|
|
|
|
|
print(' ⚠ %-52s → 「%s」' % (f, t))
|
|
|
|
|
|
|
|
|
|
|
|
# ── 3 备份残留 ───────────────────────────────────────────
|
|
|
|
|
|
print('\n【3】备份/临时残留')
|
|
|
|
|
|
junk = [f for f in FILES if re.search(r'\.bak|~$|\.orig$|\.tmp$|\.swp$|\.new$', f)]
|
|
|
|
|
|
print(' 数量 %d %s' % (len(junk), '(建议清理或纳入 .gitignore)' if junk else ''))
|
|
|
|
|
|
for f in junk[:15]:
|
|
|
|
|
|
print(' -', f)
|
|
|
|
|
|
|
|
|
|
|
|
# ── 4 术语/事实漂移 ─────────────────────────────────────
|
|
|
|
|
|
print('\n【4】术语与事实漂移(历史档案保留原文=正常,需入口说明)')
|
|
|
|
|
|
TERMS = {'业务插件(旧术语)': '业务插件', 'dsh.alotbuy.com(旧域名)': 'dsh.alotbuy.com',
|
|
|
|
|
|
'folder_plugins(已废弃)': 'folder_plugins'}
|
|
|
|
|
|
for label, k in TERMS.items():
|
|
|
|
|
|
fs = [f for f in MD if k in rd(f)]
|
|
|
|
|
|
print(' %-26s %d 个文件' % (label, len(fs)))
|
|
|
|
|
|
if fs and len(fs) <= 6:
|
|
|
|
|
|
print(' %s' % ', '.join(fs))
|
|
|
|
|
|
|
|
|
|
|
|
# ── 5 体量 ──────────────────────────────────────────────
|
|
|
|
|
|
print('\n【5】体量分布(>300 行考虑拆分;<15 行考虑合并或指针化)')
|
|
|
|
|
|
rows = sorted(((rd(f).count('\n') + 1, len(rd(f).encode()), f) for f in MD), reverse=True)
|
|
|
|
|
|
print(' 最长 8:')
|
|
|
|
|
|
for ln, by, f in rows[:8]:
|
|
|
|
|
|
print(' %5d 行 %7.1f KB %s' % (ln, by / 1024, f))
|
|
|
|
|
|
print(' 最短 5:')
|
|
|
|
|
|
for ln, by, f in rows[-5:]:
|
|
|
|
|
|
print(' %5d 行 %7.1f KB %s' % (ln, by / 1024, f))
|
|
|
|
|
|
|
|
|
|
|
|
# ── 6 引用有效性 ────────────────────────────────────────
|
|
|
|
|
|
print('\n【6】交叉引用有效性')
|
|
|
|
|
|
existing = set(num_map.keys())
|
|
|
|
|
|
bad = collections.defaultdict(list)
|
|
|
|
|
|
for f in MD:
|
|
|
|
|
|
s = rd(f)
|
|
|
|
|
|
for m in re.finditer(r'档案\s*(\d{1,2}[a-z]?)', s):
|
|
|
|
|
|
n = m.group(1)
|
|
|
|
|
|
# 键已改为字符串(支持 '37a');原先的数值范围判断 1<=n<=99 换成形状校验
|
|
|
|
|
|
if re.fullmatch(r'\d{1,2}[a-z]?', n) and n not in existing:
|
|
|
|
|
|
bad[f].append(n)
|
|
|
|
|
|
if not bad:
|
|
|
|
|
|
print(' ✓ 无悬空档案号引用')
|
|
|
|
|
|
else:
|
|
|
|
|
|
P0 = 1
|
|
|
|
|
|
for f, ns in list(bad.items())[:12]:
|
|
|
|
|
|
print(' %-48s 引用了不存在的档案号 %s' % (f, sorted(set(ns))))
|
|
|
|
|
|
|
|
|
|
|
|
# ── 7 近重复 ────────────────────────────────────────────
|
|
|
|
|
|
print('\n【7】疑似重复 / 近重复')
|
|
|
|
|
|
sig = {f: re.sub(r'\s+', '', rd(f)) for f in MD}
|
|
|
|
|
|
found = False
|
|
|
|
|
|
keys = list(sig)
|
|
|
|
|
|
for i in range(len(keys)):
|
|
|
|
|
|
for j in range(i + 1, len(keys)):
|
|
|
|
|
|
a, b = sig[keys[i]], sig[keys[j]]
|
|
|
|
|
|
if len(a) < 200 or len(b) < 200:
|
|
|
|
|
|
continue
|
|
|
|
|
|
if a == b:
|
|
|
|
|
|
print(' %-44s = %-44s 完全相同' % (keys[i], keys[j])); found = True
|
|
|
|
|
|
elif a in b or b in a:
|
|
|
|
|
|
small, big = (keys[i], keys[j]) if len(a) < len(b) else (keys[j], keys[i])
|
|
|
|
|
|
print(' %-44s ⊂ %-44s 子集(%d/%d 字符)' % (small, big, min(len(a), len(b)), max(len(a), len(b)))); found = True
|
|
|
|
|
|
else:
|
|
|
|
|
|
ga = set(a[k:k + 3] for k in range(0, len(a) - 2, 3))
|
|
|
|
|
|
gb = set(b[k:k + 3] for k in range(0, len(b) - 2, 3))
|
|
|
|
|
|
if ga and gb:
|
|
|
|
|
|
r = len(ga & gb) / min(len(ga), len(gb))
|
|
|
|
|
|
if r > 0.55:
|
|
|
|
|
|
print(' %-44s ≈ %-44s 重合 %.0f%%' % (keys[i], keys[j], r * 100)); found = True
|
|
|
|
|
|
if not found:
|
|
|
|
|
|
print(' ✓ 未发现')
|
|
|
|
|
|
|
|
|
|
|
|
# ── 8 元信息 ────────────────────────────────────────────
|
|
|
|
|
|
print('\n【8】档案头部元信息(04-调整方案/ 内)')
|
|
|
|
|
|
no_date, no_status = [], []
|
|
|
|
|
|
for f in MD:
|
|
|
|
|
|
if not f.startswith('04-调整方案/'):
|
|
|
|
|
|
continue
|
|
|
|
|
|
head = '\n'.join(rd(f).split('\n')[:14])
|
|
|
|
|
|
if not re.search(r'日期|20\d\d-\d\d-\d\d', head):
|
|
|
|
|
|
no_date.append(f)
|
|
|
|
|
|
if not re.search(r'状态', head):
|
|
|
|
|
|
no_status.append(f)
|
|
|
|
|
|
print(' 缺「日期」%d 个 %s' % (len(no_date), [os.path.basename(x) for x in no_date[:6]]))
|
|
|
|
|
|
print(' 缺「状态」%d 个 %s' % (len(no_status), [os.path.basename(x) for x in no_status[:6]]))
|
|
|
|
|
|
|
|
|
|
|
|
# ── 9 非 md ─────────────────────────────────────────────
|
|
|
|
|
|
print('\n【9】非 md 文件 %d 个(确认均属应入库的资产/脚本)' % len(OTHER))
|
|
|
|
|
|
print(' ' + (', '.join(OTHER[:12]) + (' …' if len(OTHER) > 12 else '') if OTHER else '无'))
|
|
|
|
|
|
|
|
|
|
|
|
print('\n' + '=' * 72)
|
|
|
|
|
|
print('结论:%s' % ('发现问题(见上 ⚠)' if P0 else '无 P0 级问题'))
|
|
|
|
|
|
sys.exit(1 if P0 else 0)
|