201 lines
10 KiB
Python
201 lines
10 KiB
Python
#!/usr/bin/env python3
|
||||
|
|
# -*- coding: utf-8 -*-
|
|||
|
|
"""
|
|||
|
|
docs-manifest.py — 生成文档库**机读清单**(供 AI/脚本过滤,替代"读大表格")
|
|||
|
|
|
|||
|
|
产出:
|
|||
|
|
docs-manifest.json 每份文档:编号/标题/状态/日期/字符/行数/被引用次数/分层(tier)/主题标签
|
|||
|
|
控制台 体量总览 + 分层统计 + 「常读」与「历史」清单
|
|||
|
|
|
|||
|
|
分层规则(2026-09-14 变更:由"**被引几次**"改为"**谁在引**"):
|
|||
|
|
hot 被**现行层 L1/L2**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范)反复引用 ≥3 次 —— 改东西前大概率要看
|
|||
|
|
cur 被现行层引用 1–2 次 —— 相关即看(单次引用也可能极重要,如红线类档案 07)
|
|||
|
|
warm 仅被历史档案互引(全库 ≥1,现行层 0)—— 参考
|
|||
|
|
cold 零引用 —— 历史,只在追溯时看
|
|||
|
|
doc 根级长期文档 / 资产,不参与档案分层
|
|||
|
|
⚠️ 实测教训:**L4(INDEX / 待办 / 台账)不能算现行层** —— 它们会顺带列出几乎所有档案号,
|
|||
|
|
会让判据反向失效(第一版含 L4 时 hot 从 44 抬到 60)。
|
|||
|
|
|
|||
|
|
用法:python3 scripts/docs-manifest.py [文档库根目录]
|
|||
|
|
副作用:仅写 docs-manifest.json(其余只读)
|
|||
|
|
"""
|
|||
|
|
import io, os, re, sys, json, collections
|
|||
|
|
|
|||
|
|
|
|||
|
|
def norm_status(raw):
|
|||
|
|
"""规范化档案头部的状态行 → 表里可读的一小段(≤14 字,去 ** 与尾部括注)。
|
|||
|
|
|
|||
|
|
⚠️ **权威方向**:档案头部是**源**,INDEX 表是**派生展示** —— 不允许把表里的 ❓ 回灌成 manifest 的 status,
|
|||
|
|
否则 manifest → INDEX → manifest 形成环形锁定,状态永远修不回来(2026-09-14 实测 14 篇被锁)。
|
|||
|
|
"""
|
|||
|
|
t = re.sub(r'\*{1,2}', '', raw or '').strip()
|
|||
|
|
t = re.split(r'——|||\||;|;|\s{2,}', t)[0].strip()
|
|||
|
|
t = re.sub(r'([^)]*)\s*$', '', t).strip()
|
|||
|
|
return t[:14]
|
|||
|
|
|
|||
|
|
ROOT = sys.argv[1] if len(sys.argv) > 1 else os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|||
|
|
|
|||
|
|
def rd(rel):
|
|||
|
|
try:
|
|||
|
|
return io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
|
|||
|
|
except OSError:
|
|||
|
|
return ''
|
|||
|
|
|
|||
|
|
MD = []
|
|||
|
|
for base, dirs, names in os.walk(ROOT):
|
|||
|
|
if '.git' in dirs:
|
|||
|
|
dirs.remove('.git')
|
|||
|
|
for n in names:
|
|||
|
|
if n.endswith('.md') and '.bak' not in n:
|
|||
|
|
MD.append(os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/'))
|
|||
|
|
MD.sort()
|
|||
|
|
|
|||
|
|
# ── 状态/日期:优先取 INDEX 表,其次取档案头部 ─────────────
|
|||
|
|
index_status = {}
|
|||
|
|
if os.path.exists(os.path.join(ROOT, 'INDEX.md')):
|
|||
|
|
for line in rd('INDEX.md').split('\n'):
|
|||
|
|
m = re.match(r'^\|\s*(?:04-)?(\d{1,3}[a-z]?)\s*\|', line)
|
|||
|
|
if not m:
|
|||
|
|
continue
|
|||
|
|
marks = [k for k in ('✅', '🔄', '🔧', '🧪', '🗄', '⏸', '❌') if k in line]
|
|||
|
|
dm = re.search(r'\b(\d{2}-\d{2})\b', line)
|
|||
|
|
index_status[m.group(1)] = {'status': marks[0] if marks else '?', 'date': dm.group(1) if dm else ''}
|
|||
|
|
|
|||
|
|
# ── 引用热度 ────────────────────────────────────────────
|
|||
|
|
DOMAIN_RULES = {
|
|||
|
|
'external': r'官方推荐|awesome|npm|cordis|开源导出|外部源|生态',
|
|||
|
|
'method': r'方法|决策|技能|skill|流程|工作流|协议|协作',
|
|||
|
|
'plugin': r'插件|univer|mcn|business-plugins|dsh-plugin|原生绑定',
|
|||
|
|
'ui': r'UI|前端|门户|页面|portal|界面|client bundle|渲染',
|
|||
|
|
'ops': r'运维|排障|重启|备份|配额|内存|OOM|证书|nginx|nft|执行锁|并发',
|
|||
|
|
'platform': r'平台|部署|隔离|bwrap|多租户|实例|profile|配额口径|架构|命名|重构|升级|耦合|回归清单|目标架构',
|
|||
|
|
'ops2': r'登录|401|404|竞态|冷启动|启动失败|排障|死锁',
|
|||
|
|
'ui2': r'会话提示|文案|提示条|覆盖层|面板|分区|交互',
|
|||
|
|
'method2': r'文档|信息架构|清单|机读|模板|规范|收尾|去重',
|
|||
|
|
}
|
|||
|
|
def domain_of(text):
|
|||
|
|
scores = collections.Counter()
|
|||
|
|
for name, pat in DOMAIN_RULES.items():
|
|||
|
|
scores[name] = len(re.findall(pat, text, re.I))
|
|||
|
|
best = scores.most_common(1)[0]
|
|||
|
|
if best[1] <= 0:
|
|||
|
|
return '?'
|
|||
|
|
return {'ops2': 'ops', 'ui2': 'ui', 'method2': 'method'}.get(best[0], best[0])
|
|||
|
|
|
|||
|
|
def layer_of(path, num):
|
|||
|
|
if num: return 'L5'
|
|||
|
|
if path.startswith('skills/'): return 'L3'
|
|||
|
|
if path.startswith('交接单/'): return 'L4'
|
|||
|
|
if path in ('BRIEF.md',): return 'L1'
|
|||
|
|
if path == 'CODEBUDDY.md': return 'L2'
|
|||
|
|
if path == 'INDEX.md': return 'L4'
|
|||
|
|
if path == 'README.md': return 'L2'
|
|||
|
|
if path == 'DEPLOY-本部署.md': return 'L1' # 现行部署事实
|
|||
|
|
if path == '06-工作台UI规范.md': return 'L2' # 强制基线 = 规则
|
|||
|
|
if path == '03-路线图与待办.md': return 'L4' # 状态/待办
|
|||
|
|
if path in ('BRIEF.md',): return 'L1'
|
|||
|
|
if path.startswith('docs-manifest') or path.endswith('.json'): return 'L2'
|
|||
|
|
if path.startswith('01-规划与架构'): return 'L0'
|
|||
|
|
if path.startswith('02-运维手册') or path.startswith('archive/'): return 'L5'
|
|||
|
|
if path == '04-调整方案/README.md': return 'L5' # 目录内索引 ⇒ 随其层
|
|||
|
|
return '?'
|
|||
|
|
|
|||
|
|
VIEW_DIR = os.path.join(os.path.dirname(ROOT), '.workbuddy', 'cache', 'dedupe-view')
|
|||
|
|
|
|||
|
|
|
|||
|
|
def view_field(rel):
|
|||
|
|
"""库外的"去重视图"(docs-dedupe.py 产物)—— 有则记路径,供检索/阅读优先使用。"""
|
|||
|
|
cand = os.path.join(VIEW_DIR, rel.replace('/', '__'))
|
|||
|
|
return {'dedupeView': cand} if os.path.exists(cand) else {}
|
|||
|
|
|
|||
|
|
|
|||
|
|
def num_of(f):
|
|||
|
|
m = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
|
|||
|
|
return m.group(1) if (m and f.startswith('04-调整方案/') and '/poc/' not in f) else None
|
|||
|
|
|
|||
|
|
# 引用分两份:**全库** 与 **仅 L1+L2 层**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范 = 现行事实与规则)。
|
|||
|
|
# ⚠️ 2026-09-14 实测教训:**L4(INDEX / 待办 / 台账)不能算** —— 它们会顺带列出几乎所有档案号,
|
|||
|
|
# 导致 hot 从 44 抬到 60(判据反而变糟);"被台账提到" ≠ "改东西前要看"。
|
|||
|
|
# 2026-09-14 判据变更:tier 由"被引几次"改为"**谁在引**" —— 只有被现行层引用才算 hot。
|
|||
|
|
ref = collections.Counter()
|
|||
|
|
ref_current = collections.Counter()
|
|||
|
|
for f in MD:
|
|||
|
|
lay = layer_of(f, num_of(f))
|
|||
|
|
for m in re.finditer(r'档案\s*(\d{1,2}[a-z]?)', rd(f)):
|
|||
|
|
ref[m.group(1)] += 1
|
|||
|
|
if lay in ('L1', 'L2'):
|
|||
|
|
ref_current[m.group(1)] += 1
|
|||
|
|
|
|||
|
|
# ── 域标签(检索键;未命中记 '?',由人补规则)─────────────
|
|||
|
|
items = []
|
|||
|
|
for f in MD:
|
|||
|
|
s = rd(f)
|
|||
|
|
title = next((l.strip().lstrip('#').strip() for l in s.split('\n') if l.strip().startswith('#')), '')
|
|||
|
|
mnum = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
|
|||
|
|
num = mnum.group(1) if (mnum and f.startswith('04-调整方案/') and '/poc/' not in f) else None
|
|||
|
|
head = '\n'.join(s.split('\n')[:14])
|
|||
|
|
m_tl = re.search(r'>\s*\*\*TL;DR\*\*[||::]?\s*(.+)', s)
|
|||
|
|
tldr = re.sub(r'\s+', ' ', m_tl.group(1)).strip(' ||')[:120] if m_tl else ''
|
|||
|
|
hd = re.search(r'(- 日期:\s*)([0-9]{4}-[0-9]{2}-[0-9]{2})', head)
|
|||
|
|
hs = re.search(r'状态\*{0,2}\s*[::]\s*(.{1,40})', head)
|
|||
|
|
hs_val = norm_status(hs.group(1)) if hs else ''
|
|||
|
|
st_idx = (index_status.get(num, {}) or {}).get('status') or ''
|
|||
|
|
if st_idx in ('?', '❓', ''):
|
|||
|
|
st_idx = '' # ❓/? 一律视为「缺」—— 防 INDEX↔manifest 环形锁定(2026-09-14)
|
|||
|
|
n_ref = ref.get(num, 0) if num else 0
|
|||
|
|
n_cur = ref_current.get(num, 0) if num else 0
|
|||
|
|
if not num:
|
|||
|
|
tier = 'doc' # 根级长期文档 / 资产:不参与"档案分层"
|
|||
|
|
elif n_cur >= 3:
|
|||
|
|
tier = 'hot' # 被**现行层反复引用** ⇒ 改东西前大概率要看
|
|||
|
|
elif n_cur >= 1:
|
|||
|
|
tier = 'cur' # 被现行层引用 1–2 次 ⇒ 相关即看(单次引用≠不重要,如红线类档案 07)
|
|||
|
|
elif n_ref >= 1:
|
|||
|
|
tier = 'warm' # 仅被历史档案互引 ⇒ 参考
|
|||
|
|
else:
|
|||
|
|
tier = 'cold' # 零引用
|
|||
|
|
items.append({
|
|||
|
|
'path': f,
|
|||
|
|
'num': num,
|
|||
|
|
'title': title[:80],
|
|||
|
|
'status': hs_val or st_idx or '?',
|
|||
|
|
'date': (hd.group(2) if hd else (index_status.get(num, {}) or {}).get('date', '') or ''),
|
|||
|
|
'chars': len(s),
|
|||
|
|
'lines': s.count('\n') + 1,
|
|||
|
|
'refs': n_ref,
|
|||
|
|
'refsCurrent': n_cur,
|
|||
|
|
'tier': tier,
|
|||
|
|
'layer': layer_of(f, num),
|
|||
|
|
'domain': domain_of(title + '\n' + head),
|
|||
|
|
'tldr': tldr,
|
|||
|
|
**view_field(f),
|
|||
|
|
})
|
|||
|
|
|
|||
|
|
out = {
|
|||
|
|
'generatedFrom': 'scripts/docs-manifest.py',
|
|||
|
|
'counts': {'files': len(items), 'chars': sum(i['chars'] for i in items)},
|
|||
|
|
'tiers': {t: sum(1 for i in items if i['tier'] == t) for t in ('hot', 'cur', 'warm', 'cold', 'doc')},
|
|||
|
|
'domains': dict(collections.Counter(i['domain'] for i in items)),
|
|||
|
|
'layers': dict(collections.Counter(i['layer'] for i in items)),
|
|||
|
|
'items': items,
|
|||
|
|
}
|
|||
|
|
io.open(os.path.join(ROOT, 'docs-manifest.json'), 'w', encoding='utf-8', newline='\n').write(
|
|||
|
|
json.dumps(out, ensure_ascii=False, indent=1) + '\n')
|
|||
|
|
|
|||
|
|
arch = [i for i in items if i['num'] is not None]
|
|||
|
|
print('文档 %d 份 / %s 字符(≈%s tokens)' % (len(items), format(out['counts']['chars'], ','),
|
|||
|
|
format(int(out['counts']['chars'] * 0.7), ',')))
|
|||
|
|
print('档案 %d 份 | 分层:hot %d / warm %d / cold %d | 根级文档 %d'
|
|||
|
|
% (len(arch), out['tiers']['hot'], out['tiers']['warm'], out['tiers']['cold'], out['tiers']['doc']))
|
|||
|
|
oversized = sorted([i for i in items if i['chars'] > 30000], key=lambda x: -x['chars'])
|
|||
|
|
if oversized:
|
|||
|
|
print('\n【⚠️ 单篇 > 30 KB】%d 篇(约定上限 30 KB;历史只报不改,新档案超限须拆):' % len(oversized))
|
|||
|
|
for i in oversized:
|
|||
|
|
print(' %6.1f KB %-58s' % (i['chars'] / 1024, os.path.basename(i['path'])[:56]))
|
|||
|
|
print('\n【hot】常读(引用 ≥8 次)—— 改东西前大概率要看:')
|
|||
|
|
for i in sorted([x for x in arch if x['tier'] == 'hot'], key=lambda x: -x['refs']):
|
|||
|
|
print(' 档案 %-3s %2d 次 %-52s %6d 字符' % (i['num'], i['refs'], os.path.basename(i['path'])[:50], i['chars']))
|
|||
|
|
print('\n【cold】零引用(历史候选,可只留索引行):')
|
|||
|
|
for i in sorted([x for x in arch if x['tier'] == 'cold'], key=lambda x: x['num']):
|
|||
|
|
print(' 档案 %-3s %s' % (i['num'], os.path.basename(i['path'])))
|
|||
|
|
print('\n→ 已写出 docs-manifest.json')
|