#!/usr/bin/env python3 # -*- coding: utf-8 -*- """ docs-manifest.py — 生成文档库**机读清单**(供 AI/脚本过滤,替代"读大表格") 产出: docs-manifest.json 每份文档:编号/标题/状态/日期/字符/行数/被引用次数/分层(tier)/主题标签 控制台 体量总览 + 分层统计 + 「常读」与「历史」清单 分层规则(2026-09-14 变更:由"**被引几次**"改为"**谁在引**"): hot 被**现行层 L1/L2**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范)反复引用 ≥3 次 —— 改东西前大概率要看 cur 被现行层引用 1–2 次 —— 相关即看(单次引用也可能极重要,如红线类档案 07) warm 仅被历史档案互引(全库 ≥1,现行层 0)—— 参考 cold 零引用 —— 历史,只在追溯时看 doc 根级长期文档 / 资产,不参与档案分层 ⚠️ 实测教训:**L4(INDEX / 待办 / 台账)不能算现行层** —— 它们会顺带列出几乎所有档案号, 会让判据反向失效(第一版含 L4 时 hot 从 44 抬到 60)。 用法:python3 07-scripts/docs-manifest.py [文档库根目录] 副作用:仅写 docs-manifest.json(其余只读) """ import io, os, re, sys, json, collections def norm_status(raw): """规范化档案头部的状态行 → 表里可读的一小段(≤14 字,去 ** 与尾部括注)。 ⚠️ **权威方向**:档案头部是**源**,INDEX 表是**派生展示** —— 不允许把表里的 ❓ 回灌成 manifest 的 status, 否则 manifest → INDEX → manifest 形成环形锁定,状态永远修不回来(2026-09-14 实测 14 篇被锁)。 """ t = re.sub(r'\*{1,2}', '', raw or '').strip() t = re.split(r'——|||\||;|;|\s{2,}', t)[0].strip() t = re.sub(r'([^)]*)\s*$', '', t).strip() return t[:14] ROOT = sys.argv[1] if len(sys.argv) > 1 else os.path.dirname(os.path.dirname(os.path.abspath(__file__))) def rd(rel): try: return io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read() except OSError: return '' MD = [] for base, dirs, names in os.walk(ROOT): if '.git' in dirs: dirs.remove('.git') for n in names: if n.endswith('.md') and '.bak' not in n: MD.append(os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/')) MD.sort() # ── 状态/日期:优先取 INDEX 表,其次取档案头部 ───────────── index_status = {} if os.path.exists(os.path.join(ROOT, 'INDEX.md')): for line in rd('INDEX.md').split('\n'): m = re.match(r'^\|\s*(?:04-)?(\d{1,3}[a-z]?)\s*\|', line) if not m: continue marks = [k for k in ('✅', '🔄', '🔧', '🧪', '🗄', '⏸', '❌') if k in line] dm = re.search(r'\b(\d{2}-\d{2})\b', line) index_status[m.group(1)] = {'status': marks[0] if marks else '?', 'date': dm.group(1) if dm else ''} # ── 引用热度 ──────────────────────────────────────────── DOMAIN_RULES = { 'external': r'官方推荐|awesome|npm|cordis|开源导出|外部源|生态', 'method': r'方法|决策|技能|skill|流程|工作流|协议|协作', 'plugin': r'插件|univer|mcn|business-plugins|dsh-plugin|原生绑定', 'ui': r'UI|前端|门户|页面|portal|界面|client bundle|渲染', '06-ops': r'运维|排障|重启|备份|配额|内存|OOM|证书|nginx|nft|执行锁|并发', 'platform': r'平台|部署|隔离|bwrap|多租户|实例|profile|配额口径|架构|命名|重构|升级|耦合|回归清单|目标架构', 'ops2': r'登录|401|404|竞态|冷启动|启动失败|排障|死锁', 'ui2': r'会话提示|文案|提示条|覆盖层|面板|分区|交互', 'method2': r'文档|信息架构|清单|机读|模板|规范|收尾|去重', } def domain_of(text): scores = collections.Counter() for name, pat in DOMAIN_RULES.items(): scores[name] = len(re.findall(pat, text, re.I)) best = scores.most_common(1)[0] if best[1] <= 0: return '?' return {'ops2': '06-ops', 'ui2': 'ui', 'method2': 'method'}.get(best[0], best[0]) def layer_of(path, num): if num: return 'L5' if path.startswith('08-skills/'): return 'L3' if path.startswith('05-交接单/'): return 'L4' if path in ('BRIEF.md',): return 'L1' if path == 'CODEBUDDY.md': return 'L2' if path == 'INDEX.md': return 'L4' if path == 'README.md': return 'L2' if path == 'DEPLOY-本部署.md': return 'L1' # 现行部署事实 if path == '01-规范/06-工作台UI规范.md': return 'L2' # 强制基线 = 规则 if path == '01-规范/03-路线图与待办.md': return 'L4' # 状态/待办 if path in ('BRIEF.md',): return 'L1' if path.startswith('docs-manifest') or path.endswith('.json'): return 'L2' if path.startswith('01-规范/01-规划与架构'): return 'L0' if path.startswith('01-规范/02-运维手册') or path.startswith('09-archive/'): return 'L5' if path == '04-调整方案/README.md': return 'L5' # 目录内索引 ⇒ 随其层 return '?' VIEW_DIR = os.path.join(os.path.dirname(ROOT), '.workbuddy', 'cache', 'dedupe-view') def view_field(rel): """库外的"去重视图"(docs-dedupe.py 产物)—— 有则记路径,供检索/阅读优先使用。""" cand = os.path.join(VIEW_DIR, rel.replace('/', '__')) return {'dedupeView': cand} if os.path.exists(cand) else {} def num_of(f): m = re.match(r'^(\d+[a-z]?)-', os.path.basename(f)) return m.group(1) if (m and f.startswith('04-调整方案/') and '/poc/' not in f) else None # 引用分两份:**全库** 与 **仅 L1+L2 层**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范 = 现行事实与规则)。 # ⚠️ 2026-09-14 实测教训:**L4(INDEX / 待办 / 台账)不能算** —— 它们会顺带列出几乎所有档案号, # 导致 hot 从 44 抬到 60(判据反而变糟);"被台账提到" ≠ "改东西前要看"。 # 2026-09-14 判据变更:tier 由"被引几次"改为"**谁在引**" —— 只有被现行层引用才算 hot。 ref = collections.Counter() ref_current = collections.Counter() for f in MD: lay = layer_of(f, num_of(f)) for m in re.finditer(r'档案\s*(\d{1,2}[a-z]?)', rd(f)): ref[m.group(1)] += 1 if lay in ('L1', 'L2'): ref_current[m.group(1)] += 1 # ── 域标签(检索键;未命中记 '?',由人补规则)───────────── items = [] for f in MD: s = rd(f) title = next((l.strip().lstrip('#').strip() for l in s.split('\n') if l.strip().startswith('#')), '') mnum = re.match(r'^(\d+[a-z]?)-', os.path.basename(f)) num = mnum.group(1) if (mnum and f.startswith('04-调整方案/') and '/poc/' not in f) else None head = '\n'.join(s.split('\n')[:14]) m_tl = re.search(r'>\s*\*\*TL;DR\*\*[||::]?\s*(.+)', s) tldr = re.sub(r'\s+', ' ', m_tl.group(1)).strip(' ||')[:120] if m_tl else '' hd = re.search(r'(- 日期:\s*)([0-9]{4}-[0-9]{2}-[0-9]{2})', head) hs = re.search(r'状态\*{0,2}\s*[::]\s*(.{1,40})', head) hs_val = norm_status(hs.group(1)) if hs else '' st_idx = (index_status.get(num, {}) or {}).get('status') or '' if st_idx in ('?', '❓', ''): st_idx = '' # ❓/? 一律视为「缺」—— 防 INDEX↔manifest 环形锁定(2026-09-14) n_ref = ref.get(num, 0) if num else 0 n_cur = ref_current.get(num, 0) if num else 0 if not num: tier = 'doc' # 根级长期文档 / 资产:不参与"档案分层" elif n_cur >= 3: tier = 'hot' # 被**现行层反复引用** ⇒ 改东西前大概率要看 elif n_cur >= 1: tier = 'cur' # 被现行层引用 1–2 次 ⇒ 相关即看(单次引用≠不重要,如红线类档案 07) elif n_ref >= 1: tier = 'warm' # 仅被历史档案互引 ⇒ 参考 else: tier = 'cold' # 零引用 items.append({ 'path': f, 'num': num, 'title': title[:80], 'status': hs_val or st_idx or '?', 'date': (hd.group(2) if hd else (index_status.get(num, {}) or {}).get('date', '') or ''), 'chars': len(s), 'lines': s.count('\n') + 1, 'refs': n_ref, 'refsCurrent': n_cur, 'tier': tier, 'layer': layer_of(f, num), 'domain': domain_of(title + '\n' + head), 'tldr': tldr, **view_field(f), }) out = { 'generatedFrom': '07-scripts/docs-manifest.py', 'counts': {'files': len(items), 'chars': sum(i['chars'] for i in items)}, 'tiers': {t: sum(1 for i in items if i['tier'] == t) for t in ('hot', 'cur', 'warm', 'cold', 'doc')}, 'domains': dict(collections.Counter(i['domain'] for i in items)), 'layers': dict(collections.Counter(i['layer'] for i in items)), 'items': items, } io.open(os.path.join(ROOT, 'docs-manifest.json'), 'w', encoding='utf-8', newline='\n').write( json.dumps(out, ensure_ascii=False, indent=1) + '\n') arch = [i for i in items if i['num'] is not None] print('文档 %d 份 / %s 字符(≈%s tokens)' % (len(items), format(out['counts']['chars'], ','), format(int(out['counts']['chars'] * 0.7), ','))) print('档案 %d 份 | 分层:hot %d / warm %d / cold %d | 根级文档 %d' % (len(arch), out['tiers']['hot'], out['tiers']['warm'], out['tiers']['cold'], out['tiers']['doc'])) oversized = sorted([i for i in items if i['chars'] > 30000], key=lambda x: -x['chars']) if oversized: print('\n【⚠️ 单篇 > 30 KB】%d 篇(约定上限 30 KB;历史只报不改,新档案超限须拆):' % len(oversized)) for i in oversized: print(' %6.1f KB %-58s' % (i['chars'] / 1024, os.path.basename(i['path'])[:56])) print('\n【hot】常读(引用 ≥8 次)—— 改东西前大概率要看:') for i in sorted([x for x in arch if x['tier'] == 'hot'], key=lambda x: -x['refs']): print(' 档案 %-3s %2d 次 %-52s %6d 字符' % (i['num'], i['refs'], os.path.basename(i['path'])[:50], i['chars'])) print('\n【cold】零引用(历史候选,可只留索引行):') for i in sorted([x for x in arch if x['tier'] == 'cold'], key=lambda x: x['num']): print(' 档案 %-3s %s' % (i['num'], os.path.basename(i['path']))) print('\n→ 已写出 docs-manifest.json')