build / build-and-scan (push) Waiting to run
文档库:目录改为编号制(01-规范/02-架构设计/03-数据库/04-调整方案/
05-交接单/06-ops/07-scripts/08-skills/09-archive),顶层散文件归入 01-规范/;
INDEX.md 与 docs-manifest.json 重刷(档案 146 篇);旧目录名引用全量对齐。
IM 线:src/im/**(SDK / hub / store / presence / ws / gateway-token)、
src/web/routes/im.ts、src/db/plugin-data/**、src/supervisor/plugin-assembly.ts
及对应 test/**。
插件线:poc/{im-agent-bridge,im-connection-gateway,im-conversation-tabs,
business-plugins-im,carbon-mcp-probe}、src/web/routes/{sessions,overlay-device}.ts、
src/net/relay/{device-grant,instance-credential}.ts。
仓库卫生:清出 40 个历史误入库 / 已改名文件(34 个交接单归档 + 6 个旧结构,
本地均有副本);dsh-server-docs/.gitignore 补 tmp/;交接单不入库(政策)。
202 lines
10 KiB
Python
202 lines
10 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
docs-manifest.py — 生成文档库**机读清单**(供 AI/脚本过滤,替代"读大表格")
|
||
|
||
产出:
|
||
docs-manifest.json 每份文档:编号/标题/状态/日期/字符/行数/被引用次数/分层(tier)/主题标签
|
||
控制台 体量总览 + 分层统计 + 「常读」与「历史」清单
|
||
|
||
分层规则(2026-09-14 变更:由"**被引几次**"改为"**谁在引**"):
|
||
hot 被**现行层 L1/L2**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范)反复引用 ≥3 次 —— 改东西前大概率要看
|
||
cur 被现行层引用 1–2 次 —— 相关即看(单次引用也可能极重要,如红线类档案 07)
|
||
warm 仅被历史档案互引(全库 ≥1,现行层 0)—— 参考
|
||
cold 零引用 —— 历史,只在追溯时看
|
||
doc 根级长期文档 / 资产,不参与档案分层
|
||
⚠️ 实测教训:**L4(INDEX / 待办 / 台账)不能算现行层** —— 它们会顺带列出几乎所有档案号,
|
||
会让判据反向失效(第一版含 L4 时 hot 从 44 抬到 60)。
|
||
|
||
用法:python3 07-scripts/docs-manifest.py [文档库根目录]
|
||
副作用:仅写 docs-manifest.json(其余只读)
|
||
"""
|
||
import io, os, re, sys, json, collections
|
||
|
||
|
||
def norm_status(raw):
|
||
"""规范化档案头部的状态行 → 表里可读的一小段(≤14 字,去 ** 与尾部括注)。
|
||
|
||
⚠️ **权威方向**:档案头部是**源**,INDEX 表是**派生展示** —— 不允许把表里的 ❓ 回灌成 manifest 的 status,
|
||
否则 manifest → INDEX → manifest 形成环形锁定,状态永远修不回来(2026-09-14 实测 14 篇被锁)。
|
||
"""
|
||
t = re.sub(r'\*{1,2}', '', raw or '').strip()
|
||
t = re.split(r'——|||\||;|;|\s{2,}', t)[0].strip()
|
||
t = re.sub(r'([^)]*)\s*$', '', t).strip()
|
||
return t[:14]
|
||
|
||
ROOT = sys.argv[1] if len(sys.argv) > 1 else os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
|
||
def rd(rel):
|
||
try:
|
||
return io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
|
||
except OSError:
|
||
return ''
|
||
|
||
MD = []
|
||
for base, dirs, names in os.walk(ROOT):
|
||
if '.git' in dirs:
|
||
dirs.remove('.git')
|
||
for n in names:
|
||
if n.endswith('.md') and '.bak' not in n:
|
||
MD.append(os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/'))
|
||
MD.sort()
|
||
|
||
# ── 状态/日期:优先取 INDEX 表,其次取档案头部 ─────────────
|
||
index_status = {}
|
||
if os.path.exists(os.path.join(ROOT, 'INDEX.md')):
|
||
for line in rd('INDEX.md').split('\n'):
|
||
m = re.match(r'^\|\s*(?:04-)?(\d{1,3}[a-z]?)\s*\|', line)
|
||
if not m:
|
||
continue
|
||
marks = [k for k in ('✅', '🔄', '🔧', '🧪', '🗄', '⏸', '❌') if k in line]
|
||
dm = re.search(r'\b(\d{2}-\d{2})\b', line)
|
||
index_status[m.group(1)] = {'status': marks[0] if marks else '?', 'date': dm.group(1) if dm else ''}
|
||
|
||
# ── 引用热度 ────────────────────────────────────────────
|
||
DOMAIN_RULES = {
|
||
'external': r'官方推荐|awesome|npm|cordis|开源导出|外部源|生态',
|
||
'method': r'方法|决策|技能|skill|流程|工作流|协议|协作',
|
||
'plugin': r'插件|univer|mcn|business-plugins|dsh-plugin|原生绑定',
|
||
'ui': r'UI|前端|门户|页面|portal|界面|client bundle|渲染',
|
||
'06-ops': r'运维|排障|重启|备份|配额|内存|OOM|证书|nginx|nft|执行锁|并发',
|
||
'platform': r'平台|部署|隔离|bwrap|多租户|实例|profile|配额口径|架构|命名|重构|升级|耦合|回归清单|目标架构',
|
||
'ops2': r'登录|401|404|竞态|冷启动|启动失败|排障|死锁',
|
||
'ui2': r'会话提示|文案|提示条|覆盖层|面板|分区|交互',
|
||
'method2': r'文档|信息架构|清单|机读|模板|规范|收尾|去重',
|
||
}
|
||
def domain_of(text):
|
||
scores = collections.Counter()
|
||
for name, pat in DOMAIN_RULES.items():
|
||
scores[name] = len(re.findall(pat, text, re.I))
|
||
best = scores.most_common(1)[0]
|
||
if best[1] <= 0:
|
||
return '?'
|
||
return {'ops2': '06-ops', 'ui2': 'ui', 'method2': 'method'}.get(best[0], best[0])
|
||
|
||
def layer_of(path, num):
|
||
if num: return 'L5'
|
||
if path.startswith('08-skills/'): return 'L3'
|
||
if path.startswith('05-交接单/'): return 'L4'
|
||
if path in ('BRIEF.md',): return 'L1'
|
||
if path == 'CODEBUDDY.md': return 'L2'
|
||
if path == 'INDEX.md': return 'L4'
|
||
if path == 'README.md': return 'L2'
|
||
if path == 'DEPLOY-本部署.md': return 'L1' # 现行部署事实
|
||
if path == '01-规范/06-工作台UI规范.md': return 'L2' # 强制基线 = 规则
|
||
if path == '01-规范/03-路线图与待办.md': return 'L4' # 状态/待办
|
||
if path in ('BRIEF.md',): return 'L1'
|
||
if path.startswith('docs-manifest') or path.endswith('.json'): return 'L2'
|
||
if path.startswith('01-规范/01-规划与架构'): return 'L0'
|
||
if path.startswith('01-规范/02-运维手册') or path.startswith('09-archive/'): return 'L5'
|
||
if path == '04-调整方案/README.md': return 'L5' # 目录内索引 ⇒ 随其层
|
||
return '?'
|
||
|
||
VIEW_DIR = os.path.join(os.path.dirname(ROOT), '.workbuddy', 'cache', 'dedupe-view')
|
||
|
||
|
||
def view_field(rel):
|
||
"""库外的"去重视图"(docs-dedupe.py 产物)—— 有则记路径,供检索/阅读优先使用。"""
|
||
cand = os.path.join(VIEW_DIR, rel.replace('/', '__'))
|
||
return {'dedupeView': cand} if os.path.exists(cand) else {}
|
||
|
||
|
||
def num_of(f):
|
||
m = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
|
||
return m.group(1) if (m and f.startswith('04-调整方案/') and '/poc/' not in f) else None
|
||
|
||
# 引用分两份:**全库** 与 **仅 L1+L2 层**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范 = 现行事实与规则)。
|
||
# ⚠️ 2026-09-14 实测教训:**L4(INDEX / 待办 / 台账)不能算** —— 它们会顺带列出几乎所有档案号,
|
||
# 导致 hot 从 44 抬到 60(判据反而变糟);"被台账提到" ≠ "改东西前要看"。
|
||
# 2026-09-14 判据变更:tier 由"被引几次"改为"**谁在引**" —— 只有被现行层引用才算 hot。
|
||
ref = collections.Counter()
|
||
ref_current = collections.Counter()
|
||
for f in MD:
|
||
lay = layer_of(f, num_of(f))
|
||
for m in re.finditer(r'档案\s*(\d{1,2}[a-z]?)', rd(f)):
|
||
ref[m.group(1)] += 1
|
||
if lay in ('L1', 'L2'):
|
||
ref_current[m.group(1)] += 1
|
||
|
||
# ── 域标签(检索键;未命中记 '?',由人补规则)─────────────
|
||
items = []
|
||
for f in MD:
|
||
s = rd(f)
|
||
title = next((l.strip().lstrip('#').strip() for l in s.split('\n') if l.strip().startswith('#')), '')
|
||
mnum = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
|
||
num = mnum.group(1) if (mnum and f.startswith('04-调整方案/') and '/poc/' not in f) else None
|
||
head = '\n'.join(s.split('\n')[:14])
|
||
m_tl = re.search(r'>\s*\*\*TL;DR\*\*[||::]?\s*(.+)', s)
|
||
tldr = re.sub(r'\s+', ' ', m_tl.group(1)).strip(' ||')[:120] if m_tl else ''
|
||
hd = re.search(r'(- 日期:\s*)([0-9]{4}-[0-9]{2}-[0-9]{2})', head)
|
||
hs = re.search(r'状态\*{0,2}\s*[::]\s*(.{1,40})', head)
|
||
hs_val = norm_status(hs.group(1)) if hs else ''
|
||
st_idx = (index_status.get(num, {}) or {}).get('status') or ''
|
||
if st_idx in ('?', '❓', ''):
|
||
st_idx = '' # ❓/? 一律视为「缺」—— 防 INDEX↔manifest 环形锁定(2026-09-14)
|
||
n_ref = ref.get(num, 0) if num else 0
|
||
n_cur = ref_current.get(num, 0) if num else 0
|
||
if not num:
|
||
tier = 'doc' # 根级长期文档 / 资产:不参与"档案分层"
|
||
elif n_cur >= 3:
|
||
tier = 'hot' # 被**现行层反复引用** ⇒ 改东西前大概率要看
|
||
elif n_cur >= 1:
|
||
tier = 'cur' # 被现行层引用 1–2 次 ⇒ 相关即看(单次引用≠不重要,如红线类档案 07)
|
||
elif n_ref >= 1:
|
||
tier = 'warm' # 仅被历史档案互引 ⇒ 参考
|
||
else:
|
||
tier = 'cold' # 零引用
|
||
items.append({
|
||
'path': f,
|
||
'num': num,
|
||
'title': title[:80],
|
||
'status': hs_val or st_idx or '?',
|
||
'date': (hd.group(2) if hd else (index_status.get(num, {}) or {}).get('date', '') or ''),
|
||
'chars': len(s),
|
||
'lines': s.count('\n') + 1,
|
||
'refs': n_ref,
|
||
'refsCurrent': n_cur,
|
||
'tier': tier,
|
||
'layer': layer_of(f, num),
|
||
'domain': domain_of(title + '\n' + head),
|
||
'tldr': tldr,
|
||
**view_field(f),
|
||
})
|
||
|
||
out = {
|
||
'generatedFrom': '07-scripts/docs-manifest.py',
|
||
'counts': {'files': len(items), 'chars': sum(i['chars'] for i in items)},
|
||
'tiers': {t: sum(1 for i in items if i['tier'] == t) for t in ('hot', 'cur', 'warm', 'cold', 'doc')},
|
||
'domains': dict(collections.Counter(i['domain'] for i in items)),
|
||
'layers': dict(collections.Counter(i['layer'] for i in items)),
|
||
'items': items,
|
||
}
|
||
io.open(os.path.join(ROOT, 'docs-manifest.json'), 'w', encoding='utf-8', newline='\n').write(
|
||
json.dumps(out, ensure_ascii=False, indent=1) + '\n')
|
||
|
||
arch = [i for i in items if i['num'] is not None]
|
||
print('文档 %d 份 / %s 字符(≈%s tokens)' % (len(items), format(out['counts']['chars'], ','),
|
||
format(int(out['counts']['chars'] * 0.7), ',')))
|
||
print('档案 %d 份 | 分层:hot %d / warm %d / cold %d | 根级文档 %d'
|
||
% (len(arch), out['tiers']['hot'], out['tiers']['warm'], out['tiers']['cold'], out['tiers']['doc']))
|
||
oversized = sorted([i for i in items if i['chars'] > 30000], key=lambda x: -x['chars'])
|
||
if oversized:
|
||
print('\n【⚠️ 单篇 > 30 KB】%d 篇(约定上限 30 KB;历史只报不改,新档案超限须拆):' % len(oversized))
|
||
for i in oversized:
|
||
print(' %6.1f KB %-58s' % (i['chars'] / 1024, os.path.basename(i['path'])[:56]))
|
||
print('\n【hot】常读(引用 ≥8 次)—— 改东西前大概率要看:')
|
||
for i in sorted([x for x in arch if x['tier'] == 'hot'], key=lambda x: -x['refs']):
|
||
print(' 档案 %-3s %2d 次 %-52s %6d 字符' % (i['num'], i['refs'], os.path.basename(i['path'])[:50], i['chars']))
|
||
print('\n【cold】零引用(历史候选,可只留索引行):')
|
||
for i in sorted([x for x in arch if x['tier'] == 'cold'], key=lambda x: x['num']):
|
||
print(' 档案 %-3s %s' % (i['num'], os.path.basename(i['path'])))
|
||
print('\n→ 已写出 docs-manifest.json')
|