Files
dsh_shenxian/dsh-server-docs/07-scripts/docs-manifest.py
T
admin e6207aa691
build / build-and-scan (push) Waiting to run
chore(仓库对齐): 文档库结构治理 + IM/插件线落地
文档库:目录改为编号制(01-规范/02-架构设计/03-数据库/04-调整方案/
05-交接单/06-ops/07-scripts/08-skills/09-archive),顶层散文件归入 01-规范/;
INDEX.md 与 docs-manifest.json 重刷(档案 146 篇);旧目录名引用全量对齐。

IM 线:src/im/**(SDK / hub / store / presence / ws / gateway-token)、
src/web/routes/im.ts、src/db/plugin-data/**、src/supervisor/plugin-assembly.ts
及对应 test/**。

插件线:poc/{im-agent-bridge,im-connection-gateway,im-conversation-tabs,
business-plugins-im,carbon-mcp-probe}、src/web/routes/{sessions,overlay-device}.ts、
src/net/relay/{device-grant,instance-credential}.ts。

仓库卫生:清出 40 个历史误入库 / 已改名文件(34 个交接单归档 + 6 个旧结构,
本地均有副本);dsh-server-docs/.gitignore 补 tmp/;交接单不入库(政策)。
2026-09-24 07:25:16 +08:00

202 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
docs-manifest.py — 生成文档库**机读清单**(供 AI/脚本过滤,替代"读大表格")
产出:
docs-manifest.json 每份文档:编号/标题/状态/日期/字符/行数/被引用次数/分层(tier)/主题标签
控制台 体量总览 + 分层统计 + 「常读」与「历史」清单
分层规则(2026-09-14 变更:由"**被引几次**"改为"**谁在引**"):
hot 被**现行层 L1/L2**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范)反复引用 ≥3 次 —— 改东西前大概率要看
cur 被现行层引用 1–2 次 —— 相关即看(单次引用也可能极重要,如红线类档案 07)
warm 仅被历史档案互引(全库 ≥1,现行层 0)—— 参考
cold 零引用 —— 历史,只在追溯时看
doc 根级长期文档 / 资产,不参与档案分层
⚠️ 实测教训:**L4(INDEX / 待办 / 台账)不能算现行层** —— 它们会顺带列出几乎所有档案号,
会让判据反向失效(第一版含 L4 时 hot 从 44 抬到 60)。
用法:python3 07-scripts/docs-manifest.py [文档库根目录]
副作用:仅写 docs-manifest.json(其余只读)
"""
import io, os, re, sys, json, collections
def norm_status(raw):
"""规范化档案头部的状态行 → 表里可读的一小段(≤14 字,去 ** 与尾部括注)。
⚠️ **权威方向**:档案头部是**源**,INDEX 表是**派生展示** —— 不允许把表里的 ❓ 回灌成 manifest 的 status,
否则 manifest → INDEX → manifest 形成环形锁定,状态永远修不回来(2026-09-14 实测 14 篇被锁)。
"""
t = re.sub(r'\*{1,2}', '', raw or '').strip()
t = re.split(r'——|||\||;|;|\s{2,}', t)[0].strip()
t = re.sub(r'([^)]*)\s*$', '', t).strip()
return t[:14]
ROOT = sys.argv[1] if len(sys.argv) > 1 else os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
def rd(rel):
try:
return io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
except OSError:
return ''
MD = []
for base, dirs, names in os.walk(ROOT):
if '.git' in dirs:
dirs.remove('.git')
for n in names:
if n.endswith('.md') and '.bak' not in n:
MD.append(os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/'))
MD.sort()
# ── 状态/日期:优先取 INDEX 表,其次取档案头部 ─────────────
index_status = {}
if os.path.exists(os.path.join(ROOT, 'INDEX.md')):
for line in rd('INDEX.md').split('\n'):
m = re.match(r'^\|\s*(?:04-)?(\d{1,3}[a-z]?)\s*\|', line)
if not m:
continue
marks = [k for k in ('✅', '🔄', '🔧', '🧪', '🗄', '⏸', '❌') if k in line]
dm = re.search(r'\b(\d{2}-\d{2})\b', line)
index_status[m.group(1)] = {'status': marks[0] if marks else '?', 'date': dm.group(1) if dm else ''}
# ── 引用热度 ────────────────────────────────────────────
DOMAIN_RULES = {
'external': r'官方推荐|awesome|npm|cordis|开源导出|外部源|生态',
'method': r'方法|决策|技能|skill|流程|工作流|协议|协作',
'plugin': r'插件|univer|mcn|business-plugins|dsh-plugin|原生绑定',
'ui': r'UI|前端|门户|页面|portal|界面|client bundle|渲染',
'06-ops': r'运维|排障|重启|备份|配额|内存|OOM|证书|nginx|nft|执行锁|并发',
'platform': r'平台|部署|隔离|bwrap|多租户|实例|profile|配额口径|架构|命名|重构|升级|耦合|回归清单|目标架构',
'ops2': r'登录|401|404|竞态|冷启动|启动失败|排障|死锁',
'ui2': r'会话提示|文案|提示条|覆盖层|面板|分区|交互',
'method2': r'文档|信息架构|清单|机读|模板|规范|收尾|去重',
}
def domain_of(text):
scores = collections.Counter()
for name, pat in DOMAIN_RULES.items():
scores[name] = len(re.findall(pat, text, re.I))
best = scores.most_common(1)[0]
if best[1] <= 0:
return '?'
return {'ops2': '06-ops', 'ui2': 'ui', 'method2': 'method'}.get(best[0], best[0])
def layer_of(path, num):
if num: return 'L5'
if path.startswith('08-skills/'): return 'L3'
if path.startswith('05-交接单/'): return 'L4'
if path in ('BRIEF.md',): return 'L1'
if path == 'CODEBUDDY.md': return 'L2'
if path == 'INDEX.md': return 'L4'
if path == 'README.md': return 'L2'
if path == 'DEPLOY-本部署.md': return 'L1' # 现行部署事实
if path == '01-规范/06-工作台UI规范.md': return 'L2' # 强制基线 = 规则
if path == '01-规范/03-路线图与待办.md': return 'L4' # 状态/待办
if path in ('BRIEF.md',): return 'L1'
if path.startswith('docs-manifest') or path.endswith('.json'): return 'L2'
if path.startswith('01-规范/01-规划与架构'): return 'L0'
if path.startswith('01-规范/02-运维手册') or path.startswith('09-archive/'): return 'L5'
if path == '04-调整方案/README.md': return 'L5' # 目录内索引 ⇒ 随其层
return '?'
VIEW_DIR = os.path.join(os.path.dirname(ROOT), '.workbuddy', 'cache', 'dedupe-view')
def view_field(rel):
"""库外的"去重视图"(docs-dedupe.py 产物)—— 有则记路径,供检索/阅读优先使用。"""
cand = os.path.join(VIEW_DIR, rel.replace('/', '__'))
return {'dedupeView': cand} if os.path.exists(cand) else {}
def num_of(f):
m = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
return m.group(1) if (m and f.startswith('04-调整方案/') and '/poc/' not in f) else None
# 引用分两份:**全库** 与 **仅 L1+L2 层**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范 = 现行事实与规则)。
# ⚠️ 2026-09-14 实测教训:**L4(INDEX / 待办 / 台账)不能算** —— 它们会顺带列出几乎所有档案号,
# 导致 hot 从 44 抬到 60(判据反而变糟);"被台账提到" ≠ "改东西前要看"。
# 2026-09-14 判据变更:tier 由"被引几次"改为"**谁在引**" —— 只有被现行层引用才算 hot。
ref = collections.Counter()
ref_current = collections.Counter()
for f in MD:
lay = layer_of(f, num_of(f))
for m in re.finditer(r'档案\s*(\d{1,2}[a-z]?)', rd(f)):
ref[m.group(1)] += 1
if lay in ('L1', 'L2'):
ref_current[m.group(1)] += 1
# ── 域标签(检索键;未命中记 '?',由人补规则)─────────────
items = []
for f in MD:
s = rd(f)
title = next((l.strip().lstrip('#').strip() for l in s.split('\n') if l.strip().startswith('#')), '')
mnum = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
num = mnum.group(1) if (mnum and f.startswith('04-调整方案/') and '/poc/' not in f) else None
head = '\n'.join(s.split('\n')[:14])
m_tl = re.search(r'>\s*\*\*TL;DR\*\*[||::]?\s*(.+)', s)
tldr = re.sub(r'\s+', ' ', m_tl.group(1)).strip(' ||')[:120] if m_tl else ''
hd = re.search(r'(- 日期:\s*)([0-9]{4}-[0-9]{2}-[0-9]{2})', head)
hs = re.search(r'状态\*{0,2}\s*[::]\s*(.{1,40})', head)
hs_val = norm_status(hs.group(1)) if hs else ''
st_idx = (index_status.get(num, {}) or {}).get('status') or ''
if st_idx in ('?', '❓', ''):
st_idx = '' # ❓/? 一律视为「缺」—— 防 INDEX↔manifest 环形锁定(2026-09-14)
n_ref = ref.get(num, 0) if num else 0
n_cur = ref_current.get(num, 0) if num else 0
if not num:
tier = 'doc' # 根级长期文档 / 资产:不参与"档案分层"
elif n_cur >= 3:
tier = 'hot' # 被**现行层反复引用** ⇒ 改东西前大概率要看
elif n_cur >= 1:
tier = 'cur' # 被现行层引用 1–2 次 ⇒ 相关即看(单次引用≠不重要,如红线类档案 07)
elif n_ref >= 1:
tier = 'warm' # 仅被历史档案互引 ⇒ 参考
else:
tier = 'cold' # 零引用
items.append({
'path': f,
'num': num,
'title': title[:80],
'status': hs_val or st_idx or '?',
'date': (hd.group(2) if hd else (index_status.get(num, {}) or {}).get('date', '') or ''),
'chars': len(s),
'lines': s.count('\n') + 1,
'refs': n_ref,
'refsCurrent': n_cur,
'tier': tier,
'layer': layer_of(f, num),
'domain': domain_of(title + '\n' + head),
'tldr': tldr,
**view_field(f),
})
out = {
'generatedFrom': '07-scripts/docs-manifest.py',
'counts': {'files': len(items), 'chars': sum(i['chars'] for i in items)},
'tiers': {t: sum(1 for i in items if i['tier'] == t) for t in ('hot', 'cur', 'warm', 'cold', 'doc')},
'domains': dict(collections.Counter(i['domain'] for i in items)),
'layers': dict(collections.Counter(i['layer'] for i in items)),
'items': items,
}
io.open(os.path.join(ROOT, 'docs-manifest.json'), 'w', encoding='utf-8', newline='\n').write(
json.dumps(out, ensure_ascii=False, indent=1) + '\n')
arch = [i for i in items if i['num'] is not None]
print('文档 %d 份 / %s 字符(≈%s tokens)' % (len(items), format(out['counts']['chars'], ','),
format(int(out['counts']['chars'] * 0.7), ',')))
print('档案 %d 份 | 分层:hot %d / warm %d / cold %d | 根级文档 %d'
% (len(arch), out['tiers']['hot'], out['tiers']['warm'], out['tiers']['cold'], out['tiers']['doc']))
oversized = sorted([i for i in items if i['chars'] > 30000], key=lambda x: -x['chars'])
if oversized:
print('\n【⚠️ 单篇 > 30 KB】%d 篇(约定上限 30 KB;历史只报不改,新档案超限须拆):' % len(oversized))
for i in oversized:
print(' %6.1f KB %-58s' % (i['chars'] / 1024, os.path.basename(i['path'])[:56]))
print('\n【hot】常读(引用 ≥8 次)—— 改东西前大概率要看:')
for i in sorted([x for x in arch if x['tier'] == 'hot'], key=lambda x: -x['refs']):
print(' 档案 %-3s %2d 次 %-52s %6d 字符' % (i['num'], i['refs'], os.path.basename(i['path'])[:50], i['chars']))
print('\n【cold】零引用(历史候选,可只留索引行):')
for i in sorted([x for x in arch if x['tier'] == 'cold'], key=lambda x: x['num']):
print(' 档案 %-3s %s' % (i['num'], os.path.basename(i['path'])))
print('\n→ 已写出 docs-manifest.json')