Files
dsh_ai1net_server/dsh-server-docs/scripts/docs-manifest.py
T

201 lines
10 KiB
Python
Raw Normal View History

#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
docs-manifest.py — 生成文档库**机读清单**(供 AI/脚本过滤,替代"读大表格")
产出:
docs-manifest.json 每份文档:编号/标题/状态/日期/字符/行数/被引用次数/分层(tier)/主题标签
控制台 体量总览 + 分层统计 + 「常读」与「历史」清单
分层规则(2026-09-14 变更:由"**被引几次**"改为"**谁在引**"):
hot 被**现行层 L1/L2**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范)反复引用 ≥3 次 —— 改东西前大概率要看
cur 被现行层引用 1–2 次 —— 相关即看(单次引用也可能极重要,如红线类档案 07)
warm 仅被历史档案互引(全库 ≥1,现行层 0)—— 参考
cold 零引用 —— 历史,只在追溯时看
doc 根级长期文档 / 资产,不参与档案分层
⚠️ 实测教训:**L4(INDEX / 待办 / 台账)不能算现行层** —— 它们会顺带列出几乎所有档案号,
会让判据反向失效(第一版含 L4 时 hot 从 44 抬到 60)。
用法:python3 scripts/docs-manifest.py [文档库根目录]
副作用:仅写 docs-manifest.json(其余只读)
"""
import io, os, re, sys, json, collections
def norm_status(raw):
"""规范化档案头部的状态行 → 表里可读的一小段(≤14 字,去 ** 与尾部括注)。
⚠️ **权威方向**:档案头部是**源**,INDEX 表是**派生展示** —— 不允许把表里的 ❓ 回灌成 manifest 的 status,
否则 manifest → INDEX → manifest 形成环形锁定,状态永远修不回来(2026-09-14 实测 14 篇被锁)。
"""
t = re.sub(r'\*{1,2}', '', raw or '').strip()
t = re.split(r'——|||\||;|;|\s{2,}', t)[0].strip()
t = re.sub(r'([^)]*)\s*$', '', t).strip()
return t[:14]
ROOT = sys.argv[1] if len(sys.argv) > 1 else os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
def rd(rel):
try:
return io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
except OSError:
return ''
MD = []
for base, dirs, names in os.walk(ROOT):
if '.git' in dirs:
dirs.remove('.git')
for n in names:
if n.endswith('.md') and '.bak' not in n:
MD.append(os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/'))
MD.sort()
# ── 状态/日期:优先取 INDEX 表,其次取档案头部 ─────────────
index_status = {}
if os.path.exists(os.path.join(ROOT, 'INDEX.md')):
for line in rd('INDEX.md').split('\n'):
m = re.match(r'^\|\s*(?:04-)?(\d{1,3}[a-z]?)\s*\|', line)
if not m:
continue
marks = [k for k in ('✅', '🔄', '🔧', '🧪', '🗄', '⏸', '❌') if k in line]
dm = re.search(r'\b(\d{2}-\d{2})\b', line)
index_status[m.group(1)] = {'status': marks[0] if marks else '?', 'date': dm.group(1) if dm else ''}
# ── 引用热度 ────────────────────────────────────────────
DOMAIN_RULES = {
'external': r'官方推荐|awesome|npm|cordis|开源导出|外部源|生态',
'method': r'方法|决策|技能|skill|流程|工作流|协议|协作',
'plugin': r'插件|univer|mcn|business-plugins|dsh-plugin|原生绑定',
'ui': r'UI|前端|门户|页面|portal|界面|client bundle|渲染',
'ops': r'运维|排障|重启|备份|配额|内存|OOM|证书|nginx|nft|执行锁|并发',
'platform': r'平台|部署|隔离|bwrap|多租户|实例|profile|配额口径|架构|命名|重构|升级|耦合|回归清单|目标架构',
'ops2': r'登录|401|404|竞态|冷启动|启动失败|排障|死锁',
'ui2': r'会话提示|文案|提示条|覆盖层|面板|分区|交互',
'method2': r'文档|信息架构|清单|机读|模板|规范|收尾|去重',
}
def domain_of(text):
scores = collections.Counter()
for name, pat in DOMAIN_RULES.items():
scores[name] = len(re.findall(pat, text, re.I))
best = scores.most_common(1)[0]
if best[1] <= 0:
return '?'
return {'ops2': 'ops', 'ui2': 'ui', 'method2': 'method'}.get(best[0], best[0])
def layer_of(path, num):
if num: return 'L5'
if path.startswith('skills/'): return 'L3'
if path.startswith('交接单/'): return 'L4'
if path in ('BRIEF.md',): return 'L1'
if path == 'CODEBUDDY.md': return 'L2'
if path == 'INDEX.md': return 'L4'
if path == 'README.md': return 'L2'
if path == 'DEPLOY-本部署.md': return 'L1' # 现行部署事实
if path == '06-工作台UI规范.md': return 'L2' # 强制基线 = 规则
if path == '03-路线图与待办.md': return 'L4' # 状态/待办
if path in ('BRIEF.md',): return 'L1'
if path.startswith('docs-manifest') or path.endswith('.json'): return 'L2'
if path.startswith('01-规划与架构'): return 'L0'
if path.startswith('02-运维手册') or path.startswith('archive/'): return 'L5'
if path == '04-调整方案/README.md': return 'L5' # 目录内索引 ⇒ 随其层
return '?'
VIEW_DIR = os.path.join(os.path.dirname(ROOT), '.workbuddy', 'cache', 'dedupe-view')
def view_field(rel):
"""库外的"去重视图"(docs-dedupe.py 产物)—— 有则记路径,供检索/阅读优先使用。"""
cand = os.path.join(VIEW_DIR, rel.replace('/', '__'))
return {'dedupeView': cand} if os.path.exists(cand) else {}
def num_of(f):
m = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
return m.group(1) if (m and f.startswith('04-调整方案/') and '/poc/' not in f) else None
# 引用分两份:**全库** 与 **仅 L1+L2 层**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范 = 现行事实与规则)。
# ⚠️ 2026-09-14 实测教训:**L4(INDEX / 待办 / 台账)不能算** —— 它们会顺带列出几乎所有档案号,
# 导致 hot 从 44 抬到 60(判据反而变糟);"被台账提到" ≠ "改东西前要看"。
# 2026-09-14 判据变更:tier 由"被引几次"改为"**谁在引**" —— 只有被现行层引用才算 hot。
ref = collections.Counter()
ref_current = collections.Counter()
for f in MD:
lay = layer_of(f, num_of(f))
for m in re.finditer(r'档案\s*(\d{1,2}[a-z]?)', rd(f)):
ref[m.group(1)] += 1
if lay in ('L1', 'L2'):
ref_current[m.group(1)] += 1
# ── 域标签(检索键;未命中记 '?',由人补规则)─────────────
items = []
for f in MD:
s = rd(f)
title = next((l.strip().lstrip('#').strip() for l in s.split('\n') if l.strip().startswith('#')), '')
mnum = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
num = mnum.group(1) if (mnum and f.startswith('04-调整方案/') and '/poc/' not in f) else None
head = '\n'.join(s.split('\n')[:14])
m_tl = re.search(r'>\s*\*\*TL;DR\*\*[||::]?\s*(.+)', s)
tldr = re.sub(r'\s+', ' ', m_tl.group(1)).strip(' ||')[:120] if m_tl else ''
hd = re.search(r'(- 日期:\s*)([0-9]{4}-[0-9]{2}-[0-9]{2})', head)
hs = re.search(r'状态\*{0,2}\s*[::]\s*(.{1,40})', head)
hs_val = norm_status(hs.group(1)) if hs else ''
st_idx = (index_status.get(num, {}) or {}).get('status') or ''
if st_idx in ('?', '❓', ''):
st_idx = '' # ❓/? 一律视为「缺」—— 防 INDEX↔manifest 环形锁定(2026-09-14)
n_ref = ref.get(num, 0) if num else 0
n_cur = ref_current.get(num, 0) if num else 0
if not num:
tier = 'doc' # 根级长期文档 / 资产:不参与"档案分层"
elif n_cur >= 3:
tier = 'hot' # 被**现行层反复引用** ⇒ 改东西前大概率要看
elif n_cur >= 1:
tier = 'cur' # 被现行层引用 1–2 次 ⇒ 相关即看(单次引用≠不重要,如红线类档案 07)
elif n_ref >= 1:
tier = 'warm' # 仅被历史档案互引 ⇒ 参考
else:
tier = 'cold' # 零引用
items.append({
'path': f,
'num': num,
'title': title[:80],
'status': hs_val or st_idx or '?',
'date': (hd.group(2) if hd else (index_status.get(num, {}) or {}).get('date', '') or ''),
'chars': len(s),
'lines': s.count('\n') + 1,
'refs': n_ref,
'refsCurrent': n_cur,
'tier': tier,
'layer': layer_of(f, num),
'domain': domain_of(title + '\n' + head),
'tldr': tldr,
**view_field(f),
})
out = {
'generatedFrom': 'scripts/docs-manifest.py',
'counts': {'files': len(items), 'chars': sum(i['chars'] for i in items)},
'tiers': {t: sum(1 for i in items if i['tier'] == t) for t in ('hot', 'cur', 'warm', 'cold', 'doc')},
'domains': dict(collections.Counter(i['domain'] for i in items)),
'layers': dict(collections.Counter(i['layer'] for i in items)),
'items': items,
}
io.open(os.path.join(ROOT, 'docs-manifest.json'), 'w', encoding='utf-8', newline='\n').write(
json.dumps(out, ensure_ascii=False, indent=1) + '\n')
arch = [i for i in items if i['num'] is not None]
print('文档 %d 份 / %s 字符(≈%s tokens)' % (len(items), format(out['counts']['chars'], ','),
format(int(out['counts']['chars'] * 0.7), ',')))
print('档案 %d 份 | 分层:hot %d / warm %d / cold %d | 根级文档 %d'
% (len(arch), out['tiers']['hot'], out['tiers']['warm'], out['tiers']['cold'], out['tiers']['doc']))
oversized = sorted([i for i in items if i['chars'] > 30000], key=lambda x: -x['chars'])
if oversized:
print('\n【⚠️ 单篇 > 30 KB】%d 篇(约定上限 30 KB;历史只报不改,新档案超限须拆):' % len(oversized))
for i in oversized:
print(' %6.1f KB %-58s' % (i['chars'] / 1024, os.path.basename(i['path'])[:56]))
print('\n【hot】常读(引用 ≥8 次)—— 改东西前大概率要看:')
for i in sorted([x for x in arch if x['tier'] == 'hot'], key=lambda x: -x['refs']):
print(' 档案 %-3s %2d 次 %-52s %6d 字符' % (i['num'], i['refs'], os.path.basename(i['path'])[:50], i['chars']))
print('\n【cold】零引用(历史候选,可只留索引行):')
for i in sorted([x for x in arch if x['tier'] == 'cold'], key=lambda x: x['num']):
print(' 档案 %-3s %s' % (i['num'], os.path.basename(i['path'])))
print('\n→ 已写出 docs-manifest.json')