Files
dsh_ai1net_server/dsh-server-docs/scripts/docs-manifest.py
T
admin 5ad755116e chore(docs): 文档库并入代码仓(R4 选 a)+ 索引/台账跟进
1) dsh-server-docs/ 从工作区(原 E:\...\aliyun-dsh-server\dsh-server-docs)**整体并入本仓**,
   保留目录名 ⇒ 仓库内 dsh-server-docs/... 的相对引用天然继续有效;旧目录(含其 .git)已归档到
   工作区 _中间产物_待清理/,未随本提交带入。
2) .gitattributes:新增 `dsh-server-docs/** -text` —— 原文档库是 `* -text` + autocrlf=false,
   必须保持纯 LF,否则会被本仓的 CRLF 规则翻掉。
3) 活引用里的绝对路径已全部改到新位置(docs 的 INDEX / README / scripts / skills + 用户级 skills
   + ~/.workbuddy/settings.json 的 hooks);历史档案(04-调整方案/、archive/)按「只增不改」未动。
   ⚠️ hooks 路径改动需「完全重启会话」才生效(配置是会话启动快照)。
4) 交接单/T08:新增 §16「生产整体切换执行记录」(形态 / 落地动作 / **4 个只有真上线才暴露的真 bug** /
   验收证据 / 回滚命令 / 残留项);台账 T08 行 → 已完成并归档;03-路线图 §二 登记 T08 收尾项。
5) 统一称谓:**「本机」只指跑 WorkBuddy 的开发机**,47 / 106 一律写「远程服务器」。
2026-09-15 18:47:13 +08:00

202 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
docs-manifest.py — 生成文档库**机读清单**(供 AI/脚本过滤,替代"读大表格")
产出:
docs-manifest.json 每份文档:编号/标题/状态/日期/字符/行数/被引用次数/分层(tier)/主题标签
控制台 体量总览 + 分层统计 + 「常读」与「历史」清单
分层规则(2026-09-14 变更:由"**被引几次**"改为"**谁在引**"):
hot 被**现行层 L1/L2**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范)反复引用 ≥3 次 —— 改东西前大概率要看
cur 被现行层引用 1–2 次 —— 相关即看(单次引用也可能极重要,如红线类档案 07)
warm 仅被历史档案互引(全库 ≥1,现行层 0)—— 参考
cold 零引用 —— 历史,只在追溯时看
doc 根级长期文档 / 资产,不参与档案分层
⚠️ 实测教训:**L4(INDEX / 待办 / 台账)不能算现行层** —— 它们会顺带列出几乎所有档案号,
会让判据反向失效(第一版含 L4 时 hot 从 44 抬到 60)。
用法:python3 scripts/docs-manifest.py [文档库根目录]
副作用:仅写 docs-manifest.json(其余只读)
"""
import io, os, re, sys, json, collections
def norm_status(raw):
"""规范化档案头部的状态行 → 表里可读的一小段(≤14 字,去 ** 与尾部括注)。
⚠️ **权威方向**:档案头部是**源**,INDEX 表是**派生展示** —— 不允许把表里的 ❓ 回灌成 manifest 的 status,
否则 manifest → INDEX → manifest 形成环形锁定,状态永远修不回来(2026-09-14 实测 14 篇被锁)。
"""
t = re.sub(r'\*{1,2}', '', raw or '').strip()
t = re.split(r'——|||\||;|;|\s{2,}', t)[0].strip()
t = re.sub(r'([^)]*)\s*$', '', t).strip()
return t[:14]
ROOT = sys.argv[1] if len(sys.argv) > 1 else os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
def rd(rel):
try:
return io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
except OSError:
return ''
MD = []
for base, dirs, names in os.walk(ROOT):
if '.git' in dirs:
dirs.remove('.git')
for n in names:
if n.endswith('.md') and '.bak' not in n:
MD.append(os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/'))
MD.sort()
# ── 状态/日期:优先取 INDEX 表,其次取档案头部 ─────────────
index_status = {}
if os.path.exists(os.path.join(ROOT, 'INDEX.md')):
for line in rd('INDEX.md').split('\n'):
m = re.match(r'^\|\s*(?:04-)?(\d{1,3}[a-z]?)\s*\|', line)
if not m:
continue
marks = [k for k in ('✅', '🔄', '🔧', '🧪', '🗄', '⏸', '❌') if k in line]
dm = re.search(r'\b(\d{2}-\d{2})\b', line)
index_status[m.group(1)] = {'status': marks[0] if marks else '?', 'date': dm.group(1) if dm else ''}
# ── 引用热度 ────────────────────────────────────────────
DOMAIN_RULES = {
'external': r'官方推荐|awesome|npm|cordis|开源导出|外部源|生态',
'method': r'方法|决策|技能|skill|流程|工作流|协议|协作',
'plugin': r'插件|univer|mcn|business-plugins|dsh-plugin|原生绑定',
'ui': r'UI|前端|门户|页面|portal|界面|client bundle|渲染',
'ops': r'运维|排障|重启|备份|配额|内存|OOM|证书|nginx|nft|执行锁|并发',
'platform': r'平台|部署|隔离|bwrap|多租户|实例|profile|配额口径|架构|命名|重构|升级|耦合|回归清单|目标架构',
'ops2': r'登录|401|404|竞态|冷启动|启动失败|排障|死锁',
'ui2': r'会话提示|文案|提示条|覆盖层|面板|分区|交互',
'method2': r'文档|信息架构|清单|机读|模板|规范|收尾|去重',
}
def domain_of(text):
scores = collections.Counter()
for name, pat in DOMAIN_RULES.items():
scores[name] = len(re.findall(pat, text, re.I))
best = scores.most_common(1)[0]
if best[1] <= 0:
return '?'
return {'ops2': 'ops', 'ui2': 'ui', 'method2': 'method'}.get(best[0], best[0])
def layer_of(path, num):
if num: return 'L5'
if path.startswith('skills/'): return 'L3'
if path.startswith('交接单/'): return 'L4'
if path in ('BRIEF.md',): return 'L1'
if path == 'CODEBUDDY.md': return 'L2'
if path == 'INDEX.md': return 'L4'
if path == 'README.md': return 'L2'
if path == 'DEPLOY-本部署.md': return 'L1' # 现行部署事实
if path == '06-工作台UI规范.md': return 'L2' # 强制基线 = 规则
if path == '03-路线图与待办.md': return 'L4' # 状态/待办
if path in ('BRIEF.md',): return 'L1'
if path.startswith('docs-manifest') or path.endswith('.json'): return 'L2'
if path.startswith('01-规划与架构'): return 'L0'
if path.startswith('02-运维手册') or path.startswith('archive/'): return 'L5'
if path == '04-调整方案/README.md': return 'L5' # 目录内索引 ⇒ 随其层
return '?'
VIEW_DIR = os.path.join(os.path.dirname(ROOT), '.workbuddy', 'cache', 'dedupe-view')
def view_field(rel):
"""库外的"去重视图"(docs-dedupe.py 产物)—— 有则记路径,供检索/阅读优先使用。"""
cand = os.path.join(VIEW_DIR, rel.replace('/', '__'))
return {'dedupeView': cand} if os.path.exists(cand) else {}
def num_of(f):
m = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
return m.group(1) if (m and f.startswith('04-调整方案/') and '/poc/' not in f) else None
# 引用分两份:**全库** 与 **仅 L1+L2 层**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范 = 现行事实与规则)。
# ⚠️ 2026-09-14 实测教训:**L4(INDEX / 待办 / 台账)不能算** —— 它们会顺带列出几乎所有档案号,
# 导致 hot 从 44 抬到 60(判据反而变糟);"被台账提到" ≠ "改东西前要看"。
# 2026-09-14 判据变更:tier 由"被引几次"改为"**谁在引**" —— 只有被现行层引用才算 hot。
ref = collections.Counter()
ref_current = collections.Counter()
for f in MD:
lay = layer_of(f, num_of(f))
for m in re.finditer(r'档案\s*(\d{1,2}[a-z]?)', rd(f)):
ref[m.group(1)] += 1
if lay in ('L1', 'L2'):
ref_current[m.group(1)] += 1
# ── 域标签(检索键;未命中记 '?',由人补规则)─────────────
items = []
for f in MD:
s = rd(f)
title = next((l.strip().lstrip('#').strip() for l in s.split('\n') if l.strip().startswith('#')), '')
mnum = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
num = mnum.group(1) if (mnum and f.startswith('04-调整方案/') and '/poc/' not in f) else None
head = '\n'.join(s.split('\n')[:14])
m_tl = re.search(r'>\s*\*\*TL;DR\*\*[||::]?\s*(.+)', s)
tldr = re.sub(r'\s+', ' ', m_tl.group(1)).strip(' ||')[:120] if m_tl else ''
hd = re.search(r'(- 日期:\s*)([0-9]{4}-[0-9]{2}-[0-9]{2})', head)
hs = re.search(r'状态\*{0,2}\s*[::]\s*(.{1,40})', head)
hs_val = norm_status(hs.group(1)) if hs else ''
st_idx = (index_status.get(num, {}) or {}).get('status') or ''
if st_idx in ('?', '❓', ''):
st_idx = '' # ❓/? 一律视为「缺」—— 防 INDEX↔manifest 环形锁定(2026-09-14)
n_ref = ref.get(num, 0) if num else 0
n_cur = ref_current.get(num, 0) if num else 0
if not num:
tier = 'doc' # 根级长期文档 / 资产:不参与"档案分层"
elif n_cur >= 3:
tier = 'hot' # 被**现行层反复引用** ⇒ 改东西前大概率要看
elif n_cur >= 1:
tier = 'cur' # 被现行层引用 1–2 次 ⇒ 相关即看(单次引用≠不重要,如红线类档案 07)
elif n_ref >= 1:
tier = 'warm' # 仅被历史档案互引 ⇒ 参考
else:
tier = 'cold' # 零引用
items.append({
'path': f,
'num': num,
'title': title[:80],
'status': hs_val or st_idx or '?',
'date': (hd.group(2) if hd else (index_status.get(num, {}) or {}).get('date', '') or ''),
'chars': len(s),
'lines': s.count('\n') + 1,
'refs': n_ref,
'refsCurrent': n_cur,
'tier': tier,
'layer': layer_of(f, num),
'domain': domain_of(title + '\n' + head),
'tldr': tldr,
**view_field(f),
})
out = {
'generatedFrom': 'scripts/docs-manifest.py',
'counts': {'files': len(items), 'chars': sum(i['chars'] for i in items)},
'tiers': {t: sum(1 for i in items if i['tier'] == t) for t in ('hot', 'cur', 'warm', 'cold', 'doc')},
'domains': dict(collections.Counter(i['domain'] for i in items)),
'layers': dict(collections.Counter(i['layer'] for i in items)),
'items': items,
}
io.open(os.path.join(ROOT, 'docs-manifest.json'), 'w', encoding='utf-8', newline='\n').write(
json.dumps(out, ensure_ascii=False, indent=1) + '\n')
arch = [i for i in items if i['num'] is not None]
print('文档 %d 份 / %s 字符(≈%s tokens)' % (len(items), format(out['counts']['chars'], ','),
format(int(out['counts']['chars'] * 0.7), ',')))
print('档案 %d 份 | 分层:hot %d / warm %d / cold %d | 根级文档 %d'
% (len(arch), out['tiers']['hot'], out['tiers']['warm'], out['tiers']['cold'], out['tiers']['doc']))
oversized = sorted([i for i in items if i['chars'] > 30000], key=lambda x: -x['chars'])
if oversized:
print('\n【⚠️ 单篇 > 30 KB】%d 篇(约定上限 30 KB;历史只报不改,新档案超限须拆):' % len(oversized))
for i in oversized:
print(' %6.1f KB %-58s' % (i['chars'] / 1024, os.path.basename(i['path'])[:56]))
print('\n【hot】常读(引用 ≥8 次)—— 改东西前大概率要看:')
for i in sorted([x for x in arch if x['tier'] == 'hot'], key=lambda x: -x['refs']):
print(' 档案 %-3s %2d 次 %-52s %6d 字符' % (i['num'], i['refs'], os.path.basename(i['path'])[:50], i['chars']))
print('\n【cold】零引用(历史候选,可只留索引行):')
for i in sorted([x for x in arch if x['tier'] == 'cold'], key=lambda x: x['num']):
print(' 档案 %-3s %s' % (i['num'], os.path.basename(i['path'])))
print('\n→ 已写出 docs-manifest.json')