build / build-and-scan (push) Waiting to run
文档库:目录改为编号制(01-规范/02-架构设计/03-数据库/04-调整方案/
05-交接单/06-ops/07-scripts/08-skills/09-archive),顶层散文件归入 01-规范/;
INDEX.md 与 docs-manifest.json 重刷(档案 146 篇);旧目录名引用全量对齐。
IM 线:src/im/**(SDK / hub / store / presence / ws / gateway-token)、
src/web/routes/im.ts、src/db/plugin-data/**、src/supervisor/plugin-assembly.ts
及对应 test/**。
插件线:poc/{im-agent-bridge,im-connection-gateway,im-conversation-tabs,
business-plugins-im,carbon-mcp-probe}、src/web/routes/{sessions,overlay-device}.ts、
src/net/relay/{device-grant,instance-credential}.ts。
仓库卫生:清出 40 个历史误入库 / 已改名文件(34 个交接单归档 + 6 个旧结构,
本地均有副本);dsh-server-docs/.gitignore 补 tmp/;交接单不入库(政策)。
166 lines
7.3 KiB
Python
166 lines
7.3 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""docs-archive-index.py — 让 INDEX.md 的**档案清单表**变成派生件(根治"漏登记")
|
||
|
||
背景(2026-09-14 实测):档案清单**靠手写**,已漏 82–88 共 7 篇;而 `docs-manifest.py`
|
||
已经能机读全部档案(号/状态/tier/域/tldr)。本脚本把两者接上:
|
||
docs-manifest.json ──┐
|
||
archive-summaries.json ─┴─→ INDEX.md §二 的档案表(原地替换,不搬家)
|
||
|
||
设计要点
|
||
1. **不丢手写内容**:首次运行会把现有表里手写的「一句话」**抽取到 `archive-summaries.json`**
|
||
(人工可编辑的映射文件);之后表的摘要优先级 = summaries → manifest.tldr → 标题。
|
||
2. **原地替换**:只替换 `| 04-NN | … |` 那一段连续行,**表头与根级编号行(01/02/03/06)不动**;
|
||
块边界用 BEGIN/END 注释标记,第二次起按标记整块重生成。
|
||
3. **缺失可见**:状态取不到显 `❓`、摘要取不到显 `—` —— 让"没写好头部"的档案**在表里看得见**。
|
||
|
||
用法:
|
||
python3 07-scripts/docs-archive-index.py # 只打印(不写任何文件)
|
||
python3 07-scripts/docs-archive-index.py --write # 写 archive-summaries.json + INDEX.md
|
||
退出码:0 = 一致或已刷新;1 = 有差异且未加 --write;2 = 结构异常
|
||
"""
|
||
import io
|
||
import json
|
||
import os
|
||
import re
|
||
import sys
|
||
|
||
ROOT = sys.argv[1] if len(sys.argv) > 1 and not sys.argv[1].startswith('-') else \
|
||
os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
INDEX = os.path.join(ROOT, 'INDEX.md')
|
||
SUMS = os.path.join(ROOT, 'archive-summaries.json')
|
||
MANIFEST = os.path.join(ROOT, 'docs-manifest.json')
|
||
|
||
B = '<!-- BEGIN archive-index (generated by 07-scripts/docs-archive-index.py — 勿手改) -->'
|
||
E = '<!-- END archive-index -->'
|
||
HEADER = ['| 号 | 状态 | 一句话(**机器生成**;摘要存 `archive-summaries.json`)|', '|---|---|---|']
|
||
ARCH_RE = re.compile(r'^\|\s*04-(\d{1,3}[a-z]?)\s*\|\s*(\S+)\s*\|\s*(.*?)\s*\|\s*$')
|
||
ROW_RE = re.compile(r'^\|\s*04-(\d{1,3}[a-z]?)\s*\|') # 宽松:也认手写表的坏行(缺尾竖线/带 CR)
|
||
|
||
|
||
def rd(path):
|
||
try:
|
||
return io.open(path, encoding='utf-8', newline='').read()
|
||
except OSError:
|
||
return ''
|
||
|
||
|
||
def sort_key(num):
|
||
return (int(re.sub(r'\D', '', num) or 0), num)
|
||
|
||
|
||
def harvest(index_text, existing):
|
||
"""把现有表里手写的「一句话」抽进 summaries(只补空缺,不覆盖已有)。"""
|
||
out = dict(existing)
|
||
for line in index_text.split('\n'):
|
||
m = ARCH_RE.match(line)
|
||
if m and m.group(1) not in out:
|
||
out[m.group(1)] = m.group(3)
|
||
return out
|
||
|
||
|
||
def body(manifest, sums):
|
||
rows = {}
|
||
for i in sorted([x for x in manifest['items'] if x.get('num')], key=lambda x: sort_key(x['num'])):
|
||
num = i['num']
|
||
status = i['status'] if i['status'] not in ('?', '') else '❓'
|
||
text = (sums.get(num) or i.get('tldr') or
|
||
re.sub(r'^\d{1,3}[a-z]?-', '', i.get('title') or '').strip() or '—')
|
||
rows[num] = '| 04-%s | %s | %s |' % (num, status, text[:110])
|
||
return rows
|
||
|
||
|
||
def splice(index_text, gen):
|
||
"""逐行替换(表里 04-* 行与 `—`/根级行是交错的),并按号补插缺失档案。"""
|
||
lines = index_text.split('\n')
|
||
nums = sorted(gen, key=sort_key)
|
||
out, used, dup = [], set(), []
|
||
|
||
def emit_before(limit):
|
||
for n in nums:
|
||
if n not in used and (limit is None or sort_key(n) < sort_key(limit)):
|
||
out.append(gen[n])
|
||
used.add(n)
|
||
|
||
for line in lines:
|
||
m = ROW_RE.match(line)
|
||
if m is None:
|
||
out.append(line)
|
||
continue
|
||
emit_before(m.group(1))
|
||
if m.group(1) in gen:
|
||
if m.group(1) not in used: # 首次出现 → 用生成行
|
||
out.append(gen[m.group(1)])
|
||
used.add(m.group(1))
|
||
else: # 重复的手写行 → 丢弃(自动去重)
|
||
dup.append((m.group(1), line[:60]))
|
||
else:
|
||
out.append(line) # 表里独有的号(如空号)→ 保留
|
||
emit_before(None)
|
||
if dup:
|
||
print(' 自动丢弃重复行 %d 条:%s' % (len(dup), [d[0] for d in dup]))
|
||
for n in nums:
|
||
if n not in used:
|
||
out.append(gen[n])
|
||
return '\n'.join(out)
|
||
|
||
|
||
def main():
|
||
if not os.path.exists(MANIFEST):
|
||
raise SystemExit('ERROR: 先跑 07-scripts/docs-manifest.py 生成 docs-manifest.json')
|
||
|
||
# ── 顺序断言(2026-09-14 加):派生链 manifest → 本脚本,顺序错会**静默**产出新旧混合 ──
|
||
newest, _ad = 0.0, os.path.join(ROOT, '04-调整方案')
|
||
if os.path.isdir(_ad):
|
||
for _n in os.listdir(_ad):
|
||
if _n.endswith('.md'):
|
||
try:
|
||
newest = max(newest, os.path.getmtime(os.path.join(_ad, _n)))
|
||
except OSError:
|
||
pass
|
||
stale_min = (newest - os.path.getmtime(MANIFEST)) / 60.0
|
||
if stale_min > 1:
|
||
msg = ('⚠️ 顺序警告:docs-manifest.json 比 04-调整方案/ 最新档案旧 %.0f 分钟 ⇒ 先跑 '
|
||
'07-scripts/docs-manifest.py,否则本表用的是旧数据' % stale_min)
|
||
if '--write' in sys.argv and '--force' not in sys.argv:
|
||
raise SystemExit(msg + '\n (确认要带旧数据刷新就加 --force)')
|
||
print(msg)
|
||
manifest = json.loads(rd(MANIFEST))
|
||
index_text = rd(INDEX)
|
||
old = {}
|
||
if os.path.exists(SUMS):
|
||
try:
|
||
old = json.loads(rd(SUMS))
|
||
except ValueError:
|
||
raise SystemExit('ERROR: archive-summaries.json 不是合法 JSON')
|
||
sums = harvest(index_text, old)
|
||
rows = body(manifest, sums)
|
||
new_text = splice(index_text, rows)
|
||
changed = new_text != index_text or sums != old
|
||
n_sum = sum(1 for k in rows if sums.get(k))
|
||
_miss = sorted([k for k, v in rows.items() if '❓' in v], key=sort_key)
|
||
print('档案 %d 篇 | 摘要:手写/已存 %d | 机器兜底 %d | 状态缺失(❓) %d(%.0f%%)'
|
||
% (len(rows), n_sum, len(rows) - n_sum, len(_miss),
|
||
100.0 * len(_miss) / max(1, len(rows))))
|
||
if _miss:
|
||
print(' 缺失名单:%s' % ', '.join('04-' + m for m in _miss))
|
||
if len(_miss) / max(1, len(rows)) > 0.10:
|
||
print(' ⚠️ 缺失率 >10%% ⇒ **新档案**头部必须写「- 状态:…」;历史档案按「只增不改」不回改正文,'
|
||
'可在文末「修正(YYYY-MM-DD)」节补一行状态 ⇒ 下一轮由 manifest 从头部取到')
|
||
if '--write' in sys.argv:
|
||
io.open(SUMS, 'w', encoding='utf-8', newline='\n').write(
|
||
json.dumps(dict(sorted(sums.items(), key=lambda kv: sort_key(kv[0]))),
|
||
ensure_ascii=False, indent=1) + '\n')
|
||
if new_text != index_text:
|
||
io.open(INDEX, 'w', encoding='utf-8', newline='').write(new_text)
|
||
print('已刷新 INDEX.md 档案表 + archive-summaries.json')
|
||
else:
|
||
print('INDEX.md 已是最新(仅刷新 archive-summaries.json)')
|
||
return 0
|
||
print('(只读模式)表内容与 INDEX.md %s' % ('一致' if not changed else '不一致,加 --write 刷新'))
|
||
return 1 if changed else 0
|
||
|
||
|
||
if __name__ == '__main__':
|
||
raise SystemExit(main())
|