Files
admin e6207aa691
build / build-and-scan (push) Waiting to run
chore(仓库对齐): 文档库结构治理 + IM/插件线落地
文档库:目录改为编号制(01-规范/02-架构设计/03-数据库/04-调整方案/
05-交接单/06-ops/07-scripts/08-skills/09-archive),顶层散文件归入 01-规范/;
INDEX.md 与 docs-manifest.json 重刷(档案 146 篇);旧目录名引用全量对齐。

IM 线:src/im/**(SDK / hub / store / presence / ws / gateway-token)、
src/web/routes/im.ts、src/db/plugin-data/**、src/supervisor/plugin-assembly.ts
及对应 test/**。

插件线:poc/{im-agent-bridge,im-connection-gateway,im-conversation-tabs,
business-plugins-im,carbon-mcp-probe}、src/web/routes/{sessions,overlay-device}.ts、
src/net/relay/{device-grant,instance-credential}.ts。

仓库卫生:清出 40 个历史误入库 / 已改名文件(34 个交接单归档 + 6 个旧结构,
本地均有副本);dsh-server-docs/.gitignore 补 tmp/;交接单不入库(政策)。
2026-09-24 07:25:16 +08:00

123 lines
5.4 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""docs-dedupe.py — 巨型档案的**重复块体检 + 去重视图生成**(不改原文)
背景(2026-09-14 实测)
* `04-调整方案/82-…md` = **2024 行 / 91 KB**,其中同一份「§一–§八」**重复 16 次**;
* 但本库铁律是 **L5 冻结:历史档案不回改** ⇒ 不能"直接删重复"。
⇒ 本工具遵守铁律:**只读原文 → 生成"去重视图"到库外 + 报告变体差异**;
原文仅允许在**文末追加**「修正(YYYY-MM-DD)」小节(本库明文许可)。
做三件事
1. **块级重复统计**:按 `#` / `##` / `###` 切块,归一化后哈希 ⇒ 报"哪些标题重复几次";
2. **变体检测**:同一标题的多次出现若**内容不同**(hash 不同),逐个列出(避免"以为一样其实有改动");
3. **去重视图**:写入 `--view-out`(默认库外 `.workbuddy/cache/dedupe-view/<名>.md`),
每组保留**信息最全的那一份**(字符数最大,并列取首次),其余位置用一行占位注释替代。
用法
python3 07-scripts/docs-dedupe.py <相对路径> # 只报告
python3 07-scripts/docs-dedupe.py <相对路径> --view-out=路径 # 报告 + 写视图
退出码:0 = 无重复;1 = 有重复(可生成视图);2 = 用法/文件错误
"""
import io
import os
import re
import sys
import hashlib
import collections
DOCS = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
WS = os.path.dirname(DOCS)
HEAD = re.compile(r'^(#{1,3})\s+(.*)$')
def blocks(lines):
"""按 #/##/### 切块;返回 [(起始行, 级别, 标题, 正文行列表)],含块前导内容。"""
out, cur = [], None
pre = []
for i, line in enumerate(lines, 1):
m = HEAD.match(line)
if m:
if cur is None:
cur = [i, len(m.group(1)), m.group(2).strip(), list(pre)]
else:
out.append(cur + [i - 1])
cur = [i, len(m.group(1)), m.group(2).strip(), []]
elif cur is None:
pre.append(line)
else:
cur[3].append(line)
if cur is not None:
out.append(cur + [len(lines)])
return out, pre
def norm(level, title, body):
text = re.sub(r'\s+', ' ', '\n'.join(body)).strip()
return hashlib.sha1(('%d|%s|%s' % (level, title, text)).encode('utf-8')).hexdigest()[:12]
def main(argv):
files = [a for a in argv if not a.startswith('--')]
if not files:
print(__doc__)
return 2
rel = files[0]
path = os.path.join(DOCS, rel)
if not os.path.exists(path):
print('ERROR: 找不到 %s' % path)
return 2
raw = io.open(path, encoding='utf-8', newline='').read()
lines = raw.split('\n')
bs, pre = blocks(lines)
groups = collections.defaultdict(list)
for start, level, title, body, end in bs:
groups[(level, title, norm(level, title, body))].append((start, end, body))
by_title = collections.defaultdict(list)
for (level, title, h), occ in groups.items():
by_title[title].append((h, occ))
dup_titles = {t: v for t, v in by_title.items() if sum(len(o) for _, o in v) > 1}
total_dup_blocks = sum(len(o) - 1 for v in dup_titles.values() for _, o in v)
print('文件 %s:%d 行 / %d 字符|块 %d 个|**重复标题 %d 个,冗余块 %d 个**'
% (rel, len(lines), len(raw), len(bs), len(dup_titles), total_dup_blocks))
for title, variants in sorted(dup_titles.items(),
key=lambda kv: -sum(len(o) for _, o in kv[1]))[:12]:
times = sum(len(o) for _, o in variants)
vinfo = '|'.join('变体%d×%d次' % (i + 1, len(o)) for i, (_, o) in enumerate(variants))
mark = ' ⚠️内容有差异' if len(variants) > 1 else ''
print(' %2d× %-46s %s%s' % (times, title[:44], vinfo, mark))
view_out = next((a.split('=', 1)[1] for a in argv if a.startswith('--view-out=')), None)
if view_out is None:
default = os.path.join(WS, '.workbuddy', 'cache', 'dedupe-view',
rel.replace('/', '__'))
view_out = default if '--view-out' in str(argv) else None
if view_out:
keep = {}
for title, variants in dup_titles.items():
allocc = [(h, start, end, body) for h, o in variants for start, end, body in o]
allocc.sort(key=lambda x: (-sum(len(l) + 1 for l in x[3]), x[1]))
keep[title] = allocc[0]
out, skipped = [], 0
for start, level, title, body, end in bs:
cur = (start, end, body)
keeper = keep.get(title)
if keeper is not None and (start, end, body) != (keeper[1], keeper[2], keeper[3]):
skipped += 1
out.append('%s<!-- 去重省略:同「%s」块(原文 L%d–L%d),内容与 L%d–L%d 那份一致 -->'
% ('#' * level + ' ', title, start, end, keeper[1], keeper[2]))
else:
out.append('\n'.join(['#' * level + ' ' + title] + body))
os.makedirs(os.path.dirname(view_out), exist_ok=True)
io.open(view_out, 'w', encoding='utf-8', newline='\n').write('\n'.join(out) + '\n')
print('\n去重视图已写:%s(省略 %d 块,%.1f KB → %.1f KB)'
% (view_out, skipped, len(raw) / 1024, os.path.getsize(view_out) / 1024))
return 1 if dup_titles else 0
if __name__ == '__main__':
raise SystemExit(main(sys.argv[1:]))