chore(docs): 文档库并入代码仓(R4 选 a)+ 索引/台账跟进
1) dsh-server-docs/ 从工作区(原 E:\...\aliyun-dsh-server\dsh-server-docs)**整体并入本仓**,
保留目录名 ⇒ 仓库内 dsh-server-docs/... 的相对引用天然继续有效;旧目录(含其 .git)已归档到
工作区 _中间产物_待清理/,未随本提交带入。
2) .gitattributes:新增 `dsh-server-docs/** -text` —— 原文档库是 `* -text` + autocrlf=false,
必须保持纯 LF,否则会被本仓的 CRLF 规则翻掉。
3) 活引用里的绝对路径已全部改到新位置(docs 的 INDEX / README / scripts / skills + 用户级 skills
+ ~/.workbuddy/settings.json 的 hooks);历史档案(04-调整方案/、archive/)按「只增不改」未动。
⚠️ hooks 路径改动需「完全重启会话」才生效(配置是会话启动快照)。
4) 交接单/T08:新增 §16「生产整体切换执行记录」(形态 / 落地动作 / **4 个只有真上线才暴露的真 bug** /
验收证据 / 回滚命令 / 残留项);台账 T08 行 → 已完成并归档;03-路线图 §二 登记 T08 收尾项。
5) 统一称谓:**「本机」只指跑 WorkBuddy 的开发机**,47 / 106 一律写「远程服务器」。
This commit is contained in:
1 parent
c70d5d860e
commit
5ad755116e
173 files changed
+27632
No files matched your search
@@ -0,0 +1,90 @@
|
||||
/**
|
||||
* 会话时序分析器(只输出时间戳/事件类型/时延,不输出任何消息正文)
|
||||
* 用法:node analyze-session.mjs <session.jsonl.zstd>
|
||||
*/
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { zstdDecompressSync } from 'node:zlib'
|
||||
|
||||
const file = process.argv[2]
|
||||
const raw = readFileSync(file)
|
||||
|
||||
// 可能是多帧拼接:按 zstd magic 切分逐帧解压
|
||||
const MAGIC = Buffer.from([0x28, 0xb5, 0x2f, 0xfd])
|
||||
const offsets = []
|
||||
for (let i = 0; i + 4 <= raw.length; i++) {
|
||||
if (raw.compare(MAGIC, 0, 4, i, i + 4) === 0) offsets.push(i)
|
||||
}
|
||||
let text = ''
|
||||
const frames = offsets.length > 0 ? offsets : [0]
|
||||
for (let f = 0; f < frames.length; f++) {
|
||||
const start = frames[f]
|
||||
const end = f + 1 < frames.length ? frames[f + 1] : raw.length
|
||||
try {
|
||||
text += zstdDecompressSync(raw.subarray(start, end)).toString('utf8')
|
||||
} catch (e) {
|
||||
// 单帧整体解压兜底
|
||||
}
|
||||
}
|
||||
if (text === '') {
|
||||
try {
|
||||
text = zstdDecompressSync(raw).toString('utf8')
|
||||
} catch (e) {
|
||||
console.log('decompress failed:', e.message)
|
||||
process.exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
const lines = text.split('\n').filter((l) => l.trim() !== '')
|
||||
console.log(`帧数=${frames.length} 行数=${lines.length} 解压字节=${text.length}`)
|
||||
|
||||
const pick = (obj, names, depth = 0) => {
|
||||
if (obj === null || typeof obj !== 'object' || depth > 3) return undefined
|
||||
for (const n of names) if (typeof obj[n] !== 'undefined') return obj[n]
|
||||
for (const v of Object.values(obj)) {
|
||||
const hit = pick(v, names, depth + 1)
|
||||
if (hit !== undefined) return hit
|
||||
}
|
||||
return undefined
|
||||
}
|
||||
const TS_KEYS = ['ts', 'time', 'timestamp', 'at', 'createdAt', 'created_at', 'mtime', 'date']
|
||||
const KIND_KEYS = ['kind', 'type', 'event', 'role', 'name', 'tag']
|
||||
|
||||
const events = []
|
||||
const kinds = new Map()
|
||||
for (const line of lines) {
|
||||
let o
|
||||
try {
|
||||
o = JSON.parse(line)
|
||||
} catch {
|
||||
continue
|
||||
}
|
||||
const ts = pick(o, TS_KEYS)
|
||||
const kind = pick(o, KIND_KEYS)
|
||||
const k = typeof kind === 'string' ? kind : JSON.stringify(kind ?? '?')
|
||||
kinds.set(k, (kinds.get(k) ?? 0) + 1)
|
||||
const t = typeof ts === 'number' ? (ts < 1e12 ? ts * 1000 : ts) : typeof ts === 'string' ? Date.parse(ts) : NaN
|
||||
events.push({ t: Number.isNaN(t) ? undefined : t, k })
|
||||
}
|
||||
|
||||
console.log('\n== 事件类型分布(TOP 20)==')
|
||||
;[...kinds.entries()].sort((a, b) => b[1] - a[1]).slice(0, 20).forEach(([k, c]) => console.log(` ${c}\t${k}`))
|
||||
|
||||
const withTs = events.filter((e) => e.t !== undefined)
|
||||
console.log(`\n带时间戳事件: ${withTs.length}/${events.length}`)
|
||||
if (withTs.length > 1) {
|
||||
const span = (withTs[withTs.length - 1].t - withTs[0].t) / 1000
|
||||
console.log(`时间跨度: ${span.toFixed(1)}s(约 ${(span / 60).toFixed(1)} 分钟)`)
|
||||
|
||||
const gaps = []
|
||||
for (let i = 1; i < withTs.length; i++) gaps.push({ d: (withTs[i].t - withTs[i - 1].t) / 1000, a: withTs[i - 1].k, b: withTs[i].k })
|
||||
|
||||
console.log('\n== 最大间隔 TOP 12(秒 | 前事件 → 后事件)==')
|
||||
gaps.sort((x, y) => y.d - x.d).slice(0, 12).forEach((g) => console.log(` ${g.d.toFixed(1)}\t${g.a} → ${g.b}`))
|
||||
|
||||
const over5 = gaps.filter((g) => g.d > 5)
|
||||
console.log(`\n>5s 的空档数: ${over5.length};>30s: ${gaps.filter((g) => g.d > 30).length};>60s: ${gaps.filter((g) => g.d > 60).length}`)
|
||||
|
||||
// 会话首尾时间(便于与外部日志对齐)
|
||||
const fmt = (ms) => new Date(ms).toISOString().replace('T', ' ').slice(0, 19)
|
||||
console.log(`\n首事件: ${fmt(withTs[0].t)} 末事件: ${fmt(withTs[withTs.length - 1].t)}`)
|
||||
}
|
||||
@@ -0,0 +1,71 @@
|
||||
/**
|
||||
* 会话每轮指标:TTFT(turn/start → 首个流式 chunk)、轮时长、工具耗时
|
||||
* 只输出时间与类型,不输出任何正文。
|
||||
* 用法:node analyze-turn.mjs <session.jsonl.zstd>
|
||||
*/
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { zstdDecompressSync } from 'node:zlib'
|
||||
|
||||
const raw = readFileSync(process.argv[2])
|
||||
const MAGIC = Buffer.from([0x28, 0xb5, 0x2f, 0xfd])
|
||||
const offs = []
|
||||
for (let i = 0; i + 4 <= raw.length; i++) if (raw.compare(MAGIC, 0, 4, i, i + 4) === 0) offs.push(i)
|
||||
let text = ''
|
||||
const frames = offs.length > 0 ? offs : [0]
|
||||
for (let f = 0; f < frames.length; f++) {
|
||||
const s = frames[f], e = f + 1 < frames.length ? frames[f + 1] : raw.length
|
||||
try { text += zstdDecompressSync(raw.subarray(s, e)).toString('utf8') } catch {}
|
||||
}
|
||||
if (text === '') { try { text = zstdDecompressSync(raw).toString('utf8') } catch (e) { console.log('fail', e.message); process.exit(1) } }
|
||||
|
||||
const pick = (o, keys, d = 0) => {
|
||||
if (o === null || typeof o !== 'object' || d > 3) return undefined
|
||||
for (const k of keys) if (typeof o[k] !== 'undefined') return o[k]
|
||||
for (const v of Object.values(o)) { const h = pick(v, keys, d + 1); if (h !== undefined) return h }
|
||||
return undefined
|
||||
}
|
||||
const ev = []
|
||||
for (const line of text.split('\n')) {
|
||||
if (line.trim() === '') continue
|
||||
let o; try { o = JSON.parse(line) } catch { continue }
|
||||
const kind = pick(o, ['kind', 'type', 'event'])
|
||||
const ts = pick(o, ['ts', 'time', 'timestamp', 'at', 'createdAt'])
|
||||
const t = typeof ts === 'number' ? (ts < 1e12 ? ts * 1000 : ts) : typeof ts === 'string' ? Date.parse(ts) : NaN
|
||||
if (typeof kind === 'string') ev.push({ k: kind, t: Number.isNaN(t) ? undefined : t })
|
||||
}
|
||||
|
||||
const fmt = (ms) => new Date(ms).toISOString().slice(11, 19) + 'Z'
|
||||
let turns = 0
|
||||
for (let i = 0; i < ev.length; i++) {
|
||||
if (ev[i].k !== 'turn/start') continue
|
||||
turns++
|
||||
const t0 = ev[i].t
|
||||
let end, firstChunk, firstText
|
||||
for (let j = i + 1; j < ev.length; j++) {
|
||||
if (ev[j].k === 'turn/end' && end === undefined) { end = ev[j].t; break }
|
||||
if (firstChunk === undefined && (ev[j].k === 'assistant/chunk' || ev[j].k === 'reasoning-chunks') && ev[j].t !== undefined) firstChunk = ev[j].t
|
||||
if (firstText === undefined && ev[j].k === 'text-chunks' && ev[j].t !== undefined) firstText = ev[j].t
|
||||
}
|
||||
const ttft = t0 !== undefined && firstChunk !== undefined ? ((firstChunk - t0) / 1000).toFixed(1) : '?'
|
||||
const ttftText = t0 !== undefined && firstText !== undefined ? ((firstText - t0) / 1000).toFixed(1) : '?'
|
||||
const dur = t0 !== undefined && end !== undefined ? ((end - t0) / 1000).toFixed(1) : '?'
|
||||
console.log(`轮 ${String(turns).padStart(2)} ${t0 !== undefined ? fmt(t0) : '?'} TTFT=${ttft}s 首文本=${ttftText}s 轮时长=${dur}s`)
|
||||
}
|
||||
|
||||
// 工具耗时
|
||||
const tools = ev.map((e, i) => ({ e, i })).filter((x) => x.e.k === 'tool/call')
|
||||
const dt = []
|
||||
for (const { i } of tools) {
|
||||
for (let j = i + 1; j < ev.length; j++) {
|
||||
if (ev[j].k === 'tool/result') {
|
||||
if (ev[i].t !== undefined && ev[j].t !== undefined) dt.push((ev[j].t - ev[i].t) / 1000)
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
if (dt.length > 0) {
|
||||
dt.sort((a, b) => b - a)
|
||||
const avg = dt.reduce((a, b) => a + b, 0) / dt.length
|
||||
console.log(`\n工具调用 ${dt.length} 次;平均 ${avg.toFixed(1)}s;最慢 5 个: ${dt.slice(0, 5).map((x) => x.toFixed(1) + 's').join(', ')}`)
|
||||
}
|
||||
console.log(`\n合计轮数: ${turns}`)
|
||||
@@ -0,0 +1,61 @@
|
||||
#!/usr/bin/env python3
|
||||
"""列出 Cloudflare 账号可见的 zone(可选某 zone 的 DNS 记录)——只读。
|
||||
用法: python3 cf-dns.py [zone名]
|
||||
凭据: /etc/cloudflare.ini 的 dns_cloudflare_api_token(不打印 token 本身)
|
||||
"""
|
||||
import json
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import urllib.request
|
||||
|
||||
CRED = '/etc/cloudflare.ini'
|
||||
API = 'https://api.cloudflare.com/client/v4'
|
||||
|
||||
|
||||
def token():
|
||||
txt = open(CRED, 'r', encoding='utf-8').read()
|
||||
m = re.search(r'dns_cloudflare_api_token\s*=\s*([A-Za-z0-9_\-]+)', txt)
|
||||
if not m:
|
||||
sys.exit('未在 %s 找到 dns_cloudflare_api_token' % CRED)
|
||||
return m.group(1)
|
||||
|
||||
|
||||
def get(path, tok):
|
||||
req = urllib.request.Request(API + path, headers={'Authorization': 'Bearer ' + tok})
|
||||
with urllib.request.urlopen(req, timeout=20) as resp:
|
||||
return json.loads(resp.read().decode('utf-8'))
|
||||
|
||||
|
||||
def short(v, n=44):
|
||||
v = str(v)
|
||||
return v if len(v) <= n else v[: n - 3] + '...'
|
||||
|
||||
|
||||
def main():
|
||||
tok = token()
|
||||
zones = (get('/zones?per_page=100', tok).get('result') or [])
|
||||
print('可见 zone (%d):' % len(zones))
|
||||
for z in zones:
|
||||
print(' - %s status=%s id=%s...' % (z['name'], z['status'], z['id'][:8]))
|
||||
|
||||
if len(sys.argv) < 2:
|
||||
return
|
||||
want = sys.argv[1]
|
||||
zid = ''
|
||||
for z in zones:
|
||||
if z['name'] == want:
|
||||
zid = z['id']
|
||||
if zid == '':
|
||||
print('\nzone %s 不在该 token 权限内(无法管理其 DNS/证书)' % want)
|
||||
return
|
||||
|
||||
recs = (get('/zones/%s/dns_records?per_page=100' % zid, tok).get('result') or [])
|
||||
print('\n=== %s 的 DNS 记录(%d 条)===' % (want, len(recs)))
|
||||
for r in sorted(recs, key=lambda x: (x['type'], x['name'])):
|
||||
prox = 'on' if r.get('proxied') else 'off'
|
||||
print(' %-6s %-34s -> %-46s proxied=%s' % (r['type'], r['name'], short(r['content']), prox))
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,86 @@
|
||||
#!/usr/bin/env python3
|
||||
"""探测 _acme-challenge 名称是否被通配 CNAME 劫持(只读+临时记录,用完即删)。
|
||||
用法: python3 cf-probe.py <zone名> [_acme-challenge.<zone>]
|
||||
"""
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
import urllib.request
|
||||
|
||||
CRED = '/etc/cloudflare.ini'
|
||||
API = 'https://api.cloudflare.com/client/v4'
|
||||
|
||||
|
||||
def token():
|
||||
txt = open(CRED, 'r', encoding='utf-8').read()
|
||||
m = re.search(r'dns_cloudflare_api_token\s*=\s*([A-Za-z0-9_\-]+)', txt)
|
||||
if not m:
|
||||
sys.exit('未找到 token')
|
||||
return m.group(1)
|
||||
|
||||
|
||||
def req(method, path, tok, body=None):
|
||||
data = json.dumps(body).encode('utf-8') if body is not None else None
|
||||
r = urllib.request.Request(API + path, data=data, method=method,
|
||||
headers={'Authorization': 'Bearer ' + tok,
|
||||
'Content-Type': 'application/json'})
|
||||
with urllib.request.urlopen(r, timeout=20) as resp:
|
||||
return json.loads(resp.read().decode('utf-8'))
|
||||
|
||||
|
||||
def doh(name, typ, server='https://dns.google/resolve'):
|
||||
url = '%s?name=%s&type=%s' % (server, name, typ)
|
||||
r = urllib.request.Request(url, headers={'accept': 'application/dns-json'})
|
||||
with urllib.request.urlopen(r, timeout=20) as resp:
|
||||
return json.loads(resp.read().decode('utf-8'))
|
||||
|
||||
|
||||
def show(title, name, typ):
|
||||
print('--- %s ---' % title)
|
||||
for srv, label in [('https://dns.google/resolve', 'Google'), ('https://cloudflare-dns.com/dns-query', 'CF')]:
|
||||
try:
|
||||
d = doh(name, typ, srv)
|
||||
ans = d.get('Answer') or []
|
||||
print(' [%s] Status=%s' % (label, d.get('Status')))
|
||||
if not ans:
|
||||
print(' (无 Answer)')
|
||||
for a in ans:
|
||||
print(' type=%-6s %s' % (a.get('type'), str(a.get('data'))[:90]))
|
||||
except Exception as e:
|
||||
print(' [%s] 查询失败: %s' % (label, e))
|
||||
|
||||
|
||||
def main():
|
||||
zone = sys.argv[1] if len(sys.argv) > 1 else 'alotbuy.com'
|
||||
name = sys.argv[2] if len(sys.argv) > 2 else '_acme-challenge.' + zone
|
||||
tok = token()
|
||||
|
||||
zid = ''
|
||||
for z in (req('GET', '/zones?name=' + zone, tok).get('result') or []):
|
||||
zid = z['id']
|
||||
if zid == '':
|
||||
sys.exit('zone 不在权限内')
|
||||
|
||||
print('=== 创建临时 TXT: %s ===' % name)
|
||||
created = req('POST', '/zones/%s/dns_records' % zid, tok,
|
||||
{'type': 'TXT', 'name': name, 'content': 'probe-test-value-12345', 'ttl': 120})
|
||||
if not created.get('success'):
|
||||
print(' 创建失败:', created.get('errors'))
|
||||
return
|
||||
rid = created['result']['id']
|
||||
print(' 已创建 id=%s' % rid)
|
||||
|
||||
try:
|
||||
for wait in (3, 10, 30):
|
||||
time.sleep(wait if wait == 3 else wait - 3)
|
||||
print('\n=== 等待累计 ~%ds 后查询 ===' % wait)
|
||||
show('TXT 查询', name, 'TXT')
|
||||
finally:
|
||||
print('\n=== 清理临时记录 ===')
|
||||
d = req('DELETE', '/zones/%s/dns_records/%s' % (zid, rid), tok)
|
||||
print(' 删除成功:', d.get('success'))
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,165 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""docs-archive-index.py — 让 INDEX.md 的**档案清单表**变成派生件(根治"漏登记")
|
||||
|
||||
背景(2026-09-14 实测):档案清单**靠手写**,已漏 82–88 共 7 篇;而 `docs-manifest.py`
|
||||
已经能机读全部档案(号/状态/tier/域/tldr)。本脚本把两者接上:
|
||||
docs-manifest.json ──┐
|
||||
archive-summaries.json ─┴─→ INDEX.md §二 的档案表(原地替换,不搬家)
|
||||
|
||||
设计要点
|
||||
1. **不丢手写内容**:首次运行会把现有表里手写的「一句话」**抽取到 `archive-summaries.json`**
|
||||
(人工可编辑的映射文件);之后表的摘要优先级 = summaries → manifest.tldr → 标题。
|
||||
2. **原地替换**:只替换 `| 04-NN | … |` 那一段连续行,**表头与根级编号行(01/02/03/06)不动**;
|
||||
块边界用 BEGIN/END 注释标记,第二次起按标记整块重生成。
|
||||
3. **缺失可见**:状态取不到显 `❓`、摘要取不到显 `—` —— 让"没写好头部"的档案**在表里看得见**。
|
||||
|
||||
用法:
|
||||
python3 scripts/docs-archive-index.py # 只打印(不写任何文件)
|
||||
python3 scripts/docs-archive-index.py --write # 写 archive-summaries.json + INDEX.md
|
||||
退出码:0 = 一致或已刷新;1 = 有差异且未加 --write;2 = 结构异常
|
||||
"""
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
ROOT = sys.argv[1] if len(sys.argv) > 1 and not sys.argv[1].startswith('-') else \
|
||||
os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
INDEX = os.path.join(ROOT, 'INDEX.md')
|
||||
SUMS = os.path.join(ROOT, 'archive-summaries.json')
|
||||
MANIFEST = os.path.join(ROOT, 'docs-manifest.json')
|
||||
|
||||
B = '<!-- BEGIN archive-index (generated by scripts/docs-archive-index.py — 勿手改) -->'
|
||||
E = '<!-- END archive-index -->'
|
||||
HEADER = ['| 号 | 状态 | 一句话(**机器生成**;摘要存 `archive-summaries.json`)|', '|---|---|---|']
|
||||
ARCH_RE = re.compile(r'^\|\s*04-(\d{1,3}[a-z]?)\s*\|\s*(\S+)\s*\|\s*(.*?)\s*\|\s*$')
|
||||
ROW_RE = re.compile(r'^\|\s*04-(\d{1,3}[a-z]?)\s*\|') # 宽松:也认手写表的坏行(缺尾竖线/带 CR)
|
||||
|
||||
|
||||
def rd(path):
|
||||
try:
|
||||
return io.open(path, encoding='utf-8', newline='').read()
|
||||
except OSError:
|
||||
return ''
|
||||
|
||||
|
||||
def sort_key(num):
|
||||
return (int(re.sub(r'\D', '', num) or 0), num)
|
||||
|
||||
|
||||
def harvest(index_text, existing):
|
||||
"""把现有表里手写的「一句话」抽进 summaries(只补空缺,不覆盖已有)。"""
|
||||
out = dict(existing)
|
||||
for line in index_text.split('\n'):
|
||||
m = ARCH_RE.match(line)
|
||||
if m and m.group(1) not in out:
|
||||
out[m.group(1)] = m.group(3)
|
||||
return out
|
||||
|
||||
|
||||
def body(manifest, sums):
|
||||
rows = {}
|
||||
for i in sorted([x for x in manifest['items'] if x.get('num')], key=lambda x: sort_key(x['num'])):
|
||||
num = i['num']
|
||||
status = i['status'] if i['status'] not in ('?', '') else '❓'
|
||||
text = (sums.get(num) or i.get('tldr') or
|
||||
re.sub(r'^\d{1,3}[a-z]?-', '', i.get('title') or '').strip() or '—')
|
||||
rows[num] = '| 04-%s | %s | %s |' % (num, status, text[:110])
|
||||
return rows
|
||||
|
||||
|
||||
def splice(index_text, gen):
|
||||
"""逐行替换(表里 04-* 行与 `—`/根级行是交错的),并按号补插缺失档案。"""
|
||||
lines = index_text.split('\n')
|
||||
nums = sorted(gen, key=sort_key)
|
||||
out, used, dup = [], set(), []
|
||||
|
||||
def emit_before(limit):
|
||||
for n in nums:
|
||||
if n not in used and (limit is None or sort_key(n) < sort_key(limit)):
|
||||
out.append(gen[n])
|
||||
used.add(n)
|
||||
|
||||
for line in lines:
|
||||
m = ROW_RE.match(line)
|
||||
if m is None:
|
||||
out.append(line)
|
||||
continue
|
||||
emit_before(m.group(1))
|
||||
if m.group(1) in gen:
|
||||
if m.group(1) not in used: # 首次出现 → 用生成行
|
||||
out.append(gen[m.group(1)])
|
||||
used.add(m.group(1))
|
||||
else: # 重复的手写行 → 丢弃(自动去重)
|
||||
dup.append((m.group(1), line[:60]))
|
||||
else:
|
||||
out.append(line) # 表里独有的号(如空号)→ 保留
|
||||
emit_before(None)
|
||||
if dup:
|
||||
print(' 自动丢弃重复行 %d 条:%s' % (len(dup), [d[0] for d in dup]))
|
||||
for n in nums:
|
||||
if n not in used:
|
||||
out.append(gen[n])
|
||||
return '\n'.join(out)
|
||||
|
||||
|
||||
def main():
|
||||
if not os.path.exists(MANIFEST):
|
||||
raise SystemExit('ERROR: 先跑 scripts/docs-manifest.py 生成 docs-manifest.json')
|
||||
|
||||
# ── 顺序断言(2026-09-14 加):派生链 manifest → 本脚本,顺序错会**静默**产出新旧混合 ──
|
||||
newest, _ad = 0.0, os.path.join(ROOT, '04-调整方案')
|
||||
if os.path.isdir(_ad):
|
||||
for _n in os.listdir(_ad):
|
||||
if _n.endswith('.md'):
|
||||
try:
|
||||
newest = max(newest, os.path.getmtime(os.path.join(_ad, _n)))
|
||||
except OSError:
|
||||
pass
|
||||
stale_min = (newest - os.path.getmtime(MANIFEST)) / 60.0
|
||||
if stale_min > 1:
|
||||
msg = ('⚠️ 顺序警告:docs-manifest.json 比 04-调整方案/ 最新档案旧 %.0f 分钟 ⇒ 先跑 '
|
||||
'scripts/docs-manifest.py,否则本表用的是旧数据' % stale_min)
|
||||
if '--write' in sys.argv and '--force' not in sys.argv:
|
||||
raise SystemExit(msg + '\n (确认要带旧数据刷新就加 --force)')
|
||||
print(msg)
|
||||
manifest = json.loads(rd(MANIFEST))
|
||||
index_text = rd(INDEX)
|
||||
old = {}
|
||||
if os.path.exists(SUMS):
|
||||
try:
|
||||
old = json.loads(rd(SUMS))
|
||||
except ValueError:
|
||||
raise SystemExit('ERROR: archive-summaries.json 不是合法 JSON')
|
||||
sums = harvest(index_text, old)
|
||||
rows = body(manifest, sums)
|
||||
new_text = splice(index_text, rows)
|
||||
changed = new_text != index_text or sums != old
|
||||
n_sum = sum(1 for k in rows if sums.get(k))
|
||||
_miss = sorted([k for k, v in rows.items() if '❓' in v], key=sort_key)
|
||||
print('档案 %d 篇 | 摘要:手写/已存 %d | 机器兜底 %d | 状态缺失(❓) %d(%.0f%%)'
|
||||
% (len(rows), n_sum, len(rows) - n_sum, len(_miss),
|
||||
100.0 * len(_miss) / max(1, len(rows))))
|
||||
if _miss:
|
||||
print(' 缺失名单:%s' % ', '.join('04-' + m for m in _miss))
|
||||
if len(_miss) / max(1, len(rows)) > 0.10:
|
||||
print(' ⚠️ 缺失率 >10%% ⇒ **新档案**头部必须写「- 状态:…」;历史档案按「只增不改」不回改正文,'
|
||||
'可在文末「修正(YYYY-MM-DD)」节补一行状态 ⇒ 下一轮由 manifest 从头部取到')
|
||||
if '--write' in sys.argv:
|
||||
io.open(SUMS, 'w', encoding='utf-8', newline='\n').write(
|
||||
json.dumps(dict(sorted(sums.items(), key=lambda kv: sort_key(kv[0]))),
|
||||
ensure_ascii=False, indent=1) + '\n')
|
||||
if new_text != index_text:
|
||||
io.open(INDEX, 'w', encoding='utf-8', newline='').write(new_text)
|
||||
print('已刷新 INDEX.md 档案表 + archive-summaries.json')
|
||||
else:
|
||||
print('INDEX.md 已是最新(仅刷新 archive-summaries.json)')
|
||||
return 0
|
||||
print('(只读模式)表内容与 INDEX.md %s' % ('一致' if not changed else '不一致,加 --write 刷新'))
|
||||
return 1 if changed else 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,179 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
docs-audit.py — 文档库质量扫描(只读,零副作用)
|
||||
|
||||
用途:一次性回答「文档是否清晰、无歧义、需要精简」这类问题,输出**可判定**的问题清单,
|
||||
避免靠人逐篇读。用于文档库根目录,也可用于任何 md 目录。
|
||||
|
||||
检查项(均可机器判定):
|
||||
1 档案编号冲突(同一编号被多份档案占用)
|
||||
2 标题号 ≠ 文件名号
|
||||
3 备份/临时残留(.bak / .orig / ~ 等,会污染对账与阅读)
|
||||
4 术语与事实漂移(旧术语、旧域名、已废弃表名)——历史档案保留原文属正常,需在入口说明
|
||||
5 体量分布(超长文件 → 考虑拆分;过短文件 → 考虑合并/指针)
|
||||
6 交叉引用有效性(引用「档案 NN」/「04-调整方案/NN-」是否存在)
|
||||
7 疑似重复/近重复(同一主题两处维护;含"子集包含"识别)
|
||||
8 元信息规范(档案头部是否含 日期 / 状态)
|
||||
9 非 md 文件的入库合理性提示
|
||||
|
||||
用法:
|
||||
python3 scripts/docs-audit.py [文档库根目录,默认取本脚本的上一级]
|
||||
|
||||
退出码:0 = 无 P0 级问题;1 = 发现编号冲突或失效引用(便于接 CI)。
|
||||
"""
|
||||
import io, os, re, sys, collections
|
||||
|
||||
ROOT = sys.argv[1] if len(sys.argv) > 1 else os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
P0 = 0
|
||||
|
||||
def rd(rel):
|
||||
return io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
|
||||
|
||||
def all_files():
|
||||
out = []
|
||||
for base, dirs, names in os.walk(ROOT):
|
||||
if '.git' in dirs:
|
||||
dirs.remove('.git')
|
||||
for n in names:
|
||||
out.append(os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/'))
|
||||
return sorted(out)
|
||||
|
||||
FILES = all_files()
|
||||
MD = [f for f in FILES if f.endswith('.md')]
|
||||
OTHER = [f for f in FILES if not f.endswith('.md')]
|
||||
|
||||
print('=' * 72)
|
||||
print('文档库质量扫描 root=%s' % ROOT)
|
||||
print('总文件 %d(md %d / 其他 %d)' % (len(FILES), len(MD), len(OTHER)))
|
||||
print('=' * 72)
|
||||
|
||||
# ── 1&2 编号 ──────────────────────────────────────────────
|
||||
num_map = collections.defaultdict(list)
|
||||
mismatch = []
|
||||
titles = {}
|
||||
for f in MD:
|
||||
s = rd(f)
|
||||
title = next((l.strip() for l in s.split('\n') if l.strip().startswith('#')), '')
|
||||
titles[f] = title
|
||||
mf = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
|
||||
mt = re.match(r'^#\s*(?:调整方案\s*)?(\d+[a-z]?)\s*[·、..\-]', title)
|
||||
if mf:
|
||||
num_map[mf.group(1)].append(f)
|
||||
if mf and mt and mf.group(1) != mt.group(1):
|
||||
mismatch.append((f, title[:70]))
|
||||
|
||||
print('\n【1】档案编号冲突')
|
||||
conf = {n: fs for n, fs in num_map.items() if len(fs) > 1}
|
||||
if not conf:
|
||||
print(' ✓ 无冲突')
|
||||
for n, fs in sorted(conf.items()):
|
||||
# 顶层 NN-x.md 与 04-调整方案/NN-x.md 属两套体系,不算冲突
|
||||
top = [x for x in fs if '/' not in x]
|
||||
arch = [x for x in fs if x.startswith('04-调整方案/')]
|
||||
if top and arch and len(fs) == 2:
|
||||
print(' · 编号 %s:根级 vs 04-调整方案(**两套体系,非冲突**,但入口须写明)' % n)
|
||||
continue
|
||||
P0 = 1
|
||||
print(' ⚠ 编号 %s 被 %d 份档案占用:' % (n, len(fs)))
|
||||
for x in fs:
|
||||
print(' %-56s 标题「%s」' % (x, titles[x][:44]))
|
||||
|
||||
print('\n【2】标题号 ≠ 文件名号')
|
||||
if not mismatch:
|
||||
print(' ✓ 无')
|
||||
for f, t in mismatch:
|
||||
P0 = 1
|
||||
print(' ⚠ %-52s → 「%s」' % (f, t))
|
||||
|
||||
# ── 3 备份残留 ───────────────────────────────────────────
|
||||
print('\n【3】备份/临时残留')
|
||||
junk = [f for f in FILES if re.search(r'\.bak|~$|\.orig$|\.tmp$|\.swp$|\.new$', f)]
|
||||
print(' 数量 %d %s' % (len(junk), '(建议清理或纳入 .gitignore)' if junk else ''))
|
||||
for f in junk[:15]:
|
||||
print(' -', f)
|
||||
|
||||
# ── 4 术语/事实漂移 ─────────────────────────────────────
|
||||
print('\n【4】术语与事实漂移(历史档案保留原文=正常,需入口说明)')
|
||||
TERMS = {'业务插件(旧术语)': '业务插件', 'dsh.alotbuy.com(旧域名)': 'dsh.alotbuy.com',
|
||||
'folder_plugins(已废弃)': 'folder_plugins'}
|
||||
for label, k in TERMS.items():
|
||||
fs = [f for f in MD if k in rd(f)]
|
||||
print(' %-26s %d 个文件' % (label, len(fs)))
|
||||
if fs and len(fs) <= 6:
|
||||
print(' %s' % ', '.join(fs))
|
||||
|
||||
# ── 5 体量 ──────────────────────────────────────────────
|
||||
print('\n【5】体量分布(>300 行考虑拆分;<15 行考虑合并或指针化)')
|
||||
rows = sorted(((rd(f).count('\n') + 1, len(rd(f).encode()), f) for f in MD), reverse=True)
|
||||
print(' 最长 8:')
|
||||
for ln, by, f in rows[:8]:
|
||||
print(' %5d 行 %7.1f KB %s' % (ln, by / 1024, f))
|
||||
print(' 最短 5:')
|
||||
for ln, by, f in rows[-5:]:
|
||||
print(' %5d 行 %7.1f KB %s' % (ln, by / 1024, f))
|
||||
|
||||
# ── 6 引用有效性 ────────────────────────────────────────
|
||||
print('\n【6】交叉引用有效性')
|
||||
existing = set(num_map.keys())
|
||||
bad = collections.defaultdict(list)
|
||||
for f in MD:
|
||||
s = rd(f)
|
||||
for m in re.finditer(r'档案\s*(\d{1,2}[a-z]?)', s):
|
||||
n = m.group(1)
|
||||
# 键已改为字符串(支持 '37a');原先的数值范围判断 1<=n<=99 换成形状校验
|
||||
if re.fullmatch(r'\d{1,2}[a-z]?', n) and n not in existing:
|
||||
bad[f].append(n)
|
||||
if not bad:
|
||||
print(' ✓ 无悬空档案号引用')
|
||||
else:
|
||||
P0 = 1
|
||||
for f, ns in list(bad.items())[:12]:
|
||||
print(' %-48s 引用了不存在的档案号 %s' % (f, sorted(set(ns))))
|
||||
|
||||
# ── 7 近重复 ────────────────────────────────────────────
|
||||
print('\n【7】疑似重复 / 近重复')
|
||||
sig = {f: re.sub(r'\s+', '', rd(f)) for f in MD}
|
||||
found = False
|
||||
keys = list(sig)
|
||||
for i in range(len(keys)):
|
||||
for j in range(i + 1, len(keys)):
|
||||
a, b = sig[keys[i]], sig[keys[j]]
|
||||
if len(a) < 200 or len(b) < 200:
|
||||
continue
|
||||
if a == b:
|
||||
print(' %-44s = %-44s 完全相同' % (keys[i], keys[j])); found = True
|
||||
elif a in b or b in a:
|
||||
small, big = (keys[i], keys[j]) if len(a) < len(b) else (keys[j], keys[i])
|
||||
print(' %-44s ⊂ %-44s 子集(%d/%d 字符)' % (small, big, min(len(a), len(b)), max(len(a), len(b)))); found = True
|
||||
else:
|
||||
ga = set(a[k:k + 3] for k in range(0, len(a) - 2, 3))
|
||||
gb = set(b[k:k + 3] for k in range(0, len(b) - 2, 3))
|
||||
if ga and gb:
|
||||
r = len(ga & gb) / min(len(ga), len(gb))
|
||||
if r > 0.55:
|
||||
print(' %-44s ≈ %-44s 重合 %.0f%%' % (keys[i], keys[j], r * 100)); found = True
|
||||
if not found:
|
||||
print(' ✓ 未发现')
|
||||
|
||||
# ── 8 元信息 ────────────────────────────────────────────
|
||||
print('\n【8】档案头部元信息(04-调整方案/ 内)')
|
||||
no_date, no_status = [], []
|
||||
for f in MD:
|
||||
if not f.startswith('04-调整方案/'):
|
||||
continue
|
||||
head = '\n'.join(rd(f).split('\n')[:14])
|
||||
if not re.search(r'日期|20\d\d-\d\d-\d\d', head):
|
||||
no_date.append(f)
|
||||
if not re.search(r'状态', head):
|
||||
no_status.append(f)
|
||||
print(' 缺「日期」%d 个 %s' % (len(no_date), [os.path.basename(x) for x in no_date[:6]]))
|
||||
print(' 缺「状态」%d 个 %s' % (len(no_status), [os.path.basename(x) for x in no_status[:6]]))
|
||||
|
||||
# ── 9 非 md ─────────────────────────────────────────────
|
||||
print('\n【9】非 md 文件 %d 个(确认均属应入库的资产/脚本)' % len(OTHER))
|
||||
print(' ' + (', '.join(OTHER[:12]) + (' …' if len(OTHER) > 12 else '') if OTHER else '无'))
|
||||
|
||||
print('\n' + '=' * 72)
|
||||
print('结论:%s' % ('发现问题(见上 ⚠)' if P0 else '无 P0 级问题'))
|
||||
sys.exit(1 if P0 else 0)
|
||||
@@ -0,0 +1,161 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
docs-consistency.py —— 文档库「事实一致性」校验(只读,可复跑)
|
||||
|
||||
为什么需要它:`docs-audit.py` 查的是**结构问题**(编号冲突 / 悬空引用 / 重复子集),
|
||||
**不查事实是否与现状一致** → 过时值会一直躺在库里没人发现。
|
||||
|
||||
两类检查:
|
||||
|
||||
【1】写死的取值 —— 会随并行改动过期,应改为"复跑取号"
|
||||
· "下一号 = NN"(档案编号)
|
||||
· 旧本机身份 `maidou` / `/c/Users/<user>/.workbuddy`(已迁 `E:\\ProgramData\\.workbuddy`)
|
||||
|
||||
【2】跨页取值不一致 —— 同一个事实键在多个「承诺现行」文件里取值不同
|
||||
(首跑实证:`INDEX.md` 写下一号=72、`README.md` 写 69 → 同类事实两处打架)
|
||||
|
||||
判据:**「承诺现行」的文件不许出现已废止/写死/互相矛盾的取值。**
|
||||
- 承诺现行(必查):`BRIEF.md` / `INDEX.md` / `README.md` / `CODEBUDDY.md` / `DEPLOY-本部署.md`
|
||||
/ `03-路线图与待办.md` / `06-工作台UI规范.md` / `交接单/**` / `skills/**`
|
||||
- 豁免(历史事实,只增不改):`04-调整方案/**`、`archive/**`、`01-规划与架构.md`、`02-运维手册.md`
|
||||
—— 它们写的时候那个值是对的,回改反而破坏历史。
|
||||
|
||||
⚠️ **刻意不查**「旧域名 dsh.alotbuy.com」「旧配额 512M」:这两者在库里几乎都是
|
||||
"旧域已 301" / "512M→384M" 的**合法历史表述**,正则无法可靠区分,误报率过高。
|
||||
改由人工在改域名/改配额时顺手核(档案 22 / 58 是权威)。
|
||||
|
||||
用法:
|
||||
python3 scripts/docs-consistency.py
|
||||
退出码:0 = 无问题;1 = 有违背;2 = 环境错误
|
||||
"""
|
||||
import argparse
|
||||
import io
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from collections import defaultdict
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
CURRENT_PREFIXES = (
|
||||
'BRIEF.md', 'INDEX.md', 'README.md', 'CODEBUDDY.md', 'DEPLOY-本部署.md',
|
||||
'03-路线图与待办.md', '06-工作台UI规范.md', '交接单/', 'skills/',
|
||||
)
|
||||
|
||||
# 【1】写死的取值:正则 → (说明, 修复建议)
|
||||
HARDCODED = [
|
||||
('档案「下一号」写死', r'下一号\s*[==]\s*\*{0,2}\d+', '改为"复跑取号,勿写死"(并行改动会打穿)'),
|
||||
('旧本机用户名 maidou', r'maidou', '现行 Administrator'),
|
||||
('旧技能/配置目录', r'[/\\][cC][/\\]Users[/\\][^/\\\s]+[/\\]\.workbuddy',
|
||||
'现行 E:\\ProgramData\\.workbuddy(已迁 E 盘)'),
|
||||
]
|
||||
|
||||
# 【2】跨页一致性:事实键 → 抽取正则(捕获组即取值)
|
||||
CROSS_FACTS = {
|
||||
'档案下一号': r'下一号\s*[==]\s*\*{0,2}(\d+)',
|
||||
'代码 HEAD': r'代码 HEAD\s*[` ]?\s*\*{0,2}([0-9a-f]{7,10})',
|
||||
'实例 MemoryMax': r'MemoryMax\s*[==]?\s*\*{0,2}(\d{3,4})\s*(?:MiB|M\b)?',
|
||||
}
|
||||
|
||||
|
||||
def read(p):
|
||||
try:
|
||||
return io.open(p, encoding='utf-8', errors='replace').read()
|
||||
except Exception:
|
||||
return ''
|
||||
|
||||
|
||||
def collect():
|
||||
out = []
|
||||
for root, dirs, files in os.walk(ROOT):
|
||||
dirs[:] = [d for d in dirs if d not in ('.git', 'node_modules', '__pycache__')]
|
||||
for f in files:
|
||||
if f.endswith(('.md', '.py', '.sh', '.cjs', '.mjs', '.js')):
|
||||
out.append(os.path.relpath(os.path.join(root, f), ROOT).replace('\\', '/'))
|
||||
return sorted(out)
|
||||
|
||||
|
||||
def is_current(rel):
|
||||
return any(rel == p or rel.startswith(p) for p in CURRENT_PREFIXES)
|
||||
|
||||
|
||||
def is_quoted(line, start, end):
|
||||
"""匹配是否被包住 —— 包住的内容视为**引用/举例**(如 T02 记的"下一号 = 20"已归零、
|
||||
文档里把 `maidou` 当反例引用),不是当前断言,不算违规。
|
||||
|
||||
三类包裹:`" "` / `“ ”`(引文)与 `` ` ` ``(代码字面量)。"""
|
||||
pre = line[:start].rstrip()
|
||||
post = line[end:].lstrip()
|
||||
return bool(pre and pre[-1] in '"“”`\'') and bool(post and post[0] in '"“”`\'')
|
||||
|
||||
|
||||
def scan(rx, text):
|
||||
"""返回一个文件里「不在引号内」的匹配数。"""
|
||||
n = 0
|
||||
for line in text.splitlines():
|
||||
for m in rx.finditer(line):
|
||||
if is_quoted(line, m.start(), m.end()):
|
||||
continue
|
||||
n += 1
|
||||
return n
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.parse_args()
|
||||
|
||||
files = collect()
|
||||
cur = [f for f in files if is_current(f)]
|
||||
print('文档库根:%s' % ROOT)
|
||||
print('文件总数 %d | 承诺现行 %d 个(历史豁免 %d 个)\n' % (len(files), len(cur), len(files) - len(cur)))
|
||||
|
||||
bad = 0
|
||||
|
||||
print('──【1】写死的取值 ──')
|
||||
for label, pat, fix in HARDCODED:
|
||||
rx = re.compile(pat)
|
||||
hits = [(r, scan(rx, read(os.path.join(ROOT, r)))) for r in cur]
|
||||
hits = [h for h in hits if h[1]]
|
||||
print(' 【%s】→ %s' % (label, fix))
|
||||
if not hits:
|
||||
print(' ✓ 无')
|
||||
else:
|
||||
bad += len(hits)
|
||||
for r, n in sorted(hits, key=lambda x: -x[1])[:8]:
|
||||
print(' ⚠ %-52s %d 处' % (r[:52], n))
|
||||
if len(hits) > 8:
|
||||
print(' …还有 %d 个文件' % (len(hits) - 8))
|
||||
print()
|
||||
|
||||
print('──【2】跨页取值一致性 ──')
|
||||
for name, pat in CROSS_FACTS.items():
|
||||
rx = re.compile(pat)
|
||||
vals = defaultdict(list)
|
||||
for r in cur:
|
||||
txt = read(os.path.join(ROOT, r))
|
||||
for line in txt.splitlines():
|
||||
for m in rx.finditer(line):
|
||||
if is_quoted(line, m.start(), m.end()):
|
||||
continue
|
||||
vals[m.group(1)].append(r)
|
||||
print(' 【%s】' % name)
|
||||
if len(vals) <= 1:
|
||||
print(' ✓ 取值唯一:%s' % (list(vals)[0] if vals else '(未出现)'))
|
||||
else:
|
||||
bad += 1
|
||||
for v, rs in sorted(vals.items()):
|
||||
print(' ⚠ 取值 %s ← %s' % (v, ', '.join(sorted(set(rs))[:4])))
|
||||
print(' → 同一事实多处取值不一致,请校正为同一个权威值(或改为"复跑取号")')
|
||||
print()
|
||||
|
||||
print('=' * 64)
|
||||
if bad:
|
||||
print('结论:**%d 项需处理**' % bad)
|
||||
return 1
|
||||
print('结论:承诺现行的文件与现行值一致 ✓')
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,122 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""docs-dedupe.py — 巨型档案的**重复块体检 + 去重视图生成**(不改原文)
|
||||
|
||||
背景(2026-09-14 实测)
|
||||
* `04-调整方案/82-…md` = **2024 行 / 91 KB**,其中同一份「§一–§八」**重复 16 次**;
|
||||
* 但本库铁律是 **L5 冻结:历史档案不回改** ⇒ 不能"直接删重复"。
|
||||
⇒ 本工具遵守铁律:**只读原文 → 生成"去重视图"到库外 + 报告变体差异**;
|
||||
原文仅允许在**文末追加**「修正(YYYY-MM-DD)」小节(本库明文许可)。
|
||||
|
||||
做三件事
|
||||
1. **块级重复统计**:按 `#` / `##` / `###` 切块,归一化后哈希 ⇒ 报"哪些标题重复几次";
|
||||
2. **变体检测**:同一标题的多次出现若**内容不同**(hash 不同),逐个列出(避免"以为一样其实有改动");
|
||||
3. **去重视图**:写入 `--view-out`(默认库外 `.workbuddy/cache/dedupe-view/<名>.md`),
|
||||
每组保留**信息最全的那一份**(字符数最大,并列取首次),其余位置用一行占位注释替代。
|
||||
|
||||
用法
|
||||
python3 scripts/docs-dedupe.py <相对路径> # 只报告
|
||||
python3 scripts/docs-dedupe.py <相对路径> --view-out=路径 # 报告 + 写视图
|
||||
退出码:0 = 无重复;1 = 有重复(可生成视图);2 = 用法/文件错误
|
||||
"""
|
||||
import io
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import hashlib
|
||||
import collections
|
||||
|
||||
DOCS = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
WS = os.path.dirname(DOCS)
|
||||
HEAD = re.compile(r'^(#{1,3})\s+(.*)$')
|
||||
|
||||
|
||||
def blocks(lines):
|
||||
"""按 #/##/### 切块;返回 [(起始行, 级别, 标题, 正文行列表)],含块前导内容。"""
|
||||
out, cur = [], None
|
||||
pre = []
|
||||
for i, line in enumerate(lines, 1):
|
||||
m = HEAD.match(line)
|
||||
if m:
|
||||
if cur is None:
|
||||
cur = [i, len(m.group(1)), m.group(2).strip(), list(pre)]
|
||||
else:
|
||||
out.append(cur + [i - 1])
|
||||
cur = [i, len(m.group(1)), m.group(2).strip(), []]
|
||||
elif cur is None:
|
||||
pre.append(line)
|
||||
else:
|
||||
cur[3].append(line)
|
||||
if cur is not None:
|
||||
out.append(cur + [len(lines)])
|
||||
return out, pre
|
||||
|
||||
|
||||
def norm(level, title, body):
|
||||
text = re.sub(r'\s+', ' ', '\n'.join(body)).strip()
|
||||
return hashlib.sha1(('%d|%s|%s' % (level, title, text)).encode('utf-8')).hexdigest()[:12]
|
||||
|
||||
|
||||
def main(argv):
|
||||
files = [a for a in argv if not a.startswith('--')]
|
||||
if not files:
|
||||
print(__doc__)
|
||||
return 2
|
||||
rel = files[0]
|
||||
path = os.path.join(DOCS, rel)
|
||||
if not os.path.exists(path):
|
||||
print('ERROR: 找不到 %s' % path)
|
||||
return 2
|
||||
raw = io.open(path, encoding='utf-8', newline='').read()
|
||||
lines = raw.split('\n')
|
||||
bs, pre = blocks(lines)
|
||||
|
||||
groups = collections.defaultdict(list)
|
||||
for start, level, title, body, end in bs:
|
||||
groups[(level, title, norm(level, title, body))].append((start, end, body))
|
||||
|
||||
by_title = collections.defaultdict(list)
|
||||
for (level, title, h), occ in groups.items():
|
||||
by_title[title].append((h, occ))
|
||||
|
||||
dup_titles = {t: v for t, v in by_title.items() if sum(len(o) for _, o in v) > 1}
|
||||
total_dup_blocks = sum(len(o) - 1 for v in dup_titles.values() for _, o in v)
|
||||
print('文件 %s:%d 行 / %d 字符|块 %d 个|**重复标题 %d 个,冗余块 %d 个**'
|
||||
% (rel, len(lines), len(raw), len(bs), len(dup_titles), total_dup_blocks))
|
||||
for title, variants in sorted(dup_titles.items(),
|
||||
key=lambda kv: -sum(len(o) for _, o in kv[1]))[:12]:
|
||||
times = sum(len(o) for _, o in variants)
|
||||
vinfo = '|'.join('变体%d×%d次' % (i + 1, len(o)) for i, (_, o) in enumerate(variants))
|
||||
mark = ' ⚠️内容有差异' if len(variants) > 1 else ''
|
||||
print(' %2d× %-46s %s%s' % (times, title[:44], vinfo, mark))
|
||||
|
||||
view_out = next((a.split('=', 1)[1] for a in argv if a.startswith('--view-out=')), None)
|
||||
if view_out is None:
|
||||
default = os.path.join(WS, '.workbuddy', 'cache', 'dedupe-view',
|
||||
rel.replace('/', '__'))
|
||||
view_out = default if '--view-out' in str(argv) else None
|
||||
if view_out:
|
||||
keep = {}
|
||||
for title, variants in dup_titles.items():
|
||||
allocc = [(h, start, end, body) for h, o in variants for start, end, body in o]
|
||||
allocc.sort(key=lambda x: (-sum(len(l) + 1 for l in x[3]), x[1]))
|
||||
keep[title] = allocc[0]
|
||||
out, skipped = [], 0
|
||||
for start, level, title, body, end in bs:
|
||||
cur = (start, end, body)
|
||||
keeper = keep.get(title)
|
||||
if keeper is not None and (start, end, body) != (keeper[1], keeper[2], keeper[3]):
|
||||
skipped += 1
|
||||
out.append('%s<!-- 去重省略:同「%s」块(原文 L%d–L%d),内容与 L%d–L%d 那份一致 -->'
|
||||
% ('#' * level + ' ', title, start, end, keeper[1], keeper[2]))
|
||||
else:
|
||||
out.append('\n'.join(['#' * level + ' ' + title] + body))
|
||||
os.makedirs(os.path.dirname(view_out), exist_ok=True)
|
||||
io.open(view_out, 'w', encoding='utf-8', newline='\n').write('\n'.join(out) + '\n')
|
||||
print('\n去重视图已写:%s(省略 %d 块,%.1f KB → %.1f KB)'
|
||||
% (view_out, skipped, len(raw) / 1024, os.path.getsize(view_out) / 1024))
|
||||
return 1 if dup_titles else 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
raise SystemExit(main(sys.argv[1:]))
|
||||
@@ -0,0 +1,130 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""docs-index-stats.py — 从 INDEX.md §二 清单表**算**出状态分布,并可就地刷新「状态摘要」行。
|
||||
|
||||
为什么需要它
|
||||
────────────
|
||||
「状态摘要」原本是**人工手写**的硬数字,档案一多必然漂移(2026-09-13 实测:摘要写「档案 72 份」,
|
||||
而表格实际 76 行)。本脚本把摘要变成**机器生成**:读表 → 统计 → 打印;`--write` 时回写该行,
|
||||
并顺带做**图例自检**;**完整保持文件原有行尾**(CRLF 文件不会被转成 LF)(表格用到的图标必须在「图例」行里有定义,缺则补)。
|
||||
|
||||
用法
|
||||
────
|
||||
python3 scripts/docs-index-stats.py # 只打印 + 一致性判定
|
||||
python3 scripts/docs-index-stats.py --write # 就地刷新「状态摘要」行(必要时补图例)
|
||||
退出码:0 = 一致(或已刷新);1 = 漂移且未加 --write;2 = 结构异常。
|
||||
"""
|
||||
|
||||
import io
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import collections
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
INDEX = os.path.join(ROOT, "INDEX.md")
|
||||
|
||||
ORDER = ["✅", "🔄", "🧪", "📝", "🔍", "📋", "🟡", "🗄", "🔧", "⚠️"]
|
||||
LEGEND_OF = {
|
||||
"✅": "已落地", "🔄": "维护中", "🧪": "PoC", "📝": "待开发", "🔍": "核查完成",
|
||||
"📋": "评估", "🟡": "保留兜底", "🗄": "归档", "🔧": "修复", "⚠️": "警示",
|
||||
}
|
||||
HEADER_RE = re.compile(r"^\|\s*号\s*\|\s*状态\s*\|")
|
||||
LEAF_RE = re.compile(r"^0\d$")
|
||||
|
||||
|
||||
def parse(index_path):
|
||||
"""返回 (lines, arch_rows, leaf_rows, counter, other_count)。"""
|
||||
# ⚠️ 必须 newline=""(通用换行会把 CRLF 静默转成 LF —— 2026-09-13 踩过两次)
|
||||
raw = io.open(index_path, encoding="utf-8", newline="").read()
|
||||
lines = raw.split("\n")
|
||||
try:
|
||||
start = next(i for i, l in enumerate(lines) if HEADER_RE.match(l))
|
||||
except StopIteration:
|
||||
raise SystemExit("ERROR: INDEX.md 里找不到「| 号 | 状态 | 一句话 |」表头")
|
||||
arch, leaf, other = [], [], 0
|
||||
for l in lines[start + 2:]:
|
||||
if not l.startswith("|"):
|
||||
break
|
||||
cells = [x.strip() for x in l.split("|")]
|
||||
if len(cells) < 4 or not cells[1]:
|
||||
continue
|
||||
no, st = cells[1], cells[2]
|
||||
if no.startswith("04-"):
|
||||
arch.append((no, st))
|
||||
elif LEAF_RE.match(no):
|
||||
leaf.append((no, st))
|
||||
else:
|
||||
other += 1
|
||||
cnt = collections.Counter(st for _, st in arch)
|
||||
cnt.update(st for _, st in leaf)
|
||||
eol = "\r\n" if raw.count("\r\n") > 0 else "\n"
|
||||
return lines, arch, leaf, cnt, other, eol
|
||||
|
||||
|
||||
def summary_text(cnt, n_arch, leaf_names, other):
|
||||
parts = ["%s %d" % (k, cnt[k]) for k in ORDER if cnt.get(k)]
|
||||
unmarked = sum(v for k, v in cnt.items() if k not in ORDER)
|
||||
if unmarked:
|
||||
parts.append("未标记 %d" % unmarked)
|
||||
leaf = (",另含根级编号 %d 条(%s)" % (len(leaf_names), "/".join(leaf_names))) if leaf_names else ""
|
||||
return (
|
||||
"> **状态摘要**(**机器生成,勿手改**):档案 **%d** 份(`04-*`)%s,"
|
||||
"另有非编号行 %d 条(README / INDEX / 技能 / poc 等)—— %s。"
|
||||
"复跑 `python3 scripts/docs-index-stats.py` 取数,`--write` 就地刷新本行。"
|
||||
% (n_arch, leaf, other, " | ".join(parts))
|
||||
)
|
||||
|
||||
|
||||
def main():
|
||||
write = "--write" in sys.argv
|
||||
lines, arch, leaf, cnt, other, eol = parse(INDEX)
|
||||
leaf_names = [n for n, _ in leaf]
|
||||
new = summary_text(cnt, len(arch), leaf_names, other)
|
||||
|
||||
print("档案 04-* %d 份 | 根级编号 %s | 非编号行 %d 条" % (len(arch), "/".join(leaf_names) or "—", other))
|
||||
for k in ORDER:
|
||||
if cnt.get(k):
|
||||
print(" %s %-6s %d" % (k, LEGEND_OF.get(k, ""), cnt[k]))
|
||||
unmarked = sum(v for k, v in cnt.items() if k not in ORDER)
|
||||
if unmarked:
|
||||
print(" 未标记 %d" % unmarked)
|
||||
|
||||
changed = False
|
||||
|
||||
# ── 图例自检 ─────────────────────────────────────────────────────────
|
||||
li = [i for i, l in enumerate(lines) if l.startswith("> 图例:")]
|
||||
if li:
|
||||
legend = lines[li[0]]
|
||||
missing = [k for k in cnt if k not in legend]
|
||||
if missing:
|
||||
print("⚠️ 图例缺图标:%s" % " ".join(missing))
|
||||
if write:
|
||||
lines[li[0]] = legend.rstrip().rstrip("|") + "".join(
|
||||
"|%s%s" % (k, LEGEND_OF.get(k, "")) for k in missing
|
||||
)
|
||||
changed = True
|
||||
print("✓ 已补进图例行")
|
||||
|
||||
# ── 摘要行 ───────────────────────────────────────────────────────────
|
||||
idx = [i for i, l in enumerate(lines) if l.startswith("> **状态摘要**")]
|
||||
if not idx:
|
||||
print("ERROR: 未找到「状态摘要」行", file=sys.stderr)
|
||||
return 2
|
||||
old = lines[idx[0]]
|
||||
if old.strip() == new.strip() and not changed:
|
||||
print("✓ 摘要与表格一致")
|
||||
return 0
|
||||
if not write:
|
||||
print("✗ 摘要与表格不一致(加 --write 刷新)")
|
||||
print(" 旧: %s" % old.strip()[:90])
|
||||
print(" 新: %s" % new.strip()[:90])
|
||||
return 1
|
||||
lines[idx[0]] = new
|
||||
io.open(INDEX, "w", encoding="utf-8", newline="").write(eol.join(lines))
|
||||
print("✓ 已刷新 INDEX.md(摘要行%s)" % ("+图例" if changed else ""))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,201 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
docs-manifest.py — 生成文档库**机读清单**(供 AI/脚本过滤,替代"读大表格")
|
||||
|
||||
产出:
|
||||
docs-manifest.json 每份文档:编号/标题/状态/日期/字符/行数/被引用次数/分层(tier)/主题标签
|
||||
控制台 体量总览 + 分层统计 + 「常读」与「历史」清单
|
||||
|
||||
分层规则(2026-09-14 变更:由"**被引几次**"改为"**谁在引**"):
|
||||
hot 被**现行层 L1/L2**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范)反复引用 ≥3 次 —— 改东西前大概率要看
|
||||
cur 被现行层引用 1–2 次 —— 相关即看(单次引用也可能极重要,如红线类档案 07)
|
||||
warm 仅被历史档案互引(全库 ≥1,现行层 0)—— 参考
|
||||
cold 零引用 —— 历史,只在追溯时看
|
||||
doc 根级长期文档 / 资产,不参与档案分层
|
||||
⚠️ 实测教训:**L4(INDEX / 待办 / 台账)不能算现行层** —— 它们会顺带列出几乎所有档案号,
|
||||
会让判据反向失效(第一版含 L4 时 hot 从 44 抬到 60)。
|
||||
|
||||
用法:python3 scripts/docs-manifest.py [文档库根目录]
|
||||
副作用:仅写 docs-manifest.json(其余只读)
|
||||
"""
|
||||
import io, os, re, sys, json, collections
|
||||
|
||||
|
||||
def norm_status(raw):
|
||||
"""规范化档案头部的状态行 → 表里可读的一小段(≤14 字,去 ** 与尾部括注)。
|
||||
|
||||
⚠️ **权威方向**:档案头部是**源**,INDEX 表是**派生展示** —— 不允许把表里的 ❓ 回灌成 manifest 的 status,
|
||||
否则 manifest → INDEX → manifest 形成环形锁定,状态永远修不回来(2026-09-14 实测 14 篇被锁)。
|
||||
"""
|
||||
t = re.sub(r'\*{1,2}', '', raw or '').strip()
|
||||
t = re.split(r'——|||\||;|;|\s{2,}', t)[0].strip()
|
||||
t = re.sub(r'([^)]*)\s*$', '', t).strip()
|
||||
return t[:14]
|
||||
|
||||
ROOT = sys.argv[1] if len(sys.argv) > 1 else os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
def rd(rel):
|
||||
try:
|
||||
return io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
|
||||
except OSError:
|
||||
return ''
|
||||
|
||||
MD = []
|
||||
for base, dirs, names in os.walk(ROOT):
|
||||
if '.git' in dirs:
|
||||
dirs.remove('.git')
|
||||
for n in names:
|
||||
if n.endswith('.md') and '.bak' not in n:
|
||||
MD.append(os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/'))
|
||||
MD.sort()
|
||||
|
||||
# ── 状态/日期:优先取 INDEX 表,其次取档案头部 ─────────────
|
||||
index_status = {}
|
||||
if os.path.exists(os.path.join(ROOT, 'INDEX.md')):
|
||||
for line in rd('INDEX.md').split('\n'):
|
||||
m = re.match(r'^\|\s*(?:04-)?(\d{1,3}[a-z]?)\s*\|', line)
|
||||
if not m:
|
||||
continue
|
||||
marks = [k for k in ('✅', '🔄', '🔧', '🧪', '🗄', '⏸', '❌') if k in line]
|
||||
dm = re.search(r'\b(\d{2}-\d{2})\b', line)
|
||||
index_status[m.group(1)] = {'status': marks[0] if marks else '?', 'date': dm.group(1) if dm else ''}
|
||||
|
||||
# ── 引用热度 ────────────────────────────────────────────
|
||||
DOMAIN_RULES = {
|
||||
'external': r'官方推荐|awesome|npm|cordis|开源导出|外部源|生态',
|
||||
'method': r'方法|决策|技能|skill|流程|工作流|协议|协作',
|
||||
'plugin': r'插件|univer|mcn|business-plugins|dsh-plugin|原生绑定',
|
||||
'ui': r'UI|前端|门户|页面|portal|界面|client bundle|渲染',
|
||||
'ops': r'运维|排障|重启|备份|配额|内存|OOM|证书|nginx|nft|执行锁|并发',
|
||||
'platform': r'平台|部署|隔离|bwrap|多租户|实例|profile|配额口径|架构|命名|重构|升级|耦合|回归清单|目标架构',
|
||||
'ops2': r'登录|401|404|竞态|冷启动|启动失败|排障|死锁',
|
||||
'ui2': r'会话提示|文案|提示条|覆盖层|面板|分区|交互',
|
||||
'method2': r'文档|信息架构|清单|机读|模板|规范|收尾|去重',
|
||||
}
|
||||
def domain_of(text):
|
||||
scores = collections.Counter()
|
||||
for name, pat in DOMAIN_RULES.items():
|
||||
scores[name] = len(re.findall(pat, text, re.I))
|
||||
best = scores.most_common(1)[0]
|
||||
if best[1] <= 0:
|
||||
return '?'
|
||||
return {'ops2': 'ops', 'ui2': 'ui', 'method2': 'method'}.get(best[0], best[0])
|
||||
|
||||
def layer_of(path, num):
|
||||
if num: return 'L5'
|
||||
if path.startswith('skills/'): return 'L3'
|
||||
if path.startswith('交接单/'): return 'L4'
|
||||
if path in ('BRIEF.md',): return 'L1'
|
||||
if path == 'CODEBUDDY.md': return 'L2'
|
||||
if path == 'INDEX.md': return 'L4'
|
||||
if path == 'README.md': return 'L2'
|
||||
if path == 'DEPLOY-本部署.md': return 'L1' # 现行部署事实
|
||||
if path == '06-工作台UI规范.md': return 'L2' # 强制基线 = 规则
|
||||
if path == '03-路线图与待办.md': return 'L4' # 状态/待办
|
||||
if path in ('BRIEF.md',): return 'L1'
|
||||
if path.startswith('docs-manifest') or path.endswith('.json'): return 'L2'
|
||||
if path.startswith('01-规划与架构'): return 'L0'
|
||||
if path.startswith('02-运维手册') or path.startswith('archive/'): return 'L5'
|
||||
if path == '04-调整方案/README.md': return 'L5' # 目录内索引 ⇒ 随其层
|
||||
return '?'
|
||||
|
||||
VIEW_DIR = os.path.join(os.path.dirname(ROOT), '.workbuddy', 'cache', 'dedupe-view')
|
||||
|
||||
|
||||
def view_field(rel):
|
||||
"""库外的"去重视图"(docs-dedupe.py 产物)—— 有则记路径,供检索/阅读优先使用。"""
|
||||
cand = os.path.join(VIEW_DIR, rel.replace('/', '__'))
|
||||
return {'dedupeView': cand} if os.path.exists(cand) else {}
|
||||
|
||||
|
||||
def num_of(f):
|
||||
m = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
|
||||
return m.group(1) if (m and f.startswith('04-调整方案/') and '/poc/' not in f) else None
|
||||
|
||||
# 引用分两份:**全库** 与 **仅 L1+L2 层**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范 = 现行事实与规则)。
|
||||
# ⚠️ 2026-09-14 实测教训:**L4(INDEX / 待办 / 台账)不能算** —— 它们会顺带列出几乎所有档案号,
|
||||
# 导致 hot 从 44 抬到 60(判据反而变糟);"被台账提到" ≠ "改东西前要看"。
|
||||
# 2026-09-14 判据变更:tier 由"被引几次"改为"**谁在引**" —— 只有被现行层引用才算 hot。
|
||||
ref = collections.Counter()
|
||||
ref_current = collections.Counter()
|
||||
for f in MD:
|
||||
lay = layer_of(f, num_of(f))
|
||||
for m in re.finditer(r'档案\s*(\d{1,2}[a-z]?)', rd(f)):
|
||||
ref[m.group(1)] += 1
|
||||
if lay in ('L1', 'L2'):
|
||||
ref_current[m.group(1)] += 1
|
||||
|
||||
# ── 域标签(检索键;未命中记 '?',由人补规则)─────────────
|
||||
items = []
|
||||
for f in MD:
|
||||
s = rd(f)
|
||||
title = next((l.strip().lstrip('#').strip() for l in s.split('\n') if l.strip().startswith('#')), '')
|
||||
mnum = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
|
||||
num = mnum.group(1) if (mnum and f.startswith('04-调整方案/') and '/poc/' not in f) else None
|
||||
head = '\n'.join(s.split('\n')[:14])
|
||||
m_tl = re.search(r'>\s*\*\*TL;DR\*\*[||::]?\s*(.+)', s)
|
||||
tldr = re.sub(r'\s+', ' ', m_tl.group(1)).strip(' ||')[:120] if m_tl else ''
|
||||
hd = re.search(r'(- 日期:\s*)([0-9]{4}-[0-9]{2}-[0-9]{2})', head)
|
||||
hs = re.search(r'状态\*{0,2}\s*[::]\s*(.{1,40})', head)
|
||||
hs_val = norm_status(hs.group(1)) if hs else ''
|
||||
st_idx = (index_status.get(num, {}) or {}).get('status') or ''
|
||||
if st_idx in ('?', '❓', ''):
|
||||
st_idx = '' # ❓/? 一律视为「缺」—— 防 INDEX↔manifest 环形锁定(2026-09-14)
|
||||
n_ref = ref.get(num, 0) if num else 0
|
||||
n_cur = ref_current.get(num, 0) if num else 0
|
||||
if not num:
|
||||
tier = 'doc' # 根级长期文档 / 资产:不参与"档案分层"
|
||||
elif n_cur >= 3:
|
||||
tier = 'hot' # 被**现行层反复引用** ⇒ 改东西前大概率要看
|
||||
elif n_cur >= 1:
|
||||
tier = 'cur' # 被现行层引用 1–2 次 ⇒ 相关即看(单次引用≠不重要,如红线类档案 07)
|
||||
elif n_ref >= 1:
|
||||
tier = 'warm' # 仅被历史档案互引 ⇒ 参考
|
||||
else:
|
||||
tier = 'cold' # 零引用
|
||||
items.append({
|
||||
'path': f,
|
||||
'num': num,
|
||||
'title': title[:80],
|
||||
'status': hs_val or st_idx or '?',
|
||||
'date': (hd.group(2) if hd else (index_status.get(num, {}) or {}).get('date', '') or ''),
|
||||
'chars': len(s),
|
||||
'lines': s.count('\n') + 1,
|
||||
'refs': n_ref,
|
||||
'refsCurrent': n_cur,
|
||||
'tier': tier,
|
||||
'layer': layer_of(f, num),
|
||||
'domain': domain_of(title + '\n' + head),
|
||||
'tldr': tldr,
|
||||
**view_field(f),
|
||||
})
|
||||
|
||||
out = {
|
||||
'generatedFrom': 'scripts/docs-manifest.py',
|
||||
'counts': {'files': len(items), 'chars': sum(i['chars'] for i in items)},
|
||||
'tiers': {t: sum(1 for i in items if i['tier'] == t) for t in ('hot', 'cur', 'warm', 'cold', 'doc')},
|
||||
'domains': dict(collections.Counter(i['domain'] for i in items)),
|
||||
'layers': dict(collections.Counter(i['layer'] for i in items)),
|
||||
'items': items,
|
||||
}
|
||||
io.open(os.path.join(ROOT, 'docs-manifest.json'), 'w', encoding='utf-8', newline='\n').write(
|
||||
json.dumps(out, ensure_ascii=False, indent=1) + '\n')
|
||||
|
||||
arch = [i for i in items if i['num'] is not None]
|
||||
print('文档 %d 份 / %s 字符(≈%s tokens)' % (len(items), format(out['counts']['chars'], ','),
|
||||
format(int(out['counts']['chars'] * 0.7), ',')))
|
||||
print('档案 %d 份 | 分层:hot %d / warm %d / cold %d | 根级文档 %d'
|
||||
% (len(arch), out['tiers']['hot'], out['tiers']['warm'], out['tiers']['cold'], out['tiers']['doc']))
|
||||
oversized = sorted([i for i in items if i['chars'] > 30000], key=lambda x: -x['chars'])
|
||||
if oversized:
|
||||
print('\n【⚠️ 单篇 > 30 KB】%d 篇(约定上限 30 KB;历史只报不改,新档案超限须拆):' % len(oversized))
|
||||
for i in oversized:
|
||||
print(' %6.1f KB %-58s' % (i['chars'] / 1024, os.path.basename(i['path'])[:56]))
|
||||
print('\n【hot】常读(引用 ≥8 次)—— 改东西前大概率要看:')
|
||||
for i in sorted([x for x in arch if x['tier'] == 'hot'], key=lambda x: -x['refs']):
|
||||
print(' 档案 %-3s %2d 次 %-52s %6d 字符' % (i['num'], i['refs'], os.path.basename(i['path'])[:50], i['chars']))
|
||||
print('\n【cold】零引用(历史候选,可只留索引行):')
|
||||
for i in sorted([x for x in arch if x['tier'] == 'cold'], key=lambda x: x['num']):
|
||||
print(' 档案 %-3s %s' % (i['num'], os.path.basename(i['path'])))
|
||||
print('\n→ 已写出 docs-manifest.json')
|
||||
@@ -0,0 +1,142 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""docs-search.py — 文档库**带语义的全文检索**(零依赖、不建索引、每次实时扫)
|
||||
|
||||
为什么需要它(2026-09-14 实测)
|
||||
* 库已到 **118 篇 / 95 万字符 ≈ 67 万 token** ⇒ 不可能"读全库";
|
||||
* 只能 `grep` 时,**搜到的结果分不清"现行值"还是"历史值"** —— 这正是 2026-09-12 踩过的坑
|
||||
(旧配额活在 15 个文件里,被当成事实用)。
|
||||
⇒ 本脚本把 `docs-manifest.json` 的 **(层 L0–L5 / 域 / tier / 状态)** 标注接进检索结果,
|
||||
并给 `--current` 一键**排除历史层**,让"查现行事实"这件事**结果可判**。
|
||||
|
||||
用法
|
||||
python3 scripts/docs-search.py 配额 # 全库搜「配额」
|
||||
python3 scripts/docs-search.py 配额 --current # **只搜现行层(排除 L5 档案 / archive)**
|
||||
python3 scripts/docs-search.py 插件 域 # 多词 = AND
|
||||
python3 scripts/docs-search.py glibc --domain plugin --layer L5
|
||||
python3 scripts/docs-search.py 内存 --json | jq . # 机读输出
|
||||
选项
|
||||
--current 只搜 L0–L4(排除 L5 与 archive/)—— **查现行值请默认加它**
|
||||
--layer L1,L2 限定层(L0–L5)
|
||||
--domain plugin 限定域(platform/plugin/ui/ops/external/method)
|
||||
--limit N 最多返回多少篇(默认 12)
|
||||
--context N 每篇显示多少条命中行(默认 3,0 = 只统计)
|
||||
--json 输出 JSON(供脚本消费)
|
||||
退出码:0 = 有命中;1 = 无命中;2 = 用法/环境错误
|
||||
"""
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
HISTORY_PREFIXES = ('04-调整方案/', 'archive/', '02-运维手册')
|
||||
WEIGHT = {'hot': 1.0, 'cur': 0.9, 'warm': 0.8, 'doc': 0.7, 'cold': 0.5}
|
||||
|
||||
|
||||
def load_meta():
|
||||
p = os.path.join(ROOT, 'docs-manifest.json')
|
||||
if not os.path.exists(p):
|
||||
raise SystemExit('ERROR: 缺 docs-manifest.json —— 先跑 scripts/docs-manifest.py')
|
||||
out = {}
|
||||
for i in json.loads(io.open(p, encoding='utf-8').read())['items']:
|
||||
out[i['path']] = i
|
||||
return out
|
||||
|
||||
|
||||
def walk():
|
||||
for base, dirs, names in os.walk(ROOT):
|
||||
if '.git' in dirs:
|
||||
dirs.remove('.git')
|
||||
for n in sorted(names):
|
||||
if n.endswith('.md') and '.bak' not in n:
|
||||
yield os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/')
|
||||
|
||||
|
||||
def main(argv):
|
||||
VALUE_OPTS = {'--layer', '--domain', '--limit', '--context'}
|
||||
terms, opts, flags = [], {}, set()
|
||||
i = 0
|
||||
while i < len(argv):
|
||||
a = argv[i]
|
||||
if a in VALUE_OPTS and i + 1 < len(argv):
|
||||
opts[a] = argv[i + 1]
|
||||
i += 2
|
||||
continue
|
||||
if a.startswith('--') and '=' in a:
|
||||
k, v = a.split('=', 1)
|
||||
opts[k] = v
|
||||
elif a.startswith('--'):
|
||||
flags.add(a)
|
||||
else:
|
||||
terms.append(a)
|
||||
i += 1
|
||||
flag = lambda k: k in flags # noqa: E731
|
||||
opt = lambda k, d=None: opts.get(k, d) # noqa: E731
|
||||
if not terms:
|
||||
print(__doc__)
|
||||
return 2
|
||||
terms = [t.lower() for t in terms]
|
||||
meta = load_meta()
|
||||
cur = flag('--current')
|
||||
want_layers = set((opt('--layer') or '').split(',')) - {''}
|
||||
want_domain = opt('--domain')
|
||||
limit = int(opt('--limit', '12'))
|
||||
ctx = int(opt('--context', '3'))
|
||||
|
||||
hits = []
|
||||
for rel in walk():
|
||||
if cur and (rel.startswith(HISTORY_PREFIXES) or rel == 'archive'):
|
||||
continue
|
||||
m = meta.get(rel, {})
|
||||
layer, domain = m.get('layer', '?'), m.get('domain', '?')
|
||||
if want_layers and layer not in want_layers:
|
||||
continue
|
||||
if want_domain and domain != want_domain:
|
||||
continue
|
||||
try:
|
||||
text = io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
|
||||
except OSError:
|
||||
continue
|
||||
low = text.lower()
|
||||
counts = [low.count(t) for t in terms]
|
||||
if not all(c > 0 for c in counts): # AND 语义
|
||||
continue
|
||||
total = sum(counts)
|
||||
head = (m.get('title') or '') + ' ' + (m.get('tldr') or '')
|
||||
boost = 1.4 if any(t in head.lower() for t in terms) else 1.0
|
||||
lines = text.split('\n')
|
||||
shown = [(n + 1, lines[n].strip()[:150]) for n in range(len(lines))
|
||||
if all(t in lines[n].lower() for t in terms)][:ctx]
|
||||
hits.append({
|
||||
'path': rel, 'layer': layer, 'domain': domain,
|
||||
'tier': m.get('tier', '?'), 'status': m.get('status', '?'),
|
||||
'hits': total, 'score': round(total * WEIGHT.get(m.get('tier'), 0.7) * boost, 1),
|
||||
'title': m.get('title', ''), 'samples': shown,
|
||||
'view': bool(m.get('dedupeView')),
|
||||
})
|
||||
hits.sort(key=lambda x: -x['score'])
|
||||
hits = hits[:limit]
|
||||
|
||||
if flag('--json'):
|
||||
print(json.dumps({'terms': terms, 'current_only': cur, 'results': hits},
|
||||
ensure_ascii=False, indent=1))
|
||||
else:
|
||||
scope = '仅现行层(L0–L4)' if cur else '全库(含历史 L5)'
|
||||
print('检索 %r | %s | 命中 %d 篇%s'
|
||||
% (' + '.join(terms), scope, len(hits), '' if not cur else ' (查现行值建议保持 --current)'))
|
||||
for h in hits:
|
||||
print('\n[%s] %s | %s/%s | %s | %d 次'
|
||||
% (h['status'], h['path'], h['layer'], h['domain'], h['tier'], h['hits']))
|
||||
if h['title']:
|
||||
print(' %s' % h['title'][:110])
|
||||
if h.get('view'):
|
||||
print(' ↳ 本篇有**去重视图**(体积更小,优先读它)')
|
||||
for ln, text in h['samples']:
|
||||
print(' L%-5d %s' % (ln, text))
|
||||
return 0 if hits else 1
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
raise SystemExit(main(sys.argv[1:]))
|
||||
@@ -0,0 +1,122 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""docs-shrink-guard.py — 防「共享文件被整文件重写抹掉别人的行」
|
||||
|
||||
为什么需要(2026-09-14 实测)
|
||||
* `MEMORY.md` 被并行会话**整文件重写**,抹掉了本会话刚写进去的两行;
|
||||
* 本库的规矩是「共享文件只用 Edit 精确替换,禁整文件 Write」—— 但**只靠人守纪律**,没有机制兜底。
|
||||
⇒ 本脚本给"纪律"加一条**机械兜底**:记录每个文件的**行数快照**,下次运行时若某文件
|
||||
**行数骤降**(默认 >30% 且绝对减少 >20 行),就把它当成"疑似被整文件重写"报警。
|
||||
|
||||
覆盖范围(两处共享热区):
|
||||
* 文档库 `dsh-server-docs/**/*.md`
|
||||
* 工作区记忆 `.workbuddy/memory/*.md`
|
||||
快照落在**库外**(`<工作区>/.workbuddy/cache/docs-lines.json`)—— 放库内会污染 `docs-sync-check` 对账。
|
||||
|
||||
用法
|
||||
python3 scripts/docs-shrink-guard.py # 比对并报告(不改快照)
|
||||
python3 scripts/docs-shrink-guard.py --write # 比对 + 刷新快照(改完文件后跑)
|
||||
python3 scripts/docs-shrink-guard.py --baseline # 只建档不比对(首次)
|
||||
python3 scripts/docs-shrink-guard.py --allow-shrink <路径子串> # 合法重构白名单(不报警)
|
||||
退出码:0 = 无骤降;1 = 发现骤降(需人看一眼 diff);2 = 用法错误
|
||||
"""
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
DOCS = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
WS = os.path.dirname(DOCS)
|
||||
SNAP = os.path.join(WS, '.workbuddy', 'cache', 'docs-lines.json')
|
||||
DROP_RATIO, DROP_MIN = 0.30, 20
|
||||
|
||||
|
||||
def collect():
|
||||
out = {}
|
||||
roots = [(DOCS, 'docs'), (os.path.join(WS, '.workbuddy', 'memory'), 'memory')]
|
||||
for root, tag in roots:
|
||||
if not os.path.isdir(root):
|
||||
continue
|
||||
for base, dirs, names in os.walk(root):
|
||||
if '.git' in dirs:
|
||||
dirs.remove('.git')
|
||||
if os.sep + 'cache' in base:
|
||||
continue
|
||||
for n in sorted(names):
|
||||
if not n.endswith('.md') or '.bak' in n:
|
||||
continue
|
||||
p = os.path.join(base, n)
|
||||
rel = os.path.relpath(p, WS).replace('\\', '/')
|
||||
try:
|
||||
out[rel] = sum(1 for _ in io.open(p, encoding='utf-8', errors='replace'))
|
||||
except OSError:
|
||||
continue
|
||||
return out
|
||||
|
||||
|
||||
def main(argv):
|
||||
now = collect()
|
||||
allow = [argv[i + 1] for i, a in enumerate(argv)
|
||||
if a == '--allow-shrink' and i + 1 < len(argv)]
|
||||
old = {}
|
||||
if os.path.exists(SNAP):
|
||||
try:
|
||||
old = json.loads(io.open(SNAP, encoding='utf-8').read())
|
||||
except ValueError:
|
||||
print('⚠️ 快照损坏,按首次处理')
|
||||
if '--baseline' in argv or not old:
|
||||
if '--write' in argv or '--baseline' in argv:
|
||||
os.makedirs(os.path.dirname(SNAP), exist_ok=True)
|
||||
io.open(SNAP, 'w', encoding='utf-8', newline='\n').write(
|
||||
json.dumps(now, ensure_ascii=False, indent=1, sort_keys=True) + '\n')
|
||||
print('已建基线:%d 个文件 → %s' % (len(now), SNAP))
|
||||
return 0
|
||||
print('无快照(首次)。跑 `--baseline` 建档后再用。')
|
||||
return 2
|
||||
|
||||
shrunk, added, removed, grew = [], [], [], []
|
||||
for rel, n in now.items():
|
||||
if rel not in old:
|
||||
added.append(rel)
|
||||
elif n < old[rel]:
|
||||
if old[rel] - n > DROP_MIN and (old[rel] - n) / old[rel] > DROP_RATIO:
|
||||
if any(a in rel for a in allow):
|
||||
print(' ↳ 合法重构(白名单):%s %d → %d 行' % (rel, old[rel], n))
|
||||
else:
|
||||
shrunk.append((rel, old[rel], n))
|
||||
elif old[rel] - n > 0:
|
||||
grew.append((rel, old[rel], n, 'down'))
|
||||
for rel in old:
|
||||
if rel not in now:
|
||||
removed.append(rel)
|
||||
for rel, n in now.items():
|
||||
if rel in old and n > old[rel]:
|
||||
grew.append((rel, old[rel], n, 'up'))
|
||||
|
||||
print('文件 %d(新增 %d / 删除 %d)| 行数变化 %d' % (len(now), len(added), len(removed), len(grew)))
|
||||
for rel in added[:8]:
|
||||
print(' + %s' % rel)
|
||||
for rel in removed[:8]:
|
||||
print(' - %s(消失?)' % rel)
|
||||
for rel, o, n, d in sorted(grew, key=lambda x: -abs(x[1] - x[2]))[:6]:
|
||||
print(' %s %s %d → %d 行' % ('↑' if d == 'up' else '↓', rel, o, n))
|
||||
|
||||
bad = 0
|
||||
if shrunk:
|
||||
bad = 1
|
||||
print('\n⚠️ **疑似被整文件重写(行数骤降)** —— 请看一眼 `git diff` 或与上一位写入者核对:')
|
||||
for rel, o, n in shrunk:
|
||||
print(' %s %d → %d 行(-%.0f%%)' % (rel, o, n, (o - n) / o * 100))
|
||||
print(' 本库规矩:共享文件只用 Edit 精确替换,禁整文件 Write(见 CODEBUDDY.md)')
|
||||
if '--write' in argv:
|
||||
os.makedirs(os.path.dirname(SNAP), exist_ok=True)
|
||||
io.open(SNAP, 'w', encoding='utf-8', newline='\n').write(
|
||||
json.dumps(now, ensure_ascii=False, indent=1, sort_keys=True) + '\n')
|
||||
print('已刷新快照。')
|
||||
elif bad:
|
||||
print('(未刷新快照:确认无误后跑 `--write`)')
|
||||
return bad
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
raise SystemExit(main(sys.argv[1:]))
|
||||
@@ -0,0 +1,115 @@
|
||||
#!/usr/bin/env bash
|
||||
# docs-sync-check.sh — 对账「本机文档镜像」与「服务器文档库」
|
||||
#
|
||||
# 用法:
|
||||
# bash scripts/docs-sync-check.sh # 报告双端差异
|
||||
# DOCS_REMOTE=bt-server bash scripts/docs-sync-check.sh
|
||||
#
|
||||
# 可覆盖的环境变量:
|
||||
# DOCS_LOCAL_DIR 本机镜像目录(默认 D:/github/dsh_shenxian/dsh-server-docs)
|
||||
# DOCS_REMOTE ssh 别名或 host(默认 bt-server)
|
||||
# DOCS_REMOTE_DIR 服务器文档目录(默认 /opt/dsh/docs)
|
||||
#
|
||||
# 退出码:0 = 完全一致;1 = 存在差异;2 = 环境/连接错误
|
||||
set -uo pipefail
|
||||
|
||||
LOCAL_DIR="${DOCS_LOCAL_DIR:-D:/github/dsh_shenxian/dsh-server-docs}"
|
||||
REMOTE="${DOCS_REMOTE:-bt-server}"
|
||||
REMOTE_DIR="${DOCS_REMOTE_DIR:-/opt/dsh/docs}"
|
||||
|
||||
[ -d "$LOCAL_DIR" ] || { echo "ERROR: 本机目录不存在: $LOCAL_DIR" >&2; exit 2; }
|
||||
|
||||
tmp_l="$(mktemp)"; tmp_r="$(mktemp)"
|
||||
trap 'rm -f "$tmp_l" "$tmp_r" 2>/dev/null || true' EXIT
|
||||
|
||||
# ── 本机哈希:优先「一次 Python 算完」───────────────────────────────
|
||||
# 历史坑(2026-09-12 实测):Git Bash 下 `find | while read` **逐文件 spawn** `md5sum` / `cut`
|
||||
# (132 文件 ≈ 264 次进程启动)→ 全量对账 **2 分 15 秒**,慢到被调用方(handoff-guard 信息模式)当成"挂死"。
|
||||
# Python 一次遍历 → 秒级。探测顺序:$DSH_PY → python3 → python → 本机兜底绝对路径。
|
||||
PY="${DSH_PY:-}"
|
||||
if [ -z "$PY" ]; then
|
||||
for c in python3 python; do
|
||||
if command -v "$c" >/dev/null 2>&1; then PY="$c"; break; fi
|
||||
done
|
||||
fi
|
||||
if [ -z "$PY" ] && [ -x "E:/ProgramData/.workbuddy/binaries/python/versions/3.13.12/python.exe" ]; then
|
||||
PY="E:/ProgramData/.workbuddy/binaries/python/versions/3.13.12/python.exe"
|
||||
fi
|
||||
|
||||
hash_local() {
|
||||
if [ -n "$PY" ]; then
|
||||
( cd "$LOCAL_DIR" && "$PY" -c '
|
||||
import os, sys, hashlib
|
||||
try:
|
||||
sys.stdout.reconfigure(encoding="utf-8", newline="\n")
|
||||
except Exception:
|
||||
pass
|
||||
SKIPF = (".git/", "交接单/.doing-", "交接单/.exec-lock")
|
||||
SKIPD = (".git", "交接单/.doing-", "交接单/.exec-lock")
|
||||
rows = []
|
||||
for root, dirs, files in os.walk("."):
|
||||
# ⚠️ 不能用 os.path.relpath 拼 rel —— Windows 上它会把「末尾带点」的名字规范化掉
|
||||
# (实测:`INDEX.md.bak-….` 变成无点的名字 → open 失败 → 该文件被静默漏掉 → 制造假差异)。
|
||||
r = root.replace(os.sep, "/")
|
||||
pfx = "" if r in (".", "") else (r[2:] if r.startswith("./") else r)
|
||||
dirs[:] = [d for d in dirs if not (((pfx + "/") if pfx else "") + d).startswith(SKIPD)]
|
||||
for fn in files:
|
||||
rel = (pfx + "/" + fn) if pfx else fn
|
||||
if fn == ".DS_Store" or any(rel.startswith(p) for p in SKIPF):
|
||||
continue
|
||||
if os.path.islink(rel):
|
||||
continue # 与 find -type f 同口径
|
||||
try:
|
||||
with open(rel, "rb") as fh:
|
||||
h = hashlib.md5(fh.read()).hexdigest()
|
||||
except Exception:
|
||||
# Windows 会把「末尾带点 / 保留名」等非常规文件名规范化掉 → open 直接 ENOENT。
|
||||
# 用 \\?\ 前缀绕过,**必须拼未规范化的绝对路径**(os.path.abspath 会 strip 末尾点 → 前缀失效)。
|
||||
try:
|
||||
with open("\\\\?\\" + os.getcwd() + "\\" + rel.replace("/", "\\"), "rb") as fh:
|
||||
h = hashlib.md5(fh.read()).hexdigest()
|
||||
except Exception:
|
||||
continue
|
||||
rows.append((rel, h))
|
||||
rows.sort(key=lambda r: r[0].encode("utf-8")) # 与 LC_ALL=C sort 同口径
|
||||
for rel, h in rows:
|
||||
print(rel + "\t" + h)
|
||||
' )
|
||||
else
|
||||
echo "WARN: 未找到 python,回退逐文件 md5sum(会慢 ~2 分钟)" >&2
|
||||
( cd "$LOCAL_DIR" && find . -type f ! -path './.git/*' ! -name '.DS_Store' \
|
||||
! -path './交接单/.doing-*' ! -path './交接单/.exec-lock*' | LC_ALL=C sort | while read -r f; do
|
||||
printf '%s\t%s\n' "${f#./}" "$(md5sum "$f" | cut -d' ' -f1)"
|
||||
done )
|
||||
fi
|
||||
}
|
||||
|
||||
hash_remote() {
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" \
|
||||
"cd '$REMOTE_DIR' 2>/dev/null && find . -type f ! -path './.git/*' ! -name '.DS_Store' ! -path './交接单/.doing-*' ! -path './交接单/.exec-lock*' | LC_ALL=C sort | while read -r f; do printf '%s\t%s\n' \"\${f#./}\" \"\$(md5sum \"\$f\" | cut -d' ' -f1)\"; done" \
|
||||
2>/dev/null
|
||||
}
|
||||
|
||||
hash_local > "$tmp_l"
|
||||
hash_remote > "$tmp_r"
|
||||
|
||||
if [ ! -s "$tmp_r" ]; then
|
||||
echo "ERROR: 无法读取服务器目录 $REMOTE:$REMOTE_DIR(ssh 失败或目录不存在)" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
awk -F'\t' '
|
||||
NR==FNR { l[$1]=$2; next }
|
||||
{ r[$1]=$2 }
|
||||
END {
|
||||
same=0
|
||||
for (k in l) {
|
||||
if (k in r) { if (l[k]==r[k]) same++; else { diff++; print "⚠️ 内容不一致 " k } }
|
||||
else { only_l++; print "⬆️ 仅本地(待推送) " k }
|
||||
}
|
||||
for (k in r) if (!(k in l)) { only_r++; print "⬇️ 仅服务器(待拉取) " k }
|
||||
printf "\n—— 汇总 ——\n 一致: %d\n 内容不一致: %d\n 仅本地: %d\n 仅服务器: %d\n 本地文件总数: %d / 服务器文件总数: %d\n", same, diff+0, only_l+0, only_r+0, length(l), length(r)
|
||||
if ((diff+0)+(only_l+0)+(only_r+0) > 0) { print "\n结果: 存在差异 ❌"; exit 1 }
|
||||
print "\n结果: 双端一致 ✅"
|
||||
}
|
||||
' "$tmp_l" "$tmp_r"
|
||||
@@ -0,0 +1,175 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
extract-user-voice.py —— 把一个工作区的**全部历史会话**里的「用户真实发言」抽成一份可读清单。
|
||||
|
||||
为什么要它:做「决策方法 / 协作方式」复盘时,需要用**用户的原话全集**做底料,
|
||||
而不是只靠档案与日志(那是会话的结构化沉淀,会丢掉"用户否决了什么、纠正了什么")。
|
||||
本脚本只读、不改任何会话文件。
|
||||
|
||||
输出到 stdout(可重定向到自己想放的路径;**不要写进文档库目录**,否则会被 docs-sync-check 计成"仅本地"):
|
||||
|
||||
python3 scripts/extract-user-voice.py # 自动定位当前工作区
|
||||
python3 scripts/extract-user-voice.py --project <dir名> # 指定 ~/.workbuddy/projects/<dir名>
|
||||
python3 scripts/extract-user-voice.py --full # 打印全文(默认每条截断 88 字)
|
||||
python3 scripts/extract-user-voice.py --needle 关键词 # 只列含关键词的发言(找某个决策的来龙去脉)
|
||||
|
||||
会话记录位置(WorkBuddy):`<配置目录>/projects/<把 cwd 的 : \\ / 换成 ->/<sessionId>.jsonl`
|
||||
- 配置目录判定链:`WORKBUDDY_CONFIG_DIR ?? CODEBUDDY_CONFIG_DIR ?? ~/.workbuddy`
|
||||
- 行格式:`{"type":"message","role":"user","content":[{"type":"input_text","text":"..."}]}`
|
||||
- ⚠️ 用户发言里会带一大段 `<system-reminder ...>` 前言(user_info / identity_context),
|
||||
必须剥掉才是**用户真实说的话**。
|
||||
"""
|
||||
import argparse
|
||||
import datetime
|
||||
import glob
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
SREM = re.compile(r'<system-reminder.*?</system-reminder>', re.S)
|
||||
TAGS = re.compile(r'</?(user_query|cb_summary|conversation_history_summary)>')
|
||||
WS = re.compile(r'\s+')
|
||||
|
||||
|
||||
def config_dir():
|
||||
for k in ('WORKBUDDY_CONFIG_DIR', 'CODEBUDDY_CONFIG_DIR'):
|
||||
v = os.environ.get(k)
|
||||
if v:
|
||||
return os.path.expanduser(v)
|
||||
return os.path.join(os.path.expanduser('~'), '.workbuddy')
|
||||
|
||||
|
||||
def project_dir_name(cwd):
|
||||
"""WorkBuddy 的目录名规则(实测):盘符**小写** + ':' 去掉,随后把分隔符各换成 '-'。
|
||||
例:E:\\ProgramData\\AI技能\\aliyun-dsh-server → e-ProgramData-AI技能-aliyun-dsh-server
|
||||
(路径中段的字母大小写与空格**保留**)。"""
|
||||
p = os.path.abspath(cwd)
|
||||
drive, rest = os.path.splitdrive(p)
|
||||
return (drive.rstrip(':').lower() + re.sub(r'[\\/]', '-', rest)).strip('-')
|
||||
|
||||
|
||||
def _norm(s):
|
||||
"""把目录名归一化,用于模糊比较(忽略盘符大小写、多余连字符)。"""
|
||||
return re.sub(r'-+', '-', s).lower()
|
||||
|
||||
|
||||
def resolve_project(root, cwd, explicit):
|
||||
"""返回 (目录名, 绝对路径)。显式 --project 只用它;否则从 cwd 起**逐级向上**试,
|
||||
全部落空后再对 projects/ 下的实际目录做一次归一化模糊匹配。"""
|
||||
if explicit:
|
||||
return (explicit, os.path.join(root, explicit))
|
||||
|
||||
tried = []
|
||||
cur = os.path.abspath(cwd)
|
||||
while True:
|
||||
name = project_dir_name(cur)
|
||||
tried.append(name)
|
||||
p = os.path.join(root, name)
|
||||
if os.path.isdir(p):
|
||||
return (name, p)
|
||||
parent = os.path.dirname(cur)
|
||||
if parent == cur:
|
||||
break
|
||||
cur = parent
|
||||
|
||||
# 模糊兜底:按归一化名比对实际存在的目录
|
||||
try:
|
||||
existing = [d for d in os.listdir(root) if os.path.isdir(os.path.join(root, d))]
|
||||
except OSError:
|
||||
existing = []
|
||||
for want in tried:
|
||||
for d in existing:
|
||||
if _norm(d) == _norm(want):
|
||||
return (d, os.path.join(root, d))
|
||||
return (tried[0], os.path.join(root, tried[0]))
|
||||
|
||||
|
||||
def fmt_ts(v):
|
||||
if isinstance(v, (int, float)):
|
||||
v = v / 1000.0 if v > 1e11 else v
|
||||
try:
|
||||
return datetime.datetime.fromtimestamp(v).strftime('%m-%d %H:%M')
|
||||
except Exception:
|
||||
return '?'
|
||||
return str(v)[:16] if v else '?'
|
||||
|
||||
|
||||
def user_text(line):
|
||||
"""从一行 jsonl 里取出「用户真实发言」;不是用户消息则返回 None。"""
|
||||
try:
|
||||
o = json.loads(line)
|
||||
except Exception:
|
||||
return None
|
||||
if o.get('type') != 'message' or o.get('role') != 'user':
|
||||
return None
|
||||
txt = ''
|
||||
for b in (o.get('content') or []):
|
||||
if isinstance(b, dict) and b.get('type') in ('input_text', 'text'):
|
||||
txt += b.get('text', '')
|
||||
txt = SREM.sub('', txt).strip()
|
||||
txt = TAGS.sub('', txt).strip()
|
||||
txt = WS.sub(' ', txt)
|
||||
return (fmt_ts(o.get('timestamp')), txt) if txt else None
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument('--project', help='projects 下的目录名(默认按当前 cwd 推导)')
|
||||
ap.add_argument('--cwd', default=os.getcwd(), help='用于推导项目名的 cwd')
|
||||
ap.add_argument('--full', action='store_true', help='打印全文(不截断)')
|
||||
ap.add_argument('--needle', help='只列含该关键词的发言')
|
||||
a = ap.parse_args()
|
||||
|
||||
root = os.path.join(config_dir(), 'projects')
|
||||
name, pdir = resolve_project(root, a.cwd, a.project)
|
||||
if not os.path.isdir(pdir):
|
||||
sys.stderr.write('找不到项目会话目录:%s\n' % pdir)
|
||||
sys.stderr.write('可用目录(%s):\n' % root)
|
||||
for d in sorted(os.listdir(root))[:40]:
|
||||
sys.stderr.write(' %s\n' % d)
|
||||
return 2
|
||||
|
||||
files = sorted(glob.glob(os.path.join(pdir, '*.jsonl')))
|
||||
if not files:
|
||||
sys.stderr.write('该目录下没有 .jsonl 会话文件:%s\n' % pdir)
|
||||
return 2
|
||||
|
||||
total = 0
|
||||
print('项目会话目录:%s' % pdir)
|
||||
print('会话文件 %d 个\n' % len(files))
|
||||
|
||||
print('===== 各会话概览 =====')
|
||||
per = []
|
||||
for f in files:
|
||||
msgs = []
|
||||
with io.open(f, encoding='utf-8', errors='replace') as fh:
|
||||
for line in fh:
|
||||
r = user_text(line)
|
||||
if r:
|
||||
msgs.append(r)
|
||||
per.append((os.path.basename(f[:-6]), msgs))
|
||||
total += len(msgs)
|
||||
span = ('%s → %s' % (msgs[0][0], msgs[-1][0])) if msgs else '-'
|
||||
print(' %-38s 用户发言 %4d 条 %s' % (os.path.basename(f)[:38], len(msgs), span))
|
||||
print('\n合计用户真实发言 = %d 条\n' % total)
|
||||
|
||||
for sid, msgs in per:
|
||||
if a.needle:
|
||||
msgs = [m for m in msgs if a.needle in m[1]]
|
||||
if not msgs:
|
||||
continue
|
||||
print('===== %s(命中 %d 条)=====' % (sid, len(msgs)))
|
||||
else:
|
||||
print('===== %s(%d 条)=====' % (sid, len(msgs)))
|
||||
for i, (t, m) in enumerate(msgs, 1):
|
||||
body = m if a.full else (m[:88] + ('…' if len(m) > 88 else ''))
|
||||
print('%4d %s %s' % (i, t, body))
|
||||
print()
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,292 @@
|
||||
#!/usr/bin/env bash
|
||||
# handoff-guard.sh — 开工 / 推送前的「并行冲突预检」(只读;--claim 除外,它只建一个锁目录)
|
||||
#
|
||||
# 为什么需要它:本库由**多个 AI 会话并行**读写,而「先读后改 / Edit 增量 / 改完 commit」这类
|
||||
# 约定全部依赖“人记得做”。本脚本把判据变成**一条命令 + 退出码**,不靠记忆。
|
||||
#
|
||||
# 关键设计:**mtime 只作提示、不作判定**(它分不清“谁改的”)。真正的判定来自四处:
|
||||
# ① 全局执行锁(交接单/.exec-lock)—— **一粗**:同一时刻只允许一个执行会话动「文档/代码/服务器」
|
||||
# ① 单级占用锁(交接单/.doing-<单号>)—— **一细**:这个单归谁做(供台账 / 接管使用)
|
||||
# ② 越界改动(不在我声明清单里的未提交文件)—— 推送前检查,防“顺手重放别人的半成品”
|
||||
# ④ 双端一致性(docs-sync-check.sh)—— 推送前检查,防“幽灵文件”(推回了别人已移走的文件)
|
||||
#
|
||||
# 两级锁的关系:**先抢全局锁 → 再占单级锁**;释放时**先放单级、再放全局**。
|
||||
# 单级锁允许“两个会话各做一单”(冲突域不重叠时);全局锁则彻底禁止并行执行。
|
||||
# ⇒ 本库现状(共享入口文件多 + 要动服务器)建议**默认只跑一个执行会话**,即始终持全局锁。
|
||||
#
|
||||
# 用法:
|
||||
# bash scripts/handoff-guard.sh --claim-exec "exec-session-B" # 【第一步】抢全局执行锁
|
||||
# bash scripts/handoff-guard.sh --claim T03 "exec-session-B" # 【第二步】占单级锁
|
||||
# bash scripts/handoff-guard.sh --release T03 # 完工:先放单级
|
||||
# bash scripts/handoff-guard.sh --release-exec # 再放全局
|
||||
# bash scripts/handoff-guard.sh # 看全局状态(信息模式)
|
||||
# ME="exec-session-B" MINE="交接单/T03-*.md" bash scripts/handoff-guard.sh T03 # 开工检查(严格)
|
||||
# ME="exec-session-B" MINE="..." PUSH=1 bash scripts/handoff-guard.sh T03 # 推送前检查
|
||||
#
|
||||
# 环境变量:ME(我是谁 —— 用来判断锁是不是自己的;不设则一律按“别人的锁”处理)
|
||||
# MINE(我本次要改的文件,空格分隔,支持 * 通配)
|
||||
# PUSH=1 启用推送前硬判定(④ 幽灵文件)
|
||||
# GUARD_WINDOW(分钟,默认 30)|SKIP_SYNC=1 跳过 ④
|
||||
# 退出码:0 = 放行;1 = 命中硬冲突/硬判定;2 = 环境错误
|
||||
set -uo pipefail
|
||||
|
||||
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
cd "$ROOT" || exit 2
|
||||
LOCKDIR="$ROOT/交接单/.doing-"
|
||||
LOCKEXEC="$ROOT/交接单/.exec-lock"
|
||||
ME="${ME:-}"
|
||||
|
||||
# ── 全局执行锁:占位 / 释放 ──────────────────────────────
|
||||
if [ "${1:-}" = "--claim-exec" ]; then
|
||||
OWNER="${2:-${ME:-$(whoami)}}"
|
||||
if mkdir "$LOCKEXEC" 2>/dev/null; then
|
||||
printf '%s\n开始:%s\n在做:%s\n' "$OWNER" "$(date '+%m-%d %H:%M')" "${3:-(未声明单号)}" > "$LOCKEXEC/OWNER"
|
||||
echo "✓ 已持全局执行锁($OWNER)"
|
||||
echo " ⛔ **锁的生命周期 = 任务的生命周期**(2026-09-14 用户明令):执行完成 → 必须 \`--release-exec\` 才算完成;"
|
||||
echo " 禁止抢锁做一半、不解锁就结束回合/会话(本库无心跳,带锁结束 = 把所有人挡在门外)。中途要停 ⇒ 先释放再停。"
|
||||
exit 0
|
||||
fi
|
||||
echo "✗ 抢锁失败:已有执行会话在跑 ——
|
||||
占用者:$(sed -n '1p' "$LOCKEXEC/OWNER" 2>/dev/null || echo '?') $(sed -n '2p' "$LOCKEXEC/OWNER" 2>/dev/null)
|
||||
$(sed -n '3p' "$LOCKEXEC/OWNER" 2>/dev/null)
|
||||
→ **停手**:等它做完(它会 --release-exec)。⛔ **不得人工删锁、不得接管**(**R9**,用户 2026-09-12 明令)——
|
||||
抢不到锁 = **停手 + 报告用户**;**锁的处置权只属于用户本人**" >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ "${1:-}" = "--release-exec" ]; then
|
||||
rm -rf "$LOCKEXEC" && echo "✓ 已释放全局执行锁" || { echo "✗ 释放失败" >&2; exit 1; }
|
||||
exit 0
|
||||
fi
|
||||
|
||||
|
||||
# ── 占位 / 释放 ─────────────────────────────────────────
|
||||
if [ "${1:-}" = "--claim" ]; then
|
||||
T="${2:-}"; OWNER="${3:-$(whoami)@$(date +%H:%M)}"
|
||||
[ -n "$T" ] || { echo "用法:--claim <单号> [占用者]" >&2; exit 2; }
|
||||
if mkdir "$LOCKDIR$T" 2>/dev/null; then
|
||||
printf '%s\n' "$OWNER" > "$LOCKDIR$T/OWNER"
|
||||
echo "✓ 已占位:交接单/.doing-$T($OWNER)—— 完工请 --release $T"
|
||||
exit 0
|
||||
fi
|
||||
echo "✗ 占位失败:交接单/.doing-$T 已存在(占用者 $(cat "$LOCKDIR$T/OWNER" 2>/dev/null || echo '?')) → 停手" >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ "${1:-}" = "--release" ]; then
|
||||
T="${2:-}"; [ -n "$T" ] || { echo "用法:--release <单号>" >&2; exit 2; }
|
||||
rm -rf "$LOCKDIR$T" && echo "✓ 已释放:交接单/.doing-$T" || { echo "✗ 释放失败" >&2; exit 1; }
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ── 预检 ────────────────────────────────────────────────
|
||||
MY_TASK="${1:-}"
|
||||
WINDOW="${GUARD_WINDOW:-30}"
|
||||
MINE="${MINE:-}"
|
||||
PUSH="${PUSH:-0}"
|
||||
VERDICT=0
|
||||
HARD=() # 硬失败原因
|
||||
[ -n "$MINE" ] && STRICT=1 || STRICT=0
|
||||
|
||||
echo "=================================================================="
|
||||
echo "并行冲突预检 root=$ROOT"
|
||||
echo "本次任务 = ${MY_TASK:-(未声明)}"
|
||||
echo "我是谁(ME) = ${ME:-(未声明 → 任何锁都按"别人的"处理)}"
|
||||
echo "我声明要改 = ${MINE:-(未声明 → 仅信息模式)}"
|
||||
echo "模式 = $([ "$PUSH" = "1" ] && echo 推送前检查 || echo 开工检查)"
|
||||
echo "=================================================================="
|
||||
|
||||
in_mine() { [ -z "$MINE" ] && return 1; for p in $MINE; do case "$1" in $p) return 0;; esac; done; return 1; }
|
||||
|
||||
# ── ① 占用锁(硬判定)────────────────────────────────────
|
||||
echo
|
||||
echo "【1】占用锁(交接单/.doing-*)"
|
||||
shopt -s nullglob
|
||||
LOCKS=("$LOCKDIR"*)
|
||||
if [ ${#LOCKS[@]} -eq 0 ]; then
|
||||
echo " ✓ 无人占用 —— ⚠ 这不是「可以开工」,是「**你快去抢**」:"
|
||||
echo " 开工前先:ME=\"<你的会话名>\" bash scripts/handoff-guard.sh --claim <单号> \"<你的会话名>\""
|
||||
else
|
||||
for L in "${LOCKS[@]}"; do
|
||||
n="$(basename "$L")"; o="$(cat "$L/OWNER" 2>/dev/null || echo '(未写 OWNER)')"
|
||||
if [ -n "$MY_TASK" ] && [ "$n" = ".doing-$MY_TASK" ]; then
|
||||
echo " · $n ← 你自己占的($o)"
|
||||
else
|
||||
echo " ⚠ $n 被占用:$o → 冲突域重叠就别开工"
|
||||
VERDICT=1; HARD+=("① 别的会话持有占用锁 $n($o)")
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
# ── ①b 锁 ↔ 台账一致性(2026-09-12 新增)────────────────
|
||||
# 实证:T03 10:14 就被占了,但台账那行直到 12:59 还写着「待执行」→ 别人看台账会以为**没人做**,
|
||||
# 于是可能重复开工(这不是文件写冲突,而是"状态不可见"造成的重复劳动)。
|
||||
if [ ${#LOCKS[@]} -gt 0 ]; then
|
||||
for L in "${LOCKS[@]}"; do
|
||||
n="$(basename "$L")"; T="${n#.doing-}"
|
||||
row="$(grep -m1 "^| \`$T-" 交接单/README.md 2>/dev/null || true)"
|
||||
if [ -z "$row" ]; then
|
||||
echo " ⚠ $T 有占用锁,但 \`交接单/README.md §一\` 里**没有这一行** → 补上(新增单)"
|
||||
VERDICT=1; HARD+=("①b $T 有锁但台账缺行")
|
||||
elif printf '%s' "$row" | grep -qE "执行中|已完成|已归档"; then
|
||||
echo " · $T 台账已标「执行中/已完成」✓"
|
||||
else
|
||||
echo " ⚠ $T 已被占用($n),但台账那行仍写作「$(printf '%s' "$row" | awk -F'|' '{print $3}' | tr -d ' ')」"
|
||||
echo " → **台账滞后 = 别人可能重复开工**:立刻把该行状态改成「🔄 执行中」"
|
||||
VERDICT=1; HARD+=("①b $T 有锁但台账未标「执行中」(重复开工风险)")
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
# ── ①c 全局执行锁(2026-09-12 新增:同一时刻只允许一个执行会话)────
|
||||
echo
|
||||
echo "【1c】全局执行锁(交接单/.exec-lock)"
|
||||
if [ ! -d "$LOCKEXEC" ]; then
|
||||
echo " ✓ 无全局锁 —— ⚠️ 这**不是「可以开工」,是「你快去抢」**:"
|
||||
echo " 开工前必须先:bash scripts/handoff-guard.sh --claim-exec \"<你的会话名>\""
|
||||
echo " (「环境干净」≠「没人动过」;抢锁是原子的,抢到才是你的开工许可)"
|
||||
else
|
||||
O1="$(sed -n '1p' "$LOCKEXEC/OWNER" 2>/dev/null)"
|
||||
O2="$(sed -n '2p' "$LOCKEXEC/OWNER" 2>/dev/null)"
|
||||
O3="$(sed -n '3p' "$LOCKEXEC/OWNER" 2>/dev/null)"
|
||||
if [ -n "$ME" ] && [ "$O1" = "$ME" ]; then
|
||||
echo " · 全局锁是**你自己**持有的($ME)—— 完工记得 --release-exec"
|
||||
else
|
||||
echo " ⚠ 全局锁被占用:$O1 $O2"
|
||||
[ -n "$O3" ] && echo " $O3"
|
||||
echo " → **同一时刻只允许一个执行会话**:要动「文档 / 代码 / 服务器」就先等它释放;"
|
||||
echo " ⛔ **不得人工删锁 / 不得接管**(R9):锁只能由持有者自己 --release-exec ——"
|
||||
echo " 抢不到 = 停手 + 报告用户;读 OWNER 仅用于「用户已点头、且用户自己撤锁之后」的续做"
|
||||
VERDICT=1; HARD+=("①c 别的会话持有全局执行锁($O1)→ 不允许并行执行")
|
||||
fi
|
||||
fi
|
||||
|
||||
# ── ①d 服务器侧操作锁(2026-09-12 新增,T04):平台高危操作的跨会话互斥 ────
|
||||
echo
|
||||
echo "【1d】服务器侧操作锁(/opt/dsh/state/.op-lock)"
|
||||
if [ "${SKIP_OPLOCK:-0}" = "1" ]; then
|
||||
echo " · 已跳过(SKIP_OPLOCK=1)"
|
||||
else
|
||||
OPLOCK_OUT="$(ssh -o BatchMode=yes -o ConnectTimeout=8 "${OP_LOCK_REMOTE:-bt-server}" \
|
||||
"if [ -d /opt/dsh/state/.op-lock ]; then ls -1 /opt/dsh/state/.op-lock 2>/dev/null | grep -v '^README\$' || true; else echo __NOLOCKDIR__; fi" 2>/dev/null)"; OPLOCK_RC=$?
|
||||
if [ "$OPLOCK_RC" -ne 0 ]; then
|
||||
echo " · 服务器不可达(离线)→ **只提示、不失败**;但要动线上时须先恢复可见性再动手"
|
||||
elif [ "$OPLOCK_OUT" = "__NOLOCKDIR__" ]; then
|
||||
echo " ⚠ 锁根目录不存在:/opt/dsh/state/.op-lock(T04 应已建立 → 需复查)"
|
||||
VERDICT=1; HARD+=("①d 服务器侧锁根目录缺失")
|
||||
elif [ -z "$OPLOCK_OUT" ]; then
|
||||
echo " ✓ 无平台操作锁(此刻没有会话在动线上)"
|
||||
else
|
||||
echo " ⚠ 有会话正在动线上:"
|
||||
for L in $OPLOCK_OUT; do
|
||||
echo " 🔴 $L"
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=8 "${OP_LOCK_REMOTE:-bt-server}" \
|
||||
"sed 's/^/ /' '/opt/dsh/state/.op-lock/$L/OWNER' 2>/dev/null" 2>/dev/null || true
|
||||
done
|
||||
echo " → 凡「重启 / drain / 改实例 env·quota / 批量铺插件 / 改 nginx·nft·证书」类操作,"
|
||||
echo " 开工前必须先占位(bash scripts/op-lock.sh claim <操作名> \"<影响面>\");"
|
||||
echo " 占位失败 = 有会话在动线上 → **停手**。"
|
||||
VERDICT=1; HARD+=("①d 有会话持有服务器侧操作锁($OPLOCK_OUT)")
|
||||
fi
|
||||
fi
|
||||
|
||||
# ── ② 越界改动(推送前硬判定)────────────────────────────
|
||||
echo
|
||||
echo "【2】未提交改动(含未跟踪)"
|
||||
CHANGED="$(git -c core.quotepath=false status --short 2>/dev/null | awk '{print $NF}' | grep -vE '^交接单/\.(doing-|exec-lock)' || true)"
|
||||
OUTSIDE=(); INSIDE=()
|
||||
if [ -z "$CHANGED" ]; then
|
||||
echo " ✓ 工作区干净"
|
||||
else
|
||||
while IFS= read -r f; do
|
||||
[ -z "$f" ] && continue
|
||||
if in_mine "$f"; then INSIDE+=("$f"); else OUTSIDE+=("$f"); fi
|
||||
done <<< "$CHANGED"
|
||||
echo " · 我声明的:${#INSIDE[@]} 个"
|
||||
if [ ${#OUTSIDE[@]} -eq 0 ]; then
|
||||
echo " ✓ 无越界改动(没有别人的半成品混在里面)"
|
||||
else
|
||||
echo " ⚠ 越界(别人的/我没声明):${#OUTSIDE[@]} 个"
|
||||
for f in "${OUTSIDE[@]:0:6}"; do echo " $f"; done
|
||||
[ ${#OUTSIDE[@]} -gt 6 ] && echo " …还有 $(( ${#OUTSIDE[@]} - 6 )) 个"
|
||||
echo " → **推送时只能 scp 自己声明的文件,切勿 'git add -A'**(本项不构成硬失败;"
|
||||
echo " 真正的推送硬判定在【4】——那里能精确看出「你正要推什么」)"
|
||||
fi
|
||||
fi
|
||||
|
||||
# ── ③ 热点提示(仅提示,不作判定)────────────────────────
|
||||
echo
|
||||
echo "【3】近 $WINDOW 分钟被改动的文件(**仅提示**:mtime 分不清谁改的)"
|
||||
HOT="$(find . -type f -mmin "-$WINDOW" \
|
||||
-not -path './.git/*' -not -path '*/node_modules/*' -not -name '*.bak*' \
|
||||
-not -path './交接单/.doing-*' -not -path './交接单/.exec-lock*' 2>/dev/null | sed 's|^\./||' | LC_ALL=C sort)"
|
||||
if [ -z "$HOT" ]; then
|
||||
echo " ✓ 无(这块是「冷」的)"
|
||||
else
|
||||
HIT=(); OTHER=""
|
||||
while IFS= read -r f; do
|
||||
[ -z "$f" ] && continue
|
||||
if in_mine "$f"; then HIT+=("$f"); else OTHER="$OTHER$f"$'\n'; fi
|
||||
done <<< "$HOT"
|
||||
if [ ${#HIT[@]} -gt 0 ]; then
|
||||
echo " ⚠ **我的目标文件近期被改动过**(可能是你自己,也可能是别人)→ 改前务必先读最新内容:"
|
||||
for f in "${HIT[@]}"; do echo " $f"; done
|
||||
else
|
||||
echo " · 我的目标文件均未被近期改动"
|
||||
fi
|
||||
echo " · 其它近期被改动的文件:$(printf '%s' "$OTHER" | grep -c . || true) 个(与你无关,仅供感知全库热度)"
|
||||
[ "$WINDOW" -gt 10 ] && echo " 想更锐利:GUARD_WINDOW=10 再跑一次"
|
||||
fi
|
||||
|
||||
# ── ④ 双端一致性(推送前硬判定)──────────────────────────
|
||||
echo
|
||||
echo "【4】双端一致性(docs-sync-check.sh)"
|
||||
if [ "${SKIP_SYNC:-0}" = "1" ]; then
|
||||
echo " (SKIP_SYNC=1,跳过)"
|
||||
elif [ -f scripts/docs-sync-check.sh ]; then
|
||||
# 全量对账耗时(2026-09-12 实测):优化前 **2m15s**(Git Bash 逐文件 spawn md5sum → 曾被当成"挂死"),
|
||||
# 优化后 **≈28s**。这里再加硬超时兜底,避免对账自身卡住时把整个 guard 拖死;急用时 `SKIP_SYNC=1` 跳过。
|
||||
if command -v timeout >/dev/null 2>&1; then
|
||||
SYNC="$(timeout 300 bash scripts/docs-sync-check.sh 2>/dev/null)"
|
||||
else
|
||||
SYNC="$(bash scripts/docs-sync-check.sh 2>/dev/null)"
|
||||
fi
|
||||
printf '%s\n' "$SYNC" | tail -6
|
||||
# 关键:对账结果里的「仅本地(待推送)」= 一次朴素推送**实际会推上去**的东西。
|
||||
# 其中只要有一个「不在我声明清单里」,就是幽灵文件信号(2026-09-12 事故:把已被对方归档的
|
||||
# T02 又推回服务器)→ 推送前硬失败。
|
||||
ONLYL="$(printf '%s\n' "$SYNC" | sed -n 's/.*仅本地(待推送) //p')"
|
||||
if [ -n "$ONLYL" ]; then
|
||||
GHOST=(); GOOD=0
|
||||
while IFS= read -r f; do
|
||||
[ -z "$f" ] && continue
|
||||
if in_mine "$f"; then GOOD=$((GOOD+1)); else GHOST+=("$f"); fi
|
||||
done <<< "$ONLYL"
|
||||
echo " · 仅本地(= 一次朴素推送会推上去的):$GOOD 个已声明 + ${#GHOST[@]} 个未声明"
|
||||
if [ ${#GHOST[@]} -gt 0 ]; then
|
||||
echo " ⚠ 未声明却「仅本地」——**幽灵文件风险**:"
|
||||
for f in "${GHOST[@]}"; do echo " $f"; done
|
||||
echo " → 它可能是别人**刚归档/移走**的文件(你今天就是这么推回去的)"
|
||||
if [ "$PUSH" = "1" ]; then
|
||||
VERDICT=1; HARD+=("④ 有 ${#GHOST[@]} 个「仅本地」文件不在你的推送清单里 → 停手核实后再推")
|
||||
else
|
||||
echo " (开工阶段仅提示;推送前请带 PUSH=1 复跑)"
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
else
|
||||
echo " (未找到 scripts/docs-sync-check.sh)"
|
||||
fi
|
||||
|
||||
# ── 结论 ────────────────────────────────────────────────
|
||||
echo
|
||||
echo "=================================================================="
|
||||
if [ "$VERDICT" -ne 0 ]; then
|
||||
echo "结论:**不可放行** ——"
|
||||
for r in "${HARD[@]}"; do echo " $r"; done
|
||||
elif [ "$STRICT" -eq 0 ]; then
|
||||
echo "结论:未声明改动清单 → 仅为信息输出,**不构成放行依据**(开工请带 MINE=\"...\")"
|
||||
echo " ⚠️ 且「无锁」不等于「可以开工」—— 开工前必须先 --claim-exec 抢锁(见上方【1c】)"
|
||||
echo " (2026-09-12 实证:把「无锁」读成「可以动手」,导致两个会话同时改库)"
|
||||
else
|
||||
echo "结论:未命中硬冲突 → 可以继续(仍须遵守:Edit 增量 / 改前先读 / 只推自己的文件)"
|
||||
fi
|
||||
exit "$VERDICT"
|
||||
@@ -0,0 +1,180 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
lock-guard-hook.py — 「无锁不许改库」的强制钩子(WorkBuddy PreToolUse / SessionStart)
|
||||
|
||||
为什么需要它
|
||||
────────────
|
||||
2026-09-12 实证:`handoff-guard.sh` 输出「无全局锁」被会话读成「环境干净,可以开工」
|
||||
(正确读法是「你快去抢锁」),结果**两个会话同时改了本库**。措辞已在 guard 与
|
||||
`交接单/README.md` 里补正,但**约定拦不住不看的人** —— 本脚本是强制层。
|
||||
|
||||
作用域(**刻意收窄:对其他项目零影响**)
|
||||
────────────
|
||||
仅当 `Write` / `Edit` 的目标路径落在下列根之内,才做锁判定;其余一律放行:
|
||||
· 文档库 <DSH_DOCS_ROOT>(默认 E:\\ProgramData\\AI技能\\aliyun-dsh-server\\dsh-server-docs)
|
||||
· 代码库 <DSH_CODE_REPO>(默认 D:\\github\\dsh_shenxian)
|
||||
⚠️ 例外:路径中含 `.workbuddy` 目录段的一律放行 —— 会话记忆 / 自动化工作数据属“运行态数据”,不是“仓库内容”。(2026-09-13 实测:「代码仓三方同步(DSH)」自动化写自己的 memory/*.md 时被本钩子 deny,属误伤)
|
||||
|
||||
判据
|
||||
────
|
||||
`<文档库根>/交接单/.exec-lock` 存在 = 有人持锁 → 放行(归属由台账 OWNER 承担);
|
||||
不存在 = **deny**,并把可直接粘贴的抢锁命令回给 Agent。
|
||||
|
||||
为什么**不拦 Bash**
|
||||
──────────────────
|
||||
① 抢锁命令本身必须能跑(否则把自己锁死 —— 拿不到锁就永远开不了工);
|
||||
② Bash 写文件是少数派、且破坏面可见(git status);
|
||||
③ 宁可留一个显式的安全阀,也不要一个可能导致死锁的强制层。
|
||||
|
||||
用法(settings.json 的 hooks 段,见同目录 README 或档案 73)
|
||||
"PreToolUse": [{ "matcher": "Write|Edit", "hooks": [{ "type": "command", "command": "<python> <此脚本>", "timeout": 10 }] }]
|
||||
"SessionStart":[{ "matcher": "startup", "hooks": [{ "type": "command", "command": "<python> <此脚本>", "timeout": 10 }] }]
|
||||
|
||||
退出码:始终 0(判定通过 JSON 输出表达);脚本自身异常也放行,绝不误伤。
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
|
||||
DOCS_ROOT = os.environ.get("DSH_DOCS_ROOT", r"D:\github\dsh_shenxian\dsh-server-docs")
|
||||
CODE_REPO = os.environ.get("DSH_CODE_REPO", r"D:\github\dsh_shenxian")
|
||||
LOCK_DIR = os.path.join(DOCS_ROOT, "交接单", ".exec-lock")
|
||||
OWNER_FILE = os.path.join(LOCK_DIR, "OWNER")
|
||||
|
||||
GUARD_CMD = 'bash scripts/handoff-guard.sh --claim-exec "<你的会话名>"'
|
||||
|
||||
|
||||
def norm(p: str) -> str:
|
||||
try:
|
||||
return os.path.normcase(os.path.normpath(p))
|
||||
except Exception:
|
||||
return p
|
||||
|
||||
|
||||
PROTECTED = [norm(DOCS_ROOT), norm(CODE_REPO)]
|
||||
|
||||
|
||||
def is_protected(path: str) -> bool:
|
||||
if not path:
|
||||
return False
|
||||
n = norm(path)
|
||||
# 会话记忆 / 自动化工作数据不是「仓库内容」:写它不该被本钩子拦(2026-09-13 新增)
|
||||
if ".workbuddy" in n.split(os.sep):
|
||||
return False
|
||||
for root in PROTECTED:
|
||||
if n == root or n.startswith(root + os.sep):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def lock_owner() -> str:
|
||||
try:
|
||||
with open(OWNER_FILE, encoding="utf-8") as f:
|
||||
return (f.readline() or "").strip() or "(未写 OWNER)"
|
||||
except Exception:
|
||||
return "(未写 OWNER)"
|
||||
|
||||
|
||||
def out(obj: dict) -> None:
|
||||
sys.stdout.write(json.dumps(obj, ensure_ascii=False))
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
LOG_FILE = os.environ.get(
|
||||
"DSH_LOCK_HOOK_LOG", os.path.join(os.path.dirname(DOCS_ROOT), ".workbuddy", "lock-hook.log")
|
||||
)
|
||||
|
||||
|
||||
def hook_log(event: str, detail: str) -> None:
|
||||
"""低频自证日志:只在 SessionStart 与 deny 时写一行 —— 用来回答「hook 到底有没有被触发」。
|
||||
|
||||
为什么需要:hook 配置是**启动时缓存**的,改完必须完全重启才加载;没有日志就只能靠猜。
|
||||
写入失败一律静默(hook 绝不能因为自己出问题而干扰工作)。
|
||||
"""
|
||||
try:
|
||||
os.makedirs(os.path.dirname(LOG_FILE), exist_ok=True)
|
||||
with open(LOG_FILE, "a", encoding="utf-8") as f:
|
||||
f.write(f"{time.strftime('%Y-%m-%d %H:%M:%S')}\t{event}\t{detail}\n")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def main() -> None:
|
||||
try:
|
||||
try:
|
||||
# ⚠️ 必须走 buffer 显式 UTF-8:本机环境设了 PYTHONUTF8=1,但只要有人给脚本加 `-E`
|
||||
# 就会被屏蔽 ⇒ sys.stdin 回退 cp936 ⇒ 含中文路径(`E:\ProgramData\AI技能\…`)的
|
||||
# payload 解析即炸。而本函数是 **fail-open**(读不懂就放行)⇒ 会**静默失效**:
|
||||
# 锁守卫不再拦人,却没有任何痕迹。2026-09-15 实测(同款坑已在 stop-dialog-guard.py 踩过)。
|
||||
raw = sys.stdin.buffer.read().decode("utf-8", "replace")
|
||||
except Exception:
|
||||
raw = sys.stdin.read()
|
||||
payload = json.loads(raw) if raw.strip() else {}
|
||||
except Exception as e:
|
||||
# 不留痕 = 失效无声(判不出"没被调用"与"被静默放行")⇒ 必须记一行
|
||||
hook_log("payload-unparsable", "%s: %s" % (type(e).__name__, str(e)[:80]))
|
||||
return # 读不懂 payload → 放行(fail-open 方向正确,但不能无声)
|
||||
|
||||
tool = payload.get("tool_name")
|
||||
event = payload.get("hook_event_name") or payload.get("hook_event") or ""
|
||||
|
||||
# ── SessionStart:只提示(该事件的输出只给用户看,不会进 Agent 上下文)──
|
||||
if event == "SessionStart" or (tool is None and "source" in payload):
|
||||
src = payload.get("source", "?")
|
||||
if os.path.isdir(LOCK_DIR):
|
||||
who = lock_owner()
|
||||
msg = f"🔐 全局执行锁【已被占用】:{who} —— 同一时刻只允许一个执行会话动「文档/代码/服务器」。"
|
||||
hook_log("SessionStart", f"已被占用 owner={who} source={src}")
|
||||
elif os.path.isdir(DOCS_ROOT):
|
||||
msg = "🔐 全局执行锁【空闲】—— 但「空闲」≠「可以开工」:动手前先抢锁 → " + GUARD_CMD
|
||||
hook_log("SessionStart", f"空闲 source={src}")
|
||||
else:
|
||||
msg = ""
|
||||
if msg:
|
||||
out({"systemMessage": msg, "suppressOutput": True})
|
||||
return
|
||||
|
||||
# ── PreToolUse:真正的强制点 ──
|
||||
if tool not in ("Write", "Edit", "MultiEdit", "NotebookEdit"):
|
||||
return
|
||||
ti = payload.get("tool_input") or {}
|
||||
fp = ti.get("file_path") or ti.get("path") or ti.get("notebook_path") or ""
|
||||
if not is_protected(fp):
|
||||
return
|
||||
if os.path.isdir(LOCK_DIR):
|
||||
return # 有人持锁 → 放行(归属由 OWNER / 台账承担)
|
||||
if not os.path.isdir(DOCS_ROOT):
|
||||
return # 库不在这台机器上 → 与本约定无关,放行
|
||||
|
||||
hook_log("PreToolUse-deny", f"{tool} {fp}")
|
||||
|
||||
reason = (
|
||||
"⛔ 被「锁机制」拦下:本仓库(文档库 / 代码库)当前**无人持全局执行锁**,"
|
||||
"不允许直接修改。\n\n"
|
||||
"「无锁」的正确读法 = **「你快去抢」**,不是「可以开工」"
|
||||
"(2026-09-12 实证:两个会话把「无锁」读成「环境干净」→ 同时改了本库)。\n\n"
|
||||
f"请先执行(Bash,不受本钩子限制):\n cd \"{DOCS_ROOT}\"\n {GUARD_CMD}\n\n"
|
||||
"抢到 = 开工许可;**抢不到 = 有会话在跑 → 停手**(输出会告诉你占用者与在做哪单)。\n"
|
||||
"若确实只需追加一行、且已确认无人在动,可由用户裁定后临时移除本钩子。"
|
||||
)
|
||||
out(
|
||||
{
|
||||
"hookSpecificOutput": {
|
||||
"hookEventName": "PreToolUse",
|
||||
"permissionDecision": "deny",
|
||||
"permissionDecisionReason": reason,
|
||||
},
|
||||
"systemMessage": "⛔ 已拦下一次无锁写入(本仓库要求先抢锁)",
|
||||
"suppressOutput": True,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
try:
|
||||
main()
|
||||
except Exception:
|
||||
pass # 任何异常都不阻断工作
|
||||
sys.exit(0)
|
||||
@@ -0,0 +1,65 @@
|
||||
#!/usr/bin/env python3
|
||||
"""nginx 访问日志慢请求/大响应分析(只读)
|
||||
用法: python3 nginx-slow.py <logfile> [host子串] [最近N行] [起始时间标记]
|
||||
"""
|
||||
import re
|
||||
import sys
|
||||
|
||||
log = sys.argv[1]
|
||||
host = sys.argv[2] if len(sys.argv) > 2 else ''
|
||||
tail_lines = int(sys.argv[3]) if len(sys.argv) > 3 else 20000
|
||||
since = sys.argv[4] if len(sys.argv) > 4 else ''
|
||||
|
||||
LINE = re.compile(
|
||||
r'^(?P<ip>\S+) \S+ \S+ \[(?P<time>[^\]]+)\] "(?P<req>[^"]*)" (?P<status>\d{3}) (?P<bytes>\d+)'
|
||||
r'(?: "[^"]*" "[^"]*")?(?P<rest>.*)$'
|
||||
)
|
||||
|
||||
rows = []
|
||||
with open(log, 'r', errors='replace') as fh:
|
||||
lines = fh.readlines()[-tail_lines:]
|
||||
|
||||
for line in lines:
|
||||
if host and host not in line:
|
||||
continue
|
||||
if since and since not in line:
|
||||
continue
|
||||
m = LINE.match(line)
|
||||
if not m:
|
||||
continue
|
||||
times = re.findall(r'(\d+\.\d+)', m.group('rest'))
|
||||
rt = float(times[-1]) if times else None
|
||||
rows.append(
|
||||
{
|
||||
'ip': m.group('ip'),
|
||||
'time': m.group('time'),
|
||||
'req': m.group('req'),
|
||||
'status': int(m.group('status')),
|
||||
'bytes': int(m.group('bytes')),
|
||||
'rt': rt,
|
||||
}
|
||||
)
|
||||
|
||||
print(f'匹配请求数: {len(rows)}')
|
||||
if not rows:
|
||||
sys.exit(0)
|
||||
|
||||
withrt = [r for r in rows if r['rt'] is not None]
|
||||
if withrt:
|
||||
ts = [r['rt'] for r in withrt]
|
||||
slow = [r for r in withrt if r['rt'] > 1]
|
||||
print(f'耗时: 平均 {sum(ts)/len(ts):.3f}s 中位 {sorted(ts)[len(ts)//2]:.3f}s 最大 {max(ts):.2f}s >1s 的 {len(slow)} 个')
|
||||
print('\n== 最慢 12 个 ==')
|
||||
for r in sorted(withrt, key=lambda x: -x['rt'])[:12]:
|
||||
print(f" {r['rt']:8.2f}s {r['status']} {r['bytes']:>9} B {r['time'][12:]} {r['req']}")
|
||||
|
||||
print('\n== 响应体最大 8 个 ==')
|
||||
for r in sorted(rows, key=lambda x: -x['bytes'])[:8]:
|
||||
rt = f"{r['rt']:.2f}s" if r['rt'] is not None else ' ? '
|
||||
print(f" {r['bytes']:>9} B {rt} {r['status']} {r['req']}")
|
||||
|
||||
from collections import Counter
|
||||
c = Counter(r['req'].split(' ')[1].split('?')[0] if ' ' in r['req'] else r['req'] for r in rows)
|
||||
print('\n== 路径 TOP 10 ==')
|
||||
for path, n in c.most_common(10):
|
||||
print(f' {n:5} {path}')
|
||||
@@ -0,0 +1,96 @@
|
||||
#!/usr/bin/env bash
|
||||
# op-lock.sh — 「平台高危操作锁」(服务器侧第二把锁)
|
||||
#
|
||||
# 为什么需要它:所有会话**共用一台服务器**,而下面这些操作跨会话没有任何互斥 ——
|
||||
# 重启 dshs / drain 实例 scope / 改实例 env·MemoryMax / 批量铺插件 /
|
||||
# 改 nginx vhost·证书·nft
|
||||
# 文档冲突靠 git status / mtime 能看出来,**服务器态变更看不出来**
|
||||
# (systemctl 不会告诉你 10 分钟前谁重启过)→ 只能靠显式锁。
|
||||
#
|
||||
# 与文档库锁的关系:`交接单/.doing-<单>` + `.exec-lock` 管「文档与单子」;本锁管「生产态」。
|
||||
# 顺序:**先抢全局执行锁 → 再占本锁**。
|
||||
#
|
||||
# 用法:
|
||||
# bash scripts/op-lock.sh claim <操作名> "<影响面/时长>" [占用者]
|
||||
# bash scripts/op-lock.sh release <操作名>
|
||||
# bash scripts/op-lock.sh status
|
||||
#
|
||||
# 判据:`mkdir` 原子 —— 成功 = 你拿到;报 File exists = 有别的会话正在动线上 → **停手**。
|
||||
# 退出码:0 = 成功/无人占用;1 = 已被占用(claim 失败)或环境错误。
|
||||
set -uo pipefail
|
||||
|
||||
REMOTE="${OP_LOCK_REMOTE:-bt-server}"
|
||||
LOCKROOT="${OP_LOCK_DIR:-/opt/dsh/state/.op-lock}"
|
||||
OP="${2:-}"
|
||||
WHO="${4:-${ME:-unknown-session}}"
|
||||
NOW="$(date '+%Y-%m-%d %H:%M')"
|
||||
|
||||
usage() {
|
||||
sed -n '3,20p' "$0" | sed 's/^# \{0,1\}//'
|
||||
exit 2
|
||||
}
|
||||
|
||||
case "${1:-}" in
|
||||
claim)
|
||||
[ -n "$OP" ] || usage
|
||||
IMPACT="${3:-(未声明影响面 —— 补上:谁会被断、断多久)}"
|
||||
if ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" "mkdir '$LOCKROOT/$OP' 2>/dev/null"; then
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" \
|
||||
"printf '占用者:%s\n开始时间:%s\n操作摘要:%s\n影响面:%s\n' '$WHO' '$NOW' '$OP' '$IMPACT' > '$LOCKROOT/$OP/OWNER'" || true
|
||||
echo "✅ 已占位:$LOCKROOT/$OP(占用者 $WHO,$NOW)"
|
||||
if [ "$WHO" = "unknown-session" ]; then
|
||||
echo " ⚠ 未提供会话名 → 已记作 unknown-session。后果有两个(2026-09-12 实测踩到):"
|
||||
echo " ① handoff-guard 的【1d】会把它当成**别人的锁**(连你自己都被挡住);"
|
||||
echo " ② release 的归属校验会拒绝你。请改用:"
|
||||
echo " ME=\"<会话名>\" bash scripts/op-lock.sh claim <操作名> \"<影响面>\""
|
||||
fi
|
||||
echo " 完工请:bash scripts/op-lock.sh release $OP"
|
||||
exit 0
|
||||
fi
|
||||
echo "❌ 占位失败 —— 该锁已存在(有会话正在动线上)→ 停手。" >&2
|
||||
bash "$0" status >&2 || true
|
||||
exit 1
|
||||
;;
|
||||
release)
|
||||
[ -n "$OP" ] || usage
|
||||
# 归属校验(R9,2026-09-12 补):锁只能由**持有者自己**释放 ——
|
||||
# 否则 release 就成了绕过 R9 的后门(AI 读一下 status 拿到操作名,就能把别人的锁 rm 掉)。
|
||||
RWHO="${4:-${ME:-unknown-session}}"
|
||||
ROWN="$(ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" \
|
||||
"sed -n '1p' '$LOCKROOT/$OP/OWNER' 2>/dev/null" 2>/dev/null | sed 's/^占用者://')"
|
||||
if [ -z "$ROWN" ]; then
|
||||
echo "❌ 拒绝释放:「$OP」没有 OWNER 记录(无法确认归属)。" >&2
|
||||
echo " 按 **R9**:锁的处置权只属于用户本人 —— 请报告用户,由用户处置。" >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ "$ROWN" != "$RWHO" ] && [ "${FORCE:-0}" != "1" ]; then
|
||||
echo "❌ 拒绝释放:「$OP」的占用者是「$ROWN」,而你是「$RWHO」。" >&2
|
||||
echo " 按 **R9**(禁止接管 / 删锁):抢不到锁 = 停手 + 报告用户;**锁的处置权只属于用户本人**。" >&2
|
||||
echo " (确需处置:由**用户本人**执行;或用户明确点头后用 FORCE=1,并在回报里写明理由。)" >&2
|
||||
exit 1
|
||||
fi
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" "rm -rf '$LOCKROOT/$OP'" || { echo "❌ 释放失败(ssh 不可达?)" >&2; exit 1; }
|
||||
echo "✅ 已释放:$OP(归属已核:$ROWN)"
|
||||
;;
|
||||
status)
|
||||
echo "—— 平台操作锁($REMOTE:$LOCKROOT)——"
|
||||
OUT="$(ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" \
|
||||
"ls -1 '$LOCKROOT' 2>/dev/null | grep -v '^README$'" 2>/dev/null)"
|
||||
if [ -z "$OUT" ]; then
|
||||
if ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" "test -d '$LOCKROOT'" 2>/dev/null; then
|
||||
echo " (无人占用 ✓)"
|
||||
else
|
||||
echo " ⚠ 锁根目录不存在:$LOCKROOT(需先建立)" >&2
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
for L in $OUT; do
|
||||
echo " 🔴 $L"
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" \
|
||||
"sed 's/^/ /' '$LOCKROOT/$L/OWNER' 2>/dev/null" || true
|
||||
done
|
||||
exit 1
|
||||
fi
|
||||
;;
|
||||
*) usage ;;
|
||||
esac
|
||||
@@ -0,0 +1,133 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* 插件兼容性预检 v2 —— PoC(只读)
|
||||
*
|
||||
* A. 平台内置 @deepseek-ai/* 的「真实导出符号」清单(判据来源)
|
||||
* B. 候选池每个 tgz:解包 → 依赖声明 + bundle 内 import 语句
|
||||
* C. 逐条比对 → 判定「兼容 / 不兼容 / 需装后复核」
|
||||
*
|
||||
* 用法:node plugin-compat-check.mjs
|
||||
*/
|
||||
import { readFileSync, existsSync, readdirSync, mkdtempSync, rmSync, statSync } from 'node:fs'
|
||||
import { join } from 'node:path'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { execFileSync } from 'node:child_process'
|
||||
|
||||
const DSH_ROOT = process.env.DSH_ROOT || '/usr/local/lib/node_modules/@deepseek-ai/dsh'
|
||||
const NESTED = join(DSH_ROOT, 'node_modules', '@deepseek-ai')
|
||||
const POOL = process.env.POOL || '/var/lib/dshs/business-plugins'
|
||||
|
||||
/* ---------- A. 平台 exports 真值 ---------- */
|
||||
const platform = new Map() // pkgName -> { version, keys:Set }
|
||||
for (const name of readdirSync(NESTED)) {
|
||||
if (name.startsWith('.')) continue
|
||||
const pj = join(NESTED, name, 'package.json')
|
||||
if (!existsSync(pj)) continue
|
||||
const meta = JSON.parse(readFileSync(pj, 'utf8'))
|
||||
let entry = null
|
||||
for (const cand of ['lib/index.js', 'dist/index.js', 'index.js']) {
|
||||
const f = join(NESTED, name, cand)
|
||||
if (existsSync(f)) { entry = f; break }
|
||||
}
|
||||
let keys = null
|
||||
if (entry) { try { keys = new Set(Object.keys(await import(entry))) } catch { /* ignore */ } }
|
||||
platform.set(meta.name, { version: meta.version, keys })
|
||||
}
|
||||
console.log(`[A] 平台内置 @deepseek-ai 包 ${platform.size} 个(其中有导出符号清单的 ${[...platform.values()].filter((v) => v.keys).length} 个)`)
|
||||
|
||||
const llm = platform.get('@deepseek-ai/dsh-llm')
|
||||
console.log(` · dsh-llm@${llm?.version} 导出 ${llm?.keys?.size ?? 0} 个符号;含 assertNever? ${llm?.keys?.has('assertNever') ? '✅' : '❌(= anysearch 崩溃判据成立)'}`)
|
||||
|
||||
/* ---------- B+C. 逐个 tgz 预检 ---------- */
|
||||
const IMPORT_RE = /(?:import|export)\s*(?:\*\s*as\s*[\w$]+|\{([^}]*)\})\s*from\s*["'](@deepseek-ai\/[^"'/]+(?:\/[^"']*)?)["']/g
|
||||
|
||||
function scanFileImports(file, base) {
|
||||
const text = readFileSync(file, 'utf8')
|
||||
const hits = []
|
||||
let m
|
||||
IMPORT_RE.lastIndex = 0
|
||||
while ((m = IMPORT_RE.exec(text))) {
|
||||
const spec = m[1] || ''
|
||||
// 具名导入:取每个符号(去掉 as 别名)
|
||||
const named = spec.split(',').map((s) => s.trim().split(/\s+as\s+/)[0].trim()).filter((s) => s && !s.startsWith('type '))
|
||||
hits.push({ pkg: m[2], named, file: file.slice(base.length + 1) })
|
||||
}
|
||||
return hits
|
||||
}
|
||||
|
||||
function walkJs(dir, base, out = []) {
|
||||
for (const e of readdirSync(dir, { withFileTypes: true })) {
|
||||
const p = join(dir, e.name)
|
||||
if (e.isDirectory()) { if (e.name !== 'node_modules') walkJs(p, base, out) }
|
||||
else if (/\.(js|mjs|cjs)$/.test(e.name)) out.push(p)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
const tgzs = existsSync(POOL) ? readdirSync(POOL).filter((f) => f.endsWith('.tgz')) : []
|
||||
console.log(`\n[B] 候选池 tgz ${tgzs.length} 个:${tgzs.join(', ') || '(无)'}\n`)
|
||||
|
||||
for (const tgz of tgzs) {
|
||||
const full = join(POOL, tgz)
|
||||
const tmp = mkdtempSync(join(tmpdir(), 'compat-'))
|
||||
console.log('━'.repeat(72))
|
||||
console.log(`📦 ${tgz} (${(statSync(full).size / 1024).toFixed(0)} KB)`)
|
||||
try {
|
||||
execFileSync('tar', ['-xzf', full, '-C', tmp], { stdio: 'pipe' })
|
||||
} catch (e) {
|
||||
console.log(` ✗ 解包失败: ${String(e.message).slice(0, 100)}`)
|
||||
rmSync(tmp, { recursive: true, force: true })
|
||||
continue
|
||||
}
|
||||
|
||||
// package.json
|
||||
const pkgRoot = join(tmp, 'package')
|
||||
const pj = join(pkgRoot, 'package.json')
|
||||
let name = tgz, version = '?'
|
||||
const deps = {}
|
||||
if (existsSync(pj)) {
|
||||
const meta = JSON.parse(readFileSync(pj, 'utf8'))
|
||||
name = meta.name || name
|
||||
version = meta.version || '?'
|
||||
Object.assign(deps, meta.dependencies || {}, meta.peerDependencies || {})
|
||||
}
|
||||
console.log(` 包名 ${name}@${version}`)
|
||||
|
||||
// 依赖里的 @deepseek-ai
|
||||
const dsDeps = Object.entries(deps).filter(([k]) => k.startsWith('@deepseek-ai/'))
|
||||
if (dsDeps.length) {
|
||||
console.log(` 声明的 @deepseek-ai 依赖 ${dsDeps.length} 个:`)
|
||||
for (const [d, rng] of dsDeps) {
|
||||
const p = platform.get(d)
|
||||
const verdict = !p ? '⚠️ 平台无此包' : (p.version === rng.replace(/^[\^~=]/, '') ? '✅ 版本一致' : `⚠️ 平台为 ${p.version},插件要求 ${rng}`)
|
||||
console.log(` ${d} ${rng} → ${verdict}`)
|
||||
}
|
||||
} else {
|
||||
console.log(' 未声明任何 @deepseek-ai 依赖')
|
||||
}
|
||||
|
||||
// bundle 内 import
|
||||
const files = walkJs(pkgRoot, pkgRoot)
|
||||
const hits = files.flatMap((f) => scanFileImports(f, pkgRoot))
|
||||
console.log(` 扫描 ${files.length} 个 js 文件,命中 ${hits.length} 条 @deepseek-ai 导入`)
|
||||
|
||||
const problems = []
|
||||
for (const h of hits) {
|
||||
const p = platform.get(h.pkg)
|
||||
if (!p || !p.keys) continue
|
||||
const missing = h.named.filter((n) => n && !p.keys.has(n))
|
||||
if (missing.length) problems.push({ ...h, missing })
|
||||
}
|
||||
if (problems.length) {
|
||||
console.log(' 🔴 不兼容命中:')
|
||||
for (const pr of problems) console.log(` ${pr.pkg} 缺少导出 ${pr.missing.join(', ')} ← ${pr.file}`)
|
||||
} else if (hits.length) {
|
||||
console.log(' 🟢 静态可判部分:全部符号在平台包中存在')
|
||||
} else {
|
||||
console.log(' ⚪ 静态无法判定(插件未直接 import 平台包 → 需「装后扫 node_modules」复核)')
|
||||
}
|
||||
rmSync(tmp, { recursive: true, force: true })
|
||||
}
|
||||
console.log('\n' + '━'.repeat(72))
|
||||
console.log('说明:静态判定只覆盖「插件自身 bundle 里的 import」。若插件把平台 API 的调用藏在依赖包内')
|
||||
console.log(' (如 anysearch 的 @deepseek-ai/dsh-tool-web),必须做「装后扫 profile node_modules」才能抓到。')
|
||||
@@ -0,0 +1,111 @@
|
||||
/**
|
||||
* 会话取证 · 画像:元信息/权限档位/用户消息/12 类错误模式扫描/工具清单(不输出正文以外内容)
|
||||
* guest 会话问题分析(只读)
|
||||
* 用法:node analyze-guest.mjs <session.jsonl.zstd> [--full]
|
||||
*/
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { zstdDecompressSync } from 'node:zlib'
|
||||
|
||||
const file = process.argv[2]
|
||||
const raw = readFileSync(file)
|
||||
const MAGIC = Buffer.from([0x28, 0xb5, 0x2f, 0xfd])
|
||||
const offs = []
|
||||
for (let i = 0; i + 4 <= raw.length; i++) if (raw.compare(MAGIC, 0, 4, i, i + 4) === 0) offs.push(i)
|
||||
let text = ''
|
||||
const frames = offs.length ? offs : [0]
|
||||
for (let f = 0; f < frames.length; f++) {
|
||||
const s = frames[f], e = f + 1 < frames.length ? frames[f + 1] : raw.length
|
||||
try { text += zstdDecompressSync(raw.subarray(s, e)).toString('utf8') } catch {}
|
||||
}
|
||||
const evs = []
|
||||
for (const l of text.split('\n')) {
|
||||
if (!l.trim()) continue
|
||||
try { evs.push(JSON.parse(l)) } catch {}
|
||||
}
|
||||
const ts = (ms) => new Date(Number(ms)).toLocaleString('zh-CN', { hour12: false, timeZone: 'Asia/Shanghai' })
|
||||
|
||||
// ── 1) 元信息 ─────────────────────────────────────────
|
||||
const meta = evs.find((e) => e.type === 'session')
|
||||
console.log('=== 元信息 ===')
|
||||
console.log(' id:', meta?.id, '| cwd:', meta?.cwd)
|
||||
console.log(' 创建:', ts(meta?.createdAt), '| 事件数:', evs.length, '| 解压:', text.length, '字符')
|
||||
const presets = evs.filter((e) => e.type === 'permission/preset').map((e) => e.data?.preset)
|
||||
const modes = evs.filter((e) => e.type === 'sandbox/mode').map((e) => e.data?.mode)
|
||||
const pols = evs.filter((e) => e.type === 'approval/policy').map((e) => e.data?.policy)
|
||||
console.log(' permission preset:', presets.join(','), '| sandbox mode:', modes.join(','), '| approval:', pols.join(','))
|
||||
const turns = evs.filter((e) => e.type === 'turn/start').map((e) => Number(e.data?.turn))
|
||||
console.log(' turn 数:', turns.length, '| 最后事件:', ts(evs[evs.length - 1]?.time ?? evs[evs.length - 1]?.time0))
|
||||
|
||||
// ── 2) 类型直方图(正确格式)──────────────────────────
|
||||
const hist = new Map()
|
||||
for (const e of evs) hist.set(e.type, (hist.get(e.type) ?? 0) + 1)
|
||||
console.log('\n=== 事件类型(Top 25)===')
|
||||
for (const [t, n] of [...hist].sort((a, b) => b[1] - a[1]).slice(0, 25)) console.log(' ' + String(n).padStart(5) + ' ' + t)
|
||||
|
||||
// ── 3) 用户消息 ───────────────────────────────────────
|
||||
const textOf = (d) => {
|
||||
if (typeof d === 'string') return d
|
||||
if (Array.isArray(d)) return d.map(textOf).join('')
|
||||
if (d && typeof d === 'object') {
|
||||
if (typeof d.text === 'string') return d.text
|
||||
if (d.content) return textOf(d.content)
|
||||
return ''
|
||||
}
|
||||
return ''
|
||||
}
|
||||
console.log('\n=== 用户消息(共 ' + evs.filter((e) => e.type === 'user/message').length + ' 条)===')
|
||||
for (const e of evs.filter((e) => e.type === 'user/message')) {
|
||||
const t = textOf(e.data?.content).replace(/\s+/g, ' ').slice(0, 260)
|
||||
console.log(` [${ts(e.time)}] ${t}`)
|
||||
}
|
||||
|
||||
// ── 4) 错误/拒绝模式扫描 ──────────────────────────────
|
||||
const PATTERNS = [
|
||||
['沙箱后端不可用', /no sandbox backend is usable/],
|
||||
['拒绝无沙箱运行', /refusing to run the command unconfined/],
|
||||
['权限被拒', /(?:\bdenied\b|Permission denied|EACCES|not permitted)/],
|
||||
['文件不存在', /\bENOENT\b|No such file or directory|command not found/i],
|
||||
['只读文件系统', /Read-only file system|EROFS/],
|
||||
['超时', /\btimed? ?out\b|ETIMEDOUT|timeout/i],
|
||||
['解析失败/DNS', /Could not resolve host|EAI_AGAIN|getaddrinfo/],
|
||||
['401/鉴权', /401|unauthorized|authentication required/i],
|
||||
['崩溃/熔断', /crash-restart|circuit-open/],
|
||||
['Python 缺失', /python3?: (?:command )?not found|python3 不存在/],
|
||||
['工具失败关键词', /tool (?:call )?failed|Tool failed|工具执行失败/],
|
||||
['中文报错词', /报错|失败|不可用|无法执行|被拒绝|不允许/],
|
||||
]
|
||||
console.log('\n=== 错误/限制模式扫描 ===')
|
||||
for (const [name, re] of PATTERNS) {
|
||||
const hits = []
|
||||
for (const e of evs) {
|
||||
const s = JSON.stringify(e)
|
||||
if (re.test(s)) hits.push(e)
|
||||
}
|
||||
if (!hits.length) { console.log(` ${name}: 0`); continue }
|
||||
const kinds = new Map()
|
||||
for (const h of hits) kinds.set(h.type, (kinds.get(h.type) ?? 0) + 1)
|
||||
console.log(` ${name}: ${hits.length} 次 → ${[...kinds].map(([k, v]) => k + '×' + v).join(', ')}`)
|
||||
for (const h of hits.slice(0, 2)) {
|
||||
const m = JSON.stringify(h).match(re)
|
||||
const i = JSON.stringify(h).indexOf(m[0])
|
||||
console.log(` · [${ts(h.time ?? h.time0)}] …` + JSON.stringify(h).slice(Math.max(0, i - 130), i + 130).replace(/\\n/g, ' '))
|
||||
}
|
||||
}
|
||||
|
||||
// ── 5) 工具调用清单(tool 相关事件的 name/command)────
|
||||
console.log('\n=== 工具调用(含 tool 的事件,取 name/command/tool)===')
|
||||
const toolEvs = evs.filter((e) => /tool/i.test(e.type))
|
||||
const names = new Map()
|
||||
const fails = []
|
||||
for (const e of toolEvs) {
|
||||
const s = JSON.stringify(e)
|
||||
const nm = (s.match(/"name":"([^"]{2,40})"/) ?? [])[1] ?? (s.match(/"tool":"([^"]{2,40})"/) ?? [])[1] ?? '(?)'
|
||||
names.set(nm, (names.get(nm) ?? 0) + 1)
|
||||
if (/"(?:isError|ok|error|status)":(?:true|false|"[^"]*")/.test(s)) {
|
||||
if (/"isError":true|"ok":false|"status":"(?:error|failed)"/.test(s)) fails.push(e)
|
||||
}
|
||||
}
|
||||
console.log(' 工具事件类型数:', toolEvs.length)
|
||||
for (const [n, c] of [...names].sort((a, b) => b[1] - a[1]).slice(0, 20)) console.log(' ' + String(c).padStart(4) + ' ' + n)
|
||||
console.log(' 标记失败的工具事件:', fails.length)
|
||||
for (const f of fails.slice(0, 5)) console.log(' · [' + ts(f.time ?? f.time0) + '] ' + JSON.stringify(f).slice(0, 300))
|
||||
@@ -0,0 +1,59 @@
|
||||
/**
|
||||
* 会话取证 · 逐次工具结果分类(沙箱拒绝 / 文件策略拒绝 / 命令错误 / OK)+ 审批事件时间线
|
||||
* 精确分类:每次 bash 调用的结果性质(沙箱拒绝 / 命令错误 / 成功)
|
||||
* 用法:node classify-bash.mjs <session.jsonl.zstd>
|
||||
*/
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { zstdDecompressSync } from 'node:zlib'
|
||||
|
||||
const raw = readFileSync(process.argv[2])
|
||||
const MAGIC = Buffer.from([0x28, 0xb5, 0x2f, 0xfd])
|
||||
const offs = []
|
||||
for (let i = 0; i + 4 <= raw.length; i++) if (raw.compare(MAGIC, 0, 4, i, i + 4) === 0) offs.push(i)
|
||||
let text = ''
|
||||
for (let f = 0; f < (offs.length ? offs : [0]).length; f++) {
|
||||
const s = offs[f], e = f + 1 < offs.length ? offs[f + 1] : raw.length
|
||||
try { text += zstdDecompressSync(raw.subarray(s, e)).toString('utf8') } catch {}
|
||||
}
|
||||
const evs = []
|
||||
for (const l of text.split('\n')) { if (!l.trim()) continue; try { evs.push(JSON.parse(l)) } catch {} }
|
||||
const t = (ms) => new Date(Number(ms)).toLocaleString('zh-CN', { hour12: false, timeZone: 'Asia/Shanghai' })
|
||||
|
||||
// tool/call:id → {name, args(截断)}
|
||||
const callInfo = new Map()
|
||||
for (const e of evs) {
|
||||
if (e.type !== 'tool/call') continue
|
||||
const s = JSON.stringify(e.data ?? {})
|
||||
const id = (s.match(/"callId":"([^"]+)"/) ?? [])[1]
|
||||
const name = (s.match(/"name":"([^"]+)"/) ?? [])[1]
|
||||
if (!id) continue
|
||||
const cmd = (s.match(/"command":"((?:[^"\\]|\\.)*)"/) ?? [])[1] ?? ''
|
||||
callInfo.set(id, { name, cmd: cmd.replace(/\\n/g, ' ').slice(0, 90) })
|
||||
}
|
||||
// tool/result:id → 文本
|
||||
const rows = []
|
||||
for (const e of evs) {
|
||||
if (e.type !== 'tool/result') continue
|
||||
const s = JSON.stringify(e.data ?? {})
|
||||
const id = (s.match(/"toolCallId":"([^"]+)"/) ?? [])[1]
|
||||
const info = callInfo.get(id) ?? { name: '?', cmd: '' }
|
||||
const txt = (e.data?.message?.content ?? []).map((c) => (c.content ?? []).map((x) => x.text ?? '').join('')).join('\n')
|
||||
const flat = txt.replace(/\s+/g, ' ').trim()
|
||||
let kind = 'OK'
|
||||
if (/sandbox mode .* is requested but no sandbox backend/.test(flat)) kind = '沙箱拒绝'
|
||||
else if (/file access denied under/.test(flat)) kind = '文件策略拒绝'
|
||||
else if (/^Error:|^error:|"error"|Command failed|exit code [1-9]/.test(flat)) kind = '命令错误'
|
||||
rows.push({ time: e.time, name: info.name, cmd: info.cmd, kind, head: flat.slice(0, 150) })
|
||||
}
|
||||
console.log('=== 逐次 tool/result 分类(共 %d)===', rows.length)
|
||||
for (const r of rows) {
|
||||
if (r.name !== 'bash' && r.kind === 'OK') continue
|
||||
console.log(' %s %-6s %-10s %s', t(r.time), r.name, r.kind, (r.cmd || r.head).slice(0, 110))
|
||||
}
|
||||
const byKind = {}
|
||||
for (const r of rows) byKind[r.kind] = (byKind[r.kind] ?? 0) + 1
|
||||
console.log('\n汇总:', JSON.stringify(byKind))
|
||||
const sb = rows.filter((r) => r.kind === '沙箱拒绝')
|
||||
console.log('\n沙箱拒绝时间线: %s → %s(共 %d 次)', t(sb[0]?.time), t(sb[sb.length - 1]?.time), sb.length)
|
||||
const after = sb.filter((r) => r.time > 1789116 * 1000).length
|
||||
console.log(' 其中 22:16 切档位之后:', after, '次')
|
||||
@@ -0,0 +1,29 @@
|
||||
/** 批量列出会话档位:id / 创建时间 / preset / sandbox mode / approval / turn 数 */
|
||||
* 会话取证 · 批量对比各会话 preset/sandbox/approval(定位"老会话档位未跟随平台默认"的关键工具)
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { zstdDecompressSync } from 'node:zlib'
|
||||
|
||||
const MAGIC = Buffer.from([0x28, 0xb5, 0x2f, 0xfd])
|
||||
const t = (ms) => new Date(Number(ms)).toLocaleString('zh-CN', { hour12: false, timeZone: 'Asia/Shanghai' })
|
||||
|
||||
for (const file of process.argv.slice(2)) {
|
||||
const raw = readFileSync(file)
|
||||
const offs = []
|
||||
for (let i = 0; i + 4 <= raw.length; i++) if (raw.compare(MAGIC, 0, 4, i, i + 4) === 0) offs.push(i)
|
||||
let text = ''
|
||||
for (let f = 0; f < (offs.length ? offs : [0]).length; f++) {
|
||||
const s = offs[f], e = f + 1 < offs.length ? offs[f + 1] : raw.length
|
||||
try { text += zstdDecompressSync(raw.subarray(s, e)).toString('utf8') } catch {}
|
||||
}
|
||||
const evs = []
|
||||
for (const l of text.split('\n')) { if (!l.trim()) continue; try { evs.push(JSON.parse(l)) } catch {} }
|
||||
const meta = evs.find((e) => e.type === 'session')
|
||||
const uniq = (ty, f) => [...new Set(evs.filter((e) => e.type === ty).map(f))].join(',')
|
||||
const tl = text.match(/"text":"([^"]{0,60})"/g)?.slice(0, 1)?.join('') ?? ''
|
||||
console.log(
|
||||
'%s | 创建 %s | preset=%s | sandbox=%s | approval=%s | turns=%d | %d 事件',
|
||||
(meta?.id ?? '?').slice(-12),
|
||||
t(meta?.createdAt), uniq('permission/preset', (e) => e.data?.preset), uniq('sandbox/mode', (e) => e.data?.mode),
|
||||
uniq('approval/policy', (e) => e.data?.policy), evs.filter((e) => e.type === 'turn/start').length, evs.length,
|
||||
)
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
/**
|
||||
* 会话取证 · schema 探测:解压多帧 zstd → 事件类型直方图 + 每类样本键结构(先摸清格式)
|
||||
* 会话 schema 探测(只读):输出事件类型直方图 + 少量样本键结构
|
||||
* 用法:node probe-schema.mjs <session.jsonl.zstd>
|
||||
*/
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { zstdDecompressSync } from 'node:zlib'
|
||||
|
||||
const raw = readFileSync(process.argv[2])
|
||||
const MAGIC = Buffer.from([0x28, 0xb5, 0x2f, 0xfd])
|
||||
const offs = []
|
||||
for (let i = 0; i + 4 <= raw.length; i++) if (raw.compare(MAGIC, 0, 4, i, i + 4) === 0) offs.push(i)
|
||||
let text = ''
|
||||
const frames = offs.length ? offs : [0]
|
||||
for (let f = 0; f < frames.length; f++) {
|
||||
const s = frames[f], e = f + 1 < frames.length ? frames[f + 1] : raw.length
|
||||
try { text += zstdDecompressSync(raw.subarray(s, e)).toString('utf8') } catch {}
|
||||
}
|
||||
const lines = text.split('\n').filter((l) => l.trim())
|
||||
console.log('解压后 %d 行 / %d 字符 / zstd 帧 %d', lines.length, text.length, frames.length)
|
||||
|
||||
const types = new Map()
|
||||
const samples = new Map()
|
||||
for (const l of lines) {
|
||||
let o
|
||||
try { o = JSON.parse(l) } catch { continue }
|
||||
const t = o.type ?? o.event ?? o.kind ?? '(no-type)'
|
||||
types.set(t, (types.get(t) ?? 0) + 1)
|
||||
if (!samples.has(t)) samples.set(t, o)
|
||||
}
|
||||
console.log('\n=== 事件类型直方图 ===')
|
||||
for (const [t, n] of [...types].sort((a, b) => b[1] - a[1])) console.log(' %-28s %d', t, n)
|
||||
|
||||
console.log('\n=== 每类样本(键 + 截断值) ===')
|
||||
for (const [t, o] of samples) {
|
||||
console.log('--- %s ---', t)
|
||||
const dump = (v, depth = 0) => {
|
||||
if (v === null) return 'null'
|
||||
if (Array.isArray(v)) return `[${v.length} items${v.length ? ': ' + dump(v[0], depth + 1) : ''}]`
|
||||
if (typeof v === 'object') {
|
||||
if (depth > 1) return '{…}'
|
||||
return '{' + Object.entries(v).slice(0, 10).map(([k, x]) => `${k}: ${dump(x, depth + 1)}`).join(', ') + '}'
|
||||
}
|
||||
const s = String(v)
|
||||
return JSON.stringify(s.length > 90 ? s.slice(0, 90) + '…' : s)
|
||||
}
|
||||
console.log(' ' + dump(o).slice(0, 1200))
|
||||
}
|
||||
@@ -0,0 +1,334 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""stop-dialog-guard.py —— 「禁止用征询句收尾」的 Stop 钩子(WorkBuddy / CodeBuddy)
|
||||
|
||||
为什么需要它
|
||||
────────────
|
||||
2026-09-15 实测:本工作区日志里 `tool=AskUserQuestion` 调用数 = 09-12: 43 / 09-13: 3 / **09-14: 0 / 09-15: 0**
|
||||
⇒ 既有「提问闸门」(PreToolUse + matcher ^AskUserQuestion$)**拦的是几乎不走的工具面**,
|
||||
而真实的上抛("要我接着做吗 / 请确认 / 说一声即可")发生在**正文里** —— 没有任何机制覆盖。
|
||||
|
||||
本钩子 = 覆盖那条面:**每次回复结束时**读 transcript 的**最后一条 assistant 文本**,
|
||||
只扫**收尾段**(最后两行有效内容)里的征询句式;命中 → 返回 `{"continue": false, "reason": …}`
|
||||
让 Agent **继续一轮并自我纠正**(把该自己做的事做掉,或改写成「需要你拍板」一节)。
|
||||
|
||||
安全设计(都不许省)
|
||||
────────────────────
|
||||
1. **自作用域**:只在 `transcript_path` 落在本工作区(`aliyun-dsh-server`)时生效,其他项目一律放行。
|
||||
2. **防死循环**:输入里的 `stop_hook_active == true` 时**不再阻拦**(官方语义:本次停止已由 stop hook 触发过)。
|
||||
3. **绝不添乱**:任何异常 → 静默放行(exit 0)。判定只在**收尾段**做,避免正文引用规则时误伤。
|
||||
4. **性能**:只读转录**末尾 256 KB**(实测整库最大转录 31.9 MB、全文读 14 MB ≈ 832 ms ⇒ 不可接受),只看 stdin + 该文件。
|
||||
5. **防跑飞**:同一会话 600 秒内最多拦**一次**。
|
||||
6. **急停双闸**(无需卸载/重启):env `DSH_STOP_GUARD_OFF=1`,或新建 `<工作区>/.workbuddy/stop-guard.disabled`。
|
||||
7. **低频自证日志**:命中才写一行(`<工作区>/.workbuddy/stop-dialog-guard.log`),用来回答"到底有没有触发"。
|
||||
|
||||
退出码:始终 0;决策通过 stdout 的 JSON 表达。
|
||||
安装(settings.json 的 hooks 段 · 见档案 73 / 99):
|
||||
"Stop": [{ "hooks": [{ "type": "command",
|
||||
"command": "\"<python>\" \"<此脚本>\"", "timeout": 10 }] }]
|
||||
⚠️ hooks 是**应用启动时快照** ⇒ 装完必须**完全重启 WorkBuddy**;桌面版无 /hooks 面板,等效。
|
||||
⛔ **安装命令不要给本脚本加 `-E`(或任何会屏蔽 PYTHONUTF8 的 flag)**:本机环境本就设了
|
||||
`PYTHONUTF8=1` / `PYTHONIOENCODING=utf-8`,而 `-E` 会把它们**全部忽略** ⇒ stdin 回退 **cp936** ⇒
|
||||
含中文的 payload 解码即炸。本脚本现已改为走 `buffer` 显式 UTF-8(读写都加固),但**不要靠加固兜底**,
|
||||
装的时候也别再引入新雷。(2026-09-15 实测:`-S -E` 曾让本钩子"看起来从未被调用"整整一天。)
|
||||
"""
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
|
||||
SCOPE = 'aliyun-dsh-server' # 只对本工作区生效
|
||||
LOG_REL = os.path.join('.workbuddy', 'stop-dialog-guard.log')
|
||||
# 兜底工作区:用于"每次调用必留痕"(万一宿主没给 cwd、也没设 CODEBUDDY_PROJECT_DIR)
|
||||
# 本脚本位于 <工作区>/dsh-server-docs/scripts/ ⇒ 往上三级即工作区
|
||||
WS_FALLBACK = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
# 只扫「收尾段」:出现这些就是"把该自己做的事甩回给用户"
|
||||
PATTERNS = [
|
||||
r'要我(再|接着|继续|现在)?[^。!?\n]{0,20}吗',
|
||||
r'要不要我[^。!?\n]{0,20}',
|
||||
r'是否要我[^。!?\n]{0,20}',
|
||||
r'需要我[^。!?\n]{0,20}吗',
|
||||
r'请确认[^。!?\n]{0,16}',
|
||||
r'要不要(继续|现在做|我来)[^。!?\n]{0,20}',
|
||||
r'是否(继续|需要我)[^。!?\n]{0,20}',
|
||||
r'说一声即可',
|
||||
r'你看(怎么办|怎么弄|要不要)',
|
||||
r'你(决定|拍板)一下',
|
||||
]
|
||||
RE_BAN = re.compile('|'.join(PATTERNS))
|
||||
|
||||
REASON = (
|
||||
'⛔ 收尾句是**征询句**,但按本平台规则(`CODEBUDDY.md §1`「回话前自检」+ `dsh-feature-first §5.3` 铁律 3)'
|
||||
'先重判三问:① 命中**真门禁**吗(不可逆破坏性操作 / 边界外六类)?没命中 → **删掉这句,自己做完,改成陈述句**("我接着做 X");'
|
||||
'② 是不是在把已经定下来的事再问一遍?是 → 删;③ 这件事用户有客观可判的优劣吗?没有 → 才允许问,且**一轮只问这一句**,'
|
||||
'并写进 `dsh-feature-first §5.1` 结论骨架的「**需要你拍板**」一节 —— 该节必须是**整条回复的最后一节**、'
|
||||
'且**逐条编号**(有序段落)(2026-09-15 用户明令:「放在最后,别隐藏在回复内容中间」「按照有序段落展示」),'
|
||||
'用**陈述句**列"各候选的**优点 / 缺点** + 我的倾向",不要用征询句。'
|
||||
'⚠️ 上抛前先过**取舍筛** —— 某个候选**只有优点 / 只有缺点** ⇒ **自己拍掉、不要问**;'
|
||||
'且候选**竖排成段**(A / B / C 各占一行),⛔ 不横排、不做成表格的列(2026-09-15 用户明令)。'
|
||||
)
|
||||
|
||||
|
||||
TAIL_BYTES = 262144
|
||||
MAX_BYTES = 4194304 # 扩窗上限 4 MB(防"巨行"时无限读) # 只读末尾 256 KB(实测:整库最大转录 31.9 MB;全文读 14 MB = 832 ms/轮,不可接受)
|
||||
|
||||
|
||||
def transcribe_last_assistant(path):
|
||||
"""返回最后一条 assistant 文本(**从尾部向后分块读**;读不到返回 '')。
|
||||
|
||||
⚠️ 为什么不是"一次读末尾 256 KB":一条 assistant 记录可能本身就 > 256 KB
|
||||
(长回复 / 被回显的工具输出),此时尾窗会切在 JSON 行中间 ⇒ `json.loads` 失败 ⇒ **静默漏判**。
|
||||
做法:从尾部按 TAIL_BYTES 递增扩窗(上限 MAX_BYTES),**直到至少解析出一条 assistant 记录**。
|
||||
常见情形(小消息)只花一次 256 KB 读,成本可忽略。
|
||||
"""
|
||||
try:
|
||||
size = os.path.getsize(path)
|
||||
except OSError:
|
||||
return ''
|
||||
with io.open(path, 'rb') as f:
|
||||
window = TAIL_BYTES
|
||||
while True:
|
||||
start = max(0, size - window)
|
||||
f.seek(start)
|
||||
raw = f.read().decode('utf-8', 'replace')
|
||||
lines = raw.split('\n')
|
||||
if start > 0:
|
||||
lines = lines[1:] # 丢弃被截断的首行
|
||||
for line in reversed(lines):
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
rec = json.loads(line)
|
||||
except ValueError:
|
||||
continue
|
||||
if rec.get('type') != 'message' or rec.get('role') != 'assistant':
|
||||
continue
|
||||
chunks = [c['text'] for c in (rec.get('content') or [])
|
||||
if isinstance(c, dict) and isinstance(c.get('text'), str)]
|
||||
chunks += [c for c in (rec.get('content') or []) if isinstance(c, str)]
|
||||
if chunks:
|
||||
return '\n'.join(chunks)
|
||||
if start == 0 or window >= MAX_BYTES:
|
||||
# 放行,但**留痕**(A18:静默失败是负债)——可能是一条 >MAX_BYTES 的巨型记录
|
||||
try:
|
||||
os.environ.setdefault('_DSH_SG_MISS', '1')
|
||||
r0 = os.environ.get('CODEBUDDY_PROJECT_DIR') or os.environ.get('DSH_WORKSPACE') or ''
|
||||
if r0:
|
||||
log(r0, '未能解析(窗口 %d 字节仍无 assistant 记录)' % window)
|
||||
except Exception:
|
||||
pass
|
||||
return ''
|
||||
window = min(window * 4, MAX_BYTES)
|
||||
|
||||
|
||||
def tail_lines(text, n=2):
|
||||
out = [l.strip() for l in text.strip().split('\n') if l.strip()]
|
||||
return '\n'.join(out[-n:])
|
||||
|
||||
|
||||
# 转述/引用豁免:收尾行里带引号或"引用/规则/写着/禁"等词 ⇒ 是在复述规则,不是在问用户
|
||||
RE_QUOTE = re.compile(r'[「」“”"\']|引用|规则|写着|禁')
|
||||
|
||||
|
||||
RATE_WINDOW = 600 # 秒;同一会话两次「阻止停止」的最小间隔
|
||||
|
||||
|
||||
def _rate_limited(root, sid, peek=False):
|
||||
"""同一会话 RATE_WINDOW 秒内已拦过 ⇒ 本次直接放行(防连续多轮被拦)。"""
|
||||
if not root or not sid:
|
||||
return False
|
||||
p = os.path.join(root, '.workbuddy', 'cache', 'stop-guard-fires.json')
|
||||
try:
|
||||
d = json.loads(io.open(p, encoding='utf-8').read()) if os.path.exists(p) else {}
|
||||
except Exception:
|
||||
d = {}
|
||||
now = time.time()
|
||||
if now - float(d.get(sid, 0) or 0) < RATE_WINDOW:
|
||||
return True
|
||||
d = {k: v for k, v in d.items() if now - float(v or 0) < 86400} # 只留 1 天
|
||||
d[sid] = now
|
||||
try:
|
||||
os.makedirs(os.path.dirname(p), exist_ok=True)
|
||||
io.open(p, 'w', encoding='utf-8', newline='\n').write(json.dumps(d))
|
||||
except Exception:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def log(root, detail):
|
||||
try:
|
||||
p = os.path.join(root, LOG_REL)
|
||||
os.makedirs(os.path.dirname(p), exist_ok=True)
|
||||
with io.open(p, 'a', encoding='utf-8') as f:
|
||||
f.write('%s\t%s\n' % (time.strftime('%Y-%m-%d %H:%M:%S'), detail))
|
||||
lines = io.open(p, encoding='utf-8').read().split('\n') # 上限 300 行,超出截半(防膨胀)
|
||||
if len(lines) > 300:
|
||||
io.open(p, 'w', encoding='utf-8', newline='\n').write('\n'.join(lines[-150:]))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _entry_log(payload, raw_len):
|
||||
"""⚠️ **每次被调用必留痕**(含"payload 解析失败 / 未进作用域 / 被急停"三种静默情形)。
|
||||
|
||||
2026-09-15 教训:原实现只在**通过全部守卫之后**才写日志 ⇒ 日志缺失时**无法区分**
|
||||
「宿主根本没调用」与「调用了但被静默 return」—— 而这两者的处置**完全相反**
|
||||
(前者要卸载、后者要放宽作用域判据)。凡"要判有没有被调用"的探针,必须**入口即留痕**。
|
||||
"""
|
||||
try:
|
||||
p = payload if isinstance(payload, dict) else {}
|
||||
tp = str(p.get('transcript_path') or '')
|
||||
root = (os.environ.get('CODEBUDDY_PROJECT_DIR') or os.environ.get('DSH_WORKSPACE')
|
||||
or p.get('cwd') or WS_FALLBACK)
|
||||
log(str(root), 'entry|event=%s|cwd=%s|in_scope=%s|tp=%s|keys=%s|stdin_len=%s'
|
||||
% (p.get('hook_event_name') or '(parse-fail)', p.get('cwd') or '-',
|
||||
SCOPE in tp, (tp[-80:] if tp else '-'),
|
||||
(','.join(sorted(p.keys()))[:120] or '-'), raw_len))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _read_stdin_text():
|
||||
"""**显式按 UTF-8 读 stdin** —— 不要用 `sys.stdin.read()`。
|
||||
|
||||
⚠️ 2026-09-15 实测定位:本脚本的安装形态是 `python -S -E <脚本>`,而 **`-E` 会忽略
|
||||
`PYTHONUTF8=1` / `PYTHONIOENCODING=utf-8`** ⇒ `sys.stdin.encoding` 回退成 **cp936**;
|
||||
钩子 payload 里**必然含中文**(用户的提示词)⇒ 文本模式读取抛
|
||||
`UnicodeDecodeError: 'gbk' codec can't decode byte 0x80` ⇒ **钩子静默不生效、日志为空**,
|
||||
表象却是"宿主好像没调用钩子"(实为本地炸在解码上,白排查一轮)。
|
||||
读 `buffer` 即与 flag / locale 完全无关。
|
||||
"""
|
||||
try:
|
||||
return sys.stdin.buffer.read().decode('utf-8', 'replace')
|
||||
except Exception:
|
||||
try:
|
||||
return sys.stdin.read()
|
||||
except Exception:
|
||||
return ''
|
||||
|
||||
|
||||
def _emit(obj):
|
||||
"""**显式按 UTF-8 写 stdout**(同理:cp936 下 `ensure_ascii=False` 的中文 / `⛔` 会 UnicodeEncodeError)。"""
|
||||
data = json.dumps(obj, ensure_ascii=False).encode('utf-8')
|
||||
try:
|
||||
sys.stdout.buffer.write(data)
|
||||
sys.stdout.buffer.flush()
|
||||
except Exception: # 极端兜底:退回文本写(可能丢非 GBK 字符,但不至于静默不输出)
|
||||
try:
|
||||
sys.stdout.write(data.decode('utf-8', 'replace'))
|
||||
sys.stdout.flush()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────
|
||||
# 第二方案:`UserPromptSubmit`(2026-09-15 加)
|
||||
# 背景:本版 WorkBuddy **不调用 `Stop` 钩子**(实测:留痕已开、探针句已验证会命中、日志仍空)⇒ 改用
|
||||
# `UserPromptSubmit`(输入同样带 `transcript_path`,且能通过 `additionalContext` 注入上下文)。
|
||||
# 两级模式(**改一个文本文件即可切换,无需重启** —— 脚本内容每次调用现读):
|
||||
# probe :只写日志(零风险、可判定"有没有被调用")
|
||||
# inject :若**上一轮回复的收尾是征询句** ⇒ 注入一段上下文,让下一轮自我纠正
|
||||
# 模式文件:<工作区>/.workbuddy/stop-guard-mode (内容含 "inject" 即切到 inject,否则 probe)
|
||||
CONTEXT = (
|
||||
'⛔ 【上一轮收尾自检】你上一条回复的**最后一行是征询句**("要我…吗 / 要不要我 / 请确认 / 说一声即可"类),'
|
||||
'这属于本平台**被禁的形态**(`CODEBUDDY.md §1`「回话前自检」)。本轮的处置:'
|
||||
'① 若那件事本来就该你自己拍 —— **直接做完**,用陈述句交代;'
|
||||
'② 若确实命中真门禁(不可逆破坏性操作 / 边界外六类)—— 写进 `dsh-feature-first §5.1` 结论骨架的'
|
||||
'「**需要你拍板**」一节,该节必须是**整条回复的最后一节**、**逐条编号**,且**每个候选写明优点 / 缺点**、**候选竖排成段**(A / B / C 各占一行,⛔ 不横排、不做成表格的列)(陈述句,不要用征询句);'
|
||||
'⚠️ 若某候选**只有优点或只有缺点** ⇒ **那不该问**,自己拍掉;'
|
||||
'③ 顺带按红线 **R11** 复核:这个改动有没有让项目某一维度**净变差**。'
|
||||
)
|
||||
|
||||
|
||||
def mode_of(root):
|
||||
try:
|
||||
m = io.open(os.path.join(root or '.', '.workbuddy', 'stop-guard-mode'), encoding='utf-8').read()
|
||||
except OSError:
|
||||
m = ''
|
||||
return 'inject' if 'inject' in m else 'probe'
|
||||
|
||||
|
||||
def user_prompt_mode(payload):
|
||||
"""UserPromptSubmit:probe=只记日志;inject=命中则注入上下文(不阻断提示词)。"""
|
||||
tp = str(payload.get('transcript_path') or '')
|
||||
if SCOPE not in tp:
|
||||
return
|
||||
if os.environ.get('DSH_STOP_GUARD_OFF'):
|
||||
return
|
||||
root0 = os.environ.get('CODEBUDDY_PROJECT_DIR') or os.environ.get('DSH_WORKSPACE') or ''
|
||||
if root0 and os.path.exists(os.path.join(root0, '.workbuddy', 'stop-guard.disabled')):
|
||||
return
|
||||
mode = mode_of(root0)
|
||||
text = transcribe_last_assistant(tp)
|
||||
tail = tail_lines(text, 1) if text else ''
|
||||
hit = bool(tail) and not RE_QUOTE.search(tail) and bool(RE_BAN.search(tail))
|
||||
log(root0 or '.', 'invoked(user-prompt)|mode=%s|上轮收尾=征询句:%s|%s'
|
||||
% (mode, hit, (tail.replace('\n', ' ')[:60] if tail else '(取不到上一轮文本)')))
|
||||
if hit and mode == 'inject':
|
||||
_emit({'hookSpecificOutput': {'hookEventName': 'UserPromptSubmit',
|
||||
'additionalContext': CONTEXT}})
|
||||
|
||||
|
||||
def main():
|
||||
raw = _read_stdin_text() # ⚠️ 必须走 buffer:`-E` 下 sys.stdin 是 cp936(见 _read_stdin_text 注释)
|
||||
payload = None
|
||||
if raw.strip():
|
||||
try:
|
||||
payload = json.loads(raw)
|
||||
except ValueError:
|
||||
payload = None
|
||||
_entry_log(payload, len(raw)) # ⚠️ 先留痕,再判作用域(否则"没被调用"与"静默失配"分不开)
|
||||
if payload is None:
|
||||
return
|
||||
if (payload.get('hook_event_name') or '') == 'UserPromptSubmit': # 第二方案分派
|
||||
return user_prompt_mode(payload)
|
||||
tp = str(payload.get('transcript_path') or '')
|
||||
if SCOPE not in tp: # 作用域外 → 放行
|
||||
return
|
||||
if os.environ.get('DSH_STOP_GUARD_OFF'): # 急停(环境变量)→ 放行
|
||||
return
|
||||
root0 = os.environ.get('CODEBUDDY_PROJECT_DIR') or os.environ.get('DSH_WORKSPACE') or ''
|
||||
if root0 and os.path.exists(os.path.join(root0, '.workbuddy', 'stop-guard.disabled')):
|
||||
return # 急停(闸刀文件)→ 放行
|
||||
if payload.get('stop_hook_active'): # 防死循环 → 放行
|
||||
return
|
||||
text = transcribe_last_assistant(tp)
|
||||
if not text:
|
||||
return
|
||||
sid = str(payload.get('session_id') or '')
|
||||
if os.environ.get('DSH_SG_DEBUG'): # 调试:每次调用都留痕(用于验证宿主是否真的调用本钩子)
|
||||
log(root0 or '.', 'invoked|scope=%s|tail_active=%s' % (SCOPE in tp, bool(payload.get('stop_hook_active'))))
|
||||
if _rate_limited(root0, sid, peek=True): # 只查不记账
|
||||
return
|
||||
if os.environ.get('DSH_SG_LOG_ALL', '1') != '0': # ★本工作区内**每次调用都留痕**(可判定"有没有被调用")
|
||||
log(root0 or '.', 'invoked|tail_active=%s' % bool(payload.get('stop_hook_active')))
|
||||
tail = tail_lines(text, 1) # 只看**最后一行**:命中面越窄,误报越少
|
||||
if RE_QUOTE.search(tail): # 复述/引用规则 → 不是收尾提问
|
||||
return
|
||||
m = RE_BAN.search(tail)
|
||||
if not m:
|
||||
return
|
||||
root = (os.environ.get('CODEBUDDY_PROJECT_DIR') or os.environ.get('DSH_WORKSPACE') or '')
|
||||
_rate_limited(root0, sid) # 命中才记账(同一会话 10 分钟最多拦 1 次)
|
||||
log(root if os.path.isdir(root) else '.', 'stop-dialog-guard 命中:%s | 收尾:%s'
|
||||
% (m.group(0), tail.replace('\n', ' ')[:80]))
|
||||
_emit({'continue': False, 'reason': REASON})
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
try:
|
||||
main()
|
||||
except Exception:
|
||||
# ⚠️ 钩子绝不能因自身故障干扰会话 ⇒ 仍放行,但**必须留痕**(A18:静默失败是负债;
|
||||
# 2026-09-15 实证:本文件的 `except: pass` 曾把 `NameError: out is not defined` 藏住半小时)
|
||||
try:
|
||||
import traceback
|
||||
_r = os.environ.get('CODEBUDDY_PROJECT_DIR') or os.environ.get('DSH_WORKSPACE') or '.'
|
||||
log(_r, 'EXCEPTION|%s' % traceback.format_exc().strip().split('\n')[-1][:120])
|
||||
except Exception:
|
||||
pass
|
||||
sys.exit(0)
|
||||
Reference in new issue
Block a user