文档库:目录改为编号制(01-规范/02-架构设计/03-数据库/04-调整方案/
05-交接单/06-ops/07-scripts/08-skills/09-archive),顶层散文件归入 01-规范/;
INDEX.md 与 docs-manifest.json 重刷(档案 146 篇);旧目录名引用全量对齐。
IM 线:src/im/**(SDK / hub / store / presence / ws / gateway-token)、
src/web/routes/im.ts、src/db/plugin-data/**、src/supervisor/plugin-assembly.ts
及对应 test/**。
插件线:poc/{im-agent-bridge,im-connection-gateway,im-conversation-tabs,
business-plugins-im,carbon-mcp-probe}、src/web/routes/{sessions,overlay-device}.ts、
src/net/relay/{device-grant,instance-credential}.ts。
仓库卫生:清出 40 个历史误入库 / 已改名文件(34 个交接单归档 + 6 个旧结构,
本地均有副本);dsh-server-docs/.gitignore 补 tmp/;交接单不入库(政策)。
This commit is contained in:
1 parent
3d8f50e366
commit
e6207aa691
239 files changed
+34477
-14633
No files matched your search
@@ -0,0 +1,90 @@
|
||||
/**
|
||||
* 会话时序分析器(只输出时间戳/事件类型/时延,不输出任何消息正文)
|
||||
* 用法:node analyze-session.mjs <session.jsonl.zstd>
|
||||
*/
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { zstdDecompressSync } from 'node:zlib'
|
||||
|
||||
const file = process.argv[2]
|
||||
const raw = readFileSync(file)
|
||||
|
||||
// 可能是多帧拼接:按 zstd magic 切分逐帧解压
|
||||
const MAGIC = Buffer.from([0x28, 0xb5, 0x2f, 0xfd])
|
||||
const offsets = []
|
||||
for (let i = 0; i + 4 <= raw.length; i++) {
|
||||
if (raw.compare(MAGIC, 0, 4, i, i + 4) === 0) offsets.push(i)
|
||||
}
|
||||
let text = ''
|
||||
const frames = offsets.length > 0 ? offsets : [0]
|
||||
for (let f = 0; f < frames.length; f++) {
|
||||
const start = frames[f]
|
||||
const end = f + 1 < frames.length ? frames[f + 1] : raw.length
|
||||
try {
|
||||
text += zstdDecompressSync(raw.subarray(start, end)).toString('utf8')
|
||||
} catch (e) {
|
||||
// 单帧整体解压兜底
|
||||
}
|
||||
}
|
||||
if (text === '') {
|
||||
try {
|
||||
text = zstdDecompressSync(raw).toString('utf8')
|
||||
} catch (e) {
|
||||
console.log('decompress failed:', e.message)
|
||||
process.exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
const lines = text.split('\n').filter((l) => l.trim() !== '')
|
||||
console.log(`帧数=${frames.length} 行数=${lines.length} 解压字节=${text.length}`)
|
||||
|
||||
const pick = (obj, names, depth = 0) => {
|
||||
if (obj === null || typeof obj !== 'object' || depth > 3) return undefined
|
||||
for (const n of names) if (typeof obj[n] !== 'undefined') return obj[n]
|
||||
for (const v of Object.values(obj)) {
|
||||
const hit = pick(v, names, depth + 1)
|
||||
if (hit !== undefined) return hit
|
||||
}
|
||||
return undefined
|
||||
}
|
||||
const TS_KEYS = ['ts', 'time', 'timestamp', 'at', 'createdAt', 'created_at', 'mtime', 'date']
|
||||
const KIND_KEYS = ['kind', 'type', 'event', 'role', 'name', 'tag']
|
||||
|
||||
const events = []
|
||||
const kinds = new Map()
|
||||
for (const line of lines) {
|
||||
let o
|
||||
try {
|
||||
o = JSON.parse(line)
|
||||
} catch {
|
||||
continue
|
||||
}
|
||||
const ts = pick(o, TS_KEYS)
|
||||
const kind = pick(o, KIND_KEYS)
|
||||
const k = typeof kind === 'string' ? kind : JSON.stringify(kind ?? '?')
|
||||
kinds.set(k, (kinds.get(k) ?? 0) + 1)
|
||||
const t = typeof ts === 'number' ? (ts < 1e12 ? ts * 1000 : ts) : typeof ts === 'string' ? Date.parse(ts) : NaN
|
||||
events.push({ t: Number.isNaN(t) ? undefined : t, k })
|
||||
}
|
||||
|
||||
console.log('\n== 事件类型分布(TOP 20)==')
|
||||
;[...kinds.entries()].sort((a, b) => b[1] - a[1]).slice(0, 20).forEach(([k, c]) => console.log(` ${c}\t${k}`))
|
||||
|
||||
const withTs = events.filter((e) => e.t !== undefined)
|
||||
console.log(`\n带时间戳事件: ${withTs.length}/${events.length}`)
|
||||
if (withTs.length > 1) {
|
||||
const span = (withTs[withTs.length - 1].t - withTs[0].t) / 1000
|
||||
console.log(`时间跨度: ${span.toFixed(1)}s(约 ${(span / 60).toFixed(1)} 分钟)`)
|
||||
|
||||
const gaps = []
|
||||
for (let i = 1; i < withTs.length; i++) gaps.push({ d: (withTs[i].t - withTs[i - 1].t) / 1000, a: withTs[i - 1].k, b: withTs[i].k })
|
||||
|
||||
console.log('\n== 最大间隔 TOP 12(秒 | 前事件 → 后事件)==')
|
||||
gaps.sort((x, y) => y.d - x.d).slice(0, 12).forEach((g) => console.log(` ${g.d.toFixed(1)}\t${g.a} → ${g.b}`))
|
||||
|
||||
const over5 = gaps.filter((g) => g.d > 5)
|
||||
console.log(`\n>5s 的空档数: ${over5.length};>30s: ${gaps.filter((g) => g.d > 30).length};>60s: ${gaps.filter((g) => g.d > 60).length}`)
|
||||
|
||||
// 会话首尾时间(便于与外部日志对齐)
|
||||
const fmt = (ms) => new Date(ms).toISOString().replace('T', ' ').slice(0, 19)
|
||||
console.log(`\n首事件: ${fmt(withTs[0].t)} 末事件: ${fmt(withTs[withTs.length - 1].t)}`)
|
||||
}
|
||||
@@ -0,0 +1,71 @@
|
||||
/**
|
||||
* 会话每轮指标:TTFT(turn/start → 首个流式 chunk)、轮时长、工具耗时
|
||||
* 只输出时间与类型,不输出任何正文。
|
||||
* 用法:node analyze-turn.mjs <session.jsonl.zstd>
|
||||
*/
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { zstdDecompressSync } from 'node:zlib'
|
||||
|
||||
const raw = readFileSync(process.argv[2])
|
||||
const MAGIC = Buffer.from([0x28, 0xb5, 0x2f, 0xfd])
|
||||
const offs = []
|
||||
for (let i = 0; i + 4 <= raw.length; i++) if (raw.compare(MAGIC, 0, 4, i, i + 4) === 0) offs.push(i)
|
||||
let text = ''
|
||||
const frames = offs.length > 0 ? offs : [0]
|
||||
for (let f = 0; f < frames.length; f++) {
|
||||
const s = frames[f], e = f + 1 < frames.length ? frames[f + 1] : raw.length
|
||||
try { text += zstdDecompressSync(raw.subarray(s, e)).toString('utf8') } catch {}
|
||||
}
|
||||
if (text === '') { try { text = zstdDecompressSync(raw).toString('utf8') } catch (e) { console.log('fail', e.message); process.exit(1) } }
|
||||
|
||||
const pick = (o, keys, d = 0) => {
|
||||
if (o === null || typeof o !== 'object' || d > 3) return undefined
|
||||
for (const k of keys) if (typeof o[k] !== 'undefined') return o[k]
|
||||
for (const v of Object.values(o)) { const h = pick(v, keys, d + 1); if (h !== undefined) return h }
|
||||
return undefined
|
||||
}
|
||||
const ev = []
|
||||
for (const line of text.split('\n')) {
|
||||
if (line.trim() === '') continue
|
||||
let o; try { o = JSON.parse(line) } catch { continue }
|
||||
const kind = pick(o, ['kind', 'type', 'event'])
|
||||
const ts = pick(o, ['ts', 'time', 'timestamp', 'at', 'createdAt'])
|
||||
const t = typeof ts === 'number' ? (ts < 1e12 ? ts * 1000 : ts) : typeof ts === 'string' ? Date.parse(ts) : NaN
|
||||
if (typeof kind === 'string') ev.push({ k: kind, t: Number.isNaN(t) ? undefined : t })
|
||||
}
|
||||
|
||||
const fmt = (ms) => new Date(ms).toISOString().slice(11, 19) + 'Z'
|
||||
let turns = 0
|
||||
for (let i = 0; i < ev.length; i++) {
|
||||
if (ev[i].k !== 'turn/start') continue
|
||||
turns++
|
||||
const t0 = ev[i].t
|
||||
let end, firstChunk, firstText
|
||||
for (let j = i + 1; j < ev.length; j++) {
|
||||
if (ev[j].k === 'turn/end' && end === undefined) { end = ev[j].t; break }
|
||||
if (firstChunk === undefined && (ev[j].k === 'assistant/chunk' || ev[j].k === 'reasoning-chunks') && ev[j].t !== undefined) firstChunk = ev[j].t
|
||||
if (firstText === undefined && ev[j].k === 'text-chunks' && ev[j].t !== undefined) firstText = ev[j].t
|
||||
}
|
||||
const ttft = t0 !== undefined && firstChunk !== undefined ? ((firstChunk - t0) / 1000).toFixed(1) : '?'
|
||||
const ttftText = t0 !== undefined && firstText !== undefined ? ((firstText - t0) / 1000).toFixed(1) : '?'
|
||||
const dur = t0 !== undefined && end !== undefined ? ((end - t0) / 1000).toFixed(1) : '?'
|
||||
console.log(`轮 ${String(turns).padStart(2)} ${t0 !== undefined ? fmt(t0) : '?'} TTFT=${ttft}s 首文本=${ttftText}s 轮时长=${dur}s`)
|
||||
}
|
||||
|
||||
// 工具耗时
|
||||
const tools = ev.map((e, i) => ({ e, i })).filter((x) => x.e.k === 'tool/call')
|
||||
const dt = []
|
||||
for (const { i } of tools) {
|
||||
for (let j = i + 1; j < ev.length; j++) {
|
||||
if (ev[j].k === 'tool/result') {
|
||||
if (ev[i].t !== undefined && ev[j].t !== undefined) dt.push((ev[j].t - ev[i].t) / 1000)
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
if (dt.length > 0) {
|
||||
dt.sort((a, b) => b - a)
|
||||
const avg = dt.reduce((a, b) => a + b, 0) / dt.length
|
||||
console.log(`\n工具调用 ${dt.length} 次;平均 ${avg.toFixed(1)}s;最慢 5 个: ${dt.slice(0, 5).map((x) => x.toFixed(1) + 's').join(', ')}`)
|
||||
}
|
||||
console.log(`\n合计轮数: ${turns}`)
|
||||
@@ -0,0 +1,323 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""bash-output-guard —— `PreToolUse(Bash)`:在**命令执行前**拦下"会灌爆上下文"的读命令。
|
||||
|
||||
为什么需要(2026-09-15 实测)
|
||||
────────────────────────────
|
||||
对话历史是 **append-only**:一次工具调用在转录里落两条记录 —— `function_call`(命令)
|
||||
+ `function_call_result`(**输出正文**);**输出一旦生成就永久留在 messages 里,且每轮全量重发**。
|
||||
实测某会话 **553 轮 × 平均 42 万 token = 2.33 亿 input**,而 output 仅占 **0.35%**。
|
||||
⇒ 想真正"把大输出从上下文里去掉",**唯一的时机就是它被生成之前**(模型无权删自己的历史)。
|
||||
|
||||
边界(重要)
|
||||
──────────
|
||||
* **只拦"读"类高风险命令**,且**必须给出等价限流写法**(教而非堵)。
|
||||
* **判不准就放行**(fail-open)—— 本机不是沙箱,但这类命令本身不破坏数据,误拦的代价只是麻烦。
|
||||
* 急停:env `DSH_OUTPUT_GUARD_OFF=1`,或新建 `<工作区>/.workbuddy/bash-guard.disabled`。
|
||||
* 命中才写一行日志 `<工作区>/.workbuddy/bash-guard.log`(用于调误报,上限 300 行)。
|
||||
|
||||
安装(`settings.json` 的 hooks 段 · **新增一条** · ⚠️ **需完全重启才生效**)
|
||||
────────────────────────────────────────────────────────────────────
|
||||
"PreToolUse": [ …,
|
||||
{ "matcher": "Bash",
|
||||
"hooks": [{ "type": "command",
|
||||
"command": "\"<python>\" \"<此脚本>\"", "timeout": 10 }] } ]
|
||||
⛔ **别给本脚本加 `-E`**:`-E` 会屏蔽 `PYTHONUTF8`/`PYTHONIOENCODING` ⇒ stdin 回退 cp936 ⇒
|
||||
含中文的 payload 解析失败且**静默 fail-open**(同目录 `stop-dialog-guard.py` 已因此"白排查一天")。
|
||||
"""
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
|
||||
LOG_REL = os.path.join('.workbuddy', 'bash-guard.log')
|
||||
SCOPE = 'aliyun-dsh-server'
|
||||
# 兜底工作区:脚本位于 <工作区>/dsh-server-docs/07-scripts/ ⇒ 上溯三级
|
||||
WS_FALLBACK = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
# (是否 core, 正则, 名称, 更省的写法)
|
||||
# ⚠️ 设计原则:**不能全拦** —— 误拦挡住正事 ⇒ 我会来回试/绕路 ⇒ 反而更贵;漏拦只是多花点 token。
|
||||
# core=True = 几乎必然巨大、误拦率≈0 ⇒ **默认就拦**
|
||||
# core=False = 经常确实需要全量 ⇒ **仅 hard 模式拦**(默认放行)
|
||||
RULES = [
|
||||
# cat:**按目标文件大小判定**(小文件不拦)—— 这是防日常误伤的关键,见 cat_bigfile()
|
||||
(True, '__CAT_BIGFILE__',
|
||||
'`cat` 大文件', '改用 `head -30 文件` / `sed -n \'1,30p\' 文件` / `wc -l 文件`'),
|
||||
# ⚠️ 2026-09-15 **回放今日 12 个会话的 4049 条真实命令**:本规则命中 62 次,抽样 **29/29 全是窄范围**
|
||||
# (单文件 / 具体目录),**0 条从根**;连 `grep -c` 这种本来只有几行的也被误伤。
|
||||
# 旧正则 `\bgrep\b[^|;]*-[a-zA-Z]*[rR]` 还有 bug:`[^|;]*` 会匹配到**路径里的 `-server`**
|
||||
# (`-s`+`erve`+`r`)⇒ 凡路径含 `aliyun-dsh-server` 就必命中。
|
||||
# ⇒ **误伤 >> 收益** ⇒ 降为 loose(仅 `hard` 模式拦),并把正则收紧到「选项紧跟 grep」。
|
||||
(False, r'\bgrep\s+-[a-zA-Z]*[rR]\b',
|
||||
'递归 grep', '改用 `grep -rn … | head -30`,或 `grep -rc …`(只要计数)'),
|
||||
# ⚠️ 只拦 `-R`(递归全树);**不拦 `ls -la`**(含 l 会误伤日常);选项必须紧跟 ls(同样防路径误匹配)
|
||||
(True, r'\bls\s+-[a-zA-Z]*R\b',
|
||||
'`ls -R` 递归全树', '改用 `ls -la 目录 | head -20`,或 `ls 目录 | wc -l`'),
|
||||
(True, r'\bfind\s+(/[A-Za-z]|[A-Za-z]:)',
|
||||
'`find` 从盘符/根起全树扫', '改用 `find 具体目录 -maxdepth 3 … | head -20`'),
|
||||
(True, r'\bjournalctl\b(?!.*(-n\s*\d|head|--since))',
|
||||
'`journalctl` 无行数/时间限制', '改用 `journalctl -u X -n 50 --no-pager`(`--since` 也算限流,放行)'),
|
||||
(True, r'\bdmesg\b(?!.*head)',
|
||||
'`dmesg` 无行数限制', '改用 `dmesg | tail -30`'),
|
||||
(False, r'(^|[|;&]\s*)rg\b(?!.*\|)',
|
||||
'递归 rg 无管道限流', '改用 `rg … | head -30`,或 `rg -c …`'),
|
||||
(False, r'\bgit\s+(log|diff)\b(?!.*(-n\s*\d|--max-count|head))',
|
||||
'`git log/diff` 无行数限制', '改用 `git log --oneline -5` / `git diff --stat`'),
|
||||
(False, r'\b(du|tree)\b[^|;]*\s(/|[A-Za-z]:)',
|
||||
'`du/tree` 从根起', '改用 `du -sh 具体目录` / `tree -L 2 目录 | head -40`'),
|
||||
]
|
||||
# 命令里已有限流 ⇒ 放行(只按"最外层"有就好了)
|
||||
SAFE = re.compile(r'\|\s*(head|tail|wc|grep\s+-c|cut|sed\s+-n|awk|uniq|sort\s+-u)\b')
|
||||
|
||||
|
||||
def _read_stdin():
|
||||
"""⚠️ 必须走 buffer 显式 UTF-8(`-E` 下 sys.stdin 是 cp936)。"""
|
||||
try:
|
||||
return sys.stdin.buffer.read().decode('utf-8', 'replace')
|
||||
except Exception:
|
||||
try:
|
||||
return sys.stdin.read()
|
||||
except Exception:
|
||||
return ''
|
||||
|
||||
|
||||
def _emit(obj):
|
||||
data = json.dumps(obj, ensure_ascii=False).encode('utf-8')
|
||||
try:
|
||||
sys.stdout.buffer.write(data)
|
||||
sys.stdout.buffer.flush()
|
||||
except Exception:
|
||||
try:
|
||||
sys.stdout.write(data.decode('utf-8', 'replace'))
|
||||
sys.stdout.flush()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def log(root, detail):
|
||||
try:
|
||||
p = os.path.join(root, LOG_REL)
|
||||
os.makedirs(os.path.dirname(p), exist_ok=True)
|
||||
with io.open(p, 'a', encoding='utf-8', newline='\n') as f:
|
||||
f.write('%s\t%s\n' % (time.strftime('%Y-%m-%d %H:%M:%S'), detail))
|
||||
lines = io.open(p, encoding='utf-8').read().split('\n')
|
||||
if len(lines) > 300:
|
||||
io.open(p, 'w', encoding='utf-8', newline='\n').write('\n'.join(lines[-150:]))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
CAT_BIG = 200 * 1024 # `cat` 目标文件 > 200 KB 才值得拦(小文件放行 ⇒ 防日常误伤)
|
||||
READ_BIG = 400 * 1024 # `Read` 目标文件 > 400 KB ⇒ 至少几万 token ⇒ 拦(2026-09-15 实测补)
|
||||
# ↑ 为什么补 Read:本钩子最初只管 Bash,但实测 `Read` 是**第二大输出源** ——
|
||||
# 某会话 44 次 / 累计 238 K 字符、单条最大 33 K 字符(≈13 K token);另一会话 136 次 / 217 K 字符,
|
||||
# 且**完全没有被任何机制覆盖**。⇒ "想把大输出从上下文里去掉",就必须覆盖所有会灌大输出的工具。
|
||||
|
||||
|
||||
def read_too_big(payload):
|
||||
"""`Read` 一个大文件 ⇒ True。取不到大小(文件不存在/相对路径)⇒ False(放行)。"""
|
||||
ti = payload.get('tool_input') or {}
|
||||
p = ti.get('file_path') or ti.get('path') or ''
|
||||
if not isinstance(p, str) or not p:
|
||||
return False
|
||||
try:
|
||||
return os.path.getsize(p) > READ_BIG
|
||||
except OSError:
|
||||
return False
|
||||
|
||||
|
||||
def cat_bigfile(cmd):
|
||||
"""`cat <文件>` 且目标文件 > CAT_BIG ⇒ True。取不到大小(相对路径/不存在/管道源)⇒ False(放行)。"""
|
||||
m = re.search(r'(?:^|[|;&]\s*)cat\s+(?:-[A-Za-z]+\s+)*([^\s|>;&]+)', cmd)
|
||||
if not m:
|
||||
return False
|
||||
p = m.group(1).strip('"\'')
|
||||
if p in ('-', '/dev/null', '/dev/stdin'):
|
||||
return False
|
||||
try:
|
||||
return os.path.getsize(p) > CAT_BIG
|
||||
except OSError:
|
||||
return False
|
||||
|
||||
|
||||
def segments(cmd):
|
||||
"""按 shell 分隔符切段(`;` `&&` `||` `|` 换行)—— 只对**每段的开头**做匹配。
|
||||
|
||||
⚠️ 为什么不匹配整条命令文本(2026-09-15 实测教训):那样**引号里的字符串也会被拦** ——
|
||||
例如 `printf '...ls -laR...'`、测试脚本、把命令写进文档,全都会误拦(阻塞面过大)。
|
||||
按段匹配既保住真拦(`cd x && ls -laR` 的第二段以 `ls` 开头 ⇒ 照样拦),又不误伤"只是提到"。
|
||||
"""
|
||||
return [s.strip() for s in re.split(r'(?:&&|\|\||;|\n|\|)', cmd) if s.strip()]
|
||||
|
||||
|
||||
# ── 批量活检测(2026-09-15 用户选 A)─────────────────────────────
|
||||
# 要打断的是:**"一句话需求 → 上百次工具调用"** —— 每次调用都把「命令 + 结果」永久留在上下文里。
|
||||
# 实测(会话 7057685c):逐处改注释 ⇒ Edit 142 次、痕迹 13.4 万字符;若改用一个脚本 ⇒ 约 8 千字符(省 94%)。
|
||||
# 实现:`PreToolUse` 里给 Edit/Write/Bash 记账;到档位且**同质** ⇒ deny 一刀并给出替代做法(我改路子)。
|
||||
BATCH_N = 20
|
||||
BATCH_TOOLS = ('Edit', 'Write', 'Bash')
|
||||
|
||||
|
||||
def _batch_state(root):
|
||||
p = os.path.join(root, '.workbuddy', 'batch-guard.json')
|
||||
try:
|
||||
d = json.load(io.open(p, encoding='utf-8'))
|
||||
except Exception:
|
||||
d = {}
|
||||
return p, d
|
||||
|
||||
|
||||
def _homogeneous(tool, st):
|
||||
"""同质性:Edit ⇒ 最近 20 次里同一文件出现 ≥3 次(= 同一文件反复改);Bash ⇒ 命令都很短。"""
|
||||
if tool == 'Edit':
|
||||
paths = st.get('paths') or []
|
||||
if not paths:
|
||||
return False
|
||||
c = {}
|
||||
for x in paths[-20:]:
|
||||
c[x] = c.get(x, 0) + 1
|
||||
return max(c.values()) >= 3
|
||||
if tool == 'Bash':
|
||||
lens = st.get('lens') or []
|
||||
return bool(lens) and (sum(lens[-20:]) / max(len(lens[-20:]), 1)) < 300
|
||||
return True
|
||||
|
||||
|
||||
def batch_check(root, payload, log_fn):
|
||||
"""到档位 + 同质 ⇒ 返回提示串(并记账,每档只提示一次);否则 ''。"""
|
||||
sid = str(payload.get('session_id') or '')
|
||||
tool = payload.get('tool_name') or ''
|
||||
if not sid or tool not in BATCH_TOOLS:
|
||||
return ''
|
||||
p, d = _batch_state(root)
|
||||
st = d.setdefault(sid, {})
|
||||
st[tool] = int(st.get(tool, 0)) + 1
|
||||
n = st[tool]
|
||||
ti = payload.get('tool_input') or {}
|
||||
if tool == 'Edit':
|
||||
fp = str(ti.get('file_path') or '')
|
||||
if fp:
|
||||
st['paths'] = (st.get('paths') or [])[-40:] + [fp]
|
||||
elif tool == 'Bash':
|
||||
st['lens'] = (st.get('lens') or [])[-40:] + [len(str(ti.get('command') or ''))]
|
||||
fired = st.setdefault('fired', [])
|
||||
hit = ''
|
||||
if n in (BATCH_N, BATCH_N * 2, BATCH_N * 4) and ('%s@%d' % (tool, n)) not in fired and _homogeneous(tool, st):
|
||||
fired.append('%s@%d' % (tool, n))
|
||||
files = len(set(st.get('paths') or []))
|
||||
hit = (
|
||||
'🛠【批量活提示】本会话 `%s` 已调用 **%d 次**%s ⇒ 这更像**批量活**。\n'
|
||||
'✅ 改用**一个脚本一次做完**,只回「改了 N 处 / 失败 M 处」—— **别逐处调用**:'
|
||||
'逐次调用会把 N 份「命令 + 结果」**永久留在会话上下文里、之后每轮全量重发**。\n'
|
||||
'ℹ️ 实测参照:某会话逐处改 142 次 ⇒ 痕迹 **13.4 万字符**;改用一个脚本 ⇒ 约 **8 千字符(省 94%%)**。'
|
||||
'改完**统一验一次**即可,不必改一处验一处。'
|
||||
% (tool, n, ('(改到的文件只有 %d 个)' % files) if (tool == 'Edit' and files) else ''))
|
||||
log_fn(str(root), 'DENY|批量活提示 %s@%d' % (tool, n))
|
||||
try:
|
||||
if len(d) > 30:
|
||||
for k in list(d)[:-30]:
|
||||
d.pop(k, None)
|
||||
io.open(p, 'w', encoding='utf-8', newline='\n').write(json.dumps(d, ensure_ascii=False))
|
||||
except Exception:
|
||||
pass
|
||||
return hit
|
||||
|
||||
|
||||
def reason_for(cmd, hard=False):
|
||||
segs = segments(cmd)
|
||||
for core, pat, why, fix in RULES:
|
||||
if not core and not hard: # loose 条只在 hard 模式生效
|
||||
continue
|
||||
for s in segs:
|
||||
if pat == '__CAT_BIGFILE__':
|
||||
if cat_bigfile(s):
|
||||
return why, fix
|
||||
continue
|
||||
if re.match(pat, s): # ← match(段首),不是 search(全串)
|
||||
return why, fix
|
||||
return None, None
|
||||
|
||||
|
||||
def main():
|
||||
raw = _read_stdin()
|
||||
payload = None
|
||||
if raw.strip():
|
||||
try:
|
||||
payload = json.loads(raw)
|
||||
except ValueError:
|
||||
payload = None
|
||||
root = (os.environ.get('CODEBUDDY_PROJECT_DIR') or os.environ.get('DSH_WORKSPACE')
|
||||
or (isinstance(payload, dict) and payload.get('cwd')) or WS_FALLBACK)
|
||||
tp = str((payload or {}).get('transcript_path') or '') if isinstance(payload, dict) else ''
|
||||
if isinstance(payload, dict):
|
||||
# 入口即留痕(只记 event,低频;便于判"有没有被调用")
|
||||
log(str(root), 'entry|event=%s|in_scope=%s' % (payload.get('hook_event_name') or '(parse-fail)', SCOPE in tp))
|
||||
if not isinstance(payload, dict):
|
||||
return
|
||||
if (payload.get('hook_event_name') or '') != 'PreToolUse':
|
||||
return
|
||||
tool = payload.get('tool_name') or ''
|
||||
if tool not in ('Bash', 'Read', 'Edit', 'Write'): # Bash/Read=拦大输出;Edit/Write=批量活记账
|
||||
return
|
||||
if os.environ.get('DSH_OUTPUT_GUARD_OFF'):
|
||||
return
|
||||
try:
|
||||
if os.path.exists(os.path.join(root, '.workbuddy', 'bash-guard.disabled')):
|
||||
return
|
||||
except Exception:
|
||||
pass
|
||||
if tool in ('Edit', 'Write'): # 批量活检测:deny 一刀,换掉"上百次逐处调用"
|
||||
chk = batch_check(str(root), payload, log)
|
||||
if chk:
|
||||
_emit({'hookSpecificOutput': {'hookEventName': 'PreToolUse',
|
||||
'permissionDecision': 'deny',
|
||||
'permissionDecisionReason': chk}})
|
||||
return
|
||||
if tool == 'Read':
|
||||
if read_too_big(payload):
|
||||
fp = str(((payload.get('tool_input') or {}).get('file_path')) or '')
|
||||
log(str(root), 'DENY|Read 大文件|%s' % fp[:110])
|
||||
_emit({'hookSpecificOutput': {
|
||||
'hookEventName': 'PreToolUse', 'permissionDecision': 'deny',
|
||||
'permissionDecisionReason': (
|
||||
'💰 拦下:**Read 大文件**(`%s`)—— 它的**全文会写进会话历史、之后每轮全量重发**。\n'
|
||||
'✅ 分段读:`Read` 加 `offset` / `limit`(例 `limit: 120`)只取你要的那段;'
|
||||
'或先 `grep -n "关键词" 文件 | head -20` 定位行号,再读那几行。\n'
|
||||
'ℹ️ 确实要通读 ⇒ 用 Bash 分批 `sed -n \'1,200p\' 文件`;急停 env `DSH_OUTPUT_GUARD_OFF=1` '
|
||||
'或 `.workbuddy/bash-guard.disabled`。' % fp[:80])}})
|
||||
return
|
||||
cmd = ((payload.get('tool_input') or {}).get('command')) or ''
|
||||
if not cmd or SAFE.search(cmd):
|
||||
return
|
||||
try:
|
||||
md = io.open(os.path.join(root, '.workbuddy', 'bash-guard-mode'),
|
||||
encoding='utf-8').read().lower()
|
||||
except OSError:
|
||||
md = ''
|
||||
if 'off' in md:
|
||||
return
|
||||
why, fix = reason_for(cmd, hard=('hard' in md))
|
||||
if not why:
|
||||
return
|
||||
log(str(root), 'DENY|%s|%s' % (why, cmd.replace('\n', ' ')[:110]))
|
||||
_emit({'hookSpecificOutput': {
|
||||
'hookEventName': 'PreToolUse',
|
||||
'permissionDecision': 'deny',
|
||||
'permissionDecisionReason': (
|
||||
'💰 拦下:**%s** —— 这类命令的输出会**永久留在会话上下文里、每轮全量重发**'
|
||||
'(实测某会话 553 轮 × 平均 42 万 token = 2.33 亿 input,'
|
||||
'output 仅占 0.35%%)。\n'
|
||||
'✅ 换成限流写法再发:%s\n'
|
||||
'ℹ️ 若确实需要全量:**先落盘再只读关键行**(`… > /tmp/x.txt 2>&1` 然后 `sed -n \'1,40p\' /tmp/x.txt`);'
|
||||
'急停用 env `DSH_OUTPUT_GUARD_OFF=1` 或新建 `<工作区>/.workbuddy/bash-guard.disabled`。'
|
||||
% (why, fix))}})
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
try:
|
||||
main()
|
||||
except Exception:
|
||||
pass # fail-open:本钩子只为省积分,绝不因自身异常阻断正常工作
|
||||
sys.exit(0)
|
||||
@@ -0,0 +1,61 @@
|
||||
#!/usr/bin/env python3
|
||||
"""列出 Cloudflare 账号可见的 zone(可选某 zone 的 DNS 记录)——只读。
|
||||
用法: python3 cf-dns.py [zone名]
|
||||
凭据: /etc/cloudflare.ini 的 dns_cloudflare_api_token(不打印 token 本身)
|
||||
"""
|
||||
import json
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import urllib.request
|
||||
|
||||
CRED = '/etc/cloudflare.ini'
|
||||
API = 'https://api.cloudflare.com/client/v4'
|
||||
|
||||
|
||||
def token():
|
||||
txt = open(CRED, 'r', encoding='utf-8').read()
|
||||
m = re.search(r'dns_cloudflare_api_token\s*=\s*([A-Za-z0-9_\-]+)', txt)
|
||||
if not m:
|
||||
sys.exit('未在 %s 找到 dns_cloudflare_api_token' % CRED)
|
||||
return m.group(1)
|
||||
|
||||
|
||||
def get(path, tok):
|
||||
req = urllib.request.Request(API + path, headers={'Authorization': 'Bearer ' + tok})
|
||||
with urllib.request.urlopen(req, timeout=20) as resp:
|
||||
return json.loads(resp.read().decode('utf-8'))
|
||||
|
||||
|
||||
def short(v, n=44):
|
||||
v = str(v)
|
||||
return v if len(v) <= n else v[: n - 3] + '...'
|
||||
|
||||
|
||||
def main():
|
||||
tok = token()
|
||||
zones = (get('/zones?per_page=100', tok).get('result') or [])
|
||||
print('可见 zone (%d):' % len(zones))
|
||||
for z in zones:
|
||||
print(' - %s status=%s id=%s...' % (z['name'], z['status'], z['id'][:8]))
|
||||
|
||||
if len(sys.argv) < 2:
|
||||
return
|
||||
want = sys.argv[1]
|
||||
zid = ''
|
||||
for z in zones:
|
||||
if z['name'] == want:
|
||||
zid = z['id']
|
||||
if zid == '':
|
||||
print('\nzone %s 不在该 token 权限内(无法管理其 DNS/证书)' % want)
|
||||
return
|
||||
|
||||
recs = (get('/zones/%s/dns_records?per_page=100' % zid, tok).get('result') or [])
|
||||
print('\n=== %s 的 DNS 记录(%d 条)===' % (want, len(recs)))
|
||||
for r in sorted(recs, key=lambda x: (x['type'], x['name'])):
|
||||
prox = 'on' if r.get('proxied') else 'off'
|
||||
print(' %-6s %-34s -> %-46s proxied=%s' % (r['type'], r['name'], short(r['content']), prox))
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,88 @@
|
||||
#!/usr/bin/env python3
|
||||
"""探测 _acme-challenge 名称是否被通配 CNAME 劫持(只读+临时记录,用完即删)。
|
||||
用法: python3 cf-probe.py <zone名> [_acme-challenge.<zone>]
|
||||
"""
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
import urllib.request
|
||||
|
||||
CRED = '/etc/cloudflare.ini'
|
||||
# 2026-09-19 域迁 ai1net.com 后其 zone 走独立凭据(与旧域 token 不通用)
|
||||
CREDS = {'ai1net.com': '/etc/cloudflare-ai1net.ini'}
|
||||
API = 'https://api.cloudflare.com/client/v4'
|
||||
|
||||
|
||||
def token(zone=None):
|
||||
txt = open(CREDS.get(zone or '', CRED), 'r', encoding='utf-8').read()
|
||||
m = re.search(r'dns_cloudflare_api_token\s*=\s*([A-Za-z0-9_\-]+)', txt)
|
||||
if not m:
|
||||
sys.exit('未找到 token')
|
||||
return m.group(1)
|
||||
|
||||
|
||||
def req(method, path, tok, body=None):
|
||||
data = json.dumps(body).encode('utf-8') if body is not None else None
|
||||
r = urllib.request.Request(API + path, data=data, method=method,
|
||||
headers={'Authorization': 'Bearer ' + tok,
|
||||
'Content-Type': 'application/json'})
|
||||
with urllib.request.urlopen(r, timeout=20) as resp:
|
||||
return json.loads(resp.read().decode('utf-8'))
|
||||
|
||||
|
||||
def doh(name, typ, server='https://dns.google/resolve'):
|
||||
url = '%s?name=%s&type=%s' % (server, name, typ)
|
||||
r = urllib.request.Request(url, headers={'accept': 'application/dns-json'})
|
||||
with urllib.request.urlopen(r, timeout=20) as resp:
|
||||
return json.loads(resp.read().decode('utf-8'))
|
||||
|
||||
|
||||
def show(title, name, typ):
|
||||
print('--- %s ---' % title)
|
||||
for srv, label in [('https://dns.google/resolve', 'Google'), ('https://cloudflare-dns.com/dns-query', 'CF')]:
|
||||
try:
|
||||
d = doh(name, typ, srv)
|
||||
ans = d.get('Answer') or []
|
||||
print(' [%s] Status=%s' % (label, d.get('Status')))
|
||||
if not ans:
|
||||
print(' (无 Answer)')
|
||||
for a in ans:
|
||||
print(' type=%-6s %s' % (a.get('type'), str(a.get('data'))[:90]))
|
||||
except Exception as e:
|
||||
print(' [%s] 查询失败: %s' % (label, e))
|
||||
|
||||
|
||||
def main():
|
||||
zone = sys.argv[1] if len(sys.argv) > 1 else 'ai1net.com'
|
||||
name = sys.argv[2] if len(sys.argv) > 2 else '_acme-challenge.' + zone
|
||||
tok = token(zone)
|
||||
|
||||
zid = ''
|
||||
for z in (req('GET', '/zones?name=' + zone, tok).get('result') or []):
|
||||
zid = z['id']
|
||||
if zid == '':
|
||||
sys.exit('zone 不在权限内')
|
||||
|
||||
print('=== 创建临时 TXT: %s ===' % name)
|
||||
created = req('POST', '/zones/%s/dns_records' % zid, tok,
|
||||
{'type': 'TXT', 'name': name, 'content': 'probe-test-value-12345', 'ttl': 120})
|
||||
if not created.get('success'):
|
||||
print(' 创建失败:', created.get('errors'))
|
||||
return
|
||||
rid = created['result']['id']
|
||||
print(' 已创建 id=%s' % rid)
|
||||
|
||||
try:
|
||||
for wait in (3, 10, 30):
|
||||
time.sleep(wait if wait == 3 else wait - 3)
|
||||
print('\n=== 等待累计 ~%ds 后查询 ===' % wait)
|
||||
show('TXT 查询', name, 'TXT')
|
||||
finally:
|
||||
print('\n=== 清理临时记录 ===')
|
||||
d = req('DELETE', '/zones/%s/dns_records/%s' % (zid, rid), tok)
|
||||
print(' 删除成功:', d.get('success'))
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,165 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""docs-archive-index.py — 让 INDEX.md 的**档案清单表**变成派生件(根治"漏登记")
|
||||
|
||||
背景(2026-09-14 实测):档案清单**靠手写**,已漏 82–88 共 7 篇;而 `docs-manifest.py`
|
||||
已经能机读全部档案(号/状态/tier/域/tldr)。本脚本把两者接上:
|
||||
docs-manifest.json ──┐
|
||||
archive-summaries.json ─┴─→ INDEX.md §二 的档案表(原地替换,不搬家)
|
||||
|
||||
设计要点
|
||||
1. **不丢手写内容**:首次运行会把现有表里手写的「一句话」**抽取到 `archive-summaries.json`**
|
||||
(人工可编辑的映射文件);之后表的摘要优先级 = summaries → manifest.tldr → 标题。
|
||||
2. **原地替换**:只替换 `| 04-NN | … |` 那一段连续行,**表头与根级编号行(01/02/03/06)不动**;
|
||||
块边界用 BEGIN/END 注释标记,第二次起按标记整块重生成。
|
||||
3. **缺失可见**:状态取不到显 `❓`、摘要取不到显 `—` —— 让"没写好头部"的档案**在表里看得见**。
|
||||
|
||||
用法:
|
||||
python3 07-scripts/docs-archive-index.py # 只打印(不写任何文件)
|
||||
python3 07-scripts/docs-archive-index.py --write # 写 archive-summaries.json + INDEX.md
|
||||
退出码:0 = 一致或已刷新;1 = 有差异且未加 --write;2 = 结构异常
|
||||
"""
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
ROOT = sys.argv[1] if len(sys.argv) > 1 and not sys.argv[1].startswith('-') else \
|
||||
os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
INDEX = os.path.join(ROOT, 'INDEX.md')
|
||||
SUMS = os.path.join(ROOT, 'archive-summaries.json')
|
||||
MANIFEST = os.path.join(ROOT, 'docs-manifest.json')
|
||||
|
||||
B = '<!-- BEGIN archive-index (generated by 07-scripts/docs-archive-index.py — 勿手改) -->'
|
||||
E = '<!-- END archive-index -->'
|
||||
HEADER = ['| 号 | 状态 | 一句话(**机器生成**;摘要存 `archive-summaries.json`)|', '|---|---|---|']
|
||||
ARCH_RE = re.compile(r'^\|\s*04-(\d{1,3}[a-z]?)\s*\|\s*(\S+)\s*\|\s*(.*?)\s*\|\s*$')
|
||||
ROW_RE = re.compile(r'^\|\s*04-(\d{1,3}[a-z]?)\s*\|') # 宽松:也认手写表的坏行(缺尾竖线/带 CR)
|
||||
|
||||
|
||||
def rd(path):
|
||||
try:
|
||||
return io.open(path, encoding='utf-8', newline='').read()
|
||||
except OSError:
|
||||
return ''
|
||||
|
||||
|
||||
def sort_key(num):
|
||||
return (int(re.sub(r'\D', '', num) or 0), num)
|
||||
|
||||
|
||||
def harvest(index_text, existing):
|
||||
"""把现有表里手写的「一句话」抽进 summaries(只补空缺,不覆盖已有)。"""
|
||||
out = dict(existing)
|
||||
for line in index_text.split('\n'):
|
||||
m = ARCH_RE.match(line)
|
||||
if m and m.group(1) not in out:
|
||||
out[m.group(1)] = m.group(3)
|
||||
return out
|
||||
|
||||
|
||||
def body(manifest, sums):
|
||||
rows = {}
|
||||
for i in sorted([x for x in manifest['items'] if x.get('num')], key=lambda x: sort_key(x['num'])):
|
||||
num = i['num']
|
||||
status = i['status'] if i['status'] not in ('?', '') else '❓'
|
||||
text = (sums.get(num) or i.get('tldr') or
|
||||
re.sub(r'^\d{1,3}[a-z]?-', '', i.get('title') or '').strip() or '—')
|
||||
rows[num] = '| 04-%s | %s | %s |' % (num, status, text[:110])
|
||||
return rows
|
||||
|
||||
|
||||
def splice(index_text, gen):
|
||||
"""逐行替换(表里 04-* 行与 `—`/根级行是交错的),并按号补插缺失档案。"""
|
||||
lines = index_text.split('\n')
|
||||
nums = sorted(gen, key=sort_key)
|
||||
out, used, dup = [], set(), []
|
||||
|
||||
def emit_before(limit):
|
||||
for n in nums:
|
||||
if n not in used and (limit is None or sort_key(n) < sort_key(limit)):
|
||||
out.append(gen[n])
|
||||
used.add(n)
|
||||
|
||||
for line in lines:
|
||||
m = ROW_RE.match(line)
|
||||
if m is None:
|
||||
out.append(line)
|
||||
continue
|
||||
emit_before(m.group(1))
|
||||
if m.group(1) in gen:
|
||||
if m.group(1) not in used: # 首次出现 → 用生成行
|
||||
out.append(gen[m.group(1)])
|
||||
used.add(m.group(1))
|
||||
else: # 重复的手写行 → 丢弃(自动去重)
|
||||
dup.append((m.group(1), line[:60]))
|
||||
else:
|
||||
out.append(line) # 表里独有的号(如空号)→ 保留
|
||||
emit_before(None)
|
||||
if dup:
|
||||
print(' 自动丢弃重复行 %d 条:%s' % (len(dup), [d[0] for d in dup]))
|
||||
for n in nums:
|
||||
if n not in used:
|
||||
out.append(gen[n])
|
||||
return '\n'.join(out)
|
||||
|
||||
|
||||
def main():
|
||||
if not os.path.exists(MANIFEST):
|
||||
raise SystemExit('ERROR: 先跑 07-scripts/docs-manifest.py 生成 docs-manifest.json')
|
||||
|
||||
# ── 顺序断言(2026-09-14 加):派生链 manifest → 本脚本,顺序错会**静默**产出新旧混合 ──
|
||||
newest, _ad = 0.0, os.path.join(ROOT, '04-调整方案')
|
||||
if os.path.isdir(_ad):
|
||||
for _n in os.listdir(_ad):
|
||||
if _n.endswith('.md'):
|
||||
try:
|
||||
newest = max(newest, os.path.getmtime(os.path.join(_ad, _n)))
|
||||
except OSError:
|
||||
pass
|
||||
stale_min = (newest - os.path.getmtime(MANIFEST)) / 60.0
|
||||
if stale_min > 1:
|
||||
msg = ('⚠️ 顺序警告:docs-manifest.json 比 04-调整方案/ 最新档案旧 %.0f 分钟 ⇒ 先跑 '
|
||||
'07-scripts/docs-manifest.py,否则本表用的是旧数据' % stale_min)
|
||||
if '--write' in sys.argv and '--force' not in sys.argv:
|
||||
raise SystemExit(msg + '\n (确认要带旧数据刷新就加 --force)')
|
||||
print(msg)
|
||||
manifest = json.loads(rd(MANIFEST))
|
||||
index_text = rd(INDEX)
|
||||
old = {}
|
||||
if os.path.exists(SUMS):
|
||||
try:
|
||||
old = json.loads(rd(SUMS))
|
||||
except ValueError:
|
||||
raise SystemExit('ERROR: archive-summaries.json 不是合法 JSON')
|
||||
sums = harvest(index_text, old)
|
||||
rows = body(manifest, sums)
|
||||
new_text = splice(index_text, rows)
|
||||
changed = new_text != index_text or sums != old
|
||||
n_sum = sum(1 for k in rows if sums.get(k))
|
||||
_miss = sorted([k for k, v in rows.items() if '❓' in v], key=sort_key)
|
||||
print('档案 %d 篇 | 摘要:手写/已存 %d | 机器兜底 %d | 状态缺失(❓) %d(%.0f%%)'
|
||||
% (len(rows), n_sum, len(rows) - n_sum, len(_miss),
|
||||
100.0 * len(_miss) / max(1, len(rows))))
|
||||
if _miss:
|
||||
print(' 缺失名单:%s' % ', '.join('04-' + m for m in _miss))
|
||||
if len(_miss) / max(1, len(rows)) > 0.10:
|
||||
print(' ⚠️ 缺失率 >10%% ⇒ **新档案**头部必须写「- 状态:…」;历史档案按「只增不改」不回改正文,'
|
||||
'可在文末「修正(YYYY-MM-DD)」节补一行状态 ⇒ 下一轮由 manifest 从头部取到')
|
||||
if '--write' in sys.argv:
|
||||
io.open(SUMS, 'w', encoding='utf-8', newline='\n').write(
|
||||
json.dumps(dict(sorted(sums.items(), key=lambda kv: sort_key(kv[0]))),
|
||||
ensure_ascii=False, indent=1) + '\n')
|
||||
if new_text != index_text:
|
||||
io.open(INDEX, 'w', encoding='utf-8', newline='').write(new_text)
|
||||
print('已刷新 INDEX.md 档案表 + archive-summaries.json')
|
||||
else:
|
||||
print('INDEX.md 已是最新(仅刷新 archive-summaries.json)')
|
||||
return 0
|
||||
print('(只读模式)表内容与 INDEX.md %s' % ('一致' if not changed else '不一致,加 --write 刷新'))
|
||||
return 1 if changed else 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,181 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
docs-audit.py — 文档库质量扫描(只读,零副作用)
|
||||
|
||||
用途:一次性回答「文档是否清晰、无歧义、需要精简」这类问题,输出**可判定**的问题清单,
|
||||
避免靠人逐篇读。用于文档库根目录,也可用于任何 md 目录。
|
||||
|
||||
检查项(均可机器判定):
|
||||
1 档案编号冲突(同一编号被多份档案占用)
|
||||
2 标题号 ≠ 文件名号
|
||||
3 备份/临时残留(.bak / .orig / ~ 等,会污染对账与阅读)
|
||||
4 术语与事实漂移(旧术语、旧域名、已废弃表名)——历史档案保留原文属正常,需在入口说明
|
||||
5 体量分布(超长文件 → 考虑拆分;过短文件 → 考虑合并/指针)
|
||||
6 交叉引用有效性(引用「档案 NN」/「04-调整方案/NN-」是否存在)
|
||||
7 疑似重复/近重复(同一主题两处维护;含"子集包含"识别)
|
||||
8 元信息规范(档案头部是否含 日期 / 状态)
|
||||
9 非 md 文件的入库合理性提示
|
||||
|
||||
用法:
|
||||
python3 07-scripts/docs-audit.py [文档库根目录,默认取本脚本的上一级]
|
||||
|
||||
退出码:0 = 无 P0 级问题;1 = 发现编号冲突或失效引用(便于接 CI)。
|
||||
"""
|
||||
import io, os, re, sys, collections
|
||||
|
||||
ROOT = sys.argv[1] if len(sys.argv) > 1 else os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
P0 = 0
|
||||
|
||||
def rd(rel):
|
||||
return io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
|
||||
|
||||
def all_files():
|
||||
out = []
|
||||
for base, dirs, names in os.walk(ROOT):
|
||||
if '.git' in dirs:
|
||||
dirs.remove('.git')
|
||||
for n in names:
|
||||
out.append(os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/'))
|
||||
return sorted(out)
|
||||
|
||||
FILES = all_files()
|
||||
MD = [f for f in FILES if f.endswith('.md')]
|
||||
OTHER = [f for f in FILES if not f.endswith('.md')]
|
||||
|
||||
print('=' * 72)
|
||||
print('文档库质量扫描 root=%s' % ROOT)
|
||||
print('总文件 %d(md %d / 其他 %d)' % (len(FILES), len(MD), len(OTHER)))
|
||||
print('=' * 72)
|
||||
|
||||
# ── 1&2 编号 ──────────────────────────────────────────────
|
||||
num_map = collections.defaultdict(list)
|
||||
mismatch = []
|
||||
titles = {}
|
||||
for f in MD:
|
||||
s = rd(f)
|
||||
title = next((l.strip() for l in s.split('\n') if l.strip().startswith('#')), '')
|
||||
titles[f] = title
|
||||
mf = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
|
||||
mt = re.match(r'^#\s*(?:调整方案\s*)?(\d+[a-z]?)\s*[·、..\-]', title)
|
||||
if mf and ("/" not in f or f.startswith("04-调整方案/")):
|
||||
# ⛔ 只认「根级 NN-x.md」与「04-调整方案/NN-x.md」两套体系;
|
||||
# 08-skills/**、01-规范/** 等目录下的编号文件(如 07-并行调度详解.md)不是档案
|
||||
num_map[mf.group(1)].append(f)
|
||||
if mf and mt and mf.group(1) != mt.group(1):
|
||||
mismatch.append((f, title[:70]))
|
||||
|
||||
print('\n【1】档案编号冲突')
|
||||
conf = {n: fs for n, fs in num_map.items() if len(fs) > 1}
|
||||
if not conf:
|
||||
print(' ✓ 无冲突')
|
||||
for n, fs in sorted(conf.items()):
|
||||
# 顶层 NN-x.md 与 04-调整方案/NN-x.md 属两套体系,不算冲突
|
||||
top = [x for x in fs if '/' not in x]
|
||||
arch = [x for x in fs if x.startswith('04-调整方案/')]
|
||||
if top and arch and len(fs) == 2:
|
||||
print(' · 编号 %s:根级 vs 调整方案(**两套体系,非冲突**,但入口须写明)' % n)
|
||||
continue
|
||||
P0 = 1
|
||||
print(' ⚠ 编号 %s 被 %d 份档案占用:' % (n, len(fs)))
|
||||
for x in fs:
|
||||
print(' %-56s 标题「%s」' % (x, titles[x][:44]))
|
||||
|
||||
print('\n【2】标题号 ≠ 文件名号')
|
||||
if not mismatch:
|
||||
print(' ✓ 无')
|
||||
for f, t in mismatch:
|
||||
P0 = 1
|
||||
print(' ⚠ %-52s → 「%s」' % (f, t))
|
||||
|
||||
# ── 3 备份残留 ───────────────────────────────────────────
|
||||
print('\n【3】备份/临时残留')
|
||||
junk = [f for f in FILES if re.search(r'\.bak|~$|\.orig$|\.tmp$|\.swp$|\.new$', f)]
|
||||
print(' 数量 %d %s' % (len(junk), '(建议清理或纳入 .gitignore)' if junk else ''))
|
||||
for f in junk[:15]:
|
||||
print(' -', f)
|
||||
|
||||
# ── 4 术语/事实漂移 ─────────────────────────────────────
|
||||
print('\n【4】术语与事实漂移(历史档案保留原文=正常,需入口说明)')
|
||||
TERMS = {'业务插件(旧术语)': '业务插件', 'dsh.alotbuy.com(旧域名)': 'dsh.alotbuy.com',
|
||||
'folder_plugins(已废弃)': 'folder_plugins'}
|
||||
for label, k in TERMS.items():
|
||||
fs = [f for f in MD if k in rd(f)]
|
||||
print(' %-26s %d 个文件' % (label, len(fs)))
|
||||
if fs and len(fs) <= 6:
|
||||
print(' %s' % ', '.join(fs))
|
||||
|
||||
# ── 5 体量 ──────────────────────────────────────────────
|
||||
print('\n【5】体量分布(>300 行考虑拆分;<15 行考虑合并或指针化)')
|
||||
rows = sorted(((rd(f).count('\n') + 1, len(rd(f).encode()), f) for f in MD), reverse=True)
|
||||
print(' 最长 8:')
|
||||
for ln, by, f in rows[:8]:
|
||||
print(' %5d 行 %7.1f KB %s' % (ln, by / 1024, f))
|
||||
print(' 最短 5:')
|
||||
for ln, by, f in rows[-5:]:
|
||||
print(' %5d 行 %7.1f KB %s' % (ln, by / 1024, f))
|
||||
|
||||
# ── 6 引用有效性 ────────────────────────────────────────
|
||||
print('\n【6】交叉引用有效性')
|
||||
existing = set(num_map.keys())
|
||||
bad = collections.defaultdict(list)
|
||||
for f in MD:
|
||||
s = rd(f)
|
||||
for m in re.finditer(r'档案\s*(\d{1,2}[a-z]?)', s):
|
||||
n = m.group(1)
|
||||
# 键已改为字符串(支持 '37a');原先的数值范围判断 1<=n<=99 换成形状校验
|
||||
if re.fullmatch(r'\d{1,2}[a-z]?', n) and n not in existing:
|
||||
bad[f].append(n)
|
||||
if not bad:
|
||||
print(' ✓ 无悬空档案号引用')
|
||||
else:
|
||||
P0 = 1
|
||||
for f, ns in list(bad.items())[:12]:
|
||||
print(' %-48s 引用了不存在的档案号 %s' % (f, sorted(set(ns))))
|
||||
|
||||
# ── 7 近重复 ────────────────────────────────────────────
|
||||
print('\n【7】疑似重复 / 近重复')
|
||||
sig = {f: re.sub(r'\s+', '', rd(f)) for f in MD}
|
||||
found = False
|
||||
keys = list(sig)
|
||||
for i in range(len(keys)):
|
||||
for j in range(i + 1, len(keys)):
|
||||
a, b = sig[keys[i]], sig[keys[j]]
|
||||
if len(a) < 200 or len(b) < 200:
|
||||
continue
|
||||
if a == b:
|
||||
print(' %-44s = %-44s 完全相同' % (keys[i], keys[j])); found = True
|
||||
elif a in b or b in a:
|
||||
small, big = (keys[i], keys[j]) if len(a) < len(b) else (keys[j], keys[i])
|
||||
print(' %-44s ⊂ %-44s 子集(%d/%d 字符)' % (small, big, min(len(a), len(b)), max(len(a), len(b)))); found = True
|
||||
else:
|
||||
ga = set(a[k:k + 3] for k in range(0, len(a) - 2, 3))
|
||||
gb = set(b[k:k + 3] for k in range(0, len(b) - 2, 3))
|
||||
if ga and gb:
|
||||
r = len(ga & gb) / min(len(ga), len(gb))
|
||||
if r > 0.55:
|
||||
print(' %-44s ≈ %-44s 重合 %.0f%%' % (keys[i], keys[j], r * 100)); found = True
|
||||
if not found:
|
||||
print(' ✓ 未发现')
|
||||
|
||||
# ── 8 元信息 ────────────────────────────────────────────
|
||||
print('\n【8】档案头部元信息(04-调整方案/ 内)')
|
||||
no_date, no_status = [], []
|
||||
for f in MD:
|
||||
if not f.startswith('04-调整方案/'):
|
||||
continue
|
||||
head = '\n'.join(rd(f).split('\n')[:14])
|
||||
if not re.search(r'日期|20\d\d-\d\d-\d\d', head):
|
||||
no_date.append(f)
|
||||
if not re.search(r'状态', head):
|
||||
no_status.append(f)
|
||||
print(' 缺「日期」%d 个 %s' % (len(no_date), [os.path.basename(x) for x in no_date[:6]]))
|
||||
print(' 缺「状态」%d 个 %s' % (len(no_status), [os.path.basename(x) for x in no_status[:6]]))
|
||||
|
||||
# ── 9 非 md ─────────────────────────────────────────────
|
||||
print('\n【9】非 md 文件 %d 个(确认均属应入库的资产/脚本)' % len(OTHER))
|
||||
print(' ' + (', '.join(OTHER[:12]) + (' …' if len(OTHER) > 12 else '') if OTHER else '无'))
|
||||
|
||||
print('\n' + '=' * 72)
|
||||
print('结论:%s' % ('发现问题(见上 ⚠)' if P0 else '无 P0 级问题'))
|
||||
sys.exit(1 if P0 else 0)
|
||||
@@ -0,0 +1,161 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
docs-consistency.py —— 文档库「事实一致性」校验(只读,可复跑)
|
||||
|
||||
为什么需要它:`docs-audit.py` 查的是**结构问题**(编号冲突 / 悬空引用 / 重复子集),
|
||||
**不查事实是否与现状一致** → 过时值会一直躺在库里没人发现。
|
||||
|
||||
两类检查:
|
||||
|
||||
【1】写死的取值 —— 会随并行改动过期,应改为"复跑取号"
|
||||
· "下一号 = NN"(档案编号)
|
||||
· 旧本机身份 `maidou` / `/c/Users/<user>/.workbuddy`(已迁 `E:\\ProgramData\\.workbuddy`)
|
||||
|
||||
【2】跨页取值不一致 —— 同一个事实键在多个「承诺现行」文件里取值不同
|
||||
(首跑实证:`INDEX.md` 写下一号=72、`README.md` 写 69 → 同类事实两处打架)
|
||||
|
||||
判据:**「承诺现行」的文件不许出现已废止/写死/互相矛盾的取值。**
|
||||
- 承诺现行(必查):`BRIEF.md` / `INDEX.md` / `README.md` / `CODEBUDDY.md` / `DEPLOY-本部署.md`
|
||||
/ `01-规范/03-路线图与待办.md` / `01-规范/06-工作台UI规范.md` / `05-交接单/**` / `08-skills/**`
|
||||
- 豁免(历史事实,只增不改):`04-调整方案/**`、`09-archive/**`、`01-规范/01-规划与架构.md`、`01-规范/02-运维手册.md`
|
||||
—— 它们写的时候那个值是对的,回改反而破坏历史。
|
||||
|
||||
⚠️ **刻意不查**「旧域名 `dsh.alotbuy.com` / `alotbuy.com`」「旧配额 512M」:这些在库里几乎都是
|
||||
"旧域已 301" / "512M→384M" 的**合法历史表述**,正则无法可靠区分,误报率过高。
|
||||
改由人工在改域名/改配额时顺手核(档案 22 / 58 是权威;2026-09-19 域迁 `ai1net.com` 后旧域仅作过渡装置)。
|
||||
|
||||
用法:
|
||||
python3 07-scripts/docs-consistency.py
|
||||
退出码:0 = 无问题;1 = 有违背;2 = 环境错误
|
||||
"""
|
||||
import argparse
|
||||
import io
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from collections import defaultdict
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
CURRENT_PREFIXES = (
|
||||
'BRIEF.md', 'INDEX.md', 'README.md', 'CODEBUDDY.md', 'DEPLOY-本部署.md',
|
||||
'01-规范/03-路线图与待办.md', '01-规范/06-工作台UI规范.md', '05-交接单/', '08-skills/',
|
||||
)
|
||||
|
||||
# 【1】写死的取值:正则 → (说明, 修复建议)
|
||||
HARDCODED = [
|
||||
('档案「下一号」写死', r'下一号\s*[==]\s*\*{0,2}\d+', '改为"复跑取号,勿写死"(并行改动会打穿)'),
|
||||
('旧本机用户名 maidou', r'maidou', '现行 Administrator'),
|
||||
('旧技能/配置目录', r'[/\\][cC][/\\]Users[/\\][^/\\\s]+[/\\]\.workbuddy',
|
||||
'现行 E:\\ProgramData\\.workbuddy(已迁 E 盘)'),
|
||||
]
|
||||
|
||||
# 【2】跨页一致性:事实键 → 抽取正则(捕获组即取值)
|
||||
CROSS_FACTS = {
|
||||
'档案下一号': r'下一号\s*[==]\s*\*{0,2}(\d+)',
|
||||
'代码 HEAD': r'代码 HEAD\s*[` ]?\s*\*{0,2}([0-9a-f]{7,10})',
|
||||
'实例 MemoryMax': r'MemoryMax\s*[==]?\s*\*{0,2}(\d{3,4})\s*(?:MiB|M\b)?',
|
||||
}
|
||||
|
||||
|
||||
def read(p):
|
||||
try:
|
||||
return io.open(p, encoding='utf-8', errors='replace').read()
|
||||
except Exception:
|
||||
return ''
|
||||
|
||||
|
||||
def collect():
|
||||
out = []
|
||||
for root, dirs, files in os.walk(ROOT):
|
||||
dirs[:] = [d for d in dirs if d not in ('.git', 'node_modules', '__pycache__')]
|
||||
for f in files:
|
||||
if f.endswith(('.md', '.py', '.sh', '.cjs', '.mjs', '.js')):
|
||||
out.append(os.path.relpath(os.path.join(root, f), ROOT).replace('\\', '/'))
|
||||
return sorted(out)
|
||||
|
||||
|
||||
def is_current(rel):
|
||||
return any(rel == p or rel.startswith(p) for p in CURRENT_PREFIXES)
|
||||
|
||||
|
||||
def is_quoted(line, start, end):
|
||||
"""匹配是否被包住 —— 包住的内容视为**引用/举例**(如 T02 记的"下一号 = 20"已归零、
|
||||
文档里把 `maidou` 当反例引用),不是当前断言,不算违规。
|
||||
|
||||
三类包裹:`" "` / `“ ”`(引文)与 `` ` ` ``(代码字面量)。"""
|
||||
pre = line[:start].rstrip()
|
||||
post = line[end:].lstrip()
|
||||
return bool(pre and pre[-1] in '"“”`\'') and bool(post and post[0] in '"“”`\'')
|
||||
|
||||
|
||||
def scan(rx, text):
|
||||
"""返回一个文件里「不在引号内」的匹配数。"""
|
||||
n = 0
|
||||
for line in text.splitlines():
|
||||
for m in rx.finditer(line):
|
||||
if is_quoted(line, m.start(), m.end()):
|
||||
continue
|
||||
n += 1
|
||||
return n
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.parse_args()
|
||||
|
||||
files = collect()
|
||||
cur = [f for f in files if is_current(f)]
|
||||
print('文档库根:%s' % ROOT)
|
||||
print('文件总数 %d | 承诺现行 %d 个(历史豁免 %d 个)\n' % (len(files), len(cur), len(files) - len(cur)))
|
||||
|
||||
bad = 0
|
||||
|
||||
print('──【1】写死的取值 ──')
|
||||
for label, pat, fix in HARDCODED:
|
||||
rx = re.compile(pat)
|
||||
hits = [(r, scan(rx, read(os.path.join(ROOT, r)))) for r in cur]
|
||||
hits = [h for h in hits if h[1]]
|
||||
print(' 【%s】→ %s' % (label, fix))
|
||||
if not hits:
|
||||
print(' ✓ 无')
|
||||
else:
|
||||
bad += len(hits)
|
||||
for r, n in sorted(hits, key=lambda x: -x[1])[:8]:
|
||||
print(' ⚠ %-52s %d 处' % (r[:52], n))
|
||||
if len(hits) > 8:
|
||||
print(' …还有 %d 个文件' % (len(hits) - 8))
|
||||
print()
|
||||
|
||||
print('──【2】跨页取值一致性 ──')
|
||||
for name, pat in CROSS_FACTS.items():
|
||||
rx = re.compile(pat)
|
||||
vals = defaultdict(list)
|
||||
for r in cur:
|
||||
txt = read(os.path.join(ROOT, r))
|
||||
for line in txt.splitlines():
|
||||
for m in rx.finditer(line):
|
||||
if is_quoted(line, m.start(), m.end()):
|
||||
continue
|
||||
vals[m.group(1)].append(r)
|
||||
print(' 【%s】' % name)
|
||||
if len(vals) <= 1:
|
||||
print(' ✓ 取值唯一:%s' % (list(vals)[0] if vals else '(未出现)'))
|
||||
else:
|
||||
bad += 1
|
||||
for v, rs in sorted(vals.items()):
|
||||
print(' ⚠ 取值 %s ← %s' % (v, ', '.join(sorted(set(rs))[:4])))
|
||||
print(' → 同一事实多处取值不一致,请校正为同一个权威值(或改为"复跑取号")')
|
||||
print()
|
||||
|
||||
print('=' * 64)
|
||||
if bad:
|
||||
print('结论:**%d 项需处理**' % bad)
|
||||
return 1
|
||||
print('结论:承诺现行的文件与现行值一致 ✓')
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,122 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""docs-dedupe.py — 巨型档案的**重复块体检 + 去重视图生成**(不改原文)
|
||||
|
||||
背景(2026-09-14 实测)
|
||||
* `04-调整方案/82-…md` = **2024 行 / 91 KB**,其中同一份「§一–§八」**重复 16 次**;
|
||||
* 但本库铁律是 **L5 冻结:历史档案不回改** ⇒ 不能"直接删重复"。
|
||||
⇒ 本工具遵守铁律:**只读原文 → 生成"去重视图"到库外 + 报告变体差异**;
|
||||
原文仅允许在**文末追加**「修正(YYYY-MM-DD)」小节(本库明文许可)。
|
||||
|
||||
做三件事
|
||||
1. **块级重复统计**:按 `#` / `##` / `###` 切块,归一化后哈希 ⇒ 报"哪些标题重复几次";
|
||||
2. **变体检测**:同一标题的多次出现若**内容不同**(hash 不同),逐个列出(避免"以为一样其实有改动");
|
||||
3. **去重视图**:写入 `--view-out`(默认库外 `.workbuddy/cache/dedupe-view/<名>.md`),
|
||||
每组保留**信息最全的那一份**(字符数最大,并列取首次),其余位置用一行占位注释替代。
|
||||
|
||||
用法
|
||||
python3 07-scripts/docs-dedupe.py <相对路径> # 只报告
|
||||
python3 07-scripts/docs-dedupe.py <相对路径> --view-out=路径 # 报告 + 写视图
|
||||
退出码:0 = 无重复;1 = 有重复(可生成视图);2 = 用法/文件错误
|
||||
"""
|
||||
import io
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import hashlib
|
||||
import collections
|
||||
|
||||
DOCS = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
WS = os.path.dirname(DOCS)
|
||||
HEAD = re.compile(r'^(#{1,3})\s+(.*)$')
|
||||
|
||||
|
||||
def blocks(lines):
|
||||
"""按 #/##/### 切块;返回 [(起始行, 级别, 标题, 正文行列表)],含块前导内容。"""
|
||||
out, cur = [], None
|
||||
pre = []
|
||||
for i, line in enumerate(lines, 1):
|
||||
m = HEAD.match(line)
|
||||
if m:
|
||||
if cur is None:
|
||||
cur = [i, len(m.group(1)), m.group(2).strip(), list(pre)]
|
||||
else:
|
||||
out.append(cur + [i - 1])
|
||||
cur = [i, len(m.group(1)), m.group(2).strip(), []]
|
||||
elif cur is None:
|
||||
pre.append(line)
|
||||
else:
|
||||
cur[3].append(line)
|
||||
if cur is not None:
|
||||
out.append(cur + [len(lines)])
|
||||
return out, pre
|
||||
|
||||
|
||||
def norm(level, title, body):
|
||||
text = re.sub(r'\s+', ' ', '\n'.join(body)).strip()
|
||||
return hashlib.sha1(('%d|%s|%s' % (level, title, text)).encode('utf-8')).hexdigest()[:12]
|
||||
|
||||
|
||||
def main(argv):
|
||||
files = [a for a in argv if not a.startswith('--')]
|
||||
if not files:
|
||||
print(__doc__)
|
||||
return 2
|
||||
rel = files[0]
|
||||
path = os.path.join(DOCS, rel)
|
||||
if not os.path.exists(path):
|
||||
print('ERROR: 找不到 %s' % path)
|
||||
return 2
|
||||
raw = io.open(path, encoding='utf-8', newline='').read()
|
||||
lines = raw.split('\n')
|
||||
bs, pre = blocks(lines)
|
||||
|
||||
groups = collections.defaultdict(list)
|
||||
for start, level, title, body, end in bs:
|
||||
groups[(level, title, norm(level, title, body))].append((start, end, body))
|
||||
|
||||
by_title = collections.defaultdict(list)
|
||||
for (level, title, h), occ in groups.items():
|
||||
by_title[title].append((h, occ))
|
||||
|
||||
dup_titles = {t: v for t, v in by_title.items() if sum(len(o) for _, o in v) > 1}
|
||||
total_dup_blocks = sum(len(o) - 1 for v in dup_titles.values() for _, o in v)
|
||||
print('文件 %s:%d 行 / %d 字符|块 %d 个|**重复标题 %d 个,冗余块 %d 个**'
|
||||
% (rel, len(lines), len(raw), len(bs), len(dup_titles), total_dup_blocks))
|
||||
for title, variants in sorted(dup_titles.items(),
|
||||
key=lambda kv: -sum(len(o) for _, o in kv[1]))[:12]:
|
||||
times = sum(len(o) for _, o in variants)
|
||||
vinfo = '|'.join('变体%d×%d次' % (i + 1, len(o)) for i, (_, o) in enumerate(variants))
|
||||
mark = ' ⚠️内容有差异' if len(variants) > 1 else ''
|
||||
print(' %2d× %-46s %s%s' % (times, title[:44], vinfo, mark))
|
||||
|
||||
view_out = next((a.split('=', 1)[1] for a in argv if a.startswith('--view-out=')), None)
|
||||
if view_out is None:
|
||||
default = os.path.join(WS, '.workbuddy', 'cache', 'dedupe-view',
|
||||
rel.replace('/', '__'))
|
||||
view_out = default if '--view-out' in str(argv) else None
|
||||
if view_out:
|
||||
keep = {}
|
||||
for title, variants in dup_titles.items():
|
||||
allocc = [(h, start, end, body) for h, o in variants for start, end, body in o]
|
||||
allocc.sort(key=lambda x: (-sum(len(l) + 1 for l in x[3]), x[1]))
|
||||
keep[title] = allocc[0]
|
||||
out, skipped = [], 0
|
||||
for start, level, title, body, end in bs:
|
||||
cur = (start, end, body)
|
||||
keeper = keep.get(title)
|
||||
if keeper is not None and (start, end, body) != (keeper[1], keeper[2], keeper[3]):
|
||||
skipped += 1
|
||||
out.append('%s<!-- 去重省略:同「%s」块(原文 L%d–L%d),内容与 L%d–L%d 那份一致 -->'
|
||||
% ('#' * level + ' ', title, start, end, keeper[1], keeper[2]))
|
||||
else:
|
||||
out.append('\n'.join(['#' * level + ' ' + title] + body))
|
||||
os.makedirs(os.path.dirname(view_out), exist_ok=True)
|
||||
io.open(view_out, 'w', encoding='utf-8', newline='\n').write('\n'.join(out) + '\n')
|
||||
print('\n去重视图已写:%s(省略 %d 块,%.1f KB → %.1f KB)'
|
||||
% (view_out, skipped, len(raw) / 1024, os.path.getsize(view_out) / 1024))
|
||||
return 1 if dup_titles else 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
raise SystemExit(main(sys.argv[1:]))
|
||||
@@ -0,0 +1,130 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""docs-index-stats.py — 从 INDEX.md §二 清单表**算**出状态分布,并可就地刷新「状态摘要」行。
|
||||
|
||||
为什么需要它
|
||||
────────────
|
||||
「状态摘要」原本是**人工手写**的硬数字,档案一多必然漂移(2026-09-13 实测:摘要写「档案 72 份」,
|
||||
而表格实际 76 行)。本脚本把摘要变成**机器生成**:读表 → 统计 → 打印;`--write` 时回写该行,
|
||||
并顺带做**图例自检**;**完整保持文件原有行尾**(CRLF 文件不会被转成 LF)(表格用到的图标必须在「图例」行里有定义,缺则补)。
|
||||
|
||||
用法
|
||||
────
|
||||
python3 07-scripts/docs-index-stats.py # 只打印 + 一致性判定
|
||||
python3 07-scripts/docs-index-stats.py --write # 就地刷新「状态摘要」行(必要时补图例)
|
||||
退出码:0 = 一致(或已刷新);1 = 漂移且未加 --write;2 = 结构异常。
|
||||
"""
|
||||
|
||||
import io
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import collections
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
INDEX = os.path.join(ROOT, "INDEX.md")
|
||||
|
||||
ORDER = ["✅", "🔄", "🧪", "📝", "🔍", "📋", "🟡", "🗄", "🔧", "⚠️"]
|
||||
LEGEND_OF = {
|
||||
"✅": "已落地", "🔄": "维护中", "🧪": "PoC", "📝": "待开发", "🔍": "核查完成",
|
||||
"📋": "评估", "🟡": "保留兜底", "🗄": "归档", "🔧": "修复", "⚠️": "警示",
|
||||
}
|
||||
HEADER_RE = re.compile(r"^\|\s*号\s*\|\s*状态\s*\|")
|
||||
LEAF_RE = re.compile(r"^0\d$")
|
||||
|
||||
|
||||
def parse(index_path):
|
||||
"""返回 (lines, arch_rows, leaf_rows, counter, other_count)。"""
|
||||
# ⚠️ 必须 newline=""(通用换行会把 CRLF 静默转成 LF —— 2026-09-13 踩过两次)
|
||||
raw = io.open(index_path, encoding="utf-8", newline="").read()
|
||||
lines = raw.split("\n")
|
||||
try:
|
||||
start = next(i for i, l in enumerate(lines) if HEADER_RE.match(l))
|
||||
except StopIteration:
|
||||
raise SystemExit("ERROR: INDEX.md 里找不到「| 号 | 状态 | 一句话 |」表头")
|
||||
arch, leaf, other = [], [], 0
|
||||
for l in lines[start + 2:]:
|
||||
if not l.startswith("|"):
|
||||
break
|
||||
cells = [x.strip() for x in l.split("|")]
|
||||
if len(cells) < 4 or not cells[1]:
|
||||
continue
|
||||
no, st = cells[1], cells[2]
|
||||
if no.startswith("04-"):
|
||||
arch.append((no, st))
|
||||
elif LEAF_RE.match(no):
|
||||
leaf.append((no, st))
|
||||
else:
|
||||
other += 1
|
||||
cnt = collections.Counter(st for _, st in arch)
|
||||
cnt.update(st for _, st in leaf)
|
||||
eol = "\r\n" if raw.count("\r\n") > 0 else "\n"
|
||||
return lines, arch, leaf, cnt, other, eol
|
||||
|
||||
|
||||
def summary_text(cnt, n_arch, leaf_names, other):
|
||||
parts = ["%s %d" % (k, cnt[k]) for k in ORDER if cnt.get(k)]
|
||||
unmarked = sum(v for k, v in cnt.items() if k not in ORDER)
|
||||
if unmarked:
|
||||
parts.append("未标记 %d" % unmarked)
|
||||
leaf = (",另含根级编号 %d 条(%s)" % (len(leaf_names), "/".join(leaf_names))) if leaf_names else ""
|
||||
return (
|
||||
"> **状态摘要**(**机器生成,勿手改**):档案 **%d** 份(`04-*`)%s,"
|
||||
"另有非编号行 %d 条(README / INDEX / 技能 / poc 等)—— %s。"
|
||||
"复跑 `python3 07-scripts/docs-index-stats.py` 取数,`--write` 就地刷新本行。"
|
||||
% (n_arch, leaf, other, " | ".join(parts))
|
||||
)
|
||||
|
||||
|
||||
def main():
|
||||
write = "--write" in sys.argv
|
||||
lines, arch, leaf, cnt, other, eol = parse(INDEX)
|
||||
leaf_names = [n for n, _ in leaf]
|
||||
new = summary_text(cnt, len(arch), leaf_names, other)
|
||||
|
||||
print("档案 04-* %d 份 | 根级编号 %s | 非编号行 %d 条" % (len(arch), "/".join(leaf_names) or "—", other))
|
||||
for k in ORDER:
|
||||
if cnt.get(k):
|
||||
print(" %s %-6s %d" % (k, LEGEND_OF.get(k, ""), cnt[k]))
|
||||
unmarked = sum(v for k, v in cnt.items() if k not in ORDER)
|
||||
if unmarked:
|
||||
print(" 未标记 %d" % unmarked)
|
||||
|
||||
changed = False
|
||||
|
||||
# ── 图例自检 ─────────────────────────────────────────────────────────
|
||||
li = [i for i, l in enumerate(lines) if l.startswith("> 图例:")]
|
||||
if li:
|
||||
legend = lines[li[0]]
|
||||
missing = [k for k in cnt if k not in legend]
|
||||
if missing:
|
||||
print("⚠️ 图例缺图标:%s" % " ".join(missing))
|
||||
if write:
|
||||
lines[li[0]] = legend.rstrip().rstrip("|") + "".join(
|
||||
"|%s%s" % (k, LEGEND_OF.get(k, "")) for k in missing
|
||||
)
|
||||
changed = True
|
||||
print("✓ 已补进图例行")
|
||||
|
||||
# ── 摘要行 ───────────────────────────────────────────────────────────
|
||||
idx = [i for i, l in enumerate(lines) if l.startswith("> **状态摘要**")]
|
||||
if not idx:
|
||||
print("ERROR: 未找到「状态摘要」行", file=sys.stderr)
|
||||
return 2
|
||||
old = lines[idx[0]]
|
||||
if old.strip() == new.strip() and not changed:
|
||||
print("✓ 摘要与表格一致")
|
||||
return 0
|
||||
if not write:
|
||||
print("✗ 摘要与表格不一致(加 --write 刷新)")
|
||||
print(" 旧: %s" % old.strip()[:90])
|
||||
print(" 新: %s" % new.strip()[:90])
|
||||
return 1
|
||||
lines[idx[0]] = new
|
||||
io.open(INDEX, "w", encoding="utf-8", newline="").write(eol.join(lines))
|
||||
print("✓ 已刷新 INDEX.md(摘要行%s)" % ("+图例" if changed else ""))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,201 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
docs-manifest.py — 生成文档库**机读清单**(供 AI/脚本过滤,替代"读大表格")
|
||||
|
||||
产出:
|
||||
docs-manifest.json 每份文档:编号/标题/状态/日期/字符/行数/被引用次数/分层(tier)/主题标签
|
||||
控制台 体量总览 + 分层统计 + 「常读」与「历史」清单
|
||||
|
||||
分层规则(2026-09-14 变更:由"**被引几次**"改为"**谁在引**"):
|
||||
hot 被**现行层 L1/L2**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范)反复引用 ≥3 次 —— 改东西前大概率要看
|
||||
cur 被现行层引用 1–2 次 —— 相关即看(单次引用也可能极重要,如红线类档案 07)
|
||||
warm 仅被历史档案互引(全库 ≥1,现行层 0)—— 参考
|
||||
cold 零引用 —— 历史,只在追溯时看
|
||||
doc 根级长期文档 / 资产,不参与档案分层
|
||||
⚠️ 实测教训:**L4(INDEX / 待办 / 台账)不能算现行层** —— 它们会顺带列出几乎所有档案号,
|
||||
会让判据反向失效(第一版含 L4 时 hot 从 44 抬到 60)。
|
||||
|
||||
用法:python3 07-scripts/docs-manifest.py [文档库根目录]
|
||||
副作用:仅写 docs-manifest.json(其余只读)
|
||||
"""
|
||||
import io, os, re, sys, json, collections
|
||||
|
||||
|
||||
def norm_status(raw):
|
||||
"""规范化档案头部的状态行 → 表里可读的一小段(≤14 字,去 ** 与尾部括注)。
|
||||
|
||||
⚠️ **权威方向**:档案头部是**源**,INDEX 表是**派生展示** —— 不允许把表里的 ❓ 回灌成 manifest 的 status,
|
||||
否则 manifest → INDEX → manifest 形成环形锁定,状态永远修不回来(2026-09-14 实测 14 篇被锁)。
|
||||
"""
|
||||
t = re.sub(r'\*{1,2}', '', raw or '').strip()
|
||||
t = re.split(r'——|||\||;|;|\s{2,}', t)[0].strip()
|
||||
t = re.sub(r'([^)]*)\s*$', '', t).strip()
|
||||
return t[:14]
|
||||
|
||||
ROOT = sys.argv[1] if len(sys.argv) > 1 else os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
def rd(rel):
|
||||
try:
|
||||
return io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
|
||||
except OSError:
|
||||
return ''
|
||||
|
||||
MD = []
|
||||
for base, dirs, names in os.walk(ROOT):
|
||||
if '.git' in dirs:
|
||||
dirs.remove('.git')
|
||||
for n in names:
|
||||
if n.endswith('.md') and '.bak' not in n:
|
||||
MD.append(os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/'))
|
||||
MD.sort()
|
||||
|
||||
# ── 状态/日期:优先取 INDEX 表,其次取档案头部 ─────────────
|
||||
index_status = {}
|
||||
if os.path.exists(os.path.join(ROOT, 'INDEX.md')):
|
||||
for line in rd('INDEX.md').split('\n'):
|
||||
m = re.match(r'^\|\s*(?:04-)?(\d{1,3}[a-z]?)\s*\|', line)
|
||||
if not m:
|
||||
continue
|
||||
marks = [k for k in ('✅', '🔄', '🔧', '🧪', '🗄', '⏸', '❌') if k in line]
|
||||
dm = re.search(r'\b(\d{2}-\d{2})\b', line)
|
||||
index_status[m.group(1)] = {'status': marks[0] if marks else '?', 'date': dm.group(1) if dm else ''}
|
||||
|
||||
# ── 引用热度 ────────────────────────────────────────────
|
||||
DOMAIN_RULES = {
|
||||
'external': r'官方推荐|awesome|npm|cordis|开源导出|外部源|生态',
|
||||
'method': r'方法|决策|技能|skill|流程|工作流|协议|协作',
|
||||
'plugin': r'插件|univer|mcn|business-plugins|dsh-plugin|原生绑定',
|
||||
'ui': r'UI|前端|门户|页面|portal|界面|client bundle|渲染',
|
||||
'06-ops': r'运维|排障|重启|备份|配额|内存|OOM|证书|nginx|nft|执行锁|并发',
|
||||
'platform': r'平台|部署|隔离|bwrap|多租户|实例|profile|配额口径|架构|命名|重构|升级|耦合|回归清单|目标架构',
|
||||
'ops2': r'登录|401|404|竞态|冷启动|启动失败|排障|死锁',
|
||||
'ui2': r'会话提示|文案|提示条|覆盖层|面板|分区|交互',
|
||||
'method2': r'文档|信息架构|清单|机读|模板|规范|收尾|去重',
|
||||
}
|
||||
def domain_of(text):
|
||||
scores = collections.Counter()
|
||||
for name, pat in DOMAIN_RULES.items():
|
||||
scores[name] = len(re.findall(pat, text, re.I))
|
||||
best = scores.most_common(1)[0]
|
||||
if best[1] <= 0:
|
||||
return '?'
|
||||
return {'ops2': '06-ops', 'ui2': 'ui', 'method2': 'method'}.get(best[0], best[0])
|
||||
|
||||
def layer_of(path, num):
|
||||
if num: return 'L5'
|
||||
if path.startswith('08-skills/'): return 'L3'
|
||||
if path.startswith('05-交接单/'): return 'L4'
|
||||
if path in ('BRIEF.md',): return 'L1'
|
||||
if path == 'CODEBUDDY.md': return 'L2'
|
||||
if path == 'INDEX.md': return 'L4'
|
||||
if path == 'README.md': return 'L2'
|
||||
if path == 'DEPLOY-本部署.md': return 'L1' # 现行部署事实
|
||||
if path == '01-规范/06-工作台UI规范.md': return 'L2' # 强制基线 = 规则
|
||||
if path == '01-规范/03-路线图与待办.md': return 'L4' # 状态/待办
|
||||
if path in ('BRIEF.md',): return 'L1'
|
||||
if path.startswith('docs-manifest') or path.endswith('.json'): return 'L2'
|
||||
if path.startswith('01-规范/01-规划与架构'): return 'L0'
|
||||
if path.startswith('01-规范/02-运维手册') or path.startswith('09-archive/'): return 'L5'
|
||||
if path == '04-调整方案/README.md': return 'L5' # 目录内索引 ⇒ 随其层
|
||||
return '?'
|
||||
|
||||
VIEW_DIR = os.path.join(os.path.dirname(ROOT), '.workbuddy', 'cache', 'dedupe-view')
|
||||
|
||||
|
||||
def view_field(rel):
|
||||
"""库外的"去重视图"(docs-dedupe.py 产物)—— 有则记路径,供检索/阅读优先使用。"""
|
||||
cand = os.path.join(VIEW_DIR, rel.replace('/', '__'))
|
||||
return {'dedupeView': cand} if os.path.exists(cand) else {}
|
||||
|
||||
|
||||
def num_of(f):
|
||||
m = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
|
||||
return m.group(1) if (m and f.startswith('04-调整方案/') and '/poc/' not in f) else None
|
||||
|
||||
# 引用分两份:**全库** 与 **仅 L1+L2 层**(BRIEF / CODEBUDDY / DEPLOY / 06-UI规范 = 现行事实与规则)。
|
||||
# ⚠️ 2026-09-14 实测教训:**L4(INDEX / 待办 / 台账)不能算** —— 它们会顺带列出几乎所有档案号,
|
||||
# 导致 hot 从 44 抬到 60(判据反而变糟);"被台账提到" ≠ "改东西前要看"。
|
||||
# 2026-09-14 判据变更:tier 由"被引几次"改为"**谁在引**" —— 只有被现行层引用才算 hot。
|
||||
ref = collections.Counter()
|
||||
ref_current = collections.Counter()
|
||||
for f in MD:
|
||||
lay = layer_of(f, num_of(f))
|
||||
for m in re.finditer(r'档案\s*(\d{1,2}[a-z]?)', rd(f)):
|
||||
ref[m.group(1)] += 1
|
||||
if lay in ('L1', 'L2'):
|
||||
ref_current[m.group(1)] += 1
|
||||
|
||||
# ── 域标签(检索键;未命中记 '?',由人补规则)─────────────
|
||||
items = []
|
||||
for f in MD:
|
||||
s = rd(f)
|
||||
title = next((l.strip().lstrip('#').strip() for l in s.split('\n') if l.strip().startswith('#')), '')
|
||||
mnum = re.match(r'^(\d+[a-z]?)-', os.path.basename(f))
|
||||
num = mnum.group(1) if (mnum and f.startswith('04-调整方案/') and '/poc/' not in f) else None
|
||||
head = '\n'.join(s.split('\n')[:14])
|
||||
m_tl = re.search(r'>\s*\*\*TL;DR\*\*[||::]?\s*(.+)', s)
|
||||
tldr = re.sub(r'\s+', ' ', m_tl.group(1)).strip(' ||')[:120] if m_tl else ''
|
||||
hd = re.search(r'(- 日期:\s*)([0-9]{4}-[0-9]{2}-[0-9]{2})', head)
|
||||
hs = re.search(r'状态\*{0,2}\s*[::]\s*(.{1,40})', head)
|
||||
hs_val = norm_status(hs.group(1)) if hs else ''
|
||||
st_idx = (index_status.get(num, {}) or {}).get('status') or ''
|
||||
if st_idx in ('?', '❓', ''):
|
||||
st_idx = '' # ❓/? 一律视为「缺」—— 防 INDEX↔manifest 环形锁定(2026-09-14)
|
||||
n_ref = ref.get(num, 0) if num else 0
|
||||
n_cur = ref_current.get(num, 0) if num else 0
|
||||
if not num:
|
||||
tier = 'doc' # 根级长期文档 / 资产:不参与"档案分层"
|
||||
elif n_cur >= 3:
|
||||
tier = 'hot' # 被**现行层反复引用** ⇒ 改东西前大概率要看
|
||||
elif n_cur >= 1:
|
||||
tier = 'cur' # 被现行层引用 1–2 次 ⇒ 相关即看(单次引用≠不重要,如红线类档案 07)
|
||||
elif n_ref >= 1:
|
||||
tier = 'warm' # 仅被历史档案互引 ⇒ 参考
|
||||
else:
|
||||
tier = 'cold' # 零引用
|
||||
items.append({
|
||||
'path': f,
|
||||
'num': num,
|
||||
'title': title[:80],
|
||||
'status': hs_val or st_idx or '?',
|
||||
'date': (hd.group(2) if hd else (index_status.get(num, {}) or {}).get('date', '') or ''),
|
||||
'chars': len(s),
|
||||
'lines': s.count('\n') + 1,
|
||||
'refs': n_ref,
|
||||
'refsCurrent': n_cur,
|
||||
'tier': tier,
|
||||
'layer': layer_of(f, num),
|
||||
'domain': domain_of(title + '\n' + head),
|
||||
'tldr': tldr,
|
||||
**view_field(f),
|
||||
})
|
||||
|
||||
out = {
|
||||
'generatedFrom': '07-scripts/docs-manifest.py',
|
||||
'counts': {'files': len(items), 'chars': sum(i['chars'] for i in items)},
|
||||
'tiers': {t: sum(1 for i in items if i['tier'] == t) for t in ('hot', 'cur', 'warm', 'cold', 'doc')},
|
||||
'domains': dict(collections.Counter(i['domain'] for i in items)),
|
||||
'layers': dict(collections.Counter(i['layer'] for i in items)),
|
||||
'items': items,
|
||||
}
|
||||
io.open(os.path.join(ROOT, 'docs-manifest.json'), 'w', encoding='utf-8', newline='\n').write(
|
||||
json.dumps(out, ensure_ascii=False, indent=1) + '\n')
|
||||
|
||||
arch = [i for i in items if i['num'] is not None]
|
||||
print('文档 %d 份 / %s 字符(≈%s tokens)' % (len(items), format(out['counts']['chars'], ','),
|
||||
format(int(out['counts']['chars'] * 0.7), ',')))
|
||||
print('档案 %d 份 | 分层:hot %d / warm %d / cold %d | 根级文档 %d'
|
||||
% (len(arch), out['tiers']['hot'], out['tiers']['warm'], out['tiers']['cold'], out['tiers']['doc']))
|
||||
oversized = sorted([i for i in items if i['chars'] > 30000], key=lambda x: -x['chars'])
|
||||
if oversized:
|
||||
print('\n【⚠️ 单篇 > 30 KB】%d 篇(约定上限 30 KB;历史只报不改,新档案超限须拆):' % len(oversized))
|
||||
for i in oversized:
|
||||
print(' %6.1f KB %-58s' % (i['chars'] / 1024, os.path.basename(i['path'])[:56]))
|
||||
print('\n【hot】常读(引用 ≥8 次)—— 改东西前大概率要看:')
|
||||
for i in sorted([x for x in arch if x['tier'] == 'hot'], key=lambda x: -x['refs']):
|
||||
print(' 档案 %-3s %2d 次 %-52s %6d 字符' % (i['num'], i['refs'], os.path.basename(i['path'])[:50], i['chars']))
|
||||
print('\n【cold】零引用(历史候选,可只留索引行):')
|
||||
for i in sorted([x for x in arch if x['tier'] == 'cold'], key=lambda x: x['num']):
|
||||
print(' 档案 %-3s %s' % (i['num'], os.path.basename(i['path'])))
|
||||
print('\n→ 已写出 docs-manifest.json')
|
||||
@@ -0,0 +1,142 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""docs-search.py — 文档库**带语义的全文检索**(零依赖、不建索引、每次实时扫)
|
||||
|
||||
为什么需要它(2026-09-14 实测)
|
||||
* 库已到 **118 篇 / 95 万字符 ≈ 67 万 token** ⇒ 不可能"读全库";
|
||||
* 只能 `grep` 时,**搜到的结果分不清"现行值"还是"历史值"** —— 这正是 2026-09-12 踩过的坑
|
||||
(旧配额活在 15 个文件里,被当成事实用)。
|
||||
⇒ 本脚本把 `docs-manifest.json` 的 **(层 L0–L5 / 域 / tier / 状态)** 标注接进检索结果,
|
||||
并给 `--current` 一键**排除历史层**,让"查现行事实"这件事**结果可判**。
|
||||
|
||||
用法
|
||||
python3 07-scripts/docs-search.py 配额 # 全库搜「配额」
|
||||
python3 07-scripts/docs-search.py 配额 --current # **只搜现行层(排除 L5 档案 / archive)**
|
||||
python3 07-scripts/docs-search.py 插件 域 # 多词 = AND
|
||||
python3 07-scripts/docs-search.py glibc --domain plugin --layer L5
|
||||
python3 07-scripts/docs-search.py 内存 --json | jq . # 机读输出
|
||||
选项
|
||||
--current 只搜 L0–L4(排除 L5 与 09-archive/)—— **查现行值请默认加它**
|
||||
--layer L1,L2 限定层(L0–L5)
|
||||
--domain plugin 限定域(platform/plugin/ui/06-ops/external/method)
|
||||
--limit N 最多返回多少篇(默认 12)
|
||||
--context N 每篇显示多少条命中行(默认 3,0 = 只统计)
|
||||
--json 输出 JSON(供脚本消费)
|
||||
退出码:0 = 有命中;1 = 无命中;2 = 用法/环境错误
|
||||
"""
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
HISTORY_PREFIXES = ('04-调整方案/', '09-archive/', '01-规范/02-运维手册')
|
||||
WEIGHT = {'hot': 1.0, 'cur': 0.9, 'warm': 0.8, 'doc': 0.7, 'cold': 0.5}
|
||||
|
||||
|
||||
def load_meta():
|
||||
p = os.path.join(ROOT, 'docs-manifest.json')
|
||||
if not os.path.exists(p):
|
||||
raise SystemExit('ERROR: 缺 docs-manifest.json —— 先跑 07-scripts/docs-manifest.py')
|
||||
out = {}
|
||||
for i in json.loads(io.open(p, encoding='utf-8').read())['items']:
|
||||
out[i['path']] = i
|
||||
return out
|
||||
|
||||
|
||||
def walk():
|
||||
for base, dirs, names in os.walk(ROOT):
|
||||
if '.git' in dirs:
|
||||
dirs.remove('.git')
|
||||
for n in sorted(names):
|
||||
if n.endswith('.md') and '.bak' not in n:
|
||||
yield os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/')
|
||||
|
||||
|
||||
def main(argv):
|
||||
VALUE_OPTS = {'--layer', '--domain', '--limit', '--context'}
|
||||
terms, opts, flags = [], {}, set()
|
||||
i = 0
|
||||
while i < len(argv):
|
||||
a = argv[i]
|
||||
if a in VALUE_OPTS and i + 1 < len(argv):
|
||||
opts[a] = argv[i + 1]
|
||||
i += 2
|
||||
continue
|
||||
if a.startswith('--') and '=' in a:
|
||||
k, v = a.split('=', 1)
|
||||
opts[k] = v
|
||||
elif a.startswith('--'):
|
||||
flags.add(a)
|
||||
else:
|
||||
terms.append(a)
|
||||
i += 1
|
||||
flag = lambda k: k in flags # noqa: E731
|
||||
opt = lambda k, d=None: opts.get(k, d) # noqa: E731
|
||||
if not terms:
|
||||
print(__doc__)
|
||||
return 2
|
||||
terms = [t.lower() for t in terms]
|
||||
meta = load_meta()
|
||||
cur = flag('--current')
|
||||
want_layers = set((opt('--layer') or '').split(',')) - {''}
|
||||
want_domain = opt('--domain')
|
||||
limit = int(opt('--limit', '12'))
|
||||
ctx = int(opt('--context', '3'))
|
||||
|
||||
hits = []
|
||||
for rel in walk():
|
||||
if cur and (rel.startswith(HISTORY_PREFIXES) or rel == '09-archive'):
|
||||
continue
|
||||
m = meta.get(rel, {})
|
||||
layer, domain = m.get('layer', '?'), m.get('domain', '?')
|
||||
if want_layers and layer not in want_layers:
|
||||
continue
|
||||
if want_domain and domain != want_domain:
|
||||
continue
|
||||
try:
|
||||
text = io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
|
||||
except OSError:
|
||||
continue
|
||||
low = text.lower()
|
||||
counts = [low.count(t) for t in terms]
|
||||
if not all(c > 0 for c in counts): # AND 语义
|
||||
continue
|
||||
total = sum(counts)
|
||||
head = (m.get('title') or '') + ' ' + (m.get('tldr') or '')
|
||||
boost = 1.4 if any(t in head.lower() for t in terms) else 1.0
|
||||
lines = text.split('\n')
|
||||
shown = [(n + 1, lines[n].strip()[:150]) for n in range(len(lines))
|
||||
if all(t in lines[n].lower() for t in terms)][:ctx]
|
||||
hits.append({
|
||||
'path': rel, 'layer': layer, 'domain': domain,
|
||||
'tier': m.get('tier', '?'), 'status': m.get('status', '?'),
|
||||
'hits': total, 'score': round(total * WEIGHT.get(m.get('tier'), 0.7) * boost, 1),
|
||||
'title': m.get('title', ''), 'samples': shown,
|
||||
'view': bool(m.get('dedupeView')),
|
||||
})
|
||||
hits.sort(key=lambda x: -x['score'])
|
||||
hits = hits[:limit]
|
||||
|
||||
if flag('--json'):
|
||||
print(json.dumps({'terms': terms, 'current_only': cur, 'results': hits},
|
||||
ensure_ascii=False, indent=1))
|
||||
else:
|
||||
scope = '仅现行层(L0–L4)' if cur else '全库(含历史 L5)'
|
||||
print('检索 %r | %s | 命中 %d 篇%s'
|
||||
% (' + '.join(terms), scope, len(hits), '' if not cur else ' (查现行值建议保持 --current)'))
|
||||
for h in hits:
|
||||
print('\n[%s] %s | %s/%s | %s | %d 次'
|
||||
% (h['status'], h['path'], h['layer'], h['domain'], h['tier'], h['hits']))
|
||||
if h['title']:
|
||||
print(' %s' % h['title'][:110])
|
||||
if h.get('view'):
|
||||
print(' ↳ 本篇有**去重视图**(体积更小,优先读它)')
|
||||
for ln, text in h['samples']:
|
||||
print(' L%-5d %s' % (ln, text))
|
||||
return 0 if hits else 1
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
raise SystemExit(main(sys.argv[1:]))
|
||||
@@ -0,0 +1,122 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""docs-shrink-guard.py — 防「共享文件被整文件重写抹掉别人的行」
|
||||
|
||||
为什么需要(2026-09-14 实测)
|
||||
* `MEMORY.md` 被并行会话**整文件重写**,抹掉了本会话刚写进去的两行;
|
||||
* 本库的规矩是「共享文件只用 Edit 精确替换,禁整文件 Write」—— 但**只靠人守纪律**,没有机制兜底。
|
||||
⇒ 本脚本给"纪律"加一条**机械兜底**:记录每个文件的**行数快照**,下次运行时若某文件
|
||||
**行数骤降**(默认 >30% 且绝对减少 >20 行),就把它当成"疑似被整文件重写"报警。
|
||||
|
||||
覆盖范围(两处共享热区):
|
||||
* 文档库 `dsh-server-docs/**/*.md`
|
||||
* 工作区记忆 `.workbuddy/memory/*.md`
|
||||
快照落在**库外**(`<工作区>/.workbuddy/cache/docs-lines.json`)—— 放库内会污染 `docs-sync-check` 对账。
|
||||
|
||||
用法
|
||||
python3 07-scripts/docs-shrink-guard.py # 比对并报告(不改快照)
|
||||
python3 07-scripts/docs-shrink-guard.py --write # 比对 + 刷新快照(改完文件后跑)
|
||||
python3 07-scripts/docs-shrink-guard.py --baseline # 只建档不比对(首次)
|
||||
python3 07-scripts/docs-shrink-guard.py --allow-shrink <路径子串> # 合法重构白名单(不报警)
|
||||
退出码:0 = 无骤降;1 = 发现骤降(需人看一眼 diff);2 = 用法错误
|
||||
"""
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
DOCS = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
WS = os.path.dirname(DOCS)
|
||||
SNAP = os.path.join(WS, '.workbuddy', 'cache', 'docs-lines.json')
|
||||
DROP_RATIO, DROP_MIN = 0.30, 20
|
||||
|
||||
|
||||
def collect():
|
||||
out = {}
|
||||
roots = [(DOCS, 'docs'), (os.path.join(WS, '.workbuddy', 'memory'), 'memory')]
|
||||
for root, tag in roots:
|
||||
if not os.path.isdir(root):
|
||||
continue
|
||||
for base, dirs, names in os.walk(root):
|
||||
if '.git' in dirs:
|
||||
dirs.remove('.git')
|
||||
if os.sep + 'cache' in base:
|
||||
continue
|
||||
for n in sorted(names):
|
||||
if not n.endswith('.md') or '.bak' in n:
|
||||
continue
|
||||
p = os.path.join(base, n)
|
||||
rel = os.path.relpath(p, WS).replace('\\', '/')
|
||||
try:
|
||||
out[rel] = sum(1 for _ in io.open(p, encoding='utf-8', errors='replace'))
|
||||
except OSError:
|
||||
continue
|
||||
return out
|
||||
|
||||
|
||||
def main(argv):
|
||||
now = collect()
|
||||
allow = [argv[i + 1] for i, a in enumerate(argv)
|
||||
if a == '--allow-shrink' and i + 1 < len(argv)]
|
||||
old = {}
|
||||
if os.path.exists(SNAP):
|
||||
try:
|
||||
old = json.loads(io.open(SNAP, encoding='utf-8').read())
|
||||
except ValueError:
|
||||
print('⚠️ 快照损坏,按首次处理')
|
||||
if '--baseline' in argv or not old:
|
||||
if '--write' in argv or '--baseline' in argv:
|
||||
os.makedirs(os.path.dirname(SNAP), exist_ok=True)
|
||||
io.open(SNAP, 'w', encoding='utf-8', newline='\n').write(
|
||||
json.dumps(now, ensure_ascii=False, indent=1, sort_keys=True) + '\n')
|
||||
print('已建基线:%d 个文件 → %s' % (len(now), SNAP))
|
||||
return 0
|
||||
print('无快照(首次)。跑 `--baseline` 建档后再用。')
|
||||
return 2
|
||||
|
||||
shrunk, added, removed, grew = [], [], [], []
|
||||
for rel, n in now.items():
|
||||
if rel not in old:
|
||||
added.append(rel)
|
||||
elif n < old[rel]:
|
||||
if old[rel] - n > DROP_MIN and (old[rel] - n) / old[rel] > DROP_RATIO:
|
||||
if any(a in rel for a in allow):
|
||||
print(' ↳ 合法重构(白名单):%s %d → %d 行' % (rel, old[rel], n))
|
||||
else:
|
||||
shrunk.append((rel, old[rel], n))
|
||||
elif old[rel] - n > 0:
|
||||
grew.append((rel, old[rel], n, 'down'))
|
||||
for rel in old:
|
||||
if rel not in now:
|
||||
removed.append(rel)
|
||||
for rel, n in now.items():
|
||||
if rel in old and n > old[rel]:
|
||||
grew.append((rel, old[rel], n, 'up'))
|
||||
|
||||
print('文件 %d(新增 %d / 删除 %d)| 行数变化 %d' % (len(now), len(added), len(removed), len(grew)))
|
||||
for rel in added[:8]:
|
||||
print(' + %s' % rel)
|
||||
for rel in removed[:8]:
|
||||
print(' - %s(消失?)' % rel)
|
||||
for rel, o, n, d in sorted(grew, key=lambda x: -abs(x[1] - x[2]))[:6]:
|
||||
print(' %s %s %d → %d 行' % ('↑' if d == 'up' else '↓', rel, o, n))
|
||||
|
||||
bad = 0
|
||||
if shrunk:
|
||||
bad = 1
|
||||
print('\n⚠️ **疑似被整文件重写(行数骤降)** —— 请看一眼 `git diff` 或与上一位写入者核对:')
|
||||
for rel, o, n in shrunk:
|
||||
print(' %s %d → %d 行(-%.0f%%)' % (rel, o, n, (o - n) / o * 100))
|
||||
print(' 本库规矩:共享文件只用 Edit 精确替换,禁整文件 Write(见 CODEBUDDY.md)')
|
||||
if '--write' in argv:
|
||||
os.makedirs(os.path.dirname(SNAP), exist_ok=True)
|
||||
io.open(SNAP, 'w', encoding='utf-8', newline='\n').write(
|
||||
json.dumps(now, ensure_ascii=False, indent=1, sort_keys=True) + '\n')
|
||||
print('已刷新快照。')
|
||||
elif bad:
|
||||
print('(未刷新快照:确认无误后跑 `--write`)')
|
||||
return bad
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
raise SystemExit(main(sys.argv[1:]))
|
||||
@@ -0,0 +1,116 @@
|
||||
#!/usr/bin/env bash
|
||||
# docs-sync-check.sh — 对账「本机文档镜像」与「服务器文档库」
|
||||
#
|
||||
# 用法:
|
||||
# bash 07-scripts/docs-sync-check.sh # 报告双端差异
|
||||
# DOCS_REMOTE=bt-server bash 07-scripts/docs-sync-check.sh
|
||||
#
|
||||
# 可覆盖的环境变量:
|
||||
# DOCS_LOCAL_DIR 本机镜像目录(默认 D:/github/dsh_shenxian/dsh-server-docs)
|
||||
# DOCS_REMOTE ssh 别名或 host(默认 bt-server)
|
||||
# DOCS_REMOTE_DIR 服务器文档目录(默认 /opt/dsh/docs)
|
||||
#
|
||||
# 退出码:0 = 完全一致;1 = 存在差异;2 = 环境/连接错误
|
||||
set -uo pipefail
|
||||
|
||||
LOCAL_DIR="${DOCS_LOCAL_DIR:-D:/github/dsh_shenxian/dsh-server-docs}"
|
||||
REMOTE="${DOCS_REMOTE:-bt-server}"
|
||||
REMOTE_DIR="${DOCS_REMOTE_DIR:-/opt/dsh/docs}"
|
||||
|
||||
[ -d "$LOCAL_DIR" ] || { echo "ERROR: 本机目录不存在: $LOCAL_DIR" >&2; exit 2; }
|
||||
|
||||
tmp_l="$(mktemp)"; tmp_r="$(mktemp)"
|
||||
trap 'rm -f "$tmp_l" "$tmp_r" 2>/dev/null || true' EXIT
|
||||
|
||||
# ── 本机哈希:优先「一次 Python 算完」───────────────────────────────
|
||||
# 历史坑(2026-09-12 实测):Git Bash 下 `find | while read` **逐文件 spawn** `md5sum` / `cut`
|
||||
# (132 文件 ≈ 264 次进程启动)→ 全量对账 **2 分 15 秒**,慢到被调用方(handoff-guard 信息模式)当成"挂死"。
|
||||
# Python 一次遍历 → 秒级。探测顺序:$DSH_PY → python3 → python → 本机兜底绝对路径。
|
||||
PY="${DSH_PY:-}"
|
||||
if [ -z "$PY" ]; then
|
||||
for c in python3 python; do
|
||||
if command -v "$c" >/dev/null 2>&1; then PY="$c"; break; fi
|
||||
done
|
||||
fi
|
||||
if [ -z "$PY" ] && [ -x "E:/ProgramData/.workbuddy/binaries/python/versions/3.13.12/python.exe" ]; then
|
||||
PY="E:/ProgramData/.workbuddy/binaries/python/versions/3.13.12/python.exe"
|
||||
fi
|
||||
|
||||
hash_local() {
|
||||
if [ -n "$PY" ]; then
|
||||
( cd "$LOCAL_DIR" && "$PY" -c '
|
||||
import os, sys, hashlib
|
||||
try:
|
||||
sys.stdout.reconfigure(encoding="utf-8", newline="\n")
|
||||
except Exception:
|
||||
pass
|
||||
SKIPF = (".git/", "tmp/", "05-交接单/.doing-", "05-交接单/.exec-lock", "05-交接单/.locks/")
|
||||
SKIPD = (".git", "tmp", "05-交接单/.doing-", "05-交接单/.exec-lock", "05-交接单/.locks")
|
||||
rows = []
|
||||
for root, dirs, files in os.walk("."):
|
||||
# ⚠️ 不能用 os.path.relpath 拼 rel —— Windows 上它会把「末尾带点」的名字规范化掉
|
||||
# (实测:`INDEX.md.bak-….` 变成无点的名字 → open 失败 → 该文件被静默漏掉 → 制造假差异)。
|
||||
r = root.replace(os.sep, "/")
|
||||
pfx = "" if r in (".", "") else (r[2:] if r.startswith("./") else r)
|
||||
dirs[:] = [d for d in dirs if not (((pfx + "/") if pfx else "") + d).startswith(SKIPD)]
|
||||
for fn in files:
|
||||
rel = (pfx + "/" + fn) if pfx else fn
|
||||
if fn == ".DS_Store" or any(rel.startswith(p) for p in SKIPF):
|
||||
continue
|
||||
if os.path.islink(rel):
|
||||
continue # 与 find -type f 同口径
|
||||
try:
|
||||
with open(rel, "rb") as fh:
|
||||
h = hashlib.md5(fh.read()).hexdigest()
|
||||
except Exception:
|
||||
# Windows 会把「末尾带点 / 保留名」等非常规文件名规范化掉 → open 直接 ENOENT。
|
||||
# 用 \\?\ 前缀绕过,**必须拼未规范化的绝对路径**(os.path.abspath 会 strip 末尾点 → 前缀失效)。
|
||||
try:
|
||||
with open("\\\\?\\" + os.getcwd() + "\\" + rel.replace("/", "\\"), "rb") as fh:
|
||||
h = hashlib.md5(fh.read()).hexdigest()
|
||||
except Exception:
|
||||
continue
|
||||
rows.append((rel, h))
|
||||
rows.sort(key=lambda r: r[0].encode("utf-8")) # 与 LC_ALL=C sort 同口径
|
||||
for rel, h in rows:
|
||||
print(rel + "\t" + h)
|
||||
' )
|
||||
else
|
||||
echo "WARN: 未找到 python,回退逐文件 md5sum(会慢 ~2 分钟)" >&2
|
||||
( cd "$LOCAL_DIR" && find . -type f ! -path './.git/*' ! -name '.DS_Store' \
|
||||
! -path './tmp/*' ! -path '*/.locks/*' \
|
||||
! -path './05-交接单/.doing-*' ! -path './05-交接单/.exec-lock*' | LC_ALL=C sort | while read -r f; do
|
||||
printf '%s\t%s\n' "${f#./}" "$(md5sum "$f" | cut -d' ' -f1)"
|
||||
done )
|
||||
fi
|
||||
}
|
||||
|
||||
hash_remote() {
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" \
|
||||
"cd '$REMOTE_DIR' 2>/dev/null && find . -type f ! -path './.git/*' ! -name '.DS_Store' ! -path './tmp/*' ! -path '*/.locks/*' ! -path './05-交接单/.doing-*' ! -path './05-交接单/.exec-lock*' | LC_ALL=C sort | while read -r f; do printf '%s\t%s\n' \"\${f#./}\" \"\$(md5sum \"\$f\" | cut -d' ' -f1)\"; done" \
|
||||
2>/dev/null
|
||||
}
|
||||
|
||||
hash_local > "$tmp_l"
|
||||
hash_remote > "$tmp_r"
|
||||
|
||||
if [ ! -s "$tmp_r" ]; then
|
||||
echo "ERROR: 无法读取服务器目录 $REMOTE:$REMOTE_DIR(ssh 失败或目录不存在)" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
awk -F'\t' '
|
||||
NR==FNR { l[$1]=$2; next }
|
||||
{ r[$1]=$2 }
|
||||
END {
|
||||
same=0
|
||||
for (k in l) {
|
||||
if (k in r) { if (l[k]==r[k]) same++; else { diff++; print "⚠️ 内容不一致 " k } }
|
||||
else { only_l++; print "⬆️ 仅本地(待推送) " k }
|
||||
}
|
||||
for (k in r) if (!(k in l)) { only_r++; print "⬇️ 仅服务器(待拉取) " k }
|
||||
printf "\n—— 汇总 ——\n 一致: %d\n 内容不一致: %d\n 仅本地: %d\n 仅服务器: %d\n 本地文件总数: %d / 服务器文件总数: %d\n", same, diff+0, only_l+0, only_r+0, length(l), length(r)
|
||||
if ((diff+0)+(only_l+0)+(only_r+0) > 0) { print "\n结果: 存在差异 ❌"; exit 1 }
|
||||
print "\n结果: 双端一致 ✅"
|
||||
}
|
||||
' "$tmp_l" "$tmp_r"
|
||||
@@ -0,0 +1,175 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
extract-user-voice.py —— 把一个工作区的**全部历史会话**里的「用户真实发言」抽成一份可读清单。
|
||||
|
||||
为什么要它:做「决策方法 / 协作方式」复盘时,需要用**用户的原话全集**做底料,
|
||||
而不是只靠档案与日志(那是会话的结构化沉淀,会丢掉"用户否决了什么、纠正了什么")。
|
||||
本脚本只读、不改任何会话文件。
|
||||
|
||||
输出到 stdout(可重定向到自己想放的路径;**不要写进文档库目录**,否则会被 docs-sync-check 计成"仅本地"):
|
||||
|
||||
python3 07-scripts/extract-user-voice.py # 自动定位当前工作区
|
||||
python3 07-scripts/extract-user-voice.py --project <dir名> # 指定 ~/.workbuddy/projects/<dir名>
|
||||
python3 07-scripts/extract-user-voice.py --full # 打印全文(默认每条截断 88 字)
|
||||
python3 07-scripts/extract-user-voice.py --needle 关键词 # 只列含关键词的发言(找某个决策的来龙去脉)
|
||||
|
||||
会话记录位置(WorkBuddy):`<配置目录>/projects/<把 cwd 的 : \\ / 换成 ->/<sessionId>.jsonl`
|
||||
- 配置目录判定链:`WORKBUDDY_CONFIG_DIR ?? CODEBUDDY_CONFIG_DIR ?? ~/.workbuddy`
|
||||
- 行格式:`{"type":"message","role":"user","content":[{"type":"input_text","text":"..."}]}`
|
||||
- ⚠️ 用户发言里会带一大段 `<system-reminder ...>` 前言(user_info / identity_context),
|
||||
必须剥掉才是**用户真实说的话**。
|
||||
"""
|
||||
import argparse
|
||||
import datetime
|
||||
import glob
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
SREM = re.compile(r'<system-reminder.*?</system-reminder>', re.S)
|
||||
TAGS = re.compile(r'</?(user_query|cb_summary|conversation_history_summary)>')
|
||||
WS = re.compile(r'\s+')
|
||||
|
||||
|
||||
def config_dir():
|
||||
for k in ('WORKBUDDY_CONFIG_DIR', 'CODEBUDDY_CONFIG_DIR'):
|
||||
v = os.environ.get(k)
|
||||
if v:
|
||||
return os.path.expanduser(v)
|
||||
return os.path.join(os.path.expanduser('~'), '.workbuddy')
|
||||
|
||||
|
||||
def project_dir_name(cwd):
|
||||
"""WorkBuddy 的目录名规则(实测):盘符**小写** + ':' 去掉,随后把分隔符各换成 '-'。
|
||||
例:E:\\ProgramData\\AI技能\\aliyun-dsh-server → e-ProgramData-AI技能-aliyun-dsh-server
|
||||
(路径中段的字母大小写与空格**保留**)。"""
|
||||
p = os.path.abspath(cwd)
|
||||
drive, rest = os.path.splitdrive(p)
|
||||
return (drive.rstrip(':').lower() + re.sub(r'[\\/]', '-', rest)).strip('-')
|
||||
|
||||
|
||||
def _norm(s):
|
||||
"""把目录名归一化,用于模糊比较(忽略盘符大小写、多余连字符)。"""
|
||||
return re.sub(r'-+', '-', s).lower()
|
||||
|
||||
|
||||
def resolve_project(root, cwd, explicit):
|
||||
"""返回 (目录名, 绝对路径)。显式 --project 只用它;否则从 cwd 起**逐级向上**试,
|
||||
全部落空后再对 projects/ 下的实际目录做一次归一化模糊匹配。"""
|
||||
if explicit:
|
||||
return (explicit, os.path.join(root, explicit))
|
||||
|
||||
tried = []
|
||||
cur = os.path.abspath(cwd)
|
||||
while True:
|
||||
name = project_dir_name(cur)
|
||||
tried.append(name)
|
||||
p = os.path.join(root, name)
|
||||
if os.path.isdir(p):
|
||||
return (name, p)
|
||||
parent = os.path.dirname(cur)
|
||||
if parent == cur:
|
||||
break
|
||||
cur = parent
|
||||
|
||||
# 模糊兜底:按归一化名比对实际存在的目录
|
||||
try:
|
||||
existing = [d for d in os.listdir(root) if os.path.isdir(os.path.join(root, d))]
|
||||
except OSError:
|
||||
existing = []
|
||||
for want in tried:
|
||||
for d in existing:
|
||||
if _norm(d) == _norm(want):
|
||||
return (d, os.path.join(root, d))
|
||||
return (tried[0], os.path.join(root, tried[0]))
|
||||
|
||||
|
||||
def fmt_ts(v):
|
||||
if isinstance(v, (int, float)):
|
||||
v = v / 1000.0 if v > 1e11 else v
|
||||
try:
|
||||
return datetime.datetime.fromtimestamp(v).strftime('%m-%d %H:%M')
|
||||
except Exception:
|
||||
return '?'
|
||||
return str(v)[:16] if v else '?'
|
||||
|
||||
|
||||
def user_text(line):
|
||||
"""从一行 jsonl 里取出「用户真实发言」;不是用户消息则返回 None。"""
|
||||
try:
|
||||
o = json.loads(line)
|
||||
except Exception:
|
||||
return None
|
||||
if o.get('type') != 'message' or o.get('role') != 'user':
|
||||
return None
|
||||
txt = ''
|
||||
for b in (o.get('content') or []):
|
||||
if isinstance(b, dict) and b.get('type') in ('input_text', 'text'):
|
||||
txt += b.get('text', '')
|
||||
txt = SREM.sub('', txt).strip()
|
||||
txt = TAGS.sub('', txt).strip()
|
||||
txt = WS.sub(' ', txt)
|
||||
return (fmt_ts(o.get('timestamp')), txt) if txt else None
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument('--project', help='projects 下的目录名(默认按当前 cwd 推导)')
|
||||
ap.add_argument('--cwd', default=os.getcwd(), help='用于推导项目名的 cwd')
|
||||
ap.add_argument('--full', action='store_true', help='打印全文(不截断)')
|
||||
ap.add_argument('--needle', help='只列含该关键词的发言')
|
||||
a = ap.parse_args()
|
||||
|
||||
root = os.path.join(config_dir(), 'projects')
|
||||
name, pdir = resolve_project(root, a.cwd, a.project)
|
||||
if not os.path.isdir(pdir):
|
||||
sys.stderr.write('找不到项目会话目录:%s\n' % pdir)
|
||||
sys.stderr.write('可用目录(%s):\n' % root)
|
||||
for d in sorted(os.listdir(root))[:40]:
|
||||
sys.stderr.write(' %s\n' % d)
|
||||
return 2
|
||||
|
||||
files = sorted(glob.glob(os.path.join(pdir, '*.jsonl')))
|
||||
if not files:
|
||||
sys.stderr.write('该目录下没有 .jsonl 会话文件:%s\n' % pdir)
|
||||
return 2
|
||||
|
||||
total = 0
|
||||
print('项目会话目录:%s' % pdir)
|
||||
print('会话文件 %d 个\n' % len(files))
|
||||
|
||||
print('===== 各会话概览 =====')
|
||||
per = []
|
||||
for f in files:
|
||||
msgs = []
|
||||
with io.open(f, encoding='utf-8', errors='replace') as fh:
|
||||
for line in fh:
|
||||
r = user_text(line)
|
||||
if r:
|
||||
msgs.append(r)
|
||||
per.append((os.path.basename(f[:-6]), msgs))
|
||||
total += len(msgs)
|
||||
span = ('%s → %s' % (msgs[0][0], msgs[-1][0])) if msgs else '-'
|
||||
print(' %-38s 用户发言 %4d 条 %s' % (os.path.basename(f)[:38], len(msgs), span))
|
||||
print('\n合计用户真实发言 = %d 条\n' % total)
|
||||
|
||||
for sid, msgs in per:
|
||||
if a.needle:
|
||||
msgs = [m for m in msgs if a.needle in m[1]]
|
||||
if not msgs:
|
||||
continue
|
||||
print('===== %s(命中 %d 条)=====' % (sid, len(msgs)))
|
||||
else:
|
||||
print('===== %s(%d 条)=====' % (sid, len(msgs)))
|
||||
for i, (t, m) in enumerate(msgs, 1):
|
||||
body = m if a.full else (m[:88] + ('…' if len(m) > 88 else ''))
|
||||
print('%4d %s %s' % (i, t, body))
|
||||
print()
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,562 @@
|
||||
#!/usr/bin/env bash
|
||||
# handoff-guard.sh — 开工 / 推送前的「并行冲突预检」(只读;--claim 除外,它只建一个锁目录)
|
||||
#
|
||||
# 为什么需要它:本库由**多个 AI 会话并行**读写,而「先读后改 / Edit 增量 / 改完 commit」这类
|
||||
# 约定全部依赖“人记得做”。本脚本把判据变成**一条命令 + 退出码**,不靠记忆。
|
||||
#
|
||||
# 关键设计:**mtime 只作提示、不作判定**(它分不清“谁改的”)。真正的判定来自四处:
|
||||
# ① 全局执行锁(05-交接单/.exec-lock)—— **一粗**:同一时刻只允许一个执行会话动「文档/代码/服务器」
|
||||
# ① 单级占用锁(05-交接单/.doing-<单号>)—— **一细**:这个单归谁做(供台账 / 接管使用)
|
||||
# ② 越界改动(不在我声明清单里的未提交文件)—— 推送前检查,防“顺手重放别人的半成品”
|
||||
# ④ 双端一致性(docs-sync-check.sh)—— 推送前检查,防“幽灵文件”(推回了别人已移走的文件)
|
||||
#
|
||||
# 两级锁的关系:**先抢全局锁 → 再占单级锁**;释放时**先放单级、再放全局**。
|
||||
# 单级锁允许“两个会话各做一单”(冲突域不重叠时);全局锁则彻底禁止并行执行。
|
||||
# ⇒ 本库现状(共享入口文件多 + 要动服务器)建议**默认只跑一个执行会话**,即始终持全局锁。
|
||||
#
|
||||
# 用法:
|
||||
# bash 07-scripts/handoff-guard.sh --claim-exec "exec-session-B" # 【第一步】抢全局执行锁
|
||||
# bash 07-scripts/handoff-guard.sh --claim T03 "exec-session-B" # 【第二步】占单级锁
|
||||
# bash 07-scripts/handoff-guard.sh --release T03 # 完工:先放单级
|
||||
# bash 07-scripts/handoff-guard.sh --release-exec # 再放全局
|
||||
# bash 07-scripts/handoff-guard.sh # 看全局状态(信息模式)
|
||||
# ME="exec-session-B" MINE="05-交接单/T03-*.md" bash 07-scripts/handoff-guard.sh T03 # 开工检查(严格)
|
||||
# ME="exec-session-B" MINE="..." PUSH=1 bash 07-scripts/handoff-guard.sh T03 # 推送前检查
|
||||
#
|
||||
# 环境变量:ME(我是谁 —— 用来判断锁是不是自己的;不设则一律按“别人的锁”处理)
|
||||
# MINE(我本次要改的文件,空格分隔,支持 * 通配)
|
||||
# PUSH=1 启用推送前硬判定(④ 幽灵文件)
|
||||
# GUARD_WINDOW(分钟,默认 30)|SKIP_SYNC=1 跳过 ④
|
||||
# 退出码:0 = 放行;1 = 命中硬冲突/硬判定;2 = 环境错误
|
||||
set -uo pipefail
|
||||
|
||||
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
cd "$ROOT" || exit 2
|
||||
LOCKDIR="$ROOT/05-交接单/.doing-"
|
||||
LOCKEXEC="$ROOT/05-交接单/.exec-lock"
|
||||
MELOCK="$ROOT/05-交接单/.me-lock"
|
||||
ME="${ME:-}"
|
||||
|
||||
# ══════════════════════════════════════════════════════════════════════════
|
||||
# 域锁(2026-09-22 加 · 用户令「多个会话并行开发,全局锁导致无法并行」)
|
||||
#
|
||||
# 要害:`.exec-lock` 是**布尔量**(存在/不存在),不携带"我占了哪些资源"⇒ 判不出交集
|
||||
# ⇒ 只能全有全无。域锁让锁**携带资源声明**,于是能算交集。
|
||||
#
|
||||
# 三类资源 × 三种持锁时长:
|
||||
# 域锁 `--claim-exec <会话> --domains <域...>` 全程持有(层①/②/③ 不动 export 签名)
|
||||
# 骨架锁 `--claim-skeleton <资源名>` 秒级(层④/迁移号/挂载点/package.json)
|
||||
# 发布锁 `--claim-publish` 秒级(commit/push/scp/npm run build)
|
||||
#
|
||||
# 兼容:`.exec-lock` 分支**原样保留** ⇒ 未升级的会话照旧可用(hook 是启动时快照)。
|
||||
# ══════════════════════════════════════════════════════════════════════════
|
||||
LOCKSROOT="$ROOT/05-交接单/.locks"
|
||||
GATELOCK="$LOCKSROOT/.gate" # 秒级临界区:算交集时必须串行,否则两个会话可能同时判定"无冲突"
|
||||
|
||||
# 域字符串规范化 ⇒ 域键。**必须与 lock-guard-hook.py 的 domain_key() 同规则**,
|
||||
# 否则 shell 侧写进去的键和 Python 侧算出来的键对不上 ⇒ 域锁静默失效(假绿)。
|
||||
# 规则:在路径里找第一个「已知根段」(src/web/07-scripts/08-skills/... 或工作区目录名),
|
||||
# 取它和它后一段作域键;找不到就取末两段。
|
||||
# ⚠️ 顺序有意义:**先长后短**,与 lock-guard-hook.py 的 _DOMAIN_SEGS 逐字一致。
|
||||
# `dsh-server-docs/scripts` 必须锚成 `dsh-server-docs/scripts`(不是 `07-scripts/...`)
|
||||
_ANCHOR_SEGS="dsh-server-docs aliyun-dsh-server src poc web test docs scripts skills 交接单"
|
||||
norm_domain() {
|
||||
d="${1//,/ }"
|
||||
for one in $d; do
|
||||
one="${one//\\//}"; one="${one%/}"
|
||||
# 逐段找**第一个**锚点段(用 awk 保证取"首个",与 Python 的 for-seg-in-_DOMAIN_SEGS 一致)
|
||||
_key="$(printf '%s' "$one" | awk -v anchors="$_ANCHOR_SEGS" '
|
||||
BEGIN { n = split(anchors, A, " ") }
|
||||
{ m = split($0, P, "/"); hit = 0
|
||||
for (i = 1; i <= m && !hit; i++)
|
||||
for (j = 1; j <= n; j++)
|
||||
if (tolower(P[i]) == A[j]) { print A[j] "/" P[i+1]; hit = 1; break }
|
||||
if (!hit) print (m >= 2 ? P[m-1] "/" P[m] : $0)
|
||||
}')"
|
||||
printf '%s\n' "$_key" | tr 'A-Z' 'a-z'
|
||||
done
|
||||
}
|
||||
|
||||
# 列出所有别人的域(排除自己的锁)
|
||||
# ⚠️ 判定必须与 lock-guard-hook.py 的 _is_mine() 同规则:**会话名 与 session_id 同时吻合** 才算自己。
|
||||
# 只比会话名 ⇒ 自己抢第二把域锁时会被自己挡住(2026-09-22 实测踩过,报"冲突域被占用"且占用者=自己)。
|
||||
# 只比 session_id ⇒ 同进程复用时会假绿(同进程里 CODEBUDDY_SESSION_ID 恒为真实值)。
|
||||
_my_owner() {
|
||||
# OWNER 文件行序:1=会话名|2=开始:|3=会话:<session_id>|4=ME:<会话名>
|
||||
_n="$(sed -n '1p' "$1/OWNER" 2>/dev/null || echo '')"
|
||||
_id="$(sed -n '3p' "$1/OWNER" 2>/dev/null | sed 's/^会话://' || echo '')"
|
||||
if [ -n "${CODEBUDDY_SESSION_ID:-}" ]; then
|
||||
[ "$_n" = "${OWNER:-}" ] && [ "$_id" = "$CODEBUDDY_SESSION_ID" ] && return 0
|
||||
return 1
|
||||
fi
|
||||
# 无 session_id(人工跑)⇒ 退回只比会话名
|
||||
[ -n "${OWNER:-}" ] && [ "$_n" = "$OWNER" ] && return 0
|
||||
[ -n "$ME" ] && [ "$_n" = "$ME" ] && return 0
|
||||
return 1
|
||||
}
|
||||
others_domains() {
|
||||
[ -d "$LOCKSROOT" ] || return 0
|
||||
for L in "$LOCKSROOT"/*; do
|
||||
[ -d "$L" ] || continue
|
||||
[ "$(basename "$L")" = ".gate" ] && continue
|
||||
[ "$(basename "$L")" = ".migrations" ] && continue
|
||||
_my_owner "$L" && continue
|
||||
o="$(sed -n '1p' "$L/OWNER" 2>/dev/null || echo '?')"
|
||||
if [ -f "$L/DOMAINS" ]; then
|
||||
while IFS= read -r d; do
|
||||
[ -n "$d" ] && printf '%s\t%s\n' "$d" "$o"
|
||||
done < "$L/DOMAINS"
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
if [ "${1:-}" = "--claim-exec" ]; then
|
||||
OWNER="${2:-${ME:-$(whoami)}}"
|
||||
shift 2 2>/dev/null || shift 1 2>/dev/null || true
|
||||
DOMAINS=()
|
||||
# ★ 统一清理守卫(2026-09-22 实测教训):
|
||||
# bash 的 trap 是**单一槽位** —— 先后 trap 两次,后者**覆盖**前者。
|
||||
# 旧写法 "临时文件一个 trap + .gate 一个 trap" ⇒ 后注册的把前一个冲掉 ⇒ 必然漏一个。
|
||||
# 故改成**同一个函数清两样**,全流程只注册一次。
|
||||
# ⚠️ 更关键的坑:`.gate` 可能**是别人的**(临界区被别人占着,我们在 while 里等)。
|
||||
# 若无脑 rmdir ⇒ **把别人的临界区删掉** ⇒ 破坏互斥(R9 精神:⛔ 不碰别人的锁)。
|
||||
# ⇒ 守卫只在"**确实由本次调用创建了 .gate**"时才删它(_GATE_MINE 标志)。
|
||||
_DOMTMP="$ROOT/.domains.tmp.$$"
|
||||
_GATE_MINE=0
|
||||
_cleanup_guard() {
|
||||
rm -f "$_DOMTMP" 2>/dev/null || true
|
||||
[ "$_GATE_MINE" = 1 ] && { rmdir "$GATELOCK" 2>/dev/null || true; }
|
||||
}
|
||||
trap _cleanup_guard EXIT INT TERM HUP
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--domains) shift; while [ $# -gt 0 ] && [ "${1#--}" = "$1" ]; do
|
||||
# ⚠️ 不用 mktemp:本机 mktemp 出 MSYS 路径(/c/...),被当相对路径 ⇒ 落到仓库里
|
||||
norm_domain "$1" > "$_DOMTMP"
|
||||
while IFS= read -r _d; do [ -n "$_d" ] && DOMAINS+=("$_d"); done < "$_DOMTMP"
|
||||
shift; done;;
|
||||
*) shift;;
|
||||
esac
|
||||
done
|
||||
rm -f "$_DOMTMP" 2>/dev/null || true
|
||||
|
||||
if [ ${#DOMAINS[@]} -eq 0 ]; then
|
||||
# 无域声明 ⇒ 退化为旧行为:全局独占(⛔ 不静默放宽)
|
||||
if mkdir "$LOCKEXEC" 2>/dev/null; then
|
||||
printf '%s\n开始:%s\n在做:%s\n' "$OWNER" "$(date '+%m-%d %H:%M')" "${3:-(未声明单号)}" > "$LOCKEXEC/OWNER"
|
||||
echo "✓ 已持全局执行锁($OWNER)—— ⚠️ **未声明域** ⇒ 退化为全局独占(旧行为)"
|
||||
echo " 要用域锁并行:--claim-exec \"$OWNER\" --domains <域...>"
|
||||
echo " ⛔ **锁的生命周期 = 任务的生命周期**(2026-09-14 用户明令):执行完成 → 必须 \`--release-exec\` 才算完成;"
|
||||
echo " 禁止抢锁做一半、不解锁就结束回合/会话(本库无心跳,带锁结束 = 把所有人挡在门外)。中途要停 ⇒ 先释放再停。"
|
||||
exit 0
|
||||
fi
|
||||
echo "✗ 抢锁失败:已有执行会话在跑 ——
|
||||
占用者:$(sed -n '1p' "$LOCKEXEC/OWNER" 2>/dev/null || echo '?') $(sed -n '2p' "$LOCKEXEC/OWNER" 2>/dev/null)
|
||||
$(sed -n '3p' "$LOCKEXEC/OWNER" 2>/dev/null)
|
||||
→ **停手**:等它做完(它会 --release-exec)。⛔ **不得人工删锁、不得接管**(**R9**,用户 2026-09-12 明令)——
|
||||
抢不到锁 = **停手 + 报告用户**;**锁的处置权只属于用户本人**" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# ── 有域声明 ⇒ 域锁路径 ──
|
||||
mkdir -p "$LOCKSROOT" 2>/dev/null
|
||||
# 旧全局锁仍在 ⇒ 尊重它(兼容期:在跑的会话用旧 hook)
|
||||
if [ -d "$LOCKEXEC" ]; then
|
||||
O1="$(sed -n '1p' "$LOCKEXEC/OWNER" 2>/dev/null)"
|
||||
if [ "$O1" != "$OWNER" ]; then
|
||||
echo "✗ 抢域锁失败:**旧的全局锁**仍被占($O1)—— 它未升级到域锁,仍是全局独占。" >&2
|
||||
echo " → 停手等它 --release-exec(⛔ 不得删锁,R9)" >&2
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
# ★ 临界区:查冲突 + 建锁必须原子 ⇒ 用 .gate 串行化(否则两会话可能同时判"无冲突")
|
||||
# ⚠️ 实测坑(2026-09-22):临界区内被 SIGTERM/SIGPIPE 打断(如管道 `| head` 提前关闭)
|
||||
# ⇒ `.gate` 泄漏 ⇒ 后续所有抢锁卡满 10 s 才报错。
|
||||
# ⛔ 清理已由开头注册的 `_cleanup_guard` 统一负责(trap 单槽位,此处**不得**再 trap 覆盖)。
|
||||
|
||||
_tries=0
|
||||
while ! mkdir "$GATELOCK" 2>/dev/null; do
|
||||
_tries=$((_tries+1))
|
||||
if [ "$_tries" -gt 100 ]; then
|
||||
echo "✗ 抢域锁失败:临界区 .gate 被长时间占用(疑似有会话崩在临界区内)" >&2
|
||||
echo " → 报告用户;⛔ 不得删 .gate(R9)" >&2
|
||||
# ⛔ 此处 `.gate` 是**别人的**:`_GATE_MINE` 仍为 0 ⇒ 守卫不会删它,可直接退。
|
||||
exit 1
|
||||
fi
|
||||
sleep 0.1
|
||||
done
|
||||
_GATE_MINE=1 # ★ 本调用成功创建了临界区 ⇒ 授权守卫在退出时清理它
|
||||
|
||||
CONFLICT=""
|
||||
for d in "${DOMAINS[@]}"; do
|
||||
hit="$(others_domains | awk -F'\t' -v want="$d" '$1==want {print $2; exit}')"
|
||||
[ -n "$hit" ] && CONFLICT="${CONFLICT} · 域「$d」已被「$hit」占用"$'\n'
|
||||
done
|
||||
|
||||
if [ -n "$CONFLICT" ]; then
|
||||
rmdir "$GATELOCK" 2>/dev/null
|
||||
echo "✗ 抢域锁失败:**冲突域被占用** ——
|
||||
$CONFLICT → **停手**:等对方释放,或**改做不重叠的域**。⛔ 不得删锁/接管(R9)" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# 锁目录名 = 会话名的**可逆安全映射**(一会话一把锁,域清单写在 DOMAINS 里)
|
||||
# ⚠️ 同会话二次抢不同域 ⇒ **必须合并**,⛔ 不能 rm -rf 重建
|
||||
# (重建会把第一把域的声明冲掉 ⇒ 别人就能抢进来 = 假绿;2026-09-22 实测踩过)
|
||||
# 🔴 2026-09-22 实测修复:旧写法 `tr -c 'A-Za-z0-9._-' '_'` 把**所有非 ASCII 字符压成 `_`**
|
||||
# ⇒ 会话「域锁-A」「域锁-B」「域锁-C」全部映射成同名目录 ⇒ 互相误判成"自己"⇒ **假绿**。
|
||||
# 本平台会话名**全是中文** ⇒ 这是必现缺陷,不是边界情况。
|
||||
# 改法:ASCII 安全的直接留用(可读性 + 兼容旧锁目录),否则附加 `cksum` 摘要(唯一、确定、无需外部依赖)。
|
||||
_safe="$(printf '%s' "$OWNER" | tr -c 'A-Za-z0-9._-' '_')"
|
||||
case "$OWNER" in
|
||||
*[!A-Za-z0-9._-]*)
|
||||
_safe="${_safe:0:40}-$(printf '%s' "$OWNER" | cksum | cut -d' ' -f1)"
|
||||
;;
|
||||
esac
|
||||
MYLOCK="$LOCKSROOT/$_safe"
|
||||
if [ -d "$MYLOCK" ]; then
|
||||
_merged="$MYLOCK/.DOMAINS.new"
|
||||
cat "$MYLOCK/DOMAINS" 2>/dev/null > "$_merged" || true
|
||||
for d in "${DOMAINS[@]}"; do
|
||||
grep -qxF "$d" "$_merged" 2>/dev/null || printf '%s\n' "$d" >> "$_merged"
|
||||
done
|
||||
sort -u "$_merged" -o "$_merged" 2>/dev/null || true
|
||||
mv "$_merged" "$MYLOCK/DOMAINS" 2>/dev/null
|
||||
rmdir "$GATELOCK" 2>/dev/null
|
||||
echo "✓ 已持域锁($OWNER)—— 合并到已有锁"
|
||||
echo " 我的域($(wc -l < "$MYLOCK/DOMAINS" | tr -d ' ') 个):"
|
||||
while IFS= read -r d; do [ -n "$d" ] && echo " · $d"; done < "$MYLOCK/DOMAINS"
|
||||
echo " ⛔ 锁的生命周期 = 任务的生命周期:完成 → 必须 \`--release-exec\`。中途要停 ⇒ 先释放再停。"
|
||||
exit 0
|
||||
fi
|
||||
if mkdir "$MYLOCK" 2>/dev/null; then
|
||||
printf '%s\n' "$OWNER" > "$MYLOCK/OWNER"
|
||||
printf "开始:%s\n会话:%s\nME:%s\n" "$(date "+%m-%d %H:%M")" "${CODEBUDDY_SESSION_ID:-未取到}" "$OWNER" >> "$MYLOCK/OWNER"
|
||||
printf '%s\n' "${DOMAINS[@]}" > "$MYLOCK/DOMAINS"
|
||||
rmdir "$GATELOCK" 2>/dev/null
|
||||
echo "✓ 已持域锁($OWNER)"
|
||||
echo " 我的域(${#DOMAINS[@]} 个):"
|
||||
for d in "${DOMAINS[@]}"; do echo " · $d"; done
|
||||
echo " ⛔ 锁的生命周期 = 任务的生命周期:完成 → 必须 \`--release-exec\`。中途要停 ⇒ 先释放再停。"
|
||||
exit 0
|
||||
fi
|
||||
rmdir "$GATELOCK" 2>/dev/null
|
||||
echo "✗ 抢域锁失败:建锁失败($MYLOCK)" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [ "${1:-}" = "--release-exec" ]; then
|
||||
_who="${ME:-}"
|
||||
_rel=0
|
||||
if [ -d "$LOCKSROOT" ]; then
|
||||
for L in "$LOCKSROOT"/*; do
|
||||
[ -d "$L" ] || continue
|
||||
[ "$(basename "$L")" = ".gate" ] && continue
|
||||
[ "$(basename "$L")" = ".migrations" ] && continue
|
||||
o="$(sed -n '1p' "$L/OWNER" 2>/dev/null)"
|
||||
if [ -z "$_who" ] || [ "$o" = "$_who" ]; then
|
||||
rm -rf "$L" && echo "✓ 已释放域锁($o)" && _rel=1
|
||||
fi
|
||||
done
|
||||
fi
|
||||
if [ -d "$LOCKEXEC" ]; then
|
||||
o1="$(sed -n '1p' "$LOCKEXEC/OWNER" 2>/dev/null)"
|
||||
if [ -z "$_who" ] || [ "$o1" = "$_who" ]; then
|
||||
rm -rf "$LOCKEXEC" && echo "✓ 已释放全局执行锁" && _rel=1
|
||||
fi
|
||||
fi
|
||||
[ "$_rel" = 1 ] || echo "· 无可释放的锁(或非本人持有)"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ── 骨架锁:层④/迁移号/唯一挂载点 ⇒ 秒级独占 ──────────────
|
||||
if [ "${1:-}" = "--claim-skeleton" ]; then
|
||||
RES="${2:-}"; OWNER="${3:-${ME:-$(whoami)}}"
|
||||
[ -n "$RES" ] || { echo "用法:--claim-skeleton <资源名> [占用者]" >&2; exit 2; }
|
||||
mkdir -p "$LOCKSROOT/.migrations" 2>/dev/null
|
||||
SK="$LOCKSROOT/.migrations/$RES"
|
||||
if mkdir "$SK" 2>/dev/null; then
|
||||
printf '%s\n%s\n' "$OWNER" "$(date '+%m-%d %H:%M')" > "$SK/OWNER"
|
||||
echo "✓ 已领骨架资源:$RES($OWNER)—— 完后请 --release-skeleton $RES"
|
||||
exit 0
|
||||
fi
|
||||
echo "✗ 领骨架资源失败:$RES 已被占($(sed -n '1p' "$SK/OWNER" 2>/dev/null || echo '?'))→ 停手或换名(如迁移号 +1)" >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ "${1:-}" = "--release-skeleton" ]; then
|
||||
RES="${2:-}"; [ -n "$RES" ] || { echo "用法:--release-skeleton <资源名>" >&2; exit 2; }
|
||||
rm -rf "$LOCKSROOT/.migrations/$RES" && echo "✓ 已释放骨架资源:$RES"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ── 发布锁:commit / push / scp / build ⇒ 秒级独占(仓库级,不可并行)────
|
||||
if [ "${1:-}" = "--claim-publish" ]; then
|
||||
OWNER="${2:-${ME:-$(whoami)}}"
|
||||
mkdir -p "$LOCKSROOT" 2>/dev/null
|
||||
PB="$LOCKSROOT/.publish"
|
||||
if mkdir "$PB" 2>/dev/null; then
|
||||
printf '%s\n%s\n' "$OWNER" "$(date '+%m-%d %H:%M')" > "$PB/OWNER"
|
||||
echo "✓ 已持发布锁($OWNER)—— commit/push/scp/build 期间不可并行;完后立刻 --release-publish"
|
||||
exit 0
|
||||
fi
|
||||
echo "✗ 发布锁被占($(sed -n '1p' "$PB/OWNER" 2>/dev/null || echo '?'))→ 停手等它释放" >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ "${1:-}" = "--release-publish" ]; then
|
||||
rm -rf "$LOCKSROOT/.publish" && echo "✓ 已释放发布锁"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ── 锁总览(域锁 + 骨架锁 + 发布锁 + 旧全局锁)────────────
|
||||
if [ "${1:-}" = "--locks" ]; then
|
||||
echo "—— 锁总览 ——"
|
||||
[ -d "$LOCKEXEC" ] && { echo " [全局锁] $(sed -n '1p' "$LOCKEXEC/OWNER" 2>/dev/null)(未升级会话,全局独占)"; } || echo " [全局锁] 空闲"
|
||||
if [ -d "$LOCKSROOT" ]; then
|
||||
for L in "$LOCKSROOT"/*; do
|
||||
[ -d "$L" ] || continue
|
||||
b="$(basename "$L")"
|
||||
case "$b" in .gate) continue;; .migrations)
|
||||
for M in "$LOCKSROOT/.migrations"/*; do
|
||||
[ -d "$M" ] || continue
|
||||
echo " [骨架] $(basename "$M") ← $(sed -n '1p' "$M/OWNER" 2>/dev/null)"
|
||||
done; continue;; .publish)
|
||||
echo " [发布] $(sed -n '1p' "$L/OWNER" 2>/dev/null) ← 期间不可并行 commit/push/scp"; continue;; esac
|
||||
echo " [域锁] $(sed -n '1p' "$L/OWNER" 2>/dev/null)|域:$(tr '\n' ' ' < "$L/DOMAINS" 2>/dev/null)"
|
||||
done
|
||||
fi
|
||||
exit 0
|
||||
fi
|
||||
|
||||
|
||||
# ── 占位 / 释放 ─────────────────────────────────────────
|
||||
if [ "${1:-}" = "--claim" ]; then
|
||||
T="${2:-}"; OWNER="${3:-$(whoami)@$(date +%H:%M)}"
|
||||
[ -n "$T" ] || { echo "用法:--claim <单号> [占用者]" >&2; exit 2; }
|
||||
if mkdir "$LOCKDIR$T" 2>/dev/null; then
|
||||
printf '%s\n' "$OWNER" > "$LOCKDIR$T/OWNER"
|
||||
echo "✓ 已占位:05-交接单/.doing-$T($OWNER)—— 完工请 --release $T"
|
||||
exit 0
|
||||
fi
|
||||
echo "✗ 占位失败:05-交接单/.doing-$T 已存在(占用者 $(cat "$LOCKDIR$T/OWNER" 2>/dev/null || echo '?')) → 停手" >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ "${1:-}" = "--release" ]; then
|
||||
T="${2:-}"; [ -n "$T" ] || { echo "用法:--release <单号>" >&2; exit 2; }
|
||||
rm -rf "$LOCKDIR$T" && echo "✓ 已释放:05-交接单/.doing-$T" || { echo "✗ 释放失败" >&2; exit 1; }
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ── 预检 ────────────────────────────────────────────────
|
||||
MY_TASK="${1:-}"
|
||||
WINDOW="${GUARD_WINDOW:-30}"
|
||||
MINE="${MINE:-}"
|
||||
PUSH="${PUSH:-0}"
|
||||
VERDICT=0
|
||||
HARD=() # 硬失败原因
|
||||
[ -n "$MINE" ] && STRICT=1 || STRICT=0
|
||||
|
||||
echo "=================================================================="
|
||||
echo "并行冲突预检 root=$ROOT"
|
||||
echo "本次任务 = ${MY_TASK:-(未声明)}"
|
||||
echo "我是谁(ME) = ${ME:-(未声明 → 任何锁都按"别人的"处理)}"
|
||||
echo "我声明要改 = ${MINE:-(未声明 → 仅信息模式)}"
|
||||
echo "模式 = $([ "$PUSH" = "1" ] && echo 推送前检查 || echo 开工检查)"
|
||||
echo "=================================================================="
|
||||
|
||||
in_mine() { [ -z "$MINE" ] && return 1; for p in $MINE; do case "$1" in $p) return 0;; esac; done; return 1; }
|
||||
|
||||
# ── ① 占用锁(硬判定)────────────────────────────────────
|
||||
echo
|
||||
echo "【1】占用锁(05-交接单/.doing-*)"
|
||||
shopt -s nullglob
|
||||
LOCKS=("$LOCKDIR"*)
|
||||
if [ ${#LOCKS[@]} -eq 0 ]; then
|
||||
echo " ✓ 无人占用 —— ⚠ 这不是「可以开工」,是「**你快去抢**」:"
|
||||
echo " 开工前先:ME=\"<你的会话名>\" bash 07-scripts/handoff-guard.sh --claim <单号> \"<你的会话名>\""
|
||||
else
|
||||
for L in "${LOCKS[@]}"; do
|
||||
n="$(basename "$L")"; o="$(cat "$L/OWNER" 2>/dev/null || echo '(未写 OWNER)')"
|
||||
if [ -n "$MY_TASK" ] && [ "$n" = ".doing-$MY_TASK" ]; then
|
||||
echo " · $n ← 你自己占的($o)"
|
||||
else
|
||||
echo " ⚠ $n 被占用:$o → 冲突域重叠就别开工"
|
||||
VERDICT=1; HARD+=("① 别的会话持有占用锁 $n($o)")
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
# ── ①b 锁 ↔ 台账一致性(2026-09-12 新增)────────────────
|
||||
# 实证:T03 10:14 就被占了,但台账那行直到 12:59 还写着「待执行」→ 别人看台账会以为**没人做**,
|
||||
# 于是可能重复开工(这不是文件写冲突,而是"状态不可见"造成的重复劳动)。
|
||||
if [ ${#LOCKS[@]} -gt 0 ]; then
|
||||
for L in "${LOCKS[@]}"; do
|
||||
n="$(basename "$L")"; T="${n#.doing-}"
|
||||
row="$(grep -m1 "^| \`$T-" 05-交接单/README.md 2>/dev/null || true)"
|
||||
if [ -z "$row" ]; then
|
||||
echo " ⚠ $T 有占用锁,但 \`05-交接单/README.md §一\` 里**没有这一行** → 补上(新增单)"
|
||||
VERDICT=1; HARD+=("①b $T 有锁但台账缺行")
|
||||
elif printf '%s' "$row" | grep -qE "执行中|已完成|已归档"; then
|
||||
echo " · $T 台账已标「执行中/已完成」✓"
|
||||
else
|
||||
echo " ⚠ $T 已被占用($n),但台账那行仍写作「$(printf '%s' "$row" | awk -F'|' '{print $3}' | tr -d ' ')」"
|
||||
echo " → **台账滞后 = 别人可能重复开工**:立刻把该行状态改成「🔄 执行中」"
|
||||
VERDICT=1; HARD+=("①b $T 有锁但台账未标「执行中」(重复开工风险)")
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
# ── ①c 全局执行锁(2026-09-12 新增:同一时刻只允许一个执行会话)────
|
||||
echo
|
||||
echo "【1c】全局执行锁(05-交接单/.exec-lock)"
|
||||
if [ ! -d "$LOCKEXEC" ]; then
|
||||
echo " ✓ 无全局锁 —— ⚠️ 这**不是「可以开工」,是「你快去抢」**:"
|
||||
echo " 开工前必须先:bash 07-scripts/handoff-guard.sh --claim-exec \"<你的会话名>\""
|
||||
echo " (「环境干净」≠「没人动过」;抢锁是原子的,抢到才是你的开工许可)"
|
||||
else
|
||||
O1="$(sed -n '1p' "$LOCKEXEC/OWNER" 2>/dev/null)"
|
||||
O2="$(sed -n '2p' "$LOCKEXEC/OWNER" 2>/dev/null)"
|
||||
O3="$(sed -n '3p' "$LOCKEXEC/OWNER" 2>/dev/null)"
|
||||
if [ -n "$ME" ] && [ "$O1" = "$ME" ]; then
|
||||
echo " · 全局锁是**你自己**持有的($ME)—— 完工记得 --release-exec"
|
||||
else
|
||||
echo " ⚠ 全局锁被占用:$O1 $O2"
|
||||
[ -n "$O3" ] && echo " $O3"
|
||||
echo " → **同一时刻只允许一个执行会话**:要动「文档 / 代码 / 服务器」就先等它释放;"
|
||||
echo " ⛔ **不得人工删锁 / 不得接管**(R9):锁只能由持有者自己 --release-exec ——"
|
||||
echo " 抢不到 = 停手 + 报告用户;读 OWNER 仅用于「用户已点头、且用户自己撤锁之后」的续做"
|
||||
VERDICT=1; HARD+=("①c 别的会话持有全局执行锁($O1)→ 不允许并行执行")
|
||||
fi
|
||||
fi
|
||||
|
||||
# ── ①d 服务器侧操作锁(2026-09-12 新增,T04):平台高危操作的跨会话互斥 ────
|
||||
echo
|
||||
echo "【1d】服务器侧操作锁(/opt/dsh/state/.op-lock)"
|
||||
if [ "${SKIP_OPLOCK:-0}" = "1" ]; then
|
||||
echo " · 已跳过(SKIP_OPLOCK=1)"
|
||||
else
|
||||
OPLOCK_OUT="$(ssh -o BatchMode=yes -o ConnectTimeout=8 "${OP_LOCK_REMOTE:-bt-server}" \
|
||||
"if [ -d /opt/dsh/state/.op-lock ]; then ls -1 /opt/dsh/state/.op-lock 2>/dev/null | grep -v '^README\$' || true; else echo __NOLOCKDIR__; fi" 2>/dev/null)"; OPLOCK_RC=$?
|
||||
if [ "$OPLOCK_RC" -ne 0 ]; then
|
||||
echo " · 服务器不可达(离线)→ **只提示、不失败**;但要动线上时须先恢复可见性再动手"
|
||||
elif [ "$OPLOCK_OUT" = "__NOLOCKDIR__" ]; then
|
||||
echo " ⚠ 锁根目录不存在:/opt/dsh/state/.op-lock(T04 应已建立 → 需复查)"
|
||||
VERDICT=1; HARD+=("①d 服务器侧锁根目录缺失")
|
||||
elif [ -z "$OPLOCK_OUT" ]; then
|
||||
echo " ✓ 无平台操作锁(此刻没有会话在动线上)"
|
||||
else
|
||||
echo " ⚠ 有会话正在动线上:"
|
||||
for L in $OPLOCK_OUT; do
|
||||
echo " 🔴 $L"
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=8 "${OP_LOCK_REMOTE:-bt-server}" \
|
||||
"sed 's/^/ /' '/opt/dsh/state/.op-lock/$L/OWNER' 2>/dev/null" 2>/dev/null || true
|
||||
done
|
||||
echo " → 凡「重启 / drain / 改实例 env·quota / 批量铺插件 / 改 nginx·nft·证书」类操作,"
|
||||
echo " 开工前必须先占位(bash 07-scripts/op-lock.sh claim <操作名> \"<影响面>\");"
|
||||
echo " 占位失败 = 有会话在动线上 → **停手**。"
|
||||
VERDICT=1; HARD+=("①d 有会话持有服务器侧操作锁($OPLOCK_OUT)")
|
||||
fi
|
||||
fi
|
||||
|
||||
# ── ② 越界改动(推送前硬判定)────────────────────────────
|
||||
echo
|
||||
echo "【2】未提交改动(含未跟踪)"
|
||||
CHANGED="$(git -c core.quotepath=false status --short 2>/dev/null | awk '{print $NF}' | grep -vE '^05-交接单/\.(doing-|exec-lock)' || true)"
|
||||
OUTSIDE=(); INSIDE=()
|
||||
if [ -z "$CHANGED" ]; then
|
||||
echo " ✓ 工作区干净"
|
||||
else
|
||||
while IFS= read -r f; do
|
||||
[ -z "$f" ] && continue
|
||||
if in_mine "$f"; then INSIDE+=("$f"); else OUTSIDE+=("$f"); fi
|
||||
done <<< "$CHANGED"
|
||||
echo " · 我声明的:${#INSIDE[@]} 个"
|
||||
if [ ${#OUTSIDE[@]} -eq 0 ]; then
|
||||
echo " ✓ 无越界改动(没有别人的半成品混在里面)"
|
||||
else
|
||||
echo " ⚠ 越界(别人的/我没声明):${#OUTSIDE[@]} 个"
|
||||
for f in "${OUTSIDE[@]:0:6}"; do echo " $f"; done
|
||||
[ ${#OUTSIDE[@]} -gt 6 ] && echo " …还有 $(( ${#OUTSIDE[@]} - 6 )) 个"
|
||||
echo " → **推送时只能 scp 自己声明的文件,切勿 'git add -A'**(本项不构成硬失败;"
|
||||
echo " 真正的推送硬判定在【4】——那里能精确看出「你正要推什么」)"
|
||||
fi
|
||||
fi
|
||||
|
||||
# ── ③ 热点提示(仅提示,不作判定)────────────────────────
|
||||
echo
|
||||
echo "【3】近 $WINDOW 分钟被改动的文件(**仅提示**:mtime 分不清谁改的)"
|
||||
HOT="$(find . -type f -mmin "-$WINDOW" \
|
||||
-not -path './.git/*' -not -path '*/node_modules/*' -not -name '*.bak*' \
|
||||
-not -path './05-交接单/.doing-*' -not -path './05-交接单/.exec-lock*' 2>/dev/null | sed 's|^\./||' | LC_ALL=C sort)"
|
||||
if [ -z "$HOT" ]; then
|
||||
echo " ✓ 无(这块是「冷」的)"
|
||||
else
|
||||
HIT=(); OTHER=""
|
||||
while IFS= read -r f; do
|
||||
[ -z "$f" ] && continue
|
||||
if in_mine "$f"; then HIT+=("$f"); else OTHER="$OTHER$f"$'\n'; fi
|
||||
done <<< "$HOT"
|
||||
if [ ${#HIT[@]} -gt 0 ]; then
|
||||
echo " ⚠ **我的目标文件近期被改动过**(可能是你自己,也可能是别人)→ 改前务必先读最新内容:"
|
||||
for f in "${HIT[@]}"; do echo " $f"; done
|
||||
else
|
||||
echo " · 我的目标文件均未被近期改动"
|
||||
fi
|
||||
echo " · 其它近期被改动的文件:$(printf '%s' "$OTHER" | grep -c . || true) 个(与你无关,仅供感知全库热度)"
|
||||
[ "$WINDOW" -gt 10 ] && echo " 想更锐利:GUARD_WINDOW=10 再跑一次"
|
||||
fi
|
||||
|
||||
# ── ④ 双端一致性(推送前硬判定)──────────────────────────
|
||||
echo
|
||||
echo "【4】双端一致性(docs-sync-check.sh)"
|
||||
if [ "${SKIP_SYNC:-0}" = "1" ]; then
|
||||
echo " (SKIP_SYNC=1,跳过)"
|
||||
elif [ -f 07-scripts/docs-sync-check.sh ]; then
|
||||
# 全量对账耗时(2026-09-12 实测):优化前 **2m15s**(Git Bash 逐文件 spawn md5sum → 曾被当成"挂死"),
|
||||
# 优化后 **≈28s**。这里再加硬超时兜底,避免对账自身卡住时把整个 guard 拖死;急用时 `SKIP_SYNC=1` 跳过。
|
||||
if command -v timeout >/dev/null 2>&1; then
|
||||
SYNC="$(timeout 300 bash 07-scripts/docs-sync-check.sh 2>/dev/null)"
|
||||
else
|
||||
SYNC="$(bash 07-scripts/docs-sync-check.sh 2>/dev/null)"
|
||||
fi
|
||||
printf '%s\n' "$SYNC" | tail -6
|
||||
# 关键:对账结果里的「仅本地(待推送)」= 一次朴素推送**实际会推上去**的东西。
|
||||
# 其中只要有一个「不在我声明清单里」,就是幽灵文件信号(2026-09-12 事故:把已被对方归档的
|
||||
# T02 又推回服务器)→ 推送前硬失败。
|
||||
ONLYL="$(printf '%s\n' "$SYNC" | sed -n 's/.*仅本地(待推送) //p')"
|
||||
if [ -n "$ONLYL" ]; then
|
||||
GHOST=(); GOOD=0
|
||||
while IFS= read -r f; do
|
||||
[ -z "$f" ] && continue
|
||||
if in_mine "$f"; then GOOD=$((GOOD+1)); else GHOST+=("$f"); fi
|
||||
done <<< "$ONLYL"
|
||||
echo " · 仅本地(= 一次朴素推送会推上去的):$GOOD 个已声明 + ${#GHOST[@]} 个未声明"
|
||||
if [ ${#GHOST[@]} -gt 0 ]; then
|
||||
echo " ⚠ 未声明却「仅本地」——**幽灵文件风险**:"
|
||||
for f in "${GHOST[@]}"; do echo " $f"; done
|
||||
echo " → 它可能是别人**刚归档/移走**的文件(你今天就是这么推回去的)"
|
||||
if [ "$PUSH" = "1" ]; then
|
||||
VERDICT=1; HARD+=("④ 有 ${#GHOST[@]} 个「仅本地」文件不在你的推送清单里 → 停手核实后再推")
|
||||
else
|
||||
echo " (开工阶段仅提示;推送前请带 PUSH=1 复跑)"
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
else
|
||||
echo " (未找到 07-scripts/docs-sync-check.sh)"
|
||||
fi
|
||||
|
||||
# ── 结论 ────────────────────────────────────────────────
|
||||
echo
|
||||
echo "=================================================================="
|
||||
if [ "$VERDICT" -ne 0 ]; then
|
||||
echo "结论:**不可放行** ——"
|
||||
for r in "${HARD[@]}"; do echo " $r"; done
|
||||
elif [ "$STRICT" -eq 0 ]; then
|
||||
echo "结论:未声明改动清单 → 仅为信息输出,**不构成放行依据**(开工请带 MINE=\"...\")"
|
||||
echo " ⚠️ 且「无锁」不等于「可以开工」—— 开工前必须先 --claim-exec 抢锁(见上方【1c】)"
|
||||
echo " (2026-09-12 实证:把「无锁」读成「可以动手」,导致两个会话同时改库)"
|
||||
else
|
||||
echo "结论:未命中硬冲突 → 可以继续(仍须遵守:Edit 增量 / 改前先读 / 只推自己的文件)"
|
||||
fi
|
||||
exit "$VERDICT"
|
||||
@@ -0,0 +1,68 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""handoff-status.py — 一条命令看清 05-交接单 里哪些单子【待执行】/【已完成】/【无状态】。
|
||||
|
||||
为什么需要它:`05-交接单/交接单-已完成/` 那层目录摊平后,**"目录位置"不再表达状态**。
|
||||
状态改由**机读字段**承载(与布局解耦,摊平/搬家都不失效):
|
||||
① 单子头部 `- 状态:…` 行(模板已有)→ ② 回退取 `05-交接单/README.md` 台账表的「状态」列
|
||||
|
||||
用法:python 07-scripts/handoff-status.py
|
||||
"""
|
||||
import os
|
||||
import re
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
D = os.path.join(ROOT, "05-交接单")
|
||||
README = os.path.join(D, "README.md")
|
||||
|
||||
HEAD_STATUS = re.compile(r"^[\s>*\-|#]*\*{0,2}状态\*{0,2}\s*[::]\s*(.+?)\s*$", re.M)
|
||||
DONE_KW = ("已完成", "已落地", "已归档", "已收口", "已闭合", "已实施", "已上线",
|
||||
"已修复", "已释放", "已结束", "已通过", "全部完成", "全绿", "✅")
|
||||
WAIT_KW = ("待执行", "待开工", "进行中", "在途", "待验证", "待跑", "规划中", "未开工",
|
||||
"部分执行", "受阻", "未做", "设计定稿", "🔄", "⏳", "⏸", "🟡")
|
||||
|
||||
table = {}
|
||||
if os.path.exists(README):
|
||||
with open(README, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
if not line.startswith("|"):
|
||||
continue
|
||||
cells = [c.strip() for c in line.strip().strip("|").split("|")]
|
||||
if len(cells) < 2 or ".md" not in cells[0]:
|
||||
continue
|
||||
table[cells[0].strip("`").split("/")[-1].strip()] = cells[1]
|
||||
|
||||
groups = {"待执行": [], "已完成": [], "无状态": []}
|
||||
targets = []
|
||||
for root2, dirs2, _f in os.walk(D):
|
||||
dirs2[:] = [x for x in dirs2 if not x.startswith(".")]
|
||||
for f2 in _f:
|
||||
if f2.endswith(".md") and f2 != "README.md" and not f2.startswith("."):
|
||||
rel = os.path.relpath(os.path.join(root2, f2), D).replace("\\", "/")
|
||||
targets.append((rel, os.path.join(root2, f2)))
|
||||
for n, path in sorted(targets):
|
||||
base = n.split("/")[-1]
|
||||
with open(path, encoding="utf-8", errors="ignore") as fh:
|
||||
head = "".join(fh.readlines()[:30])
|
||||
m = HEAD_STATUS.search(head)
|
||||
if m:
|
||||
s, src = m.group(1)[:44], "头部"
|
||||
elif base in table:
|
||||
s, src = table[base][:44], "台账表"
|
||||
else:
|
||||
s, src = "", ""
|
||||
if not s:
|
||||
groups["无状态"].append((n, "—"))
|
||||
elif any(k in s for k in DONE_KW):
|
||||
groups["已完成"].append((n, "%s ·%s" % (s, src)))
|
||||
elif any(k in s for k in WAIT_KW):
|
||||
groups["待执行"].append((n, "%s ·%s" % (s, src)))
|
||||
else:
|
||||
groups["无状态"].append((n, "%s ·%s未识别" % (s, src)))
|
||||
|
||||
for g in ("待执行", "已完成", "无状态"):
|
||||
print("【%s】%d 个" % (g, len(groups[g])))
|
||||
for n, s in groups[g]:
|
||||
print(" %-44s %s" % (n[:44], s))
|
||||
print()
|
||||
print("合计 %d 个单子(不含 README.md)" % sum(len(v) for v in groups.values()))
|
||||
@@ -0,0 +1,380 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
lock-guard-hook.py — 「无锁不许改库」的强制钩子(WorkBuddy PreToolUse / SessionStart)
|
||||
|
||||
为什么需要它
|
||||
────────────
|
||||
2026-09-12 实证:`handoff-guard.sh` 输出「无全局锁」被会话读成「环境干净,可以开工」
|
||||
(正确读法是「你快去抢锁」),结果**两个会话同时改了本库**。措辞已在 guard 与
|
||||
`05-交接单/README.md` 里补正,但**约定拦不住不看的人** —— 本脚本是强制层。
|
||||
|
||||
作用域(**刻意收窄:对其他项目零影响**)
|
||||
────────────
|
||||
仅当 `Write` / `Edit` 的目标路径落在下列根之内,才做锁判定;其余一律放行:
|
||||
· 文档库 <DSH_DOCS_ROOT>(默认 E:\\ProgramData\\AI技能\\aliyun-dsh-server\\dsh-server-docs)
|
||||
· 代码库 <DSH_CODE_REPO>(默认 D:\\github\\dsh_shenxian)
|
||||
⚠️ 例外:路径中含 `.workbuddy` 目录段的一律放行 —— 会话记忆 / 自动化工作数据属“运行态数据”,不是“仓库内容”。(2026-09-13 实测:「代码仓三方同步(DSH)」自动化写自己的 memory/*.md 时被本钩子 deny,属误伤)
|
||||
|
||||
判据
|
||||
────
|
||||
`<文档库根>/05-交接单/.exec-lock` 存在 = 有人持锁 → 放行(归属由台账 OWNER 承担);
|
||||
不存在 = **deny**,并把可直接粘贴的抢锁命令回给 Agent。
|
||||
|
||||
为什么**不拦 Bash**
|
||||
──────────────────
|
||||
① 抢锁命令本身必须能跑(否则把自己锁死 —— 拿不到锁就永远开不了工);
|
||||
② Bash 写文件是少数派、且破坏面可见(git status);
|
||||
③ 宁可留一个显式的安全阀,也不要一个可能导致死锁的强制层。
|
||||
|
||||
用法(settings.json 的 hooks 段,见同目录 README 或档案 73)
|
||||
"PreToolUse": [{ "matcher": "Write|Edit", "hooks": [{ "type": "command", "command": "<python> <此脚本>", "timeout": 10 }] }]
|
||||
"SessionStart":[{ "matcher": "startup", "hooks": [{ "type": "command", "command": "<python> <此脚本>", "timeout": 10 }] }]
|
||||
|
||||
退出码:始终 0(判定通过 JSON 输出表达);脚本自身异常也放行,绝不误伤。
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
|
||||
DOCS_ROOT = os.environ.get("DSH_DOCS_ROOT", r"D:\github\dsh_shenxian\dsh-server-docs")
|
||||
CODE_REPO = os.environ.get("DSH_CODE_REPO", r"D:\github\dsh_shenxian")
|
||||
LOCK_DIR = os.path.join(DOCS_ROOT, "05-交接单", ".exec-lock")
|
||||
OWNER_FILE = os.path.join(LOCK_DIR, "OWNER")
|
||||
LOCKS_ROOT = os.path.join(DOCS_ROOT, "05-交接单", ".locks")
|
||||
|
||||
GUARD_CMD = 'bash 07-scripts/handoff-guard.sh --claim-exec "<你的会话名>"'
|
||||
GUARD_DOMAIN_CMD = 'bash 07-scripts/handoff-guard.sh --claim-exec "<你的会话名>" --domains <域...>'
|
||||
|
||||
# ⚠️ 顺序有意义:**先长后短**,与 handoff-guard.sh 的 _ANCHOR_SEGS 逐字一致。
|
||||
# 两侧只要顺序或集合不同 ⇒ 同一文件算出不同域键 ⇒ 域锁静默失效(假绿)。
|
||||
_DOMAIN_SEGS = ("dsh-server-docs", "aliyun-dsh-server", "src", "poc", "web", "test", "docs", "07-scripts", "08-skills", "05-交接单")
|
||||
|
||||
|
||||
def _owner_text(d: str) -> str:
|
||||
try:
|
||||
with open(os.path.join(d, "OWNER"), encoding="utf-8") as f:
|
||||
return f.read()
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def _my_name(payload: dict) -> str:
|
||||
"""本会话名 —— hook 进程拿不到 `ME`(那是 bash 侧变量),依次尝试:
|
||||
① payload/env 的 `DSH_SESSION_NAME`(handoff-guard 抢锁时会写进 OWNER)
|
||||
② env `ME`(本地手动调用时)
|
||||
③ OWNER 里**唯一**写着本 session_id 的那把锁(真多会话下 id 唯一,可安全用它反查)
|
||||
"""
|
||||
for k in ("DSH_SESSION_NAME",):
|
||||
v = os.environ.get(k, "")
|
||||
if v:
|
||||
return v
|
||||
v = os.environ.get("ME", "")
|
||||
if v:
|
||||
return v
|
||||
return ""
|
||||
|
||||
|
||||
def _name_owner_of_id(ids: set, exclude: str = "") -> str:
|
||||
"""用 session_id 反查"我的那把锁"的会话名(仅当它唯一时才返回,避免多锁歧义)。"""
|
||||
if not ids or not os.path.isdir(LOCKS_ROOT):
|
||||
return ""
|
||||
hits = []
|
||||
for name in os.listdir(LOCKS_ROOT):
|
||||
if name.startswith("."):
|
||||
continue
|
||||
d = os.path.join(LOCKS_ROOT, name)
|
||||
if not os.path.isdir(d):
|
||||
continue
|
||||
txt = _owner_text(d)
|
||||
if any(i and i in txt for i in ids):
|
||||
first = (txt.splitlines() or [""])[0].strip()
|
||||
if first and first != exclude:
|
||||
hits.append(first)
|
||||
hits = list(dict.fromkeys(hits))
|
||||
return hits[0] if len(hits) == 1 else ""
|
||||
|
||||
|
||||
def _is_mine(owner_txt: str, ids: set, cwd: str, my_name: str = "") -> bool:
|
||||
"""归属判据 —— ⚠️ 会话名与 session_id **同时**吻合才算我的。
|
||||
|
||||
2026-09-22 实测踩到:同一进程内用 `ME=会话A/会话B` 模拟两会话时,`CODEBUDDY_SESSION_ID`
|
||||
是**真实环境变量**、两份 OWNER 都写着同一个 id ⇒ 只按 id 判会把两会话都认作"自己"
|
||||
⇒ 域隔离**全部放行**(假绿,最危险的失效形态)。
|
||||
⇒ 判据收紧为:**会话名相等** 且(id 命中 或 未取到 id)。
|
||||
"""
|
||||
first = (owner_txt.splitlines() or [""])[0].strip()
|
||||
id_hit = bool(ids) and any(i and i in owner_txt for i in ids)
|
||||
if my_name:
|
||||
return first == my_name and id_hit
|
||||
if id_hit:
|
||||
return True
|
||||
if cwd:
|
||||
nc = norm(cwd)
|
||||
if nc and nc in norm(owner_txt):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _iter_domain_locks():
|
||||
"""遍历所有域锁目录(跳过 .gate / .migrations / .publish)。"""
|
||||
if not os.path.isdir(LOCKS_ROOT):
|
||||
return
|
||||
for name in os.listdir(LOCKS_ROOT):
|
||||
if name.startswith("."):
|
||||
continue
|
||||
d = os.path.join(LOCKS_ROOT, name)
|
||||
if not os.path.isdir(d):
|
||||
continue
|
||||
try:
|
||||
with open(os.path.join(d, "DOMAINS"), encoding="utf-8") as f:
|
||||
doms = [ln.strip() for ln in f if ln.strip()]
|
||||
except Exception:
|
||||
continue
|
||||
yield d, _owner_text(d), doms
|
||||
|
||||
|
||||
def domain_key(path: str) -> str:
|
||||
"""路径 → 域键(前两段,与 handoff-guard.sh norm_domain 同规则 ⇒ 粒度到目录,宁可保守)。"""
|
||||
n = norm(path).replace("\\", "/")
|
||||
parts = [p for p in n.split("/") if p]
|
||||
for seg in _DOMAIN_SEGS:
|
||||
if seg in parts:
|
||||
i = parts.index(seg)
|
||||
return "/".join(parts[i:i + 2]).lower()
|
||||
return "/".join(parts[-2:]).lower() if len(parts) >= 2 else n.lower()
|
||||
|
||||
|
||||
def has_my_domain_lock(ids: set, cwd: str, my_name: str = "") -> bool:
|
||||
mn = my_name or _name_owner_of_id(ids)
|
||||
for d, ot, _ in _iter_domain_locks():
|
||||
if _is_mine(ot, ids, cwd, mn):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def blocked_by_domain(path: str, ids: set, cwd: str, my_name: str = "") -> tuple:
|
||||
"""目标是否落在**别人**的域内 ⇒ (占用者, 冲突域)。自己的域锁一律放行。"""
|
||||
key = domain_key(path)
|
||||
mn = my_name or _name_owner_of_id(ids)
|
||||
for d, ot, doms in _iter_domain_locks():
|
||||
if _is_mine(ot, ids, cwd, mn):
|
||||
continue
|
||||
if key and key in doms:
|
||||
return ((ot.splitlines()[0] if ot else "?"), key)
|
||||
return ("", "")
|
||||
|
||||
|
||||
def norm(p: str) -> str:
|
||||
try:
|
||||
return os.path.normcase(os.path.normpath(p))
|
||||
except Exception:
|
||||
return p
|
||||
|
||||
|
||||
PROTECTED = [norm(DOCS_ROOT), norm(CODE_REPO)]
|
||||
|
||||
|
||||
def is_protected(path: str) -> bool:
|
||||
if not path:
|
||||
return False
|
||||
n = norm(path)
|
||||
# 会话记忆 / 自动化工作数据不是「仓库内容」:写它不该被本钩子拦(2026-09-13 新增)
|
||||
if ".workbuddy" in n.split(os.sep):
|
||||
return False
|
||||
for root in PROTECTED:
|
||||
if n == root or n.startswith(root + os.sep):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def lock_owner() -> str:
|
||||
try:
|
||||
with open(OWNER_FILE, encoding="utf-8") as f:
|
||||
return (f.readline() or "").strip() or "(未写 OWNER)"
|
||||
except Exception:
|
||||
return "(未写 OWNER)"
|
||||
|
||||
|
||||
def out(obj: dict) -> None:
|
||||
# ⚠️ 2026-09-16 修复(**根因**):必须走 **`sys.stdout.buffer` 写 bytes**,⛔ 不能走文本层
|
||||
# (`sys.stdout.write(str)`)。
|
||||
# 实证:宿主 spawn 本脚本时 stdout 编码**不保证是 UTF-8**(本机 `PYTHONUTF8=1` 在 hook 环境下
|
||||
# 不保证生效),而 deny 文案里含 `⛔` / `🔐` 等字符 ⇒ 文本层写抛 `UnicodeEncodeError` ⇒
|
||||
# 异常冒泡、**stdout 为空** ⇒ 宿主**收不到 deny** ⇒ **静默放行**。
|
||||
# 而 `hook_log()` 已经把 `PreToolUse-deny` 写进日志 ⇒ 表现成 **「日志里有 DENY,但拦不住」**
|
||||
# (2026-09-16 受控自测:无锁时 Write 文档库 ⇒ 文件真被创建)。
|
||||
# 对照:`bash-output-guard.py` 用 `sys.stdout.buffer.write(bytes)` ⇒ **实测拦得住**。
|
||||
# 📌 教训:本脚本的 `stdin` 早前已按 buffer 加固(见 main 顶部注释),**stdout 漏了** ——
|
||||
# 「读写要同时加固」,只修一半等于没修。
|
||||
data = json.dumps(obj, ensure_ascii=False).encode('utf-8')
|
||||
try:
|
||||
sys.stdout.buffer.write(data)
|
||||
sys.stdout.buffer.flush()
|
||||
except Exception:
|
||||
try:
|
||||
sys.stdout.write(data.decode('utf-8', 'replace'))
|
||||
sys.stdout.flush()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
LOG_FILE = os.environ.get(
|
||||
"DSH_LOCK_HOOK_LOG", os.path.join(os.path.dirname(DOCS_ROOT), ".workbuddy", "lock-hook.log")
|
||||
)
|
||||
|
||||
|
||||
def hook_log(event: str, detail: str) -> None:
|
||||
"""低频自证日志:只在 SessionStart 与 deny 时写一行 —— 用来回答「hook 到底有没有被触发」。
|
||||
|
||||
为什么需要:hook 配置是**启动时缓存**的,改完必须完全重启才加载;没有日志就只能靠猜。
|
||||
写入失败一律静默(hook 绝不能因为自己出问题而干扰工作)。
|
||||
"""
|
||||
try:
|
||||
os.makedirs(os.path.dirname(LOG_FILE), exist_ok=True)
|
||||
with open(LOG_FILE, "a", encoding="utf-8") as f:
|
||||
f.write(f"{time.strftime('%Y-%m-%d %H:%M:%S')}\t{event}\t{detail}\n")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def main() -> None:
|
||||
try:
|
||||
try:
|
||||
# ⚠️ 必须走 buffer 显式 UTF-8:本机环境设了 PYTHONUTF8=1,但只要有人给脚本加 `-E`
|
||||
# 就会被屏蔽 ⇒ sys.stdin 回退 cp936 ⇒ 含中文路径(`E:\ProgramData\AI技能\…`)的
|
||||
# payload 解析即炸。而本函数是 **fail-open**(读不懂就放行)⇒ 会**静默失效**:
|
||||
# 锁守卫不再拦人,却没有任何痕迹。2026-09-15 实测(同款坑已在 stop-dialog-guard.py 踩过)。
|
||||
raw = sys.stdin.buffer.read().decode("utf-8", "replace")
|
||||
except Exception:
|
||||
raw = sys.stdin.read()
|
||||
payload = json.loads(raw) if raw.strip() else {}
|
||||
except Exception as e:
|
||||
# 不留痕 = 失效无声(判不出"没被调用"与"被静默放行")⇒ 必须记一行
|
||||
hook_log("payload-unparsable", "%s: %s" % (type(e).__name__, str(e)[:80]))
|
||||
return # 读不懂 payload → 放行(fail-open 方向正确,但不能无声)
|
||||
|
||||
tool = payload.get("tool_name")
|
||||
event = payload.get("hook_event_name") or payload.get("hook_event") or ""
|
||||
|
||||
# ★ 入口即留痕(2026-09-16 加)—— 与 stop-dialog-guard.py / skill-load-guard.py 同款。
|
||||
# 为什么不只记 SessionStart 与 deny:2026-09-16 实测 —— 重启后本钩子**一行都没写**,
|
||||
# 当时无法区分"没被宿主调用"与"被静默放行",只能靠猜(最后查明:SessionStart 的
|
||||
# `matcher: "startup"` 不匹配"恢复会话",钩子压根没被执行)。
|
||||
# 留痕里带上**实际收到的字段名**(keys):下次宿主字段一变,日志里立刻能看出来,
|
||||
# 而不是等某天发现守卫早已失效。低频:每次调用一行。
|
||||
hook_log("entry", "event=%s|tool=%s|keys=%s" % (
|
||||
event or "?", tool or "-", ",".join(sorted(payload.keys()))[:220]))
|
||||
|
||||
# ── SessionStart:只提示(该事件的输出只给用户看,不会进 Agent 上下文)──
|
||||
if event == "SessionStart" or (tool is None and "source" in payload):
|
||||
src = payload.get("source", "?")
|
||||
_sid = payload.get("session_id", "")
|
||||
_ids = {_sid} if _sid else {x for x in (os.environ.get("CODEBUDDY_SESSION_ID", ""),) if x}
|
||||
_cwd = payload.get("cwd") or ""
|
||||
if has_my_domain_lock(_ids, _cwd):
|
||||
msg = "🔐 你持有域锁 —— 只有落进**别人**域的写入会被拦;锁总览:`handoff-guard.sh --locks`"
|
||||
hook_log("SessionStart", f"本会话持域锁 source={src}")
|
||||
elif os.path.isdir(LOCK_DIR):
|
||||
who = lock_owner()
|
||||
msg = f"🔐 全局执行锁【已被占用】:{who} —— 但**域锁可绕开它**:不重叠的域仍能并行。"
|
||||
hook_log("SessionStart", f"旧全局锁被占 owner={who} source={src}")
|
||||
elif os.path.isdir(DOCKS := os.path.join(DOCS_ROOT, "05-交接单", ".locks")) and os.listdir(DOCKS):
|
||||
msg = "🔐 已有域锁在跑 —— 开工前先抢你要的域(不重叠即可并行):" + GUARD_DOMAIN_CMD
|
||||
hook_log("SessionStart", f"有域锁无全局锁 source={src}")
|
||||
elif os.path.isdir(DOCS_ROOT):
|
||||
msg = "🔐 无锁【空闲】—— 「空闲」≠「可以开工」:先判定可锁定范围再抢锁 →\n " + GUARD_DOMAIN_CMD
|
||||
hook_log("SessionStart", f"空闲 source={src}")
|
||||
else:
|
||||
msg = ""
|
||||
if msg:
|
||||
out({"systemMessage": msg, "suppressOutput": True})
|
||||
return
|
||||
|
||||
# ── PreToolUse:真正的强制点 ──
|
||||
if tool not in ("Write", "Edit", "MultiEdit", "NotebookEdit"):
|
||||
return
|
||||
ti = payload.get("tool_input") or {}
|
||||
fp = ti.get("file_path") or ti.get("path") or ti.get("notebook_path") or ""
|
||||
if not is_protected(fp):
|
||||
return
|
||||
if not os.path.isdir(DOCS_ROOT):
|
||||
return # 库不在这台机器上 → 与本约定无关,放行
|
||||
|
||||
sid = payload.get("session_id") or ""
|
||||
cwd = payload.get("cwd") or ""
|
||||
# ⚠️ **payload 有 session_id ⇒ 只用它**,⛔ 不要与 env 取并集!
|
||||
# 并集会让"payload 里的伪造/过期 id"叠加进程 env 里的真实 id ⇒ 假会话被判成"我" ⇒ 放行(实测踩过)。
|
||||
# env 仅作**缺省兜底**(payload 未带 session_id 时)。
|
||||
ids = {sid} if sid else {x for x in (os.environ.get("CODEBUDDY_SESSION_ID", ""),) if x}
|
||||
my_name = payload.get("session_name") or _my_name(payload)
|
||||
|
||||
# ① 旧全局锁存在 ⇒ 兼容:只对**未持域锁**的会话放行(旧行为),
|
||||
# 已持域锁的会话仍走 ② 做域隔离(2026-09-22 修正)
|
||||
# ⚠️ 修正原因:首版让「旧锁存在 ⇒ 一律放行」⇒ 域隔离**完全失效**(实测命中)。
|
||||
legacy_open = os.path.isdir(LOCK_DIR)
|
||||
if legacy_open and not has_my_domain_lock(ids, cwd, my_name):
|
||||
return
|
||||
|
||||
# ② 域锁路径(2026-09-22 加):自己持域锁 ⇒ 只看是否落进**别人**的域
|
||||
if has_my_domain_lock(ids, cwd, my_name):
|
||||
who, dom = blocked_by_domain(fp, ids, cwd, my_name)
|
||||
if not who:
|
||||
return # 在我自己的域内(或无人占该域)⇒ 放行
|
||||
hook_log("PreToolUse-deny-domain", f"{tool} {fp} ← 域 {dom} 属 {who}")
|
||||
reason = (
|
||||
"⛔ 被「域锁」拦下:目标文件落进了**别的会话**的冲突域。\n\n"
|
||||
f"冲突域:{dom}\n占用会话:{who}\n\n"
|
||||
"域锁的设计意图 = **不重叠的域可以并行**,重叠的域必须串行。\n"
|
||||
"你有两个合规选择(⛔ 都不得删对方锁,R9):\n"
|
||||
" ① 等对方释放该域;\n"
|
||||
" ② **改做不重叠的域**(这才是域锁的价值所在)。\n\n"
|
||||
f"查全部锁:bash 07-scripts/handoff-guard.sh --locks"
|
||||
)
|
||||
out({
|
||||
"hookSpecificOutput": {
|
||||
"hookEventName": "PreToolUse",
|
||||
"permissionDecision": "deny",
|
||||
"permissionDecisionReason": reason,
|
||||
},
|
||||
"systemMessage": f"⛔ 已拦下跨域写入(域 {dom} 属 {who})",
|
||||
"suppressOutput": True,
|
||||
})
|
||||
return
|
||||
|
||||
# ③ 既无全局锁、也无自己的域锁 ⇒ 拒写(与旧行为一致)
|
||||
hook_log("PreToolUse-deny", f"{tool} {fp}")
|
||||
|
||||
reason = (
|
||||
"⛔ 被「锁机制」拦下:本仓库(文档库 / 代码库)当前**你没有持任何锁**,"
|
||||
"不允许直接修改。\n\n"
|
||||
"「无锁」的正确读法 = **「你快去抢」**,不是「可以开工」"
|
||||
"(2026-09-12 实证:两个会话把「无锁」读成「环境干净」→ 同时改了本库)。\n\n"
|
||||
"请先执行(Bash,不受本钩子限制)——**两种锁选一种**:\n\n"
|
||||
" 【域锁】只改局部的模块/文件(推荐,可与他人并行)\n"
|
||||
f" cd \"{DOCS_ROOT}\"\n {GUARD_DOMAIN_CMD}\n\n"
|
||||
" 【全局锁】无域可声明,或不确定(旧行为,独占)\n"
|
||||
f" cd \"{DOCS_ROOT}\"\n {GUARD_CMD}\n\n"
|
||||
"抢到 = 开工许可;**抢不到 = 冲突域被占 → 停手或改做不重叠的域**。\n"
|
||||
"开工前先判定可锁定范围:`bash 07-scripts/preflight-lock.sh \"<会话名>\" <目标文件...>`。"
|
||||
)
|
||||
out(
|
||||
{
|
||||
"hookSpecificOutput": {
|
||||
"hookEventName": "PreToolUse",
|
||||
"permissionDecision": "deny",
|
||||
"permissionDecisionReason": reason,
|
||||
},
|
||||
"systemMessage": "⛔ 已拦下一次无锁写入(本仓库要求先抢锁)",
|
||||
"suppressOutput": True,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
try:
|
||||
main()
|
||||
except Exception:
|
||||
pass # 任何异常都不阻断工作
|
||||
sys.exit(0)
|
||||
@@ -0,0 +1,65 @@
|
||||
#!/usr/bin/env python3
|
||||
"""nginx 访问日志慢请求/大响应分析(只读)
|
||||
用法: python3 nginx-slow.py <logfile> [host子串] [最近N行] [起始时间标记]
|
||||
"""
|
||||
import re
|
||||
import sys
|
||||
|
||||
log = sys.argv[1]
|
||||
host = sys.argv[2] if len(sys.argv) > 2 else ''
|
||||
tail_lines = int(sys.argv[3]) if len(sys.argv) > 3 else 20000
|
||||
since = sys.argv[4] if len(sys.argv) > 4 else ''
|
||||
|
||||
LINE = re.compile(
|
||||
r'^(?P<ip>\S+) \S+ \S+ \[(?P<time>[^\]]+)\] "(?P<req>[^"]*)" (?P<status>\d{3}) (?P<bytes>\d+)'
|
||||
r'(?: "[^"]*" "[^"]*")?(?P<rest>.*)$'
|
||||
)
|
||||
|
||||
rows = []
|
||||
with open(log, 'r', errors='replace') as fh:
|
||||
lines = fh.readlines()[-tail_lines:]
|
||||
|
||||
for line in lines:
|
||||
if host and host not in line:
|
||||
continue
|
||||
if since and since not in line:
|
||||
continue
|
||||
m = LINE.match(line)
|
||||
if not m:
|
||||
continue
|
||||
times = re.findall(r'(\d+\.\d+)', m.group('rest'))
|
||||
rt = float(times[-1]) if times else None
|
||||
rows.append(
|
||||
{
|
||||
'ip': m.group('ip'),
|
||||
'time': m.group('time'),
|
||||
'req': m.group('req'),
|
||||
'status': int(m.group('status')),
|
||||
'bytes': int(m.group('bytes')),
|
||||
'rt': rt,
|
||||
}
|
||||
)
|
||||
|
||||
print(f'匹配请求数: {len(rows)}')
|
||||
if not rows:
|
||||
sys.exit(0)
|
||||
|
||||
withrt = [r for r in rows if r['rt'] is not None]
|
||||
if withrt:
|
||||
ts = [r['rt'] for r in withrt]
|
||||
slow = [r for r in withrt if r['rt'] > 1]
|
||||
print(f'耗时: 平均 {sum(ts)/len(ts):.3f}s 中位 {sorted(ts)[len(ts)//2]:.3f}s 最大 {max(ts):.2f}s >1s 的 {len(slow)} 个')
|
||||
print('\n== 最慢 12 个 ==')
|
||||
for r in sorted(withrt, key=lambda x: -x['rt'])[:12]:
|
||||
print(f" {r['rt']:8.2f}s {r['status']} {r['bytes']:>9} B {r['time'][12:]} {r['req']}")
|
||||
|
||||
print('\n== 响应体最大 8 个 ==')
|
||||
for r in sorted(rows, key=lambda x: -x['bytes'])[:8]:
|
||||
rt = f"{r['rt']:.2f}s" if r['rt'] is not None else ' ? '
|
||||
print(f" {r['bytes']:>9} B {rt} {r['status']} {r['req']}")
|
||||
|
||||
from collections import Counter
|
||||
c = Counter(r['req'].split(' ')[1].split('?')[0] if ' ' in r['req'] else r['req'] for r in rows)
|
||||
print('\n== 路径 TOP 10 ==')
|
||||
for path, n in c.most_common(10):
|
||||
print(f' {n:5} {path}')
|
||||
@@ -0,0 +1,96 @@
|
||||
#!/usr/bin/env bash
|
||||
# op-lock.sh — 「平台高危操作锁」(服务器侧第二把锁)
|
||||
#
|
||||
# 为什么需要它:所有会话**共用一台服务器**,而下面这些操作跨会话没有任何互斥 ——
|
||||
# 重启 dshs / drain 实例 scope / 改实例 env·MemoryMax / 批量铺插件 /
|
||||
# 改 nginx vhost·证书·nft
|
||||
# 文档冲突靠 git status / mtime 能看出来,**服务器态变更看不出来**
|
||||
# (systemctl 不会告诉你 10 分钟前谁重启过)→ 只能靠显式锁。
|
||||
#
|
||||
# 与文档库锁的关系:`05-交接单/.doing-<单>` + `.exec-lock` 管「文档与单子」;本锁管「生产态」。
|
||||
# 顺序:**先抢全局执行锁 → 再占本锁**。
|
||||
#
|
||||
# 用法:
|
||||
# bash 07-scripts/op-lock.sh claim <操作名> "<影响面/时长>" [占用者]
|
||||
# bash 07-scripts/op-lock.sh release <操作名>
|
||||
# bash 07-scripts/op-lock.sh status
|
||||
#
|
||||
# 判据:`mkdir` 原子 —— 成功 = 你拿到;报 File exists = 有别的会话正在动线上 → **停手**。
|
||||
# 退出码:0 = 成功/无人占用;1 = 已被占用(claim 失败)或环境错误。
|
||||
set -uo pipefail
|
||||
|
||||
REMOTE="${OP_LOCK_REMOTE:-bt-server}"
|
||||
LOCKROOT="${OP_LOCK_DIR:-/opt/dsh/state/.op-lock}"
|
||||
OP="${2:-}"
|
||||
WHO="${4:-${ME:-unknown-session}}"
|
||||
NOW="$(date '+%Y-%m-%d %H:%M')"
|
||||
|
||||
usage() {
|
||||
sed -n '3,20p' "$0" | sed 's/^# \{0,1\}//'
|
||||
exit 2
|
||||
}
|
||||
|
||||
case "${1:-}" in
|
||||
claim)
|
||||
[ -n "$OP" ] || usage
|
||||
IMPACT="${3:-(未声明影响面 —— 补上:谁会被断、断多久)}"
|
||||
if ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" "mkdir '$LOCKROOT/$OP' 2>/dev/null"; then
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" \
|
||||
"printf '占用者:%s\n开始时间:%s\n操作摘要:%s\n影响面:%s\n' '$WHO' '$NOW' '$OP' '$IMPACT' > '$LOCKROOT/$OP/OWNER'" || true
|
||||
echo "✅ 已占位:$LOCKROOT/$OP(占用者 $WHO,$NOW)"
|
||||
if [ "$WHO" = "unknown-session" ]; then
|
||||
echo " ⚠ 未提供会话名 → 已记作 unknown-session。后果有两个(2026-09-12 实测踩到):"
|
||||
echo " ① handoff-guard 的【1d】会把它当成**别人的锁**(连你自己都被挡住);"
|
||||
echo " ② release 的归属校验会拒绝你。请改用:"
|
||||
echo " ME=\"<会话名>\" bash 07-scripts/op-lock.sh claim <操作名> \"<影响面>\""
|
||||
fi
|
||||
echo " 完工请:bash 07-scripts/op-lock.sh release $OP"
|
||||
exit 0
|
||||
fi
|
||||
echo "❌ 占位失败 —— 该锁已存在(有会话正在动线上)→ 停手。" >&2
|
||||
bash "$0" status >&2 || true
|
||||
exit 1
|
||||
;;
|
||||
release)
|
||||
[ -n "$OP" ] || usage
|
||||
# 归属校验(R9,2026-09-12 补):锁只能由**持有者自己**释放 ——
|
||||
# 否则 release 就成了绕过 R9 的后门(AI 读一下 status 拿到操作名,就能把别人的锁 rm 掉)。
|
||||
RWHO="${4:-${ME:-unknown-session}}"
|
||||
ROWN="$(ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" \
|
||||
"sed -n '1p' '$LOCKROOT/$OP/OWNER' 2>/dev/null" 2>/dev/null | sed 's/^占用者://')"
|
||||
if [ -z "$ROWN" ]; then
|
||||
echo "❌ 拒绝释放:「$OP」没有 OWNER 记录(无法确认归属)。" >&2
|
||||
echo " 按 **R9**:锁的处置权只属于用户本人 —— 请报告用户,由用户处置。" >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ "$ROWN" != "$RWHO" ] && [ "${FORCE:-0}" != "1" ]; then
|
||||
echo "❌ 拒绝释放:「$OP」的占用者是「$ROWN」,而你是「$RWHO」。" >&2
|
||||
echo " 按 **R9**(禁止接管 / 删锁):抢不到锁 = 停手 + 报告用户;**锁的处置权只属于用户本人**。" >&2
|
||||
echo " (确需处置:由**用户本人**执行;或用户明确点头后用 FORCE=1,并在回报里写明理由。)" >&2
|
||||
exit 1
|
||||
fi
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" "rm -rf '$LOCKROOT/$OP'" || { echo "❌ 释放失败(ssh 不可达?)" >&2; exit 1; }
|
||||
echo "✅ 已释放:$OP(归属已核:$ROWN)"
|
||||
;;
|
||||
status)
|
||||
echo "—— 平台操作锁($REMOTE:$LOCKROOT)——"
|
||||
OUT="$(ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" \
|
||||
"ls -1 '$LOCKROOT' 2>/dev/null | grep -v '^README$'" 2>/dev/null)"
|
||||
if [ -z "$OUT" ]; then
|
||||
if ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" "test -d '$LOCKROOT'" 2>/dev/null; then
|
||||
echo " (无人占用 ✓)"
|
||||
else
|
||||
echo " ⚠ 锁根目录不存在:$LOCKROOT(需先建立)" >&2
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
for L in $OUT; do
|
||||
echo " 🔴 $L"
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 "$REMOTE" \
|
||||
"sed 's/^/ /' '$LOCKROOT/$L/OWNER' 2>/dev/null" || true
|
||||
done
|
||||
exit 1
|
||||
fi
|
||||
;;
|
||||
*) usage ;;
|
||||
esac
|
||||
@@ -0,0 +1,133 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* 插件兼容性预检 v2 —— PoC(只读)
|
||||
*
|
||||
* A. 平台内置 @deepseek-ai/* 的「真实导出符号」清单(判据来源)
|
||||
* B. 候选池每个 tgz:解包 → 依赖声明 + bundle 内 import 语句
|
||||
* C. 逐条比对 → 判定「兼容 / 不兼容 / 需装后复核」
|
||||
*
|
||||
* 用法:node plugin-compat-check.mjs
|
||||
*/
|
||||
import { readFileSync, existsSync, readdirSync, mkdtempSync, rmSync, statSync } from 'node:fs'
|
||||
import { join } from 'node:path'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { execFileSync } from 'node:child_process'
|
||||
|
||||
const DSH_ROOT = process.env.DSH_ROOT || '/usr/local/lib/node_modules/@deepseek-ai/dsh'
|
||||
const NESTED = join(DSH_ROOT, 'node_modules', '@deepseek-ai')
|
||||
const POOL = process.env.POOL || '/var/lib/dshs/business-plugins'
|
||||
|
||||
/* ---------- A. 平台 exports 真值 ---------- */
|
||||
const platform = new Map() // pkgName -> { version, keys:Set }
|
||||
for (const name of readdirSync(NESTED)) {
|
||||
if (name.startsWith('.')) continue
|
||||
const pj = join(NESTED, name, 'package.json')
|
||||
if (!existsSync(pj)) continue
|
||||
const meta = JSON.parse(readFileSync(pj, 'utf8'))
|
||||
let entry = null
|
||||
for (const cand of ['lib/index.js', 'dist/index.js', 'index.js']) {
|
||||
const f = join(NESTED, name, cand)
|
||||
if (existsSync(f)) { entry = f; break }
|
||||
}
|
||||
let keys = null
|
||||
if (entry) { try { keys = new Set(Object.keys(await import(entry))) } catch { /* ignore */ } }
|
||||
platform.set(meta.name, { version: meta.version, keys })
|
||||
}
|
||||
console.log(`[A] 平台内置 @deepseek-ai 包 ${platform.size} 个(其中有导出符号清单的 ${[...platform.values()].filter((v) => v.keys).length} 个)`)
|
||||
|
||||
const llm = platform.get('@deepseek-ai/dsh-llm')
|
||||
console.log(` · dsh-llm@${llm?.version} 导出 ${llm?.keys?.size ?? 0} 个符号;含 assertNever? ${llm?.keys?.has('assertNever') ? '✅' : '❌(= anysearch 崩溃判据成立)'}`)
|
||||
|
||||
/* ---------- B+C. 逐个 tgz 预检 ---------- */
|
||||
const IMPORT_RE = /(?:import|export)\s*(?:\*\s*as\s*[\w$]+|\{([^}]*)\})\s*from\s*["'](@deepseek-ai\/[^"'/]+(?:\/[^"']*)?)["']/g
|
||||
|
||||
function scanFileImports(file, base) {
|
||||
const text = readFileSync(file, 'utf8')
|
||||
const hits = []
|
||||
let m
|
||||
IMPORT_RE.lastIndex = 0
|
||||
while ((m = IMPORT_RE.exec(text))) {
|
||||
const spec = m[1] || ''
|
||||
// 具名导入:取每个符号(去掉 as 别名)
|
||||
const named = spec.split(',').map((s) => s.trim().split(/\s+as\s+/)[0].trim()).filter((s) => s && !s.startsWith('type '))
|
||||
hits.push({ pkg: m[2], named, file: file.slice(base.length + 1) })
|
||||
}
|
||||
return hits
|
||||
}
|
||||
|
||||
function walkJs(dir, base, out = []) {
|
||||
for (const e of readdirSync(dir, { withFileTypes: true })) {
|
||||
const p = join(dir, e.name)
|
||||
if (e.isDirectory()) { if (e.name !== 'node_modules') walkJs(p, base, out) }
|
||||
else if (/\.(js|mjs|cjs)$/.test(e.name)) out.push(p)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
const tgzs = existsSync(POOL) ? readdirSync(POOL).filter((f) => f.endsWith('.tgz')) : []
|
||||
console.log(`\n[B] 候选池 tgz ${tgzs.length} 个:${tgzs.join(', ') || '(无)'}\n`)
|
||||
|
||||
for (const tgz of tgzs) {
|
||||
const full = join(POOL, tgz)
|
||||
const tmp = mkdtempSync(join(tmpdir(), 'compat-'))
|
||||
console.log('━'.repeat(72))
|
||||
console.log(`📦 ${tgz} (${(statSync(full).size / 1024).toFixed(0)} KB)`)
|
||||
try {
|
||||
execFileSync('tar', ['-xzf', full, '-C', tmp], { stdio: 'pipe' })
|
||||
} catch (e) {
|
||||
console.log(` ✗ 解包失败: ${String(e.message).slice(0, 100)}`)
|
||||
rmSync(tmp, { recursive: true, force: true })
|
||||
continue
|
||||
}
|
||||
|
||||
// package.json
|
||||
const pkgRoot = join(tmp, 'package')
|
||||
const pj = join(pkgRoot, 'package.json')
|
||||
let name = tgz, version = '?'
|
||||
const deps = {}
|
||||
if (existsSync(pj)) {
|
||||
const meta = JSON.parse(readFileSync(pj, 'utf8'))
|
||||
name = meta.name || name
|
||||
version = meta.version || '?'
|
||||
Object.assign(deps, meta.dependencies || {}, meta.peerDependencies || {})
|
||||
}
|
||||
console.log(` 包名 ${name}@${version}`)
|
||||
|
||||
// 依赖里的 @deepseek-ai
|
||||
const dsDeps = Object.entries(deps).filter(([k]) => k.startsWith('@deepseek-ai/'))
|
||||
if (dsDeps.length) {
|
||||
console.log(` 声明的 @deepseek-ai 依赖 ${dsDeps.length} 个:`)
|
||||
for (const [d, rng] of dsDeps) {
|
||||
const p = platform.get(d)
|
||||
const verdict = !p ? '⚠️ 平台无此包' : (p.version === rng.replace(/^[\^~=]/, '') ? '✅ 版本一致' : `⚠️ 平台为 ${p.version},插件要求 ${rng}`)
|
||||
console.log(` ${d} ${rng} → ${verdict}`)
|
||||
}
|
||||
} else {
|
||||
console.log(' 未声明任何 @deepseek-ai 依赖')
|
||||
}
|
||||
|
||||
// bundle 内 import
|
||||
const files = walkJs(pkgRoot, pkgRoot)
|
||||
const hits = files.flatMap((f) => scanFileImports(f, pkgRoot))
|
||||
console.log(` 扫描 ${files.length} 个 js 文件,命中 ${hits.length} 条 @deepseek-ai 导入`)
|
||||
|
||||
const problems = []
|
||||
for (const h of hits) {
|
||||
const p = platform.get(h.pkg)
|
||||
if (!p || !p.keys) continue
|
||||
const missing = h.named.filter((n) => n && !p.keys.has(n))
|
||||
if (missing.length) problems.push({ ...h, missing })
|
||||
}
|
||||
if (problems.length) {
|
||||
console.log(' 🔴 不兼容命中:')
|
||||
for (const pr of problems) console.log(` ${pr.pkg} 缺少导出 ${pr.missing.join(', ')} ← ${pr.file}`)
|
||||
} else if (hits.length) {
|
||||
console.log(' 🟢 静态可判部分:全部符号在平台包中存在')
|
||||
} else {
|
||||
console.log(' ⚪ 静态无法判定(插件未直接 import 平台包 → 需「装后扫 node_modules」复核)')
|
||||
}
|
||||
rmSync(tmp, { recursive: true, force: true })
|
||||
}
|
||||
console.log('\n' + '━'.repeat(72))
|
||||
console.log('说明:静态判定只覆盖「插件自身 bundle 里的 import」。若插件把平台 API 的调用藏在依赖包内')
|
||||
console.log(' (如 anysearch 的 @deepseek-ai/dsh-tool-web),必须做「装后扫 profile node_modules」才能抓到。')
|
||||
@@ -0,0 +1,137 @@
|
||||
#!/usr/bin/env bash
|
||||
# preflight-lock.sh — 【开工门禁】「先判定可锁定范围,确认锁定后再处理」
|
||||
#
|
||||
# 用户口径(2026-09-22):
|
||||
# 「会话执行任务前先判断,当前任务涉及范围是否都可锁定,确认锁定后开始处理」
|
||||
#
|
||||
# 与 handoff-guard.sh 的关系:**门禁(本脚本,先判定)→ 抢锁(handoff-guard.sh,后动手)**。
|
||||
# ⛔ 本脚本**不建任何锁**,只做判定与提示;它**不替换** handoff-guard.sh。
|
||||
#
|
||||
# 在 CODEBUDDY.md §6(并发纪律)里是**开工第 0 步**:判定全部可锁 ⇒ 才去 --claim-exec。
|
||||
#
|
||||
# 用法:
|
||||
# bash preflight-lock.sh "<会话名>" <目标文件1> [目标文件2 ...]
|
||||
#
|
||||
# 退出码:0 = 全部可锁且已锁(可以开工)|1 = 存在不可锁项(停手)|2 = 用法错误
|
||||
#
|
||||
set -uo pipefail
|
||||
|
||||
ME="${1:-}"; shift || true
|
||||
if [ -z "$ME" ] || [ $# -eq 0 ]; then
|
||||
echo "用法:bash preflight-lock.sh \"<会话名>\" <目标文件...>" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
DOCS_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
CODE_ROOT="${DSH_CODE_REPO:-D:/github/dsh_shenxian}"
|
||||
WS_ROOT="${DSH_WS_ROOT:-E:/ProgramData/AIProject/aliyun-dsh-server}"
|
||||
LOCKROOT="$DOCS_ROOT/05-交接单/.locks"
|
||||
|
||||
# ── 机制层(⛔ 不受 architecture.md 的 src/** 分层管辖,但**全平台共用**)────────
|
||||
# 判据:这些文件任一改动都会改变**所有会话**的行为 ⇒ 与层④同性质(全局影响)。
|
||||
# 2026-09-22 实测踩到:首版门禁把本任务自己的 4 个目标文件全判进【D】未归类 ⇒ 误报。
|
||||
declare -a MECHANISM_RE=(
|
||||
'^07-scripts/handoff-guard\.sh$' # 全局锁本体
|
||||
'^07-scripts/lock-guard-hook\.py$' # 无锁拒写钩子
|
||||
'^07-scripts/op-lock\.sh$' # 服务器侧操作锁
|
||||
'^07-scripts/(skill-load-guard|stop-dialog-guard)\.py$'
|
||||
'^07-scripts/docs-sync-check\.sh$' # 推送前对账(所有会话共用)
|
||||
'^07-scripts/check-layering\.mjs$' # 分层 ratchet
|
||||
'^07-scripts/layering-baseline\.json$'
|
||||
'^state\.py$' # 所有会话的开机第 0 步
|
||||
'^CODEBUDDY\.md$' # 每会话自动加载的规则
|
||||
)
|
||||
|
||||
# ── 不可锁资源清单(命中 ⇒ 必须走"秒级独占"或串行排队)──────────────
|
||||
# 判据来源:architecture.md R1–R4 层号 + docs/tsconfig.json outDir
|
||||
declare -a NONLOCKABLE_RE=(
|
||||
'^src/(config|crypto|isolation|index)\.ts$' # 层④ 零业务依赖 · 全层依赖
|
||||
'^src/db/schema\.ts$' # 迁移号 = 全局单调序列
|
||||
'^src/web/server\.ts$' # 唯一挂载注册点
|
||||
'^package\.json$' # 依赖账本单一
|
||||
'^07-scripts/layering-baseline\.json$' # 分层 ratchet 单一
|
||||
)
|
||||
|
||||
declare -a SHARED_RE=(
|
||||
'^lib/' # 构建产物(outDir=lib)· 全局独占
|
||||
'^node_modules/'
|
||||
)
|
||||
|
||||
warn_shared=(); need_skeleton=(); ok_files=(); unknown=(); mechanism=()
|
||||
|
||||
for f in "$@"; do
|
||||
rel="${f//\\//}"
|
||||
# 剥掉已知根前缀(顺序要紧:先长后短,否则残留中间段 ⇒ 正则永不匹配)
|
||||
rel="${rel#*aliyun-dsh-server/}"
|
||||
rel="${rel#*dsh-server-docs/}"
|
||||
rel="${rel#*dsh_shenxian/}"
|
||||
rel="${rel#./}"
|
||||
matched=0
|
||||
for re in "${MECHANISM_RE[@]}"; do
|
||||
if printf '%s' "$rel" | grep -qE "$re"; then mechanism+=("$rel"); matched=1; break; fi
|
||||
done
|
||||
[ "$matched" = 1 ] && continue
|
||||
for re in "${NONLOCKABLE_RE[@]}"; do
|
||||
if printf '%s' "$rel" | grep -qE "$re"; then need_skeleton+=("$rel"); matched=1; break; fi
|
||||
done
|
||||
[ "$matched" = 1 ] && continue
|
||||
for re in "${SHARED_RE[@]}"; do
|
||||
if printf '%s' "$rel" | grep -qE "$re"; then warn_shared+=("$rel"); matched=1; break; fi
|
||||
done
|
||||
[ "$matched" = 1 ] && continue
|
||||
if printf '%s' "$rel" | grep -qE '^(src|poc|test|web|docs|scripts|交接单|skills)/'; then
|
||||
ok_files+=("$rel")
|
||||
else
|
||||
unknown+=("$rel")
|
||||
fi
|
||||
done
|
||||
|
||||
echo "══════════ 开工门禁 · 可锁定范围判定 ══════════"
|
||||
echo "会话:$ME"
|
||||
echo
|
||||
|
||||
echo "【A】可独立锁定(域锁,全程持有):${#ok_files[@]} 个"
|
||||
for f in "${ok_files[@]:0:8}"; do echo " ✓ $f"; done
|
||||
[ ${#ok_files[@]} -gt 8 ] && echo " …还有 $(( ${#ok_files[@]} - 8 )) 个"
|
||||
|
||||
echo
|
||||
echo "【B】不可全程锁定(层④/迁移号/唯一注册点)→ 改走**秒级独占**:${#need_skeleton[@]} 个"
|
||||
for f in "${need_skeleton[@]}"; do echo " ⚠ $f"; done
|
||||
[ ${#need_skeleton[@]} -eq 0 ] && echo " (无)"
|
||||
|
||||
echo
|
||||
echo "【E】机制层(⛔ 不受 src 分层管辖,但**全平台共用** ⇒ 与层④同性质):${#mechanism[@]} 个"
|
||||
for f in "${mechanism[@]}"; do echo " ◆ $f"; done
|
||||
[ ${#mechanism[@]} -eq 0 ] && echo " (无)"
|
||||
[ ${#mechanism[@]} -gt 0 ] && echo " ⚠ 改这些 = 改所有会话的行为 ⇒ 必须: ①独占②兼容旧行为③改后跑自证"
|
||||
|
||||
echo
|
||||
echo "【C】共享资源(构建产物 / 依赖树)→ 纳入发布锁:${#warn_shared[@]} 个"
|
||||
for f in "${warn_shared[@]:0:4}"; do echo " ⚠ $f"; done
|
||||
[ ${#warn_shared[@]} -eq 0 ] && echo " (无)"
|
||||
|
||||
if [ ${#unknown[@]} -gt 0 ]; then
|
||||
echo
|
||||
echo "【D】未归类路径(⛔ 必须先想清"属哪个域",R2):${#unknown[@]} 个"
|
||||
for f in "${unknown[@]:0:6}"; do echo " ? $f"; done
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "════════════════════════════════════════════"
|
||||
rc=0
|
||||
[ ${#unknown[@]} -gt 0 ] && rc=1
|
||||
[ ${#mechanism[@]} -gt 0 ] && rc=1
|
||||
if [ "$rc" = 1 ]; then
|
||||
echo "结论:**暂不可并行开工** ——"
|
||||
[ ${#unknown[@]} -gt 0 ] && echo " · 【D】${#unknown[@]} 个未归类路径:先定域(R2)"
|
||||
if [ ${#mechanism[@]} -gt 0 ]; then
|
||||
echo " · 【E】${#mechanism[@]} 个机制层文件:**全平台共用** ⇒ ⛔ 不得与其他会话并行"
|
||||
echo " ⇒ 仅当**确认无其他会话在跑**时,才允许独占开工;否则停手等窗口"
|
||||
fi
|
||||
exit 1
|
||||
fi
|
||||
echo "结论:**可以锁定** ——"
|
||||
echo " · ${#ok_files[@]} 个走域锁|${#need_skeleton[@]} 个走秒级独占|${#warn_shared[@]} 个并发布锁"
|
||||
echo " · 下一步:bash handoff-guard.sh --claim-exec \"$ME\" --domains <由 A 组反推的域>"
|
||||
echo " ⛔ 本脚本是**提示性门禁**:hook 只拦 Write/Edit,Bash 写入不受限(有意安全阀)"
|
||||
exit 0
|
||||
@@ -0,0 +1,111 @@
|
||||
/**
|
||||
* 会话取证 · 画像:元信息/权限档位/用户消息/12 类错误模式扫描/工具清单(不输出正文以外内容)
|
||||
* guest 会话问题分析(只读)
|
||||
* 用法:node analyze-guest.mjs <session.jsonl.zstd> [--full]
|
||||
*/
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { zstdDecompressSync } from 'node:zlib'
|
||||
|
||||
const file = process.argv[2]
|
||||
const raw = readFileSync(file)
|
||||
const MAGIC = Buffer.from([0x28, 0xb5, 0x2f, 0xfd])
|
||||
const offs = []
|
||||
for (let i = 0; i + 4 <= raw.length; i++) if (raw.compare(MAGIC, 0, 4, i, i + 4) === 0) offs.push(i)
|
||||
let text = ''
|
||||
const frames = offs.length ? offs : [0]
|
||||
for (let f = 0; f < frames.length; f++) {
|
||||
const s = frames[f], e = f + 1 < frames.length ? frames[f + 1] : raw.length
|
||||
try { text += zstdDecompressSync(raw.subarray(s, e)).toString('utf8') } catch {}
|
||||
}
|
||||
const evs = []
|
||||
for (const l of text.split('\n')) {
|
||||
if (!l.trim()) continue
|
||||
try { evs.push(JSON.parse(l)) } catch {}
|
||||
}
|
||||
const ts = (ms) => new Date(Number(ms)).toLocaleString('zh-CN', { hour12: false, timeZone: 'Asia/Shanghai' })
|
||||
|
||||
// ── 1) 元信息 ─────────────────────────────────────────
|
||||
const meta = evs.find((e) => e.type === 'session')
|
||||
console.log('=== 元信息 ===')
|
||||
console.log(' id:', meta?.id, '| cwd:', meta?.cwd)
|
||||
console.log(' 创建:', ts(meta?.createdAt), '| 事件数:', evs.length, '| 解压:', text.length, '字符')
|
||||
const presets = evs.filter((e) => e.type === 'permission/preset').map((e) => e.data?.preset)
|
||||
const modes = evs.filter((e) => e.type === 'sandbox/mode').map((e) => e.data?.mode)
|
||||
const pols = evs.filter((e) => e.type === 'approval/policy').map((e) => e.data?.policy)
|
||||
console.log(' permission preset:', presets.join(','), '| sandbox mode:', modes.join(','), '| approval:', pols.join(','))
|
||||
const turns = evs.filter((e) => e.type === 'turn/start').map((e) => Number(e.data?.turn))
|
||||
console.log(' turn 数:', turns.length, '| 最后事件:', ts(evs[evs.length - 1]?.time ?? evs[evs.length - 1]?.time0))
|
||||
|
||||
// ── 2) 类型直方图(正确格式)──────────────────────────
|
||||
const hist = new Map()
|
||||
for (const e of evs) hist.set(e.type, (hist.get(e.type) ?? 0) + 1)
|
||||
console.log('\n=== 事件类型(Top 25)===')
|
||||
for (const [t, n] of [...hist].sort((a, b) => b[1] - a[1]).slice(0, 25)) console.log(' ' + String(n).padStart(5) + ' ' + t)
|
||||
|
||||
// ── 3) 用户消息 ───────────────────────────────────────
|
||||
const textOf = (d) => {
|
||||
if (typeof d === 'string') return d
|
||||
if (Array.isArray(d)) return d.map(textOf).join('')
|
||||
if (d && typeof d === 'object') {
|
||||
if (typeof d.text === 'string') return d.text
|
||||
if (d.content) return textOf(d.content)
|
||||
return ''
|
||||
}
|
||||
return ''
|
||||
}
|
||||
console.log('\n=== 用户消息(共 ' + evs.filter((e) => e.type === 'user/message').length + ' 条)===')
|
||||
for (const e of evs.filter((e) => e.type === 'user/message')) {
|
||||
const t = textOf(e.data?.content).replace(/\s+/g, ' ').slice(0, 260)
|
||||
console.log(` [${ts(e.time)}] ${t}`)
|
||||
}
|
||||
|
||||
// ── 4) 错误/拒绝模式扫描 ──────────────────────────────
|
||||
const PATTERNS = [
|
||||
['沙箱后端不可用', /no sandbox backend is usable/],
|
||||
['拒绝无沙箱运行', /refusing to run the command unconfined/],
|
||||
['权限被拒', /(?:\bdenied\b|Permission denied|EACCES|not permitted)/],
|
||||
['文件不存在', /\bENOENT\b|No such file or directory|command not found/i],
|
||||
['只读文件系统', /Read-only file system|EROFS/],
|
||||
['超时', /\btimed? ?out\b|ETIMEDOUT|timeout/i],
|
||||
['解析失败/DNS', /Could not resolve host|EAI_AGAIN|getaddrinfo/],
|
||||
['401/鉴权', /401|unauthorized|authentication required/i],
|
||||
['崩溃/熔断', /crash-restart|circuit-open/],
|
||||
['Python 缺失', /python3?: (?:command )?not found|python3 不存在/],
|
||||
['工具失败关键词', /tool (?:call )?failed|Tool failed|工具执行失败/],
|
||||
['中文报错词', /报错|失败|不可用|无法执行|被拒绝|不允许/],
|
||||
]
|
||||
console.log('\n=== 错误/限制模式扫描 ===')
|
||||
for (const [name, re] of PATTERNS) {
|
||||
const hits = []
|
||||
for (const e of evs) {
|
||||
const s = JSON.stringify(e)
|
||||
if (re.test(s)) hits.push(e)
|
||||
}
|
||||
if (!hits.length) { console.log(` ${name}: 0`); continue }
|
||||
const kinds = new Map()
|
||||
for (const h of hits) kinds.set(h.type, (kinds.get(h.type) ?? 0) + 1)
|
||||
console.log(` ${name}: ${hits.length} 次 → ${[...kinds].map(([k, v]) => k + '×' + v).join(', ')}`)
|
||||
for (const h of hits.slice(0, 2)) {
|
||||
const m = JSON.stringify(h).match(re)
|
||||
const i = JSON.stringify(h).indexOf(m[0])
|
||||
console.log(` · [${ts(h.time ?? h.time0)}] …` + JSON.stringify(h).slice(Math.max(0, i - 130), i + 130).replace(/\\n/g, ' '))
|
||||
}
|
||||
}
|
||||
|
||||
// ── 5) 工具调用清单(tool 相关事件的 name/command)────
|
||||
console.log('\n=== 工具调用(含 tool 的事件,取 name/command/tool)===')
|
||||
const toolEvs = evs.filter((e) => /tool/i.test(e.type))
|
||||
const names = new Map()
|
||||
const fails = []
|
||||
for (const e of toolEvs) {
|
||||
const s = JSON.stringify(e)
|
||||
const nm = (s.match(/"name":"([^"]{2,40})"/) ?? [])[1] ?? (s.match(/"tool":"([^"]{2,40})"/) ?? [])[1] ?? '(?)'
|
||||
names.set(nm, (names.get(nm) ?? 0) + 1)
|
||||
if (/"(?:isError|ok|error|status)":(?:true|false|"[^"]*")/.test(s)) {
|
||||
if (/"isError":true|"ok":false|"status":"(?:error|failed)"/.test(s)) fails.push(e)
|
||||
}
|
||||
}
|
||||
console.log(' 工具事件类型数:', toolEvs.length)
|
||||
for (const [n, c] of [...names].sort((a, b) => b[1] - a[1]).slice(0, 20)) console.log(' ' + String(c).padStart(4) + ' ' + n)
|
||||
console.log(' 标记失败的工具事件:', fails.length)
|
||||
for (const f of fails.slice(0, 5)) console.log(' · [' + ts(f.time ?? f.time0) + '] ' + JSON.stringify(f).slice(0, 300))
|
||||
@@ -0,0 +1,59 @@
|
||||
/**
|
||||
* 会话取证 · 逐次工具结果分类(沙箱拒绝 / 文件策略拒绝 / 命令错误 / OK)+ 审批事件时间线
|
||||
* 精确分类:每次 bash 调用的结果性质(沙箱拒绝 / 命令错误 / 成功)
|
||||
* 用法:node classify-bash.mjs <session.jsonl.zstd>
|
||||
*/
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { zstdDecompressSync } from 'node:zlib'
|
||||
|
||||
const raw = readFileSync(process.argv[2])
|
||||
const MAGIC = Buffer.from([0x28, 0xb5, 0x2f, 0xfd])
|
||||
const offs = []
|
||||
for (let i = 0; i + 4 <= raw.length; i++) if (raw.compare(MAGIC, 0, 4, i, i + 4) === 0) offs.push(i)
|
||||
let text = ''
|
||||
for (let f = 0; f < (offs.length ? offs : [0]).length; f++) {
|
||||
const s = offs[f], e = f + 1 < offs.length ? offs[f + 1] : raw.length
|
||||
try { text += zstdDecompressSync(raw.subarray(s, e)).toString('utf8') } catch {}
|
||||
}
|
||||
const evs = []
|
||||
for (const l of text.split('\n')) { if (!l.trim()) continue; try { evs.push(JSON.parse(l)) } catch {} }
|
||||
const t = (ms) => new Date(Number(ms)).toLocaleString('zh-CN', { hour12: false, timeZone: 'Asia/Shanghai' })
|
||||
|
||||
// tool/call:id → {name, args(截断)}
|
||||
const callInfo = new Map()
|
||||
for (const e of evs) {
|
||||
if (e.type !== 'tool/call') continue
|
||||
const s = JSON.stringify(e.data ?? {})
|
||||
const id = (s.match(/"callId":"([^"]+)"/) ?? [])[1]
|
||||
const name = (s.match(/"name":"([^"]+)"/) ?? [])[1]
|
||||
if (!id) continue
|
||||
const cmd = (s.match(/"command":"((?:[^"\\]|\\.)*)"/) ?? [])[1] ?? ''
|
||||
callInfo.set(id, { name, cmd: cmd.replace(/\\n/g, ' ').slice(0, 90) })
|
||||
}
|
||||
// tool/result:id → 文本
|
||||
const rows = []
|
||||
for (const e of evs) {
|
||||
if (e.type !== 'tool/result') continue
|
||||
const s = JSON.stringify(e.data ?? {})
|
||||
const id = (s.match(/"toolCallId":"([^"]+)"/) ?? [])[1]
|
||||
const info = callInfo.get(id) ?? { name: '?', cmd: '' }
|
||||
const txt = (e.data?.message?.content ?? []).map((c) => (c.content ?? []).map((x) => x.text ?? '').join('')).join('\n')
|
||||
const flat = txt.replace(/\s+/g, ' ').trim()
|
||||
let kind = 'OK'
|
||||
if (/sandbox mode .* is requested but no sandbox backend/.test(flat)) kind = '沙箱拒绝'
|
||||
else if (/file access denied under/.test(flat)) kind = '文件策略拒绝'
|
||||
else if (/^Error:|^error:|"error"|Command failed|exit code [1-9]/.test(flat)) kind = '命令错误'
|
||||
rows.push({ time: e.time, name: info.name, cmd: info.cmd, kind, head: flat.slice(0, 150) })
|
||||
}
|
||||
console.log('=== 逐次 tool/result 分类(共 %d)===', rows.length)
|
||||
for (const r of rows) {
|
||||
if (r.name !== 'bash' && r.kind === 'OK') continue
|
||||
console.log(' %s %-6s %-10s %s', t(r.time), r.name, r.kind, (r.cmd || r.head).slice(0, 110))
|
||||
}
|
||||
const byKind = {}
|
||||
for (const r of rows) byKind[r.kind] = (byKind[r.kind] ?? 0) + 1
|
||||
console.log('\n汇总:', JSON.stringify(byKind))
|
||||
const sb = rows.filter((r) => r.kind === '沙箱拒绝')
|
||||
console.log('\n沙箱拒绝时间线: %s → %s(共 %d 次)', t(sb[0]?.time), t(sb[sb.length - 1]?.time), sb.length)
|
||||
const after = sb.filter((r) => r.time > 1789116 * 1000).length
|
||||
console.log(' 其中 22:16 切档位之后:', after, '次')
|
||||
@@ -0,0 +1,29 @@
|
||||
/** 批量列出会话档位:id / 创建时间 / preset / sandbox mode / approval / turn 数 */
|
||||
* 会话取证 · 批量对比各会话 preset/sandbox/approval(定位"老会话档位未跟随平台默认"的关键工具)
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { zstdDecompressSync } from 'node:zlib'
|
||||
|
||||
const MAGIC = Buffer.from([0x28, 0xb5, 0x2f, 0xfd])
|
||||
const t = (ms) => new Date(Number(ms)).toLocaleString('zh-CN', { hour12: false, timeZone: 'Asia/Shanghai' })
|
||||
|
||||
for (const file of process.argv.slice(2)) {
|
||||
const raw = readFileSync(file)
|
||||
const offs = []
|
||||
for (let i = 0; i + 4 <= raw.length; i++) if (raw.compare(MAGIC, 0, 4, i, i + 4) === 0) offs.push(i)
|
||||
let text = ''
|
||||
for (let f = 0; f < (offs.length ? offs : [0]).length; f++) {
|
||||
const s = offs[f], e = f + 1 < offs.length ? offs[f + 1] : raw.length
|
||||
try { text += zstdDecompressSync(raw.subarray(s, e)).toString('utf8') } catch {}
|
||||
}
|
||||
const evs = []
|
||||
for (const l of text.split('\n')) { if (!l.trim()) continue; try { evs.push(JSON.parse(l)) } catch {} }
|
||||
const meta = evs.find((e) => e.type === 'session')
|
||||
const uniq = (ty, f) => [...new Set(evs.filter((e) => e.type === ty).map(f))].join(',')
|
||||
const tl = text.match(/"text":"([^"]{0,60})"/g)?.slice(0, 1)?.join('') ?? ''
|
||||
console.log(
|
||||
'%s | 创建 %s | preset=%s | sandbox=%s | approval=%s | turns=%d | %d 事件',
|
||||
(meta?.id ?? '?').slice(-12),
|
||||
t(meta?.createdAt), uniq('permission/preset', (e) => e.data?.preset), uniq('sandbox/mode', (e) => e.data?.mode),
|
||||
uniq('approval/policy', (e) => e.data?.policy), evs.filter((e) => e.type === 'turn/start').length, evs.length,
|
||||
)
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
/**
|
||||
* 会话取证 · schema 探测:解压多帧 zstd → 事件类型直方图 + 每类样本键结构(先摸清格式)
|
||||
* 会话 schema 探测(只读):输出事件类型直方图 + 少量样本键结构
|
||||
* 用法:node probe-schema.mjs <session.jsonl.zstd>
|
||||
*/
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { zstdDecompressSync } from 'node:zlib'
|
||||
|
||||
const raw = readFileSync(process.argv[2])
|
||||
const MAGIC = Buffer.from([0x28, 0xb5, 0x2f, 0xfd])
|
||||
const offs = []
|
||||
for (let i = 0; i + 4 <= raw.length; i++) if (raw.compare(MAGIC, 0, 4, i, i + 4) === 0) offs.push(i)
|
||||
let text = ''
|
||||
const frames = offs.length ? offs : [0]
|
||||
for (let f = 0; f < frames.length; f++) {
|
||||
const s = frames[f], e = f + 1 < frames.length ? frames[f + 1] : raw.length
|
||||
try { text += zstdDecompressSync(raw.subarray(s, e)).toString('utf8') } catch {}
|
||||
}
|
||||
const lines = text.split('\n').filter((l) => l.trim())
|
||||
console.log('解压后 %d 行 / %d 字符 / zstd 帧 %d', lines.length, text.length, frames.length)
|
||||
|
||||
const types = new Map()
|
||||
const samples = new Map()
|
||||
for (const l of lines) {
|
||||
let o
|
||||
try { o = JSON.parse(l) } catch { continue }
|
||||
const t = o.type ?? o.event ?? o.kind ?? '(no-type)'
|
||||
types.set(t, (types.get(t) ?? 0) + 1)
|
||||
if (!samples.has(t)) samples.set(t, o)
|
||||
}
|
||||
console.log('\n=== 事件类型直方图 ===')
|
||||
for (const [t, n] of [...types].sort((a, b) => b[1] - a[1])) console.log(' %-28s %d', t, n)
|
||||
|
||||
console.log('\n=== 每类样本(键 + 截断值) ===')
|
||||
for (const [t, o] of samples) {
|
||||
console.log('--- %s ---', t)
|
||||
const dump = (v, depth = 0) => {
|
||||
if (v === null) return 'null'
|
||||
if (Array.isArray(v)) return `[${v.length} items${v.length ? ': ' + dump(v[0], depth + 1) : ''}]`
|
||||
if (typeof v === 'object') {
|
||||
if (depth > 1) return '{…}'
|
||||
return '{' + Object.entries(v).slice(0, 10).map(([k, x]) => `${k}: ${dump(x, depth + 1)}`).join(', ') + '}'
|
||||
}
|
||||
const s = String(v)
|
||||
return JSON.stringify(s.length > 90 ? s.slice(0, 90) + '…' : s)
|
||||
}
|
||||
console.log(' ' + dump(o).slice(0, 1200))
|
||||
}
|
||||
@@ -0,0 +1,263 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""skill-load-guard.py —— 「用户点名了方法 ⇒ 强制加载技能」的 UserPromptSubmit 钩子
|
||||
|
||||
为什么需要它
|
||||
────────────
|
||||
2026-09-16 复盘会话 `ddea70b7`(「确认guest用户数据迁移到106服务器」):
|
||||
用户 U6 明确说「D1–D6 先做哪些 **按照你的规划执行**,中间有问题**参考决策方法**」,
|
||||
但该会话 265 次工具调用里 **`Skill` 计数 = 0** —— 那份方法论**从未进入上下文**。
|
||||
AI 仍按常驻判据把两个**非门禁**问题("106 旧控制面要不要停" / "D1–D6 先做哪些")上抛给用户,
|
||||
而用户随后 45 秒 / 58 秒内自己给出了答案(Q1 的答案还超出了 AI 给的选项空间)。
|
||||
|
||||
⇒ 与 `stop-dialog-guard.py` 同族问题:**规则写了,但没有任何机制保证它在正确的时机被取用**。
|
||||
(那个钩子处理"回复收尾时的征询句",本钩子处理"用户点名时的方法加载",两者互补。)
|
||||
|
||||
本钩子 = 覆盖"用户点名"这条面:**每次用户提交时**扫输入,命中点名关键词 ⇒
|
||||
通过 `additionalContext` 注入一条**紧邻用户消息**的强制指令
|
||||
(位置比静态规则文件显著得多 —— 长会话里 CODEBUDDY.md 的硬要求容易被"稀释")。
|
||||
|
||||
安全设计(都不许省)
|
||||
────────────────────
|
||||
1. **自作用域**:只在 cwd 落在本工作区(`aliyun-dsh-server`)时生效,其他项目一律放行。
|
||||
2. **绝不添乱**:任何异常 → 静默放行(exit 0);本钩子**从不拦用户**,只在命中时**追加一段提醒**。
|
||||
3. **防跑飞**:同一会话 300 秒内最多注入一次。
|
||||
4. **急停双闸**(无需卸载/重启):env `DSH_SKILL_GUARD_OFF=1`,或 `<工作区>/.workbuddy/skill-guard.disabled`。
|
||||
5. **低频自证日志**:命中才写一行到 `<工作区>/.workbuddy/skill-load-guard.log` —— 用来回答"到底有没有触发"。
|
||||
6. **显式 UTF-8**:读写都走 `buffer`,不依赖 `PYTHONUTF8`(`-E` 会屏蔽它 ⇒ cp936 ⇒ 含中文静默炸)。
|
||||
|
||||
退出码:始终 0;决策通过 stdout 的 JSON 表达。
|
||||
安装(`settings.json` 的 hooks 段 · 与既有钩子同族,**并列挂在同一个 UserPromptSubmit 的 hooks 数组里**):
|
||||
"UserPromptSubmit": [{ "hooks": [
|
||||
{ "type": "command", "command": "\"<python>\" \"<此脚本>\"", "timeout": 10 },
|
||||
{ "type": "command", "command": "\"<python>\" -S -E \"<stop-dialog-guard.py>\"", "timeout": 10 }
|
||||
]}]
|
||||
⚠️ hooks 是**应用启动时快照** ⇒ 装完必须**完全重启 WorkBuddy**(关窗 ≠ 退出)才加载。
|
||||
"""
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
|
||||
# ── 作用域(2026-09-22 由单值扩为**多值 + env 可覆盖**)─────────────────────────
|
||||
# 起因:用户 2026-09-22 拍板「B」—— 让本钩子覆盖本机全部会话区,而不是只管 aliyun-dsh-server。
|
||||
# 形态刻意与 `dsh-ai1net-desktop/.workbuddy/guard/sync-scoped-guards.py` 生成的副本**同形**
|
||||
# (`_scopes()` + `_in_scope()`),这样两侧可互换、副本的机械变换规则不再需要改写本文件。
|
||||
# ⛔ 不写死绝对路径:只比工作区**目录名**,换机器 / 改盘符都不受影响。
|
||||
# ⚠️ env `DSH_GUARD_SCOPES`(逗号分隔)可覆盖;设为空串 ⇒ 只走默认表。
|
||||
_SCOPES_DEFAULT = (
|
||||
'aliyun-dsh-server', # 平台主工作区(本脚本所在的项目)
|
||||
'dsh-ai1net-desktop', # 桌面壳 / 开源导出线
|
||||
)
|
||||
_log_rel = os.path.join('.workbuddy', 'skill-load-guard.log')
|
||||
|
||||
|
||||
def _workdirs():
|
||||
"""作用域标记元组(env `DSH_GUARD_SCOPES` 优先,逗号分隔)。"""
|
||||
raw = os.environ.get('DSH_GUARD_SCOPES')
|
||||
if raw is None:
|
||||
return _SCOPES_DEFAULT
|
||||
return tuple(s.strip() for s in raw.split(',') if s.strip())
|
||||
|
||||
|
||||
def _in_scope(s):
|
||||
"""目录名匹配即算命中(与副本同形;比绝对路径稳,不受盘符/用户名影响)。"""
|
||||
return any(x in str(s or '') for x in _workdirs())
|
||||
|
||||
|
||||
LOG_REL = _log_rel
|
||||
STATE_REL = os.path.join('.workbuddy', '.skill-load-guard.state')
|
||||
DISABLE_REL = os.path.join('.workbuddy', 'skill-guard.disabled')
|
||||
COOLDOWN = 300 # 同会话注入冷却(秒)
|
||||
|
||||
# 用户「点名方法 / 点名规则」的词表 —— 命中任一即注入。
|
||||
# ── 分组 1:点名「决策方法」(2026-09-16 立,最稳的一条)─────────────────────
|
||||
# ── 分组 2:点名「通用作业规则」(2026-09-22 用户拍板 B 后新增)─────────────────
|
||||
# 目的:`agent-operating-rules`(跨工作区通用规则技能)过去**完全靠 description 匹配**,
|
||||
# 是所有技能里最容易被漏的一个 —— 加三条窄词给它一条机制通道。
|
||||
TRIGGERS = (
|
||||
# 分组 1 —— 决策方法族(→ 加载 `dsh-decision-method`)
|
||||
'决策方法',
|
||||
'自行决策', '自主决策', '自己决策', '自己拿主意',
|
||||
'别问我', '不要问我', '不用问我', '按你的规划', '按你的判断',
|
||||
'参考决策', '决策方法论',
|
||||
# 分组 2 —— 作业规则族(→ 加载 `agent-operating-rules`)
|
||||
'按规则来', '按规则做', '按作业规则', '遵守规则', '按规矩来',
|
||||
)
|
||||
|
||||
# 命中词 → 该加载哪个技能(2026-09-22 加:从"只会推 dsh-decision-method"扩为按命中词分流)
|
||||
_LOAD_RULES = ('按规则来', '按规则做', '按作业规则', '遵守规则', '按规矩来')
|
||||
|
||||
|
||||
def _read_stdin():
|
||||
try:
|
||||
raw = sys.stdin.buffer.read()
|
||||
except Exception:
|
||||
return {}
|
||||
if not raw:
|
||||
return {}
|
||||
for enc in ('utf-8', 'utf-8-sig', 'gbk'):
|
||||
try:
|
||||
return json.loads(raw.decode(enc))
|
||||
except Exception:
|
||||
continue
|
||||
return {}
|
||||
|
||||
|
||||
def _emit(obj):
|
||||
try:
|
||||
sys.stdout.buffer.write(json.dumps(obj, ensure_ascii=False).encode('utf-8'))
|
||||
sys.stdout.buffer.flush()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _workdir(obj):
|
||||
for k in ('cwd', 'project_dir', 'workspace', 'projectDir'):
|
||||
v = obj.get(k)
|
||||
if isinstance(v, str) and v:
|
||||
return v
|
||||
for k in ('CODEBUDDY_PROJECT_DIR', 'WORKSPACE_FOLDER', 'PWD'):
|
||||
v = os.environ.get(k)
|
||||
if v:
|
||||
return v
|
||||
# 兜底:本脚本位于 <工作区>/dsh-server-docs/07-scripts/
|
||||
try:
|
||||
return os.path.abspath(os.path.join(os.path.dirname(os.path.abspath(__file__)), '..', '..'))
|
||||
except Exception:
|
||||
return ''
|
||||
|
||||
|
||||
def _norm_path(p):
|
||||
"""把 MSYS / Git-Bash 风格路径规范成 Windows 风格(`/e/foo` → `E:/foo`)。
|
||||
|
||||
动机(2026-09-22 实测,**正在持续发生**):**Windows 原生 python 会把
|
||||
`/e/ProgramData/x` 解释成「当前盘根 + 相对路径」= `e\\ProgramData\\x`**
|
||||
⇒ 若当前盘是 E,就落到 `E:\\e\\ProgramData\\x`。
|
||||
实证:E 盘根长出影子目录 `E:\\e\\ProgramData\\AI技能\\<ws>\\.workbuddy\\`(355 文件 / 29 MB,仍在增长)。
|
||||
|
||||
⛔ 同源铁律:本工作区已有「**Python exe 不认 `/e/…` ⇒ 传 `E:/…`**」。
|
||||
"""
|
||||
try:
|
||||
s = str(p or '')
|
||||
m = re.match(r'^/([A-Za-z])(/.*)?$', s)
|
||||
if m:
|
||||
return m.group(1).upper() + ':' + (m.group(2) or '/')
|
||||
except Exception:
|
||||
pass
|
||||
return p
|
||||
|
||||
|
||||
def _log(workdir, line):
|
||||
try:
|
||||
p = os.path.join(_norm_path(workdir), LOG_REL)
|
||||
with io.open(p, 'a', encoding='utf-8', newline='\n') as f:
|
||||
f.write('%s %s\n' % (time.strftime('%Y-%m-%d %H:%M:%S'), line))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _cooling(workdir, sid):
|
||||
"""同一会话 COOLDOWN 秒内不重复注入。"""
|
||||
try:
|
||||
p = os.path.join(workdir, STATE_REL)
|
||||
now = time.time()
|
||||
last_sid, last_ts = '', 0.0
|
||||
if os.path.exists(p):
|
||||
try:
|
||||
with io.open(p, 'r', encoding='utf-8') as f:
|
||||
parts = f.read().strip().split('\t')
|
||||
last_sid, last_ts = parts[0], float(parts[1] or 0)
|
||||
except Exception:
|
||||
pass
|
||||
if last_sid == sid and (now - last_ts) < COOLDOWN:
|
||||
return True
|
||||
with io.open(p, 'w', encoding='utf-8', newline='\n') as f:
|
||||
f.write('%s\t%.3f\n' % (sid, now))
|
||||
except Exception:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def main():
|
||||
if os.environ.get('DSH_SKILL_GUARD_OFF') == '1':
|
||||
return
|
||||
obj = _read_stdin()
|
||||
if not obj:
|
||||
return
|
||||
|
||||
workdir = _norm_path(_workdir(obj))
|
||||
if not workdir or not _in_scope(workdir):
|
||||
return # 自作用域:表内工作区之外的会话一律放行
|
||||
if os.path.exists(os.path.join(workdir, DISABLE_REL)):
|
||||
return
|
||||
|
||||
prompt = ''
|
||||
for k in ('prompt', 'user_prompt', 'userPrompt', 'message', 'text'):
|
||||
v = obj.get(k)
|
||||
if isinstance(v, str) and v.strip():
|
||||
prompt = v
|
||||
break
|
||||
|
||||
sid = str(obj.get('session_id') or obj.get('sessionId') or 'unknown')
|
||||
# ★ 入口即留痕(2026-09-16 加)—— 否则"没命中就没日志",永远无法回答
|
||||
# "它到底有没有被宿主调用"。与 stop-dialog-guard.py 同款设计;每次用户提交一行。
|
||||
_log(workdir, 'entry|event=UserPromptSubmit|in_scope=True|sid=%s|prompt_len=%d'
|
||||
% (sid[:8], len(prompt)))
|
||||
if not prompt:
|
||||
return
|
||||
|
||||
hits = [w for w in TRIGGERS if w in prompt]
|
||||
if not hits:
|
||||
return
|
||||
|
||||
if _cooling(workdir, sid):
|
||||
return
|
||||
|
||||
seg = re.sub(r'\s+', ' ', prompt)
|
||||
pos = min(seg.find(w) for w in hits)
|
||||
excerpt = seg[max(0, pos - 40): pos + 80].strip()
|
||||
|
||||
# ── 按命中词分流(2026-09-22):规则族与决策族给不同指令 ──────────────────
|
||||
hit_rules = [w for w in hits if w in _LOAD_RULES]
|
||||
if hit_rules:
|
||||
head = '检测到用户本轮**点名了作业规则**(命中:%s)。' % '、'.join(hit_rules)
|
||||
body = (
|
||||
'⛔ **不要凭记忆代替、也不要以"我已经知道规则"为由跳过**:\n'
|
||||
' **先调用 Skill 工具加载 `agent-operating-rules`**(通用作业规则 · 含"什么时候自决 / 什么时候上抛 / 回复排版 / 说话方式 / 不越界 / 多棒接力"),\n'
|
||||
' 然后才开始作答。若本轮涉及功能需求判类型 / 该不该上抛,**同时加载 `dsh-feature-first`**。\n\n'
|
||||
'为什么强制:该技能是**跨工作区通用**的(不绑定任何具体项目),'
|
||||
'但它与其他技能一样**靠 description 匹配按需加载 —— 没有机制保证它在任何工作区都被取用**'
|
||||
'(它的 description 里已写「当你在任何一个工作区开始任务…」,仍可能漏)。'
|
||||
'本条注入即补上这个缺口。\n'
|
||||
'(若判定本轮确实与作业规则无关,可在作答中一句话说明后继续 —— 但**不要静默跳过加载**。)'
|
||||
)
|
||||
else:
|
||||
head = '检测到用户本轮**点名了方法**(命中:%s)。' % '、'.join(hits)
|
||||
body = (
|
||||
'⛔ **不要凭记忆代替、也不要以"我已经知道判据"为由跳过**:\n'
|
||||
' **先调用 Skill 工具加载 `dsh-decision-method`**(决策方法论 · 含 §4.5 规则冲突裁决顺序 + 上抛前三问),\n'
|
||||
' 然后才开始作答。若本轮涉及功能需求判类型 / 该不该上抛,**同时加载 `dsh-feature-first`**。\n\n'
|
||||
'为什么强制:2026-09-16 实测(会话 ddea70b7)—— 用户点名「参考决策方法」后,'
|
||||
'AI 全程 `Skill` 调用 **0 次**,仍按旧判据把两个**非门禁**问题上抛给用户,'
|
||||
'用户 45 / 58 秒后自己给出了答案。**规则写在文件里 ≠ 会在正确的时机被取用。**\n'
|
||||
'(若判定本轮确实与决策无关,可在作答中一句话说明后继续 —— 但**不要静默跳过加载**。)'
|
||||
)
|
||||
|
||||
ctx = '【技能加载闸门 · 机制层强制】%s\n' '原文片段:「…%s…」\n\n' '%s' % (head, excerpt, body)
|
||||
|
||||
_emit({'hookSpecificOutput': {
|
||||
'hookEventName': 'UserPromptSubmit',
|
||||
'additionalContext': ctx,
|
||||
}})
|
||||
_log(workdir, 'HIT sid=%s hits=%s' % (sid[:8], ','.join(hits)))
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
try:
|
||||
main()
|
||||
except Exception:
|
||||
pass
|
||||
sys.exit(0)
|
||||
@@ -0,0 +1,620 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""stop-dialog-guard.py —— 「禁止用征询句收尾」的 Stop 钩子(WorkBuddy / CodeBuddy)
|
||||
|
||||
为什么需要它
|
||||
────────────
|
||||
2026-09-15 实测:本工作区日志里 `tool=AskUserQuestion` 调用数 = 09-12: 43 / 09-13: 3 / **09-14: 0 / 09-15: 0**
|
||||
⇒ 既有「提问闸门」(PreToolUse + matcher ^AskUserQuestion$)**拦的是几乎不走的工具面**,
|
||||
而真实的上抛("要我接着做吗 / 请确认 / 说一声即可")发生在**正文里** —— 没有任何机制覆盖。
|
||||
|
||||
本钩子 = 覆盖那条面:**每次回复结束时**读 transcript 的**最后一条 assistant 文本**,
|
||||
只扫**收尾段**(最后两行有效内容)里的征询句式;命中 → 返回 `{"continue": false, "reason": …}`
|
||||
让 Agent **继续一轮并自我纠正**(把该自己做的事做掉,或改写成「需要你拍板」一节)。
|
||||
|
||||
安全设计(都不许省)
|
||||
────────────────────
|
||||
1. **自作用域**:只在 `transcript_path` 落在本工作区(`aliyun-dsh-server`)时生效,其他项目一律放行。
|
||||
2. **防死循环**:输入里的 `stop_hook_active == true` 时**不再阻拦**(官方语义:本次停止已由 stop hook 触发过)。
|
||||
3. **绝不添乱**:任何异常 → 静默放行(exit 0)。判定只在**收尾段**做,避免正文引用规则时误伤。
|
||||
4. **性能**:只读转录**末尾 256 KB**(实测整库最大转录 31.9 MB、全文读 14 MB ≈ 832 ms ⇒ 不可接受),只看 stdin + 该文件。
|
||||
5. **防跑飞**:同一会话 600 秒内最多拦**一次**。
|
||||
6. **急停双闸**(无需卸载/重启):env `DSH_STOP_GUARD_OFF=1`,或新建 `<工作区>/.workbuddy/stop-guard.disabled`。
|
||||
7. **低频自证日志**:命中才写一行(`<工作区>/.workbuddy/stop-dialog-guard.log`),用来回答"到底有没有触发"。
|
||||
|
||||
退出码:始终 0;决策通过 stdout 的 JSON 表达。
|
||||
安装(settings.json 的 hooks 段 · 见档案 73 / 99):
|
||||
"Stop": [{ "hooks": [{ "type": "command",
|
||||
"command": "\"<python>\" \"<此脚本>\"", "timeout": 10 }] }]
|
||||
⚠️ hooks 是**应用启动时快照** ⇒ 装完必须**完全重启 WorkBuddy**;桌面版无 /hooks 面板,等效。
|
||||
⛔ **安装命令不要给本脚本加 `-E`(或任何会屏蔽 PYTHONUTF8 的 flag)**:本机环境本就设了
|
||||
`PYTHONUTF8=1` / `PYTHONIOENCODING=utf-8`,而 `-E` 会把它们**全部忽略** ⇒ stdin 回退 **cp936** ⇒
|
||||
含中文的 payload 解码即炸。本脚本现已改为走 `buffer` 显式 UTF-8(读写都加固),但**不要靠加固兜底**,
|
||||
装的时候也别再引入新雷。(2026-09-15 实测:`-S -E` 曾让本钩子"看起来从未被调用"整整一天。)
|
||||
"""
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
|
||||
# ── 作用域(2026-09-22 由单值扩为**多值 + env 可覆盖**)─────────────────────────
|
||||
# 起因:用户 2026-09-22 拍板「B」—— 让本钩子覆盖本机全部会话区,而不是只管 aliyun-dsh-server。
|
||||
# 形态刻意与 `dsh-ai1net-desktop/.workbuddy/guard/sync-scoped-guards.py` 生成的副本**同形**
|
||||
# (`_scopes()` + `_in_scope()`),这样两侧可互换、副本的机械变换规则不再需要改写本文件。
|
||||
# ⛔ 不写死绝对路径:只比工作区**目录名**,换机器 / 改盘符都不受影响。
|
||||
# ⚠️ env `DSH_GUARD_SCOPES`(逗号分隔)可覆盖;设为空串 ⇒ 只走默认表。
|
||||
_SCOPES_DEFAULT = (
|
||||
'aliyun-dsh-server', # 平台主工作区(本脚本所在的项目)
|
||||
'dsh-ai1net-desktop', # 桌面壳 / 开源导出线
|
||||
)
|
||||
|
||||
|
||||
def _scopes():
|
||||
"""作用域标记元组(env `DSH_GUARD_SCOPES` 优先,逗号分隔)。"""
|
||||
raw = os.environ.get('DSH_GUARD_SCOPES')
|
||||
if raw is None:
|
||||
return _SCOPES_DEFAULT
|
||||
return tuple(s.strip() for s in raw.split(',') if s.strip())
|
||||
|
||||
|
||||
def _in_scope(s):
|
||||
"""目录名匹配即算命中(与副本同形;比绝对路径稳,不受盘符/用户名影响)。"""
|
||||
return any(x in str(s or '') for x in _scopes())
|
||||
|
||||
|
||||
LOG_REL = os.path.join('.workbuddy', 'stop-dialog-guard.log')
|
||||
# 兜底工作区:用于"每次调用必留痕"(万一宿主没给 cwd、也没设 CODEBUDDY_PROJECT_DIR)
|
||||
# 本脚本位于 <工作区>/dsh-server-docs/07-scripts/ ⇒ 往上三级即工作区
|
||||
WS_FALLBACK = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
# 只扫「收尾段」:出现这些就是"把该自己做的事甩回给用户"
|
||||
PATTERNS = [
|
||||
r'要我(再|接着|继续|现在)?[^。!?\n]{0,20}吗',
|
||||
r'要不要我[^。!?\n]{0,20}',
|
||||
r'是否要我[^。!?\n]{0,20}',
|
||||
r'需要我[^。!?\n]{0,20}吗',
|
||||
r'请确认[^。!?\n]{0,16}',
|
||||
r'要不要(继续|现在做|我来)[^。!?\n]{0,20}',
|
||||
r'是否(继续|需要我)[^。!?\n]{0,20}',
|
||||
r'说一声即可',
|
||||
r'你看(怎么办|怎么弄|要不要)',
|
||||
r'你(决定|拍板)一下',
|
||||
]
|
||||
RE_BAN = re.compile('|'.join(PATTERNS))
|
||||
|
||||
REASON = (
|
||||
'⛔ 收尾句是**征询句**,但按本平台规则(`CODEBUDDY.md §1`「回话前自检」+ `dsh-feature-first §5.3` 铁律 3)'
|
||||
'先重判三问:① 命中**真门禁**吗(不可逆破坏性操作 / 边界外六类)?没命中 → **删掉这句,自己做完,改成陈述句**("我接着做 X");'
|
||||
'② 是不是在把已经定下来的事再问一遍?是 → 删;③ 这件事用户有客观可判的优劣吗?没有 → 才允许问,且**一轮只问这一句**,'
|
||||
'并写进 `dsh-feature-first §5.1` 结论骨架的「**需要你拍板**」一节 —— 该节必须是**整条回复的最后一节**、'
|
||||
'且**逐条编号**(有序段落)(2026-09-15 用户明令:「放在最后,别隐藏在回复内容中间」「按照有序段落展示」),'
|
||||
'用**陈述句**列"各候选的**优点 / 缺点** + 我的倾向",不要用征询句。'
|
||||
'⚠️ 上抛前先过**取舍筛** —— 某个候选**只有优点 / 只有缺点** ⇒ **自己拍掉、不要问**;'
|
||||
'且候选**竖排成段**(A / B / C 各占一行),⛔ 不横排、不做成表格的列(2026-09-15 用户明令)。'
|
||||
)
|
||||
|
||||
|
||||
TAIL_BYTES = 262144
|
||||
MAX_BYTES = 4194304 # 扩窗上限 4 MB(防"巨行"时无限读) # 只读末尾 256 KB(实测:整库最大转录 31.9 MB;全文读 14 MB = 832 ms/轮,不可接受)
|
||||
|
||||
|
||||
def transcribe_last_assistant(path):
|
||||
"""返回最后一条 assistant 文本(**从尾部向后分块读**;读不到返回 '')。
|
||||
|
||||
⚠️ 为什么不是"一次读末尾 256 KB":一条 assistant 记录可能本身就 > 256 KB
|
||||
(长回复 / 被回显的工具输出),此时尾窗会切在 JSON 行中间 ⇒ `json.loads` 失败 ⇒ **静默漏判**。
|
||||
做法:从尾部按 TAIL_BYTES 递增扩窗(上限 MAX_BYTES),**直到至少解析出一条 assistant 记录**。
|
||||
常见情形(小消息)只花一次 256 KB 读,成本可忽略。
|
||||
"""
|
||||
try:
|
||||
size = os.path.getsize(path)
|
||||
except OSError:
|
||||
return ''
|
||||
with io.open(path, 'rb') as f:
|
||||
window = TAIL_BYTES
|
||||
while True:
|
||||
start = max(0, size - window)
|
||||
f.seek(start)
|
||||
raw = f.read().decode('utf-8', 'replace')
|
||||
lines = raw.split('\n')
|
||||
if start > 0:
|
||||
lines = lines[1:] # 丢弃被截断的首行
|
||||
for line in reversed(lines):
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
rec = json.loads(line)
|
||||
except ValueError:
|
||||
continue
|
||||
if rec.get('type') != 'message' or rec.get('role') != 'assistant':
|
||||
continue
|
||||
chunks = [c['text'] for c in (rec.get('content') or [])
|
||||
if isinstance(c, dict) and isinstance(c.get('text'), str)]
|
||||
chunks += [c for c in (rec.get('content') or []) if isinstance(c, str)]
|
||||
if chunks:
|
||||
return '\n'.join(chunks)
|
||||
if start == 0 or window >= MAX_BYTES:
|
||||
# 放行,但**留痕**(A18:静默失败是负债)——可能是一条 >MAX_BYTES 的巨型记录
|
||||
try:
|
||||
os.environ.setdefault('_DSH_SG_MISS', '1')
|
||||
r0 = os.environ.get('CODEBUDDY_PROJECT_DIR') or os.environ.get('DSH_WORKSPACE') or ''
|
||||
if r0:
|
||||
log(r0, '未能解析(窗口 %d 字节仍无 assistant 记录)' % window)
|
||||
except Exception:
|
||||
pass
|
||||
return ''
|
||||
window = min(window * 4, MAX_BYTES)
|
||||
|
||||
|
||||
def tail_lines(text, n=2):
|
||||
out = [l.strip() for l in text.strip().split('\n') if l.strip()]
|
||||
return '\n'.join(out[-n:])
|
||||
|
||||
|
||||
# 转述/引用豁免:收尾行里带引号或"引用/规则/写着/禁"等词 ⇒ 是在复述规则,不是在问用户
|
||||
RE_QUOTE = re.compile(r'[「」“”"\']|引用|规则|写着|禁')
|
||||
|
||||
|
||||
RATE_WINDOW = 600 # 秒;同一会话两次「阻止停止」的最小间隔
|
||||
|
||||
|
||||
def _rate_limited(root, sid, peek=False):
|
||||
"""同一会话 RATE_WINDOW 秒内已拦过 ⇒ 本次直接放行(防连续多轮被拦)。"""
|
||||
if not root or not sid:
|
||||
return False
|
||||
p = os.path.join(_norm_path(root), '.workbuddy', 'cache', 'stop-guard-fires.json')
|
||||
try:
|
||||
d = json.loads(io.open(p, encoding='utf-8').read()) if os.path.exists(p) else {}
|
||||
except Exception:
|
||||
d = {}
|
||||
now = time.time()
|
||||
if now - float(d.get(sid, 0) or 0) < RATE_WINDOW:
|
||||
return True
|
||||
d = {k: v for k, v in d.items() if now - float(v or 0) < 86400} # 只留 1 天
|
||||
d[sid] = now
|
||||
try:
|
||||
os.makedirs(os.path.dirname(p), exist_ok=True)
|
||||
io.open(p, 'w', encoding='utf-8', newline='\n').write(json.dumps(d))
|
||||
except Exception:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _norm_path(p):
|
||||
"""把 MSYS / Git-Bash 风格路径规范成 Windows 风格(`/e/foo` → `E:/foo`)。
|
||||
|
||||
动机(2026-09-22 实测,**正在持续发生**):**Windows 原生 python 会把
|
||||
`/e/ProgramData/x` 解释成「当前盘根 + 相对路径」= `e\\ProgramData\\x`**
|
||||
⇒ 若当前盘是 E,就落到 `E:\\e\\ProgramData\\x`。
|
||||
实证:E 盘根长出影子目录 `E:\\e\\ProgramData\\AI技能\\<ws>\\.workbuddy\\`,
|
||||
内含本 hook 的 `stop-dialog-guard.log`(2496 B,最后写入 09-22 06:17)与
|
||||
浏览器 `_devlogs/pud-f1/`(Chromium user-data-dir)—— 共 355 文件 / 29 MB,且**仍在增长**。
|
||||
|
||||
⛔ 同源铁律:本工作区已有「**Python exe 不认 `/e/…` ⇒ 传 `E:/…`**」。
|
||||
"""
|
||||
try:
|
||||
s = str(p or '')
|
||||
m = re.match(r'^/([A-Za-z])(/.*)?$', s)
|
||||
if m:
|
||||
return m.group(1).upper() + ':' + (m.group(2) or '/')
|
||||
except Exception:
|
||||
pass
|
||||
return p
|
||||
|
||||
|
||||
def log(root, detail):
|
||||
try:
|
||||
p = os.path.join(_norm_path(root), LOG_REL)
|
||||
os.makedirs(os.path.dirname(p), exist_ok=True)
|
||||
with io.open(p, 'a', encoding='utf-8') as f:
|
||||
f.write('%s\t%s\n' % (time.strftime('%Y-%m-%d %H:%M:%S'), detail))
|
||||
lines = io.open(p, encoding='utf-8').read().split('\n') # 上限 300 行,超出截半(防膨胀)
|
||||
if len(lines) > 300:
|
||||
io.open(p, 'w', encoding='utf-8', newline='\n').write('\n'.join(lines[-150:]))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _entry_log(payload, raw_len):
|
||||
"""⚠️ **每次被调用必留痕**(含"payload 解析失败 / 未进作用域 / 被急停"三种静默情形)。
|
||||
|
||||
2026-09-15 教训:原实现只在**通过全部守卫之后**才写日志 ⇒ 日志缺失时**无法区分**
|
||||
「宿主根本没调用」与「调用了但被静默 return」—— 而这两者的处置**完全相反**
|
||||
(前者要卸载、后者要放宽作用域判据)。凡"要判有没有被调用"的探针,必须**入口即留痕**。
|
||||
"""
|
||||
try:
|
||||
p = payload if isinstance(payload, dict) else {}
|
||||
tp = str(p.get('transcript_path') or '')
|
||||
root = (os.environ.get('CODEBUDDY_PROJECT_DIR') or os.environ.get('DSH_WORKSPACE')
|
||||
or p.get('cwd') or WS_FALLBACK)
|
||||
log(str(root), 'entry|event=%s|cwd=%s|in_scope=%s|tp=%s|keys=%s|stdin_len=%s'
|
||||
% (p.get('hook_event_name') or '(parse-fail)', p.get('cwd') or '-',
|
||||
_in_scope(tp), (tp[-80:] if tp else '-'),
|
||||
(','.join(sorted(p.keys()))[:120] or '-'), raw_len))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _read_stdin_text():
|
||||
"""**显式按 UTF-8 读 stdin** —— 不要用 `sys.stdin.read()`。
|
||||
|
||||
⚠️ 2026-09-15 实测定位:本脚本的安装形态是 `python -S -E <脚本>`,而 **`-E` 会忽略
|
||||
`PYTHONUTF8=1` / `PYTHONIOENCODING=utf-8`** ⇒ `sys.stdin.encoding` 回退成 **cp936**;
|
||||
钩子 payload 里**必然含中文**(用户的提示词)⇒ 文本模式读取抛
|
||||
`UnicodeDecodeError: 'gbk' codec can't decode byte 0x80` ⇒ **钩子静默不生效、日志为空**,
|
||||
表象却是"宿主好像没调用钩子"(实为本地炸在解码上,白排查一轮)。
|
||||
读 `buffer` 即与 flag / locale 完全无关。
|
||||
"""
|
||||
try:
|
||||
return sys.stdin.buffer.read().decode('utf-8', 'replace')
|
||||
except Exception:
|
||||
try:
|
||||
return sys.stdin.read()
|
||||
except Exception:
|
||||
return ''
|
||||
|
||||
|
||||
def _emit(obj):
|
||||
"""**显式按 UTF-8 写 stdout**(同理:cp936 下 `ensure_ascii=False` 的中文 / `⛔` 会 UnicodeEncodeError)。"""
|
||||
data = json.dumps(obj, ensure_ascii=False).encode('utf-8')
|
||||
try:
|
||||
sys.stdout.buffer.write(data)
|
||||
sys.stdout.buffer.flush()
|
||||
except Exception: # 极端兜底:退回文本写(可能丢非 GBK 字符,但不至于静默不输出)
|
||||
try:
|
||||
sys.stdout.write(data.decode('utf-8', 'replace'))
|
||||
sys.stdout.flush()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────
|
||||
# 第二方案:`UserPromptSubmit`(2026-09-15 加)
|
||||
# 背景:本版 WorkBuddy **不调用 `Stop` 钩子**(实测:留痕已开、探针句已验证会命中、日志仍空)⇒ 改用
|
||||
# `UserPromptSubmit`(输入同样带 `transcript_path`,且能通过 `additionalContext` 注入上下文)。
|
||||
# 两级模式(**改一个文本文件即可切换,无需重启** —— 脚本内容每次调用现读):
|
||||
# probe :只写日志(零风险、可判定"有没有被调用")
|
||||
# inject :若**上一轮回复的收尾是征询句** ⇒ 注入一段上下文,让下一轮自我纠正
|
||||
# 模式文件:<工作区>/.workbuddy/stop-guard-mode (内容含 "inject" 即切到 inject,否则 probe)
|
||||
CONTEXT = (
|
||||
'⛔ 【上一轮收尾自检】你上一条回复的**最后一行是征询句**("要我…吗 / 要不要我 / 请确认 / 说一声即可"类),'
|
||||
'这属于本平台**被禁的形态**(`CODEBUDDY.md §1`「回话前自检」)。本轮的处置:'
|
||||
'① 若那件事本来就该你自己拍 —— **直接做完**,用陈述句交代;'
|
||||
'② 若确实命中真门禁(不可逆破坏性操作 / 边界外六类)—— 写进 `dsh-feature-first §5.1` 结论骨架的'
|
||||
'「**需要你拍板**」一节,该节必须是**整条回复的最后一节**、**逐条编号**,且**每个候选写明优点 / 缺点**、**候选竖排成段**(A / B / C 各占一行,⛔ 不横排、不做成表格的列)(陈述句,不要用征询句);'
|
||||
'⚠️ 若某候选**只有优点或只有缺点** ⇒ **那不该问**,自己拍掉;'
|
||||
'③ 顺带按红线 **R11** 复核:这个改动有没有让项目某一维度**净变差**。'
|
||||
)
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────
|
||||
# 会话预算(2026-09-15 用户选 A 案 · 提交 03c8363):到 **12 万 token / 80 次工具调用** ⇒ 提醒开新会话
|
||||
# ⚠️ 2026-09-16 更正:本行原写「~15 万 token / ~120 次工具调用」,与代码值(`BUDGET_TOKENS=120000` /
|
||||
# `BUDGET_TOOLS=80`)**不符** —— 提交 `03c8363` 的 message 里就写的是 `120000/80`,代码从未用过 15 万/120
|
||||
# ⇒ 判定为注释笔误,已按代码更正(避免后人据注释误判告警点)。
|
||||
# 数值本身由 AI 在实现层自定(属 §1「性能与资源调参」= 边界内);**方案(A 案四招)才是用户 09-15 拍板的**。
|
||||
# 为什么加进本脚本、而不新装一个 hook:`settings.json` 的 hooks 是**应用启动时快照**(新增条目要重启),
|
||||
# 而**脚本内容每次调用现读** ⇒ 改这里即刻生效。数据源 = 转录 `type=function_call` → `message.usage.input_tokens`
|
||||
# (最近一条 usage 即"当前上下文体量",精确,不靠估算)。
|
||||
# 依据(实测某会话):上下文 5.2 万 → 59.2 万;累计 input 1.93 亿 / output 51.9 万(**371:1**);
|
||||
# 其中 33 次缓存失效,每次都把 ~50 万 token **按全价**重算 ⇒ 会话越长,单次失效越贵。
|
||||
# ★ 2026-09-16 改为**分级 + 去重**(原为单一阈值 120000/80,超了之后每轮都报 ⇒ 变成"狼来了")
|
||||
# 设计要点:**同一级别只报一次** ⇒ 跨级才再提醒,既不麻木也不失警。
|
||||
BUDGET_TOKENS = 120000 # 一级:轻提示(保留原名,兼容既有引用)
|
||||
BUDGET_STRONG = 200000 # 二级:建议收口
|
||||
BUDGET_FORCE = 300000 # 三级:**强制收口**(先落盘、出接续包,再开新会话)
|
||||
BUDGET_TOOLS = 80
|
||||
ALERT_LEVEL_REL = os.path.join('.workbuddy', '.budget-alert-level')
|
||||
|
||||
# 为什么三级不是"自动开新会话"(2026-09-16 与用户讨论后定):
|
||||
# ① **技术上做不到** —— hook 只有 `additionalContext`(注入)与 `permissionDecision`(拦工具)两种输出,
|
||||
# 事件只有 SessionStart / PreToolUse / UserPromptSubmit / Stop,**没有"创建/切换会话"的能力**;
|
||||
# AI 自身也只能在会话内行动。⇒ "自动开"这一半无法实现。
|
||||
# ② **设计上不该做** —— 新会话 = 上下文清零 ⇒ **在途状态全丢**;而"该带走什么"只有 AI 判断得了
|
||||
# ⇒ 顺序必须是 **先收口(落盘 + 接续包)→ 再由用户开新会话**,反了就是"突然失忆"。
|
||||
# ③ 因此本级**不做"自动开"**,做**"强制收口"**:把状态固化成文档,让下一个会话能无损接上。
|
||||
LV_PREFIX = {
|
||||
1: '',
|
||||
2: '⚠️ 上下文已过 **20 万**,本轮收口后建议开新会话。\n'
|
||||
' ⚠️ 本轮若你**登记了自动接续(automation)/ 要开新会话** ⇒ **必须在给用户的回复里用陈述句说明**(别悄悄做掉)。\n',
|
||||
3: ('🔴 **已过 30 万 ⇒ 进入强制收口模式**(先落盘、再接续):\n'
|
||||
' ① 把在途状态写进 `.workbuddy/memory/`(今日日志 + 必要的 MEMORY.md 条目);\n'
|
||||
' ② 产出**接续包**:目标 / 已完成 / 在途 / 未完成 / 下一步 / 关键决定 / 回滚点'
|
||||
'(模板见工作区根 `会话接续规范_20260916.md §3.1.1`);\n'
|
||||
' ③ **把接续登记成一次性 automation**(调 `automation_update`,+2 分钟触发)'
|
||||
'—— 这是"自动接续"唯一可用的通道(钩子不能建会话、不能建自动化);\n'
|
||||
' prompt **照 `会话接续规范 §3.2.1` 模板**:⓪ 先跑 `state.py` + 接续包路径 + 开机四步 + 工具调用上限;\n'
|
||||
' ⛔ prompt 里**不许复制任务细节**("细节的唯一来源是交接单/接续包";实测:抄细节的 prompt 会挤掉"开机四步",'
|
||||
'那轮跑了 40 次调用 / 9.37 分);\n'
|
||||
' 🔴 **prompt 必须带「接续包 md5」**(`md5sum <接续包路径>`)—— 新会话开工前会重算校验,'
|
||||
'不符即停、报告「口径已更新,需重新接续」(2026-09-16 实测:登记 17:01 → 触发 17:03,'
|
||||
'而原会话一路工作到 18:21、17:2x–17:29 还在改判断 ⇒ 新旧两张皮);\n'
|
||||
' 🔴 **登记后若你还要改接续包 / 改关键判断 ⇒ 回来撤销或重登记**这条 automation ——'
|
||||
'⛔ 不重登记 = 下一棒按旧口径开工且无从知道;\n'
|
||||
' ④ ★**必做 · 告知用户**:在**给用户的最后一条回复里**(用陈述句,不是征询句)写明这次自动接续 —— '
|
||||
'例如「已登记自动接续:约 N 分钟后自动开新会话继续(**不需要你操作**);接续点 = `<文件>`;'
|
||||
'若想自己开,口令 = `<state.py 口令>`」。\n'
|
||||
' ⛔ **这不是可选项**:新建会话 / 新建自动化是**用户可感知的状态变更**'
|
||||
'(提问闸门 A 类原话:「AI 会不会悄悄改他的设置」)—— 悄悄做掉不告知,'
|
||||
'用户会在会话列表里凭空看见多出一个会话而不知何来。\n'
|
||||
' 📌 2026-09-16 用户实测反馈:「**他在最后一个回复结尾没说这个事,导致我不知道**」⇒ 本条即由此而来。\n'
|
||||
' ⑤ ③ 若做不成(工具不可用等)⇒ 兜底同样要告知:「请开新会话,接续点在 X」,**由用户开**。\n'
|
||||
' ⛔ 不要在本会话继续开新任务 —— 每多跑一轮,成本按当前水位线性放大。\n'),
|
||||
}
|
||||
|
||||
|
||||
def _alert_level(tokens, ncalls):
|
||||
"""0=未超 1=轻 2=建议收口 3=强制收口"""
|
||||
t = tokens or 0
|
||||
if t >= BUDGET_FORCE:
|
||||
return 3
|
||||
if t >= BUDGET_STRONG:
|
||||
return 2
|
||||
if t >= BUDGET_TOKENS:
|
||||
return 1
|
||||
if ncalls is not None and ncalls >= BUDGET_TOOLS:
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
def _last_level(root, sid):
|
||||
"""读「本会话」已报到的级别。**按 sid 区分** —— 否则新会话会被上一会话的状态挡住 ⇒ **漏报**。"""
|
||||
try:
|
||||
with io.open(os.path.join(_norm_path(root), ALERT_LEVEL_REL), encoding='utf-8') as f:
|
||||
s = (f.read() or '').strip()
|
||||
ps = s.split('\t')
|
||||
if len(ps) == 2 and ps[0] == sid:
|
||||
return int(ps[1] or 0)
|
||||
except Exception:
|
||||
pass
|
||||
return 0
|
||||
|
||||
|
||||
def _set_level(root, sid, lv):
|
||||
try:
|
||||
p = os.path.join(_norm_path(root), ALERT_LEVEL_REL)
|
||||
os.makedirs(os.path.dirname(p), exist_ok=True)
|
||||
with io.open(p, 'w', encoding='utf-8', newline='\n') as f:
|
||||
f.write('%s\t%d' % (sid, lv))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def session_budget(path):
|
||||
"""返回 (当前上下文 token, 工具调用累计次数);读不到返回 (None, None)。"""
|
||||
try:
|
||||
if os.path.getsize(path) > 64 * 1024 * 1024:
|
||||
return None, None
|
||||
except OSError:
|
||||
return None, None
|
||||
last_in, n_calls, prev_in = None, 0, None
|
||||
try:
|
||||
with io.open(path, encoding='utf-8', errors='replace') as f:
|
||||
for ln in f:
|
||||
if '"function_call"' not in ln:
|
||||
continue
|
||||
n_calls += 1
|
||||
if '"usage"' not in ln:
|
||||
continue
|
||||
try:
|
||||
o = json.loads(ln)
|
||||
except ValueError:
|
||||
continue
|
||||
if o.get('type') != 'function_call':
|
||||
continue
|
||||
u = ((o.get('message') or {}).get('usage')) or {}
|
||||
v = u.get('input_tokens')
|
||||
if v:
|
||||
prev_in, last_in = last_in, int(v)
|
||||
except OSError:
|
||||
return None, None, None
|
||||
return last_in, n_calls, prev_in
|
||||
|
||||
|
||||
def budget_note(tokens, ncalls, prev=None, root='', sid=''):
|
||||
"""分级提醒;**同一级别只注入一次**(跨级才再报)—— 治"超阈值后每轮都报"导致的麻木。
|
||||
未超阈值时,单轮增量异常(≥4 万 token)仍提醒。"""
|
||||
# ★ 分级 + 去重(2026-09-16):本级已报过 ⇒ 静默(**按会话 id 区分**,新会话重新计数)
|
||||
lv = _alert_level(tokens, ncalls)
|
||||
if lv and lv <= _last_level(root, sid):
|
||||
return ''
|
||||
if lv:
|
||||
_set_level(root, sid, lv)
|
||||
over_t = tokens is not None and tokens >= BUDGET_TOKENS
|
||||
over_n = ncalls is not None and ncalls >= BUDGET_TOOLS
|
||||
grew = (tokens is not None and prev is not None and tokens - prev >= 40000) if not (over_t or over_n) else False
|
||||
if not (over_t or over_n or grew):
|
||||
return ''
|
||||
which = ('上下文 %s token' % tokens) if over_t else (
|
||||
'工具调用 %s 次' % ncalls) if over_n else ('**上一轮新增 %s token**' % (tokens - prev))
|
||||
return (
|
||||
LV_PREFIX.get(lv, '') +
|
||||
'💰 【会话预算告警】本会话已到 **%s**(一级阈值 %s token / %s 次工具调用)。'
|
||||
'代价机制(2026-09-15 实测):对话历史是**追加式**的 —— 一次工具调用的输出(`function_call_result`)'
|
||||
'**永久留在历史里、每轮全量重发**(某会话 553 轮 × 平均 42 万 = **2.33 亿 input**,output 仅 0.35%%)。'
|
||||
'**处置:本轮收口后开新会话**(新会话起点约 5 万 ⇒ 每轮降到 1/12);'
|
||||
'并把该记的状态写进记忆 + 按 `dsh-change-workflow` 交付门禁落交接。'
|
||||
'⚠️ 压增长的三条硬纪律:**❶ 大输出先落盘、只读关键行**(`> /tmp/x.txt` 后 `sed -n`);'
|
||||
'**❷ 命令层限流**(`| head -30` / `| cut -c1-120` / `grep -c` 代替 `grep`);'
|
||||
'**❸ 让脚本内部聚合、只 print 摘要** —— ⛔ 禁 `cat` 大文件、无 `head` 的 `grep -r`、`ls -laR`。'
|
||||
% (which, BUDGET_TOKENS, BUDGET_TOOLS)
|
||||
)
|
||||
|
||||
|
||||
def guard_health(root):
|
||||
"""**自动发现并上报**:读 `bash-guard.log`,同一规则重复命中 ≥3 次 ⇒ 提示"可能误伤"。
|
||||
|
||||
为什么要有:门禁自己不会喊疼 —— 只有"写日志"没人看。让每轮都跑的脚本顺带体检,
|
||||
机制问题才能在**下一次用户发言时**浮出来(而不是等人发现"AI 怎么老做不成事")。
|
||||
"""
|
||||
try:
|
||||
p = os.path.join(_norm_path(root), '.workbuddy', 'bash-guard.log')
|
||||
rows = [l for l in io.open(p, encoding='utf-8').read().split('\n') if 'DENY|' in l][-40:]
|
||||
except OSError:
|
||||
return ''
|
||||
if len(rows) < 3:
|
||||
return ''
|
||||
cnt = {}
|
||||
for l in rows:
|
||||
try:
|
||||
k = l.split('DENY|')[1].split('|')[0].strip()
|
||||
except IndexError:
|
||||
continue
|
||||
cnt[k] = cnt.get(k, 0) + 1
|
||||
if not cnt:
|
||||
return ''
|
||||
why, n = max(cnt.items(), key=lambda kv: kv[1])
|
||||
if n < 3:
|
||||
return ''
|
||||
return ('⚠️【门禁自检】最近 %d 次 Bash 里有 **%d 次**因「%s」被拦(同一规则重复命中)⇒ 先怀疑**误伤**,'
|
||||
'不是谁的操作有问题。处置:① 换等价限流写法;② 若确认误伤 ⇒ 把 `off` 写进 '
|
||||
'`.workbuddy/bash-guard-mode`(或 env `DSH_OUTPUT_GUARD_OFF=1`),并**主动上报用户**'
|
||||
'(规则该不该收窄是人的决定)。' % (len(rows), n, why))
|
||||
|
||||
|
||||
def path_health(root):
|
||||
"""**自动发现并上报**:hooks 指向的脚本、文档库位置,是否还在。
|
||||
|
||||
来历(2026-09-15 实测事故):文档库被整目录搬到 `_中间产物_待清理/` ⇒ 宿主 hooks 仍指向旧位置 ⇒
|
||||
**锁闸门静默失效**;当时唯一线索是"`lock-hook.log` 今天 0 条 PreToolUse",而**没人会主动去数**。
|
||||
⇒ 让每轮都跑的脚本顺带体检,机制失效能在**下一次用户发言时**自己浮出来。
|
||||
|
||||
判据(任一命中即报):① **从 `settings.json` 现读** hooks 的 command,抽出其中的 `.py` 逐个 `os.path.exists`
|
||||
(配置里是权威指向 ⇒ 能发现"指向了不存在的文件");② 文档库在两处候选位置**都不存在**。
|
||||
取不到配置 ⇒ 跳过该项(**fail-open**,绝不因体检本身误报)。
|
||||
"""
|
||||
bad = []
|
||||
cfg = os.path.join(os.environ.get('WORKBUDDY_CONFIG_DIR')
|
||||
or os.path.join(os.path.expanduser('~'), '.workbuddy'), 'settings.json')
|
||||
seen = 0
|
||||
try:
|
||||
d = json.load(io.open(cfg, encoding='utf-8'))
|
||||
for ev, arr in (d.get('hooks') or {}).items():
|
||||
for blk in (arr or []):
|
||||
for h in ((blk or {}).get('hooks') or []):
|
||||
c = h.get('command') or ''
|
||||
if c:
|
||||
seen += 1
|
||||
for m in re.finditer(r'([A-Za-z]:[\\/][^"\']*?\.py)', c):
|
||||
p = m.group(1)
|
||||
if not os.path.exists(p):
|
||||
bad.append('hooks「%s」指向的 `%s` 不存在' % (ev, os.path.basename(p)))
|
||||
except Exception:
|
||||
return '' # 读不到配置 ⇒ fail-open,绝不因体检本身误报
|
||||
if seen == 0:
|
||||
bad.append('hooks 配置里**没有任何 command 条目** ⇒ 钩子可能被清空 / 被整段覆盖')
|
||||
if not bad:
|
||||
return ''
|
||||
uniq = []
|
||||
for x in bad:
|
||||
if x not in uniq:
|
||||
uniq.append(x)
|
||||
return ('🚨【路径自检】%s ⇒ **机制可能已静默失效**。处置:**立刻上报用户**,并核对 `settings.json` 的 '
|
||||
'hooks 路径与文档库当前位置(2026-09-15 同类事故:文档库被整目录搬走,锁闸门失效一整天无人察觉)。'
|
||||
% ';'.join(uniq[:3]))
|
||||
|
||||
|
||||
def mode_of(root):
|
||||
try:
|
||||
m = io.open(os.path.join(root or '.', '.workbuddy', 'stop-guard-mode'), encoding='utf-8').read()
|
||||
except OSError:
|
||||
m = ''
|
||||
return 'inject' if 'inject' in m else 'probe'
|
||||
|
||||
|
||||
def user_prompt_mode(payload):
|
||||
"""UserPromptSubmit:probe=只记日志;inject=命中则注入上下文(不阻断提示词)。"""
|
||||
tp = str(payload.get('transcript_path') or '')
|
||||
if not _in_scope(tp):
|
||||
return
|
||||
if os.environ.get('DSH_STOP_GUARD_OFF'):
|
||||
return
|
||||
root0 = os.environ.get('CODEBUDDY_PROJECT_DIR') or os.environ.get('DSH_WORKSPACE') or ''
|
||||
if root0 and os.path.exists(os.path.join(_norm_path(root0 or '.'), '.workbuddy', 'stop-guard.disabled')):
|
||||
return
|
||||
mode = mode_of(root0)
|
||||
text = transcribe_last_assistant(tp)
|
||||
tail = tail_lines(text, 1) if text else ''
|
||||
hit = bool(tail) and not RE_QUOTE.search(tail) and bool(RE_BAN.search(tail))
|
||||
toks, ncalls, prev = session_budget(tp)
|
||||
note = budget_note(toks, ncalls, prev, root0 or '.', str(payload.get('session_id') or ''))
|
||||
gh = guard_health(root0 or '.')
|
||||
ph = path_health(root0 or '.')
|
||||
log(root0 or '.', 'invoked(user-prompt)|mode=%s|上轮收尾=征询句:%s|上下文=%s tok(+%s)|工具=%s 次|预算告警=%s|门禁自检=%s|路径自检=%s|%s'
|
||||
% (mode, hit, toks, (toks - prev) if (toks and prev) else '-', ncalls, bool(note), bool(gh), bool(ph),
|
||||
(tail.replace('\n', ' ')[:60] if tail else '(取不到上一轮文本)')))
|
||||
# ⚠️ 2026-09-15 用户选 C:**取消**「会话预算」注入 —— 当时的症结是"每轮都报、太吵"。
|
||||
# ★ 2026-09-16 恢复**分级注入**:`budget_note` 已改为「分级 + 同会话跨级才报一次」(见其 docstring),
|
||||
# 噪音症结已消解;且用户当日提出「上下文超 30 万要不要自动收口」⇒
|
||||
# **AI 必须先知道水位才谈得上收口** ⇒ 恢复注入(一级静默无关、二级建议、三级强制收口)。
|
||||
if mode == 'inject' and (hit or gh or ph or note):
|
||||
ctx = ''
|
||||
for seg in (CONTEXT if hit else '', note, gh, ph):
|
||||
if seg:
|
||||
ctx += ('\n\n' + seg) if ctx else seg
|
||||
_emit({'hookSpecificOutput': {'hookEventName': 'UserPromptSubmit',
|
||||
'additionalContext': ctx}})
|
||||
|
||||
|
||||
def main():
|
||||
raw = _read_stdin_text() # ⚠️ 必须走 buffer:`-E` 下 sys.stdin 是 cp936(见 _read_stdin_text 注释)
|
||||
payload = None
|
||||
if raw.strip():
|
||||
try:
|
||||
payload = json.loads(raw)
|
||||
except ValueError:
|
||||
payload = None
|
||||
_entry_log(payload, len(raw)) # ⚠️ 先留痕,再判作用域(否则"没被调用"与"静默失配"分不开)
|
||||
if payload is None:
|
||||
return
|
||||
if (payload.get('hook_event_name') or '') == 'UserPromptSubmit': # 第二方案分派
|
||||
return user_prompt_mode(payload)
|
||||
tp = str(payload.get('transcript_path') or '')
|
||||
if not _in_scope(tp): # 作用域外 → 放行
|
||||
return
|
||||
if os.environ.get('DSH_STOP_GUARD_OFF'): # 急停(环境变量)→ 放行
|
||||
return
|
||||
root0 = os.environ.get('CODEBUDDY_PROJECT_DIR') or os.environ.get('DSH_WORKSPACE') or ''
|
||||
if root0 and os.path.exists(os.path.join(_norm_path(root0 or '.'), '.workbuddy', 'stop-guard.disabled')):
|
||||
return # 急停(闸刀文件)→ 放行
|
||||
if payload.get('stop_hook_active'): # 防死循环 → 放行
|
||||
return
|
||||
text = transcribe_last_assistant(tp)
|
||||
if not text:
|
||||
return
|
||||
sid = str(payload.get('session_id') or '')
|
||||
if os.environ.get('DSH_SG_DEBUG'): # 调试:每次调用都留痕(用于验证宿主是否真的调用本钩子)
|
||||
log(root0 or '.', 'invoked|scope=%s|tail_active=%s' % (_in_scope(tp), bool(payload.get('stop_hook_active'))))
|
||||
if _rate_limited(root0, sid, peek=True): # 只查不记账
|
||||
return
|
||||
if os.environ.get('DSH_SG_LOG_ALL', '1') != '0': # ★本工作区内**每次调用都留痕**(可判定"有没有被调用")
|
||||
log(root0 or '.', 'invoked|tail_active=%s' % bool(payload.get('stop_hook_active')))
|
||||
tail = tail_lines(text, 1) # 只看**最后一行**:命中面越窄,误报越少
|
||||
if RE_QUOTE.search(tail): # 复述/引用规则 → 不是收尾提问
|
||||
return
|
||||
m = RE_BAN.search(tail)
|
||||
if not m:
|
||||
return
|
||||
root = (os.environ.get('CODEBUDDY_PROJECT_DIR') or os.environ.get('DSH_WORKSPACE') or '')
|
||||
_rate_limited(root0, sid) # 命中才记账(同一会话 10 分钟最多拦 1 次)
|
||||
log(root if os.path.isdir(root) else '.', 'stop-dialog-guard 命中:%s | 收尾:%s'
|
||||
% (m.group(0), tail.replace('\n', ' ')[:80]))
|
||||
_emit({'continue': False, 'reason': REASON})
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
try:
|
||||
main()
|
||||
except Exception:
|
||||
# ⚠️ 钩子绝不能因自身故障干扰会话 ⇒ 仍放行,但**必须留痕**(A18:静默失败是负债;
|
||||
# 2026-09-15 实证:本文件的 `except: pass` 曾把 `NameError: out is not defined` 藏住半小时)
|
||||
try:
|
||||
import traceback
|
||||
_r = os.environ.get('CODEBUDDY_PROJECT_DIR') or os.environ.get('DSH_WORKSPACE') or '.'
|
||||
log(_r, 'EXCEPTION|%s' % traceback.format_exc().strip().split('\n')[-1][:120])
|
||||
except Exception:
|
||||
pass
|
||||
sys.exit(0)
|
||||
Reference in new issue
Block a user