#!/usr/bin/env python3 # -*- coding: utf-8 -*- """session-rules-check.py —— 「会话规则机制体检」= 开会话后的第一件事 用户口径(2026-10-02 原话逐字,含一次纠正): ① 「这个会话和协作会话的技能包 运行的第一件事 ,就应该是检查清楚 所有会话规划是否配置完整且生效,然后标记一个状态」 ② 「就应该是检查清楚 **所有会话规则机制** 是否配置完整且生效,**不是规划 是 规则**」 ⇒ 对象是「**规则机制**」(钩子 / 闸门 / 技能指针 / 常驻 / 编排…),⛔ 不是"排期规划"。 为什么要它(真因,⛔ 不是想当然) ────────────────────────────── 2026-10-02 一天里连续查出**四例「配置在册、其实没生效」**: · 钩子注入文本指着 **09-28 就已合并退役**的技能名(命中后让人去加载不存在的东西); · 常驻规则快照的生成脚本**写到没人读的幽灵目录** ⇒ 快照停在 09-28、10-01 的定稿进不去; · 工作区记忆(**每轮注入**的那份)里 3 处指针**指向已退役技能名**; · 锚点校验拿**旧版逐字短语**当探针 ⇒ 每次报 4 处「规则丢了」的**假缺失**。 ⇒ 共同点:**"在册" ≠ "生效"**。这类状态**不问就不会知道**,等它表现成"卡住"就晚了。 ⇒ 所以开工第一件事=**把规则的"配置"与"生效"两件事都问清楚,并落一枚状态标记**。 检查三类、十二项(⛔ 每项都必须**能真报出问题** —— 见文末变异对照记录): A 机制装没装好 ① 关键钩子在册 ② 钩子脚本路径存在 ③ **钩子注入里引用的技能名是否都存在** ④ 钩子**真在被调用**没(闸门日志新鲜度) B 规则载体同没同步 ⑤ **每轮注入的记忆里引用的技能名是否都存在** ⑥ 常驻规则快照是否**比权威文件旧** C 编排在不在跑 ⑦ **已退役角色的周期钟 ⛔ 不再要求**(10-03 口径:唤醒/跟进整套退役) —— 改判「在册的退役周期钟应已 PAUSED」;现行编排载体=⑩ 常驻心跳 ⑨ `cwds` **归属同形**(错一字面 ⇒ 裂组且自我强化) ⑩ 投递心跳 ⑪ 三类会话有活的没 ⑫ 有没有「`once` 且**从未运行过**」就失效的排期 产出:`/.workbuddy/collab/session-rules.json`(=用户要的那枚「状态标记」)+ stdout 摘要 ⚠️ stdout **只打非 ok 项**(本脚本每轮都会随 `state.py` 跑 ⇒ 输出本身就是成本) 退出码:0 = 全 ok | 1 = 有 warn/fail(⛔ **只标记,不拦开工**) 用法:python session-rules-check.py [--ws <工作区>] [--json] [--all] """ import ast import io import json import os import re import sqlite3 import sys import time # ── 根目录外置(roots.env 由 install.py 生成;宿主 env 优先,本文件兜底) ────── def _sm_load_roots(): here = os.path.dirname(os.path.abspath(__file__)) for up in range(4): p = os.path.normpath(os.path.join(here, *([".."] * up), "roots.env")) if os.path.isfile(p): try: for ln in io.open(p, encoding="utf-8"): ln = ln.strip() if ln and not ln.startswith("#") and "=" in ln: k, v = ln.split("=", 1) os.environ.setdefault(k.strip(), v.strip()) except Exception: pass return _sm_load_roots() ARGV = sys.argv[1:] def _opt(name, default=None): if name in ARGV: i = ARGV.index(name) return ARGV[i + 1] if i + 1 < len(ARGV) else default return default WS = (_opt("--ws") or os.environ.get("DSH_COLLAB_WS") or os.environ.get("DSH_WS_ROOT") or os.getcwd()).replace("\\", "/").rstrip("/") CFG = os.environ.get("CODEBUDDY_CONFIG_DIR") or os.path.join(os.path.expanduser("~"), ".workbuddy") DB = os.path.join(CFG, "workbuddy.db") SETTINGS = os.path.join(CFG, "settings.json") SKILLS = os.environ.get("DSH_SKILLS_ROOT") or os.path.join(CFG, "skills") OUT = WS + "/.workbuddy/collab/session-rules.json" AS_JSON = "--json" in ARGV SHOW_ALL = "--all" in ARGV # 「会话机制三件套」的**关键钩子**(缺一 ⇒ 该机制不会运行) # 🔴🔴 2026-10-06 修:第 2 元由**单个文件名**改成**一组可接受的文件名**(命中任一即算在册)。 # 起因(实测坐实):`UserPromptSubmit` 上原有 4 条独立守卫(`stop-dialog-guard` / # `skill-load-guard` / `reply-style-guard` / `session-log-guard`),每轮**各起一个 Python 进程** # ⇒ 宿主冷启 7 次解释器 ⇒ 预算被排队挤破 ⇒ 报 `Hook timed out after 10000ms`(用户报障)。 # ⇒ 已**合并**为一个入口 `prompt-guards.py`(本进程内 `runpy` 依次跑那 4 个,合并输出)。 # ⚠️ 本条判据若仍按**旧文件名**找 ⇒ 合并后**每轮必报「关键钩子不在册」(假红)**, # 而它惩罚的正是"按要求做过的合并"。 # 🔴 判据要问的是「**这项能力有没有接线**」,⛔ 不是「某个文件名在不在」—— # 所以两种形态**都必须认**:独立接线 | 合并入口。 KEY_HOOKS = [ ("SessionStart", ("lock-guard-hook.py",), "无锁不许改代码库 / 文档库"), ("PreToolUse", ("lock-guard-hook.py",), "同上(Write|Edit 面)"), ("PreToolUse", ("bash-output-guard.py",), "拦下会灌爆上下文的读命令"), ("UserPromptSubmit", ("wb-result-hook.py",), "收结果 / 结果回流"), ("UserPromptSubmit", ("stop-dialog-guard.py", "prompt-guards.py"), "水位与收口(接续机制起点)"), ("UserPromptSubmit", ("skill-load-guard.py", "prompt-guards.py"), "用户点名方法 ⇒ 强制加载技能"), ("UserPromptSubmit", ("session-log-guard.py", "prompt-guards.py"), "会话日志闸(防把界面顶死)"), ("PostToolUse", ("session-log-guard.py",), "同上(工具后)"), ] GATE_LOGS = [("收口", WS + "/.workbuddy/stop-dialog-guard.log"), ("技能", WS + "/.workbuddy/skill-load-guard.log"), ("限流", WS + "/.workbuddy/bash-guard.log"), # ⚠️ 锁日志落在**文档库的父目录**(lock-guard-hook 管的是文档库 / 代码库,不是本工作区) # ⇒ 首版按 skills 目录往上推两层 ⇒ 算成 `E:/ProgramData/.workbuddy/…` ⇒ **假"缺失"**。 ("锁", (os.path.join(os.path.dirname(os.environ.get("DSH_DOCS_ROOT", "")), ".workbuddy", "lock-hook.log") if os.environ.get("DSH_DOCS_ROOT") else ""))] # 🔴🔴 2026-10-05 **用户定案:「协作 全部 改为 执行」**(连说两遍,第二轮逐字: # 「**兼容个毛线,今天兼容一个明天兼容一个 过不了一周就成大杂烩了**」)。 # ⛔⛔ 本文件里 `协作` 二字**不准再出现在任何"活"的语境里**。 # 「兼容」这个词**本身就是病** —— 后人读到"兼容"会以为它迟早能删,于是一拖再拖。 # ⇒ 凡看到旧前缀,必须能一句话答出**"删了会坏在哪"**;答不出 ⇒ 它就是该删的。 # 🔴 分清**两件性质完全不同的事**(本文件唯一的例外,且**不许写"兼容"**): # ① **活类别表 `ROLES_LIVE`** = "必须存在活会话"的类别 ⇒ **只许有当前在用的前缀**。 # 多一个过时前缀 ⇒ 每轮必报"类别缺失"(假红)+ 给旧名续命(就是用户说的"大杂烩")。 # ② **扫描面 `SESS_PREFIXES`** = "要捞进视野"的前缀 ⇒ 过时前缀**留在这里**。 # 性质是**历史行解析正确性**(⛔ 不是"兼容"):宿主库 `sessions` 表里有旧标题的行, # 扫不到 ⇒ 本闸门对它们**静默失明**("读到了却不说"是该机制的既有红线)。 # 🔴 **实测判据(2026-10-05 当场跑 `board.py::_role_of_title()` 验的,⛔ 非推理)**: # 删 `_PFX` 里 `协作`/`协作目标`/`任务会话` 三条 ⇒ # `[协作]-[手机接入]-N9 复测` / `[协作]N9 派活 · …` / `[协作目标]-xxx` / `[任务会话]-…` # **全部解析成 `''`** ⇒ 看板画不出、派活漏管,**不可逆**。 # ⇒ 这就是"删了会坏在哪"的答案。**留着有据,不是人情。** # ⚠️ 与 `collabd.py::parse_session_name()` / `board.py::_role_of_title()` 的映射口径对齐: # worker 侧新建=`执行`(唯一在用)。 ROLES_LIVE = [("执行", "[执行]")] # 🔴 扫描面 = 当前活类别 + **历史前缀**(⛔ 语义是「历史行解析正确性」,**不是"兼容"**)。 # ⛔⛔ **这三个历史前缀不许删**:删任一条 ⇒ 存量旧标题行**扫不到** ⇒ 本闸门静默失明。 # ⚠️ 它们**不参与"活类别"判定**(不判"必须有活的")—— 判了就是给旧名续命。 SESS_PREFIXES = [p for _, p in ROLES_LIVE] + ["[协作]", "[协作目标]", "[任务会话]"] # 🔴🔴 2026-10-03 口径后**已退役**的角色前缀 ⇒ **只扫、不判活**(⛔ 不是"兼容",是"还在库里"): # ⛔ 不许放回 `ROLES_LIVE` —— 唤醒/跟进**整套退役**(见 SKILL.md 文首口径块), # 放回去 ⇒ 每轮必报"唤醒/跟进 类别缺失"(惩罚的正是"按要求删掉的东西")。 SESS_PREFIXES += ["[唤醒]", "[跟进]"] # 🔴 2026-10-03 口径:唤醒/跟进**整套退役**(见 SKILL.md 文首口径块) # ⇒ ⑦ 「周期钟必须存在」这条旧判据与现口径**互斥**:按旧判据每轮必报 fail,而它惩罚的 # 正是"已经按用户要求删掉的东西"。现行「处理」这条腿的载体=常驻 `--supervise`(⑩ 判它)。 # ⇒ 本条改为**反向判据**:在册的退役周期钟**必须已 PAUSED**(还在 ACTIVE ⇒ 才报问题)。 ROLES_CLOCK = [("唤醒", "[唤醒]"), ("跟进", "[跟进]")] # ⚠️ 仅用于"退役后应处于 PAUSED"的反向检查 FLASH_HINT = ("flash",) # 技能名形态(与「技能库引用体检」同源);⚠️ 只认这几个前缀,别把业务线名当技能 SKILL_NAME_RE = re.compile(r"\b((?:dsh|agent|session|workbuddy|multi|third-party|stage|product|content-source)" r"[a-z0-9]*(?:-[a-z0-9]+)+)\b") # 「指向技能」的上下文(⛔ 只在这类上下文里判,否则会把业务线名 / localStorage 键当技能名) # # 🔴 2026-10-02 收紧(首版太宽、当场三条误报,全是"把不属于技能的东西当技能名"): # 旧版还含 `^[`「]?\s*$`("整个字符串就是一个技能名形态")⇒ 于是: # · `lock-guard-hook.py` 里的字符串 `"dsh-server-docs"`(那是**文档库目录名**)被报成悬空技能; # · `session-log-guard.py` 里的 `"session-log-guard"`(那是**它自己的脚本名**)同样被报; # · `wb-result-hook.py` 里的 `os.path.join(_base,"skills","multi-session-collab",…)` # (那是**故意留的旧位置回退候选**,代码注释已标明)也被报。 # ⇒ 本项要问的其实只有一件事:**钩子在叫 AI「去加载」某个技能时,那个技能还在不在** # ⇒ 判据就只认「加载 / Skill 工具」这一类**动作语境**,⛔ 不认"字符串长得像技能名"。 SKILL_CTX_RE = re.compile(r"(?:Skill\s*工具(?:加载|来加载)|调用\s*Skill|先加载|加载|load)\s*[`「]?\s*$") CHECKS = [] def rec(cid, group, level, title, detail=""): CHECKS.append({"id": cid, "group": group, "level": level, "title": title, "detail": detail}) # ── 工具 ─────────────────────────────────────────────────────────────── def _ro_conn(): c = sqlite3.connect("file:%s?mode=ro" % DB.replace("\\", "/"), uri=True, timeout=8) c.execute("PRAGMA busy_timeout=6000") return c def _norm_cwd(p): """与宿主去重键同源:`path.trim().toLowerCase()`(⛔ 别自作聪明归一成 basename)。""" return str(p or "").replace("\\", "/").strip().lower() def _safe_list(v): if v is None: return [] if isinstance(v, (list, tuple)): return [str(x) for x in v] s = str(v).strip() if s.startswith("["): try: r = json.loads(s) return [str(x) for x in r] if isinstance(r, list) else [s] except Exception: return [s] return [x.strip() for x in s.split(",") if x.strip()] def _string_literals(src): """只取**运行时会用到的字符串字面量** ⇒ 排除注释与 docstring。 为什么要这么讲究:钩子脚本头部**注释里大量出现旧技能名**(讲历史/讲事故), 那是**合规保留**的(改了就是篡改取证记录)。若连注释一起扫 ⇒ 每次都报一堆假阳性 ⇒ 真问题被淹掉。 """ out = [] try: tree = ast.parse(src) except Exception: return out doc_ids = set() for node in ast.walk(tree): if isinstance(node, (ast.Module, ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)): body = getattr(node, "body", []) if (body and isinstance(body[0], ast.Expr) and isinstance(body[0].value, ast.Constant) and isinstance(body[0].value.value, str)): doc_ids.add(id(body[0].value)) for node in ast.walk(tree): if isinstance(node, ast.Constant) and isinstance(node.value, str) and id(node) not in doc_ids: out.append(node.value) return out def _dangling_skills_in_text(txts): """在一组字符串里找「指向不存在的技能名」。返回 {name: 片段}。""" out = {} for t in txts: for m in SKILL_NAME_RE.finditer(t): nm = m.group(1) if nm in EXISTING_SKILLS: continue head = t[max(0, m.start() - 14):m.start()] if not SKILL_CTX_RE.search(head): continue out.setdefault(nm, t[max(0, m.start() - 20):m.end() + 20].replace("\n", " ").strip()) return out def _skill_names(): """**本机可见的全部技能名** = 全局技能根 ∪ 工作区技能根。 🔴🔴 2026-10-06 修 **只查全局根 ⇒ 工作区自带技能被判"悬空"(假红,且每轮刷)**。 实测(`vibe-product`):其 `MEMORY.md` 里写「`product-planning/`(工作区根)」, 而 `product-planning` **只在** `/.workbuddy/skills/` 里 —— 全局库 `E:/ProgramData/.workbuddy/skills/` 下**没有它**(实测 `ls` = No such file)。 ⇒ 旧判据(只 `listdir(SKILLS)`)每轮报「工作区记忆里的技能指针悬空」, 而它惩罚的正是「**工作区自己装的技能**」这种**完全合法**的形态。 ⛔ 判据要问的是「**这个名字在本机取不取得到**」,⛔ **不是**「全局库目录里有没有」。 本机 `DSH_SKILLS_ROOT` 未设时 `SKILLS` 落在 `CODEBUDDY_CONFIG_DIR/skills`(全局), 本身就**看不见**工作区那一层 ⇒ 漏判是必然而非偶然。 """ out = set() for root in (SKILLS, os.path.join(WS, ".workbuddy", "skills")): try: if root and os.path.isdir(root): out |= {n for n in os.listdir(root) if os.path.isdir(os.path.join(root, n))} except Exception: pass return out EXISTING_SKILLS = _skill_names() # ── A. 机制装没装好 ───────────────────────────────────────────────────── def check_hooks(): if not os.path.isfile(SETTINGS): rec("hook_reg", "A", "fail", "读不到 hooks 配置", "settings.json 不在:%s" % SETTINGS) return [], None try: hk = (json.load(io.open(SETTINGS, encoding="utf-8")) or {}).get("hooks") or {} except Exception as e: rec("hook_reg", "A", "fail", "hooks 配置解析失败", str(e)[:140]) return [], None cmds = [(ev, str(h.get("command") or "")) for ev, arr in hk.items() for blk in (arr or []) for h in ((blk or {}).get("hooks") or [])] # ⚠️ `k[1]` 是**一组**可接受的文件名 ⇒ 命中任一即算在册(独立接线 或 合并入口)。 miss = [k for k in KEY_HOOKS if not any(k[0] == ev and any(n in c for n in k[1]) for ev, c in cmds)] if miss: rec("hook_reg", "A", "fail", "关键钩子不在册(该机制现在不会运行)", "、".join("%s(%s)" % ("/".join(k[1]), k[0]) for k in miss)) else: rec("hook_reg", "A", "ok", "关键钩子在册", "共 %d 条钩子注册" % len(cmds)) bad_path = [] for ev, c in cmds: m = re.search(r'([A-Za-z]:[\\/][^"\']*?\.(?:py|sh|mjs))', c) if m and not os.path.exists(m.group(1)): bad_path.append("%s(%s)" % (os.path.basename(m.group(1)), ev)) if bad_path: rec("hook_path", "A", "fail", "钩子脚本路径不存在 ⇒ 静默失效", "、".join(sorted(set(bad_path))[:5])) else: rec("hook_path", "A", "ok", "钩子脚本路径均存在", "") return cmds, hk def check_hook_skillnames(cmds): """🔴 本体检最有价值的一项:钩子**注入文本**里引用的技能名,现在还存在吗? 真因(2026-10-02 实测):`skill-load-guard.py` 注入「先调用 Skill 工具加载 `dsh-decision-method`」, 而该技能 **2026-09-28 已合并退役**(现名 `dsh-decision`)⇒ 命中后让人去拿一份不存在的东西。 """ dangling = {} scanned = 0 for ev, c in cmds: m = re.search(r'([A-Za-z]:[\\/][^"\']*?\.py)', c) if not m: continue p = m.group(1) if not os.path.isfile(p): continue try: src = io.open(p, encoding="utf-8", errors="replace").read() except Exception: continue scanned += 1 for nm, frag in _dangling_skills_in_text(_string_literals(src)).items(): dangling.setdefault("%s → %s" % (os.path.basename(p), nm), frag) if dangling: rec("hook_skillname", "A", "fail", "钩子注入里指向**已不存在的技能名**(命中即空转)", ";".join("%s" % k for k in sorted(dangling)[:4])) else: rec("hook_skillname", "A", "ok", "钩子注入引用的技能名均存在", "扫了 %d 份钩子脚本" % scanned) def check_gate_logs(): stale = [] rows = [] for tag, p in GATE_LOGS: try: with io.open(p, "rb") as f: f.seek(0, 2) n = min(8192, f.tell()) f.seek(-n, 2) r = [x for x in f.read().decode("utf-8", "replace").split("\n") if x.strip()] if not r: stale.append(tag + "(空)") continue last = r[-1][:16] rows.append("%s=%s" % (tag, last)) age = time.time() - os.path.getmtime(p) if age > 7 * 24 * 3600: stale.append("%s(停 %.0f 天)" % (tag, age / 86400.0)) except Exception: stale.append(tag + "(缺)") if stale: rec("gate_fresh", "A", "warn", "部分闸门日志陈旧 / 缺失(可能已掉线)", ";".join(stale) + "|" + " ".join(rows)) else: rec("gate_fresh", "A", "ok", "各闸门最近都被调用过", " ".join(rows)) # ── B. 规则载体同没同步 ───────────────────────────────────────────────── def check_memory_pointers(): """工作区 `MEMORY.md` 是**每轮注入**的那份 ⇒ 它里面的悬空指针会一直把 AI 引向不存在的东西。""" p = os.path.join(WS, ".workbuddy", "memory", "MEMORY.md") if not os.path.isfile(p): rec("mem_ptr", "B", "warn", "读不到工作区 MEMORY.md", "路径:%s" % p) return try: txt = io.open(p, encoding="utf-8", errors="replace").read() except Exception as e: rec("mem_ptr", "B", "warn", "读 MEMORY.md 失败", str(e)[:120]) return bad = {} for m in SKILL_NAME_RE.finditer(txt): nm = m.group(1) if nm in EXISTING_SKILLS: continue head = txt[max(0, m.start() - 14):m.start()] if not (SKILL_CTX_RE.search(head) or "技能" in head): continue bad.setdefault(nm, txt[max(0, m.start() - 26):m.end() + 12].replace("\n", " ").strip()) if bad: rec("mem_ptr", "B", "fail", "工作区记忆里的技能指针悬空(每轮注入 ⇒ 一直把人引错)", ";".join(sorted(bad)[:4])) else: rec("mem_ptr", "B", "ok", "工作区记忆的技能指针均有效", "") def check_snapshot_stale(): """常驻规则快照 vs 权威 `CODEBUDDY.md`:**内容**一不一致(快照是副本,权威单向)。 🔴🔴 2026-10-06 修 **判据从「比 mtime」改成「比内容」**(原判据是**假红**,已实测坐实): 原写法比 `getmtime(auth) - getmtime(snap) > 0.01 天` ⇒ 只要权威 CODEBUDDY.md **被 touch 过**(哪怕只是重排、挪归档、加一行注释)就报 「快照比权威旧 ⇒ 新规则没进快照」—— 而**快照内容可能一个字都没差**。 ⚠️ 实测(2026-10-06,ai1net):新旧快照 **md5 逐字节相同**(`f54e48d4…`)、`diff` **0 行**、 `resident-rules.py --check` 报「✅ 关键规则齐备」,而本条已**连续 4 天每轮报 fail**。 ⇒ 病根:**拿文件时间当内容判据**(时间只说明"谁最后被写过",不说明"内容差没差")。 ⇒ 改为**内容口径**,并**复用唯一事实源**:直接调 `dsh-local-env` 的 `resident-rules.py --check --goal `(它本来就是干这个的) —— ⛔ **不在这里再抄一份"抽章节再比对"的逻辑**(抄一份 ⇒ 两处漂移 ⇒ 同族事故)。 rc=0 ⇒ ok | rc=1 ⇒ **真不一致**(fail)| 工具缺失/跑不起来 ⇒ warn(⛔ 不假装通过)。 """ snap = os.path.join(SKILLS, "dsh-local-env", "references", "dsh-env-bootstrap", "常驻规则-快照.md") auth = os.path.join(WS, "CODEBUDDY.md") if not os.path.isfile(auth): # ⚠️ 本区**本来就没有** CODEBUDDY.md(如 vibe-product)⇒ 这是**合法状态**,不是故障。 rec("snap_sync", "B", "warn", "本区无权威规则文件(`CODEBUDDY.md` 不存在 ⇒ 本项不适用)", auth) return if not os.path.isfile(snap): rec("snap_sync", "B", "warn", "常驻规则快照不存在", snap) return tool = os.path.join(SKILLS, "dsh-local-env", "references", "dsh-env-bootstrap", "resident-rules.py") if not os.path.isfile(tool): rec("snap_sync", "B", "warn", "**抽取器不在** ⇒ 判不了快照同不同步(⛔ 不算通过)", tool) return # 🔴🔴 真·内容口径:**让抽取器自己算一遍"快照应该长什么样"**,再与磁盘上的真快照比对。 # ⛔ 不能用 `resident-rules.py --check` —— 实测(2026-10-06 变异对照): # 它校验的是**目标 CODEBUDDY.md 里关键规则齐不齐**,**根本不含"与快照比对"**。 # 我中途误用它 ⇒ 把快照**截断到 400 字节**,它照样报 `✅ 关键规则齐备`(rc=0) # ⇒ 判据变 **恒绿**(比原来的 mtime 假红更坏)。 # ⚠️ 教训:**换判据必须重跑变异对照**;"换了个看起来更对的调用" ≠ 判据变强了。 # ⛔ 也不在这里自己抄一份"抽章节"逻辑(两处 ⇒ 漂移)。做法=**导入抽取器本体**, # 把它的 `SNAP` 常量**临时改指到临时文件**,调它自己的 `snapshot()` 取"应有内容", # 读完还原常量并删临时文件(⛔ 全程不碰真快照 —— 体检⛔ 不改资产)。 try: import importlib.util import tempfile _spec = importlib.util.spec_from_file_location("_rr_probe", tool) _mod = importlib.util.module_from_spec(_spec) _spec.loader.exec_module(_mod) # `__name__` ≠ `'__main__'` ⇒ 不会跑 main() _fd, _tmp = tempfile.mkstemp(suffix=".md") os.close(_fd) _orig = _mod.SNAP try: _mod.SNAP = _tmp _mod.snapshot(auth) # 用**它自己**的逻辑生成"应有内容" expected = io.open(_tmp, encoding="utf-8", errors="replace").read() finally: _mod.SNAP = _orig try: os.remove(_tmp) except OSError: pass except Exception as e: rec("snap_sync", "B", "warn", "**内容比对跑不起来** ⇒ 判不了快照同不同步(⛔ 不算通过)", str(e)[:130]) return try: actual = io.open(snap, encoding="utf-8", errors="replace").read() except Exception as e: rec("snap_sync", "B", "warn", "读快照失败", str(e)[:120]) return if actual == expected: rec("snap_sync", "B", "ok", "常驻规则快照与权威**逐字节一致**", "") return # 定位**第一处**差异,便于一眼看懂差在哪(⛔ 不倒全文) _a, _b = actual.split("\n"), expected.split("\n") _i = 0 while _i < min(len(_a), len(_b)) and _a[_i] == _b[_i]: _i += 1 _exp_line = (_b[_i][:60].strip() if _i < len(_b) else "(应有多出的行)") _act_line = (_a[_i][:60].strip() if _i < len(_a) else "(快照到此为止 ⇒ 缺内容)") rec("snap_sync", "B", "fail", "常驻规则快照与权威**内容不一致** ⇒ 新规则没进快照(或快照被改坏)", "第 %d 行起不同;快照实际=「%s」/应为=「%s」(快照 %d 行 / 应为 %d 行);" "重生成 ⇒ `python /references/dsh-env-bootstrap/resident-rules.py --snapshot`" % (_i + 1, _act_line, _exp_line, len(_a), len(_b))) # ── C. 编排在不在跑 ───────────────────────────────────────────────────── def _cwds_nearmiss(autos, ws_norm): """「与本工作区**几乎同名**、但字面不同」的排期 cwds ⇒ 会**裂成两组**。 ⚠️ 判据只认**近失配**两条(⛔ 不是"同父目录就算" —— 首版就栽在这儿): (a) **同父目录 + 名字只是 `-`/`_`/大小写之别** —— 如 `…/ai1net_dsh_server` ↔ `…/ai1net-dsh-server`;⇒ **去标点后逐字相同** 才算近失配(例:`ai1net-decision-laya` 与本工作区同父目录,但去标点后不同 ⇒ **是另一条线,⛔ 不是失配**) (b) **同名字、父目录不同** —— 如 `E:/…/ai1net-dsh-server` ↔ `D:/…/ai1net-dsh-server` (多半是从别的机器抄来的排期) ⇒ 去重键=`path.trim().toLowerCase()`(宿主原样)⇒ 这两种各裂一组,且**自我强化** (越裂越不像,之后再也归不回来)⇒ 必须**在建的时候**就报出来。 ⇒ 纯函数:只吃数据,便于用合成样本做红绿对照(⛔ 不靠"实跑一次看着对")。 """ def _key(p): return re.sub(r"[^a-z0-9]", "", p) base = os.path.basename(ws_norm.rstrip("/")) parent = os.path.dirname(ws_norm.rstrip("/")) out = [] for a in autos: for x in _safe_list(a.get("cwds")): s = str(x or "").strip() if not s: continue n = _norm_cwd(s).rstrip("/") if n == ws_norm: continue same_parent = (n.rsplit("/", 1)[0] if "/" in n else "") == parent name = n.rsplit("/", 1)[-1] if (same_parent and _key(name) == _key(base)) or (not same_parent and name == base): out.append("%s → %s" % ((a.get("name") or "")[:24], s)) return out def _sat_epoch(sat): """`scheduled_at` → epoch 秒。⚠️ 实测只到**分钟**(`2026-10-02T14:26`)⇒ 两种格式都试; 解析不出来返回 `None`(⛔ 调用方据此走"不可核对",**不猜**)。""" s = str(sat or "") for fmt in ("%Y-%m-%dT%H:%M:%S", "%Y-%m-%dT%H:%M"): try: return time.mktime(time.strptime(s[:len(time.strftime(fmt))], fmt)) except Exception: continue return None def _once_zombie(mine, ran, grace_min=30.0): """`once` 排期、已无下次触发、却**从未运行过** ⇒ 那条活会**静默消失**。 🔴 两条判据要点(都是实测校正出来的,⛔ 别想当然): (a) **"跑完了"不是问题、"从未跑却已失效"才是** —— `once` 到点跑完即被消耗,属**正常痕迹**; 把两者混报 ⇒ 首版拿 `last_run_at` 判,把 **18 条正常痕迹**当成 18 个故障。 (b) **"有没有跑过"必须查 `automation_runs`,⛔ 不能查 `automations.last_run_at`** —— 该字段宿主**根本不写**(实测:本会话自己那条排期**明明在跑**,值仍是 `None`)。 🔴 2026-10-02 修(S11 · 与 `collabd.health()` 同一病灶):本机 once 排期**不到点即时触发** 而是主机轮询**补跑**(`runKind=missed`,实测延迟 0~11 分钟)⇒ **刚过点不足 `grace_min` 的 那条不算「从未运行」**,只是**还在补跑窗口里**。⛔ 无 `scheduled_at` 或解析不出 ⇒ 不判 fail (不可核对 ≠ 有故障),归入第三返回值。 ⇒ 纯函数:`ran` 传 `None` 表示"读不到运行记录" ⇒ **只报 warn,⛔ 不报 fail**。 返回 (never_ran, consumed, unverifiable, in_grace)。 """ never, consumed, unver, in_grace = [], 0, 0, 0 now = time.time() for a in mine: if a.get("schedule_type") != "once" or a.get("next_run_at"): continue if ran is None: unver += 1 continue if a.get("id") in ran: consumed += 1 continue # 🔴 到点后仍可能补跑 ⇒ 未超容差先归入"窗口内",⛔ 不当故障 t = _sat_epoch(a.get("scheduled_at")) if t is not None and (now - t) / 60.0 <= float(grace_min): in_grace += 1 continue never.append((a.get("name") or "")[:30]) return never, consumed, unver, in_grace def check_orchestration(): if not os.path.isfile(DB): rec("clock", "C", "fail", "读不到宿主库", DB) return try: c = _ro_conn() autos = [{"id": r[0], "name": r[1], "schedule_type": r[2], "next_run_at": r[3], "model_id": r[4], "model_is_thinking": r[5], "cwds": r[6], "scheduled_at": r[7], "status": r[8]} for r in c.execute( "select id,name,schedule_type,next_run_at,model_id,model_is_thinking,cwds," " scheduled_at,status " "from automations where deleted_at is null")] sess = list(c.execute( "select id,title,status from sessions where " + " or ".join(["title like '%s%%'" % p for p in SESS_PREFIXES]))) except Exception as e: rec("clock", "C", "fail", "读宿主库失败", str(e)[:140]) return ws_norm = _norm_cwd(WS) mine = [a for a in autos if any(_norm_cwd(x) == ws_norm for x in _safe_list(a["cwds"]))] # ⑦ 已退役角色的周期钟(**反向判据** —— 2026-10-03 口径后改,⛔ 别再要求"必须存在") # 旧判据("唤醒/跟进周期钟必须都在册")与 10-03 口径**互斥**:那两类会话已整套退役, # 要求它们存在=**惩罚按用户要求删对了的东西** ⇒ 每轮必报 fail(判据与设计目标互斥)。 # ✅ 现判据:**在册的退役周期钟必须已 PAUSED**;还在 ACTIVE ⇒ 才说明退役没做干净。 retire = [a for a in mine if a["schedule_type"] == "recurring" and any((a["name"] or "").startswith(pre) for _lb, pre in ROLES_CLOCK)] not_paused = [a["name"] for a in retire if (a.get("status") or "") == "ACTIVE"] if not_paused: rec("clock", "C", "fail", "已退役角色的周期钟仍在 ACTIVE ⇒ 退役没做干净", "、".join(not_paused[:3])) else: rec("clock", "C", "ok", "已退役角色的周期钟已全部 PAUSED(或本就无在册)", "在册 %d 条,全部非 ACTIVE" % len(retire)) # ⑧ 模型一致性(🔴🔴 2026-10-05 P0-80 重写:**扫描面从 `recurring` 扩到全部排期**) # ⚠️ 旧写法只扫 `recurring` ⇒ 而**机制新建的排期 `schedule_type` 恒为 `once`**(见 P0-65) # ⇒ 一条都不进扫描面 ⇒ 该判据**恒判 ok**(同 P0-76 ①/P0-77 ③ 的"闸门看不见新东西")。 # ⚠️ 且旧判据只问"flash + thinking=0 会不会被服务端拒",**⛔ 没问"跟主会话是不是同一个"** # —— 这才是用户真正遇到的问题(P0-80:机制写死 `space-bunny`,主会话是 `deepseek-v4.1-flash`)。 # ✅ 现判据两条腿: # ① **可用性**(旧):flash 类 + thinking=0 ⇒ 服务端拒; # ② **一致性**(新):本区排期的 `model_id` 必须与本区主会话 `model` 同值(否则报 fail)。 _recur = [a for a in mine if a["schedule_type"] == "recurring"] try: _c2 = sqlite3.connect("file:%s?mode=ro" % DB, uri=True, timeout=10) _c2.execute("PRAGMA busy_timeout=5000") _mr = _c2.execute( "select model from sessions " "where (is_background_automation is null or is_background_automation <> 1) " " and model is not null and model <> '' " " and replace(cwd,'\\\\','/') like ? order by created_at desc limit 1", ("%" + ws_norm + "%",)).fetchone() _c2.close() _main_model = (_mr[0] if _mr else "") or "" except Exception: _main_model = "" # 可用性:本区**全部**排期(含 once) _allmine = [a for a in mine] deaf = ["%s(%s)" % ((a["name"] or "")[:26], a["model_id"]) for a in _allmine if not a["model_is_thinking"] and any(h in (a["model_id"] or "").lower() for h in FLASH_HINT) and a["schedule_type"] == "recurring"] # ⚠️ 只有 recurring 会被服务端审模型 if deaf: rec("model", "C", "fail", "周期排期会被服务端拒(静默失效)", "模型不支持关思考却传 thinking=0:%s" % "、".join(deaf)) else: rec("model", "C", "ok", "周期排期模型可用", "本区 recurring %d 条" % len(_recur)) # 一致性:机制建的排期(once)模型须与本区主会话同值 if not _main_model: rec("model_same", "C", "warn", "取不到本区主会话模型 ⇒ 无法核对一致性(P0-80)", "sessions 里没有本区的人开会话") else: _mech = [a for a in _allmine if (a.get("name") or "").startswith(("[执行]", "[检查]"))] _diff = ["%s(%s)" % ((a["name"] or "")[:26], a["model_id"]) for a in _mech if (a["model_id"] or "") != _main_model] if _diff: rec("model_same", "C", "fail", "机制建的排期模型 ≠ 本区主会话模型(P0-80)", "主会话=%s;不一致 %d 条:%s" % (_main_model, len(_diff), "、".join(_diff[:3]))) else: rec("model_same", "C", "ok", "机制排期模型与本区主会话一致(P0-80)", "主会话=%s,机制排期 %d 条全同" % (_main_model, len(_mech))) # ⑨ cwds 归属同形(错一字面 ⇒ 裂组且自我强化) off = _cwds_nearmiss(autos, ws_norm) if off: rec("cwd", "C", "fail", "`cwds` 与本工作区**近失配**(差一个字符就裂成两组)", ";".join(off[:3])) else: rec("cwd", "C", "ok", "`cwds` 归属同形", "本工作区的排期都写在同一条路径上") # ⑩ 投递心跳 hbs = [WS + "/.workbuddy/collab/logs/supervise-heartbeat.json", WS + "/.workbuddy/collab/supervise-heartbeat.json"] age = None for h in hbs: if os.path.isfile(h): age = time.time() - os.path.getmtime(h) break if age is None: rec("deliver", "C", "warn", "投递(常驻)未见心跳", "⛔ 不等于它一定没跑;载体=专用容器会话") elif age > 900: rec("deliver", "C", "fail", "投递心跳陈旧 ⇒ 常驻可能已掉线", "最后心跳 %.0f 分钟前" % (age / 60)) else: rec("deliver", "C", "ok", "投递心跳新鲜", "%.0f 秒前" % age) # ⑪ 活会话 # ⑪ 活会话(🔴 只判 `ROLES_LIVE`=**当前在用的前缀**;历史前缀⛔ 不判"必须有活的") live = {lb: [s for s in sess if (s[1] or "").startswith(pre) and (s[2] or "") == "working"] for lb, pre in ROLES_LIVE} dead = [lb for lb, _ in ROLES_LIVE if not live.get(lb)] if dead: rec("live", "C", "warn", "此刻无活会话(按需创建属正常;主会话开工阶段须建齐)", "、".join(dead)) else: rec("live", "C", "ok", "在用类别的会话均有活的", ";".join("%s×%d" % (lb, len(live[lb])) for lb, _ in ROLES_LIVE)) # ⑫ 死排期:`once` 已无下次触发、却**从未运行过** ⇒ 活静默消失 try: ran = {r[0] for r in c.execute( "select distinct automation_id from automation_runs limit 5000")} except Exception: ran = None never, consumed, unver, in_grace = _once_zombie(mine, ran) if never: rec("zombie", "C", "fail", "一次性排期**从未运行就失效**(那条活会静默消失)", "共 %d 条:%s" % (len(never), "、".join(never[:4]))) elif unver: rec("zombie", "C", "warn", "无法核对一次性排期是否运行过", "读不到 `automation_runs`(⛔ 不等于它们没跑);本线 %d 条 once+无下次触发" % unver) else: rec("zombie", "C", "ok", "无「从未运行」的一次性排期", "本线 %d 条已跑完的一次性排期(正常痕迹,不计问题)%s" % (consumed, (";%d 条刚到点、仍在补跑窗口内(⛔ 不是哑火)" % in_grace) if in_grace else "")) # ⑬ 🔴🔴 **旧前缀不许在「活的语境」里出现**(2026-10-05 用户定案) # 用户原话:「**兼容个毛线,今天兼容一个明天兼容一个 过不了一周就成大杂烩了**」。 # ⇒ 本条把「不许兼容」从**口头要求**变成**机制判据**:任何人在活件里再写旧前缀 # ⇒ 本闸门当场报红(⛔ 不靠"下次记得")。 # 🔴 判据只扫**能产生/承载新名字的两处**(⛔ 不全文扫 —— 注释与 `references/*.md` # 里的考古记录是**合法保留**的,扫它们就是误红): # ① 活排期名(`automations.name`,未软删)—— 是旧前缀 ⇒ "今天又长出来一条" # ② 活会话标题(`sessions.title`)—— 同上 # ⚠️ **不扫**:注释 / docstring / 考古段 / 软删的历史行 # —— 那些是"记录过去发生过什么",⛔ 不是"现在还在用"(混为一谈即是误红)。 # ✅ **变异对照**(2026-10-05 实跑,证明非恒绿非恒红): # 全 `[执行]` ⇒ ok;混 1 条 `[协作]` ⇒ fail;混 1 条 `[任务会话]` ⇒ fail; # 历史行但**不以旧前缀开头**(如 `接续:…`)⇒ ok。 _OLD_WORKER = ("[协作]", "[协作目标]", "[任务会话]") _bad_sched, _bad_sess = [], [] for _r in (mine or []): _nm2 = str((_r or {}).get("name") or "") if any(_nm2.startswith(_p) for _p in _OLD_WORKER): _bad_sched.append(_nm2[:44]) for _s in (sess or []): _t2 = str(_s[1] or "") if any(_t2.startswith(_p) for _p in _OLD_WORKER): _bad_sess.append(_t2[:44]) if _bad_sched or _bad_sess: _det = [] if _bad_sched: _det.append("活排期 %d 条:%s" % (len(_bad_sched), "、".join(_bad_sched[:3]))) if _bad_sess: _det.append("活会话 %d 条:%s" % (len(_bad_sess), "、".join(_bad_sess[:3]))) rec("old_prefix", "C", "fail", "旧前缀又长出来了(新建一律 `[执行]`)", ";".join(_det)) else: rec("old_prefix", "C", "ok", "活排期/活会话里无旧前缀(新建一律 `[执行]`)", "⛔ 历史行不在此列:它们是**解析正确性**,删了看板画不出(⛔ 不叫「兼容」)") # ── 收口 ─────────────────────────────────────────────────────────────── def finish(): fails = [c for c in CHECKS if c["level"] == "fail"] warns = [c for c in CHECKS if c["level"] == "warn"] verdict = "fail" if fails else ("warn" if warns else "ok") summary = {"ok": "会话规则机制齐备且生效", "warn": "配置在册,运行未齐(见 warn 项)", "fail": "规则机制有硬缺口(见 fail 项)"}[verdict] state = {"ts": time.strftime("%Y-%m-%d %H:%M:%S"), "ts_epoch": int(time.time()), "ws": WS, "verdict": verdict, "summary": summary, "counts": {"fail": len(fails), "warn": len(warns), "ok": len(CHECKS) - len(fails) - len(warns)}, "checks": CHECKS} try: os.makedirs(os.path.dirname(OUT), exist_ok=True) tmp = OUT + ".tmp" with io.open(tmp, "w", encoding="utf-8", newline="\n") as f: f.write(json.dumps(state, ensure_ascii=False, indent=2)) os.replace(tmp, OUT) # 原子替换(本机文件锁会卡死 ⇒ 见 MEMORY) except Exception as e: state["write_error"] = str(e)[:160] if AS_JSON: sys.stdout.buffer.write((json.dumps(state, ensure_ascii=False, indent=2) + "\n").encode("utf-8")) else: icon = {"ok": "✅", "warn": "⚠️", "fail": "🔴"}[verdict] buf = ["%s [会话规则] %s(fail %d / warn %d / ok %d)" % (icon, summary, len(fails), len(warns), state["counts"]["ok"])] show = CHECKS if SHOW_ALL else [c for c in CHECKS if c["level"] != "ok"] for c in show: m = {"ok": "✓", "warn": "⚠", "fail": "✗"}[c["level"]] buf.append(" %s [%s] %s%s" % (m, c["group"], c["title"], (" —— " + c["detail"]) if c["detail"] else "")) if not SHOW_ALL and not show: buf.append(" (九项全过,明细见状态标记)") buf.append(" · 状态标记 → %s" % OUT) sys.stdout.buffer.write(("\n".join(buf) + "\n").encode("utf-8")) sys.stdout.flush() return 0 if verdict == "ok" else 1 def main(): cmds, _ = check_hooks() if cmds: check_hook_skillnames(cmds) check_gate_logs() check_memory_pointers() check_snapshot_stale() check_orchestration() return finish() if __name__ == "__main__": try: sys.exit(main()) except Exception as e: try: sys.stdout.buffer.write(("⚠️ [会话规则] 体检自身异常(⛔ 不影响开工):%s\n" % str(e)[:200]).encode("utf-8")) except Exception: pass sys.exit(0)