Files
workbuddy_skills/session-mechanism/scripts/mut_run.py
T
admin 19101acd65 init: workbuddy_skills 重建,仅收录 session-mechanism
- 按用户指示清空原有 25 技能内容,只提交 session-mechanism(57 文件)
- 附 .gitignore(产物 + 本机凭据)
- 令牌明文已脱敏(历史 .neodata_token 与 pitfalls 引用均不入库)
- 本提交为孤儿提交(父提交为空),历史自此重新开始
2026-10-05 14:13:24 +08:00

261 lines
13 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# 🔴 变异验证 · 2026-10-04 · 会话级忙判据两个新缺陷
#
# ⚠️ 纪律(来自 references/00-动手前必过.md ③④):
# ① **先验基线绿**:基线 PASS 17/FAIL 0(忙判据那条用例)。
# ② **锚在唯一位置**:`def _session_busy_conv` / `def _conv_ts` 唯一定位,⛔ 不锚"文本出现过"。
# ③ **每个变异先验「被检对象真被改了」**:比对长度+md5,⛔ 没真改就算**失败**(不许跳过分母)。
# ④ ⛔ **跳过的变异一律计入失败**。
#
# 用法:python mut_run.py
import hashlib
import importlib.util
import io
import os
import re
import shutil
import subprocess
import sys
import tempfile
HERE = os.path.dirname(os.path.abspath(__file__))
COLLABD = os.path.join(HERE, "collabd.py")
SELFTEST = os.path.join(HERE, "selftest.py")
PY = sys.executable
def md5(p):
return hashlib.md5(open(p, "rb").read()).hexdigest()
# ── 变异体定义 ────────────────────────────────────────────────────────────
# (编号, 目标判据, 人类可读描述, 变异函数)
def mut_order_revert(src):
"""M1 针对 ⑭:**丢掉 key 的时间戳分量** = 退回"纯遍历顺序"⇒ 昨天盖掉今天。
⚠️ 踩坑记录(这条本身就是变异验证的教训):第一版写成
`key[1] >= last_t[1]`(只把比较从元组换成 seq)⇒ **等价变异、恒绿**。
原因:样本里每个文件内部时间戳本来就升序,**任何"取最后一条"的写法都等价**。
⇒ 变异必须动**语义**(时间戳参与/不参与),⛔ 不能只动写法。
"""
old = " key = (ts if ts is not None else 0.0, seq)"
new = " key = (0.0, seq) # MUTANT: 丢掉时间戳 = 退回纯遍历顺序"
assert old in src, "锚点1 未命中"
return src.replace(old, new, 1)
def mut_freshness_off(src):
"""M2 针对 ⑮:把新鲜度闸阈值放大到"永不过期"(缺陷原样)。"""
old = " if last_t[0] > 0 and (time.time() - last_t[0]) > SRSM_FRESH:"
new = " if False: # MUTANT: 关掉新鲜度闸"
assert old in src, "锚点2 未命中"
return src.replace(old, new, 1)
def mut_tz_local(src):
"""M3 针对 ⑰:时间戳改按**本地时区**解析(mktime)⇒ 东八区差 8h ⇒ 恒判过期。"""
old = " t = float(calendar.timegm((int(m.group(\"Y\")), int(m.group(\"M\")), int(m.group(\"D\")),\n int(m.group(\"h\")), int(m.group(\"m\")), int(m.group(\"s\")), 0, 0, 0)))"
new = (" import calendar as _c # MUTANT: 按本地时区解析\n"
" _st = (int(m.group(\"Y\")), int(m.group(\"M\")), int(m.group(\"D\")),\n"
" int(m.group(\"h\")), int(m.group(\"m\")), int(m.group(\"s\")), 0, 0, -1)\n"
" t = float(time.mktime(_st))")
assert old in src, "锚点3 未命中"
return src.replace(old, new, 1)
def mut_tz_double_offset(src):
"""M9 针对 ⑰:额外+8h 偏移(另一种时区错法)⇒ 「2 分钟前」变「8 小时后」⇒ 恒不过期。"""
old = " ms = m.group(\"ms\")\n return t + (float(\"0.\" + ms) if ms else 0.0)"
new = (" ms = m.group(\"ms\")\n"
" return t + 28800.0 + (float(\"0.\" + ms) if ms else 0.0) # MUTANT: 多加 8h")
assert old in src, "锚点9 未命中"
return src.replace(old, new, 1)
def mut_freshness_too_hard(src):
"""M4 针对 ⑯:新鲜度闸写太狠(窗口=0)⇒ 连正在跑的也判 idle(反向回归)。"""
old = "(time.time() - last_t[0]) > SRSM_FRESH:"
new = "(time.time() - last_t[0]) > 0: # MUTANT: 闸写太狠"
assert old in src, "锚点4 未命中"
return src.replace(old, new, 1)
def mut_ts_none(src):
"""M5 针对 ⑰:时间戳一律返回 None(解析失败)⇒ 保守当 busy ⇒ ⑯ 报红。"""
old = " m = _CONV_TS.match(ln)"
new = " return None # MUTANT: 一律解析失败\n m = _CONV_TS.match(ln)"
assert old in src, "锚点5 未命中"
return src.replace(old, new, 1)
def mut_busy_always(src):
"""M8 针对 ⑯:`idle` 也被当成 busy(陈旧闸写太狠的另一种写法)⇒ ⑯/⑭ 报红。"""
old = " if last_to in (\"working\", \"planning\"):"
new = " if True: # MUTANT: 不分状态一律当在跑"
assert old in src, "锚点8 未命中"
return src.replace(old, new, 1)
def mut_order_reverse_files(src):
"""M6 针对 ⑭:把 `_conv_log` 的返回顺序反过来([昨天, 今天])⇒ 顺序依赖立刻现形。"""
old = " for cand in (\"%s.log\" % sid, \"%s.log\" % str(sid).replace(\"-\", \"\")):"
new = " for cand in (\"%s.log\" % sid,): # MUTANT: 去掉无连字符形态"
assert old in src, "锚点6 未命中"
s = src.replace(old, new, 1)
return s.replace(" for d in (time.strftime(\"%Y-%m-%d\"),\n time.strftime(\"%Y-%m-%d\", time.localtime(time.time() - 86400))):",
" for d in (time.strftime(\"%Y-%m-%d\", time.localtime(time.time() - 86400)),\n time.strftime(\"%Y-%m-%d\")):", 1)
def mut_freshness_only_first_file(src):
"""M7 针对 ⑮:陈旧判定只在**第一个文件**生效 ⇒ 单文件陈旧样本仍判 busy(⑮ 报红)。"""
old = " if last_t[0] > 0 and (time.time() - last_t[0]) > SRSM_FRESH:"
new = " if last_t[0] > 0 and (time.time() - last_t[0]) > SRSM_FRESH and len(_conv_log(sid)) > 1: # MUTANT: 陈旧判定只在多文件时生效"
assert old in src, "锚点7 未命中"
return src.replace(old, new, 1)
MUTS = [
("M1", "⑭跨日取最新", mut_order_revert),
("M2", "⑮新鲜度闸", mut_freshness_off),
("M3", "⑰UTC时区", mut_tz_local),
("M4", "⑯反向回归", mut_freshness_too_hard),
("M5", "⑰时间戳解析", mut_ts_none),
("M6", "⑭跨日取最新", mut_order_reverse_files),
("M7", "⑮新鲜度闸", mut_freshness_only_first_file),
("M8", "⑯反向回归", mut_busy_always),
("M9", "⑰UTC时区", mut_tz_double_offset),
]
def run_selftest(workdir):
"""⚠️ **必须用工作目录里的那份 `selftest.py`**,⛔ 不能用技能目录的原版绝对路径。
🔴🔴 2026-10-04 当场栽过(写在这里当护栏,别删):`selftest.py` 里
`HERE = Path(__file__).resolve().parent`,而 `imp()` 加载的是 **`HERE / "collabd.py"`**
⇒ 只改 `cwd` 没用、只传原版绝对路径也没用 ⇒ **测的一直是原版 ⇒ 恒绿 ⇒ 7/7 全是假象**。
⇒ 判据:`HERE` 指向 `workdir` 时才算数(下面 `assert` 硬查)。
"""
st = os.path.join(workdir, "selftest.py")
assert os.path.abspath(os.path.dirname(st)) == os.path.abspath(workdir), "测的不是工作目录里的 selftest"
p = subprocess.run([PY, st, "-k", "忙判据"],
capture_output=True, text=True, encoding="utf-8",
errors="replace", cwd=workdir, timeout=600)
out = p.stdout + p.stderr
# 🔴 护栏:确认真加载的是 workdir 里那份(防止以后再有人改回绝对路径)
assert workdir.replace("\\", "/") in out.replace("\\", "/"), "输出里看不到工作目录 ⇒ 加载的不是变异体"
return out
def verdict(out):
"""返回 (是否报红, 是否真找到用例)。⛔ 没找到用例 = 验证作废,不算绿。"""
ran = ("忙判据" in out) and ("投递忙判据" in out)
if not ran:
return None, False
# 该用例整体 FAIL ⇔ 汇总行 PASS 0/ FAIL >=1,或用例行前是 ✗
m = re.search(r"合计:PASS (\d+) / FAIL (\d+)", out)
if not m:
return None, True
return (int(m.group(2)) >= 1), True
def main():
print("=" * 74)
print("变异验证 · 会话级忙判据(每个新判据 ≥2 个变异体,⛔ 跳过计入失败)")
print("=" * 74)
orig_src = open(COLLABD, "rb").read().decode("utf-8")
orig_md5 = md5(COLLABD)
work = tempfile.mkdtemp(prefix="mutrun-")
shutil.copy2(COLLABD, os.path.join(work, "collabd.py"))
shutil.copy2(SELFTEST, os.path.join(work, "selftest.py"))
# 替身:把 collabd 换成变异体
with open(os.path.join(work, "collabd.py"), "w", encoding="utf-8") as f:
f.write(orig_src)
base_out = run_selftest(work)
base_red, base_ran = verdict(base_out)
print("\n【基线】mutant 未植入时的原版:")
print(" 找到用例=%s 报红=%s => %s"
% (base_ran, base_red, "✅ 绿(可谈抓得住)" if (base_ran and not base_red) else "❌ 基线不绿,验证作废"))
if not (base_ran and not base_red):
print("\n❌ 基线不绿 ⇒ 停在此处(恒红比漏网更坏:会让所有变异都'红',验证作废)。")
return 2
results = []
for mid, target, fn in MUTS:
# 还原原版再植入
with open(os.path.join(work, "collabd.py"), "w", encoding="utf-8") as f:
f.write(orig_src)
try:
mutated = fn(orig_src)
except AssertionError as e:
results.append((mid, target, "锚点未命中:%s" % e, False))
print(" ⛔ %-3s %-12s 锚点未命中 ⇒ 计入失败" % (mid, target))
continue
# ③ 验「被检对象真被改了」:长度差 + md5 + 语法
if mutated == orig_src:
results.append((mid, target, "变异体与原版相同(replace 没生效)", False))
print(" ⛔ %-3s %-12s 真没被改 ⇒ 计入失败" % (mid, target))
continue
with open(os.path.join(work, "collabd.py"), "w", encoding="utf-8") as f:
f.write(mutated)
dlen = len(mutated) - len(orig_src)
mm = md5(os.path.join(work, "collabd.py"))
cp = subprocess.run([PY, "-m", "py_compile", os.path.join(work, "collabd.py")],
capture_output=True, text=True, encoding="utf-8")
changed = (mm != orig_md5) and (dlen != 0)
if not changed:
results.append((mid, target, "被检对象未真改", False))
print(" ⛔ %-3s %-12s md5/长度未变 ⇒ 计入失败" % (mid, target))
continue
if cp.returncode != 0:
results.append((mid, target, "变异体语法错(测的是崩溃不是判据)", False))
print(" ⛔ %-3s %-12s 语法错 ⇒ 计入失败" % (mid, target))
continue
out = run_selftest(work)
# 🔴🔴 最关键的一道:**确认被检代码里真的有变异标记**。
# 上一轮 7/7 全"假绿"就是栽在这:变异体压根没被加载,测的是原版。
# ⛔ 只看"跑绿/跑红"不够 —— 必须证明**跑的那份**含变异。
marker = "# MUTANT:"
if marker not in mutated:
results.append((mid, target, "变异体缺标记(无法证明被加载)", False))
print(" ⛔ %-3s %-12s 变异体无标记 ⇒ 计入失败" % (mid, target))
continue
if marker not in open(os.path.join(work, "collabd.py"), encoding="utf-8").read():
results.append((mid, target, "工作目录里没有变异体", False))
print(" ⛔ %-3s %-12s 工作目录无变异体 ⇒ 计入失败" % (mid, target))
continue
red, ran = verdict(out)
if red is None:
results.append((mid, target, "没跑到用例(验证作废)", False))
print(" ⛔ %-3s %-12s 没跑到用例 ⇒ 计入失败" % (mid, target))
continue
ok = bool(red)
results.append((mid, target, "报红" if ok else "❌未报红(假绿)", ok))
print(" %s %-3s %-12s 真改(len%+d) ⇒ %s"
% ("✅" if ok else "❌", mid, target, dlen, "报红 ✅" if ok else "未报红 ❌ 假绿"))
# 还原
with open(os.path.join(work, "collabd.py"), "w", encoding="utf-8") as f:
f.write(orig_src)
print("\n" + "-" * 74)
npass = sum(1 for r in results if r[3])
ntot = len(results)
print("变异汇总:%d/%d 报红" % (npass, ntot))
# 每个新判据的覆盖
cov = {}
for mid, target, msg, ok in results:
cov.setdefault(target, []).append((mid, ok))
print("\n每个新判据的变异体覆盖(≥2 个):")
for target, lst in cov.items():
okc = sum(1 for _, o in lst if o)
flag = "✅" if len(lst) >= 2 and okc == len(lst) else "❌"
print(" %s %-12s 变异体 %d 个,报红 %d 个 %s"
% (flag, target, len(lst), okc, [m for m, _ in lst]))
shutil.rmtree(work, ignore_errors=True)
print("\n" + ("✅ 全部变异体按预期报红" if npass == ntot and ntot >= 2 else "❌ 有变异体未报红/被跳过"))
return 0 if npass == ntot else 1
if __name__ == "__main__":
sys.exit(main())