Files

261 lines
13 KiB
Python
Raw Permalink Normal View History

# 🔴 变异验证 · 2026-10-04 · 会话级忙判据两个新缺陷
#
# ⚠️ 纪律(来自 references/00-动手前必过.md ③④):
# ① **先验基线绿**:基线 PASS 17/FAIL 0(忙判据那条用例)。
# ② **锚在唯一位置**:`def _session_busy_conv` / `def _conv_ts` 唯一定位,⛔ 不锚"文本出现过"。
# ③ **每个变异先验「被检对象真被改了」**:比对长度+md5,⛔ 没真改就算**失败**(不许跳过分母)。
# ④ ⛔ **跳过的变异一律计入失败**。
#
# 用法:python mut_run.py
import hashlib
import importlib.util
import io
import os
import re
import shutil
import subprocess
import sys
import tempfile
HERE = os.path.dirname(os.path.abspath(__file__))
COLLABD = os.path.join(HERE, "collabd.py")
SELFTEST = os.path.join(HERE, "selftest.py")
PY = sys.executable
def md5(p):
return hashlib.md5(open(p, "rb").read()).hexdigest()
# ── 变异体定义 ────────────────────────────────────────────────────────────
# (编号, 目标判据, 人类可读描述, 变异函数)
def mut_order_revert(src):
"""M1 针对 ⑭:**丢掉 key 的时间戳分量** = 退回"纯遍历顺序"⇒ 昨天盖掉今天。
⚠️ 踩坑记录(这条本身就是变异验证的教训):第一版写成
`key[1] >= last_t[1]`(只把比较从元组换成 seq)⇒ **等价变异、恒绿**。
原因:样本里每个文件内部时间戳本来就升序,**任何"取最后一条"的写法都等价**。
⇒ 变异必须动**语义**(时间戳参与/不参与),⛔ 不能只动写法。
"""
old = " key = (ts if ts is not None else 0.0, seq)"
new = " key = (0.0, seq) # MUTANT: 丢掉时间戳 = 退回纯遍历顺序"
assert old in src, "锚点1 未命中"
return src.replace(old, new, 1)
def mut_freshness_off(src):
"""M2 针对 ⑮:把新鲜度闸阈值放大到"永不过期"(缺陷原样)。"""
old = " if last_t[0] > 0 and (time.time() - last_t[0]) > SRSM_FRESH:"
new = " if False: # MUTANT: 关掉新鲜度闸"
assert old in src, "锚点2 未命中"
return src.replace(old, new, 1)
def mut_tz_local(src):
"""M3 针对 ⑰:时间戳改按**本地时区**解析(mktime)⇒ 东八区差 8h ⇒ 恒判过期。"""
old = " t = float(calendar.timegm((int(m.group(\"Y\")), int(m.group(\"M\")), int(m.group(\"D\")),\n int(m.group(\"h\")), int(m.group(\"m\")), int(m.group(\"s\")), 0, 0, 0)))"
new = (" import calendar as _c # MUTANT: 按本地时区解析\n"
" _st = (int(m.group(\"Y\")), int(m.group(\"M\")), int(m.group(\"D\")),\n"
" int(m.group(\"h\")), int(m.group(\"m\")), int(m.group(\"s\")), 0, 0, -1)\n"
" t = float(time.mktime(_st))")
assert old in src, "锚点3 未命中"
return src.replace(old, new, 1)
def mut_tz_double_offset(src):
"""M9 针对 ⑰:额外+8h 偏移(另一种时区错法)⇒ 「2 分钟前」变「8 小时后」⇒ 恒不过期。"""
old = " ms = m.group(\"ms\")\n return t + (float(\"0.\" + ms) if ms else 0.0)"
new = (" ms = m.group(\"ms\")\n"
" return t + 28800.0 + (float(\"0.\" + ms) if ms else 0.0) # MUTANT: 多加 8h")
assert old in src, "锚点9 未命中"
return src.replace(old, new, 1)
def mut_freshness_too_hard(src):
"""M4 针对 ⑯:新鲜度闸写太狠(窗口=0)⇒ 连正在跑的也判 idle(反向回归)。"""
old = "(time.time() - last_t[0]) > SRSM_FRESH:"
new = "(time.time() - last_t[0]) > 0: # MUTANT: 闸写太狠"
assert old in src, "锚点4 未命中"
return src.replace(old, new, 1)
def mut_ts_none(src):
"""M5 针对 ⑰:时间戳一律返回 None(解析失败)⇒ 保守当 busy ⇒ ⑯ 报红。"""
old = " m = _CONV_TS.match(ln)"
new = " return None # MUTANT: 一律解析失败\n m = _CONV_TS.match(ln)"
assert old in src, "锚点5 未命中"
return src.replace(old, new, 1)
def mut_busy_always(src):
"""M8 针对 ⑯:`idle` 也被当成 busy(陈旧闸写太狠的另一种写法)⇒ ⑯/⑭ 报红。"""
old = " if last_to in (\"working\", \"planning\"):"
new = " if True: # MUTANT: 不分状态一律当在跑"
assert old in src, "锚点8 未命中"
return src.replace(old, new, 1)
def mut_order_reverse_files(src):
"""M6 针对 ⑭:把 `_conv_log` 的返回顺序反过来([昨天, 今天])⇒ 顺序依赖立刻现形。"""
old = " for cand in (\"%s.log\" % sid, \"%s.log\" % str(sid).replace(\"-\", \"\")):"
new = " for cand in (\"%s.log\" % sid,): # MUTANT: 去掉无连字符形态"
assert old in src, "锚点6 未命中"
s = src.replace(old, new, 1)
return s.replace(" for d in (time.strftime(\"%Y-%m-%d\"),\n time.strftime(\"%Y-%m-%d\", time.localtime(time.time() - 86400))):",
" for d in (time.strftime(\"%Y-%m-%d\", time.localtime(time.time() - 86400)),\n time.strftime(\"%Y-%m-%d\")):", 1)
def mut_freshness_only_first_file(src):
"""M7 针对 ⑮:陈旧判定只在**第一个文件**生效 ⇒ 单文件陈旧样本仍判 busy(⑮ 报红)。"""
old = " if last_t[0] > 0 and (time.time() - last_t[0]) > SRSM_FRESH:"
new = " if last_t[0] > 0 and (time.time() - last_t[0]) > SRSM_FRESH and len(_conv_log(sid)) > 1: # MUTANT: 陈旧判定只在多文件时生效"
assert old in src, "锚点7 未命中"
return src.replace(old, new, 1)
MUTS = [
("M1", "⑭跨日取最新", mut_order_revert),
("M2", "⑮新鲜度闸", mut_freshness_off),
("M3", "⑰UTC时区", mut_tz_local),
("M4", "⑯反向回归", mut_freshness_too_hard),
("M5", "⑰时间戳解析", mut_ts_none),
("M6", "⑭跨日取最新", mut_order_reverse_files),
("M7", "⑮新鲜度闸", mut_freshness_only_first_file),
("M8", "⑯反向回归", mut_busy_always),
("M9", "⑰UTC时区", mut_tz_double_offset),
]
def run_selftest(workdir):
"""⚠️ **必须用工作目录里的那份 `selftest.py`**,⛔ 不能用技能目录的原版绝对路径。
🔴🔴 2026-10-04 当场栽过(写在这里当护栏,别删):`selftest.py` 里
`HERE = Path(__file__).resolve().parent`,而 `imp()` 加载的是 **`HERE / "collabd.py"`**
⇒ 只改 `cwd` 没用、只传原版绝对路径也没用 ⇒ **测的一直是原版 ⇒ 恒绿 ⇒ 7/7 全是假象**。
⇒ 判据:`HERE` 指向 `workdir` 时才算数(下面 `assert` 硬查)。
"""
st = os.path.join(workdir, "selftest.py")
assert os.path.abspath(os.path.dirname(st)) == os.path.abspath(workdir), "测的不是工作目录里的 selftest"
p = subprocess.run([PY, st, "-k", "忙判据"],
capture_output=True, text=True, encoding="utf-8",
errors="replace", cwd=workdir, timeout=600)
out = p.stdout + p.stderr
# 🔴 护栏:确认真加载的是 workdir 里那份(防止以后再有人改回绝对路径)
assert workdir.replace("\\", "/") in out.replace("\\", "/"), "输出里看不到工作目录 ⇒ 加载的不是变异体"
return out
def verdict(out):
"""返回 (是否报红, 是否真找到用例)。⛔ 没找到用例 = 验证作废,不算绿。"""
ran = ("忙判据" in out) and ("投递忙判据" in out)
if not ran:
return None, False
# 该用例整体 FAIL ⇔ 汇总行 PASS 0/ FAIL >=1,或用例行前是 ✗
m = re.search(r"合计:PASS (\d+) / FAIL (\d+)", out)
if not m:
return None, True
return (int(m.group(2)) >= 1), True
def main():
print("=" * 74)
print("变异验证 · 会话级忙判据(每个新判据 ≥2 个变异体,⛔ 跳过计入失败)")
print("=" * 74)
orig_src = open(COLLABD, "rb").read().decode("utf-8")
orig_md5 = md5(COLLABD)
work = tempfile.mkdtemp(prefix="mutrun-")
shutil.copy2(COLLABD, os.path.join(work, "collabd.py"))
shutil.copy2(SELFTEST, os.path.join(work, "selftest.py"))
# 替身:把 collabd 换成变异体
with open(os.path.join(work, "collabd.py"), "w", encoding="utf-8") as f:
f.write(orig_src)
base_out = run_selftest(work)
base_red, base_ran = verdict(base_out)
print("\n【基线】mutant 未植入时的原版:")
print(" 找到用例=%s 报红=%s => %s"
% (base_ran, base_red, "✅ 绿(可谈抓得住)" if (base_ran and not base_red) else "❌ 基线不绿,验证作废"))
if not (base_ran and not base_red):
print("\n❌ 基线不绿 ⇒ 停在此处(恒红比漏网更坏:会让所有变异都'红',验证作废)。")
return 2
results = []
for mid, target, fn in MUTS:
# 还原原版再植入
with open(os.path.join(work, "collabd.py"), "w", encoding="utf-8") as f:
f.write(orig_src)
try:
mutated = fn(orig_src)
except AssertionError as e:
results.append((mid, target, "锚点未命中:%s" % e, False))
print(" ⛔ %-3s %-12s 锚点未命中 ⇒ 计入失败" % (mid, target))
continue
# ③ 验「被检对象真被改了」:长度差 + md5 + 语法
if mutated == orig_src:
results.append((mid, target, "变异体与原版相同(replace 没生效)", False))
print(" ⛔ %-3s %-12s 真没被改 ⇒ 计入失败" % (mid, target))
continue
with open(os.path.join(work, "collabd.py"), "w", encoding="utf-8") as f:
f.write(mutated)
dlen = len(mutated) - len(orig_src)
mm = md5(os.path.join(work, "collabd.py"))
cp = subprocess.run([PY, "-m", "py_compile", os.path.join(work, "collabd.py")],
capture_output=True, text=True, encoding="utf-8")
changed = (mm != orig_md5) and (dlen != 0)
if not changed:
results.append((mid, target, "被检对象未真改", False))
print(" ⛔ %-3s %-12s md5/长度未变 ⇒ 计入失败" % (mid, target))
continue
if cp.returncode != 0:
results.append((mid, target, "变异体语法错(测的是崩溃不是判据)", False))
print(" ⛔ %-3s %-12s 语法错 ⇒ 计入失败" % (mid, target))
continue
out = run_selftest(work)
# 🔴🔴 最关键的一道:**确认被检代码里真的有变异标记**。
# 上一轮 7/7 全"假绿"就是栽在这:变异体压根没被加载,测的是原版。
# ⛔ 只看"跑绿/跑红"不够 —— 必须证明**跑的那份**含变异。
marker = "# MUTANT:"
if marker not in mutated:
results.append((mid, target, "变异体缺标记(无法证明被加载)", False))
print(" ⛔ %-3s %-12s 变异体无标记 ⇒ 计入失败" % (mid, target))
continue
if marker not in open(os.path.join(work, "collabd.py"), encoding="utf-8").read():
results.append((mid, target, "工作目录里没有变异体", False))
print(" ⛔ %-3s %-12s 工作目录无变异体 ⇒ 计入失败" % (mid, target))
continue
red, ran = verdict(out)
if red is None:
results.append((mid, target, "没跑到用例(验证作废)", False))
print(" ⛔ %-3s %-12s 没跑到用例 ⇒ 计入失败" % (mid, target))
continue
ok = bool(red)
results.append((mid, target, "报红" if ok else "❌未报红(假绿)", ok))
print(" %s %-3s %-12s 真改(len%+d) ⇒ %s"
% ("✅" if ok else "❌", mid, target, dlen, "报红 ✅" if ok else "未报红 ❌ 假绿"))
# 还原
with open(os.path.join(work, "collabd.py"), "w", encoding="utf-8") as f:
f.write(orig_src)
print("\n" + "-" * 74)
npass = sum(1 for r in results if r[3])
ntot = len(results)
print("变异汇总:%d/%d 报红" % (npass, ntot))
# 每个新判据的覆盖
cov = {}
for mid, target, msg, ok in results:
cov.setdefault(target, []).append((mid, ok))
print("\n每个新判据的变异体覆盖(≥2 个):")
for target, lst in cov.items():
okc = sum(1 for _, o in lst if o)
flag = "✅" if len(lst) >= 2 and okc == len(lst) else "❌"
print(" %s %-12s 变异体 %d 个,报红 %d 个 %s"
% (flag, target, len(lst), okc, [m for m, _ in lst]))
shutil.rmtree(work, ignore_errors=True)
print("\n" + ("✅ 全部变异体按预期报红" if npass == ntot and ntot >= 2 else "❌ 有变异体未报红/被跳过"))
return 0 if npass == ntot else 1
if __name__ == "__main__":
sys.exit(main())