chore(工作区): 纳入版本控制基线(回收 411 MB 过程产物)

回收 411 MB(470 M → 58.8 M),全部经回收站,可恢复:
- 待清理/(146.2 M,含 relay 分片 128 M 与 42 项过程目录)
- tmp/(32.4 M,按接续棒命名的过程临时区)
- .workbuddy/tmp/(39.5 M)
- 4 份 workbuddy.db 冗余副本(101 M,09-23 事故的坏副本 / 抢救产物)
- tmp/im16/gw/centrifugo 二进制(63.9 M,可重下)+ 缓存残留

入库范围:常驻规则(CODEBUDDY.md / README.md / state.py)、在途接续入口与
接续包、docs/、交付物/、交接单/、归档/、scripts/、.codebuddy/、
.workbuddy/memory/;共 398 件,其中 >60 KB 的 26 件全为文档。

排除(.gitignore):tmp/、待清理/、运行态日志与缓存、*.db 与 DB 备份整目录、
打包二进制(*.tar.gz / *.tgz)、记忆修复前备份。
This commit is contained in:
admin committed 2026-09-24 07:51:03 +08:00
commit ce8e6ceed9
396 files changed
+66045

No files matched your search

+93
View File
@@ -0,0 +1,93 @@
import io, json, sqlite3, collections, re
con = sqlite3.connect(r"E:/ProgramData/.workbuddy/workbuddy.db"); con.row_factory = sqlite3.Row
autos = {}
for r in con.execute("select id,name,status,schedule_type,rrule,scheduled_at,cwds,created_at from automations"):
autos[r["id"]] = dict(r)
def walk(o, acc):
"""递归收集所有 usage 对象"""
if isinstance(o, dict):
u = o.get("usage")
if isinstance(u, dict) and "inputTokens" in u:
acc.append(u)
for v in o.values(): walk(v, acc)
elif isinstance(o, list):
for v in o: walk(v, acc)
rows = []
for r in con.execute("select automation_id, thread_id, runs_json, result_success, created_at from automation_runs order by created_at"):
aid = r["automation_id"]
acc = []
try: walk(json.loads(r["runs_json"] or "[]"), acc)
except Exception: pass
n_tool = len(re.findall(r'"toolName"', r["runs_json"] or ""))
rows.append({"aid": aid, "t": r["created_at"], "usages": acc, "n_tool": n_tool,
"name": (autos.get(aid) or {}).get("name", aid)})
print("=" * 96)
print("%-34s %-8s %-8s %-9s %-8s %s" % ("自动化", "运行次", "工具次", "积分总计", "缓存命中", "请求数"))
print("=" * 96)
agg = collections.defaultdict(lambda: {"runs":0,"tools":0,"cost":0.0,"hit":0,"miss":0,"req":0,"cat":collections.Counter()})
for x in rows:
a = agg[x["aid"]]; a["runs"] += 1; a["tools"] += x["n_tool"]
for u in x["usages"]:
a["req"] += 1
c = (u.get("cost") or {}).get("amount") or 0
a["cost"] += c
d = u.get("details") or {}
a["hit"] += d.get("promptCacheHitTokens", 0) or 0
a["miss"] += d.get("promptCacheMissTokens", 0) or 0
for k, v in (u.get("byCategory") or {}).items():
if isinstance(v, (int, float)): a["cat"][k] += v
tot = {"cost":0.0,"hit":0,"miss":0,"tools":0,"req":0}
for aid, a in sorted(agg.items(), key=lambda kv: -kv[1]["cost"]):
nm = (autos.get(aid) or {}).get("name", aid)[:32]
hr = a["hit"] / max(a["hit"] + a["miss"], 1) * 100
print("%-34s %-8d %-8d %-9.2f %-8s %d" % (nm, a["runs"], a["tools"], a["cost"], "%.1f%%" % hr, a["req"]))
tot["cost"] += a["cost"]; tot["hit"] += a["hit"]; tot["miss"] += a["miss"]; tot["tools"] += a["tools"]; tot["req"] += a["req"]
print("-" * 96)
print("合计:运行 %d 次 | 工具调用 %d | 模型请求 %d | 积分 %.2f | 缓存命中率 %.1f%%"
% (sum(a["runs"] for a in agg.values()), tot["tools"], tot["req"], tot["cost"],
tot["hit"] / max(tot["hit"] + tot["miss"], 1) * 100))
print()
print("== 固定注入构成(所有自动化请求的平均,token)==")
cats = collections.Counter()
for a in agg.values(): cats.update(a["cat"])
if cats:
s = sum(v for k, v in cats.items() if k != "version")
for k, v in cats.most_common():
if k == "version": continue
print(" %-14s %8.0f (%4.1f%%)" % (k, v / max(len(rows), 1), v / max(s, 1) * 100))
print(" %-14s %8.0f" % ("合计/请求", s / max(len(rows), 1)))
print()
print("== 逐次请求明细(前 20 条:inputTokens / credit / 缓存命中率 / 工具数)==")
n = 0
for x in rows:
for u in x["usages"]:
d = u.get("details") or {}
h, m = d.get("promptCacheHitTokens", 0) or 0, d.get("promptCacheMissTokens", 0) or 0
c = (u.get("cost") or {}).get("amount") or 0
print(" %-30s in=%-8s credit=%-7s hit=%4.1f%% tools=%d [%s]"
% (x["name"][:30], u.get("inputTokens"), round(c, 3),
h / max(h + m, 1) * 100, x["n_tool"],
__import__("datetime").datetime.fromtimestamp(x["t"] / 1000).strftime("%m-%d %H:%M")))
n += 1
if n >= 20: break
if n >= 20: break
print()
print("== 仍在跑的自动化(按计划类型)==")
for aid, a in autos.items():
if a["status"] != "ACTIVE": continue
if a["rrule"]:
per_day = {"HOURLY": 24}.get("HOURLY", 0)
iv = re.search(r"INTERVAL=(\d+)", a["rrule"])
k = 24 / int(iv.group(1)) if iv else "?"
print(" [周期] %-34s %s ⇒ 约 %.0f 次/天" % (a["name"][:34], a["rrule"], k))
else:
print(" [一次] %-34s %s" % (a["name"][:34], a["scheduled_at"]))
+47
View File
@@ -0,0 +1,47 @@
# -*- coding: utf-8 -*-
"""逐会话从转录 rawUsage 汇总真实积分(递归找 rawUsage.credit)。"""
import io, json, glob
HOME = "E:/ProgramData/.workbuddy"
SIDS = {
"3814f5fb": "覆盖网络线·归档接续(自动 @10:20)",
"e265f0cd": "覆盖网络线·接续优化版(自动 @10:45)",
"478eef8c": "决策方法-2(自动 @11:10)",
"d48a9be8": "覆盖网络线·S1 落地(自动 @14:32)",
"b08b1c35": "确认覆盖网络任务待办事项(手动新会话,基线)",
}
def walk(o, hits, path=""):
if isinstance(o, dict):
for k, v in o.items():
if k == "rawUsage" and isinstance(v, dict):
hits.append(v)
else:
walk(v, hits, path + "/" + str(k))
elif isinstance(o, list):
for x in o:
walk(x, hits, path)
print("%-38s %6s %8s %8s" % ("会话", "计费次", "积分合计", "均值"))
for short, label in SIDS.items():
hits = glob.glob(HOME + "/projects/*/" + short + "*.jsonl")
if not hits:
print(short, "无转录")
continue
ru = []
with io.open(hits[0], "r", encoding="utf-8", errors="replace", newline="") as f:
for line in f:
line = line.strip()
if not line:
continue
try:
d = json.loads(line)
except Exception:
continue
walk(d, ru)
cr = [r.get("credit") for r in ru if isinstance(r.get("credit"), (int, float))]
tot = sum(cr)
print("%-38s %6d %8.2f %8s" % (label[:36], len(cr), tot, ("%.3f" % (tot / len(cr))) if cr else "-"))
if cr:
s = sorted(cr)
print(" min=%.2f 中位=%.2f max=%.2f | 首=%.2f 末=%.2f" % (s[0], s[len(s)//2], s[-1], cr[0], cr[-1]))
+82
View File
@@ -0,0 +1,82 @@
# -*- coding: utf-8 -*-
"""在全部转录里按 ai-title 找「评估代码清理影响并整理」,并顺便给每个会话的身份卡。"""
import io
import json
import os
import re
ROOT = r"E:\ProgramData\.workbuddy\projects"
KW = ["清理", "评估", "整理"]
out = []
for proj in os.listdir(ROOT):
d = os.path.join(ROOT, proj)
if not os.path.isdir(d):
continue
for fn in os.listdir(d):
if not fn.endswith(".jsonl"):
continue
p = os.path.join(d, fn)
titles = []
n_user = 0
first_real = None
try:
with io.open(p, encoding="utf-8", errors="replace") as f:
for ln in f:
if '"ai-title"' not in ln and "<user_query>" not in ln:
continue
try:
o = json.loads(ln)
except Exception:
continue
if o.get("type") == "ai-title":
titles.append(o.get("aiTitle", ""))
if o.get("type") == "message" and o.get("role") == "user":
t = o.get("content")
s = "".join(x.get("text", "") for x in t if isinstance(x, dict)) if isinstance(t, list) else str(t)
if "<user_query>" in s and first_real is None:
first_real = " ".join(s.split())[:90]
n_user += 1
except Exception:
continue
blob = " ".join(titles)
if any(k in blob for k in KW):
out.append((os.path.getsize(p), proj, fn, titles, n_user, first_real))
print("=" * 78)
print("按标题命中「清理/评估/整理」的会话")
print("=" * 78)
for size, proj, fn, titles, nu, fr in sorted(out, key=lambda x: -x[0]):
print(" %s %6.1f MB user=%d" % (fn[:8], size / 1024 / 1024, nu))
print(" 工作区: %s" % proj)
print(" 标题 : %s" % titles)
print(" 首问 : %s" % fr)
print()
print("=" * 78)
print("本工作区全部会话的身份卡(标题 + 首问),供交叉确认")
print("=" * 78)
d = os.path.join(ROOT, "e-ProgramData-AI技能-aliyun-dsh-server")
for fn in sorted(os.listdir(d)):
if not fn.endswith(".jsonl"):
continue
p = os.path.join(d, fn)
titles, fr, last = [], None, None
with io.open(p, encoding="utf-8", errors="replace") as f:
for ln in f:
if '"ai-title"' not in ln and '"<user_query>"' not in ln:
continue
try:
o = json.loads(ln)
except Exception:
continue
if o.get("type") == "ai-title":
titles.append(o.get("aiTitle", ""))
if o.get("type") == "message" and o.get("role") == "user":
t = o.get("content")
s = "".join(x.get("text", "") for x in t if isinstance(x, dict)) if isinstance(t, list) else str(t)
if "<user_query>" in s and fr is None:
fr = " ".join(s.split())[:80]
last = o.get("timestamp")
print(" %s %5.1f MB %s" % (fn[:8], os.path.getsize(p) / 1024 / 1024, titles))
print(" 首问: %s" % fr)
print(" 末条 ts: %s" % last)
+79
View File
@@ -0,0 +1,79 @@
# -*- coding: utf-8 -*-
"""回放测试:把今天所有会话的真实 Bash 命令灌进 bash-output-guard,统计误拦率。
只打印「摘要 + DENY 明细」,不 dump 全量命令。
"""
import collections
import glob
import importlib.util
import io
import json
import os
GUARD = r"D:\github\dsh_shenxian\dsh-server-docs\scripts\bash-output-guard.py"
PROJ = r"E:\ProgramData\.workbuddy\projects\e-ProgramData-AI技能-aliyun-dsh-server"
sp = importlib.util.spec_from_file_location("g", GUARD)
g = importlib.util.module_from_spec(sp)
sp.loader.exec_module(g)
cmds = []
for f in sorted(glob.glob(os.path.join(PROJ, "*.jsonl"))):
sid = os.path.basename(f)[:8]
for ln in io.open(f, encoding="utf-8", errors="replace"):
if '"function_call"' not in ln or '"Bash"' not in ln:
continue
try:
o = json.loads(ln)
except ValueError:
continue
if o.get("type") != "function_call" or o.get("name") != "Bash":
continue
try:
a = json.loads(o.get("arguments") or "{}")
except ValueError:
continue
c = a.get("command")
if isinstance(c, str) and c.strip():
cmds.append((sid, c))
by_sid = collections.Counter(s for s, _ in cmds)
print("=== 样本 ===")
print(" Bash 命令 %d 条 | 会话 %d 个" % (len(cmds), len(by_sid)))
for s, n in by_sid.most_common():
print(" %s %d 条" % (s, n))
den, safehit = [], 0
for sid, c in cmds:
if g.SAFE.search(c): # 真链路里 main() 先过 SAFE 就放行
safehit += 1
continue
why, fix = g.reason_for(c, hard=False)
if why:
den.append((sid, why, c))
print()
print("=== soft 档(默认)判定 ===")
print(" DENY = %d / %d = **%.2f%%** (被 SAFE 限流救回 %d 条)"
% (len(den), len(cmds), 100.0 * len(den) / max(len(cmds), 1), safehit))
for w, n in collections.Counter(w for _, w, _ in den).most_common():
print(" %-22s %d" % (w, n))
print()
print("=== DENY 明细(前 30 条 · 截 108 字符)===")
for sid, why, c in den[:30]:
print(" [%s] %-20s %s" % (sid, why, c.replace("\n", " ")[:108]))
if not den:
print(" (无 DENY —— 全部放行)")
print()
print("=== hard 档(仅参考)===")
denh = []
for sid, c in cmds:
if g.SAFE.search(c):
continue
why, _ = g.reason_for(c, hard=True)
if why:
denh.append((sid, why, c))
print(" DENY = %d / %d = %.2f%%" % (len(denh), len(cmds), 100.0 * len(denh) / max(len(cmds), 1)))
for w, n in collections.Counter(w for _, w, _ in denh).most_common():
print(" %-22s %d" % (w, n))
+39
View File
@@ -0,0 +1,39 @@
#!/bin/bash
# 带校验重试的接力搬运:47:/tmp/<prefix>-* --(并行拉+大小校验+重拉)--> 本机 --高速--> 106 --> (可选)解包
# 用法: _relay2.sh <prefix> <name> [解包目标目录]
export PATH="/d/Program Files/Git/usr/bin:/c/Windows/System32:/c/Windows:$PATH"
SSH="ssh -o BatchMode=yes -o ConnectTimeout=15"
SCP="scp -q -o BatchMode=yes -o ConnectTimeout=15"
P="$1"; N="$2"; DEST="$3"
L="/e/ProgramData/AI技能/aliyun-dsh-server/_relay-$N"
mkdir -p "$L"
$SSH bt-server "stat -c '%s %n' /tmp/$P-*" > "$L/remote.txt"
TOTAL=$(wc -l < "$L/remote.txt")
echo "远端分片=$TOTAL"
for round in 1 2 3; do
: > "$L/redo.txt"
while read -r rsz rpath; do
n=$(basename "$rpath")
lsz=$(stat -c%s "$L/$n" 2>/dev/null || echo 0)
[ "$rsz" != "$lsz" ] && echo "$rpath" >> "$L/redo.txt"
done < "$L/remote.txt"
NEED=$(wc -l < "$L/redo.txt")
echo "第${round}轮: 需拉/重拉 $NEED 片"
[ "$NEED" = "0" ] && break
xargs -P8 -I% $SCP bt-server:% "$L/" < "$L/redo.txt"
done
BAD=0
while read -r rsz rpath; do
n=$(basename "$rpath"); lsz=$(stat -c%s "$L/$n" 2>/dev/null || echo 0)
if [ "$rsz" != "$lsz" ]; then echo "❌ 仍不完整: $n 本地=$lsz 远端=$rsz"; BAD=1; fi
done < "$L/remote.txt"
[ "$BAD" = "1" ] && { echo "分片校验未通过,中止"; exit 1; }
echo "✅ 分片校验通过: $TOTAL 片 / $(du -cb "$L"/$P-* | tail -1 | cut -f1) 字节"
$SSH test106 "rm -rf /tmp/$N; mkdir -p /tmp/$N"
S=$(date +%s)
$SCP "$L"/$P-* test106:/tmp/$N/
E=$(date +%s)
echo "上传106: $($SSH test106 "ls /tmp/$N | wc -l") 片 耗时$((E-S))s"
if [ -n "$DEST" ]; then
$SSH test106 "cd /tmp/$N && cat $P-* > /tmp/$N.tgz && tar xzf /tmp/$N.tgz -C $DEST && echo '解包完成 → $DEST'"
fi
+163
View File
@@ -0,0 +1,163 @@
# -*- coding: utf-8 -*-
"""对指定会话做「积分消耗」归因:
① 逐请求 input/output token 曲线(credit burn 的直接证据)
② 上下文构成(谁占的)
③ 最重的工具 / 重复读的文件
④ 用户可见输出 vs 烧掉的输入
用法: python diag-credit.py <sid前8位>
"""
import io
import json
import os
import re
import sys
from collections import Counter
PROJ = r"E:\ProgramData\.workbuddy\projects\e-ProgramData-AI技能-aliyun-dsh-server"
SID8 = sys.argv[1] if len(sys.argv) > 1 else "7057685c"
P = None
for fn in os.listdir(PROJ):
if fn.startswith(SID8) and fn.endswith(".jsonl"):
P = os.path.join(PROJ, fn)
assert P, "找不到转录 " + SID8
CPT = 2.5
def sz(x):
if isinstance(x, str):
return len(x)
try:
return len(json.dumps(x, ensure_ascii=False))
except Exception:
return len(str(x))
def txt_of(c):
if isinstance(c, list):
return "".join(x.get("text", "") for x in c if isinstance(x, dict))
return str(c or "")
titles, usages, tot = [], [], Counter()
results, calls, reads = [], [], Counter()
per_tool_chars, per_tool_n = Counter(), Counter()
asst_msgs, user_real = [], []
pending_call = {}
with io.open(P, encoding="utf-8", errors="replace") as f:
for ln in f:
if not ln.strip():
continue
try:
o = json.loads(ln)
except Exception:
continue
t = o.get("type")
if t == "ai-title":
titles.append(o.get("aiTitle", ""))
elif t == "message":
s = sz(txt_of(o.get("content")))
if o.get("role") == "user":
tot["user_msg"] += s
txt = txt_of(o.get("content"))
m = re.search(r"<user_query>(.*?)</user_query>", txt, re.S)
if m:
user_real.append((o.get("timestamp"), " ".join(m.group(1).split())[:100]))
else:
tot["asst_msg"] += s
asst_msgs.append((o.get("timestamp"), sz(txt_of(o.get("content")))))
elif t == "reasoning":
tot["reasoning"] += sz(o.get("content"))
elif t == "function_call":
s = sz(o.get("arguments"))
tot["fn_call_args"] += s
nm = o.get("name") or "?"
calls.append((nm, s))
per_tool_chars[nm] += s
per_tool_n[nm] += 1
u = ((o.get("message") or {}).get("usage")) or {}
if u:
usages.append((o.get("timestamp"), u.get("input_tokens"), u.get("output_tokens"),
u.get("cache_read_input_tokens")))
if nm == "Read":
try:
a = json.loads(o.get("arguments") or "{}")
fp = a.get("file_path") or a.get("path") or ""
if fp:
reads[os.path.basename(fp)] += 1
except Exception:
pass
elif t == "function_call_result":
s = sz(o.get("output"))
tot["fn_result"] += s
nm = o.get("name") or "?"
per_tool_chars[nm] += s
per_tool_n[nm] += 1
results.append((nm, s, " ".join(str(o.get("output"))[:100].split())))
else:
tot["other:" + str(t)] += sz(o)
print("=" * 78)
print("会话 %s 标题: %s" % (SID8, titles))
print("=" * 78)
print()
print("① 逐请求 token(input / output / 缓存读)—— 直接看积分去向")
print("=" * 78)
if usages:
ti = sum(u[1] or 0 for u in usages)
to = sum(u[2] or 0 for u in usages)
tc = sum(u[3] or 0 for u in usages)
print(" 请求数 = %d" % len(usages))
print(" Σ input = %12d token" % ti)
print(" Σ output = %12d token" % to)
print(" Σ cache_read = %12d token(%.0f%% 的 input 命中缓存)" % (tc, 100.0 * tc / max(ti, 1)))
print(" ★ input : output = %.1f : 1" % (ti / max(to, 1)))
print()
print(" 前 5 条 / 后 5 条:")
for u in usages[:5] + [None] + usages[-5:]:
if u is None:
print(" …")
else:
print(" ts=%-14s in=%-8s out=%-6s cache=%s" % (str(u[0])[:14], u[1], u[2], u[3]))
else:
print(" (无 usage)")
print()
print("② 上下文构成(转录体量归因)")
print("=" * 78)
grand = sum(tot.values())
for k, v in tot.most_common():
print(" %-30s %10d 字符 ≈%9d tok %5.1f%%" % (k, v, int(v / CPT), 100.0 * v / grand))
print()
print(" AI 正文 : 用户真实输入 比 = %.1f : 1" % (tot["asst_msg"] / max(tot["user_msg"], 1)))
print(" 工具往返占体量 = %.1f%%(fn_result + fn_call_args)"
% (100.0 * (tot["fn_result"] + tot["fn_call_args"]) / max(grand, 1)))
print()
print("③ 最重的工具(累计字符 / 次数)")
print("=" * 78)
for nm, c in per_tool_chars.most_common(12):
print(" %-18s %10d 字符 ≈%9d tok 调用 %d 次 (平均 %d 字符)"
% (nm, c, int(c / CPT), per_tool_n[nm], c // max(per_tool_n[nm], 1)))
print()
print("④ 单条最大工具结果 Top 10")
print("=" * 78)
for i, (nm, c, pv) in enumerate(sorted(results, key=lambda x: -x[1])[:10], 1):
print(" %2d) %-14s %8d 字符 ≈%6d tok | %s" % (i, nm, c, int(c / CPT), pv[:86]))
print()
print("⑤ 被重复 Read 的文件(读一次留一份,永不释放)")
print("=" * 78)
rep = [(k, v) for k, v in reads.most_common(12) if v >= 2]
for k, v in rep:
print(" %-52s %d 次" % (k[:52], v))
if not rep:
print(" (无重复读)")
print()
print("⑥ 用户真实输入(<user_query>)共 %d 条" % len(user_real))
print("=" * 78)
for ts, q in user_real[:30]:
print(" [%s] %s" % (ts, q))