# -*- coding: utf-8 -*- """对指定会话做「积分消耗」归因: ① 逐请求 input/output token 曲线(credit burn 的直接证据) ② 上下文构成(谁占的) ③ 最重的工具 / 重复读的文件 ④ 用户可见输出 vs 烧掉的输入 用法: python diag-credit.py """ import io import json import os import re import sys from collections import Counter PROJ = r"E:\ProgramData\.workbuddy\projects\e-ProgramData-AIProject-aliyun-dsh-server" SID8 = sys.argv[1] if len(sys.argv) > 1 else "7057685c" P = None for fn in os.listdir(PROJ): if fn.startswith(SID8) and fn.endswith(".jsonl"): P = os.path.join(PROJ, fn) assert P, "找不到转录 " + SID8 CPT = 2.5 def sz(x): if isinstance(x, str): return len(x) try: return len(json.dumps(x, ensure_ascii=False)) except Exception: return len(str(x)) def txt_of(c): if isinstance(c, list): return "".join(x.get("text", "") for x in c if isinstance(x, dict)) return str(c or "") titles, usages, tot = [], [], Counter() results, calls, reads = [], [], Counter() per_tool_chars, per_tool_n = Counter(), Counter() asst_msgs, user_real = [], [] pending_call = {} with io.open(P, encoding="utf-8", errors="replace") as f: for ln in f: if not ln.strip(): continue try: o = json.loads(ln) except Exception: continue t = o.get("type") if t == "ai-title": titles.append(o.get("aiTitle", "")) elif t == "message": s = sz(txt_of(o.get("content"))) if o.get("role") == "user": tot["user_msg"] += s txt = txt_of(o.get("content")) m = re.search(r"(.*?)", txt, re.S) if m: user_real.append((o.get("timestamp"), " ".join(m.group(1).split())[:100])) else: tot["asst_msg"] += s asst_msgs.append((o.get("timestamp"), sz(txt_of(o.get("content"))))) elif t == "reasoning": tot["reasoning"] += sz(o.get("content")) elif t == "function_call": s = sz(o.get("arguments")) tot["fn_call_args"] += s nm = o.get("name") or "?" calls.append((nm, s)) per_tool_chars[nm] += s per_tool_n[nm] += 1 u = ((o.get("message") or {}).get("usage")) or {} if u: usages.append((o.get("timestamp"), u.get("input_tokens"), u.get("output_tokens"), u.get("cache_read_input_tokens"))) if nm == "Read": try: a = json.loads(o.get("arguments") or "{}") fp = a.get("file_path") or a.get("path") or "" if fp: reads[os.path.basename(fp)] += 1 except Exception: pass elif t == "function_call_result": s = sz(o.get("output")) tot["fn_result"] += s nm = o.get("name") or "?" per_tool_chars[nm] += s per_tool_n[nm] += 1 results.append((nm, s, " ".join(str(o.get("output"))[:100].split()))) else: tot["other:" + str(t)] += sz(o) print("=" * 78) print("会话 %s 标题: %s" % (SID8, titles)) print("=" * 78) print() print("① 逐请求 token(input / output / 缓存读)—— 直接看积分去向") print("=" * 78) if usages: ti = sum(u[1] or 0 for u in usages) to = sum(u[2] or 0 for u in usages) tc = sum(u[3] or 0 for u in usages) print(" 请求数 = %d" % len(usages)) print(" Σ input = %12d token" % ti) print(" Σ output = %12d token" % to) print(" Σ cache_read = %12d token(%.0f%% 的 input 命中缓存)" % (tc, 100.0 * tc / max(ti, 1))) print(" ★ input : output = %.1f : 1" % (ti / max(to, 1))) print() print(" 前 5 条 / 后 5 条:") for u in usages[:5] + [None] + usages[-5:]: if u is None: print(" …") else: print(" ts=%-14s in=%-8s out=%-6s cache=%s" % (str(u[0])[:14], u[1], u[2], u[3])) else: print(" (无 usage)") print() print("② 上下文构成(转录体量归因)") print("=" * 78) grand = sum(tot.values()) for k, v in tot.most_common(): print(" %-30s %10d 字符 ≈%9d tok %5.1f%%" % (k, v, int(v / CPT), 100.0 * v / grand)) print() print(" AI 正文 : 用户真实输入 比 = %.1f : 1" % (tot["asst_msg"] / max(tot["user_msg"], 1))) print(" 工具往返占体量 = %.1f%%(fn_result + fn_call_args)" % (100.0 * (tot["fn_result"] + tot["fn_call_args"]) / max(grand, 1))) print() print("③ 最重的工具(累计字符 / 次数)") print("=" * 78) for nm, c in per_tool_chars.most_common(12): print(" %-18s %10d 字符 ≈%9d tok 调用 %d 次 (平均 %d 字符)" % (nm, c, int(c / CPT), per_tool_n[nm], c // max(per_tool_n[nm], 1))) print() print("④ 单条最大工具结果 Top 10") print("=" * 78) for i, (nm, c, pv) in enumerate(sorted(results, key=lambda x: -x[1])[:10], 1): print(" %2d) %-14s %8d 字符 ≈%6d tok | %s" % (i, nm, c, int(c / CPT), pv[:86])) print() print("⑤ 被重复 Read 的文件(读一次留一份,永不释放)") print("=" * 78) rep = [(k, v) for k, v in reads.most_common(12) if v >= 2] for k, v in rep: print(" %-52s %d 次" % (k[:52], v)) if not rep: print(" (无重复读)") print() print("⑥ 用户真实输入()共 %d 条" % len(user_real)) print("=" * 78) for ts, q in user_real[:30]: print(" [%s] %s" % (ts, q))