# -*- coding: utf-8 -*- """在全部转录里按 ai-title 找「评估代码清理影响并整理」,并顺便给每个会话的身份卡。""" import io import json import os import re ROOT = r"E:\ProgramData\.workbuddy\projects" KW = ["清理", "评估", "整理"] out = [] for proj in os.listdir(ROOT): d = os.path.join(ROOT, proj) if not os.path.isdir(d): continue for fn in os.listdir(d): if not fn.endswith(".jsonl"): continue p = os.path.join(d, fn) titles = [] n_user = 0 first_real = None try: with io.open(p, encoding="utf-8", errors="replace") as f: for ln in f: if '"ai-title"' not in ln and "" not in ln: continue try: o = json.loads(ln) except Exception: continue if o.get("type") == "ai-title": titles.append(o.get("aiTitle", "")) if o.get("type") == "message" and o.get("role") == "user": t = o.get("content") s = "".join(x.get("text", "") for x in t if isinstance(x, dict)) if isinstance(t, list) else str(t) if "" in s and first_real is None: first_real = " ".join(s.split())[:90] n_user += 1 except Exception: continue blob = " ".join(titles) if any(k in blob for k in KW): out.append((os.path.getsize(p), proj, fn, titles, n_user, first_real)) print("=" * 78) print("按标题命中「清理/评估/整理」的会话") print("=" * 78) for size, proj, fn, titles, nu, fr in sorted(out, key=lambda x: -x[0]): print(" %s %6.1f MB user=%d" % (fn[:8], size / 1024 / 1024, nu)) print(" 工作区: %s" % proj) print(" 标题 : %s" % titles) print(" 首问 : %s" % fr) print() print("=" * 78) print("本工作区全部会话的身份卡(标题 + 首问),供交叉确认") print("=" * 78) d = os.path.join(ROOT, "e-ProgramData-AI技能-aliyun-dsh-server") for fn in sorted(os.listdir(d)): if not fn.endswith(".jsonl"): continue p = os.path.join(d, fn) titles, fr, last = [], None, None with io.open(p, encoding="utf-8", errors="replace") as f: for ln in f: if '"ai-title"' not in ln and '""' not in ln: continue try: o = json.loads(ln) except Exception: continue if o.get("type") == "ai-title": titles.append(o.get("aiTitle", "")) if o.get("type") == "message" and o.get("role") == "user": t = o.get("content") s = "".join(x.get("text", "") for x in t if isinstance(x, dict)) if isinstance(t, list) else str(t) if "" in s and fr is None: fr = " ".join(s.split())[:80] last = o.get("timestamp") print(" %s %5.1f MB %s" % (fn[:8], os.path.getsize(p) / 1024 / 1024, titles)) print(" 首问: %s" % fr) print(" 末条 ts: %s" % last)