2026-09-24 07:51:03 +08:00
|
|
|
# -*- coding: utf-8 -*-
|
|
|
|
|
"""在全部转录里按 ai-title 找「评估代码清理影响并整理」,并顺便给每个会话的身份卡。"""
|
|
|
|
|
import io
|
|
|
|
|
import json
|
|
|
|
|
import os
|
|
|
|
|
import re
|
|
|
|
|
|
|
|
|
|
ROOT = r"E:\ProgramData\.workbuddy\projects"
|
|
|
|
|
KW = ["清理", "评估", "整理"]
|
|
|
|
|
out = []
|
|
|
|
|
for proj in os.listdir(ROOT):
|
|
|
|
|
d = os.path.join(ROOT, proj)
|
|
|
|
|
if not os.path.isdir(d):
|
|
|
|
|
continue
|
|
|
|
|
for fn in os.listdir(d):
|
|
|
|
|
if not fn.endswith(".jsonl"):
|
|
|
|
|
continue
|
|
|
|
|
p = os.path.join(d, fn)
|
|
|
|
|
titles = []
|
|
|
|
|
n_user = 0
|
|
|
|
|
first_real = None
|
|
|
|
|
try:
|
|
|
|
|
with io.open(p, encoding="utf-8", errors="replace") as f:
|
|
|
|
|
for ln in f:
|
|
|
|
|
if '"ai-title"' not in ln and "<user_query>" not in ln:
|
|
|
|
|
continue
|
|
|
|
|
try:
|
|
|
|
|
o = json.loads(ln)
|
|
|
|
|
except Exception:
|
|
|
|
|
continue
|
|
|
|
|
if o.get("type") == "ai-title":
|
|
|
|
|
titles.append(o.get("aiTitle", ""))
|
|
|
|
|
if o.get("type") == "message" and o.get("role") == "user":
|
|
|
|
|
t = o.get("content")
|
|
|
|
|
s = "".join(x.get("text", "") for x in t if isinstance(x, dict)) if isinstance(t, list) else str(t)
|
|
|
|
|
if "<user_query>" in s and first_real is None:
|
|
|
|
|
first_real = " ".join(s.split())[:90]
|
|
|
|
|
n_user += 1
|
|
|
|
|
except Exception:
|
|
|
|
|
continue
|
|
|
|
|
blob = " ".join(titles)
|
|
|
|
|
if any(k in blob for k in KW):
|
|
|
|
|
out.append((os.path.getsize(p), proj, fn, titles, n_user, first_real))
|
|
|
|
|
|
|
|
|
|
print("=" * 78)
|
|
|
|
|
print("按标题命中「清理/评估/整理」的会话")
|
|
|
|
|
print("=" * 78)
|
|
|
|
|
for size, proj, fn, titles, nu, fr in sorted(out, key=lambda x: -x[0]):
|
|
|
|
|
print(" %s %6.1f MB user=%d" % (fn[:8], size / 1024 / 1024, nu))
|
|
|
|
|
print(" 工作区: %s" % proj)
|
|
|
|
|
print(" 标题 : %s" % titles)
|
|
|
|
|
print(" 首问 : %s" % fr)
|
|
|
|
|
print()
|
|
|
|
|
|
|
|
|
|
print("=" * 78)
|
|
|
|
|
print("本工作区全部会话的身份卡(标题 + 首问),供交叉确认")
|
|
|
|
|
print("=" * 78)
|
2026-10-10 23:13:22 +08:00
|
|
|
d = os.path.join(ROOT, "e-ProgramData-AIProject-ai1net-dsh-server")
|
2026-09-24 07:51:03 +08:00
|
|
|
for fn in sorted(os.listdir(d)):
|
|
|
|
|
if not fn.endswith(".jsonl"):
|
|
|
|
|
continue
|
|
|
|
|
p = os.path.join(d, fn)
|
|
|
|
|
titles, fr, last = [], None, None
|
|
|
|
|
with io.open(p, encoding="utf-8", errors="replace") as f:
|
|
|
|
|
for ln in f:
|
|
|
|
|
if '"ai-title"' not in ln and '"<user_query>"' not in ln:
|
|
|
|
|
continue
|
|
|
|
|
try:
|
|
|
|
|
o = json.loads(ln)
|
|
|
|
|
except Exception:
|
|
|
|
|
continue
|
|
|
|
|
if o.get("type") == "ai-title":
|
|
|
|
|
titles.append(o.get("aiTitle", ""))
|
|
|
|
|
if o.get("type") == "message" and o.get("role") == "user":
|
|
|
|
|
t = o.get("content")
|
|
|
|
|
s = "".join(x.get("text", "") for x in t if isinstance(x, dict)) if isinstance(t, list) else str(t)
|
|
|
|
|
if "<user_query>" in s and fr is None:
|
|
|
|
|
fr = " ".join(s.split())[:80]
|
|
|
|
|
last = o.get("timestamp")
|
|
|
|
|
print(" %s %5.1f MB %s" % (fn[:8], os.path.getsize(p) / 1024 / 1024, titles))
|
|
|
|
|
print(" 首问: %s" % fr)
|
|
|
|
|
print(" 末条 ts: %s" % last)
|