Files
dsh_ai1net_server/.workbuddy/tools/find-session-by-title.py
T
admin ce8e6ceed9 chore(工作区): 纳入版本控制基线(回收 411 MB 过程产物)
回收 411 MB(470 M → 58.8 M),全部经回收站,可恢复:
- 待清理/(146.2 M,含 relay 分片 128 M 与 42 项过程目录)
- tmp/(32.4 M,按接续棒命名的过程临时区)
- .workbuddy/tmp/(39.5 M)
- 4 份 workbuddy.db 冗余副本(101 M,09-23 事故的坏副本 / 抢救产物)
- tmp/im16/gw/centrifugo 二进制(63.9 M,可重下)+ 缓存残留

入库范围:常驻规则(CODEBUDDY.md / README.md / state.py)、在途接续入口与
接续包、docs/、交付物/、交接单/、归档/、scripts/、.codebuddy/、
.workbuddy/memory/;共 398 件,其中 >60 KB 的 26 件全为文档。

排除(.gitignore):tmp/、待清理/、运行态日志与缓存、*.db 与 DB 备份整目录、
打包二进制(*.tar.gz / *.tgz)、记忆修复前备份。
2026-09-24 07:51:03 +08:00

83 lines
3.2 KiB
Python

# -*- coding: utf-8 -*-
"""在全部转录里按 ai-title 找「评估代码清理影响并整理」,并顺便给每个会话的身份卡。"""
import io
import json
import os
import re
ROOT = r"E:\ProgramData\.workbuddy\projects"
KW = ["清理", "评估", "整理"]
out = []
for proj in os.listdir(ROOT):
d = os.path.join(ROOT, proj)
if not os.path.isdir(d):
continue
for fn in os.listdir(d):
if not fn.endswith(".jsonl"):
continue
p = os.path.join(d, fn)
titles = []
n_user = 0
first_real = None
try:
with io.open(p, encoding="utf-8", errors="replace") as f:
for ln in f:
if '"ai-title"' not in ln and "<user_query>" not in ln:
continue
try:
o = json.loads(ln)
except Exception:
continue
if o.get("type") == "ai-title":
titles.append(o.get("aiTitle", ""))
if o.get("type") == "message" and o.get("role") == "user":
t = o.get("content")
s = "".join(x.get("text", "") for x in t if isinstance(x, dict)) if isinstance(t, list) else str(t)
if "<user_query>" in s and first_real is None:
first_real = " ".join(s.split())[:90]
n_user += 1
except Exception:
continue
blob = " ".join(titles)
if any(k in blob for k in KW):
out.append((os.path.getsize(p), proj, fn, titles, n_user, first_real))
print("=" * 78)
print("按标题命中「清理/评估/整理」的会话")
print("=" * 78)
for size, proj, fn, titles, nu, fr in sorted(out, key=lambda x: -x[0]):
print(" %s %6.1f MB user=%d" % (fn[:8], size / 1024 / 1024, nu))
print(" 工作区: %s" % proj)
print(" 标题 : %s" % titles)
print(" 首问 : %s" % fr)
print()
print("=" * 78)
print("本工作区全部会话的身份卡(标题 + 首问),供交叉确认")
print("=" * 78)
d = os.path.join(ROOT, "e-ProgramData-AI技能-aliyun-dsh-server")
for fn in sorted(os.listdir(d)):
if not fn.endswith(".jsonl"):
continue
p = os.path.join(d, fn)
titles, fr, last = [], None, None
with io.open(p, encoding="utf-8", errors="replace") as f:
for ln in f:
if '"ai-title"' not in ln and '"<user_query>"' not in ln:
continue
try:
o = json.loads(ln)
except Exception:
continue
if o.get("type") == "ai-title":
titles.append(o.get("aiTitle", ""))
if o.get("type") == "message" and o.get("role") == "user":
t = o.get("content")
s = "".join(x.get("text", "") for x in t if isinstance(x, dict)) if isinstance(t, list) else str(t)
if "<user_query>" in s and fr is None:
fr = " ".join(s.split())[:80]
last = o.get("timestamp")
print(" %s %5.1f MB %s" % (fn[:8], os.path.getsize(p) / 1024 / 1024, titles))
print(" 首问: %s" % fr)
print(" 末条 ts: %s" % last)