回收 411 MB(470 M → 58.8 M),全部经回收站,可恢复: - 待清理/(146.2 M,含 relay 分片 128 M 与 42 项过程目录) - tmp/(32.4 M,按接续棒命名的过程临时区) - .workbuddy/tmp/(39.5 M) - 4 份 workbuddy.db 冗余副本(101 M,09-23 事故的坏副本 / 抢救产物) - tmp/im16/gw/centrifugo 二进制(63.9 M,可重下)+ 缓存残留 入库范围:常驻规则(CODEBUDDY.md / README.md / state.py)、在途接续入口与 接续包、docs/、交付物/、交接单/、归档/、scripts/、.codebuddy/、 .workbuddy/memory/;共 398 件,其中 >60 KB 的 26 件全为文档。 排除(.gitignore):tmp/、待清理/、运行态日志与缓存、*.db 与 DB 备份整目录、 打包二进制(*.tar.gz / *.tgz)、记忆修复前备份。
83 lines
3.2 KiB
Python
83 lines
3.2 KiB
Python
# -*- coding: utf-8 -*-
|
|
"""在全部转录里按 ai-title 找「评估代码清理影响并整理」,并顺便给每个会话的身份卡。"""
|
|
import io
|
|
import json
|
|
import os
|
|
import re
|
|
|
|
ROOT = r"E:\ProgramData\.workbuddy\projects"
|
|
KW = ["清理", "评估", "整理"]
|
|
out = []
|
|
for proj in os.listdir(ROOT):
|
|
d = os.path.join(ROOT, proj)
|
|
if not os.path.isdir(d):
|
|
continue
|
|
for fn in os.listdir(d):
|
|
if not fn.endswith(".jsonl"):
|
|
continue
|
|
p = os.path.join(d, fn)
|
|
titles = []
|
|
n_user = 0
|
|
first_real = None
|
|
try:
|
|
with io.open(p, encoding="utf-8", errors="replace") as f:
|
|
for ln in f:
|
|
if '"ai-title"' not in ln and "<user_query>" not in ln:
|
|
continue
|
|
try:
|
|
o = json.loads(ln)
|
|
except Exception:
|
|
continue
|
|
if o.get("type") == "ai-title":
|
|
titles.append(o.get("aiTitle", ""))
|
|
if o.get("type") == "message" and o.get("role") == "user":
|
|
t = o.get("content")
|
|
s = "".join(x.get("text", "") for x in t if isinstance(x, dict)) if isinstance(t, list) else str(t)
|
|
if "<user_query>" in s and first_real is None:
|
|
first_real = " ".join(s.split())[:90]
|
|
n_user += 1
|
|
except Exception:
|
|
continue
|
|
blob = " ".join(titles)
|
|
if any(k in blob for k in KW):
|
|
out.append((os.path.getsize(p), proj, fn, titles, n_user, first_real))
|
|
|
|
print("=" * 78)
|
|
print("按标题命中「清理/评估/整理」的会话")
|
|
print("=" * 78)
|
|
for size, proj, fn, titles, nu, fr in sorted(out, key=lambda x: -x[0]):
|
|
print(" %s %6.1f MB user=%d" % (fn[:8], size / 1024 / 1024, nu))
|
|
print(" 工作区: %s" % proj)
|
|
print(" 标题 : %s" % titles)
|
|
print(" 首问 : %s" % fr)
|
|
print()
|
|
|
|
print("=" * 78)
|
|
print("本工作区全部会话的身份卡(标题 + 首问),供交叉确认")
|
|
print("=" * 78)
|
|
d = os.path.join(ROOT, "e-ProgramData-AI技能-aliyun-dsh-server")
|
|
for fn in sorted(os.listdir(d)):
|
|
if not fn.endswith(".jsonl"):
|
|
continue
|
|
p = os.path.join(d, fn)
|
|
titles, fr, last = [], None, None
|
|
with io.open(p, encoding="utf-8", errors="replace") as f:
|
|
for ln in f:
|
|
if '"ai-title"' not in ln and '"<user_query>"' not in ln:
|
|
continue
|
|
try:
|
|
o = json.loads(ln)
|
|
except Exception:
|
|
continue
|
|
if o.get("type") == "ai-title":
|
|
titles.append(o.get("aiTitle", ""))
|
|
if o.get("type") == "message" and o.get("role") == "user":
|
|
t = o.get("content")
|
|
s = "".join(x.get("text", "") for x in t if isinstance(x, dict)) if isinstance(t, list) else str(t)
|
|
if "<user_query>" in s and fr is None:
|
|
fr = " ".join(s.split())[:80]
|
|
last = o.get("timestamp")
|
|
print(" %s %5.1f MB %s" % (fn[:8], os.path.getsize(p) / 1024 / 1024, titles))
|
|
print(" 首问: %s" % fr)
|
|
print(" 末条 ts: %s" % last)
|