Files
dsh_ai1net_server/.workbuddy/tools/find-session-by-title.py
T
admin c1b5e4d966 chore(工作区): 全量入库 + 补齐 .gitignore(以工作区为准)
- 变更规模:新增 514 / 修改 62 / 重命名 155 / 删除 4(归档重组与文档轮次)
- .gitignore 修:`归档/**/db-cwd归一-备份-*/` —— 原规则写绝对层级(归档/db-cwd归一-…),
  目录搬进 归档/配置与备份/ 后**静默失效**,43 MB 的 DB 备份又变成未跟踪
- .gitignore 补:嵌套 git 内部数据(归档/内嵌git-20261008/、归档/skills-git-旧线-20261007/dotgit-原样移出/)
- .gitignore 补:运行态与部署副本(.workbuddy/collab/、.workbuddy/tools/、.workbuddy/.load-pending、.workbuddy/tmp-*)
- .gitignore 补:备份件(*.bak-*)
- 未跟踪文件从 2190 降到 890(其余为 归档/ 归档件与 .workbuddy/memory/ 知识文件,按口径入库)
2026-10-10 23:13:22 +08:00

83 lines
3.2 KiB
Python

# -*- coding: utf-8 -*-
"""在全部转录里按 ai-title 找「评估代码清理影响并整理」,并顺便给每个会话的身份卡。"""
import io
import json
import os
import re
ROOT = r"E:\ProgramData\.workbuddy\projects"
KW = ["清理", "评估", "整理"]
out = []
for proj in os.listdir(ROOT):
d = os.path.join(ROOT, proj)
if not os.path.isdir(d):
continue
for fn in os.listdir(d):
if not fn.endswith(".jsonl"):
continue
p = os.path.join(d, fn)
titles = []
n_user = 0
first_real = None
try:
with io.open(p, encoding="utf-8", errors="replace") as f:
for ln in f:
if '"ai-title"' not in ln and "<user_query>" not in ln:
continue
try:
o = json.loads(ln)
except Exception:
continue
if o.get("type") == "ai-title":
titles.append(o.get("aiTitle", ""))
if o.get("type") == "message" and o.get("role") == "user":
t = o.get("content")
s = "".join(x.get("text", "") for x in t if isinstance(x, dict)) if isinstance(t, list) else str(t)
if "<user_query>" in s and first_real is None:
first_real = " ".join(s.split())[:90]
n_user += 1
except Exception:
continue
blob = " ".join(titles)
if any(k in blob for k in KW):
out.append((os.path.getsize(p), proj, fn, titles, n_user, first_real))
print("=" * 78)
print("按标题命中「清理/评估/整理」的会话")
print("=" * 78)
for size, proj, fn, titles, nu, fr in sorted(out, key=lambda x: -x[0]):
print(" %s %6.1f MB user=%d" % (fn[:8], size / 1024 / 1024, nu))
print(" 工作区: %s" % proj)
print(" 标题 : %s" % titles)
print(" 首问 : %s" % fr)
print()
print("=" * 78)
print("本工作区全部会话的身份卡(标题 + 首问),供交叉确认")
print("=" * 78)
d = os.path.join(ROOT, "e-ProgramData-AIProject-ai1net-dsh-server")
for fn in sorted(os.listdir(d)):
if not fn.endswith(".jsonl"):
continue
p = os.path.join(d, fn)
titles, fr, last = [], None, None
with io.open(p, encoding="utf-8", errors="replace") as f:
for ln in f:
if '"ai-title"' not in ln and '"<user_query>"' not in ln:
continue
try:
o = json.loads(ln)
except Exception:
continue
if o.get("type") == "ai-title":
titles.append(o.get("aiTitle", ""))
if o.get("type") == "message" and o.get("role") == "user":
t = o.get("content")
s = "".join(x.get("text", "") for x in t if isinstance(x, dict)) if isinstance(t, list) else str(t)
if "<user_query>" in s and fr is None:
fr = " ".join(s.split())[:80]
last = o.get("timestamp")
print(" %s %5.1f MB %s" % (fn[:8], os.path.getsize(p) / 1024 / 1024, titles))
print(" 首问: %s" % fr)
print(" 末条 ts: %s" % last)