2026-09-15 18:47:13 +08:00
|
|
|
|
#!/usr/bin/env python3
|
|
|
|
|
|
# -*- coding: utf-8 -*-
|
|
|
|
|
|
"""
|
|
|
|
|
|
extract-user-voice.py —— 把一个工作区的**全部历史会话**里的「用户真实发言」抽成一份可读清单。
|
|
|
|
|
|
|
|
|
|
|
|
为什么要它:做「决策方法 / 协作方式」复盘时,需要用**用户的原话全集**做底料,
|
|
|
|
|
|
而不是只靠档案与日志(那是会话的结构化沉淀,会丢掉"用户否决了什么、纠正了什么")。
|
|
|
|
|
|
本脚本只读、不改任何会话文件。
|
|
|
|
|
|
|
|
|
|
|
|
输出到 stdout(可重定向到自己想放的路径;**不要写进文档库目录**,否则会被 docs-sync-check 计成"仅本地"):
|
|
|
|
|
|
|
2026-09-24 07:25:16 +08:00
|
|
|
|
python3 07-scripts/extract-user-voice.py # 自动定位当前工作区
|
|
|
|
|
|
python3 07-scripts/extract-user-voice.py --project <dir名> # 指定 ~/.workbuddy/projects/<dir名>
|
|
|
|
|
|
python3 07-scripts/extract-user-voice.py --full # 打印全文(默认每条截断 88 字)
|
|
|
|
|
|
python3 07-scripts/extract-user-voice.py --needle 关键词 # 只列含关键词的发言(找某个决策的来龙去脉)
|
2026-09-15 18:47:13 +08:00
|
|
|
|
|
|
|
|
|
|
会话记录位置(WorkBuddy):`<配置目录>/projects/<把 cwd 的 : \\ / 换成 ->/<sessionId>.jsonl`
|
|
|
|
|
|
- 配置目录判定链:`WORKBUDDY_CONFIG_DIR ?? CODEBUDDY_CONFIG_DIR ?? ~/.workbuddy`
|
|
|
|
|
|
- 行格式:`{"type":"message","role":"user","content":[{"type":"input_text","text":"..."}]}`
|
|
|
|
|
|
- ⚠️ 用户发言里会带一大段 `<system-reminder ...>` 前言(user_info / identity_context),
|
|
|
|
|
|
必须剥掉才是**用户真实说的话**。
|
|
|
|
|
|
"""
|
|
|
|
|
|
import argparse
|
|
|
|
|
|
import datetime
|
|
|
|
|
|
import glob
|
|
|
|
|
|
import io
|
|
|
|
|
|
import json
|
|
|
|
|
|
import os
|
|
|
|
|
|
import re
|
|
|
|
|
|
import sys
|
|
|
|
|
|
|
|
|
|
|
|
SREM = re.compile(r'<system-reminder.*?</system-reminder>', re.S)
|
|
|
|
|
|
TAGS = re.compile(r'</?(user_query|cb_summary|conversation_history_summary)>')
|
|
|
|
|
|
WS = re.compile(r'\s+')
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def config_dir():
|
|
|
|
|
|
for k in ('WORKBUDDY_CONFIG_DIR', 'CODEBUDDY_CONFIG_DIR'):
|
|
|
|
|
|
v = os.environ.get(k)
|
|
|
|
|
|
if v:
|
|
|
|
|
|
return os.path.expanduser(v)
|
|
|
|
|
|
return os.path.join(os.path.expanduser('~'), '.workbuddy')
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def project_dir_name(cwd):
|
|
|
|
|
|
"""WorkBuddy 的目录名规则(实测):盘符**小写** + ':' 去掉,随后把分隔符各换成 '-'。
|
|
|
|
|
|
例:E:\\ProgramData\\AI技能\\aliyun-dsh-server → e-ProgramData-AI技能-aliyun-dsh-server
|
|
|
|
|
|
(路径中段的字母大小写与空格**保留**)。"""
|
|
|
|
|
|
p = os.path.abspath(cwd)
|
|
|
|
|
|
drive, rest = os.path.splitdrive(p)
|
|
|
|
|
|
return (drive.rstrip(':').lower() + re.sub(r'[\\/]', '-', rest)).strip('-')
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _norm(s):
|
|
|
|
|
|
"""把目录名归一化,用于模糊比较(忽略盘符大小写、多余连字符)。"""
|
|
|
|
|
|
return re.sub(r'-+', '-', s).lower()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def resolve_project(root, cwd, explicit):
|
|
|
|
|
|
"""返回 (目录名, 绝对路径)。显式 --project 只用它;否则从 cwd 起**逐级向上**试,
|
|
|
|
|
|
全部落空后再对 projects/ 下的实际目录做一次归一化模糊匹配。"""
|
|
|
|
|
|
if explicit:
|
|
|
|
|
|
return (explicit, os.path.join(root, explicit))
|
|
|
|
|
|
|
|
|
|
|
|
tried = []
|
|
|
|
|
|
cur = os.path.abspath(cwd)
|
|
|
|
|
|
while True:
|
|
|
|
|
|
name = project_dir_name(cur)
|
|
|
|
|
|
tried.append(name)
|
|
|
|
|
|
p = os.path.join(root, name)
|
|
|
|
|
|
if os.path.isdir(p):
|
|
|
|
|
|
return (name, p)
|
|
|
|
|
|
parent = os.path.dirname(cur)
|
|
|
|
|
|
if parent == cur:
|
|
|
|
|
|
break
|
|
|
|
|
|
cur = parent
|
|
|
|
|
|
|
|
|
|
|
|
# 模糊兜底:按归一化名比对实际存在的目录
|
|
|
|
|
|
try:
|
|
|
|
|
|
existing = [d for d in os.listdir(root) if os.path.isdir(os.path.join(root, d))]
|
|
|
|
|
|
except OSError:
|
|
|
|
|
|
existing = []
|
|
|
|
|
|
for want in tried:
|
|
|
|
|
|
for d in existing:
|
|
|
|
|
|
if _norm(d) == _norm(want):
|
|
|
|
|
|
return (d, os.path.join(root, d))
|
|
|
|
|
|
return (tried[0], os.path.join(root, tried[0]))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def fmt_ts(v):
|
|
|
|
|
|
if isinstance(v, (int, float)):
|
|
|
|
|
|
v = v / 1000.0 if v > 1e11 else v
|
|
|
|
|
|
try:
|
|
|
|
|
|
return datetime.datetime.fromtimestamp(v).strftime('%m-%d %H:%M')
|
|
|
|
|
|
except Exception:
|
|
|
|
|
|
return '?'
|
|
|
|
|
|
return str(v)[:16] if v else '?'
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def user_text(line):
|
|
|
|
|
|
"""从一行 jsonl 里取出「用户真实发言」;不是用户消息则返回 None。"""
|
|
|
|
|
|
try:
|
|
|
|
|
|
o = json.loads(line)
|
|
|
|
|
|
except Exception:
|
|
|
|
|
|
return None
|
|
|
|
|
|
if o.get('type') != 'message' or o.get('role') != 'user':
|
|
|
|
|
|
return None
|
|
|
|
|
|
txt = ''
|
|
|
|
|
|
for b in (o.get('content') or []):
|
|
|
|
|
|
if isinstance(b, dict) and b.get('type') in ('input_text', 'text'):
|
|
|
|
|
|
txt += b.get('text', '')
|
|
|
|
|
|
txt = SREM.sub('', txt).strip()
|
|
|
|
|
|
txt = TAGS.sub('', txt).strip()
|
|
|
|
|
|
txt = WS.sub(' ', txt)
|
|
|
|
|
|
return (fmt_ts(o.get('timestamp')), txt) if txt else None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def main():
|
|
|
|
|
|
ap = argparse.ArgumentParser()
|
|
|
|
|
|
ap.add_argument('--project', help='projects 下的目录名(默认按当前 cwd 推导)')
|
|
|
|
|
|
ap.add_argument('--cwd', default=os.getcwd(), help='用于推导项目名的 cwd')
|
|
|
|
|
|
ap.add_argument('--full', action='store_true', help='打印全文(不截断)')
|
|
|
|
|
|
ap.add_argument('--needle', help='只列含该关键词的发言')
|
|
|
|
|
|
a = ap.parse_args()
|
|
|
|
|
|
|
|
|
|
|
|
root = os.path.join(config_dir(), 'projects')
|
|
|
|
|
|
name, pdir = resolve_project(root, a.cwd, a.project)
|
|
|
|
|
|
if not os.path.isdir(pdir):
|
|
|
|
|
|
sys.stderr.write('找不到项目会话目录:%s\n' % pdir)
|
|
|
|
|
|
sys.stderr.write('可用目录(%s):\n' % root)
|
|
|
|
|
|
for d in sorted(os.listdir(root))[:40]:
|
|
|
|
|
|
sys.stderr.write(' %s\n' % d)
|
|
|
|
|
|
return 2
|
|
|
|
|
|
|
|
|
|
|
|
files = sorted(glob.glob(os.path.join(pdir, '*.jsonl')))
|
|
|
|
|
|
if not files:
|
|
|
|
|
|
sys.stderr.write('该目录下没有 .jsonl 会话文件:%s\n' % pdir)
|
|
|
|
|
|
return 2
|
|
|
|
|
|
|
|
|
|
|
|
total = 0
|
|
|
|
|
|
print('项目会话目录:%s' % pdir)
|
|
|
|
|
|
print('会话文件 %d 个\n' % len(files))
|
|
|
|
|
|
|
|
|
|
|
|
print('===== 各会话概览 =====')
|
|
|
|
|
|
per = []
|
|
|
|
|
|
for f in files:
|
|
|
|
|
|
msgs = []
|
|
|
|
|
|
with io.open(f, encoding='utf-8', errors='replace') as fh:
|
|
|
|
|
|
for line in fh:
|
|
|
|
|
|
r = user_text(line)
|
|
|
|
|
|
if r:
|
|
|
|
|
|
msgs.append(r)
|
|
|
|
|
|
per.append((os.path.basename(f[:-6]), msgs))
|
|
|
|
|
|
total += len(msgs)
|
|
|
|
|
|
span = ('%s → %s' % (msgs[0][0], msgs[-1][0])) if msgs else '-'
|
|
|
|
|
|
print(' %-38s 用户发言 %4d 条 %s' % (os.path.basename(f)[:38], len(msgs), span))
|
|
|
|
|
|
print('\n合计用户真实发言 = %d 条\n' % total)
|
|
|
|
|
|
|
|
|
|
|
|
for sid, msgs in per:
|
|
|
|
|
|
if a.needle:
|
|
|
|
|
|
msgs = [m for m in msgs if a.needle in m[1]]
|
|
|
|
|
|
if not msgs:
|
|
|
|
|
|
continue
|
|
|
|
|
|
print('===== %s(命中 %d 条)=====' % (sid, len(msgs)))
|
|
|
|
|
|
else:
|
|
|
|
|
|
print('===== %s(%d 条)=====' % (sid, len(msgs)))
|
|
|
|
|
|
for i, (t, m) in enumerate(msgs, 1):
|
|
|
|
|
|
body = m if a.full else (m[:88] + ('…' if len(m) > 88 else ''))
|
|
|
|
|
|
print('%4d %s %s' % (i, t, body))
|
|
|
|
|
|
print()
|
|
|
|
|
|
return 0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
if __name__ == '__main__':
|
|
|
|
|
|
sys.exit(main())
|