Files
dsh_shenxian/dsh-server-docs/scripts/extract-user-voice.py
T
admin 5ad755116e chore(docs): 文档库并入代码仓(R4 选 a)+ 索引/台账跟进
1) dsh-server-docs/ 从工作区(原 E:\...\aliyun-dsh-server\dsh-server-docs)**整体并入本仓**,
   保留目录名 ⇒ 仓库内 dsh-server-docs/... 的相对引用天然继续有效;旧目录(含其 .git)已归档到
   工作区 _中间产物_待清理/,未随本提交带入。
2) .gitattributes:新增 `dsh-server-docs/** -text` —— 原文档库是 `* -text` + autocrlf=false,
   必须保持纯 LF,否则会被本仓的 CRLF 规则翻掉。
3) 活引用里的绝对路径已全部改到新位置(docs 的 INDEX / README / scripts / skills + 用户级 skills
   + ~/.workbuddy/settings.json 的 hooks);历史档案(04-调整方案/、archive/)按「只增不改」未动。
   ⚠️ hooks 路径改动需「完全重启会话」才生效(配置是会话启动快照)。
4) 交接单/T08:新增 §16「生产整体切换执行记录」(形态 / 落地动作 / **4 个只有真上线才暴露的真 bug** /
   验收证据 / 回滚命令 / 残留项);台账 T08 行 → 已完成并归档;03-路线图 §二 登记 T08 收尾项。
5) 统一称谓:**「本机」只指跑 WorkBuddy 的开发机**,47 / 106 一律写「远程服务器」。
2026-09-15 18:47:13 +08:00

176 lines
6.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
extract-user-voice.py —— 把一个工作区的**全部历史会话**里的「用户真实发言」抽成一份可读清单。
为什么要它:做「决策方法 / 协作方式」复盘时,需要用**用户的原话全集**做底料,
而不是只靠档案与日志(那是会话的结构化沉淀,会丢掉"用户否决了什么、纠正了什么")。
本脚本只读、不改任何会话文件。
输出到 stdout(可重定向到自己想放的路径;**不要写进文档库目录**,否则会被 docs-sync-check 计成"仅本地"):
python3 scripts/extract-user-voice.py # 自动定位当前工作区
python3 scripts/extract-user-voice.py --project <dir名> # 指定 ~/.workbuddy/projects/<dir名>
python3 scripts/extract-user-voice.py --full # 打印全文(默认每条截断 88 字)
python3 scripts/extract-user-voice.py --needle 关键词 # 只列含关键词的发言(找某个决策的来龙去脉)
会话记录位置(WorkBuddy):`<配置目录>/projects/<把 cwd 的 : \\ / 换成 ->/<sessionId>.jsonl`
- 配置目录判定链:`WORKBUDDY_CONFIG_DIR ?? CODEBUDDY_CONFIG_DIR ?? ~/.workbuddy`
- 行格式:`{"type":"message","role":"user","content":[{"type":"input_text","text":"..."}]}`
- ⚠️ 用户发言里会带一大段 `<system-reminder ...>` 前言(user_info / identity_context),
必须剥掉才是**用户真实说的话**。
"""
import argparse
import datetime
import glob
import io
import json
import os
import re
import sys
SREM = re.compile(r'<system-reminder.*?</system-reminder>', re.S)
TAGS = re.compile(r'</?(user_query|cb_summary|conversation_history_summary)>')
WS = re.compile(r'\s+')
def config_dir():
for k in ('WORKBUDDY_CONFIG_DIR', 'CODEBUDDY_CONFIG_DIR'):
v = os.environ.get(k)
if v:
return os.path.expanduser(v)
return os.path.join(os.path.expanduser('~'), '.workbuddy')
def project_dir_name(cwd):
"""WorkBuddy 的目录名规则(实测):盘符**小写** + ':' 去掉,随后把分隔符各换成 '-'。
例:E:\\ProgramData\\AI技能\\aliyun-dsh-server → e-ProgramData-AI技能-aliyun-dsh-server
(路径中段的字母大小写与空格**保留**)。"""
p = os.path.abspath(cwd)
drive, rest = os.path.splitdrive(p)
return (drive.rstrip(':').lower() + re.sub(r'[\\/]', '-', rest)).strip('-')
def _norm(s):
"""把目录名归一化,用于模糊比较(忽略盘符大小写、多余连字符)。"""
return re.sub(r'-+', '-', s).lower()
def resolve_project(root, cwd, explicit):
"""返回 (目录名, 绝对路径)。显式 --project 只用它;否则从 cwd 起**逐级向上**试,
全部落空后再对 projects/ 下的实际目录做一次归一化模糊匹配。"""
if explicit:
return (explicit, os.path.join(root, explicit))
tried = []
cur = os.path.abspath(cwd)
while True:
name = project_dir_name(cur)
tried.append(name)
p = os.path.join(root, name)
if os.path.isdir(p):
return (name, p)
parent = os.path.dirname(cur)
if parent == cur:
break
cur = parent
# 模糊兜底:按归一化名比对实际存在的目录
try:
existing = [d for d in os.listdir(root) if os.path.isdir(os.path.join(root, d))]
except OSError:
existing = []
for want in tried:
for d in existing:
if _norm(d) == _norm(want):
return (d, os.path.join(root, d))
return (tried[0], os.path.join(root, tried[0]))
def fmt_ts(v):
if isinstance(v, (int, float)):
v = v / 1000.0 if v > 1e11 else v
try:
return datetime.datetime.fromtimestamp(v).strftime('%m-%d %H:%M')
except Exception:
return '?'
return str(v)[:16] if v else '?'
def user_text(line):
"""从一行 jsonl 里取出「用户真实发言」;不是用户消息则返回 None。"""
try:
o = json.loads(line)
except Exception:
return None
if o.get('type') != 'message' or o.get('role') != 'user':
return None
txt = ''
for b in (o.get('content') or []):
if isinstance(b, dict) and b.get('type') in ('input_text', 'text'):
txt += b.get('text', '')
txt = SREM.sub('', txt).strip()
txt = TAGS.sub('', txt).strip()
txt = WS.sub(' ', txt)
return (fmt_ts(o.get('timestamp')), txt) if txt else None
def main():
ap = argparse.ArgumentParser()
ap.add_argument('--project', help='projects 下的目录名(默认按当前 cwd 推导)')
ap.add_argument('--cwd', default=os.getcwd(), help='用于推导项目名的 cwd')
ap.add_argument('--full', action='store_true', help='打印全文(不截断)')
ap.add_argument('--needle', help='只列含该关键词的发言')
a = ap.parse_args()
root = os.path.join(config_dir(), 'projects')
name, pdir = resolve_project(root, a.cwd, a.project)
if not os.path.isdir(pdir):
sys.stderr.write('找不到项目会话目录:%s\n' % pdir)
sys.stderr.write('可用目录(%s):\n' % root)
for d in sorted(os.listdir(root))[:40]:
sys.stderr.write(' %s\n' % d)
return 2
files = sorted(glob.glob(os.path.join(pdir, '*.jsonl')))
if not files:
sys.stderr.write('该目录下没有 .jsonl 会话文件:%s\n' % pdir)
return 2
total = 0
print('项目会话目录:%s' % pdir)
print('会话文件 %d 个\n' % len(files))
print('===== 各会话概览 =====')
per = []
for f in files:
msgs = []
with io.open(f, encoding='utf-8', errors='replace') as fh:
for line in fh:
r = user_text(line)
if r:
msgs.append(r)
per.append((os.path.basename(f[:-6]), msgs))
total += len(msgs)
span = ('%s → %s' % (msgs[0][0], msgs[-1][0])) if msgs else '-'
print(' %-38s 用户发言 %4d 条 %s' % (os.path.basename(f)[:38], len(msgs), span))
print('\n合计用户真实发言 = %d 条\n' % total)
for sid, msgs in per:
if a.needle:
msgs = [m for m in msgs if a.needle in m[1]]
if not msgs:
continue
print('===== %s(命中 %d 条)=====' % (sid, len(msgs)))
else:
print('===== %s(%d 条)=====' % (sid, len(msgs)))
for i, (t, m) in enumerate(msgs, 1):
body = m if a.full else (m[:88] + ('…' if len(m) > 88 else ''))
print('%4d %s %s' % (i, t, body))
print()
return 0
if __name__ == '__main__':
sys.exit(main())