1) dsh-server-docs/ 从工作区(原 E:\...\aliyun-dsh-server\dsh-server-docs)**整体并入本仓**,
保留目录名 ⇒ 仓库内 dsh-server-docs/... 的相对引用天然继续有效;旧目录(含其 .git)已归档到
工作区 _中间产物_待清理/,未随本提交带入。
2) .gitattributes:新增 `dsh-server-docs/** -text` —— 原文档库是 `* -text` + autocrlf=false,
必须保持纯 LF,否则会被本仓的 CRLF 规则翻掉。
3) 活引用里的绝对路径已全部改到新位置(docs 的 INDEX / README / scripts / skills + 用户级 skills
+ ~/.workbuddy/settings.json 的 hooks);历史档案(04-调整方案/、archive/)按「只增不改」未动。
⚠️ hooks 路径改动需「完全重启会话」才生效(配置是会话启动快照)。
4) 交接单/T08:新增 §16「生产整体切换执行记录」(形态 / 落地动作 / **4 个只有真上线才暴露的真 bug** /
验收证据 / 回滚命令 / 残留项);台账 T08 行 → 已完成并归档;03-路线图 §二 登记 T08 收尾项。
5) 统一称谓:**「本机」只指跑 WorkBuddy 的开发机**,47 / 106 一律写「远程服务器」。
176 lines
6.6 KiB
Python
176 lines
6.6 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
extract-user-voice.py —— 把一个工作区的**全部历史会话**里的「用户真实发言」抽成一份可读清单。
|
||
|
||
为什么要它:做「决策方法 / 协作方式」复盘时,需要用**用户的原话全集**做底料,
|
||
而不是只靠档案与日志(那是会话的结构化沉淀,会丢掉"用户否决了什么、纠正了什么")。
|
||
本脚本只读、不改任何会话文件。
|
||
|
||
输出到 stdout(可重定向到自己想放的路径;**不要写进文档库目录**,否则会被 docs-sync-check 计成"仅本地"):
|
||
|
||
python3 scripts/extract-user-voice.py # 自动定位当前工作区
|
||
python3 scripts/extract-user-voice.py --project <dir名> # 指定 ~/.workbuddy/projects/<dir名>
|
||
python3 scripts/extract-user-voice.py --full # 打印全文(默认每条截断 88 字)
|
||
python3 scripts/extract-user-voice.py --needle 关键词 # 只列含关键词的发言(找某个决策的来龙去脉)
|
||
|
||
会话记录位置(WorkBuddy):`<配置目录>/projects/<把 cwd 的 : \\ / 换成 ->/<sessionId>.jsonl`
|
||
- 配置目录判定链:`WORKBUDDY_CONFIG_DIR ?? CODEBUDDY_CONFIG_DIR ?? ~/.workbuddy`
|
||
- 行格式:`{"type":"message","role":"user","content":[{"type":"input_text","text":"..."}]}`
|
||
- ⚠️ 用户发言里会带一大段 `<system-reminder ...>` 前言(user_info / identity_context),
|
||
必须剥掉才是**用户真实说的话**。
|
||
"""
|
||
import argparse
|
||
import datetime
|
||
import glob
|
||
import io
|
||
import json
|
||
import os
|
||
import re
|
||
import sys
|
||
|
||
SREM = re.compile(r'<system-reminder.*?</system-reminder>', re.S)
|
||
TAGS = re.compile(r'</?(user_query|cb_summary|conversation_history_summary)>')
|
||
WS = re.compile(r'\s+')
|
||
|
||
|
||
def config_dir():
|
||
for k in ('WORKBUDDY_CONFIG_DIR', 'CODEBUDDY_CONFIG_DIR'):
|
||
v = os.environ.get(k)
|
||
if v:
|
||
return os.path.expanduser(v)
|
||
return os.path.join(os.path.expanduser('~'), '.workbuddy')
|
||
|
||
|
||
def project_dir_name(cwd):
|
||
"""WorkBuddy 的目录名规则(实测):盘符**小写** + ':' 去掉,随后把分隔符各换成 '-'。
|
||
例:E:\\ProgramData\\AI技能\\aliyun-dsh-server → e-ProgramData-AI技能-aliyun-dsh-server
|
||
(路径中段的字母大小写与空格**保留**)。"""
|
||
p = os.path.abspath(cwd)
|
||
drive, rest = os.path.splitdrive(p)
|
||
return (drive.rstrip(':').lower() + re.sub(r'[\\/]', '-', rest)).strip('-')
|
||
|
||
|
||
def _norm(s):
|
||
"""把目录名归一化,用于模糊比较(忽略盘符大小写、多余连字符)。"""
|
||
return re.sub(r'-+', '-', s).lower()
|
||
|
||
|
||
def resolve_project(root, cwd, explicit):
|
||
"""返回 (目录名, 绝对路径)。显式 --project 只用它;否则从 cwd 起**逐级向上**试,
|
||
全部落空后再对 projects/ 下的实际目录做一次归一化模糊匹配。"""
|
||
if explicit:
|
||
return (explicit, os.path.join(root, explicit))
|
||
|
||
tried = []
|
||
cur = os.path.abspath(cwd)
|
||
while True:
|
||
name = project_dir_name(cur)
|
||
tried.append(name)
|
||
p = os.path.join(root, name)
|
||
if os.path.isdir(p):
|
||
return (name, p)
|
||
parent = os.path.dirname(cur)
|
||
if parent == cur:
|
||
break
|
||
cur = parent
|
||
|
||
# 模糊兜底:按归一化名比对实际存在的目录
|
||
try:
|
||
existing = [d for d in os.listdir(root) if os.path.isdir(os.path.join(root, d))]
|
||
except OSError:
|
||
existing = []
|
||
for want in tried:
|
||
for d in existing:
|
||
if _norm(d) == _norm(want):
|
||
return (d, os.path.join(root, d))
|
||
return (tried[0], os.path.join(root, tried[0]))
|
||
|
||
|
||
def fmt_ts(v):
|
||
if isinstance(v, (int, float)):
|
||
v = v / 1000.0 if v > 1e11 else v
|
||
try:
|
||
return datetime.datetime.fromtimestamp(v).strftime('%m-%d %H:%M')
|
||
except Exception:
|
||
return '?'
|
||
return str(v)[:16] if v else '?'
|
||
|
||
|
||
def user_text(line):
|
||
"""从一行 jsonl 里取出「用户真实发言」;不是用户消息则返回 None。"""
|
||
try:
|
||
o = json.loads(line)
|
||
except Exception:
|
||
return None
|
||
if o.get('type') != 'message' or o.get('role') != 'user':
|
||
return None
|
||
txt = ''
|
||
for b in (o.get('content') or []):
|
||
if isinstance(b, dict) and b.get('type') in ('input_text', 'text'):
|
||
txt += b.get('text', '')
|
||
txt = SREM.sub('', txt).strip()
|
||
txt = TAGS.sub('', txt).strip()
|
||
txt = WS.sub(' ', txt)
|
||
return (fmt_ts(o.get('timestamp')), txt) if txt else None
|
||
|
||
|
||
def main():
|
||
ap = argparse.ArgumentParser()
|
||
ap.add_argument('--project', help='projects 下的目录名(默认按当前 cwd 推导)')
|
||
ap.add_argument('--cwd', default=os.getcwd(), help='用于推导项目名的 cwd')
|
||
ap.add_argument('--full', action='store_true', help='打印全文(不截断)')
|
||
ap.add_argument('--needle', help='只列含该关键词的发言')
|
||
a = ap.parse_args()
|
||
|
||
root = os.path.join(config_dir(), 'projects')
|
||
name, pdir = resolve_project(root, a.cwd, a.project)
|
||
if not os.path.isdir(pdir):
|
||
sys.stderr.write('找不到项目会话目录:%s\n' % pdir)
|
||
sys.stderr.write('可用目录(%s):\n' % root)
|
||
for d in sorted(os.listdir(root))[:40]:
|
||
sys.stderr.write(' %s\n' % d)
|
||
return 2
|
||
|
||
files = sorted(glob.glob(os.path.join(pdir, '*.jsonl')))
|
||
if not files:
|
||
sys.stderr.write('该目录下没有 .jsonl 会话文件:%s\n' % pdir)
|
||
return 2
|
||
|
||
total = 0
|
||
print('项目会话目录:%s' % pdir)
|
||
print('会话文件 %d 个\n' % len(files))
|
||
|
||
print('===== 各会话概览 =====')
|
||
per = []
|
||
for f in files:
|
||
msgs = []
|
||
with io.open(f, encoding='utf-8', errors='replace') as fh:
|
||
for line in fh:
|
||
r = user_text(line)
|
||
if r:
|
||
msgs.append(r)
|
||
per.append((os.path.basename(f[:-6]), msgs))
|
||
total += len(msgs)
|
||
span = ('%s → %s' % (msgs[0][0], msgs[-1][0])) if msgs else '-'
|
||
print(' %-38s 用户发言 %4d 条 %s' % (os.path.basename(f)[:38], len(msgs), span))
|
||
print('\n合计用户真实发言 = %d 条\n' % total)
|
||
|
||
for sid, msgs in per:
|
||
if a.needle:
|
||
msgs = [m for m in msgs if a.needle in m[1]]
|
||
if not msgs:
|
||
continue
|
||
print('===== %s(命中 %d 条)=====' % (sid, len(msgs)))
|
||
else:
|
||
print('===== %s(%d 条)=====' % (sid, len(msgs)))
|
||
for i, (t, m) in enumerate(msgs, 1):
|
||
body = m if a.full else (m[:88] + ('…' if len(m) > 88 else ''))
|
||
print('%4d %s %s' % (i, t, body))
|
||
print()
|
||
return 0
|
||
|
||
|
||
if __name__ == '__main__':
|
||
sys.exit(main())
|