Files
admin e6207aa691
build / build-and-scan (push) Waiting to run
chore(仓库对齐): 文档库结构治理 + IM/插件线落地
文档库:目录改为编号制(01-规范/02-架构设计/03-数据库/04-调整方案/
05-交接单/06-ops/07-scripts/08-skills/09-archive),顶层散文件归入 01-规范/;
INDEX.md 与 docs-manifest.json 重刷(档案 146 篇);旧目录名引用全量对齐。

IM 线:src/im/**(SDK / hub / store / presence / ws / gateway-token)、
src/web/routes/im.ts、src/db/plugin-data/**、src/supervisor/plugin-assembly.ts
及对应 test/**。

插件线:poc/{im-agent-bridge,im-connection-gateway,im-conversation-tabs,
business-plugins-im,carbon-mcp-probe}、src/web/routes/{sessions,overlay-device}.ts、
src/net/relay/{device-grant,instance-credential}.ts。

仓库卫生:清出 40 个历史误入库 / 已改名文件(34 个交接单归档 + 6 个旧结构,
本地均有副本);dsh-server-docs/.gitignore 补 tmp/;交接单不入库(政策)。
2026-09-24 07:25:16 +08:00

176 lines
6.6 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
extract-user-voice.py —— 把一个工作区的**全部历史会话**里的「用户真实发言」抽成一份可读清单。
为什么要它:做「决策方法 / 协作方式」复盘时,需要用**用户的原话全集**做底料,
而不是只靠档案与日志(那是会话的结构化沉淀,会丢掉"用户否决了什么、纠正了什么")。
本脚本只读、不改任何会话文件。
输出到 stdout(可重定向到自己想放的路径;**不要写进文档库目录**,否则会被 docs-sync-check 计成"仅本地"):
python3 07-scripts/extract-user-voice.py # 自动定位当前工作区
python3 07-scripts/extract-user-voice.py --project <dir名> # 指定 ~/.workbuddy/projects/<dir名>
python3 07-scripts/extract-user-voice.py --full # 打印全文(默认每条截断 88 字)
python3 07-scripts/extract-user-voice.py --needle 关键词 # 只列含关键词的发言(找某个决策的来龙去脉)
会话记录位置(WorkBuddy):`<配置目录>/projects/<把 cwd 的 : \\ / 换成 ->/<sessionId>.jsonl`
- 配置目录判定链:`WORKBUDDY_CONFIG_DIR ?? CODEBUDDY_CONFIG_DIR ?? ~/.workbuddy`
- 行格式:`{"type":"message","role":"user","content":[{"type":"input_text","text":"..."}]}`
- ⚠️ 用户发言里会带一大段 `<system-reminder ...>` 前言(user_info / identity_context),
必须剥掉才是**用户真实说的话**。
"""
import argparse
import datetime
import glob
import io
import json
import os
import re
import sys
SREM = re.compile(r'<system-reminder.*?</system-reminder>', re.S)
TAGS = re.compile(r'</?(user_query|cb_summary|conversation_history_summary)>')
WS = re.compile(r'\s+')
def config_dir():
for k in ('WORKBUDDY_CONFIG_DIR', 'CODEBUDDY_CONFIG_DIR'):
v = os.environ.get(k)
if v:
return os.path.expanduser(v)
return os.path.join(os.path.expanduser('~'), '.workbuddy')
def project_dir_name(cwd):
"""WorkBuddy 的目录名规则(实测):盘符**小写** + ':' 去掉,随后把分隔符各换成 '-'。
例:E:\\ProgramData\\AI技能\\aliyun-dsh-server → e-ProgramData-AI技能-aliyun-dsh-server
(路径中段的字母大小写与空格**保留**)。"""
p = os.path.abspath(cwd)
drive, rest = os.path.splitdrive(p)
return (drive.rstrip(':').lower() + re.sub(r'[\\/]', '-', rest)).strip('-')
def _norm(s):
"""把目录名归一化,用于模糊比较(忽略盘符大小写、多余连字符)。"""
return re.sub(r'-+', '-', s).lower()
def resolve_project(root, cwd, explicit):
"""返回 (目录名, 绝对路径)。显式 --project 只用它;否则从 cwd 起**逐级向上**试,
全部落空后再对 projects/ 下的实际目录做一次归一化模糊匹配。"""
if explicit:
return (explicit, os.path.join(root, explicit))
tried = []
cur = os.path.abspath(cwd)
while True:
name = project_dir_name(cur)
tried.append(name)
p = os.path.join(root, name)
if os.path.isdir(p):
return (name, p)
parent = os.path.dirname(cur)
if parent == cur:
break
cur = parent
# 模糊兜底:按归一化名比对实际存在的目录
try:
existing = [d for d in os.listdir(root) if os.path.isdir(os.path.join(root, d))]
except OSError:
existing = []
for want in tried:
for d in existing:
if _norm(d) == _norm(want):
return (d, os.path.join(root, d))
return (tried[0], os.path.join(root, tried[0]))
def fmt_ts(v):
if isinstance(v, (int, float)):
v = v / 1000.0 if v > 1e11 else v
try:
return datetime.datetime.fromtimestamp(v).strftime('%m-%d %H:%M')
except Exception:
return '?'
return str(v)[:16] if v else '?'
def user_text(line):
"""从一行 jsonl 里取出「用户真实发言」;不是用户消息则返回 None。"""
try:
o = json.loads(line)
except Exception:
return None
if o.get('type') != 'message' or o.get('role') != 'user':
return None
txt = ''
for b in (o.get('content') or []):
if isinstance(b, dict) and b.get('type') in ('input_text', 'text'):
txt += b.get('text', '')
txt = SREM.sub('', txt).strip()
txt = TAGS.sub('', txt).strip()
txt = WS.sub(' ', txt)
return (fmt_ts(o.get('timestamp')), txt) if txt else None
def main():
ap = argparse.ArgumentParser()
ap.add_argument('--project', help='projects 下的目录名(默认按当前 cwd 推导)')
ap.add_argument('--cwd', default=os.getcwd(), help='用于推导项目名的 cwd')
ap.add_argument('--full', action='store_true', help='打印全文(不截断)')
ap.add_argument('--needle', help='只列含该关键词的发言')
a = ap.parse_args()
root = os.path.join(config_dir(), 'projects')
name, pdir = resolve_project(root, a.cwd, a.project)
if not os.path.isdir(pdir):
sys.stderr.write('找不到项目会话目录:%s\n' % pdir)
sys.stderr.write('可用目录(%s):\n' % root)
for d in sorted(os.listdir(root))[:40]:
sys.stderr.write(' %s\n' % d)
return 2
files = sorted(glob.glob(os.path.join(pdir, '*.jsonl')))
if not files:
sys.stderr.write('该目录下没有 .jsonl 会话文件:%s\n' % pdir)
return 2
total = 0
print('项目会话目录:%s' % pdir)
print('会话文件 %d 个\n' % len(files))
print('===== 各会话概览 =====')
per = []
for f in files:
msgs = []
with io.open(f, encoding='utf-8', errors='replace') as fh:
for line in fh:
r = user_text(line)
if r:
msgs.append(r)
per.append((os.path.basename(f[:-6]), msgs))
total += len(msgs)
span = ('%s → %s' % (msgs[0][0], msgs[-1][0])) if msgs else '-'
print(' %-38s 用户发言 %4d 条 %s' % (os.path.basename(f)[:38], len(msgs), span))
print('\n合计用户真实发言 = %d 条\n' % total)
for sid, msgs in per:
if a.needle:
msgs = [m for m in msgs if a.needle in m[1]]
if not msgs:
continue
print('===== %s(命中 %d 条)=====' % (sid, len(msgs)))
else:
print('===== %s(%d 条)=====' % (sid, len(msgs)))
for i, (t, m) in enumerate(msgs, 1):
body = m if a.full else (m[:88] + ('…' if len(m) > 88 else ''))
print('%4d %s %s' % (i, t, body))
print()
return 0
if __name__ == '__main__':
sys.exit(main())