Files
dsh_shenxian/dsh-server-docs/scripts/docs-search.py
T
admin 5ad755116e chore(docs): 文档库并入代码仓(R4 选 a)+ 索引/台账跟进
1) dsh-server-docs/ 从工作区(原 E:\...\aliyun-dsh-server\dsh-server-docs)**整体并入本仓**,
   保留目录名 ⇒ 仓库内 dsh-server-docs/... 的相对引用天然继续有效;旧目录(含其 .git)已归档到
   工作区 _中间产物_待清理/,未随本提交带入。
2) .gitattributes:新增 `dsh-server-docs/** -text` —— 原文档库是 `* -text` + autocrlf=false,
   必须保持纯 LF,否则会被本仓的 CRLF 规则翻掉。
3) 活引用里的绝对路径已全部改到新位置(docs 的 INDEX / README / scripts / skills + 用户级 skills
   + ~/.workbuddy/settings.json 的 hooks);历史档案(04-调整方案/、archive/)按「只增不改」未动。
   ⚠️ hooks 路径改动需「完全重启会话」才生效(配置是会话启动快照)。
4) 交接单/T08:新增 §16「生产整体切换执行记录」(形态 / 落地动作 / **4 个只有真上线才暴露的真 bug** /
   验收证据 / 回滚命令 / 残留项);台账 T08 行 → 已完成并归档;03-路线图 §二 登记 T08 收尾项。
5) 统一称谓:**「本机」只指跑 WorkBuddy 的开发机**,47 / 106 一律写「远程服务器」。
2026-09-15 18:47:13 +08:00

143 lines
5.8 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""docs-search.py — 文档库**带语义的全文检索**(零依赖、不建索引、每次实时扫)
为什么需要它(2026-09-14 实测)
* 库已到 **118 篇 / 95 万字符 ≈ 67 万 token** ⇒ 不可能"读全库";
* 只能 `grep` 时,**搜到的结果分不清"现行值"还是"历史值"** —— 这正是 2026-09-12 踩过的坑
(旧配额活在 15 个文件里,被当成事实用)。
⇒ 本脚本把 `docs-manifest.json` 的 **(层 L0–L5 / 域 / tier / 状态)** 标注接进检索结果,
并给 `--current` 一键**排除历史层**,让"查现行事实"这件事**结果可判**。
用法
python3 scripts/docs-search.py 配额 # 全库搜「配额」
python3 scripts/docs-search.py 配额 --current # **只搜现行层(排除 L5 档案 / archive)**
python3 scripts/docs-search.py 插件 域 # 多词 = AND
python3 scripts/docs-search.py glibc --domain plugin --layer L5
python3 scripts/docs-search.py 内存 --json | jq . # 机读输出
选项
--current 只搜 L0–L4(排除 L5 与 archive/)—— **查现行值请默认加它**
--layer L1,L2 限定层(L0–L5)
--domain plugin 限定域(platform/plugin/ui/ops/external/method)
--limit N 最多返回多少篇(默认 12)
--context N 每篇显示多少条命中行(默认 3,0 = 只统计)
--json 输出 JSON(供脚本消费)
退出码:0 = 有命中;1 = 无命中;2 = 用法/环境错误
"""
import io
import json
import os
import re
import sys
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
HISTORY_PREFIXES = ('04-调整方案/', 'archive/', '02-运维手册')
WEIGHT = {'hot': 1.0, 'cur': 0.9, 'warm': 0.8, 'doc': 0.7, 'cold': 0.5}
def load_meta():
p = os.path.join(ROOT, 'docs-manifest.json')
if not os.path.exists(p):
raise SystemExit('ERROR: 缺 docs-manifest.json —— 先跑 scripts/docs-manifest.py')
out = {}
for i in json.loads(io.open(p, encoding='utf-8').read())['items']:
out[i['path']] = i
return out
def walk():
for base, dirs, names in os.walk(ROOT):
if '.git' in dirs:
dirs.remove('.git')
for n in sorted(names):
if n.endswith('.md') and '.bak' not in n:
yield os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/')
def main(argv):
VALUE_OPTS = {'--layer', '--domain', '--limit', '--context'}
terms, opts, flags = [], {}, set()
i = 0
while i < len(argv):
a = argv[i]
if a in VALUE_OPTS and i + 1 < len(argv):
opts[a] = argv[i + 1]
i += 2
continue
if a.startswith('--') and '=' in a:
k, v = a.split('=', 1)
opts[k] = v
elif a.startswith('--'):
flags.add(a)
else:
terms.append(a)
i += 1
flag = lambda k: k in flags # noqa: E731
opt = lambda k, d=None: opts.get(k, d) # noqa: E731
if not terms:
print(__doc__)
return 2
terms = [t.lower() for t in terms]
meta = load_meta()
cur = flag('--current')
want_layers = set((opt('--layer') or '').split(',')) - {''}
want_domain = opt('--domain')
limit = int(opt('--limit', '12'))
ctx = int(opt('--context', '3'))
hits = []
for rel in walk():
if cur and (rel.startswith(HISTORY_PREFIXES) or rel == 'archive'):
continue
m = meta.get(rel, {})
layer, domain = m.get('layer', '?'), m.get('domain', '?')
if want_layers and layer not in want_layers:
continue
if want_domain and domain != want_domain:
continue
try:
text = io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
except OSError:
continue
low = text.lower()
counts = [low.count(t) for t in terms]
if not all(c > 0 for c in counts): # AND 语义
continue
total = sum(counts)
head = (m.get('title') or '') + ' ' + (m.get('tldr') or '')
boost = 1.4 if any(t in head.lower() for t in terms) else 1.0
lines = text.split('\n')
shown = [(n + 1, lines[n].strip()[:150]) for n in range(len(lines))
if all(t in lines[n].lower() for t in terms)][:ctx]
hits.append({
'path': rel, 'layer': layer, 'domain': domain,
'tier': m.get('tier', '?'), 'status': m.get('status', '?'),
'hits': total, 'score': round(total * WEIGHT.get(m.get('tier'), 0.7) * boost, 1),
'title': m.get('title', ''), 'samples': shown,
'view': bool(m.get('dedupeView')),
})
hits.sort(key=lambda x: -x['score'])
hits = hits[:limit]
if flag('--json'):
print(json.dumps({'terms': terms, 'current_only': cur, 'results': hits},
ensure_ascii=False, indent=1))
else:
scope = '仅现行层(L0–L4)' if cur else '全库(含历史 L5)'
print('检索 %r | %s | 命中 %d 篇%s'
% (' + '.join(terms), scope, len(hits), '' if not cur else ' (查现行值建议保持 --current)'))
for h in hits:
print('\n[%s] %s | %s/%s | %s | %d 次'
% (h['status'], h['path'], h['layer'], h['domain'], h['tier'], h['hits']))
if h['title']:
print(' %s' % h['title'][:110])
if h.get('view'):
print(' ↳ 本篇有**去重视图**(体积更小,优先读它)')
for ln, text in h['samples']:
print(' L%-5d %s' % (ln, text))
return 0 if hits else 1
if __name__ == '__main__':
raise SystemExit(main(sys.argv[1:]))