build / build-and-scan (push) Waiting to run
文档库:目录改为编号制(01-规范/02-架构设计/03-数据库/04-调整方案/
05-交接单/06-ops/07-scripts/08-skills/09-archive),顶层散文件归入 01-规范/;
INDEX.md 与 docs-manifest.json 重刷(档案 146 篇);旧目录名引用全量对齐。
IM 线:src/im/**(SDK / hub / store / presence / ws / gateway-token)、
src/web/routes/im.ts、src/db/plugin-data/**、src/supervisor/plugin-assembly.ts
及对应 test/**。
插件线:poc/{im-agent-bridge,im-connection-gateway,im-conversation-tabs,
business-plugins-im,carbon-mcp-probe}、src/web/routes/{sessions,overlay-device}.ts、
src/net/relay/{device-grant,instance-credential}.ts。
仓库卫生:清出 40 个历史误入库 / 已改名文件(34 个交接单归档 + 6 个旧结构,
本地均有副本);dsh-server-docs/.gitignore 补 tmp/;交接单不入库(政策)。
143 lines
5.8 KiB
Python
143 lines
5.8 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""docs-search.py — 文档库**带语义的全文检索**(零依赖、不建索引、每次实时扫)
|
||
|
||
为什么需要它(2026-09-14 实测)
|
||
* 库已到 **118 篇 / 95 万字符 ≈ 67 万 token** ⇒ 不可能"读全库";
|
||
* 只能 `grep` 时,**搜到的结果分不清"现行值"还是"历史值"** —— 这正是 2026-09-12 踩过的坑
|
||
(旧配额活在 15 个文件里,被当成事实用)。
|
||
⇒ 本脚本把 `docs-manifest.json` 的 **(层 L0–L5 / 域 / tier / 状态)** 标注接进检索结果,
|
||
并给 `--current` 一键**排除历史层**,让"查现行事实"这件事**结果可判**。
|
||
|
||
用法
|
||
python3 07-scripts/docs-search.py 配额 # 全库搜「配额」
|
||
python3 07-scripts/docs-search.py 配额 --current # **只搜现行层(排除 L5 档案 / archive)**
|
||
python3 07-scripts/docs-search.py 插件 域 # 多词 = AND
|
||
python3 07-scripts/docs-search.py glibc --domain plugin --layer L5
|
||
python3 07-scripts/docs-search.py 内存 --json | jq . # 机读输出
|
||
选项
|
||
--current 只搜 L0–L4(排除 L5 与 09-archive/)—— **查现行值请默认加它**
|
||
--layer L1,L2 限定层(L0–L5)
|
||
--domain plugin 限定域(platform/plugin/ui/06-ops/external/method)
|
||
--limit N 最多返回多少篇(默认 12)
|
||
--context N 每篇显示多少条命中行(默认 3,0 = 只统计)
|
||
--json 输出 JSON(供脚本消费)
|
||
退出码:0 = 有命中;1 = 无命中;2 = 用法/环境错误
|
||
"""
|
||
import io
|
||
import json
|
||
import os
|
||
import re
|
||
import sys
|
||
|
||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
HISTORY_PREFIXES = ('04-调整方案/', '09-archive/', '01-规范/02-运维手册')
|
||
WEIGHT = {'hot': 1.0, 'cur': 0.9, 'warm': 0.8, 'doc': 0.7, 'cold': 0.5}
|
||
|
||
|
||
def load_meta():
|
||
p = os.path.join(ROOT, 'docs-manifest.json')
|
||
if not os.path.exists(p):
|
||
raise SystemExit('ERROR: 缺 docs-manifest.json —— 先跑 07-scripts/docs-manifest.py')
|
||
out = {}
|
||
for i in json.loads(io.open(p, encoding='utf-8').read())['items']:
|
||
out[i['path']] = i
|
||
return out
|
||
|
||
|
||
def walk():
|
||
for base, dirs, names in os.walk(ROOT):
|
||
if '.git' in dirs:
|
||
dirs.remove('.git')
|
||
for n in sorted(names):
|
||
if n.endswith('.md') and '.bak' not in n:
|
||
yield os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/')
|
||
|
||
|
||
def main(argv):
|
||
VALUE_OPTS = {'--layer', '--domain', '--limit', '--context'}
|
||
terms, opts, flags = [], {}, set()
|
||
i = 0
|
||
while i < len(argv):
|
||
a = argv[i]
|
||
if a in VALUE_OPTS and i + 1 < len(argv):
|
||
opts[a] = argv[i + 1]
|
||
i += 2
|
||
continue
|
||
if a.startswith('--') and '=' in a:
|
||
k, v = a.split('=', 1)
|
||
opts[k] = v
|
||
elif a.startswith('--'):
|
||
flags.add(a)
|
||
else:
|
||
terms.append(a)
|
||
i += 1
|
||
flag = lambda k: k in flags # noqa: E731
|
||
opt = lambda k, d=None: opts.get(k, d) # noqa: E731
|
||
if not terms:
|
||
print(__doc__)
|
||
return 2
|
||
terms = [t.lower() for t in terms]
|
||
meta = load_meta()
|
||
cur = flag('--current')
|
||
want_layers = set((opt('--layer') or '').split(',')) - {''}
|
||
want_domain = opt('--domain')
|
||
limit = int(opt('--limit', '12'))
|
||
ctx = int(opt('--context', '3'))
|
||
|
||
hits = []
|
||
for rel in walk():
|
||
if cur and (rel.startswith(HISTORY_PREFIXES) or rel == '09-archive'):
|
||
continue
|
||
m = meta.get(rel, {})
|
||
layer, domain = m.get('layer', '?'), m.get('domain', '?')
|
||
if want_layers and layer not in want_layers:
|
||
continue
|
||
if want_domain and domain != want_domain:
|
||
continue
|
||
try:
|
||
text = io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
|
||
except OSError:
|
||
continue
|
||
low = text.lower()
|
||
counts = [low.count(t) for t in terms]
|
||
if not all(c > 0 for c in counts): # AND 语义
|
||
continue
|
||
total = sum(counts)
|
||
head = (m.get('title') or '') + ' ' + (m.get('tldr') or '')
|
||
boost = 1.4 if any(t in head.lower() for t in terms) else 1.0
|
||
lines = text.split('\n')
|
||
shown = [(n + 1, lines[n].strip()[:150]) for n in range(len(lines))
|
||
if all(t in lines[n].lower() for t in terms)][:ctx]
|
||
hits.append({
|
||
'path': rel, 'layer': layer, 'domain': domain,
|
||
'tier': m.get('tier', '?'), 'status': m.get('status', '?'),
|
||
'hits': total, 'score': round(total * WEIGHT.get(m.get('tier'), 0.7) * boost, 1),
|
||
'title': m.get('title', ''), 'samples': shown,
|
||
'view': bool(m.get('dedupeView')),
|
||
})
|
||
hits.sort(key=lambda x: -x['score'])
|
||
hits = hits[:limit]
|
||
|
||
if flag('--json'):
|
||
print(json.dumps({'terms': terms, 'current_only': cur, 'results': hits},
|
||
ensure_ascii=False, indent=1))
|
||
else:
|
||
scope = '仅现行层(L0–L4)' if cur else '全库(含历史 L5)'
|
||
print('检索 %r | %s | 命中 %d 篇%s'
|
||
% (' + '.join(terms), scope, len(hits), '' if not cur else ' (查现行值建议保持 --current)'))
|
||
for h in hits:
|
||
print('\n[%s] %s | %s/%s | %s | %d 次'
|
||
% (h['status'], h['path'], h['layer'], h['domain'], h['tier'], h['hits']))
|
||
if h['title']:
|
||
print(' %s' % h['title'][:110])
|
||
if h.get('view'):
|
||
print(' ↳ 本篇有**去重视图**(体积更小,优先读它)')
|
||
for ln, text in h['samples']:
|
||
print(' L%-5d %s' % (ln, text))
|
||
return 0 if hits else 1
|
||
|
||
|
||
if __name__ == '__main__':
|
||
raise SystemExit(main(sys.argv[1:]))
|