#!/usr/bin/env python3 # -*- coding: utf-8 -*- """docs-search.py — 文档库**带语义的全文检索**(零依赖、不建索引、每次实时扫) 为什么需要它(2026-09-14 实测) * 库已到 **118 篇 / 95 万字符 ≈ 67 万 token** ⇒ 不可能"读全库"; * 只能 `grep` 时,**搜到的结果分不清"现行值"还是"历史值"** —— 这正是 2026-09-12 踩过的坑 (旧配额活在 15 个文件里,被当成事实用)。 ⇒ 本脚本把 `docs-manifest.json` 的 **(层 L0–L5 / 域 / tier / 状态)** 标注接进检索结果, 并给 `--current` 一键**排除历史层**,让"查现行事实"这件事**结果可判**。 用法 python3 07-scripts/docs-search.py 配额 # 全库搜「配额」 python3 07-scripts/docs-search.py 配额 --current # **只搜现行层(排除 L5 档案 / archive)** python3 07-scripts/docs-search.py 插件 域 # 多词 = AND python3 07-scripts/docs-search.py glibc --domain plugin --layer L5 python3 07-scripts/docs-search.py 内存 --json | jq . # 机读输出 选项 --current 只搜 L0–L4(排除 L5 与 09-archive/)—— **查现行值请默认加它** --layer L1,L2 限定层(L0–L5) --domain plugin 限定域(platform/plugin/ui/06-ops/external/method) --limit N 最多返回多少篇(默认 12) --context N 每篇显示多少条命中行(默认 3,0 = 只统计) --json 输出 JSON(供脚本消费) 退出码:0 = 有命中;1 = 无命中;2 = 用法/环境错误 """ import io import json import os import re import sys ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) HISTORY_PREFIXES = ('04-调整方案/', '09-archive/', '01-规范/02-运维手册') WEIGHT = {'hot': 1.0, 'cur': 0.9, 'warm': 0.8, 'doc': 0.7, 'cold': 0.5} def load_meta(): p = os.path.join(ROOT, 'docs-manifest.json') if not os.path.exists(p): raise SystemExit('ERROR: 缺 docs-manifest.json —— 先跑 07-scripts/docs-manifest.py') out = {} for i in json.loads(io.open(p, encoding='utf-8').read())['items']: out[i['path']] = i return out def walk(): for base, dirs, names in os.walk(ROOT): if '.git' in dirs: dirs.remove('.git') for n in sorted(names): if n.endswith('.md') and '.bak' not in n: yield os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/') def main(argv): VALUE_OPTS = {'--layer', '--domain', '--limit', '--context'} terms, opts, flags = [], {}, set() i = 0 while i < len(argv): a = argv[i] if a in VALUE_OPTS and i + 1 < len(argv): opts[a] = argv[i + 1] i += 2 continue if a.startswith('--') and '=' in a: k, v = a.split('=', 1) opts[k] = v elif a.startswith('--'): flags.add(a) else: terms.append(a) i += 1 flag = lambda k: k in flags # noqa: E731 opt = lambda k, d=None: opts.get(k, d) # noqa: E731 if not terms: print(__doc__) return 2 terms = [t.lower() for t in terms] meta = load_meta() cur = flag('--current') want_layers = set((opt('--layer') or '').split(',')) - {''} want_domain = opt('--domain') limit = int(opt('--limit', '12')) ctx = int(opt('--context', '3')) hits = [] for rel in walk(): if cur and (rel.startswith(HISTORY_PREFIXES) or rel == '09-archive'): continue m = meta.get(rel, {}) layer, domain = m.get('layer', '?'), m.get('domain', '?') if want_layers and layer not in want_layers: continue if want_domain and domain != want_domain: continue try: text = io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read() except OSError: continue low = text.lower() counts = [low.count(t) for t in terms] if not all(c > 0 for c in counts): # AND 语义 continue total = sum(counts) head = (m.get('title') or '') + ' ' + (m.get('tldr') or '') boost = 1.4 if any(t in head.lower() for t in terms) else 1.0 lines = text.split('\n') shown = [(n + 1, lines[n].strip()[:150]) for n in range(len(lines)) if all(t in lines[n].lower() for t in terms)][:ctx] hits.append({ 'path': rel, 'layer': layer, 'domain': domain, 'tier': m.get('tier', '?'), 'status': m.get('status', '?'), 'hits': total, 'score': round(total * WEIGHT.get(m.get('tier'), 0.7) * boost, 1), 'title': m.get('title', ''), 'samples': shown, 'view': bool(m.get('dedupeView')), }) hits.sort(key=lambda x: -x['score']) hits = hits[:limit] if flag('--json'): print(json.dumps({'terms': terms, 'current_only': cur, 'results': hits}, ensure_ascii=False, indent=1)) else: scope = '仅现行层(L0–L4)' if cur else '全库(含历史 L5)' print('检索 %r | %s | 命中 %d 篇%s' % (' + '.join(terms), scope, len(hits), '' if not cur else ' (查现行值建议保持 --current)')) for h in hits: print('\n[%s] %s | %s/%s | %s | %d 次' % (h['status'], h['path'], h['layer'], h['domain'], h['tier'], h['hits'])) if h['title']: print(' %s' % h['title'][:110]) if h.get('view'): print(' ↳ 本篇有**去重视图**(体积更小,优先读它)') for ln, text in h['samples']: print(' L%-5d %s' % (ln, text)) return 0 if hits else 1 if __name__ == '__main__': raise SystemExit(main(sys.argv[1:]))