Files
admin e6207aa691
build / build-and-scan (push) Waiting to run
chore(仓库对齐): 文档库结构治理 + IM/插件线落地
文档库:目录改为编号制(01-规范/02-架构设计/03-数据库/04-调整方案/
05-交接单/06-ops/07-scripts/08-skills/09-archive),顶层散文件归入 01-规范/;
INDEX.md 与 docs-manifest.json 重刷(档案 146 篇);旧目录名引用全量对齐。

IM 线:src/im/**(SDK / hub / store / presence / ws / gateway-token)、
src/web/routes/im.ts、src/db/plugin-data/**、src/supervisor/plugin-assembly.ts
及对应 test/**。

插件线:poc/{im-agent-bridge,im-connection-gateway,im-conversation-tabs,
business-plugins-im,carbon-mcp-probe}、src/web/routes/{sessions,overlay-device}.ts、
src/net/relay/{device-grant,instance-credential}.ts。

仓库卫生:清出 40 个历史误入库 / 已改名文件(34 个交接单归档 + 6 个旧结构,
本地均有副本);dsh-server-docs/.gitignore 补 tmp/;交接单不入库(政策)。
2026-09-24 07:25:16 +08:00

143 lines
5.8 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""docs-search.py — 文档库**带语义的全文检索**(零依赖、不建索引、每次实时扫)
为什么需要它(2026-09-14 实测)
* 库已到 **118 篇 / 95 万字符 ≈ 67 万 token** ⇒ 不可能"读全库";
* 只能 `grep` 时,**搜到的结果分不清"现行值"还是"历史值"** —— 这正是 2026-09-12 踩过的坑
(旧配额活在 15 个文件里,被当成事实用)。
⇒ 本脚本把 `docs-manifest.json` 的 **(层 L0–L5 / 域 / tier / 状态)** 标注接进检索结果,
并给 `--current` 一键**排除历史层**,让"查现行事实"这件事**结果可判**。
用法
python3 07-scripts/docs-search.py 配额 # 全库搜「配额」
python3 07-scripts/docs-search.py 配额 --current # **只搜现行层(排除 L5 档案 / archive)**
python3 07-scripts/docs-search.py 插件 域 # 多词 = AND
python3 07-scripts/docs-search.py glibc --domain plugin --layer L5
python3 07-scripts/docs-search.py 内存 --json | jq . # 机读输出
选项
--current 只搜 L0–L4(排除 L5 与 09-archive/)—— **查现行值请默认加它**
--layer L1,L2 限定层(L0–L5)
--domain plugin 限定域(platform/plugin/ui/06-ops/external/method)
--limit N 最多返回多少篇(默认 12)
--context N 每篇显示多少条命中行(默认 3,0 = 只统计)
--json 输出 JSON(供脚本消费)
退出码:0 = 有命中;1 = 无命中;2 = 用法/环境错误
"""
import io
import json
import os
import re
import sys
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
HISTORY_PREFIXES = ('04-调整方案/', '09-archive/', '01-规范/02-运维手册')
WEIGHT = {'hot': 1.0, 'cur': 0.9, 'warm': 0.8, 'doc': 0.7, 'cold': 0.5}
def load_meta():
p = os.path.join(ROOT, 'docs-manifest.json')
if not os.path.exists(p):
raise SystemExit('ERROR: 缺 docs-manifest.json —— 先跑 07-scripts/docs-manifest.py')
out = {}
for i in json.loads(io.open(p, encoding='utf-8').read())['items']:
out[i['path']] = i
return out
def walk():
for base, dirs, names in os.walk(ROOT):
if '.git' in dirs:
dirs.remove('.git')
for n in sorted(names):
if n.endswith('.md') and '.bak' not in n:
yield os.path.relpath(os.path.join(base, n), ROOT).replace('\\', '/')
def main(argv):
VALUE_OPTS = {'--layer', '--domain', '--limit', '--context'}
terms, opts, flags = [], {}, set()
i = 0
while i < len(argv):
a = argv[i]
if a in VALUE_OPTS and i + 1 < len(argv):
opts[a] = argv[i + 1]
i += 2
continue
if a.startswith('--') and '=' in a:
k, v = a.split('=', 1)
opts[k] = v
elif a.startswith('--'):
flags.add(a)
else:
terms.append(a)
i += 1
flag = lambda k: k in flags # noqa: E731
opt = lambda k, d=None: opts.get(k, d) # noqa: E731
if not terms:
print(__doc__)
return 2
terms = [t.lower() for t in terms]
meta = load_meta()
cur = flag('--current')
want_layers = set((opt('--layer') or '').split(',')) - {''}
want_domain = opt('--domain')
limit = int(opt('--limit', '12'))
ctx = int(opt('--context', '3'))
hits = []
for rel in walk():
if cur and (rel.startswith(HISTORY_PREFIXES) or rel == '09-archive'):
continue
m = meta.get(rel, {})
layer, domain = m.get('layer', '?'), m.get('domain', '?')
if want_layers and layer not in want_layers:
continue
if want_domain and domain != want_domain:
continue
try:
text = io.open(os.path.join(ROOT, rel), encoding='utf-8', errors='replace').read()
except OSError:
continue
low = text.lower()
counts = [low.count(t) for t in terms]
if not all(c > 0 for c in counts): # AND 语义
continue
total = sum(counts)
head = (m.get('title') or '') + ' ' + (m.get('tldr') or '')
boost = 1.4 if any(t in head.lower() for t in terms) else 1.0
lines = text.split('\n')
shown = [(n + 1, lines[n].strip()[:150]) for n in range(len(lines))
if all(t in lines[n].lower() for t in terms)][:ctx]
hits.append({
'path': rel, 'layer': layer, 'domain': domain,
'tier': m.get('tier', '?'), 'status': m.get('status', '?'),
'hits': total, 'score': round(total * WEIGHT.get(m.get('tier'), 0.7) * boost, 1),
'title': m.get('title', ''), 'samples': shown,
'view': bool(m.get('dedupeView')),
})
hits.sort(key=lambda x: -x['score'])
hits = hits[:limit]
if flag('--json'):
print(json.dumps({'terms': terms, 'current_only': cur, 'results': hits},
ensure_ascii=False, indent=1))
else:
scope = '仅现行层(L0–L4)' if cur else '全库(含历史 L5)'
print('检索 %r | %s | 命中 %d 篇%s'
% (' + '.join(terms), scope, len(hits), '' if not cur else ' (查现行值建议保持 --current)'))
for h in hits:
print('\n[%s] %s | %s/%s | %s | %d 次'
% (h['status'], h['path'], h['layer'], h['domain'], h['tier'], h['hits']))
if h['title']:
print(' %s' % h['title'][:110])
if h.get('view'):
print(' ↳ 本篇有**去重视图**(体积更小,优先读它)')
for ln, text in h['samples']:
print(' L%-5d %s' % (ln, text))
return 0 if hits else 1
if __name__ == '__main__':
raise SystemExit(main(sys.argv[1:]))