Files
dsh_shenxian/dsh-server-docs/scripts/docs-index-stats.py
T

130 lines
5.5 KiB
Python
Raw Normal View History

#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""docs-index-stats.py — 从 INDEX.md §二 清单表**算**出状态分布,并可就地刷新「状态摘要」行。
为什么需要它
────────────
「状态摘要」原本是**人工手写**的硬数字,档案一多必然漂移(2026-09-13 实测:摘要写「档案 72 份」,
而表格实际 76 行)。本脚本把摘要变成**机器生成**:读表 → 统计 → 打印;`--write` 时回写该行,
并顺带做**图例自检**;**完整保持文件原有行尾**(CRLF 文件不会被转成 LF)(表格用到的图标必须在「图例」行里有定义,缺则补)。
用法
────
python3 scripts/docs-index-stats.py # 只打印 + 一致性判定
python3 scripts/docs-index-stats.py --write # 就地刷新「状态摘要」行(必要时补图例)
退出码:0 = 一致(或已刷新);1 = 漂移且未加 --write;2 = 结构异常。
"""
import io
import os
import re
import sys
import collections
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
INDEX = os.path.join(ROOT, "INDEX.md")
ORDER = ["✅", "🔄", "🧪", "📝", "🔍", "📋", "🟡", "🗄", "🔧", "⚠️"]
LEGEND_OF = {
"✅": "已落地", "🔄": "维护中", "🧪": "PoC", "📝": "待开发", "🔍": "核查完成",
"📋": "评估", "🟡": "保留兜底", "🗄": "归档", "🔧": "修复", "⚠️": "警示",
}
HEADER_RE = re.compile(r"^\|\s*号\s*\|\s*状态\s*\|")
LEAF_RE = re.compile(r"^0\d$")
def parse(index_path):
"""返回 (lines, arch_rows, leaf_rows, counter, other_count)。"""
# ⚠️ 必须 newline=""(通用换行会把 CRLF 静默转成 LF —— 2026-09-13 踩过两次)
raw = io.open(index_path, encoding="utf-8", newline="").read()
lines = raw.split("\n")
try:
start = next(i for i, l in enumerate(lines) if HEADER_RE.match(l))
except StopIteration:
raise SystemExit("ERROR: INDEX.md 里找不到「| 号 | 状态 | 一句话 |」表头")
arch, leaf, other = [], [], 0
for l in lines[start + 2:]:
if not l.startswith("|"):
break
cells = [x.strip() for x in l.split("|")]
if len(cells) < 4 or not cells[1]:
continue
no, st = cells[1], cells[2]
if no.startswith("04-"):
arch.append((no, st))
elif LEAF_RE.match(no):
leaf.append((no, st))
else:
other += 1
cnt = collections.Counter(st for _, st in arch)
cnt.update(st for _, st in leaf)
eol = "\r\n" if raw.count("\r\n") > 0 else "\n"
return lines, arch, leaf, cnt, other, eol
def summary_text(cnt, n_arch, leaf_names, other):
parts = ["%s %d" % (k, cnt[k]) for k in ORDER if cnt.get(k)]
unmarked = sum(v for k, v in cnt.items() if k not in ORDER)
if unmarked:
parts.append("未标记 %d" % unmarked)
leaf = (",另含根级编号 %d 条(%s)" % (len(leaf_names), "/".join(leaf_names))) if leaf_names else ""
return (
"> **状态摘要**(**机器生成,勿手改**):档案 **%d** 份(`04-*`)%s,"
"另有非编号行 %d 条(README / INDEX / 技能 / poc 等)—— %s。"
"复跑 `python3 scripts/docs-index-stats.py` 取数,`--write` 就地刷新本行。"
% (n_arch, leaf, other, " | ".join(parts))
)
def main():
write = "--write" in sys.argv
lines, arch, leaf, cnt, other, eol = parse(INDEX)
leaf_names = [n for n, _ in leaf]
new = summary_text(cnt, len(arch), leaf_names, other)
print("档案 04-* %d 份 | 根级编号 %s | 非编号行 %d 条" % (len(arch), "/".join(leaf_names) or "—", other))
for k in ORDER:
if cnt.get(k):
print(" %s %-6s %d" % (k, LEGEND_OF.get(k, ""), cnt[k]))
unmarked = sum(v for k, v in cnt.items() if k not in ORDER)
if unmarked:
print(" 未标记 %d" % unmarked)
changed = False
# ── 图例自检 ─────────────────────────────────────────────────────────
li = [i for i, l in enumerate(lines) if l.startswith("> 图例:")]
if li:
legend = lines[li[0]]
missing = [k for k in cnt if k not in legend]
if missing:
print("⚠️ 图例缺图标:%s" % " ".join(missing))
if write:
lines[li[0]] = legend.rstrip().rstrip("|") + "".join(
"|%s%s" % (k, LEGEND_OF.get(k, "")) for k in missing
)
changed = True
print("✓ 已补进图例行")
# ── 摘要行 ───────────────────────────────────────────────────────────
idx = [i for i, l in enumerate(lines) if l.startswith("> **状态摘要**")]
if not idx:
print("ERROR: 未找到「状态摘要」行", file=sys.stderr)
return 2
old = lines[idx[0]]
if old.strip() == new.strip() and not changed:
print("✓ 摘要与表格一致")
return 0
if not write:
print("✗ 摘要与表格不一致(加 --write 刷新)")
print(" 旧: %s" % old.strip()[:90])
print(" 新: %s" % new.strip()[:90])
return 1
lines[idx[0]] = new
io.open(INDEX, "w", encoding="utf-8", newline="").write(eol.join(lines))
print("✓ 已刷新 INDEX.md(摘要行%s)" % ("+图例" if changed else ""))
return 0
if __name__ == "__main__":
sys.exit(main())