#!/usr/bin/env python3 # -*- coding: utf-8 -*- """docs-index-stats.py — 从 INDEX.md §二 清单表**算**出状态分布,并可就地刷新「状态摘要」行。 为什么需要它 ──────────── 「状态摘要」原本是**人工手写**的硬数字,档案一多必然漂移(2026-09-13 实测:摘要写「档案 72 份」, 而表格实际 76 行)。本脚本把摘要变成**机器生成**:读表 → 统计 → 打印;`--write` 时回写该行, 并顺带做**图例自检**;**完整保持文件原有行尾**(CRLF 文件不会被转成 LF)(表格用到的图标必须在「图例」行里有定义,缺则补)。 用法 ──── python3 07-scripts/docs-index-stats.py # 只打印 + 一致性判定 python3 07-scripts/docs-index-stats.py --write # 就地刷新「状态摘要」行(必要时补图例) 退出码:0 = 一致(或已刷新);1 = 漂移且未加 --write;2 = 结构异常。 """ import io import os import re import sys import collections ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) INDEX = os.path.join(ROOT, "INDEX.md") ORDER = ["✅", "🔄", "🧪", "📝", "🔍", "📋", "🟡", "🗄", "🔧", "⚠️"] LEGEND_OF = { "✅": "已落地", "🔄": "维护中", "🧪": "PoC", "📝": "待开发", "🔍": "核查完成", "📋": "评估", "🟡": "保留兜底", "🗄": "归档", "🔧": "修复", "⚠️": "警示", } HEADER_RE = re.compile(r"^\|\s*号\s*\|\s*状态\s*\|") LEAF_RE = re.compile(r"^0\d$") def parse(index_path): """返回 (lines, arch_rows, leaf_rows, counter, other_count)。""" # ⚠️ 必须 newline=""(通用换行会把 CRLF 静默转成 LF —— 2026-09-13 踩过两次) raw = io.open(index_path, encoding="utf-8", newline="").read() lines = raw.split("\n") try: start = next(i for i, l in enumerate(lines) if HEADER_RE.match(l)) except StopIteration: raise SystemExit("ERROR: INDEX.md 里找不到「| 号 | 状态 | 一句话 |」表头") arch, leaf, other = [], [], 0 for l in lines[start + 2:]: if not l.startswith("|"): break cells = [x.strip() for x in l.split("|")] if len(cells) < 4 or not cells[1]: continue no, st = cells[1], cells[2] if no.startswith("04-"): arch.append((no, st)) elif LEAF_RE.match(no): leaf.append((no, st)) else: other += 1 cnt = collections.Counter(st for _, st in arch) cnt.update(st for _, st in leaf) eol = "\r\n" if raw.count("\r\n") > 0 else "\n" return lines, arch, leaf, cnt, other, eol def summary_text(cnt, n_arch, leaf_names, other): parts = ["%s %d" % (k, cnt[k]) for k in ORDER if cnt.get(k)] unmarked = sum(v for k, v in cnt.items() if k not in ORDER) if unmarked: parts.append("未标记 %d" % unmarked) leaf = (",另含根级编号 %d 条(%s)" % (len(leaf_names), "/".join(leaf_names))) if leaf_names else "" return ( "> **状态摘要**(**机器生成,勿手改**):档案 **%d** 份(`04-*`)%s," "另有非编号行 %d 条(README / INDEX / 技能 / poc 等)—— %s。" "复跑 `python3 07-scripts/docs-index-stats.py` 取数,`--write` 就地刷新本行。" % (n_arch, leaf, other, " | ".join(parts)) ) def main(): write = "--write" in sys.argv lines, arch, leaf, cnt, other, eol = parse(INDEX) leaf_names = [n for n, _ in leaf] new = summary_text(cnt, len(arch), leaf_names, other) print("档案 04-* %d 份 | 根级编号 %s | 非编号行 %d 条" % (len(arch), "/".join(leaf_names) or "—", other)) for k in ORDER: if cnt.get(k): print(" %s %-6s %d" % (k, LEGEND_OF.get(k, ""), cnt[k])) unmarked = sum(v for k, v in cnt.items() if k not in ORDER) if unmarked: print(" 未标记 %d" % unmarked) changed = False # ── 图例自检 ───────────────────────────────────────────────────────── li = [i for i, l in enumerate(lines) if l.startswith("> 图例:")] if li: legend = lines[li[0]] missing = [k for k in cnt if k not in legend] if missing: print("⚠️ 图例缺图标:%s" % " ".join(missing)) if write: lines[li[0]] = legend.rstrip().rstrip("|") + "".join( "|%s%s" % (k, LEGEND_OF.get(k, "")) for k in missing ) changed = True print("✓ 已补进图例行") # ── 摘要行 ─────────────────────────────────────────────────────────── idx = [i for i, l in enumerate(lines) if l.startswith("> **状态摘要**")] if not idx: print("ERROR: 未找到「状态摘要」行", file=sys.stderr) return 2 old = lines[idx[0]] if old.strip() == new.strip() and not changed: print("✓ 摘要与表格一致") return 0 if not write: print("✗ 摘要与表格不一致(加 --write 刷新)") print(" 旧: %s" % old.strip()[:90]) print(" 新: %s" % new.strip()[:90]) return 1 lines[idx[0]] = new io.open(INDEX, "w", encoding="utf-8", newline="").write(eol.join(lines)) print("✓ 已刷新 INDEX.md(摘要行%s)" % ("+图例" if changed else "")) return 0 if __name__ == "__main__": sys.exit(main())