#!/usr/bin/env python3 # -*- coding: utf-8 -*- """复述检测:找出技能包里「同一件事被写了好几处」的候选。 思路:把每份 md 拆成句子(按 。!?;\n 切),归一化(去 markdown 记号/空白), 再做字符 shingle(默认 16 字),统计同一 shingle 出现在多少个「不同位置」。 位置 = (文件, 句子序号)。跨文件或跨段落重复 = 复述嫌疑。 只读,不写任何被扫描文件。 """ import os, re, sys, collections ROOT = sys.argv[1] if len(sys.argv) > 1 else "." MINLEN = int(sys.argv[2]) if len(sys.argv) > 2 else 16 SKIP_DIRS = {"assets", "归档", "_留痕", ".git", "node_modules", "vendor"} SKIP_NAME = {"_superseded-competitor-analysis-英文商业版.md"} def norm(s: str) -> str: s = re.sub(r"`[^`]*`", "", s) # 去行内代码(路径/命令不算复述) s = re.sub(r"[*#>|\-\s`]+", "", s) # 去 markdown 记号与空白 s = re.sub(r"[,。!?;:、()()《》「」『』\[\]【】…—~~·\"'’“”]", "", s) return s def sentences(text): # 先按行,再按句号族切 out = [] for ln in text.splitlines(): ln = ln.strip() if not ln: continue for piece in re.split(r"(?<=[。!?;])", ln): p = piece.strip() if p: out.append(p) return out files = [] for dp, dns, fns in os.walk(ROOT): dns[:] = [d for d in dns if d not in SKIP_DIRS] for fn in fns: if not fn.endswith(".md"): continue if fn in SKIP_NAME: continue files.append(os.path.join(dp, fn)) shingle_map = collections.defaultdict(set) # shingle -> set(loc) sent_index = [] # loc -> (path, idx, raw) for path in files: with open(path, encoding="utf-8") as f: text = f.read() for i, raw in enumerate(sentences(text)): n = norm(raw) if len(n) < MINLEN: continue loc = (path, i) sent_index.append((loc, raw, n)) for j in range(0, len(n) - MINLEN + 1): shingle_map[n[j:j + MINLEN]].add(loc) # 每个 shingle 落在几个不同文件 / 几个不同位置 def spread(locs): return len(locs), len({p for p, _ in locs}) # 句级打标:一句里若含「跨文件或跨段」的高频 shingle,就是复述嫌疑 cand = [] for loc, raw, n in sent_index: hits = set() for j in range(0, max(1, len(n) - MINLEN + 1)): sh = n[j:j + MINLEN] locs = shingle_map.get(sh) if not locs: continue nf, nl = spread(locs) if nl >= 2: hits.update(locs) if hits: nf, nl = spread(hits) cand.append((nf, nl, loc, raw)) cand.sort(key=lambda x: (-x[1], -x[0], x[2][0])) seen = set() print(f"# 复述嫌疑(跨 ≥2 流程位置,MINLEN={MINLEN})—— 共 {len(cand)} 条\n") for nf, nl, loc, raw in cand[:120]: key = loc if key in seen: continue seen.add(key) others = sorted({p for p, _ in shingle_map.get(norm(raw)[:MINLEN], set())}) tag = "跨文件" if nl >= 2 else "" print(f"[{nl} 位置 {tag}] {loc[0]}#{loc[1]}") print(f" {raw[:160]}") print("\n# —— 出现次数最多的 16 字片段(TOP 40)——") top = sorted(((len(v), tuple(sorted(v))) for v in shingle_map.values()), key=lambda x: -x[0])[:40] for cnt, locs in top: if cnt < 3: break fileset = sorted({p for p, _ in locs}) print(f"{cnt}x 文件数{len(fileset)} | 示例: {fileset[0]}#{sorted(locs)[0][1]}")