#!/usr/bin/env python3 # -*- coding: utf-8 -*- """复述检测 v2:按「最大重复簇」聚合输出。 输出形如: 【簇 3 文件 · 5 处】关键词行 - 文件A#12 - 文件B#40 ... 只读,不写被扫描文件。 """ import os, re, sys, collections ROOT = sys.argv[1] if len(sys.argv) > 1 else "." MINLEN = int(sys.argv[2]) if len(sys.argv) > 2 else 14 SKIP_DIRS = {"assets", "归档", "_留痕", ".git", "node_modules", "vendor"} SKIP_NAME = {"_superseded-competitor-analysis-英文商业版.md"} def norm(s: str) -> str: s = re.sub(r"`[^`]*`", "", s) s = re.sub(r"[*#>|\-\s`]+", "", s) s = re.sub(r"[,。!?;:、()()《》「」『』\[\]【】…—~~·\"'’“”]", "", s) return s def sentences(text): out = [] for ln in text.splitlines(): ln = ln.strip() if not ln: continue for piece in re.split(r"(?<=[。!?;])", ln): p = piece.strip() if p: out.append(p) return out files = [] for dp, dns, fns in os.walk(ROOT): dns[:] = [d for d in dns if d not in SKIP_DIRS] for fn in fns: if fn.endswith(".md") and fn not in SKIP_NAME: files.append(os.path.join(dp, fn)) files.sort() sent = [] # (loc, raw, norm) for path in files: with open(path, encoding="utf-8") as f: for i, raw in enumerate(sentences(f.read())): n = norm(raw) if len(n) >= MINLEN: sent.append(((path, i), raw, n)) sh_map = collections.defaultdict(set) for loc, raw, n in sent: for j in range(len(n) - MINLEN + 1): sh_map[n[j:j + MINLEN]].add(loc) # 句级:收集它命中的、跨≥2文件的 shingle 所覆盖的全部位置 pairs = [] for loc, raw, n in sent: hit = set() for j in range(max(1, len(n) - MINLEN + 1)): locs = sh_map.get(n[j:j + MINLEN]) if locs and len({p for p, _ in locs}) >= 2: hit |= locs if len({p for p, _ in hit}) >= 2: pairs.append((loc, raw, hit)) # 聚簇:位置集合有交集的句子并入同一簇 clusters = [] for loc, raw, hit in pairs: placed = False for c in clusters: if c["locs"] & hit: c["locs"] |= hit c["members"].append((loc, raw)) placed = True break if not placed: clusters.append({"locs": set(hit), "members": [(loc, raw)]}) # 合并可传递的簇 changed = True while changed: changed = False for i in range(len(clusters)): for j in range(i + 1, len(clusters)): if clusters[i]["locs"] & clusters[j]["locs"]: clusters[i]["locs"] |= clusters[j]["locs"] clusters[i]["members"] += clusters[j]["members"] del clusters[j] changed = True break if changed: break clusters = [c for c in clusters if len({p for p, _ in c["locs"]}) >= 2] clusters.sort(key=lambda c: (-len({p for p, _ in c["locs"]}), -len(c["locs"]))) print(f"# 跨文件复述簇:{len(clusters)} 个(MINLEN={MINLEN})\n") for k, c in enumerate(clusters, 1): fs = sorted({p for p, _ in c["locs"]}) print(f"【簇{k}】{len(fs)} 文件 · {len(c['locs'])} 处") for loc, raw in sorted(c["members"]): print(f" {loc[0]}#{loc[1]} {raw[:110]}") print()