Files

109 lines
3.3 KiB
Python
Raw Permalink Normal View History

#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""复述检测 v2:按「最大重复簇」聚合输出。
输出形如:
【簇 3 文件 · 5 处】关键词行
- 文件A#12
- 文件B#40
...
只读,不写被扫描文件。
"""
import os, re, sys, collections
ROOT = sys.argv[1] if len(sys.argv) > 1 else "."
MINLEN = int(sys.argv[2]) if len(sys.argv) > 2 else 14
SKIP_DIRS = {"assets", "归档", "_留痕", ".git", "node_modules", "vendor"}
SKIP_NAME = {"_superseded-competitor-analysis-英文商业版.md"}
def norm(s: str) -> str:
s = re.sub(r"`[^`]*`", "", s)
s = re.sub(r"[*#>|\-\s`]+", "", s)
s = re.sub(r"[,。!?;:、()()《》「」『』\[\]【】…—~~·\"'’“”]", "", s)
return s
def sentences(text):
out = []
for ln in text.splitlines():
ln = ln.strip()
if not ln:
continue
for piece in re.split(r"(?<=[。!?;])", ln):
p = piece.strip()
if p:
out.append(p)
return out
files = []
for dp, dns, fns in os.walk(ROOT):
dns[:] = [d for d in dns if d not in SKIP_DIRS]
for fn in fns:
if fn.endswith(".md") and fn not in SKIP_NAME:
files.append(os.path.join(dp, fn))
files.sort()
sent = [] # (loc, raw, norm)
for path in files:
with open(path, encoding="utf-8") as f:
for i, raw in enumerate(sentences(f.read())):
n = norm(raw)
if len(n) >= MINLEN:
sent.append(((path, i), raw, n))
sh_map = collections.defaultdict(set)
for loc, raw, n in sent:
for j in range(len(n) - MINLEN + 1):
sh_map[n[j:j + MINLEN]].add(loc)
# 句级:收集它命中的、跨≥2文件的 shingle 所覆盖的全部位置
pairs = []
for loc, raw, n in sent:
hit = set()
for j in range(max(1, len(n) - MINLEN + 1)):
locs = sh_map.get(n[j:j + MINLEN])
if locs and len({p for p, _ in locs}) >= 2:
hit |= locs
if len({p for p, _ in hit}) >= 2:
pairs.append((loc, raw, hit))
# 聚簇:位置集合有交集的句子并入同一簇
clusters = []
for loc, raw, hit in pairs:
placed = False
for c in clusters:
if c["locs"] & hit:
c["locs"] |= hit
c["members"].append((loc, raw))
placed = True
break
if not placed:
clusters.append({"locs": set(hit), "members": [(loc, raw)]})
# 合并可传递的簇
changed = True
while changed:
changed = False
for i in range(len(clusters)):
for j in range(i + 1, len(clusters)):
if clusters[i]["locs"] & clusters[j]["locs"]:
clusters[i]["locs"] |= clusters[j]["locs"]
clusters[i]["members"] += clusters[j]["members"]
del clusters[j]
changed = True
break
if changed:
break
clusters = [c for c in clusters if len({p for p, _ in c["locs"]}) >= 2]
clusters.sort(key=lambda c: (-len({p for p, _ in c["locs"]}), -len(c["locs"])))
print(f"# 跨文件复述簇:{len(clusters)} 个(MINLEN={MINLEN})\n")
for k, c in enumerate(clusters, 1):
fs = sorted({p for p, _ in c["locs"]})
print(f"【簇{k}】{len(fs)} 文件 · {len(c['locs'])} 处")
for loc, raw in sorted(c["members"]):
print(f" {loc[0]}#{loc[1]} {raw[:110]}")
print()