109 lines
3.3 KiB
Python
109 lines
3.3 KiB
Python
#!/usr/bin/env python3
|
||||
|
|
# -*- coding: utf-8 -*-
|
|||
|
|
"""复述检测 v2:按「最大重复簇」聚合输出。
|
|||
|
|
|
|||
|
|
输出形如:
|
|||
|
|
【簇 3 文件 · 5 处】关键词行
|
|||
|
|
- 文件A#12
|
|||
|
|
- 文件B#40
|
|||
|
|
...
|
|||
|
|
只读,不写被扫描文件。
|
|||
|
|
"""
|
|||
|
|
import os, re, sys, collections
|
|||
|
|
|
|||
|
|
ROOT = sys.argv[1] if len(sys.argv) > 1 else "."
|
|||
|
|
MINLEN = int(sys.argv[2]) if len(sys.argv) > 2 else 14
|
|||
|
|
SKIP_DIRS = {"assets", "归档", "_留痕", ".git", "node_modules", "vendor"}
|
|||
|
|
SKIP_NAME = {"_superseded-competitor-analysis-英文商业版.md"}
|
|||
|
|
|
|||
|
|
|
|||
|
|
def norm(s: str) -> str:
|
|||
|
|
s = re.sub(r"`[^`]*`", "", s)
|
|||
|
|
s = re.sub(r"[*#>|\-\s`]+", "", s)
|
|||
|
|
s = re.sub(r"[,。!?;:、()()《》「」『』\[\]【】…—~~·\"'’“”]", "", s)
|
|||
|
|
return s
|
|||
|
|
|
|||
|
|
|
|||
|
|
def sentences(text):
|
|||
|
|
out = []
|
|||
|
|
for ln in text.splitlines():
|
|||
|
|
ln = ln.strip()
|
|||
|
|
if not ln:
|
|||
|
|
continue
|
|||
|
|
for piece in re.split(r"(?<=[。!?;])", ln):
|
|||
|
|
p = piece.strip()
|
|||
|
|
if p:
|
|||
|
|
out.append(p)
|
|||
|
|
return out
|
|||
|
|
|
|||
|
|
|
|||
|
|
files = []
|
|||
|
|
for dp, dns, fns in os.walk(ROOT):
|
|||
|
|
dns[:] = [d for d in dns if d not in SKIP_DIRS]
|
|||
|
|
for fn in fns:
|
|||
|
|
if fn.endswith(".md") and fn not in SKIP_NAME:
|
|||
|
|
files.append(os.path.join(dp, fn))
|
|||
|
|
files.sort()
|
|||
|
|
|
|||
|
|
sent = [] # (loc, raw, norm)
|
|||
|
|
for path in files:
|
|||
|
|
with open(path, encoding="utf-8") as f:
|
|||
|
|
for i, raw in enumerate(sentences(f.read())):
|
|||
|
|
n = norm(raw)
|
|||
|
|
if len(n) >= MINLEN:
|
|||
|
|
sent.append(((path, i), raw, n))
|
|||
|
|
|
|||
|
|
sh_map = collections.defaultdict(set)
|
|||
|
|
for loc, raw, n in sent:
|
|||
|
|
for j in range(len(n) - MINLEN + 1):
|
|||
|
|
sh_map[n[j:j + MINLEN]].add(loc)
|
|||
|
|
|
|||
|
|
# 句级:收集它命中的、跨≥2文件的 shingle 所覆盖的全部位置
|
|||
|
|
pairs = []
|
|||
|
|
for loc, raw, n in sent:
|
|||
|
|
hit = set()
|
|||
|
|
for j in range(max(1, len(n) - MINLEN + 1)):
|
|||
|
|
locs = sh_map.get(n[j:j + MINLEN])
|
|||
|
|
if locs and len({p for p, _ in locs}) >= 2:
|
|||
|
|
hit |= locs
|
|||
|
|
if len({p for p, _ in hit}) >= 2:
|
|||
|
|
pairs.append((loc, raw, hit))
|
|||
|
|
|
|||
|
|
# 聚簇:位置集合有交集的句子并入同一簇
|
|||
|
|
clusters = []
|
|||
|
|
for loc, raw, hit in pairs:
|
|||
|
|
placed = False
|
|||
|
|
for c in clusters:
|
|||
|
|
if c["locs"] & hit:
|
|||
|
|
c["locs"] |= hit
|
|||
|
|
c["members"].append((loc, raw))
|
|||
|
|
placed = True
|
|||
|
|
break
|
|||
|
|
if not placed:
|
|||
|
|
clusters.append({"locs": set(hit), "members": [(loc, raw)]})
|
|||
|
|
|
|||
|
|
# 合并可传递的簇
|
|||
|
|
changed = True
|
|||
|
|
while changed:
|
|||
|
|
changed = False
|
|||
|
|
for i in range(len(clusters)):
|
|||
|
|
for j in range(i + 1, len(clusters)):
|
|||
|
|
if clusters[i]["locs"] & clusters[j]["locs"]:
|
|||
|
|
clusters[i]["locs"] |= clusters[j]["locs"]
|
|||
|
|
clusters[i]["members"] += clusters[j]["members"]
|
|||
|
|
del clusters[j]
|
|||
|
|
changed = True
|
|||
|
|
break
|
|||
|
|
if changed:
|
|||
|
|
break
|
|||
|
|
|
|||
|
|
clusters = [c for c in clusters if len({p for p, _ in c["locs"]}) >= 2]
|
|||
|
|
clusters.sort(key=lambda c: (-len({p for p, _ in c["locs"]}), -len(c["locs"])))
|
|||
|
|
|
|||
|
|
print(f"# 跨文件复述簇:{len(clusters)} 个(MINLEN={MINLEN})\n")
|
|||
|
|
for k, c in enumerate(clusters, 1):
|
|||
|
|
fs = sorted({p for p, _ in c["locs"]})
|
|||
|
|
print(f"【簇{k}】{len(fs)} 文件 · {len(c['locs'])} 处")
|
|||
|
|
for loc, raw in sorted(c["members"]):
|
|||
|
|
print(f" {loc[0]}#{loc[1]} {raw[:110]}")
|
|||
|
|
print()
|