Files
contentm_agent/tmp/dup_scan.py
T

100 lines
3.5 KiB
Python
Raw Normal View History

#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""复述检测:找出技能包里「同一件事被写了好几处」的候选。
思路:把每份 md 拆成句子(按 。!?;\n 切),归一化(去 markdown 记号/空白),
再做字符 shingle(默认 16 字),统计同一 shingle 出现在多少个「不同位置」。
位置 = (文件, 句子序号)。跨文件或跨段落重复 = 复述嫌疑。
只读,不写任何被扫描文件。
"""
import os, re, sys, collections
ROOT = sys.argv[1] if len(sys.argv) > 1 else "."
MINLEN = int(sys.argv[2]) if len(sys.argv) > 2 else 16
SKIP_DIRS = {"assets", "归档", "_留痕", ".git", "node_modules", "vendor"}
SKIP_NAME = {"_superseded-competitor-analysis-英文商业版.md"}
def norm(s: str) -> str:
s = re.sub(r"`[^`]*`", "", s) # 去行内代码(路径/命令不算复述)
s = re.sub(r"[*#>|\-\s`]+", "", s) # 去 markdown 记号与空白
s = re.sub(r"[,。!?;:、()()《》「」『』\[\]【】…—~~·\"'’“”]", "", s)
return s
def sentences(text):
# 先按行,再按句号族切
out = []
for ln in text.splitlines():
ln = ln.strip()
if not ln:
continue
for piece in re.split(r"(?<=[。!?;])", ln):
p = piece.strip()
if p:
out.append(p)
return out
files = []
for dp, dns, fns in os.walk(ROOT):
dns[:] = [d for d in dns if d not in SKIP_DIRS]
for fn in fns:
if not fn.endswith(".md"):
continue
if fn in SKIP_NAME:
continue
files.append(os.path.join(dp, fn))
shingle_map = collections.defaultdict(set) # shingle -> set(loc)
sent_index = [] # loc -> (path, idx, raw)
for path in files:
with open(path, encoding="utf-8") as f:
text = f.read()
for i, raw in enumerate(sentences(text)):
n = norm(raw)
if len(n) < MINLEN:
continue
loc = (path, i)
sent_index.append((loc, raw, n))
for j in range(0, len(n) - MINLEN + 1):
shingle_map[n[j:j + MINLEN]].add(loc)
# 每个 shingle 落在几个不同文件 / 几个不同位置
def spread(locs):
return len(locs), len({p for p, _ in locs})
# 句级打标:一句里若含「跨文件或跨段」的高频 shingle,就是复述嫌疑
cand = []
for loc, raw, n in sent_index:
hits = set()
for j in range(0, max(1, len(n) - MINLEN + 1)):
sh = n[j:j + MINLEN]
locs = shingle_map.get(sh)
if not locs:
continue
nf, nl = spread(locs)
if nl >= 2:
hits.update(locs)
if hits:
nf, nl = spread(hits)
cand.append((nf, nl, loc, raw))
cand.sort(key=lambda x: (-x[1], -x[0], x[2][0]))
seen = set()
print(f"# 复述嫌疑(跨 ≥2 流程位置,MINLEN={MINLEN})—— 共 {len(cand)} 条\n")
for nf, nl, loc, raw in cand[:120]:
key = loc
if key in seen:
continue
seen.add(key)
others = sorted({p for p, _ in shingle_map.get(norm(raw)[:MINLEN], set())})
tag = "跨文件" if nl >= 2 else ""
print(f"[{nl} 位置 {tag}] {loc[0]}#{loc[1]}")
print(f" {raw[:160]}")
print("\n# —— 出现次数最多的 16 字片段(TOP 40)——")
top = sorted(((len(v), tuple(sorted(v))) for v in shingle_map.values()),
key=lambda x: -x[0])[:40]
for cnt, locs in top:
if cnt < 3:
break
fileset = sorted({p for p, _ in locs})
print(f"{cnt}x 文件数{len(fileset)} | 示例: {fileset[0]}#{sorted(locs)[0][1]}")