Files
contentm_agent/tmp/dup_scan.py
T
WorkBuddy df56c2c137 初始化提交:contentm_agent 工作区全量快照
内容分四块:
1、产品规划产出 —— MCN 短视频整合营销工作台的①段五份(1a 需求/1b 竞品/1c 画像/1d 策略/1e 场景)、②段两份(2a 功能/2b 布局)、③段界面(DESIGN.md 契约与令牌表 + mcn-workbench.html 原型 + 实测/会诊/审查三份 + 23 张闸门截图)。
2、开源竞品调研 —— 5 个内容工作台项目的取证原始件与 1b 系列分析文档。
3、参考资料 —— 竞品视频抽帧 1145 张 + 2 个源视频 + 功能点截图。
4、机制侧 —— 协作脚本与状态台账、工作区记忆日志、抽帧/OCR 脚本。

.gitignore 只排运行时日志、脚本备份副本与一次性探针输出,其余按原样入库。
2026-10-08 08:13:02 +08:00

101 lines
3.5 KiB
Python

#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""复述检测:找出技能包里「同一件事被写了好几处」的候选。
思路:把每份 md 拆成句子(按 。!?;\n 切),归一化(去 markdown 记号/空白),
再做字符 shingle(默认 16 字),统计同一 shingle 出现在多少个「不同位置」。
位置 = (文件, 句子序号)。跨文件或跨段落重复 = 复述嫌疑。
只读,不写任何被扫描文件。
"""
import os, re, sys, collections
ROOT = sys.argv[1] if len(sys.argv) > 1 else "."
MINLEN = int(sys.argv[2]) if len(sys.argv) > 2 else 16
SKIP_DIRS = {"assets", "归档", "_留痕", ".git", "node_modules", "vendor"}
SKIP_NAME = {"_superseded-competitor-analysis-英文商业版.md"}
def norm(s: str) -> str:
s = re.sub(r"`[^`]*`", "", s) # 去行内代码(路径/命令不算复述)
s = re.sub(r"[*#>|\-\s`]+", "", s) # 去 markdown 记号与空白
s = re.sub(r"[,。!?;:、()()《》「」『』\[\]【】…—~~·\"'’“”]", "", s)
return s
def sentences(text):
# 先按行,再按句号族切
out = []
for ln in text.splitlines():
ln = ln.strip()
if not ln:
continue
for piece in re.split(r"(?<=[。!?;])", ln):
p = piece.strip()
if p:
out.append(p)
return out
files = []
for dp, dns, fns in os.walk(ROOT):
dns[:] = [d for d in dns if d not in SKIP_DIRS]
for fn in fns:
if not fn.endswith(".md"):
continue
if fn in SKIP_NAME:
continue
files.append(os.path.join(dp, fn))
shingle_map = collections.defaultdict(set) # shingle -> set(loc)
sent_index = [] # loc -> (path, idx, raw)
for path in files:
with open(path, encoding="utf-8") as f:
text = f.read()
for i, raw in enumerate(sentences(text)):
n = norm(raw)
if len(n) < MINLEN:
continue
loc = (path, i)
sent_index.append((loc, raw, n))
for j in range(0, len(n) - MINLEN + 1):
shingle_map[n[j:j + MINLEN]].add(loc)
# 每个 shingle 落在几个不同文件 / 几个不同位置
def spread(locs):
return len(locs), len({p for p, _ in locs})
# 句级打标:一句里若含「跨文件或跨段」的高频 shingle,就是复述嫌疑
cand = []
for loc, raw, n in sent_index:
hits = set()
for j in range(0, max(1, len(n) - MINLEN + 1)):
sh = n[j:j + MINLEN]
locs = shingle_map.get(sh)
if not locs:
continue
nf, nl = spread(locs)
if nl >= 2:
hits.update(locs)
if hits:
nf, nl = spread(hits)
cand.append((nf, nl, loc, raw))
cand.sort(key=lambda x: (-x[1], -x[0], x[2][0]))
seen = set()
print(f"# 复述嫌疑(跨 ≥2 流程位置,MINLEN={MINLEN})—— 共 {len(cand)} 条\n")
for nf, nl, loc, raw in cand[:120]:
key = loc
if key in seen:
continue
seen.add(key)
others = sorted({p for p, _ in shingle_map.get(norm(raw)[:MINLEN], set())})
tag = "跨文件" if nl >= 2 else ""
print(f"[{nl} 位置 {tag}] {loc[0]}#{loc[1]}")
print(f" {raw[:160]}")
print("\n# —— 出现次数最多的 16 字片段(TOP 40)——")
top = sorted(((len(v), tuple(sorted(v))) for v in shingle_map.values()),
key=lambda x: -x[0])[:40]
for cnt, locs in top:
if cnt < 3:
break
fileset = sorted({p for p, _ in locs})
print(f"{cnt}x 文件数{len(fileset)} | 示例: {fileset[0]}#{sorted(locs)[0][1]}")