内容分四块: 1、产品规划产出 —— MCN 短视频整合营销工作台的①段五份(1a 需求/1b 竞品/1c 画像/1d 策略/1e 场景)、②段两份(2a 功能/2b 布局)、③段界面(DESIGN.md 契约与令牌表 + mcn-workbench.html 原型 + 实测/会诊/审查三份 + 23 张闸门截图)。 2、开源竞品调研 —— 5 个内容工作台项目的取证原始件与 1b 系列分析文档。 3、参考资料 —— 竞品视频抽帧 1145 张 + 2 个源视频 + 功能点截图。 4、机制侧 —— 协作脚本与状态台账、工作区记忆日志、抽帧/OCR 脚本。 .gitignore 只排运行时日志、脚本备份副本与一次性探针输出,其余按原样入库。
101 lines
3.5 KiB
Python
101 lines
3.5 KiB
Python
#!/usr/bin/env python3
|
|
# -*- coding: utf-8 -*-
|
|
"""复述检测:找出技能包里「同一件事被写了好几处」的候选。
|
|
|
|
思路:把每份 md 拆成句子(按 。!?;\n 切),归一化(去 markdown 记号/空白),
|
|
再做字符 shingle(默认 16 字),统计同一 shingle 出现在多少个「不同位置」。
|
|
位置 = (文件, 句子序号)。跨文件或跨段落重复 = 复述嫌疑。
|
|
|
|
只读,不写任何被扫描文件。
|
|
"""
|
|
import os, re, sys, collections
|
|
|
|
ROOT = sys.argv[1] if len(sys.argv) > 1 else "."
|
|
MINLEN = int(sys.argv[2]) if len(sys.argv) > 2 else 16
|
|
SKIP_DIRS = {"assets", "归档", "_留痕", ".git", "node_modules", "vendor"}
|
|
SKIP_NAME = {"_superseded-competitor-analysis-英文商业版.md"}
|
|
|
|
def norm(s: str) -> str:
|
|
s = re.sub(r"`[^`]*`", "", s) # 去行内代码(路径/命令不算复述)
|
|
s = re.sub(r"[*#>|\-\s`]+", "", s) # 去 markdown 记号与空白
|
|
s = re.sub(r"[,。!?;:、()()《》「」『』\[\]【】…—~~·\"'’“”]", "", s)
|
|
return s
|
|
|
|
def sentences(text):
|
|
# 先按行,再按句号族切
|
|
out = []
|
|
for ln in text.splitlines():
|
|
ln = ln.strip()
|
|
if not ln:
|
|
continue
|
|
for piece in re.split(r"(?<=[。!?;])", ln):
|
|
p = piece.strip()
|
|
if p:
|
|
out.append(p)
|
|
return out
|
|
|
|
files = []
|
|
for dp, dns, fns in os.walk(ROOT):
|
|
dns[:] = [d for d in dns if d not in SKIP_DIRS]
|
|
for fn in fns:
|
|
if not fn.endswith(".md"):
|
|
continue
|
|
if fn in SKIP_NAME:
|
|
continue
|
|
files.append(os.path.join(dp, fn))
|
|
|
|
shingle_map = collections.defaultdict(set) # shingle -> set(loc)
|
|
sent_index = [] # loc -> (path, idx, raw)
|
|
for path in files:
|
|
with open(path, encoding="utf-8") as f:
|
|
text = f.read()
|
|
for i, raw in enumerate(sentences(text)):
|
|
n = norm(raw)
|
|
if len(n) < MINLEN:
|
|
continue
|
|
loc = (path, i)
|
|
sent_index.append((loc, raw, n))
|
|
for j in range(0, len(n) - MINLEN + 1):
|
|
shingle_map[n[j:j + MINLEN]].add(loc)
|
|
|
|
# 每个 shingle 落在几个不同文件 / 几个不同位置
|
|
def spread(locs):
|
|
return len(locs), len({p for p, _ in locs})
|
|
|
|
# 句级打标:一句里若含「跨文件或跨段」的高频 shingle,就是复述嫌疑
|
|
cand = []
|
|
for loc, raw, n in sent_index:
|
|
hits = set()
|
|
for j in range(0, max(1, len(n) - MINLEN + 1)):
|
|
sh = n[j:j + MINLEN]
|
|
locs = shingle_map.get(sh)
|
|
if not locs:
|
|
continue
|
|
nf, nl = spread(locs)
|
|
if nl >= 2:
|
|
hits.update(locs)
|
|
if hits:
|
|
nf, nl = spread(hits)
|
|
cand.append((nf, nl, loc, raw))
|
|
|
|
cand.sort(key=lambda x: (-x[1], -x[0], x[2][0]))
|
|
seen = set()
|
|
print(f"# 复述嫌疑(跨 ≥2 流程位置,MINLEN={MINLEN})—— 共 {len(cand)} 条\n")
|
|
for nf, nl, loc, raw in cand[:120]:
|
|
key = loc
|
|
if key in seen:
|
|
continue
|
|
seen.add(key)
|
|
others = sorted({p for p, _ in shingle_map.get(norm(raw)[:MINLEN], set())})
|
|
tag = "跨文件" if nl >= 2 else ""
|
|
print(f"[{nl} 位置 {tag}] {loc[0]}#{loc[1]}")
|
|
print(f" {raw[:160]}")
|
|
print("\n# —— 出现次数最多的 16 字片段(TOP 40)——")
|
|
top = sorted(((len(v), tuple(sorted(v))) for v in shingle_map.values()),
|
|
key=lambda x: -x[0])[:40]
|
|
for cnt, locs in top:
|
|
if cnt < 3:
|
|
break
|
|
fileset = sorted({p for p, _ in locs})
|
|
print(f"{cnt}x 文件数{len(fileset)} | 示例: {fileset[0]}#{sorted(locs)[0][1]}")
|