内容分四块: 1、产品规划产出 —— MCN 短视频整合营销工作台的①段五份(1a 需求/1b 竞品/1c 画像/1d 策略/1e 场景)、②段两份(2a 功能/2b 布局)、③段界面(DESIGN.md 契约与令牌表 + mcn-workbench.html 原型 + 实测/会诊/审查三份 + 23 张闸门截图)。 2、开源竞品调研 —— 5 个内容工作台项目的取证原始件与 1b 系列分析文档。 3、参考资料 —— 竞品视频抽帧 1145 张 + 2 个源视频 + 功能点截图。 4、机制侧 —— 协作脚本与状态台账、工作区记忆日志、抽帧/OCR 脚本。 .gitignore 只排运行时日志、脚本备份副本与一次性探针输出,其余按原样入库。
110 lines
3.3 KiB
Python
110 lines
3.3 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""复述检测 v2:按「最大重复簇」聚合输出。
|
||
|
||
输出形如:
|
||
【簇 3 文件 · 5 处】关键词行
|
||
- 文件A#12
|
||
- 文件B#40
|
||
...
|
||
只读,不写被扫描文件。
|
||
"""
|
||
import os, re, sys, collections
|
||
|
||
ROOT = sys.argv[1] if len(sys.argv) > 1 else "."
|
||
MINLEN = int(sys.argv[2]) if len(sys.argv) > 2 else 14
|
||
SKIP_DIRS = {"assets", "归档", "_留痕", ".git", "node_modules", "vendor"}
|
||
SKIP_NAME = {"_superseded-competitor-analysis-英文商业版.md"}
|
||
|
||
|
||
def norm(s: str) -> str:
|
||
s = re.sub(r"`[^`]*`", "", s)
|
||
s = re.sub(r"[*#>|\-\s`]+", "", s)
|
||
s = re.sub(r"[,。!?;:、()()《》「」『』\[\]【】…—~~·\"'’“”]", "", s)
|
||
return s
|
||
|
||
|
||
def sentences(text):
|
||
out = []
|
||
for ln in text.splitlines():
|
||
ln = ln.strip()
|
||
if not ln:
|
||
continue
|
||
for piece in re.split(r"(?<=[。!?;])", ln):
|
||
p = piece.strip()
|
||
if p:
|
||
out.append(p)
|
||
return out
|
||
|
||
|
||
files = []
|
||
for dp, dns, fns in os.walk(ROOT):
|
||
dns[:] = [d for d in dns if d not in SKIP_DIRS]
|
||
for fn in fns:
|
||
if fn.endswith(".md") and fn not in SKIP_NAME:
|
||
files.append(os.path.join(dp, fn))
|
||
files.sort()
|
||
|
||
sent = [] # (loc, raw, norm)
|
||
for path in files:
|
||
with open(path, encoding="utf-8") as f:
|
||
for i, raw in enumerate(sentences(f.read())):
|
||
n = norm(raw)
|
||
if len(n) >= MINLEN:
|
||
sent.append(((path, i), raw, n))
|
||
|
||
sh_map = collections.defaultdict(set)
|
||
for loc, raw, n in sent:
|
||
for j in range(len(n) - MINLEN + 1):
|
||
sh_map[n[j:j + MINLEN]].add(loc)
|
||
|
||
# 句级:收集它命中的、跨≥2文件的 shingle 所覆盖的全部位置
|
||
pairs = []
|
||
for loc, raw, n in sent:
|
||
hit = set()
|
||
for j in range(max(1, len(n) - MINLEN + 1)):
|
||
locs = sh_map.get(n[j:j + MINLEN])
|
||
if locs and len({p for p, _ in locs}) >= 2:
|
||
hit |= locs
|
||
if len({p for p, _ in hit}) >= 2:
|
||
pairs.append((loc, raw, hit))
|
||
|
||
# 聚簇:位置集合有交集的句子并入同一簇
|
||
clusters = []
|
||
for loc, raw, hit in pairs:
|
||
placed = False
|
||
for c in clusters:
|
||
if c["locs"] & hit:
|
||
c["locs"] |= hit
|
||
c["members"].append((loc, raw))
|
||
placed = True
|
||
break
|
||
if not placed:
|
||
clusters.append({"locs": set(hit), "members": [(loc, raw)]})
|
||
|
||
# 合并可传递的簇
|
||
changed = True
|
||
while changed:
|
||
changed = False
|
||
for i in range(len(clusters)):
|
||
for j in range(i + 1, len(clusters)):
|
||
if clusters[i]["locs"] & clusters[j]["locs"]:
|
||
clusters[i]["locs"] |= clusters[j]["locs"]
|
||
clusters[i]["members"] += clusters[j]["members"]
|
||
del clusters[j]
|
||
changed = True
|
||
break
|
||
if changed:
|
||
break
|
||
|
||
clusters = [c for c in clusters if len({p for p, _ in c["locs"]}) >= 2]
|
||
clusters.sort(key=lambda c: (-len({p for p, _ in c["locs"]}), -len(c["locs"])))
|
||
|
||
print(f"# 跨文件复述簇:{len(clusters)} 个(MINLEN={MINLEN})\n")
|
||
for k, c in enumerate(clusters, 1):
|
||
fs = sorted({p for p, _ in c["locs"]})
|
||
print(f"【簇{k}】{len(fs)} 文件 · {len(c['locs'])} 处")
|
||
for loc, raw in sorted(c["members"]):
|
||
print(f" {loc[0]}#{loc[1]} {raw[:110]}")
|
||
print()
|