Files
WorkBuddy df56c2c137 初始化提交:contentm_agent 工作区全量快照
内容分四块:
1、产品规划产出 —— MCN 短视频整合营销工作台的①段五份(1a 需求/1b 竞品/1c 画像/1d 策略/1e 场景)、②段两份(2a 功能/2b 布局)、③段界面(DESIGN.md 契约与令牌表 + mcn-workbench.html 原型 + 实测/会诊/审查三份 + 23 张闸门截图)。
2、开源竞品调研 —— 5 个内容工作台项目的取证原始件与 1b 系列分析文档。
3、参考资料 —— 竞品视频抽帧 1145 张 + 2 个源视频 + 功能点截图。
4、机制侧 —— 协作脚本与状态台账、工作区记忆日志、抽帧/OCR 脚本。

.gitignore 只排运行时日志、脚本备份副本与一次性探针输出,其余按原样入库。
2026-10-08 08:13:02 +08:00

110 lines
3.3 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""复述检测 v2:按「最大重复簇」聚合输出。
输出形如:
【簇 3 文件 · 5 处】关键词行
- 文件A#12
- 文件B#40
...
只读,不写被扫描文件。
"""
import os, re, sys, collections
ROOT = sys.argv[1] if len(sys.argv) > 1 else "."
MINLEN = int(sys.argv[2]) if len(sys.argv) > 2 else 14
SKIP_DIRS = {"assets", "归档", "_留痕", ".git", "node_modules", "vendor"}
SKIP_NAME = {"_superseded-competitor-analysis-英文商业版.md"}
def norm(s: str) -> str:
s = re.sub(r"`[^`]*`", "", s)
s = re.sub(r"[*#>|\-\s`]+", "", s)
s = re.sub(r"[,。!?;:、()()《》「」『』\[\]【】…—~~·\"'’“”]", "", s)
return s
def sentences(text):
out = []
for ln in text.splitlines():
ln = ln.strip()
if not ln:
continue
for piece in re.split(r"(?<=[。!?;])", ln):
p = piece.strip()
if p:
out.append(p)
return out
files = []
for dp, dns, fns in os.walk(ROOT):
dns[:] = [d for d in dns if d not in SKIP_DIRS]
for fn in fns:
if fn.endswith(".md") and fn not in SKIP_NAME:
files.append(os.path.join(dp, fn))
files.sort()
sent = [] # (loc, raw, norm)
for path in files:
with open(path, encoding="utf-8") as f:
for i, raw in enumerate(sentences(f.read())):
n = norm(raw)
if len(n) >= MINLEN:
sent.append(((path, i), raw, n))
sh_map = collections.defaultdict(set)
for loc, raw, n in sent:
for j in range(len(n) - MINLEN + 1):
sh_map[n[j:j + MINLEN]].add(loc)
# 句级:收集它命中的、跨≥2文件的 shingle 所覆盖的全部位置
pairs = []
for loc, raw, n in sent:
hit = set()
for j in range(max(1, len(n) - MINLEN + 1)):
locs = sh_map.get(n[j:j + MINLEN])
if locs and len({p for p, _ in locs}) >= 2:
hit |= locs
if len({p for p, _ in hit}) >= 2:
pairs.append((loc, raw, hit))
# 聚簇:位置集合有交集的句子并入同一簇
clusters = []
for loc, raw, hit in pairs:
placed = False
for c in clusters:
if c["locs"] & hit:
c["locs"] |= hit
c["members"].append((loc, raw))
placed = True
break
if not placed:
clusters.append({"locs": set(hit), "members": [(loc, raw)]})
# 合并可传递的簇
changed = True
while changed:
changed = False
for i in range(len(clusters)):
for j in range(i + 1, len(clusters)):
if clusters[i]["locs"] & clusters[j]["locs"]:
clusters[i]["locs"] |= clusters[j]["locs"]
clusters[i]["members"] += clusters[j]["members"]
del clusters[j]
changed = True
break
if changed:
break
clusters = [c for c in clusters if len({p for p, _ in c["locs"]}) >= 2]
clusters.sort(key=lambda c: (-len({p for p, _ in c["locs"]}), -len(c["locs"])))
print(f"# 跨文件复述簇:{len(clusters)} 个(MINLEN={MINLEN})\n")
for k, c in enumerate(clusters, 1):
fs = sorted({p for p, _ in c["locs"]})
print(f"【簇{k}】{len(fs)} 文件 · {len(c['locs'])} 处")
for loc, raw in sorted(c["members"]):
print(f" {loc[0]}#{loc[1]} {raw[:110]}")
print()