初始化提交:contentm_agent 工作区全量快照

内容分四块:
1、产品规划产出 —— MCN 短视频整合营销工作台的①段五份(1a 需求/1b 竞品/1c 画像/1d 策略/1e 场景)、②段两份(2a 功能/2b 布局)、③段界面(DESIGN.md 契约与令牌表 + mcn-workbench.html 原型 + 实测/会诊/审查三份 + 23 张闸门截图)。
2、开源竞品调研 —— 5 个内容工作台项目的取证原始件与 1b 系列分析文档。
3、参考资料 —— 竞品视频抽帧 1145 张 + 2 个源视频 + 功能点截图。
4、机制侧 —— 协作脚本与状态台账、工作区记忆日志、抽帧/OCR 脚本。

.gitignore 只排运行时日志、脚本备份副本与一次性探针输出,其余按原样入库。
This commit is contained in:
WorkBuddy committed 2026-10-08 08:13:02 +08:00
commit df56c2c137
1773 files changed
+205840

No files matched your search

+109
View File
@@ -0,0 +1,109 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""复述检测 v2:按「最大重复簇」聚合输出。
输出形如:
【簇 3 文件 · 5 处】关键词行
- 文件A#12
- 文件B#40
...
只读,不写被扫描文件。
"""
import os, re, sys, collections
ROOT = sys.argv[1] if len(sys.argv) > 1 else "."
MINLEN = int(sys.argv[2]) if len(sys.argv) > 2 else 14
SKIP_DIRS = {"assets", "归档", "_留痕", ".git", "node_modules", "vendor"}
SKIP_NAME = {"_superseded-competitor-analysis-英文商业版.md"}
def norm(s: str) -> str:
s = re.sub(r"`[^`]*`", "", s)
s = re.sub(r"[*#>|\-\s`]+", "", s)
s = re.sub(r"[,。!?;:、()()《》「」『』\[\]【】…—~~·\"'’“”]", "", s)
return s
def sentences(text):
out = []
for ln in text.splitlines():
ln = ln.strip()
if not ln:
continue
for piece in re.split(r"(?<=[。!?;])", ln):
p = piece.strip()
if p:
out.append(p)
return out
files = []
for dp, dns, fns in os.walk(ROOT):
dns[:] = [d for d in dns if d not in SKIP_DIRS]
for fn in fns:
if fn.endswith(".md") and fn not in SKIP_NAME:
files.append(os.path.join(dp, fn))
files.sort()
sent = [] # (loc, raw, norm)
for path in files:
with open(path, encoding="utf-8") as f:
for i, raw in enumerate(sentences(f.read())):
n = norm(raw)
if len(n) >= MINLEN:
sent.append(((path, i), raw, n))
sh_map = collections.defaultdict(set)
for loc, raw, n in sent:
for j in range(len(n) - MINLEN + 1):
sh_map[n[j:j + MINLEN]].add(loc)
# 句级:收集它命中的、跨≥2文件的 shingle 所覆盖的全部位置
pairs = []
for loc, raw, n in sent:
hit = set()
for j in range(max(1, len(n) - MINLEN + 1)):
locs = sh_map.get(n[j:j + MINLEN])
if locs and len({p for p, _ in locs}) >= 2:
hit |= locs
if len({p for p, _ in hit}) >= 2:
pairs.append((loc, raw, hit))
# 聚簇:位置集合有交集的句子并入同一簇
clusters = []
for loc, raw, hit in pairs:
placed = False
for c in clusters:
if c["locs"] & hit:
c["locs"] |= hit
c["members"].append((loc, raw))
placed = True
break
if not placed:
clusters.append({"locs": set(hit), "members": [(loc, raw)]})
# 合并可传递的簇
changed = True
while changed:
changed = False
for i in range(len(clusters)):
for j in range(i + 1, len(clusters)):
if clusters[i]["locs"] & clusters[j]["locs"]:
clusters[i]["locs"] |= clusters[j]["locs"]
clusters[i]["members"] += clusters[j]["members"]
del clusters[j]
changed = True
break
if changed:
break
clusters = [c for c in clusters if len({p for p, _ in c["locs"]}) >= 2]
clusters.sort(key=lambda c: (-len({p for p, _ in c["locs"]}), -len(c["locs"])))
print(f"# 跨文件复述簇:{len(clusters)} 个(MINLEN={MINLEN})\n")
for k, c in enumerate(clusters, 1):
fs = sorted({p for p, _ in c["locs"]})
print(f"【簇{k}】{len(fs)} 文件 · {len(c['locs'])} 处")
for loc, raw in sorted(c["members"]):
print(f" {loc[0]}#{loc[1]} {raw[:110]}")
print()