初始化提交:contentm_agent 工作区全量快照
内容分四块: 1、产品规划产出 —— MCN 短视频整合营销工作台的①段五份(1a 需求/1b 竞品/1c 画像/1d 策略/1e 场景)、②段两份(2a 功能/2b 布局)、③段界面(DESIGN.md 契约与令牌表 + mcn-workbench.html 原型 + 实测/会诊/审查三份 + 23 张闸门截图)。 2、开源竞品调研 —— 5 个内容工作台项目的取证原始件与 1b 系列分析文档。 3、参考资料 —— 竞品视频抽帧 1145 张 + 2 个源视频 + 功能点截图。 4、机制侧 —— 协作脚本与状态台账、工作区记忆日志、抽帧/OCR 脚本。 .gitignore 只排运行时日志、脚本备份副本与一次性探针输出,其余按原样入库。
This commit is contained in:
commit
df56c2c137
1773 files changed
+205840
No files matched your search
@@ -0,0 +1,109 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""复述检测 v2:按「最大重复簇」聚合输出。
|
||||
|
||||
输出形如:
|
||||
【簇 3 文件 · 5 处】关键词行
|
||||
- 文件A#12
|
||||
- 文件B#40
|
||||
...
|
||||
只读,不写被扫描文件。
|
||||
"""
|
||||
import os, re, sys, collections
|
||||
|
||||
ROOT = sys.argv[1] if len(sys.argv) > 1 else "."
|
||||
MINLEN = int(sys.argv[2]) if len(sys.argv) > 2 else 14
|
||||
SKIP_DIRS = {"assets", "归档", "_留痕", ".git", "node_modules", "vendor"}
|
||||
SKIP_NAME = {"_superseded-competitor-analysis-英文商业版.md"}
|
||||
|
||||
|
||||
def norm(s: str) -> str:
|
||||
s = re.sub(r"`[^`]*`", "", s)
|
||||
s = re.sub(r"[*#>|\-\s`]+", "", s)
|
||||
s = re.sub(r"[,。!?;:、()()《》「」『』\[\]【】…—~~·\"'’“”]", "", s)
|
||||
return s
|
||||
|
||||
|
||||
def sentences(text):
|
||||
out = []
|
||||
for ln in text.splitlines():
|
||||
ln = ln.strip()
|
||||
if not ln:
|
||||
continue
|
||||
for piece in re.split(r"(?<=[。!?;])", ln):
|
||||
p = piece.strip()
|
||||
if p:
|
||||
out.append(p)
|
||||
return out
|
||||
|
||||
|
||||
files = []
|
||||
for dp, dns, fns in os.walk(ROOT):
|
||||
dns[:] = [d for d in dns if d not in SKIP_DIRS]
|
||||
for fn in fns:
|
||||
if fn.endswith(".md") and fn not in SKIP_NAME:
|
||||
files.append(os.path.join(dp, fn))
|
||||
files.sort()
|
||||
|
||||
sent = [] # (loc, raw, norm)
|
||||
for path in files:
|
||||
with open(path, encoding="utf-8") as f:
|
||||
for i, raw in enumerate(sentences(f.read())):
|
||||
n = norm(raw)
|
||||
if len(n) >= MINLEN:
|
||||
sent.append(((path, i), raw, n))
|
||||
|
||||
sh_map = collections.defaultdict(set)
|
||||
for loc, raw, n in sent:
|
||||
for j in range(len(n) - MINLEN + 1):
|
||||
sh_map[n[j:j + MINLEN]].add(loc)
|
||||
|
||||
# 句级:收集它命中的、跨≥2文件的 shingle 所覆盖的全部位置
|
||||
pairs = []
|
||||
for loc, raw, n in sent:
|
||||
hit = set()
|
||||
for j in range(max(1, len(n) - MINLEN + 1)):
|
||||
locs = sh_map.get(n[j:j + MINLEN])
|
||||
if locs and len({p for p, _ in locs}) >= 2:
|
||||
hit |= locs
|
||||
if len({p for p, _ in hit}) >= 2:
|
||||
pairs.append((loc, raw, hit))
|
||||
|
||||
# 聚簇:位置集合有交集的句子并入同一簇
|
||||
clusters = []
|
||||
for loc, raw, hit in pairs:
|
||||
placed = False
|
||||
for c in clusters:
|
||||
if c["locs"] & hit:
|
||||
c["locs"] |= hit
|
||||
c["members"].append((loc, raw))
|
||||
placed = True
|
||||
break
|
||||
if not placed:
|
||||
clusters.append({"locs": set(hit), "members": [(loc, raw)]})
|
||||
|
||||
# 合并可传递的簇
|
||||
changed = True
|
||||
while changed:
|
||||
changed = False
|
||||
for i in range(len(clusters)):
|
||||
for j in range(i + 1, len(clusters)):
|
||||
if clusters[i]["locs"] & clusters[j]["locs"]:
|
||||
clusters[i]["locs"] |= clusters[j]["locs"]
|
||||
clusters[i]["members"] += clusters[j]["members"]
|
||||
del clusters[j]
|
||||
changed = True
|
||||
break
|
||||
if changed:
|
||||
break
|
||||
|
||||
clusters = [c for c in clusters if len({p for p, _ in c["locs"]}) >= 2]
|
||||
clusters.sort(key=lambda c: (-len({p for p, _ in c["locs"]}), -len(c["locs"])))
|
||||
|
||||
print(f"# 跨文件复述簇:{len(clusters)} 个(MINLEN={MINLEN})\n")
|
||||
for k, c in enumerate(clusters, 1):
|
||||
fs = sorted({p for p, _ in c["locs"]})
|
||||
print(f"【簇{k}】{len(fs)} 文件 · {len(c['locs'])} 处")
|
||||
for loc, raw in sorted(c["members"]):
|
||||
print(f" {loc[0]}#{loc[1]} {raw[:110]}")
|
||||
print()
|
||||
Reference in new issue
Block a user