内容分四块: 1、产品规划产出 —— MCN 短视频整合营销工作台的①段五份(1a 需求/1b 竞品/1c 画像/1d 策略/1e 场景)、②段两份(2a 功能/2b 布局)、③段界面(DESIGN.md 契约与令牌表 + mcn-workbench.html 原型 + 实测/会诊/审查三份 + 23 张闸门截图)。 2、开源竞品调研 —— 5 个内容工作台项目的取证原始件与 1b 系列分析文档。 3、参考资料 —— 竞品视频抽帧 1145 张 + 2 个源视频 + 功能点截图。 4、机制侧 —— 协作脚本与状态台账、工作区记忆日志、抽帧/OCR 脚本。 .gitignore 只排运行时日志、脚本备份副本与一次性探针输出,其余按原样入库。
120 lines
5.0 KiB
Python
120 lines
5.0 KiB
Python
# -*- coding: utf-8 -*-
|
||
"""取证脚本(第 5 棒):补拉「文到 AI」公开可得面。
|
||
|
||
对象:wendaoai/wendao-content-workbench(闭源仓库,全仓仅 5 blob)。
|
||
策略:
|
||
- raw.githubusercontent.com 本机通道不稳 ⇒ 一律走 GitHub `contents` API 取正文(返回 base64);
|
||
- 另拉 releases / tags / languages / contributors / 组织仓库列表 等元数据面;
|
||
- 产物落 取证/wendao/,逐件记 HTTP 状态与字节数(⛔ 不吃静默失败)。
|
||
只读 GitHub 公开 API;不下载发布包、不安装依赖。
|
||
"""
|
||
import base64
|
||
import json
|
||
import os
|
||
import sys
|
||
import time
|
||
import urllib.error
|
||
import urllib.request
|
||
|
||
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
||
|
||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||
OUT = os.path.join(HERE, "wendao")
|
||
os.makedirs(OUT, exist_ok=True)
|
||
|
||
FULL = "wendaoai/wendao-content-workbench"
|
||
ORG = "wendaoai"
|
||
BRANCH = "main"
|
||
|
||
HDRS = {
|
||
"User-Agent": "wb-research/1.0",
|
||
"Accept": "application/vnd.github+json",
|
||
}
|
||
|
||
LOG = []
|
||
|
||
|
||
def get_json(url):
|
||
req = urllib.request.Request(url, headers=HDRS)
|
||
with urllib.request.urlopen(req, timeout=45) as r:
|
||
return json.loads(r.read().decode("utf-8")), r.status
|
||
|
||
|
||
def fetch_contents(path, out_name):
|
||
"""contents API 取正文;返回 (ok, bytes, note)。"""
|
||
url = "https://api.github.com/repos/%s/contents/%s?ref=%s" % (FULL, path, BRANCH)
|
||
try:
|
||
obj, status = get_json(url)
|
||
raw = base64.b64decode(obj.get("content", ""))
|
||
with open(os.path.join(OUT, out_name), "wb") as f:
|
||
f.write(raw)
|
||
rec = {"path": path, "out": out_name, "http": status, "bytes": len(raw),
|
||
"sha": obj.get("sha"), "fetched_at": time.strftime("%Y-%m-%dT%H:%M:%S")}
|
||
LOG.append(rec)
|
||
print(" OK %-32s %7d B http=%s" % (path, len(raw), status))
|
||
return True
|
||
except urllib.error.HTTPError as e:
|
||
LOG.append({"path": path, "http": e.code, "error": "HTTP %s" % e.code})
|
||
print(" ERR %-32s HTTP %s" % (path, e.code))
|
||
return False
|
||
except Exception as e: # noqa: BLE001
|
||
LOG.append({"path": path, "error": repr(e)})
|
||
print(" ERR %-32s %r" % (path, e))
|
||
return False
|
||
|
||
|
||
def fetch_api(url, out_name, note):
|
||
try:
|
||
obj, status = get_json(url)
|
||
with open(os.path.join(OUT, out_name), "w", encoding="utf-8") as f:
|
||
json.dump(obj, f, ensure_ascii=False, indent=1)
|
||
n = len(obj) if isinstance(obj, list) else "-"
|
||
LOG.append({"api": note, "out": out_name, "http": status,
|
||
"items": n, "fetched_at": time.strftime("%Y-%m-%dT%H:%M:%S")})
|
||
print(" OK %-32s items=%s http=%s" % (note, n, status))
|
||
return obj
|
||
except urllib.error.HTTPError as e:
|
||
LOG.append({"api": note, "http": e.code, "error": "HTTP %s" % e.code})
|
||
print(" ERR %-32s HTTP %s" % (note, e.code))
|
||
return None
|
||
except Exception as e: # noqa: BLE001
|
||
LOG.append({"api": note, "error": repr(e)})
|
||
print(" ERR %-32s %r" % (note, e))
|
||
return None
|
||
|
||
|
||
print("== 1) 仓内 5 个 blob 的正文(contents API)==")
|
||
# 依据第 1 棒 wendao.tree.json:全仓共 5 个 blob
|
||
fetch_contents("README.md", "wen_README.md")
|
||
fetch_contents("NOTICE.md", "wen_NOTICE.md")
|
||
fetch_contents("RELEASE_NOTES_0.0.113.md", "wen_RELEASE_NOTES_0.0.113.md")
|
||
fetch_contents(".gitignore", "wen_gitignore.txt")
|
||
|
||
print("\n== 2) assets 目录清单(含 desktop-home.png)==")
|
||
fetch_api("https://api.github.com/repos/%s/contents/assets?ref=%s" % (FULL, BRANCH),
|
||
"wen_assets_dir.json", "contents:assets")
|
||
|
||
print("\n== 3) 元数据面:releases / tags / languages / contributors ==")
|
||
rel = fetch_api("https://api.github.com/repos/%s/releases" % FULL, "wen_releases.json", "releases")
|
||
fetch_api("https://api.github.com/repos/%s/tags" % FULL, "wen_tags.json", "tags")
|
||
fetch_api("https://api.github.com/repos/%s/languages" % FULL, "wen_languages.json", "languages")
|
||
fetch_api("https://api.github.com/repos/%s/contributors?per_page=100" % FULL,
|
||
"wen_contributors.json", "contributors")
|
||
fetch_api("https://api.github.com/repos/%s/branches" % FULL, "wen_branches.json", "branches")
|
||
fetch_api("https://api.github.com/repos/%s/commits?per_page=100" % FULL,
|
||
"wen_commits.json", "commits")
|
||
fetch_api("https://api.github.com/repos/%s/forks" % FULL, "wen_forks.json", "forks")
|
||
|
||
print("\n== 4) 组织 wendaoai 的公开仓库列表(旁证:是否还有其它仓)==")
|
||
fetch_api("https://api.github.com/users/%s/repos?per_page=100&sort=updated" % ORG,
|
||
"wen_org_repos.json", "org:repos")
|
||
fetch_api("https://api.github.com/users/%s" % ORG, "wen_org.json", "org:profile")
|
||
|
||
with open(os.path.join(OUT, "_fetch_log.json"), "w", encoding="utf-8") as f:
|
||
json.dump({"full_name": FULL, "branch": BRANCH,
|
||
"fetched_at": time.strftime("%Y-%m-%dT%H:%M:%S"),
|
||
"items": LOG}, f, ensure_ascii=False, indent=1)
|
||
|
||
ok = sum(1 for x in LOG if "error" not in x)
|
||
print("\nDONE -> %s 成功 %d / 共 %d" % (OUT, ok, len(LOG)))
|