Files

119 lines
5.0 KiB
Python
Raw Permalink Normal View History

# -*- coding: utf-8 -*-
"""取证脚本(第 5 棒):补拉「文到 AI」公开可得面。
对象:wendaoai/wendao-content-workbench(闭源仓库,全仓仅 5 blob)。
策略:
- raw.githubusercontent.com 本机通道不稳 ⇒ 一律走 GitHub `contents` API 取正文(返回 base64);
- 另拉 releases / tags / languages / contributors / 组织仓库列表 等元数据面;
- 产物落 取证/wendao/,逐件记 HTTP 状态与字节数(⛔ 不吃静默失败)。
只读 GitHub 公开 API;不下载发布包、不安装依赖。
"""
import base64
import json
import os
import sys
import time
import urllib.error
import urllib.request
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
HERE = os.path.dirname(os.path.abspath(__file__))
OUT = os.path.join(HERE, "wendao")
os.makedirs(OUT, exist_ok=True)
FULL = "wendaoai/wendao-content-workbench"
ORG = "wendaoai"
BRANCH = "main"
HDRS = {
"User-Agent": "wb-research/1.0",
"Accept": "application/vnd.github+json",
}
LOG = []
def get_json(url):
req = urllib.request.Request(url, headers=HDRS)
with urllib.request.urlopen(req, timeout=45) as r:
return json.loads(r.read().decode("utf-8")), r.status
def fetch_contents(path, out_name):
"""contents API 取正文;返回 (ok, bytes, note)。"""
url = "https://api.github.com/repos/%s/contents/%s?ref=%s" % (FULL, path, BRANCH)
try:
obj, status = get_json(url)
raw = base64.b64decode(obj.get("content", ""))
with open(os.path.join(OUT, out_name), "wb") as f:
f.write(raw)
rec = {"path": path, "out": out_name, "http": status, "bytes": len(raw),
"sha": obj.get("sha"), "fetched_at": time.strftime("%Y-%m-%dT%H:%M:%S")}
LOG.append(rec)
print(" OK %-32s %7d B http=%s" % (path, len(raw), status))
return True
except urllib.error.HTTPError as e:
LOG.append({"path": path, "http": e.code, "error": "HTTP %s" % e.code})
print(" ERR %-32s HTTP %s" % (path, e.code))
return False
except Exception as e: # noqa: BLE001
LOG.append({"path": path, "error": repr(e)})
print(" ERR %-32s %r" % (path, e))
return False
def fetch_api(url, out_name, note):
try:
obj, status = get_json(url)
with open(os.path.join(OUT, out_name), "w", encoding="utf-8") as f:
json.dump(obj, f, ensure_ascii=False, indent=1)
n = len(obj) if isinstance(obj, list) else "-"
LOG.append({"api": note, "out": out_name, "http": status,
"items": n, "fetched_at": time.strftime("%Y-%m-%dT%H:%M:%S")})
print(" OK %-32s items=%s http=%s" % (note, n, status))
return obj
except urllib.error.HTTPError as e:
LOG.append({"api": note, "http": e.code, "error": "HTTP %s" % e.code})
print(" ERR %-32s HTTP %s" % (note, e.code))
return None
except Exception as e: # noqa: BLE001
LOG.append({"api": note, "error": repr(e)})
print(" ERR %-32s %r" % (note, e))
return None
print("== 1) 仓内 5 个 blob 的正文(contents API)==")
# 依据第 1 棒 wendao.tree.json:全仓共 5 个 blob
fetch_contents("README.md", "wen_README.md")
fetch_contents("NOTICE.md", "wen_NOTICE.md")
fetch_contents("RELEASE_NOTES_0.0.113.md", "wen_RELEASE_NOTES_0.0.113.md")
fetch_contents(".gitignore", "wen_gitignore.txt")
print("\n== 2) assets 目录清单(含 desktop-home.png)==")
fetch_api("https://api.github.com/repos/%s/contents/assets?ref=%s" % (FULL, BRANCH),
"wen_assets_dir.json", "contents:assets")
print("\n== 3) 元数据面:releases / tags / languages / contributors ==")
rel = fetch_api("https://api.github.com/repos/%s/releases" % FULL, "wen_releases.json", "releases")
fetch_api("https://api.github.com/repos/%s/tags" % FULL, "wen_tags.json", "tags")
fetch_api("https://api.github.com/repos/%s/languages" % FULL, "wen_languages.json", "languages")
fetch_api("https://api.github.com/repos/%s/contributors?per_page=100" % FULL,
"wen_contributors.json", "contributors")
fetch_api("https://api.github.com/repos/%s/branches" % FULL, "wen_branches.json", "branches")
fetch_api("https://api.github.com/repos/%s/commits?per_page=100" % FULL,
"wen_commits.json", "commits")
fetch_api("https://api.github.com/repos/%s/forks" % FULL, "wen_forks.json", "forks")
print("\n== 4) 组织 wendaoai 的公开仓库列表(旁证:是否还有其它仓)==")
fetch_api("https://api.github.com/users/%s/repos?per_page=100&sort=updated" % ORG,
"wen_org_repos.json", "org:repos")
fetch_api("https://api.github.com/users/%s" % ORG, "wen_org.json", "org:profile")
with open(os.path.join(OUT, "_fetch_log.json"), "w", encoding="utf-8") as f:
json.dump({"full_name": FULL, "branch": BRANCH,
"fetched_at": time.strftime("%Y-%m-%dT%H:%M:%S"),
"items": LOG}, f, ensure_ascii=False, indent=1)
ok = sum(1 for x in LOG if "error" not in x)
print("\nDONE -> %s 成功 %d / 共 %d" % (OUT, ok, len(LOG)))