# -*- coding: utf-8 -*- """取证脚本:拉取 5 个目标仓库的元数据 / README / 文件树。 只读 GitHub 公开 API,产物落本目录 api/ 下。运行一次即留痕(含取证时间)。 """ import json, os, sys, time, urllib.request, urllib.error sys.stdout.reconfigure(encoding="utf-8", errors="replace") HERE = os.path.dirname(os.path.abspath(__file__)) OUT = os.path.join(HERE, "api") os.makedirs(OUT, exist_ok=True) REPOS = [ ("easel", "ZJU-REAL/Easel"), ("wendao", "wendaoai/wendao-content-workbench"), ("opencreator", "krillinai/OpenCreator"), ("postiz", "gitroomhq/postiz-app"), ("postsider", "lumizone/postsider"), ] HDRS = {"User-Agent": "wb-research/1.0", "Accept": "application/vnd.github+json"} def get(url): req = urllib.request.Request(url, headers=HDRS) with urllib.request.urlopen(req, timeout=40) as r: return json.loads(r.read().decode("utf-8")) def save(name, obj): p = os.path.join(OUT, name) with open(p, "w", encoding="utf-8") as f: json.dump(obj, f, ensure_ascii=False, indent=1) return p meta_all = {} for slug, full in REPOS: rec = {"full_name": full, "fetched_at": time.strftime("%Y-%m-%dT%H:%M:%S")} try: m = get("https://api.github.com/repos/" + full) save(slug + ".repo.json", m) rec["meta"] = {k: m.get(k) for k in [ "full_name", "description", "language", "stargazers_count", "forks_count", "open_issues_count", "default_branch", "created_at", "pushed_at", "size", "homepage", "archived", "fork"]} rec["meta"]["license"] = (m.get("license") or {}).get("spdx_id") rec["meta"]["topics"] = m.get("topics") branch = m.get("default_branch") or "main" try: rdm = get("https://api.github.com/repos/%s/readme" % full) save(slug + ".readme.json", rdm) import base64 txt = base64.b64decode(rdm.get("content", "")).decode("utf-8", "replace") with open(os.path.join(OUT, slug + ".README.md"), "w", encoding="utf-8") as f: f.write(txt) rec["readme_path"] = rdm.get("path") rec["readme_bytes"] = len(txt.encode("utf-8")) except urllib.error.HTTPError as e: rec["readme_error"] = "HTTP %s" % e.code try: tr = get("https://api.github.com/repos/%s/git/trees/%s?recursive=1" % (full, branch)) save(slug + ".tree.json", tr) paths = [t["path"] for t in tr.get("tree", []) if t.get("type") == "blob"] rec["tree_total_blobs"] = len(paths) rec["tree_truncated"] = tr.get("truncated") tops = sorted({p.split("/")[0] for p in paths}) rec["tree_top_level"] = tops except urllib.error.HTTPError as e: rec["tree_error"] = "HTTP %s" % e.code except urllib.error.HTTPError as e: rec["error"] = "HTTP %s %s" % (e.code, e.reason) except Exception as e: rec["error"] = repr(e) meta_all[slug] = rec print("[%s] %s" % (slug, json.dumps(rec, ensure_ascii=False)[:600])) time.sleep(0.6) save("_summary.json", meta_all) print("\nDONE -> " + OUT)