257 lines
8.6 KiB
Python
257 lines
8.6 KiB
Python
#!/usr/bin/env python3
|
||
"""
|
||
批量获取抖音榜单数据(近30天)
|
||
- 每日作品热榜 (likesRank) — 按点赞总量排名
|
||
- 每日点赞榜 (hotContentRank source=每日点赞飙升榜) — 按单日新增点赞排名
|
||
- 7日点赞榜 (hotContentRank source=七日点赞飙升榜) — 按七日新增点赞排名
|
||
|
||
输出:RedFox数据/榜单数据/YYYY-MM/ 目录,JSON格式保留完整原始响应
|
||
|
||
用法:
|
||
python batch_fetch_rankings.py
|
||
"""
|
||
|
||
import os, json, time, sys
|
||
from datetime import datetime, timedelta
|
||
from urllib.request import Request, urlopen
|
||
from urllib.error import HTTPError, URLError
|
||
|
||
# ── 配置 ──
|
||
API_KEY = os.environ.get("REDFOX_API_KEY", "ak_662511b4e9a74de59dc0e77d9edbca43")
|
||
LIKES_RANK_URL = "https://redfox.hk/story/api/dy/search/likesRank"
|
||
HOT_CONTENT_URL = "https://redfox.hk/story/api/dy/search/hotContentRank"
|
||
OUTPUT_BASE = "D:/github/技能百宝箱/ThirdPartySkills/RedFox数据/榜单数据"
|
||
|
||
# 近30天
|
||
END_DATE = datetime(2026, 8, 9)
|
||
DAYS = 30
|
||
REQUEST_DELAY = 0.5 # 每次请求间隔秒数,避免限流
|
||
|
||
|
||
def post_json(url, body):
|
||
"""发送 POST 请求,返回解析后的 JSON"""
|
||
data = json.dumps(body).encode("utf-8")
|
||
req = Request(url, data=data, method="POST")
|
||
req.add_header("X-API-KEY", API_KEY)
|
||
req.add_header("Content-Type", "application/json")
|
||
req.add_header("Accept", "application/json")
|
||
with urlopen(req, timeout=30) as resp:
|
||
return json.loads(resp.read().decode("utf-8"))
|
||
|
||
|
||
def save_json(data, directory, filename):
|
||
"""保存 JSON 文件,返回 (路径, KB数, 记录数)"""
|
||
os.makedirs(directory, exist_ok=True)
|
||
path = os.path.join(directory, filename)
|
||
with open(path, 'w', encoding='utf-8') as f:
|
||
json.dump(data, f, ensure_ascii=False, indent=2)
|
||
size_kb = os.path.getsize(path) / 1024
|
||
return path, size_kb
|
||
|
||
|
||
def fetch_daily_likes_rank(date_str):
|
||
"""每日作品热榜:按点赞总量排名"""
|
||
body = {
|
||
"source": "抖音每日热门作品榜-GitHub",
|
||
"type": "全部",
|
||
"startTime": date_str,
|
||
"endTime": date_str
|
||
}
|
||
return post_json(LIKES_RANK_URL, body)
|
||
|
||
|
||
def fetch_daily_surge(date_str):
|
||
"""每日点赞榜:按单日新增点赞排名"""
|
||
body = {
|
||
"source": "抖音每日点赞飙升榜",
|
||
"type": "全部",
|
||
"startTime": date_str
|
||
}
|
||
return post_json(HOT_CONTENT_URL, body)
|
||
|
||
|
||
def fetch_weekly_surge(date_str):
|
||
"""7日点赞榜:按七日新增点赞排名"""
|
||
body = {
|
||
"source": "抖音七日点赞飙升榜-GitHub",
|
||
"type": "全部",
|
||
"startTime": date_str
|
||
}
|
||
return post_json(HOT_CONTENT_URL, body)
|
||
|
||
|
||
def main():
|
||
print(f"🚀 开始批量获取近{DAYS}天榜单数据")
|
||
print(f" 时间范围:{(END_DATE - timedelta(days=DAYS-1)).strftime('%Y-%m-%d')} ~ {END_DATE.strftime('%Y-%m-%d')}")
|
||
print(f" 输出目录:{OUTPUT_BASE}")
|
||
print()
|
||
|
||
stats = {"success": 0, "empty": 0, "error": 0, "total_records": 0, "total_kb": 0.0}
|
||
|
||
# ── 第一轮:每日作品热榜(30天)──
|
||
print("=" * 60)
|
||
print("📊 一、每日作品热榜(likesRank)")
|
||
print("=" * 60)
|
||
for i in range(DAYS):
|
||
d = END_DATE - timedelta(days=i)
|
||
date_str = d.strftime('%Y-%m-%d')
|
||
month = d.strftime('%Y-%m')
|
||
out_dir = os.path.join(OUTPUT_BASE, month)
|
||
fname = f"每日作品热榜_{date_str}.json"
|
||
|
||
try:
|
||
resp = fetch_daily_likes_rank(date_str)
|
||
code = resp.get("code", -1)
|
||
items = resp.get("data", [])
|
||
if isinstance(items, list):
|
||
count = len(items)
|
||
elif isinstance(items, dict):
|
||
count = len(items.get("dailyRank", items.get("weeklyRank", [])))
|
||
else:
|
||
count = 0
|
||
|
||
if code != 2000:
|
||
print(f" ⚠️ {date_str} API错误码={code} msg={resp.get('msg', '')}")
|
||
stats["error"] += 1
|
||
time.sleep(REQUEST_DELAY)
|
||
continue
|
||
|
||
if count == 0:
|
||
print(f" 📭 {date_str} 暂无数据")
|
||
stats["empty"] += 1
|
||
time.sleep(REQUEST_DELAY)
|
||
continue
|
||
|
||
path, kb = save_json(resp, out_dir, fname)
|
||
stats["success"] += 1
|
||
stats["total_records"] += count
|
||
stats["total_kb"] += kb
|
||
print(f" ✅ {date_str} {count}条 {kb:.1f}KB → {fname}")
|
||
|
||
except HTTPError as e:
|
||
print(f" ❌ {date_str} HTTP {e.code}")
|
||
stats["error"] += 1
|
||
except URLError as e:
|
||
print(f" ❌ {date_str} 网络错误: {e.reason}")
|
||
stats["error"] += 1
|
||
except Exception as e:
|
||
print(f" ❌ {date_str} {e}")
|
||
stats["error"] += 1
|
||
|
||
time.sleep(REQUEST_DELAY)
|
||
|
||
print()
|
||
|
||
# ── 第二轮:每日点赞榜(30天)──
|
||
print("=" * 60)
|
||
print("📊 二、每日点赞榜(hotContentRank - 日飙升)")
|
||
print("=" * 60)
|
||
for i in range(DAYS):
|
||
d = END_DATE - timedelta(days=i)
|
||
date_str = d.strftime('%Y-%m-%d')
|
||
month = d.strftime('%Y-%m')
|
||
out_dir = os.path.join(OUTPUT_BASE, month)
|
||
fname = f"每日点赞榜_{date_str}.json"
|
||
|
||
try:
|
||
resp = fetch_daily_surge(date_str)
|
||
code = resp.get("code", -1)
|
||
data_obj = resp.get("data", {})
|
||
items = data_obj.get("dailyRank", []) if isinstance(data_obj, dict) else []
|
||
count = len(items)
|
||
|
||
if code != 2000:
|
||
print(f" ⚠️ {date_str} API错误码={code} msg={resp.get('msg', '')}")
|
||
stats["error"] += 1
|
||
time.sleep(REQUEST_DELAY)
|
||
continue
|
||
|
||
if count == 0:
|
||
print(f" 📭 {date_str} 暂无数据")
|
||
stats["empty"] += 1
|
||
time.sleep(REQUEST_DELAY)
|
||
continue
|
||
|
||
path, kb = save_json(resp, out_dir, fname)
|
||
stats["success"] += 1
|
||
stats["total_records"] += count
|
||
stats["total_kb"] += kb
|
||
print(f" ✅ {date_str} {count}条 {kb:.1f}KB → {fname}")
|
||
|
||
except HTTPError as e:
|
||
print(f" ❌ {date_str} HTTP {e.code}")
|
||
stats["error"] += 1
|
||
except URLError as e:
|
||
print(f" ❌ {date_str} 网络错误: {e.reason}")
|
||
stats["error"] += 1
|
||
except Exception as e:
|
||
print(f" ❌ {date_str} {e}")
|
||
stats["error"] += 1
|
||
|
||
time.sleep(REQUEST_DELAY)
|
||
|
||
print()
|
||
|
||
# ── 第三轮:7日点赞榜(30天)──
|
||
print("=" * 60)
|
||
print("📊 三、7日点赞榜(hotContentRank - 周飙升)")
|
||
print("=" * 60)
|
||
for i in range(DAYS):
|
||
d = END_DATE - timedelta(days=i)
|
||
date_str = d.strftime('%Y-%m-%d')
|
||
month = d.strftime('%Y-%m')
|
||
out_dir = os.path.join(OUTPUT_BASE, month)
|
||
fname = f"7日点赞榜_{date_str}.json"
|
||
|
||
try:
|
||
resp = fetch_weekly_surge(date_str)
|
||
code = resp.get("code", -1)
|
||
data_obj = resp.get("data", {})
|
||
items = data_obj.get("weeklyRank", []) if isinstance(data_obj, dict) else []
|
||
count = len(items)
|
||
|
||
if code != 2000:
|
||
print(f" ⚠️ {date_str} API错误码={code} msg={resp.get('msg', '')}")
|
||
stats["error"] += 1
|
||
time.sleep(REQUEST_DELAY)
|
||
continue
|
||
|
||
if count == 0:
|
||
print(f" 📭 {date_str} 暂无数据")
|
||
stats["empty"] += 1
|
||
time.sleep(REQUEST_DELAY)
|
||
continue
|
||
|
||
path, kb = save_json(resp, out_dir, fname)
|
||
stats["success"] += 1
|
||
stats["total_records"] += count
|
||
stats["total_kb"] += kb
|
||
print(f" ✅ {date_str} {count}条 {kb:.1f}KB → {fname}")
|
||
|
||
except HTTPError as e:
|
||
print(f" ❌ {date_str} HTTP {e.code}")
|
||
stats["error"] += 1
|
||
except URLError as e:
|
||
print(f" ❌ {date_str} 网络错误: {e.reason}")
|
||
stats["error"] += 1
|
||
except Exception as e:
|
||
print(f" ❌ {date_str} {e}")
|
||
stats["error"] += 1
|
||
|
||
time.sleep(REQUEST_DELAY)
|
||
|
||
# ── 汇总 ──
|
||
print()
|
||
print("=" * 60)
|
||
print("📋 汇总统计")
|
||
print("=" * 60)
|
||
print(f" ✅ 成功: {stats['success']} 次")
|
||
print(f" 📭 空数据: {stats['empty']} 次")
|
||
print(f" ❌ 错误: {stats['error']} 次")
|
||
print(f" 📊 总记录数: {stats['total_records']} 条")
|
||
print(f" 💾 总数据量: {stats['total_kb']:.1f} KB ({stats['total_kb']/1024:.2f} MB)")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|