Files
contentm_agent/extract_frames.py
T

92 lines
3.6 KiB
Python
Raw Normal View History

# -*- coding: utf-8 -*-
"""按 aigc-idea-impression 抽帧 SOP 提取视频关键帧(双因素 + 首尾 + 5s 长间隔补帧)。
规避原 SOP bash 版的 bc 依赖与 Windows CRLF 坑:全部用 Python 调度 ffmpeg/ffprobe。
用法:
python extract_frames.py --dir <视频目录(自动找首个.mp4)> --out <帧输出目录> --ff <ffmpeg bin 目录>
"""
import argparse, glob, os, re, subprocess, sys
def run(cmd):
return subprocess.run(cmd, capture_output=True, text=True, encoding="utf-8", errors="replace")
def probe(video, ffprobe):
r = run([ffprobe, "-v", "error", "-select_streams", "v:0",
"-show_entries", "stream=width,height", "-show_entries", "format=duration",
"-of", "default=nk=1:nw=1", video])
vals = [x.strip() for x in r.stdout.split("\n") if x.strip()]
w = int(vals[0]); h = int(vals[1]); dur = float(vals[2])
return w, h, dur
def scene_times(video, vf, ffmpeg):
"""对给定 -vf 跑一次 scene 检测,解析所有 pts_time。"""
r = run([ffmpeg, "-i", video, "-vf", vf + ",showinfo", "-vsync", "vfr", "-f", "null", "-"])
out = set()
for line in r.stderr.splitlines():
m = re.search(r"pts_time:([0-9.]+)", line)
if m:
out.add(round(float(m.group(1)), 4))
return sorted(out)
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--dir", required=True, help="含 .mp4 的目录")
ap.add_argument("--out", required=True, help="帧输出目录")
ap.add_argument("--ff", default=r"D:\ComfyUI_windows\ffmpeg-8.0.1-full_build\bin",
help="ffmpeg/ffprobe 所在 bin 目录")
args = ap.parse_args()
ffmpeg = os.path.join(args.ff, "ffmpeg.exe")
ffprobe = os.path.join(args.ff, "ffprobe.exe")
vids = sorted(glob.glob(os.path.join(args.dir, "*.mp4")))
if not vids:
print("✗ 目录里没有 .mp4:%s" % args.dir); sys.exit(1)
video = vids[0]
print("视频:%s" % video)
w, h, dur = probe(video, ffprobe)
print("分辨率=%dx%d 时长=%.2fs" % (w, h, dur))
# 字幕区 = 底部 15%(横屏 B 站常用;竖屏改 1/4 即可)
sh = max(1, int(h * 0.15)); sy = h - sh
crop_vf = "crop=%d:%d:0:%d" % (w, sh, sy)
t_a = scene_times(video, "select='gt(scene,0.3)'", ffmpeg) # ① 画面变化
t_b = scene_times(video, "%s,select='gt(scene,0.1)'" % crop_vf, ffmpeg) # ② 字幕变化
t_c = scene_times(video, "select='gt(scene,0.05)'", ffmpeg) # ③ 低阈值补漏
print("scene 0.3=%d 字幕区 0.1=%d 0.05=%d" % (len(t_a), len(t_b), len(t_c)))
times = sorted(set(t_a) | set(t_b) | set(t_c))
times.append(0.0) # 首帧
times.append(max(0.0, round(dur - 0.3, 3))) # 尾帧
times = sorted(set(times))
# ④ 长间隔强制补帧(>5s 补中点)
filled = []
for i, t in enumerate(times):
filled.append(t)
if i > 0:
gap = t - times[i - 1]
if gap > 5:
filled.append(round((t + times[i - 1]) / 2, 3))
times = sorted(set(filled))
print("最终帧数(含首尾+补帧):%d" % len(times))
os.makedirs(args.out, exist_ok=True)
for i, t in enumerate(times, 1):
tt = t
if t > dur:
tt = max(0.0, dur - 0.3)
outp = os.path.join(args.out, "c_%04d_%.3fs.jpg" % (i, t))
r = run([ffmpeg, "-y", "-ss", "%.3f" % tt, "-i", video,
"-frames:v", "1", "-q:v", "2", outp])
if r.returncode != 0:
print(" ✗ 帧 %d 失败:%s" % (i, r.stderr[-200:]))
print("✓ 抽帧完成 → %s" % args.out)
if __name__ == "__main__":
main()