# -*- coding: utf-8 -*- """按《视频抽帧与字幕规范》严格重抽关键帧。 与 v1 的差异: - 字幕区 scene 检测严格使用底部 15%(B 站 1920x1080 -> 162px;按比例适配其他分辨率)。 - 新增字幕区低阈值 0.05 作为安全网,防止缓变换句被漏检。 - 输出 frame_times.json,记录每帧来自哪种检测(便于核对)。 - 保留首帧 0s、尾帧 DUR-0.3s、>5s 长间隔补中点的规范步骤。 用法: python extract_frames_v2.py --dir <视频目录> --out <帧输出目录> --ff """ import argparse, glob, json, os, re, subprocess, sys def run(cmd): return subprocess.run(cmd, capture_output=True, text=True, encoding="utf-8", errors="replace") def probe(video, ffprobe): r = run([ffprobe, "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=width,height", "-show_entries", "format=duration", "-of", "default=nk=1:nw=1", video]) vals = [x.strip() for x in r.stdout.split("\n") if x.strip()] return int(vals[0]), int(vals[1]), float(vals[2]) def scene_times(video, vf, ffmpeg): """跑一次 scene 检测并返回所有 pts_time(秒)。""" r = run([ffmpeg, "-i", video, "-vf", vf + ",showinfo", "-vsync", "vfr", "-f", "null", "-"]) out = set() for line in r.stderr.splitlines(): m = re.search(r"pts_time:([0-9.]+)", line) if m: out.add(round(float(m.group(1)), 4)) return sorted(out) def main(): ap = argparse.ArgumentParser() ap.add_argument("--dir", required=True, help="含 .mp4 的目录") ap.add_argument("--out", required=True, help="帧输出目录") ap.add_argument("--ff", default=r"D:\ComfyUI_windows\ffmpeg-8.0.1-full_build\bin", help="ffmpeg/ffprobe 所在 bin 目录") args = ap.parse_args() ffmpeg = os.path.join(args.ff, "ffmpeg.exe") ffprobe = os.path.join(args.ff, "ffprobe.exe") vids = sorted(glob.glob(os.path.join(args.dir, "*.mp4"))) if not vids: print("✗ 目录里没有 .mp4:%s" % args.dir) sys.exit(1) video = vids[0] print("视频:%s" % video) w, h, dur = probe(video, ffprobe) print("分辨率=%dx%d 时长=%.3fs" % (w, h, dur)) # 字幕区用于 scene 检测:底部 15%(规范参考值) sub_h = max(1, int(round(h * 0.15))) sub_y = h - sub_h crop_vf = "crop=%d:%d:0:%d" % (w, sub_h, sub_y) print("字幕区裁剪(检测用):%s" % crop_vf) # ① 画面变化 t_scene_03 = scene_times(video, "select='gt(scene,0.3)'", ffmpeg) # ② 字幕变化(规范核心) t_sub_01 = scene_times(video, "%s,select='gt(scene,0.1)'" % crop_vf, ffmpeg) # ③ 低阈值补漏(规范标准步骤) t_scene_005 = scene_times(video, "select='gt(scene,0.05)'", ffmpeg) # 额外安全网:字幕区低阈值,捕获缓变换句 t_sub_005 = scene_times(video, "%s,select='gt(scene,0.05)'" % crop_vf, ffmpeg) print("scene 0.3=%d 字幕区 0.1=%d 全局 0.05=%d 字幕区 0.05=%d" % (len(t_scene_03), len(t_sub_01), len(t_scene_005), len(t_sub_005))) # 合并 + 首尾 all_times = sorted(set(t_scene_03) | set(t_sub_01) | set(t_scene_005) | set(t_sub_005)) all_times.append(0.0) all_times.append(max(0.0, round(dur - 0.3, 3))) all_times = sorted(set(all_times)) # >5s 长间隔补中点 filled = [] prev = None for t in all_times: if prev is not None and (t - prev) > 5: filled.append(round((t + prev) / 2, 3)) filled.append(t) prev = t times = sorted(set(filled)) print("最终帧数(含首尾+补帧):%d" % len(times)) # 记录来源(用于调试/核对) source_map = {} for t in times: tags = [] if t == 0.0: tags.append("start") if abs(t - (dur - 0.3)) < 0.001: tags.append("end") if t in t_scene_03: tags.append("scene03") if t in t_sub_01: tags.append("sub01") if t in t_scene_005: tags.append("scene005") if t in t_sub_005: tags.append("sub005") if "start" not in tags and "end" not in tags and all(t not in s for s in (t_scene_03, t_sub_01, t_scene_005, t_sub_005)): tags.append("fill") source_map["%.3f" % t] = tags os.makedirs(args.out, exist_ok=True) for i, t in enumerate(times, 1): tt = t if t <= dur else max(0.0, dur - 0.3) outp = os.path.join(args.out, "c_%04d_%.3fs.jpg" % (i, t)) r = run([ffmpeg, "-y", "-ss", "%.3f" % tt, "-i", video, "-frames:v", "1", "-q:v", "2", outp]) if r.returncode != 0: print(" ✗ 帧 %d 失败:%s" % (i, r.stderr[-200:])) meta = { "video": video, "width": w, "height": h, "duration": dur, "subtitle_crop": crop_vf, "counts": { "scene_0.3": len(t_scene_03), "sub_0.1": len(t_sub_01), "scene_0.05": len(t_scene_005), "sub_0.05": len(t_sub_005), "final": len(times) }, "frame_times": source_map } with open(os.path.join(args.out, "frame_times.json"), "w", encoding="utf-8", newline="\n") as f: json.dump(meta, f, ensure_ascii=False, indent=2) print("✓ 抽帧完成 → %s" % args.out) if __name__ == "__main__": main()