#!/opt/homebrew/bin/python3.11 """ Cut Key Frame Extraction - Extract representative frames from each scene for VLM analysis For each scene in cut.json, extracts the middle frame as a key frame. Output: {uuid}_scene_{n}.jpg files in output directory Usage: python cut_key_frame.py --file-uuid abc123 --video /path/to/video.mp4 --cut-json /path/to/cut.json python cut_key_frame.py --file-uuid abc123 --video /path/to/video.mp4 --cut-json /path/to/cut.json --output-dir /custom/output Output: {output_dir}/{uuid}_scene_1.jpg {output_dir}/{uuid}_scene_2.jpg ... """ import argparse import json import subprocess import sys from pathlib import Path def extract_frame(video_path: str, frame_number: int, output_path: str) -> bool: """ Extract a single frame from video using ffmpeg. Args: video_path: Path to video file frame_number: Frame number to extract (0-indexed) output_path: Output path for the frame Returns: True if successful, False otherwise """ cmd = [ "ffmpeg", "-y", "-v", "quiet", "-i", video_path, "-vf", f"select=eq(n\\,{frame_number})", "-vframes", "1", "-q:v", "2", output_path ] result = subprocess.run(cmd, capture_output=True) return result.returncode == 0 def extract_scene_key_frames( file_uuid: str, video_path: str, cut_json_path: str, output_dir: str, ) -> dict: """ Extract key frames from each scene in cut.json. Args: file_uuid: File UUID video_path: Path to video file cut_json_path: Path to cut.json output_dir: Output directory for key frames Returns: Dict with scenes processed and output paths """ # Read cut.json with open(cut_json_path, 'r') as f: cut_data = json.load(f) scenes = cut_data.get("scenes", []) if not scenes: print(f"No scenes found in {cut_json_path}", file=sys.stderr) return {"scenes": [], "output_dir": output_dir} fps = cut_data.get("fps", 24.0) output_path = Path(output_dir) output_path.mkdir(parents=True, exist_ok=True) results = [] for scene in scenes: scene_number = scene.get("scene_number", 0) start_frame = scene.get("start_frame", 0) end_frame = scene.get("end_frame", 0) # Extract middle frame middle_frame = (start_frame + end_frame) // 2 # Output path output_file = output_path / f"{file_uuid}_scene_{scene_number}.jpg" # Extract frame success = extract_frame(video_path, middle_frame, str(output_file)) results.append({ "scene_number": scene_number, "middle_frame": middle_frame, "start_frame": start_frame, "end_frame": end_frame, "output_path": str(output_file), "success": success, }) if success: print(f"[CUT_KEY_FRAME] Scene {scene_number}: frame {middle_frame} -> {output_file}") else: print(f"[CUT_KEY_FRAME] Scene {scene_number}: FAILED to extract frame {middle_frame}", file=sys.stderr) return { "file_uuid": file_uuid, "total_scenes": len(scenes), "scenes": results, "output_dir": str(output_dir), } def main(): parser = argparse.ArgumentParser(description="Extract key frames from scenes for VLM analysis") parser.add_argument("--file-uuid", "-u", required=True, help="File UUID") parser.add_argument("--video", "-v", required=True, help="Video file path") parser.add_argument("--cut-json", "-c", required=True, help="cut.json path") parser.add_argument("--output-dir", "-o", default=None, help="Output directory (default: same as cut.json)") parser.add_argument("--json", "-j", action="store_true", help="Output as JSON") args = parser.parse_args() # Default output dir to same as cut.json output_dir = args.output_dir or str(Path(args.cut_json).parent) result = extract_scene_key_frames( args.file_uuid, args.video, args.cut_json, output_dir, ) if args.json: print(json.dumps(result, indent=2)) else: print(f"Extracted {result['total_scenes']} scene key frames to {result['output_dir']}") if __name__ == "__main__": main()