#!/opt/homebrew/bin/python3.11 """ MediaPipe Pose with Face Frame Alignment 1. Load frames with faces from Apple Vision face_traced.json 2. Run MediaPipe pose only on those frames 3. Filter poses aligned with face bboxes Usage: python3 scripts/mediapipe_pose_aligned.py --video /path/to/video.mp4 --file-uuid --output-dir /path/to/output Output: {uuid}.pose.mediapipe.aligned.json """ import argparse import json import os import sys import time from pathlib import Path try: import cv2 import mediapipe as mp import numpy as np from mediapipe.tasks.python.vision import PoseLandmarker, PoseLandmarkerOptions from mediapipe.tasks.python.core.base_options import BaseOptions except ImportError as e: print(f"Missing dependency: {e}", file=sys.stderr) sys.exit(1) LANDMARK_NAMES = [ "nose", "left_eye_inner", "left_eye", "left_eye_outer", "right_eye_inner", "right_eye", "right_eye_outer", "left_ear", "right_ear", "mouth_left", "mouth_right", "left_shoulder", "right_shoulder", "left_elbow", "right_elbow", "left_wrist", "right_wrist", "left_pinky", "right_pinky", "left_index", "right_index", "left_thumb", "right_thumb", "left_hip", "right_hip", "left_knee", "right_knee", "left_ankle", "right_ankle", "left_heel", "right_heel", "left_foot_index", "right_foot_index", ] def point_in_bbox(x, y, bbox): """Check if point is inside bbox.""" return bbox['x'] <= x <= bbox['x'] + bbox['width'] and bbox['y'] <= y <= bbox['y'] + bbox['height'] def face_keypoints_aligned(keypoints, face_bboxes): """Check if nose, left_eye, right_eye all in same face bbox.""" kp_dict = {kp['name']: kp for kp in keypoints} required = ['nose', 'left_eye', 'right_eye'] if not all(k in kp_dict for k in required): return False, None for bbox in face_bboxes: all_in = all(point_in_bbox(kp_dict[k]['x'], kp_dict[k]['y'], bbox) for k in required) if all_in: return True, bbox return False, None def process_video( video_path: str, face_json_path: str, output_path: str, file_uuid: str, ) -> dict: """ Process video with MediaPipe pose on frames with faces. """ # Load face data print(f"[pose_aligned] Loading face data: {face_json_path}") with open(face_json_path) as f: face_data = json.load(f) face_frames = face_data.get('frames', {}) print(f"[pose_aligned] Frames with faces: {len(face_frames)}") # Download model model_path = os.path.expanduser("~/.mediapipe/models/pose_landmarker_heavy.task") if not os.path.exists(model_path): os.makedirs(os.path.dirname(model_path), exist_ok=True) print(f"[pose_aligned] Downloading model...") import urllib.request url = "https://storage.googleapis.com/mediapipe-models/pose_landmarker/pose_landmarker_heavy/float16/1/pose_landmarker_heavy.task" urllib.request.urlretrieve(url, model_path) # Initialize pose detector options = PoseLandmarkerOptions( base_options=BaseOptions(model_asset_path=model_path), running_mode=mp.tasks.vision.RunningMode.VIDEO, ) detector = PoseLandmarker.create_from_options(options) # Open video cap = cv2.VideoCapture(video_path) if not cap.isOpened(): print(f"[pose_aligned] Cannot open video", file=sys.stderr) return {"error": "Cannot open video"} fps = cap.get(cv2.CAP_PROP_FPS) total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) print(f"[pose_aligned] Video: {total_frames} frames, {fps:.2f} fps, {width}x{height}") frames_data = [] total_poses = 0 aligned_poses = 0 start_time = time.time() # Process only frames with faces face_frame_nums = sorted(int(k) for k in face_frames.keys()) for i, frame_num in enumerate(face_frame_nums): # Seek to frame cap.set(cv2.CAP_PROP_POS_FRAMES, frame_num) ret, frame = cap.read() if not ret: continue # Get face bboxes for this frame face_frame = face_frames[str(frame_num)] face_bboxes = [] for face in face_frame.get('faces', []): face_bboxes.append({ 'x': face['x'], 'y': face['y'], 'width': face['width'], 'height': face['height'] }) # Detect pose rgb_frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) mp_image = mp.Image(mp.ImageFormat.SRGB, rgb_frame) results = detector.detect_for_video(mp_image, int(frame_num * 1000 / fps)) if results.pose_landmarks: persons = [] for pose_landmarks in results.pose_landmarks: total_poses += 1 # Convert landmarks keypoints = [] for idx, landmark in enumerate(pose_landmarks): name = LANDMARK_NAMES[idx] if idx < len(LANDMARK_NAMES) else f"landmark_{idx}" kp = { "name": name, "x": landmark.x * width, "y": landmark.y * height, "z": landmark.z if hasattr(landmark, 'z') else 0, "confidence": landmark.visibility if hasattr(landmark, 'visibility') else 1.0, } keypoints.append(kp) # Check alignment aligned, matched_bbox = face_keypoints_aligned(keypoints, face_bboxes) if aligned: aligned_poses += 1 face_keypoints = [kp for kp in keypoints if kp['name'] in ['nose', 'left_eye', 'right_eye']] persons.append({ "keypoints": keypoints, "face_keypoints": face_keypoints, "matched_face_bbox": matched_bbox, }) if persons: frames_data.append({ "frame": frame_num, "timestamp": frame_num / fps, "persons": persons, }) if (i + 1) % 500 == 0: elapsed = time.time() - start_time print(f"[pose_aligned] Processed {i+1}/{len(face_frame_nums)} frames, {aligned_poses} aligned poses ({elapsed:.1f}s)") cap.release() detector.close() elapsed = time.time() - start_time # Build output output = { "file_uuid": file_uuid, "processor": "mediapipe_pose_aligned", "fps": fps, "total_frames": total_frames, "face_frames_processed": len(face_frame_nums), "total_poses_detected": total_poses, "aligned_poses": aligned_poses, "alignment_rate": f"{aligned_poses / total_poses * 100:.1f}%" if total_poses > 0 else "0%", "frames_with_aligned_pose": len(frames_data), "elapsed_seconds": round(elapsed, 2), "frames": frames_data, } # Save with open(output_path, "w") as f: json.dump(output, f) print(f"\n[pose_aligned] Saved: {output_path}") print(f"[pose_aligned] Total poses detected: {total_poses}") print(f"[pose_aligned] Aligned poses: {aligned_poses} ({output['alignment_rate']})") print(f"[pose_aligned] Frames with aligned pose: {len(frames_data)}") print(f"[pose_aligned] Elapsed: {elapsed:.1f}s") return output def main(): parser = argparse.ArgumentParser(description="MediaPipe pose aligned with face frames") parser.add_argument("--video", "-v", required=True, help="Video file path") parser.add_argument("--file-uuid", "-u", required=True, help="File UUID") parser.add_argument("--output-dir", "-o", default="/Users/accusys/momentry/output", help="Output directory") args = parser.parse_args() output_path = Path(args.output_dir) face_json = output_path / f"{args.file_uuid}.face_traced.json" if not face_json.exists(): print(f"[pose_aligned] Face file not found: {face_json}", file=sys.stderr) sys.exit(1) output_file = output_path / f"{args.file_uuid}.pose.mediapipe.aligned.json" result = process_video( args.video, str(face_json), str(output_file), args.file_uuid, ) if __name__ == "__main__": main()