39a2cbc65b
- get_face_groups_handler: COALESCE(tp.name, tn.label) for name consistency - sync_file_status: compare JSON vs pre_chunks (not chunk table) - face consistency: compare frames.len() not total_faces - cleanup 2 ghost records with NULL file_name/file_path - replace identity_agent with face_dedup in pipeline stages - remove identity_agent_api.rs and all references - update required_processors to match actual processors - update AGENTS.md with team responsibilities - add Studio pipeline changes documentation
219 lines
7.7 KiB
Python
219 lines
7.7 KiB
Python
#!/opt/homebrew/bin/python3.11
|
|
"""
|
|
MediaPipe Pose Processor - Using MediaPipe Pose Landmarker
|
|
|
|
Detects human pose with 33 keypoints, including face landmarks (nose, eyes).
|
|
Coordinates are normalized (0-1), converted to pixel coordinates.
|
|
|
|
Usage:
|
|
python3 scripts/mediapipe_pose_processor.py --video /path/to/video.mp4 --file-uuid <uuid> --output-dir /path/to/output
|
|
|
|
Output:
|
|
{uuid}.pose.mediapipe.json
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import sys
|
|
import time
|
|
from pathlib import Path
|
|
|
|
try:
|
|
import cv2
|
|
import mediapipe as mp
|
|
import numpy as np
|
|
from mediapipe.tasks.python.vision import PoseLandmarker, PoseLandmarkerOptions
|
|
from mediapipe.tasks.python.core.base_options import BaseOptions
|
|
except ImportError as e:
|
|
print(f"Missing dependency: {e}", file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
|
|
# MediaPipe Pose landmark names (33 keypoints)
|
|
LANDMARK_NAMES = [
|
|
"nose", "left_eye_inner", "left_eye", "left_eye_outer",
|
|
"right_eye_inner", "right_eye", "right_eye_outer",
|
|
"left_ear", "right_ear", "mouth_left", "mouth_right",
|
|
"left_shoulder", "right_shoulder", "left_elbow", "right_elbow",
|
|
"left_wrist", "right_wrist", "left_pinky", "right_pinky",
|
|
"left_index", "right_index", "left_thumb", "right_thumb",
|
|
"left_hip", "right_hip", "left_knee", "right_knee",
|
|
"left_ankle", "right_ankle", "left_heel", "right_heel",
|
|
"left_foot_index", "right_foot_index",
|
|
]
|
|
|
|
|
|
def process_video(
|
|
video_path: str,
|
|
output_path: str,
|
|
file_uuid: str,
|
|
sample_interval: int = 3, # Match Apple Vision default
|
|
) -> dict:
|
|
"""
|
|
Process video with MediaPipe Pose.
|
|
|
|
Args:
|
|
video_path: Path to video file
|
|
output_path: Output JSON path
|
|
file_uuid: File UUID
|
|
sample_interval: Process every N frames
|
|
|
|
Returns:
|
|
Dict with pose data
|
|
"""
|
|
# Download model if not exists
|
|
model_path = os.path.expanduser("~/.mediapipe/models/pose_landmarker_heavy.task")
|
|
if not os.path.exists(model_path):
|
|
os.makedirs(os.path.dirname(model_path), exist_ok=True)
|
|
print(f"[mediapipe_pose] Downloading model...")
|
|
import urllib.request
|
|
url = "https://storage.googleapis.com/mediapipe-models/pose_landmarker/pose_landmarker_heavy/float16/1/pose_landmarker_heavy.task"
|
|
urllib.request.urlretrieve(url, model_path)
|
|
print(f"[mediapipe_pose] Model downloaded to {model_path}")
|
|
|
|
# Initialize Pose Landmarker
|
|
options = PoseLandmarkerOptions(
|
|
base_options=BaseOptions(model_asset_path=model_path),
|
|
running_mode=mp.tasks.vision.RunningMode.VIDEO,
|
|
)
|
|
detector = PoseLandmarker.create_from_options(options)
|
|
|
|
# Open video
|
|
cap = cv2.VideoCapture(video_path)
|
|
if not cap.isOpened():
|
|
print(f"[mediapipe_pose] Cannot open video: {video_path}", file=sys.stderr)
|
|
return {"error": "Cannot open video"}
|
|
|
|
fps = cap.get(cv2.CAP_PROP_FPS)
|
|
total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
|
|
width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
|
|
height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
|
|
|
|
print(f"[mediapipe_pose] Video: {total_frames} frames, {fps:.2f} fps, {width}x{height}")
|
|
print(f"[mediapipe_pose] Processing every {sample_interval} frames...")
|
|
|
|
frames_data = []
|
|
frame_num = 0
|
|
processed_count = 0
|
|
|
|
start_time = time.time()
|
|
|
|
while True:
|
|
ret, frame = cap.read()
|
|
if not ret:
|
|
break
|
|
|
|
# Process every N frames
|
|
if frame_num % sample_interval == 0:
|
|
# Convert BGR to RGB
|
|
rgb_frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
|
|
|
|
# Create MediaPipe Image
|
|
mp_image = mp.Image(mp.ImageFormat.SRGB, rgb_frame)
|
|
|
|
# Detect pose
|
|
results = detector.detect_for_video(mp_image, int(frame_num * 1000 / fps))
|
|
|
|
if results.pose_landmarks:
|
|
for pose_landmarks in results.pose_landmarks:
|
|
keypoints = []
|
|
face_keypoints = []
|
|
|
|
for idx, landmark in enumerate(pose_landmarks):
|
|
name = LANDMARK_NAMES[idx] if idx < len(LANDMARK_NAMES) else f"landmark_{idx}"
|
|
kp = {
|
|
"name": name,
|
|
"x": landmark.x * width,
|
|
"y": landmark.y * height,
|
|
"z": landmark.z if hasattr(landmark, 'z') else 0,
|
|
"confidence": landmark.visibility if hasattr(landmark, 'visibility') else 1.0,
|
|
}
|
|
keypoints.append(kp)
|
|
|
|
if name in ["nose", "left_eye", "right_eye"]:
|
|
face_keypoints.append(kp)
|
|
|
|
# Calculate bbox from all keypoints
|
|
valid_kps = [kp for kp in keypoints if kp["confidence"] > 0.3]
|
|
if valid_kps:
|
|
x_coords = [kp["x"] for kp in valid_kps]
|
|
y_coords = [kp["y"] for kp in valid_kps]
|
|
bbox = {
|
|
"x": min(x_coords),
|
|
"y": min(y_coords),
|
|
"width": max(x_coords) - min(x_coords),
|
|
"height": max(y_coords) - min(y_coords),
|
|
}
|
|
else:
|
|
bbox = {"x": 0, "y": 0, "width": 0, "height": 0}
|
|
|
|
frames_data.append({
|
|
"frame": frame_num,
|
|
"timestamp": frame_num / fps,
|
|
"persons": [{
|
|
"keypoints": keypoints,
|
|
"bbox": bbox,
|
|
"face_keypoints": face_keypoints,
|
|
}],
|
|
})
|
|
|
|
processed_count += 1
|
|
|
|
if processed_count % 100 == 0:
|
|
elapsed = time.time() - start_time
|
|
print(f"[mediapipe_pose] Processed {processed_count} poses ({elapsed:.1f}s)")
|
|
|
|
frame_num += 1
|
|
|
|
cap.release()
|
|
detector.close()
|
|
|
|
elapsed = time.time() - start_time
|
|
|
|
# Build output
|
|
output = {
|
|
"file_uuid": file_uuid,
|
|
"processor": "mediapipe_pose",
|
|
"fps": fps,
|
|
"frame_count": total_frames,
|
|
"sample_interval": sample_interval,
|
|
"total_poses": len(frames_data),
|
|
"elapsed_seconds": round(elapsed, 2),
|
|
"frames": frames_data,
|
|
}
|
|
|
|
# Save
|
|
with open(output_path, "w") as f:
|
|
json.dump(output, f)
|
|
|
|
print(f"[mediapipe_pose] Saved: {output_path}")
|
|
print(f"[mediapipe_pose] Total poses: {len(frames_data)}")
|
|
print(f"[mediapipe_pose] Elapsed: {elapsed:.1f}s")
|
|
|
|
return output
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(description="MediaPipe Pose Processor")
|
|
parser.add_argument("--video", "-v", required=True, help="Video file path")
|
|
parser.add_argument("--file-uuid", "-u", required=True, help="File UUID")
|
|
parser.add_argument("--output-dir", "-o", default="/Users/accusys/momentry/output", help="Output directory")
|
|
parser.add_argument("--sample-interval", "-s", type=int, default=3, help="Process every N frames")
|
|
args = parser.parse_args()
|
|
|
|
output_path = Path(args.output_dir) / f"{args.file_uuid}.pose.mediapipe.json"
|
|
|
|
result = process_video(
|
|
args.video,
|
|
str(output_path),
|
|
args.file_uuid,
|
|
args.sample_interval,
|
|
)
|
|
|
|
if "error" not in result:
|
|
print(f"\n[mediapipe_pose] Done. Output: {output_path}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main() |