fix: face group name read consistency, sync_file_status fix, cleanup ghost records, identity_agent replaced with face_dedup
- get_face_groups_handler: COALESCE(tp.name, tn.label) for name consistency - sync_file_status: compare JSON vs pre_chunks (not chunk table) - face consistency: compare frames.len() not total_faces - cleanup 2 ghost records with NULL file_name/file_path - replace identity_agent with face_dedup in pipeline stages - remove identity_agent_api.rs and all references - update required_processors to match actual processors - update AGENTS.md with team responsibilities - add Studio pipeline changes documentation
This commit is contained in:
@@ -0,0 +1,246 @@
|
||||
#!/opt/homebrew/bin/python3.11
|
||||
"""
|
||||
MediaPipe Pose with Face Frame Alignment
|
||||
|
||||
1. Load frames with faces from Apple Vision face_traced.json
|
||||
2. Run MediaPipe pose only on those frames
|
||||
3. Filter poses aligned with face bboxes
|
||||
|
||||
Usage:
|
||||
python3 scripts/mediapipe_pose_aligned.py --video /path/to/video.mp4 --file-uuid <uuid> --output-dir /path/to/output
|
||||
|
||||
Output:
|
||||
{uuid}.pose.mediapipe.aligned.json
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
import cv2
|
||||
import mediapipe as mp
|
||||
import numpy as np
|
||||
from mediapipe.tasks.python.vision import PoseLandmarker, PoseLandmarkerOptions
|
||||
from mediapipe.tasks.python.core.base_options import BaseOptions
|
||||
except ImportError as e:
|
||||
print(f"Missing dependency: {e}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
LANDMARK_NAMES = [
|
||||
"nose", "left_eye_inner", "left_eye", "left_eye_outer",
|
||||
"right_eye_inner", "right_eye", "right_eye_outer",
|
||||
"left_ear", "right_ear", "mouth_left", "mouth_right",
|
||||
"left_shoulder", "right_shoulder", "left_elbow", "right_elbow",
|
||||
"left_wrist", "right_wrist", "left_pinky", "right_pinky",
|
||||
"left_index", "right_index", "left_thumb", "right_thumb",
|
||||
"left_hip", "right_hip", "left_knee", "right_knee",
|
||||
"left_ankle", "right_ankle", "left_heel", "right_heel",
|
||||
"left_foot_index", "right_foot_index",
|
||||
]
|
||||
|
||||
|
||||
def point_in_bbox(x, y, bbox):
|
||||
"""Check if point is inside bbox."""
|
||||
return bbox['x'] <= x <= bbox['x'] + bbox['width'] and bbox['y'] <= y <= bbox['y'] + bbox['height']
|
||||
|
||||
|
||||
def face_keypoints_aligned(keypoints, face_bboxes):
|
||||
"""Check if nose, left_eye, right_eye all in same face bbox."""
|
||||
kp_dict = {kp['name']: kp for kp in keypoints}
|
||||
|
||||
required = ['nose', 'left_eye', 'right_eye']
|
||||
if not all(k in kp_dict for k in required):
|
||||
return False, None
|
||||
|
||||
for bbox in face_bboxes:
|
||||
all_in = all(point_in_bbox(kp_dict[k]['x'], kp_dict[k]['y'], bbox) for k in required)
|
||||
if all_in:
|
||||
return True, bbox
|
||||
|
||||
return False, None
|
||||
|
||||
|
||||
def process_video(
|
||||
video_path: str,
|
||||
face_json_path: str,
|
||||
output_path: str,
|
||||
file_uuid: str,
|
||||
) -> dict:
|
||||
"""
|
||||
Process video with MediaPipe pose on frames with faces.
|
||||
"""
|
||||
# Load face data
|
||||
print(f"[pose_aligned] Loading face data: {face_json_path}")
|
||||
with open(face_json_path) as f:
|
||||
face_data = json.load(f)
|
||||
|
||||
face_frames = face_data.get('frames', {})
|
||||
print(f"[pose_aligned] Frames with faces: {len(face_frames)}")
|
||||
|
||||
# Download model
|
||||
model_path = os.path.expanduser("~/.mediapipe/models/pose_landmarker_heavy.task")
|
||||
if not os.path.exists(model_path):
|
||||
os.makedirs(os.path.dirname(model_path), exist_ok=True)
|
||||
print(f"[pose_aligned] Downloading model...")
|
||||
import urllib.request
|
||||
url = "https://storage.googleapis.com/mediapipe-models/pose_landmarker/pose_landmarker_heavy/float16/1/pose_landmarker_heavy.task"
|
||||
urllib.request.urlretrieve(url, model_path)
|
||||
|
||||
# Initialize pose detector
|
||||
options = PoseLandmarkerOptions(
|
||||
base_options=BaseOptions(model_asset_path=model_path),
|
||||
running_mode=mp.tasks.vision.RunningMode.VIDEO,
|
||||
)
|
||||
detector = PoseLandmarker.create_from_options(options)
|
||||
|
||||
# Open video
|
||||
cap = cv2.VideoCapture(video_path)
|
||||
if not cap.isOpened():
|
||||
print(f"[pose_aligned] Cannot open video", file=sys.stderr)
|
||||
return {"error": "Cannot open video"}
|
||||
|
||||
fps = cap.get(cv2.CAP_PROP_FPS)
|
||||
total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
|
||||
width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
|
||||
height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
|
||||
|
||||
print(f"[pose_aligned] Video: {total_frames} frames, {fps:.2f} fps, {width}x{height}")
|
||||
|
||||
frames_data = []
|
||||
total_poses = 0
|
||||
aligned_poses = 0
|
||||
|
||||
start_time = time.time()
|
||||
|
||||
# Process only frames with faces
|
||||
face_frame_nums = sorted(int(k) for k in face_frames.keys())
|
||||
|
||||
for i, frame_num in enumerate(face_frame_nums):
|
||||
# Seek to frame
|
||||
cap.set(cv2.CAP_PROP_POS_FRAMES, frame_num)
|
||||
ret, frame = cap.read()
|
||||
|
||||
if not ret:
|
||||
continue
|
||||
|
||||
# Get face bboxes for this frame
|
||||
face_frame = face_frames[str(frame_num)]
|
||||
face_bboxes = []
|
||||
for face in face_frame.get('faces', []):
|
||||
face_bboxes.append({
|
||||
'x': face['x'], 'y': face['y'],
|
||||
'width': face['width'], 'height': face['height']
|
||||
})
|
||||
|
||||
# Detect pose
|
||||
rgb_frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
|
||||
mp_image = mp.Image(mp.ImageFormat.SRGB, rgb_frame)
|
||||
results = detector.detect_for_video(mp_image, int(frame_num * 1000 / fps))
|
||||
|
||||
if results.pose_landmarks:
|
||||
persons = []
|
||||
|
||||
for pose_landmarks in results.pose_landmarks:
|
||||
total_poses += 1
|
||||
|
||||
# Convert landmarks
|
||||
keypoints = []
|
||||
for idx, landmark in enumerate(pose_landmarks):
|
||||
name = LANDMARK_NAMES[idx] if idx < len(LANDMARK_NAMES) else f"landmark_{idx}"
|
||||
kp = {
|
||||
"name": name,
|
||||
"x": landmark.x * width,
|
||||
"y": landmark.y * height,
|
||||
"z": landmark.z if hasattr(landmark, 'z') else 0,
|
||||
"confidence": landmark.visibility if hasattr(landmark, 'visibility') else 1.0,
|
||||
}
|
||||
keypoints.append(kp)
|
||||
|
||||
# Check alignment
|
||||
aligned, matched_bbox = face_keypoints_aligned(keypoints, face_bboxes)
|
||||
|
||||
if aligned:
|
||||
aligned_poses += 1
|
||||
face_keypoints = [kp for kp in keypoints if kp['name'] in ['nose', 'left_eye', 'right_eye']]
|
||||
|
||||
persons.append({
|
||||
"keypoints": keypoints,
|
||||
"face_keypoints": face_keypoints,
|
||||
"matched_face_bbox": matched_bbox,
|
||||
})
|
||||
|
||||
if persons:
|
||||
frames_data.append({
|
||||
"frame": frame_num,
|
||||
"timestamp": frame_num / fps,
|
||||
"persons": persons,
|
||||
})
|
||||
|
||||
if (i + 1) % 500 == 0:
|
||||
elapsed = time.time() - start_time
|
||||
print(f"[pose_aligned] Processed {i+1}/{len(face_frame_nums)} frames, {aligned_poses} aligned poses ({elapsed:.1f}s)")
|
||||
|
||||
cap.release()
|
||||
detector.close()
|
||||
|
||||
elapsed = time.time() - start_time
|
||||
|
||||
# Build output
|
||||
output = {
|
||||
"file_uuid": file_uuid,
|
||||
"processor": "mediapipe_pose_aligned",
|
||||
"fps": fps,
|
||||
"total_frames": total_frames,
|
||||
"face_frames_processed": len(face_frame_nums),
|
||||
"total_poses_detected": total_poses,
|
||||
"aligned_poses": aligned_poses,
|
||||
"alignment_rate": f"{aligned_poses / total_poses * 100:.1f}%" if total_poses > 0 else "0%",
|
||||
"frames_with_aligned_pose": len(frames_data),
|
||||
"elapsed_seconds": round(elapsed, 2),
|
||||
"frames": frames_data,
|
||||
}
|
||||
|
||||
# Save
|
||||
with open(output_path, "w") as f:
|
||||
json.dump(output, f)
|
||||
|
||||
print(f"\n[pose_aligned] Saved: {output_path}")
|
||||
print(f"[pose_aligned] Total poses detected: {total_poses}")
|
||||
print(f"[pose_aligned] Aligned poses: {aligned_poses} ({output['alignment_rate']})")
|
||||
print(f"[pose_aligned] Frames with aligned pose: {len(frames_data)}")
|
||||
print(f"[pose_aligned] Elapsed: {elapsed:.1f}s")
|
||||
|
||||
return output
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="MediaPipe pose aligned with face frames")
|
||||
parser.add_argument("--video", "-v", required=True, help="Video file path")
|
||||
parser.add_argument("--file-uuid", "-u", required=True, help="File UUID")
|
||||
parser.add_argument("--output-dir", "-o", default="/Users/accusys/momentry/output", help="Output directory")
|
||||
args = parser.parse_args()
|
||||
|
||||
output_path = Path(args.output_dir)
|
||||
face_json = output_path / f"{args.file_uuid}.face_traced.json"
|
||||
|
||||
if not face_json.exists():
|
||||
print(f"[pose_aligned] Face file not found: {face_json}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
output_file = output_path / f"{args.file_uuid}.pose.mediapipe.aligned.json"
|
||||
|
||||
result = process_video(
|
||||
args.video,
|
||||
str(face_json),
|
||||
str(output_file),
|
||||
args.file_uuid,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user