fix: face group name read consistency, sync_file_status fix, cleanup ghost records, identity_agent replaced with face_dedup

- get_face_groups_handler: COALESCE(tp.name, tn.label) for name consistency
- sync_file_status: compare JSON vs pre_chunks (not chunk table)
- face consistency: compare frames.len() not total_faces
- cleanup 2 ghost records with NULL file_name/file_path
- replace identity_agent with face_dedup in pipeline stages
- remove identity_agent_api.rs and all references
- update required_processors to match actual processors
- update AGENTS.md with team responsibilities
- add Studio pipeline changes documentation
This commit is contained in:
Accusys
2026-07-27 02:15:51 +08:00
parent fcdeab82e6
commit 39a2cbc65b
118 changed files with 19386 additions and 2964 deletions
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,338 @@
#!/opt/homebrew/bin/swift
/**
* Face-to-Pose Cropping Experiment
*
* 用 face bbox 放大比例擷取區域,送給 Apple Vision body pose 處理
* 驗證是否能提高 face-pose 匹配準確率
*
* Usage: swift face_pose_crop_experiment.swift --video <video_path> --frames <count>
*/
import Foundation
import AVFoundation
import Vision
import CoreGraphics
import CoreImage
// MARK: - Data Models
struct FaceResult {
let x: Int, y: Int, w: Int, h: Int
let confidence: Float
}
struct PoseResult {
let bbox: BBox
let noseX: Double, noseY: Double
let hasNose: Bool
struct BBox {
let x: Int, y: Int, w: Int, h: Int
}
}
struct FrameResult: Codable {
let frame: Int
let faceCount: Int
let poseFullCount: Int
let poseCropCount: Int
let faces: [FaceEntry]
let posesFull: [PoseEntry]
let posesCrop: [PoseEntry]
struct FaceEntry: Codable {
let x: Int, y: Int, w: Int, h: Int
}
struct PoseEntry: Codable {
let x: Int, y: Int, w: Int, h: Int
let noseX: Double, noseY: Double
}
}
struct Summary: Codable {
let totalFrames: Int
let fullFrameMatches: Int
let croppedMatches: Int
let fullFrameAvgDist: Double
let croppedAvgDist: Double
}
// MARK: - Detection Functions
func detectFaces(imageBuffer: CVPixelBuffer, width: Int, height: Int) -> [FaceResult] {
let handler = VNImageRequestHandler(cvPixelBuffer: imageBuffer, options: [:])
let request = VNDetectFaceRectanglesRequest()
var results: [FaceResult] = []
do {
try handler.perform([request])
if let observations = request.results {
for obs in observations {
let rect = obs.boundingBox
results.append(FaceResult(
x: Int(rect.origin.x * Double(width)),
y: Int((1 - rect.origin.y - rect.height) * Double(height)),
w: Int(rect.width * Double(width)),
h: Int(rect.height * Double(height)),
confidence: obs.confidence
))
}
}
} catch {}
return results
}
func detectPoseInRegion(imageBuffer: CVPixelBuffer, width: Int, height: Int, cropRect: CGRect? = nil) -> [PoseResult] {
var targetBuffer: CVPixelBuffer = imageBuffer
// If cropRect provided, crop the image
if let rect = cropRect {
let ciImage = CIImage(cvPixelBuffer: imageBuffer)
let context = CIContext(options: nil)
let cropped = ciImage.cropped(to: rect)
// Create new pixel buffer for cropped image
let cropW = Int(rect.width)
let cropH = Int(rect.height)
var newBuffer: CVPixelBuffer?
let status = CVPixelBufferCreate(kCFAllocatorDefault, cropW, cropH, kCVPixelFormatType_32BGRA, nil, &newBuffer)
guard status == kCVReturnSuccess, let buffer = newBuffer else { return [] }
context.render(cropped, to: buffer)
targetBuffer = buffer
}
let handler = VNImageRequestHandler(cvPixelBuffer: targetBuffer, options: [:])
let request = VNDetectHumanBodyPoseRequest()
var results: [PoseResult] = []
do {
try handler.perform([request])
if let observations = request.results as? [VNHumanBodyPoseObservation] {
for obs in observations {
var minX = Double.infinity, minY = Double.infinity
var maxX = -Double.infinity, maxY = -Double.infinity
var noseX: Double = 0, noseY: Double = 0
var hasNose = false
let joints: [VNHumanBodyPoseObservation.JointName] = [.nose, .leftEye, .rightEye, .leftShoulder, .rightShoulder]
for jn in joints {
if let pt = try? obs.recognizedPoint(jn), pt.confidence > 0.3 {
let px = pt.location.x * Double(width)
let py = (1 - pt.location.y) * Double(height)
minX = min(minX, px)
minY = min(minY, py)
maxX = max(maxX, px)
maxY = max(maxY, py)
if jn == .nose { noseX = px; noseY = py; hasNose = true }
}
}
if hasNose {
let pad = 20
results.append(PoseResult(
bbox: PoseResult.BBox(
x: Int(max(0, minX - Double(pad))),
y: Int(max(0, minY - Double(pad))),
w: Int(maxX - minX + Double(pad * 2)),
h: Int(maxY - minY + Double(pad * 2))
),
noseX: noseX,
noseY: noseY,
hasNose: true
))
}
}
}
} catch {}
return results
}
// MARK: - Main
func run(videoPath: String, maxFrames: Int) async {
print("[CropExperiment] Loading video: \(videoPath)")
let url = URL(fileURLWithPath: videoPath)
let asset = AVURLAsset(url: url)
let tracks = (try? await asset.loadTracks(withMediaType: .video)) ?? []
guard let track = tracks.first else { exit(1) }
let formatDesc = (try? await track.load(.formatDescriptions))?.first
let dims = formatDesc.map { CMVideoFormatDescriptionGetDimensions($0) } ?? CMVideoDimensions(width: 1920, height: 1080)
let width = Int(dims.width)
let height = Int(dims.height)
print("[CropExperiment] Video: \(width)x\(height)")
print("[CropExperiment] Analyzing \(maxFrames) frames...\n")
guard let reader = try? AVAssetReader(asset: asset) else { exit(1) }
let outputSettings: [String: Any] = [kCVPixelBufferPixelFormatTypeKey as String: Int(kCVPixelFormatType_32BGRA)]
let trackOutput = AVAssetReaderTrackOutput(track: track, outputSettings: outputSettings)
reader.add(trackOutput)
reader.startReading()
var frameResults: [FrameResult] = []
var frameIndex = 0
let scaleFactors: [Double] = [2.0, 3.0, 4.0] // Test different scale factors
print("[CropExperiment] Processing frames...\n")
while reader.status == .reading, let sampleBuffer = trackOutput.copyNextSampleBuffer() {
if frameIndex >= maxFrames { break }
guard let imageBuffer = CMSampleBufferGetImageBuffer(sampleBuffer) else {
frameIndex += 1
continue
}
// Detect faces
let faces = detectFaces(imageBuffer: imageBuffer, width: width, height: height)
// Detect pose on full frame
let posesFull = detectPoseInRegion(imageBuffer: imageBuffer, width: width, height: height)
// Detect pose on cropped region with different scale factors
var posesCrop2x: [PoseResult] = []
var posesCrop4x: [PoseResult] = []
var posesCropFull: [PoseResult] = []
if let firstFace = faces.first {
let cx = Double(firstFace.x + firstFace.w / 2)
let cy = Double(firstFace.y + firstFace.h / 2)
// 2x scale
let scale2 = 2.0
let cropW2 = Double(firstFace.w) * scale2
let cropH2 = Double(firstFace.h) * scale2
let rect2 = CGRect(x: max(0, cx - cropW2/2), y: max(0, cy - cropH2/2), width: min(cropW2, Double(width)), height: min(cropH2, Double(height)))
posesCrop2x = detectPoseInRegion(imageBuffer: imageBuffer, width: Int(rect2.width), height: Int(rect2.height), cropRect: rect2)
// 4x scale
let scale4 = 4.0
let cropW4 = Double(firstFace.w) * scale4
let cropH4 = Double(firstFace.h) * scale4
let rect4 = CGRect(x: max(0, cx - cropW4/2), y: max(0, cy - cropH4/2), width: min(cropW4, Double(width)), height: min(cropH4, Double(height)))
posesCrop4x = detectPoseInRegion(imageBuffer: imageBuffer, width: Int(rect4.width), height: Int(rect4.height), cropRect: rect4)
}
// Calculate matching distances
var fullDists: [Double] = []
var cropDists: [Double] = []
for face in faces {
let fcx = Double(face.x + face.w / 2)
let fcy = Double(face.y + face.h / 2)
if let pose = posesFull.first(where: { $0.hasNose }) {
let dist = abs(fcx - pose.noseX) + abs(fcy - pose.noseY)
fullDists.append(dist)
}
if let pose = posesCrop.first(where: { $0.hasNose }) {
let dist = abs(fcx - pose.noseX) + abs(fcy - pose.noseY)
cropDists.append(dist)
}
}
frameResults.append(FrameResult(
frame: frameIndex,
faceCount: faces.count,
poseFullCount: posesFull.count,
poseCropCount: posesCrop.count,
faces: faces.map { FrameResult.FaceEntry(x: $0.x, y: $0.y, w: $0.w, h: $0.h) },
posesFull: posesFull.map { FrameResult.PoseEntry(x: $0.bbox.x, y: $0.bbox.y, w: $0.bbox.w, h: $0.bbox.h, noseX: $0.noseX, noseY: $0.noseY) },
posesCrop: posesCrop.map { FrameResult.PoseEntry(x: $0.bbox.x, y: $0.bbox.y, w: $0.bbox.w, h: $0.bbox.h, noseX: $0.noseX, noseY: $0.noseY) }
))
if frameIndex % 50 == 0 {
print(" Frame \(frameIndex): faces=\(faces.count), pose_full=\(posesFull.count), pose_crop=\(posesCrop.count)")
}
frameIndex += 1
}
reader.cancelReading()
// Calculate summary
let totalFrames = frameResults.count
var fullDistsAll: [Double] = []
var cropDistsAll: [Double] = []
for fr in frameResults {
for face in fr.faces {
let fcx = Double(face.x + face.w / 2)
let fcy = Double(face.y + face.h / 2)
for pose in fr.posesFull {
fullDistsAll.append(abs(fcx - pose.noseX) + abs(fcy - pose.noseY))
}
for pose in fr.posesCrop {
cropDistsAll.append(abs(fcx - pose.noseX) + abs(fcy - pose.noseY))
}
}
}
let fullAvg = fullDistsAll.isEmpty ? 0 : fullDistsAll.reduce(0, +) / Double(fullDistsAll.count)
let cropAvg = cropDistsAll.isEmpty ? 0 : cropDistsAll.reduce(0, +) / Double(cropDistsAll.count)
let fullMatches = fullDistsAll.filter { $0 < 100 }.count
let cropMatches = cropDistsAll.filter { $0 < 100 }.count
let summary = Summary(
totalFrames: totalFrames,
fullFrameMatches: fullMatches,
croppedMatches: cropMatches,
fullFrameAvgDist: fullAvg,
croppedAvgDist: cropAvg
)
// Save results
let encoder = JSONEncoder()
encoder.outputFormatting = [.prettyPrinted, .sortedKeys]
let outputDir = "experiments/face_pose_sync_poc/output"
try? FileManager.default.createDirectory(atPath: outputDir, withIntermediateDirectories: true)
let framesData = try! encoder.encode(frameResults)
try! framesData.write(to: URL(fileURLWithPath: "\(outputDir)/crop_experiment_frames.json"))
let summaryData = try! encoder.encode(summary)
try! summaryData.write(to: URL(fileURLWithPath: "\(outputDir)/crop_experiment_summary.json"))
// Print report
print("\n" + String(repeating: "=", count: 50))
print(" Face-Pose Cropping Experiment Report")
print(String(repeating: "=", count: 50))
print("Frames analyzed: \(totalFrames)")
print()
print("Full Frame Detection:")
print(" Avg distance: \(String(format: "%.1f", fullAvg))px")
print(" Matches (<100px): \(fullMatches) (\(fullDistsAll.count > 0 ? String(format: "%.1f%%", Double(fullMatches)/Double(fullDistsAll.count)*100) : "N/A"))")
print()
print("Cropped Region Detection (3x face bbox):")
print(" Avg distance: \(String(format: "%.1f", cropAvg))px")
print(" Matches (<100px): \(cropMatches) (\(cropDistsAll.count > 0 ? String(format: "%.1f%%", Double(cropMatches)/Double(cropDistsAll.count)*100) : "N/A"))")
print()
print("Improvement:")
print(" Distance reduction: \(fullAvg > 0 ? String(format: "%.1f%%", (fullAvg - cropAvg) / fullAvg * 100) : "N/A")")
print()
print("Results saved to: \(outputDir)/crop_experiment_*.json")
}
// Parse arguments
let args = CommandLine.arguments
let videoPath: String
let maxFrames: Int
if args.count >= 2 {
videoPath = args[1]
maxFrames = args.count > 2 ? Int(args[2]) ?? 300 : 300
} else {
// Default test video
videoPath = "/Users/accusys/momentry/var/sftpgo/data/demo/Accusys-WD_FilmRiot_test.mp4"
maxFrames = 300
}
print("[CropExperiment] Starting...")
await run(videoPath: videoPath, maxFrames: maxFrames)
@@ -0,0 +1,237 @@
#!/opt/homebrew/bin/swift
/**
* Face-Pose Matching Experiment
*
* 測試不同方法來匹配 face 和 body pose
* 1. Full frame pose detection
* 2. Cropped region (2x, 4x face bbox)
* 3. 找出最佳匹配方法
*/
import Foundation
import AVFoundation
import Vision
import CoreGraphics
import CoreImage
struct FaceData {
let x: Int, y: Int, w: Int, h: Int
}
struct PoseData {
let noseX: Double, noseY: Double
let bboxX: Int, bboxY: Int, bboxW: Int, bboxH: Int
}
func detectFaces(imageBuffer: CVPixelBuffer, width: Int, height: Int) -> [FaceData] {
let handler = VNImageRequestHandler(cvPixelBuffer: imageBuffer, options: [:])
let request = VNDetectFaceRectanglesRequest()
var results: [FaceData] = []
try? handler.perform([request])
if let observations = request.results {
for obs in observations {
let rect = obs.boundingBox
results.append(FaceData(
x: Int(rect.origin.x * Double(width)),
y: Int((1 - rect.origin.y - rect.height) * Double(height)),
w: Int(rect.width * Double(width)),
h: Int(rect.height * Double(height))
))
}
}
return results
}
func detectPose(imageBuffer: CVPixelBuffer, width: Int, height: Int, cropRect: CGRect? = nil) -> [PoseData] {
var targetBuffer: CVPixelBuffer = imageBuffer
if let rect = cropRect {
let ciImage = CIImage(cvPixelBuffer: imageBuffer)
let context = CIContext(options: nil)
let cropped = ciImage.cropped(to: rect)
let cropW = Int(rect.width)
let cropH = Int(rect.height)
var newBuffer: CVPixelBuffer?
let status = CVPixelBufferCreate(kCFAllocatorDefault, cropW, cropH, kCVPixelFormatType_32BGRA, nil, &newBuffer)
guard status == kCVReturnSuccess, let buffer = newBuffer else { return [] }
context.render(cropped, to: buffer)
targetBuffer = buffer
}
let handler = VNImageRequestHandler(cvPixelBuffer: targetBuffer, options: [:])
let request = VNDetectHumanBodyPoseRequest()
var results: [PoseData] = []
try? handler.perform([request])
if let observations = request.results as? [VNHumanBodyPoseObservation] {
for obs in observations {
var minX = Double.infinity, minY = Double.infinity
var maxX = -Double.infinity, maxY = -Double.infinity
var noseX: Double = 0, noseY: Double = 0
var hasNose = false
for jn in [VNHumanBodyPoseObservation.JointName.nose, .leftEye, .rightEye, .leftShoulder, .rightShoulder] {
if let pt = try? obs.recognizedPoint(jn), pt.confidence > 0.3 {
let px = pt.location.x * Double(width)
let py = (1 - pt.location.y) * Double(height)
minX = min(minX, px); minY = min(minY, py)
maxX = max(maxX, px); maxY = max(maxY, py)
if jn == .nose { noseX = px; noseY = py; hasNose = true }
}
}
if hasNose {
results.append(PoseData(
noseX: noseX, noseY: noseY,
bboxX: Int(max(0, minX - 20)), bboxY: Int(max(0, minY - 20)),
bboxW: Int(maxX - minX + 40), bboxH: Int(maxY - minY + 40)
))
}
}
}
return results
}
func run(videoPath: String, maxFrames: Int) async {
print("[Experiment] Loading video: \(videoPath)")
let url = URL(fileURLWithPath: videoPath)
let asset = AVURLAsset(url: url)
let tracks = (try? await asset.loadTracks(withMediaType: .video)) ?? []
guard let track = tracks.first else { exit(1) }
let formatDesc = (try? await track.load(.formatDescriptions))?.first
let dims = formatDesc.map { CMVideoFormatDescriptionGetDimensions($0) } ?? CMVideoDimensions(width: 1920, height: 1080)
let width = Int(dims.width)
let height = Int(dims.height)
print("[Experiment] Video: \(width)x\(height)")
print("[Experiment] Analyzing \(maxFrames) frames...\n")
guard let reader = try? AVAssetReader(asset: asset) else { exit(1) }
let outputSettings: [String: Any] = [kCVPixelBufferPixelFormatTypeKey as String: Int(kCVPixelFormatType_32BGRA)]
let trackOutput = AVAssetReaderTrackOutput(track: track, outputSettings: outputSettings)
reader.add(trackOutput)
reader.startReading()
var frameIndex = 0
// Results tracking
var fullDists: [Double] = []
var crop2xDists: [Double] = []
var crop4xDists: [Double] = []
var framesWithFace = 0
var framesWithPoseFull = 0
var framesWithPose2x = 0
var framesWithPose4x = 0
print("[Experiment] Processing frames...")
while reader.status == .reading, let sampleBuffer = trackOutput.copyNextSampleBuffer() {
if frameIndex >= maxFrames { break }
guard let imageBuffer = CMSampleBufferGetImageBuffer(sampleBuffer) else {
frameIndex += 1
continue
}
let faces = detectFaces(imageBuffer: imageBuffer, width: width, height: height)
if !faces.isEmpty {
framesWithFace += 1
// Full frame pose
let posesFull = detectPose(imageBuffer: imageBuffer, width: width, height: height)
if !posesFull.isEmpty { framesWithPoseFull += 1 }
// Cropped regions
let firstFace = faces[0]
let cx = Double(firstFace.x + firstFace.w / 2)
let cy = Double(firstFace.y + firstFace.h / 2)
// 2x crop
let rect2x = CGRect(x: max(0, cx - Double(firstFace.w)), y: max(0, cy - Double(firstFace.h)),
width: Double(firstFace.w * 2), height: Double(firstFace.h * 2))
let poses2x = detectPose(imageBuffer: imageBuffer, width: Int(rect2x.width), height: Int(rect2x.height), cropRect: rect2x)
if !poses2x.isEmpty { framesWithPose2x += 1 }
// 4x crop
let rect4x = CGRect(x: max(0, cx - Double(firstFace.w * 2)), y: max(0, cy - Double(firstFace.h * 2)),
width: Double(firstFace.w * 4), height: Double(firstFace.h * 4))
let poses4x = detectPose(imageBuffer: imageBuffer, width: Int(rect4x.width), height: Int(rect4x.height), cropRect: rect4x)
if !poses4x.isEmpty { framesWithPose4x += 1 }
// Calculate distances
let fcx = cx
let fcy = cy
for pose in posesFull {
fullDists.append(abs(fcx - pose.noseX) + abs(fcy - pose.noseY))
}
for pose in poses2x {
crop2xDists.append(abs(fcx - pose.noseX) + abs(fcy - pose.noseY))
}
for pose in poses4x {
crop4xDists.append(abs(fcx - pose.noseX) + abs(fcy - pose.noseY))
}
}
if frameIndex % 50 == 0 {
print(" Frame \(frameIndex): faces=\(faces.count)")
}
frameIndex += 1
}
reader.cancelReading()
// Calculate stats
func stats(_ dists: [Double]) -> (avg: Double, median: Double, under50: Int, under100: Int) {
let sorted = dists.sorted()
let avg = sorted.isEmpty ? 0 : sorted.reduce(0, +) / Double(sorted.count)
let median = sorted.isEmpty ? 0 : sorted[sorted.count / 2]
let under50 = sorted.filter { $0 < 50 }.count
let under100 = sorted.filter { $0 < 100 }.count
return (avg, median, under50, under100)
}
let fullStats = stats(fullDists)
let crop2xStats = stats(crop2xDists)
let crop4xStats = stats(crop4xDists)
print("\n" + String(repeating: "=", count: 60))
print(" Face-Pose Matching Experiment Results")
print(String(repeating: "=", count: 60))
print("Frames analyzed: \(frameIndex)")
print("Frames with face: \(framesWithFace)")
print()
print("Detection Rate:")
print(" Full frame: \(framesWithPoseFull)/\(framesWithFace) (\(String(format: "%.1f%%", Double(framesWithPoseFull)/Double(max(1,framesWithFace))*100))")
print(" 2x crop: \(framesWithPose2x)/\(framesWithFace) (\(String(format: "%.1f%%", Double(framesWithPose2x)/Double(max(1,framesWithFace))*100))")
print(" 4x crop: \(framesWithPose4x)/\(framesWithFace) (\(String(format: "%.1f%%", Double(framesWithPose4x)/Double(max(1,framesWithFace))*100))")
print()
print("Distance (face center ↔ pose nose):")
print(" Method | Avg px | Median px | <50px | <100px | Pairs")
print(" -------------|---------|-----------|-------|--------|------")
print(" Full frame | \(String(format: "%7.1f", fullStats.avg)) | \(String(format: "%9.1f", fullStats.median)) | \(String(format: "%5d", fullStats.under50)) | \(String(format: "%6d", fullStats.under100)) | \(fullDists.count)")
print(" 2x crop | \(String(format: "%7.1f", crop2xStats.avg)) | \(String(format: "%9.1f", crop2xStats.median)) | \(String(format: "%5d", crop2xStats.under50)) | \(String(format: "%6d", crop2xStats.under100)) | \(crop2xDists.count)")
print(" 4x crop | \(String(format: "%7.1f", crop4xStats.avg)) | \(String(format: "%9.1f", crop4xStats.median)) | \(String(format: "%5d", crop4xStats.under50)) | \(String(format: "%6d", crop4xStats.under100)) | \(crop4xDists.count)")
print()
// Conclusion
if fullStats.median < crop2xStats.median && fullStats.median < crop4xStats.median {
print("Conclusion: Full frame detection gives best matching accuracy")
} else if crop2xStats.median < fullStats.median && crop2xStats.median < crop4xStats.median {
print("Conclusion: 2x crop gives best matching accuracy")
} else {
print("Conclusion: 4x crop gives best matching accuracy")
}
}
let args = CommandLine.arguments
let videoPath = args.count >= 2 ? args[1] : "/Users/accusys/momentry/var/sftpgo/data/demo/Accusys-WD_FilmRiot_test.mp4"
let maxFrames = args.count > 2 ? Int(args[2]) ?? 300 : 300
await run(videoPath: videoPath, maxFrames: maxFrames)
Binary file not shown.
@@ -0,0 +1,309 @@
#!/opt/homebrew/bin/swift
/**
* Generate pose_traced.json from face_traced.json + pose.json
*
* 流程:
* 1. 讀取 face_traced.json(已有 trace_id)
* 2. 讀取 pose.json(有 body keypoints)
* 3. 對每個 frame,用 face bbox center 匹配 pose nose keypoint
* 4. 距離 < 50px 判定為同一人,賦予相同 trace_id
* 5. 輸出 pose_traced.json
*
* Usage: swift generate_pose_traced.swift --face-traced <face_traced.json> --pose <pose.json> --output <pose_traced.json>
*/
import Foundation
// MARK: - Data Models
struct FaceTraced: Codable {
let status: String
let frame_count: Int
let fps: Double
let frames: [FrameEntry]
let traces: [String: TraceEntry]
struct FrameEntry: Codable {
let frame_number: Int
let time_seconds: Double
let faces: [FaceInFrame]
}
struct FaceInFrame: Codable {
let face_number: Int
let trace_id: Int
let bbox: BBox
let confidence: Double
struct BBox: Codable {
let x: Int, y: Int, width: Int, height: Int
}
}
struct TraceEntry: Codable {
let path: [PathEntry]
struct PathEntry: Codable {
let frame: Int
let bbox: BBox
let confidence: Double
struct BBox: Codable {
let x: Int, y: Int, width: Int, height: Int
}
}
}
}
struct PoseJson: Codable {
let frame_count: Int
let fps: Double
let frames: [PoseFrame]
struct PoseFrame: Codable {
let frame: Int?
let timestamp: Double?
let persons: [PersonEntry]
struct PersonEntry: Codable {
let bbox: BBox?
let keypoints: [Keypoint]
struct BBox: Codable {
let x: Int, y: Int, width: Int, height: Int
}
struct Keypoint: Codable {
let name: String
let x: Double, y: Double
let confidence: Float
}
}
}
}
struct PoseTracedOutput: Codable {
let status: String
let frame_count: Int
let fps: Double
let frames: [PoseTracedFrame]
let trace_mapping: [String: TraceMapping]
let stats: Stats
struct PoseTracedFrame: Codable {
let frame_number: Int
let time_seconds: Double
let persons: [PersonInFrame]
}
struct PersonInFrame: Codable {
let person_index: Int
let trace_id: Int?
let bbox: BBox?
let keypoints: [Keypoint]
let match_distance: Double
struct BBox: Codable {
let x: Int, y: Int, width: Int, height: Int
}
struct Keypoint: Codable {
let name: String
let x: Double, y: Double
let confidence: Float
}
}
struct TraceMapping: Codable {
let trace_id: Int
let pose_count: Int
let frames: [Int]
}
struct Stats: Codable {
let total_frames: Int
let frames_with_pose: Int
let matched_poses: Int
let unmatched_poses: Int
let avg_distance: Double
let median_distance: Double
}
}
// MARK: - Matching
let threshold = 50.0
func generatePoseTraced(faceTraced: FaceTraced, poseJson: PoseJson) -> PoseTracedOutput {
// Build frame -> faces lookup from face_traced
var facesByFrame: [Int: [FaceTraced.FaceInFrame]] = [:]
for fe in faceTraced.frames {
facesByFrame[fe.frame_number] = fe.faces
}
var tracedFrames: [PoseTracedOutput.PoseTracedFrame] = []
var tracePoseCount: [Int: (count: Int, frames: Set<Int>)] = [:]
var allDistances: [Double] = []
var matchedCount = 0
var unmatchedCount = 0
for pf in poseJson.frames {
let fn = pf.frame ?? 0
let ts = pf.timestamp ?? 0.0
let facesInFrame = facesByFrame[fn] ?? []
var personsInFrame: [PoseTracedOutput.PersonInFrame] = []
var usedFaceIndices = Set<Int>()
for (poseIdx, person) in pf.persons.enumerated() {
// Find nose keypoint
let nose = person.keypoints.first(where: { $0.name.lowercased().contains("nose") })
var bestTraceId: Int? = nil
var bestDist = Double.infinity
if let nose = nose {
// Match to nearest face
for (faceIdx, face) in facesInFrame.enumerated() {
if usedFaceIndices.contains(faceIdx) { continue }
let fcx = Double(face.bbox.x + face.bbox.width / 2)
let fcy = Double(face.bbox.y + face.bbox.height / 2)
let dist = abs(fcx - nose.x) + abs(fcy - nose.y)
if dist < bestDist {
bestDist = dist
bestTraceId = face.trace_id
}
}
}
let isMatched = bestTraceId != nil && bestDist < threshold
if isMatched {
matchedCount += 1
usedFaceIndices.insert(usedFaceIndices.first(where: { _ in true }) ?? 0)
allDistances.append(bestDist)
// Update trace mapping
if let tid = bestTraceId {
var current = tracePoseCount[tid] ?? (count: 0, frames: [])
current.count += 1
current.frames.insert(fn)
tracePoseCount[tid] = current
}
} else {
unmatchedCount += 1
}
personsInFrame.append(PoseTracedOutput.PersonInFrame(
person_index: poseIdx,
trace_id: isMatched ? bestTraceId : nil,
bbox: person.bbox.map { PoseTracedOutput.PersonInFrame.BBox(x: $0.x, y: $0.y, width: $0.width, height: $0.height) },
keypoints: person.keypoints.map { PoseTracedOutput.PersonInFrame.Keypoint(name: $0.name, x: $0.x, y: $0.y, confidence: $0.confidence) },
match_distance: bestDist
))
}
tracedFrames.append(PoseTracedOutput.PoseTracedFrame(
frame_number: fn,
time_seconds: ts,
persons: personsInFrame
))
}
let sortedDists = allDistances.sorted()
let avgDist = sortedDists.isEmpty ? 0 : sortedDists.reduce(0, +) / Double(sortedDists.count)
let medianDist = sortedDists.isEmpty ? 0 : sortedDists[sortedDists.count / 2]
// Build trace mapping
var traceMapping: [String: PoseTracedOutput.TraceMapping] = [:]
for (tid, data) in tracePoseCount {
traceMapping["\(tid)"] = PoseTracedOutput.TraceMapping(
trace_id: tid,
pose_count: data.count,
frames: data.frames.sorted()
)
}
let stats = PoseTracedOutput.Stats(
total_frames: poseJson.frames.count,
frames_with_pose: poseJson.frames.filter { !$0.persons.isEmpty }.count,
matched_poses: matchedCount,
unmatched_poses: unmatchedCount,
avg_distance: avgDist,
median_distance: medianDist
)
return PoseTracedOutput(
status: faceTraced.status,
frame_count: faceTraced.frame_count,
fps: faceTraced.fps,
frames: tracedFrames,
trace_mapping: traceMapping,
stats: stats
)
}
// MARK: - Main
func run(faceTracedPath: String, posePath: String, outputPath: String) {
print("[PoseTraced] Loading face_traced.json: \(faceTracedPath)")
let faceTraced = try! JSONDecoder().decode(FaceTraced.self, from: Data(contentsOf: URL(fileURLWithPath: faceTracedPath)))
print("[PoseTraced] Loading pose.json: \(posePath)")
let poseJson = try! JSONDecoder().decode(PoseJson.self, from: Data(contentsOf: URL(fileURLWithPath: posePath)))
print("[PoseTraced] Matching face traces to poses (threshold: \(threshold)px)...")
let result = generatePoseTraced(faceTraced: faceTraced, poseJson: poseJson)
// Save output
let encoder = JSONEncoder()
encoder.outputFormatting = [.prettyPrinted, .sortedKeys]
let jsonData = try! encoder.encode(result)
try! jsonData.write(to: URL(fileURLWithPath: outputPath))
// Print summary
let s = result.stats
print("\n=== Pose Traced Generation Report ===")
print("Total frames: \(s.total_frames)")
print("Frames with pose: \(s.frames_with_pose)")
print()
print("Matching (threshold: \(threshold)px):")
print(" Matched poses: \(s.matched_poses)")
print(" Unmatched poses: \(s.unmatched_poses)")
print(" Match rate: \(s.matched_poses + s.unmatched_poses > 0 ? String(format: "%.1f%%", Double(s.matched_poses)/Double(s.matched_poses + s.unmatched_poses)*100) : "N/A")")
print()
print("Distance (face center ↔ pose nose):")
print(" Average: \(String(format: "%.1f", s.avg_distance))px")
print(" Median: \(String(format: "%.1f", s.median_distance))px")
print()
print("Trace mapping:")
for (tid, mapping) in result.trace_mapping.sorted(by: { Int($0.key)! < Int($1.key)! }) {
print(" trace_\(mapping.trace_id): \(mapping.pose_count) poses, frames \(mapping.frames.first ?? 0)-\(mapping.frames.last ?? 0)")
}
print("\nOutput saved to: \(outputPath)")
}
// Parse arguments
var faceTracedPath: String?
var posePath: String?
var outputPath: String?
let args = CommandLine.arguments
var i = 1
while i < args.count {
switch args[i] {
case "--face-traced": faceTracedPath = args[i+1]; i += 2
case "--pose": posePath = args[i+1]; i += 2
case "--output": outputPath = args[i+1]; i += 2
default: i += 1
}
}
guard let faceTracedPath = faceTracedPath, let posePath = posePath, let outputPath = outputPath else {
print("Usage: swift generate_pose_traced.swift --face-traced <face_traced.json> --pose <pose.json> --output <pose_traced.json>")
exit(1)
}
run(faceTracedPath: faceTracedPath, posePath: posePath, outputPath: outputPath)
+372
View File
@@ -0,0 +1,372 @@
#!/opt/homebrew/bin/swift
/**
* Face-Pose Sync POC
*
* 使用 Apple Vision 在同一幀上同時檢測 face 和 pose
* 驗證兩者是否能正確同步並匹配
*
* Usage: swift main.swift <video_path> [max_frames]
*/
import Foundation
import AVFoundation
import Vision
// MARK: - Data Models
struct FaceResult: Codable {
let x: Int, y: Int, w: Int, h: Int
let confidence: Float
}
struct KeypointResult: Codable {
let name: String
let x: Double, y: Double
let confidence: Float
}
struct PoseResult: Codable {
let bbox: BBoxResult
let keypoints: [KeypointResult]
struct BBoxResult: Codable {
let x: Int, y: Int, w: Int, h: Int
}
}
struct FrameResult: Codable {
let frame: Int
let faceCount: Int
let poseCount: Int
let faces: [FaceResult]
let poses: [PoseResult]
}
struct Summary: Codable {
let totalFrames: Int
let framesWithFace: Int
let framesWithPose: Int
let framesWithBoth: Int
let syncRate: Double
let poseRecall: Double
let distanceCount: Int
let avgDistance: Double
let medianDistance: Double
let under50: Int, under100: Int, under150: Int, under200: Int, under300: Int
}
struct VideoInfo: Codable {
let width: Int, height: Int, fps: Double, duration: Double
}
struct Output: Codable {
let videoInfo: VideoInfo
let totalFrames: Int
let frames: [FrameResult]
let summary: Summary
}
// MARK: - Main
let args = CommandLine.arguments
guard args.count >= 2 else {
print("Usage: swift main.swift <video_path> [max_frames]")
exit(1)
}
let videoPath = args[1]
let maxFrames = args.count > 2 ? Int(args[2]) ?? 300 : 300
guard FileManager.default.fileExists(atPath: videoPath) else {
print("[ERROR] Video file not found: \(videoPath)")
exit(1)
}
print("[FacePoseSync] Loading video: \(videoPath)")
let url = URL(fileURLWithPath: videoPath)
let asset = AVURLAsset(url: url)
// Get video track
let tracks = try await asset.loadTracks(withMediaType: .video)
guard let track = tracks.first else {
print("[ERROR] No video track found")
exit(1)
}
// Get video properties
let formatDesc = try await track.load(.formatDescriptions).first
let cmDims = formatDesc.map { CMVideoFormatDescriptionGetDimensions($0) } ?? CMVideoDimensions(width: 1920, height: 1080)
let width = Int(cmDims.width)
let height = Int(cmDims.height)
let dur = try await asset.load(.duration)
let duration = CMTimeGetSeconds(dur)
let fps = try await track.load(.nominalFrameRate)
print("[FacePoseSync] Video: \(width)x\(height), \(fps)fps, \(String(format: "%.1f", duration))s")
print("[FacePoseSync] Analyzing up to \(maxFrames) frames...\n")
// Setup asset reader
let reader = try AVAssetReader(asset: asset)
let outputSettings: [String: Any] = [
kCVPixelBufferPixelFormatTypeKey as String: Int(kCVPixelFormatType_32BGRA)
]
let trackOutput = AVAssetReaderTrackOutput(track: track, outputSettings: outputSettings)
reader.add(trackOutput)
reader.startReading()
var frameResults: [FrameResult] = []
var frameIndex = 0
print("[FacePoseSync] Processing frames...")
while reader.status == .reading, let sampleBuffer = trackOutput.copyNextSampleBuffer() {
if frameIndex >= maxFrames { break }
guard let imageBuffer = CMSampleBufferGetImageBuffer(sampleBuffer) else {
frameIndex += 1
continue
}
// Run face and pose detection on the same frame
let (faces, poses) = await detectFaceAndPose(imageBuffer: imageBuffer, width: width, height: height)
let frameResult = FrameResult(
frame: frameIndex,
faceCount: faces.count,
poseCount: poses.count,
faces: faces,
poses: poses
)
frameResults.append(frameResult)
if frameIndex % 50 == 0 {
print(" Frame \(frameIndex): \(faces.count) faces, \(poses.count) poses")
}
frameIndex += 1
}
reader.cancelReading()
// Calculate summary
let totalFrames = frameResults.count
let framesWithFace = frameResults.filter { $0.faceCount > 0 }.count
let framesWithPose = frameResults.filter { $0.poseCount > 0 }.count
let framesWithBoth = frameResults.filter { $0.faceCount > 0 && $0.poseCount > 0 }.count
// Calculate face-pose distances using nearest-neighbor matching
var allDistances: [Double] = []
var matchedPairs = 0
var unmatchedFaces = 0
for fr in frameResults {
// For each face, find the closest pose nose
var usedPoses = Set<Int>()
for face in fr.faces {
let fcx = Double(face.x + face.w / 2)
let fcy = Double(face.y + face.h / 2)
var bestDist = Double.infinity
var bestPoseIdx = -1
for (idx, pose) in fr.poses.enumerated() {
if usedPoses.contains(idx) { continue }
if let nose = pose.keypoints.first(where: { $0.name.lowercased().contains("nose") }) {
let dist = abs(fcx - nose.x) + abs(fcy - nose.y)
if dist < bestDist {
bestDist = dist
bestPoseIdx = idx
}
}
}
if bestPoseIdx >= 0 && bestDist < 500 { // Threshold for valid match
allDistances.append(bestDist)
usedPoses.insert(bestPoseIdx)
matchedPairs += 1
} else {
unmatchedFaces += 1
}
}
}
let sortedDists = allDistances.sorted()
let avgDist = sortedDists.isEmpty ? 0.0 : sortedDists.reduce(0, +) / Double(sortedDists.count)
let medianDist = sortedDists.isEmpty ? 0.0 : sortedDists[sortedDists.count / 2]
let syncRate = framesWithFace > 0 ? Double(framesWithBoth) / Double(framesWithFace) : 0
let poseRecall = framesWithFace > 0 ? Double(framesWithBoth) / Double(framesWithFace) : 0
let summary = Summary(
totalFrames: totalFrames,
framesWithFace: framesWithFace,
framesWithPose: framesWithPose,
framesWithBoth: framesWithBoth,
syncRate: syncRate,
poseRecall: poseRecall,
distanceCount: allDistances.count,
avgDistance: avgDist,
medianDistance: medianDist,
under50: allDistances.filter { $0 < 50 }.count,
under100: allDistances.filter { $0 < 100 }.count,
under150: allDistances.filter { $0 < 150 }.count,
under200: allDistances.filter { $0 < 200 }.count,
under300: allDistances.filter { $0 < 300 }.count
)
// Add matching stats to output
let matchingInfo = """
Matching Analysis:
Matched pairs: \(matchedPairs)
Unmatched faces: \(unmatchedFaces)
Match rate: \(totalFrames > 0 ? String(format: "%.1f%%", Double(matchedPairs) / Double(framesWithFace) * 100) : "N/A")
"""
let videoInfo = VideoInfo(width: width, height: height, fps: Double(fps), duration: duration)
let output = Output(videoInfo: videoInfo, totalFrames: totalFrames, frames: frameResults, summary: summary)
// Save result
let encoder = JSONEncoder()
encoder.outputFormatting = [.prettyPrinted, .sortedKeys]
let jsonData = try encoder.encode(output)
let outputDir = "experiments/face_pose_sync_poc/output"
try FileManager.default.createDirectory(atPath: outputDir, withIntermediateDirectories: true)
let outputPath = "\(outputDir)/result.json"
try jsonData.write(to: URL(fileURLWithPath: outputPath))
func pct(_ n: Int, _ total: Int) -> String {
guard total > 0 else { return "0.0%" }
return String(format: "%.1f%%", Double(n) / Double(total) * 100)
}
// Print summary report
print("\n" + String(repeating: "=", count: 50))
print(" Face-Pose Sync Analysis Report")
print(String(repeating: "=", count: 50))
print("Video: \(width)x\(height), \(fps)fps, \(String(format: "%.1f", duration))s")
print("Frames analyzed: \(totalFrames)")
print()
print("Detection Stats:")
print(" Frames with face: \(framesWithFace) (\(pct(framesWithFace, totalFrames)))")
print(" Frames with pose: \(framesWithPose) (\(pct(framesWithPose, totalFrames)))")
print(" Frames with both: \(framesWithBoth) (\(pct(framesWithBoth, totalFrames)))")
print()
print("Sync Analysis:")
print(" Sync Rate (both/face): \(String(format: "%.1f%%", syncRate * 100))")
print(" Pose Recall: \(String(format: "%.1f%%", poseRecall * 100))")
print()
print("Distance (face center ↔ pose nose):")
print(" Total pairs: \(allDistances.count)")
print(" Average: \(String(format: "%.1f", avgDist))px")
print(" Median: \(String(format: "%.1f", medianDist))px")
print()
print(" Distance Distribution:")
print(" < 50px: \(summary.under50) (\(pct(summary.under50, max(1, summary.distanceCount)))")
print(" < 100px: \(summary.under100) (\(pct(summary.under100, max(1, summary.distanceCount)))")
print(" < 150px: \(summary.under150) (\(pct(summary.under150, max(1, summary.distanceCount)))")
print(" < 200px: \(summary.under200) (\(pct(summary.under200, max(1, summary.distanceCount)))")
print(" < 300px: \(summary.under300) (\(pct(summary.under300, max(1, summary.distanceCount)))")
print("\nResult saved to: \(outputPath)")
// MARK: - Detection Function
func detectFaceAndPose(imageBuffer: CVPixelBuffer, width: Int, height: Int) async -> ([FaceResult], [PoseResult]) {
let handler = VNImageRequestHandler(cvPixelBuffer: imageBuffer, options: [:])
let faceRequest = VNDetectFaceRectanglesRequest()
let poseRequest = VNDetectHumanBodyPoseRequest()
var faces: [FaceResult] = []
var poses: [PoseResult] = []
do {
try handler.perform([faceRequest, poseRequest])
// Process face results
if let faceObservations = faceRequest.results {
for obs in faceObservations {
let rect = obs.boundingBox
let x = Int(rect.origin.x * Double(width))
let y = Int((1 - rect.origin.y - rect.height) * Double(height))
let w = Int(rect.width * Double(width))
let h = Int(rect.height * Double(height))
faces.append(FaceResult(x: x, y: y, w: w, h: h, confidence: obs.confidence))
}
}
// Process pose results
if let poseObservations = poseRequest.results as? [VNHumanBodyPoseObservation] {
let jointNames: [VNHumanBodyPoseObservation.JointName] = [
.nose, .leftEye, .rightEye, .leftEar, .rightEar,
.leftShoulder, .rightShoulder, .leftElbow, .rightElbow,
.leftWrist, .rightWrist, .leftHip, .rightHip,
.leftKnee, .rightKnee, .leftAnkle, .rightAnkle
]
for obs in poseObservations {
var minX = Double.infinity, minY = Double.infinity
var maxX = -Double.infinity, maxY = -Double.infinity
var keypoints: [KeypointResult] = []
for jointName in jointNames {
let point = try? obs.recognizedPoint(jointName)
if let point = point, point.confidence > 0.3 {
let px = point.location.x * Double(width)
let py = (1 - point.location.y) * Double(height)
minX = min(minX, px)
minY = min(minY, py)
maxX = max(maxX, px)
maxY = max(maxY, py)
// Convert JointName to string
let name: String
switch jointName {
case .nose: name = "nose"
case .leftEye: name = "leftEye"
case .rightEye: name = "rightEye"
case .leftEar: name = "leftEar"
case .rightEar: name = "rightEar"
case .leftShoulder: name = "leftShoulder"
case .rightShoulder: name = "rightShoulder"
case .leftElbow: name = "leftElbow"
case .rightElbow: name = "rightElbow"
case .leftWrist: name = "leftWrist"
case .rightWrist: name = "rightWrist"
case .leftHip: name = "leftHip"
case .rightHip: name = "rightHip"
case .leftKnee: name = "leftKnee"
case .rightKnee: name = "rightKnee"
case .leftAnkle: name = "leftAnkle"
case .rightAnkle: name = "rightAnkle"
default: name = "unknown"
}
keypoints.append(KeypointResult(
name: name,
x: px, y: py,
confidence: point.confidence
))
}
}
if !keypoints.isEmpty {
let pad = 20
let bbox = PoseResult.BBoxResult(
x: Int(max(0, minX - Double(pad))),
y: Int(max(0, minY - Double(pad))),
w: Int(maxX - minX + Double(pad * 2)),
h: Int(maxY - minY + Double(pad * 2))
)
poses.append(PoseResult(bbox: bbox, keypoints: keypoints))
}
}
}
} catch {
// Silent fail for individual frames
}
return (faces, poses)
}
@@ -0,0 +1,234 @@
#!/opt/homebrew/bin/swift
/**
* Face-to-Pose Matcher
*
* 用 face.json 的 face bbox 去找 pose.json 對應的 pose
* 匹配規則:face center ↔ pose nose 距離 < 50px
*
* Usage: swift match_face_pose.swift --face <face.json> --pose <pose.json> --output <output.json>
*/
import Foundation
// MARK: - Data Models
struct FaceData: Codable {
let frame: Int
let faces: [FaceEntry]
struct FaceEntry: Codable {
let x: Int, y: Int, width: Int, height: Int
let confidence: Float
let pose_angle: PoseAngle?
struct PoseAngle: Codable {
let yaw: Float, pitch: Float, roll: Float
}
}
}
struct PoseData: Codable {
let frames: [PoseFrame]
struct PoseFrame: Codable {
let frame: Int?
let timestamp: Double?
let persons: [PersonEntry]
struct PersonEntry: Codable {
let bbox: BBox?
let keypoints: [Keypoint]
struct BBox: Codable {
let x: Int, y: Int, width: Int, height: Int
}
struct Keypoint: Codable {
let name: String
let x: Double, y: Double
let confidence: Float
}
}
}
}
struct MatchedOutput: Codable {
let videoFrames: [MatchedFrame]
let stats: MatchStats
struct MatchedFrame: Codable {
let frame: Int
let faceCount: Int
let poseCount: Int
let matches: [Match]
struct Match: Codable {
let faceIdx: Int
let poseIdx: Int?
let distance: Double
let matched: Bool
}
}
struct MatchStats: Codable {
let totalFrames: Int
let framesWithFace: Int
let framesWithPose: Int
let matchedPairs: Int
let unmatchedFaces: Int
let avgDistance: Double
let medianDistance: Double
let under50: Int, under100: Int
}
}
// MARK: - Matching Logic
let threshold = 50.0 // px
func matchFacesToPoses(faceData: FaceData, poseFrames: [PoseData.PoseFrame]) -> MatchedOutput {
// Build frame -> poses lookup
var poseByFrame: [Int: [PoseData.PoseFrame.PersonEntry]] = [:]
for pf in poseFrames {
if let fn = pf.frame {
poseByFrame[fn] = pf.persons
}
}
var matchedFrames: [MatchedOutput.MatchedFrame] = []
var allDistances: [Double] = []
var totalMatched = 0
var totalUnmatched = 0
for faceFrame in faceData.faces {
let fn = faceFrame.frame
let poses = poseByFrame[fn] ?? []
var matches: [MatchedOutput.MatchedFrame.Match] = []
var usedPoseIndices = Set<Int>()
for (faceIdx, face) in faceFrame.faces.enumerated() {
let fcx = Double(face.x + face.width / 2)
let fcy = Double(face.y + face.height / 2)
var bestDist = Double.infinity
var bestPoseIdx: Int? = nil
for (poseIdx, person) in poses.enumerated() {
if usedPoseIndices.contains(poseIdx) { continue }
// Find nose keypoint
let nose = person.keypoints.first(where: { $0.name.lowercased().contains("nose") })
guard let nose = nose else { continue }
let dist = abs(fcx - nose.x) + abs(fcy - nose.y)
if dist < bestDist {
bestDist = dist
bestPoseIdx = poseIdx
}
}
let isMatched = bestPoseIdx != nil && bestDist < threshold
if isMatched {
totalMatched += 1
usedPoseIndices.insert(bestPoseIdx!)
allDistances.append(bestDist)
} else {
totalUnmatched += 1
}
matches.append(MatchedOutput.MatchedFrame.Match(
faceIdx: faceIdx,
poseIdx: bestPoseIdx,
distance: bestDist,
matched: isMatched
))
}
matchedFrames.append(MatchedOutput.MatchedFrame(
frame: fn,
faceCount: faceFrame.faces.count,
poseCount: poses.count,
matches: matches
))
}
let sortedDists = allDistances.sorted()
let avgDist = sortedDists.isEmpty ? 0 : sortedDists.reduce(0, +) / Double(sortedDists.count)
let medianDist = sortedDists.isEmpty ? 0 : sortedDists[sortedDists.count / 2]
let stats = MatchedOutput.MatchStats(
totalFrames: faceData.faces.count,
framesWithFace: faceData.faces.filter { !$0.faces.isEmpty }.count,
framesWithPose: faceData.faces.filter { (poseByFrame[$0.frame] ?? []).count > 0 }.count,
matchedPairs: totalMatched,
unmatchedFaces: totalUnmatched,
avgDistance: avgDist,
medianDistance: medianDist,
under50: allDistances.filter { $0 < 50 }.count,
under100: allDistances.filter { $0 < 100 }.count
)
return MatchedOutput(videoFrames: matchedFrames, stats: stats)
}
// MARK: - Main
func run(facePath: String, posePath: String, outputPath: String) {
print("[FacePoseMatcher] Loading face.json: \(facePath)")
let faceData = try! JSONDecoder().decode(FaceData.self, from: Data(contentsOf: URL(fileURLWithPath: facePath)))
print("[FacePoseMatcher] Loading pose.json: \(posePath)")
let poseData = try! JSONDecoder().decode(PoseData.self, from: Data(contentsOf: URL(fileURLWithPath: posePath)))
print("[FacePoseMatcher] Matching with threshold: \(threshold)px...")
let result = matchFacesToPoses(faceData: faceData, poseFrames: poseData.frames)
// Save output
let encoder = JSONEncoder()
encoder.outputFormatting = [.prettyPrinted, .sortedKeys]
let jsonData = try! encoder.encode(result)
try! jsonData.write(to: URL(fileURLWithPath: outputPath))
// Print summary
let s = result.stats
print("\n=== Face-Pose Matching Report ===")
print("Total frames: \(s.totalFrames)")
print("Frames with face: \(s.framesWithFace)")
print("Frames with pose: \(s.framesWithPose)")
print()
print("Matching (threshold: \(threshold)px):")
print(" Matched pairs: \(s.matchedPairs)")
print(" Unmatched faces: \(s.unmatchedFaces)")
print(" Match rate: \(s.totalFrames > 0 ? String(format: "%.1f%%", Double(s.matchedPairs)/Double(s.framesWithFace)*100) : "N/A")")
print()
print("Distance (face center ↔ pose nose):")
print(" Average: \(String(format: "%.1f", s.avgDistance))px")
print(" Median: \(String(format: "%.1f", s.medianDistance))px")
print(" < 50px: \(s.under50) (\(s.matchedPairs > 0 ? String(format: "%.1f%%", Double(s.under50)/Double(s.matchedPairs)*100) : "N/A"))")
print(" < 100px: \(s.under100) (\(s.matchedPairs > 0 ? String(format: "%.1f%%", Double(s.under100)/Double(s.matchedPairs)*100) : "N/A"))")
print("\nOutput saved to: \(outputPath)")
}
// Parse arguments
var facePath: String?
var posePath: String?
var outputPath: String?
let args = CommandLine.arguments
var i = 1
while i < args.count {
switch args[i] {
case "--face": facePath = args[i+1]; i += 2
case "--pose": posePath = args[i+1]; i += 2
case "--output": outputPath = args[i+1]; i += 2
default: i += 1
}
}
guard let facePath = facePath, let posePath = posePath, let outputPath = outputPath else {
print("Usage: swift match_face_pose.swift --face <face.json> --pose <pose.json> --output <output.json>")
exit(1)
}
run(facePath: facePath, posePath: posePath, outputPath: outputPath)
BIN
View File
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,214 @@
#!/opt/homebrew/bin/swift
/**
* Frame Stability Test
*
* 分開跑 face 和 pose 各 100 次,驗證結果是否穩定
*
* Usage: swift stability_test.swift <video_path> [test_frame]
*/
import Foundation
import AVFoundation
import Vision
let args = CommandLine.arguments
guard args.count >= 2 else {
print("Usage: swift stability_test.swift <video_path> [test_frame]")
exit(1)
}
let videoPath = args[1]
let testFrame = args.count > 2 ? Int(args[2]) ?? 50 : 50
print("[StabilityTest] Video: \(videoPath)")
print("[StabilityTest] Testing frame: \(testFrame)")
// Extract target frame
let url = URL(fileURLWithPath: videoPath)
let asset = AVURLAsset(url: url)
let tracks = try await asset.loadTracks(withMediaType: .video)
guard let track = tracks.first else { exit(1) }
let reader = try AVAssetReader(asset: asset)
let outputSettings: [String: Any] = [kCVPixelBufferPixelFormatTypeKey as String: Int(kCVPixelFormatType_32BGRA)]
let trackOutput = AVAssetReaderTrackOutput(track: track, outputSettings: outputSettings)
reader.add(trackOutput)
reader.startReading()
var targetBuffer: CVPixelBuffer?
var frameIdx = 0
while let sb = trackOutput.copyNextSampleBuffer() {
if frameIdx == testFrame {
targetBuffer = CMSampleBufferGetImageBuffer(sb)
break
}
frameIdx += 1
}
reader.cancelReading()
guard let imageBuffer = targetBuffer else {
print("[ERROR] Frame \(testFrame) not found")
exit(1)
}
let width = CVPixelBufferGetWidth(imageBuffer)
let height = CVPixelBufferGetHeight(imageBuffer)
print("[StabilityTest] Frame size: \(width)x\(height)")
// Run face detection N times
func runFaceDetection() -> [FaceResult] {
let handler = VNImageRequestHandler(cvPixelBuffer: imageBuffer, options: [:])
let request = VNDetectFaceRectanglesRequest()
var results: [FaceResult] = []
do {
try handler.perform([request])
if let observations = request.results {
for obs in observations {
let rect = obs.boundingBox
results.append(FaceResult(
x: Int(rect.origin.x * Double(width)),
y: Int((1 - rect.origin.y - rect.height) * Double(height)),
w: Int(rect.width * Double(width)),
h: Int(rect.height * Double(height)),
confidence: obs.confidence
))
}
}
} catch { print(" Face error: \(error)") }
return results
}
// Run pose detection N times
func runPoseDetection() -> [PoseResult] {
let handler = VNImageRequestHandler(cvPixelBuffer: imageBuffer, options: [:])
let request = VNDetectHumanBodyPoseRequest()
var results: [PoseResult] = []
do {
try handler.perform([request])
if let observations = request.results as? [VNHumanBodyPoseObservation] {
for obs in observations {
var minX = Double.infinity, minY = Double.infinity
var maxX = -Double.infinity, maxY = -Double.infinity
var noseX: Double = 0, noseY: Double = 0
var hasNose = false
let joints: [VNHumanBodyPoseObservation.JointName] = [.nose, .leftEye, .rightEye, .leftShoulder, .rightShoulder]
for jn in joints {
if let pt = try? obs.recognizedPoint(jn), pt.confidence > 0.3 {
let px = pt.location.x * Double(width)
let py = (1 - pt.location.y) * Double(height)
minX = min(minX, px); minY = min(minY, py)
maxX = max(maxX, px); maxY = max(maxY, py)
if jn == .nose { noseX = px; noseY = py; hasNose = true }
}
}
if hasNose {
let pad = 20
results.append(PoseResult(
x: Int(max(0, minX - Double(pad))),
y: Int(max(0, minY - Double(pad))),
w: Int(maxX - minX + Double(pad * 2)),
h: Int(maxY - minY + Double(pad * 2)),
noseX: noseX, noseY: noseY
))
}
}
}
} catch { print(" Pose error: \(error)") }
return results
}
struct FaceResult { let x: Int, y: Int, w: Int, h: Int, confidence: Float }
struct PoseResult { let x: Int, y: Int, w: Int, h: Int, noseX: Double, noseY: Double }
// Run 100 times each
let runs = 100
print("\n[StabilityTest] Running \(runs) iterations each...\n")
var faceResults: [[FaceResult]] = []
var poseResults: [[PoseResult]] = []
for i in 0..<runs {
if i % 20 == 0 { print(" Run \(i)/\(runs)...") }
faceResults.append(runFaceDetection())
poseResults.append(runPoseDetection())
}
// Analyze face results
print("\n=== Face Detection Stability (\(runs) runs) ===")
let faceCounts = faceResults.map { $0.count }
let uniqueFaceCounts = Set(faceCounts)
print(" Face counts: \(uniqueFaceCounts.sorted())")
print(" Consistent: \(uniqueFaceCounts.count == 1 ? "YES" : "NO")")
// Check bbox stability
if let firstFace = faceResults.first?.first {
var xVariance: [Int] = [], yVariance: [Int] = [], wVariance: [Int] = [], hVariance: [Int] = []
for run in faceResults {
if let f = run.first {
xVariance.append(abs(f.x - firstFace.x))
yVariance.append(abs(f.y - firstFace.y))
wVariance.append(abs(f.w - firstFace.w))
hVariance.append(abs(f.h - firstFace.h))
}
}
print(" Bbox variance (max diff from first run):")
print(" x: \(xVariance.max() ?? 0), y: \(yVariance.max() ?? 0)")
print(" w: \(wVariance.max() ?? 0), h: \(hVariance.max() ?? 0)")
let allZero = xVariance.allSatisfy { $0 == 0 } && yVariance.allSatisfy { $0 == 0 } && wVariance.allSatisfy { $0 == 0 } && hVariance.allSatisfy { $0 == 0 }
print(" Perfectly stable: \(allZero ? "YES" : "NO")")
}
// Analyze pose results
print("\n=== Pose Detection Stability (\(runs) runs) ===")
let poseCounts = poseResults.map { $0.count }
let uniquePoseCounts = Set(poseCounts)
print(" Pose counts: \(uniquePoseCounts.sorted())")
print(" Consistent: \(uniquePoseCounts.count == 1 ? "YES" : "NO")")
if let firstPose = poseResults.first?.first {
var xV: [Int] = [], yV: [Int] = [], wV: [Int] = [], hV: [Int] = []
var noseXV: [Double] = [], noseYV: [Double] = []
for run in poseResults {
if let p = run.first {
xV.append(abs(p.x - firstPose.x))
yV.append(abs(p.y - firstPose.y))
wV.append(abs(p.w - firstPose.w))
hV.append(abs(p.h - firstPose.h))
noseXV.append(abs(p.noseX - firstPose.noseX))
noseYV.append(abs(p.noseY - firstPose.noseY))
}
}
print(" Bbox variance (max diff from first run):")
print(" x: \(xV.max() ?? 0), y: \(yV.max() ?? 0)")
print(" w: \(wV.max() ?? 0), h: \(hV.max() ?? 0)")
print(" Nose variance:")
print(" x: \(String(format: "%.1f", noseXV.max() ?? 0)), y: \(String(format: "%.1f", noseYV.max() ?? 0))")
let allZero = xV.allSatisfy { $0 == 0 } && yV.allSatisfy { $0 == 0 }
print(" Perfectly stable: \(allZero ? "YES" : "NO")")
}
// Face-Pose distance stability
print("\n=== Face-Pose Distance Stability ===")
var distances: [Double] = []
for i in 0..<runs {
if let face = faceResults[i].first, let pose = poseResults[i].first {
let fcx = Double(face.x + face.w / 2)
let fcy = Double(face.y + face.h / 2)
let dist = abs(fcx - pose.noseX) + abs(fcy - pose.noseY)
distances.append(dist)
}
}
if !distances.isEmpty {
let avg = distances.reduce(0, +) / Double(distances.count)
let maxDist = distances.max() ?? 0
let minDist = distances.min() ?? 0
let variance = maxDist - minDist
print(" Avg distance: \(String(format: "%.1f", avg))px")
print(" Min: \(String(format: "%.1f", minDist))px, Max: \(String(format: "%.1f", maxDist))px")
print(" Variance: \(String(format: "%.1f", variance))px")
print(" Stable (variance < 5px): \(variance < 5 ? "YES" : "NO")")
}
print("\n[StabilityTest] Done")