fix: face group name read consistency, sync_file_status fix, cleanup ghost records, identity_agent replaced with face_dedup

- get_face_groups_handler: COALESCE(tp.name, tn.label) for name consistency
- sync_file_status: compare JSON vs pre_chunks (not chunk table)
- face consistency: compare frames.len() not total_faces
- cleanup 2 ghost records with NULL file_name/file_path
- replace identity_agent with face_dedup in pipeline stages
- remove identity_agent_api.rs and all references
- update required_processors to match actual processors
- update AGENTS.md with team responsibilities
- add Studio pipeline changes documentation
This commit is contained in:
Accusys
2026-07-27 02:15:51 +08:00
parent fcdeab82e6
commit 39a2cbc65b
118 changed files with 19386 additions and 2964 deletions
+16
View File
@@ -126,5 +126,21 @@ let package = Package(
path: ".",
sources: ["swift_face_pose.swift"]
),
.executableTarget(
name: "swift_pose_expansion",
dependencies: [
.product(name: "ArgumentParser", package: "swift-argument-parser"),
],
path: ".",
sources: ["swift_pose_expansion.swift"]
),
.executableTarget(
name: "swift_appearance_expansion",
dependencies: [
.product(name: "ArgumentParser", package: "swift-argument-parser"),
],
path: ".",
sources: ["swift_appearance_expansion.swift"]
),
]
)
@@ -0,0 +1,332 @@
import Foundation
import Vision
import ArgumentParser
import AVFoundation
/// Swift Appearance Expansion Processor V2
///
/// Reads pose.json and extracts colors at keypoint positions.
///
/// Algorithm:
/// 1. Load pose.json, get frames with trace_id
/// 2. For each pose frame, extract colors at keypoint positions
/// 3. Record overall brightness for lighting adjustment
/// 4. Expand outward, stop when 3 consecutive frames have low similarity
/// 5. Output at 8Hz sampling
///
/// Key Concepts:
/// - Appearance = colors at body part positions (head, torso, legs, feet)
/// - Used for tracking and agent search ("person wearing red shirt")
/// - Approximate colors are sufficient for top-K search
@main
struct SwiftAppearanceExpansion: ParsableCommand {
@Argument(help: "Video file path")
var videoPath: String
@Argument(help: "Input pose.json path")
var posePath: String
@Argument(help: "Output appearance.json path")
var outputPath: String
@Option(name: .long, help: "UUID for logging")
var uuid: String = ""
@Option(name: .long, help: "Consecutive miss threshold (default: 3)")
var missThreshold: Int = 3
@Option(name: .long, help: "Color sampling radius (default: 15)")
var colorRadius: Int = 15
mutating func run() throws {
let startTime = Date()
print("[AppearanceExpansion] Starting appearance extraction from pose: \(videoPath)")
// Load pose.json
guard let poseData = try? Data(contentsOf: URL(fileURLWithPath: posePath)) else {
print("[AppearanceExpansion] ERROR: Cannot read \(posePath)")
return
}
guard let poseJson = try? JSONSerialization.jsonObject(with: poseData) as? [String: Any] else {
print("[AppearanceExpansion] ERROR: Invalid JSON in \(posePath)")
return
}
// Extract frames with trace_id from pose
var poseFrameDict: [Int: [String: Any]] = [:] // frame -> pose data with trace_id
if let frames = poseJson["frames"] as? [[String: Any]] {
for frameData in frames {
guard let frameNum = frameData["frame"] as? Int else { continue }
poseFrameDict[frameNum] = frameData
}
}
print("[AppearanceExpansion] Found \(poseFrameDict.count) pose frames in \(posePath)")
if poseFrameDict.isEmpty {
print("[AppearanceExpansion] No pose frames found, skipping")
let emptyOutput: [String: Any] = ["frame_count": 0, "fps": 0.0, "frames": []]
let jsonData = try JSONSerialization.data(withJSONObject: emptyOutput, options: [])
try jsonData.write(to: URL(fileURLWithPath: outputPath))
return
}
// Get video info
let url = URL(fileURLWithPath: videoPath)
let asset = AVAsset(url: url)
guard let videoTrack = asset.tracks(withMediaType: .video).first else {
print("[AppearanceExpansion] ERROR: No video track")
return
}
let fps = videoTrack.nominalFrameRate
let duration = CMTimeGetSeconds(asset.duration)
let totalFrames = Int(duration * Double(fps))
let sampleInterval = max(1, Int(floor(Double(fps) / 8.0)))
print("[AppearanceExpansion] Video: \(fps)fps, \(totalFrames) frames, 8Hz interval=\(sampleInterval)")
// Track appearance frames
var appearanceFrameDict: [Int: [String: Any]] = [:]
// Setup asset reader
let outputSettings: [String: Any] = [
kCVPixelBufferPixelFormatTypeKey as String: kCVPixelFormatType_32BGRA
]
let reader = try AVAssetReader(asset: asset)
let trackOutput = AVAssetReaderTrackOutput(track: videoTrack, outputSettings: outputSettings)
trackOutput.alwaysCopiesSampleData = false
reader.add(trackOutput)
guard reader.startReading() else {
print("[AppearanceExpansion] ERROR: Cannot start reader")
return
}
// Process frames
var frameIndex = 0
while let sampleBuffer = trackOutput.copyNextSampleBuffer() {
defer { frameIndex += 1 }
guard let pixelBuffer = CMSampleBufferGetImageBuffer(sampleBuffer) else {
continue
}
// Check if this is a pose frame
if let poseData = poseFrameDict[frameIndex] {
let seconds = Double(frameIndex) / Double(fps)
// Extract colors at keypoint positions
let colors = extractColorsAtKeypoints(pixelBuffer: pixelBuffer, poseData: poseData, radius: colorRadius)
// Calculate overall brightness
let brightness = calculateBrightness(pixelBuffer: pixelBuffer)
// Get trace_id from pose (inherit)
let traceId = poseData["trace_id"] as? Int ?? 0
appearanceFrameDict[frameIndex] = [
"frame": frameIndex,
"timestamp": seconds,
"trace_id": traceId,
"brightness": brightness,
"colors": colors
]
}
// Progress logging
if frameIndex % 5000 == 0 {
let elapsed = Date().timeIntervalSince(startTime)
print("[AppearanceExpansion] Frame \(frameIndex)/\(totalFrames), \(appearanceFrameDict.count) appearances, \(Int(elapsed))s")
fflush(stdout)
}
}
reader.cancelReading()
print("[AppearanceExpansion] Extraction done: \(appearanceFrameDict.count) frames with appearance")
// 8Hz sampling output
var outputFrames: [[String: Any]] = []
let sortedAppearanceFrames = appearanceFrameDict.keys.sorted()
var targetFrame = 0
while targetFrame < totalFrames {
// Find closest appearance frame to target
var closestFrame: Int? = nil
var closestDist = Int.max
for appFrame in sortedAppearanceFrames {
let dist = abs(appFrame - targetFrame)
if dist < closestDist && dist <= sampleInterval {
closestDist = dist
closestFrame = appFrame
}
}
if let cf = closestFrame, let data = appearanceFrameDict[cf] {
outputFrames.append(data)
}
targetFrame += sampleInterval
}
// Write output
let output: [String: Any] = [
"frame_count": outputFrames.count,
"fps": Double(fps),
"frames": outputFrames
]
let jsonData = try JSONSerialization.data(withJSONObject: output, options: [])
try jsonData.write(to: URL(fileURLWithPath: outputPath))
let elapsed = Date().timeIntervalSince(startTime)
print("[AppearanceExpansion] Done: \(outputFrames.count) frames at 8Hz, \(String(format: "%.1f", elapsed))s → \(outputPath)")
}
/// Extract colors at keypoint positions
func extractColorsAtKeypoints(pixelBuffer: CVPixelBuffer, poseData: [String: Any], radius: Int) -> [String: [Int]] {
let imgW = CVPixelBufferGetWidth(pixelBuffer)
let imgH = CVPixelBufferGetHeight(pixelBuffer)
CVPixelBufferLockBaseAddress(pixelBuffer, .readOnly)
defer { CVPixelBufferUnlockBaseAddress(pixelBuffer, .readOnly) }
guard let baseAddress = CVPixelBufferGetBaseAddress(pixelBuffer) else {
return [:]
}
let bytesPerRow = CVPixelBufferGetBytesPerRow(pixelBuffer)
let buffer = baseAddress.bindMemory(to: UInt8.self, capacity: bytesPerRow * imgH)
var colors: [String: [Int]] = [:]
// Define keypoint groups for body parts
let bodyParts: [String: [String]] = [
"head": ["nose", "left_eye", "right_eye", "left_ear", "right_ear"],
"torso": ["left_shoulder", "right_shoulder"],
"legs": ["left_hip", "right_hip", "left_knee", "right_knee"],
"feet": ["left_ankle", "right_ankle"]
]
// Extract color for each body part
for (partName, keypointNames) in bodyParts {
var totalR = 0, totalG = 0, totalB = 0
var count = 0
// Get persons array from pose data
if let persons = poseData["persons"] as? [[String: Any]] {
for person in persons {
if let keypoints = person["keypoints"] as? [[String: Any]] {
for kp in keypoints {
guard let name = kp["name"] as? String,
keypointNames.contains(name),
let x = kp["x"] as? Double,
let y = kp["y"] as? Double,
let confidence = kp["confidence"] as? Double,
confidence > 0.3 else { continue }
// Get average color around keypoint
let color = getAverageColor(
buffer: buffer,
bytesPerRow: bytesPerRow,
imgW: imgW,
imgH: imgH,
centerX: Int(x),
centerY: Int(y),
radius: radius
)
totalR += color.0
totalG += color.1
totalB += color.2
count += 1
}
}
}
}
if count > 0 {
colors[partName] = [totalR / count, totalG / count, totalB / count]
}
}
return colors
}
/// Get average color around a position
func getAverageColor(
buffer: UnsafeMutablePointer<UInt8>,
bytesPerRow: Int,
imgW: Int,
imgH: Int,
centerX: Int,
centerY: Int,
radius: Int
) -> (Int, Int, Int) {
let x1 = max(0, centerX - radius)
let x2 = min(imgW - 1, centerX + radius)
let y1 = max(0, centerY - radius)
let y2 = min(imgH - 1, centerY + radius)
var totalR = 0, totalG = 0, totalB = 0, count = 0
for y in y1...y2 {
let rowStart = y * bytesPerRow
for x in x1...x2 {
let offset = rowStart + x * 4
totalB += Int(buffer[offset])
totalG += Int(buffer[offset + 1])
totalR += Int(buffer[offset + 2])
count += 1
}
}
if count > 0 {
return (totalR / count, totalG / count, totalB / count)
}
return (0, 0, 0)
}
/// Calculate overall brightness of frame
func calculateBrightness(pixelBuffer: CVPixelBuffer) -> Double {
let imgW = CVPixelBufferGetWidth(pixelBuffer)
let imgH = CVPixelBufferGetHeight(pixelBuffer)
CVPixelBufferLockBaseAddress(pixelBuffer, .readOnly)
defer { CVPixelBufferUnlockBaseAddress(pixelBuffer, .readOnly) }
guard let baseAddress = CVPixelBufferGetBaseAddress(pixelBuffer) else {
return 0.0
}
let bytesPerRow = CVPixelBufferGetBytesPerRow(pixelBuffer)
let buffer = baseAddress.bindMemory(to: UInt8.self, capacity: bytesPerRow * imgH)
var totalBrightness = 0.0
var count = 0
// Sample every 10 pixels for speed
for y in stride(from: 0, to: imgH, by: 10) {
let rowStart = y * bytesPerRow
for x in stride(from: 0, to: imgW, by: 10) {
let offset = rowStart + x * 4
let b = Double(buffer[offset])
let g = Double(buffer[offset + 1])
let r = Double(buffer[offset + 2])
// Calculate luminance
let luminance = 0.299 * r + 0.587 * g + 0.114 * b
totalBrightness += luminance / 255.0
count += 1
}
}
return count > 0 ? totalBrightness / Double(count) : 0.0
}
}
@@ -0,0 +1,376 @@
import Foundation
import Vision
import ArgumentParser
import AVFoundation
/// Swift Pose Expansion Processor V2
///
/// Reads face_traced.json (from face tracking) and expands pose detection from trace frames.
/// Inherits trace_id from face traces for proper tracking continuity.
///
/// Algorithm:
/// 1. Load face_traced.json, extract frames grouped by trace_id
/// 2. For each trace_id, start from face frames and expand forward/backward
/// 3. Stop expansion when 3 consecutive frames have no pose detection
/// 4. Associate each pose with the nearest face's trace_id
/// 5. Output pose.json at 8Hz sampling (floor(fps/8) interval)
@main
struct SwiftPoseExpansion: ParsableCommand {
@Argument(help: "Video file path")
var videoPath: String
@Argument(help: "Input face_traced.json path")
var faceTracedPath: String
@Argument(help: "Output pose.json path")
var outputPath: String
@Option(name: .long, help: "UUID for logging")
var uuid: String = ""
@Option(name: .long, help: "Consecutive miss threshold to stop expansion (default: 3)")
var missThreshold: Int = 3
mutating func run() throws {
let startTime = Date()
print("[PoseExpansion] Starting pose expansion from face traces: \(videoPath)")
// Load face_traced.json
guard let faceData = try? Data(contentsOf: URL(fileURLWithPath: faceTracedPath)) else {
print("[PoseExpansion] ERROR: Cannot read \(faceTracedPath)")
return
}
guard let faceJson = try? JSONSerialization.jsonObject(with: faceData) as? [String: Any] else {
print("[PoseExpansion] ERROR: Invalid JSON in \(faceTracedPath)")
return
}
// Extract frames with trace_id mapping
// frameToTraces: frame -> [(trace_id, x, y, w, h)]
var frameToTraces: [Int: [(traceId: Int, x: Double, y: Double, w: Double, h: Double)]] = [:]
var allTraceIds: Set<Int> = []
// Handle both dict and list format
if let framesDict = faceJson["frames"] as? [String: Any] {
for (frameStr, frameData) in framesDict {
guard let frameNum = Int(frameStr) else { continue }
if let faces = (frameData as? [String: Any])?["faces"] as? [[String: Any]] {
for face in faces {
if let traceId = face["trace_id"] as? Int, traceId > 0 {
allTraceIds.insert(traceId)
let bbox = face["bbox"] as? [String: Any]
let x = bbox?["x"] as? Double ?? face["x"] as? Double ?? 0
let y = bbox?["y"] as? Double ?? face["y"] as? Double ?? 0
let w = bbox?["width"] as? Double ?? face["width"] as? Double ?? 0
let h = bbox?["height"] as? Double ?? face["height"] as? Double ?? 0
frameToTraces[frameNum, default: []].append((traceId, x, y, w, h))
}
}
}
}
} else if let framesList = faceJson["frames"] as? [[String: Any]] {
for frameData in framesList {
guard let frameNum = frameData["frame"] as? Int else { continue }
if let faces = frameData["faces"] as? [[String: Any]] {
for face in faces {
if let traceId = face["trace_id"] as? Int, traceId > 0 {
allTraceIds.insert(traceId)
let bbox = face["bbox"] as? [String: Any]
let x = bbox?["x"] as? Double ?? face["x"] as? Double ?? 0
let y = bbox?["y"] as? Double ?? face["y"] as? Double ?? 0
let w = bbox?["width"] as? Double ?? face["width"] as? Double ?? 0
let h = bbox?["height"] as? Double ?? face["height"] as? Double ?? 0
frameToTraces[frameNum, default: []].append((traceId, x, y, w, h))
}
}
}
}
}
print("[PoseExpansion] Found \(allTraceIds.count) traces, \(frameToTraces.count) frames in \(faceTracedPath)")
if frameToTraces.isEmpty {
print("[PoseExpansion] No traces found, skipping pose expansion")
let emptyOutput: [String: Any] = ["frame_count": 0, "fps": 0.0, "frames": []]
let jsonData = try JSONSerialization.data(withJSONObject: emptyOutput, options: [])
try jsonData.write(to: URL(fileURLWithPath: outputPath))
return
}
// Get video info
let url = URL(fileURLWithPath: videoPath)
let asset = AVAsset(url: url)
guard let videoTrack = asset.tracks(withMediaType: .video).first else {
print("[PoseExpansion] ERROR: No video track")
return
}
let fps = videoTrack.nominalFrameRate
let duration = CMTimeGetSeconds(asset.duration)
let totalFrames = Int(duration * Double(fps))
let sampleInterval = max(1, Int(floor(Double(fps) / 8.0)))
print("[PoseExpansion] Video: \(fps)fps, \(totalFrames) frames, 8Hz interval=\(sampleInterval)")
// Build set of all face frames
let allFaceFrameSet = Set(frameToTraces.keys)
// Track which frames have pose with trace_id
var poseFrameDict: [Int: [String: Any]] = [:]
// Setup asset reader
let outputSettings: [String: Any] = [
kCVPixelBufferPixelFormatTypeKey as String: kCVPixelFormatType_32BGRA
]
let reader = try AVAssetReader(asset: asset)
let trackOutput = AVAssetReaderTrackOutput(track: videoTrack, outputSettings: outputSettings)
trackOutput.alwaysCopiesSampleData = false
reader.add(trackOutput)
guard reader.startReading() else {
print("[PoseExpansion] ERROR: Cannot start reader")
return
}
// Process frames
var frameIndex = 0
var consecutiveMisses = 0
var activeTraceIds: Set<Int> = [] // Currently active traces being expanded
while let sampleBuffer = trackOutput.copyNextSampleBuffer() {
defer { frameIndex += 1 }
guard let pixelBuffer = CMSampleBufferGetImageBuffer(sampleBuffer) else {
continue
}
// Check if this frame has face traces
let faceTraces = frameToTraces[frameIndex]
let isFaceFrame = faceTraces != nil && !faceTraces!.isEmpty
if isFaceFrame, let traces = faceTraces {
// Update active traces
for ft in traces {
activeTraceIds.insert(ft.traceId)
}
consecutiveMisses = 0
}
// Check if we should process this frame
let shouldProcess = isFaceFrame ||
(consecutiveMisses < missThreshold * sampleInterval && !activeTraceIds.isEmpty)
if shouldProcess {
let poseResult = detectPose(pixelBuffer: pixelBuffer)
if poseResult.hasPose {
let seconds = Double(frameIndex) / Double(fps)
// Determine trace_id for this pose
var traceId = 0
if isFaceFrame, let traces = faceTraces {
// Use the trace_id from face (may need bbox matching for multi-person)
// For now, use the first trace_id found
traceId = traces.first?.traceId ?? 0
} else {
// Inherit from nearest face frame with active trace
let nearestFaceFrame = findNearestFaceFrame(
frameIndex: frameIndex,
frameToTraces: frameToTraces,
activeTraceIds: activeTraceIds
)
if let nearest = nearestFaceFrame, let traces = frameToTraces[nearest] {
traceId = traces.first?.traceId ?? 0
}
}
poseFrameDict[frameIndex] = [
"frame": frameIndex,
"timestamp": seconds,
"trace_id": traceId,
"persons": poseResult.persons
]
consecutiveMisses = 0
} else {
consecutiveMisses += 1
}
}
// Progress logging
if frameIndex % 5000 == 0 {
let elapsed = Date().timeIntervalSince(startTime)
print("[PoseExpansion] Frame \(frameIndex)/\(totalFrames), \(poseFrameDict.count) poses, \(Int(elapsed))s")
fflush(stdout)
}
}
reader.cancelReading()
print("[PoseExpansion] Detection done: \(poseFrameDict.count) frames with pose")
// 8Hz sampling output
var outputFrames: [[String: Any]] = []
let sortedPoseFrames = poseFrameDict.keys.sorted()
var targetFrame = 0
while targetFrame < totalFrames {
// Find closest pose frame to target
var closestFrame: Int? = nil
var closestDist = Int.max
for poseFrame in sortedPoseFrames {
let dist = abs(poseFrame - targetFrame)
if dist < closestDist && dist <= sampleInterval {
closestDist = dist
closestFrame = poseFrame
}
}
if let cf = closestFrame, let data = poseFrameDict[cf] {
outputFrames.append(data)
}
targetFrame += sampleInterval
}
// Write output
let output: [String: Any] = [
"frame_count": outputFrames.count,
"fps": Double(fps),
"frames": outputFrames
]
let jsonData = try JSONSerialization.data(withJSONObject: output, options: [])
try jsonData.write(to: URL(fileURLWithPath: outputPath))
let elapsed = Date().timeIntervalSince(startTime)
print("[PoseExpansion] Done: \(outputFrames.count) frames at 8Hz, \(String(format: "%.1f", elapsed))s → \(outputPath)")
}
/// Find nearest face frame with active trace
func findNearestFaceFrame(
frameIndex: Int,
frameToTraces: [Int: [(traceId: Int, x: Double, y: Double, w: Double, h: Double)]],
activeTraceIds: Set<Int>
) -> Int? {
var nearestFrame: Int? = nil
var nearestDist = Int.max
for (frameNum, traces) in frameToTraces {
// Check if this frame has an active trace
let hasActiveTrace = traces.contains { activeTraceIds.contains($0.traceId) }
if hasActiveTrace {
let dist = abs(frameNum - frameIndex)
if dist < nearestDist {
nearestDist = dist
nearestFrame = frameNum
}
}
}
return nearestFrame
}
func detectPose(pixelBuffer: CVPixelBuffer) -> (hasPose: Bool, persons: [[String: Any]]) {
let imgW = CGFloat(CVPixelBufferGetWidth(pixelBuffer))
let imgH = CGFloat(CVPixelBufferGetHeight(pixelBuffer))
let handler = VNImageRequestHandler(cvPixelBuffer: pixelBuffer, options: [:])
let bodyReq = VNDetectHumanBodyPoseRequest()
do {
try handler.perform([bodyReq])
} catch {
return (false, [])
}
let jointNames: [VNHumanBodyPoseObservation.JointName] = [
.nose, .leftEye, .rightEye, .leftEar, .rightEar,
.neck, .root,
.leftShoulder, .rightShoulder,
.leftElbow, .rightElbow,
.leftWrist, .rightWrist,
.leftHip, .rightHip,
.leftKnee, .rightKnee,
.leftAnkle, .rightAnkle,
]
var persons: [[String: Any]] = []
let poses = bodyReq.results ?? []
for pose in poses {
var keypoints: [[String: Any]] = []
var minX = CGFloat.greatestFiniteMagnitude
var minY = CGFloat.greatestFiniteMagnitude
var maxX: CGFloat = 0
var maxY: CGFloat = 0
for joint in jointNames {
if let point = try? pose.recognizedPoint(joint) {
let desc = String(describing: joint.rawValue)
var rawName = desc
.replacingOccurrences(of: "VNRecognizedPointKey(_rawValue: ", with: "")
.replacingOccurrences(of: ")", with: "")
.trimmingCharacters(in: .whitespaces)
let nameMap: [String: String] = [
"head_joint": "nose",
"left_eye_joint": "left_eye",
"right_eye_joint": "right_eye",
"left_ear_joint": "left_ear",
"right_ear_joint": "right_ear",
"neck_1_joint": "neck",
"left_shoulder_1_joint": "left_shoulder",
"right_shoulder_1_joint": "right_shoulder",
"left_elbow_1_joint": "left_elbow",
"right_elbow_1_joint": "right_elbow",
"left_hand_joint": "left_wrist",
"right_hand_joint": "right_wrist",
"left_hip_1_joint": "left_hip",
"right_hip_1_joint": "right_hip",
"left_knee_1_joint": "left_knee",
"right_knee_1_joint": "right_knee",
"left_ankle_1_joint": "left_ankle",
"right_ankle_1_joint": "right_ankle",
"center_hip_joint": "root",
]
if let mapped = nameMap[rawName] {
rawName = mapped
}
let px = point.location.x * CGFloat(imgW)
let py = CGFloat(imgH) - point.location.y * CGFloat(imgH)
keypoints.append([
"name": rawName.isEmpty ? "\(joint)" : rawName,
"x": px,
"y": py,
"confidence": point.confidence,
])
if point.confidence > 0.1 {
minX = min(minX, px)
minY = min(minY, py)
maxX = max(maxX, px)
maxY = max(maxY, py)
}
}
}
var bbox: [String: Any] = ["x": 0, "y": 0, "width": 0, "height": 0]
if maxX > minX {
bbox = [
"x": Int(minX),
"y": Int(minY),
"width": Int(maxX - minX),
"height": Int(maxY - minY),
]
}
persons.append(["keypoints": keypoints, "bbox": bbox])
}
return (!persons.isEmpty, persons)
}
}