39a2cbc65b
- get_face_groups_handler: COALESCE(tp.name, tn.label) for name consistency - sync_file_status: compare JSON vs pre_chunks (not chunk table) - face consistency: compare frames.len() not total_faces - cleanup 2 ghost records with NULL file_name/file_path - replace identity_agent with face_dedup in pipeline stages - remove identity_agent_api.rs and all references - update required_processors to match actual processors - update AGENTS.md with team responsibilities - add Studio pipeline changes documentation
150 lines
5.4 KiB
Python
150 lines
5.4 KiB
Python
#!/usr/bin/env python3
|
|
"""One-time backfill: generate .profile.json for all registered files."""
|
|
|
|
import json
|
|
import os
|
|
import sys
|
|
import subprocess
|
|
from datetime import datetime, timezone
|
|
|
|
OUTPUT_DIR = os.environ.get("MOMENTRY_OUTPUT_DIR", "/Users/accusys/momentry/output")
|
|
DB_URL = os.environ.get("DATABASE_URL", "postgres://accusys@localhost:5432/momentry")
|
|
PSQL = "/opt/homebrew/Cellar/libpq/18.4/bin/psql"
|
|
|
|
|
|
def query_db(sql):
|
|
result = subprocess.run(
|
|
[PSQL, "-U", "accusys", "-d", "momentry", "-t", "-A", "-c", sql],
|
|
capture_output=True, text=True
|
|
)
|
|
if result.returncode != 0:
|
|
print(f"DB error: {result.stderr}", file=sys.stderr)
|
|
return []
|
|
lines = result.stdout.strip().split("\n")
|
|
return [line for line in lines if line.strip()]
|
|
|
|
|
|
def extract_key_frame(video_path, duration, output_dir, file_uuid):
|
|
"""Extract a representative frame using ffmpeg, save as JPG."""
|
|
seek_time = duration * 0.1 if duration > 0 else 1.0
|
|
out_path = os.path.join(output_dir, f"{file_uuid}.key_frame.jpg")
|
|
try:
|
|
result = subprocess.run(
|
|
[
|
|
"ffmpeg", "-y", "-ss", f"{seek_time:.2f}",
|
|
"-i", video_path,
|
|
"-vframes", "1",
|
|
"-vf", "scale=640:-1",
|
|
"-q:v", "5",
|
|
out_path
|
|
],
|
|
capture_output=True, timeout=30
|
|
)
|
|
if result.returncode == 0 and os.path.exists(out_path):
|
|
return f"{file_uuid}.key_frame.jpg"
|
|
except Exception as e:
|
|
print(f" key_frame extraction failed: {e}", file=sys.stderr)
|
|
return None
|
|
|
|
|
|
def main():
|
|
print(f"Output dir: {OUTPUT_DIR}")
|
|
print(f"Looking for files without .profile.json...")
|
|
|
|
# Get all registered files
|
|
rows = query_db(
|
|
"SELECT file_uuid, COALESCE(file_name, ''), COALESCE(file_path, ''), "
|
|
"COALESCE(file_type, 'unknown'), COALESCE(content_hash, ''), "
|
|
"COALESCE(duration, 0), COALESCE(width, 0), COALESCE(height, 0), "
|
|
"COALESCE(fps, 0), COALESCE(total_frames, 0) "
|
|
"FROM videos ORDER BY created_at"
|
|
)
|
|
|
|
created = 0
|
|
skipped = 0
|
|
for row in rows:
|
|
parts = row.split("|")
|
|
if len(parts) < 10:
|
|
continue
|
|
file_uuid, file_name, file_path, file_type, content_hash, \
|
|
duration, width, height, fps, total_frames = parts[:10]
|
|
|
|
# Compute total_frames from duration * fps if DB value is 0
|
|
db_total_frames = int(total_frames)
|
|
duration_f = float(duration)
|
|
fps_f = float(fps)
|
|
if db_total_frames <= 0 and duration_f > 0 and fps_f > 0:
|
|
computed_frames = int(duration_f * fps_f)
|
|
db_total_frames = computed_frames
|
|
query_db(
|
|
f"UPDATE videos SET total_frames = {db_total_frames} WHERE file_uuid = '{file_uuid}'"
|
|
)
|
|
|
|
profile_path = os.path.join(OUTPUT_DIR, f"{file_uuid}.profile.json")
|
|
if os.path.exists(profile_path):
|
|
# Patch existing profiles with total_frames=0
|
|
with open(profile_path) as pf:
|
|
existing = json.load(pf)
|
|
if existing.get("metadata", {}).get("total_frames") == 0 and duration_f > 0 and fps_f > 0:
|
|
existing["metadata"]["total_frames"] = db_total_frames
|
|
with open(profile_path, "w") as pf:
|
|
json.dump(existing, pf, indent=2, ensure_ascii=False)
|
|
print(f" [patched] {file_uuid}: total_frames 0 → {db_total_frames}")
|
|
skipped += 1
|
|
continue
|
|
|
|
now = datetime.now(timezone.utc).isoformat()
|
|
parent = os.path.dirname(file_path) if file_path else ""
|
|
|
|
profile = {
|
|
"version": "1.0",
|
|
"file_uuid": file_uuid,
|
|
"file_name": file_name,
|
|
"file_type": file_type,
|
|
"birth": {
|
|
"mac_address": "",
|
|
"birthday": now,
|
|
"original_path": parent,
|
|
"original_filename": file_name,
|
|
"canonical_path": file_path,
|
|
"content_hash": content_hash if content_hash else None
|
|
},
|
|
"current": {
|
|
"path": file_path,
|
|
"file_name": file_name,
|
|
"file_type": file_type
|
|
},
|
|
"history": [{
|
|
"action": "backfilled",
|
|
"timestamp": now,
|
|
"path": file_path,
|
|
"file_name": file_name
|
|
}],
|
|
"metadata": {
|
|
"duration": duration_f,
|
|
"width": int(width),
|
|
"height": int(height),
|
|
"fps": fps_f,
|
|
"total_frames": db_total_frames
|
|
} if duration_f > 0 or int(width) > 0 else None,
|
|
"key_frame": None # will be filled below
|
|
}
|
|
|
|
# Extract key_frame for video files that exist on disk
|
|
if file_type == "video" and file_path and os.path.exists(file_path):
|
|
print(f" Extracting key_frame for {file_name}...")
|
|
kf = extract_key_frame(file_path, float(duration), OUTPUT_DIR, file_uuid)
|
|
if kf:
|
|
profile["key_frame"] = kf
|
|
|
|
with open(profile_path, "w") as f:
|
|
json.dump(profile, f, indent=2, ensure_ascii=False)
|
|
created += 1
|
|
print(f" Created: {file_uuid}.profile.json ({file_name or 'ZOMBIE'})")
|
|
|
|
print(f"\nDone: {created} created, {skipped} skipped (already exist)")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|