fix: face group name read consistency, sync_file_status fix, cleanup ghost records, identity_agent replaced with face_dedup

- get_face_groups_handler: COALESCE(tp.name, tn.label) for name consistency
- sync_file_status: compare JSON vs pre_chunks (not chunk table)
- face consistency: compare frames.len() not total_faces
- cleanup 2 ghost records with NULL file_name/file_path
- replace identity_agent with face_dedup in pipeline stages
- remove identity_agent_api.rs and all references
- update required_processors to match actual processors
- update AGENTS.md with team responsibilities
- add Studio pipeline changes documentation
This commit is contained in:
Accusys
2026-07-27 02:15:51 +08:00
parent fcdeab82e6
commit 39a2cbc65b
118 changed files with 19386 additions and 2964 deletions
+286
View File
@@ -0,0 +1,286 @@
#!/opt/homebrew/bin/python3.11
"""
Embedding Model Evaluation - Compare embeddinggemma vs nomic-embed-text-v2-moe
Usage:
python3 scripts/eval_embedding_models.py --file-uuid <uuid> --output-dir /path/to/output
Metrics:
1. Accuracy: Semantic similarity ranking quality
2. Speed: Latency per embedding
3. Dimension: Vector size
"""
import argparse
import json
import math
import os
import sys
import time
from pathlib import Path
try:
import requests
except ImportError:
print("requests not installed: pip install requests", file=sys.stderr)
sys.exit(1)
# Embedding endpoints
EMBED_A_URL = "http://localhost:11436/v1/embeddings" # embeddinggemma
EMBED_B_URL = "http://localhost:11434/api/embed" # nomic-embed-text-v2-moe
EMBED_B_MODEL = "nomic-embed-text-v2-moe"
# Test queries (Chinese, English, Mixed)
TEST_QUERIES = [
{"query": "穿西裝的男人", "lang": "zh"},
{"query": "室內辦公室", "lang": "zh"},
{"query": "雪景", "lang": "zh"},
{"query": "持槍的人", "lang": "zh"},
{"query": "woman in white dress", "lang": "en"},
{"query": "outdoor scene night", "lang": "en"},
{"query": "person holding object", "lang": "en"},
{"query": "穿著 formal 的男人", "lang": "mixed"},
]
def get_embedding_a(text: str) -> tuple:
"""Get embedding from embeddinggemma (port 11436)."""
start = time.time()
try:
resp = requests.post(EMBED_A_URL, json={"input": text}, timeout=30)
resp.raise_for_status()
data = resp.json()
elapsed = time.time() - start
return data["data"][0]["embedding"], elapsed
except Exception as e:
print(f"[eval] embeddinggemma error: {e}", file=sys.stderr)
return [], 0
def get_embedding_b(text: str) -> tuple:
"""Get embedding from nomic-embed-text-v2-moe (port 11434)."""
start = time.time()
try:
resp = requests.post(EMBED_B_URL, json={"model": EMBED_B_MODEL, "input": text}, timeout=30)
resp.raise_for_status()
data = resp.json()
elapsed = time.time() - start
return data["embeddings"][0], elapsed
except Exception as e:
print(f"[eval] nomic error: {e}", file=sys.stderr)
return [], 0
def cosine_similarity(a: list, b: list) -> float:
"""Calculate cosine similarity."""
if not a or not b or len(a) != len(b):
return 0.0
dot = sum(x * y for x, y in zip(a, b))
norm_a = math.sqrt(sum(x * x for x in a))
norm_b = math.sqrt(sum(y * y for y in b))
return dot / (norm_a * norm_b) if norm_a > 0 and norm_b > 0 else 0.0
def load_vlm_descriptions(file_uuid: str, output_dir: str) -> list:
"""Load VLM descriptions from trace/scene/interval profiles."""
descriptions = []
output_path = Path(output_dir)
# Load trace profiles
trace_dir = output_path / file_uuid
if trace_dir.exists():
for trace_path in sorted(trace_dir.glob("trace_*")):
profile_path = trace_path / "trace_profile.json"
if profile_path.exists():
with open(profile_path) as f:
profile = json.load(f)
desc = profile.get("vlm_description", "")
if desc:
descriptions.append({
"id": f"trace_{profile.get('trace_id', 0)}",
"type": "trace",
"text": desc,
})
# Load scene profiles
scene_profile = output_path / f"{file_uuid}_scene_profile.json"
if scene_profile.exists():
with open(scene_profile) as f:
data = json.load(f)
for scene in data.get("scenes", []):
desc = scene.get("vlm_description", "")
if desc:
descriptions.append({
"id": f"scene_{scene.get('scene_number', 0)}",
"type": "scene",
"text": desc,
})
# Load interval profiles
interval_profile = output_path / f"{file_uuid}_interval_profile.json"
if interval_profile.exists():
with open(interval_profile) as f:
data = json.load(f)
for interval in data.get("intervals", []):
desc = interval.get("vlm_description", "")
if desc:
descriptions.append({
"id": f"interval_{interval.get('interval_index', 0)}",
"type": "interval",
"text": desc,
"timestamp_sec": interval.get("timestamp_sec", 0),
})
return descriptions
def evaluate_model(get_embedding_fn, name: str, descriptions: list, queries: list) -> dict:
"""Evaluate a single model."""
print(f"\n[eval] Evaluating {name}...")
results = {
"model": name,
"dimension": None,
"avg_latency_ms": 0,
"total_embeddings": 0,
"test_results": [],
}
# Embed all VLM descriptions
vlm_embeddings = []
total_latency = 0
for i, desc in enumerate(descriptions[:100]): # Limit to 100 for speed
emb, latency = get_embedding_fn(desc["text"])
total_latency += latency
if emb:
vlm_embeddings.append({
"id": desc["id"],
"type": desc["type"],
"text": desc["text"],
"embedding": emb,
})
if results["dimension"] is None:
results["dimension"] = len(emb)
if (i + 1) % 20 == 0:
print(f"[eval] Embedded {i+1}/{min(len(descriptions), 100)}...")
results["total_embeddings"] = len(vlm_embeddings)
if vlm_embeddings:
results["avg_latency_ms"] = round(total_latency / len(vlm_embeddings) * 1000, 1)
# Test queries
for test in queries:
query_emb, latency = get_embedding_fn(test["query"])
if not query_emb:
continue
# Find top-5 similar
similarities = []
for vlm in vlm_embeddings:
sim = cosine_similarity(query_emb, vlm["embedding"])
similarities.append({
"id": vlm["id"],
"type": vlm["type"],
"text": vlm["text"][:100],
"score": round(sim, 4),
})
similarities.sort(key=lambda x: x["score"], reverse=True)
top5 = similarities[:5]
results["test_results"].append({
"query": test["query"],
"lang": test["lang"],
"latency_ms": round(latency * 1000, 1),
"top5": top5,
})
return results
def main():
parser = argparse.ArgumentParser(description="Embedding model evaluation")
parser.add_argument("--file-uuid", "-u", help="File UUID for VLM data")
parser.add_argument("--output-dir", "-o", default="/Users/accusys/momentry/output", help="Output directory")
parser.add_argument("--limit", "-l", type=int, default=100, help="Max VLM descriptions to embed")
args = parser.parse_args()
print("=" * 70)
print("Embedding Model Evaluation")
print("=" * 70)
# Load VLM descriptions
descriptions = []
if args.file_uuid:
descriptions = load_vlm_descriptions(args.file_uuid, args.output_dir)
print(f"\n[eval] Loaded {len(descriptions)} VLM descriptions from {args.file_uuid}")
if not descriptions:
print("[eval] No VLM descriptions found. Using sample data...")
descriptions = [
{"id": "sample_1", "type": "sample", "text": "A person wearing a red shirt and black pants standing in an office."},
{"id": "sample_2", "type": "sample", "text": "Two people in a meeting room, one wearing glasses and formal attire."},
{"id": "sample_3", "type": "sample", "text": "A woman holding a small brown dog outdoors on a sunny day."},
{"id": "sample_4", "type": "sample", "text": "Night scene on a busy street with cars and pedestrians."},
{"id": "sample_5", "type": "sample", "text": "Person in casual clothing sitting at a desk in an office."},
]
# Evaluate Model A (embeddinggemma)
results_a = evaluate_model(get_embedding_a, "embeddinggemma", descriptions, TEST_QUERIES)
# Evaluate Model B (nomic-embed-text-v2-moe)
results_b = evaluate_model(get_embedding_b, "nomic-embed-text-v2-moe", descriptions, TEST_QUERIES)
# Print comparison
print("\n" + "=" * 70)
print("COMPARISON RESULTS")
print("=" * 70)
print(f"\n| Metric | embeddinggemma | nomic-embed-text-v2-moe |")
print(f"|--------|----------------|--------------------------|")
print(f"| Dimension | {results_a.get('dimension', 'N/A')} | {results_b.get('dimension', 'N/A')} |")
print(f"| Avg Latency | {results_a.get('avg_latency_ms', 'N/A')}ms | {results_b.get('avg_latency_ms', 'N/A')}ms |")
print(f"| Total Embedded | {results_a.get('total_embeddings', 0)} | {results_b.get('total_embeddings', 0)} |")
# Show test query results
print("\n" + "-" * 70)
print("TOP-5 RESULTS PER QUERY")
print("-" * 70)
for i, test in enumerate(TEST_QUERIES):
print(f"\nQuery: {test['query']} ({test['lang']})")
if i < len(results_a.get("test_results", [])):
print(f" embeddinggemma Top-5:")
for r in results_a["test_results"][i]["top5"]:
print(f" {r['id']}: {r['score']:.4f} - {r['text'][:50]}...")
if i < len(results_b.get("test_results", [])):
print(f" nomic Top-5:")
for r in results_b["test_results"][i]["top5"]:
print(f" {r['id']}: {r['score']:.4f} - {r['text'][:50]}...")
# Save results
output = {
"embeddinggemma": results_a,
"nomic-embed-text-v2-moe": results_b,
"comparison": {
"dimension_a": results_a.get("dimension"),
"dimension_b": results_b.get("dimension"),
"latency_diff_ms": (results_b.get("avg_latency_ms", 0) or 0) - (results_a.get("avg_latency_ms", 0) or 0),
},
"queries": TEST_QUERIES,
}
output_file = "embedding_eval_results.json"
with open(output_file, "w") as f:
json.dump(output, f, indent=2)
print(f"\n[eval] Results saved to: {output_file}")
if __name__ == "__main__":
main()