39a2cbc65b
- get_face_groups_handler: COALESCE(tp.name, tn.label) for name consistency - sync_file_status: compare JSON vs pre_chunks (not chunk table) - face consistency: compare frames.len() not total_faces - cleanup 2 ghost records with NULL file_name/file_path - replace identity_agent with face_dedup in pipeline stages - remove identity_agent_api.rs and all references - update required_processors to match actual processors - update AGENTS.md with team responsibilities - add Studio pipeline changes documentation
286 lines
10 KiB
Python
286 lines
10 KiB
Python
#!/opt/homebrew/bin/python3.11
|
|
"""
|
|
Embedding Model Evaluation - Compare embeddinggemma vs nomic-embed-text-v2-moe
|
|
|
|
Usage:
|
|
python3 scripts/eval_embedding_models.py --file-uuid <uuid> --output-dir /path/to/output
|
|
|
|
Metrics:
|
|
1. Accuracy: Semantic similarity ranking quality
|
|
2. Speed: Latency per embedding
|
|
3. Dimension: Vector size
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import math
|
|
import os
|
|
import sys
|
|
import time
|
|
from pathlib import Path
|
|
|
|
try:
|
|
import requests
|
|
except ImportError:
|
|
print("requests not installed: pip install requests", file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
# Embedding endpoints
|
|
EMBED_A_URL = "http://localhost:11436/v1/embeddings" # embeddinggemma
|
|
EMBED_B_URL = "http://localhost:11434/api/embed" # nomic-embed-text-v2-moe
|
|
EMBED_B_MODEL = "nomic-embed-text-v2-moe"
|
|
|
|
# Test queries (Chinese, English, Mixed)
|
|
TEST_QUERIES = [
|
|
{"query": "穿西裝的男人", "lang": "zh"},
|
|
{"query": "室內辦公室", "lang": "zh"},
|
|
{"query": "雪景", "lang": "zh"},
|
|
{"query": "持槍的人", "lang": "zh"},
|
|
{"query": "woman in white dress", "lang": "en"},
|
|
{"query": "outdoor scene night", "lang": "en"},
|
|
{"query": "person holding object", "lang": "en"},
|
|
{"query": "穿著 formal 的男人", "lang": "mixed"},
|
|
]
|
|
|
|
|
|
def get_embedding_a(text: str) -> tuple:
|
|
"""Get embedding from embeddinggemma (port 11436)."""
|
|
start = time.time()
|
|
try:
|
|
resp = requests.post(EMBED_A_URL, json={"input": text}, timeout=30)
|
|
resp.raise_for_status()
|
|
data = resp.json()
|
|
elapsed = time.time() - start
|
|
return data["data"][0]["embedding"], elapsed
|
|
except Exception as e:
|
|
print(f"[eval] embeddinggemma error: {e}", file=sys.stderr)
|
|
return [], 0
|
|
|
|
|
|
def get_embedding_b(text: str) -> tuple:
|
|
"""Get embedding from nomic-embed-text-v2-moe (port 11434)."""
|
|
start = time.time()
|
|
try:
|
|
resp = requests.post(EMBED_B_URL, json={"model": EMBED_B_MODEL, "input": text}, timeout=30)
|
|
resp.raise_for_status()
|
|
data = resp.json()
|
|
elapsed = time.time() - start
|
|
return data["embeddings"][0], elapsed
|
|
except Exception as e:
|
|
print(f"[eval] nomic error: {e}", file=sys.stderr)
|
|
return [], 0
|
|
|
|
|
|
def cosine_similarity(a: list, b: list) -> float:
|
|
"""Calculate cosine similarity."""
|
|
if not a or not b or len(a) != len(b):
|
|
return 0.0
|
|
dot = sum(x * y for x, y in zip(a, b))
|
|
norm_a = math.sqrt(sum(x * x for x in a))
|
|
norm_b = math.sqrt(sum(y * y for y in b))
|
|
return dot / (norm_a * norm_b) if norm_a > 0 and norm_b > 0 else 0.0
|
|
|
|
|
|
def load_vlm_descriptions(file_uuid: str, output_dir: str) -> list:
|
|
"""Load VLM descriptions from trace/scene/interval profiles."""
|
|
descriptions = []
|
|
output_path = Path(output_dir)
|
|
|
|
# Load trace profiles
|
|
trace_dir = output_path / file_uuid
|
|
if trace_dir.exists():
|
|
for trace_path in sorted(trace_dir.glob("trace_*")):
|
|
profile_path = trace_path / "trace_profile.json"
|
|
if profile_path.exists():
|
|
with open(profile_path) as f:
|
|
profile = json.load(f)
|
|
desc = profile.get("vlm_description", "")
|
|
if desc:
|
|
descriptions.append({
|
|
"id": f"trace_{profile.get('trace_id', 0)}",
|
|
"type": "trace",
|
|
"text": desc,
|
|
})
|
|
|
|
# Load scene profiles
|
|
scene_profile = output_path / f"{file_uuid}_scene_profile.json"
|
|
if scene_profile.exists():
|
|
with open(scene_profile) as f:
|
|
data = json.load(f)
|
|
for scene in data.get("scenes", []):
|
|
desc = scene.get("vlm_description", "")
|
|
if desc:
|
|
descriptions.append({
|
|
"id": f"scene_{scene.get('scene_number', 0)}",
|
|
"type": "scene",
|
|
"text": desc,
|
|
})
|
|
|
|
# Load interval profiles
|
|
interval_profile = output_path / f"{file_uuid}_interval_profile.json"
|
|
if interval_profile.exists():
|
|
with open(interval_profile) as f:
|
|
data = json.load(f)
|
|
for interval in data.get("intervals", []):
|
|
desc = interval.get("vlm_description", "")
|
|
if desc:
|
|
descriptions.append({
|
|
"id": f"interval_{interval.get('interval_index', 0)}",
|
|
"type": "interval",
|
|
"text": desc,
|
|
"timestamp_sec": interval.get("timestamp_sec", 0),
|
|
})
|
|
|
|
return descriptions
|
|
|
|
|
|
def evaluate_model(get_embedding_fn, name: str, descriptions: list, queries: list) -> dict:
|
|
"""Evaluate a single model."""
|
|
print(f"\n[eval] Evaluating {name}...")
|
|
|
|
results = {
|
|
"model": name,
|
|
"dimension": None,
|
|
"avg_latency_ms": 0,
|
|
"total_embeddings": 0,
|
|
"test_results": [],
|
|
}
|
|
|
|
# Embed all VLM descriptions
|
|
vlm_embeddings = []
|
|
total_latency = 0
|
|
|
|
for i, desc in enumerate(descriptions[:100]): # Limit to 100 for speed
|
|
emb, latency = get_embedding_fn(desc["text"])
|
|
total_latency += latency
|
|
|
|
if emb:
|
|
vlm_embeddings.append({
|
|
"id": desc["id"],
|
|
"type": desc["type"],
|
|
"text": desc["text"],
|
|
"embedding": emb,
|
|
})
|
|
|
|
if results["dimension"] is None:
|
|
results["dimension"] = len(emb)
|
|
|
|
if (i + 1) % 20 == 0:
|
|
print(f"[eval] Embedded {i+1}/{min(len(descriptions), 100)}...")
|
|
|
|
results["total_embeddings"] = len(vlm_embeddings)
|
|
if vlm_embeddings:
|
|
results["avg_latency_ms"] = round(total_latency / len(vlm_embeddings) * 1000, 1)
|
|
|
|
# Test queries
|
|
for test in queries:
|
|
query_emb, latency = get_embedding_fn(test["query"])
|
|
if not query_emb:
|
|
continue
|
|
|
|
# Find top-5 similar
|
|
similarities = []
|
|
for vlm in vlm_embeddings:
|
|
sim = cosine_similarity(query_emb, vlm["embedding"])
|
|
similarities.append({
|
|
"id": vlm["id"],
|
|
"type": vlm["type"],
|
|
"text": vlm["text"][:100],
|
|
"score": round(sim, 4),
|
|
})
|
|
|
|
similarities.sort(key=lambda x: x["score"], reverse=True)
|
|
top5 = similarities[:5]
|
|
|
|
results["test_results"].append({
|
|
"query": test["query"],
|
|
"lang": test["lang"],
|
|
"latency_ms": round(latency * 1000, 1),
|
|
"top5": top5,
|
|
})
|
|
|
|
return results
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(description="Embedding model evaluation")
|
|
parser.add_argument("--file-uuid", "-u", help="File UUID for VLM data")
|
|
parser.add_argument("--output-dir", "-o", default="/Users/accusys/momentry/output", help="Output directory")
|
|
parser.add_argument("--limit", "-l", type=int, default=100, help="Max VLM descriptions to embed")
|
|
args = parser.parse_args()
|
|
|
|
print("=" * 70)
|
|
print("Embedding Model Evaluation")
|
|
print("=" * 70)
|
|
|
|
# Load VLM descriptions
|
|
descriptions = []
|
|
if args.file_uuid:
|
|
descriptions = load_vlm_descriptions(args.file_uuid, args.output_dir)
|
|
print(f"\n[eval] Loaded {len(descriptions)} VLM descriptions from {args.file_uuid}")
|
|
|
|
if not descriptions:
|
|
print("[eval] No VLM descriptions found. Using sample data...")
|
|
descriptions = [
|
|
{"id": "sample_1", "type": "sample", "text": "A person wearing a red shirt and black pants standing in an office."},
|
|
{"id": "sample_2", "type": "sample", "text": "Two people in a meeting room, one wearing glasses and formal attire."},
|
|
{"id": "sample_3", "type": "sample", "text": "A woman holding a small brown dog outdoors on a sunny day."},
|
|
{"id": "sample_4", "type": "sample", "text": "Night scene on a busy street with cars and pedestrians."},
|
|
{"id": "sample_5", "type": "sample", "text": "Person in casual clothing sitting at a desk in an office."},
|
|
]
|
|
|
|
# Evaluate Model A (embeddinggemma)
|
|
results_a = evaluate_model(get_embedding_a, "embeddinggemma", descriptions, TEST_QUERIES)
|
|
|
|
# Evaluate Model B (nomic-embed-text-v2-moe)
|
|
results_b = evaluate_model(get_embedding_b, "nomic-embed-text-v2-moe", descriptions, TEST_QUERIES)
|
|
|
|
# Print comparison
|
|
print("\n" + "=" * 70)
|
|
print("COMPARISON RESULTS")
|
|
print("=" * 70)
|
|
print(f"\n| Metric | embeddinggemma | nomic-embed-text-v2-moe |")
|
|
print(f"|--------|----------------|--------------------------|")
|
|
print(f"| Dimension | {results_a.get('dimension', 'N/A')} | {results_b.get('dimension', 'N/A')} |")
|
|
print(f"| Avg Latency | {results_a.get('avg_latency_ms', 'N/A')}ms | {results_b.get('avg_latency_ms', 'N/A')}ms |")
|
|
print(f"| Total Embedded | {results_a.get('total_embeddings', 0)} | {results_b.get('total_embeddings', 0)} |")
|
|
|
|
# Show test query results
|
|
print("\n" + "-" * 70)
|
|
print("TOP-5 RESULTS PER QUERY")
|
|
print("-" * 70)
|
|
|
|
for i, test in enumerate(TEST_QUERIES):
|
|
print(f"\nQuery: {test['query']} ({test['lang']})")
|
|
|
|
if i < len(results_a.get("test_results", [])):
|
|
print(f" embeddinggemma Top-5:")
|
|
for r in results_a["test_results"][i]["top5"]:
|
|
print(f" {r['id']}: {r['score']:.4f} - {r['text'][:50]}...")
|
|
|
|
if i < len(results_b.get("test_results", [])):
|
|
print(f" nomic Top-5:")
|
|
for r in results_b["test_results"][i]["top5"]:
|
|
print(f" {r['id']}: {r['score']:.4f} - {r['text'][:50]}...")
|
|
|
|
# Save results
|
|
output = {
|
|
"embeddinggemma": results_a,
|
|
"nomic-embed-text-v2-moe": results_b,
|
|
"comparison": {
|
|
"dimension_a": results_a.get("dimension"),
|
|
"dimension_b": results_b.get("dimension"),
|
|
"latency_diff_ms": (results_b.get("avg_latency_ms", 0) or 0) - (results_a.get("avg_latency_ms", 0) or 0),
|
|
},
|
|
"queries": TEST_QUERIES,
|
|
}
|
|
|
|
output_file = "embedding_eval_results.json"
|
|
with open(output_file, "w") as f:
|
|
json.dump(output, f, indent=2)
|
|
|
|
print(f"\n[eval] Results saved to: {output_file}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main() |