fix: face group name read consistency, sync_file_status fix, cleanup ghost records, identity_agent replaced with face_dedup
- get_face_groups_handler: COALESCE(tp.name, tn.label) for name consistency - sync_file_status: compare JSON vs pre_chunks (not chunk table) - face consistency: compare frames.len() not total_faces - cleanup 2 ghost records with NULL file_name/file_path - replace identity_agent with face_dedup in pipeline stages - remove identity_agent_api.rs and all references - update required_processors to match actual processors - update AGENTS.md with team responsibilities - add Studio pipeline changes documentation
This commit is contained in:
Executable
+330
@@ -0,0 +1,330 @@
|
||||
#!/opt/homebrew/bin/python3.11
|
||||
"""
|
||||
Scene VLM Caption - Generate VLM descriptions for scene key frames
|
||||
|
||||
Analyzes scene key frames ({uuid}_scene_N.jpg) using VLM (llava:7b).
|
||||
|
||||
Usage:
|
||||
python scene_vlm_caption.py --file-uuid abc123 --output-dir /path/to/output
|
||||
python scene_vlm_caption.py --scene-dir /path/to/output --scene-number 1
|
||||
|
||||
Output:
|
||||
{output_dir}/{uuid}_scene_profile.json with:
|
||||
- scenes: [{scene_number, vlm_description, vlm_location, ...}]
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
import requests
|
||||
except ImportError:
|
||||
print("requests not installed: pip install requests", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def encode_image(image_path: str) -> str:
|
||||
"""Encode image to base64."""
|
||||
with open(image_path, "rb") as f:
|
||||
return base64.b64encode(f.read()).decode("utf-8")
|
||||
|
||||
|
||||
def call_vlm(image_path: str, prompt: str, model: str = "llava:7b", ollama_url: str = "http://localhost:11434") -> str:
|
||||
"""Call Ollama VLM API."""
|
||||
image_b64 = encode_image(image_path)
|
||||
|
||||
payload = {
|
||||
"model": model,
|
||||
"prompt": prompt,
|
||||
"images": [image_b64],
|
||||
"stream": False,
|
||||
"options": {"num_predict": 100}
|
||||
}
|
||||
|
||||
try:
|
||||
resp = requests.post(f"{ollama_url}/api/generate", json=payload, timeout=30)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
return data.get("response", "").strip()
|
||||
except Exception as e:
|
||||
print(f"[vlm] API error: {e}", file=sys.stderr)
|
||||
return ""
|
||||
|
||||
|
||||
def get_embedding(text: str, model: str = "nomic-embed-text-v2-moe", ollama_url: str = "http://localhost:11434") -> list:
|
||||
"""Get embedding from Ollama."""
|
||||
try:
|
||||
resp = requests.post(
|
||||
f"{ollama_url}/api/embed",
|
||||
json={"model": model, "input": text},
|
||||
timeout=30,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
return data.get("embeddings", [[]])[0]
|
||||
except Exception as e:
|
||||
print(f"[vlm] Embedding error: {e}", file=sys.stderr)
|
||||
return []
|
||||
|
||||
|
||||
def store_to_qdrant(analysis: dict, file_uuid: str, scene_number: int, qdrant_url: str = "http://localhost:6333", qdrant_api_key: str = None) -> bool:
|
||||
"""Store VLM results to Qdrant _vlm collection."""
|
||||
description = analysis.get("vlm_description", "")
|
||||
if not description:
|
||||
return False
|
||||
|
||||
# Get embedding
|
||||
embedding = get_embedding(description)
|
||||
if not embedding:
|
||||
print(f"[vlm] Failed to get embedding for scene_{scene_number}", file=sys.stderr)
|
||||
return False
|
||||
|
||||
# Generate point ID
|
||||
import hashlib
|
||||
point_id = int(hashlib.md5(f"{file_uuid}_scene_{scene_number}".encode()).hexdigest()[:16], 16)
|
||||
|
||||
# Build payload
|
||||
payload = {
|
||||
"type": "scene",
|
||||
"file_uuid": file_uuid,
|
||||
"scene_number": scene_number,
|
||||
**analysis,
|
||||
}
|
||||
|
||||
# Upsert to Qdrant
|
||||
try:
|
||||
headers = {}
|
||||
if qdrant_api_key:
|
||||
headers["api-key"] = qdrant_api_key
|
||||
|
||||
resp = requests.put(
|
||||
f"{qdrant_url}/collections/_vlm/points?wait=true",
|
||||
json={
|
||||
"points": [{
|
||||
"id": point_id,
|
||||
"vector": embedding,
|
||||
"payload": payload,
|
||||
}]
|
||||
},
|
||||
headers=headers,
|
||||
timeout=30,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
print(f"[vlm] Stored to Qdrant: scene_{scene_number}")
|
||||
return True
|
||||
except Exception as e:
|
||||
print(f"[vlm] Qdrant error: {e}", file=sys.stderr)
|
||||
return False
|
||||
|
||||
|
||||
def analyze_scene(image_path: str, model: str = "llava:7b") -> dict:
|
||||
"""
|
||||
Analyze a scene key frame with VLM.
|
||||
|
||||
Returns:
|
||||
Dict with VLM analysis results
|
||||
"""
|
||||
if not Path(image_path).exists():
|
||||
print(f"[vlm] Image not found: {image_path}", file=sys.stderr)
|
||||
return {}
|
||||
|
||||
print(f"[vlm] Analyzing {Path(image_path).name}...")
|
||||
|
||||
# Prompt 1: Scene description
|
||||
desc_prompt = "Describe this scene briefly. Include: location type, main objects, people count, activity. If uncertain, say 'unclear'. Do not guess."
|
||||
description = call_vlm(image_path, desc_prompt, model)
|
||||
|
||||
# Prompt 2: Lighting
|
||||
light_prompt = "What is the lighting? Answer one word: day, night, indoor-light, mixed, or unknown."
|
||||
lighting = call_vlm(image_path, light_prompt, model).lower().strip()
|
||||
|
||||
# Prompt 3: Location classification
|
||||
loc_prompt = "Classify the location. Answer in JSON: {\"location\": \"indoor/outdoor/unknown\", \"setting\": \"office/street/home/nature/studio/unknown\"}. Use 'unknown' if uncertain."
|
||||
loc_raw = call_vlm(image_path, loc_prompt, model)
|
||||
|
||||
loc_data = {}
|
||||
try:
|
||||
loc_clean = loc_raw.replace("```json", "").replace("```", "").strip()
|
||||
parsed = json.loads(loc_clean)
|
||||
if isinstance(parsed, dict):
|
||||
loc_data = parsed
|
||||
else:
|
||||
loc_data = {}
|
||||
except:
|
||||
loc_data = {}
|
||||
|
||||
# Prompt 4: Weather (for outdoor scenes)
|
||||
weather_prompt = "If outdoor, what is the weather? Answer one word: sunny, cloudy, rainy, night, or unknown. If indoor, answer 'indoor'."
|
||||
weather = call_vlm(image_path, weather_prompt, model).lower().strip()
|
||||
|
||||
# Prompt 5: People count
|
||||
people_prompt = "How many people are visible? Answer a number or 'unclear'."
|
||||
people_count = call_vlm(image_path, people_prompt, model).strip()
|
||||
|
||||
# Prompt 6: Objects/vehicles
|
||||
objects_prompt = "What notable objects or vehicles are visible? Answer in JSON: {\"vehicles\": [\"car\", \"bus\", etc.], \"objects\": [\"table\", \"chair\", etc.]}. Use empty lists if none or unclear."
|
||||
objects_raw = call_vlm(image_path, objects_prompt, model)
|
||||
|
||||
objects_data = {}
|
||||
try:
|
||||
objects_clean = objects_raw.replace("```json", "").replace("```", "").strip()
|
||||
parsed = json.loads(objects_clean)
|
||||
if isinstance(parsed, dict):
|
||||
objects_data = parsed
|
||||
else:
|
||||
objects_data = {"vehicles": [], "objects": []}
|
||||
except:
|
||||
objects_data = {"vehicles": [], "objects": []}
|
||||
|
||||
# Prompt 7: Plants
|
||||
plants_prompt = "What plants are visible? Answer in JSON: {\"has_plants\": true/false, \"plants\": [\"list recognizable names or brief descriptions\"]. Use empty list if none or unclear.}"
|
||||
plants_raw = call_vlm(image_path, plants_prompt, model)
|
||||
|
||||
plants_data = {}
|
||||
try:
|
||||
plants_clean = plants_raw.replace("```json", "").replace("```", "").strip()
|
||||
parsed = json.loads(plants_clean)
|
||||
if isinstance(parsed, dict):
|
||||
plants_data = parsed
|
||||
else:
|
||||
plants_data = {"has_plants": False, "plants": []}
|
||||
except:
|
||||
plants_data = {"has_plants": False, "plants": []}
|
||||
|
||||
# Prompt 8: Animals
|
||||
animals_prompt = "What animals are visible? Answer in JSON: {\"has_animals\": true/false, \"animals\": [\"list recognizable names or brief descriptions\"]. Use empty list if none or unclear.}"
|
||||
animals_raw = call_vlm(image_path, animals_prompt, model)
|
||||
|
||||
animals_data = {}
|
||||
try:
|
||||
animals_clean = animals_raw.replace("```json", "").replace("```", "").strip()
|
||||
parsed = json.loads(animals_clean)
|
||||
if isinstance(parsed, dict):
|
||||
animals_data = parsed
|
||||
else:
|
||||
animals_data = {"has_animals": False, "animals": []}
|
||||
except:
|
||||
animals_data = {"has_animals": False, "animals": []}
|
||||
|
||||
# Prompt 9: Tags
|
||||
tags_prompt = "List 5 tags for this scene, comma-separated. Only include what is clearly visible. Examples: office, street, sunny, crowd, nature."
|
||||
tags_raw = call_vlm(image_path, tags_prompt, model)
|
||||
tags = [t.strip() for t in tags_raw.replace(",", " ").split() if t.strip()][:5]
|
||||
|
||||
return {
|
||||
"vlm_description": description,
|
||||
"vlm_lighting": lighting,
|
||||
"vlm_location": loc_data.get("location", "unknown"),
|
||||
"vlm_setting": loc_data.get("setting", "unknown"),
|
||||
"vlm_weather": weather,
|
||||
"vlm_people_count": people_count,
|
||||
"vlm_vehicles": objects_data.get("vehicles", []),
|
||||
"vlm_objects": objects_data.get("objects", []),
|
||||
"vlm_has_plants": plants_data.get("has_plants", False),
|
||||
"vlm_plants": plants_data.get("plants", []),
|
||||
"vlm_has_animals": animals_data.get("has_animals", False),
|
||||
"vlm_animals": animals_data.get("animals", []),
|
||||
"vlm_tags": tags,
|
||||
"vlm_model": model,
|
||||
}
|
||||
|
||||
|
||||
def analyze_all_scenes(file_uuid: str, output_dir: str, model: str = "llava:7b", store_qdrant: bool = True) -> dict:
|
||||
"""
|
||||
Analyze all scene key frames for a file.
|
||||
|
||||
Returns:
|
||||
Summary dict
|
||||
"""
|
||||
output_path = Path(output_dir)
|
||||
|
||||
# Find all scene images
|
||||
scene_images = sorted(output_path.glob(f"{file_uuid}_scene_*.jpg"))
|
||||
|
||||
if not scene_images:
|
||||
print(f"[vlm] No scene images found: {file_uuid}_scene_*.jpg in {output_dir}", file=sys.stderr)
|
||||
return {"error": "No scene images"}
|
||||
|
||||
results = []
|
||||
qdrant_api_key = os.environ.get("QDRANT_API_KEY")
|
||||
|
||||
for scene_img in scene_images:
|
||||
# Extract scene number from filename
|
||||
scene_number = int(scene_img.stem.split("_scene_")[1])
|
||||
|
||||
analysis = analyze_scene(str(scene_img), model)
|
||||
|
||||
if analysis:
|
||||
# Store to Qdrant
|
||||
if store_qdrant:
|
||||
store_to_qdrant(analysis, file_uuid, scene_number, qdrant_api_key=qdrant_api_key)
|
||||
|
||||
results.append({
|
||||
"scene_number": scene_number,
|
||||
"image_path": str(scene_img),
|
||||
**analysis,
|
||||
})
|
||||
|
||||
# Save profile
|
||||
profile = {
|
||||
"file_uuid": file_uuid,
|
||||
"total_scenes": len(results),
|
||||
"model": model,
|
||||
"scenes": results,
|
||||
}
|
||||
|
||||
profile_path = output_path / f"{file_uuid}_scene_profile.json"
|
||||
with open(profile_path, "w") as f:
|
||||
json.dump(profile, f, indent=2)
|
||||
|
||||
print(f"[vlm] Saved profile: {profile_path}")
|
||||
|
||||
return profile
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="VLM caption generation for scene key frames")
|
||||
parser.add_argument("--file-uuid", "-u", help="File UUID (analyze all scenes)")
|
||||
parser.add_argument("--scene-dir", "-d", help="Scene directory (contains {uuid}_scene_N.jpg)")
|
||||
parser.add_argument("--scene-number", "-n", type=int, help="Single scene number")
|
||||
parser.add_argument("--output-dir", "-o", default="/Users/accusys/momentry/output", help="Output directory")
|
||||
parser.add_argument("--model", "-m", default="llava:7b", help="VLM model name")
|
||||
parser.add_argument("--json", "-j", action="store_true", help="Output as JSON")
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.file_uuid:
|
||||
result = analyze_all_scenes(args.file_uuid, args.output_dir, args.model)
|
||||
elif args.scene_dir and args.scene_number is not None:
|
||||
# Find file_uuid from directory
|
||||
scene_dir = Path(args.scene_dir)
|
||||
file_uuid = None
|
||||
for f in scene_dir.glob("*_scene_*.jpg"):
|
||||
file_uuid = f.stem.split("_scene_")[0]
|
||||
break
|
||||
|
||||
if not file_uuid:
|
||||
print("Cannot determine file_uuid from scene_dir", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
scene_img = scene_dir / f"{file_uuid}_scene_{args.scene_number}.jpg"
|
||||
result = analyze_scene(str(scene_img), args.model)
|
||||
else:
|
||||
parser.error("Requires --file-uuid or (--scene-dir + --scene-number)")
|
||||
|
||||
if args.json:
|
||||
print(json.dumps(result, indent=2))
|
||||
else:
|
||||
if "scenes" in result:
|
||||
print(f"Analyzed {len(result['scenes'])} scenes")
|
||||
elif "vlm_description" in result:
|
||||
print(f"Description: {result['vlm_description']}")
|
||||
print(f"Location: {result.get('vlm_location', 'unknown')}")
|
||||
print(f"Tags: {result.get('vlm_tags', [])}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user