fix: use environment variables for VLM/LLM endpoints and models

- trace_vlm_caption.py: use MOMENTRY_LLM_VISION_URL and MOMENTRY_LLM_VISION_MODEL
- scene_vlm_caption.py: use MOMENTRY_LLM_VISION_URL and MOMENTRY_LLM_VISION_MODEL
- Changed from Ollama /api/generate to OpenAI-compatible /v1/chat/completions format
- Added embedding server environment variables
This commit is contained in:
Accusys
2026-07-27 13:49:08 +08:00
parent 39a2cbc65b
commit 5d0c771a2b
2 changed files with 56 additions and 30 deletions
+28 -15
View File
@@ -2,7 +2,7 @@
"""
Scene VLM Caption - Generate VLM descriptions for scene key frames
Analyzes scene key frames ({uuid}_scene_N.jpg) using VLM (llava:7b).
Analyzes scene key frames ({uuid}_scene_N.jpg) using VLM.
Usage:
python scene_vlm_caption.py --file-uuid abc123 --output-dir /path/to/output
@@ -11,6 +11,10 @@ Usage:
Output:
{output_dir}/{uuid}_scene_profile.json with:
- scenes: [{scene_number, vlm_description, vlm_location, ...}]
Environment Variables:
MOMENTRY_LLM_VISION_URL - VLM endpoint (default: http://localhost:8091/v1/chat/completions)
MOMENTRY_LLM_VISION_MODEL - VLM model (default: llava-v1.6-vicuna-13b)
"""
import argparse
@@ -20,6 +24,10 @@ import os
import sys
from pathlib import Path
# VLM configuration from environment variables
VLM_URL = os.environ.get("MOMENTRY_LLM_VISION_URL", "http://localhost:8091/v1/chat/completions")
VLM_MODEL = os.environ.get("MOMENTRY_LLM_VISION_MODEL", "llava-v1.6-vicuna-13b")
try:
import requests
except ImportError:
@@ -33,39 +41,44 @@ def encode_image(image_path: str) -> str:
return base64.b64encode(f.read()).decode("utf-8")
def call_vlm(image_path: str, prompt: str, model: str = "llava:7b", ollama_url: str = "http://localhost:11434") -> str:
"""Call Ollama VLM API."""
def call_vlm(image_path: str, prompt: str) -> str:
"""Call VLM API using OpenAI-compatible format."""
image_b64 = encode_image(image_path)
payload = {
"model": model,
"prompt": prompt,
"images": [image_b64],
"stream": False,
"options": {"num_predict": 100}
"model": VLM_MODEL,
"messages": [
{"role": "user", "content": [
{"type": "text", "text": prompt},
{"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{image_b64}"}}
]}
],
"max_tokens": 100,
}
try:
resp = requests.post(f"{ollama_url}/api/generate", json=payload, timeout=30)
resp = requests.post(VLM_URL, json=payload, timeout=30)
resp.raise_for_status()
data = resp.json()
return data.get("response", "").strip()
return data.get("choices", [{}])[0].get("message", {}).get("content", "").strip()
except Exception as e:
print(f"[vlm] API error: {e}", file=sys.stderr)
return ""
def get_embedding(text: str, model: str = "nomic-embed-text-v2-moe", ollama_url: str = "http://localhost:11434") -> list:
"""Get embedding from Ollama."""
def get_embedding(text: str) -> list:
"""Get embedding from embedding server."""
embed_url = os.environ.get("MOMENTRY_EMBEDDING_URL", "http://localhost:11436/v1/embeddings")
embed_model = os.environ.get("MOMENTRY_EMBEDDING_MODEL", "embeddinggemma-300m")
try:
resp = requests.post(
f"{ollama_url}/api/embed",
json={"model": model, "input": text},
embed_url,
json={"model": embed_model, "input": text},
timeout=30,
)
resp.raise_for_status()
data = resp.json()
return data.get("embeddings", [[]])[0]
return data.get("data", [{}])[0].get("embedding", [])
except Exception as e:
print(f"[vlm] Embedding error: {e}", file=sys.stderr)
return []