feat: add cluster-agent endpoint, fix OCR labeling, fix ingestion blocking

- Add POST /api/v1/file/:file_uuid/cluster-agent endpoint for on-demand face clustering
- Fix OCR chunks being labeled as ASRX: use ChunkType.as_str() instead of {:?}
- Rule 1 now deletes old chunks before re-inserting to avoid stale data
- Add fallback face_traced.json when store_traced_faces.py fails
- ingestion_complete now handles status='error' to unblock jobs
- VLM describe tool now uses trace_id with pre-extracted face crops
This commit is contained in:
Accusys
2026-07-20 21:48:27 +08:00
parent 244af51edf
commit 7dc910e2de
8 changed files with 654 additions and 57 deletions
+113
View File
@@ -1,5 +1,6 @@
use base64::{engine::general_purpose::STANDARD as BASE64, Engine};
use serde_json;
use std::time::Duration;
use crate::core::db::qdrant_db::QdrantDb;
use crate::core::db::schema;
@@ -1233,3 +1234,115 @@ pub async fn exec_search_by_appearance(
Err("Color search output not found".to_string())
}
}
fn face_crop_path(file_uuid: &str, trace_id: i32) -> Option<std::path::PathBuf> {
let base = std::env::var("MOMENTRY_OUTPUT_DIR")
.unwrap_or_else(|_| "/Users/accusys/momentry/output".to_string());
let dir = std::path::PathBuf::from(base)
.join(".faces")
.join(file_uuid)
.join(trace_id.to_string());
if !dir.exists() {
return None;
}
let mut entries: Vec<_> = match std::fs::read_dir(&dir) {
Ok(e) => e.filter_map(|e| e.ok()).collect(),
Err(_) => return None,
};
entries.sort_by_key(|e| e.file_name());
entries.first().map(|e| e.path())
}
pub async fn exec_vlm_describe(
pool: &sqlx::PgPool,
args: &serde_json::Value,
) -> Result<String, String> {
let file_uuid = args.get("file_uuid").and_then(|v| v.as_str()).unwrap_or("");
let trace_id = args.get("trace_id").and_then(|v| v.as_i64()).unwrap_or(0) as i32;
let prompt = args
.get("prompt")
.and_then(|v| v.as_str())
.unwrap_or("Describe this person's clothing and appearance. Focus on colors, clothing type, and any distinctive visual features.");
if file_uuid.is_empty() {
return Ok(serde_json::json!({"error": "file_uuid is required"}).to_string());
}
if trace_id <= 0 {
return Ok(serde_json::json!({"error": "trace_id is required and must be > 0"}).to_string());
}
let crop_path = face_crop_path(file_uuid, trace_id)
.ok_or_else(|| format!("No face crop found for {} trace {}", file_uuid, trace_id))?;
let jpeg_bytes = std::fs::read(&crop_path)
.map_err(|e| format!("Failed to read face crop: {}", e))?;
let videos = schema::table_name("videos");
let fps: f64 = sqlx::query_scalar(&format!(
"SELECT COALESCE(fps, 25.0) FROM {} WHERE file_uuid = $1",
videos
))
.bind(file_uuid)
.fetch_optional(pool)
.await
.map_err(|e| e.to_string())?
.unwrap_or(25.0);
let frame_name = crop_path
.file_stem()
.and_then(|s| s.to_str())
.and_then(|s| s.parse::<i64>().ok())
.unwrap_or(0);
let timestamp_secs = frame_name as f64 / fps;
let base64_img = BASE64.encode(&jpeg_bytes);
let ollama_url = std::env::var("OLLAMA_URL")
.unwrap_or_else(|_| "http://localhost:11434".to_string());
let model = std::env::var("VLM_MODEL").unwrap_or_else(|_| "llava".to_string());
let body = serde_json::json!({
"model": model,
"prompt": prompt,
"images": [base64_img],
"stream": false,
"options": {
"num_predict": 80
}
});
let client = reqwest::Client::builder()
.timeout(Duration::from_secs(30))
.build()
.map_err(|e| format!("Failed to create HTTP client: {}", e))?;
let resp = client
.post(format!("{}/api/generate", ollama_url))
.json(&body)
.send()
.await
.map_err(|e| format!("Ollama request failed: {}", e))?;
let resp_json: serde_json::Value = resp
.json()
.await
.map_err(|e| format!("Failed to parse Ollama response: {}", e))?;
let description = resp_json
.get("response")
.and_then(|v| v.as_str())
.unwrap_or("No description returned")
.to_string();
Ok(serde_json::json!({
"tool": "vlm_describe",
"result": {
"file_uuid": file_uuid,
"trace_id": trace_id,
"frame": frame_name,
"description": description,
"time_sec": (timestamp_secs * 100.0).round() / 100.0
}
})
.to_string())
}
+9 -1
View File
@@ -30,6 +30,14 @@ pub async fn execute_rule1(db: &PostgresDb, file_uuid: &str, fps: f64) -> Result
let mut count = 0;
let mut tx = pool.begin().await?;
// Delete existing chunks for this file before re-inserting
let chunk_table = schema::table_name("chunk");
sqlx::query(&format!("DELETE FROM {} WHERE file_uuid = $1", chunk_table))
.bind(file_uuid)
.execute(&mut *tx)
.await?;
info!("Rule 1: Deleted old chunks for video {}", file_uuid);
// Phase 1: ASRX segments (pure speech, NO OCR merge)
for seg in asr_segments.iter() {
// Skip chunks with no text
@@ -97,7 +105,7 @@ pub async fn execute_rule1(db: &PostgresDb, file_uuid: &str, fps: f64) -> Result
file_id as i32,
file_uuid.to_string(),
format!("{}", count),
ChunkType::Sentence,
ChunkType::Ocr,
ChunkRule::Rule1,
start_time,
end_time,
+25 -47
View File
@@ -824,7 +824,7 @@ pub struct PostgresCache {
#[derive(Debug, serde::Serialize, sqlx::FromRow)]
pub struct SemanticSearchResult {
pub id: i32,
pub file_uuid: Option<String>, // Added for global search
pub file_uuid: Option<String>,
pub scene_order: i32,
pub start_frame: i64,
pub end_frame: i64,
@@ -836,6 +836,7 @@ pub struct SemanticSearchResult {
pub metadata: Option<serde_json::Value>,
pub similarity: Option<f64>,
pub content: Option<serde_json::Value>,
pub chunk_type: String,
}
/// Result structure for child chunks
@@ -2519,7 +2520,8 @@ impl PostgresDb {
text_content, \
metadata, \
(1 - (embedding <=> $1::vector)) as similarity, \
content \
content, \
COALESCE(chunk_type, 'sentence') as chunk_type \
FROM {} \
WHERE file_uuid = $2 AND chunk_type IN ('sentence', 'story_parent', 'llm_parent') AND embedding IS NOT NULL \
ORDER BY embedding <=> $1::vector \
@@ -2557,7 +2559,8 @@ impl PostgresDb {
text_content, \
metadata, \
(1 - (embedding <=> $1::vector)) as similarity, \
content \
content, \
COALESCE(chunk_type, 'sentence') as chunk_type \
FROM {} \
WHERE chunk_type IN ('sentence', 'story_parent', 'llm_parent') AND embedding IS NOT NULL \
ORDER BY embedding <=> $1::vector \
@@ -2590,7 +2593,8 @@ impl PostgresDb {
text_content as text_content, \
metadata, \
1.0::float8 as similarity, \
content \
content, \
COALESCE(chunk_type, 'sentence') as chunk_type \
FROM {} \
WHERE file_uuid = $1 AND chunk_id = $2 AND embedding IS NOT NULL \
LIMIT 1",
@@ -2622,7 +2626,8 @@ impl PostgresDb {
text_content as text_content, \
metadata, \
1.0::float8 as similarity, \
content \
content, \
COALESCE(chunk_type, 'sentence') as chunk_type \
FROM {} \
WHERE file_uuid = $1 AND chunk_id = $2 \
LIMIT 1",
@@ -2885,7 +2890,7 @@ impl PostgresDb {
tx: &mut sqlx::Transaction<'_, sqlx::Postgres>,
) -> Result<()> {
let table = schema::table_name("chunk");
let ct_str = format!("{:?}", chunk.chunk_type).to_lowercase();
let ct_str = chunk.chunk_type.as_str();
let fps = chunk.fps;
let start_time = chunk.start_frame as f64 / chunk.fps;
let end_time = chunk.end_frame as f64 / chunk.fps;
@@ -3419,45 +3424,18 @@ impl PostgresDb {
let like = format!("%{}%", query.replace('%', "%%"));
use sqlx::Row;
// Check if query contains CJK characters
let has_cjk = query.chars().any(|c| {
('\u{4E00}'..='\u{9FFF}').contains(&c)
|| ('\u{3040}'..='\u{309F}').contains(&c)
|| ('\u{30A0}'..='\u{30FF}').contains(&c)
|| ('\u{AC00}'..='\u{D7AF}').contains(&c)
});
let sql = if has_cjk {
// CJK/Korean: use ILIKE position-based ranking
format!(
"SELECT chunk_id, file_uuid, chunk_type, text_content, start_time, end_time, \
(1.0 - (POSITION(LOWER($1) IN LOWER(text_content))::float8 / NULLIF(LENGTH(text_content), 0)::float8))::float8 as score \
FROM {} \
WHERE text_content ILIKE $2 AND text_content != '' \
{}\
ORDER BY score DESC \
LIMIT $3",
table,
if file_uuid.is_some() { "AND file_uuid = $4 " } else { "" }
)
} else {
// English: use PostgreSQL full-text search
format!(
"SELECT chunk_id, file_uuid, chunk_type, text_content, start_time, end_time, \
CASE \
WHEN to_tsvector('english', text_content) @@ plainto_tsquery('english', $1) \
THEN ts_rank(to_tsvector('english', text_content), plainto_tsquery('english', $1))::float8 \
ELSE 0.1::float8 \
END as score \
FROM {} \
WHERE text_content ILIKE $2 AND text_content != '' \
{}\
ORDER BY score DESC \
LIMIT $3",
table,
if file_uuid.is_some() { "AND file_uuid = $4 " } else { "" }
)
};
// Simple LIKE-based matching (no FTS stemming)
let sql = format!(
"SELECT chunk_id, file_uuid, chunk_type, text_content, start_time, end_time, \
(1.0 - (POSITION(LOWER($1) IN LOWER(text_content))::float8 / NULLIF(LENGTH(text_content), 0)::float8))::float8 as score \
FROM {} \
WHERE text_content ILIKE $2 AND text_content != '' \
{}\
ORDER BY score DESC \
LIMIT $3",
table,
if file_uuid.is_some() { "AND file_uuid = $4 " } else { "" }
);
let rows = if let Some(u) = file_uuid {
sqlx::query(&sql)
@@ -4309,7 +4287,7 @@ impl PostgresDb {
pub async fn store_chunk(&self, chunk: &crate::core::chunk::types::Chunk) -> Result<()> {
let table = schema::table_name("chunk");
let ct_str = format!("{:?}", chunk.chunk_type).to_lowercase();
let ct_str = chunk.chunk_type.as_str();
let start_time = chunk.start_frame as f64 / chunk.fps;
let end_time = chunk.end_frame as f64 / chunk.fps;
sqlx::query(&format!(
@@ -4550,7 +4528,7 @@ impl Database for PostgresDb {
impl crate::core::db::ChunkStore for PostgresDb {
async fn store_chunk(&self, chunk: &crate::core::chunk::types::Chunk) -> Result<()> {
let table = schema::table_name("chunk");
let ct_str = format!("{:?}", chunk.chunk_type).to_lowercase();
let ct_str = chunk.chunk_type.as_str();
let start_time = chunk.start_frame as f64 / chunk.fps;
let end_time = chunk.end_frame as f64 / chunk.fps;
sqlx::query(&format!(