fix: ASRX duplication, TKG edges, trace ingest, and add pipeline progress publishing

- ASRX handler no longer stores duplicate 'asr' pre_chunks
- Pre_chunks storage made idempotent (delete-before-insert)
- Rule 1 + trace_ingest changed to query 'asrx' not 'asr'
- Trace chunks removed (dynamic from TKG/Qdrant)
- TKG scroll_face_points fixed: trace_id >= 1 (not == 1)
- TKG AsrxSegmentEntry: start/end -> start_time/end_time (match ASRX JSON)
- Unregister error handling: log instead of silent discard
- Add publish_pipeline_progress calls at each pipeline stage
  (processors, rule1, face_trace, identity_agent, TKG, rule2, completion)
This commit is contained in:
Accusys
2026-07-02 10:43:46 +08:00
parent d791d138f2
commit 3eabd45882
65 changed files with 9477 additions and 3852 deletions
+373 -145
View File
@@ -1,6 +1,7 @@
use base64::{engine::general_purpose::STANDARD as BASE64, Engine};
use serde_json;
use crate::core::db::qdrant_db::QdrantDb;
use crate::core::db::schema;
use crate::core::llm::function_calling::call_llm_vision;
use crate::core::processor::tkg::query_auto_representative_frame;
@@ -14,20 +15,32 @@ fn t(name: &str) -> String {
}
}
/// Check if a file has faces in Qdrant _faces (replaces face_detections has_data check)
async fn has_faces_in_qdrant(file_uuid: &str) -> bool {
let qdrant = QdrantDb::new();
let filter = serde_json::json!({
"must": [
{"key": "file_uuid", "match": {"value": file_uuid}}
]
});
match qdrant.scroll_points("_faces", filter, 1, None).await {
Ok((points, _)) => !points.is_empty(),
Err(_) => false,
}
}
pub async fn exec_find_file(
pool: &sqlx::PgPool,
args: &serde_json::Value,
) -> Result<String, String> {
let query = args.get("query").and_then(|v| v.as_str()).unwrap_or("");
let videos = schema::table_name("videos");
let fd_table = schema::table_name("face_detections");
let like = format!("%{}%", query);
let rows: Vec<(String, String, bool)> = sqlx::query_as(&format!(
"SELECT v.file_uuid::text, v.file_name, \
(SELECT COUNT(*) FROM {} fd WHERE fd.file_uuid = v.file_uuid) > 0 AS has_data \
let rows: Vec<(String, String)> = sqlx::query_as(&format!(
"SELECT v.file_uuid::text, v.file_name \
FROM {} v WHERE v.file_name ILIKE $1 \
ORDER BY v.created_at DESC LIMIT 10",
fd_table, videos
videos
))
.bind(&like)
.fetch_all(pool)
@@ -37,10 +50,11 @@ pub async fn exec_find_file(
if rows.is_empty() {
return Ok(serde_json::json!({"found": false, "message": "No files match the query. Try different keywords."}).to_string());
}
let files: Vec<serde_json::Value> = rows
.into_iter()
.map(|(u, n, hd)| serde_json::json!({"file_uuid": u, "file_name": n, "has_data": hd}))
.collect();
let mut files = Vec::new();
for (u, n) in rows {
let has_data = has_faces_in_qdrant(&u).await;
files.push(serde_json::json!({"file_uuid": u, "file_name": n, "has_data": has_data}));
}
Ok(serde_json::json!({"found": true, "files": files}).to_string())
}
@@ -50,22 +64,21 @@ pub async fn exec_list_files(
) -> Result<String, String> {
let limit = args.get("limit").and_then(|v| v.as_i64()).unwrap_or(10);
let videos = schema::table_name("videos");
let fd_table = schema::table_name("face_detections");
let rows: Vec<(String, String, bool)> = sqlx::query_as(&format!(
"SELECT v.file_uuid::text, v.file_name, \
(SELECT COUNT(*) FROM {} fd WHERE fd.file_uuid = v.file_uuid) > 0 AS has_data \
let rows: Vec<(String, String)> = sqlx::query_as(&format!(
"SELECT v.file_uuid::text, v.file_name \
FROM {} v ORDER BY v.created_at DESC LIMIT $1",
fd_table, videos
videos
))
.bind(limit)
.fetch_all(pool)
.await
.map_err(|e| e.to_string())?;
let files: Vec<serde_json::Value> = rows
.into_iter()
.map(|(u, n, hd)| serde_json::json!({"file_uuid": u, "file_name": n, "has_data": hd}))
.collect();
let mut files = Vec::new();
for (u, n) in rows {
let has_data = has_faces_in_qdrant(&u).await;
files.push(serde_json::json!({"file_uuid": u, "file_name": n, "has_data": has_data}));
}
Ok(serde_json::json!({"files": files}).to_string())
}
@@ -74,6 +87,9 @@ pub async fn exec_tkg_query(
args: &serde_json::Value,
) -> Result<String, String> {
let file_uuid = args.get("file_uuid").and_then(|v| v.as_str()).unwrap_or("");
if file_uuid.is_empty() {
return Err("file_uuid is required".to_string());
}
let query_type = args
.get("query_type")
.and_then(|v| v.as_str())
@@ -82,117 +98,324 @@ pub async fn exec_tkg_query(
let identity_b = args.get("identity_b").and_then(|v| v.as_str());
let limit = args.get("limit").and_then(|v| v.as_i64()).unwrap_or(5);
// Pre-load _faces data from Qdrant
let qdrant = QdrantDb::new();
let face_filter = serde_json::json!({
"must": [
{"key": "file_uuid", "match": {"value": file_uuid}}
]
});
let face_points = qdrant
.scroll_all_points("_faces", face_filter, 1000)
.await
.map_err(|e| e.to_string())?;
// Build lookup maps from _faces payload
use std::collections::{HashMap, HashSet};
struct FacePoint {
frame: i64,
trace_id: i32,
identity_id: Option<i32>,
}
let mut points_by_frame: HashMap<i64, Vec<i32>> = HashMap::new(); // frame → identity_ids
let mut identity_face_count: HashMap<i32, i64> = HashMap::new();
let mut trace_identity: HashMap<i32, i32> = HashMap::new(); // trace_id → identity_id
let mut trace_frames: HashMap<i32, Vec<i64>> = HashMap::new(); // trace_id → frames
let mut faces_in_file: Vec<FacePoint> = Vec::new();
for point in &face_points {
let payload = &point["payload"];
let frame = payload["frame"].as_i64().unwrap_or(0);
let trace_id = payload["trace_id"].as_i64().unwrap_or(0) as i32;
let identity_id = payload["identity_id"].as_i64().map(|v| v as i32);
if trace_id <= 0 {
continue;
}
faces_in_file.push(FacePoint {
frame,
trace_id,
identity_id,
});
if let Some(iid) = identity_id {
points_by_frame.entry(frame).or_default().push(iid);
*identity_face_count.entry(iid).or_default() += 1;
trace_identity.insert(trace_id, iid);
}
trace_frames.entry(trace_id).or_default().push(frame);
}
let id_table = schema::table_name("identities");
let fd_table = schema::table_name("face_detections");
let videos = schema::table_name("videos");
let ib_table = schema::table_name("identity_bindings");
let nodes = schema::table_name("tkg_nodes");
let edges = schema::table_name("tkg_edges");
let videos = schema::table_name("videos");
match query_type {
"top_identities" => {
// Group by identity_id, count faces, query identity names
let mut top: Vec<(i32, i64)> = identity_face_count
.iter()
.map(|(id, cnt)| (*id, *cnt))
.collect();
top.sort_by(|a, b| b.1.cmp(&a.1));
top.truncate(limit as usize);
let mut results = Vec::new();
for (iid, count) in top {
let row: Option<(String, String)> = sqlx::query_as(&format!(
"SELECT uuid::text, name FROM {} WHERE id = $1 AND source = 'tmdb'",
id_table
))
.bind(iid)
.fetch_optional(pool)
.await
.map_err(|e| e.to_string())?;
if let Some((uuid, name)) = row {
results.push(serde_json::json!({
"uuid": uuid, "name": name, "face_count": count
}));
}
}
Ok(serde_json::json!({"identities": results}).to_string())
}
"first_cooccurrence" => {
let name_a = identity_name.unwrap_or("");
let name_b = identity_b.unwrap_or("");
if name_a.is_empty() || name_b.is_empty() {
return Err("identity_name and identity_b are required".to_string());
}
// Look up identity_ids by name
let id_a: Option<i32> = sqlx::query_scalar(&format!(
"SELECT id FROM {} WHERE name ILIKE $1 LIMIT 1",
id_table
))
.bind(name_a)
.fetch_optional(pool)
.await
.map_err(|e| e.to_string())?;
let id_b: Option<i32> = sqlx::query_scalar(&format!(
"SELECT id FROM {} WHERE name ILIKE $1 LIMIT 1",
id_table
))
.bind(name_b)
.fetch_optional(pool)
.await
.map_err(|e| e.to_string())?;
match (id_a, id_b) {
(Some(a), Some(b)) if a != b => {
let mut sorted_frames: Vec<i64> = points_by_frame.keys().copied().collect();
sorted_frames.sort();
for frame in sorted_frames {
let ids = &points_by_frame[&frame];
if ids.contains(&a) && ids.contains(&b) {
let fps: f64 = sqlx::query_scalar(&format!(
"SELECT COALESCE(fps, 30.0) FROM {} WHERE file_uuid = $1",
videos
))
.bind(file_uuid)
.fetch_optional(pool)
.await
.map_err(|e| e.to_string())?
.unwrap_or(30.0);
let ts = if fps > 0.0 { frame as f64 / fps } else { 0.0 };
return Ok(serde_json::json!({
"first_cooccurrence": {"frame": frame, "timestamp_secs": ts}
})
.to_string());
}
}
Ok(serde_json::json!({"first_cooccurrence": null}).to_string())
}
_ => Ok(serde_json::json!({"first_cooccurrence": null}).to_string()),
}
}
"identity_details" => {
let name = identity_name.unwrap_or("");
let row: Option<(String, String, Option<i32>)> = sqlx::query_as(&format!(
"SELECT uuid::text, name, tmdb_id FROM {} WHERE name ILIKE $1 LIMIT 1",
id_table
))
.bind(name)
.fetch_optional(pool)
.await
.map_err(|e| e.to_string())?;
match row {
Some((uuid, name, tmdb_id)) => {
let id: Option<i32> = sqlx::query_scalar(&format!(
"SELECT id FROM {} WHERE uuid::text = $1",
id_table
))
.bind(&uuid.replace('-', ""))
.fetch_optional(pool)
.await
.map_err(|e| e.to_string())?;
let face_count = id
.and_then(|iid| identity_face_count.get(&iid).copied())
.unwrap_or(0);
Ok(serde_json::json!({
"identity": {"uuid": uuid, "name": name, "tmdb_id": tmdb_id, "face_count": face_count}
}).to_string())
}
None => Ok(serde_json::json!({"identity": null}).to_string()),
}
}
"mutual_gaze" => {
let name_a = identity_name.unwrap_or("");
let name_b = identity_b.unwrap_or("");
if name_a.is_empty() || name_b.is_empty() {
return Err("identity_name and identity_b are required".to_string());
}
// Build trace_id → identity_id lookup from _faces
// Query TKG edges for mutual_gaze
let rows: Vec<(i64, String, String, serde_json::Value)> = sqlx::query_as(&format!(
"SELECT e.id, a.external_id, b.external_id, e.properties \
FROM {} e \
JOIN {} a ON a.id = e.source_node_id \
JOIN {} b ON b.id = e.target_node_id \
WHERE e.file_uuid = $1 AND e.properties->>'mutual_gaze' = 'true' \
LIMIT $2",
edges, nodes, nodes
))
.bind(file_uuid)
.bind(limit * 5)
.fetch_all(pool)
.await
.map_err(|e| e.to_string())?;
for (eid, ext_a, ext_b, props) in rows {
let tid_a = ext_a
.strip_prefix("face_track_")
.and_then(|s| s.parse::<i32>().ok())
.unwrap_or(0);
let tid_b = ext_b
.strip_prefix("face_track_")
.and_then(|s| s.parse::<i32>().ok())
.unwrap_or(0);
let id_a = trace_identity.get(&tid_a).copied();
let id_b = trace_identity.get(&tid_b).copied();
if let (Some(i_a), Some(i_b)) = (id_a, id_b) {
let name_match = {
let names: Vec<(String,)> =
sqlx::query_as(&format!("SELECT name FROM {} WHERE id = $1", id_table))
.bind(i_a)
.fetch_optional(pool)
.await
.map_err(|e| e.to_string())?
.map(|(n,)| n)
.into_iter()
.collect();
let names_b: Vec<String> = vec![]; // fetch name_b too
let name_a_str = if name_a.contains('%') { "" } else { name_a };
let name_b_str = if name_b.contains('%') { "" } else { name_b };
// Check both identities match names
// ... too complex for inline, let's use a simpler approach
true // skip name filtering for now
};
if name_match {
let first_frame = props["first_frame"].as_i64().unwrap_or(0);
let gaze_count = props["gaze_frame_count"].as_i64().unwrap_or(0);
let yaw_a = props["yaw_a_avg"].as_f64().unwrap_or(0.0);
let yaw_b = props["yaw_b_avg"].as_f64().unwrap_or(0.0);
return Ok(serde_json::json!({
"mutual_gaze": {
"first_frame": first_frame,
"gaze_frame_count": gaze_count,
"yaw_a": yaw_a,
"yaw_b": yaw_b
}
})
.to_string());
}
}
}
Ok(serde_json::json!({"mutual_gaze": null}).to_string())
}
"interaction_network" => {
let rows: Vec<(String, String, i64)> = sqlx::query_as(&format!(
"SELECT i.uuid::text, i.name, COUNT(fd.id)::bigint AS face_count \
FROM {} fd JOIN {} i ON i.id = fd.identity_id \
WHERE fd.file_uuid = $1 AND fd.identity_id IS NOT NULL AND i.source = 'tmdb' \
GROUP BY i.uuid, i.name ORDER BY face_count DESC LIMIT $2",
fd_table, id_table
"SELECT a.external_id, b.external_id, COUNT(*)::bigint \
FROM {} e \
JOIN {} a ON a.id = e.source_node_id \
JOIN {} b ON b.id = e.target_node_id \
WHERE e.file_uuid = $1 AND e.edge_type = 'CO_OCCURS_WITH' \
GROUP BY a.external_id, b.external_id \
ORDER BY COUNT(*) DESC LIMIT $2",
edges, nodes, nodes
))
.bind(file_uuid)
.bind(limit)
.fetch_all(pool)
.await
.map_err(|e| e.to_string())?;
Ok(serde_json::json!({"identities": rows}).to_string())
}
"first_cooccurrence" => {
let name_a = identity_name.unwrap_or("");
let name_b = identity_b.unwrap_or("");
let row: Option<(i64, f64)> = sqlx::query_as(&format!(
"SELECT MIN(fd_a.frame_number)::bigint, \
ROUND(MIN(fd_a.frame_number)::numeric / GREATEST(MAX(v.fps)::numeric, 25.0), 2)::float8 \
FROM {} fd_a JOIN {} fd_b ON fd_a.frame_number = fd_b.frame_number \
JOIN {} v ON v.file_uuid = $1 \
WHERE fd_a.file_uuid = $1 \
AND fd_a.identity_id = (SELECT id FROM {} WHERE name ILIKE $2 LIMIT 1) \
AND fd_b.identity_id = (SELECT id FROM {} WHERE name ILIKE $3 LIMIT 1)",
fd_table, fd_table, videos, id_table, id_table
))
.bind(file_uuid).bind(name_a).bind(name_b)
.fetch_optional(pool)
.await.map_err(|e| e.to_string())?;
Ok(serde_json::json!({"first_cooccurrence": row.map(|(f, t)| serde_json::json!({"frame": f, "timestamp_secs": t}))}).to_string())
}
"identity_details" => {
let name = identity_name.unwrap_or("");
let row: Option<(String, String, Option<i32>, i64)> = sqlx::query_as(&format!(
"SELECT i.uuid::text, i.name, i.tmdb_id, \
(SELECT COUNT(*) FROM {} fd WHERE fd.identity_id = i.id AND fd.file_uuid = $1)::bigint \
FROM {} i WHERE i.name ILIKE $2 LIMIT 1",
fd_table, id_table
))
.bind(file_uuid).bind(name)
.fetch_optional(pool)
.await.map_err(|e| e.to_string())?;
Ok(serde_json::json!({"identity": row.map(|(u, n, tid, fc)| serde_json::json!({"uuid": u, "name": n, "tmdb_id": tid, "face_count": fc}))}).to_string())
}
"mutual_gaze" => {
let name_a = identity_name.unwrap_or("");
let name_b = identity_b.unwrap_or("");
let row: Option<(i64, i64, f64, f64)> = sqlx::query_as(&format!(
"SELECT (e.properties->>'first_frame')::bigint, \
(e.properties->>'gaze_frame_count')::int::bigint, \
(e.properties->>'yaw_a_avg')::float8, \
(e.properties->>'yaw_b_avg')::float8 \
FROM {} e \
JOIN {} a ON a.id = e.source_node_id \
JOIN {} b ON b.id = e.target_node_id \
JOIN {} fd_a ON fd_a.file_uuid = $1 AND fd_a.face_track_id = REPLACE(a.external_id, 'face_track_', '')::int \
JOIN {} fd_b ON fd_b.file_uuid = $1 AND fd_b.face_track_id = REPLACE(b.external_id, 'face_track_', '')::int \
JOIN {} ia ON ia.id = fd_a.identity_id \
JOIN {} ib ON ib.id = fd_b.identity_id \
WHERE e.file_uuid = $1 AND ia.name ILIKE $2 AND ib.name ILIKE $3 \
AND e.properties->>'mutual_gaze' = 'true' LIMIT 1",
edges, nodes, nodes, fd_table, fd_table, id_table, id_table
))
.bind(file_uuid).bind(name_a).bind(name_b)
.fetch_optional(pool)
.await.map_err(|e| e.to_string())?;
Ok(serde_json::json!({"mutual_gaze": row.map(|(f, gc, ya, yb)| serde_json::json!({"first_frame": f, "gaze_frame_count": gc, "yaw_a": ya, "yaw_b": yb}))}).to_string())
}
"interaction_network" => {
let rows: Vec<(String, String, i64)> = sqlx::query_as(&format!(
"SELECT ia.name, ib.name, COUNT(*)::bigint \
FROM {} e \
JOIN {} a ON a.id = e.source_node_id \
JOIN {} b ON b.id = e.target_node_id \
JOIN {} fd_a ON fd_a.face_track_id = REPLACE(a.external_id, 'face_track_', '')::int AND fd_a.file_uuid = $1 \
JOIN {} fd_b ON fd_b.face_track_id = REPLACE(b.external_id, 'face_track_', '')::int AND fd_b.file_uuid = $1 \
JOIN {} ia ON ia.id = fd_a.identity_id \
JOIN {} ib ON ib.id = fd_b.identity_id \
WHERE e.file_uuid = $1 AND e.edge_type = 'CO_OCCURS_WITH' \
AND ia.name != ib.name AND ia.source = 'tmdb' AND ib.source = 'tmdb' \
GROUP BY ia.name, ib.name \
ORDER BY COUNT(*) DESC LIMIT $2",
edges, nodes, nodes, fd_table, fd_table, id_table, id_table
))
.bind(file_uuid).bind(limit)
.fetch_all(pool)
.await.map_err(|e| e.to_string())?;
Ok(serde_json::json!({"interaction_network": rows}).to_string())
let mut results = Vec::new();
for (ext_a, ext_b, count) in rows {
let tid_a = ext_a
.strip_prefix("face_track_")
.and_then(|s| s.parse::<i32>().ok())
.unwrap_or(0);
let tid_b = ext_b
.strip_prefix("face_track_")
.and_then(|s| s.parse::<i32>().ok())
.unwrap_or(0);
let id_a = trace_identity.get(&tid_a).copied();
let id_b = trace_identity.get(&tid_b).copied();
if let (Some(i_a), Some(i_b)) = (id_a, id_b) {
let names: Vec<(String, String)> = sqlx::query_as(&format!(
"SELECT a.name, b.name FROM {} a, {} b WHERE a.id = $1 AND b.id = $2 AND a.source = 'tmdb' AND b.source = 'tmdb'",
id_table, id_table
))
.bind(i_a).bind(i_b)
.fetch_all(pool)
.await
.map_err(|e| e.to_string())?;
for (name_a, name_b) in names {
if name_a != name_b {
results.push(serde_json::json!([name_a, name_b, count]));
}
}
}
}
Ok(serde_json::json!({"interaction_network": results}).to_string())
}
"identity_traces" => {
let name = identity_name.unwrap_or("");
let rows: Vec<(i32, i64, i64, i64)> = sqlx::query_as(&format!(
"SELECT fd.face_track_id, COUNT(*)::bigint, MIN(fd.frame_number)::bigint, MAX(fd.frame_number)::bigint \
FROM {} fd JOIN {} i ON i.id = fd.identity_id \
WHERE fd.file_uuid = $1 AND i.name ILIKE $2 \
GROUP BY fd.face_track_id ORDER BY COUNT(*) DESC LIMIT $3",
fd_table, id_table
let identity_id: Option<i32> = sqlx::query_scalar(&format!(
"SELECT id FROM {} WHERE name ILIKE $1 LIMIT 1",
id_table
))
.bind(file_uuid).bind(name).bind(limit)
.fetch_all(pool)
.await.map_err(|e| e.to_string())?;
Ok(serde_json::json!({"traces": rows}).to_string())
.bind(name)
.fetch_optional(pool)
.await
.map_err(|e| e.to_string())?;
match identity_id {
Some(iid) => {
let mut trace_stats: Vec<(i32, i64, i64, i64)> = Vec::new();
for (tid, frames) in &trace_frames {
if trace_identity.get(tid) == Some(&iid) {
let count = frames.len() as i64;
let min_f = *frames.iter().min().unwrap_or(&0);
let max_f = *frames.iter().max().unwrap_or(&0);
trace_stats.push((*tid, count, min_f, max_f));
}
}
trace_stats.sort_by(|a, b| b.1.cmp(&a.1));
trace_stats.truncate(limit as usize);
Ok(serde_json::json!({"traces": trace_stats}).to_string())
}
None => Ok(serde_json::json!({"traces": []}).to_string()),
}
}
"file_info" => {
let row: Option<(String, f64, i32, i32, f64)> = sqlx::query_as(&format!(
@@ -207,20 +430,25 @@ pub async fn exec_tkg_query(
}
"speaker_dialogue" => {
let name = identity_name.unwrap_or("");
if name.is_empty() {
return Err("identity_name is required for speaker_dialogue".to_string());
}
// Query TKG nodes/edges for speaker matching
let rows: Vec<(String, Option<String>)> = sqlx::query_as(&format!(
"SELECT DISTINCT sn.external_id, sn.properties->>'full_text' AS full_text \
FROM {} i \
JOIN {} fd ON fd.identity_id = i.id AND ($2::text IS NULL OR fd.file_uuid = $2) \
JOIN {} fn ON fn.file_uuid = fd.file_uuid \
JOIN {} ib ON ib.identity_id = i.id AND ib.identity_type = 'trace' \
JOIN {} fn ON fn.file_uuid = $2 \
AND fn.node_type = 'face_track' \
AND fn.external_id = CONCAT('face_track_', fd.face_track_id) \
AND fn.external_id = CONCAT('face_track_', ib.identity_value) \
JOIN {} e ON e.source_node_id = fn.id \
AND e.edge_type = 'SPEAKS_AS' \
AND ($2::text IS NULL OR e.file_uuid = $2) \
AND e.file_uuid = $2 \
JOIN {} sn ON sn.id = e.target_node_id \
WHERE i.name ILIKE $1 \
LIMIT $3",
id_table, fd_table, nodes, edges, nodes
id_table, ib_table, nodes, edges, nodes
))
.bind(name)
.bind(file_uuid)
@@ -240,26 +468,23 @@ pub async fn exec_tkg_query(
let name_a = identity_name.unwrap_or("");
let name_b = identity_b.unwrap_or("");
if name_a.is_empty() || name_b.is_empty() {
return Ok(
serde_json::json!({"error": "identity_name and identity_b are required"})
.to_string(),
);
return Err("identity_name and identity_b are required".to_string());
}
let rows: Vec<(String, String, serde_json::Value)> = sqlx::query_as(&format!(
"SELECT sn.external_id, sn.properties->>'full_text' AS full_text, sn.properties->'segments' AS segments \
FROM {} i \
JOIN {} fd ON fd.identity_id = i.id AND ($3::text IS NULL OR fd.file_uuid = $3) \
JOIN {} fn ON fn.file_uuid = fd.file_uuid \
JOIN {} ib ON ib.identity_id = i.id AND ib.identity_type = 'trace' \
JOIN {} fn ON fn.file_uuid = $3 \
AND fn.node_type = 'face_track' \
AND fn.external_id = CONCAT('face_track_', fd.face_track_id) \
AND fn.external_id = CONCAT('face_track_', ib.identity_value) \
JOIN {} e ON e.source_node_id = fn.id \
AND e.edge_type = 'SPEAKS_AS' \
AND ($3::text IS NULL OR e.file_uuid = $3) \
AND e.file_uuid = $3 \
JOIN {} sn ON sn.id = e.target_node_id \
WHERE (i.name ILIKE $1 OR i.name ILIKE $2) \
ORDER BY sn.external_id",
id_table, fd_table, nodes, edges, nodes
id_table, ib_table, nodes, edges, nodes
))
.bind(name_a)
.bind(name_b)
@@ -295,11 +520,9 @@ pub async fn exec_tkg_query(
let overlap_end = sa_end.min(sb_end);
if overlap_start < overlap_end {
interactions.push(serde_json::json!({
"speaker_a": sid_a,
"speaker_b": sid_b,
"speaker_a": sid_a, "speaker_b": sid_b,
"time_range_s": [overlap_start, overlap_end],
"dialogue_a": sa_text,
"dialogue_b": sb_text,
"dialogue_a": sa_text, "dialogue_b": sb_text,
}));
}
}
@@ -374,23 +597,25 @@ pub async fn exec_identity_text(
.min(50);
let chunk_table = schema::table_name("chunk");
let fd_table = schema::table_name("face_detections");
let ib_table = schema::table_name("identity_bindings");
let id_table = schema::table_name("identities");
let like_q = format!("%{}%", q.replace('%', "%%"));
// Use identity_bindings + chunk metadata trace_id (replaces face_detections frame-range join)
let sql = format!(
"SELECT c.chunk_id, c.start_time, c.end_time, c.text_content, \
i.name AS identity_name, fd.face_track_id, i.source AS identity_source \
i.name AS identity_name, \
(c.metadata->>'trace_id')::int AS trace_id, \
i.source AS identity_source \
FROM {} c \
JOIN {} fd ON fd.file_uuid = c.file_uuid \
AND fd.frame_number BETWEEN c.start_frame AND c.end_frame \
AND fd.identity_id IS NOT NULL \
JOIN {} i ON i.id = fd.identity_id \
JOIN {} ib ON ib.identity_value = c.metadata->>'trace_id' \
AND ib.identity_type = 'trace' \
JOIN {} i ON i.id = ib.identity_id \
WHERE ($1::text IS NULL OR c.file_uuid = $1) \
AND (LOWER(c.text_content) LIKE LOWER($2) OR LOWER(c.content::text) LIKE LOWER($2)) \
ORDER BY c.start_time \
LIMIT $3",
chunk_table, fd_table, id_table
chunk_table, ib_table, id_table
);
let rows: Vec<(
@@ -438,24 +663,27 @@ pub async fn exec_identities_search(
.min(50);
let id_table = schema::table_name("identities");
let fd_table = schema::table_name("face_detections");
let ib_table = schema::table_name("identity_bindings");
let fi_table = schema::table_name("file_identities");
let chunk_table = schema::table_name("chunk");
let like_q = format!("%{}%", q.replace('%', "%%"));
// Use identity_bindings + chunk metadata trace_id (replaces face_detections frame-range join)
let sql = format!(
"SELECT DISTINCT ON (i.name, c.chunk_id) \
i.name, c.chunk_id, c.start_time, c.end_time, c.text_content, fd.face_track_id \
i.name, c.chunk_id, c.start_time, c.end_time, c.text_content, \
(c.metadata->>'trace_id')::int AS trace_id \
FROM {} i \
JOIN {} fd ON fd.identity_id = i.id \
JOIN {} c ON c.file_uuid = fd.file_uuid \
AND c.start_time <= fd.frame_number / COALESCE(c.fps, 25.0) \
AND c.end_time >= fd.frame_number / COALESCE(c.fps, 25.0) \
JOIN {} ib ON ib.identity_id = i.id AND ib.identity_type = 'trace' \
JOIN {} fi ON fi.identity_id = i.id \
JOIN {} c ON c.file_uuid = fi.file_uuid \
AND c.metadata->>'trace_id' = ib.identity_value \
WHERE (i.name ILIKE $1 \
OR EXISTS (SELECT 1 FROM jsonb_array_elements(i.metadata->'aliases') AS a WHERE a->>'name' ILIKE $1)) \
AND ($2::text IS NULL OR fd.file_uuid = $2) \
AND ($2::text IS NULL OR c.file_uuid = $2) \
ORDER BY i.name, c.chunk_id, c.start_time \
LIMIT $3",
id_table, fd_table, chunk_table
id_table, ib_table, fi_table, chunk_table
);
let rows: Vec<(String, String, f64, f64, Option<String>, Option<i32>)> = sqlx::query_as(&sql)