feat: score-based search, LLM re-ranking endpoint, video title search, pipeline module

Core search changes:
- Replace RRF with score-based merge (max of semantic/keyword/identity)
- Add video title ILIKE search for brand/name queries (score 0.9)
- Add /api/v1/search/llm-smart endpoint with Gemma 4 re-ranking
- Fix LLM JSON parsing (markdown fences, empty responses)

Infrastructure:
- Rebuild Qdrant collection (clear 347K contaminated points)
- Add dotenv loading to main.rs for config parity
- Implement store_pre_chunk in postgres_db.rs

Pipeline module (WordPress):
- store-asrx, rule1, vectorize, phase1, complete endpoints
- CLI commands for pipeline operations

Docs:
- SEARCH_SCORE_IMPROVEMENT.md (score-based merge proposal)
This commit is contained in:
Accusys
2026-06-04 07:40:41 +08:00
parent e1572907ae
commit 834b0d4865
14 changed files with 835 additions and 31 deletions
+172
View File
@@ -0,0 +1,172 @@
use anyhow::{Context, Result};
use crate::core::chunk::rule1_ingest;
use crate::core::config;
use crate::core::db::postgres_db::PostgresDb;
use crate::core::db::qdrant_db::QdrantDb;
use crate::core::db::schema;
use crate::core::db::VectorPayload;
use crate::core::embedding::Embedder;
use crate::core::processor::asrx::AsrxResult;
use crate::core::processor::PythonExecutor;
use crate::core::storage::output_dir::OutputDir;
pub async fn store_asrx_chunks(db: &PostgresDb, uuid: &str) -> Result<()> {
let output_dir = OutputDir::new();
let asrx_path = output_dir.get_output_path(uuid, "asrx.json");
let json_str = std::fs::read_to_string(&asrx_path)
.with_context(|| format!("ASRX file not found: {:?}", asrx_path))?;
let result: AsrxResult = serde_json::from_str(&json_str)
.context("Failed to parse ASRX JSON")?;
let segments_count = result.segments.len();
let mut pre_chunks = Vec::new();
let mut speaker_detections = Vec::new();
for (i, segment) in result.segments.iter().enumerate() {
let data = serde_json::json!({
"text": segment.text,
"speaker_id": segment.speaker_id,
"timestamp": segment.start_time,
});
pre_chunks.push((i as i64, Some(segment.start_time), data, None, None));
speaker_detections.push((
segment.speaker_id.clone().unwrap_or_default(),
segment.start_time,
segment.end_time,
segment.text.clone(),
None::<String>,
0.0,
));
}
db.store_raw_pre_chunks_batch(uuid, "asrx", &pre_chunks).await?;
db.store_raw_pre_chunks_batch(uuid, "asr", &pre_chunks).await?;
db.store_speaker_detections_batch(uuid, &speaker_detections).await?;
println!("Stored {} ASRX pre-chunks for {}", segments_count, uuid);
Ok(())
}
pub async fn execute_rule1(db: &PostgresDb, uuid: &str) -> Result<usize> {
let video = db.get_video_by_uuid(uuid)
.await?
.context("Video not found")?;
let fps = video.fps;
let count = rule1_ingest::execute_rule1(db, uuid, fps).await
.context("Rule 1 ingestion failed")?;
println!("Rule 1 completed: {} chunks inserted for {}", count, uuid);
Ok(count)
}
pub async fn vectorize_chunks(uuid: &str) -> Result<()> {
let db = PostgresDb::new(&config::DATABASE_URL).await?;
let qdrant = QdrantDb::new();
let embedder = Embedder::new("embeddinggemma-300m".to_string());
let chunk_table = schema::table_name("chunk");
let rows = sqlx::query_as::<_, (String, String, String, i64, i64, f64, f64, String)>(
&format!(
"SELECT chunk_id, chunk_type, text_content, start_frame, end_frame, \
start_time, end_time, content::text \
FROM {} WHERE file_uuid = $1 AND chunk_type = 'sentence' \
AND embedding IS NULL \
AND (text_content IS NOT NULL AND text_content != '') \
ORDER BY id",
chunk_table
),
)
.bind(uuid)
.fetch_all(db.pool())
.await?;
if rows.is_empty() {
println!("No sentence chunks to vectorize for {}", uuid);
return Ok(());
}
let total = rows.len();
let mut stored = 0usize;
for (chunk_id, _chunk_type, text, start_frame, end_frame, start_time, end_time, _content_str) in &rows {
if text.is_empty() {
continue;
}
match embedder.embed_document(text).await {
Ok(vector) => {
if let Err(e) = db.store_vector(chunk_id, &vector, uuid).await {
eprintln!("PG store failed for {}: {}", chunk_id, e);
continue;
}
let payload = VectorPayload {
file_uuid: uuid.to_string(),
chunk_id: chunk_id.clone(),
chunk_type: "sentence".to_string(),
start_frame: *start_frame,
end_frame: *end_frame,
start_time: *start_time,
end_time: *end_time,
text: Some(text.clone()),
};
if let Err(e) = qdrant.upsert_vector(chunk_id, &vector, payload).await {
eprintln!("Qdrant upsert failed for {}: {}", chunk_id, e);
continue;
}
stored += 1;
if stored % 50 == 0 {
println!("Vectorized {}/{} chunks for {}", stored, total, uuid);
}
}
Err(e) => {
eprintln!("Embedding failed for {}: {}", chunk_id, e);
}
}
}
println!("Vectorization complete: {}/{} vectors for {}", stored, total, uuid);
Ok(())
}
pub async fn run_phase1(uuid: &str) -> Result<()> {
let executor = PythonExecutor::new()
.context("Failed to create PythonExecutor")?;
executor
.run(
"release_pack.py",
&["--phase", "1", "--file-uuid", uuid],
None,
"RELEASE_P1",
Some(std::time::Duration::from_secs(120)),
)
.await
.context("Phase 1 release pack failed")?;
println!("Phase 1 release packaged for {}", uuid);
Ok(())
}
pub async fn mark_complete(db: &PostgresDb, uuid: &str) -> Result<()> {
use crate::core::db::MonitorJobStatus;
use crate::core::db::VideoStatus;
let job_id = sqlx::query_scalar::<_, i32>(
&format!("SELECT id FROM {} WHERE uuid = $1 LIMIT 1", schema::table_name("monitor_jobs")),
)
.bind(uuid)
.fetch_optional(db.pool())
.await?;
if let Some(job_id) = job_id {
db.update_job_status(job_id, MonitorJobStatus::Completed).await?;
println!("Job {} marked as completed", job_id);
}
db.update_video_status(uuid, VideoStatus::Completed).await?;
println!("Video {} marked as completed", uuid);
Ok(())
}