From 5e83ee7dac740064d28187657a0208e166f32635 Mon Sep 17 00:00:00 2001 From: Accusys Date: Sun, 19 Jul 2026 13:33:31 +0800 Subject: [PATCH] fix: keyword search for OCR text MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Fix OCR confidence threshold: 0.5 → 0.2 - Fix fetch_ocr_texts frame calculation (use start_frame from DB) - Fix text_match filter to use text_content instead of summary - Process all merged results (not just top 30) to include keyword results - Add logging for keyword search debugging Fixes issue where OCR text like 'AUDIO MONITORING' and 'thunderbolt' could not be found via keyword search. --- src/api/search.rs | 38 +++++++++++++++++++++++++++------- src/core/chunk/rule1_ingest.rs | 5 ++--- 2 files changed, 33 insertions(+), 10 deletions(-) diff --git a/src/api/search.rs b/src/api/search.rs index d9abbbe..095e660 100644 --- a/src/api/search.rs +++ b/src/api/search.rs @@ -191,10 +191,16 @@ pub async fn smart_search( .search_bm25(&req.query, req.file_uuid.as_deref(), fetch_limit as i64) .await { - Ok(rows) => rows - .into_iter() - .map(|r| (r.file_uuid, r.chunk_id, r.combined_score)) - .collect(), + Ok(rows) => { + tracing::info!( + "Smart search: Keyword search returned {} hits for '{}'", + rows.len(), + req.query + ); + rows.into_iter() + .map(|r| (r.file_uuid, r.chunk_id, r.combined_score)) + .collect() + } Err(e) => { tracing::warn!("Keyword search (bm25) failed: {}", e); vec![] @@ -379,10 +385,22 @@ pub async fn smart_search( .unwrap_or(std::cmp::Ordering::Equal) }); + tracing::info!( + "Smart search: Merged {} results (semantic={}, keyword={}, title={}, identity={})", + merged.len(), + semantic_results.len(), + keyword_results.len(), + title_results.len(), + identity_results.len() + ); + // 6. Enrich top results from PG and build final response let query_lower = req.query.to_lowercase(); let mut final_results = Vec::new(); - for mr in ranked.iter().take(limit * 3) { + // Take all merged results to ensure keyword results are processed + // (keyword results often have low scores and rank last) + let take_count = ranked.len(); + for mr in ranked.iter().take(take_count) { // 取更多結果以便過濾 // Use no_embedding version for keyword results, regular for semantic let pg_opt = if mr.keyword_score.is_some() && mr.semantic_score.is_none() { @@ -397,13 +415,19 @@ pub async fn smart_search( let is_keyword_only = mr.keyword_score.is_some() && mr.semantic_score.is_none(); if !is_keyword_only { // 關鍵字過濾: CJK 用子字串匹配,英文用單詞邊界匹配 - let summary_lower = pg.summary.to_lowercase(); + // 使用 text_content 或 summary 進行匹配 + let match_text = if pg.summary.is_empty() { + pg.text_content.clone().unwrap_or_default() + } else { + pg.summary.clone() + }; + let summary_lower = match_text.to_lowercase(); let query_words: Vec = query_lower .split_whitespace() .map(|s| s.to_string()) .collect(); - let text_match = !pg.summary.is_empty() && { + let text_match = !match_text.is_empty() && { let has_cjk = |s: &str| -> bool { s.chars().any(|c| { ('\u{4E00}'..='\u{9FFF}').contains(&c) diff --git a/src/core/chunk/rule1_ingest.rs b/src/core/chunk/rule1_ingest.rs index 847b499..3117cf5 100644 --- a/src/core/chunk/rule1_ingest.rs +++ b/src/core/chunk/rule1_ingest.rs @@ -7,7 +7,7 @@ use sqlx::{PgPool, Row}; use std::collections::BTreeMap; use tracing::{info, warn}; -const OCR_CONFIDENCE_THRESHOLD: f64 = 0.5; +const OCR_CONFIDENCE_THRESHOLD: f64 = 0.2; pub async fn execute_rule1(db: &PostgresDb, file_uuid: &str, fps: f64) -> Result { let pool = db.pool(); @@ -219,11 +219,10 @@ async fn fetch_ocr_texts( let mut map: BTreeMap> = BTreeMap::new(); for row in rows { - let start_time: f64 = row.try_get("start_time").unwrap_or(0.0); + let start_frame: i64 = row.try_get("start_frame").unwrap_or(0); let end_time_raw: Option = row.try_get("end_time").ok(); let data: Value = row.try_get("data").unwrap_or(Value::Null); - let start_frame = (start_time * fps) as i64; let end_frame = end_time_raw .filter(|t| *t > 0.0) .map(|t| (t * fps) as i64)