fix: keyword search for OCR text
- Fix OCR confidence threshold: 0.5 → 0.2 - Fix fetch_ocr_texts frame calculation (use start_frame from DB) - Fix text_match filter to use text_content instead of summary - Process all merged results (not just top 30) to include keyword results - Add logging for keyword search debugging Fixes issue where OCR text like 'AUDIO MONITORING' and 'thunderbolt' could not be found via keyword search.
This commit is contained in:
+31
-7
@@ -191,10 +191,16 @@ pub async fn smart_search(
|
||||
.search_bm25(&req.query, req.file_uuid.as_deref(), fetch_limit as i64)
|
||||
.await
|
||||
{
|
||||
Ok(rows) => rows
|
||||
.into_iter()
|
||||
.map(|r| (r.file_uuid, r.chunk_id, r.combined_score))
|
||||
.collect(),
|
||||
Ok(rows) => {
|
||||
tracing::info!(
|
||||
"Smart search: Keyword search returned {} hits for '{}'",
|
||||
rows.len(),
|
||||
req.query
|
||||
);
|
||||
rows.into_iter()
|
||||
.map(|r| (r.file_uuid, r.chunk_id, r.combined_score))
|
||||
.collect()
|
||||
}
|
||||
Err(e) => {
|
||||
tracing::warn!("Keyword search (bm25) failed: {}", e);
|
||||
vec![]
|
||||
@@ -379,10 +385,22 @@ pub async fn smart_search(
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
});
|
||||
|
||||
tracing::info!(
|
||||
"Smart search: Merged {} results (semantic={}, keyword={}, title={}, identity={})",
|
||||
merged.len(),
|
||||
semantic_results.len(),
|
||||
keyword_results.len(),
|
||||
title_results.len(),
|
||||
identity_results.len()
|
||||
);
|
||||
|
||||
// 6. Enrich top results from PG and build final response
|
||||
let query_lower = req.query.to_lowercase();
|
||||
let mut final_results = Vec::new();
|
||||
for mr in ranked.iter().take(limit * 3) {
|
||||
// Take all merged results to ensure keyword results are processed
|
||||
// (keyword results often have low scores and rank last)
|
||||
let take_count = ranked.len();
|
||||
for mr in ranked.iter().take(take_count) {
|
||||
// 取更多結果以便過濾
|
||||
// Use no_embedding version for keyword results, regular for semantic
|
||||
let pg_opt = if mr.keyword_score.is_some() && mr.semantic_score.is_none() {
|
||||
@@ -397,13 +415,19 @@ pub async fn smart_search(
|
||||
let is_keyword_only = mr.keyword_score.is_some() && mr.semantic_score.is_none();
|
||||
if !is_keyword_only {
|
||||
// 關鍵字過濾: CJK 用子字串匹配,英文用單詞邊界匹配
|
||||
let summary_lower = pg.summary.to_lowercase();
|
||||
// 使用 text_content 或 summary 進行匹配
|
||||
let match_text = if pg.summary.is_empty() {
|
||||
pg.text_content.clone().unwrap_or_default()
|
||||
} else {
|
||||
pg.summary.clone()
|
||||
};
|
||||
let summary_lower = match_text.to_lowercase();
|
||||
let query_words: Vec<String> = query_lower
|
||||
.split_whitespace()
|
||||
.map(|s| s.to_string())
|
||||
.collect();
|
||||
|
||||
let text_match = !pg.summary.is_empty() && {
|
||||
let text_match = !match_text.is_empty() && {
|
||||
let has_cjk = |s: &str| -> bool {
|
||||
s.chars().any(|c| {
|
||||
('\u{4E00}'..='\u{9FFF}').contains(&c)
|
||||
|
||||
@@ -7,7 +7,7 @@ use sqlx::{PgPool, Row};
|
||||
use std::collections::BTreeMap;
|
||||
use tracing::{info, warn};
|
||||
|
||||
const OCR_CONFIDENCE_THRESHOLD: f64 = 0.5;
|
||||
const OCR_CONFIDENCE_THRESHOLD: f64 = 0.2;
|
||||
|
||||
pub async fn execute_rule1(db: &PostgresDb, file_uuid: &str, fps: f64) -> Result<usize> {
|
||||
let pool = db.pool();
|
||||
@@ -219,11 +219,10 @@ async fn fetch_ocr_texts(
|
||||
|
||||
let mut map: BTreeMap<i64, Vec<String>> = BTreeMap::new();
|
||||
for row in rows {
|
||||
let start_time: f64 = row.try_get("start_time").unwrap_or(0.0);
|
||||
let start_frame: i64 = row.try_get("start_frame").unwrap_or(0);
|
||||
let end_time_raw: Option<f64> = row.try_get("end_time").ok();
|
||||
let data: Value = row.try_get("data").unwrap_or(Value::Null);
|
||||
|
||||
let start_frame = (start_time * fps) as i64;
|
||||
let end_frame = end_time_raw
|
||||
.filter(|t| *t > 0.0)
|
||||
.map(|t| (t * fps) as i64)
|
||||
|
||||
Reference in New Issue
Block a user