fix: keyword search for OCR text
- Fix OCR confidence threshold: 0.5 → 0.2 - Fix fetch_ocr_texts frame calculation (use start_frame from DB) - Fix text_match filter to use text_content instead of summary - Process all merged results (not just top 30) to include keyword results - Add logging for keyword search debugging Fixes issue where OCR text like 'AUDIO MONITORING' and 'thunderbolt' could not be found via keyword search.
This commit is contained in:
+31
-7
@@ -191,10 +191,16 @@ pub async fn smart_search(
|
|||||||
.search_bm25(&req.query, req.file_uuid.as_deref(), fetch_limit as i64)
|
.search_bm25(&req.query, req.file_uuid.as_deref(), fetch_limit as i64)
|
||||||
.await
|
.await
|
||||||
{
|
{
|
||||||
Ok(rows) => rows
|
Ok(rows) => {
|
||||||
.into_iter()
|
tracing::info!(
|
||||||
.map(|r| (r.file_uuid, r.chunk_id, r.combined_score))
|
"Smart search: Keyword search returned {} hits for '{}'",
|
||||||
.collect(),
|
rows.len(),
|
||||||
|
req.query
|
||||||
|
);
|
||||||
|
rows.into_iter()
|
||||||
|
.map(|r| (r.file_uuid, r.chunk_id, r.combined_score))
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
tracing::warn!("Keyword search (bm25) failed: {}", e);
|
tracing::warn!("Keyword search (bm25) failed: {}", e);
|
||||||
vec![]
|
vec![]
|
||||||
@@ -379,10 +385,22 @@ pub async fn smart_search(
|
|||||||
.unwrap_or(std::cmp::Ordering::Equal)
|
.unwrap_or(std::cmp::Ordering::Equal)
|
||||||
});
|
});
|
||||||
|
|
||||||
|
tracing::info!(
|
||||||
|
"Smart search: Merged {} results (semantic={}, keyword={}, title={}, identity={})",
|
||||||
|
merged.len(),
|
||||||
|
semantic_results.len(),
|
||||||
|
keyword_results.len(),
|
||||||
|
title_results.len(),
|
||||||
|
identity_results.len()
|
||||||
|
);
|
||||||
|
|
||||||
// 6. Enrich top results from PG and build final response
|
// 6. Enrich top results from PG and build final response
|
||||||
let query_lower = req.query.to_lowercase();
|
let query_lower = req.query.to_lowercase();
|
||||||
let mut final_results = Vec::new();
|
let mut final_results = Vec::new();
|
||||||
for mr in ranked.iter().take(limit * 3) {
|
// Take all merged results to ensure keyword results are processed
|
||||||
|
// (keyword results often have low scores and rank last)
|
||||||
|
let take_count = ranked.len();
|
||||||
|
for mr in ranked.iter().take(take_count) {
|
||||||
// 取更多結果以便過濾
|
// 取更多結果以便過濾
|
||||||
// Use no_embedding version for keyword results, regular for semantic
|
// Use no_embedding version for keyword results, regular for semantic
|
||||||
let pg_opt = if mr.keyword_score.is_some() && mr.semantic_score.is_none() {
|
let pg_opt = if mr.keyword_score.is_some() && mr.semantic_score.is_none() {
|
||||||
@@ -397,13 +415,19 @@ pub async fn smart_search(
|
|||||||
let is_keyword_only = mr.keyword_score.is_some() && mr.semantic_score.is_none();
|
let is_keyword_only = mr.keyword_score.is_some() && mr.semantic_score.is_none();
|
||||||
if !is_keyword_only {
|
if !is_keyword_only {
|
||||||
// 關鍵字過濾: CJK 用子字串匹配,英文用單詞邊界匹配
|
// 關鍵字過濾: CJK 用子字串匹配,英文用單詞邊界匹配
|
||||||
let summary_lower = pg.summary.to_lowercase();
|
// 使用 text_content 或 summary 進行匹配
|
||||||
|
let match_text = if pg.summary.is_empty() {
|
||||||
|
pg.text_content.clone().unwrap_or_default()
|
||||||
|
} else {
|
||||||
|
pg.summary.clone()
|
||||||
|
};
|
||||||
|
let summary_lower = match_text.to_lowercase();
|
||||||
let query_words: Vec<String> = query_lower
|
let query_words: Vec<String> = query_lower
|
||||||
.split_whitespace()
|
.split_whitespace()
|
||||||
.map(|s| s.to_string())
|
.map(|s| s.to_string())
|
||||||
.collect();
|
.collect();
|
||||||
|
|
||||||
let text_match = !pg.summary.is_empty() && {
|
let text_match = !match_text.is_empty() && {
|
||||||
let has_cjk = |s: &str| -> bool {
|
let has_cjk = |s: &str| -> bool {
|
||||||
s.chars().any(|c| {
|
s.chars().any(|c| {
|
||||||
('\u{4E00}'..='\u{9FFF}').contains(&c)
|
('\u{4E00}'..='\u{9FFF}').contains(&c)
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ use sqlx::{PgPool, Row};
|
|||||||
use std::collections::BTreeMap;
|
use std::collections::BTreeMap;
|
||||||
use tracing::{info, warn};
|
use tracing::{info, warn};
|
||||||
|
|
||||||
const OCR_CONFIDENCE_THRESHOLD: f64 = 0.5;
|
const OCR_CONFIDENCE_THRESHOLD: f64 = 0.2;
|
||||||
|
|
||||||
pub async fn execute_rule1(db: &PostgresDb, file_uuid: &str, fps: f64) -> Result<usize> {
|
pub async fn execute_rule1(db: &PostgresDb, file_uuid: &str, fps: f64) -> Result<usize> {
|
||||||
let pool = db.pool();
|
let pool = db.pool();
|
||||||
@@ -219,11 +219,10 @@ async fn fetch_ocr_texts(
|
|||||||
|
|
||||||
let mut map: BTreeMap<i64, Vec<String>> = BTreeMap::new();
|
let mut map: BTreeMap<i64, Vec<String>> = BTreeMap::new();
|
||||||
for row in rows {
|
for row in rows {
|
||||||
let start_time: f64 = row.try_get("start_time").unwrap_or(0.0);
|
let start_frame: i64 = row.try_get("start_frame").unwrap_or(0);
|
||||||
let end_time_raw: Option<f64> = row.try_get("end_time").ok();
|
let end_time_raw: Option<f64> = row.try_get("end_time").ok();
|
||||||
let data: Value = row.try_get("data").unwrap_or(Value::Null);
|
let data: Value = row.try_get("data").unwrap_or(Value::Null);
|
||||||
|
|
||||||
let start_frame = (start_time * fps) as i64;
|
|
||||||
let end_frame = end_time_raw
|
let end_frame = end_time_raw
|
||||||
.filter(|t| *t > 0.0)
|
.filter(|t| *t > 0.0)
|
||||||
.map(|t| (t * fps) as i64)
|
.map(|t| (t * fps) as i64)
|
||||||
|
|||||||
Reference in New Issue
Block a user