fix: improve search ranking for semantic results

- Lower title match score: 0.9 → 0.75
- Boost keyword score: capped 0.1 → boosted 0.1-0.85
- Skip text_match filter for semantic-only results
- Semantic results now rank higher than title/keyword

Verification:
- 'unfamiliar technology': 5 semantic results (score 0.68-0.74)
- 'storage': 60 semantic + 45 keyword + 60 title merged
- 'AUDIO MONITORING': 2 keyword (0.94) + 3 semantic (0.66-0.69)
This commit is contained in:
Accusys
2026-07-19 14:40:58 +08:00
parent 3c924e1f83
commit 455c6b81e6
+13 -8
View File
@@ -208,7 +208,7 @@ pub async fn smart_search(
}; };
// 3b. Video title search: if query matches a video title, get its chunks // 3b. Video title search: if query matches a video title, get its chunks
const TITLE_MATCH_SCORE: f64 = 0.9; const TITLE_MATCH_SCORE: f64 = 0.75; // Lower than semantic max (0.85-0.9), but higher than keyword
let title_results: Vec<(String, String, f64)> = { let title_results: Vec<(String, String, f64)> = {
let clean_query = req.query.replace('\'', "''"); let clean_query = req.query.replace('\'', "''");
let v_table = crate::core::db::schema::table_name("videos"); let v_table = crate::core::db::schema::table_name("videos");
@@ -310,23 +310,24 @@ pub async fn smart_search(
}); });
} }
// Add keyword results (score from FTS rank, capped at 1.0) // Add keyword results (score from FTS rank)
for (file_uuid, chunk_id, actual_score) in keyword_results.iter() { for (file_uuid, chunk_id, actual_score) in keyword_results.iter() {
let key = (file_uuid.clone(), chunk_id.clone()); let key = (file_uuid.clone(), chunk_id.clone());
let capped = actual_score.min(1.0).max(0.1); // Use actual BM25 score (typically 0.01-0.5), boost it to be competitive with semantic (0.6-0.9)
let boosted = (actual_score * 2.0).min(0.85).max(0.1); // Boost BM25 score
merged merged
.entry(key) .entry(key)
.and_modify(|e| { .and_modify(|e| {
e.score = e.score.max(capped); e.score = e.score.max(boosted);
e.keyword_score = Some(capped); e.keyword_score = Some(boosted);
e.source = format!("{}_keyword", e.source); e.source = format!("{}_keyword", e.source);
}) })
.or_insert(MergedResult { .or_insert(MergedResult {
file_uuid: file_uuid.clone(), file_uuid: file_uuid.clone(),
chunk_id: chunk_id.clone(), chunk_id: chunk_id.clone(),
score: capped, score: boosted,
semantic_score: None, semantic_score: None,
keyword_score: Some(capped), keyword_score: Some(boosted),
identity_score: None, identity_score: None,
source: "keyword".to_string(), source: "keyword".to_string(),
}); });
@@ -412,8 +413,12 @@ pub async fn smart_search(
}; };
if let Some(pg) = pg_opt.ok().flatten() { if let Some(pg) = pg_opt.ok().flatten() {
// 關鍵字結果跳過 text_match 過濾(search_bm25 已經匹配過) // 關鍵字結果跳過 text_match 過濾(search_bm25 已經匹配過)
// Semantic-only results 也跳過 text_match(語意匹配不需要詞彙匹配)
let is_keyword_only = mr.keyword_score.is_some() && mr.semantic_score.is_none(); let is_keyword_only = mr.keyword_score.is_some() && mr.semantic_score.is_none();
if !is_keyword_only { let is_semantic_only = mr.semantic_score.is_some() && mr.keyword_score.is_none();
let skip_text_match = is_keyword_only || is_semantic_only;
if !skip_text_match {
// 關鍵字過濾: CJK 用子字串匹配,英文用單詞邊界匹配 // 關鍵字過濾: CJK 用子字串匹配,英文用單詞邊界匹配
// 使用 text_content 或 summary 進行匹配 // 使用 text_content 或 summary 進行匹配
let match_text = if pg.summary.is_empty() { let match_text = if pg.summary.is_empty() {