From ff9978f74ca195e8205daffc73c0faed91adf975 Mon Sep 17 00:00:00 2001 From: Max Freedom Pollard <272618364+MaxFreedomPollard@users.noreply.github.com> Date: Mon, 7 Sep 2026 05:56:57 -0400 Subject: [PATCH] fix(search): stop a repeated URL from faking cross-engine agreement `WebSearch::search` fuses the engines with Reciprocal Rank Fusion, which sums one term per ranked list. `merge` added a term for every result instead, so an engine that listed the same canonical URL twice had both of its ranks counted: 1/61 + 1/62 = 0.0325 That is what two independent engines agreeing at rank 1 are worth (2/61 = 0.0328), produced from a single list. Agreement across engines is the only ranking signal this federation has, and a repeat inside one engine forges it. The repeats come from the repository's own canonicalization, not from exotic input. `canonicalize` in `search/engine.rs` drops the fragment, the trailing slash and the `utm_*`, `gclid`, `fbclid` and `mc_*` parameters, and `redirected_target` unwraps the Bing, DuckDuckGo and Google redirector links, so rows that are visibly distinct on one result page collapse onto one URL. Nothing dedupes an engine's own list before `merge` sees it. `merge` already knew the rule: the engine label was guarded with `!existing.engines.contains(&engine)`. Put the score behind the same guard, so each engine contributes its best rank once. Cross-engine merging and the longest-snippet rule are untouched. --- server/src/search/federation.rs | 83 ++++++++++++++++++++++++++++++++- 1 file changed, 82 insertions(+), 1 deletion(-) diff --git a/server/src/search/federation.rs b/server/src/search/federation.rs index e7bdbd3..94b79bc 100644 --- a/server/src/search/federation.rs +++ b/server/src/search/federation.rs @@ -122,8 +122,12 @@ fn merge(merged: &mut HashMap, engine: &'static str, results: let score = 1.0 / (RRF_K + rank as f64 + 1.0); match merged.get_mut(&result.url) { Some(existing) => { - existing.score += score; + // Reciprocal Rank Fusion sums one term per ranked list. One + // engine listing a URL more than once is still one list, so + // only its best rank counts; a later repeat contributes + // nothing but its snippet. if !existing.engines.contains(&engine) { + existing.score += score; existing.engines.push(engine); } if result.chunk.len() > existing.chunk.len() { @@ -137,3 +141,80 @@ fn merge(merged: &mut HashMap, engine: &'static str, results: } } } + +#[cfg(test)] +mod tests { + use super::*; + + fn hit(url: &str, engine: &'static str, chunk: &str) -> SearchHit { + SearchHit::new("title", url, chunk, vec![engine]) + } + + fn rrf(rank: usize) -> f64 { + 1.0 / (RRF_K + rank as f64 + 1.0) + } + + #[test] + fn an_engine_that_lists_one_url_twice_contributes_a_single_rrf_term() { + let mut merged = HashMap::new(); + merge( + &mut merged, + "duckduckgo", + vec![ + hit("https://example.com/p", "duckduckgo", "short"), + hit("https://example.com/p", "duckduckgo", "longer snippet"), + ], + ); + + let entry = &merged["https://example.com/p"]; + assert_eq!(entry.engines, vec!["duckduckgo"]); + assert_eq!(entry.score, rrf(0)); + assert_eq!(entry.chunk, "longer snippet"); + } + + #[test] + fn two_engines_that_agree_on_a_url_still_sum_both_ranks() { + let mut merged = HashMap::new(); + merge( + &mut merged, + "google", + vec![hit("https://example.com/p", "google", "a")], + ); + merge( + &mut merged, + "bing", + vec![ + hit("https://example.com/other", "bing", "b"), + hit("https://example.com/p", "bing", "c"), + ], + ); + + let entry = &merged["https://example.com/p"]; + assert_eq!(entry.engines, vec!["google", "bing"]); + assert_eq!(entry.score, rrf(0) + rrf(1)); + } + + #[test] + fn agreement_between_engines_outranks_one_engine_repeating_itself() { + let mut merged = HashMap::new(); + merge( + &mut merged, + "duckduckgo", + vec![ + hit("https://example.com/repeated", "duckduckgo", "a"), + hit("https://example.com/repeated", "duckduckgo", "b"), + hit("https://example.com/agreed", "duckduckgo", "c"), + ], + ); + merge( + &mut merged, + "brave", + vec![hit("https://example.com/agreed", "brave", "d")], + ); + + assert!( + merged["https://example.com/agreed"].score + > merged["https://example.com/repeated"].score + ); + } +}