Fusion min-max normalises over every keyword match, so hybrid_search asked BM25 for a ranked list of the whole corpus: a hash insert per posting, then a sort of every match, then the merge sorted every candidate again to keep k. - BM25Index::scores returns every match unsorted, accumulated in a dense array (contributions are strictly positive, so zero means untouched). search() is built on it with the bounded heap. - merge_vector_keyword partitions out its top k (select_nth) and orders only those, with the same score-then-id order. - Both hybrid paths use scores(). Rankings are identical (equivalence tests for both changes). p50 0.24 -> 0.07 ms (1K), 2.1 -> 0.49 ms (10K), 23 -> 4.65 ms (100K). The harness gains --fusion-study, which measured the alternative — capping the keyword pool — and found it changes the top-10 for most queries (overlap 0.83-0.92, different #1 for 10-35%) for only a 2x saving. Not adopted. Co-Authored-By: Claude Fable 5.1 <[email protected]>
190 lines
6.9 KiB
Rust
190 lines
6.9 KiB
Rust
//! Search and agents_md methods for HDF5Memory.
|
|
|
|
use std::path::Path;
|
|
|
|
use crate::bm25;
|
|
use crate::hybrid;
|
|
use crate::{HDF5Memory, MAX_ACTIVATION_WEIGHT, MemoryError, Result, SearchResult};
|
|
|
|
impl HDF5Memory {
|
|
/// Vector + keyword scoring stage of [`HDF5Memory::hybrid_search`].
|
|
///
|
|
/// Without the `hnsw` feature this is a full linear cosine scan (the exact
|
|
/// previous behaviour, also used as the correctness oracle in tests). With
|
|
/// `hnsw` enabled and an index available, the vector candidates come from an
|
|
/// approximate-nearest-neighbour search over an over-fetched pool, then merge
|
|
/// with BM25 via the shared [`hybrid::merge_vector_keyword`].
|
|
#[cfg(feature = "hnsw")]
|
|
fn vector_keyword_search(
|
|
&mut self,
|
|
query_embedding: &[f32],
|
|
query_text: &str,
|
|
bm25: &bm25::BM25Index,
|
|
vector_weight: f32,
|
|
keyword_weight: f32,
|
|
k: usize,
|
|
) -> Vec<(usize, f32)> {
|
|
self.ensure_hnsw_fresh();
|
|
match self.hnsw.as_ref() {
|
|
Some(index) if !index.is_empty() && index.dimension() == query_embedding.len() => {
|
|
// Over-fetch so the merge sees a useful vector pool; cosine
|
|
// distance from the index converts back to similarity (1 - d).
|
|
let pool = (k * 8).max(64);
|
|
let vec_scores: Vec<(usize, f32)> = index
|
|
.search(query_embedding, pool, pool)
|
|
.into_iter()
|
|
.map(|(id, dist)| (id, 1.0 - dist))
|
|
.collect();
|
|
// Fusion normalises over every keyword match, so it needs all
|
|
// the scores — but not ranked.
|
|
let kw_scores = bm25.scores(query_text);
|
|
hybrid::merge_vector_keyword(
|
|
vec_scores,
|
|
kw_scores,
|
|
vector_weight,
|
|
keyword_weight,
|
|
k,
|
|
)
|
|
}
|
|
_ => hybrid::hybrid_search(
|
|
query_embedding,
|
|
query_text,
|
|
&self.cache.embeddings,
|
|
&self.cache.chunks,
|
|
&self.cache.tombstones,
|
|
bm25,
|
|
vector_weight,
|
|
keyword_weight,
|
|
k,
|
|
),
|
|
}
|
|
}
|
|
|
|
#[cfg(not(feature = "hnsw"))]
|
|
fn vector_keyword_search(
|
|
&mut self,
|
|
query_embedding: &[f32],
|
|
query_text: &str,
|
|
bm25: &bm25::BM25Index,
|
|
vector_weight: f32,
|
|
keyword_weight: f32,
|
|
k: usize,
|
|
) -> Vec<(usize, f32)> {
|
|
hybrid::hybrid_search(
|
|
query_embedding,
|
|
query_text,
|
|
&self.cache.embeddings,
|
|
&self.cache.chunks,
|
|
&self.cache.tombstones,
|
|
bm25,
|
|
vector_weight,
|
|
keyword_weight,
|
|
k,
|
|
)
|
|
}
|
|
|
|
/// Perform hybrid search combining cosine vector similarity and BM25 keyword search.
|
|
pub fn hybrid_search(
|
|
&mut self,
|
|
query_embedding: &[f32],
|
|
query_text: &str,
|
|
vector_weight: f32,
|
|
keyword_weight: f32,
|
|
k: usize,
|
|
) -> Vec<SearchResult> {
|
|
// The keyword index lives for the life of the store and is updated
|
|
// incrementally. Take it out for the duration of the call so the
|
|
// vector stage can borrow `self` mutably, then put it back.
|
|
self.ensure_bm25_fresh();
|
|
let bm25 = self.bm25.take().expect("ensure_bm25_fresh leaves an index");
|
|
let scored = self.vector_keyword_search(
|
|
query_embedding,
|
|
query_text,
|
|
&bm25,
|
|
vector_weight,
|
|
keyword_weight,
|
|
k,
|
|
);
|
|
let mut results: Vec<SearchResult> = scored
|
|
.into_iter()
|
|
.map(|(idx, score)| {
|
|
let w = self.cache.activation_weights[idx];
|
|
SearchResult {
|
|
score: score * w.sqrt(),
|
|
chunk: self.cache.chunks[idx].clone(),
|
|
index: idx,
|
|
timestamp: self.cache.timestamps[idx],
|
|
source_channel: self.cache.source_channels[idx].clone(),
|
|
activation: w,
|
|
}
|
|
})
|
|
.collect();
|
|
// Ties broken by index so results (and therefore which records get
|
|
// boosted) don't depend on HashMap iteration order upstream.
|
|
results.sort_by(|a, b| {
|
|
b.score
|
|
.partial_cmp(&a.score)
|
|
.unwrap_or(std::cmp::Ordering::Equal)
|
|
.then(a.index.cmp(&b.index))
|
|
});
|
|
|
|
// Only reinforce records that actually matched. When fewer than `k`
|
|
// records are relevant, the rest of the list is zero-score filler;
|
|
// boosting it would teach the store that arbitrary records are
|
|
// important just because they were nearby in iteration order.
|
|
let hit_indices: Vec<usize> = results
|
|
.iter()
|
|
.filter(|r| r.score > 0.0)
|
|
.map(|r| r.index)
|
|
.collect();
|
|
self.apply_hebbian_boost(&hit_indices);
|
|
self.bm25 = Some(bm25);
|
|
|
|
results
|
|
}
|
|
|
|
/// Reinforce the records a query returned. The new weights are persisted by
|
|
/// the next checkpoint (any write that flushes, `flush_wal`, or drop) — not
|
|
/// by rewriting the whole store inside the query, which is what made
|
|
/// `hybrid_search` cost O(store size) in disk I/O. They are a ranking hint,
|
|
/// not user data: a crash before the next checkpoint only forgets the
|
|
/// boosts since the last one.
|
|
fn apply_hebbian_boost(&mut self, hit_indices: &[usize]) {
|
|
if hit_indices.is_empty() || self.config.hebbian_boost == 0.0 {
|
|
return;
|
|
}
|
|
for &idx in hit_indices {
|
|
let w = &mut self.cache.activation_weights[idx];
|
|
*w = (*w + self.config.hebbian_boost).min(MAX_ACTIVATION_WEIGHT);
|
|
}
|
|
self.activations_dirty = true;
|
|
}
|
|
|
|
/// Get the chunk text for a memory entry by index.
|
|
pub fn get_chunk(&self, index: usize) -> Option<&str> {
|
|
if index < self.cache.chunks.len() && self.cache.tombstones[index] == 0 {
|
|
Some(&self.cache.chunks[index])
|
|
} else {
|
|
None
|
|
}
|
|
}
|
|
|
|
/// Generate an AGENTS.md string from current memory state.
|
|
pub fn generate_agents_md(&self) -> String {
|
|
crate::agents_md::generate(&self.config, &self.cache, &self.sessions, &self.knowledge)
|
|
}
|
|
|
|
/// Write AGENTS.md to disk alongside the .h5 file.
|
|
pub fn write_agents_md(&self) -> Result<()> {
|
|
let md = self.generate_agents_md();
|
|
let md_path = self.config.path.with_extension("agents.md");
|
|
std::fs::write(&md_path, md).map_err(MemoryError::Io)
|
|
}
|
|
|
|
/// Read AGENTS.md from disk (if it exists).
|
|
pub fn read_agents_md(path: &Path) -> Result<String> {
|
|
let md_path = path.with_extension("agents.md");
|
|
std::fs::read_to_string(&md_path).map_err(MemoryError::Io)
|
|
}
|
|
}
|