Files
tantivy/src/aggregation/value_source/mod.rs
T
Paul Masurel 090f12157c Prepare ValueSource for computed text sources
- `ValueSource::term_dictionary()` lets a source resolve its own term
  ords (`TermOrdDictionary`, implemented by the sstable dictionary).
  Terms and cardinality resolve ords through it; a registered name never
  uses the physical dictionary of the same field.
- `ValueSource::memory_consumption()`: the growth observed across
  `load_block` calls is charged to the aggregation memory limits.
- `ValueSourceProvider::for_segment` receives the column types the
  aggregation accepts.
- A registered source whose type is not allowed is treated as absent
  instead of silently falling back to the physical field.
- composite and multi_terms reject registered sources explicitly.
2026-10-05 15:15:19 +02:00

232 lines
7.6 KiB
Rust

mod block_accessor;
mod value_source_registry;
#[cfg(test)]
pub(crate) mod tests;
use std::borrow::Borrow;
use std::io;
pub(crate) use block_accessor::ColumnBlockAccessor;
use columnar::{Cardinality, Column, ColumnType, ColumnValues, Dictionary, RowId, StrColumn};
pub use value_source_registry::{ValueSourceProvider, ValueSourceRegistry};
use crate::DocId;
/// A source of values for a block of documents.
pub trait ValueSource: std::fmt::Debug {
/// Logical type of the encoded values returned by this source.
///
/// Numeric values use the corresponding monotonic `u64` mapping; string/bytes
/// values are dictionary ordinals and IP addresses use the compact space ord.
fn column_type(&self) -> ColumnType;
/// Loads the values for `docs` into `values`.
///
/// Precondition: `docs` has to be strictly increasing.
///
/// The output buffers are reused across blocks: on entry, `values` and `docids`
/// hold stale data from a previous call. Implementations must clear their
/// content (not append to it).
///
/// On return, depending on the returned `Cardinality`:
/// - `Full`: `values.len() == docs.len()` and `values[i]` is the value of `docs[i]`. `docids`
/// is left unspecified and must not be read by the caller.
/// - `Optional` / `Multivalued`: `docids.len() == values.len()` and `values[i]` is a value of
/// `docids[i]`. `docids` only contains docs from `docs`. A doc is repeated once per value.
///
/// `row_ids` is scratch the implementation may use freely.
///
/// Takes `&mut self` so that implementations can keep per-segment state (caches, scratch
/// buffers) across blocks. Each source is owned by a single collector.
fn load_block(
&mut self,
docs: &[DocId],
values: &mut Vec<u64>,
docids: &mut Vec<DocId>,
row_ids: &mut Vec<RowId>,
) -> Cardinality;
/// Returns the physical column, if this source is backed by one.
fn as_column(&self) -> Option<&Column<u64>> {
None
}
/// Global value bounds, for fast paths that need to size or clamp something up front.
fn bounds(&self) -> Option<(u64, u64)> {
let column = self.as_column()?;
Some((column.min_value(), column.max_value()))
}
/// Returns the physical text column, if this source is backed by one.
///
/// For the paths that need the full sstable dictionary (regex search, streaming all of the
/// terms) rather than resolving ords. `Some` implies that `term_dictionary()` is `Some` and
/// that its ords are sorted with the terms.
fn as_str_column(&self) -> Option<&StrColumn> {
None
}
/// Dictionary resolving the term ords returned by `load_block`, for `Str` sources.
///
/// Every `Str` source with values must return its dictionary. `None` is only acceptable for
/// a source without any value (e.g. the empty-column shim of an absent field).
///
/// Contract: the ords returned by earlier `load_block` calls remain valid. The dictionary
/// may grow as more blocks are loaded.
fn term_dictionary(&self) -> Option<&dyn TermOrdDictionary> {
None
}
/// Heap memory owned by the source, in bytes.
///
/// Sources can grow while loading blocks (e.g. a dictionary built on the fly). The growth
/// observed across `load_block` calls is charged to the aggregation memory limits.
fn memory_consumption(&self) -> usize {
0
}
}
/// Resolves the term ords of a `Str` [`ValueSource`] back into terms.
pub trait TermOrdDictionary {
/// Returns true if the order of the ords matches the lexicographic order of the terms.
///
/// Aggregations relying on that property (e.g. a terms aggregation ordered by `_key`) must
/// check it.
fn ords_sorted_with_terms(&self) -> bool;
/// Number of terms in the dictionary. Valid ords are `0..num_terms()`.
fn num_terms(&self) -> u64;
/// Calls `callback` with the term associated with each ord, in the order of `sorted_ords`.
///
/// Precondition: `sorted_ords` is sorted in ascending order.
///
/// Returns false if an ord was not found in the dictionary.
fn sorted_ords_to_term_cb(
&self,
sorted_ords: &[u64],
callback: &mut dyn FnMut(&[u8]),
) -> io::Result<bool>;
}
impl TermOrdDictionary for Dictionary {
fn ords_sorted_with_terms(&self) -> bool {
true
}
fn num_terms(&self) -> u64 {
Dictionary::num_terms(self) as u64
}
fn sorted_ords_to_term_cb(
&self,
sorted_ords: &[u64],
callback: &mut dyn FnMut(&[u8]),
) -> io::Result<bool> {
Dictionary::sorted_ords_to_term_cb(self, sorted_ords, callback)
}
}
// Lenient columns have erased their logical type; the tuple retains it alongside the values.
impl<ColumnRef: Borrow<Column<u64>> + std::fmt::Debug> ValueSource for (ColumnRef, ColumnType) {
#[inline]
fn column_type(&self) -> ColumnType {
self.1
}
#[inline]
fn load_block(
&mut self,
docs: &[DocId],
values: &mut Vec<u64>,
docids: &mut Vec<DocId>,
row_ids: &mut Vec<RowId>,
) -> Cardinality {
let column = self.0.borrow();
let cardinality = column.index.get_cardinality();
if cardinality.is_full() {
load_full_column_values(docs, &*column.values, values);
} else {
docids.clear();
row_ids.clear();
column.row_ids_for_docs(docs, docids, row_ids);
values.resize(row_ids.len(), 0u64);
column.values.get_vals(row_ids, values);
}
cardinality
}
#[inline]
fn as_column(&self) -> Option<&Column<u64>> {
Some(self.0.borrow())
}
}
// A physical text column: the values are the term ords of its dictionary.
impl ValueSource for StrColumn {
#[inline]
fn column_type(&self) -> ColumnType {
ColumnType::Str
}
#[inline]
fn load_block(
&mut self,
docs: &[DocId],
values: &mut Vec<u64>,
docids: &mut Vec<DocId>,
row_ids: &mut Vec<RowId>,
) -> Cardinality {
(self.ords(), ColumnType::Str).load_block(docs, values, docids, row_ids)
}
#[inline]
fn as_column(&self) -> Option<&Column<u64>> {
Some(self.ords())
}
fn as_str_column(&self) -> Option<&StrColumn> {
Some(self)
}
fn term_dictionary(&self) -> Option<&dyn TermOrdDictionary> {
Some(self.dictionary())
}
}
/// `docs` has to be sorted ascending and free of duplicates.
#[inline]
fn load_full_column_values(
docs: &[DocId],
column_values: &dyn ColumnValues<u64>,
values: &mut Vec<u64>,
) {
// Skip the resize when already the right length (common case: fixed-size blocks).
if values.len() != docs.len() {
values.resize(docs.len(), 0u64);
}
// When the docs form a contiguous ascending run we can fetch the values as a single range.
// This lets codecs (e.g. bitpacked) bulk-decode the slice instead of gathering value-by-value.
if is_contiguous(docs) {
column_values.get_range(docs[0] as u64, values);
} else {
column_values.get_vals(docs, values);
}
}
/// Returns true if `docs` is a contiguous ascending run `[d, d + 1, ..., d + n - 1]`.
///
/// `docs` has to be sorted ascending and free of duplicates.
#[inline]
fn is_contiguous(docs: &[u32]) -> bool {
let (Some(&first), Some(&last)) = (docs.first(), docs.last()) else {
return false;
};
debug_assert!(
docs.windows(2).all(|w| w[0] < w[1]),
"fetch_block requires docs sorted ascending without duplicates"
);
(last - first) as usize + 1 == docs.len()
}