mod block_accessor; mod value_source_registry; #[cfg(test)] pub(crate) mod tests; use std::borrow::Borrow; use std::io; pub(crate) use block_accessor::ColumnBlockAccessor; use columnar::{Cardinality, Column, ColumnType, ColumnValues, Dictionary, RowId, StrColumn}; pub use value_source_registry::{ValueSourceProvider, ValueSourceRegistry}; use crate::DocId; /// A source of values for a block of documents. pub trait ValueSource: std::fmt::Debug { /// Logical type of the encoded values returned by this source. /// /// Numeric values use the corresponding monotonic `u64` mapping; string/bytes /// values are dictionary ordinals and IP addresses use the compact space ord. fn column_type(&self) -> ColumnType; /// Loads the values for `docs` into `values`. /// /// Precondition: `docs` has to be strictly increasing. /// /// The output buffers are reused across blocks: on entry, `values` and `docids` /// hold stale data from a previous call. Implementations must clear their /// content (not append to it). /// /// On return, depending on the returned `Cardinality`: /// - `Full`: `values.len() == docs.len()` and `values[i]` is the value of `docs[i]`. `docids` /// is left unspecified and must not be read by the caller. /// - `Optional` / `Multivalued`: `docids.len() == values.len()` and `values[i]` is a value of /// `docids[i]`. `docids` only contains docs from `docs`. A doc is repeated once per value. /// /// `row_ids` is scratch the implementation may use freely. /// /// Takes `&mut self` so that implementations can keep per-segment state (caches, scratch /// buffers) across blocks. Each source is owned by a single collector. fn load_block( &mut self, docs: &[DocId], values: &mut Vec, docids: &mut Vec, row_ids: &mut Vec, ) -> Cardinality; /// Returns the physical column, if this source is backed by one. fn as_column(&self) -> Option<&Column> { None } /// Global value bounds, for fast paths that need to size or clamp something up front. fn bounds(&self) -> Option<(u64, u64)> { let column = self.as_column()?; Some((column.min_value(), column.max_value())) } /// Returns the physical text column, if this source is backed by one. /// /// For the paths that need the full sstable dictionary (regex search, streaming all of the /// terms) rather than resolving ords. `Some` implies that `term_dictionary()` is `Some` and /// that its ords are sorted with the terms. fn as_str_column(&self) -> Option<&StrColumn> { None } /// Dictionary resolving the term ords returned by `load_block`, for `Str` sources. /// /// Every `Str` source with values must return its dictionary. `None` is only acceptable for /// a source without any value (e.g. the empty-column shim of an absent field). /// /// Contract: the ords returned by earlier `load_block` calls remain valid. The dictionary /// may grow as more blocks are loaded. fn term_dictionary(&self) -> Option<&dyn TermOrdDictionary> { None } /// Heap memory owned by the source, in bytes. /// /// Sources can grow while loading blocks (e.g. a dictionary built on the fly). The growth /// observed across `load_block` calls is charged to the aggregation memory limits. fn memory_consumption(&self) -> usize { 0 } } /// Resolves the term ords of a `Str` [`ValueSource`] back into terms. pub trait TermOrdDictionary { /// Returns true if the order of the ords matches the lexicographic order of the terms. /// /// Aggregations relying on that property (e.g. a terms aggregation ordered by `_key`) must /// check it. fn ords_sorted_with_terms(&self) -> bool; /// Number of terms in the dictionary. Valid ords are `0..num_terms()`. fn num_terms(&self) -> u64; /// Calls `callback` with the term associated with each ord, in the order of `sorted_ords`. /// /// Precondition: `sorted_ords` is sorted in ascending order. /// /// Returns false if an ord was not found in the dictionary. fn sorted_ords_to_term_cb( &self, sorted_ords: &[u64], callback: &mut dyn FnMut(&[u8]), ) -> io::Result; } impl TermOrdDictionary for Dictionary { fn ords_sorted_with_terms(&self) -> bool { true } fn num_terms(&self) -> u64 { Dictionary::num_terms(self) as u64 } fn sorted_ords_to_term_cb( &self, sorted_ords: &[u64], callback: &mut dyn FnMut(&[u8]), ) -> io::Result { Dictionary::sorted_ords_to_term_cb(self, sorted_ords, callback) } } // Lenient columns have erased their logical type; the tuple retains it alongside the values. impl> + std::fmt::Debug> ValueSource for (ColumnRef, ColumnType) { #[inline] fn column_type(&self) -> ColumnType { self.1 } #[inline] fn load_block( &mut self, docs: &[DocId], values: &mut Vec, docids: &mut Vec, row_ids: &mut Vec, ) -> Cardinality { let column = self.0.borrow(); let cardinality = column.index.get_cardinality(); if cardinality.is_full() { load_full_column_values(docs, &*column.values, values); } else { docids.clear(); row_ids.clear(); column.row_ids_for_docs(docs, docids, row_ids); values.resize(row_ids.len(), 0u64); column.values.get_vals(row_ids, values); } cardinality } #[inline] fn as_column(&self) -> Option<&Column> { Some(self.0.borrow()) } } // A physical text column: the values are the term ords of its dictionary. impl ValueSource for StrColumn { #[inline] fn column_type(&self) -> ColumnType { ColumnType::Str } #[inline] fn load_block( &mut self, docs: &[DocId], values: &mut Vec, docids: &mut Vec, row_ids: &mut Vec, ) -> Cardinality { (self.ords(), ColumnType::Str).load_block(docs, values, docids, row_ids) } #[inline] fn as_column(&self) -> Option<&Column> { Some(self.ords()) } fn as_str_column(&self) -> Option<&StrColumn> { Some(self) } fn term_dictionary(&self) -> Option<&dyn TermOrdDictionary> { Some(self.dictionary()) } } /// `docs` has to be sorted ascending and free of duplicates. #[inline] fn load_full_column_values( docs: &[DocId], column_values: &dyn ColumnValues, values: &mut Vec, ) { // Skip the resize when already the right length (common case: fixed-size blocks). if values.len() != docs.len() { values.resize(docs.len(), 0u64); } // When the docs form a contiguous ascending run we can fetch the values as a single range. // This lets codecs (e.g. bitpacked) bulk-decode the slice instead of gathering value-by-value. if is_contiguous(docs) { column_values.get_range(docs[0] as u64, values); } else { column_values.get_vals(docs, values); } } /// Returns true if `docs` is a contiguous ascending run `[d, d + 1, ..., d + n - 1]`. /// /// `docs` has to be sorted ascending and free of duplicates. #[inline] fn is_contiguous(docs: &[u32]) -> bool { let (Some(&first), Some(&last)) = (docs.first(), docs.last()) else { return false; }; debug_assert!( docs.windows(2).all(|w| w[0] < w[1]), "fetch_block requires docs sorted ascending without duplicates" ); (last - first) as usize + 1 == docs.len() }