mirror of
https://github.com/GreptimeTeam/greptimedb.git
synced 2026-10-02 18:15:36 +00:00
* perf(index): build bloom filters from element hashes without per-token allocation Signed-off-by: Dennis Zhuang <killme2008@gmail.com> * perf(index): speed up inverted index building with hashed buffers and sync pushes Signed-off-by: Dennis Zhuang <killme2008@gmail.com> * chore(index): use BuildHasher::hash_one for element hashes Signed-off-by: Dennis Zhuang <killme2008@gmail.com> * chore(index): require callers to act on spill requests Signed-off-by: Dennis Zhuang <killme2008@gmail.com> * fix(index): insert bloom hashes one by one to keep segment set capacity bounded Signed-off-by: Dennis Zhuang <killme2008@gmail.com> * test(index): add index build and bloom search benchmarks Signed-off-by: Dennis Zhuang <killme2008@gmail.com> * test(index): keep applier setup out of the bloom search benchmark Signed-off-by: Dennis Zhuang <killme2008@gmail.com> * fix(index): skip empty spills and keep the old inverted sort memory estimate Signed-off-by: Dennis Zhuang <killme2008@gmail.com> * fix(index): stream fulltext token hashes and test spill dispatch Signed-off-by: Dennis Zhuang <killme2008@gmail.com> * docs(index): note non-ASCII tokens still allocate in analyze_text_hashes Signed-off-by: Dennis Zhuang <killme2008@gmail.com> * docs(index): narrow the allocation note to case-insensitive non-ASCII tokens Signed-off-by: Dennis Zhuang <killme2008@gmail.com> --------- Signed-off-by: Dennis Zhuang <killme2008@gmail.com>
57 lines
2.2 KiB
Rust
57 lines
2.2 KiB
Rust
// Copyright 2023 Greptime Team
|
|
//
|
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
// you may not use this file except in compliance with the License.
|
|
// You may obtain a copy of the License at
|
|
//
|
|
// http://www.apache.org/licenses/LICENSE-2.0
|
|
//
|
|
// Unless required by applicable law or agreed to in writing, software
|
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
// See the License for the specific language governing permissions and
|
|
// limitations under the License.
|
|
|
|
pub mod sort;
|
|
pub mod sort_create;
|
|
|
|
use async_trait::async_trait;
|
|
|
|
use crate::BytesRef;
|
|
use crate::bitmap::BitmapType;
|
|
use crate::inverted_index::error::Result;
|
|
use crate::inverted_index::format::writer::InvertedIndexWriter;
|
|
|
|
/// `InvertedIndexCreator` provides functionality to construct an inverted index
|
|
#[async_trait]
|
|
pub trait InvertedIndexCreator: Send {
|
|
/// Adds a value to the named index. A `None` value represents an absence of data (null)
|
|
///
|
|
/// It should be equivalent to calling `push_with_name_n` with `n = 1`
|
|
#[must_use = "a true result requires calling `spill` before pushing more"]
|
|
fn push_with_name(&mut self, index_name: &str, value: Option<BytesRef<'_>>) -> bool {
|
|
self.push_with_name_n(index_name, value, 1)
|
|
}
|
|
|
|
/// Buffers `n` identical values for the named index. `None` values represent absence of
|
|
/// data (null).
|
|
///
|
|
/// Returns true when buffered data exceeds the memory limit; the caller must then call
|
|
/// [`InvertedIndexCreator::spill`] before pushing more. Pushing is synchronous so the
|
|
/// per-row path does not allocate a future.
|
|
#[must_use = "a true result requires calling `spill` before pushing more"]
|
|
fn push_with_name_n(&mut self, index_name: &str, value: Option<BytesRef<'_>>, n: usize)
|
|
-> bool;
|
|
|
|
/// Moves the buffers that asked for it to external storage.
|
|
async fn spill(&mut self) -> Result<()>;
|
|
|
|
/// Finalizes the index creation process, ensuring all data is properly indexed and stored
|
|
/// in the provided writer
|
|
async fn finish(
|
|
&mut self,
|
|
writer: &mut dyn InvertedIndexWriter,
|
|
bitmap_type: BitmapType,
|
|
) -> Result<()>;
|
|
}
|