From 59962097d047acbd7b6c6da74877ace82146fdce Mon Sep 17 00:00:00 2001 From: Naveen Aiathurai Date: Fri, 16 Jun 2023 08:03:55 +0530 Subject: [PATCH] fix: #2078 return error when tokenizer not found while indexing (#2093) * fix: #2078 return error when tokenizer not found while indexing * chore: formatting issues * chore: fix review comments --- src/indexer/segment_writer.rs | 56 ++++++++++++++++++++++++++++------- 1 file changed, 46 insertions(+), 10 deletions(-) diff --git a/src/indexer/segment_writer.rs b/src/indexer/segment_writer.rs index 9ccb9810a..2436641fa 100644 --- a/src/indexer/segment_writer.rs +++ b/src/indexer/segment_writer.rs @@ -15,7 +15,7 @@ use crate::postings::{ use crate::schema::{FieldEntry, FieldType, Schema, Term, Value, DATE_TIME_PRECISION_INDEXED}; use crate::store::{StoreReader, StoreWriter}; use crate::tokenizer::{FacetTokenizer, PreTokenizedStream, TextAnalyzer, Tokenizer}; -use crate::{DocId, Document, Opstamp, SegmentComponent}; +use crate::{DocId, Document, Opstamp, SegmentComponent, TantivyError}; /// Computes the initial size of the hash table. /// @@ -98,14 +98,18 @@ impl SegmentWriter { } _ => None, }; - text_options - .and_then(|text_index_option| { - let tokenizer_name = &text_index_option.tokenizer(); - tokenizer_manager.get(tokenizer_name) - }) - .unwrap_or_default() + let tokenizer_name = text_options + .and_then(|text_index_option| Some(text_index_option.tokenizer())) + .unwrap_or("default"); + + tokenizer_manager.get(tokenizer_name).ok_or_else(|| { + TantivyError::SchemaError(format!( + "Error getting tokenizer for field: {}", + field_entry.name() + )) + }) }) - .collect(); + .collect::, _>>()?; Ok(SegmentWriter { max_doc: 0, ctx: IndexingContext::new(table_size), @@ -438,7 +442,9 @@ fn remap_and_write( #[cfg(test)] mod tests { - use std::path::Path; + use std::path::{Path, PathBuf}; + + use tempfile::TempDir; use super::compute_initial_table_size; use crate::collector::Count; @@ -446,7 +452,9 @@ mod tests { use crate::directory::RamDirectory; use crate::postings::TermInfo; use crate::query::PhraseQuery; - use crate::schema::{IndexRecordOption, Schema, Type, STORED, STRING, TEXT}; + use crate::schema::{ + IndexRecordOption, Schema, TextFieldIndexing, TextOptions, Type, STORED, STRING, TEXT, + }; use crate::store::{Compressor, StoreReader, StoreWriter}; use crate::time::format_description::well_known::Rfc3339; use crate::time::OffsetDateTime; @@ -900,4 +908,32 @@ mod tests { postings.positions(&mut positions); assert_eq!(positions, &[4]); //< as opposed to 3 if we had a position length of 1. } + + #[test] + fn test_show_error_when_tokenizer_not_registered() { + let text_field_indexing = TextFieldIndexing::default() + .set_tokenizer("custom_en") + .set_index_option(IndexRecordOption::WithFreqsAndPositions); + let text_options = TextOptions::default() + .set_indexing_options(text_field_indexing) + .set_stored(); + let mut schema_builder = Schema::builder(); + schema_builder.add_text_field("title", text_options); + let schema = schema_builder.build(); + let tempdir = TempDir::new().unwrap(); + let tempdir_path = PathBuf::from(tempdir.path()); + Index::create_in_dir(&tempdir_path, schema).unwrap(); + let index = Index::open_in_dir(tempdir_path).unwrap(); + let schema = index.schema(); + let mut index_writer = index.writer(50_000_000).unwrap(); + let title = schema.get_field("title").unwrap(); + let mut document = Document::default(); + document.add_text(title, "The Old Man and the Sea"); + index_writer.add_document(document).unwrap(); + let error = index_writer.commit().unwrap_err(); + assert_eq!( + error.to_string(), + "Schema error: 'Error getting tokenizer for field: title'" + ); + } }