macos ci

fix Term documentation (#655 )
u64-based fields are actually 4+8=12 bytes long
2026-01-11 03:22:55 +00:00 · 2019-09-13 10:16:42 +09:00 · 2019-09-11 18:49:35 +09:00 · 2019-09-11 17:12:08 +09:00 · 2019-09-09 06:36:04 +09:00 · 2019-09-07 19:40:21 +09:00
13 changed files with 36 additions and 44 deletions
--- a/.travis.yml
+++ b/.travis.yml
@@ -41,8 +41,8 @@ matrix:
    - env: TARGET=x86_64-unknown-linux-gnu CODECOV=1 #UPLOAD_DOCS=1
    # - env: TARGET=x86_64-unknown-linux-musl CODECOV=1
    # OSX
-    #- env: TARGET=x86_64-apple-darwin
-    #  os: osx
+    - env: TARGET=x86_64-apple-darwin
+      os: osx

 before_install:
  - set -e
--- a/query-grammar/src/occur.rs
+++ b/query-grammar/src/occur.rs
@@ -1,3 +1,6 @@
+use std::fmt;
+use std::fmt::Write;
+
 /// Defines whether a term in a query must be present,
 /// should be present or must not be present.
 #[derive(Debug, Clone, Hash, Copy, Eq, PartialEq)]
@@ -18,7 +21,7 @@ impl Occur {
    /// - `Should` => '?',
    /// - `Must` => '+'
    /// - `Not` => '-'
-    pub fn to_char(self) -> char {
+    fn to_char(self) -> char {
        match self {
            Occur::Should => '?',
            Occur::Must => '+',
@@ -47,3 +50,9 @@ impl Occur {
        }
    }
 }
+
+impl fmt::Display for Occur {
+    fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
+        f.write_char(self.to_char())
+    }
+}
--- a/query-grammar/src/user_input_ast.rs
+++ b/query-grammar/src/user_input_ast.rs
@@ -151,7 +151,7 @@ impl fmt::Debug for UserInputAST {
                Ok(())
            }
            UserInputAST::Unary(ref occur, ref subquery) => {
-                write!(formatter, "{}({:?})", occur.to_char(), subquery)
+                write!(formatter, "{}({:?})", occur, subquery)
            }
            UserInputAST::Leaf(ref subquery) => write!(formatter, "{:?}", subquery),
        }
--- a/src/directory/mmap_directory.rs
+++ b/src/directory/mmap_directory.rs
@@ -265,7 +265,7 @@ impl MmapDirectoryInner {
            }
        }
        if let Some(watch_wrapper) = self.watcher.write().unwrap().as_mut() {
-            return Ok(watch_wrapper.watch(watch_callback));
+            Ok(watch_wrapper.watch(watch_callback))
        } else {
            unreachable!("At this point, watch wrapper is supposed to be initialized");
        }
--- a/src/fastfield/multivalued/writer.rs
+++ b/src/fastfield/multivalued/writer.rs
@@ -5,8 +5,8 @@ use crate::postings::UnorderedTermId;
 use crate::schema::{Document, Field};
 use crate::termdict::TermOrdinal;
 use crate::DocId;
+use fnv::FnvHashMap;
 use itertools::Itertools;
-use std::collections::HashMap;
 use std::io;

 /// Writer for multi-valued (as in, more than one value per document)
@@ -102,7 +102,7 @@ impl MultiValueIntFastFieldWriter {
    pub fn serialize(
        &self,
        serializer: &mut FastFieldSerializer,
-        mapping_opt: Option<&HashMap<UnorderedTermId, TermOrdinal>>,
+        mapping_opt: Option<&FnvHashMap<UnorderedTermId, TermOrdinal>>,
    ) -> io::Result<()> {
        {
            // writing the offset index
--- a/src/fastfield/writer.rs
+++ b/src/fastfield/writer.rs
@@ -6,6 +6,7 @@ use crate::fastfield::{BytesFastFieldWriter, FastFieldSerializer};
 use crate::postings::UnorderedTermId;
 use crate::schema::{Cardinality, Document, Field, FieldType, Schema};
 use crate::termdict::TermOrdinal;
+use fnv::FnvHashMap;
 use std::collections::HashMap;
 use std::io;

@@ -116,7 +117,7 @@ impl FastFieldsWriter {
    pub fn serialize(
        &self,
        serializer: &mut FastFieldSerializer,
-        mapping: &HashMap<Field, HashMap<UnorderedTermId, TermOrdinal>>,
+        mapping: &HashMap<Field, FnvHashMap<UnorderedTermId, TermOrdinal>>,
    ) -> io::Result<()> {
        for field_writer in &self.single_value_writers {
            field_writer.serialize(serializer)?;
--- a/src/postings/postings_writer.rs
+++ b/src/postings/postings_writer.rs
@@ -12,6 +12,7 @@ use crate::tokenizer::TokenStream;
 use crate::tokenizer::{Token, MAX_TOKEN_LEN};
 use crate::DocId;
 use crate::Result;
+use fnv::FnvHashMap;
 use std::collections::HashMap;
 use std::io;
 use std::marker::PhantomData;
@@ -127,12 +128,12 @@ impl MultiFieldPostingsWriter {
    pub fn serialize(
        &self,
        serializer: &mut InvertedIndexSerializer,
-    ) -> Result<HashMap<Field, HashMap<UnorderedTermId, TermOrdinal>>> {
+    ) -> Result<HashMap<Field, FnvHashMap<UnorderedTermId, TermOrdinal>>> {
        let mut term_offsets: Vec<(&[u8], Addr, UnorderedTermId)> =
            self.term_index.iter().collect();
        term_offsets.sort_unstable_by_key(|&(k, _, _)| k);

-        let mut unordered_term_mappings: HashMap<Field, HashMap<UnorderedTermId, TermOrdinal>> =
+        let mut unordered_term_mappings: HashMap<Field, FnvHashMap<UnorderedTermId, TermOrdinal>> =
            HashMap::new();

        let field_offsets = make_field_partition(&term_offsets);
@@ -147,7 +148,7 @@ impl MultiFieldPostingsWriter {
                    let unordered_term_ids = term_offsets[start..stop]
                        .iter()
                        .map(|&(_, _, bucket)| bucket);
-                    let mapping: HashMap<UnorderedTermId, TermOrdinal> = unordered_term_ids
+                    let mapping: FnvHashMap<UnorderedTermId, TermOrdinal> = unordered_term_ids
                        .enumerate()
                        .map(|(term_ord, unord_term_id)| {
                            (unord_term_id as UnorderedTermId, term_ord as TermOrdinal)
--- a/src/postings/serializer.rs
+++ b/src/postings/serializer.rs
@@ -141,10 +141,7 @@ impl<'a> FieldSerializer<'a> {
            FieldType::Str(ref text_options) => {
                if let Some(text_indexing_options) = text_options.get_indexing_options() {
                    let index_option = text_indexing_options.index_option();
-                    (
-                        index_option.is_termfreq_enabled(),
-                        index_option.is_position_enabled(),
-                    )
+                    (index_option.has_freq(), index_option.has_positions())
                } else {
                    (false, false)
                }
--- a/src/query/intersection.rs
+++ b/src/query/intersection.rs
@@ -45,7 +45,7 @@ pub fn intersect_scorers(mut scorers: Vec<Box<dyn Scorer>>) -> Box<dyn Scorer> {
    })
 }

-/// Creates a `DocSet` that iterator through the intersection of two `DocSet`s.
+/// Creates a `DocSet` that iterate through the intersection of two or more `DocSet`s.
 pub struct Intersection<TDocSet: DocSet, TOtherDocSet: DocSet = Box<dyn Scorer>> {
    left: TDocSet,
    right: TDocSet,
--- a/src/query/intersection_two.rs
+++ b/src/query/intersection_two.rs
@@ -5,7 +5,7 @@ use Score;
 use SkipResult;


-/// Creates a `DocSet` that iterator through the intersection of two `DocSet`s.
+/// Creates a `DocSet` that iterate through the intersection of two `DocSet`s.
 pub struct IntersectionTwoTerms<TDocSet> {
    left: TDocSet,
    right: TDocSet
--- a/src/query/union.rs
+++ b/src/query/union.rs
@@ -28,7 +28,7 @@ where
    }
 }

-/// Creates a `DocSet` that iterator through the intersection of two `DocSet`s.
+/// Creates a `DocSet` that iterate through the union of two or more `DocSet`s.
 pub struct Union<TScorer, TScoreCombiner = DoNothingCombiner> {
    docsets: Vec<TScorer>,
    bitsets: Box<[TinySet; HORIZON_NUM_TINYBITSETS]>,
--- a/src/schema/index_record_option.rs
+++ b/src/schema/index_record_option.rs
@@ -29,22 +29,6 @@ pub enum IndexRecordOption {
 }

 impl IndexRecordOption {
-    /// Returns true iff the term frequency will be encoded.
-    pub fn is_termfreq_enabled(self) -> bool {
-        match self {
-            IndexRecordOption::WithFreqsAndPositions | IndexRecordOption::WithFreqs => true,
-            _ => false,
-        }
-    }
-
-    /// Returns true iff the term positions within the document are stored as well.
-    pub fn is_position_enabled(self) -> bool {
-        match self {
-            IndexRecordOption::WithFreqsAndPositions => true,
-            _ => false,
-        }
-    }
-
    /// Returns true iff this option includes encoding
    /// term frequencies.
    pub fn has_freq(self) -> bool {
--- a/src/schema/term.rs
+++ b/src/schema/term.rs
@@ -22,10 +22,10 @@ impl Term {
    /// Builds a term given a field, and a i64-value
    ///
    /// Assuming the term has a field id of 1, and a i64 value of 3234,
-    /// the Term will have 8 bytes.
+    /// the Term will have 12 bytes.
    ///
    /// The first four byte are dedicated to storing the field id as a u64.
-    /// The 4 following bytes are encoding the u64 value.
+    /// The 8 following bytes are encoding the u64 value.
    pub fn from_field_i64(field: Field, val: i64) -> Term {
        let val_u64: u64 = common::i64_to_u64(val);
        Term::from_field_u64(field, val_u64)
@@ -33,11 +33,11 @@ impl Term {

    /// Builds a term given a field, and a f64-value
    ///
-    /// Assuming the term has a field id of 1, and a u64 value of 3234,
-    /// the Term will have 8 bytes. <= this is wrong
+    /// Assuming the term has a field id of 1, and a f64 value of 1.5,
+    /// the Term will have 12 bytes.
    ///
    /// The first four byte are dedicated to storing the field id as a u64.
-    /// The 4 following bytes are encoding the u64 value.
+    /// The 8 following bytes are encoding the f64 as a u64 value.
    pub fn from_field_f64(field: Field, val: f64) -> Term {
        let val_u64: u64 = common::f64_to_u64(val);
        Term::from_field_u64(field, val_u64)
@@ -46,10 +46,10 @@ impl Term {
    /// Builds a term given a field, and a DateTime value
    ///
    /// Assuming the term has a field id of 1, and a timestamp i64 value of 3234,
-    /// the Term will have 8 bytes.
+    /// the Term will have 12 bytes.
    ///
    /// The first four byte are dedicated to storing the field id as a u64.
-    /// The 4 following bytes are encoding the DateTime as i64 timestamp value.
+    /// The 8 following bytes are encoding the DateTime as i64 timestamp value.
    pub fn from_field_date(field: Field, val: &DateTime) -> Term {
        let val_timestamp = val.timestamp();
        Term::from_field_i64(field, val_timestamp)
@@ -82,10 +82,10 @@ impl Term {
    /// Builds a term given a field, and a u64-value
    ///
    /// Assuming the term has a field id of 1, and a u64 value of 3234,
-    /// the Term will have 8 bytes.
+    /// the Term will have 12 bytes.
    ///
    /// The first four byte are dedicated to storing the field id as a u64.
-    /// The 4 following bytes are encoding the u64 value.
+    /// The 8 following bytes are encoding the u64 value.
    pub fn from_field_u64(field: Field, val: u64) -> Term {
        let mut term = Term(vec![0u8; INT_TERM_LEN]);
        term.set_field(field);
@@ -182,7 +182,7 @@ where
    ///
    /// # Panics
    /// ... or returns an invalid value
-    /// if the term is not a `i64` field.
+    /// if the term is not a `f64` field.
    pub fn get_f64(&self) -> f64 {
        common::u64_to_f64(BigEndian::read_u64(&self.0.as_ref()[4..]))
    }
Author	SHA1	Message	Date
Paul Masurel	f2b8c030d5	macos ci	2019-09-13 10:16:42 +09:00
fdb-hiroshima	7e08e0047b	fix Term documentation (#655 ) u64-based fields are actually 4+8=12 bytes long	2019-09-11 18:49:35 +09:00
fdb-hiroshima	1a817f117f	fix documentation error (#654 ) Union missdocumented as doing an intersection Union and Intersection can hold more than 2 DocSets	2019-09-11 17:12:08 +09:00
petr-tik	2ec19b21ae	Remove unnecessary duplicate methods (#650 ) Closes #649 Spotted by @imor	2019-09-09 06:36:04 +09:00
Raminder Singh	141f5a93f7	Using FnvHashMap for mapping UnorderedTermId to TermOrdinal. Fixes #507 (#647 ) * Using FnvHashMap for mapping UnorderedTermId to TermOrdinal. Fixes #507 * Fixed cargo fmt errors	2019-09-07 19:40:21 +09:00
Paul Masurel	df47d55cd2	Occur debug interface (#648 )	2019-09-07 15:08:45 +09:00
Raminder Singh	5e579fd6b7	Fixed clippy warning: unneeded return statement (#646 )	2019-09-07 10:14:37 +09:00