Compare commits

..

1 Commits

Author SHA1 Message Date
Gatefixer 238028767a docs(node): clarify full-text search filtering 2026-08-05 22:27:18 +00:00
6 changed files with 91 additions and 137 deletions
+22 -10
View File
@@ -541,7 +541,17 @@ where(predicate): this
A filter statement to be applied to this query.
The filter should be supplied as an SQL query string. For example:
Filters are applied before full-text and vector searches by default; no
separate prefilter call is needed. For vector searches only, use
[VectorQuery#postfilter](VectorQuery.md#postfilter) to apply the filter after the search.
The filter should be supplied as an SQL query string.
Filtering performance can often be improved by creating a scalar index
on the filter column(s).
Calling this multiple times combines the filters with a logical AND rather
than replacing the previous filter.
#### Parameters
@@ -551,18 +561,20 @@ The filter should be supplied as an SQL query string. For example:
`this`
#### Example
#### Examples
```ts
x > 10
y > 0 AND y < 100
x > 5 OR y = 'test'
const results = await table
.search("puppy", "fts")
.where("meta = 'foo'")
.limit(10)
.toArray();
```
Filtering performance can often be improved by creating a scalar index
on the filter column(s).
Calling this multiple times combines the filters with a logical AND rather
than replacing the previous filter.
```ts
query.where("x > 10");
query.where("y > 0 AND y < 100");
query.where("x > 5 OR y = 'test'");
```
#### Inherited from
+25 -10
View File
@@ -560,6 +560,9 @@ postfilter(): VectorQuery
If this is called then filtering will happen after the vector search instead of
before.
This method is only available for vector search queries. Full-text search
filters are always applied before the search.
By default filtering will be performed before the vector search. This is how
filtering is typically understood to work. This prefilter step does add some
additional latency. Creating a scalar index on the filter column(s) can
@@ -790,7 +793,17 @@ where(predicate): this
A filter statement to be applied to this query.
The filter should be supplied as an SQL query string. For example:
Filters are applied before full-text and vector searches by default; no
separate prefilter call is needed. For vector searches only, use
[VectorQuery#postfilter](VectorQuery.md#postfilter) to apply the filter after the search.
The filter should be supplied as an SQL query string.
Filtering performance can often be improved by creating a scalar index
on the filter column(s).
Calling this multiple times combines the filters with a logical AND rather
than replacing the previous filter.
#### Parameters
@@ -800,18 +813,20 @@ The filter should be supplied as an SQL query string. For example:
`this`
#### Example
#### Examples
```ts
x > 10
y > 0 AND y < 100
x > 5 OR y = 'test'
const results = await table
.search("puppy", "fts")
.where("meta = 'foo'")
.limit(10)
.toArray();
```
Filtering performance can often be improved by creating a scalar index
on the filter column(s).
Calling this multiple times combines the filters with a logical AND rather
than replacing the previous filter.
```ts
query.where("x > 10");
query.where("y > 0 AND y < 100");
query.where("x > 5 OR y = 'test'");
```
#### Inherited from
+17
View File
@@ -47,5 +47,22 @@ test("filtering examples", async () => {
.limit(5)
.toArray();
// --8<-- [end:orderby_search]
const ftsTable = await db.createTable("myFts", [
{ text: "Frodo was a happy puppy", category: "pet" },
{ text: "A puppy training guide", category: "guide" },
{ text: "There are several kittens playing", category: "pet" },
]);
await ftsTable.createIndex("text", { config: lancedb.Index.fts() });
// --8<-- [start:fts_prefilter]
const ftsResults = await ftsTable
.search("puppy", "fts")
.where("category = 'pet'")
.limit(10)
.toArray();
// --8<-- [end:fts_prefilter]
expect(ftsResults).toHaveLength(1);
expect(ftsResults[0].category).toBe("pet");
});
});
+24 -5
View File
@@ -363,17 +363,33 @@ export class StandardQueryBase<
/**
* A filter statement to be applied to this query.
*
* The filter should be supplied as an SQL query string. For example:
* @example
* x > 10
* y > 0 AND y < 100
* x > 5 OR y = 'test'
* Filters are applied before full-text and vector searches by default; no
* separate prefilter call is needed. For vector searches only, use
* {@link VectorQuery#postfilter} to apply the filter after the search.
*
* The filter should be supplied as an SQL query string.
*
* Filtering performance can often be improved by creating a scalar index
* on the filter column(s).
*
* Calling this multiple times combines the filters with a logical AND rather
* than replacing the previous filter.
*
* @example Filter a full-text search
* ```ts
* const results = await table
* .search("puppy", "fts")
* .where("meta = 'foo'")
* .limit(10)
* .toArray();
* ```
*
* @example SQL filter expressions
* ```ts
* query.where("x > 10");
* query.where("y > 0 AND y < 100");
* query.where("x > 5 OR y = 'test'");
* ```
*/
where(predicate: string): this {
this.doCall((inner: NativeQueryType) => inner.onlyIf(predicate));
@@ -669,6 +685,9 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
* If this is called then filtering will happen after the vector search instead of
* before.
*
* This method is only available for vector search queries. Full-text search
* filters are always applied before the search.
*
* By default filtering will be performed before the vector search. This is how
* filtering is typically understood to work. This prefilter step does add some
* additional latency. Creating a scalar index on the filter column(s) can
+3 -56
View File
@@ -3,7 +3,6 @@
from typing import List
from urllib.parse import unquote, urlparse
import numpy as np
@@ -126,20 +125,9 @@ class InstructorEmbeddingFunction(TextEmbeddingFunction):
@weak_lru(maxsize=1)
def get_model(self):
huggingface_hub = attempt_import_or_raise("huggingface_hub", "huggingface-hub")
missing = object()
original_cached_download = getattr(huggingface_hub, "cached_download", missing)
if original_cached_download is missing:
huggingface_hub.cached_download = _cached_download(huggingface_hub)
try:
instructor_embedding = attempt_import_or_raise(
"InstructorEmbedding", "InstructorEmbedding"
)
finally:
if original_cached_download is missing:
del huggingface_hub.cached_download
instructor_embedding = attempt_import_or_raise(
"InstructorEmbedding", "InstructorEmbedding"
)
torch = attempt_import_or_raise("torch", "torch")
model = instructor_embedding.INSTRUCTOR(self.name)
@@ -152,44 +140,3 @@ class InstructorEmbeddingFunction(TextEmbeddingFunction):
model, {torch.nn.Linear}, dtype=torch.qint8
)
return model
def _cached_download(huggingface_hub):
"""Provide the legacy download API used by sentence-transformers 2.2.x."""
def cached_download(
*,
url,
cache_dir=None,
force_filename=None,
library_name=None,
library_version=None,
user_agent=None,
use_auth_token=None,
**_,
):
path = urlparse(url).path.lstrip("/")
try:
repo_id, resolved_path = path.split("/resolve/", maxsplit=1)
revision, filename = resolved_path.split("/", maxsplit=1)
except ValueError as err:
raise ValueError(f"Unsupported Hugging Face Hub URL: {url}") from err
repo_id = unquote(repo_id)
revision = unquote(revision)
filename = unquote(filename)
# sentence-transformers derives force_filename from this Hub path with
# os.path.join. Using the URL path beneath local_dir produces the same
# local destination without sending Windows separators to the Hub.
return huggingface_hub.hf_hub_download(
repo_id=repo_id,
filename=filename,
revision=revision,
local_dir=cache_dir,
library_name=library_name,
library_version=library_version,
user_agent=user_agent,
token=use_auth_token,
)
return cached_download
-56
View File
@@ -1,11 +1,8 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
import ntpath
import os
import pickle
import sys
from types import ModuleType
from typing import List, Optional, Union
from unittest.mock import MagicMock, patch
@@ -525,59 +522,6 @@ def test_embedding_function_safe_model_dump(embedding_type):
)
def test_instructor_embedding_supports_huggingface_hub_without_cached_download(
tmp_path, monkeypatch
):
from lancedb.embeddings.instructor import InstructorEmbeddingFunction
hub_download = MagicMock(return_value="/cache/1_Pooling/config.json")
huggingface_hub = ModuleType("huggingface_hub")
huggingface_hub.hf_hub_download = hub_download
torch = ModuleType("torch")
monkeypatch.setitem(sys.modules, "huggingface_hub", huggingface_hub)
monkeypatch.setitem(sys.modules, "torch", torch)
monkeypatch.delitem(sys.modules, "InstructorEmbedding", raising=False)
monkeypatch.syspath_prepend(str(tmp_path))
(tmp_path / "InstructorEmbedding.py").write_text(
"from huggingface_hub import cached_download\n\n"
"class INSTRUCTOR:\n"
" def __init__(self, name):\n"
" self.name = name\n"
)
embedding = InstructorEmbeddingFunction.create(show_progress_bar=False)
instructor_model = embedding.get_model()
assert instructor_model.name == "hkunlp/instructor-base"
assert not hasattr(huggingface_hub, "cached_download")
instructor_embedding = sys.modules["InstructorEmbedding"]
path = instructor_embedding.cached_download(
url=(
"https://huggingface.co/hkunlp/instructor-base/resolve/abc123/"
"1_Pooling/config.json"
),
cache_dir="/cache",
force_filename=ntpath.join("1_Pooling", "config.json"),
library_name="sentence-transformers",
library_version="2.2.2",
use_auth_token="token",
)
assert path == "/cache/1_Pooling/config.json"
hub_download.assert_called_once_with(
repo_id="hkunlp/instructor-base",
filename="1_Pooling/config.json",
revision="abc123",
local_dir="/cache",
library_name="sentence-transformers",
library_version="2.2.2",
user_agent=None,
token="token",
)
@patch("time.sleep")
def test_retry(mock_sleep):
test_function = MagicMock(side_effect=[Exception] * 9 + ["result"])