Compare commits

...

19 Commits

Author SHA1 Message Date
Lance Release 3fd322a93a Bump version: 0.35.0-beta.1 → 0.35.0-beta.2 2026-07-14 23:27:49 +00:00
LanceDB Robot d8f0982ee8 chore: update lance dependency to v9.0.0-beta.23 (#3665)
Updates the Rust workspace Lance dependencies and Java lance-core from
v9.0.0-beta.19 to v9.0.0-beta.23.

No compatibility fixes were required; strict workspace Clippy and Rust
formatting pass. Lance tag:
https://github.com/lance-format/lance/releases/tag/v9.0.0-beta.23

---------

Co-authored-by: Jack Ye <yezhaoqin@gmail.com>
2026-07-14 16:26:56 -07:00
dependabot[bot] 7276c34c51 chore(deps): bump the rust-minor-patch group across 1 directory with 6 updates (#3658)
Bumps the rust-minor-patch group with 6 updates in the / directory:

| Package | From | To |
| --- | --- | --- |
| [regex](https://github.com/rust-lang/regex) | `1.12.4` | `1.13.0` |
| [bytes](https://github.com/tokio-rs/bytes) | `1.12.0` | `1.12.1` |
| [uuid](https://github.com/uuid-rs/uuid) | `1.23.4` | `1.23.5` |
| [http-body](https://github.com/hyperium/http-body) | `1.0.1` | `1.1.0`
|
| [napi](https://github.com/napi-rs/napi-rs) | `3.10.3` | `3.10.5` |
| [napi-derive](https://github.com/napi-rs/napi-rs) | `3.5.9` | `3.5.10`
|


Updates `regex` from 1.12.4 to 1.13.0
<details>
<summary>Changelog</summary>
<p><em>Sourced from <a
href="https://github.com/rust-lang/regex/blob/master/CHANGELOG.md">regex's
changelog</a>.</em></p>
<blockquote>
<h1>1.13.0 (2026-07-09)</h1>
<p>This release includes a new API, a <code>regex!</code> macro, for
lazy compilation of
a regex from a string literal. If you use regexes a lot, it's likely
you've
already written one exactly like it. The new macro can be used like
this:</p>
<pre lang="rust"><code>use regex::regex;
<p>fn is_match(line: &amp;str) -&gt; bool {<br />
// The regex will be compiled approximately once and reused
automatically.<br />
// This avoids the footgun of using <code>Regex::new</code> here, which
would<br />
// guarantee that it would be compiled every time this routine is
called.<br />
// This would likely make this routine much slower than it needs to
be.<br />
regex!(r&quot;bar|baz&quot;).is_match(line)<br />
}</p>
<p>let hay = &quot;<br />
path/to/foo:54:Blue Harvest<br />
path/to/bar:90:Something, Something, Something, Dark Side<br />
path/to/baz:3:It's a Trap!<br />
&quot;;</p>
<p>let matches = hay.lines().filter(|line| is_match(line)).count();<br
/>
assert_eq!(matches, 2);<br />
</code></pre></p>
<p>Improvements:</p>
<ul>
<li><a
href="https://redirect.github.com/rust-lang/regex/issues/709">#709</a>:
Add a new <code>regex!</code> macro for efficient and automatic reuse of
a compiled regex.</li>
</ul>
</blockquote>
</details>
<details>
<summary>Commits</summary>
<ul>
<li><a
href="https://github.com/rust-lang/regex/commit/926af2e68eca3ce089815790541cf50759ba2c59"><code>926af2e</code></a>
1.13.0</li>
<li><a
href="https://github.com/rust-lang/regex/commit/7d941a93561430cd259bb9ceb84cc66f33ae7be8"><code>7d941a9</code></a>
regex-automata-0.4.15</li>
<li><a
href="https://github.com/rust-lang/regex/commit/e358341229ebd5feb9a78d8cc85b459c3c7b6600"><code>e358341</code></a>
api: add <code>regex!</code> macro for lazy compilation</li>
<li><a
href="https://github.com/rust-lang/regex/commit/c42033379c8760105ef90287f319de73d1572242"><code>c420333</code></a>
automata: disable miri on a couple doc tests</li>
<li><a
href="https://github.com/rust-lang/regex/commit/b9d2cf724f89754ea879b6c223d2292c4d3e2dd3"><code>b9d2cf7</code></a>
github: add FUNDING link</li>
<li><a
href="https://github.com/rust-lang/regex/commit/0858006b1460ba781deda54b8d2b01b3f9f949f7"><code>0858006</code></a>
docs: add AI policy for contributors</li>
<li><a
href="https://github.com/rust-lang/regex/commit/468fc64ecd6493caaca40dbe8319c31c5c08a83d"><code>468fc64</code></a>
automata: reject dense DFA start states that are match states</li>
<li>See full diff in <a
href="https://github.com/rust-lang/regex/compare/1.12.4...1.13.0">compare
view</a></li>
</ul>
</details>
<br />

Updates `bytes` from 1.12.0 to 1.12.1
<details>
<summary>Release notes</summary>
<p><em>Sourced from <a
href="https://github.com/tokio-rs/bytes/releases">bytes's
releases</a>.</em></p>
<blockquote>
<h2>Bytes v1.12.1</h2>
<h1>1.12.1 (July 8th, 2026)</h1>
<h3>Fixed</h3>
<ul>
<li>Properly handle when <code>Box::new</code> panics (<a
href="https://redirect.github.com/tokio-rs/bytes/issues/837">#837</a>)</li>
</ul>
</blockquote>
</details>
<details>
<summary>Changelog</summary>
<p><em>Sourced from <a
href="https://github.com/tokio-rs/bytes/blob/master/CHANGELOG.md">bytes's
changelog</a>.</em></p>
<blockquote>
<h1>1.12.1 (July 8th, 2026)</h1>
<h3>Fixed</h3>
<ul>
<li>Properly handle when <code>Box::new</code> panics (<a
href="https://redirect.github.com/tokio-rs/bytes/issues/837">#837</a>)</li>
</ul>
</blockquote>
</details>
<details>
<summary>Commits</summary>
<ul>
<li><a
href="https://github.com/tokio-rs/bytes/commit/76c0fbb54ed4336caf9d2311658a2f4a5627c21d"><code>76c0fbb</code></a>
Release bytes v1.12.1 (<a
href="https://redirect.github.com/tokio-rs/bytes/issues/838">#838</a>)</li>
<li><a
href="https://github.com/tokio-rs/bytes/commit/924c82bf0053cb13a0fb5165925d564622b2092f"><code>924c82b</code></a>
Handle unwinding from Box::new (<a
href="https://redirect.github.com/tokio-rs/bytes/issues/837">#837</a>)</li>
<li>See full diff in <a
href="https://github.com/tokio-rs/bytes/compare/v1.12.0...v1.12.1">compare
view</a></li>
</ul>
</details>
<br />

Updates `uuid` from 1.23.4 to 1.23.5
<details>
<summary>Release notes</summary>
<p><em>Sourced from <a
href="https://github.com/uuid-rs/uuid/releases">uuid's
releases</a>.</em></p>
<blockquote>
<h2>v1.23.5</h2>
<h2>What's Changed</h2>
<ul>
<li>doc: Fix broken link by <a
href="https://github.com/frostyplanet"><code>@​frostyplanet</code></a>
in <a
href="https://redirect.github.com/uuid-rs/uuid/pull/891">uuid-rs/uuid#891</a></li>
<li>perf: Optimize UUID hex parsing and formatting by <a
href="https://github.com/geeknoid"><code>@​geeknoid</code></a> in <a
href="https://redirect.github.com/uuid-rs/uuid/pull/894">uuid-rs/uuid#894</a></li>
<li>Prepare for 1.23.5 release by <a
href="https://github.com/KodrAus"><code>@​KodrAus</code></a> in <a
href="https://redirect.github.com/uuid-rs/uuid/pull/895">uuid-rs/uuid#895</a></li>
</ul>
<h2>New Contributors</h2>
<ul>
<li><a href="https://github.com/geeknoid"><code>@​geeknoid</code></a>
made their first contribution in <a
href="https://redirect.github.com/uuid-rs/uuid/pull/894">uuid-rs/uuid#894</a></li>
</ul>
<p><strong>Full Changelog</strong>: <a
href="https://github.com/uuid-rs/uuid/compare/v1.23.4...v1.23.5">https://github.com/uuid-rs/uuid/compare/v1.23.4...v1.23.5</a></p>
</blockquote>
</details>
<details>
<summary>Commits</summary>
<ul>
<li><a
href="https://github.com/uuid-rs/uuid/commit/5dc6b3d1a995e6244a386740588c8d094ca30690"><code>5dc6b3d</code></a>
Merge pull request <a
href="https://redirect.github.com/uuid-rs/uuid/issues/895">#895</a> from
uuid-rs/cargo/v1.23.5</li>
<li><a
href="https://github.com/uuid-rs/uuid/commit/5a7dfe50e2a2cf41a9d4330e00971e891bcb990f"><code>5a7dfe5</code></a>
prepare for 1.23.5 release</li>
<li><a
href="https://github.com/uuid-rs/uuid/commit/9b4bfc8fe359e24638eccf6c6be424c25ad6ba8c"><code>9b4bfc8</code></a>
Merge pull request <a
href="https://redirect.github.com/uuid-rs/uuid/issues/894">#894</a> from
geeknoid/main</li>
<li><a
href="https://github.com/uuid-rs/uuid/commit/5acc5a550ef1ccec951f1d2618b33e1171a88b9e"><code>5acc5a5</code></a>
perf: Optimize UUID hex parsing and formatting</li>
<li><a
href="https://github.com/uuid-rs/uuid/commit/1e5d8679542d2bb15412a86839006dc01f680a51"><code>1e5d867</code></a>
Merge pull request <a
href="https://redirect.github.com/uuid-rs/uuid/issues/891">#891</a> from
frostyplanet/doc</li>
<li><a
href="https://github.com/uuid-rs/uuid/commit/49310f04afd83b7d7667c1e6d7f26f93f46cedda"><code>49310f0</code></a>
doc: Fix broken link</li>
<li>See full diff in <a
href="https://github.com/uuid-rs/uuid/compare/v1.23.4...v1.23.5">compare
view</a></li>
</ul>
</details>
<br />

Updates `http-body` from 1.0.1 to 1.1.0
<details>
<summary>Commits</summary>
<ul>
<li><a
href="https://github.com/hyperium/http-body/commit/3396328602f7b147ae7b13f022c2b94dff9434e3"><code>3396328</code></a>
http-body v1.1.0</li>
<li><a
href="https://github.com/hyperium/http-body/commit/2fb78de9c875c364b7eb1a1a117acc3b83ffb13a"><code>2fb78de</code></a>
chore: bump license year (<a
href="https://redirect.github.com/hyperium/http-body/issues/170">#170</a>)</li>
<li><a
href="https://github.com/hyperium/http-body/commit/b16554b604e598466f6ae5a2689d637230d56d3e"><code>b16554b</code></a>
chore(ci): bump checkout to v7</li>
<li><a
href="https://github.com/hyperium/http-body/commit/c0c53caee7b5192e83cd2bcd273f66419b8acedc"><code>c0c53ca</code></a>
chore(ci): use msrv aware update for msrv job</li>
<li><a
href="https://github.com/hyperium/http-body/commit/5ed15d2c3d10592c82c4bab30c2cda060831bc47"><code>5ed15d2</code></a>
tests: fix clippy::double_parens</li>
<li><a
href="https://github.com/hyperium/http-body/commit/c8cb37f9ce2f8723b25e1ef1a9f6cb63ef1f9c54"><code>c8cb37f</code></a>
Derive <code>Copy</code> trait to <code>SizeHint</code> struct (<a
href="https://redirect.github.com/hyperium/http-body/issues/164">#164</a>)</li>
<li><a
href="https://github.com/hyperium/http-body/commit/915d6d5cbb5406b09f1d95978096094a1d35d5bf"><code>915d6d5</code></a>
feat(util): add <code>InspectErr</code>, <code>InspectFrame</code>
combinators (<a
href="https://redirect.github.com/hyperium/http-body/issues/161">#161</a>)</li>
<li><a
href="https://github.com/hyperium/http-body/commit/0fc0a9415cff00df921c2e8b5b6bbcb9e1a34263"><code>0fc0a94</code></a>
docs: fix broken intradoc links (<a
href="https://redirect.github.com/hyperium/http-body/issues/162">#162</a>)</li>
<li><a
href="https://github.com/hyperium/http-body/commit/5a849d49dc8ddba3382cead6d0368264fae5d827"><code>5a849d4</code></a>
chore: add FUNDING.yml</li>
<li><a
href="https://github.com/hyperium/http-body/commit/1a91851246be2ed913d6ace3f5cc18acf0d1d332"><code>1a91851</code></a>
feat: impl <code>Add</code> for <code>SizeHint</code>'s (<a
href="https://redirect.github.com/hyperium/http-body/issues/156">#156</a>)</li>
<li>Additional commits viewable in <a
href="https://github.com/hyperium/http-body/compare/v1.0.1...v1.1.0">compare
view</a></li>
</ul>
</details>
<br />

Updates `napi` from 3.10.3 to 3.10.5
<details>
<summary>Release notes</summary>
<p><em>Sourced from <a
href="https://github.com/napi-rs/napi-rs/releases">napi's
releases</a>.</em></p>
<blockquote>
<h2>napi-v3.10.5</h2>
<h3>Fixed</h3>
<ul>
<li><em>(napi)</em> release FunctionRef off the JS thread via the
custom-GC TSFN (<a
href="https://redirect.github.com/napi-rs/napi-rs/pull/3394">#3394</a>)</li>
</ul>
<h2>napi-v3.10.4</h2>
<h3>Fixed</h3>
<ul>
<li><em>(cli)</em> align build and project configuration (<a
href="https://redirect.github.com/napi-rs/napi-rs/pull/3387">#3387</a>)</li>
</ul>
<h3>Other</h3>
<ul>
<li><em>(readme)</em> point sponsors image at napi.rs/sponsors.svg (<a
href="https://redirect.github.com/napi-rs/napi-rs/pull/3379">#3379</a>)</li>
</ul>
</blockquote>
</details>
<details>
<summary>Commits</summary>
<ul>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/970988341eb7f859d2df6da1fb7b12f404a2123e"><code>9709883</code></a>
chore(napi): release v3.10.5 (<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3395">#3395</a>)</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/c931c97a82ad9da42e86c141ce92cbe322930585"><code>c931c97</code></a>
fix(napi): release FunctionRef off the JS thread via the custom-GC TSFN
(<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3394">#3394</a>)</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/3812aa748caeb1fdb72d773564827a23307b81d8"><code>3812aa7</code></a>
chore: release (<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3380">#3380</a>)</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/ce5677944b8e66e44396b435dcb154122b2b8732"><code>ce56779</code></a>
chore(release): publish</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/b9825c713ff4f871a47c8be897db9859508f4bd5"><code>b9825c7</code></a>
fix(derive): defer receiver borrow until argument conversion (<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3392">#3392</a>)</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/aa49714ed8a5619d65407ceb4ad9e79a1ee5b332"><code>aa49714</code></a>
fix(cli): align build and project configuration (<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3387">#3387</a>)</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/68cbb8d63a73d4c740c4c1c9b61b82c88e13f8b7"><code>68cbb8d</code></a>
chore(deps): update yarn to v4.17.1 (<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3385">#3385</a>)</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/3069f442c30ce3d02e218a29a865ae89d3f50847"><code>3069f44</code></a>
fix(sys): fall back to libnode.dll for symbol loading on MSVC targets
(<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3384">#3384</a>)</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/b0157131dc4086debffd321db318eb2c6c905401"><code>b015713</code></a>
fix(cli): validate cross-compilation flags upfront and document them
accurate...</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/81a35ce09c67765cdfdc06b909318e10d1345193"><code>81a35ce</code></a>
chore(deps): update dependency oxc-parser to ^0.139.0 (<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3382">#3382</a>)</li>
<li>Additional commits viewable in <a
href="https://github.com/napi-rs/napi-rs/compare/napi-v3.10.3...napi-v3.10.5">compare
view</a></li>
</ul>
</details>
<br />

Updates `napi-derive` from 3.5.9 to 3.5.10
<details>
<summary>Release notes</summary>
<p><em>Sourced from <a
href="https://github.com/napi-rs/napi-rs/releases">napi-derive's
releases</a>.</em></p>
<blockquote>
<h2>napi-derive-v3.5.10</h2>
<h3>Other</h3>
<ul>
<li>updated the following local packages: napi-derive-backend</li>
</ul>
</blockquote>
</details>
<details>
<summary>Commits</summary>
<ul>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/3812aa748caeb1fdb72d773564827a23307b81d8"><code>3812aa7</code></a>
chore: release (<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3380">#3380</a>)</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/ce5677944b8e66e44396b435dcb154122b2b8732"><code>ce56779</code></a>
chore(release): publish</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/b9825c713ff4f871a47c8be897db9859508f4bd5"><code>b9825c7</code></a>
fix(derive): defer receiver borrow until argument conversion (<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3392">#3392</a>)</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/aa49714ed8a5619d65407ceb4ad9e79a1ee5b332"><code>aa49714</code></a>
fix(cli): align build and project configuration (<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3387">#3387</a>)</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/68cbb8d63a73d4c740c4c1c9b61b82c88e13f8b7"><code>68cbb8d</code></a>
chore(deps): update yarn to v4.17.1 (<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3385">#3385</a>)</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/3069f442c30ce3d02e218a29a865ae89d3f50847"><code>3069f44</code></a>
fix(sys): fall back to libnode.dll for symbol loading on MSVC targets
(<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3384">#3384</a>)</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/b0157131dc4086debffd321db318eb2c6c905401"><code>b015713</code></a>
fix(cli): validate cross-compilation flags upfront and document them
accurate...</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/81a35ce09c67765cdfdc06b909318e10d1345193"><code>81a35ce</code></a>
chore(deps): update dependency oxc-parser to ^0.139.0 (<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3382">#3382</a>)</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/4bff1272b0c045117c74f541afe9d7b47852181e"><code>4bff127</code></a>
docs(readme): point sponsors image at napi.rs/sponsors.svg (<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3379">#3379</a>)</li>
<li><a
href="https://github.com/napi-rs/napi-rs/commit/1ac467e06e71f78b983630926c7908894d08e496"><code>1ac467e</code></a>
chore(napi): release v3.10.3 (<a
href="https://redirect.github.com/napi-rs/napi-rs/issues/3376">#3376</a>)</li>
<li>Additional commits viewable in <a
href="https://github.com/napi-rs/napi-rs/compare/napi-derive-v3.5.9...napi-derive-v3.5.10">compare
view</a></li>
</ul>
</details>
<br />

Signed-off-by: dependabot[bot] <support@github.com>
Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
2026-07-14 14:54:33 -07:00
kid 1918d1a3b6 fix(rust): skip embedding functions for empty batches (#3646)
Fixes #3174
Also fixes #3645

Empty record batches now append correctly typed empty embedding arrays
without invoking embedding providers. This avoids OpenAI requests with
an invalid empty input while preserving source-column validation and
the non-empty execution paths.

As a small cleanup, the single- and multi-embedding code paths now share
a single upfront lookup of their source columns ("input_columns")
instead
of each path looking them up independently. Also moves `lance-testing`
from regular dependencies to dev-dependencies where it belongs.

Tests run:
- `cargo fmt --all -- --check`
- `cargo test --quiet -p lancedb --lib
empty_batch_skips_embedding_functions`
- `cargo test --quiet -p lancedb --lib
empty_batch_still_validates_source_column`
- `cargo test --quiet -p lancedb --lib
test_create_empty_table_with_embeddings`
- `cargo check --quiet -p lancedb --features remote --tests --examples`
- `cargo clippy --quiet -p lancedb --features remote --tests --examples`
- `cargo test --quiet -p lancedb --lib`
- `cargo test --quiet --features remote --tests`
2026-07-14 14:46:31 -07:00
Prashanth Rao 3b626efa47 fix(python): fill bad vector values element-wise (#3613)
## Summary

Fix `on_bad_vectors="fill"` so it replaces only invalid or missing
vector values instead of replacing the entire vector row.

Fixes #3026.

## Reasoning

The old Python sanitizer detected whether a vector row was bad at row
granularity. For `fill`, it then used that row-level flag to replace the
whole vector with `[fill_value] * dim`. That meant an input like `[1.0,
NaN, 3.0]` became `[0.0, 0.0, 0.0]`, even though the documented and more
useful behavior is to preserve valid values and fill only the bad
element.

I checked whether this should be a Rust-side fix so TypeScript users
would benefit too. Today, Rust core exposes `NaNVectorBehavior::{Error,
Keep}` for rejecting or keeping NaN vectors, while the Python
`on_bad_vectors` API (`error`, `drop`, `fill`, `null`) is implemented in
the Python ingestion sanitizer before data reaches Rust. TypeScript does
not expose the Python `on_bad_vectors="fill"` behavior today. Moving
this exact behavior to Rust would be a broader cross-language API
change, so this PR keeps the fix scoped to the currently affected Python
API.

## What changed

- Added a small helper that fills bad vector rows by preserving valid
elements, replacing NaN elements with `fill_value`, truncating vectors
longer than the expected dimension, and padding short vectors with
`fill_value`.
- Kept the existing fast path unchanged: the helper only runs after bad
vectors are detected and `on_bad_vectors="fill"` is selected.
- Updated sanitizer and table tests to assert element-wise NaN
replacement and short-vector padding for both `create_table` and `add`.

## Validation

- `uv run ruff format .`
- `uv run ruff check .`
- `cd python && uv run --no-sync pytest
python/tests/test_util.py::test_handle_bad_vectors_jagged
python/tests/test_util.py::test_handle_bad_vectors_nan
python/tests/test_table.py::test_create_with_nans
python/tests/test_table.py::test_add_with_nans -vv`

Targeted pytest result: `10 passed`.

## Why this fix is Python-side (and not Rust)

The problematic behavior lives in Python’s `on_bad_vectors` sanitizer,
before data is handed off to Rust. Rust currently only exposes
`NaNVectorBehavior::{Error, Keep}` for add operations, while Python has
the richer `on_bad_vectors={"error","drop","fill","null"}` API.
TypeScript does not currently expose the Python-style fill behavior, so
moving this exact fix into Rust would require designing a broader
cross-language bad-vector handling API.

This PR keeps the change scoped to the existing affected surface:
Python’s `on_bad_vectors="fill"` path. This way, Python users
immediately benefit.
2026-07-14 13:43:17 -07:00
Prashanth Rao 137eac9b50 docs: add LanceDB agent skill for portable pipelines (#3662)
## What the new agent skill covers

We want to help users _easily_ write LanceDB pipelines to bring their
data in from other places, no matter whether they use LanceDB OSS or
Enterprise.

The `lancedb` set of skills contains guidance for agents on the
following:
- Distinguishes local and remote table capabilities.
- Promotes bounded reads using `select()` and `limit()`.
- Prevents accidental full-table materialization.
- Documents correct Python sync/async scan APIs.
- Recommends validated Python schemas and batched ingestion.
- Provides indexing, query-tuning, diagnostics, and maintenance
guidance.
- Documents the Enterprise table-name cache issue: avoid immediately
reusing a dropped or overwritten table name; write to a fresh name and
rename after propagation.
- Adds Python and TypeScript API, pattern, and performance references.
- Adds a heuristic scanner for potentially unsafe Python and TypeScript
materialization patterns.

This change only adds agent documentation and tooling: no LanceDB
runtime code, Rust code, SDK APIs, dependencies, or CI configuration are
modified.

## Context

The LanceDB agent skill was accidentally pushed directly to `main` in
`8ea78e3fbcb26718112ab4ddec55a91804b869d3`, bypassing the normal review
workflow. That commit was reverted on `main` by `c12a6dce` so the
protected branch is back to its prior content.
2026-07-14 16:34:38 -04:00
Jack Ye 06b53c97d6 feat: add table FTS query tokenization (#3659)
## Summary
- add table-level FTS query tokenization returning token text and
position
- use the native index tokenizer for local tables and remote index
metadata for remote tables
- expose sync and async Python table wrappers with focused coverage
2026-07-14 10:59:33 -07:00
Will Jones 711e05619b perf: skip Dataset::index_statistics() for all index types (#3346)
`Dataset::index_statistics()` loads index files and does meaningful CPU
work to serialize low-level info. Most fields
`NativeTable::index_stats()` needs are available from manifest metadata
via `Dataset::describe_indices()`, which is much cheaper.

`NativeTable::index_stats()` now:

- Calls `describe_indices()` filtered by name; returns `Ok(None)` if no
match.
- Parses `distance_type` from `description.details()` JSON (the
`VectorIndexDetails` proto stored in the manifest by recent Lance
versions).
- Falls back to `index_statistics()` only for vector indices where
`details()` returns no `distance_type` — this handles older Lance
datasets that didn't write `VectorIndexDetails`.
- `Unknown` index types (e.g. Lance's internal `FragReuseIndex`) are
explicitly filtered out of `list_indices` rather than erroring.

## Test plan
- [x] `test_create_scalar_index` — asserts `index_type`,
`distance_type`, and `num_unindexed_rows > 0` after adding rows
post-index
- [x] `test_create_fm_index`, `test_create_bitmap_index`,
`test_create_label_list_index` — added `index_stats` assertions
- [x] IvfPq, IvfHnswPq, IvfHnswSq, IvfHnswFlat tests assert
`distance_type == Some(L2)`
- [x] `test_list_indices_skip_frag_reuse` — FragReuseIndex is filtered
by the Unknown guard in `list_indices`

---------

Co-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-07-14 09:45:50 -07:00
Weston Pace afc0e5f497 chore: upgrade spin dependency in lock file to avoid yanked version (#3663) 2026-07-14 09:01:01 -07:00
prrao87 c12a6dce9f Revert "add LanceDB agent skill for portable pipelines"
This reverts commit 8ea78e3fbc.
2026-07-14 10:57:36 -04:00
prrao87 8ea78e3fbc add LanceDB agent skill for portable pipelines 2026-07-14 10:02:13 -04:00
kid 40238d240a fix(python): preserve phrase semantics in sync queries (#3654)
## Summary

- serialize sync phrase queries consistently for execution and query
plans
- restore the documented no-argument hybrid `phrase_query()` behavior
- keep reranker input as the original user text without mutating the
builder

Fixes #3653.

## Testing

- `python/.venv/bin/python -m pytest <8 focused test nodes> -q` (`8
passed`)
- `python/.venv/bin/python -m ruff format --check
python/python/lancedb/query.py python/python/tests/test_fts.py
python/python/tests/test_hybrid_query.py`
- `python/.venv/bin/python -m ruff check .`
- `git diff --check origin/main...HEAD`

The complete hybrid module and the real native FTS phrase test were not
completed
in the current PyO3 runtime environment: both stalled in the native
`lancedb.connect()` fixture and were interrupted without an assertion
failure.
2026-07-13 23:44:35 -07:00
dependabot[bot] 60428e1a32 chore(deps): bump rand from 0.9.4 to 0.10.1 (#3648)
Bumps [rand](https://github.com/rust-random/rand) from 0.9.4 to 0.10.1.
<details>
<summary>Changelog</summary>
<p><em>Sourced from <a
href="https://github.com/rust-random/rand/blob/master/CHANGELOG.md">rand's
changelog</a>.</em></p>
<blockquote>
<h2>[0.10.1] — 2026-02-11</h2>
<p>This release includes a fix for a soundness bug; see <a
href="https://redirect.github.com/rust-random/rand/issues/1763">#1763</a>.</p>
<h3>Changes</h3>
<ul>
<li>Document panic behavior of <code>make_rng</code> and add
<code>#[track_caller]</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1761">#1761</a>)</li>
<li>Deprecate feature <code>log</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1763">#1763</a>)</li>
</ul>
<p><a
href="https://redirect.github.com/rust-random/rand/issues/1761">#1761</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1761">rust-random/rand#1761</a>
<a
href="https://redirect.github.com/rust-random/rand/issues/1763">#1763</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1763">rust-random/rand#1763</a></p>
<h2>[0.10.0] - 2026-02-08</h2>
<h3>Changes</h3>
<ul>
<li>The dependency on <code>rand_chacha</code> has been replaced with a
dependency on <code>chacha20</code>. This changes the implementation
behind <code>StdRng</code>, but the output remains the same. There may
be some API breakage when using the ChaCha-types directly as these are
now the ones in <code>chacha20</code> instead of
<code>rand_chacha</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1642">#1642</a>).</li>
<li>Rename fns <code>IndexedRandom::choose_multiple</code> -&gt;
<code>sample</code>, <code>choose_multiple_array</code> -&gt;
<code>sample_array</code>, <code>choose_multiple_weighted</code> -&gt;
<code>sample_weighted</code>, struct <code>SliceChooseIter</code> -&gt;
<code>IndexedSamples</code> and fns
<code>IteratorRandom::choose_multiple</code> -&gt; <code>sample</code>,
<code>choose_multiple_fill</code> -&gt; <code>sample_fill</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1632">#1632</a>)</li>
<li>Use Edition 2024 and MSRV 1.85 (<a
href="https://redirect.github.com/rust-random/rand/issues/1653">#1653</a>)</li>
<li>Let <code>Fill</code> be implemented for element types, not
sliceable types (<a
href="https://redirect.github.com/rust-random/rand/issues/1652">#1652</a>)</li>
<li>Fix <code>OsError::raw_os_error</code> on UEFI targets by returning
<code>Option&lt;usize&gt;</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1665">#1665</a>)</li>
<li>Replace fn <code>TryRngCore::read_adapter(..) -&gt;
RngReadAdapter</code> with simpler struct <code>RngReader</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1669">#1669</a>)</li>
<li>Remove fns <code>SeedableRng::from_os_rng</code>,
<code>try_from_os_rng</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1674">#1674</a>)</li>
<li>Remove <code>Clone</code> support for <code>StdRng</code>,
<code>ReseedingRng</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1677">#1677</a>)</li>
<li>Use <code>postcard</code> instead of <code>bincode</code> to test
the serde feature (<a
href="https://redirect.github.com/rust-random/rand/issues/1693">#1693</a>)</li>
<li>Avoid excessive allocation in <code>IteratorRandom::sample</code>
when <code>amount</code> is much larger than iterator size (<a
href="https://redirect.github.com/rust-random/rand/issues/1695">#1695</a>)</li>
<li>Rename <code>os_rng</code> -&gt; <code>sys_rng</code>,
<code>OsRng</code> -&gt; <code>SysRng</code>, <code>OsError</code> -&gt;
<code>SysError</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1697">#1697</a>)</li>
<li>Rename <code>Rng</code> -&gt; <code>RngExt</code> as upstream
<code>rand_core</code> has renamed <code>RngCore</code> -&gt;
<code>Rng</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1717">#1717</a>)</li>
</ul>
<h3>Additions</h3>
<ul>
<li>Add fns <code>IndexedRandom::choose_iter</code>,
<code>choose_weighted_iter</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1632">#1632</a>)</li>
<li>Pub export <code>Xoshiro128PlusPlus</code>,
<code>Xoshiro256PlusPlus</code> prngs (<a
href="https://redirect.github.com/rust-random/rand/issues/1649">#1649</a>)</li>
<li>Pub export <code>ChaCha8Rng</code>, <code>ChaCha12Rng</code>,
<code>ChaCha20Rng</code> behind <code>chacha</code> feature (<a
href="https://redirect.github.com/rust-random/rand/issues/1659">#1659</a>)</li>
<li>Fn <code>rand::make_rng() -&gt; R where R: SeedableRng</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1734">#1734</a>)</li>
</ul>
<h3>Removals</h3>
<ul>
<li>Removed <code>ReseedingRng</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1722">#1722</a>)</li>
<li>Removed unused feature &quot;nightly&quot; (<a
href="https://redirect.github.com/rust-random/rand/issues/1732">#1732</a>)</li>
<li>Removed feature <code>small_rng</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1732">#1732</a>)</li>
</ul>
<p><a
href="https://redirect.github.com/rust-random/rand/issues/1632">#1632</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1632">rust-random/rand#1632</a>
<a
href="https://redirect.github.com/rust-random/rand/issues/1642">#1642</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1642">rust-random/rand#1642</a>
<a
href="https://redirect.github.com/rust-random/rand/issues/1649">#1649</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1649">rust-random/rand#1649</a>
<a
href="https://redirect.github.com/rust-random/rand/issues/1652">#1652</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1652">rust-random/rand#1652</a>
<a
href="https://redirect.github.com/rust-random/rand/issues/1653">#1653</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1653">rust-random/rand#1653</a>
<a
href="https://redirect.github.com/rust-random/rand/issues/1659">#1659</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1659">rust-random/rand#1659</a>
<a
href="https://redirect.github.com/rust-random/rand/issues/1665">#1665</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1665">rust-random/rand#1665</a>
<a
href="https://redirect.github.com/rust-random/rand/issues/1669">#1669</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1669">rust-random/rand#1669</a>
<a
href="https://redirect.github.com/rust-random/rand/issues/1674">#1674</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1674">rust-random/rand#1674</a>
<a
href="https://redirect.github.com/rust-random/rand/issues/1677">#1677</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1677">rust-random/rand#1677</a>
<a
href="https://redirect.github.com/rust-random/rand/issues/1693">#1693</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1693">rust-random/rand#1693</a>
<a
href="https://redirect.github.com/rust-random/rand/issues/1695">#1695</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1695">rust-random/rand#1695</a>
<a
href="https://redirect.github.com/rust-random/rand/issues/1697">#1697</a>:
<a
href="https://redirect.github.com/rust-random/rand/pull/1697">rust-random/rand#1697</a></p>
<!-- raw HTML omitted -->
</blockquote>
<p>... (truncated)</p>
</details>
<details>
<summary>Commits</summary>
<ul>
<li><a
href="https://github.com/rust-random/rand/commit/27ff4cb7ced3122a1f677fc248c1a07e59ddc8cd"><code>27ff4cb</code></a>
Prepare v0.10.1: deprecate feature <code>log</code> (<a
href="https://redirect.github.com/rust-random/rand/issues/1763">#1763</a>)</li>
<li><a
href="https://github.com/rust-random/rand/commit/98d06386dc4e1d1c89a91f4e483d571921c29ecf"><code>98d0638</code></a>
make_rng: document panic and add #[track_caller] (<a
href="https://redirect.github.com/rust-random/rand/issues/1761">#1761</a>)</li>
<li><a
href="https://github.com/rust-random/rand/commit/54e5eaaa7ac11af3aa60b5ccc486182189e6f9ef"><code>54e5eaa</code></a>
Fix doc error (<a
href="https://redirect.github.com/rust-random/rand/issues/1758">#1758</a>)</li>
<li><a
href="https://github.com/rust-random/rand/commit/1ce4c080186730595a8d464591d17aac22a42252"><code>1ce4c08</code></a>
Bump itoa from 1.0.17 to 1.0.18 in the all-deps group (<a
href="https://redirect.github.com/rust-random/rand/issues/1756">#1756</a>)</li>
<li><a
href="https://github.com/rust-random/rand/commit/ccb734b9c22891a19f11be125c2f09a43809b08e"><code>ccb734b</code></a>
docs: fix typo in doc comment (<a
href="https://redirect.github.com/rust-random/rand/issues/1754">#1754</a>)</li>
<li><a
href="https://github.com/rust-random/rand/commit/357eb7de9c9c80184449e8b515c821e48cf4df74"><code>357eb7d</code></a>
Bump libc from 0.2.182 to 0.2.183 in the all-deps group (<a
href="https://redirect.github.com/rust-random/rand/issues/1753">#1753</a>)</li>
<li><a
href="https://github.com/rust-random/rand/commit/5e77fe5d61b886988cae67b6d8fb09e405845c63"><code>5e77fe5</code></a>
Fix trait references in documentation (<a
href="https://redirect.github.com/rust-random/rand/issues/1752">#1752</a>)</li>
<li><a
href="https://github.com/rust-random/rand/commit/da891850ab2b38f4322ec140ae29d305dfb162c3"><code>da89185</code></a>
Bump the all-deps group with 3 updates (<a
href="https://redirect.github.com/rust-random/rand/issues/1751">#1751</a>)</li>
<li><a
href="https://github.com/rust-random/rand/commit/50516ff45c3675d9c2d247e70bc8db691ed8366d"><code>50516ff</code></a>
Bump the all-deps group with 2 updates (<a
href="https://redirect.github.com/rust-random/rand/issues/1749">#1749</a>)</li>
<li><a
href="https://github.com/rust-random/rand/commit/fd71de97fdc7050b9a2d8384f5f8afce7d991ca3"><code>fd71de9</code></a>
Bump the all-deps group with 2 updates (<a
href="https://redirect.github.com/rust-random/rand/issues/1747">#1747</a>)</li>
<li>Additional commits viewable in <a
href="https://github.com/rust-random/rand/compare/0.9.4...0.10.1">compare
view</a></li>
</ul>
</details>
<br />


[![Dependabot compatibility
score](https://dependabot-badges.githubapp.com/badges/compatibility_score?dependency-name=rand&package-manager=cargo&previous-version=0.9.4&new-version=0.10.1)](https://docs.github.com/en/github/managing-security-vulnerabilities/about-dependabot-security-updates#about-compatibility-scores)

Dependabot will resolve any conflicts with this PR as long as you don't
alter it yourself. You can also trigger a rebase manually by commenting
`@dependabot rebase`.

[//]: # (dependabot-automerge-start)
[//]: # (dependabot-automerge-end)

---

<details>
<summary>Dependabot commands and options</summary>
<br />

You can trigger Dependabot actions by commenting on this PR:
- `@dependabot rebase` will rebase this PR
- `@dependabot recreate` will recreate this PR, overwriting any edits
that have been made to it
- `@dependabot show <dependency name> ignore conditions` will show all
of the ignore conditions of the specified dependency
- `@dependabot ignore this major version` will close this PR and stop
Dependabot creating any more for this major version (unless you reopen
the PR or upgrade to it yourself)
- `@dependabot ignore this minor version` will close this PR and stop
Dependabot creating any more for this minor version (unless you reopen
the PR or upgrade to it yourself)
- `@dependabot ignore this dependency` will close this PR and stop
Dependabot creating any more for this dependency (unless you reopen the
PR or upgrade to it yourself)


</details>

Signed-off-by: dependabot[bot] <support@github.com>
Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
2026-07-13 16:01:21 -07:00
Mateusz Szewczyk 5b982f2f05 feat(python): added support for WatsonxReranker component (#3642)
## Summary

Adds `WatsonxReranker` to the Python bindings, integrating the [IBM
watsonx.ai text rerank
API](https://cloud.ibm.com/docs/apis/watsonx-ai#text-rerank) via the
`ibm_watsonx_ai` SDK (`pip install ibm-watsonx-ai`).

## Parameters

| Parameter | Default | Description |
|---|---|---|
| `model_name` | `"cross-encoder/ms-marco-minilm-l-12-v2"` | Rerank
model ID |
| `column` | `"text"` | Table column used as document input |
| `top_n` | `None` | Return only the top-n results |
| `return_score` | `"relevance"` | `"relevance"` or `"all"` |
| `api_key` | `None` | Falls back to `WATSONX_API_KEY` env var |
| `project_id` | `None` | Falls back to `WATSONX_PROJECT_ID` env var —
mutually exclusive with `space_id` |
| `space_id` | `None` | Falls back to `WATSONX_SPACE_ID` env var —
mutually exclusive with `project_id` |
| `url` | `None` | Defaults to `https://us-south.ml.cloud.ibm.com` |
| `truncate_input_tokens` | `None` | Token truncation limit |

## Usage

```python
from lancedb.rerankers import WatsonxReranker

# credentials from environment variables
reranker = WatsonxReranker()

# or passed explicitly
reranker = WatsonxReranker(
    api_key="<key>",
    project_id="<project-id>",   # or space_id="<space-id>"
    top_n=5,
)
```

## Testing

Integration test added in `test_rerankers.py`, skipped unless
`WATSONX_API_KEY` and one of `WATSONX_PROJECT_ID` / `WATSONX_SPACE_ID`
are set.
2026-07-13 15:58:32 -07:00
Will Jones cde48fad95 ci: remove CODEOWNERS file (#3655)
The CODEOWNERS file added in #3312 automatically requests reviewers on
every PR — the `*` default owner routes all changes to two reviewers.
This is mostly noise for contributors, and we prefer a single requested
reviewer per PR.

Remove the file.

Reverts #3312.

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-13 14:22:40 -07:00
Mark McDonald 1f2068b9fe fix(python): gemini batching, user agent and variable dims (#3618)
Carrying over from #2915, this patch introduces:
* Single-API call batching support for Gemini embeddings (up to 100 at a
time, the API limit)
* A versioned user agent header for Gemini API calls
* Support for [variable embedding dimension
size](https://ai.google.dev/gemini-api/docs/embeddings#control-embedding-size)
(Gemini is MRL trained)
2026-07-13 12:28:33 -07:00
kid 7527890607 fix(python): preserve zero distance bounds in hybrid search (#3652)
## Summary

- preserve explicit `0.0` distance bounds in synchronous hybrid search
- distinguish omitted `None` endpoints from zero-valued endpoints when
configuring the vector child query
- add a public end-to-end regression test for a zero upper bound

## Testing

- `cd python && uv run --extra tests pytest
python/tests/test_hybrid_query.py -q`
- `uv run --project python ruff format --check
python/python/lancedb/query.py python/python/tests/test_hybrid_query.py`
- `uv run --project python ruff check .`

Fixes #3651
2026-07-13 12:28:26 -07:00
Drew Gallardo a548e59d49 feat(python): blob v2 fetch API (#3578)
Python bindings for blob v2 read on **local** tables. Rust read APIs
landed in #3562.

This PR wires `fetch_blob_files`, `fetch_blobs`, v2
query/`to_pandas(blob_mode="bytes")`, and hidden `_rowid` metadata so
`fetch_*` works from query hits without exposing `_rowid` in the column
list.

**Cloud:** `RemoteTable.fetch_blobs` / `fetch_blob_files` raise
`NotImplementedError` until Phalanx ships the server route (separate
track; not blocking local merge).

### Primary path: lazy file handles

```python
table = db.create_table("videos", schema=pa.schema([
    pa.field("id", pa.int64()),
    lancedb.blob("video"),
]))
table.add([{"id": 1, "video": open("clip.mp4", "rb").read()}])

hits = table.search().select(["id", "video"]).to_arrow()
handle = table.fetch_blob_files("video", hits)[0]

# seek + partial read — PyAV / decoders can use the handle
handle.seek(frame_offset)
chunk = handle.read_range(0, 65536)
```

`BlobFile` exposes `seek`, `read`, `read_range`, `read_up_to`, and works
with `BufferedReader`.

### When you want full bytes

```python
blobs = table.fetch_blobs("video", hits)  # eager materialize, null-aligned
df = table.to_pandas(blob_mode="bytes")   # descriptors → bytes in pandas
```

### `_rowid` (join key, not user `id`)

Fetch needs Lance row ids. For v2 blob queries we auto-inject `_rowid`,
stash it in Arrow schema metadata on `to_arrow()`, and drop the visible
column unless you pass `.with_row_id(True)`.

v1 legacy blobs (`lance-encoding:blob`) unchanged; fetch on v1 raises
the migration error.

## Test plan

- [x] `./scripts/test-blob.sh python` (105 passed in worktree)
- [x] `fetch_blob_files` lazy read, seek, partial read, null alignment,
cross-fragment dups
- [x] hybrid query → `fetch_blobs` / `fetch_blob_files`
- [ ] Will re-review after seek/`BlobFile` commit (`d77ab1a6`)

---------

Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-10 12:54:16 -07:00
Lance Release 104fc5a08e Bump version: 0.32.0-beta.0 → 0.32.0-beta.1 2026-07-10 16:13:35 +00:00
72 changed files with 4970 additions and 420 deletions
+9 -1
View File
@@ -1,6 +1,14 @@
---
name: lancedb-branch-ops
description: Branch management for LanceDB tables via the REST API. Use this skill whenever someone wants to create, delete, list, or switch branches on a LanceDB table — or needs to make sure a write (metadata update, index build, etc.) lands on a specific branch instead of main. Invoke it even without the word "branch" if context makes clear they want an experimental copy of a table, want to isolate changes, or want to confirm a mutation didn't touch main. Covers: branches/list, branches/create, branches/delete, and passing "branch" in describe/update_field_metadata/create_index to target a non-main version.
description: >-
Manage LanceDB table branches through the REST API: list, create, and delete
branches; target schema reads, field-metadata updates, and index creation to a
named branch; and verify that branch changes remain isolated from main. Use
when a task involves branch lifecycle, an experimental or isolated table
version, directing an operation to a non-main branch, or confirming that a
mutation did not affect main. This skill also explains that LanceDB has no
checkout operation; each request selects its target branch in the request
body.
---
## Goal
+79
View File
@@ -0,0 +1,79 @@
---
name: lancedb
description: Use when writing, reviewing, debugging, or documenting LanceDB pipelines in Python or TypeScript, especially code that should work across local LanceDB OSS tables and remote LanceDB Enterprise/Cloud tables. Helps avoid non-portable full-table materialization, choose idiomatic query/search patterns, and apply LanceDB performance defaults for ingestion, indexing, filtering, and diagnostics.
---
# Building LanceDB Pipelines
Use this skill to produce LanceDB pipelines that are portable between local and remote tables (for LanceDB Enterprise/Cloud) and idiomatic for the selected SDK.
## LanceDB Table Modes
LanceDB has two common execution modes:
- **Local table**: embedded, open source, in-process LanceDB. The client opens data from a local path or object storage URI and executes queries in the application process.
- **Remote table**: LanceDB Enterprise/Cloud table opened through a `db://...` URI. The data may be very large, commonly backed by object storage, and queried through a remote service.
Do NOT assume local-only table helpers exist on remote tables. If the user asks for LanceDB Enterprise, Cloud, `db://...`, production remote access, or a remote table, focus on the remote table path: use `search()` / `query()`, keep reads bounded with `select()` and `limit()`, and avoid table-level full materialization APIs.
## Workflow
1. Identify the SDK: Python, TypeScript, or both.
2. Identify the table mode: local/embedded OSS, remote Enterprise/Cloud, or portable across both. If the user says "LanceDB Enterprise", choose the remote table path.
3. Read the matching language branch before writing or changing code:
- Python patterns: `references/python/patterns.md`
- Python API quick reference: `references/python/api_reference.md`
- Python performance guidance: `references/python/performance.md`
- TypeScript patterns: `references/typescript/patterns.md`
- TypeScript API quick reference: `references/typescript/api_reference.md`
- TypeScript performance guidance: `references/typescript/performance.md`
4. Start with `patterns.md` for the selected SDK. Read `api_reference.md` when choosing method names or return collectors. Read `performance.md` when the task involves ingestion, indexing, filtering, query tuning, diagnostics, or large datasets.
5. For Python schemas, favor Pydantic models and validate records before writing. Use PyArrow schemas when Arrow-native, streaming, or highly dynamic data makes them materially better suited.
6. Prefer `search()` or `query()` builders with explicit `select()` and `limit()` for reads.
7. Avoid table-level full materialization in remote or portable code. This is the main local-vs-remote read pitfall.
8. After a successful embedded OSS ingestion, call `table.optimize()`. Do not call it for Enterprise/Cloud; remote maintenance is automatic.
9. For remote Enterprise/Cloud writes, never drop-then-reuse or `mode="overwrite"` the same table name — see "Enterprise: never drop-then-reuse the same table name" below. This is the main local-vs-remote write pitfall.
10. If reviewing an existing file or repo, run `scripts/check_materialization.py` on the relevant paths and inspect each finding before editing.
11. Cross-check unfamiliar or non-trivial API claims against the source tree instead of relying on memory.
## Core Portability Rule
Do not write code that assumes a local table API will exist on a remote table. Remote tables can be very large, so whole-table materialization helpers are intentionally unavailable or unsafe.
This does **not** mean result conversion is forbidden. Bounded query/search result collection is normal:
- Python: `table.search(...).select([...]).limit(10).to_pandas()`
- TypeScript: `await table.search(...).select([...]).limit(10).toArray()`
The unsafe pattern is table-level or unbounded collection, plus local-only dataset escape hatches in remote code:
- Python: `table.to_pandas()`, `table.to_arrow()`, `table.to_polars()`; `table.to_lance()` is local/OSS-only dataset access, not materialization
- TypeScript: `await table.toArrow()`, `await table.query().toArray()` without `limit()`
## Enterprise: never drop-then-reuse the same table name
LanceDB Enterprise/Cloud splits a **control plane** (DDL: create/drop/rename) from a **data plane** (query nodes that serve reads). Query nodes cache the resolved dataset for a table name for up to `table_cache_ttl`**default 300 seconds (5 minutes)**. After you drop or overwrite a table, the control plane updates immediately but the data plane keeps serving the *old* dataset until that cache entry expires. During the window the two planes disagree.
The failure this causes: you `drop_table("t")` then immediately `create_table("t", ...)` (or `create_table("t", ..., mode="overwrite")`). The DDL returns success, but every query against `t` returns **`500 Internal Server Error`** (the query node resolves the stale/deleted dataset), and a fresh `describe` may still show the *old* schema/version. It looks like your write silently failed; it didn't — the name is cached.
**`mode="overwrite"` has the same problem** — it is a drop+create of the same name under the hood.
Rules for portable Enterprise ingestion:
1. **Never reuse a table name you just dropped/overwrote within the cache TTL.** Do not use `mode="overwrite"` to replace an existing Enterprise table in place.
2. To (re)load data, **write to a fresh table name** (e.g. `<table>_v2`, or a run-stamped suffix). A brand-new name has no cached data-plane entry, so writes and reads work immediately.
3. Before creating, `list_tables()` and **fail loudly if the name already exists** rather than overwriting — prompt for a new name.
4. To land on a specific final name that is currently occupied by an old table: drop the old table, **wait out the TTL (~5 min), then `rename_table(fresh_name, final_name)`**. Renaming onto a name whose old dataset is still cached hits the same race, so the wait is mandatory. `rename_table` is a supported control-plane op.
5. When you hand a table name back to a human, tell them which step still needs the propagation wait (usually: "the old `t` was dropped; run the rename in ~5 minutes").
This is Enterprise/Cloud-specific. Local/OSS tables have no separate data plane, so `mode="overwrite"` and immediate same-name reuse are fine there.
## Script
Run the scanner when reviewing or modifying an existing codebase:
```bash
python skills/lancedb/scripts/check_materialization.py path/to/file_or_dir
```
The script reports likely unsafe full-table materialization in Python and TypeScript. Treat results as review prompts, not automatic proof of a bug.
@@ -0,0 +1,105 @@
# Python API Reference
Quick method reference for Python LanceDB code. Cross-check source for non-trivial claims.
## Connect
```python
import lancedb
db = lancedb.connect("./camelot-db") # local/OSS
db = lancedb.connect("db://my-db", api_key=api_key, region=region) # remote
```
**Place the local database directory next to the script/entrypoint that opens it** (i.e. resolve the path relative to the script, `Path(__file__).parent / "camelot-db"`), not buried under a shared `data/` folder. The Lance dataset is the database, not a data file — keeping it beside its code makes ownership obvious and paths stable regardless of the working directory the script is launched from.
**Do not name the directory `lancedb`** (e.g. `./lancedb`, `./data/lancedb`). It collides with the imported `lancedb` package name, which is confusing to read and easy to shadow in scripts. Give it a name derived from the repo or dataset with a clear prefix/suffix — for example `./<dataset>-db`, `./<repo>_lancedb`, or `./vectordb`.
Async:
```python
db = await lancedb.connect_async("./camelot-db")
```
## Table Reads
| Task | Preferred API |
| --- | --- |
| Vector search | `table.search(query_vector).limit(k)` |
| Full scan with filters/projection (sync) | `table.search().where(...).select(...).limit(...)` |
| Full scan with filters/projection (async) | `table.query().where(...).select(...).limit(...)` |
| Filter | `.where("col > 10")` |
| Projection | `.select(["id", "text"])` |
| Bound result count | `.limit(20)` |
| Collect bounded result as Python objects (default, no extra deps) | `.to_list()` on query/search result |
| Collect bounded result as Arrow (default, `pyarrow` always available) | `.to_arrow()` on query/search result |
| Collect bounded result as pandas (only if project uses pandas) | `.to_pandas()` on query/search result |
| Collect bounded result as Polars (only if project uses polars) | `.to_polars()` on query/search result |
## Sync vs Async Scan API
The plain-scan entry point differs between the sync and async clients. **Verified against `lancedb` 0.34.0** — re-check if the pinned version changes:
- **Sync** (`lancedb.connect(...)`): the table has **no `.query()` method**. Use `.search()` with no argument for a plain scan; it returns a query builder that supports `.where()`, `.select()`, `.limit()`, and the `.to_list()` / `.to_arrow()` / `.to_pandas()` / `.to_polars()` collectors.
```python
rows = table.search().where("status = 'ready'").select(["id", "text"]).limit(20).to_list()
```
- **Async** (`lancedb.connect_async(...)`): the table has **both** `.query()` and `.search()`. Use `.query()` for a plain scan.
```python
rows = await async_table.query().where("status = 'ready'").select(["id", "text"]).limit(20).to_list()
```
Do not call `table.query()` on a sync table — it raises `AttributeError`.
## Local vs Remote Table Methods
| API | Local table | Remote table | Agent guidance |
| --- | --- | --- | --- |
| `table.search(...)` | Yes | Yes | Preferred read path (sync + async) |
| `table.query()` | Async only | Async only | Sync scan path is `table.search()`; `.query()` is the async scan builder |
| `table.to_pandas()` | Yes | No / unsafe for portability | Avoid in portable code |
| `table.to_arrow()` | Yes | No / unsafe for portability | Avoid in portable code |
| `table.to_polars()` | Yes | No / unsafe for portability | Avoid in portable code |
| `table.to_lance()` | Yes | No | Local/OSS escape hatch only |
## Indexes
Use `create_index(...)` for vector indexes and modern index configs. Use scalar indexes for filtered or merge keys.
Common calls:
```python
table.create_index("vector")
table.create_scalar_index("status")
table.create_fts_index("text")
```
Check source docs before specifying advanced index config names or parameters.
## Filtering And Recall Knobs
```python
table.search(query_vector).where("status = 'ready'") # pre-filter by default
table.search(query_vector).where("status = 'ready'", prefilter=False)
table.search(query_vector).limit(10).refine_factor(20)
table.search(query_vector).limit(10).nprobes(50)
```
Use post-filtering only when fewer than `limit` results are acceptable.
## Diagnostics
```python
print(table.search(query_vector).where("year > 2000").limit(10).analyze_plan())
print(table.index_stats("vector_idx"))
```
Use these before changing indexes or search tuning.
## Maintenance
```python
table.optimize()
```
Call this after every successful local/OSS ingestion. It handles compaction, cleanup of old versions according to retention, and index optimization. Do not add this for LanceDB Enterprise/Cloud remote tables; Enterprise handles compaction and cleanup automatically from cluster configuration.
@@ -0,0 +1,173 @@
# Python Patterns
Use these patterns when writing Python code with `lancedb`.
## Before Writing Code
Choose the output type from what the project actually depends on. **Do not assume `pandas` or `polars` is installed** — they are heavy dependencies that many LanceDB projects do not use. `pyarrow`, by contrast, ships as a LanceDB dependency and is always available, so it is a safe default to lean on.
Default output (after applying `select()` and `limit()`):
- **Python objects**: `.to_list()` — a list of dicts, no extra dependencies. Prefer this for scripts, examples, and agent-generated code unless there is a reason to do otherwise.
- **PyArrow**: `.to_arrow()` — a `pyarrow.Table`, when the surrounding code is Arrow-native or you need columnar/zero-copy handoff.
Only reach for a DataFrame when the project *already* declares that dependency:
- Pandas projects (pandas in `pyproject.toml`/requirements): `.to_pandas()`.
- Polars projects (polars declared): `.to_polars()`.
If unsure, check the dependency manifest or the imports in surrounding files. When in doubt, use `.to_list()` or `.to_arrow()`.
## Schema Design and Validation
Favor `LanceModel` and Pydantic validation for Python schemas. They keep field
types readable, validate source records before a write, and map directly to a
LanceDB schema. Use `Vector(dimension)` for fixed-size vectors:
```python
from lancedb.pydantic import LanceModel, Vector
class Document(LanceModel):
id: int
text: str
vector: Vector(384, nullable=False)
rows = [Document.model_validate(row) for row in source_rows]
table = db.create_table("documents", schema=Document)
table.add(rows)
```
Use PyArrow schemas instead when the pipeline is already Arrow-native, needs
record-batch streaming, or has runtime schema requirements that would make a
Pydantic model harder to understand. Declare Pydantic as a direct project
dependency when application code imports it, even if LanceDB also depends on it.
## Recommended Patterns
### Bounded search or query
Use this for application reads, examples, notebooks, and agent-generated scripts:
```python
results = (
table.search(query_vector)
.where("status = 'ready'")
.select(["id", "text"])
.limit(20)
.to_list() # or .to_arrow(); .to_pandas()/.to_polars() only if the project uses them
)
```
Why: `search()` works across local and remote tables and on both the sync and async clients. `select()` avoids fetching unused columns. `limit()` prevents accidental full-table reads. `.to_list()` and `.to_arrow()` avoid assuming pandas/polars is installed (see "Before Writing Code").
For a **plain scan** (no query vector), the entry point differs by client:
```python
# Sync client: no .query() method — use .search() with no argument.
rows = table.search().where("status = 'ready'").select(["id", "text"]).limit(20).to_list()
# Async client: use .query().
rows = await async_table.query().where("status = 'ready'").select(["id", "text"]).limit(20).to_list()
```
`table.query()` on a sync table raises `AttributeError` (verified on `lancedb` 0.34.0). See the "Sync vs Async Scan API" section in `api_reference.md`.
### Bounded query result conversion
It is fine to collect bounded query/search results:
```python
arrow_table = table.search().select(["id"]).limit(100).to_arrow() # sync plain scan
rows = table.search(query_vector).limit(10).to_list()
df = table.search(query_vector).limit(10).to_pandas() # only if pandas is a project dep
```
### Local-only Lance dataset API
`table.to_lance()` does not itself materialize the full dataset. It returns the underlying `lance.LanceDataset`, making the table accessible through the PyLance dataset API. Use it when the task is explicitly local/OSS and needs Lance dataset methods not exposed by LanceDB:
```python
# Local/OSS only: RemoteTable does not expose table.to_lance().
ds = table.to_lance()
for batch in ds.to_batches(columns=["id", "text"], batch_size=10_000):
process(batch)
```
### Async Python
Keep the same shape and bound the result before collecting:
```python
results = await (
async_table.query()
.where("status = 'ready'")
.select(["id", "text"])
.limit(20)
.to_list() # or .to_arrow()
)
```
## Anti-Patterns
**Avoid the following anti-patterns in your code.**
### Table-level full materialization
Avoid whole-table collectors in portable or large-table code:
```python
df = table.to_pandas()
arrow_table = table.to_arrow()
polars_df = table.to_polars()
```
Why: local tables expose these whole-table collectors, but remote tables intentionally do not — a remote production table can be far larger than a local development table, so it is easy to accidentally pull the entire table into memory.
`table.to_lance()` is different: it is not a full materialization call, but it is still local/OSS-only and should not appear in code meant to run against remote Enterprise tables.
### Unbounded result collection
Avoid query/search collection without a meaningful limit:
```python
rows = table.search().to_list() # unbounded plain scan
rows = table.search(query_vector).to_list() # unbounded vector search
```
Prefer `select(...).limit(...)` before collecting; for large reads, stream in batches instead.
### Per-row writes
Avoid loops that write one row per call:
```python
for row in rows:
table.add([row]) # one commit + fragment per row
```
Each `add()` creates a new version and fragment. Pass the whole batch in a single call, or chunk very large inputs:
```python
table.add(rows) # single commit
# for very large inputs, add batches of several thousand rows
```
After the final successful write to an embedded OSS table, call
`table.optimize()`. Skip this for Enterprise/Cloud tables because their
maintenance is automatic.
### Drop-then-reuse the same table name (Enterprise/Cloud)
Avoid dropping or overwriting a remote table and then reusing that name right away:
```python
db.drop_table("my_table")
table = db.create_table("my_table", data=rows) # reads 500 for ~5 min
table = db.create_table("my_table", data=rows, mode="overwrite") # same problem
```
Why: Enterprise/Cloud splits DDL (control plane) from query serving (data plane). The data plane caches the dataset behind a table name for up to `table_cache_ttl` (default 300s / 5 min), so after a drop/overwrite the DDL succeeds but queries against the reused name return `500 Internal Server Error` until the cache expires — and a fresh `describe` may still show the old schema. Instead, write to a **fresh name**, use `list_tables()` and fail if it already exists, then `rename_table(fresh, final)` onto the final name only after the old table's drop has propagated (~5 min). See the "Enterprise: never drop-then-reuse the same table name" section in `SKILL.md`. Local/OSS tables have no separate data plane — overwrite freely there.
### Guessing performance fixes
Avoid changing `nprobes`, `refine_factor`, or index types before checking the query plan and index stats. Diagnose first, then tune one knob at a time.
@@ -0,0 +1,131 @@
# Python Performance Guidance
Use this when writing Python code that ingests data, queries large tables, builds indexes, or investigates latency.
## Ingestion
### Recommended: validate schemas and records with Pydantic
Favor `LanceModel` for readable Python schema definitions and validate source
records before writing. Use PyArrow directly for Arrow-native or streaming
pipelines where it is the clearer representation.
```python
from lancedb.pydantic import LanceModel, Vector
class Document(LanceModel):
id: int
text: str
vector: Vector(384, nullable=False)
rows = [Document.model_validate(row) for row in source_rows]
table = db.create_table("documents", schema=Document)
table.add(rows)
```
### Recommended: bulk ingestion for materialized data
```python
table.add(arrow_table)
table.add(df)
table.add(pa.dataset("data/", format="parquet"))
```
For very large initial loads, create the table empty first, then call `add(...)`. Passing data directly to `create_table(name, data)` can skip the auto-parallel write path.
### Recommended: iterator ingestion for generated or streamed data
```python
def batches():
for raw in source:
vectors = model.encode(raw["text"])
yield pa.RecordBatch.from_pydict({**raw, "vector": vectors})
table.add(batches())
```
Use chunks of several thousand rows or more when practical. Tiny batches and per-row writes create many small fragments.
### Anti-pattern: per-row `add()`
```python
for row in rows:
table.add([row])
```
Each call creates a version and fragment. This slows ingestion and later queries.
## Indexing
- Build a vector index once brute-force vector search becomes too slow. As a rule of thumb, local brute force is fine below roughly 100K vectors; beyond that, build an index.
- Use `IVF_PQ` as the general-purpose default. Enterprise builds this automatically.
- Use scalar indexes for filtered columns and merge/upsert keys.
- Use `BTREE` for mostly distinct numeric/string/temporal columns, `BITMAP` for booleans and low-cardinality columns, and `LABEL_LIST` for list membership queries.
- Keep full-text defaults unless phrase queries require position data.
## Querying
Always be explicit:
```python
table.search(query_vector).select(["id", "title"]).limit(20)
```
- `select()` reduces bytes read and transferred.
- `limit()` prevents accidental full-table materialization.
- Pre-filtering is the default and guarantees returned rows satisfy the predicate.
- Use post-filtering only when fewer than `limit` results are acceptable.
## Recall Tuning
Tune one knob at a time:
- Quantized indexes: raise `refine_factor` to rescore more candidates on full vectors.
- HNSW-backed indexes: raise `ef`; start around `1.5 * k`, increase toward `10 * k` if recall is short.
- IVF candidate breadth: `nprobes` is auto-tuned; override only when a selective pre-filter leaves too few neighbors.
## Maintenance
After every successful embedded OSS/local ingestion, call `table.optimize()`.
Do not add this to LanceDB Enterprise/Cloud remote table code; remote compaction
and cleanup are handled automatically based on the Enterprise cluster
configuration.
Why local maintenance is needed:
- Frequent writes can create many small fragments. Queries then need to scan across more files, which can increase latency.
- Updates, deletes, and appends create new table versions. Old versions are retained for time travel and rollback, which can grow disk usage.
- Indexes may have newly added rows that are not yet fully optimized into the index structure.
For local/OSS tables, run `optimize()` after the final successful ingestion
write. Also run it after later batches of update/delete operations or on a
regular maintenance schedule:
```python
table.optimize()
```
If the user wants more aggressive local disk cleanup, pass a shorter cleanup retention window:
```python
from datetime import timedelta
table.optimize(cleanup_older_than=timedelta(days=1))
```
Do not use very short cleanup windows when the application depends on time travel, rollback, or old versions.
## Diagnostics
Before changing code or indexes, inspect:
```python
print(table.search(query_vector).where("year > 2000").limit(10).analyze_plan())
print(table.index_stats("vector_idx"))
```
Look for high scan bytes, missing indexes, fragmented data, and unindexed rows.
## Python Multiprocessing
When using multiprocessing, use `spawn` rather than `fork`. LanceDB is multi-threaded internally, and `fork` plus a multi-threaded process is unsafe.
@@ -0,0 +1,78 @@
# TypeScript API Reference
Quick method reference for TypeScript LanceDB code. Cross-check source for non-trivial claims.
## Connect
```typescript
import * as lancedb from "@lancedb/lancedb";
const db = await lancedb.connect("./camelot-db");
```
**Place the local database directory next to the script/entrypoint that opens it** (resolve the path relative to the module, e.g. via `import.meta.dirname` / `__dirname`), not buried under a shared `data/` folder. The Lance dataset is the database, not a data file — keeping it beside its code makes ownership obvious and paths stable regardless of the working directory the script is launched from.
**Do not name the directory `lancedb`** (e.g. `./lancedb`, `./data/lancedb`). It collides with the imported `lancedb` package/namespace, which is confusing to read. Give it a name derived from the repo or dataset with a clear prefix/suffix — for example `./<dataset>-db`, `./<repo>_lancedb`, or `./vectordb`.
Remote connections use `db://...` plus Enterprise/Cloud credentials and deployment settings. Check current source/docs for exact connection options.
## Table Reads
| Task | Preferred API |
| --- | --- |
| Vector search | `table.search(queryVector).limit(k)` |
| Full scan with filters/projection | `table.query().where(...).select(...).limit(...)` |
| Filter | `.where("col > 10")` |
| Projection | `.select(["id", "text"])` |
| Bound result count | `.limit(20)` |
| Collect bounded result as objects | `.toArray()` on query/search result |
| Collect bounded result as Arrow | `.toArrow()` on query/search result |
| Stream result batches | `for await (const batch of table.query()...)` |
## Local vs Remote Safety
| API | Agent guidance |
| --- | --- |
| `table.search(...)` | Preferred read path |
| `table.query()` | Preferred scan/filter path |
| `await table.toArrow()` | Avoid in portable or large-table code |
| `await table.query().toArray()` with no `limit()` | Avoid; unbounded collection |
| `await table.query().toArrow()` with no `limit()` | Avoid; unbounded collection |
## Indexes
```typescript
await table.createIndex("vector");
await table.createIndex("status");
```
Use vector indexes for large vector search workloads and scalar indexes for filtered columns or merge/upsert keys. Check source/docs before specifying advanced index options.
## Filtering And Recall Knobs
```typescript
await table.search(queryVector).where("status = 'ready'").limit(10).toArray();
await table.search(queryVector).limit(10).refineFactor(20).toArray();
await table.search(queryVector).limit(10).nprobes(50).toArray();
await table.search(queryVector).limit(10).ef(100).toArray();
await table.search(queryVector).where("status = 'ready'").postfilter().limit(10).toArray();
```
Use `postfilter()` only when fewer than `limit` results are acceptable.
## Diagnostics
```typescript
console.log(await table.search(queryVector).where("year > 2000").limit(10).analyzePlan());
console.log(await table.indexStats("vector_idx"));
```
Use these before changing indexes or search tuning.
## Maintenance
```typescript
await table.optimize();
```
Call this after every successful local/OSS ingestion. It handles compaction, cleanup of old versions according to retention, and index optimization. Do not add this for LanceDB Enterprise/Cloud remote tables; Enterprise handles compaction and cleanup automatically from cluster configuration.
@@ -0,0 +1,100 @@
# TypeScript Patterns
Use these patterns when writing TypeScript code with `@lancedb/lancedb`.
## Recommended Patterns
### Bounded query
Use this for application reads, scripts, and examples:
```typescript
const rows = await table
.query()
.where("status = 'ready'")
.select(["id", "text"])
.limit(20)
.toArray();
```
### Bounded vector search
```typescript
const rows = await table
.search(queryVector)
.select(["id", "text"])
.limit(20)
.toArray();
```
### Batch streaming for larger reads
When the task needs many rows, avoid collecting everything at once:
```typescript
for await (const batch of table
.query()
.where("status = 'ready'")
.select(["id", "text"])
.limit(10_000)) {
process(batch);
}
```
## Anti-Patterns
**Avoid the following anti-patterns in your code.**
### Table-level full materialization
Avoid whole-table collectors in portable or large-table code:
```typescript
const tableArrow = await table.toArrow();
```
Why: local tables expose these whole-table collectors, but remote tables intentionally do not — a remote production table can be far larger than a local development table, so it is easy to accidentally pull the entire table into memory.
### Unbounded result collection
Avoid query/search collection without a meaningful limit:
```typescript
const rows = await table.query().toArray(); // unbounded plain scan
const rows = await table.search(queryVector).toArray(); // unbounded vector search
```
Prefer `select(...).limit(...)` before collecting; for large reads, stream in batches instead.
### Per-row writes
Avoid loops that write one row per call:
```typescript
for (const row of rows) {
await table.add([row]); // one commit + fragment per row
}
```
Each `add()` creates a new version and fragment. Pass the whole batch in a single call, or chunk very large inputs:
```typescript
await table.add(rows); // single commit
// for very large inputs, add in chunks of several thousand rows
```
### Drop-then-reuse the same table name (Enterprise/Cloud)
Avoid dropping or overwriting a remote table and then reusing that name right away:
```typescript
await db.dropTable("my_table");
const table = await db.createTable("my_table", rows); // reads 500 for ~5 min
const table = await db.createTable("my_table", rows, { mode: "overwrite" }); // same problem
```
Why: Enterprise/Cloud splits DDL (control plane) from query serving (data plane). The data plane caches the dataset behind a table name for up to `table_cache_ttl` (default 300s / 5 min), so after a drop/overwrite the DDL succeeds but queries against the reused name return `500 Internal Server Error` until the cache expires — and a fresh `describe` may still show the old schema. Instead, write to a **fresh name**, use `tableNames()` and fail if it already exists, then `renameTable(fresh, final)` onto the final name only after the old table's drop has propagated (~5 min). See the "Enterprise: never drop-then-reuse the same table name" section in `SKILL.md`. Local/OSS tables have no separate data plane — overwrite freely there.
### Guessing performance fixes
Avoid changing `nprobes`, `refineFactor`, `ef`, or index settings before checking `analyzePlan()` and `indexStats(...)`. Diagnose first, then tune one knob at a time.
@@ -0,0 +1,78 @@
# TypeScript Performance Guidance
Use this when writing TypeScript code that ingests data, queries large tables, builds indexes, or investigates latency.
## Ingestion
- Prefer bulk or batched writes.
- Avoid per-row write loops; they create many small commits/fragments.
- For generated data, accumulate reasonable batches before adding.
- For file-backed data, prefer APIs that stream from Arrow/Parquet-style inputs when available.
## Indexing
- Build a vector index once brute-force vector search becomes too slow. As a rule of thumb, local brute force is fine below roughly 100K vectors; beyond that, build an index.
- Use the general-purpose vector index defaults unless the task has explicit recall/latency requirements.
- Build scalar indexes for filtered columns and merge/upsert keys.
- Use full-text index phrase options only when phrase queries require them.
## Querying
Always be explicit:
```typescript
await table.search(queryVector).select(["id", "title"]).limit(20).toArray();
```
- `select()` reduces bytes read and transferred.
- `limit()` prevents accidental full-table collection.
- Pre-filtering is the default behavior. Use `postfilter()` only when fewer than `limit` results are acceptable.
## Recall Tuning
Tune one knob at a time:
- Quantized indexes: raise `refineFactor(...)` to rescore more candidates on full vectors.
- HNSW-backed indexes: raise `ef(...)`; start around `1.5 * k`, increase toward `10 * k` if recall is short.
- IVF candidate breadth: `nprobes(...)` is usually auto-tuned; override only when a selective pre-filter leaves too few neighbors.
## Maintenance
After every successful embedded OSS/local ingestion, call `table.optimize()`.
Do not add this to LanceDB Enterprise/Cloud remote table code; remote compaction
and cleanup are handled automatically based on the Enterprise cluster
configuration.
Why local maintenance is needed:
- Frequent writes can create many small fragments. Queries then need to scan across more files, which can increase latency.
- Updates, deletes, and appends create new table versions. Old versions are retained for time travel and rollback, which can grow disk usage.
- Indexes may have newly added rows that are not yet fully optimized into the index structure.
For local/OSS tables, run `optimize()` after the final successful ingestion
write. Also run it after later batches of update/delete operations or on a
regular maintenance schedule:
```typescript
await table.optimize();
```
If the user wants more aggressive local disk cleanup, pass a shorter cleanup retention window:
```typescript
const olderThan = new Date(Date.now() - 24 * 60 * 60 * 1000);
await table.optimize({ cleanupOlderThan: olderThan });
```
Do not use very short cleanup windows when the application depends on time travel, rollback, or old versions.
## Diagnostics
Before changing code or indexes, inspect:
```typescript
console.log(await table.search(queryVector).where("year > 2000").limit(10).analyzePlan());
console.log(await table.indexStats("vector_idx"));
```
Look for high scan cost, missing indexes, fragmented data, and unindexed rows.
@@ -0,0 +1,135 @@
#!/usr/bin/env python3
"""Scan Python and TypeScript for likely unsafe LanceDB materialization."""
from __future__ import annotations
import argparse
import re
import sys
from dataclasses import dataclass
from pathlib import Path
PY_FULL_TABLE = re.compile(r"\b\w+\.(to_pandas|to_arrow|to_polars)\s*\(")
TS_TABLE_TO_ARROW = re.compile(r"\b\w+\.toArrow\s*\(")
TS_QUERY_COLLECTOR = re.compile(r"\.query\s*\(\s*\)[\s\S]*?\.to(Array|Arrow)\s*\(")
@dataclass
class Finding:
path: Path
line: int
message: str
text: str
def iter_files(paths: list[Path]) -> list[Path]:
files: list[Path] = []
for path in paths:
if path.is_dir():
files.extend(
p
for p in path.rglob("*")
if p.suffix in {".py", ".ts", ".tsx"} and "node_modules" not in p.parts
)
elif path.suffix in {".py", ".ts", ".tsx"}:
files.append(path)
return sorted(set(files))
def line_number(text: str, offset: int) -> int:
return text.count("\n", 0, offset) + 1
def scan_python(path: Path, text: str) -> list[Finding]:
findings: list[Finding] = []
for match in PY_FULL_TABLE.finditer(text):
line_start = text.rfind("\n", 0, match.start()) + 1
line_end = text.find("\n", match.start())
if line_end == -1:
line_end = len(text)
line = text[line_start:line_end].strip()
if ".search(" in line or ".query(" in line:
continue
findings.append(
Finding(
path,
line_number(text, match.start()),
f"Review Python `{match.group(1)}()` call; table-level materialization is not portable to remote tables.",
line,
)
)
return findings
def statement_around(text: str, start: int, end: int) -> str:
before = max(text.rfind(";", 0, start), text.rfind("\n\n", 0, start))
after_candidates = [pos for pos in (text.find(";", end), text.find("\n\n", end)) if pos != -1]
after = min(after_candidates) if after_candidates else len(text)
return text[before + 1 : after].strip()
def scan_typescript(path: Path, text: str) -> list[Finding]:
findings: list[Finding] = []
for match in TS_TABLE_TO_ARROW.finditer(text):
stmt = statement_around(text, match.start(), match.end())
if ".query(" in stmt or ".search(" in stmt:
continue
findings.append(
Finding(
path,
line_number(text, match.start()),
"Review TypeScript `table.toArrow()`-style call; table-level materialization is not portable for large/remote tables.",
stmt.splitlines()[0].strip(),
)
)
for match in TS_QUERY_COLLECTOR.finditer(text):
stmt = statement_around(text, match.start(), match.end())
if ".limit(" in stmt:
continue
findings.append(
Finding(
path,
line_number(text, match.start()),
"Review unbounded TypeScript query collection; add `limit()` or stream batches.",
stmt.splitlines()[0].strip(),
)
)
return findings
def scan_file(path: Path) -> list[Finding]:
text = path.read_text(encoding="utf-8", errors="replace")
if path.suffix == ".py":
return scan_python(path, text)
if path.suffix in {".ts", ".tsx"}:
return scan_typescript(path, text)
return []
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("paths", nargs="+", type=Path)
parser.add_argument(
"--no-fail", action="store_true", help="Always exit 0 after reporting findings."
)
args = parser.parse_args()
findings: list[Finding] = []
for path in iter_files(args.paths):
findings.extend(scan_file(path))
for finding in findings:
print(f"{finding.path}:{finding.line}: {finding.message}")
print(f" {finding.text}")
if findings:
print(
f"\n{len(findings)} finding(s). Review manually; bounded query result conversion may be OK."
)
return 0 if args.no_fail or not findings else 1
if __name__ == "__main__":
sys.exit(main())
+1 -1
View File
@@ -1,5 +1,5 @@
[tool.bumpversion]
current_version = "0.32.0-beta.0"
current_version = "0.32.0-beta.1"
parse = """(?x)
(?P<major>0|[1-9]\\d*)\\.
(?P<minor>0|[1-9]\\d*)\\.
-21
View File
@@ -1,21 +0,0 @@
# CODEOWNERS
#
# These owners will be the default owners for everything in the repo.
# They will be requested for review when someone opens a pull request.
#
# See https://docs.github.com/en/repositories/managing-your-repositorys-settings-and-features/customizing-your-repository/about-code-owners
# Default owners for everything
* @jackye1995 @wjones127
# Release and publish workflows — changes here can affect supply chain security
/.github/workflows/ @jackye1995 @wjones127 @Xuanwo
# Remote client and auth — sensitive networking and auth code
/rust/lancedb/src/remote/ @jackye1995 @wjones127
# Python FFI boundary
/python/src/ @jackye1995 @wjones127 @AyushExel
# NodeJS FFI boundary
/nodejs/src/ @jackye1995 @wjones127
+20 -4
View File
@@ -125,10 +125,26 @@ jobs:
- uses: rui314/setup-mold@v1
- name: Make Swap
run: |
sudo fallocate -l 16G /swapfile
sudo chmod 600 /swapfile
sudo mkswap /swapfile
sudo swapon /swapfile
swapfile=/swapfile
min_swap_bytes=$((15 * 1024 * 1024 * 1024))
active_swap_bytes="$(sudo swapon --show=NAME,SIZE --bytes --noheadings | awk '$1 == "/swapfile" { print $2 }')"
if [ -n "$active_swap_bytes" ]; then
if [ "$active_swap_bytes" -ge "$min_swap_bytes" ]; then
echo "/swapfile is already active with enough space; skipping swap creation"
exit 0
fi
echo "/swapfile is already active but smaller than 16G; using /mnt/lancedb-swapfile"
swapfile=/mnt/lancedb-swapfile
fi
if sudo swapon --show=NAME --noheadings | grep -Fxq "$swapfile"; then
echo "$swapfile is already active; skipping swap creation"
exit 0
fi
sudo rm -f "$swapfile"
sudo fallocate -l 16G "$swapfile"
sudo chmod 600 "$swapfile"
sudo mkswap "$swapfile"
sudo swapon "$swapfile"
- name: Build
run: cargo build --profile ci --all-features --tests --locked --examples
- name: Run feature tests
Generated
+115 -115
View File
@@ -666,7 +666,7 @@ dependencies = [
"http 0.2.12",
"http 1.4.2",
"http-body 0.4.6",
"http-body 1.0.1",
"http-body 1.1.0",
"percent-encoding",
"pin-project-lite",
"tracing",
@@ -774,7 +774,7 @@ dependencies = [
"hmac 0.13.0",
"http 0.2.12",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"lru",
"percent-encoding",
"regex-lite",
@@ -907,7 +907,7 @@ dependencies = [
"crc-fast",
"hex",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"http-body-util",
"md-5 0.11.0",
"pin-project-lite",
@@ -941,7 +941,7 @@ dependencies = [
"futures-core",
"futures-util",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"http-body-util",
"percent-encoding",
"pin-project-lite",
@@ -1025,7 +1025,7 @@ dependencies = [
"http 0.2.12",
"http 1.4.2",
"http-body 0.4.6",
"http-body 1.0.1",
"http-body 1.1.0",
"http-body-util",
"pin-project-lite",
"pin-utils",
@@ -1086,7 +1086,7 @@ dependencies = [
"http 0.2.12",
"http 1.4.2",
"http-body 0.4.6",
"http-body 1.0.1",
"http-body 1.1.0",
"http-body-util",
"itoa",
"num-integer",
@@ -1133,7 +1133,7 @@ dependencies = [
"bytes",
"futures-util",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"http-body-util",
"hyper 1.9.0",
"hyper-util",
@@ -1166,7 +1166,7 @@ dependencies = [
"bytes",
"futures-util",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"http-body-util",
"mime",
"pin-project-lite",
@@ -1463,9 +1463,9 @@ checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b"
[[package]]
name = "bytes"
version = "1.12.0"
version = "1.12.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8ae3f5d315924270530207e2a68396c3cc547f6dca3fbdca317cfb1a51edb593"
checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04"
[[package]]
name = "bytes-utils"
@@ -1491,7 +1491,7 @@ dependencies = [
"memmap2 0.9.10",
"num-traits",
"num_cpus",
"rand 0.9.4",
"rand 0.9.5",
"rand_distr 0.5.1",
"rayon",
"safetensors",
@@ -1527,7 +1527,7 @@ dependencies = [
"candle-nn",
"fancy-regex",
"num-traits",
"rand 0.9.4",
"rand 0.9.5",
"rayon",
"serde",
"serde_json",
@@ -1961,7 +1961,7 @@ dependencies = [
"crc",
"digest 0.10.7",
"rustversion",
"spin 0.10.0",
"spin 0.10.1",
]
[[package]]
@@ -2326,7 +2326,7 @@ dependencies = [
"log",
"object_store",
"parking_lot",
"rand 0.9.4",
"rand 0.9.5",
"regex",
"sqlparser 0.61.0",
"tempfile",
@@ -2441,7 +2441,7 @@ dependencies = [
"itertools 0.14.0",
"log",
"object_store",
"rand 0.9.4",
"rand 0.9.5",
"tokio",
"url",
]
@@ -2541,7 +2541,7 @@ dependencies = [
"log",
"object_store",
"parking_lot",
"rand 0.9.4",
"rand 0.9.5",
"tempfile",
"url",
]
@@ -2606,7 +2606,7 @@ dependencies = [
"md-5 0.10.6",
"memchr",
"num-traits",
"rand 0.9.4",
"rand 0.9.5",
"regex",
"sha2 0.10.9",
"unicode-segmentation",
@@ -3384,7 +3384,7 @@ checksum = "719a903cc23e4a89e87962c2a80fdb45cdaad0983a89bd150bb57b4c8571a7d5"
dependencies = [
"half",
"num-traits",
"rand 0.9.4",
"rand 0.9.5",
"rand_distr 0.5.1",
]
@@ -3429,11 +3429,11 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c"
[[package]]
name = "fsst"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow-array",
"rand 0.9.4",
"rand 0.9.5",
]
[[package]]
@@ -3797,7 +3797,7 @@ dependencies = [
"hostname",
"prost",
"prost-types",
"rand 0.9.4",
"rand 0.9.5",
"reqwest 0.12.28",
"serde",
"thiserror 2.0.18",
@@ -3868,7 +3868,7 @@ dependencies = [
"cfg-if 1.0.4",
"crunchy",
"num-traits",
"rand 0.9.4",
"rand 0.9.5",
"rand_distr 0.5.1",
"zerocopy",
]
@@ -3961,7 +3961,7 @@ dependencies = [
"libc",
"log",
"num_cpus",
"rand 0.9.4",
"rand 0.9.5",
"reqwest 0.12.28",
"serde",
"serde_json",
@@ -4065,9 +4065,9 @@ dependencies = [
[[package]]
name = "http-body"
version = "1.0.1"
version = "1.1.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1efedce1fb8e6913f23e0c92de8e62cd5b772a67e7b3946df930a62566c93184"
checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c"
dependencies = [
"bytes",
"http 1.4.2",
@@ -4082,7 +4082,7 @@ dependencies = [
"bytes",
"futures-core",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"pin-project-lite",
]
@@ -4149,7 +4149,7 @@ dependencies = [
"futures-core",
"h2 0.4.14",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"httparse",
"httpdate",
"itoa",
@@ -4215,7 +4215,7 @@ dependencies = [
"futures-channel",
"futures-util",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"hyper 1.9.0",
"ipnet",
"libc",
@@ -4728,7 +4728,7 @@ dependencies = [
"nom 8.0.0",
"num-traits",
"ordered-float 5.3.0",
"rand 0.9.4",
"rand 0.9.5",
"serde",
"serde_json",
"zmij",
@@ -4780,8 +4780,8 @@ checksum = "e037a2e1d8d5fdbd49b16a4ea09d5d6401c1f29eca5ff29d03d3824dba16256a"
[[package]]
name = "lance"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arc-swap",
"arrow",
@@ -4837,7 +4837,7 @@ dependencies = [
"prost",
"prost-build",
"prost-types",
"rand 0.9.4",
"rand 0.9.5",
"rayon",
"roaring",
"rustc-hash",
@@ -4855,8 +4855,8 @@ dependencies = [
[[package]]
name = "lance-arrow"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow-array",
"arrow-buffer",
@@ -4872,13 +4872,13 @@ dependencies = [
"half",
"jsonb",
"num-traits",
"rand 0.9.4",
"rand 0.9.5",
]
[[package]]
name = "lance-arrow-scalar"
version = "58.0.0"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow-array",
"arrow-buffer",
@@ -4892,7 +4892,7 @@ dependencies = [
[[package]]
name = "lance-arrow-stats"
version = "58.0.0"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow-array",
"arrow-schema",
@@ -4901,8 +4901,8 @@ dependencies = [
[[package]]
name = "lance-bitpacking"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrayref",
"crunchy",
@@ -4912,8 +4912,8 @@ dependencies = [
[[package]]
name = "lance-core"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow-array",
"arrow-buffer",
@@ -4936,7 +4936,7 @@ dependencies = [
"object_store",
"pin-project",
"prost",
"rand 0.9.4",
"rand 0.9.5",
"roaring",
"serde_json",
"snafu 0.9.0",
@@ -4951,8 +4951,8 @@ dependencies = [
[[package]]
name = "lance-datafusion"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow",
"arrow-array",
@@ -4982,8 +4982,8 @@ dependencies = [
[[package]]
name = "lance-datagen"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow",
"arrow-array",
@@ -4993,15 +4993,15 @@ dependencies = [
"futures",
"half",
"hex",
"rand 0.9.4",
"rand 0.9.5",
"rand_distr 0.5.1",
"rand_xoshiro",
]
[[package]]
name = "lance-derive"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"proc-macro2",
"quote",
@@ -5010,8 +5010,8 @@ dependencies = [
[[package]]
name = "lance-encoding"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow-arith",
"arrow-array",
@@ -5036,7 +5036,7 @@ dependencies = [
"num-traits",
"prost",
"prost-build",
"rand 0.9.4",
"rand 0.9.5",
"strum 0.26.3",
"tokio",
"tracing",
@@ -5046,8 +5046,8 @@ dependencies = [
[[package]]
name = "lance-file"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow-arith",
"arrow-array",
@@ -5077,8 +5077,8 @@ dependencies = [
[[package]]
name = "lance-index"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arc-swap",
"arrow",
@@ -5126,7 +5126,7 @@ dependencies = [
"prost",
"prost-build",
"prost-types",
"rand 0.9.4",
"rand 0.9.5",
"rand_distr 0.5.1",
"rangemap",
"rayon",
@@ -5143,8 +5143,8 @@ dependencies = [
[[package]]
name = "lance-io"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow",
"arrow-arith",
@@ -5176,7 +5176,7 @@ dependencies = [
"path_abs",
"pin-project",
"prost",
"rand 0.9.4",
"rand 0.9.5",
"serde",
"tempfile",
"tokio",
@@ -5186,8 +5186,8 @@ dependencies = [
[[package]]
name = "lance-linalg"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow-array",
"arrow-buffer",
@@ -5197,14 +5197,14 @@ dependencies = [
"lance-arrow",
"lance-core",
"num-traits",
"rand 0.9.4",
"rand 0.9.5",
"rayon",
]
[[package]]
name = "lance-namespace"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow",
"async-trait",
@@ -5216,8 +5216,8 @@ dependencies = [
[[package]]
name = "lance-namespace-impls"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow",
"arrow-ipc",
@@ -5241,7 +5241,7 @@ dependencies = [
"log",
"object_store",
"quick-xml 0.40.1",
"rand 0.9.4",
"rand 0.9.5",
"reqwest 0.12.28",
"roaring",
"serde",
@@ -5271,8 +5271,8 @@ dependencies = [
[[package]]
name = "lance-select"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow-array",
"arrow-buffer",
@@ -5287,8 +5287,8 @@ dependencies = [
[[package]]
name = "lance-table"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow",
"arrow-array",
@@ -5312,7 +5312,7 @@ dependencies = [
"prost",
"prost-build",
"prost-types",
"rand 0.9.4",
"rand 0.9.5",
"rangemap",
"roaring",
"semver",
@@ -5327,8 +5327,8 @@ dependencies = [
[[package]]
name = "lance-testing"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"arrow-array",
"arrow-schema",
@@ -5336,13 +5336,13 @@ dependencies = [
"lance-arrow",
"num-traits",
"pprof 0.15.0",
"rand 0.9.4",
"rand 0.9.5",
]
[[package]]
name = "lance-tokenizer"
version = "9.0.0-beta.19"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.19#8f0e6d3a7c53438275b134c0ac1afbc80600616e"
version = "9.0.0-beta.23"
source = "git+https://github.com/lance-format/lance.git?tag=v9.0.0-beta.23#0acc51eb8f013985395bf3ac7f0ef4f8a23a377d"
dependencies = [
"icu_segmenter",
"jieba-rs",
@@ -5355,7 +5355,7 @@ dependencies = [
[[package]]
name = "lancedb"
version = "0.32.0-beta.0"
version = "0.32.0-beta.1"
dependencies = [
"ahash",
"anyhow",
@@ -5394,7 +5394,7 @@ dependencies = [
"half",
"hf-hub",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"lance",
"lance-arrow",
"lance-core",
@@ -5420,7 +5420,7 @@ dependencies = [
"polars",
"polars-arrow",
"pprof 0.14.1",
"rand 0.9.4",
"rand 0.9.5",
"random_word",
"regex",
"reqwest 0.12.28",
@@ -5443,7 +5443,7 @@ dependencies = [
[[package]]
name = "lancedb-nodejs"
version = "0.32.0-beta.0"
version = "0.32.0-beta.1"
dependencies = [
"arrow-array",
"arrow-buffer",
@@ -5468,7 +5468,7 @@ dependencies = [
[[package]]
name = "lancedb-python"
version = "0.35.0-beta.0"
version = "0.35.0-beta.1"
dependencies = [
"arrow",
"async-trait",
@@ -5500,7 +5500,7 @@ version = "1.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe"
dependencies = [
"spin 0.9.8",
"spin 0.9.9",
]
[[package]]
@@ -5895,7 +5895,7 @@ dependencies = [
"ordered-float 4.6.0",
"quanta",
"radix_trie",
"rand 0.9.4",
"rand 0.9.5",
"rand_xoshiro",
"sketches-ddsketch",
]
@@ -6041,9 +6041,9 @@ dependencies = [
[[package]]
name = "napi"
version = "3.10.3"
version = "3.10.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0c71997d6f7ad4a756966e452426848ac27d3b37a295302d63afbbcce0270f93"
checksum = "6826e5ddc15589b2d68c8ad5321c18e85d40488e93e32962f362e572669bccf6"
dependencies = [
"bitflags 2.11.1",
"chrono",
@@ -6066,9 +6066,9 @@ checksum = "c9c366d2c8c60b86fa632df75f745509b52f9128f91a6bad4c796e44abb505e1"
[[package]]
name = "napi-derive"
version = "3.5.9"
version = "3.5.10"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d4ba572deef53e2c386759a8c2014175a62679d74ff83adc205c8bc0e0285727"
checksum = "b0fe526e81c105d3640516fcde83909dd1afe757c0d7a15af58830b5bc0fb9a1"
dependencies = [
"convert_case",
"ctor 1.0.5",
@@ -6080,9 +6080,9 @@ dependencies = [
[[package]]
name = "napi-derive-backend"
version = "5.1.1"
version = "5.1.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ddd961eb2aa8965e3f29722d754f3a86907eb1984e2fbcbe3fe87b9a02d6bfba"
checksum = "514281397bcddd9ea9a876c7a21a57bff2374237a000ca9a64ea0211ec1993e2"
dependencies = [
"convert_case",
"proc-macro2",
@@ -6093,9 +6093,9 @@ dependencies = [
[[package]]
name = "napi-sys"
version = "3.2.2"
version = "3.2.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1f5bcdf71abd3a50d00b49c1c2c75251cb3c913777d6139cd37dabc093a5e400"
checksum = "73e43cf2eb0bd1bf95a43c07c076ebd2da5d1e015a71c3d201faeffffcc0ecac"
dependencies = [
"libloading",
]
@@ -6458,7 +6458,7 @@ dependencies = [
"bytes",
"futures",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"jiff",
"log",
"md-5 0.11.0",
@@ -7482,7 +7482,7 @@ dependencies = [
"nix",
"once_cell",
"smallvec",
"spin 0.10.0",
"spin 0.10.1",
"symbolic-demangle",
"tempfile",
"thiserror 1.0.69",
@@ -7504,7 +7504,7 @@ dependencies = [
"nix",
"once_cell",
"smallvec",
"spin 0.10.0",
"spin 0.10.1",
"symbolic-demangle",
"tempfile",
"thiserror 2.0.18",
@@ -7809,7 +7809,7 @@ dependencies = [
"bytes",
"getrandom 0.3.4",
"lru-slab",
"rand 0.9.4",
"rand 0.9.5",
"ring",
"rustc-hash",
"rustls 0.23.40",
@@ -7894,9 +7894,9 @@ dependencies = [
[[package]]
name = "rand"
version = "0.9.4"
version = "0.9.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "44c5af06bb1b7d3216d91932aed5265164bf384dc89cd6ba05cf59a35f5f76ea"
checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41"
dependencies = [
"rand_chacha 0.9.0",
"rand_core 0.9.5",
@@ -7974,7 +7974,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6a8615d50dcf34fa31f7ab52692afec947c4dd0ab803cc87cb3b0b4570ff7463"
dependencies = [
"num-traits",
"rand 0.9.4",
"rand 0.9.5",
]
[[package]]
@@ -8138,9 +8138,9 @@ dependencies = [
[[package]]
name = "regex"
version = "1.12.4"
version = "1.13.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f1292b7759ae1cb9ec195452d1390a074f0cd8541ab7a5a8c31cd6db45d4a6ba"
checksum = "2a0e75113e14dc5acb068cd0786884f214f1312650a3d36d269f5c4f3cdee8a2"
dependencies = [
"aho-corasick",
"memchr",
@@ -8327,7 +8327,7 @@ dependencies = [
"futures-util",
"h2 0.4.14",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"http-body-util",
"hyper 1.9.0",
"hyper-rustls 0.27.9",
@@ -8371,7 +8371,7 @@ dependencies = [
"futures-core",
"futures-util",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"http-body-util",
"hyper 1.9.0",
"hyper-rustls 0.27.9",
@@ -9296,15 +9296,15 @@ dependencies = [
[[package]]
name = "spin"
version = "0.9.8"
version = "0.9.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6980e8d7511241f8acf4aebddbb1ff938df5eebe98691418c4468d0b72a96a67"
checksum = "3763264f6b73151db08c50ff20d7d8a0b8796e021cdea7ceedad07b80155fa0e"
[[package]]
name = "spin"
version = "0.10.0"
version = "0.10.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d5fe4ccb98d9c292d56fec89a5e07da7fc4cf0dc11e156b41793132775d3e591"
checksum = "023a211cb3138dbc438680b32560ad89f699977624c9f8dbb95a47d5b4c07dd3"
dependencies = [
"lock_api",
]
@@ -10016,7 +10016,7 @@ dependencies = [
"bytes",
"h2 0.4.14",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"http-body-util",
"hyper 1.9.0",
"hyper-timeout",
@@ -10072,7 +10072,7 @@ dependencies = [
"bitflags 2.11.1",
"bytes",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"http-body-util",
"pin-project-lite",
"tower-layer",
@@ -10092,7 +10092,7 @@ dependencies = [
"futures-core",
"futures-util",
"http 1.4.2",
"http-body 1.0.1",
"http-body 1.1.0",
"http-body-util",
"pin-project-lite",
"tokio",
@@ -10215,7 +10215,7 @@ version = "2.1.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9ea3136b675547379c4bd395ca6b938e5ad3c3d20fad76e7fe85f9e0d011419c"
dependencies = [
"rand 0.9.4",
"rand 0.9.5",
]
[[package]]
@@ -10380,9 +10380,9 @@ checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821"
[[package]]
name = "uuid"
version = "1.23.4"
version = "1.23.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bf80a72845275afea99e7f2b434723d3bc7e38470fcd1c7ed39a599c73319a53"
checksum = "ea5fab0d6c3c01ae70085a09cb03d4c7a1d6314e2b3e075392783396d724ca0a"
dependencies = [
"getrandom 0.4.2",
"js-sys",
+14 -14
View File
@@ -13,20 +13,20 @@ categories = ["database-implementations"]
rust-version = "1.91.0"
[workspace.dependencies]
lance = { "version" = "=9.0.0-beta.19", default-features = false, "tag" = "v9.0.0-beta.19", "git" = "https://github.com/lance-format/lance.git" }
lance-core = { "version" = "=9.0.0-beta.19", "tag" = "v9.0.0-beta.19", "git" = "https://github.com/lance-format/lance.git" }
lance-datagen = { "version" = "=9.0.0-beta.19", "tag" = "v9.0.0-beta.19", "git" = "https://github.com/lance-format/lance.git" }
lance-file = { "version" = "=9.0.0-beta.19", "tag" = "v9.0.0-beta.19", "git" = "https://github.com/lance-format/lance.git" }
lance-io = { "version" = "=9.0.0-beta.19", default-features = false, "tag" = "v9.0.0-beta.19", "git" = "https://github.com/lance-format/lance.git" }
lance-index = { "version" = "=9.0.0-beta.19", "tag" = "v9.0.0-beta.19", "git" = "https://github.com/lance-format/lance.git" }
lance-linalg = { "version" = "=9.0.0-beta.19", "tag" = "v9.0.0-beta.19", "git" = "https://github.com/lance-format/lance.git" }
lance-namespace = { "version" = "=9.0.0-beta.19", "tag" = "v9.0.0-beta.19", "git" = "https://github.com/lance-format/lance.git" }
lance-namespace-impls = { "version" = "=9.0.0-beta.19", default-features = false, "tag" = "v9.0.0-beta.19", "git" = "https://github.com/lance-format/lance.git" }
lance-table = { "version" = "=9.0.0-beta.19", "tag" = "v9.0.0-beta.19", "git" = "https://github.com/lance-format/lance.git" }
lance-testing = { "version" = "=9.0.0-beta.19", "tag" = "v9.0.0-beta.19", "git" = "https://github.com/lance-format/lance.git" }
lance-datafusion = { "version" = "=9.0.0-beta.19", "tag" = "v9.0.0-beta.19", "git" = "https://github.com/lance-format/lance.git" }
lance-encoding = { "version" = "=9.0.0-beta.19", "tag" = "v9.0.0-beta.19", "git" = "https://github.com/lance-format/lance.git" }
lance-arrow = { "version" = "=9.0.0-beta.19", "tag" = "v9.0.0-beta.19", "git" = "https://github.com/lance-format/lance.git" }
lance = { "version" = "=9.0.0-beta.23", default-features = false, "tag" = "v9.0.0-beta.23", "git" = "https://github.com/lance-format/lance.git" }
lance-core = { "version" = "=9.0.0-beta.23", "tag" = "v9.0.0-beta.23", "git" = "https://github.com/lance-format/lance.git" }
lance-datagen = { "version" = "=9.0.0-beta.23", "tag" = "v9.0.0-beta.23", "git" = "https://github.com/lance-format/lance.git" }
lance-file = { "version" = "=9.0.0-beta.23", "tag" = "v9.0.0-beta.23", "git" = "https://github.com/lance-format/lance.git" }
lance-io = { "version" = "=9.0.0-beta.23", default-features = false, "tag" = "v9.0.0-beta.23", "git" = "https://github.com/lance-format/lance.git" }
lance-index = { "version" = "=9.0.0-beta.23", "tag" = "v9.0.0-beta.23", "git" = "https://github.com/lance-format/lance.git" }
lance-linalg = { "version" = "=9.0.0-beta.23", "tag" = "v9.0.0-beta.23", "git" = "https://github.com/lance-format/lance.git" }
lance-namespace = { "version" = "=9.0.0-beta.23", "tag" = "v9.0.0-beta.23", "git" = "https://github.com/lance-format/lance.git" }
lance-namespace-impls = { "version" = "=9.0.0-beta.23", default-features = false, "tag" = "v9.0.0-beta.23", "git" = "https://github.com/lance-format/lance.git" }
lance-table = { "version" = "=9.0.0-beta.23", "tag" = "v9.0.0-beta.23", "git" = "https://github.com/lance-format/lance.git" }
lance-testing = { "version" = "=9.0.0-beta.23", "tag" = "v9.0.0-beta.23", "git" = "https://github.com/lance-format/lance.git" }
lance-datafusion = { "version" = "=9.0.0-beta.23", "tag" = "v9.0.0-beta.23", "git" = "https://github.com/lance-format/lance.git" }
lance-encoding = { "version" = "=9.0.0-beta.23", "tag" = "v9.0.0-beta.23", "git" = "https://github.com/lance-format/lance.git" }
lance-arrow = { "version" = "=9.0.0-beta.23", "tag" = "v9.0.0-beta.23", "git" = "https://github.com/lance-format/lance.git" }
ahash = "0.8"
# Note that this one does not include pyarrow
arrow = { version = "58.0.0", optional = false }
+1 -1
View File
@@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`:
<dependency>
<groupId>com.lancedb</groupId>
<artifactId>lancedb-core</artifactId>
<version>0.32.0-beta.0</version>
<version>0.32.0-beta.1</version>
</dependency>
```
+26
View File
@@ -934,6 +934,32 @@ Return the table as an arrow table
***
### tokenize()
```ts
abstract tokenize(query, options): Promise<FtsToken[]>
```
Tokenize a full-text search query using the tokenizer configured on an FTS index.
Specify exactly one of `column` or `indexName`.
Model-backed tokenizers such as `jieba/*` and `lindera/*` are rebuilt in
the client process from index metadata. For remote tables, this means the
same tokenizer model files must also exist locally.
#### Parameters
* **query**: `string`
* **options**: [`TokenizeTableOptions`](../type-aliases/TokenizeTableOptions.md)
#### Returns
`Promise`&lt;[`FtsToken`](../interfaces/FtsToken.md)[]&gt;
***
### unsetLsmWriteSpec()
```ts
+26
View File
@@ -0,0 +1,26 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / tokenize
# Function: tokenize()
```ts
function tokenize(query, options?): Promise<FtsToken[]>
```
Tokenize a full-text search query using an explicit tokenizer.
This does not require a table or FTS index. The tokenizer options match
[Index.fts](../classes/Index.md#fts).
## Parameters
* **query**: `string`
* **options?**: `Partial`&lt;[`TokenizeOptions`](../interfaces/TokenizeOptions.md)&gt;
## Returns
`Promise`&lt;[`FtsToken`](../interfaces/FtsToken.md)[]&gt;
+5
View File
@@ -72,6 +72,7 @@
- [FragmentStatistics](interfaces/FragmentStatistics.md)
- [FragmentSummaryStats](interfaces/FragmentSummaryStats.md)
- [FtsOptions](interfaces/FtsOptions.md)
- [FtsToken](interfaces/FtsToken.md)
- [FullTextQuery](interfaces/FullTextQuery.md)
- [FullTextSearchOptions](interfaces/FullTextSearchOptions.md)
- [HnswPqOptions](interfaces/HnswPqOptions.md)
@@ -107,6 +108,7 @@
- [TimeoutConfig](interfaces/TimeoutConfig.md)
- [TlsConfig](interfaces/TlsConfig.md)
- [TokenResponse](interfaces/TokenResponse.md)
- [TokenizeOptions](interfaces/TokenizeOptions.md)
- [UpdateFieldMetadataResult](interfaces/UpdateFieldMetadataResult.md)
- [UpdateOptions](interfaces/UpdateOptions.md)
- [UpdateResult](interfaces/UpdateResult.md)
@@ -116,6 +118,7 @@
## Type Aliases
- [BaseTokenizer](type-aliases/BaseTokenizer.md)
- [Data](type-aliases/Data.md)
- [DataLike](type-aliases/DataLike.md)
- [FieldLike](type-aliases/FieldLike.md)
@@ -125,6 +128,7 @@
- [RecordBatchLike](type-aliases/RecordBatchLike.md)
- [SchemaLike](type-aliases/SchemaLike.md)
- [TableLike](type-aliases/TableLike.md)
- [TokenizeTableOptions](type-aliases/TokenizeTableOptions.md)
## Functions
@@ -135,3 +139,4 @@
- [makeArrowTable](functions/makeArrowTable.md)
- [packBits](functions/packBits.md)
- [permutationBuilder](functions/permutationBuilder.md)
- [tokenize](functions/tokenize.md)
+5 -1
View File
@@ -23,7 +23,7 @@ whether to remove punctuation
### baseTokenizer?
```ts
optional baseTokenizer: "raw" | "simple" | "whitespace" | "ngram";
optional baseTokenizer: BaseTokenizer;
```
The tokenizer to use when building the index.
@@ -37,6 +37,10 @@ The following tokenizers are available:
"raw" - Raw tokenizer. This tokenizer does not split the text into tokens and indexes the entire text as a single token.
"icu" - ICU dictionary-based word segmentation.
"icu/split" - ICU segmentation with simple-style delimiter splitting.
***
### language?
+29
View File
@@ -0,0 +1,29 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / FtsToken
# Interface: FtsToken
Token produced by the tokenizer configured on a full-text search index.
## Properties
### position
```ts
position: number;
```
Token position used by full-text query matching.
***
### text
```ts
text: string;
```
Token text after tokenizer filters have been applied.
+109
View File
@@ -0,0 +1,109 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / TokenizeOptions
# Interface: TokenizeOptions
Options for tokenizing a full-text search query without a table index.
## Properties
### asciiFolding?
```ts
optional asciiFolding: boolean;
```
Whether to fold ASCII characters.
***
### baseTokenizer?
```ts
optional baseTokenizer: BaseTokenizer;
```
The tokenizer to use. The default is "simple".
***
### language?
```ts
optional language: string;
```
Language for stemming and stop words.
***
### lowercase?
```ts
optional lowercase: boolean;
```
Whether to lowercase tokens.
***
### maxTokenLength?
```ts
optional maxTokenLength: number;
```
Maximum token length; tokens longer than this are ignored.
***
### ngramMaxLength?
```ts
optional ngramMaxLength: number;
```
N-gram maximum length.
***
### ngramMinLength?
```ts
optional ngramMinLength: number;
```
N-gram minimum length.
***
### prefixOnly?
```ts
optional prefixOnly: boolean;
```
Whether to only emit token prefixes for the n-gram tokenizer.
***
### removeStopWords?
```ts
optional removeStopWords: boolean;
```
Whether to remove stop words.
***
### stem?
```ts
optional stem: boolean;
```
Whether to stem tokens.
+19
View File
@@ -0,0 +1,19 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / BaseTokenizer
# Type Alias: BaseTokenizer
```ts
type BaseTokenizer:
| "simple"
| "whitespace"
| "raw"
| "ngram"
| "icu"
| "icu/split"
| `jieba/${string}`
| `lindera/${string}`;
```
@@ -0,0 +1,11 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / TokenizeTableOptions
# Type Alias: TokenizeTableOptions
```ts
type TokenizeTableOptions: object | object;
```
+1 -1
View File
@@ -8,7 +8,7 @@
<parent>
<groupId>com.lancedb</groupId>
<artifactId>lancedb-parent</artifactId>
<version>0.32.0-beta.0</version>
<version>0.32.0-beta.1</version>
<relativePath>../pom.xml</relativePath>
</parent>
+2 -2
View File
@@ -6,7 +6,7 @@
<groupId>com.lancedb</groupId>
<artifactId>lancedb-parent</artifactId>
<version>0.32.0-beta.0</version>
<version>0.32.0-beta.1</version>
<packaging>pom</packaging>
<name>${project.artifactId}</name>
<description>LanceDB Java SDK Parent POM</description>
@@ -28,7 +28,7 @@
<properties>
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
<arrow.version>15.0.0</arrow.version>
<lance-core.version>9.0.0-beta.19</lance-core.version>
<lance-core.version>9.0.0-beta.23</lance-core.version>
<spotless.skip>false</spotless.skip>
<spotless.version>2.30.0</spotless.version>
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
+1 -1
View File
@@ -1,7 +1,7 @@
[package]
name = "lancedb-nodejs"
edition.workspace = true
version = "0.32.0-beta.0"
version = "0.32.0-beta.1"
publish = false
license.workspace = true
description.workspace = true
+70
View File
@@ -16,6 +16,7 @@ import {
PhraseQuery,
Table,
connect,
tokenize,
} from "../lancedb";
import {
Table as ArrowTable,
@@ -2307,6 +2308,75 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
expect(results2[0].text).toBe(data[1].text);
});
test("tokenizes FTS queries by column or index name", async () => {
const db = await connect(tmpDir.name);
const data = [
{
text: "Running in cafés",
japanese: "Hello, こんにちは世界!",
vector: [0.1, 0.2, 0.3],
},
];
const table = await db.createTable("test", data);
await table.createIndex("text", {
config: Index.fts({ baseTokenizer: "simple" }),
});
await table.createIndex("japanese", {
config: Index.fts({
baseTokenizer: "icu",
stem: false,
removeStopWords: false,
}),
name: "japanese_icu_idx",
});
await expect(table.tokenize("hello", {} as never)).rejects.toThrow(
"Specify exactly one",
);
await expect(
table.tokenize("hello", {
column: "text",
indexName: "text_idx",
} as never),
).rejects.toThrow("Specify exactly one");
const simpleTokens = await table.tokenize("Running in cafés", {
column: "text",
});
expect(simpleTokens).toEqual([
{ text: "run", position: 0 },
{ text: "cafe", position: 2 },
]);
const icuTokens = await table.tokenize("Hello, こんにちは世界!", {
indexName: "japanese_icu_idx",
});
expect(icuTokens).toEqual([
{ text: "hello", position: 0 },
{ text: "こんにちは", position: 1 },
{ text: "世界", position: 2 },
]);
const directSimpleTokens = await tokenize("Running in cafés", {
baseTokenizer: "simple",
});
expect(directSimpleTokens).toEqual([
{ text: "run", position: 0 },
{ text: "cafe", position: 2 },
]);
const directIcuTokens = await tokenize("Hello, こんにちは世界!", {
baseTokenizer: "icu",
stem: false,
removeStopWords: false,
});
expect(directIcuTokens).toEqual([
{ text: "hello", position: 0 },
{ text: "こんにちは", position: 1 },
{ text: "世界", position: 2 },
]);
});
test("full text search fast search", async () => {
const db = await connect(tmpDir.name);
const data = [{ text: "hello world", vector: [0.1, 0.2, 0.3], id: 1 }];
+68
View File
@@ -13,9 +13,12 @@ import {
Connection as LanceDbConnection,
JsHeaderProvider as NativeJsHeaderProvider,
Session,
tokenize as nativeTokenize,
} from "./native.js";
import { HeaderProvider } from "./header";
import type { BaseTokenizer } from "./indices";
import type { FtsToken } from "./table";
// Re-export native header provider for use with connectWithHeaderProvider
export { JsHeaderProvider as NativeJsHeaderProvider } from "./native.js";
@@ -114,6 +117,7 @@ export {
HnswPqOptions,
HnswSqOptions,
FtsOptions,
BaseTokenizer,
} from "./indices";
export {
@@ -124,6 +128,8 @@ export {
OptimizeOptions,
Version,
WriteProgress,
FtsToken,
TokenizeTableOptions,
LsmWriteSpec,
ColumnAlteration,
FieldMetadataUpdate,
@@ -155,6 +161,68 @@ export {
} from "./arrow";
export { IntoSql, packBits } from "./util";
/**
* Options for tokenizing a full-text search query without a table index.
*/
export interface TokenizeOptions {
/**
* The tokenizer to use. The default is "simple".
*/
baseTokenizer?: BaseTokenizer;
/** Language for stemming and stop words. */
language?: string;
/** Maximum token length; tokens longer than this are ignored. */
maxTokenLength?: number;
/** Whether to lowercase tokens. */
lowercase?: boolean;
/** Whether to stem tokens. */
stem?: boolean;
/** Whether to remove stop words. */
removeStopWords?: boolean;
/** Whether to fold ASCII characters. */
asciiFolding?: boolean;
/** N-gram minimum length. */
ngramMinLength?: number;
/** N-gram maximum length. */
ngramMaxLength?: number;
/** Whether to only emit token prefixes for the n-gram tokenizer. */
prefixOnly?: boolean;
}
/**
* Tokenize a full-text search query using an explicit tokenizer.
*
* This does not require a table or FTS index. The tokenizer options match
* {@link Index.fts}.
*/
export async function tokenize(
query: string,
options?: Partial<TokenizeOptions>,
): Promise<FtsToken[]> {
return await nativeTokenize(
query,
options?.baseTokenizer,
options?.language,
options?.maxTokenLength,
options?.lowercase,
options?.stem,
options?.removeStopWords,
options?.asciiFolding,
options?.ngramMinLength,
options?.ngramMaxLength,
options?.prefixOnly,
);
}
/**
* Connect to a LanceDB instance at the given URI.
*
+15 -1
View File
@@ -486,6 +486,16 @@ export interface IvfFlatOptions {
sampleRate?: number;
}
export type BaseTokenizer =
| "simple"
| "whitespace"
| "raw"
| "ngram"
| "icu"
| "icu/split"
| `jieba/${string}`
| `lindera/${string}`;
/**
* Options to create a full text search index
*/
@@ -509,8 +519,12 @@ export interface FtsOptions {
* "whitespace" - Whitespace tokenizer. This tokenizer splits the text into tokens using whitespace as a delimiter.
*
* "raw" - Raw tokenizer. This tokenizer does not split the text into tokens and indexes the entire text as a single token.
*
* "icu" - ICU dictionary-based word segmentation.
*
* "icu/split" - ICU segmentation with simple-style delimiter splitting.
*/
baseTokenizer?: "simple" | "whitespace" | "raw" | "ngram";
baseTokenizer?: BaseTokenizer;
/**
* language for stemming and stop words
+44
View File
@@ -158,6 +158,26 @@ export interface Version {
metadata: Record<string, string>;
}
/** Token produced by the tokenizer configured on a full-text search index. */
export interface FtsToken {
/** Token text after tokenizer filters have been applied. */
text: string;
/** Token position used by full-text query matching. */
position: number;
}
export type TokenizeTableOptions =
| {
/** FTS-indexed column whose tokenizer should be used. */
column: string;
indexName?: never;
}
| {
/** Name of the FTS index whose tokenizer should be used. */
indexName: string;
column?: never;
};
/**
* Specification selecting Lance's MemWAL LSM-style write path for
* `mergeInsert`.
@@ -716,6 +736,19 @@ export abstract class Table {
abstract optimize(options?: Partial<OptimizeOptions>): Promise<OptimizeStats>;
/** List all indices that have been created with {@link Table.createIndex} */
abstract listIndices(): Promise<IndexConfig[]>;
/**
* Tokenize a full-text search query using the tokenizer configured on an FTS index.
*
* Specify exactly one of `column` or `indexName`.
*
* Model-backed tokenizers such as `jieba/*` and `lindera/*` are rebuilt in
* the client process from index metadata. For remote tables, this means the
* same tokenizer model files must also exist locally.
*/
abstract tokenize(
query: string,
options: TokenizeTableOptions,
): Promise<FtsToken[]>;
/** Return the table as an arrow table */
abstract toArrow(): Promise<ArrowTable>;
@@ -1173,6 +1206,17 @@ export class LocalTable extends Table {
return await this.inner.listIndices();
}
async tokenize(
query: string,
options: TokenizeTableOptions,
): Promise<FtsToken[]> {
return await this.inner.tokenize(
query,
options?.column,
options?.indexName,
);
}
async toArrow(): Promise<ArrowTable> {
return await this.query().toArrow();
}
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@lancedb/lancedb-darwin-arm64",
"version": "0.32.0-beta.0",
"version": "0.32.0-beta.1",
"os": ["darwin"],
"cpu": ["arm64"],
"main": "lancedb.darwin-arm64.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@lancedb/lancedb-linux-arm64-gnu",
"version": "0.32.0-beta.0",
"version": "0.32.0-beta.1",
"os": ["linux"],
"cpu": ["arm64"],
"main": "lancedb.linux-arm64-gnu.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@lancedb/lancedb-linux-arm64-musl",
"version": "0.32.0-beta.0",
"version": "0.32.0-beta.1",
"os": ["linux"],
"cpu": ["arm64"],
"main": "lancedb.linux-arm64-musl.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@lancedb/lancedb-linux-x64-gnu",
"version": "0.32.0-beta.0",
"version": "0.32.0-beta.1",
"os": ["linux"],
"cpu": ["x64"],
"main": "lancedb.linux-x64-gnu.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@lancedb/lancedb-linux-x64-musl",
"version": "0.32.0-beta.0",
"version": "0.32.0-beta.1",
"os": ["linux"],
"cpu": ["x64"],
"main": "lancedb.linux-x64-musl.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@lancedb/lancedb-win32-arm64-msvc",
"version": "0.32.0-beta.0",
"version": "0.32.0-beta.1",
"os": [
"win32"
],
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@lancedb/lancedb-win32-x64-msvc",
"version": "0.32.0-beta.0",
"version": "0.32.0-beta.1",
"os": ["win32"],
"cpu": ["x64"],
"main": "lancedb.win32-x64-msvc.node",
+2 -2
View File
@@ -1,12 +1,12 @@
{
"name": "@lancedb/lancedb",
"version": "0.32.0-beta.0",
"version": "0.32.0-beta.1",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "@lancedb/lancedb",
"version": "0.32.0-beta.0",
"version": "0.32.0-beta.1",
"cpu": [
"x64",
"arm64"
+1 -1
View File
@@ -11,7 +11,7 @@
"ann"
],
"private": false,
"version": "0.32.0-beta.0",
"version": "0.32.0-beta.1",
"main": "dist/index.js",
"exports": {
".": "./dist/index.js",
+62
View File
@@ -9,8 +9,11 @@ use lancedb::index::vector::{
IvfFlatIndexBuilder, IvfHnswPqIndexBuilder, IvfHnswSqIndexBuilder, IvfPqIndexBuilder,
IvfRqIndexBuilder,
};
use lancedb::tokenize as lancedb_tokenize;
use napi_derive::napi;
use crate::error::NapiErrorExt;
use crate::table::FtsToken;
use crate::util::parse_distance_type;
#[napi]
@@ -30,6 +33,65 @@ impl Index {
}
}
#[napi(catch_unwind)]
#[allow(dead_code, clippy::too_many_arguments)]
pub fn tokenize(
query: String,
base_tokenizer: Option<String>,
language: Option<String>,
max_token_length: Option<u32>,
lower_case: Option<bool>,
stem: Option<bool>,
remove_stop_words: Option<bool>,
ascii_folding: Option<bool>,
ngram_min_length: Option<u32>,
ngram_max_length: Option<u32>,
prefix_only: Option<bool>,
) -> napi::Result<Vec<FtsToken>> {
let mut opts = FtsIndexBuilder::default();
if let Some(base_tokenizer) = base_tokenizer {
opts = opts.base_tokenizer(base_tokenizer);
}
if let Some(language) = language {
opts = opts.language(&language).map_err(|_| {
napi::Error::from_reason(format!(
"LanceDB does not support the requested language: '{}'",
language
))
})?;
}
if let Some(max_token_length) = max_token_length {
opts = opts.max_token_length(Some(max_token_length as usize));
}
if let Some(lower_case) = lower_case {
opts = opts.lower_case(lower_case);
}
if let Some(stem) = stem {
opts = opts.stem(stem);
}
if let Some(remove_stop_words) = remove_stop_words {
opts = opts.remove_stop_words(remove_stop_words);
}
if let Some(ascii_folding) = ascii_folding {
opts = opts.ascii_folding(ascii_folding);
}
if let Some(ngram_min_length) = ngram_min_length {
opts = opts.ngram_min_length(ngram_min_length);
}
if let Some(ngram_max_length) = ngram_max_length {
opts = opts.ngram_max_length(ngram_max_length);
}
if let Some(prefix_only) = prefix_only {
opts = opts.ngram_prefix_only(prefix_only);
}
Ok(lancedb_tokenize(&query, &opts)
.default_error()?
.into_iter()
.map(FtsToken::from)
.collect())
}
#[napi]
impl Index {
#[napi(factory)]
+41 -2
View File
@@ -8,8 +8,8 @@ use chrono::{DateTime, Utc};
use lancedb::ipc::{ipc_file_to_batches, ipc_file_to_schema};
use lancedb::table::{
AddDataMode, ColumnAlteration as LanceColumnAlteration, Duration,
FieldMetadataUpdate as LanceFieldMetadataUpdate, NewColumnTransform, OptimizeAction,
OptimizeOptions, Ref, Table as LanceDbTable,
FieldMetadataUpdate as LanceFieldMetadataUpdate, FtsToken as LanceDbFtsToken,
NewColumnTransform, OptimizeAction, OptimizeOptions, Ref, Table as LanceDbTable,
};
use napi::bindgen_prelude::*;
use napi::threadsafe_function::{ThreadsafeFunction, ThreadsafeFunctionCallMode};
@@ -574,6 +574,27 @@ impl Table {
.collect::<Vec<_>>())
}
#[napi(catch_unwind)]
pub async fn tokenize(
&self,
query: String,
column: Option<String>,
index_name: Option<String>,
) -> napi::Result<Vec<FtsToken>> {
let table = self.inner_ref()?;
let tokens = match (column.as_deref(), index_name.as_deref()) {
(Some(_), Some(_)) | (None, None) => {
return Err(napi::Error::from_reason(
"Specify exactly one of 'column' or 'indexName'",
));
}
(Some(column), None) => table.tokenize_with_column(&query, column).await,
(None, Some(index_name)) => table.tokenize(&query, index_name).await,
}
.default_error()?;
Ok(tokens.into_iter().map(FtsToken::from).collect())
}
#[napi(catch_unwind)]
pub async fn index_stats(&self, index_name: String) -> napi::Result<Option<IndexStatistics>> {
let tbl = self.inner_ref()?;
@@ -681,6 +702,24 @@ impl From<lancedb::index::IndexConfig> for IndexConfig {
}
}
#[napi(object)]
/// A token produced by the tokenizer configured on a full-text search index.
pub struct FtsToken {
/// The token text after the index tokenizer has applied its filters.
pub text: String,
/// The token position used by full-text query matching.
pub position: u32,
}
impl From<LanceDbFtsToken> for FtsToken {
fn from(token: LanceDbFtsToken) -> Self {
Self {
text: token.text,
position: token.position,
}
}
}
/// Specification selecting Lance's MemWAL LSM-style write path for
/// `mergeInsert`.
///
+1 -1
View File
@@ -1,5 +1,5 @@
[tool.bumpversion]
current_version = "0.35.0-beta.1"
current_version = "0.35.0-beta.2"
parse = """(?x)
(?P<major>0|[1-9]\\d*)\\.
(?P<minor>0|[1-9]\\d*)\\.
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "lancedb-python"
version = "0.35.0-beta.1"
version = "0.35.0-beta.2"
publish = false
edition.workspace = true
description = "Python bindings for LanceDB"
+43 -2
View File
@@ -6,19 +6,22 @@ import importlib.metadata
import os
from concurrent.futures import ThreadPoolExecutor
from datetime import timedelta
from typing import Dict, Optional, Union, Any, List
from typing import Dict, Optional, Union, Any, List, Iterable
__version__ = importlib.metadata.version("lancedb")
from ._lancedb import connect as lancedb_connect
from ._lancedb import FtsToken
from ._lancedb import tokenize as _tokenize
from .common import URI, sanitize_uri
from urllib.parse import urlparse
from .db import AsyncConnection, DBConnection, LanceDBConnection
from .remote import ClientConfig
from .remote.db import RemoteDBConnection
from .expr import Expr, col, lit, func
from .schema import vector
from .schema import blob, vector, BlobType
from .table import AsyncTable, Table
from .types import BaseTokenizerType
from ._lancedb import Session
from .namespace import (
connect_namespace,
@@ -246,6 +249,40 @@ def connect(
)
def tokenize(
query: str,
*,
base_tokenizer: BaseTokenizerType = "simple",
language: str = "English",
max_token_length: Optional[int] = 40,
lower_case: bool = True,
stem: bool = True,
remove_stop_words: bool = True,
ascii_folding: bool = True,
ngram_min_length: int = 3,
ngram_max_length: int = 3,
prefix_only: bool = False,
) -> Iterable[FtsToken]:
"""Tokenize a full-text search query using an explicit tokenizer.
This does not require a table or FTS index. The tokenizer options match
:class:`lancedb.index.FTS`.
"""
return _tokenize(
query,
base_tokenizer=base_tokenizer,
language=language,
max_token_length=max_token_length,
lower_case=lower_case,
stem=stem,
remove_stop_words=remove_stop_words,
ascii_folding=ascii_folding,
ngram_min_length=ngram_min_length,
ngram_max_length=ngram_max_length,
prefix_only=prefix_only,
)
WORKER_PROPERTY_PREFIX = "_lancedb_worker_"
@@ -456,17 +493,21 @@ async def connect_async(
__all__ = [
"connect",
"connect_async",
"tokenize",
"connect_namespace",
"connect_namespace_async",
"AsyncConnection",
"AsyncLanceNamespaceDBConnection",
"AsyncTable",
"FtsToken",
"col",
"Expr",
"func",
"lit",
"URI",
"sanitize_uri",
"blob",
"BlobType",
"vector",
"DBConnection",
"LanceDBConnection",
+420
View File
@@ -0,0 +1,420 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
"""Blob fetch API and v2 projection helpers."""
from __future__ import annotations
import io
from collections.abc import Awaitable, Callable, Iterable
from typing import TYPE_CHECKING, Optional, Union
import pyarrow as pa
from .expr import Expr
from .schema import blob_v2_column_paths
from .types import BlobMode, QueryProjection, QueryProjectionSpec
from .util import get_uri_scheme
if TYPE_CHECKING:
from _typeshed import WriteableBuffer
from .remote.table import RemoteTable
from .table import AsyncTable, Table
BLOB_MODE_TO_HANDLING = {
"lazy": "blobs_descriptions",
"bytes": "all_binary",
"descriptions": "blobs_descriptions",
}
ROW_ID_FIELD_NAME = "_lance_row_id"
FetchBlobsSync = Callable[[str, pa.Table], pa.Array | pa.ChunkedArray]
FetchBlobsAsync = Callable[[str, pa.Table], Awaitable[pa.Array | pa.ChunkedArray]]
class BlobFile(io.RawIOBase):
"""Seekable lazy handle from :meth:`~lancedb.table.Table.fetch_blob_files`.
Bytes load on ``read`` or ``read_range``, not when the handle is opened.
Use :meth:`aread` from async code.
"""
def __init__(self, inner) -> None:
self._inner = inner
async def aread(self) -> bytes:
return await self._inner.read()
def close(self) -> None:
self._inner.close()
@property
def closed(self) -> bool:
return self._inner.is_closed()
def readable(self) -> bool:
return True
def seekable(self) -> bool:
return True
def seek(self, offset: int, whence: int = io.SEEK_SET) -> int:
if whence == io.SEEK_SET:
self._inner.seek(offset)
elif whence == io.SEEK_CUR:
self._inner.seek(self._inner.tell() + offset)
elif whence == io.SEEK_END:
self._inner.seek(self._inner.size() + offset)
else:
raise ValueError(f"invalid whence: {whence}")
return self._inner.tell()
def tell(self) -> int:
return self._inner.tell()
def size(self) -> int:
return self._inner.size()
def readall(self) -> bytes:
return self._inner.read_bytes()
def read(self, size: int = -1) -> bytes:
if size == -1:
return self._inner.read_bytes()
return super().read(size)
def read_range(self, offset: int, length: int) -> bytes:
return self._inner.read_range(offset, length)
def readinto(self, b: WriteableBuffer) -> int:
view = memoryview(b).cast("B")
chunk = self._inner.read_up_to(len(view))
view[: len(chunk)] = chunk
return len(chunk)
def __repr__(self) -> str:
return f"<BlobFile size={self.size()}>"
def validate_blob_mode(blob_mode: BlobMode) -> None:
if blob_mode not in BLOB_MODE_TO_HANDLING:
modes = ", ".join(repr(mode) for mode in BLOB_MODE_TO_HANDLING)
raise ValueError(f"blob_mode must be one of {modes}, got {blob_mode!r}")
def supports_blob_auto_row_id(table: Table | AsyncTable | RemoteTable) -> bool:
"""Blob auto row-id applies to native tables, not LanceDB Cloud."""
from .remote.table import RemoteTable
if isinstance(table, RemoteTable):
return False
inner = getattr(table, "_inner", None)
if inner is not None:
uri = inner.database().uri
if isinstance(uri, str) and get_uri_scheme(uri) == "db":
return False
return True
def projection_includes_blob_column(
projection: QueryProjection,
blob_columns: Iterable[str],
) -> bool:
columns = set(blob_columns)
if not columns:
return False
if projection is None:
return True
for output, source in _iter_projection_pairs(projection):
if output in columns or source in columns:
return True
return False
def blob_v2_projection_sources(
schema: pa.Schema,
projection: QueryProjection,
) -> dict[str, str]:
blob_columns = blob_v2_column_paths(schema)
if not blob_columns:
return {}
columns = set(blob_columns)
if projection is None:
return {column: column for column in blob_columns}
return {
output: source
for output, source in _iter_projection_pairs(projection)
if source in columns
}
def v2_projection_needs_row_id(
schema: pa.Schema,
projection: QueryProjection,
*,
with_row_id: bool,
) -> bool:
if with_row_id:
return False
return projection_includes_blob_column(projection, blob_v2_column_paths(schema))
def blob_auto_row_id_for_scan(
table: Table | AsyncTable | RemoteTable,
schema: pa.Schema,
projection: QueryProjection,
*,
with_row_id: bool | None,
) -> bool:
if with_row_id is not None:
return False
if not supports_blob_auto_row_id(table):
return False
return v2_projection_needs_row_id(schema, projection, with_row_id=False)
def finalize_blob_query_table(
tbl: pa.Table,
*,
user_requested_row_id: bool,
blob_auto_row_id: bool,
blob_paths: Iterable[str] = (),
) -> pa.Table:
if user_requested_row_id or not blob_auto_row_id:
return tbl
return stash_auto_row_ids(tbl, blob_paths)
async def replace_v2_blob_columns_with_bytes(
tbl: pa.Table,
blob_sources: dict[str, str],
fetch_blobs: FetchBlobsAsync,
) -> pa.Table:
for output_name, source_name in blob_sources.items():
if output_name not in tbl.column_names:
continue
blobs = await fetch_blobs(source_name, tbl)
tbl = _set_blob_column(tbl, output_name, blobs)
return tbl
def replace_v2_blob_columns_with_bytes_sync(
tbl: pa.Table,
blob_sources: dict[str, str],
fetch_blobs: FetchBlobsSync,
) -> pa.Table:
for output_name, source_name in blob_sources.items():
if output_name not in tbl.column_names:
continue
blobs = fetch_blobs(source_name, tbl)
tbl = _set_blob_column(tbl, output_name, blobs)
return tbl
def stash_auto_row_ids(tbl: pa.Table, blob_paths: Iterable[str]) -> pa.Table:
if "_rowid" not in tbl.column_names:
raise ValueError("query result has no '_rowid' column to hide")
present_paths = [p for p in blob_paths if p.split(".")[0] in tbl.column_names]
if not present_paths:
raise ValueError("query result has no blob v2 column to carry a row id")
row_ids = tbl["_rowid"]
if isinstance(row_ids, pa.ChunkedArray):
row_ids = row_ids.combine_chunks()
row_ids = row_ids.cast(pa.uint64())
for path in present_paths:
tbl = _embed_row_id_in_column(tbl, path, row_ids)
return tbl.drop_columns(["_rowid"])
def read_row_ids_from_hits(hits: pa.Table, blob_column: str) -> list[int]:
if "_rowid" in hits.column_names:
return hits["_rowid"].to_pylist()
try:
leaf = _leaf_struct_column(hits, blob_column)
if ROW_ID_FIELD_NAME in leaf.type.names:
return leaf.field(ROW_ID_FIELD_NAME).to_pylist()
except KeyError:
pass
# blob_column is the source name; aliased projections use the output name in hits.
row_ids = _find_row_id_in_any_column(hits)
if row_ids is not None:
return row_ids
raise ValueError(
f"query result has no '_rowid' column and no '{ROW_ID_FIELD_NAME}' "
f"field on blob column '{blob_column}'. Pass fresh blob query "
"results, call .with_row_id(True), or pass a list of row ids."
)
def _find_row_id_in_any_column(tbl: pa.Table) -> Optional[list[int]]:
for name in tbl.column_names:
column = tbl.column(name)
if isinstance(column, pa.ChunkedArray):
column = column.combine_chunks()
row_ids = _find_row_id_in_struct(column)
if row_ids is not None:
return row_ids
return None
def _find_row_id_in_struct(array: pa.Array) -> Optional[list[int]]:
if not pa.types.is_struct(array.type):
return None
if ROW_ID_FIELD_NAME in array.type.names:
return array.field(ROW_ID_FIELD_NAME).to_pylist()
for i in range(array.type.num_fields):
row_ids = _find_row_id_in_struct(array.field(i))
if row_ids is not None:
return row_ids
return None
def _iter_projection_pairs(
projection: QueryProjectionSpec,
) -> Iterable[tuple[str, str]]:
if isinstance(projection, dict):
for name, expr in projection.items():
if isinstance(expr, str):
yield name, expr
elif isinstance(expr, Expr):
yield name, expr.to_sql()
return
for column in projection:
if isinstance(column, str):
yield column, column
elif isinstance(column, tuple) and len(column) == 2:
name, expr = column
if isinstance(expr, str):
yield name, expr
elif isinstance(expr, Expr):
yield name, expr.to_sql()
def _set_blob_column(tbl: pa.Table, output_name: str, blobs: pa.Array) -> pa.Table:
index = tbl.schema.get_field_index(output_name)
return tbl.set_column(index, pa.field(output_name, blobs.type), [blobs])
def _embed_row_id_in_column(tbl: pa.Table, path: str, row_ids: pa.Array) -> pa.Table:
def add_row_id(children: list, child_fields: list) -> None:
children.append(row_ids)
child_fields.append(pa.field(ROW_ID_FIELD_NAME, pa.uint64(), nullable=False))
return _transform_struct_column(tbl, path, add_row_id)
def strip_auto_row_ids(tbl: pa.Table, blob_paths: Iterable[str]) -> pa.Table:
"""Remove any `_lance_row_id` field embedded in blob descriptor structs.
For read-only descriptor views (`blob_mode="descriptions"`) that never
fetch bytes, so have no use for the row id.
"""
def drop_row_id(children: list, child_fields: list) -> None:
for i, field in enumerate(child_fields):
if field.name == ROW_ID_FIELD_NAME:
del children[i], child_fields[i]
return
for path in blob_paths:
if path.split(".")[0] not in tbl.column_names:
continue
tbl = _transform_struct_column(tbl, path, drop_row_id)
return tbl
def _transform_struct_column(
tbl: pa.Table, path: str, leaf_transform: Callable[[list, list], None]
) -> pa.Table:
top_name, *rest = path.split(".")
top_index = tbl.schema.get_field_index(top_name)
top_field = tbl.schema.field(top_index)
top_array = tbl.column(top_name)
if isinstance(top_array, pa.ChunkedArray):
top_array = top_array.combine_chunks()
new_array, new_field = _rebuild_struct(top_array, top_field, rest, leaf_transform)
return tbl.set_column(top_index, new_field, new_array)
def _rebuild_struct(
struct_array: pa.StructArray,
struct_field: pa.Field,
remaining_path: list[str],
leaf_transform: Callable[[list, list], None],
) -> tuple[pa.StructArray, pa.Field]:
null_mask = struct_array.is_null()
if not remaining_path:
children = [struct_array.field(i) for i in range(struct_array.type.num_fields)]
child_fields = list(struct_array.type)
leaf_transform(children, child_fields)
new_array = pa.StructArray.from_arrays(
children, fields=child_fields, mask=null_mask
)
else:
child_name = remaining_path[0]
child_index = struct_array.type.get_field_index(child_name)
child_array = struct_array.field(child_index)
child_field = struct_array.type.field(child_index)
new_child_array, new_child_field = _rebuild_struct(
child_array, child_field, remaining_path[1:], leaf_transform
)
children = []
child_fields = []
for i in range(struct_array.type.num_fields):
field = struct_array.type.field(i)
if field.name == child_name:
children.append(new_child_array)
child_fields.append(new_child_field)
else:
children.append(struct_array.field(i))
child_fields.append(field)
new_array = pa.StructArray.from_arrays(
children, fields=child_fields, mask=null_mask
)
new_field = pa.field(
struct_field.name,
new_array.type,
nullable=struct_field.nullable,
metadata=struct_field.metadata,
)
return new_array, new_field
def _leaf_struct_column(tbl: pa.Table, path: str) -> pa.StructArray:
parts = path.split(".")
column = tbl.column(parts[0])
if isinstance(column, pa.ChunkedArray):
column = column.combine_chunks()
for part in parts[1:]:
column = column.field(part)
return column
def _normalize_blob_row_ids(
row_ids: Union[list[int], pa.Table], blob_column: str
) -> list[int]:
if isinstance(row_ids, pa.Table):
return read_row_ids_from_hits(row_ids, blob_column)
if isinstance(row_ids, (pa.Array, pa.ChunkedArray)):
raise ValueError(
"pass a query table with _rowid, not a column array "
"(use fetch_blobs('image', hits), not fetch_blobs('image', hits['image']))"
)
return list(row_ids)
def _wrap_blob_files(handles: Iterable[object]) -> list[Optional[BlobFile]]:
return [BlobFile(handle) if handle is not None else None for handle in handles]
+44
View File
@@ -25,6 +25,7 @@ from lance_namespace import (
ListTablesResponse,
)
from .remote import ClientConfig
from .types import BaseTokenizerType
IvfHnswPq: type[HnswPq] = HnswPq
IvfHnswSq: type[HnswSq] = HnswSq
@@ -48,6 +49,20 @@ class MetricDescription:
def register_lancedb_metrics_recorder() -> bool: ...
def lancedb_metrics_catalog() -> List[MetricDescription]: ...
def snapshot_lancedb_metrics() -> List[MetricPoint]: ...
def tokenize(
query: str,
*,
base_tokenizer: BaseTokenizerType = "simple",
language: str = "English",
max_token_length: Optional[int] = 40,
lower_case: bool = True,
stem: bool = True,
remove_stop_words: bool = True,
ascii_folding: bool = True,
ngram_min_length: int = 3,
ngram_max_length: int = 3,
prefix_only: bool = False,
) -> List["FtsToken"]: ...
class PyExpr:
"""A type-safe DataFusion expression node (Rust-side handle)."""
@@ -181,6 +196,17 @@ class Connection(object):
self,
) -> Dict[str, Any]: ...
class BlobFile:
async def read(self) -> bytes: ...
def read_bytes(self) -> bytes: ...
def close(self) -> None: ...
def is_closed(self) -> bool: ...
def seek(self, position: int) -> None: ...
def tell(self) -> int: ...
def size(self) -> int: ...
def read_range(self, offset: int, length: int) -> bytes: ...
def read_up_to(self, length: int) -> bytes: ...
class Table:
def name(self) -> str: ...
def __repr__(self) -> str: ...
@@ -227,6 +253,13 @@ class Table:
async def prewarm_index(self, index_name: str) -> None: ...
async def prewarm_data(self, columns: Optional[List[str]] = None) -> None: ...
async def list_indices(self) -> list[IndexConfig]: ...
async def tokenize(
self,
query: str,
*,
column: Optional[str] = None,
index_name: Optional[str] = None,
) -> list[FtsToken]: ...
async def delete(self, filter: Union[str, PyExpr]) -> DeleteResult: ...
async def add_columns(self, columns: list[tuple[str, str]]) -> AddColumnsResult: ...
async def add_columns_with_schema(self, schema: pa.Schema) -> AddColumnsResult: ...
@@ -258,6 +291,13 @@ class Table:
def query(self) -> Query: ...
def take_offsets(self, offsets: list[int]) -> TakeQuery: ...
def take_row_ids(self, row_ids: list[int]) -> TakeQuery: ...
async def blob_columns(self) -> list[str]: ...
async def fetch_blobs(
self, column: str, row_ids: list[int]
) -> pa.LargeBinaryArray: ...
async def fetch_blob_files(
self, column: str, row_ids: list[int]
) -> list[Optional[BlobFile]]: ...
def vector_search(self) -> VectorQuery: ...
class Tags:
@@ -493,6 +533,10 @@ class MergeResult:
num_attempts: int
num_rows: int
class FtsToken:
text: str
position: int
class LsmWriteSpec:
"""Specification selecting Lance's MemWAL LSM-style write path for
`merge_insert`."""
+23 -10
View File
@@ -4,7 +4,7 @@
import os
from functools import cached_property
from typing import List, Union
from typing import List, Optional, Union
import numpy as np
@@ -15,6 +15,8 @@ from .base import TextEmbeddingFunction
from .registry import register
from .utils import TEXT, api_key_not_found_help
EMBEDDING_BATCH_SIZE = 100
@register("gemini-text")
class GeminiText(TextEmbeddingFunction):
@@ -81,6 +83,7 @@ class GeminiText(TextEmbeddingFunction):
"""
name: str = "gemini-embedding-001"
dim: Optional[int] = None
query_task_type: str = "retrieval_query"
source_task_type: str = "retrieval_document"
@@ -93,6 +96,8 @@ class GeminiText(TextEmbeddingFunction):
model_config["ignored_types"] = (cached_property,)
def ndims(self):
if self.dim:
return self.dim
# TODO: fix hardcoding
return 768
@@ -133,22 +138,22 @@ class GeminiText(TextEmbeddingFunction):
contents.append({"parts": [{"text": text}]})
# Build config
config_kwargs = {}
config_kwargs = {"output_dimensionality": self.ndims()}
if task_type:
config_kwargs["task_type"] = task_type.upper() # API expects uppercase
# Call embed_content for each content
config = types.EmbedContentConfig(**config_kwargs) if config_kwargs else None
# Call embed_content in groups of at most EMBEDDING_BATCH_SIZE docs at a time
embeddings = []
for content in contents:
config = (
types.EmbedContentConfig(**config_kwargs) if config_kwargs else None
)
for i in range(0, len(contents), EMBEDDING_BATCH_SIZE):
chunk = contents[i : i + EMBEDDING_BATCH_SIZE]
response = self.client.models.embed_content(
model=self.name,
contents=content,
contents=chunk,
config=config,
)
embeddings.append(response.embeddings[0].values)
embeddings.extend([np.array(e.values) for e in response.embeddings])
return embeddings
@@ -160,5 +165,13 @@ class GeminiText(TextEmbeddingFunction):
api_key_not_found_help("google")
from google import genai as genai_module
from lancedb import __version__
return genai_module.Client(api_key=os.environ.get("GOOGLE_API_KEY"))
return genai_module.Client(
api_key=os.environ.get("GOOGLE_API_KEY"),
http_options={
"headers": {
"x-goog-api-client": f"lancedb/{__version__}",
}
},
)
+2
View File
@@ -127,6 +127,8 @@ class FTS:
- "whitespace": Split text by whitespace, but not punctuation.
- "raw": No tokenization. The entire text is treated as a single token.
- "ngram": N-gram tokenizer for substring-style matching.
- "icu": ICU dictionary-based word segmentation.
- "icu/split": ICU segmentation with simple-style delimiter splitting.
- "jieba/*": Jieba tokenizer loaded from Lance's language model home.
- "lindera/*": Lindera tokenizer loaded from Lance's language model home.
language : str, default "English"
+300 -79
View File
@@ -15,10 +15,12 @@ from typing import (
List,
Literal,
Optional,
Protocol,
Tuple,
Type,
TypeVar,
Union,
runtime_checkable,
)
import deprecation
@@ -39,15 +41,21 @@ from .expr import Expr
from .rerankers.base import Reranker
from .rerankers.rrf import RRFReranker
from .rerankers.util import check_reranker_result
from .schema import is_blob_like_field, schema_has_blob_field
from .util import flatten_columns
BlobMode = Literal["lazy", "bytes", "descriptions"]
_BLOB_MODE_TO_HANDLING = {
"lazy": "blobs_descriptions",
"bytes": "all_binary",
"descriptions": "blobs_descriptions",
}
from ._blob import (
BLOB_MODE_TO_HANDLING,
FetchBlobsAsync,
FetchBlobsSync,
blob_auto_row_id_for_scan,
blob_v2_projection_sources,
finalize_blob_query_table,
replace_v2_blob_columns_with_bytes,
replace_v2_blob_columns_with_bytes_sync,
supports_blob_auto_row_id,
validate_blob_mode,
)
from .types import BlobMode, QueryProjection
if TYPE_CHECKING:
import sys
@@ -73,25 +81,22 @@ if TYPE_CHECKING:
T = TypeVar("T", bound="LanceModel")
def _validate_blob_mode(blob_mode: BlobMode) -> None:
if blob_mode not in _BLOB_MODE_TO_HANDLING:
modes = ", ".join(repr(mode) for mode in _BLOB_MODE_TO_HANDLING)
raise ValueError(f"blob_mode must be one of {modes}, got {blob_mode!r}")
@runtime_checkable
class _LanceScanner(Protocol):
projected_schema: pa.Schema | None
schema: pa.Schema | None
def to_pandas(self, blob_mode: BlobMode | None = ..., **kwargs) -> pd.DataFrame: ...
def _field_is_blob(field: pa.Field) -> bool:
metadata = field.metadata or {}
return metadata.get(b"lance-encoding:blob") == b"true" or (
metadata.get("lance-encoding:blob") == "true"
)
def to_pyarrow(self): ...
def to_table(self) -> pa.Table: ...
def _schema_has_blob_field(schema: pa.Schema) -> bool:
return any(_field_is_blob(field) for field in schema)
def to_reader(self): ...
def _blob_mode_requires_native_pandas(blob_mode: BlobMode, schema: pa.Schema) -> bool:
return blob_mode in _BLOB_MODE_TO_HANDLING and _schema_has_blob_field(schema)
return blob_mode in BLOB_MODE_TO_HANDLING and schema_has_blob_field(schema)
def _unsupported_blob_pandas_error(reason: str) -> RuntimeError:
@@ -140,13 +145,7 @@ def _combine_where(
return f"({existing_sql}) AND ({new_sql})"
def _projection_to_scanner_kwargs(
columns: Optional[
Union[
List[str], List[Tuple[str, Union[str, Expr]]], Dict[str, Union[str, Expr]]
]
],
) -> Dict[str, Any]:
def _projection_to_scanner_kwargs(columns: QueryProjection) -> Dict[str, Any]:
if columns is None:
return {}
if isinstance(columns, list):
@@ -171,7 +170,11 @@ def _projection_to_scanner_kwargs(
def _scanner_kwargs_for_query(
query: Query, blob_mode: BlobMode, dataset: Optional[Any] = None
query: Query,
blob_mode: BlobMode,
dataset: Optional[Any] = None,
*,
with_row_id: Optional[bool] = None,
) -> Dict[str, Any]:
fragments = _scanner_fragments_for_query(query, dataset)
kwargs = {
@@ -179,10 +182,10 @@ def _scanner_kwargs_for_query(
"filter": _filter_to_sql(query.filter),
"limit": query.limit,
"offset": query.offset,
"with_row_id": query.with_row_id,
"with_row_id": with_row_id if with_row_id is not None else query.with_row_id,
"with_row_address": query.with_row_address,
"fast_search": query.fast_search,
"blob_handling": _BLOB_MODE_TO_HANDLING[blob_mode],
"blob_handling": BLOB_MODE_TO_HANDLING[blob_mode],
"fragments": fragments,
}
return {key: value for key, value in kwargs.items() if value is not None}
@@ -215,11 +218,11 @@ def _scanner_fragments_for_query(query: Query, dataset: Optional[Any]) -> Option
def _ensure_lazy_blob_frame(
df: "pd.DataFrame", schema: pa.Schema, blob_mode: BlobMode
) -> "pd.DataFrame":
if blob_mode != "lazy" or not _schema_has_blob_field(schema) or len(df) == 0:
if blob_mode != "lazy" or not schema_has_blob_field(schema) or len(df) == 0:
return df
for field in schema:
if not _field_is_blob(field) or field.name not in df.columns:
if not is_blob_like_field(field) or field.name not in df.columns:
continue
value = df[field.name].iloc[0]
if value is not None and not hasattr(value, "readall"):
@@ -229,7 +232,7 @@ def _ensure_lazy_blob_frame(
return df
def _scanner_to_table(scanner: Any) -> pa.Table:
def _scanner_to_table(scanner: _LanceScanner) -> pa.Table:
if hasattr(scanner, "to_pyarrow"):
reader = scanner.to_pyarrow()
return reader.read_all()
@@ -239,7 +242,9 @@ def _scanner_to_table(scanner: Any) -> pa.Table:
return reader.read_all()
def _scanner_to_pandas(scanner: Any, blob_mode: BlobMode, **kwargs) -> "pd.DataFrame":
def _scanner_to_pandas(
scanner: _LanceScanner, blob_mode: BlobMode, **kwargs
) -> pd.DataFrame:
schema = getattr(scanner, "projected_schema", None)
if schema is None:
schema = getattr(scanner, "schema", None)
@@ -260,13 +265,71 @@ def _scanner_to_pandas(scanner: Any, blob_mode: BlobMode, **kwargs) -> "pd.DataF
return df
tbl = _scanner_to_table(scanner)
if blob_mode == "lazy" and _schema_has_blob_field(tbl.schema):
if blob_mode == "lazy" and schema_has_blob_field(tbl.schema):
raise _unsupported_blob_pandas_error(
"the Lance scanner does not expose to_pandas"
)
return tbl.to_pandas(**kwargs)
def _finish_plain_scan_pandas(
scanner: _LanceScanner,
*,
blob_mode: BlobMode,
blob_sources: dict[str, str],
fetch_blobs: FetchBlobsSync,
strip_auto_row_id: bool,
flatten: Optional[Union[int, bool]],
**kwargs,
) -> pd.DataFrame:
if blob_sources:
tbl = _scanner_to_table(scanner)
tbl = replace_v2_blob_columns_with_bytes_sync(tbl, blob_sources, fetch_blobs)
if strip_auto_row_id and "_rowid" in tbl.column_names:
tbl = tbl.drop_columns(["_rowid"])
if flatten is not None:
tbl = flatten_columns(tbl, flatten)
return tbl.to_pandas(**kwargs)
if flatten is not None:
tbl = flatten_columns(_scanner_to_table(scanner), flatten)
if strip_auto_row_id and "_rowid" in tbl.column_names:
tbl = tbl.drop_columns(["_rowid"])
return tbl.to_pandas(**kwargs)
df = _scanner_to_pandas(scanner, blob_mode, **kwargs)
if strip_auto_row_id and "_rowid" in df.columns:
return df.drop(columns=["_rowid"])
return df
async def _finish_plain_scan_pandas_async(
scanner: _LanceScanner,
*,
blob_mode: BlobMode,
blob_sources: dict[str, str],
fetch_blobs: FetchBlobsAsync,
strip_auto_row_id: bool,
flatten: Optional[Union[int, bool]],
**kwargs,
) -> pd.DataFrame:
if blob_sources:
tbl = _scanner_to_table(scanner)
tbl = await replace_v2_blob_columns_with_bytes(tbl, blob_sources, fetch_blobs)
if strip_auto_row_id and "_rowid" in tbl.column_names:
tbl = tbl.drop_columns(["_rowid"])
if flatten is not None:
tbl = flatten_columns(tbl, flatten)
return tbl.to_pandas(**kwargs)
if flatten is not None:
tbl = flatten_columns(_scanner_to_table(scanner), flatten)
if strip_auto_row_id and "_rowid" in tbl.column_names:
tbl = tbl.drop_columns(["_rowid"])
return tbl.to_pandas(**kwargs)
df = _scanner_to_pandas(scanner, blob_mode, **kwargs)
if strip_auto_row_id and "_rowid" in df.columns:
return df.drop(columns=["_rowid"])
return df
# Pydantic validation function for vector queries
def ensure_vector_query(
val: Any,
@@ -674,7 +737,7 @@ class Query(pydantic.BaseModel):
distance_type: Optional[str] = None
# which columns to return in the results (dict values may be str or Expr)
columns: Optional[Union[List[str], Dict[str, Union[str, Expr]]]] = None
columns: QueryProjection = None
# minimum number of IVF partitions to search
#
@@ -958,7 +1021,7 @@ class LanceQueryBuilder(ABC):
Forwarded to pyarrow.Table.to_pandas after query execution and
optional flattening.
"""
_validate_blob_mode(blob_mode)
validate_blob_mode(blob_mode)
output_schema = getattr(self, "output_schema", None)
if output_schema is not None:
schema = output_schema()
@@ -1017,6 +1080,11 @@ class LanceQueryBuilder(ABC):
Execute the query and return the results as a pyarrow
[RecordBatchReader](https://arrow.apache.org/docs/python/generated/pyarrow.RecordBatchReader.html)
For v2 blob projections, ``to_batches`` keeps the auto ``_rowid``
column visible so batch consumers can call ``fetch_blobs``. Use
``to_arrow``, ``to_list``, or ``to_pandas`` if you want LanceDB to hide
auto row ids in the final collected result.
Parameters
----------
batch_size: int
@@ -1195,6 +1263,42 @@ class LanceQueryBuilder(ABC):
self._with_row_id = with_row_id
return self
def _user_requested_row_id(self) -> bool:
return self._with_row_id is True
def _blob_auto_row_id_enabled(self) -> bool:
if not supports_blob_auto_row_id(self._table):
return False
return blob_auto_row_id_for_scan(
self._table,
self._table.schema,
self._columns,
with_row_id=self._with_row_id,
)
def _scan_needs_row_id(self) -> bool:
return self._user_requested_row_id() or self._blob_auto_row_id_enabled()
def _query_for_scan(self) -> Query:
query = self.to_query_object()
if self._scan_needs_row_id():
query.with_row_id = True
return query
def _finalize_blob_query_table(self, tbl: pa.Table) -> pa.Table:
blob_auto_row_id = self._blob_auto_row_id_enabled()
blob_paths = (
blob_v2_projection_sources(self._table.schema, self._columns).keys()
if blob_auto_row_id
else ()
)
return finalize_blob_query_table(
tbl,
user_requested_row_id=self._user_requested_row_id(),
blob_auto_row_id=blob_auto_row_id,
blob_paths=blob_paths,
)
def with_row_address(self, with_row_address: bool = True) -> Self:
"""Set whether to return row addresses.
@@ -1371,13 +1475,29 @@ class LanceQueryBuilder(ABC):
return None
dataset = self._table.to_lance()
scanner = dataset.scanner(
**_scanner_kwargs_for_query(query, blob_mode, dataset)
blob_auto_row_id = self._blob_auto_row_id_enabled()
blob_sources = (
blob_v2_projection_sources(self._table.schema, query.columns)
if blob_mode == "bytes"
else {}
)
scanner = dataset.scanner(
**_scanner_kwargs_for_query(
query,
"descriptions" if blob_sources else blob_mode,
dataset,
with_row_id=query.with_row_id or blob_auto_row_id or bool(blob_sources),
)
)
return _finish_plain_scan_pandas(
scanner,
blob_mode=blob_mode,
blob_sources=blob_sources,
fetch_blobs=self._table.fetch_blobs,
strip_auto_row_id=blob_auto_row_id,
flatten=flatten,
**kwargs,
)
if flatten is not None:
tbl = flatten_columns(_scanner_to_table(scanner), flatten)
return tbl.to_pandas(**kwargs)
return _scanner_to_pandas(scanner, blob_mode, **kwargs)
@abstractmethod
def to_query_object(self) -> Query:
@@ -1625,7 +1745,9 @@ class LanceVectorQueryBuilder(LanceQueryBuilder):
The maximum time to wait for the query to complete.
If None, wait indefinitely.
"""
return self.to_batches(timeout=timeout).read_all()
return self._finalize_blob_query_table(
self.to_batches(timeout=timeout).read_all()
)
def to_query_object(self) -> Query:
"""
@@ -1685,7 +1807,7 @@ class LanceVectorQueryBuilder(LanceQueryBuilder):
vector = self._query if isinstance(self._query, list) else self._query.tolist()
if isinstance(vector[0], np.ndarray):
vector = [v.tolist() for v in vector]
query = self.to_query_object()
query = self._query_for_scan()
result_set = self._table._execute_query(
query, batch_size=batch_size, timeout=timeout
)
@@ -1829,8 +1951,7 @@ class LanceFtsQueryBuilder(LanceQueryBuilder):
Parameters
----------
phrase_query: bool, default True
If True, then the query will be wrapped in quotes and
double quotes replaced by single quotes.
If True, then an unquoted string query will be wrapped in quotes.
Returns
-------
@@ -1840,6 +1961,21 @@ class LanceFtsQueryBuilder(LanceQueryBuilder):
self._phrase_query = phrase_query
return self
def _query_with_phrase_semantics(self) -> str | FullTextQuery:
query = self._query
if not self._phrase_query:
return query
if isinstance(query, str):
if not query.startswith('"') or not query.endswith('"'):
return f'"{query}"'
return query
if isinstance(query, PhraseQuery):
return query
raise TypeError(
"phrase_query() requires a string or PhraseQuery, "
f"got {type(query).__name__}"
)
def fast_search(self) -> LanceFtsQueryBuilder:
"""
Skip a flat search of unindexed data. This will improve
@@ -1864,7 +2000,7 @@ class LanceFtsQueryBuilder(LanceQueryBuilder):
fragments=self._fragments,
fragment_ids=self._fragment_ids,
full_text_query=FullTextSearchQuery(
query=self._query, columns=self._fts_columns
query=self._query_with_phrase_semantics(), columns=self._fts_columns
),
offset=self._offset,
fast_search=self._fast_search,
@@ -1882,22 +2018,13 @@ class LanceFtsQueryBuilder(LanceQueryBuilder):
def to_arrow(self, *, timeout: Optional[timedelta] = None) -> pa.Table:
self._table._ensure_no_legacy_fts_index()
query = self._query
if self._phrase_query:
if isinstance(query, str):
if not query.startswith('"') or not query.endswith('"'):
self._query = f'"{query}"'
elif isinstance(query, FullTextQuery) and not isinstance(
query, PhraseQuery
):
raise TypeError("Please use PhraseQuery for phrase queries.")
query = self.to_query_object()
query = self._query_for_scan()
results = self._table._execute_query(query, timeout=timeout)
results = results.read_all()
if self._reranker is not None:
results = self._reranker.rerank_fts(self._query, results)
check_reranker_result(results)
return results
return self._finalize_blob_query_table(results)
def to_batches(
self, /, batch_size: Optional[int] = None, timeout: Optional[timedelta] = None
@@ -1925,7 +2052,9 @@ class LanceFtsQueryBuilder(LanceQueryBuilder):
class LanceEmptyQueryBuilder(LanceQueryBuilder):
def to_arrow(self, *, timeout: Optional[timedelta] = None) -> pa.Table:
return self.to_batches(timeout=timeout).read_all()
return self._finalize_blob_query_table(
self.to_batches(timeout=timeout).read_all()
)
def to_query_object(self) -> Query:
return Query(
@@ -1947,7 +2076,7 @@ class LanceEmptyQueryBuilder(LanceQueryBuilder):
def to_batches(
self, /, batch_size: Optional[int] = None, timeout: Optional[timedelta] = None
) -> pa.RecordBatchReader:
query = self.to_query_object()
query = self._query_for_scan()
return self._table._execute_query(query, batch_size=batch_size, timeout=timeout)
def rerank(self, reranker: Reranker) -> LanceEmptyQueryBuilder:
@@ -2019,14 +2148,13 @@ class LanceHybridQueryBuilder(LanceQueryBuilder):
return vector_query, text_query
def phrase_query(self, phrase_query: bool = None) -> LanceHybridQueryBuilder:
def phrase_query(self, phrase_query: bool = True) -> LanceHybridQueryBuilder:
"""Set whether to use phrase query.
Parameters
----------
phrase_query: bool, default True
If True, then the query will be wrapped in quotes and
double quotes replaced by single quotes.
If True, then an unquoted string query will be wrapped in quotes.
Returns
-------
@@ -2051,15 +2179,25 @@ class LanceHybridQueryBuilder(LanceQueryBuilder):
fts_results = fts_future.result()
vector_results = vector_future.result()
return self._combine_hybrid_results(
results = self._combine_hybrid_results(
fts_results=fts_results,
vector_results=vector_results,
norm=self._norm,
fts_query=self._fts_query._query,
reranker=self._reranker,
limit=self._limit,
with_row_ids=self._with_row_id,
with_row_ids=True,
)
return self._finish_hybrid_results(results)
def _finish_hybrid_results(self, results: pa.Table) -> pa.Table:
if self._user_requested_row_id():
return results
if self._blob_auto_row_id_enabled():
return self._finalize_blob_query_table(results)
if "_rowid" in results.column_names:
return results.drop(["_rowid"])
return results
@staticmethod
def _combine_hybrid_results(
@@ -2500,7 +2638,7 @@ class LanceHybridQueryBuilder(LanceQueryBuilder):
self._vector_query.ef(self._ef)
if self._bypass_vector_index:
self._vector_query.bypass_vector_index()
if self._lower_bound or self._upper_bound:
if self._lower_bound is not None or self._upper_bound is not None:
self._vector_query.distance_range(
lower_bound=self._lower_bound, upper_bound=self._upper_bound
)
@@ -2530,6 +2668,9 @@ class AsyncQueryBase(object):
self._with_row_address = None
self._fragments = None
self._fragment_ids = None
self._with_row_id = None
self._blob_auto_row_id = False
self._blob_paths: tuple[str, ...] = ()
def to_query_object(self) -> Query:
"""
@@ -2539,11 +2680,46 @@ class AsyncQueryBase(object):
python and more easily serializable.
"""
query = Query.from_inner(self._inner.to_query_request())
query.with_row_id = self._user_requested_row_id()
query.with_row_address = self._with_row_address
query.fragments = self._fragments
query.fragment_ids = self._fragment_ids
return query
def _user_requested_row_id(self) -> bool:
return self._with_row_id is True
def _blob_auto_row_id_enabled(self) -> bool:
return self._blob_auto_row_id
def _finalize_blob_query_table(self, tbl: pa.Table) -> pa.Table:
return finalize_blob_query_table(
tbl,
user_requested_row_id=self._user_requested_row_id(),
blob_auto_row_id=self._blob_auto_row_id_enabled(),
blob_paths=self._blob_paths,
)
async def _maybe_add_blob_row_id(self) -> None:
if self._table is None or not supports_blob_auto_row_id(self._table):
self._blob_auto_row_id = False
self._blob_paths = ()
return
req = self._inner.to_query_request()
schema = await self._table.schema()
self._blob_auto_row_id = blob_auto_row_id_for_scan(
self._table,
schema,
req.select,
with_row_id=self._with_row_id,
)
if not self._blob_auto_row_id:
self._blob_paths = ()
return
self._blob_paths = tuple(blob_v2_projection_sources(schema, req.select).keys())
self._inner.with_row_id()
def select(self, columns: Union[List[str], dict[str, str]]) -> Self:
"""
Return only the specified columns.
@@ -2596,6 +2772,7 @@ class AsyncQueryBase(object):
"""
Include the _rowid column in the results.
"""
self._with_row_id = True
self._inner.with_row_id()
return self
@@ -2642,6 +2819,7 @@ class AsyncQueryBase(object):
If not specified, no timeout is applied. If the query does not
complete within the specified time, an error will be raised.
"""
await self._maybe_add_blob_row_id()
return AsyncRecordBatchReader(
await self._inner.execute(
max_batch_length=max_batch_length, timeout=timeout
@@ -2672,8 +2850,8 @@ class AsyncQueryBase(object):
complete within the specified time, an error will be raised.
"""
batch_iter = await self.to_batches(timeout=timeout)
return pa.Table.from_batches(
await batch_iter.read_all(), schema=batch_iter.schema
return self._finalize_blob_query_table(
pa.Table.from_batches(await batch_iter.read_all(), schema=batch_iter.schema)
)
async def to_list(self, timeout: Optional[timedelta] = None) -> List[dict]:
@@ -2740,7 +2918,7 @@ class AsyncQueryBase(object):
Forwarded to pyarrow.Table.to_pandas after query execution and
optional flattening.
"""
_validate_blob_mode(blob_mode)
validate_blob_mode(blob_mode)
if hasattr(self._inner, "output_schema"):
schema = await self.output_schema()
if _blob_mode_requires_native_pandas(blob_mode, schema):
@@ -2781,14 +2959,36 @@ class AsyncQueryBase(object):
if not _query_is_plain_scan(query):
return None
schema = await self._table.schema()
blob_auto_row_id = blob_auto_row_id_for_scan(
self._table,
schema,
query.columns,
with_row_id=self._with_row_id,
)
blob_sources = (
blob_v2_projection_sources(schema, query.columns)
if blob_mode == "bytes"
else {}
)
dataset = await self._table._to_lance()
scanner = dataset.scanner(
**_scanner_kwargs_for_query(query, blob_mode, dataset)
**_scanner_kwargs_for_query(
query,
"descriptions" if blob_sources else blob_mode,
dataset,
with_row_id=query.with_row_id or blob_auto_row_id or bool(blob_sources),
)
)
return await _finish_plain_scan_pandas_async(
scanner,
blob_mode=blob_mode,
blob_sources=blob_sources,
fetch_blobs=self._table.fetch_blobs,
strip_auto_row_id=blob_auto_row_id,
flatten=flatten,
**kwargs,
)
if flatten is not None:
tbl = flatten_columns(_scanner_to_table(scanner), flatten)
return tbl.to_pandas(**kwargs)
return _scanner_to_pandas(scanner, blob_mode, **kwargs)
async def to_polars(
self,
@@ -3573,9 +3773,24 @@ class AsyncHybridQuery(AsyncStandardQuery, AsyncVectorQueryBase):
fts_query = AsyncFTSQuery(self._inner.to_fts_query(), self._table)
vec_query = AsyncVectorQuery(self._inner.to_vector_query(), self._table)
# save the row ID choice that was made on the query builder and force it
# to actually fetch the row ids because we need this for reranking
with_row_ids = self._inner.get_with_row_id()
req = fts_query._inner.to_query_request()
blob_auto_row_id = False
blob_paths: tuple[str, ...] = ()
if self._table is not None and supports_blob_auto_row_id(self._table):
schema = await self._table.schema()
blob_auto_row_id = blob_auto_row_id_for_scan(
self._table,
schema,
req.select,
with_row_id=self._with_row_id,
)
if blob_auto_row_id:
blob_paths = tuple(
blob_v2_projection_sources(schema, req.select).keys()
)
self._blob_auto_row_id = blob_auto_row_id
self._blob_paths = blob_paths
fts_query.with_row_id()
vec_query.with_row_id()
@@ -3591,8 +3806,14 @@ class AsyncHybridQuery(AsyncStandardQuery, AsyncVectorQueryBase):
fts_query=fts_query.get_query(),
reranker=self._reranker,
limit=self._inner.get_limit(),
with_row_ids=with_row_ids,
with_row_ids=True,
)
if (
not self._user_requested_row_id()
and not blob_auto_row_id
and "_rowid" in result.column_names
):
result = result.drop(["_rowid"])
return AsyncRecordBatchReader(result, max_batch_length=max_batch_length)
+31
View File
@@ -28,6 +28,7 @@ from lancedb._lancedb import (
UpdateFieldMetadataResult,
DeleteResult,
DropColumnsResult,
FtsToken,
IndexConfig,
LsmWriteSpec,
MergeResult,
@@ -244,6 +245,23 @@ class RemoteTable(Table):
"""List all the indices on the table"""
return LOOP.run(self._table.list_indices())
def tokenize(
self,
query: str,
*,
column: Optional[str] = None,
index_name: Optional[str] = None,
) -> Iterable[FtsToken]:
"""Tokenize a query using the tokenizer configured on an FTS index.
Model-backed tokenizers such as ``jieba/*`` and ``lindera/*`` are
rebuilt in the client process from index metadata, so the same tokenizer
model files must exist locally.
"""
return LOOP.run(
self._table.tokenize(query, column=column, index_name=index_name)
)
def index_stats(self, index_uuid: str) -> Optional[IndexStatistics]:
"""List all the stats of a specified index"""
return LOOP.run(self._table.index_stats(index_uuid))
@@ -994,6 +1012,19 @@ class RemoteTable(Table):
"migrate_v2_manifest_paths() is not supported on the LanceDB Cloud"
)
def blob_columns(self) -> list[str]:
raise NotImplementedError(
"blob_columns() is not yet supported on the LanceDB Cloud"
)
def fetch_blobs(self, column: str, row_ids) -> pa.LargeBinaryArray:
raise NotImplementedError("fetch_blobs() is not supported on LanceDB Cloud")
def fetch_blob_files(self, column: str, row_ids):
raise NotImplementedError(
"fetch_blob_files() is not supported on LanceDB Cloud"
)
def head(self, n=5) -> pa.Table:
"""
Return the first `n` rows of the table.
@@ -12,6 +12,7 @@ from .rrf import RRFReranker
from .mrr import MRRReranker
from .answerdotai import AnswerdotaiRerankers
from .voyageai import VoyageAIReranker
from .watsonx import WatsonxReranker
__all__ = [
"Reranker",
@@ -25,4 +26,5 @@ __all__ = [
"AnswerdotaiRerankers",
"VoyageAIReranker",
"MRRReranker",
"WatsonxReranker",
]
+180
View File
@@ -0,0 +1,180 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
import os
from functools import cached_property
from typing import Dict, Optional
import pyarrow as pa
from ..util import attempt_import_or_raise
from .base import Reranker
DEFAULT_WATSONX_URL = "https://us-south.ml.cloud.ibm.com"
class WatsonxReranker(Reranker):
"""
Reranks the results using the IBM watsonx.ai Rerank API.
Uses the ``ibm_watsonx_ai`` SDK (``Rerank.generate``) under the hood.
API Docs:
https://cloud.ibm.com/docs/apis/watsonx-ai#text-rerank
Supported rerank models:
https://dataplatform.cloud.ibm.com/docs/content/wsj/analyze-data/fm-models-embed.html?context=wx#rerank
Parameters
----------
model_name : str, default "cross-encoder/ms-marco-minilm-l-12-v2"
The ID of the rerank model to use.
column : str, default "text"
The name of the column to use as input to the reranker.
top_n : int, optional
Return only the top-n results. If ``None``, all results are returned.
return_score : str, default "relevance"
Options are ``"relevance"`` or ``"all"``.
api_key : str, optional
IBM Cloud API key. Falls back to the ``WATSONX_API_KEY`` environment
variable when not provided.
project_id : str, optional
watsonx.ai project ID. Falls back to the ``WATSONX_PROJECT_ID``
environment variable when not provided. Mutually exclusive with
``space_id`` — exactly one must be supplied.
space_id : str, optional
watsonx.ai deployment space ID. Falls back to the ``WATSONX_SPACE_ID``
environment variable when not provided. Mutually exclusive with
``project_id`` — exactly one must be supplied.
url : str, optional
watsonx.ai service URL. Defaults to
``"https://us-south.ml.cloud.ibm.com"``.
truncate_input_tokens : int, optional
Truncate each input to this many tokens before scoring. Passed
directly to the ``parameters`` dict of ``Rerank.generate``.
"""
def __init__(
self,
model_name: str = "cross-encoder/ms-marco-minilm-l-12-v2",
column: str = "text",
top_n: Optional[int] = None,
return_score: str = "relevance",
api_key: Optional[str] = None,
project_id: Optional[str] = None,
space_id: Optional[str] = None,
url: Optional[str] = None,
truncate_input_tokens: Optional[int] = None,
):
super().__init__(return_score)
self.model_name = model_name
self.column = column
self.top_n = top_n
self.api_key = api_key
self.project_id = project_id
self.space_id = space_id
self.url = url
self.truncate_input_tokens = truncate_input_tokens
def __str__(self) -> str:
return f"WatsonxReranker(model_name={self.model_name})"
@cached_property
def _client(self):
ibm_watsonx_ai = attempt_import_or_raise("ibm_watsonx_ai")
ibm_watsonx_ai_foundation_models = attempt_import_or_raise(
"ibm_watsonx_ai.foundation_models"
)
# --- credentials ---
api_key = self.api_key or os.environ.get("WATSONX_API_KEY")
if not api_key:
raise ValueError(
"WATSONX_API_KEY not set. Either set it in your environment or "
"pass it as `api_key` argument to WatsonxReranker."
)
credentials = ibm_watsonx_ai.Credentials(
api_key=api_key,
url=self.url or DEFAULT_WATSONX_URL,
)
# --- project_id / space_id (exactly one required) ---
project_id = self.project_id or os.environ.get("WATSONX_PROJECT_ID")
space_id = self.space_id or os.environ.get("WATSONX_SPACE_ID")
if project_id and space_id:
raise ValueError("Provide either `project_id` or `space_id`, not both.")
if not project_id and not space_id:
raise ValueError(
"Either WATSONX_PROJECT_ID or WATSONX_SPACE_ID must be set. "
"Pass one as an argument to WatsonxReranker or set the corresponding "
"environment variable."
)
kwargs: Dict = dict(model_id=self.model_name, credentials=credentials)
if project_id:
kwargs["project_id"] = project_id
else:
kwargs["space_id"] = space_id
return ibm_watsonx_ai_foundation_models.Rerank(**kwargs)
def _build_params(self) -> Dict:
"""Build the ``parameters`` dict forwarded to ``Rerank.generate``."""
return_options: Dict = {"inputs": True}
if self.top_n is not None:
return_options["top_n"] = self.top_n
params: Dict = {"return_options": return_options}
if self.truncate_input_tokens is not None:
params["truncate_input_tokens"] = self.truncate_input_tokens
return params
def _rerank(self, result_set: pa.Table, query: str) -> pa.Table:
result_set = self._handle_empty_results(result_set)
if len(result_set) == 0:
return result_set
docs = result_set[self.column].to_pylist()
response = self._client.generate(
query=query,
inputs=docs,
params=self._build_params(),
)
results = response["results"]
indices, scores = zip(
*[(result["index"], result["score"]) for result in results]
)
result_set = result_set.take(list(indices))
result_set = result_set.append_column(
"_relevance_score", pa.array(scores, type=pa.float32())
)
return result_set
def rerank_hybrid(
self,
query: str,
vector_results: pa.Table,
fts_results: pa.Table,
) -> pa.Table:
if self.score == "all":
combined_results = self._merge_and_keep_scores(vector_results, fts_results)
else:
combined_results = self.merge_results(vector_results, fts_results)
combined_results = self._rerank(combined_results, query)
if self.score == "relevance":
combined_results = self._keep_relevance_score(combined_results)
return combined_results
def rerank_vector(self, query: str, vector_results: pa.Table) -> pa.Table:
vector_results = self._rerank(vector_results, query)
if self.score == "relevance":
vector_results = vector_results.drop_columns(["_distance"])
return vector_results
def rerank_fts(self, query: str, fts_results: pa.Table) -> pa.Table:
fts_results = self._rerank(fts_results, query)
if self.score == "relevance":
fts_results = fts_results.drop_columns(["_score"])
return fts_results
+125 -1
View File
@@ -2,10 +2,134 @@
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
"""Schema related utilities."""
"""Schema helpers for Lance blob columns."""
import pyarrow as pa
_BLOB_EXTENSION_NAME = "lance.blob.v2"
_BLOB_V1_KEY = "lance-encoding:blob"
_ARROW_EXT_NAME_KEY = "ARROW:extension:name"
class BlobType(pa.ExtensionType):
"""PyArrow extension type for a Lance blob v2 column.
Queries return descriptors; call :meth:`~lancedb.table.Table.fetch_blob_files`
for lazy reads or :meth:`~lancedb.table.Table.fetch_blobs` for eager bytes.
"""
def __init__(self) -> None:
storage_type = pa.struct(
[
pa.field("data", pa.large_binary(), nullable=True),
pa.field("uri", pa.utf8(), nullable=True),
pa.field("position", pa.uint64(), nullable=True),
pa.field("size", pa.uint64(), nullable=True),
]
)
super().__init__(storage_type, _BLOB_EXTENSION_NAME)
def __arrow_ext_serialize__(self) -> bytes:
return b""
@classmethod
def __arrow_ext_deserialize__(
cls, storage_type: pa.DataType, serialized: bytes
) -> "BlobType":
return cls()
def __reduce__(self):
# Ensure pickle round-trips on older pyarrow (apache/arrow#35599).
return type(self).__arrow_ext_deserialize__, (
self.storage_type,
self.__arrow_ext_serialize__(),
)
try:
pa.register_extension_type(BlobType()) # type: ignore[arg-type]
except pa.ArrowKeyError:
pass
def _metadata_value(metadata: dict, key: str):
return metadata.get(key.encode()) or metadata.get(key)
def _metadata_marks_blob_v2(metadata: dict) -> bool:
if not metadata:
return False
extension_name = _metadata_value(metadata, _ARROW_EXT_NAME_KEY)
return extension_name in (_BLOB_EXTENSION_NAME, _BLOB_EXTENSION_NAME.encode())
def _metadata_marks_legacy_blob(metadata: dict) -> bool:
if not metadata:
return False
return _metadata_value(metadata, _BLOB_V1_KEY) in ("true", b"true")
def is_blob_v2_field(field: pa.Field) -> bool:
"""Return True if `field` declares a blob v2 extension column."""
field_type = field.type
if (
isinstance(field_type, pa.ExtensionType)
and field_type.extension_name == _BLOB_EXTENSION_NAME
):
return True
return _metadata_marks_blob_v2(field.metadata or {})
def is_blob_like_field(field: pa.Field) -> bool:
"""Blob detection for ``to_pandas(blob_mode=...)`` and scanner paths only.
Matches v2 extension fields on table schema, legacy ``lance-encoding:blob``
storage columns, and v2 query descriptor fields (the engine tags those with
the same metadata). Not used for fetch or auto ``_rowid``.
"""
return is_blob_v2_field(field) or _metadata_marks_legacy_blob(field.metadata or {})
def _collect_blob_paths(schema: pa.Schema, is_blob) -> list[str]:
paths: list[str] = []
def walk(fields, prefix: str) -> None:
for field in fields:
path = f"{prefix}.{field.name}" if prefix else field.name
if is_blob(field):
paths.append(path)
elif pa.types.is_struct(field.type):
walk(field.type, path)
elif (
pa.types.is_list(field.type)
or pa.types.is_large_list(field.type)
or pa.types.is_fixed_size_list(field.type)
):
walk([field.type.value_field], path)
walk(schema, "")
return paths
def blob_column_paths(schema: pa.Schema) -> list[str]:
"""Dotted paths of blob-like columns (v2 extension or legacy metadata)."""
return _collect_blob_paths(schema, is_blob_like_field)
def blob_v2_column_paths(schema: pa.Schema) -> list[str]:
return _collect_blob_paths(schema, is_blob_v2_field)
def schema_has_blob_field(schema: pa.Schema) -> bool:
return bool(blob_column_paths(schema))
def blob(name: str, nullable: bool = True) -> pa.Field:
"""Create a Lance blob v2 column field."""
return pa.field(name, BlobType(), nullable=nullable)
def vector(dimension: int, value_type: pa.DataType = pa.float32()) -> pa.DataType:
"""A help function to create a vector type.
+190 -34
View File
@@ -29,6 +29,14 @@ from urllib.parse import urlparse
from lancedb.scannable import _register_optional_converters, to_scannable
from . import __version__
from ._blob import (
BlobFile,
_normalize_blob_row_ids,
_wrap_blob_files,
strip_auto_row_ids,
validate_blob_mode,
)
from .types import BlobMode
from lancedb.arrow import peek_reader
from lancedb.background_loop import LOOP, embedding_executor
from .dependencies import (
@@ -88,10 +96,7 @@ from .util import (
value_to_sql,
)
from .index import lang_mapping
BlobMode = Literal["lazy", "bytes", "descriptions"]
_VALID_BLOB_MODES = ("lazy", "bytes", "descriptions")
from .schema import blob_v2_column_paths, schema_has_blob_field
def _should_push_down_query_table(
@@ -100,23 +105,6 @@ def _should_push_down_query_table(
return namespace_client is not None and "QueryTable" in pushdown_operations
def _validate_blob_mode(blob_mode: BlobMode) -> None:
if blob_mode not in _VALID_BLOB_MODES:
modes = ", ".join(repr(mode) for mode in _VALID_BLOB_MODES)
raise ValueError(f"blob_mode must be one of {modes}, got {blob_mode!r}")
def _field_is_blob(field: pa.Field) -> bool:
metadata = field.metadata or {}
return metadata.get(b"lance-encoding:blob") == b"true" or (
metadata.get("lance-encoding:blob") == "true"
)
def _schema_has_blob_field(schema: pa.Schema) -> bool:
return any(_field_is_blob(field) for field in schema)
_MODEL_BACKED_TOKENIZER_PREFIXES = ("jieba", "lindera")
_MODEL_BACKED_TOKENIZER_ERRORS = (
"unknown base tokenizer",
@@ -185,6 +173,7 @@ if TYPE_CHECKING:
UpdateFieldMetadataResult,
DeleteResult,
DropColumnsResult,
FtsToken,
LsmWriteSpec,
MergeResult,
UpdateResult,
@@ -1159,6 +1148,8 @@ class Table(ABC):
- "whitespace": Split text by whitespace, but not punctuation.
- "raw": No tokenization. The entire text is treated as a single token.
- "ngram": N-Gram tokenizer.
- "icu": ICU dictionary-based word segmentation.
- "icu/split": ICU segmentation with simple-style delimiter splitting.
- "jieba/*": Jieba tokenizer loaded from Lance's language model home.
- "lindera/*": Lindera tokenizer loaded from Lance's language model home.
language : str, default "English"
@@ -1523,6 +1514,31 @@ class Table(ABC):
A query object that can be executed to get the rows.
"""
@abstractmethod
def blob_columns(self) -> list[str]:
"""Names of the blob v2 columns declared on this table."""
@abstractmethod
def fetch_blobs(
self, column: str, row_ids: Union[list[int], pa.Table]
) -> pa.LargeBinaryArray:
"""Materialize full blob bytes for ``column`` at the given rows.
Convenience for small payloads. For large values use
:meth:`fetch_blob_files`.
"""
@abstractmethod
def fetch_blob_files(
self, column: str, row_ids: Union[list[int], pa.Table]
) -> "list[Optional[BlobFile]]":
"""Open lazy, seekable :class:`~lancedb._blob.BlobFile` handles.
Prefer this over :meth:`fetch_blobs` for large payloads. ``row_ids`` is
a ``list[int]`` or query ``pyarrow.Table`` with ``_rowid`` (or stashed
row-id metadata). Null rows are ``None``. Local tables only.
"""
@abstractmethod
def _execute_query(
self,
@@ -1786,6 +1802,24 @@ class Table(ABC):
[Table.create_index][lancedb.table.Table.create_index]
"""
@abstractmethod
def tokenize(
self,
query: str,
*,
column: Optional[str] = None,
index_name: Optional[str] = None,
) -> Iterable[FtsToken]:
"""
Tokenize a query using the tokenizer configured on an FTS index.
Specify exactly one of ``column`` or ``index_name``.
Model-backed tokenizers such as ``jieba/*`` and ``lindera/*`` are
rebuilt in the client process from index metadata. For remote tables,
this means the same tokenizer model files must also exist locally.
"""
@abstractmethod
def index_stats(self, index_name: str) -> Optional[IndexStatistics]:
"""
@@ -2204,6 +2238,19 @@ class LanceTable(Table):
def take_row_ids(self, row_ids: list[int]) -> LanceTakeQueryBuilder:
return LanceTakeQueryBuilder(self._table.take_row_ids(row_ids))
def blob_columns(self) -> list[str]:
return LOOP.run(self._table.blob_columns())
def fetch_blobs(
self, column: str, row_ids: Union[list[int], pa.Table]
) -> pa.LargeBinaryArray:
return LOOP.run(self._table.fetch_blobs(column, row_ids))
def fetch_blob_files(
self, column: str, row_ids: Union[list[int], pa.Table]
) -> "list[Optional[BlobFile]]":
return LOOP.run(self._table.fetch_blob_files(column, row_ids))
@property
def tags(self) -> Tags:
"""Tag management for the table.
@@ -2399,9 +2446,14 @@ class LanceTable(Table):
-------
pd.DataFrame
"""
_validate_blob_mode(blob_mode)
if blob_mode == "descriptions" or not _schema_has_blob_field(self.schema):
return self.to_arrow().to_pandas(**kwargs)
validate_blob_mode(blob_mode)
if blob_mode == "descriptions" or not schema_has_blob_field(self.schema):
arrow_tbl = self.to_arrow()
if blob_mode == "descriptions":
arrow_tbl = strip_auto_row_ids(
arrow_tbl, blob_v2_column_paths(self.schema)
)
return arrow_tbl.to_pandas(**kwargs)
if (
blob_mode == "lazy"
@@ -2410,6 +2462,9 @@ class LanceTable(Table):
):
return self.to_arrow().to_pandas(**kwargs)
if blob_mode == "bytes" and blob_v2_column_paths(self.schema):
return self.search().to_pandas(blob_mode=blob_mode, **kwargs)
return self.to_lance().to_pandas(blob_mode=blob_mode, **kwargs)
def to_arrow(self) -> pa.Table:
@@ -3710,6 +3765,26 @@ class LanceTable(Table):
"""
return LOOP.run(self._table.list_indices())
def tokenize(
self,
query: str,
*,
column: Optional[str] = None,
index_name: Optional[str] = None,
) -> Iterable[FtsToken]:
"""
Tokenize a query using the tokenizer configured on an FTS index.
Specify exactly one of ``column`` or ``index_name``.
Model-backed tokenizers such as ``jieba/*`` and ``lindera/*`` are
rebuilt in the client process from index metadata. For remote tables,
this means the same tokenizer model files must also exist locally.
"""
return LOOP.run(
self._table.tokenize(query, column=column, index_name=index_name)
)
def index_stats(self, index_name: str) -> Optional[IndexStatistics]:
"""
Retrieve statistics about an index
@@ -4079,17 +4154,58 @@ def _handle_bad_vector_column(
raise ValueError(
"`fill_value` must not be None if `on_bad_vectors` is 'fill'"
)
vec_arr = pc.if_else(
is_bad,
pa.scalar([fill_value] * dim, type=vec_arr.type),
vec_arr,
)
vec_arr = _fill_bad_vector_values(vec_arr, dim, fill_value)
else:
raise ValueError(f"Invalid value for on_bad_vectors: {on_bad_vectors}")
return data.set_column(position, vector_column_name, vec_arr)
def _fill_bad_vector_values(
arr: Union[pa.Array, pa.ChunkedArray],
dim: int,
fill_value: float,
) -> pa.Array:
if not isinstance(arr, pa.ChunkedArray):
arr = pa.chunked_array([arr])
arr = arr.combine_chunks()
# A fixed-size slice truncates long vectors and pads short vectors with nulls.
# Slice an array marking the original child nulls in parallel so padding nulls
# can be distinguished from null values that were already present.
sliced = pc.list_slice(arr, 0, dim, return_fixed_size_list=True)
child_nulls = pc.is_null(arr.values)
parent_nulls = pc.is_null(arr)
if pa.types.is_list(arr.type):
original_child_nulls = pa.ListArray.from_arrays(
arr.offsets, child_nulls, mask=parent_nulls
)
elif pa.types.is_large_list(arr.type):
original_child_nulls = pa.LargeListArray.from_arrays(
arr.offsets, child_nulls, mask=parent_nulls
)
else:
original_child_nulls = pa.FixedSizeListArray.from_arrays(
child_nulls, arr.type.list_size, mask=parent_nulls
)
sliced_child_nulls = pc.list_slice(
original_child_nulls, 0, dim, return_fixed_size_list=True
)
needs_fill = pc.is_null(sliced_child_nulls.values)
values = sliced.values
if pa.types.is_floating(values.type):
values_for_nan_check = (
values.cast(pa.float32()) if pa.types.is_float16(values.type) else values
)
needs_fill = pc.or_kleene(needs_fill, pc.is_nan(values_for_nan_check))
fill_scalar = pa.scalar(fill_value).cast(values.type)
filled_values = pc.if_else(needs_fill, fill_scalar, values)
filled = pa.FixedSizeListArray.from_arrays(filled_values, dim)
return filled.cast(arr.type)
def has_nan_values(arr: Union[pa.ListArray, pa.ChunkedArray]) -> pa.BooleanArray:
if isinstance(arr, pa.ChunkedArray):
values = pa.chunked_array([chunk.flatten() for chunk in arr.chunks])
@@ -4539,14 +4655,18 @@ class AsyncTable:
-------
pd.DataFrame
"""
_validate_blob_mode(blob_mode)
if blob_mode == "descriptions" or not _schema_has_blob_field(
await self.schema()
):
return (await self.to_arrow()).to_pandas(**kwargs)
validate_blob_mode(blob_mode)
schema = await self.schema()
if blob_mode == "descriptions" or not schema_has_blob_field(schema):
arrow_tbl = await self.to_arrow()
if blob_mode == "descriptions":
arrow_tbl = strip_auto_row_ids(arrow_tbl, blob_v2_column_paths(schema))
return arrow_tbl.to_pandas(**kwargs)
if blob_mode == "lazy" and get_uri_scheme(await self.uri()) == "memory":
return (await self.to_arrow()).to_pandas(**kwargs)
if blob_mode == "bytes" and blob_v2_column_paths(schema):
return await self.query().to_pandas(blob_mode=blob_mode, **kwargs)
return (await self._to_lance()).to_pandas(blob_mode=blob_mode, **kwargs)
async def to_arrow(self) -> pa.Table:
@@ -5647,6 +5767,24 @@ class AsyncTable:
"""
return AsyncTakeQuery(self._inner.take_row_ids(row_ids), self)
async def blob_columns(self) -> list[str]:
return await self._inner.blob_columns()
async def fetch_blobs(
self, column: str, row_ids: Union[list[int], pa.Table]
) -> pa.LargeBinaryArray:
return await self._inner.fetch_blobs(
column, _normalize_blob_row_ids(row_ids, column)
)
async def fetch_blob_files(
self, column: str, row_ids: Union[list[int], pa.Table]
) -> "list[Optional[BlobFile]]":
handles = await self._inner.fetch_blob_files(
column, _normalize_blob_row_ids(row_ids, column)
)
return _wrap_blob_files(handles)
@property
def tags(self) -> AsyncTags:
"""Tag management for the dataset.
@@ -5748,6 +5886,24 @@ class AsyncTable:
"""
return await self._inner.list_indices()
async def tokenize(
self,
query: str,
*,
column: Optional[str] = None,
index_name: Optional[str] = None,
) -> Iterable[FtsToken]:
"""
Tokenize a query using the tokenizer configured on an FTS index.
Specify exactly one of ``column`` or ``index_name``.
Model-backed tokenizers such as ``jieba/*`` and ``lindera/*`` are
rebuilt in the client process from index metadata. For remote tables,
this means the same tokenizer model files must also exist locally.
"""
return await self._inner.tokenize(query, column=column, index_name=index_name)
async def index_stats(self, index_name: str) -> Optional[IndexStatistics]:
"""
Retrieve statistics about an index
+17 -2
View File
@@ -1,11 +1,24 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
from typing import Literal
from __future__ import annotations
from typing import Dict, List, Literal, Optional, Tuple, Union
from .expr import Expr
# Query type literals
QueryType = Literal["vector", "fts", "hybrid", "auto"]
BlobMode = Literal["lazy", "bytes", "descriptions"]
QueryProjectionSpec = Union[
List[str],
List[Tuple[str, Union[str, Expr]]],
Dict[str, Union[str, Expr]],
]
QueryProjection = Optional[QueryProjectionSpec]
# Distance type literals
DistanceType = Literal["l2", "cosine", "dot"]
DistanceTypeWithHamming = Literal["l2", "cosine", "dot", "hamming"]
@@ -42,5 +55,7 @@ IndexType = Literal[
]
# Tokenizer literals
BuiltinTokenizerType = Literal["simple", "raw", "whitespace", "ngram"]
BuiltinTokenizerType = Literal[
"simple", "raw", "whitespace", "ngram", "icu", "icu/split"
]
BaseTokenizerType = BuiltinTokenizerType | str
+562
View File
@@ -0,0 +1,562 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
import io
import pyarrow as pa
import pyarrow.compute as pc
import pytest
import lancedb
from lancedb._blob import read_row_ids_from_hits, stash_auto_row_ids
from lancedb.index import FTS
from lancedb.schema import blob_column_paths, blob_v2_column_paths
def _blob_table(name, rows):
db = lancedb.connect("memory:///")
schema = pa.schema([pa.field("id", pa.int64()), lancedb.blob("image")])
table = db.create_table(name, schema=schema)
table.add(rows)
return table
def _blob_array(name, values):
blob_type = lancedb.blob(name).type
storage_type = blob_type.storage_type
storage = pa.StructArray.from_arrays(
[
pa.array(values, type=pa.large_binary()),
pa.array([None] * len(values), type=pa.string()),
pa.array([None] * len(values), type=pa.uint64()),
pa.array([None] * len(values), type=pa.uint64()),
],
fields=list(storage_type),
)
return pa.ExtensionArray.from_storage(blob_type, storage)
def _row_ids_by_id(table):
hits = table.search().with_row_id(True).limit(1000).to_arrow()
assert "_rowid" in hits.column_names
return dict(zip(hits["id"].to_pylist(), hits["_rowid"].to_pylist()))
def test_blob_factory_declares_v2_field():
field = lancedb.blob("image")
assert isinstance(field.type, pa.ExtensionType)
assert field.type.extension_name == "lance.blob.v2"
def test_blob_v2_column_paths_include_list_children():
schema = pa.schema(
[
pa.field("id", pa.int64()),
pa.field("info", pa.struct([lancedb.blob("blob")])),
pa.field("images", pa.list_(lancedb.blob("image"))),
pa.field("large_images", pa.large_list(lancedb.blob("large_image"))),
pa.field(
"fixed_images",
pa.list_(lancedb.blob("fixed_image"), list_size=2),
),
]
)
assert blob_v2_column_paths(schema) == [
"info.blob",
"images.image",
"large_images.large_image",
"fixed_images.fixed_image",
]
def _legacy_v1_table(name):
db = lancedb.connect("memory:///")
schema = pa.schema(
[
pa.field("id", pa.int64()),
pa.field(
"legacy", pa.large_binary(), metadata={"lance-encoding:blob": "true"}
),
]
)
table = db.create_table(name, schema=schema)
table.add([{"id": 1, "legacy": b"old"}])
return table
def test_blob_v2_column_paths_exclude_legacy_metadata():
schema = pa.schema(
[
pa.field("id", pa.int64()),
lancedb.blob("image"),
pa.field(
"legacy", pa.large_binary(), metadata={"lance-encoding:blob": "true"}
),
]
)
assert blob_v2_column_paths(schema) == ["image"]
assert blob_column_paths(schema) == ["image", "legacy"]
def test_blob_v2_paths_match_blob_columns():
table = _blob_table("paths_match", [{"id": 1, "image": b"x"}])
assert blob_v2_column_paths(table.schema) == table.blob_columns()
db = lancedb.connect("memory:///")
info = pa.StructArray.from_arrays(
[
pa.array(["first"], type=pa.string()),
_blob_array("blob", [b"nested"]),
],
names=["name", "blob"],
)
data = pa.Table.from_arrays(
[pa.array([1], type=pa.int64()), info],
names=["id", "info"],
)
nested = db.create_table("nested_paths", data=data)
assert blob_v2_column_paths(nested.schema) == nested.blob_columns()
def test_auto_row_id_stash_round_trip():
table = _blob_table(
"stash_round_trip",
[{"id": 1, "image": b"alpha"}, {"id": 2, "image": b"beta"}],
)
hits = table.search().with_row_id(True).limit(10).to_arrow()
row_ids = hits["_rowid"].to_pylist()
stashed = stash_auto_row_ids(hits, ["image"])
assert "_rowid" not in stashed.column_names
assert stashed.schema.field("image").metadata == hits.schema.field("image").metadata
assert read_row_ids_from_hits(stashed, "image") == row_ids
def test_blob_query_omits_auto_row_id():
table = _blob_table("rowid", [{"id": 1, "image": b"x"}])
hits = table.search().limit(10).to_arrow()
assert "_rowid" not in hits.column_names
def test_blob_query_explicit_row_id_opt_in():
table = _blob_table("explicit_rowid", [{"id": 1, "image": b"x"}])
hits = table.search().with_row_id(True).limit(10).to_arrow()
assert "_rowid" in hits.column_names
def test_table_to_pandas_descriptions_mode_omits_row_id():
table = _blob_table("descriptions_no_leak", [{"id": 1, "image": b"x"}])
df = table.to_pandas(blob_mode="descriptions")
descriptor = df["image"].iloc[0]
assert "_lance_row_id" not in descriptor
assert set(descriptor.keys()) == {"kind", "position", "size", "blob_id", "blob_uri"}
@pytest.mark.asyncio
async def test_async_table_to_pandas_descriptions_mode_omits_row_id():
db = await lancedb.connect_async("memory:///")
schema = pa.schema([pa.field("id", pa.int64()), lancedb.blob("image")])
table = await db.create_table("descriptions_no_leak_async", schema=schema)
await table.add([{"id": 1, "image": b"x"}])
df = await table.to_pandas(blob_mode="descriptions")
descriptor = df["image"].iloc[0]
assert "_lance_row_id" not in descriptor
assert set(descriptor.keys()) == {"kind", "position", "size", "blob_id", "blob_uri"}
def test_fetch_blobs_round_trip():
table = _blob_table(
"round_trip",
[{"id": 1, "image": b"alpha"}, {"id": 2, "image": b"beta"}],
)
by_id = _row_ids_by_id(table)
blobs = table.fetch_blobs("image", [by_id[1], by_id[2]])
assert [blobs[0].as_py(), blobs[1].as_py()] == [b"alpha", b"beta"]
def test_fetch_blobs_accepts_query_result():
table = _blob_table("from_result", [{"id": 1, "image": b"gamma"}])
hits = table.search().limit(10).to_arrow()
assert "_rowid" not in hits.column_names
blobs = table.fetch_blobs("image", hits)
assert {blobs[i].as_py() for i in range(len(blobs))} == {b"gamma"}
def test_fetch_blobs_null_alignment():
table = _blob_table(
"nulls",
[{"id": 1, "image": b"present"}, {"id": 2, "image": None}],
)
by_id = _row_ids_by_id(table)
request = [by_id[1], by_id[2], by_id[1]]
blobs = table.fetch_blobs("image", request)
assert len(blobs) == len(request)
assert blobs[0].as_py() == b"present"
assert blobs[1].as_py() is None
assert blobs[2].as_py() == b"present"
def test_fetch_blobs_nested_path():
db = lancedb.connect("memory:///")
info = pa.StructArray.from_arrays(
[
pa.array(["first", "second"], type=pa.string()),
_blob_array("blob", [b"nested-alpha", b"nested-beta"]),
],
names=["name", "blob"],
)
data = pa.Table.from_arrays(
[pa.array([1, 2], type=pa.int64()), info],
names=["id", "info"],
)
table = db.create_table("nested", data=data)
by_id = _row_ids_by_id(table)
blobs = table.fetch_blobs("info.blob", [by_id[1], by_id[2]])
assert [blobs[0].as_py(), blobs[1].as_py()] == [b"nested-alpha", b"nested-beta"]
def test_fetch_blob_files_lazy_read():
payload = b"lazy-read" * 100
table = _blob_table("lazy", [{"id": 1, "image": payload}])
by_id = _row_ids_by_id(table)
handles = table.fetch_blob_files("image", [by_id[1]])
assert len(handles) == 1
assert handles[0].read() == payload
def test_fetch_blob_files_null_alignment():
table = _blob_table(
"lazy_nulls",
[{"id": 1, "image": b"here"}, {"id": 2, "image": None}],
)
by_id = _row_ids_by_id(table)
handles = table.fetch_blob_files("image", [by_id[2], by_id[1]])
assert len(handles) == 2
assert handles[0] is None
assert handles[1].read() == b"here"
def test_fetch_blobs_rejects_non_blob_column():
table = _blob_table("reject", [{"id": 1, "image": b"x"}])
with pytest.raises(ValueError, match="not a blob column"):
table.fetch_blobs("id", [0])
def test_legacy_v1_query_omits_auto_row_id():
table = _legacy_v1_table("legacy_v1")
hits = table.search().select(["legacy"]).limit(10).to_arrow()
assert "_rowid" not in hits.column_names
def test_fetch_blobs_rejects_legacy_v1_column():
table = _legacy_v1_table("legacy_fetch")
with pytest.raises(ValueError, match="legacy blob column.*blob v2"):
table.fetch_blobs("legacy", [0])
@pytest.mark.asyncio
async def test_async_fetch_blob_files_lazy_read():
db = await lancedb.connect_async("memory:///")
schema = pa.schema([pa.field("id", pa.int64()), lancedb.blob("image")])
table = await db.create_table("async_lazy", schema=schema)
payload = b"async-lazy" * 100
await table.add([{"id": 1, "image": payload}])
hits = (
await table.query().select({"image_alias": "image"}).limit(10).to_arrow()
).combine_chunks()
assert "_rowid" not in hits.column_names
handles = await table.fetch_blob_files("image", hits)
assert len(handles) == 1
assert await handles[0].aread() == payload
def test_fetch_blobs_from_query_result_without_row_id_raises():
table = _blob_table("no_rowid", [{"id": 1, "image": b"x"}])
hits = table.search().select(["id"]).to_arrow()
assert "_rowid" not in hits.column_names
with pytest.raises(ValueError, match="_rowid"):
table.fetch_blobs("image", hits)
_HYBRID_BLOB_SCHEMA = pa.schema(
[
pa.field("id", pa.int64()),
pa.field("text", pa.utf8()),
pa.field("vector", pa.list_(pa.float32(), list_size=2)),
lancedb.blob("image"),
]
)
_HYBRID_BLOB_ROWS = [
{"id": 1, "text": "hello alpha", "vector": [1.0, 0.0], "image": b"alpha"},
{"id": 2, "text": "hello beta", "vector": [0.9, 0.1], "image": b"beta"},
{"id": 3, "text": "other", "vector": [0.0, 1.0], "image": b"other"},
]
def _hybrid_blob_table(db):
table = db.create_table("hybrid_blob_fetch", schema=_HYBRID_BLOB_SCHEMA)
table.add(_HYBRID_BLOB_ROWS)
table.create_index("text", config=FTS(with_position=False))
return table
async def _hybrid_blob_table_async(db):
table = await db.create_table("hybrid_blob_fetch_async", schema=_HYBRID_BLOB_SCHEMA)
await table.add(_HYBRID_BLOB_ROWS)
await table.create_index("text", config=FTS(with_position=False))
return table
def test_blob_v2_hybrid_fetch_blobs():
table = _hybrid_blob_table(lancedb.connect("memory:///"))
hits = (
table.search(query_type="hybrid")
.vector([1.0, 0.0])
.text("hello")
.select(["id", "image"])
.limit(2)
.to_arrow()
)
assert "_rowid" not in hits.column_names
assert "_lance_row_id" in hits.schema.field("image").type.names
blobs = table.fetch_blobs("image", hits)
assert {blobs[i].as_py() for i in range(len(blobs))} == {b"alpha", b"beta"}
@pytest.mark.asyncio
async def test_blob_v2_hybrid_fetch_blobs_async():
db = await lancedb.connect_async("memory:///hybrid_blob_fetch_async")
table = await _hybrid_blob_table_async(db)
hits = await (
table.query()
.nearest_to([1.0, 0.0])
.nearest_to_text("hello")
.select(["id", "image"])
.limit(2)
.to_arrow()
)
assert "_rowid" not in hits.column_names
assert "_lance_row_id" in hits.schema.field("image").type.names
blobs = await table.fetch_blobs("image", hits)
assert {blobs[i].as_py() for i in range(len(blobs))} == {b"alpha", b"beta"}
def test_blob_file_seek_read_and_read_range():
payload = _identifiable_payload(1024)
table = _blob_table("seek_read", [{"id": 1, "image": payload}])
by_id = _row_ids_by_id(table)
handle = table.fetch_blob_files("image", [by_id[1]])[0]
assert handle.seek(100) == 100
assert handle.read(16) == payload[100:116]
handle.seek(100)
assert handle.read_range(500, 8) == payload[500:508]
assert handle.tell() == 100
with pytest.raises(ValueError, match="whence"):
handle.seek(0, 99)
def test_fetch_blob_files_from_query_partial_read():
payload = _identifiable_payload(65536)
table = _blob_table("query_partial", [{"id": 1, "image": payload}])
hits = table.search().select(["id", "image"]).limit(1).to_arrow()
assert "_rowid" not in hits.column_names
handle = table.fetch_blob_files("image", hits)[0]
assert handle.size() == 65536
assert handle.read_range(0, 128) == payload[:128]
assert handle.tell() == 0
assert handle.seek(40000) == 40000
assert handle.read(16) == payload[40000:40016]
def test_blob_file_buffered_reader():
payload = _identifiable_payload(4096)
table = _blob_table("buffered_reader", [{"id": 1, "image": payload}])
hits = table.search().select(["id", "image"]).limit(1).to_arrow()
handle = table.fetch_blob_files("image", hits)[0]
reader = io.BufferedReader(handle)
assert reader.read(8) == payload[:8]
assert reader.read(8) == payload[8:16]
assert reader.read() == payload[16:]
def test_fetch_blob_files_cross_fragment_nulls_and_dups():
db = lancedb.connect("memory:///")
schema = pa.schema([pa.field("id", pa.int64()), lancedb.blob("image")])
table = db.create_table("cross_fragment", schema=schema)
table.add([{"id": 1, "image": b"alpha"}])
table.add([{"id": 2, "image": None}, {"id": 3, "image": b"beta"}])
by_id = _row_ids_by_id(table)
request = [by_id[3], by_id[2], by_id[1], by_id[3]]
handles = table.fetch_blob_files("image", request)
assert len(handles) == 4
assert handles[1] is None
assert handles[0].read() == b"beta"
assert handles[2].read() == b"alpha"
assert handles[3].seek(1) == 1
assert handles[3].read() == b"eta"
def test_blob_file_pyav_decode_seek(tmp_path):
av = pytest.importorskip("av")
import fractions
clip = tmp_path / "clip.mp4"
with av.open(str(clip), mode="w") as container:
stream = container.add_stream("mpeg4", rate=5)
stream.width, stream.height, stream.pix_fmt = 32, 32, "yuv420p"
stream.time_base = fractions.Fraction(1, 5)
for pts in range(5):
frame = av.VideoFrame(32, 32, "yuv420p")
frame.pts = pts
container.mux(stream.encode(frame))
container.mux(stream.encode(None))
table = _blob_table("pyav", [{"id": 1, "image": clip.read_bytes()}])
hits = table.search().select(["image"]).limit(1).to_arrow()
handle = table.fetch_blob_files("image", hits)[0]
with av.open(handle) as container:
stream = container.streams.video[0]
container.seek(0)
assert next(container.decode(stream)) is not None
def test_blob_v2_hybrid_fetch_blob_files_seek():
table = _hybrid_blob_table(lancedb.connect("memory:///"))
hits = (
table.search(query_type="hybrid")
.vector([1.0, 0.0])
.text("hello")
.select(["id", "image"])
.limit(2)
.to_arrow()
)
assert "_rowid" not in hits.column_names
handles = table.fetch_blob_files("image", hits)
assert len(handles) == 2
assert {handle.read_range(0, 2) for handle in handles} == {b"al", b"be"}
first = handles[0]
assert first.seek(1) == 1
assert first.read(2) in {b"lp", b"et"}
def test_blob_file_header_sniff_from_search():
payload = b"%PDF-1.7\n" + bytes(4096)
table = _blob_table("header_sniff", [{"id": 1, "image": payload}])
hits = table.search().select(["id", "image"]).limit(1).to_arrow()
handle = table.fetch_blob_files("image", hits)[0]
assert handle.read_range(0, 4) == b"%PDF"
assert handle.tell() == 0
def test_blob_file_multiple_handles_independent_cursors():
table = _blob_table(
"multi_handle",
[{"id": 1, "image": b"first-payload"}, {"id": 2, "image": b"second-payload"}],
)
by_id = _row_ids_by_id(table)
first, second = table.fetch_blob_files("image", [by_id[1], by_id[2]])
assert first.seek(6) == 6
assert second.tell() == 0
assert first.read(7) == b"payload"
assert second.read(6) == b"second"
def test_fetch_blob_files_nested_path_seek():
db = lancedb.connect("memory:///")
info = pa.StructArray.from_arrays(
[
pa.array(["first", "second"], type=pa.string()),
_blob_array("blob", [b"nested-alpha", b"nested-beta"]),
],
names=["name", "blob"],
)
data = pa.Table.from_arrays(
[pa.array([1, 2], type=pa.int64()), info],
names=["id", "info"],
)
table = db.create_table("nested_seek", data=data)
by_id = _row_ids_by_id(table)
handle = table.fetch_blob_files("info.blob", [by_id[2]])[0]
assert handle.seek(7) == 7
assert handle.read() == b"beta"
def test_fetch_blobs_survives_sort_after_query():
table = _blob_table(
"sort_survives",
[{"id": i, "image": f"payload-{i}".encode()} for i in range(5)],
)
hits = table.search().select(["id", "image"]).to_arrow()
sort_idx = pc.sort_indices(hits["id"], sort_keys=[("id", "descending")])
sorted_hits = hits.take(sort_idx)
blobs = table.fetch_blobs("image", sorted_hits)
expected = [f"payload-{i}".encode() for i in sorted_hits["id"].to_pylist()]
assert [blobs[i].as_py() for i in range(len(blobs))] == expected
def test_fetch_blobs_survives_filter_and_sort_after_query():
table = _blob_table(
"filter_sort_survives",
[{"id": i, "image": f"payload-{i}".encode()} for i in range(5)],
)
hits = table.search().select(["id", "image"]).to_arrow()
filtered = hits.filter(pc.field("id") >= 2)
sort_idx = pc.sort_indices(filtered["id"], sort_keys=[("id", "descending")])
filtered_sorted = filtered.take(sort_idx)
blobs = table.fetch_blobs("image", filtered_sorted)
expected = [f"payload-{i}".encode() for i in filtered_sorted["id"].to_pylist()]
assert [blobs[i].as_py() for i in range(len(blobs))] == expected
def test_fetch_blob_files_survives_sort_after_query():
table = _blob_table(
"lazy_sort_survives",
[{"id": i, "image": f"payload-{i}".encode()} for i in range(5)],
)
hits = table.search().select(["id", "image"]).to_arrow()
sort_idx = pc.sort_indices(hits["id"], sort_keys=[("id", "descending")])
sorted_hits = hits.take(sort_idx)
handles = table.fetch_blob_files("image", sorted_hits)
expected = [f"payload-{i}".encode() for i in sorted_hits["id"].to_pylist()]
assert [handle.read() for handle in handles] == expected
def test_fetch_blobs_nested_path_survives_sort_after_query():
db = lancedb.connect("memory:///")
values = [f"payload-{i}".encode() for i in range(4)]
info = pa.StructArray.from_arrays(
[pa.array(["row"] * 4, type=pa.string()), _blob_array("blob", values)],
names=["name", "blob"],
)
data = pa.Table.from_arrays(
[pa.array(range(4), type=pa.int64()), info],
names=["id", "info"],
)
table = db.create_table("nested_sort_survives", data=data)
hits = table.search().to_arrow()
sort_idx = pc.sort_indices(hits["id"], sort_keys=[("id", "descending")])
sorted_hits = hits.take(sort_idx)
blobs = table.fetch_blobs("info.blob", sorted_hits)
expected = [f"payload-{i}".encode() for i in sorted_hits["id"].to_pylist()]
assert [blobs[i].as_py() for i in range(len(blobs))] == expected
def _identifiable_payload(size: int) -> bytes:
block = 256
return b"".join(bytes([i % 256]) * block for i in range(size // block))
+169
View File
@@ -786,6 +786,97 @@ def test_language(mem_db: DBConnection):
assert len(results) == 0
def test_tokenize_uses_simple_index_tokenizer(mem_db: DBConnection):
data = pa.table({"text": ["Running in cafés"], "other": ["Running in cafés"]})
table = mem_db.create_table("test_tokenize", data=data)
table.create_index("text", config=FTS(base_tokenizer="simple"))
tokens = table.tokenize("Running in cafés", column="text")
assert [(token.text, token.position) for token in tokens] == [
("run", 0),
("cafe", 2),
]
def test_tokenize_uses_icu_index_tokenizer_by_name(mem_db: DBConnection):
data = pa.table({"text": ["Hello, こんにちは世界!"]})
table = mem_db.create_table("test_tokenize_icu", data=data)
table.create_index(
"text",
config=FTS(
base_tokenizer="icu",
stem=False,
remove_stop_words=False,
),
name="text_icu_idx",
)
tokens = table.tokenize("Hello, こんにちは世界!", index_name="text_icu_idx")
assert [(token.text, token.position) for token in tokens] == [
("hello", 0),
("こんにちは", 1),
("世界", 2),
]
def test_tokenize_requires_one_selector(mem_db: DBConnection):
data = pa.table({"text": ["hello world"]})
table = mem_db.create_table("test_tokenize_selector", data=data)
table.create_index("text", config=FTS(), name="text_idx")
with pytest.raises(ValueError, match="Specify exactly one"):
table.tokenize("hello")
with pytest.raises(ValueError, match="Specify exactly one"):
table.tokenize("hello", column="text", index_name="text_idx")
def test_tokenize_requires_fts_index(mem_db: DBConnection):
data = pa.table({"text": ["hello world"]})
table = mem_db.create_table("test_tokenize_no_index", data=data)
with pytest.raises(ValueError, match="does not have a full text search index"):
table.tokenize("hello", column="text")
@pytest.mark.asyncio
async def test_tokenize_async(async_table):
await async_table.create_index("text", config=FTS())
tokens = await async_table.tokenize("Running in cafés", column="text")
assert [(token.text, token.position) for token in tokens] == [
("run", 0),
("cafe", 2),
]
def test_tokenize_uses_explicit_simple_tokenizer():
tokens = ldb.tokenize("Running in cafés", base_tokenizer="simple")
assert [(token.text, token.position) for token in tokens] == [
("run", 0),
("cafe", 2),
]
def test_tokenize_uses_explicit_icu_tokenizer():
tokens = ldb.tokenize(
"Hello, こんにちは世界!",
base_tokenizer="icu",
stem=False,
remove_stop_words=False,
)
assert [(token.text, token.position) for token in tokens] == [
("hello", 0),
("こんにちは", 1),
("世界", 2),
]
def test_fts_on_list(mem_db: DBConnection):
data = pa.table(
{
@@ -1084,6 +1175,84 @@ def test_fts_query_to_json():
assert json_str == expected
def test_fts_phrase_query_is_preserved_in_query_object():
query = LanceFtsQueryBuilder(mock.Mock(), "puppy runs").phrase_query()
query_object = query.to_query_object()
assert query_object.full_text_query.query == '"puppy runs"'
def test_fts_phrase_query_execution_preserves_user_text():
table = mock.Mock()
table.schema = pa.schema([])
table._execute_query.return_value = pa.table({"text": ["result"]}).to_reader()
class CapturingReranker:
score = "relevance"
def __init__(self):
self.queries = []
def rerank_fts(self, query, results):
self.queries.append(query)
return results.append_column("_relevance_score", [[1.0]])
reranker = CapturingReranker()
query = (
LanceFtsQueryBuilder(table, "puppy runs")
.phrase_query()
.with_row_id(False)
.rerank(reranker)
)
query.to_arrow()
backend_query = table._execute_query.call_args.args[0]
assert (
backend_query.full_text_query.query,
reranker.queries,
query._query,
) == ('"puppy runs"', ["puppy runs"], "puppy runs")
def test_fts_phrase_query_false_preserves_string():
query = LanceFtsQueryBuilder(mock.Mock(), "puppy runs").phrase_query(False)
query_object = query.to_query_object()
assert query_object.full_text_query.query == "puppy runs"
def test_fts_phrase_query_preserves_fully_quoted_string():
query = LanceFtsQueryBuilder(mock.Mock(), '"puppy runs"').phrase_query()
query_object = query.to_query_object()
assert query_object.full_text_query.query == '"puppy runs"'
def test_fts_phrase_query_preserves_structured_phrase_query():
phrase_query = PhraseQuery("puppy runs", "text")
query = LanceFtsQueryBuilder(mock.Mock(), phrase_query).phrase_query()
query_object = query.to_query_object()
assert query_object.full_text_query.query == phrase_query
def test_fts_phrase_query_rejects_other_structured_queries():
query = LanceFtsQueryBuilder(
mock.Mock(), MatchQuery("puppy", "text")
).phrase_query()
with pytest.raises(
TypeError,
match=r"phrase_query\(\) requires a string or PhraseQuery, got MatchQuery",
):
query.to_query_object()
def test_fts_fast_search(table):
table.create_fts_index("text")
+183
View File
@@ -0,0 +1,183 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
"""Unit tests for GeminiText embedding function."""
import sys
from unittest.mock import MagicMock, patch
# Mock google.genai modules before they are imported by gemini_text.py
mock_google = MagicMock()
mock_genai = MagicMock()
mock_types = MagicMock()
mock_google.genai = mock_genai
mock_genai.types = mock_types
sys.modules["google"] = mock_google
sys.modules["google.genai"] = mock_genai
sys.modules["google.genai.types"] = mock_types
import pytest # noqa: E402
import numpy as np # noqa: E402
from lancedb.embeddings import get_registry # noqa: E402
from lancedb import __version__ # noqa: E402
class TestGeminiText:
"""Tests for GeminiText model registration, configuration, and execution."""
@pytest.fixture(autouse=True)
def setup_mocks(self):
"""Set up standard mocks for google-genai Client and Config."""
# Reset mocks
mock_genai.reset_mock()
mock_types.reset_mock()
self.mock_client = MagicMock()
mock_genai.Client.return_value = self.mock_client
# Mock response for embed_content
self.mock_embedding_1 = MagicMock()
self.mock_embedding_1.values = [0.1] * 768
self.mock_embedding_2 = MagicMock()
self.mock_embedding_2.values = [0.2] * 768
self.mock_response = MagicMock()
self.mock_response.embeddings = [self.mock_embedding_1, self.mock_embedding_2]
self.mock_client.models.embed_content.return_value = self.mock_response
def test_gemini_registered(self):
"""Test that gemini-text is registered in the embedding function registry."""
registry = get_registry()
assert registry.get("gemini-text") is not None
def test_client_init_headers(self):
"""Test that Client is initialized with the partner-attribution header."""
with patch.dict("os.environ", {"GOOGLE_API_KEY": "test-key"}):
with patch("lancedb.embeddings.gemini_text.attempt_import_or_raise"):
registry = get_registry()
func = registry.get("gemini-text").create()
# Access the client property to trigger initialization
_ = func.client
mock_genai.Client.assert_called_once_with(
api_key="test-key",
http_options={
"headers": {
"x-goog-api-client": f"lancedb/{__version__}",
}
},
)
def test_generate_embeddings_batched(self):
"""Test that multiple texts are sent in a single batched API request."""
with patch.dict("os.environ", {"GOOGLE_API_KEY": "test-key"}):
with patch("lancedb.embeddings.gemini_text.attempt_import_or_raise"):
registry = get_registry()
func = registry.get("gemini-text").create()
texts = ["hello", "world"]
embeddings = func.generate_embeddings(texts)
# Check embed_content was called exactly once
self.mock_client.models.embed_content.assert_called_once()
# Verify call arguments
call_kwargs = self.mock_client.models.embed_content.call_args.kwargs
assert call_kwargs["model"] == "gemini-embedding-001"
assert len(call_kwargs["contents"]) == 2
assert call_kwargs["contents"][0] == {"parts": [{"text": "hello"}]}
assert call_kwargs["contents"][1] == {"parts": [{"text": "world"}]}
# Verify returns are correct numpy arrays
assert len(embeddings) == 2
assert isinstance(embeddings[0], np.ndarray)
assert embeddings[0].shape == (768,)
assert np.allclose(embeddings[0], 0.1)
assert np.allclose(embeddings[1], 0.2)
def test_generate_embeddings_retrieval_document(self):
"""Test that retrieval_document task type prepends the document title part."""
with patch.dict("os.environ", {"GOOGLE_API_KEY": "test-key"}):
with patch("lancedb.embeddings.gemini_text.attempt_import_or_raise"):
registry = get_registry()
func = registry.get("gemini-text").create(
source_task_type="retrieval_document"
)
texts = ["doc text"]
# We need mock to return only 1 embedding since we only pass 1 text
mock_embedding = MagicMock()
mock_embedding.values = [0.3] * 768
self.mock_response.embeddings = [mock_embedding]
embeddings = func.generate_embeddings(
texts, task_type="retrieval_document"
)
# Check call arguments for retrieval_document
call_kwargs = self.mock_client.models.embed_content.call_args.kwargs
assert call_kwargs["contents"][0] == {
"parts": [{"text": "Embedding of a document"}, {"text": "doc text"}]
}
mock_types.EmbedContentConfig.assert_called_once_with(
output_dimensionality=768, task_type="RETRIEVAL_DOCUMENT"
)
assert len(embeddings) == 1
assert np.allclose(embeddings[0], 0.3)
def test_custom_dimension(self):
"""Test that custom dimension (dim) can be configured and passed to config."""
with patch.dict("os.environ", {"GOOGLE_API_KEY": "test-key"}):
with patch("lancedb.embeddings.gemini_text.attempt_import_or_raise"):
registry = get_registry()
func = registry.get("gemini-text").create(dim=3072)
assert func.ndims() == 3072
texts = ["hello"]
mock_embedding = MagicMock()
mock_embedding.values = [0.5] * 3072
self.mock_response.embeddings = [mock_embedding]
_ = func.generate_embeddings(texts)
mock_types.EmbedContentConfig.assert_called_once_with(
output_dimensionality=3072
)
def test_generate_embeddings_chunked(self):
"""Test that generate_embeddings chunks texts into groups of 100."""
with patch.dict("os.environ", {"GOOGLE_API_KEY": "test-key"}):
with patch("lancedb.embeddings.gemini_text.attempt_import_or_raise"):
registry = get_registry()
func = registry.get("gemini-text").create()
# Passing 250 texts should make 3 calls (100, 100, 50)
texts = [f"text_{i}" for i in range(250)]
# Mock client response to return correct number of embeddings per chunk
def mock_embed_side_effect(model, contents, config=None):
mock_resp = MagicMock()
mock_embeddings = []
for _ in contents:
emb = MagicMock()
# Each embedding is length 768
emb.values = [0.1] * 768
mock_embeddings.append(emb)
mock_resp.embeddings = mock_embeddings
return mock_resp
self.mock_client.models.embed_content.side_effect = (
mock_embed_side_effect
)
embeddings = func.generate_embeddings(texts)
# embed_content should be called 3 times
assert self.mock_client.models.embed_content.call_count == 3
assert len(embeddings) == 250
+33
View File
@@ -1,6 +1,8 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
from unittest import mock
import lancedb
from lancedb.query import LanceHybridQueryBuilder
@@ -139,6 +141,20 @@ def test_hybrid_query_distance_range(sync_table: Table):
assert 0.2 <= dist.as_py() <= 0.5
def test_hybrid_query_applies_zero_upper_distance_bound(sync_table: Table):
result = (
sync_table.search(query_type="hybrid")
.vector([0.0, 0.4])
.text("elephant")
.distance_range(upper_bound=0.0)
.rerank(RRFReranker(return_score="all"))
.limit(4)
.to_arrow()
)
assert len(result) == 0
@pytest.mark.asyncio
async def test_hybrid_query_distance_range_async(table: AsyncTable):
reranker = RRFReranker(return_score="all")
@@ -177,6 +193,23 @@ async def test_analyze_plan(table: AsyncTable):
assert "metrics=" in res
def test_hybrid_phrase_query_is_preserved_in_analyze_plan():
table = mock.Mock()
analyzed_queries = []
table._analyze_plan.side_effect = lambda query: analyzed_queries.append(query) or ""
(
LanceHybridQueryBuilder(table)
.vector([0.1, 0.2])
.text("puppy runs")
.phrase_query()
.analyze_plan()
)
assert len(analyzed_queries) == 2
assert analyzed_queries[1].full_text_query.query == '"puppy runs"'
@pytest.fixture
def table_with_id(tmpdir_factory) -> Table:
tmp_path = str(tmpdir_factory.mktemp("data"))
+140 -18
View File
@@ -11,6 +11,7 @@ import lancedb
from lancedb.db import AsyncConnection
from lancedb.embeddings.base import TextEmbeddingFunction
from lancedb.embeddings.registry import get_registry, register
from lancedb.expr import col
from lancedb.index import FTS, IvfPq
import lancedb.pydantic
import numpy as np
@@ -63,11 +64,71 @@ def _blob_query_data():
)
def _create_blob_v2_query_table(db, name):
schema = pa.schema(
[
pa.field("id", pa.int64()),
pa.field("tag", pa.utf8()),
pa.field("vector", pa.list_(pa.float32(), list_size=2)),
lancedb.blob("blob"),
]
)
table = db.create_table(name, schema=schema)
table.add(
[
{"id": 1, "tag": "drop", "vector": [1.0, 0.0], "blob": b"one"},
{"id": 2, "tag": "keep", "vector": [2.0, 0.0], "blob": b"two"},
{"id": 3, "tag": "keep", "vector": [3.0, 0.0], "blob": b"three"},
{"id": 4, "tag": "keep", "vector": [4.0, 0.0], "blob": b"four"},
]
)
return table
async def _create_blob_v2_query_table_async(db, name):
schema = pa.schema(
[
pa.field("id", pa.int64()),
pa.field("tag", pa.utf8()),
pa.field("vector", pa.list_(pa.float32(), list_size=2)),
lancedb.blob("blob"),
]
)
table = await db.create_table(name, schema=schema)
await table.add(
[
{"id": 1, "tag": "drop", "vector": [1.0, 0.0], "blob": b"one"},
{"id": 2, "tag": "keep", "vector": [2.0, 0.0], "blob": b"two"},
{"id": 3, "tag": "keep", "vector": [3.0, 0.0], "blob": b"three"},
{"id": 4, "tag": "keep", "vector": [4.0, 0.0], "blob": b"four"},
]
)
return table
def _assert_lazy_blob(value, expected: bytes):
assert hasattr(value, "readall")
assert value.readall() == expected
def _assert_blob_bytes_projection(df):
assert df["id_alias"].tolist() == [3, 4]
assert df["payload"].tolist() == [b"three", b"four"]
assert df["double_id"].tolist() == [6, 8]
def _blob_query_table(db, name, blob_schema):
if blob_schema == "v1":
return db.create_table(name, _blob_query_data())
return _create_blob_v2_query_table(db, name)
async def _blob_query_table_async(db, name, blob_schema):
if blob_schema == "v1":
return await db.create_table(name, _blob_query_data())
return await _create_blob_v2_query_table_async(db, name)
@pytest.fixture(scope="module")
def table(tmpdir_factory) -> lancedb.table.Table:
tmp_path = str(tmpdir_factory.mktemp("data"))
@@ -235,10 +296,11 @@ def test_plain_scan_query_to_pandas_blob_modes(tmp_db, blob_mode):
assert not hasattr(first, "readall")
def test_plain_scan_query_to_pandas_blob_projection(tmp_db):
@pytest.mark.parametrize("blob_schema", ["v1", "v2"])
def test_plain_scan_query_to_pandas_blob_bytes_projection(tmp_db, blob_schema):
pytest.importorskip("lance")
table = tmp_db.create_table(
"test_query_to_pandas_blob_projection", _blob_query_data()
table = _blob_query_table(
tmp_db, f"test_query_to_pandas_blob_{blob_schema}_bytes", blob_schema
)
df = (
@@ -250,9 +312,8 @@ def test_plain_scan_query_to_pandas_blob_projection(tmp_db):
.to_pandas(blob_mode="bytes")
)
assert df["id_alias"].tolist() == [3, 4]
assert df["payload"].tolist() == [b"three", b"four"]
assert df["double_id"].tolist() == [6, 8]
_assert_blob_bytes_projection(df)
assert "_rowid" not in df.columns
@pytest.mark.parametrize("blob_mode", ["bytes", "descriptions"])
@@ -348,18 +409,6 @@ async def test_async_plain_scan_query_to_pandas_blob_projection(tmp_db_async):
assert lazy_df["id"].tolist() == [1]
_assert_lazy_blob(lazy_df["blob"].iloc[0], b"one")
bytes_df = await (
table.query()
.where("id >= 2")
.select({"id_alias": "id", "payload": "blob", "double_id": "id * 2"})
.limit(2)
.offset(1)
.to_pandas(blob_mode="bytes")
)
assert bytes_df["id_alias"].tolist() == [3, 4]
assert bytes_df["payload"].tolist() == [b"three", b"four"]
assert bytes_df["double_id"].tolist() == [6, 8]
desc_df = await (
table.query()
.where("id = 1")
@@ -371,6 +420,31 @@ async def test_async_plain_scan_query_to_pandas_blob_projection(tmp_db_async):
assert not hasattr(first, "readall")
@pytest.mark.asyncio
@pytest.mark.parametrize("blob_schema", ["v1", "v2"])
async def test_async_plain_scan_query_to_pandas_blob_bytes_projection(
tmp_db_async, blob_schema
):
pytest.importorskip("lance")
table = await _blob_query_table_async(
tmp_db_async,
f"test_async_query_to_pandas_blob_{blob_schema}_bytes",
blob_schema,
)
df = await (
table.query()
.where("id >= 2")
.select({"id_alias": "id", "payload": "blob", "double_id": "id * 2"})
.limit(2)
.offset(1)
.to_pandas(blob_mode="bytes")
)
_assert_blob_bytes_projection(df)
assert "_rowid" not in df.columns
@pytest.mark.asyncio
@pytest.mark.parametrize("blob_mode", ["bytes", "descriptions"])
async def test_async_plain_scan_query_to_pandas_blob_mode_does_not_collect_arrow(
@@ -502,6 +576,18 @@ def test_with_row_id(table: lancedb.table.Table):
assert rs["_rowid"].to_pylist() == [0, 1]
def test_blob_v2_query_omits_auto_row_id(tmp_db):
table = _create_blob_v2_query_table(tmp_db, "test_blob_v2_omits_auto_rowid")
query_obj = table.search().select(["id", "blob"]).limit(2).to_query_object()
assert query_obj.with_row_id is None
rs = table.search().select(["id", "blob"]).limit(2).to_arrow()
assert "_rowid" not in rs.column_names
assert rs["id"].to_pylist() == [1, 2]
def test_where_repeated_combines_with_and(table: lancedb.table.Table):
# Calling where() more than once should AND the filters together instead of
# silently replacing the previous one (regression test for #2649).
@@ -1946,3 +2032,39 @@ def test_fast_search(tmp_path):
# 2. Fast Search -> Should NOT include "LanceScan" (Uses Index)
plan = table.search(q).fast_search().explain_plan(True)
assert "LanceScan" not in plan
def test_blob_v2_with_row_id_bytes_pandas(tmp_db):
table = _create_blob_v2_query_table(tmp_db, "test_blob_v2_rowid_bytes_pandas")
df = (
table.search()
.with_row_id(True)
.select(["id", "blob"])
.to_pandas(blob_mode="bytes")
)
assert "_rowid" in df.columns
assert df["id"].tolist() == [1, 2, 3, 4]
assert df["blob"].tolist() == [b"one", b"two", b"three", b"four"]
def test_blob_v2_expr_projection_stash(tmp_db):
table = _create_blob_v2_query_table(tmp_db, "test_blob_v2_expr_projection_stash")
hits = table.search().select({"blob_alias": col("blob")}).limit(2).to_arrow()
assert "_rowid" not in hits.column_names
assert "_lance_row_id" in hits.schema.field("blob_alias").type.names
blobs = table.fetch_blobs("blob", hits)
assert [blobs[i].as_py() for i in range(len(blobs))] == [b"one", b"two"]
def test_blob_v2_to_batches_row_id(tmp_db):
table = _create_blob_v2_query_table(tmp_db, "test_blob_v2_to_batches_rowid")
hits = table.search().select(["id", "blob"]).limit(2).to_batches().read_all()
assert "_rowid" in hits.column_names
blobs = table.fetch_blobs("blob", hits)
assert [blobs[i].as_py() for i in range(len(blobs))] == [b"one", b"two"]
+17
View File
@@ -23,6 +23,7 @@ from lancedb.rerankers import (
AnswerdotaiRerankers,
VoyageAIReranker,
MRRReranker,
WatsonxReranker,
)
from lancedb.table import LanceTable
@@ -727,3 +728,19 @@ def test_linear_combination_missing_fts_is_penalised():
f"Document with FTS score (rowid 0, {scores[0]:.4f}) should beat "
f"document with no FTS match (rowid 1, {scores[1]:.4f})"
)
@pytest.mark.skipif(
os.environ.get("WATSONX_API_KEY") is None
or (
os.environ.get("WATSONX_PROJECT_ID") is None
and os.environ.get("WATSONX_SPACE_ID") is None
),
reason="WATSONX_API_KEY and one of WATSONX_PROJECT_ID / "
"WATSONX_SPACE_ID must be set",
)
def test_watsonx_reranker(tmp_path):
pytest.importorskip("ibm_watsonx_ai")
table, schema = get_test_table(tmp_path)
reranker = WatsonxReranker()
_run_test_reranker(reranker, table, "single player experience", None, schema)
+70 -12
View File
@@ -45,6 +45,32 @@ def _blob_test_data():
)
def _blob_v2_table(db: DBConnection, name: str):
schema = pa.schema([pa.field("id", pa.int64()), lancedb.blob("blob")])
table = db.create_table(name, schema=schema)
table.add([{"id": 1, "blob": b"hello"}, {"id": 2, "blob": b"world"}])
return table
async def _blob_v2_table_async(db: AsyncConnection, name: str):
schema = pa.schema([pa.field("id", pa.int64()), lancedb.blob("blob")])
table = await db.create_table(name, schema=schema)
await table.add([{"id": 1, "blob": b"hello"}, {"id": 2, "blob": b"world"}])
return table
def _blob_table(db: DBConnection, name: str, blob_schema: str):
if blob_schema == "v1":
return db.create_table(name, data=_blob_test_data())
return _blob_v2_table(db, name)
async def _blob_table_async(db: AsyncConnection, name: str, blob_schema: str):
if blob_schema == "v1":
return await db.create_table(name, data=_blob_test_data())
return await _blob_v2_table_async(db, name)
def _assert_lazy_blob(value, expected: bytes):
assert hasattr(value, "readall")
assert value.readall() == expected
@@ -107,6 +133,18 @@ def test_table_to_pandas_blob_modes(tmp_db: DBConnection, blob_mode):
assert not hasattr(first, "readall")
@pytest.mark.parametrize("blob_schema", ["v1", "v2"])
def test_table_to_pandas_blob_bytes(tmp_db: DBConnection, blob_schema):
pytest.importorskip("lance")
table = _blob_table(tmp_db, f"test_to_pandas_blob_{blob_schema}_bytes", blob_schema)
df = table.to_pandas(blob_mode="bytes")
assert list(df.columns) == ["id", "blob"]
assert df["blob"].tolist() == [b"hello", b"world"]
assert "_rowid" not in df.columns
def test_table_to_pandas_kwargs(tmp_db: DBConnection):
pd = pytest.importorskip("pandas")
data = pa.table({"id": pa.array([1, 2], pa.int64())})
@@ -118,15 +156,20 @@ def test_table_to_pandas_kwargs(tmp_db: DBConnection):
@pytest.mark.asyncio
async def test_async_table_to_pandas_blob_bytes(tmp_db_async: AsyncConnection):
@pytest.mark.parametrize("blob_schema", ["v1", "v2"])
async def test_async_table_to_pandas_blob_bytes(
tmp_db_async: AsyncConnection, blob_schema
):
pytest.importorskip("lance")
table = await tmp_db_async.create_table(
"test_async_to_pandas_blob_bytes", data=_blob_test_data()
table = await _blob_table_async(
tmp_db_async, f"test_async_to_pandas_blob_{blob_schema}_bytes", blob_schema
)
df = await table.to_pandas(blob_mode="bytes")
assert list(df.columns) == ["id", "blob"]
assert df["blob"].tolist() == [b"hello", b"world"]
assert "_rowid" not in df.columns
@pytest.mark.asyncio
@@ -1568,16 +1611,23 @@ def test_create_with_nans(mem_db: DBConnection):
"fill_test",
data=[
{"vector": [3.1, 4.1], "item": "foo", "price": 10.0},
{"vector": [2.1, 4.1], "item": "foo", "price": 9.0},
{"vector": [np.nan], "item": "bar", "price": 20.0},
{"vector": [np.nan, np.nan], "item": "bar", "price": 20.0},
{"vector": [np.nan, 5.0], "item": "bar", "price": 21.0},
{"vector": [5], "item": "bar", "price": 22.0},
],
on_bad_vectors="fill",
fill_value=0.0,
)
assert len(table) == 3
assert len(table) == 5
arrow_tbl = table.search().where("item == 'bar'").to_arrow()
v = arrow_tbl["vector"].to_pylist()[0]
assert np.allclose(v, np.array([0.0, 0.0]))
filled_vectors = {
row["price"]: row["vector"]
for row in arrow_tbl.select(["price", "vector"]).to_pylist()
}
assert np.allclose(filled_vectors[20.0], np.array([0.0, 0.0]))
assert np.allclose(filled_vectors[21.0], np.array([0.0, 5.0]))
assert np.allclose(filled_vectors[22.0], np.array([5.0, 0.0]))
def test_add_with_nans(mem_db: DBConnection):
@@ -1620,15 +1670,21 @@ def test_add_with_nans(mem_db: DBConnection):
data=[
{"vector": [3.1, 4.1], "item": "foo", "price": 10.0},
{"vector": [np.nan], "item": "bar", "price": 20.0},
{"vector": [np.nan, np.nan], "item": "bar", "price": 20.0},
{"vector": [np.nan, 5.0], "item": "bar", "price": 21.0},
{"vector": [5], "item": "bar", "price": 22.0},
],
on_bad_vectors="fill",
fill_value=0.0,
)
assert len(table) == 3
assert len(table) == 4
arrow_tbl = table.search().where("item == 'bar'").to_arrow()
v = arrow_tbl["vector"].to_pylist()[0]
assert np.allclose(v, np.array([0.0, 0.0]))
filled_vectors = {
row["price"]: row["vector"]
for row in arrow_tbl.select(["price", "vector"]).to_pylist()
}
assert np.allclose(filled_vectors[20.0], np.array([0.0, 0.0]))
assert np.allclose(filled_vectors[21.0], np.array([0.0, 5.0]))
assert np.allclose(filled_vectors[22.0], np.array([5.0, 0.0]))
def test_add_with_empty_fixed_size_list_drops_bad_rows(mem_db: DBConnection):
@@ -1789,7 +1845,9 @@ def test_on_bad_vectors_fill_preserves_arrow_nested_vector_type(mem_db: DBConnec
fill_value=0.0,
)
assert table.to_arrow()["vector"].to_pylist() == [[1.0, 2.0], [0.0, 0.0]]
vector = table.to_arrow()["vector"]
assert vector.type == pa.list_(pa.float32())
assert vector.to_pylist() == [[1.0, 2.0], [0.0, 3.0]]
@pytest.mark.parametrize(
+47 -5
View File
@@ -13,6 +13,7 @@ from lancedb.embeddings.registry import EmbeddingFunctionRegistry
from lancedb.table import (
_append_vector_columns,
_cast_to_target_schema,
_fill_bad_vector_values,
_handle_bad_vectors,
_into_pyarrow_reader,
_infer_target_schema,
@@ -287,7 +288,9 @@ def test_append_vector_columns():
@pytest.mark.parametrize("on_bad_vectors", ["error", "drop", "fill", "null"])
def test_handle_bad_vectors_jagged(on_bad_vectors):
vector = pa.array([[1.0, 2.0], [3.0], [4.0, 5.0]])
vector = pa.array(
[[1.0, 2.0], [3.0], [4.0, 5.0], [6.0, 7.0, 8.0], [None, 9.0], None]
)
schema = pa.schema({"vector": pa.list_(pa.float64())})
data = pa.table({"vector": vector}, schema=schema)
@@ -313,15 +316,54 @@ def test_handle_bad_vectors_jagged(on_bad_vectors):
).read_all()
if on_bad_vectors == "drop":
expected = pa.array([[1.0, 2.0], [4.0, 5.0]])
expected = pa.array([[1.0, 2.0], [4.0, 5.0], [None, 9.0]])
elif on_bad_vectors == "fill":
expected = pa.array([[1.0, 2.0], [42.0, 42.0], [4.0, 5.0]])
expected = pa.array(
[
[1.0, 2.0],
[3.0, 42.0],
[4.0, 5.0],
[6.0, 7.0],
[None, 9.0],
[42.0, 42.0],
]
)
elif on_bad_vectors == "null":
expected = pa.array([[1.0, 2.0], None, [4.0, 5.0]])
expected = pa.array([[1.0, 2.0], None, [4.0, 5.0], None, [None, 9.0], None])
assert output["vector"].combine_chunks() == expected
@pytest.mark.parametrize(
("vector_type", "vectors", "expected"),
[
(
pa.list_(pa.float64()),
[[1.0, float("nan")], [2.0], None, [None, 3.0], [4.0, 5.0, 6.0]],
[[1.0, 42.0], [2.0, 42.0], [42.0, 42.0], [None, 3.0], [4.0, 5.0]],
),
(
pa.large_list(pa.float64()),
[[1.0, float("nan")], [2.0], None, [None, 3.0], [4.0, 5.0, 6.0]],
[[1.0, 42.0], [2.0, 42.0], [42.0, 42.0], [None, 3.0], [4.0, 5.0]],
),
(
pa.list_(pa.float64(), 2),
[[1.0, float("nan")], None, [None, 3.0]],
[[1.0, 42.0], [42.0, 42.0], [None, 3.0]],
),
],
)
def test_fill_bad_vector_values_arrow_types(vector_type, vectors, expected):
arr = pa.array([[0.0, 0.0], *vectors, [9.0, 9.0]], type=vector_type)
arr = arr.slice(1, len(vectors))
actual = _fill_bad_vector_values(arr, dim=2, fill_value=42.0)
assert actual.type == vector_type
assert actual.to_pylist() == expected
@pytest.mark.parametrize("on_bad_vectors", ["error", "drop", "fill", "null"])
def test_handle_bad_vectors_nan(on_bad_vectors):
vector = pa.array([[1.0, float("nan")], [3.0, 4.0]])
@@ -351,7 +393,7 @@ def test_handle_bad_vectors_nan(on_bad_vectors):
if on_bad_vectors == "drop":
expected = pa.array([[3.0, 4.0]])
elif on_bad_vectors == "fill":
expected = pa.array([[42.0, 42.0], [3.0, 4.0]])
expected = pa.array([[1.0, 42.0], [3.0, 4.0]])
elif on_bad_vectors == "null":
expected = pa.array([None, [3.0, 4.0]])
+5 -2
View File
@@ -15,8 +15,8 @@ use pyo3::{
use query::{FTSQuery, HybridQuery, Query, VectorQuery};
use session::Session;
use table::{
AddColumnsResult, AddResult, AlterColumnsResult, DeleteResult, DropColumnsResult, LsmWriteSpec,
MergeResult, Table, UpdateFieldMetadataResult, UpdateResult,
AddColumnsResult, AddResult, AlterColumnsResult, DeleteResult, DropColumnsResult, FtsToken,
LsmWriteSpec, MergeResult, PyBlobFile, Table, UpdateFieldMetadataResult, UpdateResult,
};
pub mod arrow;
@@ -44,6 +44,7 @@ pub fn _lancedb(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> {
m.add_class::<Connection>()?;
m.add_class::<Session>()?;
m.add_class::<Table>()?;
m.add_class::<PyBlobFile>()?;
m.add_class::<IndexConfig>()?;
m.add_class::<Query>()?;
m.add_class::<FTSQuery>()?;
@@ -59,6 +60,7 @@ pub fn _lancedb(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> {
m.add_class::<DeleteResult>()?;
m.add_class::<DropColumnsResult>()?;
m.add_class::<UpdateResult>()?;
m.add_class::<FtsToken>()?;
m.add_class::<PyAsyncPermutationBuilder>()?;
m.add_class::<PyPermutationReader>()?;
m.add_class::<PyExpr>()?;
@@ -74,6 +76,7 @@ pub fn _lancedb(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> {
m.add_function(wrap_pyfunction!(connect, m)?)?;
m.add_function(wrap_pyfunction!(connect_namespace, m)?)?;
m.add_function(wrap_pyfunction!(connect_namespace_client, m)?)?;
m.add_function(wrap_pyfunction!(table::tokenize, m)?)?;
m.add_function(wrap_pyfunction!(permutation::async_permutation_builder, m)?)?;
m.add_function(wrap_pyfunction!(util::validate_table_name, m)?)?;
m.add_function(wrap_pyfunction!(query::fts_query_to_json, m)?)?;
+225 -5
View File
@@ -2,7 +2,7 @@
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
use std::{collections::HashMap, sync::Arc};
use crate::runtime::future_into_py;
use crate::runtime::{block_on, future_into_py};
use crate::{
connection::Connection,
error::PythonErrorExt,
@@ -12,19 +12,23 @@ use crate::{
table::scannable::PyScannable,
};
use arrow::{
array::{Array, LargeBinaryArray},
datatypes::{DataType, Schema},
ffi_stream::ArrowArrayStreamReader,
pyarrow::{FromPyArrow, PyArrowType, ToPyArrow},
};
use lancedb::blob::BlobFile;
use lancedb::index::scalar::FtsIndexBuilder;
use lancedb::table::{
AddDataMode, ColumnAlteration, Duration, FieldMetadataUpdate, NewColumnTransform,
OptimizeAction, OptimizeOptions, Ref, Table as LanceDbTable,
AddDataMode, ColumnAlteration, Duration, FieldMetadataUpdate, FtsToken as LanceDbFtsToken,
NewColumnTransform, OptimizeAction, OptimizeOptions, Ref, Table as LanceDbTable,
};
use lancedb::tokenize as lancedb_tokenize;
use pyo3::{
Bound, FromPyObject, Py, PyAny, PyRef, PyResult, Python,
exceptions::{PyRuntimeError, PyValueError},
pyclass, pymethods,
types::{IntoPyDict, PyAnyMethods, PyDict, PyDictMethods},
pyclass, pyfunction, pymethods,
types::{IntoPyDict, PyAnyMethods, PyBytes, PyDict, PyDictMethods},
};
mod scannable;
@@ -412,6 +416,150 @@ impl From<lancedb::table::DropColumnsResult> for DropColumnsResult {
}
}
/// Lazy blob handle from ``Table.fetch_blob_files``.
#[pyclass(name = "BlobFile")]
pub struct PyBlobFile {
inner: Arc<BlobFile>,
}
#[pymethods]
impl PyBlobFile {
fn read_bytes(self_: PyRef<'_, Self>) -> PyResult<Py<PyBytes>> {
let inner = self_.inner.clone();
let bytes = block_on(async move { inner.read().await })
.map_err(|e| PyRuntimeError::new_err(format!("blob read failed: {e}")))?;
Ok(PyBytes::new(self_.py(), bytes.as_ref()).unbind())
}
pub fn read(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
let inner = self_.inner.clone();
future_into_py(self_.py(), async move {
let bytes = inner
.read()
.await
.map_err(|e| PyRuntimeError::new_err(format!("blob read failed: {e}")))?;
Python::attach(|py| Ok(PyBytes::new(py, bytes.as_ref()).unbind()))
})
}
fn close(self_: PyRef<'_, Self>) -> PyResult<()> {
let inner = self_.inner.clone();
block_on(async move { inner.close().await })
.map_err(|e| PyRuntimeError::new_err(format!("blob close failed: {e}")))
}
fn is_closed(self_: PyRef<'_, Self>) -> bool {
let inner = self_.inner.clone();
block_on(async move { inner.is_closed().await })
}
fn seek(self_: PyRef<'_, Self>, position: u64) -> PyResult<()> {
let inner = self_.inner.clone();
block_on(async move { inner.seek(position).await })
.map_err(|e| PyRuntimeError::new_err(format!("blob seek failed: {e}")))
}
fn tell(self_: PyRef<'_, Self>) -> PyResult<u64> {
let inner = self_.inner.clone();
block_on(async move { inner.tell().await })
.map_err(|e| PyRuntimeError::new_err(format!("blob tell failed: {e}")))
}
fn size(self_: PyRef<'_, Self>) -> u64 {
self_.inner.size()
}
/// Read a blob-local byte range without moving the cursor.
fn read_range(self_: PyRef<'_, Self>, offset: u64, length: usize) -> PyResult<Py<PyBytes>> {
let end = offset
.checked_add(length as u64)
.ok_or_else(|| PyValueError::new_err("offset + length overflowed"))?;
let inner = self_.inner.clone();
let bytes = block_on(async move { inner.read_range(offset..end).await })
.map_err(|e| PyRuntimeError::new_err(format!("blob read_range failed: {e}")))?;
Ok(PyBytes::new(self_.py(), bytes.as_ref()).unbind())
}
fn read_up_to(self_: PyRef<'_, Self>, length: usize) -> PyResult<Py<PyBytes>> {
let inner = self_.inner.clone();
let bytes = block_on(async move { inner.read_up_to(length).await })
.map_err(|e| PyRuntimeError::new_err(format!("blob read failed: {e}")))?;
Ok(PyBytes::new(self_.py(), bytes.as_ref()).unbind())
}
}
#[pyclass(get_all, from_py_object)]
#[derive(Clone, Debug)]
pub struct FtsToken {
pub text: String,
pub position: u32,
}
#[pymethods]
impl FtsToken {
pub fn __repr__(&self) -> String {
format!("FtsToken(text={:?}, position={})", self.text, self.position)
}
}
impl From<LanceDbFtsToken> for FtsToken {
fn from(token: LanceDbFtsToken) -> Self {
Self {
text: token.text,
position: token.position,
}
}
}
#[pyfunction(signature = (
query,
*,
base_tokenizer = "simple".to_string(),
language = "English".to_string(),
max_token_length = Some(40),
lower_case = true,
stem = true,
remove_stop_words = true,
ascii_folding = true,
ngram_min_length = 3,
ngram_max_length = 3,
prefix_only = false
))]
#[allow(clippy::too_many_arguments)]
pub fn tokenize(
query: String,
base_tokenizer: String,
language: String,
max_token_length: Option<u32>,
lower_case: bool,
stem: bool,
remove_stop_words: bool,
ascii_folding: bool,
ngram_min_length: u32,
ngram_max_length: u32,
prefix_only: bool,
) -> PyResult<Vec<FtsToken>> {
let params = FtsIndexBuilder::default()
.base_tokenizer(base_tokenizer)
.language(&language)
.map_err(|_| {
PyValueError::new_err(format!(
"LanceDB does not support the requested language: '{}'",
language
))
})?
.max_token_length(max_token_length.map(|value| value as usize))
.lower_case(lower_case)
.stem(stem)
.remove_stop_words(remove_stop_words)
.ascii_folding(ascii_folding)
.ngram_min_length(ngram_min_length)
.ngram_max_length(ngram_max_length)
.ngram_prefix_only(prefix_only);
let tokens = lancedb_tokenize(&query, &params).infer_error()?;
Ok(tokens.into_iter().map(FtsToken::from).collect())
}
#[pyclass]
pub struct Table {
// We keep a copy of the name to use if the inner table is dropped
@@ -710,6 +858,29 @@ impl Table {
})
}
#[pyo3(signature = (query, *, column=None, index_name=None))]
pub fn tokenize(
self_: PyRef<'_, Self>,
query: String,
column: Option<String>,
index_name: Option<String>,
) -> PyResult<Bound<'_, PyAny>> {
let inner = self_.inner_ref()?.clone();
future_into_py(self_.py(), async move {
let tokens = match (column.as_deref(), index_name.as_deref()) {
(Some(_), Some(_)) | (None, None) => {
return Err(PyValueError::new_err(
"Specify exactly one of 'column' or 'index_name'",
));
}
(Some(column), None) => inner.tokenize_with_column(&query, column).await,
(None, Some(index_name)) => inner.tokenize(&query, index_name).await,
}
.infer_error()?;
Ok(tokens.into_iter().map(FtsToken::from).collect::<Vec<_>>())
})
}
pub fn index_stats(self_: PyRef<'_, Self>, index_name: String) -> PyResult<Bound<'_, PyAny>> {
let inner = self_.inner_ref()?.clone();
future_into_py(self_.py(), async move {
@@ -901,6 +1072,55 @@ impl Table {
))
}
/// Names of the blob v2 columns declared on this table, in declaration order.
pub fn blob_columns(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
let inner = self_.inner_ref()?.clone();
future_into_py(self_.py(), async move {
inner.blob_columns().await.infer_error()
})
}
/// Read blob bytes for `row_ids` from blob v2 column `column`.
#[pyo3(signature = (column, row_ids))]
pub fn fetch_blobs(
self_: PyRef<'_, Self>,
column: String,
row_ids: Vec<u64>,
) -> PyResult<Bound<'_, PyAny>> {
let inner = self_.inner_ref()?.clone();
future_into_py(self_.py(), async move {
let blobs: LargeBinaryArray = inner
.fetch_blobs(column.as_str(), &row_ids)
.await
.infer_error()?;
Python::attach(|py| blobs.to_data().to_pyarrow(py).map(|obj| obj.unbind()))
})
}
/// Open lazy blob handles for `row_ids` from blob v2 column `column`.
#[pyo3(signature = (column, row_ids))]
pub fn fetch_blob_files(
self_: PyRef<'_, Self>,
column: String,
row_ids: Vec<u64>,
) -> PyResult<Bound<'_, PyAny>> {
let inner = self_.inner_ref()?.clone();
future_into_py(self_.py(), async move {
let handles = inner
.fetch_blob_files(column.as_str(), &row_ids)
.await
.infer_error()?;
Ok(handles
.into_iter()
.map(|handle| {
handle.map(|file| PyBlobFile {
inner: Arc::new(file),
})
})
.collect::<Vec<_>>())
})
}
/// Optimize the on-disk data by compacting and pruning old data, for better performance.
#[pyo3(signature = (cleanup_since_ms=None, delete_unverified=None))]
pub fn optimize(
+2 -2
View File
@@ -1,6 +1,6 @@
[package]
name = "lancedb"
version = "0.32.0-beta.0"
version = "0.32.0-beta.1"
edition.workspace = true
description = "LanceDB: A serverless, low-latency vector database for AI applications"
license.workspace = true
@@ -44,7 +44,6 @@ lance-io = { workspace = true }
lance-index = { workspace = true, features = ["tokenizer-jieba", "tokenizer-lindera"] }
lance-table = { workspace = true }
lance-linalg = { workspace = true }
lance-testing = { workspace = true }
lance-encoding = { workspace = true }
lance-arrow = { workspace = true }
lance-namespace = { workspace = true }
@@ -95,6 +94,7 @@ semver = { workspace = true }
[dev-dependencies]
anyhow = "1"
lance-testing = { workspace = true }
tempfile = "3.5.0"
random_word = { version = "0.4.3", features = ["en"] }
tokio = { version = "1.23", features = ["io-util", "macros", "net", "rt-multi-thread", "sync"] }
+119 -10
View File
@@ -198,28 +198,36 @@ fn compute_embedding_arrays(
batch: &RecordBatch,
embeddings: &[(EmbeddingDefinition, Arc<dyn EmbeddingFunction>)],
) -> Result<Vec<Arc<dyn Array>>> {
if embeddings.len() == 1 {
let (fld, func) = &embeddings[0];
let input_columns = embeddings
.iter()
.map(|(fld, func)| {
let src_column =
batch
.column_by_name(&fld.source_column)
.ok_or_else(|| Error::InvalidInput {
message: format!("Source column '{}' not found", fld.source_column),
})?;
Ok((src_column.clone(), func))
})
.collect::<Result<Vec<_>>>()?;
if batch.num_rows() == 0 {
return input_columns
.iter()
.map(|(_, func)| Ok(arrow_array::new_empty_array(func.dest_type()?.as_ref())))
.collect();
}
if input_columns.len() == 1 {
let (src_column, func) = &input_columns[0];
return Ok(vec![func.compute_source_embeddings(src_column.clone())?]);
}
// Parallel path: multiple embeddings
std::thread::scope(|s| {
let handles: Vec<_> = embeddings
let handles: Vec<_> = input_columns
.iter()
.map(|(fld, func)| {
let src_column = batch.column_by_name(&fld.source_column).ok_or_else(|| {
Error::InvalidInput {
message: format!("Source column '{}' not found", fld.source_column),
}
})?;
.map(|(src_column, func)| {
let handle = s.spawn(move || func.compute_source_embeddings(src_column.clone()));
Ok(handle)
@@ -392,3 +400,104 @@ impl<R: RecordBatchReader> RecordBatchReader for WithEmbeddings<R> {
.into_rich_schema()
}
}
#[cfg(test)]
mod tests {
use std::sync::{
Arc,
atomic::{AtomicUsize, Ordering},
};
use arrow_array::{Array, ArrayRef, FixedSizeListArray, RecordBatch, StringArray};
use arrow_schema::DataType;
use super::*;
#[derive(Debug)]
struct FailingEmbedding {
calls: AtomicUsize,
}
impl EmbeddingFunction for FailingEmbedding {
fn name(&self) -> &str {
"failing"
}
fn source_type(&self) -> Result<Cow<'_, DataType>> {
Ok(Cow::Owned(DataType::Utf8))
}
fn dest_type(&self) -> Result<Cow<'_, DataType>> {
Ok(Cow::Owned(DataType::new_fixed_size_list(
DataType::Float32,
3,
false,
)))
}
fn compute_source_embeddings(&self, _source: Arc<dyn Array>) -> Result<Arc<dyn Array>> {
self.calls.fetch_add(1, Ordering::SeqCst);
Err(Error::Runtime {
message: "embedding function must not receive an empty batch".to_string(),
})
}
fn compute_query_embeddings(&self, _input: Arc<dyn Array>) -> Result<Arc<dyn Array>> {
unreachable!("query embeddings are not exercised by this test")
}
}
#[test]
fn empty_batch_skips_embedding_functions() {
let embedding_function = Arc::new(FailingEmbedding {
calls: AtomicUsize::new(0),
});
let source: ArrayRef = Arc::new(StringArray::from(Vec::<&str>::new()));
let batch = RecordBatch::try_from_iter([("text", source)]).unwrap();
let embeddings = vec![(
EmbeddingDefinition::new("text", "failing", Some("text_embedding")),
embedding_function.clone() as Arc<dyn EmbeddingFunction>,
)];
let result = compute_embeddings_for_batch(batch, &embeddings).unwrap();
assert_eq!(embedding_function.calls.load(Ordering::SeqCst), 0);
assert_eq!(result.num_rows(), 0);
let embedding = result.column_by_name("text_embedding").unwrap();
assert_eq!(
embedding.data_type(),
&DataType::new_fixed_size_list(DataType::Float32, 3, false)
);
assert_eq!(embedding.null_count(), 0);
let embedding = embedding
.as_any()
.downcast_ref::<FixedSizeListArray>()
.unwrap();
assert_eq!(embedding.len(), 0);
assert_eq!(embedding.value_length(), 3);
assert_eq!(embedding.values().len(), 0);
}
#[test]
fn empty_batch_still_validates_source_column() {
let embedding_function = Arc::new(FailingEmbedding {
calls: AtomicUsize::new(0),
});
let source: ArrayRef = Arc::new(StringArray::from(Vec::<&str>::new()));
let batch = RecordBatch::try_from_iter([("text", source)]).unwrap();
let embeddings = vec![(
EmbeddingDefinition::new("missing_column", "failing", Some("text_embedding")),
embedding_function.clone() as Arc<dyn EmbeddingFunction>,
)];
let result = compute_embeddings_for_batch(batch, &embeddings);
assert!(result.is_err());
assert!(
matches!(result.unwrap_err(), Error::InvalidInput { .. }),
"expected InvalidInput error when source column is missing"
);
assert_eq!(embedding_function.calls.load(Ordering::SeqCst), 0);
}
}
+5 -9
View File
@@ -317,6 +317,8 @@ pub enum IndexType {
// FTS
#[serde(alias = "INVERTED", alias = "Inverted")]
FTS,
/// Catch-all for index types not recognized by this version of LanceDB.
Unknown,
}
impl std::fmt::Display for IndexType {
@@ -334,6 +336,7 @@ impl std::fmt::Display for IndexType {
Self::LabelList => write!(f, "LABEL_LIST"),
Self::Fm => write!(f, "FM"),
Self::FTS => write!(f, "FTS"),
Self::Unknown => write!(f, "UNKNOWN"),
}
}
}
@@ -355,9 +358,7 @@ impl std::str::FromStr for IndexType {
"IVF_HNSW_PQ" => Ok(Self::IvfHnswPq),
"IVF_HNSW_SQ" => Ok(Self::IvfHnswSq),
"IVF_HNSW_FLAT" => Ok(Self::IvfHnswFlat),
_ => Err(Error::InvalidInput {
message: format!("the input value {} is not a valid IndexType", value),
}),
_ => Ok(Self::Unknown),
}
}
}
@@ -425,20 +426,15 @@ pub struct IndexConfig {
#[derive(Debug, Deserialize)]
pub(crate) struct IndexMetadata {
pub metric_type: Option<DistanceType>,
// Sometimes the index type is provided at this level.
pub index_type: Option<IndexType>,
}
// This struct is used to deserialize the JSON data returned from the Lance API
// Dataset::index_statistics().
// Deserializes the JSON returned by Dataset::index_statistics().
#[skip_serializing_none]
#[derive(Debug, Deserialize)]
pub(crate) struct IndexStatisticsImpl {
pub num_indexed_rows: usize,
pub num_unindexed_rows: usize,
pub indices: Vec<IndexMetadata>,
// Sometimes, the index type is provided at this level.
pub index_type: Option<IndexType>,
pub num_indices: Option<u32>,
}
+9 -1
View File
@@ -207,7 +207,15 @@ use lance_linalg::distance::DistanceType as LanceDistanceType;
/// a built-in pull-based adapter.
#[cfg(feature = "metrics")]
pub use metrics;
pub use table::Table;
pub use table::{FtsToken, Table};
/// Tokenize a full-text search query using an explicit FTS tokenizer configuration.
///
/// This does not require a table or FTS index. The tokenizer options are the
/// same [`index::scalar::FtsIndexBuilder`] values used when creating an FTS index.
pub fn tokenize(query: &str, params: &index::scalar::FtsIndexBuilder) -> Result<Vec<FtsToken>> {
table::tokenize(query, params)
}
#[derive(Debug, Copy, Clone, PartialEq, Serialize, Deserialize, Default)]
#[non_exhaustive]
+136 -2
View File
@@ -2817,8 +2817,7 @@ mod tests {
use super::*;
use crate::remote::client::{ClientConfig, RetryConfig};
use crate::table::AddDataMode;
use crate::table::FieldMetadataUpdate;
use crate::table::{AddDataMode, FieldMetadataUpdate, FtsToken};
use arrow::{array::AsArray, compute::concat_batches, datatypes::Int32Type};
use arrow_array::{Int32Array, RecordBatch, RecordBatchIterator, record_batch};
@@ -4888,6 +4887,141 @@ mod tests {
assert_eq!(text_idx.created_at, None);
}
#[tokio::test]
async fn test_tokenize_uses_remote_index_details() {
let schema = Schema::new(vec![Field::new("text", DataType::Utf8, false)]);
let index_details = serde_json::json!({
"base_tokenizer": "icu",
"language": "English",
"with_position": false,
"max_token_length": 40,
"lower_case": true,
"stem": false,
"remove_stop_words": false,
"ascii_folding": true,
})
.to_string();
let table = Table::new_with_handler("my_table", move |request| {
assert_eq!(request.method(), "POST");
match request.url().path() {
"/v1/table/my_table/describe/" => http::Response::builder()
.status(200)
.body(describe_response(&schema))
.unwrap(),
"/v1/table/my_table/index/list/" => {
let body = serde_json::json!({
"indexes": [
{
"index_name": "text_idx",
"columns": ["text"],
"index_type": "FTS",
"index_details": index_details,
},
]
});
http::Response::builder()
.status(200)
.body(serde_json::to_string(&body).unwrap())
.unwrap()
}
path => panic!("Unexpected path: {}", path),
}
});
let tokens = table
.tokenize("Hello, こんにちは世界!", "text_idx")
.await
.unwrap();
assert_eq!(
tokens,
vec![
FtsToken {
text: "hello".to_string(),
position: 0,
},
FtsToken {
text: "こんにちは".to_string(),
position: 1,
},
FtsToken {
text: "世界".to_string(),
position: 2,
},
]
);
}
#[tokio::test]
async fn test_tokenize_requires_existing_index_name() {
let schema = Schema::new(vec![Field::new("text", DataType::Utf8, false)]);
let table = Table::new_with_handler("my_table", move |request| -> http::Response<String> {
assert_eq!(request.method(), "POST");
match request.url().path() {
"/v1/table/my_table/describe/" => http::Response::builder()
.status(200)
.body(describe_response(&schema))
.unwrap(),
"/v1/table/my_table/index/list/" => {
let body = serde_json::json!({ "indexes": [] });
http::Response::builder()
.status(200)
.body(serde_json::to_string(&body).unwrap())
.unwrap()
}
path => panic!("Unexpected path: {}", path),
}
});
let err = table.tokenize("hello", "text_idx").await.unwrap_err();
assert!(matches!(
err,
Error::InvalidInput { message }
if message.contains("No index named 'text_idx'")
));
}
#[tokio::test]
async fn test_tokenize_with_column_remote_requires_index_details() {
let schema = Schema::new(vec![Field::new("text", DataType::Utf8, false)]);
let table = Table::new_with_handler("my_table", move |request| {
assert_eq!(request.method(), "POST");
match request.url().path() {
"/v1/table/my_table/describe/" => http::Response::builder()
.status(200)
.body(describe_response(&schema))
.unwrap(),
"/v1/table/my_table/index/list/" => {
let body = serde_json::json!({
"indexes": [
{
"index_name": "text_idx",
"columns": ["text"],
"index_type": "FTS",
},
]
});
http::Response::builder()
.status(200)
.body(serde_json::to_string(&body).unwrap())
.unwrap()
}
path => panic!("Unexpected path: {}", path),
}
});
let err = table
.tokenize_with_column("hello", "text")
.await
.unwrap_err();
assert!(matches!(
err,
Error::InvalidInput { message }
if message.contains("does not include tokenizer details")
));
}
#[test]
fn test_deserialize_created_at() {
#[derive(Deserialize)]
+233 -29
View File
@@ -23,10 +23,13 @@ use lance::dataset::{InsertBuilder, WriteParams};
use lance::index::DatasetIndexExt;
use lance::io::{ObjectStoreParams, WrappingObjectStore};
use lance_datafusion::utils::StreamingWriteSource;
use lance_index::IndexCriteria;
use lance_io::object_store::{LanceNamespaceStorageOptionsProvider, StorageOptionsAccessor};
pub use query::AnyQuery;
use lance::io::commit::namespace_manifest::LanceNamespaceExternalManifestStore;
use lance_index::scalar::InvertedIndexParams;
use lance_index::scalar::inverted::query::collect_query_tokens;
use lance_namespace::LanceNamespace;
use lance_namespace::error::NamespaceError;
use lance_namespace::models::DescribeTableRequest;
@@ -42,6 +45,7 @@ use std::sync::Arc;
use crate::connection::NamespaceClientPushdownOperation;
use crate::DistanceType;
use crate::data::scannable::{PeekedScannable, Scannable, estimate_write_partitions};
use crate::database::Database;
use crate::database::read_freshness::TableFreshness;
@@ -49,10 +53,10 @@ use crate::embeddings::{EmbeddingDefinition, EmbeddingRegistry, MemoryRegistry};
use crate::error::{Error, Result};
use crate::index::IndexStatistics;
use crate::index::{Index, IndexBuilder};
use crate::index::{IndexConfig, IndexStatisticsImpl};
use crate::index::{IndexConfig, IndexStatisticsImpl, IndexType};
use crate::query::{IntoQueryVector, Query, QueryExecutionOptions, TakeQuery, VectorQuery};
use crate::table::datafusion::insert::InsertExec;
use crate::utils::{PatchReadParam, PatchWriteParam};
use crate::utils::{PatchReadParam, PatchWriteParam, resolve_arrow_field_path};
use self::dataset::DatasetConsistencyWrapper;
use self::merge::MergeInsertBuilder;
@@ -471,6 +475,33 @@ impl LsmWriteSpec {
}
}
/// A token produced by the tokenizer configured on a full-text search index.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct FtsToken {
/// The token text after the index tokenizer has applied its filters.
pub text: String,
/// The token position used by full-text query matching.
pub position: u32,
}
/// Tokenize a full-text search query using an explicit FTS tokenizer configuration.
///
/// This does not require a table or FTS index. Use
/// [`crate::index::scalar::FtsIndexBuilder`] to supply the same tokenizer
/// options used when creating an FTS index.
pub fn tokenize(query: &str, params: &InvertedIndexParams) -> Result<Vec<FtsToken>> {
let mut tokenizer = params.build().map_err(|err| Error::InvalidInput {
message: format!("Failed to build tokenizer: {}", err),
})?;
let tokens = collect_query_tokens(query, &mut tokenizer);
Ok((0..tokens.len())
.map(|idx| FtsToken {
text: tokens.get_token(idx).to_string(),
position: tokens.position(idx),
})
.collect())
}
/// A trait for anything "table-like". This is used for both native tables (which target
/// Lance datasets) and remote tables (which target LanceDB cloud)
///
@@ -1659,6 +1690,111 @@ impl Table {
self.inner.list_indices().await
}
/// Tokenize a full-text search query using the tokenizer configured on an FTS index.
///
/// Model-backed tokenizers such as `jieba/*` and `lindera/*` are rebuilt in
/// the client process from index metadata. For remote tables, this means the
/// same tokenizer model files must also exist locally.
pub async fn tokenize(&self, query: &str, index_name: &str) -> Result<Vec<FtsToken>> {
let indices = self.inner.list_indices().await?;
let matches = indices
.iter()
.filter(|idx| idx.name == index_name)
.collect::<Vec<_>>();
let index = match matches.as_slice() {
[index] => *index,
[] => {
return Err(Error::InvalidInput {
message: format!("No index named '{}'", index_name),
});
}
_ => {
return Err(Error::InvalidInput {
message: format!("Index name '{}' is ambiguous", index_name),
});
}
};
if index.index_type != IndexType::FTS {
return Err(Error::InvalidInput {
message: format!("Index '{}' is not a full text search index", index_name),
});
}
self.tokenize_with_index(query, index, index_name)
}
/// Tokenize a full-text search query using the tokenizer configured on the
/// FTS index for a column.
///
/// The column must have exactly one FTS index. Model-backed tokenizers such
/// as `jieba/*` and `lindera/*` are rebuilt in the client process from
/// index metadata. For remote tables, this means the same tokenizer model
/// files must also exist locally.
pub async fn tokenize_with_column(&self, query: &str, column: &str) -> Result<Vec<FtsToken>> {
let schema = self.inner.schema().await?;
let (column, _) = resolve_arrow_field_path(schema.as_ref(), column)?;
let indices = self.inner.list_indices().await?;
let matches = indices
.iter()
.filter(|idx| {
idx.index_type == IndexType::FTS
&& idx.columns.len() == 1
&& idx.columns[0] == column
})
.collect::<Vec<_>>();
let index = match matches.as_slice() {
[index] => *index,
[] => {
return Err(Error::InvalidInput {
message: format!("Column '{}' does not have a full text search index", column),
});
}
_ => {
return Err(Error::InvalidInput {
message: format!(
"Column '{}' has multiple full text search indexes; tokenization by column is ambiguous",
column
),
});
}
};
self.tokenize(query, &index.name).await
}
fn tokenize_with_index(
&self,
query: &str,
index: &IndexConfig,
index_name: &str,
) -> Result<Vec<FtsToken>> {
let selector_description = format!("index name '{}'", index_name);
let details = index
.index_details
.as_deref()
.ok_or_else(|| Error::InvalidInput {
message: format!(
"Full text search index '{}' for {} does not include tokenizer details",
index.name, selector_description
),
})?;
let params = serde_json::from_str::<InvertedIndexParams>(details).map_err(|err| {
Error::InvalidInput {
message: format!(
"Failed to parse tokenizer details for full text search index '{}' for {}: {}",
index.name, selector_description, err
),
}
})?;
tokenize(query, &params).map_err(|err| match err {
Error::InvalidInput { message } => Error::InvalidInput {
message: format!(
"{} for full text search index '{}' for {}",
message, index.name, selector_description
),
},
err => err,
})
}
/// Get the table URI (storage location)
///
/// Returns the full storage location of the table (e.g., S3/GCS path).
@@ -2967,17 +3103,21 @@ impl BaseTable for NativeTable {
.await?
.into_iter()
.filter_map(|idx_desc| {
let index_type: crate::index::IndexType = match idx_desc.index_type().parse() {
Ok(index_type) => index_type,
Err(e) => {
log::warn!(
"Failed to parse index type for index {}: {}",
let index_type: crate::index::IndexType = idx_desc
.index_type()
.parse()
.unwrap_or(crate::index::IndexType::Unknown);
if index_type == crate::index::IndexType::Unknown {
// Internal or future index types that this version doesn't recognize
// (e.g. Lance's internal FragReuseIndex) are silently excluded from
// the user-visible index listing.
log::debug!(
"Skipping unrecognized index '{}' (type '{}') in list_indices",
idx_desc.name(),
e
idx_desc.index_type(),
);
return None;
}
};
let field_ids = idx_desc.field_ids();
let mut columns = Vec::with_capacity(field_ids.len());
@@ -3044,40 +3184,83 @@ impl BaseTable for NativeTable {
}
async fn index_stats(&self, index_name: &str) -> Result<Option<IndexStatistics>> {
let stats = match self
.dataset
.get()
.await?
.index_statistics(index_name.as_ref())
.await
{
Ok(stats) => stats,
Err(lance_core::Error::IndexNotFound { .. }) => return Ok(None),
Err(e) => return Err(Error::from(e)),
// describe_indices() reads only manifest-level metadata (no index file I/O).
// VectorIndexDetails in the manifest carries distance_type for indices written
// by recent Lance versions. For older datasets that didn't write those details
// we fall back to index_statistics() for vector index types.
let dataset = self.dataset.get().await?;
let mut descriptions = dataset
.describe_indices(Some(IndexCriteria::default().with_name(index_name)))
.await?;
let Some(description) = descriptions.pop() else {
return Ok(None);
};
let index_type: crate::index::IndexType = description
.index_type()
.parse()
.unwrap_or(crate::index::IndexType::Unknown);
let is_vector = matches!(
index_type,
crate::index::IndexType::IvfFlat
| crate::index::IndexType::IvfSq
| crate::index::IndexType::IvfPq
| crate::index::IndexType::IvfRq
| crate::index::IndexType::IvfHnswPq
| crate::index::IndexType::IvfHnswSq
| crate::index::IndexType::IvfHnswFlat
);
// details() serializes VectorIndexDetails to JSON with an uppercase "metric_type"
// field (e.g. "L2", "COSINE"). Parse it with a case-insensitive match.
let distance_type = description.details().ok().and_then(|json| {
#[derive(serde::Deserialize)]
struct Details {
metric_type: Option<String>,
}
serde_json::from_str::<Details>(&json)
.ok()
.and_then(|d| d.metric_type)
.and_then(|m| match m.to_uppercase().as_str() {
"L2" => Some(DistanceType::L2),
"COSINE" => Some(DistanceType::Cosine),
"DOT" => Some(DistanceType::Dot),
"HAMMING" => Some(DistanceType::Hamming),
_ => None,
})
});
// Older Lance datasets didn't write VectorIndexDetails, so distance_type won't
// be in the manifest. Fall back to index_statistics() only in that case.
if is_vector && distance_type.is_none() {
let stats = dataset.index_statistics(index_name).await?;
let mut stats: IndexStatisticsImpl =
serde_json::from_str(&stats).map_err(|e| Error::InvalidInput {
message: format!("error deserializing index statistics: {}", e),
})?;
let first_index = stats.indices.pop().ok_or_else(|| Error::InvalidInput {
message: "index statistics is empty".to_string(),
})?;
// Index type should be present at one of the levels.
let index_type =
stats
.index_type
.or(first_index.index_type)
.ok_or_else(|| Error::InvalidInput {
message: "index statistics was missing index type".to_string(),
})?;
Ok(Some(IndexStatistics {
return Ok(Some(IndexStatistics {
num_indexed_rows: stats.num_indexed_rows,
num_unindexed_rows: stats.num_unindexed_rows,
index_type,
distance_type: first_index.metric_type,
num_indices: stats.num_indices,
}));
}
let num_indexed_rows = description.rows_indexed() as usize;
let total_rows = dataset.count_rows(None).await?;
let num_unindexed_rows = total_rows.saturating_sub(num_indexed_rows);
Ok(Some(IndexStatistics {
num_indexed_rows,
num_unindexed_rows,
index_type,
distance_type,
num_indices: Some(description.metadata().len() as u32),
}))
}
@@ -3234,6 +3417,27 @@ mod tests {
use crate::query::{ExecutableQuery, QueryBase};
use crate::test_utils::connection::new_test_connection;
#[test]
fn test_tokenize_uses_explicit_simple_tokenizer() {
let params =
crate::index::scalar::FtsIndexBuilder::default().base_tokenizer("simple".to_string());
let tokens = crate::tokenize("Running in cafés", &params).unwrap();
assert_eq!(
tokens,
vec![
FtsToken {
text: "run".to_string(),
position: 0,
},
FtsToken {
text: "cafe".to_string(),
position: 2,
},
]
);
}
#[tokio::test]
async fn test_open() {
let tmp_dir = tempdir().unwrap();
+33
View File
@@ -589,6 +589,7 @@ mod tests {
let stats = table.index_stats(index_name).await.unwrap().unwrap();
assert_eq!(stats.num_indexed_rows, 512);
assert_eq!(stats.num_unindexed_rows, 0);
assert_eq!(stats.distance_type, Some(crate::DistanceType::L2));
}
#[tokio::test]
@@ -646,6 +647,7 @@ mod tests {
let stats = table.index_stats(index_name).await.unwrap().unwrap();
assert_eq!(stats.num_indexed_rows, 512);
assert_eq!(stats.num_unindexed_rows, 0);
assert_eq!(stats.distance_type, Some(crate::DistanceType::L2));
}
#[tokio::test]
@@ -690,6 +692,10 @@ mod tests {
assert_eq!(index.index_type, crate::index::IndexType::IvfHnswFlat);
assert_eq!(index.columns, vec!["embeddings".to_string()]);
assert_eq!(table.count_rows(None).await.unwrap(), 512);
let stats = table.index_stats(&index.name).await.unwrap().unwrap();
assert_eq!(stats.num_indexed_rows, 512);
assert_eq!(stats.num_unindexed_rows, 0);
assert_eq!(stats.distance_type, Some(crate::DistanceType::L2));
}
#[tokio::test]
@@ -747,6 +753,15 @@ mod tests {
let stats = table.index_stats(index_name).await.unwrap().unwrap();
assert_eq!(stats.num_indexed_rows, 1);
assert_eq!(stats.num_unindexed_rows, 0);
assert_eq!(stats.index_type, crate::index::IndexType::BTree);
assert_eq!(stats.distance_type, None);
// Rows added after the index was built appear as unindexed.
let new_batch = record_batch!(("i", Int32, [2])).unwrap();
table.add(new_batch).execute().await.unwrap();
let stats = table.index_stats(index_name).await.unwrap().unwrap();
assert_eq!(stats.num_indexed_rows, 1);
assert_eq!(stats.num_unindexed_rows, 1);
}
#[tokio::test]
@@ -795,6 +810,12 @@ mod tests {
.map(|b| b.num_rows())
.sum::<usize>();
assert_eq!(count, 1);
let stats = table.index_stats("text_idx").await.unwrap().unwrap();
assert_eq!(stats.num_indexed_rows, 1);
assert_eq!(stats.num_unindexed_rows, 0);
assert_eq!(stats.index_type, crate::index::IndexType::Fm);
assert_eq!(stats.distance_type, None);
}
#[tokio::test]
@@ -1188,6 +1209,12 @@ mod tests {
let index = configs_iter.next().unwrap();
assert_eq!(index.index_type, crate::index::IndexType::Bitmap);
assert_eq!(index.columns, vec!["large_data".to_string()]);
let stats = table.index_stats("category_idx").await.unwrap().unwrap();
assert_eq!(stats.num_indexed_rows, 100);
assert_eq!(stats.num_unindexed_rows, 0);
assert_eq!(stats.index_type, crate::index::IndexType::Bitmap);
assert_eq!(stats.distance_type, None);
}
#[tokio::test]
@@ -1256,6 +1283,12 @@ mod tests {
let index = index_configs.into_iter().next().unwrap();
assert_eq!(index.index_type, crate::index::IndexType::LabelList);
assert_eq!(index.columns, vec!["tags".to_string()]);
let stats = table.index_stats("tags_idx").await.unwrap().unwrap();
assert_eq!(stats.num_indexed_rows, 40);
assert_eq!(stats.num_unindexed_rows, 0);
assert_eq!(stats.index_type, crate::index::IndexType::LabelList);
assert_eq!(stats.distance_type, None);
}
#[tokio::test]