mirror of
https://github.com/lancedb/lancedb.git
synced 2026-08-27 16:38:31 +00:00
Compare commits
69 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 0e9224ce9b | |||
| 77a93fee76 | |||
| 7bb501839a | |||
| 5b347afd99 | |||
| 706a9c327f | |||
| be290447d9 | |||
| 79ba076429 | |||
| ec21e37040 | |||
| 6ba80a960c | |||
| 11f24b1df4 | |||
| 2ba7407dc3 | |||
| 607e556927 | |||
| 564e5d0d56 | |||
| dd5cb4d805 | |||
| dbc3687c7b | |||
| ec80acb668 | |||
| fc44535cee | |||
| 4048150fdd | |||
| 2922c171f7 | |||
| c5f9efefe9 | |||
| f4c668e244 | |||
| b1cfe6edb1 | |||
| 001237c7a4 | |||
| 369b10a377 | |||
| 1c3cd1d918 | |||
| 9707966943 | |||
| 62fe413a52 | |||
| 1493ece3de | |||
| e6444ecc05 | |||
| cc0139c136 | |||
| b20696ef9c | |||
| 772bdeced8 | |||
| c1a3fa7f51 | |||
| 0ba82873c5 | |||
| 3af51541a0 | |||
| 2c06a48bd8 | |||
| ac8b28c010 | |||
| 173f889d2a | |||
| 03b52e5877 | |||
| 798e5364fb | |||
| f1f34dfdd3 | |||
| 123c921c4f | |||
| 99a68db78c | |||
| 9e73d440a3 | |||
| 3956d9dbfa | |||
| 16e1967efc | |||
| 27dd92c67e | |||
| 9e2e711c7a | |||
| c3176a47ce | |||
| 7357d63e87 | |||
| 624a75edf7 | |||
| c7ea91f3ea | |||
| 8e24dd3828 | |||
| f79dc017c4 | |||
| e6ae93f52a | |||
| 3dd9c598e9 | |||
| 9e26bf3fba | |||
| 93354baf34 | |||
| 05602ec7d5 | |||
| e3b472c212 | |||
| a6418b6cb9 | |||
| dd2b11eda2 | |||
| 5a1015ba72 | |||
| 48945d0658 | |||
| 77208fd464 | |||
| b505dc1315 | |||
| 7dfdfe6401 | |||
| 4dc2d9a0f2 | |||
| 1ad6ce3a4e |
+1
-1
@@ -1,5 +1,5 @@
|
|||||||
[tool.bumpversion]
|
[tool.bumpversion]
|
||||||
current_version = "0.37.1-beta.0"
|
current_version = "0.37.1-beta.1"
|
||||||
parse = """(?x)
|
parse = """(?x)
|
||||||
(?P<major>0|[1-9]\\d*)\\.
|
(?P<major>0|[1-9]\\d*)\\.
|
||||||
(?P<minor>0|[1-9]\\d*)\\.
|
(?P<minor>0|[1-9]\\d*)\\.
|
||||||
|
|||||||
@@ -0,0 +1,222 @@
|
|||||||
|
name: Check doc links
|
||||||
|
|
||||||
|
# Checking external links is inherently noisy: third-party sites rate-limit
|
||||||
|
# automated clients, reject non-browser user agents, and go down temporarily.
|
||||||
|
# Blocking pull requests on that trades a lot of false failures for very little
|
||||||
|
# signal, so this runs on a schedule and reports findings in a single tracking
|
||||||
|
# issue instead of failing anyone's build.
|
||||||
|
on:
|
||||||
|
schedule:
|
||||||
|
- cron: "0 7 * * *"
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
# The report lives in one repository-global issue, so runs must not overlap: a
|
||||||
|
# lookup racing a create produces duplicate issues, and a healthy run closing
|
||||||
|
# the issue while a failing run only rewrites its body would leave a broken
|
||||||
|
# report closed. The group is deliberately ref-independent so that a manual
|
||||||
|
# dispatch serializes against the scheduled run.
|
||||||
|
concurrency:
|
||||||
|
group: docs-link-check
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
|
permissions: {}
|
||||||
|
|
||||||
|
env:
|
||||||
|
REPORT_TITLE: "Docs link checker report"
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
scan:
|
||||||
|
name: Scan links
|
||||||
|
runs-on: ubuntu-24.04
|
||||||
|
# lychee-action is pinned by SHA, but its wrapper downloads the lychee
|
||||||
|
# release tarball at run time without verifying a digest, and hands the
|
||||||
|
# resulting binary a GitHub token. Release assets remain replaceable, so
|
||||||
|
# that binary is confined to a job whose token can only read public
|
||||||
|
# content; everything that writes runs in the report job below.
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
outputs:
|
||||||
|
exit_code: ${{ steps.lychee.outputs.exit_code }}
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v6
|
||||||
|
with:
|
||||||
|
# workflow_dispatch can run from any ref, but the report is
|
||||||
|
# repository-global. Always measure the default branch so a manual
|
||||||
|
# run from a topic branch cannot close a report that main warrants,
|
||||||
|
# or overwrite it with branch-only findings.
|
||||||
|
ref: ${{ github.event.repository.default_branch }}
|
||||||
|
persist-credentials: false
|
||||||
|
|
||||||
|
- name: Check links
|
||||||
|
id: lychee
|
||||||
|
uses: lycheeverse/lychee-action@e7477775783ea5526144ba13e8db5eec57747ce8 # v2.9.0
|
||||||
|
with:
|
||||||
|
# Restricted to http(s) on purpose. Much of docs/src is generated
|
||||||
|
# API reference (the js/ tree comes from `npm run docs` in nodejs)
|
||||||
|
# and the hand-written pages use mkdocstrings cross-references and
|
||||||
|
# nav-relative paths that only resolve in the site mkdocs builds,
|
||||||
|
# not in this checkout, so relative links would be reported as
|
||||||
|
# broken on every run.
|
||||||
|
args: >-
|
||||||
|
--scheme https
|
||||||
|
--scheme http
|
||||||
|
--no-progress
|
||||||
|
--max-retries 3
|
||||||
|
--timeout 20
|
||||||
|
'docs/src/**/*.md'
|
||||||
|
format: json
|
||||||
|
output: ./lychee/out.json
|
||||||
|
jobSummary: false
|
||||||
|
# The report, not a red build, is the signal for broken links. The
|
||||||
|
# validation step below still fails the run if the check itself
|
||||||
|
# breaks.
|
||||||
|
fail: false
|
||||||
|
|
||||||
|
- name: Validate report
|
||||||
|
# lychee does not reserve exit code 2 for broken links: its CLI
|
||||||
|
# parser also exits 2 on an invalid option, before any link was
|
||||||
|
# checked or any report written. Only a parseable report whose
|
||||||
|
# counts agree with the exit code counts as a link verdict; anything
|
||||||
|
# else fails here, and the report job below is skipped entirely, so
|
||||||
|
# the tracking issue is never touched. Exit 2 covers timeouts as
|
||||||
|
# well as errors, and a timed-out host is exactly the transient
|
||||||
|
# unavailability this report exists to surface, so both count as
|
||||||
|
# findings. Requiring total > 0 also catches a glob that silently
|
||||||
|
# stopped matching any file.
|
||||||
|
if: steps.lychee.outputs.exit_code == 0 || steps.lychee.outputs.exit_code == 2
|
||||||
|
env:
|
||||||
|
EXIT_CODE: ${{ steps.lychee.outputs.exit_code }}
|
||||||
|
run: |
|
||||||
|
jq -e --argjson code "$EXIT_CODE" '
|
||||||
|
(.total > 0) and
|
||||||
|
(if $code == 0
|
||||||
|
then .errors == 0 and .timeouts == 0
|
||||||
|
and (.error_map | length == 0) and (.timeout_map | length == 0)
|
||||||
|
else (.errors + .timeouts) > 0
|
||||||
|
and ((.error_map | length) + (.timeout_map | length)) > 0
|
||||||
|
end)
|
||||||
|
' ./lychee/out.json
|
||||||
|
|
||||||
|
- name: Upload report
|
||||||
|
if: steps.lychee.outputs.exit_code == 2
|
||||||
|
uses: actions/upload-artifact@v7
|
||||||
|
with:
|
||||||
|
name: link-report
|
||||||
|
path: ./lychee/out.json
|
||||||
|
retention-days: 7
|
||||||
|
|
||||||
|
report:
|
||||||
|
name: Update report issue
|
||||||
|
needs: scan
|
||||||
|
runs-on: ubuntu-24.04
|
||||||
|
# Deliberately no checkout: this job needs the report artifact and the
|
||||||
|
# issues API, not the repository contents.
|
||||||
|
permissions:
|
||||||
|
issues: write
|
||||||
|
env:
|
||||||
|
EXIT_CODE: ${{ needs.scan.outputs.exit_code }}
|
||||||
|
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
steps:
|
||||||
|
- name: Classify checker result
|
||||||
|
# lychee exits 0 when every link resolves and 2 when links fail,
|
||||||
|
# both already cross-checked against the report by the scan job's
|
||||||
|
# validation step. Anything else (1 runtime, 3 bad config) means the
|
||||||
|
# check never produced a link verdict, which must surface as a failed
|
||||||
|
# run rather than be published as "broken documentation links".
|
||||||
|
run: |
|
||||||
|
case "$EXIT_CODE" in
|
||||||
|
0|2)
|
||||||
|
echo "lychee exit code $EXIT_CODE"
|
||||||
|
;;
|
||||||
|
*)
|
||||||
|
echo "::error::lychee exited with '$EXIT_CODE': the link check did not complete. Leaving the report issue untouched."
|
||||||
|
exit 1
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
|
||||||
|
- name: Find existing report issue
|
||||||
|
id: report
|
||||||
|
# Matched on title alone, and through search rather than a listing:
|
||||||
|
# the issue action applies labels in a separate call after creating the
|
||||||
|
# issue, so a label filter misses a half-created report, and this
|
||||||
|
# repository has far more open issues than one listing page holds.
|
||||||
|
# Closed issues are included because a healthy run closes the report:
|
||||||
|
# an open-only lookup would forget that identity and the next failing
|
||||||
|
# run would open a duplicate. The oldest match stays the canonical
|
||||||
|
# report and is reopened below when links break again.
|
||||||
|
run: |
|
||||||
|
match=$(gh issue list --repo "$GITHUB_REPOSITORY" --state all \
|
||||||
|
--search "in:title \"$REPORT_TITLE\" author:app/github-actions" \
|
||||||
|
--limit 50 --json number,title,state \
|
||||||
|
--jq "[.[] | select(.title == \"$REPORT_TITLE\")] | sort_by(.number) | first // empty")
|
||||||
|
echo "number=$(jq -r '.number // empty' <<<"$match")" >> "$GITHUB_OUTPUT"
|
||||||
|
echo "state=$(jq -r '.state // empty' <<<"$match")" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
|
- name: Download report
|
||||||
|
if: env.EXIT_CODE == 2
|
||||||
|
uses: actions/download-artifact@v8
|
||||||
|
with:
|
||||||
|
name: link-report
|
||||||
|
path: ./lychee
|
||||||
|
|
||||||
|
- name: Compose report
|
||||||
|
if: env.EXIT_CODE == 2
|
||||||
|
run: |
|
||||||
|
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
|
||||||
|
{
|
||||||
|
echo "Broken documentation links found by [\`$GITHUB_WORKFLOW\`]($run_url)."
|
||||||
|
echo
|
||||||
|
echo "This issue is rewritten by every scheduled run and closed automatically once all links resolve."
|
||||||
|
echo
|
||||||
|
echo "Entries can be false positives: some sites rate-limit or block automated clients while working fine in a browser. Confirm before editing the docs, and add persistent offenders to \`--exclude\` in \`.github/workflows/docs-link-check.yml\`."
|
||||||
|
echo
|
||||||
|
# Timeouts are reported alongside errors: entries land in
|
||||||
|
# timeout_map with a status text instead of an HTTP code.
|
||||||
|
jq -r '
|
||||||
|
"\(.errors) of \(.total) links failed, \(.timeouts) timed out.",
|
||||||
|
"",
|
||||||
|
([(.error_map | to_entries[]), (.timeout_map | to_entries[])]
|
||||||
|
| group_by(.key)[] |
|
||||||
|
"### Errors in \(.[0].key)",
|
||||||
|
"",
|
||||||
|
(map(.value[])[] | "* [\(.status.code // .status.text // "ERR")] <\(.url)> — \(.status.details // .status.text // "unknown error")"),
|
||||||
|
"")
|
||||||
|
' ./lychee/out.json
|
||||||
|
} > ./lychee/issue.md
|
||||||
|
|
||||||
|
- name: Reopen report issue
|
||||||
|
# A healthy run closes the report, and the issue action below only
|
||||||
|
# rewrites the body of whatever number it is given. Without an
|
||||||
|
# explicit reopen, the 2 -> 0 -> 2 sequence would keep rewriting a
|
||||||
|
# closed issue while links are broken. A CLOSED state implies the
|
||||||
|
# lookup found a canonical issue, so no separate emptiness check.
|
||||||
|
if: env.EXIT_CODE == 2 && steps.report.outputs.state == 'CLOSED'
|
||||||
|
env:
|
||||||
|
ISSUE_NUMBER: ${{ steps.report.outputs.number }}
|
||||||
|
run: |
|
||||||
|
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
|
||||||
|
gh issue reopen "$ISSUE_NUMBER" --repo "$GITHUB_REPOSITORY" \
|
||||||
|
--comment "Broken documentation links found again in [the latest run]($run_url)."
|
||||||
|
|
||||||
|
- name: Report broken links
|
||||||
|
if: env.EXIT_CODE == 2
|
||||||
|
uses: peter-evans/create-issue-from-file@fca9117c27cdc29c6c4db3b86c48e4115a786710 # v6.0.0
|
||||||
|
with:
|
||||||
|
# Empty on the first failing run, which creates the issue; afterwards
|
||||||
|
# the same issue is updated in place.
|
||||||
|
issue-number: ${{ steps.report.outputs.number }}
|
||||||
|
title: ${{ env.REPORT_TITLE }}
|
||||||
|
content-filepath: ./lychee/issue.md
|
||||||
|
labels: documentation
|
||||||
|
|
||||||
|
- name: Close report issue once links are healthy
|
||||||
|
# An OPEN state implies the lookup found a canonical issue; a report
|
||||||
|
# that is already closed needs nothing.
|
||||||
|
if: env.EXIT_CODE == 0 && steps.report.outputs.state == 'OPEN'
|
||||||
|
env:
|
||||||
|
ISSUE_NUMBER: ${{ steps.report.outputs.number }}
|
||||||
|
run: |
|
||||||
|
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
|
||||||
|
gh issue close "$ISSUE_NUMBER" --repo "$GITHUB_REPOSITORY" \
|
||||||
|
--comment "All documentation links resolved in [the latest run]($run_url)."
|
||||||
@@ -296,16 +296,18 @@ jobs:
|
|||||||
cargo update -p aws-types --precise 1.3.9
|
cargo update -p aws-types --precise 1.3.9
|
||||||
cargo update -p aws-sigv4 --precise 1.3.5
|
cargo update -p aws-sigv4 --precise 1.3.5
|
||||||
cargo update -p aws-credential-types --precise 1.2.8
|
cargo update -p aws-credential-types --precise 1.2.8
|
||||||
cargo update -p aws-smithy-checksums --precise 0.63.9
|
# aws-smithy-checksums must stay at or above 0.63.13: OpenDAL's S3
|
||||||
|
# service needs crc-fast ~1.9, and older releases pin it to ~1.3.
|
||||||
|
cargo update -p aws-smithy-checksums --precise 0.63.13
|
||||||
cargo update -p aws-smithy-runtime --precise 1.9.3
|
cargo update -p aws-smithy-runtime --precise 1.9.3
|
||||||
cargo update -p aws-smithy-http --precise 0.62.4
|
cargo update -p aws-smithy-http --precise 0.62.6
|
||||||
cargo update -p aws-smithy-eventstream --precise 0.60.12
|
cargo update -p aws-smithy-eventstream --precise 0.60.14
|
||||||
cargo update -p aws-smithy-http-client --precise 1.1.3
|
cargo update -p aws-smithy-http-client --precise 1.1.3
|
||||||
cargo update -p aws-smithy-observability --precise 0.1.4
|
cargo update -p aws-smithy-observability --precise 0.1.4
|
||||||
cargo update -p aws-smithy-query --precise 0.60.8
|
cargo update -p aws-smithy-query --precise 0.60.8
|
||||||
cargo update -p aws-smithy-runtime-api --precise 1.9.1
|
cargo update -p aws-smithy-runtime-api --precise 1.9.3
|
||||||
cargo update -p aws-smithy-async --precise 1.2.6
|
cargo update -p aws-smithy-async --precise 1.2.7
|
||||||
cargo update -p aws-smithy-types --precise 1.3.5
|
cargo update -p aws-smithy-types --precise 1.3.6
|
||||||
cargo update -p aws-smithy-xml --precise 0.60.11
|
cargo update -p aws-smithy-xml --precise 0.60.11
|
||||||
cargo update -p home --precise 0.5.9
|
cargo update -p home --precise 0.5.9
|
||||||
- name: cargo +${{ matrix.msrv }} check
|
- name: cargo +${{ matrix.msrv }} check
|
||||||
|
|||||||
@@ -92,6 +92,8 @@ Python bindings changes:
|
|||||||
* Should use `LOOP.run()` to call the corresponding `AsyncTable` method.
|
* Should use `LOOP.run()` to call the corresponding `AsyncTable` method.
|
||||||
6. Add concrete sync method to `RemoteTable` class in `python/python/lancedb/remote/table.py`.
|
6. Add concrete sync method to `RemoteTable` class in `python/python/lancedb/remote/table.py`.
|
||||||
7. Add unit test in `python/tests/test_table.py`.
|
7. Add unit test in `python/tests/test_table.py`.
|
||||||
|
8. If you added a new public class or module-level function (not just a method on an
|
||||||
|
existing class), expose it in the API reference. See "Python API reference" below.
|
||||||
|
|
||||||
TypeScript bindings changes:
|
TypeScript bindings changes:
|
||||||
|
|
||||||
@@ -103,6 +105,33 @@ TypeScript bindings changes:
|
|||||||
5. Add test in `nodejs/__test__/table.test.ts`.
|
5. Add test in `nodejs/__test__/table.test.ts`.
|
||||||
6. Run `npm run docs` to generate TypeScript documentation.
|
6. Run `npm run docs` to generate TypeScript documentation.
|
||||||
|
|
||||||
|
## Python API reference
|
||||||
|
|
||||||
|
`docs/src/python/python.md` is the entire Python API reference. It is maintained by
|
||||||
|
hand, and anything not listed there is not rendered at all, so new public classes and
|
||||||
|
module-level functions have to be added explicitly. How depends on the module:
|
||||||
|
|
||||||
|
* `lancedb.index`, `lancedb.embeddings`, `lancedb.remote`, and `lancedb.rerankers` are
|
||||||
|
rendered by a single directive each, driven by the module's `__all__`. Add the new
|
||||||
|
name to `__all__` and it appears; forget, and it is silently omitted.
|
||||||
|
* Everything else (`lancedb`, `lancedb.table`, `lancedb.query`, `lancedb.db`, ...) is
|
||||||
|
listed symbol by symbol. Add a `::: lancedb.<module>.<Name>` line to the matching
|
||||||
|
section, and remember that the page separates synchronous and asynchronous APIs.
|
||||||
|
|
||||||
|
Deliberately undocumented: concrete implementations reached through an abstract base
|
||||||
|
(`LanceTable`, `LanceDBConnection`, `RemoteDBConnection`), query base classes already
|
||||||
|
covered by `inherited_members`, and internal helpers.
|
||||||
|
|
||||||
|
Cross-references in docstrings use mkdocstrings syntax, `[text][lancedb.table.Table]`.
|
||||||
|
Plain relative links such as `[Table](Table)` do not resolve. To check your work:
|
||||||
|
|
||||||
|
```shell
|
||||||
|
pip install -r docs/requirements.txt
|
||||||
|
cd docs && PYTHONPATH=. mkdocs build
|
||||||
|
```
|
||||||
|
|
||||||
|
The docs site only builds on pushes to `main`, so this is not covered by PR CI.
|
||||||
|
|
||||||
## Review Guidelines
|
## Review Guidelines
|
||||||
|
|
||||||
Please consider the following when reviewing code contributions.
|
Please consider the following when reviewing code contributions.
|
||||||
|
|||||||
Generated
+327
-306
File diff suppressed because it is too large
Load Diff
+15
-15
@@ -13,20 +13,20 @@ categories = ["database-implementations"]
|
|||||||
rust-version = "1.91.0"
|
rust-version = "1.91.0"
|
||||||
|
|
||||||
[workspace.dependencies]
|
[workspace.dependencies]
|
||||||
lance = { "version" = "=10.0.0-beta.5", default-features = false, "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance = { "version" = "=11.0.0-beta.3", default-features = false, "tag" = "v11.0.0-beta.3", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-core = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-core = { "version" = "=11.0.0-beta.3", "tag" = "v11.0.0-beta.3", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-datagen = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-datagen = { "version" = "=11.0.0-beta.3", "tag" = "v11.0.0-beta.3", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-file = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-file = { "version" = "=11.0.0-beta.3", "tag" = "v11.0.0-beta.3", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-io = { "version" = "=10.0.0-beta.5", default-features = false, "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-io = { "version" = "=11.0.0-beta.3", default-features = false, "tag" = "v11.0.0-beta.3", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-index = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-index = { "version" = "=11.0.0-beta.3", "tag" = "v11.0.0-beta.3", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-linalg = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-linalg = { "version" = "=11.0.0-beta.3", "tag" = "v11.0.0-beta.3", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-namespace = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-namespace = { "version" = "=11.0.0-beta.3", "tag" = "v11.0.0-beta.3", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-namespace-impls = { "version" = "=10.0.0-beta.5", default-features = false, "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-namespace-impls = { "version" = "=11.0.0-beta.3", default-features = false, "tag" = "v11.0.0-beta.3", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-table = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-table = { "version" = "=11.0.0-beta.3", "tag" = "v11.0.0-beta.3", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-testing = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-testing = { "version" = "=11.0.0-beta.3", "tag" = "v11.0.0-beta.3", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-datafusion = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-datafusion = { "version" = "=11.0.0-beta.3", "tag" = "v11.0.0-beta.3", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-encoding = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-encoding = { "version" = "=11.0.0-beta.3", "tag" = "v11.0.0-beta.3", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-arrow = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-arrow = { "version" = "=11.0.0-beta.3", "tag" = "v11.0.0-beta.3", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
ahash = "0.8"
|
ahash = "0.8"
|
||||||
# Note that this one does not include pyarrow
|
# Note that this one does not include pyarrow
|
||||||
arrow = { version = "58.0.0", optional = false }
|
arrow = { version = "58.0.0", optional = false }
|
||||||
@@ -52,7 +52,7 @@ env_logger = "0.11"
|
|||||||
half = { "version" = "2.7.1", default-features = false, features = [
|
half = { "version" = "2.7.1", default-features = false, features = [
|
||||||
"num-traits",
|
"num-traits",
|
||||||
] }
|
] }
|
||||||
futures = "0"
|
futures = "0.3"
|
||||||
log = "0.4"
|
log = "0.4"
|
||||||
metrics = "0.24"
|
metrics = "0.24"
|
||||||
metrics-util = "0.19"
|
metrics-util = "0.19"
|
||||||
|
|||||||
@@ -51,6 +51,11 @@ plugins:
|
|||||||
paths: [../python/python]
|
paths: [../python/python]
|
||||||
options:
|
options:
|
||||||
docstring_style: numpy
|
docstring_style: numpy
|
||||||
|
docstring_options:
|
||||||
|
# Attributes documented in a `Parameters` section, and pydantic
|
||||||
|
# dataclasses whose `__init__` griffe cannot see statically, both
|
||||||
|
# trip this check. It reports nothing actionable here.
|
||||||
|
warn_unknown_params: false
|
||||||
heading_level: 3
|
heading_level: 3
|
||||||
show_signature_annotations: true
|
show_signature_annotations: true
|
||||||
show_root_heading: true
|
show_root_heading: true
|
||||||
|
|||||||
@@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`:
|
|||||||
<dependency>
|
<dependency>
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-core</artifactId>
|
<artifactId>lancedb-core</artifactId>
|
||||||
<version>0.37.1-beta.0</version>
|
<version>0.37.1-beta.1</version>
|
||||||
</dependency>
|
</dependency>
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# Contributing to LanceDB Typescript
|
# Contributing to LanceDB Typescript
|
||||||
|
|
||||||
This document outlines the process for contributing to LanceDB Typescript.
|
This document outlines the process for contributing to LanceDB Typescript.
|
||||||
For general contribution guidelines, see [CONTRIBUTING.md](../CONTRIBUTING.md).
|
For general contribution guidelines, see [CONTRIBUTING.md](https://github.com/lancedb/lancedb/blob/main/CONTRIBUTING.md).
|
||||||
|
|
||||||
## Project layout
|
## Project layout
|
||||||
|
|
||||||
|
|||||||
@@ -25,6 +25,27 @@ the underlying connection has been closed.
|
|||||||
|
|
||||||
## Methods
|
## Methods
|
||||||
|
|
||||||
|
### cancelJob()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract cancelJob(jobId): Promise<boolean>
|
||||||
|
```
|
||||||
|
|
||||||
|
Request cancellation of a server-side job by id.
|
||||||
|
|
||||||
|
Resolves to true if the server accepted the cancellation, false if no
|
||||||
|
such job exists. Cancelling an already-terminal job is a no-op success.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **jobId**: `string`
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`boolean`>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### cloneTable()
|
### cloneTable()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -365,6 +386,26 @@ Drop an existing table.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
### getJob()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract getJob(jobId): Promise<null | JobDescription>
|
||||||
|
```
|
||||||
|
|
||||||
|
Describe a single server-side job by id.
|
||||||
|
|
||||||
|
Resolves to `null` when the server has no such job.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **jobId**: `string`
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`null` \| [`JobDescription`](../interfaces/JobDescription.md)>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### isOpen()
|
### isOpen()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -379,6 +420,62 @@ Return true if the connection has not been closed
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
### job()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract job(jobId): Job
|
||||||
|
```
|
||||||
|
|
||||||
|
A [Job](Job.md) handle for a server-side job by id.
|
||||||
|
|
||||||
|
The handle is constructed without a server round trip; an unknown id
|
||||||
|
surfaces when the handle is used. Dropping the handle has no effect on
|
||||||
|
the job itself.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **jobId**: `string`
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
[`Job`](Job.md)
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### jobHistory()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract jobHistory(jobId?): Promise<Table<any>>
|
||||||
|
```
|
||||||
|
|
||||||
|
The lifecycle event history of a server-side job, as an Arrow table.
|
||||||
|
|
||||||
|
Lists history across all jobs when `jobId` is omitted.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **jobId?**: `string`
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`Table`<`any`>>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### listJobs()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract listJobs(): Promise<JobInfo[]>
|
||||||
|
```
|
||||||
|
|
||||||
|
List server-side jobs across the database's tables.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<[`JobInfo`](../interfaces/JobInfo.md)[]>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### listNamespaces()
|
### listNamespaces()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
|
|||||||
@@ -0,0 +1,83 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / Job
|
||||||
|
|
||||||
|
# Class: Job
|
||||||
|
|
||||||
|
A handle to an operation that may still be running.
|
||||||
|
|
||||||
|
## Constructors
|
||||||
|
|
||||||
|
### new Job()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
new Job(): Job
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
[`Job`](Job.md)
|
||||||
|
|
||||||
|
## Accessors
|
||||||
|
|
||||||
|
### id
|
||||||
|
|
||||||
|
```ts
|
||||||
|
get id(): null | string
|
||||||
|
```
|
||||||
|
|
||||||
|
Identifies the operation on the server that is running it. Operations
|
||||||
|
that run in this process have no server id. The value is opaque.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`null` \| `string`
|
||||||
|
|
||||||
|
## Methods
|
||||||
|
|
||||||
|
### cancel()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
cancel(): Promise<void>
|
||||||
|
```
|
||||||
|
|
||||||
|
Request cancellation. Cancelling a finished operation is a no-op.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`void`>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### status()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
status(): Promise<string>
|
||||||
|
```
|
||||||
|
|
||||||
|
The operation's current lifecycle state: "running", "finished",
|
||||||
|
"failed", or "cancelled".
|
||||||
|
|
||||||
|
A point snapshot; unlike [Job.wait](Job.md#wait) it does not block or reject
|
||||||
|
on a terminal failure state. States a newer server reports that this
|
||||||
|
client version does not know pass through as-is.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`string`>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### wait()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
wait(): Promise<void>
|
||||||
|
```
|
||||||
|
|
||||||
|
Wait until the operation reaches a terminal state.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`void`>
|
||||||
@@ -295,6 +295,29 @@ await table.createIndex("my_float_col");
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
### createIndexAsync()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract createIndexAsync(column, options?): Promise<Job>
|
||||||
|
```
|
||||||
|
|
||||||
|
Create an index, returning a handle to the indexing job.
|
||||||
|
|
||||||
|
The job may already be complete when returned; callers must not assume
|
||||||
|
the index exists until [Job.wait](Job.md#wait) resolves.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **column**: `string`
|
||||||
|
|
||||||
|
* **options?**: `Partial`<[`IndexOptions`](../interfaces/IndexOptions.md)>
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<[`Job`](Job.md)>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### currentBranch()
|
### currentBranch()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -408,9 +431,10 @@ Read the [LsmWriteSpec](../interfaces/LsmWriteSpec.md) currently installed on th
|
|||||||
|
|
||||||
Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
|
Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
|
||||||
spec has been set, or it was removed with [Table#unsetLsmWriteSpec](Table.md#unsetlsmwritespec)).
|
spec has been set, or it was removed with [Table#unsetLsmWriteSpec](Table.md#unsetlsmwritespec)).
|
||||||
The returned spec — including its `maintainedIndexes` and
|
The returned spec mirrors what was passed to
|
||||||
`writerConfigDefaults` — mirrors what was passed to
|
[Table#setLsmWriteSpec](Table.md#setlsmwritespec), except that `maintainedIndexes` always
|
||||||
[Table#setLsmWriteSpec](Table.md#setlsmwritespec).
|
reports the concrete list resolved when the spec was set — `undefined`
|
||||||
|
never round-trips.
|
||||||
|
|
||||||
#### Returns
|
#### Returns
|
||||||
|
|
||||||
@@ -783,6 +807,11 @@ All variants require the table to have an unenforced primary key
|
|||||||
([Table#setUnenforcedPrimaryKey](Table.md#setunenforcedprimarykey)); bucket sharding additionally
|
([Table#setUnenforcedPrimaryKey](Table.md#setunenforcedprimarykey)); bucket sharding additionally
|
||||||
requires it to be the single column being bucketed.
|
requires it to be the single column being bucketed.
|
||||||
|
|
||||||
|
Omitting `maintainedIndexes` maintains every index on the table, resolved
|
||||||
|
here, failing if one cannot be maintained — name them to install anyway.
|
||||||
|
Naming them pins an exact set, and a still-building index is rejected
|
||||||
|
rather than quietly omitted.
|
||||||
|
|
||||||
#### Parameters
|
#### Parameters
|
||||||
|
|
||||||
* **spec**: [`LsmWriteSpec`](../interfaces/LsmWriteSpec.md)
|
* **spec**: [`LsmWriteSpec`](../interfaces/LsmWriteSpec.md)
|
||||||
|
|||||||
@@ -25,6 +25,7 @@
|
|||||||
- [Connection](classes/Connection.md)
|
- [Connection](classes/Connection.md)
|
||||||
- [HeaderProvider](classes/HeaderProvider.md)
|
- [HeaderProvider](classes/HeaderProvider.md)
|
||||||
- [Index](classes/Index.md)
|
- [Index](classes/Index.md)
|
||||||
|
- [Job](classes/Job.md)
|
||||||
- [MakeArrowTableOptions](classes/MakeArrowTableOptions.md)
|
- [MakeArrowTableOptions](classes/MakeArrowTableOptions.md)
|
||||||
- [MatchQuery](classes/MatchQuery.md)
|
- [MatchQuery](classes/MatchQuery.md)
|
||||||
- [MergeInsertBuilder](classes/MergeInsertBuilder.md)
|
- [MergeInsertBuilder](classes/MergeInsertBuilder.md)
|
||||||
@@ -88,6 +89,9 @@
|
|||||||
- [IvfFlatOptions](interfaces/IvfFlatOptions.md)
|
- [IvfFlatOptions](interfaces/IvfFlatOptions.md)
|
||||||
- [IvfPqOptions](interfaces/IvfPqOptions.md)
|
- [IvfPqOptions](interfaces/IvfPqOptions.md)
|
||||||
- [IvfRqOptions](interfaces/IvfRqOptions.md)
|
- [IvfRqOptions](interfaces/IvfRqOptions.md)
|
||||||
|
- [JobDescription](interfaces/JobDescription.md)
|
||||||
|
- [JobFailureInfo](interfaces/JobFailureInfo.md)
|
||||||
|
- [JobInfo](interfaces/JobInfo.md)
|
||||||
- [ListNamespacesOptions](interfaces/ListNamespacesOptions.md)
|
- [ListNamespacesOptions](interfaces/ListNamespacesOptions.md)
|
||||||
- [ListNamespacesResponse](interfaces/ListNamespacesResponse.md)
|
- [ListNamespacesResponse](interfaces/ListNamespacesResponse.md)
|
||||||
- [LsmWriteSpec](interfaces/LsmWriteSpec.md)
|
- [LsmWriteSpec](interfaces/LsmWriteSpec.md)
|
||||||
|
|||||||
@@ -0,0 +1,66 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / JobDescription
|
||||||
|
|
||||||
|
# Interface: JobDescription
|
||||||
|
|
||||||
|
A described job from `Connection.getJob`.
|
||||||
|
|
||||||
|
## Properties
|
||||||
|
|
||||||
|
### creationMs
|
||||||
|
|
||||||
|
```ts
|
||||||
|
creationMs: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
When the job was created, in milliseconds since the epoch.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### failure?
|
||||||
|
|
||||||
|
```ts
|
||||||
|
optional failure: JobFailureInfo;
|
||||||
|
```
|
||||||
|
|
||||||
|
Why the job failed, when the job is failed and the server reports a
|
||||||
|
reason.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### jobId
|
||||||
|
|
||||||
|
```ts
|
||||||
|
jobId: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### jobType
|
||||||
|
|
||||||
|
```ts
|
||||||
|
jobType: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### specJson?
|
||||||
|
|
||||||
|
```ts
|
||||||
|
optional specJson: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
The job-type-specific specification as a JSON string, when present.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### state
|
||||||
|
|
||||||
|
```ts
|
||||||
|
state: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
Lifecycle state: "running", "finished", "failed", or "cancelled".
|
||||||
@@ -0,0 +1,33 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / JobFailureInfo
|
||||||
|
|
||||||
|
# Interface: JobFailureInfo
|
||||||
|
|
||||||
|
The server's account of why a job failed.
|
||||||
|
|
||||||
|
## Properties
|
||||||
|
|
||||||
|
### message?
|
||||||
|
|
||||||
|
```ts
|
||||||
|
optional message: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### phase?
|
||||||
|
|
||||||
|
```ts
|
||||||
|
optional phase: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### retryable?
|
||||||
|
|
||||||
|
```ts
|
||||||
|
optional retryable: boolean;
|
||||||
|
```
|
||||||
@@ -0,0 +1,58 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / JobInfo
|
||||||
|
|
||||||
|
# Interface: JobInfo
|
||||||
|
|
||||||
|
A row from `Connection.listJobs`: one server-side job.
|
||||||
|
|
||||||
|
## Properties
|
||||||
|
|
||||||
|
### createdAtMillis
|
||||||
|
|
||||||
|
```ts
|
||||||
|
createdAtMillis: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
When the job was created, in milliseconds since the epoch.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### jobId
|
||||||
|
|
||||||
|
```ts
|
||||||
|
jobId: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
The job id -- what `Connection.getJob` and `Connection.cancelJob`
|
||||||
|
accept.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### jobType
|
||||||
|
|
||||||
|
```ts
|
||||||
|
jobType: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### state
|
||||||
|
|
||||||
|
```ts
|
||||||
|
state: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
Lifecycle state: "running", "finished", "failed", or "cancelled".
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### table
|
||||||
|
|
||||||
|
```ts
|
||||||
|
table: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
The table the job runs against, without URI or namespace.
|
||||||
@@ -34,7 +34,9 @@ Bucket and identity variants: the sharding column.
|
|||||||
optional maintainedIndexes: string[];
|
optional maintainedIndexes: string[];
|
||||||
```
|
```
|
||||||
|
|
||||||
Names of indexes the MemWAL should keep up to date during writes.
|
Indexes the MemWAL keeps up to date. Omit to maintain every supported
|
||||||
|
index, resolved on install — a snapshot, so indexes created later are not
|
||||||
|
maintained. Pass `[]` for none.
|
||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
|||||||
@@ -44,4 +44,7 @@ The number of rows in the table
|
|||||||
totalBytes: number;
|
totalBytes: number;
|
||||||
```
|
```
|
||||||
|
|
||||||
The total number of bytes in the table
|
The total size, in bytes, of the table's data files, index files, and
|
||||||
|
overlay files
|
||||||
|
|
||||||
|
Read from the manifest, so this excludes deletion files and manifests.
|
||||||
|
|||||||
+114
-49
@@ -26,6 +26,18 @@ is also an [asynchronous API client](#connections-asynchronous).
|
|||||||
|
|
||||||
::: lancedb.db.DBConnection
|
::: lancedb.db.DBConnection
|
||||||
|
|
||||||
|
::: lancedb.Session
|
||||||
|
|
||||||
|
## Namespaces (Synchronous)
|
||||||
|
|
||||||
|
A namespace-backed connection resolves tables through a
|
||||||
|
[Lance namespace](https://lance-format.github.io/lance-namespace/) service instead of
|
||||||
|
listing a storage directory.
|
||||||
|
|
||||||
|
::: lancedb.connect_namespace
|
||||||
|
|
||||||
|
::: lancedb.namespace.LanceNamespaceDBConnection
|
||||||
|
|
||||||
## Tables (Synchronous)
|
## Tables (Synchronous)
|
||||||
|
|
||||||
::: lancedb.table.Table
|
::: lancedb.table.Table
|
||||||
@@ -34,8 +46,12 @@ is also an [asynchronous API client](#connections-asynchronous).
|
|||||||
|
|
||||||
::: lancedb.table.FragmentSummaryStats
|
::: lancedb.table.FragmentSummaryStats
|
||||||
|
|
||||||
|
::: lancedb.table.TableStatistics
|
||||||
|
|
||||||
::: lancedb.table.Tags
|
::: lancedb.table.Tags
|
||||||
|
|
||||||
|
::: lancedb.table.Branches
|
||||||
|
|
||||||
## Expressions
|
## Expressions
|
||||||
|
|
||||||
Type-safe expression builder for filters and projections. Use these instead
|
Type-safe expression builder for filters and projections. Use these instead
|
||||||
@@ -62,29 +78,46 @@ of raw SQL strings with [where][lancedb.query.LanceQueryBuilder.where] and
|
|||||||
|
|
||||||
::: lancedb.query.LanceHybridQueryBuilder
|
::: lancedb.query.LanceHybridQueryBuilder
|
||||||
|
|
||||||
|
::: lancedb.query.LanceEmptyQueryBuilder
|
||||||
|
|
||||||
|
::: lancedb.query.LanceTakeQueryBuilder
|
||||||
|
|
||||||
|
## Full text queries
|
||||||
|
|
||||||
|
Structured full text queries can be passed to
|
||||||
|
[Table.search][lancedb.table.Table.search] or
|
||||||
|
[AsyncTable.search][lancedb.table.AsyncTable.search] in place of a query string,
|
||||||
|
and combined with [BooleanQuery][lancedb.query.BooleanQuery].
|
||||||
|
|
||||||
|
::: lancedb.query.FullTextQuery
|
||||||
|
|
||||||
|
::: lancedb.query.MatchQuery
|
||||||
|
|
||||||
|
::: lancedb.query.PhraseQuery
|
||||||
|
|
||||||
|
::: lancedb.query.BoostQuery
|
||||||
|
|
||||||
|
::: lancedb.query.MultiMatchQuery
|
||||||
|
|
||||||
|
::: lancedb.query.BooleanQuery
|
||||||
|
|
||||||
|
::: lancedb.query.FullTextOperator
|
||||||
|
|
||||||
|
::: lancedb.query.Occur
|
||||||
|
|
||||||
## Embeddings
|
## Embeddings
|
||||||
|
|
||||||
::: lancedb.embeddings.registry.EmbeddingFunctionRegistry
|
::: lancedb.embeddings
|
||||||
|
options:
|
||||||
::: lancedb.embeddings.base.EmbeddingFunctionConfig
|
show_root_heading: false
|
||||||
|
show_root_toc_entry: false
|
||||||
::: lancedb.embeddings.base.EmbeddingFunction
|
|
||||||
|
|
||||||
::: lancedb.embeddings.base.TextEmbeddingFunction
|
|
||||||
|
|
||||||
::: lancedb.embeddings.sentence_transformers.SentenceTransformerEmbeddings
|
|
||||||
|
|
||||||
::: lancedb.embeddings.openai.OpenAIEmbeddings
|
|
||||||
|
|
||||||
::: lancedb.embeddings.open_clip.OpenClipEmbeddings
|
|
||||||
|
|
||||||
## Remote configuration
|
## Remote configuration
|
||||||
|
|
||||||
::: lancedb.remote.ClientConfig
|
::: lancedb.remote
|
||||||
|
options:
|
||||||
::: lancedb.remote.TimeoutConfig
|
show_root_heading: false
|
||||||
|
show_root_toc_entry: false
|
||||||
::: lancedb.remote.RetryConfig
|
|
||||||
|
|
||||||
## Context
|
## Context
|
||||||
|
|
||||||
@@ -122,7 +155,22 @@ tokens = list(lancedb.tokenize("acme makes searchable data",
|
|||||||
custom_stop_words=["acme"]))
|
custom_stop_words=["acme"]))
|
||||||
```
|
```
|
||||||
|
|
||||||
::: lancedb.index.FTS
|
::: lancedb.tokenize
|
||||||
|
|
||||||
|
::: lancedb.FtsToken
|
||||||
|
|
||||||
|
## Blobs
|
||||||
|
|
||||||
|
Blob columns store large binary values out of line so they can be read lazily
|
||||||
|
instead of being materialized with the rest of the row.
|
||||||
|
|
||||||
|
::: lancedb.blob
|
||||||
|
|
||||||
|
::: lancedb.BlobType
|
||||||
|
|
||||||
|
::: lancedb._blob.BlobFile
|
||||||
|
options:
|
||||||
|
show_root_full_path: false
|
||||||
|
|
||||||
## Utilities
|
## Utilities
|
||||||
|
|
||||||
@@ -130,6 +178,14 @@ tokens = list(lancedb.tokenize("acme makes searchable data",
|
|||||||
|
|
||||||
::: lancedb.merge.LanceMergeInsertBuilder
|
::: lancedb.merge.LanceMergeInsertBuilder
|
||||||
|
|
||||||
|
::: lancedb.otel.instrument_lancedb_metrics
|
||||||
|
|
||||||
|
## Exceptions
|
||||||
|
|
||||||
|
::: lancedb.exceptions.MissingValueError
|
||||||
|
|
||||||
|
::: lancedb.exceptions.MissingColumnError
|
||||||
|
|
||||||
## Integrations
|
## Integrations
|
||||||
|
|
||||||
## Pydantic
|
## Pydantic
|
||||||
@@ -138,19 +194,30 @@ tokens = list(lancedb.tokenize("acme makes searchable data",
|
|||||||
|
|
||||||
::: lancedb.pydantic.vector
|
::: lancedb.pydantic.vector
|
||||||
|
|
||||||
|
::: lancedb.pydantic.Vector
|
||||||
|
|
||||||
|
::: lancedb.pydantic.MultiVector
|
||||||
|
|
||||||
::: lancedb.pydantic.LanceModel
|
::: lancedb.pydantic.LanceModel
|
||||||
|
|
||||||
|
## PyTorch
|
||||||
|
|
||||||
|
::: lancedb.streaming.StreamingDataset
|
||||||
|
|
||||||
|
::: lancedb.permutation.permutation_builder
|
||||||
|
|
||||||
|
::: lancedb.permutation.PermutationBuilder
|
||||||
|
|
||||||
|
::: lancedb.permutation.Permutation
|
||||||
|
|
||||||
|
::: lancedb.permutation.Transforms
|
||||||
|
|
||||||
## Reranking
|
## Reranking
|
||||||
|
|
||||||
::: lancedb.rerankers.linear_combination.LinearCombinationReranker
|
::: lancedb.rerankers
|
||||||
|
options:
|
||||||
::: lancedb.rerankers.cohere.CohereReranker
|
show_root_heading: false
|
||||||
|
show_root_toc_entry: false
|
||||||
::: lancedb.rerankers.colbert.ColbertReranker
|
|
||||||
|
|
||||||
::: lancedb.rerankers.cross_encoder.CrossEncoderReranker
|
|
||||||
|
|
||||||
::: lancedb.rerankers.openai.OpenaiReranker
|
|
||||||
|
|
||||||
## Connections (Asynchronous)
|
## Connections (Asynchronous)
|
||||||
|
|
||||||
@@ -161,6 +228,12 @@ can be used to create, list, or open tables.
|
|||||||
|
|
||||||
::: lancedb.db.AsyncConnection
|
::: lancedb.db.AsyncConnection
|
||||||
|
|
||||||
|
## Namespaces (Asynchronous)
|
||||||
|
|
||||||
|
::: lancedb.connect_namespace_async
|
||||||
|
|
||||||
|
::: lancedb.namespace.AsyncLanceNamespaceDBConnection
|
||||||
|
|
||||||
## Tables (Asynchronous)
|
## Tables (Asynchronous)
|
||||||
|
|
||||||
Table hold your actual data as a collection of records / rows.
|
Table hold your actual data as a collection of records / rows.
|
||||||
@@ -169,32 +242,20 @@ Table hold your actual data as a collection of records / rows.
|
|||||||
|
|
||||||
::: lancedb.table.AsyncTags
|
::: lancedb.table.AsyncTags
|
||||||
|
|
||||||
|
::: lancedb.table.AsyncBranches
|
||||||
|
|
||||||
## Indices (Asynchronous)
|
## Indices (Asynchronous)
|
||||||
|
|
||||||
Indices can be created on a table to speed up queries. This section
|
Indices can be created on a table to speed up queries. This section
|
||||||
lists the indices that LanceDb supports.
|
lists the indices that LanceDb supports.
|
||||||
|
|
||||||
::: lancedb.index.BTree
|
::: lancedb.index
|
||||||
|
options:
|
||||||
::: lancedb.index.Bitmap
|
show_root_heading: false
|
||||||
|
show_root_toc_entry: false
|
||||||
::: lancedb.index.LabelList
|
# `lang_mapping` is defined in the module rather than imported, so it is
|
||||||
|
# picked up despite not being in `__all__`. It is an internal lookup table.
|
||||||
::: lancedb.index.FTS
|
filters: ["!^_", "!^lang_mapping$"]
|
||||||
|
|
||||||
::: lancedb.index.IvfPq
|
|
||||||
|
|
||||||
::: lancedb.index.HnswPq
|
|
||||||
|
|
||||||
::: lancedb.index.HnswSq
|
|
||||||
|
|
||||||
::: lancedb.index.IvfFlat
|
|
||||||
|
|
||||||
::: lancedb.index.IvfSq
|
|
||||||
|
|
||||||
::: lancedb.index.IvfRq
|
|
||||||
|
|
||||||
::: lancedb.index.HnswFlat
|
|
||||||
|
|
||||||
::: lancedb.table.IndexStatistics
|
::: lancedb.table.IndexStatistics
|
||||||
|
|
||||||
@@ -222,3 +283,7 @@ rows nearest to a query vector and can be created with the
|
|||||||
::: lancedb.query.AsyncHybridQuery
|
::: lancedb.query.AsyncHybridQuery
|
||||||
options:
|
options:
|
||||||
inherited_members: true
|
inherited_members: true
|
||||||
|
|
||||||
|
::: lancedb.query.AsyncTakeQuery
|
||||||
|
options:
|
||||||
|
inherited_members: true
|
||||||
|
|||||||
@@ -8,7 +8,7 @@
|
|||||||
<parent>
|
<parent>
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-parent</artifactId>
|
<artifactId>lancedb-parent</artifactId>
|
||||||
<version>0.37.1-beta.0</version>
|
<version>0.37.1-beta.1</version>
|
||||||
<relativePath>../pom.xml</relativePath>
|
<relativePath>../pom.xml</relativePath>
|
||||||
</parent>
|
</parent>
|
||||||
|
|
||||||
|
|||||||
+2
-2
@@ -6,7 +6,7 @@
|
|||||||
|
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-parent</artifactId>
|
<artifactId>lancedb-parent</artifactId>
|
||||||
<version>0.37.1-beta.0</version>
|
<version>0.37.1-beta.1</version>
|
||||||
<packaging>pom</packaging>
|
<packaging>pom</packaging>
|
||||||
<name>${project.artifactId}</name>
|
<name>${project.artifactId}</name>
|
||||||
<description>LanceDB Java SDK Parent POM</description>
|
<description>LanceDB Java SDK Parent POM</description>
|
||||||
@@ -28,7 +28,7 @@
|
|||||||
<properties>
|
<properties>
|
||||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||||
<arrow.version>15.0.0</arrow.version>
|
<arrow.version>15.0.0</arrow.version>
|
||||||
<lance-core.version>10.0.0-beta.5</lance-core.version>
|
<lance-core.version>11.0.0-beta.3</lance-core.version>
|
||||||
<spotless.skip>false</spotless.skip>
|
<spotless.skip>false</spotless.skip>
|
||||||
<spotless.version>2.30.0</spotless.version>
|
<spotless.version>2.30.0</spotless.version>
|
||||||
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
|
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# Contributing to LanceDB Typescript
|
# Contributing to LanceDB Typescript
|
||||||
|
|
||||||
This document outlines the process for contributing to LanceDB Typescript.
|
This document outlines the process for contributing to LanceDB Typescript.
|
||||||
For general contribution guidelines, see [CONTRIBUTING.md](../CONTRIBUTING.md).
|
For general contribution guidelines, see [CONTRIBUTING.md](https://github.com/lancedb/lancedb/blob/main/CONTRIBUTING.md).
|
||||||
|
|
||||||
## Project layout
|
## Project layout
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "lancedb-nodejs"
|
name = "lancedb-nodejs"
|
||||||
edition.workspace = true
|
edition.workspace = true
|
||||||
version = "0.37.1-beta.0"
|
version = "0.37.1-beta.1"
|
||||||
publish = false
|
publish = false
|
||||||
license.workspace = true
|
license.workspace = true
|
||||||
description.workspace = true
|
description.workspace = true
|
||||||
|
|||||||
@@ -197,6 +197,35 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
expect(table.getChild("d")?.toJSON()).toEqual([9n, 10n, null]);
|
expect(table.getChild("d")?.toJSON()).toEqual([9n, 10n, null]);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("will use a provided FixedSizeList schema with typed array values", function () {
|
||||||
|
const schema = new Schema([
|
||||||
|
new Field("text", new Utf8(), false),
|
||||||
|
new Field(
|
||||||
|
"vector",
|
||||||
|
new FixedSizeList(3, new Field("item", new Float32(), false)),
|
||||||
|
false,
|
||||||
|
),
|
||||||
|
]);
|
||||||
|
|
||||||
|
const table = makeArrowTable(
|
||||||
|
[
|
||||||
|
{
|
||||||
|
text: "foo",
|
||||||
|
vector: new Float32Array([1, 2, 3]),
|
||||||
|
},
|
||||||
|
],
|
||||||
|
{ schema },
|
||||||
|
);
|
||||||
|
|
||||||
|
expect(table.getChild("text")?.toJSON()).toEqual(["foo"]);
|
||||||
|
expect(
|
||||||
|
table
|
||||||
|
.getChild("vector")
|
||||||
|
?.toJSON()
|
||||||
|
.map((value) => value.toJSON()),
|
||||||
|
).toEqual([[1, 2, 3]]);
|
||||||
|
});
|
||||||
|
|
||||||
it("will assume the column `vector` is FixedSizeList<Float32> by default", async function () {
|
it("will assume the column `vector` is FixedSizeList<Float32> by default", async function () {
|
||||||
const schema = new Schema([
|
const schema = new Schema([
|
||||||
new Field("a", new Float(Precision.DOUBLE), true),
|
new Field("a", new Float(Precision.DOUBLE), true),
|
||||||
|
|||||||
@@ -11,8 +11,11 @@ import {
|
|||||||
Float16,
|
Float16,
|
||||||
Float32,
|
Float32,
|
||||||
Float64,
|
Float64,
|
||||||
|
Int32,
|
||||||
Schema,
|
Schema,
|
||||||
Utf8,
|
Utf8,
|
||||||
|
fromDataToBuffer,
|
||||||
|
tableFromIPC,
|
||||||
} from "../lancedb/arrow";
|
} from "../lancedb/arrow";
|
||||||
import { EmbeddingFunction, LanceSchema } from "../lancedb/embedding";
|
import { EmbeddingFunction, LanceSchema } from "../lancedb/embedding";
|
||||||
import { getRegistry, register } from "../lancedb/embedding/registry";
|
import { getRegistry, register } from "../lancedb/embedding/registry";
|
||||||
@@ -184,6 +187,63 @@ describe("embedding functions", () => {
|
|||||||
const vector0 = JSON.parse(JSON.stringify(arr[0].vector));
|
const vector0 = JSON.parse(JSON.stringify(arr[0].vector));
|
||||||
expect(vector0).toEqual([1, 2, 3]);
|
expect(vector0).toEqual([1, 2, 3]);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("should append generated vectors to a non-nullable schema", async () => {
|
||||||
|
@register("non_nullable_schema_test")
|
||||||
|
class MockEmbeddingFunction extends EmbeddingFunction<string> {
|
||||||
|
ndims() {
|
||||||
|
return 3;
|
||||||
|
}
|
||||||
|
embeddingDataType(): Float {
|
||||||
|
return new Float64();
|
||||||
|
}
|
||||||
|
async computeSourceEmbeddings(data: string[]) {
|
||||||
|
return data.map(() => [1, 2, 3]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const schema = new Schema([
|
||||||
|
new Field("id", new Int32()),
|
||||||
|
new Field("text", new Utf8()),
|
||||||
|
new Field("type", new Utf8()),
|
||||||
|
new Field(
|
||||||
|
"vector",
|
||||||
|
new FixedSizeList(3, new Field("item", new Float64())),
|
||||||
|
),
|
||||||
|
]);
|
||||||
|
const func = new MockEmbeddingFunction();
|
||||||
|
const db = await connect(tmpDir.name);
|
||||||
|
const table = await db.createEmptyTable("test_non_nullable", schema, {
|
||||||
|
embeddingFunction: {
|
||||||
|
function: func,
|
||||||
|
sourceColumn: "text",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
const data = [
|
||||||
|
{ id: 1, text: "Carrot", type: "vegetable" },
|
||||||
|
{ id: 2, text: "Apple", type: "fruit" },
|
||||||
|
];
|
||||||
|
const buffer = await fromDataToBuffer(
|
||||||
|
data,
|
||||||
|
undefined,
|
||||||
|
await table.schema(),
|
||||||
|
);
|
||||||
|
const generatedTable = tableFromIPC(buffer);
|
||||||
|
const vectorField = generatedTable.schema.fields.find(
|
||||||
|
(field) => field.name === "vector",
|
||||||
|
);
|
||||||
|
expect(vectorField?.nullable).toBe(false);
|
||||||
|
|
||||||
|
await table.add(data);
|
||||||
|
|
||||||
|
const rows = await table.query().toArray();
|
||||||
|
expect(rows).toHaveLength(2);
|
||||||
|
for (const row of rows) {
|
||||||
|
expect([...row.vector]).toEqual([1, 2, 3]);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
it("should error when appending to a table with an unregistered embedding function", async () => {
|
it("should error when appending to a table with an unregistered embedding function", async () => {
|
||||||
@register("mock")
|
@register("mock")
|
||||||
class MockEmbeddingFunction extends EmbeddingFunction<string> {
|
class MockEmbeddingFunction extends EmbeddingFunction<string> {
|
||||||
|
|||||||
@@ -0,0 +1,14 @@
|
|||||||
|
// SPDX-License-Identifier: Apache-2.0
|
||||||
|
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
import packageJson = require("../package.json");
|
||||||
|
|
||||||
|
describe("package metadata", () => {
|
||||||
|
it("requires Node.js type declarations compatible with the runtime", () => {
|
||||||
|
expect(packageJson.engines.node).toBe(">= 18");
|
||||||
|
expect(packageJson.peerDependencies["@types/node"]).toBe(">=18");
|
||||||
|
expect(packageJson.peerDependenciesMeta["@types/node"]).toEqual({
|
||||||
|
optional: true,
|
||||||
|
});
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -110,6 +110,81 @@ describe("Query outputSchema", () => {
|
|||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
describe("Search pagination", () => {
|
||||||
|
let tmpDir: tmp.DirResult;
|
||||||
|
let table: Table;
|
||||||
|
|
||||||
|
beforeEach(async () => {
|
||||||
|
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
||||||
|
const db = await connect(tmpDir.name);
|
||||||
|
const schema = new Schema([
|
||||||
|
new Field("id", new Int64(), false),
|
||||||
|
new Field("text", new Utf8(), false),
|
||||||
|
new Field(
|
||||||
|
"vector",
|
||||||
|
new FixedSizeList(2, new Field("item", new Float32())),
|
||||||
|
false,
|
||||||
|
),
|
||||||
|
]);
|
||||||
|
const data = makeArrowTable(
|
||||||
|
[
|
||||||
|
{ id: 1n, text: "common", vector: [0, 0] },
|
||||||
|
{ id: 2n, text: "common common", vector: [1, 1] },
|
||||||
|
{ id: 3n, text: "common common common", vector: [2, 2] },
|
||||||
|
{ id: 4n, text: "common common common common", vector: [3, 3] },
|
||||||
|
],
|
||||||
|
{ schema },
|
||||||
|
);
|
||||||
|
table = await db.createTable("test", data);
|
||||||
|
});
|
||||||
|
|
||||||
|
afterEach(() => {
|
||||||
|
tmpDir.removeCallback();
|
||||||
|
});
|
||||||
|
|
||||||
|
it("applies offset after the vector search limit", async () => {
|
||||||
|
const allResults = await table
|
||||||
|
.vectorSearch([0, 0])
|
||||||
|
.select(["id"])
|
||||||
|
.limit(4)
|
||||||
|
.toArray();
|
||||||
|
const secondPage = await table
|
||||||
|
.vectorSearch([0, 0])
|
||||||
|
.select(["id"])
|
||||||
|
.limit(2)
|
||||||
|
.offset(2)
|
||||||
|
.toArray();
|
||||||
|
|
||||||
|
expect(allResults).toHaveLength(4);
|
||||||
|
expect(secondPage).toHaveLength(2);
|
||||||
|
expect(secondPage.map((row) => row.id)).toEqual(
|
||||||
|
allResults.slice(2, 4).map((row) => row.id),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("applies offset after the full-text search limit", async () => {
|
||||||
|
await table.createIndex("text", { config: Index.fts() });
|
||||||
|
|
||||||
|
const allResults = await table
|
||||||
|
.search("common", "fts")
|
||||||
|
.select(["id"])
|
||||||
|
.limit(4)
|
||||||
|
.toArray();
|
||||||
|
const secondPage = await table
|
||||||
|
.search("common", "fts")
|
||||||
|
.select(["id"])
|
||||||
|
.limit(2)
|
||||||
|
.offset(2)
|
||||||
|
.toArray();
|
||||||
|
|
||||||
|
expect(allResults).toHaveLength(4);
|
||||||
|
expect(secondPage).toHaveLength(2);
|
||||||
|
expect(secondPage.map((row) => row.id)).toEqual(
|
||||||
|
allResults.slice(2, 4).map((row) => row.id),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
describe("Query orderBy", () => {
|
describe("Query orderBy", () => {
|
||||||
let tmpDir: tmp.DirResult;
|
let tmpDir: tmp.DirResult;
|
||||||
let table: Table;
|
let table: Table;
|
||||||
|
|||||||
@@ -170,6 +170,38 @@ describe("remote connection", () => {
|
|||||||
);
|
);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("surfaces JSON server errors from remote table operations", async () => {
|
||||||
|
await withMockDatabase(
|
||||||
|
(req, res) => {
|
||||||
|
const path = req.url ?? "";
|
||||||
|
if (path.endsWith("/describe/")) {
|
||||||
|
res.writeHead(200, { "Content-Type": "application/json" }).end(
|
||||||
|
JSON.stringify({
|
||||||
|
name: "broken_table",
|
||||||
|
version: 1,
|
||||||
|
schema: { fields: [] },
|
||||||
|
}),
|
||||||
|
);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (path.endsWith("/count_rows/")) {
|
||||||
|
res
|
||||||
|
.writeHead(400, { "Content-Type": "application/json" })
|
||||||
|
.end(JSON.stringify({ error: "count rows failed" }));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
res.writeHead(404).end();
|
||||||
|
},
|
||||||
|
async (db) => {
|
||||||
|
const table = await db.openTable("broken_table");
|
||||||
|
|
||||||
|
await expect(table.countRows()).rejects.toThrow("count rows failed");
|
||||||
|
},
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
it("should pass on requested extra headers", async () => {
|
it("should pass on requested extra headers", async () => {
|
||||||
await withMockDatabase(
|
await withMockDatabase(
|
||||||
(req, res) => {
|
(req, res) => {
|
||||||
@@ -877,3 +909,96 @@ describe("remote connection", () => {
|
|||||||
});
|
});
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
describe("remote connection jobs surface", () => {
|
||||||
|
it("lists, describes, cancels, and reads history", async () => {
|
||||||
|
const { tableFromArrays, tableToIPC } = await import("apache-arrow");
|
||||||
|
const eventsTable = tableFromArrays({ state: ["created", "succeeded"] });
|
||||||
|
const eventsBody = Buffer.from(tableToIPC(eventsTable, "stream"));
|
||||||
|
|
||||||
|
await withMockDatabase(
|
||||||
|
(req, res) => {
|
||||||
|
let body = "";
|
||||||
|
req.on("data", (chunk) => {
|
||||||
|
body += chunk;
|
||||||
|
});
|
||||||
|
req.on("end", () => {
|
||||||
|
const payload = body.length > 0 ? JSON.parse(body) : {};
|
||||||
|
if (req.url === "/v1/jobs/list") {
|
||||||
|
if (payload["page_token"] === undefined) {
|
||||||
|
res
|
||||||
|
.writeHead(200, { "Content-Type": "application/json" })
|
||||||
|
.end(
|
||||||
|
'{"jobs": [{"job_id": "job-1", "table": "t1", ' +
|
||||||
|
'"job_type": "create_index", "state": "in_progress", ' +
|
||||||
|
'"created_at_millis": 1000}], "page_token": "next"}',
|
||||||
|
);
|
||||||
|
} else {
|
||||||
|
res
|
||||||
|
.writeHead(200, { "Content-Type": "application/json" })
|
||||||
|
.end(
|
||||||
|
'{"jobs": [{"job_id": "job-2", "table": "t2", ' +
|
||||||
|
'"job_type": "create_index", "state": "succeeded", ' +
|
||||||
|
'"created_at_millis": 2000}]}',
|
||||||
|
);
|
||||||
|
}
|
||||||
|
} else if (req.url === "/v1/jobs/describe") {
|
||||||
|
if (payload["job_id"] !== "job-1") {
|
||||||
|
res.writeHead(404).end("no such job");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
res
|
||||||
|
.writeHead(200, { "Content-Type": "application/json" })
|
||||||
|
.end(
|
||||||
|
'{"job_id": "job-1", "job_type": "create_index", ' +
|
||||||
|
'"job_state": "FAILED", "creation_ms": 1000, ' +
|
||||||
|
'"spec": {"column": "vec"}, "failure": {"phase": "execute", ' +
|
||||||
|
'"message": "worker died", "retryable": true}}',
|
||||||
|
);
|
||||||
|
} else if (req.url === "/v1/jobs/cancel") {
|
||||||
|
if (payload["job_id"] !== "job-1") {
|
||||||
|
res.writeHead(404).end("no such job");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
res
|
||||||
|
.writeHead(200, { "Content-Type": "application/json" })
|
||||||
|
.end('{"job_id": "job-1"}');
|
||||||
|
} else if (req.url === "/v1/jobs/query_events") {
|
||||||
|
res
|
||||||
|
.writeHead(200, {
|
||||||
|
"Content-Type": "application/vnd.apache.arrow.stream",
|
||||||
|
})
|
||||||
|
.end(eventsBody);
|
||||||
|
} else {
|
||||||
|
res.writeHead(404).end();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
},
|
||||||
|
async (db) => {
|
||||||
|
const jobs = await db.listJobs();
|
||||||
|
expect(jobs.map((job) => job.jobId)).toEqual(["job-1", "job-2"]);
|
||||||
|
expect(jobs[0].state).toEqual("running");
|
||||||
|
expect(jobs[1].state).toEqual("finished");
|
||||||
|
|
||||||
|
const description = await db.getJob("job-1");
|
||||||
|
expect(description?.state).toEqual("failed");
|
||||||
|
expect(JSON.parse(description?.specJson ?? "")).toEqual({
|
||||||
|
column: "vec",
|
||||||
|
});
|
||||||
|
expect(description?.failure?.message).toEqual("worker died");
|
||||||
|
expect(await db.getJob("missing")).toBeNull();
|
||||||
|
|
||||||
|
expect(await db.cancelJob("job-1")).toBe(true);
|
||||||
|
expect(await db.cancelJob("missing")).toBe(false);
|
||||||
|
|
||||||
|
const history = await db.jobHistory("job-1");
|
||||||
|
expect(history.numRows).toEqual(2);
|
||||||
|
|
||||||
|
const job = db.job("job-1");
|
||||||
|
expect(job.id).toEqual("job-1");
|
||||||
|
expect(await job.status()).toEqual("failed");
|
||||||
|
await expect(job.wait()).rejects.toThrow("worker died");
|
||||||
|
},
|
||||||
|
);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|||||||
@@ -86,6 +86,44 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
await expect(table.countRows()).resolves.toBe(3);
|
await expect(table.countRows()).resolves.toBe(3);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("should support a foreign Float64 vector schema end to end", async () => {
|
||||||
|
const conn = await connect(tmpDir.name);
|
||||||
|
const schema = new arrow.Schema([
|
||||||
|
new arrow.Field("resource_id", new arrow.Int32(), false),
|
||||||
|
new arrow.Field(
|
||||||
|
"vector",
|
||||||
|
new arrow.FixedSizeList(
|
||||||
|
3,
|
||||||
|
new arrow.Field("value", new arrow.Float64(), true),
|
||||||
|
),
|
||||||
|
false,
|
||||||
|
),
|
||||||
|
]);
|
||||||
|
const data = [
|
||||||
|
{
|
||||||
|
// biome-ignore lint/style/useNamingConvention: matches the reported schema
|
||||||
|
resource_id: 0,
|
||||||
|
vector: [0.1, 0.1, 0.1],
|
||||||
|
},
|
||||||
|
];
|
||||||
|
|
||||||
|
const resources = await conn.createTable("resources", data, { schema });
|
||||||
|
|
||||||
|
const existing = await resources
|
||||||
|
.query()
|
||||||
|
.where("resource_id = 0")
|
||||||
|
.limit(1)
|
||||||
|
.toArray();
|
||||||
|
expect(existing).toHaveLength(1);
|
||||||
|
|
||||||
|
const matched = await resources
|
||||||
|
.search(Float64Array.from(data[0].vector))
|
||||||
|
.limit(1)
|
||||||
|
.toArray();
|
||||||
|
expect(matched).toHaveLength(1);
|
||||||
|
expect(matched[0]["resource_id"]).toBe(0);
|
||||||
|
});
|
||||||
|
|
||||||
it("should support branches", async () => {
|
it("should support branches", async () => {
|
||||||
await table.add([{ id: 1 }]);
|
await table.add([{ id: 1 }]);
|
||||||
expect(await table.countRows()).toBe(1);
|
expect(await table.countRows()).toBe(1);
|
||||||
@@ -239,8 +277,16 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
},
|
},
|
||||||
numIndices: 0,
|
numIndices: 0,
|
||||||
numRows: 3,
|
numRows: 3,
|
||||||
totalBytes: 44,
|
// Full on-disk size of the two data files, footers and metadata included.
|
||||||
|
totalBytes: 684,
|
||||||
});
|
});
|
||||||
|
|
||||||
|
// Index files count toward totalBytes too (only deletion files and
|
||||||
|
// manifests are excluded).
|
||||||
|
await table.createIndex("id", { config: Index.btree() });
|
||||||
|
const statsWithIndex = await table.stats();
|
||||||
|
expect(statsWithIndex.numIndices).toBe(1);
|
||||||
|
expect(statsWithIndex.totalBytes).toBeGreaterThan(684);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("should overwrite data if asked", async () => {
|
it("should overwrite data if asked", async () => {
|
||||||
@@ -851,7 +897,11 @@ describe("When creating an index", () => {
|
|||||||
afterEach(() => tmpDir.removeCallback());
|
afterEach(() => tmpDir.removeCallback());
|
||||||
|
|
||||||
it("should create a vector index on vector columns", async () => {
|
it("should create a vector index on vector columns", async () => {
|
||||||
await tbl.createIndex("vec");
|
const job = await tbl.createIndexAsync("vec");
|
||||||
|
expect(job.id).toBeNull();
|
||||||
|
await job.wait();
|
||||||
|
// Cancelling a job that already finished succeeds and does nothing.
|
||||||
|
await job.cancel();
|
||||||
|
|
||||||
// check index directory
|
// check index directory
|
||||||
const indexDir = path.join(tmpDir.name, "test.lance", "_indices");
|
const indexDir = path.join(tmpDir.name, "test.lance", "_indices");
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
// SPDX-License-Identifier: Apache-2.0
|
// SPDX-License-Identifier: Apache-2.0
|
||||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
import { tableFromIPC } from "apache-arrow";
|
||||||
import {
|
import {
|
||||||
Data,
|
Data,
|
||||||
SchemaLike,
|
SchemaLike,
|
||||||
@@ -20,6 +21,9 @@ import type {
|
|||||||
CreateNamespaceResponse,
|
CreateNamespaceResponse,
|
||||||
DescribeNamespaceResponse,
|
DescribeNamespaceResponse,
|
||||||
DropNamespaceResponse,
|
DropNamespaceResponse,
|
||||||
|
Job,
|
||||||
|
JobDescription,
|
||||||
|
JobInfo,
|
||||||
ListNamespacesResponse,
|
ListNamespacesResponse,
|
||||||
} from "./native";
|
} from "./native";
|
||||||
export type {
|
export type {
|
||||||
@@ -436,6 +440,40 @@ export abstract class Connection {
|
|||||||
newName: string,
|
newName: string,
|
||||||
options?: RenameTableOptions,
|
options?: RenameTableOptions,
|
||||||
): Promise<void>;
|
): Promise<void>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A {@link Job} handle for a server-side job by id.
|
||||||
|
*
|
||||||
|
* The handle is constructed without a server round trip; an unknown id
|
||||||
|
* surfaces when the handle is used. Dropping the handle has no effect on
|
||||||
|
* the job itself.
|
||||||
|
*/
|
||||||
|
abstract job(jobId: string): Job;
|
||||||
|
|
||||||
|
/** List server-side jobs across the database's tables. */
|
||||||
|
abstract listJobs(): Promise<JobInfo[]>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Describe a single server-side job by id.
|
||||||
|
*
|
||||||
|
* Resolves to `null` when the server has no such job.
|
||||||
|
*/
|
||||||
|
abstract getJob(jobId: string): Promise<JobDescription | null>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Request cancellation of a server-side job by id.
|
||||||
|
*
|
||||||
|
* Resolves to true if the server accepted the cancellation, false if no
|
||||||
|
* such job exists. Cancelling an already-terminal job is a no-op success.
|
||||||
|
*/
|
||||||
|
abstract cancelJob(jobId: string): Promise<boolean>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The lifecycle event history of a server-side job, as an Arrow table.
|
||||||
|
*
|
||||||
|
* Lists history across all jobs when `jobId` is omitted.
|
||||||
|
*/
|
||||||
|
abstract jobHistory(jobId?: string): Promise<ArrowTable>;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** @hideconstructor */
|
/** @hideconstructor */
|
||||||
@@ -722,6 +760,30 @@ export class LocalConnection extends Connection {
|
|||||||
options?.newNamespacePath,
|
options?.newNamespacePath,
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
job(jobId: string): Job {
|
||||||
|
return this.inner.job(jobId);
|
||||||
|
}
|
||||||
|
|
||||||
|
async listJobs(): Promise<JobInfo[]> {
|
||||||
|
return this.inner.listJobs();
|
||||||
|
}
|
||||||
|
|
||||||
|
async getJob(jobId: string): Promise<JobDescription | null> {
|
||||||
|
return this.inner.getJob(jobId);
|
||||||
|
}
|
||||||
|
|
||||||
|
async cancelJob(jobId: string): Promise<boolean> {
|
||||||
|
return this.inner.cancelJob(jobId);
|
||||||
|
}
|
||||||
|
|
||||||
|
async jobHistory(jobId?: string): Promise<ArrowTable> {
|
||||||
|
const buf = await this.inner.jobHistory(jobId);
|
||||||
|
if (buf.length === 0) {
|
||||||
|
return new ArrowTable();
|
||||||
|
}
|
||||||
|
return tableFromIPC(buf);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
@@ -85,7 +85,13 @@ export {
|
|||||||
RenameTableOptions,
|
RenameTableOptions,
|
||||||
} from "./connection";
|
} from "./connection";
|
||||||
|
|
||||||
export { Session } from "./native.js";
|
export {
|
||||||
|
Job,
|
||||||
|
JobDescription,
|
||||||
|
JobFailureInfo,
|
||||||
|
JobInfo,
|
||||||
|
Session,
|
||||||
|
} from "./native.js";
|
||||||
|
|
||||||
export {
|
export {
|
||||||
ExecutableQuery,
|
ExecutableQuery,
|
||||||
|
|||||||
+42
-4
@@ -30,6 +30,7 @@ import {
|
|||||||
DropColumnsResult,
|
DropColumnsResult,
|
||||||
IndexConfig,
|
IndexConfig,
|
||||||
IndexStatistics,
|
IndexStatistics,
|
||||||
|
Job,
|
||||||
Branches as NativeBranches,
|
Branches as NativeBranches,
|
||||||
OptimizeStats,
|
OptimizeStats,
|
||||||
TableStatistics,
|
TableStatistics,
|
||||||
@@ -196,7 +197,11 @@ export interface LsmWriteSpec {
|
|||||||
column?: string;
|
column?: string;
|
||||||
/** Bucket variant: the number of buckets, in `[1, 1024]`. */
|
/** Bucket variant: the number of buckets, in `[1, 1024]`. */
|
||||||
numBuckets?: number;
|
numBuckets?: number;
|
||||||
/** Names of indexes the MemWAL should keep up to date during writes. */
|
/**
|
||||||
|
* Indexes the MemWAL keeps up to date. Omit to maintain every supported
|
||||||
|
* index, resolved on install — a snapshot, so indexes created later are not
|
||||||
|
* maintained. Pass `[]` for none.
|
||||||
|
*/
|
||||||
maintainedIndexes?: string[];
|
maintainedIndexes?: string[];
|
||||||
/** Default `ShardWriter` configuration recorded in the MemWAL index. */
|
/** Default `ShardWriter` configuration recorded in the MemWAL index. */
|
||||||
writerConfigDefaults?: Record<string, string>;
|
writerConfigDefaults?: Record<string, string>;
|
||||||
@@ -358,6 +363,17 @@ export abstract class Table {
|
|||||||
options?: Partial<IndexOptions>,
|
options?: Partial<IndexOptions>,
|
||||||
): Promise<void>;
|
): Promise<void>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Create an index, returning a handle to the indexing job.
|
||||||
|
*
|
||||||
|
* The job may already be complete when returned; callers must not assume
|
||||||
|
* the index exists until {@link Job.wait} resolves.
|
||||||
|
*/
|
||||||
|
abstract createIndexAsync(
|
||||||
|
column: string,
|
||||||
|
options?: Partial<IndexOptions>,
|
||||||
|
): Promise<Job>;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Drop an index from the table.
|
* Drop an index from the table.
|
||||||
*
|
*
|
||||||
@@ -583,6 +599,11 @@ export abstract class Table {
|
|||||||
* All variants require the table to have an unenforced primary key
|
* All variants require the table to have an unenforced primary key
|
||||||
* ({@link Table#setUnenforcedPrimaryKey}); bucket sharding additionally
|
* ({@link Table#setUnenforcedPrimaryKey}); bucket sharding additionally
|
||||||
* requires it to be the single column being bucketed.
|
* requires it to be the single column being bucketed.
|
||||||
|
*
|
||||||
|
* Omitting `maintainedIndexes` maintains every index on the table, resolved
|
||||||
|
* here, failing if one cannot be maintained — name them to install anyway.
|
||||||
|
* Naming them pins an exact set, and a still-building index is rejected
|
||||||
|
* rather than quietly omitted.
|
||||||
* @param {LsmWriteSpec} spec The sharding spec to install.
|
* @param {LsmWriteSpec} spec The sharding spec to install.
|
||||||
* @returns {Promise<void>}
|
* @returns {Promise<void>}
|
||||||
* @example
|
* @example
|
||||||
@@ -610,9 +631,10 @@ export abstract class Table {
|
|||||||
*
|
*
|
||||||
* Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
|
* Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
|
||||||
* spec has been set, or it was removed with {@link Table#unsetLsmWriteSpec}).
|
* spec has been set, or it was removed with {@link Table#unsetLsmWriteSpec}).
|
||||||
* The returned spec — including its `maintainedIndexes` and
|
* The returned spec mirrors what was passed to
|
||||||
* `writerConfigDefaults` — mirrors what was passed to
|
* {@link Table#setLsmWriteSpec}, except that `maintainedIndexes` always
|
||||||
* {@link Table#setLsmWriteSpec}.
|
* reports the concrete list resolved when the spec was set — `undefined`
|
||||||
|
* never round-trips.
|
||||||
* @returns {Promise<LsmWriteSpec | undefined>}
|
* @returns {Promise<LsmWriteSpec | undefined>}
|
||||||
*/
|
*/
|
||||||
abstract getLsmWriteSpec(): Promise<LsmWriteSpec | undefined>;
|
abstract getLsmWriteSpec(): Promise<LsmWriteSpec | undefined>;
|
||||||
@@ -940,6 +962,22 @@ export class LocalTable extends Table {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async createIndexAsync(
|
||||||
|
column: string,
|
||||||
|
options?: Partial<IndexOptions>,
|
||||||
|
): Promise<Job> {
|
||||||
|
// biome-ignore lint/suspicious/noExplicitAny: skip
|
||||||
|
const nativeIndex = (options?.config as any)?.inner;
|
||||||
|
return await this.inner.createIndexAsync(
|
||||||
|
nativeIndex,
|
||||||
|
column,
|
||||||
|
options?.replace,
|
||||||
|
options?.waitTimeoutSeconds,
|
||||||
|
options?.name,
|
||||||
|
options?.train,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
async dropIndex(name: string): Promise<void> {
|
async dropIndex(name: string): Promise<void> {
|
||||||
await this.inner.dropIndex(name);
|
await this.inner.dropIndex(name);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-darwin-arm64",
|
"name": "@lancedb/lancedb-darwin-arm64",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.37.1-beta.1",
|
||||||
"os": ["darwin"],
|
"os": ["darwin"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.darwin-arm64.node",
|
"main": "lancedb.darwin-arm64.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.37.1-beta.1",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.linux-arm64-gnu.node",
|
"main": "lancedb.linux-arm64-gnu.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-arm64-musl",
|
"name": "@lancedb/lancedb-linux-arm64-musl",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.37.1-beta.1",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.linux-arm64-musl.node",
|
"main": "lancedb.linux-arm64-musl.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-x64-gnu",
|
"name": "@lancedb/lancedb-linux-x64-gnu",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.37.1-beta.1",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["x64"],
|
"cpu": ["x64"],
|
||||||
"main": "lancedb.linux-x64-gnu.node",
|
"main": "lancedb.linux-x64-gnu.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-x64-musl",
|
"name": "@lancedb/lancedb-linux-x64-musl",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.37.1-beta.1",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["x64"],
|
"cpu": ["x64"],
|
||||||
"main": "lancedb.linux-x64-musl.node",
|
"main": "lancedb.linux-x64-musl.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-win32-arm64-msvc",
|
"name": "@lancedb/lancedb-win32-arm64-msvc",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.37.1-beta.1",
|
||||||
"os": [
|
"os": [
|
||||||
"win32"
|
"win32"
|
||||||
],
|
],
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-win32-x64-msvc",
|
"name": "@lancedb/lancedb-win32-x64-msvc",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.37.1-beta.1",
|
||||||
"os": ["win32"],
|
"os": ["win32"],
|
||||||
"cpu": ["x64"],
|
"cpu": ["x64"],
|
||||||
"main": "lancedb.win32-x64-msvc.node",
|
"main": "lancedb.win32-x64-msvc.node",
|
||||||
|
|||||||
Generated
+8
-2
@@ -1,12 +1,12 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb",
|
"name": "@lancedb/lancedb",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.37.1-beta.1",
|
||||||
"lockfileVersion": 3,
|
"lockfileVersion": 3,
|
||||||
"requires": true,
|
"requires": true,
|
||||||
"packages": {
|
"packages": {
|
||||||
"": {
|
"": {
|
||||||
"name": "@lancedb/lancedb",
|
"name": "@lancedb/lancedb",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.37.1-beta.1",
|
||||||
"cpu": [
|
"cpu": [
|
||||||
"x64",
|
"x64",
|
||||||
"arm64"
|
"arm64"
|
||||||
@@ -55,7 +55,13 @@
|
|||||||
"openai": "4.29.2"
|
"openai": "4.29.2"
|
||||||
},
|
},
|
||||||
"peerDependencies": {
|
"peerDependencies": {
|
||||||
|
"@types/node": ">=18",
|
||||||
"apache-arrow": ">=15.0.0 <=18.1.0"
|
"apache-arrow": ">=15.0.0 <=18.1.0"
|
||||||
|
},
|
||||||
|
"peerDependenciesMeta": {
|
||||||
|
"@types/node": {
|
||||||
|
"optional": true
|
||||||
|
}
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
"node_modules/@aws-crypto/crc32": {
|
"node_modules/@aws-crypto/crc32": {
|
||||||
|
|||||||
+7
-1
@@ -11,7 +11,7 @@
|
|||||||
"ann"
|
"ann"
|
||||||
],
|
],
|
||||||
"private": false,
|
"private": false,
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.37.1-beta.1",
|
||||||
"main": "dist/index.js",
|
"main": "dist/index.js",
|
||||||
"exports": {
|
"exports": {
|
||||||
".": "./dist/index.js",
|
".": "./dist/index.js",
|
||||||
@@ -101,6 +101,12 @@
|
|||||||
"openai": "4.29.2"
|
"openai": "4.29.2"
|
||||||
},
|
},
|
||||||
"peerDependencies": {
|
"peerDependencies": {
|
||||||
|
"@types/node": ">=18",
|
||||||
"apache-arrow": ">=15.0.0 <=18.1.0"
|
"apache-arrow": ">=15.0.0 <=18.1.0"
|
||||||
|
},
|
||||||
|
"peerDependenciesMeta": {
|
||||||
|
"@types/node": {
|
||||||
|
"optional": true
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -340,6 +340,69 @@ impl Connection {
|
|||||||
self.get_inner()?.drop_all_tables(&ns).await.default_error()
|
self.get_inner()?.drop_all_tables(&ns).await.default_error()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A `Job` handle for a server-side job by id.
|
||||||
|
///
|
||||||
|
/// The handle is constructed without a server round trip; an unknown id
|
||||||
|
/// surfaces when the handle is used.
|
||||||
|
#[napi]
|
||||||
|
pub fn job(&self, job_id: String) -> napi::Result<crate::job::Job> {
|
||||||
|
let job = self.get_inner()?.job(job_id).default_error()?;
|
||||||
|
Ok(crate::job::Job::new(job))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// List server-side jobs across the database's tables.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn list_jobs(&self) -> napi::Result<Vec<crate::job::JobInfo>> {
|
||||||
|
let jobs = self.get_inner()?.list_jobs().await.default_error()?;
|
||||||
|
Ok(jobs.into_iter().map(Into::into).collect())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Describe a single server-side job by id. `null` when the server has
|
||||||
|
/// no such job.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn get_job(
|
||||||
|
&self,
|
||||||
|
job_id: String,
|
||||||
|
) -> napi::Result<Option<crate::job::JobDescription>> {
|
||||||
|
let description = self.get_inner()?.get_job(&job_id).await.default_error()?;
|
||||||
|
Ok(description.map(Into::into))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Request cancellation of a server-side job by id. Returns true if the
|
||||||
|
/// server accepted the cancellation, false if no such job exists.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn cancel_job(&self, job_id: String) -> napi::Result<bool> {
|
||||||
|
self.get_inner()?.cancel_job(&job_id).await.default_error()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The lifecycle event history of a server-side job (all jobs when
|
||||||
|
/// `job_id` is null), as an Arrow IPC stream buffer. Empty when there is
|
||||||
|
/// no history.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn job_history(&self, job_id: Option<String>) -> napi::Result<Buffer> {
|
||||||
|
let batches = self
|
||||||
|
.get_inner()?
|
||||||
|
.job_history(job_id.as_deref())
|
||||||
|
.await
|
||||||
|
.default_error()?;
|
||||||
|
let Some(first) = batches.first() else {
|
||||||
|
return Ok(Buffer::from(Vec::<u8>::new()));
|
||||||
|
};
|
||||||
|
let mut out = Vec::new();
|
||||||
|
let mut writer = arrow_ipc::writer::StreamWriter::try_new(&mut out, &first.schema())
|
||||||
|
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
|
||||||
|
for batch in &batches {
|
||||||
|
writer
|
||||||
|
.write(batch)
|
||||||
|
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
|
||||||
|
}
|
||||||
|
writer
|
||||||
|
.finish()
|
||||||
|
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
|
||||||
|
drop(writer);
|
||||||
|
Ok(Buffer::from(out))
|
||||||
|
}
|
||||||
|
|
||||||
#[napi(catch_unwind)]
|
#[napi(catch_unwind)]
|
||||||
/// Describe a namespace and return its properties.
|
/// Describe a namespace and return its properties.
|
||||||
pub async fn describe_namespace(
|
pub async fn describe_namespace(
|
||||||
|
|||||||
@@ -0,0 +1,123 @@
|
|||||||
|
// SPDX-License-Identifier: Apache-2.0
|
||||||
|
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
use napi_derive::napi;
|
||||||
|
|
||||||
|
use crate::error::NapiErrorExt;
|
||||||
|
|
||||||
|
/// A handle to an operation that may still be running.
|
||||||
|
#[napi]
|
||||||
|
pub struct Job {
|
||||||
|
inner: Arc<lancedb::Job>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Job {
|
||||||
|
pub(crate) fn new(inner: lancedb::Job) -> Self {
|
||||||
|
Self {
|
||||||
|
inner: Arc::new(inner),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[napi]
|
||||||
|
impl Job {
|
||||||
|
/// Identifies the operation on the server that is running it. Operations
|
||||||
|
/// that run in this process have no server id. The value is opaque.
|
||||||
|
#[napi(getter)]
|
||||||
|
pub fn id(&self) -> Option<String> {
|
||||||
|
self.inner.id().map(str::to_string)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The operation's current lifecycle state: "running", "finished",
|
||||||
|
/// "failed", or "cancelled".
|
||||||
|
///
|
||||||
|
/// A point snapshot; unlike {@link Job.wait} it does not block or reject
|
||||||
|
/// on a terminal failure state. States a newer server reports that this
|
||||||
|
/// client version does not know pass through as-is.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn status(&self) -> napi::Result<String> {
|
||||||
|
self.inner.status().await.default_error()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Wait until the operation reaches a terminal state.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn wait(&self) -> napi::Result<()> {
|
||||||
|
self.inner.wait().await.default_error()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Request cancellation. Cancelling a finished operation is a no-op.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn cancel(&self) -> napi::Result<()> {
|
||||||
|
self.inner.cancel().await.default_error()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A row from `Connection.listJobs`: one server-side job.
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct JobInfo {
|
||||||
|
/// The job id -- what `Connection.getJob` and `Connection.cancelJob`
|
||||||
|
/// accept.
|
||||||
|
pub job_id: String,
|
||||||
|
/// The table the job runs against, without URI or namespace.
|
||||||
|
pub table: String,
|
||||||
|
pub job_type: String,
|
||||||
|
/// Lifecycle state: "running", "finished", "failed", or "cancelled".
|
||||||
|
pub state: String,
|
||||||
|
/// When the job was created, in milliseconds since the epoch.
|
||||||
|
pub created_at_millis: i64,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<lancedb::database::JobInfo> for JobInfo {
|
||||||
|
fn from(info: lancedb::database::JobInfo) -> Self {
|
||||||
|
Self {
|
||||||
|
job_id: info.job_id,
|
||||||
|
table: info.table,
|
||||||
|
job_type: info.job_type,
|
||||||
|
state: info.state,
|
||||||
|
created_at_millis: info.created_at_millis,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The server's account of why a job failed.
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct JobFailureInfo {
|
||||||
|
pub phase: Option<String>,
|
||||||
|
pub message: Option<String>,
|
||||||
|
pub retryable: Option<bool>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A described job from `Connection.getJob`.
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct JobDescription {
|
||||||
|
pub job_id: String,
|
||||||
|
pub job_type: String,
|
||||||
|
/// Lifecycle state: "running", "finished", "failed", or "cancelled".
|
||||||
|
pub state: String,
|
||||||
|
/// When the job was created, in milliseconds since the epoch.
|
||||||
|
pub creation_ms: i64,
|
||||||
|
/// The job-type-specific specification as a JSON string, when present.
|
||||||
|
pub spec_json: Option<String>,
|
||||||
|
/// Why the job failed, when the job is failed and the server reports a
|
||||||
|
/// reason.
|
||||||
|
pub failure: Option<JobFailureInfo>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<lancedb::database::JobDescription> for JobDescription {
|
||||||
|
fn from(description: lancedb::database::JobDescription) -> Self {
|
||||||
|
Self {
|
||||||
|
job_id: description.job_id,
|
||||||
|
job_type: description.job_type,
|
||||||
|
state: description.state,
|
||||||
|
creation_ms: description.creation_ms,
|
||||||
|
spec_json: (!description.spec.is_null()).then(|| description.spec.to_string()),
|
||||||
|
failure: description.failure.map(|failure| JobFailureInfo {
|
||||||
|
phase: failure.phase,
|
||||||
|
message: failure.message,
|
||||||
|
retryable: failure.retryable,
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -11,6 +11,7 @@ mod error;
|
|||||||
mod header;
|
mod header;
|
||||||
mod index;
|
mod index;
|
||||||
mod iterator;
|
mod iterator;
|
||||||
|
mod job;
|
||||||
pub mod merge;
|
pub mod merge;
|
||||||
pub mod otel;
|
pub mod otel;
|
||||||
pub mod permutation;
|
pub mod permutation;
|
||||||
|
|||||||
+49
-9
@@ -168,6 +168,39 @@ impl Table {
|
|||||||
builder.execute().await.default_error()
|
builder.execute().await.default_error()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn create_index_async(
|
||||||
|
&self,
|
||||||
|
index: Option<&Index>,
|
||||||
|
column: String,
|
||||||
|
replace: Option<bool>,
|
||||||
|
wait_timeout_s: Option<i64>,
|
||||||
|
name: Option<String>,
|
||||||
|
train: Option<bool>,
|
||||||
|
) -> napi::Result<crate::job::Job> {
|
||||||
|
let lancedb_index = if let Some(index) = index {
|
||||||
|
index.consume()?
|
||||||
|
} else {
|
||||||
|
lancedb::index::Index::Auto
|
||||||
|
};
|
||||||
|
let mut builder = self.inner_ref()?.create_index(&[column], lancedb_index);
|
||||||
|
if let Some(replace) = replace {
|
||||||
|
builder = builder.replace(replace);
|
||||||
|
}
|
||||||
|
if let Some(timeout) = wait_timeout_s {
|
||||||
|
builder =
|
||||||
|
builder.wait_timeout(std::time::Duration::from_secs(timeout.try_into().unwrap()));
|
||||||
|
}
|
||||||
|
if let Some(name) = name {
|
||||||
|
builder = builder.name(name);
|
||||||
|
}
|
||||||
|
if let Some(train) = train {
|
||||||
|
builder = builder.train(train);
|
||||||
|
}
|
||||||
|
let job = builder.execute_async().await.default_error()?;
|
||||||
|
Ok(crate::job::Job::new(job))
|
||||||
|
}
|
||||||
|
|
||||||
#[napi(catch_unwind)]
|
#[napi(catch_unwind)]
|
||||||
pub async fn drop_index(&self, index_name: String) -> napi::Result<()> {
|
pub async fn drop_index(&self, index_name: String) -> napi::Result<()> {
|
||||||
self.inner_ref()?
|
self.inner_ref()?
|
||||||
@@ -306,7 +339,9 @@ impl Table {
|
|||||||
let transforms = NewColumnTransform::SqlExpressions(transforms);
|
let transforms = NewColumnTransform::SqlExpressions(transforms);
|
||||||
let res = self
|
let res = self
|
||||||
.inner_ref()?
|
.inner_ref()?
|
||||||
.add_columns(transforms, None)
|
.add_columns()
|
||||||
|
.transform(transforms)
|
||||||
|
.execute()
|
||||||
.await
|
.await
|
||||||
.default_error()?;
|
.default_error()?;
|
||||||
Ok(res.into())
|
Ok(res.into())
|
||||||
@@ -323,7 +358,9 @@ impl Table {
|
|||||||
let transforms = NewColumnTransform::AllNulls(schema);
|
let transforms = NewColumnTransform::AllNulls(schema);
|
||||||
let res = self
|
let res = self
|
||||||
.inner_ref()?
|
.inner_ref()?
|
||||||
.add_columns(transforms, None)
|
.add_columns()
|
||||||
|
.transform(transforms)
|
||||||
|
.execute()
|
||||||
.await
|
.await
|
||||||
.default_error()?;
|
.default_error()?;
|
||||||
Ok(res.into())
|
Ok(res.into())
|
||||||
@@ -735,7 +772,8 @@ pub struct LsmWriteSpec {
|
|||||||
pub column: Option<String>,
|
pub column: Option<String>,
|
||||||
/// Bucket variant: the number of buckets, in `[1, 1024]`.
|
/// Bucket variant: the number of buckets, in `[1, 1024]`.
|
||||||
pub num_buckets: Option<u32>,
|
pub num_buckets: Option<u32>,
|
||||||
/// Names of indexes the MemWAL should keep up to date during writes.
|
/// Indexes the MemWAL keeps up to date. Omitted resolves every
|
||||||
|
/// maintainable index on install; an empty array means none.
|
||||||
pub maintained_indexes: Option<Vec<String>>,
|
pub maintained_indexes: Option<Vec<String>>,
|
||||||
/// Default `ShardWriter` configuration recorded in the MemWAL index.
|
/// Default `ShardWriter` configuration recorded in the MemWAL index.
|
||||||
pub writer_config_defaults: Option<HashMap<String, String>>,
|
pub writer_config_defaults: Option<HashMap<String, String>>,
|
||||||
@@ -745,7 +783,6 @@ impl TryFrom<LsmWriteSpec> for lancedb::table::LsmWriteSpec {
|
|||||||
type Error = napi::Error;
|
type Error = napi::Error;
|
||||||
|
|
||||||
fn try_from(value: LsmWriteSpec) -> napi::Result<Self> {
|
fn try_from(value: LsmWriteSpec) -> napi::Result<Self> {
|
||||||
let maintained = value.maintained_indexes.unwrap_or_default();
|
|
||||||
let writer_config_defaults = value.writer_config_defaults.unwrap_or_default();
|
let writer_config_defaults = value.writer_config_defaults.unwrap_or_default();
|
||||||
let spec = match value.spec_type.as_str() {
|
let spec = match value.spec_type.as_str() {
|
||||||
"bucket" => {
|
"bucket" => {
|
||||||
@@ -772,7 +809,7 @@ impl TryFrom<LsmWriteSpec> for lancedb::table::LsmWriteSpec {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
Ok(spec
|
Ok(spec
|
||||||
.with_maintained_indexes(maintained)
|
.with_maintained_indexes(value.maintained_indexes)
|
||||||
.with_writer_config_defaults(writer_config_defaults))
|
.with_writer_config_defaults(writer_config_defaults))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -790,7 +827,7 @@ impl From<lancedb::table::LsmWriteSpec> for LsmWriteSpec {
|
|||||||
spec_type: "bucket".to_string(),
|
spec_type: "bucket".to_string(),
|
||||||
column: Some(column),
|
column: Some(column),
|
||||||
num_buckets: Some(num_buckets),
|
num_buckets: Some(num_buckets),
|
||||||
maintained_indexes: Some(maintained_indexes),
|
maintained_indexes,
|
||||||
writer_config_defaults: Some(writer_config_defaults),
|
writer_config_defaults: Some(writer_config_defaults),
|
||||||
},
|
},
|
||||||
Native::Identity {
|
Native::Identity {
|
||||||
@@ -801,7 +838,7 @@ impl From<lancedb::table::LsmWriteSpec> for LsmWriteSpec {
|
|||||||
spec_type: "identity".to_string(),
|
spec_type: "identity".to_string(),
|
||||||
column: Some(column),
|
column: Some(column),
|
||||||
num_buckets: None,
|
num_buckets: None,
|
||||||
maintained_indexes: Some(maintained_indexes),
|
maintained_indexes,
|
||||||
writer_config_defaults: Some(writer_config_defaults),
|
writer_config_defaults: Some(writer_config_defaults),
|
||||||
},
|
},
|
||||||
Native::Unsharded {
|
Native::Unsharded {
|
||||||
@@ -811,7 +848,7 @@ impl From<lancedb::table::LsmWriteSpec> for LsmWriteSpec {
|
|||||||
spec_type: "unsharded".to_string(),
|
spec_type: "unsharded".to_string(),
|
||||||
column: None,
|
column: None,
|
||||||
num_buckets: None,
|
num_buckets: None,
|
||||||
maintained_indexes: Some(maintained_indexes),
|
maintained_indexes,
|
||||||
writer_config_defaults: Some(writer_config_defaults),
|
writer_config_defaults: Some(writer_config_defaults),
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
@@ -1006,7 +1043,10 @@ impl From<lancedb::index::IndexStatistics> for IndexStatistics {
|
|||||||
|
|
||||||
#[napi(object)]
|
#[napi(object)]
|
||||||
pub struct TableStatistics {
|
pub struct TableStatistics {
|
||||||
/// The total number of bytes in the table
|
/// The total size, in bytes, of the table's data files, index files, and
|
||||||
|
/// overlay files
|
||||||
|
///
|
||||||
|
/// Read from the manifest, so this excludes deletion files and manifests.
|
||||||
pub total_bytes: i64,
|
pub total_bytes: i64,
|
||||||
|
|
||||||
/// The number of rows in the table
|
/// The number of rows in the table
|
||||||
|
|||||||
+3
-3
@@ -1,6 +1,6 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "lancedb-python"
|
name = "lancedb-python"
|
||||||
version = "0.37.1-beta.0"
|
version = "0.37.1-beta.1"
|
||||||
publish = false
|
publish = false
|
||||||
edition.workspace = true
|
edition.workspace = true
|
||||||
description = "Python bindings for LanceDB"
|
description = "Python bindings for LanceDB"
|
||||||
@@ -26,7 +26,7 @@ lance-namespace-impls.workspace = true
|
|||||||
lance-io.workspace = true
|
lance-io.workspace = true
|
||||||
env_logger.workspace = true
|
env_logger.workspace = true
|
||||||
log.workspace = true
|
log.workspace = true
|
||||||
pyo3 = { version = "0.28", features = ["extension-module", "abi3-py39", "chrono"] }
|
pyo3 = { version = "0.28", features = ["extension-module", "abi3-py310", "chrono"] }
|
||||||
chrono = { version = "0.4", default-features = false, features = ["clock"] }
|
chrono = { version = "0.4", default-features = false, features = ["clock"] }
|
||||||
pyo3-async-runtimes = { version = "0.28", features = [
|
pyo3-async-runtimes = { version = "0.28", features = [
|
||||||
"attributes",
|
"attributes",
|
||||||
@@ -43,7 +43,7 @@ libc = "0.2"
|
|||||||
[build-dependencies]
|
[build-dependencies]
|
||||||
pyo3-build-config = { version = "0.28", features = [
|
pyo3-build-config = { version = "0.28", features = [
|
||||||
"extension-module",
|
"extension-module",
|
||||||
"abi3-py39",
|
"abi3-py310",
|
||||||
] }
|
] }
|
||||||
|
|
||||||
[features]
|
[features]
|
||||||
|
|||||||
@@ -60,7 +60,7 @@ tests = [
|
|||||||
"pytest-asyncio>=0.21",
|
"pytest-asyncio>=0.21",
|
||||||
"duckdb>=0.9.0",
|
"duckdb>=0.9.0",
|
||||||
"pytz>=2023.3",
|
"pytz>=2023.3",
|
||||||
"polars>=0.19, <=1.3.0",
|
"polars>=0.19, <=1.32.3",
|
||||||
"pyarrow<25",
|
"pyarrow<25",
|
||||||
"pyarrow-stubs>=16.0",
|
"pyarrow-stubs>=16.0",
|
||||||
"pylance==9.0.0rc1",
|
"pylance==9.0.0rc1",
|
||||||
@@ -140,6 +140,7 @@ include = [
|
|||||||
"python/lancedb/remote/errors.py",
|
"python/lancedb/remote/errors.py",
|
||||||
"python/lancedb/embeddings/__init__.py",
|
"python/lancedb/embeddings/__init__.py",
|
||||||
"python/lancedb/_lancedb.pyi",
|
"python/lancedb/_lancedb.pyi",
|
||||||
|
"python/type_tests/connect.py",
|
||||||
]
|
]
|
||||||
exclude = ["python/tests/"]
|
exclude = ["python/tests/"]
|
||||||
pythonVersion = "3.13"
|
pythonVersion = "3.13"
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ from .remote import ClientConfig
|
|||||||
from .remote.db import RemoteDBConnection
|
from .remote.db import RemoteDBConnection
|
||||||
from .expr import Expr, col, lit, func
|
from .expr import Expr, col, lit, func
|
||||||
from .schema import blob, vector, BlobType
|
from .schema import blob, vector, BlobType
|
||||||
|
from .job import AsyncJob, Job
|
||||||
from .table import AsyncTable, Table
|
from .table import AsyncTable, Table
|
||||||
from .types import BaseTokenizerType
|
from .types import BaseTokenizerType
|
||||||
from ._lancedb import Session
|
from ._lancedb import Session
|
||||||
@@ -500,6 +501,7 @@ __all__ = [
|
|||||||
"connect_namespace",
|
"connect_namespace",
|
||||||
"connect_namespace_async",
|
"connect_namespace_async",
|
||||||
"AsyncConnection",
|
"AsyncConnection",
|
||||||
|
"AsyncJob",
|
||||||
"AsyncLanceNamespaceDBConnection",
|
"AsyncLanceNamespaceDBConnection",
|
||||||
"AsyncTable",
|
"AsyncTable",
|
||||||
"FtsToken",
|
"FtsToken",
|
||||||
@@ -513,6 +515,7 @@ __all__ = [
|
|||||||
"BlobType",
|
"BlobType",
|
||||||
"vector",
|
"vector",
|
||||||
"DBConnection",
|
"DBConnection",
|
||||||
|
"Job",
|
||||||
"LanceDBConnection",
|
"LanceDBConnection",
|
||||||
"LanceNamespaceDBConnection",
|
"LanceNamespaceDBConnection",
|
||||||
"RemoteDBConnection",
|
"RemoteDBConnection",
|
||||||
|
|||||||
@@ -14,14 +14,10 @@ import pyarrow as pa
|
|||||||
from .expr import Expr
|
from .expr import Expr
|
||||||
from .schema import blob_v2_column_paths
|
from .schema import blob_v2_column_paths
|
||||||
from .types import BlobMode, QueryProjection, QueryProjectionSpec
|
from .types import BlobMode, QueryProjection, QueryProjectionSpec
|
||||||
from .util import get_uri_scheme
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
from _typeshed import WriteableBuffer
|
from _typeshed import WriteableBuffer
|
||||||
|
|
||||||
from .remote.table import RemoteTable
|
|
||||||
from .table import AsyncTable, Table
|
|
||||||
|
|
||||||
BLOB_MODE_TO_HANDLING = {
|
BLOB_MODE_TO_HANDLING = {
|
||||||
"lazy": "blobs_descriptions",
|
"lazy": "blobs_descriptions",
|
||||||
"bytes": "all_binary",
|
"bytes": "all_binary",
|
||||||
@@ -104,22 +100,6 @@ def validate_blob_mode(blob_mode: BlobMode) -> None:
|
|||||||
raise ValueError(f"blob_mode must be one of {modes}, got {blob_mode!r}")
|
raise ValueError(f"blob_mode must be one of {modes}, got {blob_mode!r}")
|
||||||
|
|
||||||
|
|
||||||
def supports_blob_auto_row_id(table: Table | AsyncTable | RemoteTable) -> bool:
|
|
||||||
"""Blob auto row-id applies to native tables, not LanceDB Cloud."""
|
|
||||||
from .remote.table import RemoteTable
|
|
||||||
|
|
||||||
if isinstance(table, RemoteTable):
|
|
||||||
return False
|
|
||||||
|
|
||||||
inner = getattr(table, "_inner", None)
|
|
||||||
if inner is not None:
|
|
||||||
uri = inner.database().uri
|
|
||||||
if isinstance(uri, str) and get_uri_scheme(uri) == "db":
|
|
||||||
return False
|
|
||||||
|
|
||||||
return True
|
|
||||||
|
|
||||||
|
|
||||||
def projection_includes_blob_column(
|
def projection_includes_blob_column(
|
||||||
projection: QueryProjection,
|
projection: QueryProjection,
|
||||||
blob_columns: Iterable[str],
|
blob_columns: Iterable[str],
|
||||||
@@ -164,16 +144,14 @@ def v2_projection_needs_row_id(
|
|||||||
|
|
||||||
|
|
||||||
def blob_auto_row_id_for_scan(
|
def blob_auto_row_id_for_scan(
|
||||||
table: Table | AsyncTable | RemoteTable,
|
|
||||||
schema: pa.Schema,
|
schema: pa.Schema,
|
||||||
projection: QueryProjection,
|
projection: QueryProjection,
|
||||||
*,
|
*,
|
||||||
with_row_id: bool | None,
|
with_row_id: bool | None,
|
||||||
) -> bool:
|
) -> bool:
|
||||||
|
"""Auto row-id only applies when the caller said nothing about row ids."""
|
||||||
if with_row_id is not None:
|
if with_row_id is not None:
|
||||||
return False
|
return False
|
||||||
if not supports_blob_auto_row_id(table):
|
|
||||||
return False
|
|
||||||
return v2_projection_needs_row_id(schema, projection, with_row_id=False)
|
return v2_projection_needs_row_id(schema, projection, with_row_id=False)
|
||||||
|
|
||||||
|
|
||||||
@@ -186,6 +164,11 @@ def finalize_blob_query_table(
|
|||||||
) -> pa.Table:
|
) -> pa.Table:
|
||||||
if user_requested_row_id or not blob_auto_row_id:
|
if user_requested_row_id or not blob_auto_row_id:
|
||||||
return tbl
|
return tbl
|
||||||
|
if "_rowid" not in tbl.column_names:
|
||||||
|
# A backend that ignores the row-id request leaves nothing to stash. Hand
|
||||||
|
# back the projection as-is so fetch_blobs raises the error that names the
|
||||||
|
# ways to supply row ids, rather than failing here about a hidden column.
|
||||||
|
return tbl
|
||||||
return stash_auto_row_ids(tbl, blob_paths)
|
return stash_auto_row_ids(tbl, blob_paths)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -146,6 +146,13 @@ class Connection(object):
|
|||||||
start_after: Optional[str],
|
start_after: Optional[str],
|
||||||
limit: Optional[int],
|
limit: Optional[int],
|
||||||
) -> list[str]: ... # Deprecated: Use list_tables instead
|
) -> list[str]: ... # Deprecated: Use list_tables instead
|
||||||
|
def job(self, job_id: str) -> Job: ...
|
||||||
|
async def list_jobs(self) -> List[JobInfo]: ...
|
||||||
|
async def get_job(self, job_id: str) -> Optional[JobDescription]: ...
|
||||||
|
async def cancel_job(self, job_id: str) -> bool: ...
|
||||||
|
async def job_history(
|
||||||
|
self, job_id: Optional[str] = None
|
||||||
|
) -> List[pa.RecordBatch]: ...
|
||||||
async def create_table(
|
async def create_table(
|
||||||
self,
|
self,
|
||||||
name: str,
|
name: str,
|
||||||
@@ -209,6 +216,47 @@ class BlobFile:
|
|||||||
def read_range(self, offset: int, length: int) -> bytes: ...
|
def read_range(self, offset: int, length: int) -> bytes: ...
|
||||||
def read_up_to(self, length: int) -> bytes: ...
|
def read_up_to(self, length: int) -> bytes: ...
|
||||||
|
|
||||||
|
class Job:
|
||||||
|
@property
|
||||||
|
def id(self) -> Optional[str]: ...
|
||||||
|
async def status(self) -> str: ...
|
||||||
|
async def wait(self) -> None: ...
|
||||||
|
async def cancel(self) -> None: ...
|
||||||
|
|
||||||
|
class JobInfo:
|
||||||
|
@property
|
||||||
|
def job_id(self) -> str: ...
|
||||||
|
@property
|
||||||
|
def table(self) -> str: ...
|
||||||
|
@property
|
||||||
|
def job_type(self) -> str: ...
|
||||||
|
@property
|
||||||
|
def state(self) -> str: ...
|
||||||
|
@property
|
||||||
|
def created_at_millis(self) -> int: ...
|
||||||
|
|
||||||
|
class JobFailureInfo:
|
||||||
|
@property
|
||||||
|
def phase(self) -> Optional[str]: ...
|
||||||
|
@property
|
||||||
|
def message(self) -> Optional[str]: ...
|
||||||
|
@property
|
||||||
|
def retryable(self) -> Optional[bool]: ...
|
||||||
|
|
||||||
|
class JobDescription:
|
||||||
|
@property
|
||||||
|
def job_id(self) -> str: ...
|
||||||
|
@property
|
||||||
|
def job_type(self) -> str: ...
|
||||||
|
@property
|
||||||
|
def state(self) -> str: ...
|
||||||
|
@property
|
||||||
|
def creation_ms(self) -> int: ...
|
||||||
|
@property
|
||||||
|
def spec_json(self) -> Optional[str]: ...
|
||||||
|
@property
|
||||||
|
def failure(self) -> Optional[JobFailureInfo]: ...
|
||||||
|
|
||||||
class Table:
|
class Table:
|
||||||
def name(self) -> str: ...
|
def name(self) -> str: ...
|
||||||
def __repr__(self) -> str: ...
|
def __repr__(self) -> str: ...
|
||||||
@@ -248,6 +296,28 @@ class Table:
|
|||||||
name: Optional[str],
|
name: Optional[str],
|
||||||
train: Optional[bool],
|
train: Optional[bool],
|
||||||
): ...
|
): ...
|
||||||
|
async def create_index_async(
|
||||||
|
self,
|
||||||
|
column: str,
|
||||||
|
index: Union[
|
||||||
|
IvfFlat,
|
||||||
|
IvfSq,
|
||||||
|
IvfPq,
|
||||||
|
HnswPq,
|
||||||
|
HnswSq,
|
||||||
|
HnswFlat,
|
||||||
|
BTree,
|
||||||
|
Bitmap,
|
||||||
|
LabelList,
|
||||||
|
Fm,
|
||||||
|
FTS,
|
||||||
|
],
|
||||||
|
replace: Optional[bool],
|
||||||
|
wait_timeout: Optional[object],
|
||||||
|
*,
|
||||||
|
name: Optional[str],
|
||||||
|
train: Optional[bool],
|
||||||
|
) -> Job: ...
|
||||||
async def list_versions(self) -> List[Dict[str, Any]]: ...
|
async def list_versions(self) -> List[Dict[str, Any]]: ...
|
||||||
async def version(self) -> int: ...
|
async def version(self) -> int: ...
|
||||||
async def checkout(self, version: Union[int, str]): ...
|
async def checkout(self, version: Union[int, str]): ...
|
||||||
@@ -285,6 +355,10 @@ class Table:
|
|||||||
async def set_lsm_write_spec(self, spec: LsmWriteSpec) -> None: ...
|
async def set_lsm_write_spec(self, spec: LsmWriteSpec) -> None: ...
|
||||||
async def unset_lsm_write_spec(self) -> None: ...
|
async def unset_lsm_write_spec(self) -> None: ...
|
||||||
async def get_lsm_write_spec(self) -> Optional[LsmWriteSpec]: ...
|
async def get_lsm_write_spec(self) -> Optional[LsmWriteSpec]: ...
|
||||||
|
async def checkpoint_lsm(self) -> None: ...
|
||||||
|
async def flush_lsm(self) -> None: ...
|
||||||
|
async def compact_lsm(self) -> None: ...
|
||||||
|
async def get_lsm_stats(self, include_generation_rows: bool) -> Optional[dict]: ...
|
||||||
async def close_lsm_writers(self) -> None: ...
|
async def close_lsm_writers(self) -> None: ...
|
||||||
@property
|
@property
|
||||||
def tags(self) -> Tags: ...
|
def tags(self) -> Tags: ...
|
||||||
@@ -579,9 +653,10 @@ class LsmWriteSpec:
|
|||||||
def identity(column: str) -> "LsmWriteSpec": ...
|
def identity(column: str) -> "LsmWriteSpec": ...
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def unsharded() -> "LsmWriteSpec": ...
|
def unsharded() -> "LsmWriteSpec": ...
|
||||||
def with_maintained_indexes(self, indexes: List[str]) -> "LsmWriteSpec":
|
def with_maintained_indexes(self, indexes: Optional[List[str]]) -> "LsmWriteSpec":
|
||||||
"""Return a copy of this spec asking the MemWAL to keep the named
|
"""Set which indexes the MemWAL keeps up to date. None resolves every
|
||||||
indexes up to date as rows are appended."""
|
index on the table at install, failing if one cannot be maintained;
|
||||||
|
a list is verbatim, empty means none."""
|
||||||
...
|
...
|
||||||
def with_writer_config_defaults(self, defaults: Dict[str, str]) -> "LsmWriteSpec":
|
def with_writer_config_defaults(self, defaults: Dict[str, str]) -> "LsmWriteSpec":
|
||||||
"""Return a copy of this spec recording the given default
|
"""Return a copy of this spec recording the given default
|
||||||
@@ -596,7 +671,9 @@ class LsmWriteSpec:
|
|||||||
@property
|
@property
|
||||||
def num_buckets(self) -> Optional[int]: ...
|
def num_buckets(self) -> Optional[int]: ...
|
||||||
@property
|
@property
|
||||||
def maintained_indexes(self) -> List[str]: ...
|
def maintained_indexes(self) -> Optional[List[str]]:
|
||||||
|
"""Indexes the MemWAL keeps up to date, or None for every supported one."""
|
||||||
|
...
|
||||||
@property
|
@property
|
||||||
def writer_config_defaults(self) -> Dict[str, str]: ...
|
def writer_config_defaults(self) -> Dict[str, str]: ...
|
||||||
|
|
||||||
|
|||||||
+185
-10
@@ -45,6 +45,7 @@ from lance_namespace.errors import NamespaceNotEmptyError, TableNotFoundError
|
|||||||
|
|
||||||
from . import __version__
|
from . import __version__
|
||||||
from ._lancedb import connect as lancedb_connect # type: ignore
|
from ._lancedb import connect as lancedb_connect # type: ignore
|
||||||
|
from .job import AsyncJob, Job
|
||||||
from .table import (
|
from .table import (
|
||||||
AsyncTable,
|
AsyncTable,
|
||||||
LanceTable,
|
LanceTable,
|
||||||
@@ -63,6 +64,7 @@ if TYPE_CHECKING:
|
|||||||
from .pydantic import LanceModel
|
from .pydantic import LanceModel
|
||||||
|
|
||||||
from ._lancedb import Connection as LanceDbConnection
|
from ._lancedb import Connection as LanceDbConnection
|
||||||
|
from ._lancedb import JobDescription, JobInfo
|
||||||
from .common import DATA, URI
|
from .common import DATA, URI
|
||||||
from .embeddings import EmbeddingFunctionConfig
|
from .embeddings import EmbeddingFunctionConfig
|
||||||
from ._lancedb import Session
|
from ._lancedb import Session
|
||||||
@@ -178,6 +180,51 @@ class DBConnection(EnforceOverrides):
|
|||||||
"Namespace operations are not supported for this connection type"
|
"Namespace operations are not supported for this connection type"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def namespace_exists(self, namespace_id: List[str]) -> bool:
|
||||||
|
"""Check if a namespace exists.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
namespace_id: List[str]
|
||||||
|
The namespace identifier to check.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
bool
|
||||||
|
True if the namespace exists, False otherwise.
|
||||||
|
|
||||||
|
Raises
|
||||||
|
------
|
||||||
|
NotImplementedError
|
||||||
|
If the connection type does not support namespace operations.
|
||||||
|
"""
|
||||||
|
raise NotImplementedError(
|
||||||
|
"Namespace operations are not supported for this connection type"
|
||||||
|
)
|
||||||
|
|
||||||
|
def table_exists(self, table_id: List[str]) -> bool:
|
||||||
|
"""Check if a table exists.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
table_id: List[str]
|
||||||
|
The table identifier to check (full path including namespace
|
||||||
|
segments and table name).
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
bool
|
||||||
|
True if the table exists, False otherwise.
|
||||||
|
|
||||||
|
Raises
|
||||||
|
------
|
||||||
|
NotImplementedError
|
||||||
|
If the connection type does not support namespace operations.
|
||||||
|
"""
|
||||||
|
raise NotImplementedError(
|
||||||
|
"Namespace operations are not supported for this connection type"
|
||||||
|
)
|
||||||
|
|
||||||
def list_tables(
|
def list_tables(
|
||||||
self,
|
self,
|
||||||
namespace_path: Optional[List[str]] = None,
|
namespace_path: Optional[List[str]] = None,
|
||||||
@@ -359,7 +406,7 @@ class DBConnection(EnforceOverrides):
|
|||||||
|
|
||||||
Data is converted to Arrow before being written to disk. For maximum
|
Data is converted to Arrow before being written to disk. For maximum
|
||||||
control over how data is saved, either provide the PyArrow schema to
|
control over how data is saved, either provide the PyArrow schema to
|
||||||
convert to or else provide a [PyArrow Table](pyarrow.Table) directly.
|
convert to or else provide a [PyArrow Table][pyarrow.Table] directly.
|
||||||
|
|
||||||
>>> import pyarrow as pa
|
>>> import pyarrow as pa
|
||||||
>>> custom_schema = pa.schema([
|
>>> custom_schema = pa.schema([
|
||||||
@@ -563,6 +610,46 @@ class DBConnection(EnforceOverrides):
|
|||||||
"""
|
"""
|
||||||
raise NotImplementedError("serialize is not supported for this connection type")
|
raise NotImplementedError("serialize is not supported for this connection type")
|
||||||
|
|
||||||
|
def job(self, job_id: str) -> Job:
|
||||||
|
"""A [Job][lancedb.job.Job] handle for a server-side job by id.
|
||||||
|
|
||||||
|
The handle is constructed without a server round trip; an unknown id
|
||||||
|
surfaces when the handle is used. Dropping the handle has no effect
|
||||||
|
on the job itself.
|
||||||
|
"""
|
||||||
|
raise NotImplementedError("job is not supported for this connection type")
|
||||||
|
|
||||||
|
def list_jobs(self) -> List[JobInfo]:
|
||||||
|
"""List server-side jobs across the database's tables."""
|
||||||
|
raise NotImplementedError("list_jobs is not supported for this connection type")
|
||||||
|
|
||||||
|
def get_job(self, job_id: str) -> Optional[JobDescription]:
|
||||||
|
"""Describe a single server-side job by id.
|
||||||
|
|
||||||
|
Returns None when the server has no such job.
|
||||||
|
"""
|
||||||
|
raise NotImplementedError("get_job is not supported for this connection type")
|
||||||
|
|
||||||
|
def cancel_job(self, job_id: str) -> bool:
|
||||||
|
"""Request cancellation of a server-side job by id.
|
||||||
|
|
||||||
|
Returns True if the server accepted the cancellation, False if no
|
||||||
|
such job exists. Cancelling an already-terminal job is a no-op
|
||||||
|
success.
|
||||||
|
"""
|
||||||
|
raise NotImplementedError(
|
||||||
|
"cancel_job is not supported for this connection type"
|
||||||
|
)
|
||||||
|
|
||||||
|
def job_history(self, job_id: Optional[str] = None) -> List[pa.RecordBatch]:
|
||||||
|
"""The lifecycle event history of a server-side job, as Arrow batches.
|
||||||
|
|
||||||
|
Lists history across all jobs when `job_id` is None.
|
||||||
|
"""
|
||||||
|
raise NotImplementedError(
|
||||||
|
"job_history is not supported for this connection type"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
class LanceDBConnection(DBConnection):
|
class LanceDBConnection(DBConnection):
|
||||||
"""
|
"""
|
||||||
@@ -620,6 +707,9 @@ class LanceDBConnection(DBConnection):
|
|||||||
self._namespace_client_properties = namespace_client_properties
|
self._namespace_client_properties = namespace_client_properties
|
||||||
if _inner is not None:
|
if _inner is not None:
|
||||||
self._conn = _inner
|
self._conn = _inner
|
||||||
|
# Native-derived wrappers resolve this in their async reconstruction
|
||||||
|
# path so construction never synchronously re-enters LOOP.
|
||||||
|
self._read_consistency_interval = read_consistency_interval
|
||||||
self._cached_namespace_client = None
|
self._cached_namespace_client = None
|
||||||
return
|
return
|
||||||
|
|
||||||
@@ -669,11 +759,14 @@ class LanceDBConnection(DBConnection):
|
|||||||
# storage_options. Also, this class really shouldn't be holding any state
|
# storage_options. Also, this class really shouldn't be holding any state
|
||||||
# beyond _conn.
|
# beyond _conn.
|
||||||
self._conn = AsyncConnection(LOOP.run(do_connect()))
|
self._conn = AsyncConnection(LOOP.run(do_connect()))
|
||||||
|
# Keep property access synchronous so debugger introspection cannot wait on
|
||||||
|
# the background loop while that thread is suspended at a breakpoint.
|
||||||
|
self._read_consistency_interval = read_consistency_interval
|
||||||
self._cached_namespace_client: Optional[LanceNamespace] = None
|
self._cached_namespace_client: Optional[LanceNamespace] = None
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def read_consistency_interval(self) -> Optional[timedelta]:
|
def read_consistency_interval(self) -> Optional[timedelta]:
|
||||||
return LOOP.run(self._conn.get_read_consistency_interval())
|
return self._read_consistency_interval
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def session(self) -> Optional[Session]:
|
def session(self) -> Optional[Session]:
|
||||||
@@ -684,15 +777,19 @@ class LanceDBConnection(DBConnection):
|
|||||||
return self._conn.uri
|
return self._conn.uri
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def from_inner(cls, inner: LanceDbConnection):
|
def from_inner(
|
||||||
return cls(None, _inner=inner)
|
cls,
|
||||||
|
inner: LanceDbConnection,
|
||||||
|
read_consistency_interval: Optional[timedelta],
|
||||||
|
):
|
||||||
|
return cls(
|
||||||
|
None,
|
||||||
|
read_consistency_interval=read_consistency_interval,
|
||||||
|
_inner=inner,
|
||||||
|
)
|
||||||
|
|
||||||
def __repr__(self) -> str:
|
def __repr__(self) -> str:
|
||||||
val = f"{self.__class__.__name__}(uri={self._conn.uri!r}"
|
return f"{self.__class__.__name__}(uri={self._conn.uri!r})"
|
||||||
if self.read_consistency_interval is not None:
|
|
||||||
val += f", read_consistency_interval={repr(self.read_consistency_interval)}"
|
|
||||||
val += ")"
|
|
||||||
return val
|
|
||||||
|
|
||||||
@override
|
@override
|
||||||
def serialize(self) -> str:
|
def serialize(self) -> str:
|
||||||
@@ -1129,6 +1226,47 @@ class LanceDBConnection(DBConnection):
|
|||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|
||||||
|
@override
|
||||||
|
def job(self, job_id: str) -> Job:
|
||||||
|
"""A [Job][lancedb.job.Job] handle for a server-side job by id.
|
||||||
|
|
||||||
|
The handle is constructed without a server round trip; an unknown id
|
||||||
|
surfaces when the handle is used. Dropping the handle has no effect
|
||||||
|
on the job itself.
|
||||||
|
"""
|
||||||
|
return Job(self._conn.job(job_id))
|
||||||
|
|
||||||
|
@override
|
||||||
|
def list_jobs(self) -> List[JobInfo]:
|
||||||
|
"""List server-side jobs across the database's tables."""
|
||||||
|
return LOOP.run(self._conn.list_jobs())
|
||||||
|
|
||||||
|
@override
|
||||||
|
def get_job(self, job_id: str) -> Optional[JobDescription]:
|
||||||
|
"""Describe a single server-side job by id.
|
||||||
|
|
||||||
|
Returns None when the server has no such job.
|
||||||
|
"""
|
||||||
|
return LOOP.run(self._conn.get_job(job_id))
|
||||||
|
|
||||||
|
@override
|
||||||
|
def cancel_job(self, job_id: str) -> bool:
|
||||||
|
"""Request cancellation of a server-side job by id.
|
||||||
|
|
||||||
|
Returns True if the server accepted the cancellation, False if no
|
||||||
|
such job exists. Cancelling an already-terminal job is a no-op
|
||||||
|
success.
|
||||||
|
"""
|
||||||
|
return LOOP.run(self._conn.cancel_job(job_id))
|
||||||
|
|
||||||
|
@override
|
||||||
|
def job_history(self, job_id: Optional[str] = None) -> List[pa.RecordBatch]:
|
||||||
|
"""The lifecycle event history of a server-side job, as Arrow batches.
|
||||||
|
|
||||||
|
Lists history across all jobs when `job_id` is None.
|
||||||
|
"""
|
||||||
|
return LOOP.run(self._conn.job_history(job_id))
|
||||||
|
|
||||||
@override
|
@override
|
||||||
def namespace_client(self) -> LanceNamespace:
|
def namespace_client(self) -> LanceNamespace:
|
||||||
"""Get the equivalent namespace client for this connection.
|
"""Get the equivalent namespace client for this connection.
|
||||||
@@ -1529,7 +1667,7 @@ class AsyncConnection(object):
|
|||||||
|
|
||||||
Data is converted to Arrow before being written to disk. For maximum
|
Data is converted to Arrow before being written to disk. For maximum
|
||||||
control over how data is saved, either provide the PyArrow schema to
|
control over how data is saved, either provide the PyArrow schema to
|
||||||
convert to or else provide a [PyArrow Table](pyarrow.Table) directly.
|
convert to or else provide a [PyArrow Table][pyarrow.Table] directly.
|
||||||
|
|
||||||
>>> import pyarrow as pa
|
>>> import pyarrow as pa
|
||||||
>>> custom_schema = pa.schema([
|
>>> custom_schema = pa.schema([
|
||||||
@@ -1838,6 +1976,43 @@ class AsyncConnection(object):
|
|||||||
namespace_path = []
|
namespace_path = []
|
||||||
await self._inner.drop_all_tables(namespace_path=namespace_path)
|
await self._inner.drop_all_tables(namespace_path=namespace_path)
|
||||||
|
|
||||||
|
def job(self, job_id: str) -> AsyncJob:
|
||||||
|
"""An [AsyncJob][lancedb.job.AsyncJob] handle for a server-side job
|
||||||
|
by id.
|
||||||
|
|
||||||
|
The handle is constructed without a server round trip; an unknown id
|
||||||
|
surfaces when the handle is used. Dropping the handle has no effect
|
||||||
|
on the job itself.
|
||||||
|
"""
|
||||||
|
return AsyncJob(self._inner.job(job_id))
|
||||||
|
|
||||||
|
async def list_jobs(self) -> List[JobInfo]:
|
||||||
|
"""List server-side jobs across the database's tables."""
|
||||||
|
return await self._inner.list_jobs()
|
||||||
|
|
||||||
|
async def get_job(self, job_id: str) -> Optional[JobDescription]:
|
||||||
|
"""Describe a single server-side job by id.
|
||||||
|
|
||||||
|
Returns None when the server has no such job.
|
||||||
|
"""
|
||||||
|
return await self._inner.get_job(job_id)
|
||||||
|
|
||||||
|
async def cancel_job(self, job_id: str) -> bool:
|
||||||
|
"""Request cancellation of a server-side job by id.
|
||||||
|
|
||||||
|
Returns True if the server accepted the cancellation, False if no
|
||||||
|
such job exists. Cancelling an already-terminal job is a no-op
|
||||||
|
success.
|
||||||
|
"""
|
||||||
|
return await self._inner.cancel_job(job_id)
|
||||||
|
|
||||||
|
async def job_history(self, job_id: Optional[str] = None) -> List[pa.RecordBatch]:
|
||||||
|
"""The lifecycle event history of a server-side job, as Arrow batches.
|
||||||
|
|
||||||
|
Lists history across all jobs when `job_id` is None.
|
||||||
|
"""
|
||||||
|
return await self._inner.job_history(job_id)
|
||||||
|
|
||||||
async def namespace_client(self) -> LanceNamespace:
|
async def namespace_client(self) -> LanceNamespace:
|
||||||
"""Get the equivalent namespace client for this connection.
|
"""Get the equivalent namespace client for this connection.
|
||||||
|
|
||||||
|
|||||||
@@ -21,3 +21,32 @@ from .watsonx import WatsonxEmbeddings
|
|||||||
from .voyageai import VoyageAIEmbeddingFunction
|
from .voyageai import VoyageAIEmbeddingFunction
|
||||||
from .colpali import ColPaliEmbeddings
|
from .colpali import ColPaliEmbeddings
|
||||||
from .siglip import SigLipEmbeddings
|
from .siglip import SigLipEmbeddings
|
||||||
|
|
||||||
|
# The API reference renders this package with a single mkdocstrings directive,
|
||||||
|
# which only picks up names listed here. New embedding functions must be added
|
||||||
|
# to both the imports above and this list, or they will silently go undocumented.
|
||||||
|
__all__ = [
|
||||||
|
"EmbeddingFunction",
|
||||||
|
"EmbeddingFunctionConfig",
|
||||||
|
"TextEmbeddingFunction",
|
||||||
|
"EmbeddingFunctionRegistry",
|
||||||
|
"get_registry",
|
||||||
|
"register",
|
||||||
|
"SentenceTransformerEmbeddings",
|
||||||
|
"OpenAIEmbeddings",
|
||||||
|
"OpenClipEmbeddings",
|
||||||
|
"BedRockText",
|
||||||
|
"CohereEmbeddingFunction",
|
||||||
|
"GeminiText",
|
||||||
|
"GteEmbeddings",
|
||||||
|
"InstructorEmbeddingFunction",
|
||||||
|
"JinaEmbeddings",
|
||||||
|
"OllamaEmbeddings",
|
||||||
|
"TransformersEmbeddingFunction",
|
||||||
|
"ColbertEmbeddings",
|
||||||
|
"VoyageAIEmbeddingFunction",
|
||||||
|
"WatsonxEmbeddings",
|
||||||
|
"ColPaliEmbeddings",
|
||||||
|
"ImageBindEmbeddings",
|
||||||
|
"SigLipEmbeddings",
|
||||||
|
]
|
||||||
|
|||||||
@@ -21,20 +21,20 @@ class BedRockText(TextEmbeddingFunction):
|
|||||||
"""
|
"""
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
name: str, default "amazon.titan-embed-text-v1"
|
name : str, default "amazon.titan-embed-text-v1"
|
||||||
The model ID of the bedrock model to use. Supported models for are:
|
The model ID of the bedrock model to use. Supported models for are:
|
||||||
- amazon.titan-embed-text-v1
|
- amazon.titan-embed-text-v1
|
||||||
- cohere.embed-english-v3
|
- cohere.embed-english-v3
|
||||||
- cohere.embed-multilingual-v3
|
- cohere.embed-multilingual-v3
|
||||||
region: str, default "us-east-1"
|
region : str, default "us-east-1"
|
||||||
Optional name of the AWS Region in which the service should be called.
|
Optional name of the AWS Region in which the service should be called.
|
||||||
profile_name: str, default None
|
profile_name : str, default None
|
||||||
Optional name of the AWS profile to use for calling the Bedrock service.
|
Optional name of the AWS profile to use for calling the Bedrock service.
|
||||||
If not specified, the default profile will be used.
|
If not specified, the default profile will be used.
|
||||||
assumed_role: str, default None
|
assumed_role : str, default None
|
||||||
Optional ARN of an AWS IAM role to assume for calling the Bedrock service.
|
Optional ARN of an AWS IAM role to assume for calling the Bedrock service.
|
||||||
If not specified, the current active credentials will be used.
|
If not specified, the current active credentials will be used.
|
||||||
role_session_name: str, default "lancedb-embeddings"
|
role_session_name : str, default "lancedb-embeddings"
|
||||||
Optional name of the AWS IAM role session to use for calling the Bedrock
|
Optional name of the AWS IAM role session to use for calling the Bedrock
|
||||||
service. If not specified, "lancedb-embeddings" name will be used.
|
service. If not specified, "lancedb-embeddings" name will be used.
|
||||||
|
|
||||||
|
|||||||
@@ -22,7 +22,7 @@ class CohereEmbeddingFunction(TextEmbeddingFunction):
|
|||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
name: str, default "embed-multilingual-v2.0"
|
name : str, default "embed-multilingual-v2.0"
|
||||||
The name of the model to use. List of acceptable models:
|
The name of the model to use. List of acceptable models:
|
||||||
|
|
||||||
* embed-english-v3.0
|
* embed-english-v3.0
|
||||||
@@ -33,12 +33,14 @@ class CohereEmbeddingFunction(TextEmbeddingFunction):
|
|||||||
* embed-english-light-v2.0
|
* embed-english-light-v2.0
|
||||||
* embed-multilingual-v2.0
|
* embed-multilingual-v2.0
|
||||||
|
|
||||||
source_input_type: str, default "search_document"
|
source_input_type : str, default "search_document"
|
||||||
The input type for the source column in the database
|
The input type for the source column in the database
|
||||||
|
|
||||||
query_input_type: str, default "search_query"
|
query_input_type : str, default "search_query"
|
||||||
The input type for the query column in the database
|
The input type for the query column in the database
|
||||||
|
|
||||||
|
Notes
|
||||||
|
-----
|
||||||
Cohere supports following input types:
|
Cohere supports following input types:
|
||||||
|
|
||||||
| Input Type | Description |
|
| Input Type | Description |
|
||||||
|
|||||||
@@ -44,7 +44,7 @@ class ColPaliEmbeddings(EmbeddingFunction):
|
|||||||
The token pooling strategy to use, by default "hierarchical".
|
The token pooling strategy to use, by default "hierarchical".
|
||||||
- "hierarchical": Progressively pools tokens to reduce sequence length.
|
- "hierarchical": Progressively pools tokens to reduce sequence length.
|
||||||
- "lambda": A simpler pooling that uses a custom `pooling_func`.
|
- "lambda": A simpler pooling that uses a custom `pooling_func`.
|
||||||
pooling_func: typing.Callable, optional
|
pooling_func : typing.Callable, optional
|
||||||
A function to use for pooling when `pooling_strategy` is "lambda".
|
A function to use for pooling when `pooling_strategy` is "lambda".
|
||||||
pool_factor : int
|
pool_factor : int
|
||||||
Factor to reduce sequence length if token pooling is enabled (default 2).
|
Factor to reduce sequence length if token pooling is enabled (default 2).
|
||||||
@@ -52,7 +52,7 @@ class ColPaliEmbeddings(EmbeddingFunction):
|
|||||||
Quantization configuration for the model. (default None, bitsandbytes needed)
|
Quantization configuration for the model. (default None, bitsandbytes needed)
|
||||||
batch_size : int
|
batch_size : int
|
||||||
Batch size for processing inputs (default 2).
|
Batch size for processing inputs (default 2).
|
||||||
offload_folder: str, optional
|
offload_folder : str, optional
|
||||||
Folder to offload model weights if using CPU offloading (default None). This is
|
Folder to offload model weights if using CPU offloading (default None). This is
|
||||||
useful for large models that do not fit in memory.
|
useful for large models that do not fit in memory.
|
||||||
"""
|
"""
|
||||||
|
|||||||
@@ -48,16 +48,16 @@ class GeminiText(TextEmbeddingFunction):
|
|||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
name: str, default "gemini-embedding-001"
|
name : str, default "gemini-embedding-001"
|
||||||
The name of the model to use. Supported models include:
|
The name of the model to use. Supported models include:
|
||||||
- "gemini-embedding-001" (768 dimensions)
|
- "gemini-embedding-001" (768 dimensions)
|
||||||
|
|
||||||
Note: The legacy "models/embedding-001" format is also supported but
|
Note: The legacy "models/embedding-001" format is also supported but
|
||||||
"gemini-embedding-001" is recommended.
|
"gemini-embedding-001" is recommended.
|
||||||
|
|
||||||
query_task_type: str, default "retrieval_query"
|
query_task_type : str, default "retrieval_query"
|
||||||
Sets the task type for the queries.
|
Sets the task type for the queries.
|
||||||
source_task_type: str, default "retrieval_document"
|
source_task_type : str, default "retrieval_document"
|
||||||
Sets the task type for ingestion.
|
Sets the task type for ingestion.
|
||||||
|
|
||||||
Examples
|
Examples
|
||||||
|
|||||||
@@ -26,13 +26,13 @@ class GteEmbeddings(TextEmbeddingFunction):
|
|||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
name: str, default "thenlper/gte-large"
|
name : str, default "thenlper/gte-large"
|
||||||
The name of the model to use.
|
The name of the model to use.
|
||||||
device: str, default "cpu"
|
device : str, default "cpu"
|
||||||
Sets the device type for the model.
|
Sets the device type for the model.
|
||||||
normalize: str, default "True"
|
normalize : str, default "True"
|
||||||
Controls normalize param in encode function for the transformer.
|
Controls normalize param in encode function for the transformer.
|
||||||
mlx: bool, default False
|
mlx : bool, default False
|
||||||
Controls which model to use. False for gte-large,True for the mlx version.
|
Controls which model to use. False for gte-large,True for the mlx version.
|
||||||
|
|
||||||
Examples
|
Examples
|
||||||
|
|||||||
@@ -35,23 +35,23 @@ class InstructorEmbeddingFunction(TextEmbeddingFunction):
|
|||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
name: str
|
name : str
|
||||||
The name of the model to use. Available models are listed at
|
The name of the model to use. Available models are listed at
|
||||||
https://github.com/xlang-ai/instructor-embedding#model-list;
|
https://github.com/xlang-ai/instructor-embedding#model-list;
|
||||||
The default model is hkunlp/instructor-base
|
The default model is hkunlp/instructor-base
|
||||||
batch_size: int, default 32
|
batch_size : int, default 32
|
||||||
The batch size to use when generating embeddings
|
The batch size to use when generating embeddings
|
||||||
device: str, default "cpu"
|
device : str, default "cpu"
|
||||||
The device to use when generating embeddings
|
The device to use when generating embeddings
|
||||||
show_progress_bar: bool, default True
|
show_progress_bar : bool, default True
|
||||||
Whether to show a progress bar when generating embeddings
|
Whether to show a progress bar when generating embeddings
|
||||||
normalize_embeddings: bool, default True
|
normalize_embeddings : bool, default True
|
||||||
Whether to normalize the embeddings
|
Whether to normalize the embeddings
|
||||||
quantize: bool, default False
|
quantize : bool, default False
|
||||||
Whether to quantize the model
|
Whether to quantize the model
|
||||||
source_instruction: str, default "represent the document for retrieval"
|
source_instruction : str, default "represent the document for retrieval"
|
||||||
The instruction for the source column
|
The instruction for the source column
|
||||||
query_instruction: str, default "represent the document for retrieving the most
|
query_instruction : str, default "represent the document for retrieving the most
|
||||||
similar documents"
|
similar documents"
|
||||||
The instruction for the query
|
The instruction for the query
|
||||||
|
|
||||||
@@ -101,8 +101,7 @@ class InstructorEmbeddingFunction(TextEmbeddingFunction):
|
|||||||
|
|
||||||
@weak_lru(maxsize=1)
|
@weak_lru(maxsize=1)
|
||||||
def ndims(self):
|
def ndims(self):
|
||||||
model = self.get_model()
|
return len(self.generate_embeddings([[self.source_instruction, "foo"]])[0])
|
||||||
return model.encode("foo").shape[0]
|
|
||||||
|
|
||||||
def compute_query_embeddings(self, query: str, *args, **kwargs) -> List[np.array]:
|
def compute_query_embeddings(self, query: str, *args, **kwargs) -> List[np.array]:
|
||||||
return self.generate_embeddings([[self.query_instruction, query]])
|
return self.generate_embeddings([[self.query_instruction, query]])
|
||||||
|
|||||||
@@ -40,10 +40,10 @@ class JinaEmbeddings(EmbeddingFunction):
|
|||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
name: str, default "jina-clip-v1". Note that some models support both image
|
name : str, default "jina-clip-v1". Note that some models support both image
|
||||||
and text embeddings and some just text embedding
|
and text embeddings and some just text embedding
|
||||||
|
|
||||||
api_key: str, default None
|
api_key : str, default None
|
||||||
The api key to access Jina API. If you pass None, you can set JINA_API_KEY
|
The api key to access Jina API. If you pass None, you can set JINA_API_KEY
|
||||||
environment variable
|
environment variable
|
||||||
|
|
||||||
@@ -87,12 +87,13 @@ class JinaEmbeddings(EmbeddingFunction):
|
|||||||
if isinstance(image, bytes):
|
if isinstance(image, bytes):
|
||||||
image_dict = {"image": base64.b64encode(image).decode("utf-8")}
|
image_dict = {"image": base64.b64encode(image).decode("utf-8")}
|
||||||
elif isinstance(image, (str, Path)):
|
elif isinstance(image, (str, Path)):
|
||||||
parsed = urlparse.urlparse(image)
|
parsed = urlparse(str(image))
|
||||||
# TODO handle drive letter on windows.
|
|
||||||
PIL_Image = attempt_import_or_raise("PIL.Image", "pillow")
|
PIL_Image = attempt_import_or_raise("PIL.Image", "pillow")
|
||||||
if parsed.scheme == "file":
|
if parsed.scheme == "file":
|
||||||
pil_image = PIL_Image.open(parsed.path)
|
pil_image = PIL_Image.open(parsed.path)
|
||||||
elif parsed.scheme == "":
|
elif parsed.scheme == "" or (os.name == "nt" and len(parsed.scheme) == 1):
|
||||||
|
# A Windows drive letter parses as a one-character scheme
|
||||||
|
# ("C:\\img.png" -> scheme="c"), so treat it as a local path.
|
||||||
pil_image = PIL_Image.open(image if os.name == "nt" else parsed.path)
|
pil_image = PIL_Image.open(image if os.name == "nt" else parsed.path)
|
||||||
elif parsed.scheme.startswith("http"):
|
elif parsed.scheme.startswith("http"):
|
||||||
pil_image = PIL_Image.open(io.BytesIO(url_retrieve(image)))
|
pil_image = PIL_Image.open(io.BytesIO(url_retrieve(image)))
|
||||||
|
|||||||
@@ -21,13 +21,13 @@ class SentenceTransformerEmbeddings(TextEmbeddingFunction):
|
|||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
name: str, default "all-MiniLM-L6-v2"
|
name : str, default "all-MiniLM-L6-v2"
|
||||||
The name of the model to use.
|
The name of the model to use.
|
||||||
device: str, default "cpu"
|
device : str, default "cpu"
|
||||||
The device to use for the model
|
The device to use for the model
|
||||||
normalize: bool, default True
|
normalize : bool, default True
|
||||||
Whether to normalize the embeddings
|
Whether to normalize the embeddings
|
||||||
trust_remote_code: bool, default True
|
trust_remote_code : bool, default True
|
||||||
Whether to trust the remote code
|
Whether to trust the remote code
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|||||||
@@ -167,7 +167,7 @@ class VoyageAIEmbeddingFunction(EmbeddingFunction):
|
|||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
name: str
|
name : str
|
||||||
The name of the model to use. List of acceptable models:
|
The name of the model to use. List of acceptable models:
|
||||||
|
|
||||||
* voyage-4 (1024 dims, general-purpose and multilingual retrieval)
|
* voyage-4 (1024 dims, general-purpose and multilingual retrieval)
|
||||||
@@ -185,7 +185,7 @@ class VoyageAIEmbeddingFunction(EmbeddingFunction):
|
|||||||
* voyage-law-2
|
* voyage-law-2
|
||||||
* voyage-code-2
|
* voyage-code-2
|
||||||
|
|
||||||
output_dimension: int, optional
|
output_dimension : int, optional
|
||||||
The output dimension for models that support flexible dimensions.
|
The output dimension for models that support flexible dimensions.
|
||||||
Currently only voyage-multimodal-3.5 supports this feature.
|
Currently only voyage-multimodal-3.5 supports this feature.
|
||||||
Valid options: 256, 512, 1024 (default), 2048.
|
Valid options: 256, 512, 1024 (default), 2048.
|
||||||
|
|||||||
@@ -23,3 +23,15 @@ class MissingColumnError(KeyError):
|
|||||||
return (
|
return (
|
||||||
f"Error: Column '{self.column_name}' does not exist in the DataFrame object"
|
f"Error: Column '{self.column_name}' does not exist in the DataFrame object"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class JobFailedError(RuntimeError):
|
||||||
|
"""Exception raised when an asynchronous job reaches the failed state."""
|
||||||
|
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
class JobCancelledError(RuntimeError):
|
||||||
|
"""Exception raised when an asynchronous job was cancelled."""
|
||||||
|
|
||||||
|
pass
|
||||||
|
|||||||
@@ -219,7 +219,7 @@ class HnswPq:
|
|||||||
distance has a range of (-∞, ∞). If the vectors are normalized (i.e. their
|
distance has a range of (-∞, ∞). If the vectors are normalized (i.e. their
|
||||||
l2 norm is 1), then dot distance is equivalent to the cosine distance.
|
l2 norm is 1), then dot distance is equivalent to the cosine distance.
|
||||||
|
|
||||||
num_partitions, default sqrt(num_rows)
|
num_partitions: int, default sqrt(num_rows)
|
||||||
|
|
||||||
The number of IVF partitions to create.
|
The number of IVF partitions to create.
|
||||||
|
|
||||||
@@ -228,7 +228,7 @@ class HnswPq:
|
|||||||
will require too much memory. Each partition becomes its own HNSW graph, so
|
will require too much memory. Each partition becomes its own HNSW graph, so
|
||||||
setting this value higher reduces the peak memory use of training.
|
setting this value higher reduces the peak memory use of training.
|
||||||
|
|
||||||
num_sub_vectors, default is vector dimension / 16
|
num_sub_vectors: int, default is vector dimension / 16
|
||||||
|
|
||||||
Number of sub-vectors of PQ.
|
Number of sub-vectors of PQ.
|
||||||
|
|
||||||
@@ -244,13 +244,13 @@ class HnswPq:
|
|||||||
If the dimension is not visible by 8 then we use 1 subvector. This is not
|
If the dimension is not visible by 8 then we use 1 subvector. This is not
|
||||||
ideal and will likely result in poor performance.
|
ideal and will likely result in poor performance.
|
||||||
|
|
||||||
num_bits: int, default 8
|
num_bits: int, default 8
|
||||||
Number of bits to encode each sub-vector.
|
Number of bits to encode each sub-vector.
|
||||||
|
|
||||||
This value controls how much the sub-vectors are compressed. The more bits
|
This value controls how much the sub-vectors are compressed. The more bits
|
||||||
the more accurate the index but the slower search. Only 4 and 8 are supported.
|
the more accurate the index but the slower search. Only 4 and 8 are supported.
|
||||||
|
|
||||||
max_iterations, default 50
|
max_iterations: int, default 50
|
||||||
|
|
||||||
Max iterations to train kmeans.
|
Max iterations to train kmeans.
|
||||||
|
|
||||||
@@ -263,7 +263,7 @@ class HnswPq:
|
|||||||
those cases it is unlikely that setting this larger will lead to the index
|
those cases it is unlikely that setting this larger will lead to the index
|
||||||
converging anyways.
|
converging anyways.
|
||||||
|
|
||||||
sample_rate, default 256
|
sample_rate: int, default 256
|
||||||
|
|
||||||
The rate used to calculate the number of training vectors for kmeans.
|
The rate used to calculate the number of training vectors for kmeans.
|
||||||
|
|
||||||
@@ -279,14 +279,14 @@ class HnswPq:
|
|||||||
Increasing this value might improve the quality of the index but in
|
Increasing this value might improve the quality of the index but in
|
||||||
most cases the default should be sufficient.
|
most cases the default should be sufficient.
|
||||||
|
|
||||||
m, default 20
|
m: int, default 20
|
||||||
|
|
||||||
The number of neighbors to select for each vector in the HNSW graph.
|
The number of neighbors to select for each vector in the HNSW graph.
|
||||||
|
|
||||||
This value controls the tradeoff between search speed and accuracy.
|
This value controls the tradeoff between search speed and accuracy.
|
||||||
The higher the value the more accurate the search but the slower it will be.
|
The higher the value the more accurate the search but the slower it will be.
|
||||||
|
|
||||||
ef_construction, default 300
|
ef_construction: int, default 300
|
||||||
|
|
||||||
The number of candidates to evaluate during the construction of the HNSW graph.
|
The number of candidates to evaluate during the construction of the HNSW graph.
|
||||||
|
|
||||||
@@ -297,7 +297,7 @@ class HnswPq:
|
|||||||
This value should be set to a value that is not less than `ef` in the
|
This value should be set to a value that is not less than `ef` in the
|
||||||
search phase.
|
search phase.
|
||||||
|
|
||||||
target_partition_size, default is 1,048,576
|
target_partition_size: int, default is 1,048,576
|
||||||
|
|
||||||
The target size of each partition.
|
The target size of each partition.
|
||||||
|
|
||||||
@@ -351,7 +351,7 @@ class HnswSq:
|
|||||||
distance has a range of (-∞, ∞). If the vectors are normalized (i.e. their
|
distance has a range of (-∞, ∞). If the vectors are normalized (i.e. their
|
||||||
l2 norm is 1), then dot distance is equivalent to the cosine distance.
|
l2 norm is 1), then dot distance is equivalent to the cosine distance.
|
||||||
|
|
||||||
num_partitions, default sqrt(num_rows)
|
num_partitions: int, default sqrt(num_rows)
|
||||||
|
|
||||||
The number of IVF partitions to create.
|
The number of IVF partitions to create.
|
||||||
|
|
||||||
@@ -360,7 +360,7 @@ class HnswSq:
|
|||||||
will require too much memory. Each partition becomes its own HNSW graph, so
|
will require too much memory. Each partition becomes its own HNSW graph, so
|
||||||
setting this value higher reduces the peak memory use of training.
|
setting this value higher reduces the peak memory use of training.
|
||||||
|
|
||||||
max_iterations, default 50
|
max_iterations: int, default 50
|
||||||
|
|
||||||
Max iterations to train kmeans.
|
Max iterations to train kmeans.
|
||||||
|
|
||||||
@@ -373,7 +373,7 @@ class HnswSq:
|
|||||||
In those cases it is unlikely that setting this larger will lead to
|
In those cases it is unlikely that setting this larger will lead to
|
||||||
the index converging anyways.
|
the index converging anyways.
|
||||||
|
|
||||||
sample_rate, default 256
|
sample_rate: int, default 256
|
||||||
|
|
||||||
The rate used to calculate the number of training vectors for kmeans.
|
The rate used to calculate the number of training vectors for kmeans.
|
||||||
|
|
||||||
@@ -389,14 +389,14 @@ class HnswSq:
|
|||||||
Increasing this value might improve the quality of the index but in
|
Increasing this value might improve the quality of the index but in
|
||||||
most cases the default should be sufficient.
|
most cases the default should be sufficient.
|
||||||
|
|
||||||
m, default 20
|
m: int, default 20
|
||||||
|
|
||||||
The number of neighbors to select for each vector in the HNSW graph.
|
The number of neighbors to select for each vector in the HNSW graph.
|
||||||
|
|
||||||
This value controls the tradeoff between search speed and accuracy.
|
This value controls the tradeoff between search speed and accuracy.
|
||||||
The higher the value the more accurate the search but the slower it will be.
|
The higher the value the more accurate the search but the slower it will be.
|
||||||
|
|
||||||
ef_construction, default 300
|
ef_construction: int, default 300
|
||||||
|
|
||||||
The number of candidates to evaluate during the construction of the HNSW graph.
|
The number of candidates to evaluate during the construction of the HNSW graph.
|
||||||
|
|
||||||
@@ -407,7 +407,7 @@ class HnswSq:
|
|||||||
This value should be set to a value that is not less than `ef` in the search
|
This value should be set to a value that is not less than `ef` in the search
|
||||||
phase.
|
phase.
|
||||||
|
|
||||||
target_partition_size, default is 1,048,576
|
target_partition_size: int, default is 1,048,576
|
||||||
|
|
||||||
The target size of each partition.
|
The target size of each partition.
|
||||||
|
|
||||||
@@ -460,7 +460,7 @@ class HnswFlat:
|
|||||||
distance has a range of (-∞, ∞). If the vectors are normalized (i.e. their
|
distance has a range of (-∞, ∞). If the vectors are normalized (i.e. their
|
||||||
l2 norm is 1), then dot distance is equivalent to the cosine distance.
|
l2 norm is 1), then dot distance is equivalent to the cosine distance.
|
||||||
|
|
||||||
num_partitions, default sqrt(num_rows)
|
num_partitions: int, default sqrt(num_rows)
|
||||||
|
|
||||||
The number of IVF partitions to create.
|
The number of IVF partitions to create.
|
||||||
|
|
||||||
@@ -470,18 +470,18 @@ class HnswFlat:
|
|||||||
graph, so setting this value higher reduces the peak memory use of
|
graph, so setting this value higher reduces the peak memory use of
|
||||||
training.
|
training.
|
||||||
|
|
||||||
max_iterations, default 50
|
max_iterations: int, default 50
|
||||||
|
|
||||||
Max iterations to train kmeans.
|
Max iterations to train kmeans.
|
||||||
|
|
||||||
When training an IVF index we use kmeans to calculate the partitions.
|
When training an IVF index we use kmeans to calculate the partitions.
|
||||||
This parameter controls how many iterations of kmeans to run.
|
This parameter controls how many iterations of kmeans to run.
|
||||||
|
|
||||||
sample_rate, default 256
|
sample_rate: int, default 256
|
||||||
|
|
||||||
The rate used to calculate the number of training vectors for kmeans.
|
The rate used to calculate the number of training vectors for kmeans.
|
||||||
|
|
||||||
m, default 20
|
m: int, default 20
|
||||||
|
|
||||||
The number of neighbors to select for each vector in the HNSW graph.
|
The number of neighbors to select for each vector in the HNSW graph.
|
||||||
|
|
||||||
@@ -489,7 +489,7 @@ class HnswFlat:
|
|||||||
The higher the value the more accurate the search but the slower it
|
The higher the value the more accurate the search but the slower it
|
||||||
will be.
|
will be.
|
||||||
|
|
||||||
ef_construction, default 300
|
ef_construction: int, default 300
|
||||||
|
|
||||||
The number of candidates to evaluate during the construction of the HNSW
|
The number of candidates to evaluate during the construction of the HNSW
|
||||||
graph.
|
graph.
|
||||||
@@ -501,7 +501,7 @@ class HnswFlat:
|
|||||||
than 500. This value should be set to a value that is not less than `ef`
|
than 500. This value should be set to a value that is not less than `ef`
|
||||||
in the search phase.
|
in the search phase.
|
||||||
|
|
||||||
target_partition_size, default is 1,048,576
|
target_partition_size: int, default is 1,048,576
|
||||||
|
|
||||||
The target size of each partition.
|
The target size of each partition.
|
||||||
"""
|
"""
|
||||||
@@ -605,7 +605,7 @@ class IvfFlat:
|
|||||||
|
|
||||||
The default value is 256.
|
The default value is 256.
|
||||||
|
|
||||||
target_partition_size, default is 8192
|
target_partition_size: int, default is 8192
|
||||||
|
|
||||||
The target size of each partition.
|
The target size of each partition.
|
||||||
|
|
||||||
@@ -769,7 +769,7 @@ class IvfPq:
|
|||||||
|
|
||||||
The default value is 256.
|
The default value is 256.
|
||||||
|
|
||||||
target_partition_size, default is 8192
|
target_partition_size: int, default is 8192
|
||||||
|
|
||||||
The target size of each partition.
|
The target size of each partition.
|
||||||
|
|
||||||
@@ -830,7 +830,7 @@ class IvfRq:
|
|||||||
sample_rate: int, default 256
|
sample_rate: int, default 256
|
||||||
Controls the number of training vectors: sample_rate * num_partitions.
|
Controls the number of training vectors: sample_rate * num_partitions.
|
||||||
|
|
||||||
target_partition_size, default is 8192
|
target_partition_size: int, default is 8192
|
||||||
Target size of each partition.
|
Target size of each partition.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
@@ -845,6 +845,9 @@ class IvfRq:
|
|||||||
accelerator: Optional[str] = None
|
accelerator: Optional[str] = None
|
||||||
|
|
||||||
|
|
||||||
|
# The API reference renders this module with a single mkdocstrings directive,
|
||||||
|
# which only picks up names listed here. New public names must be added to this
|
||||||
|
# list, or they will silently go undocumented.
|
||||||
__all__ = [
|
__all__ = [
|
||||||
"BTree",
|
"BTree",
|
||||||
"IvfPq",
|
"IvfPq",
|
||||||
|
|||||||
@@ -0,0 +1,105 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
"""Handles to operations a server may run asynchronously."""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
from datetime import timedelta
|
||||||
|
from typing import Optional
|
||||||
|
|
||||||
|
from lancedb.background_loop import LOOP
|
||||||
|
|
||||||
|
from . import _lancedb
|
||||||
|
|
||||||
|
|
||||||
|
class AsyncJob:
|
||||||
|
"""A handle to an operation that may still be running.
|
||||||
|
|
||||||
|
The operation may already be complete when the handle is created.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self, inner: Optional["_lancedb.Job"]):
|
||||||
|
self._inner = inner
|
||||||
|
|
||||||
|
@property
|
||||||
|
def id(self) -> Optional[str]:
|
||||||
|
"""Identifies the operation on the server that is running it.
|
||||||
|
|
||||||
|
Returned for correlating with server logs or the jobs API. Operations
|
||||||
|
that run in this process have no server id and return `None`. The value
|
||||||
|
is opaque: parsing it or storing it to resume the job later is not
|
||||||
|
supported.
|
||||||
|
"""
|
||||||
|
return self._inner.id if self._inner is not None else None
|
||||||
|
|
||||||
|
async def status(self) -> str:
|
||||||
|
"""The operation's current lifecycle state: "running", "finished",
|
||||||
|
"failed", or "cancelled".
|
||||||
|
|
||||||
|
A point snapshot; unlike `wait` it does not block or raise on a
|
||||||
|
terminal failure state. States a newer server reports that this
|
||||||
|
client version does not know pass through as-is.
|
||||||
|
"""
|
||||||
|
if self._inner is None:
|
||||||
|
return "finished"
|
||||||
|
return await self._inner.status()
|
||||||
|
|
||||||
|
async def wait(self, timeout: Optional[timedelta] = None):
|
||||||
|
"""Wait until the operation reaches a terminal state.
|
||||||
|
|
||||||
|
Raises `JobFailedError` if the operation failed, `JobCancelledError`
|
||||||
|
if it was cancelled, and `TimeoutError` if `timeout` elapses first.
|
||||||
|
"""
|
||||||
|
if self._inner is None:
|
||||||
|
return
|
||||||
|
if timeout is None:
|
||||||
|
await self._inner.wait()
|
||||||
|
else:
|
||||||
|
await asyncio.wait_for(self._inner.wait(), timeout.total_seconds())
|
||||||
|
|
||||||
|
async def cancel(self):
|
||||||
|
"""Request cancellation. Cancelling a finished operation is a no-op."""
|
||||||
|
if self._inner is None:
|
||||||
|
return
|
||||||
|
await self._inner.cancel()
|
||||||
|
|
||||||
|
|
||||||
|
class Job:
|
||||||
|
"""Synchronous counterpart of `AsyncJob`."""
|
||||||
|
|
||||||
|
def __init__(self, inner: Optional[AsyncJob]):
|
||||||
|
self._inner = inner
|
||||||
|
|
||||||
|
@property
|
||||||
|
def id(self) -> Optional[str]:
|
||||||
|
"""Identifies the operation on the server that is running it.
|
||||||
|
|
||||||
|
See :attr:`AsyncJob.id`.
|
||||||
|
"""
|
||||||
|
return self._inner.id if self._inner is not None else None
|
||||||
|
|
||||||
|
def status(self) -> str:
|
||||||
|
"""The operation's current lifecycle state: "running", "finished",
|
||||||
|
"failed", or "cancelled".
|
||||||
|
|
||||||
|
See :meth:`AsyncJob.status`.
|
||||||
|
"""
|
||||||
|
if self._inner is None:
|
||||||
|
return "finished"
|
||||||
|
return LOOP.run(self._inner.status())
|
||||||
|
|
||||||
|
def wait(self, timeout: Optional[timedelta] = None):
|
||||||
|
"""Block until the operation reaches a terminal state.
|
||||||
|
|
||||||
|
Raises `JobFailedError` if the operation failed, `JobCancelledError`
|
||||||
|
if it was cancelled, and `TimeoutError` if `timeout` elapses first.
|
||||||
|
"""
|
||||||
|
if self._inner is None:
|
||||||
|
return
|
||||||
|
LOOP.run(self._inner.wait(timeout))
|
||||||
|
|
||||||
|
def cancel(self):
|
||||||
|
"""Request cancellation. Cancelling a finished operation is a no-op."""
|
||||||
|
if self._inner is None:
|
||||||
|
return
|
||||||
|
LOOP.run(self._inner.cancel())
|
||||||
@@ -92,8 +92,10 @@ class LanceMergeInsertBuilder(object):
|
|||||||
self._when_not_matched_by_source_delete = True
|
self._when_not_matched_by_source_delete = True
|
||||||
if isinstance(condition, Expr):
|
if isinstance(condition, Expr):
|
||||||
self._when_not_matched_by_source_condition_expr = condition._inner
|
self._when_not_matched_by_source_condition_expr = condition._inner
|
||||||
elif condition is not None:
|
self._when_not_matched_by_source_condition = None
|
||||||
|
else:
|
||||||
self._when_not_matched_by_source_condition = condition
|
self._when_not_matched_by_source_condition = condition
|
||||||
|
self._when_not_matched_by_source_condition_expr = None
|
||||||
return self
|
return self
|
||||||
|
|
||||||
def use_index(self, use_index: bool) -> LanceMergeInsertBuilder:
|
def use_index(self, use_index: bool) -> LanceMergeInsertBuilder:
|
||||||
|
|||||||
@@ -38,7 +38,11 @@ from lance_namespace_urllib3_client.models.query_table_request_vector import (
|
|||||||
QueryTableRequestVector,
|
QueryTableRequestVector,
|
||||||
)
|
)
|
||||||
from lance_namespace_urllib3_client.models.string_fts_query import StringFtsQuery
|
from lance_namespace_urllib3_client.models.string_fts_query import StringFtsQuery
|
||||||
from lance_namespace.errors import NamespaceNotEmptyError, TableNotFoundError
|
from lance_namespace.errors import (
|
||||||
|
NamespaceNotEmptyError,
|
||||||
|
NamespaceNotFoundError,
|
||||||
|
TableNotFoundError,
|
||||||
|
)
|
||||||
from lancedb._lancedb import (
|
from lancedb._lancedb import (
|
||||||
connect_namespace as _connect_namespace,
|
connect_namespace as _connect_namespace,
|
||||||
connect_namespace_client as _connect_namespace_client,
|
connect_namespace_client as _connect_namespace_client,
|
||||||
@@ -53,6 +57,8 @@ from lance_namespace import (
|
|||||||
DropNamespaceResponse,
|
DropNamespaceResponse,
|
||||||
ListNamespacesResponse,
|
ListNamespacesResponse,
|
||||||
ListTablesResponse,
|
ListTablesResponse,
|
||||||
|
NamespaceExistsRequest,
|
||||||
|
TableExistsRequest,
|
||||||
)
|
)
|
||||||
from lancedb.table import AsyncTable, LanceTable, Table
|
from lancedb.table import AsyncTable, LanceTable, Table
|
||||||
from lancedb.util import validate_table_name
|
from lancedb.util import validate_table_name
|
||||||
@@ -780,6 +786,51 @@ class LanceNamespaceDBConnection(DBConnection):
|
|||||||
"""
|
"""
|
||||||
return LOOP.run(self._inner.describe_namespace(namespace_path))
|
return LOOP.run(self._inner.describe_namespace(namespace_path))
|
||||||
|
|
||||||
|
@override
|
||||||
|
def namespace_exists(self, namespace_id: List[str]) -> bool:
|
||||||
|
"""
|
||||||
|
Check if a namespace exists.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
namespace_id : List[str]
|
||||||
|
The namespace identifier to check.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
bool
|
||||||
|
True if the namespace exists, False otherwise.
|
||||||
|
"""
|
||||||
|
request = NamespaceExistsRequest(id=namespace_id)
|
||||||
|
try:
|
||||||
|
self._namespace_client.namespace_exists(request)
|
||||||
|
return True
|
||||||
|
except NamespaceNotFoundError:
|
||||||
|
return False
|
||||||
|
|
||||||
|
@override
|
||||||
|
def table_exists(self, table_id: List[str]) -> bool:
|
||||||
|
"""
|
||||||
|
Check if a table exists.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
table_id : List[str]
|
||||||
|
The table identifier to check (full path including namespace
|
||||||
|
segments and table name).
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
bool
|
||||||
|
True if the table exists, False otherwise.
|
||||||
|
"""
|
||||||
|
request = TableExistsRequest(id=table_id)
|
||||||
|
try:
|
||||||
|
self._namespace_client.table_exists(request)
|
||||||
|
return True
|
||||||
|
except TableNotFoundError:
|
||||||
|
return False
|
||||||
|
|
||||||
@override
|
@override
|
||||||
def list_tables(
|
def list_tables(
|
||||||
self,
|
self,
|
||||||
@@ -1233,6 +1284,49 @@ class AsyncLanceNamespaceDBConnection:
|
|||||||
"""
|
"""
|
||||||
return await self._inner.describe_namespace(namespace_path)
|
return await self._inner.describe_namespace(namespace_path)
|
||||||
|
|
||||||
|
async def namespace_exists(self, namespace_id: List[str]) -> bool:
|
||||||
|
"""
|
||||||
|
Check if a namespace exists.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
namespace_id : List[str]
|
||||||
|
The namespace identifier to check.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
bool
|
||||||
|
True if the namespace exists, False otherwise.
|
||||||
|
"""
|
||||||
|
request = NamespaceExistsRequest(id=namespace_id)
|
||||||
|
try:
|
||||||
|
self._namespace_client.namespace_exists(request)
|
||||||
|
return True
|
||||||
|
except NamespaceNotFoundError:
|
||||||
|
return False
|
||||||
|
|
||||||
|
async def table_exists(self, table_id: List[str]) -> bool:
|
||||||
|
"""
|
||||||
|
Check if a table exists.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
table_id : List[str]
|
||||||
|
The table identifier to check (full path including namespace
|
||||||
|
segments and table name).
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
bool
|
||||||
|
True if the table exists, False otherwise.
|
||||||
|
"""
|
||||||
|
request = TableExistsRequest(id=table_id)
|
||||||
|
try:
|
||||||
|
self._namespace_client.table_exists(request)
|
||||||
|
return True
|
||||||
|
except TableNotFoundError:
|
||||||
|
return False
|
||||||
|
|
||||||
async def list_tables(
|
async def list_tables(
|
||||||
self,
|
self,
|
||||||
namespace_path: Optional[List[str]] = None,
|
namespace_path: Optional[List[str]] = None,
|
||||||
|
|||||||
@@ -226,7 +226,7 @@ class PermutationBuilder:
|
|||||||
|
|
||||||
async def do_execute():
|
async def do_execute():
|
||||||
inner_tbl = await self._async.execute()
|
inner_tbl = await self._async.execute()
|
||||||
return LanceTable.from_inner(inner_tbl)
|
return await LanceTable.from_inner(inner_tbl)
|
||||||
|
|
||||||
return LOOP.run(do_execute())
|
return LOOP.run(do_execute())
|
||||||
|
|
||||||
@@ -438,7 +438,8 @@ class Permutation:
|
|||||||
_reader: Optional[PermutationReader] = None,
|
_reader: Optional[PermutationReader] = None,
|
||||||
):
|
):
|
||||||
"""
|
"""
|
||||||
Internal constructor. Use [from_tables](#from_tables) instead.
|
Internal constructor. Use
|
||||||
|
[from_tables][lancedb.permutation.Permutation.from_tables] instead.
|
||||||
"""
|
"""
|
||||||
assert base_table is not None, "base_table is required"
|
assert base_table is not None, "base_table is required"
|
||||||
assert selection is not None, "selection is required"
|
assert selection is not None, "selection is required"
|
||||||
@@ -985,8 +986,9 @@ class Permutation:
|
|||||||
types. Conversion of strings, lists, and structs will require creating python
|
types. Conversion of strings, lists, and structs will require creating python
|
||||||
objects and this is not zero-copy.
|
objects and this is not zero-copy.
|
||||||
|
|
||||||
For custom formatting, use [with_transform](#with_transform) which overrides
|
For custom formatting, use
|
||||||
this method.
|
[with_transform][lancedb.permutation.Permutation.with_transform] which
|
||||||
|
overrides this method.
|
||||||
"""
|
"""
|
||||||
assert format is not None, "format is required"
|
assert format is not None, "format is required"
|
||||||
if format == "python":
|
if format == "python":
|
||||||
@@ -1061,7 +1063,8 @@ class Permutation:
|
|||||||
Note: this method returns a new permutation and does not modify `self`
|
Note: this method returns a new permutation and does not modify `self`
|
||||||
It is provided for compatibility with the huggingface Dataset API.
|
It is provided for compatibility with the huggingface Dataset API.
|
||||||
|
|
||||||
Use [with_skip](#with_skip) instead to avoid confusion.
|
Use [with_skip][lancedb.permutation.Permutation.with_skip] instead to
|
||||||
|
avoid confusion.
|
||||||
"""
|
"""
|
||||||
return self.with_skip(skip)
|
return self.with_skip(skip)
|
||||||
|
|
||||||
@@ -1084,7 +1087,8 @@ class Permutation:
|
|||||||
Note: this method returns a new permutation and does not modify `self`
|
Note: this method returns a new permutation and does not modify `self`
|
||||||
It is provided for compatibility with the huggingface Dataset API.
|
It is provided for compatibility with the huggingface Dataset API.
|
||||||
|
|
||||||
Use [with_take](#with_take) instead to avoid confusion.
|
Use [with_take][lancedb.permutation.Permutation.with_take] instead to
|
||||||
|
avoid confusion.
|
||||||
"""
|
"""
|
||||||
return self.with_take(limit)
|
return self.with_take(limit)
|
||||||
|
|
||||||
@@ -1107,7 +1111,8 @@ class Permutation:
|
|||||||
Note: this method returns a new permutation and does not modify `self`
|
Note: this method returns a new permutation and does not modify `self`
|
||||||
It is provided for compatibility with the huggingface Dataset API.
|
It is provided for compatibility with the huggingface Dataset API.
|
||||||
|
|
||||||
Use [with_repeat](#with_repeat) instead to avoid confusion.
|
Use [with_repeat][lancedb.permutation.Permutation.with_repeat] instead
|
||||||
|
to avoid confusion.
|
||||||
"""
|
"""
|
||||||
return self.with_repeat(times)
|
return self.with_repeat(times)
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1 @@
|
|||||||
|
|
||||||
@@ -153,6 +153,16 @@ def Vector(
|
|||||||
return FixedSizeList
|
return FixedSizeList
|
||||||
|
|
||||||
|
|
||||||
|
def _raise_bare_vector_error(*_args):
|
||||||
|
raise TypeError("Vector must be parameterized with a dimension, e.g. Vector(128).")
|
||||||
|
|
||||||
|
|
||||||
|
# Pydantic v1 and v2 otherwise treat the bare Vector factory as a field validator
|
||||||
|
# and inspect its signature, which produces misleading errors about internal types.
|
||||||
|
setattr(Vector, "__get_validators__", _raise_bare_vector_error)
|
||||||
|
setattr(Vector, "__get_pydantic_core_schema__", _raise_bare_vector_error)
|
||||||
|
|
||||||
|
|
||||||
def MultiVector(
|
def MultiVector(
|
||||||
dim: int, value_type: pa.DataType = pa.float32(), nullable: bool = True
|
dim: int, value_type: pa.DataType = pa.float32(), nullable: bool = True
|
||||||
) -> Type:
|
) -> Type:
|
||||||
|
|||||||
@@ -52,7 +52,6 @@ from ._blob import (
|
|||||||
finalize_blob_query_table,
|
finalize_blob_query_table,
|
||||||
replace_v2_blob_columns_with_bytes,
|
replace_v2_blob_columns_with_bytes,
|
||||||
replace_v2_blob_columns_with_bytes_sync,
|
replace_v2_blob_columns_with_bytes_sync,
|
||||||
supports_blob_auto_row_id,
|
|
||||||
validate_blob_mode,
|
validate_blob_mode,
|
||||||
)
|
)
|
||||||
from .types import BlobMode, QueryProjection
|
from .types import BlobMode, QueryProjection
|
||||||
@@ -651,7 +650,8 @@ class Query(pydantic.BaseModel):
|
|||||||
distance_type : Optional[str]
|
distance_type : Optional[str]
|
||||||
the distance type to use for vector search
|
the distance type to use for vector search
|
||||||
|
|
||||||
This can be l2 (default), cosine and dot. See [metric definitions][search] for
|
This can be l2 (default), cosine and dot. See
|
||||||
|
[metric definitions](https://lancedb.com/docs/search/vector-search/) for
|
||||||
more details.
|
more details.
|
||||||
|
|
||||||
If this is not a vector search this will be None.
|
If this is not a vector search this will be None.
|
||||||
@@ -664,8 +664,9 @@ class Query(pydantic.BaseModel):
|
|||||||
|
|
||||||
- A higher number makes search more accurate but also slower.
|
- A higher number makes search more accurate but also slower.
|
||||||
|
|
||||||
- See discussion in [Querying an ANN Index][querying-an-ann-index] for
|
- See discussion in
|
||||||
tuning advice.
|
[Querying an ANN Index](https://lancedb.com/docs/indexing/)
|
||||||
|
for tuning advice.
|
||||||
|
|
||||||
Will be None if this is not a vector search.
|
Will be None if this is not a vector search.
|
||||||
refine_factor : Optional[int]
|
refine_factor : Optional[int]
|
||||||
@@ -673,8 +674,9 @@ class Query(pydantic.BaseModel):
|
|||||||
|
|
||||||
- A higher number makes search more accurate but also slower.
|
- A higher number makes search more accurate but also slower.
|
||||||
|
|
||||||
- See discussion in [Querying an ANN Index][querying-an-ann-index] for
|
- See discussion in
|
||||||
tuning advice.
|
[Querying an ANN Index](https://lancedb.com/docs/indexing/)
|
||||||
|
for tuning advice.
|
||||||
|
|
||||||
Will be None if this is not a vector search.
|
Will be None if this is not a vector search.
|
||||||
lower_bound : Optional[float]
|
lower_bound : Optional[float]
|
||||||
@@ -1277,10 +1279,7 @@ class LanceQueryBuilder(ABC):
|
|||||||
return self._with_row_id is True
|
return self._with_row_id is True
|
||||||
|
|
||||||
def _blob_auto_row_id_enabled(self) -> bool:
|
def _blob_auto_row_id_enabled(self) -> bool:
|
||||||
if not supports_blob_auto_row_id(self._table):
|
|
||||||
return False
|
|
||||||
return blob_auto_row_id_for_scan(
|
return blob_auto_row_id_for_scan(
|
||||||
self._table,
|
|
||||||
self._table.schema,
|
self._table.schema,
|
||||||
self._columns,
|
self._columns,
|
||||||
with_row_id=self._with_row_id,
|
with_row_id=self._with_row_id,
|
||||||
@@ -1651,8 +1650,8 @@ class LanceVectorQueryBuilder(LanceQueryBuilder):
|
|||||||
Higher values will yield better recall (more likely to find vectors if
|
Higher values will yield better recall (more likely to find vectors if
|
||||||
they exist) at the expense of latency.
|
they exist) at the expense of latency.
|
||||||
|
|
||||||
See discussion in [Querying an ANN Index][querying-an-ann-index] for
|
See discussion in [Querying an ANN Index](https://lancedb.com/docs/indexing/)
|
||||||
tuning advice.
|
for tuning advice.
|
||||||
|
|
||||||
This method sets both the minimum and maximum number of probes to the same
|
This method sets both the minimum and maximum number of probes to the same
|
||||||
value. See `minimum_nprobes` and `maximum_nprobes` for more fine-grained
|
value. See `minimum_nprobes` and `maximum_nprobes` for more fine-grained
|
||||||
@@ -1752,8 +1751,8 @@ class LanceVectorQueryBuilder(LanceQueryBuilder):
|
|||||||
As an example, a refine factor of 2 will sample 2x as many vectors as
|
As an example, a refine factor of 2 will sample 2x as many vectors as
|
||||||
requested, re-ranks them, and returns the top half most relevant results.
|
requested, re-ranks them, and returns the top half most relevant results.
|
||||||
|
|
||||||
See discussion in [Querying an ANN Index][querying-an-ann-index] for
|
See discussion in [Querying an ANN Index](https://lancedb.com/docs/indexing/)
|
||||||
tuning advice.
|
for tuning advice.
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
@@ -2698,7 +2697,7 @@ class LanceHybridQueryBuilder(LanceQueryBuilder):
|
|||||||
self._fts_query.phrase_query(True)
|
self._fts_query.phrase_query(True)
|
||||||
if self._distance_type:
|
if self._distance_type:
|
||||||
self._vector_query.metric(self._distance_type)
|
self._vector_query.metric(self._distance_type)
|
||||||
if self._minimum_nprobes:
|
if self._minimum_nprobes is not None:
|
||||||
self._vector_query.minimum_nprobes(self._minimum_nprobes)
|
self._vector_query.minimum_nprobes(self._minimum_nprobes)
|
||||||
if self._maximum_nprobes is not None:
|
if self._maximum_nprobes is not None:
|
||||||
self._vector_query.maximum_nprobes(self._maximum_nprobes)
|
self._vector_query.maximum_nprobes(self._maximum_nprobes)
|
||||||
@@ -2771,7 +2770,7 @@ class AsyncQueryBase(object):
|
|||||||
)
|
)
|
||||||
|
|
||||||
async def _maybe_add_blob_row_id(self) -> None:
|
async def _maybe_add_blob_row_id(self) -> None:
|
||||||
if self._table is None or not supports_blob_auto_row_id(self._table):
|
if self._table is None:
|
||||||
self._blob_auto_row_id = False
|
self._blob_auto_row_id = False
|
||||||
self._blob_paths = ()
|
self._blob_paths = ()
|
||||||
return
|
return
|
||||||
@@ -2779,7 +2778,6 @@ class AsyncQueryBase(object):
|
|||||||
req = self._inner.to_query_request()
|
req = self._inner.to_query_request()
|
||||||
schema = await self._table.schema()
|
schema = await self._table.schema()
|
||||||
self._blob_auto_row_id = blob_auto_row_id_for_scan(
|
self._blob_auto_row_id = blob_auto_row_id_for_scan(
|
||||||
self._table,
|
|
||||||
schema,
|
schema,
|
||||||
req.select,
|
req.select,
|
||||||
with_row_id=self._with_row_id,
|
with_row_id=self._with_row_id,
|
||||||
@@ -3031,7 +3029,6 @@ class AsyncQueryBase(object):
|
|||||||
|
|
||||||
schema = await self._table.schema()
|
schema = await self._table.schema()
|
||||||
blob_auto_row_id = blob_auto_row_id_for_scan(
|
blob_auto_row_id = blob_auto_row_id_for_scan(
|
||||||
self._table,
|
|
||||||
schema,
|
schema,
|
||||||
query.columns,
|
query.columns,
|
||||||
with_row_id=self._with_row_id,
|
with_row_id=self._with_row_id,
|
||||||
@@ -3379,8 +3376,9 @@ class AsyncQuery(AsyncStandardQuery):
|
|||||||
are various ANN search parameters that will let you fine tune your recall
|
are various ANN search parameters that will let you fine tune your recall
|
||||||
accuracy vs search latency.
|
accuracy vs search latency.
|
||||||
|
|
||||||
Vector searches always have a [limit][]. If `limit` has not been called then
|
Vector searches always have a
|
||||||
a default `limit` of 10 will be used.
|
[limit][lancedb.query.AsyncVectorQuery.limit]. If `limit` has not been
|
||||||
|
called then a default `limit` of 10 will be used.
|
||||||
|
|
||||||
Typically, a single vector is passed in as the query. However, you can also
|
Typically, a single vector is passed in as the query. However, you can also
|
||||||
pass in multiple vectors. When multiple vectors are passed in, if the vector
|
pass in multiple vectors. When multiple vectors are passed in, if the vector
|
||||||
@@ -3511,8 +3509,9 @@ class AsyncFTSQuery(AsyncStandardQuery):
|
|||||||
are various ANN search parameters that will let you fine tune your recall
|
are various ANN search parameters that will let you fine tune your recall
|
||||||
accuracy vs search latency.
|
accuracy vs search latency.
|
||||||
|
|
||||||
Hybrid searches always have a [limit][]. If `limit` has not been called then
|
Hybrid searches always have a
|
||||||
a default `limit` of 10 will be used.
|
[limit][lancedb.query.AsyncHybridQuery.limit]. If `limit` has not been
|
||||||
|
called then a default `limit` of 10 will be used.
|
||||||
|
|
||||||
Typically, a single vector is passed in as the query. However, you can also
|
Typically, a single vector is passed in as the query. However, you can also
|
||||||
pass in multiple vectors. This can be useful if you want to find the nearest
|
pass in multiple vectors. This can be useful if you want to find the nearest
|
||||||
@@ -3875,10 +3874,9 @@ class AsyncHybridQuery(AsyncStandardQuery, AsyncVectorQueryBase):
|
|||||||
req = fts_query._inner.to_query_request()
|
req = fts_query._inner.to_query_request()
|
||||||
blob_auto_row_id = False
|
blob_auto_row_id = False
|
||||||
blob_paths: tuple[str, ...] = ()
|
blob_paths: tuple[str, ...] = ()
|
||||||
if self._table is not None and supports_blob_auto_row_id(self._table):
|
if self._table is not None:
|
||||||
schema = await self._table.schema()
|
schema = await self._table.schema()
|
||||||
blob_auto_row_id = blob_auto_row_id_for_scan(
|
blob_auto_row_id = blob_auto_row_id_for_scan(
|
||||||
self._table,
|
|
||||||
schema,
|
schema,
|
||||||
req.select,
|
req.select,
|
||||||
with_row_id=self._with_row_id,
|
with_row_id=self._with_row_id,
|
||||||
|
|||||||
@@ -11,6 +11,9 @@ from lancedb import __version__
|
|||||||
from .header import HeaderProvider
|
from .header import HeaderProvider
|
||||||
from .oauth import OAuthConfig, OAuthFlowType
|
from .oauth import OAuthConfig, OAuthFlowType
|
||||||
|
|
||||||
|
# The API reference renders this module with a single mkdocstrings directive,
|
||||||
|
# which only picks up names listed here. New public names must be added to this
|
||||||
|
# list, or they will silently go undocumented.
|
||||||
__all__ = [
|
__all__ = [
|
||||||
"TimeoutConfig",
|
"TimeoutConfig",
|
||||||
"RetryConfig",
|
"RetryConfig",
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ import json
|
|||||||
import logging
|
import logging
|
||||||
from concurrent.futures import ThreadPoolExecutor
|
from concurrent.futures import ThreadPoolExecutor
|
||||||
import sys
|
import sys
|
||||||
from typing import Any, Dict, Iterable, List, Optional, Union
|
from typing import TYPE_CHECKING, Any, Dict, Iterable, List, Optional, Union
|
||||||
from urllib.parse import urlparse
|
from urllib.parse import urlparse
|
||||||
import warnings
|
import warnings
|
||||||
|
|
||||||
@@ -23,6 +23,10 @@ import pyarrow as pa
|
|||||||
|
|
||||||
from ..common import DATA
|
from ..common import DATA
|
||||||
from ..db import DBConnection, LOOP
|
from ..db import DBConnection, LOOP
|
||||||
|
from ..job import Job
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from .._lancedb import JobDescription, JobInfo
|
||||||
from ..embeddings import EmbeddingFunctionConfig
|
from ..embeddings import EmbeddingFunctionConfig
|
||||||
from lance_namespace import (
|
from lance_namespace import (
|
||||||
LanceNamespace,
|
LanceNamespace,
|
||||||
@@ -415,6 +419,11 @@ class RemoteDBConnection(DBConnection):
|
|||||||
|
|
||||||
if namespace_path is None:
|
if namespace_path is None:
|
||||||
namespace_path = []
|
namespace_path = []
|
||||||
|
if storage_options is not None:
|
||||||
|
logging.info(
|
||||||
|
"storage_options is ignored in LanceDb Cloud"
|
||||||
|
" (storage is managed; set storage_options on connect() instead)"
|
||||||
|
)
|
||||||
if index_cache_size is not None:
|
if index_cache_size is not None:
|
||||||
logging.info(
|
logging.info(
|
||||||
"index_cache_size is ignored in LanceDb Cloud"
|
"index_cache_size is ignored in LanceDb Cloud"
|
||||||
@@ -684,6 +693,47 @@ class RemoteDBConnection(DBConnection):
|
|||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|
||||||
|
@override
|
||||||
|
def job(self, job_id: str) -> Job:
|
||||||
|
"""A [Job][lancedb.job.Job] handle for a server-side job by id.
|
||||||
|
|
||||||
|
The handle is constructed without a server round trip; an unknown id
|
||||||
|
surfaces when the handle is used. Dropping the handle has no effect
|
||||||
|
on the job itself.
|
||||||
|
"""
|
||||||
|
return Job(self._conn.job(job_id))
|
||||||
|
|
||||||
|
@override
|
||||||
|
def list_jobs(self) -> List["JobInfo"]:
|
||||||
|
"""List server-side jobs across the database's tables."""
|
||||||
|
return LOOP.run(self._conn.list_jobs())
|
||||||
|
|
||||||
|
@override
|
||||||
|
def get_job(self, job_id: str) -> Optional["JobDescription"]:
|
||||||
|
"""Describe a single server-side job by id.
|
||||||
|
|
||||||
|
Returns None when the server has no such job.
|
||||||
|
"""
|
||||||
|
return LOOP.run(self._conn.get_job(job_id))
|
||||||
|
|
||||||
|
@override
|
||||||
|
def cancel_job(self, job_id: str) -> bool:
|
||||||
|
"""Request cancellation of a server-side job by id.
|
||||||
|
|
||||||
|
Returns True if the server accepted the cancellation, False if no
|
||||||
|
such job exists. Cancelling an already-terminal job is a no-op
|
||||||
|
success.
|
||||||
|
"""
|
||||||
|
return LOOP.run(self._conn.cancel_job(job_id))
|
||||||
|
|
||||||
|
@override
|
||||||
|
def job_history(self, job_id: Optional[str] = None) -> List[pa.RecordBatch]:
|
||||||
|
"""The lifecycle event history of a server-side job, as Arrow batches.
|
||||||
|
|
||||||
|
Lists history across all jobs when `job_id` is None.
|
||||||
|
"""
|
||||||
|
return LOOP.run(self._conn.job_history(job_id))
|
||||||
|
|
||||||
@override
|
@override
|
||||||
def namespace_client(self) -> LanceNamespace:
|
def namespace_client(self) -> LanceNamespace:
|
||||||
"""Get the equivalent namespace client for this connection.
|
"""Get the equivalent namespace client for this connection.
|
||||||
|
|||||||
@@ -53,9 +53,9 @@ class RetryError(LanceDBClientError):
|
|||||||
"""An error that occurs when the client has exceeded the maximum number of retries.
|
"""An error that occurs when the client has exceeded the maximum number of retries.
|
||||||
|
|
||||||
The retry strategy can be adjusted by setting the
|
The retry strategy can be adjusted by setting the
|
||||||
[retry_config](lancedb.remote.ClientConfig.retry_config) in the client
|
[retry_config][lancedb.remote.ClientConfig.retry_config] in the client
|
||||||
configuration. This is passed in the `client_config` argument of
|
configuration. This is passed in the `client_config` argument of
|
||||||
[connect](lancedb.connect) and [connect_async](lancedb.connect_async).
|
[connect][lancedb.connect] and [connect_async][lancedb.connect_async].
|
||||||
|
|
||||||
The __cause__ attribute of this exception will be the last exception that
|
The __cause__ attribute of this exception will be the last exception that
|
||||||
caused the retry to fail. It will be an
|
caused the retry to fail. It will be an
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ from typing import (
|
|||||||
import warnings
|
import warnings
|
||||||
|
|
||||||
from lancedb import __version__
|
from lancedb import __version__
|
||||||
|
from lancedb._blob import BlobFile
|
||||||
|
|
||||||
from lancedb._lancedb import (
|
from lancedb._lancedb import (
|
||||||
AddColumnsResult,
|
AddColumnsResult,
|
||||||
@@ -47,6 +48,7 @@ from lancedb.index import (
|
|||||||
IvfSq,
|
IvfSq,
|
||||||
LabelList,
|
LabelList,
|
||||||
)
|
)
|
||||||
|
from lancedb.job import Job
|
||||||
from lancedb.remote.db import LOOP
|
from lancedb.remote.db import LOOP
|
||||||
from lancedb.table import IndexConfigType, KNOWN_METRICS
|
from lancedb.table import IndexConfigType, KNOWN_METRICS
|
||||||
import pyarrow as pa
|
import pyarrow as pa
|
||||||
@@ -540,6 +542,34 @@ class RemoteTable(Table):
|
|||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def create_index_async(
|
||||||
|
self,
|
||||||
|
column: str,
|
||||||
|
*,
|
||||||
|
config: IndexConfigType,
|
||||||
|
replace: Optional[bool] = None,
|
||||||
|
wait_timeout: Optional[timedelta] = None,
|
||||||
|
name: Optional[str] = None,
|
||||||
|
train: bool = True,
|
||||||
|
) -> Job:
|
||||||
|
"""Create an index, returning a handle to the indexing job.
|
||||||
|
|
||||||
|
The job may already be complete when returned; callers must not assume
|
||||||
|
the index exists until :meth:`Job.wait` returns.
|
||||||
|
"""
|
||||||
|
return Job(
|
||||||
|
LOOP.run(
|
||||||
|
self._table.create_index_async(
|
||||||
|
column,
|
||||||
|
replace=replace,
|
||||||
|
config=config,
|
||||||
|
wait_timeout=wait_timeout,
|
||||||
|
name=name,
|
||||||
|
train=train,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
def _is_legacy_create_index_call(
|
def _is_legacy_create_index_call(
|
||||||
self,
|
self,
|
||||||
first_arg: str,
|
first_arg: str,
|
||||||
@@ -580,8 +610,9 @@ class RemoteTable(Table):
|
|||||||
progress: Optional[Union[bool, Callable, Any]] = None,
|
progress: Optional[Union[bool, Callable, Any]] = None,
|
||||||
write_parallelism: Optional[int] = None,
|
write_parallelism: Optional[int] = None,
|
||||||
) -> AddResult:
|
) -> AddResult:
|
||||||
"""Add more data to the [Table](Table). It has the same API signature as
|
"""Add more data to the [Table][lancedb.table.Table].
|
||||||
the OSS version.
|
|
||||||
|
It has the same API signature as the OSS version.
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
@@ -641,7 +672,8 @@ class RemoteTable(Table):
|
|||||||
fast_search: bool = False,
|
fast_search: bool = False,
|
||||||
) -> LanceVectorQueryBuilder:
|
) -> LanceVectorQueryBuilder:
|
||||||
"""Create a search query to find the nearest neighbors
|
"""Create a search query to find the nearest neighbors
|
||||||
of the given query vector. We currently support [vector search][search]
|
of the given query vector. We currently support
|
||||||
|
[vector search](https://lancedb.com/docs/search/vector-search/)
|
||||||
|
|
||||||
All query options are defined in
|
All query options are defined in
|
||||||
[LanceVectorQueryBuilder][lancedb.query.LanceVectorQueryBuilder].
|
[LanceVectorQueryBuilder][lancedb.query.LanceVectorQueryBuilder].
|
||||||
@@ -1037,22 +1069,22 @@ class RemoteTable(Table):
|
|||||||
)
|
)
|
||||||
|
|
||||||
def blob_columns(self) -> list[str]:
|
def blob_columns(self) -> list[str]:
|
||||||
raise NotImplementedError(
|
return LOOP.run(self._table.blob_columns())
|
||||||
"blob_columns() is not yet supported on the LanceDB Cloud"
|
|
||||||
)
|
|
||||||
|
|
||||||
def fetch_blobs(self, column: str, row_ids) -> pa.LargeBinaryArray:
|
def fetch_blobs(
|
||||||
raise NotImplementedError("fetch_blobs() is not supported on LanceDB Cloud")
|
self, column: str, row_ids: Union[list[int], pa.Table]
|
||||||
|
) -> pa.LargeBinaryArray:
|
||||||
|
return LOOP.run(self._table.fetch_blobs(column, row_ids))
|
||||||
|
|
||||||
def fetch_blob_ranges(self, column: str, requests) -> pa.LargeBinaryArray:
|
def fetch_blob_ranges(self, column: str, requests) -> pa.LargeBinaryArray:
|
||||||
raise NotImplementedError(
|
raise NotImplementedError(
|
||||||
"fetch_blob_ranges() is not supported on LanceDB Cloud"
|
"fetch_blob_ranges() is not supported on LanceDB Cloud"
|
||||||
)
|
)
|
||||||
|
|
||||||
def fetch_blob_files(self, column: str, row_ids):
|
def fetch_blob_files(
|
||||||
raise NotImplementedError(
|
self, column: str, row_ids: Union[list[int], pa.Table]
|
||||||
"fetch_blob_files() is not supported on LanceDB Cloud"
|
) -> "list[Optional[BlobFile]]":
|
||||||
)
|
return LOOP.run(self._table.fetch_blob_files(column, row_ids))
|
||||||
|
|
||||||
def head(self, n=5) -> pa.Table:
|
def head(self, n=5) -> pa.Table:
|
||||||
"""
|
"""
|
||||||
|
|||||||
@@ -14,6 +14,9 @@ from .answerdotai import AnswerdotaiRerankers
|
|||||||
from .voyageai import VoyageAIReranker
|
from .voyageai import VoyageAIReranker
|
||||||
from .watsonx import WatsonxReranker
|
from .watsonx import WatsonxReranker
|
||||||
|
|
||||||
|
# The API reference renders this module with a single mkdocstrings directive,
|
||||||
|
# which only picks up names listed here. New public names must be added to this
|
||||||
|
# list, or they will silently go undocumented.
|
||||||
__all__ = [
|
__all__ = [
|
||||||
"Reranker",
|
"Reranker",
|
||||||
"CrossEncoderReranker",
|
"CrossEncoderReranker",
|
||||||
|
|||||||
+243
-39
@@ -40,6 +40,7 @@ from ._blob import (
|
|||||||
from .types import BlobMode
|
from .types import BlobMode
|
||||||
from lancedb.arrow import peek_reader
|
from lancedb.arrow import peek_reader
|
||||||
from lancedb.background_loop import LOOP, embedding_executor
|
from lancedb.background_loop import LOOP, embedding_executor
|
||||||
|
from lancedb.job import AsyncJob, Job
|
||||||
from .dependencies import (
|
from .dependencies import (
|
||||||
_check_for_hugging_face,
|
_check_for_hugging_face,
|
||||||
_check_for_lance,
|
_check_for_lance,
|
||||||
@@ -107,6 +108,11 @@ def _should_push_down_query_table(
|
|||||||
return namespace_client is not None and "QueryTable" in pushdown_operations
|
return namespace_client is not None and "QueryTable" in pushdown_operations
|
||||||
|
|
||||||
|
|
||||||
|
def _polars_predicate_pushdown_barrier(frame: Any) -> Any:
|
||||||
|
"""Return a Polars frame unchanged while blocking predicate pushdown."""
|
||||||
|
return frame
|
||||||
|
|
||||||
|
|
||||||
_MODEL_BACKED_TOKENIZER_PREFIXES = ("jieba", "lindera")
|
_MODEL_BACKED_TOKENIZER_PREFIXES = ("jieba", "lindera")
|
||||||
_MODEL_BACKED_TOKENIZER_ERRORS = (
|
_MODEL_BACKED_TOKENIZER_ERRORS = (
|
||||||
"unknown base tokenizer",
|
"unknown base tokenizer",
|
||||||
@@ -863,12 +869,18 @@ class Table(ABC):
|
|||||||
"""
|
"""
|
||||||
raise NotImplementedError
|
raise NotImplementedError
|
||||||
|
|
||||||
def to_polars(self, **kwargs) -> "pl.DataFrame":
|
def to_polars(self, **kwargs) -> "pl.LazyFrame":
|
||||||
"""Return the table as a polars.DataFrame.
|
"""Return the table as a Polars LazyFrame.
|
||||||
|
|
||||||
|
Note
|
||||||
|
----
|
||||||
|
The Polars streaming engine is not supported because it does not currently
|
||||||
|
implement Python PyArrow dataset scans. Use the default engine when collecting
|
||||||
|
this LazyFrame.
|
||||||
|
|
||||||
Returns
|
Returns
|
||||||
-------
|
-------
|
||||||
polars.DataFrame
|
polars.LazyFrame
|
||||||
"""
|
"""
|
||||||
raise NotImplementedError
|
raise NotImplementedError
|
||||||
|
|
||||||
@@ -977,6 +989,24 @@ class Table(ABC):
|
|||||||
"""
|
"""
|
||||||
raise NotImplementedError
|
raise NotImplementedError
|
||||||
|
|
||||||
|
def create_index_async(
|
||||||
|
self,
|
||||||
|
column: str,
|
||||||
|
*,
|
||||||
|
config: IndexConfigType,
|
||||||
|
replace: Optional[bool] = None,
|
||||||
|
wait_timeout: Optional[timedelta] = None,
|
||||||
|
name: Optional[str] = None,
|
||||||
|
train: bool = True,
|
||||||
|
) -> Job:
|
||||||
|
"""Create an index, returning a handle to the indexing job.
|
||||||
|
|
||||||
|
Takes the same arguments as :meth:`create_index`. The job may already
|
||||||
|
be complete when returned; callers must not assume the index exists
|
||||||
|
until :meth:`Job.wait` returns.
|
||||||
|
"""
|
||||||
|
raise NotImplementedError
|
||||||
|
|
||||||
def drop_index(self, name: str) -> None:
|
def drop_index(self, name: str) -> None:
|
||||||
"""
|
"""
|
||||||
Drop an index from the table.
|
Drop an index from the table.
|
||||||
@@ -1211,7 +1241,7 @@ class Table(ABC):
|
|||||||
progress: Optional[Union[bool, Callable, Any]] = None,
|
progress: Optional[Union[bool, Callable, Any]] = None,
|
||||||
write_parallelism: Optional[int] = None,
|
write_parallelism: Optional[int] = None,
|
||||||
) -> AddResult:
|
) -> AddResult:
|
||||||
"""Add more data to the [Table](Table).
|
"""Add more data to the [Table][lancedb.table.Table].
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
@@ -1343,8 +1373,8 @@ class Table(ABC):
|
|||||||
fts_columns: Optional[Union[str, List[str]]] = None,
|
fts_columns: Optional[Union[str, List[str]]] = None,
|
||||||
) -> LanceQueryBuilder:
|
) -> LanceQueryBuilder:
|
||||||
"""Create a search query to find the nearest neighbors
|
"""Create a search query to find the nearest neighbors
|
||||||
of the given query vector. We currently support [vector search][search]
|
of the given query vector. We currently support [vector search](https://lancedb.com/docs/search/vector-search/)
|
||||||
and [full-text search][experimental-full-text-search].
|
and [full-text search](https://lancedb.com/docs/search/full-text-search/).
|
||||||
|
|
||||||
All query options are defined in
|
All query options are defined in
|
||||||
[LanceQueryBuilder][lancedb.query.LanceQueryBuilder].
|
[LanceQueryBuilder][lancedb.query.LanceQueryBuilder].
|
||||||
@@ -1574,8 +1604,10 @@ class Table(ABC):
|
|||||||
"""Open lazy, seekable :class:`~lancedb._blob.BlobFile` handles.
|
"""Open lazy, seekable :class:`~lancedb._blob.BlobFile` handles.
|
||||||
|
|
||||||
Prefer this over :meth:`fetch_blobs` for large payloads. ``row_ids`` is
|
Prefer this over :meth:`fetch_blobs` for large payloads. ``row_ids`` is
|
||||||
a ``list[int]`` or query ``pyarrow.Table`` with ``_rowid`` (or stashed
|
a ``list[int]`` or a query ``pyarrow.Table`` carrying row identity via
|
||||||
row-id metadata). Null rows are ``None``. Local tables only.
|
``_rowid`` or a ``_lance_row_id`` field on the blob descriptor. Null
|
||||||
|
rows are ``None``. Remote tables require LanceDB Cloud server 0.5.0 or
|
||||||
|
newer.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
@@ -1778,7 +1810,7 @@ class Table(ABC):
|
|||||||
for faster reads.
|
for faster reads.
|
||||||
|
|
||||||
Arguments are passed onto Lance's
|
Arguments are passed onto Lance's
|
||||||
[compact_files][lance.dataset.DatasetOptimizer.compact_files].
|
`lance.dataset.DatasetOptimizer.compact_files`.
|
||||||
For most cases, the default should be fine.
|
For most cases, the default should be fine.
|
||||||
|
|
||||||
See Also
|
See Also
|
||||||
@@ -1832,6 +1864,8 @@ class Table(ABC):
|
|||||||
retrain: bool, default False
|
retrain: bool, default False
|
||||||
This parameter is no longer used and is deprecated.
|
This parameter is no longer used and is deprecated.
|
||||||
|
|
||||||
|
Notes
|
||||||
|
-----
|
||||||
The frequency an application should call optimize is based on the frequency of
|
The frequency an application should call optimize is based on the frequency of
|
||||||
data modifications. If data is frequently added, deleted, or updated then
|
data modifications. If data is frequently added, deleted, or updated then
|
||||||
optimize should be run frequently. A good rule of thumb is to run optimize if
|
optimize should be run frequently. A good rule of thumb is to run optimize if
|
||||||
@@ -1986,15 +2020,14 @@ class Table(ABC):
|
|||||||
change permanent you can use the `[Self::restore]` method.
|
change permanent you can use the `[Self::restore]` method.
|
||||||
|
|
||||||
Any operation that modifies the table will fail while the table is in a checked
|
Any operation that modifies the table will fail while the table is in a checked
|
||||||
out state.
|
out state. To return the table to a normal state use
|
||||||
|
`[Self::checkout_latest]`.
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
version: int | str,
|
version: int | str,
|
||||||
The version to check out. A version number (`int`) or a tag
|
The version to check out. A version number (`int`) or a tag
|
||||||
(`str`) can be provided.
|
(`str`) can be provided.
|
||||||
|
|
||||||
To return the table to a normal state use `[Self::checkout_latest]`
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
@@ -2160,11 +2193,15 @@ class LanceTable(Table):
|
|||||||
return self.name
|
return self.name
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def from_inner(cls, tbl: LanceDBTable):
|
async def from_inner(cls, tbl: LanceDBTable):
|
||||||
from .db import LanceDBConnection
|
from .db import AsyncConnection, LanceDBConnection
|
||||||
|
|
||||||
async_tbl = AsyncTable(tbl)
|
async_tbl = AsyncTable(tbl)
|
||||||
conn = LanceDBConnection.from_inner(tbl.database())
|
inner_conn = tbl.database()
|
||||||
|
read_consistency_interval = await AsyncConnection(
|
||||||
|
inner_conn
|
||||||
|
).get_read_consistency_interval()
|
||||||
|
conn = LanceDBConnection.from_inner(inner_conn, read_consistency_interval)
|
||||||
return cls(
|
return cls(
|
||||||
conn,
|
conn,
|
||||||
async_tbl.name,
|
async_tbl.name,
|
||||||
@@ -2468,13 +2505,7 @@ class LanceTable(Table):
|
|||||||
return LOOP.run(self._table.count_rows(filter))
|
return LOOP.run(self._table.count_rows(filter))
|
||||||
|
|
||||||
def __repr__(self) -> str:
|
def __repr__(self) -> str:
|
||||||
val = f"{self.__class__.__name__}(name={self.name!r}"
|
return f"{self.__class__.__name__}(name={self.name!r}, _conn={self._conn!r})"
|
||||||
if self._conn.read_consistency_interval is not None:
|
|
||||||
val += ", read_consistency_interval={!r}".format(
|
|
||||||
self._conn.read_consistency_interval
|
|
||||||
)
|
|
||||||
val += f", _conn={self._conn!r})"
|
|
||||||
return val
|
|
||||||
|
|
||||||
def __str__(self) -> str:
|
def __str__(self) -> str:
|
||||||
return self.__repr__()
|
return self.__repr__()
|
||||||
@@ -2549,6 +2580,9 @@ class LanceTable(Table):
|
|||||||
2. Currently we've disabled push-down of the filters from polars
|
2. Currently we've disabled push-down of the filters from polars
|
||||||
because polars pushdown into pyarrow uses pyarrow compute
|
because polars pushdown into pyarrow uses pyarrow compute
|
||||||
expressions rather than SQl strings (which LanceDB supports)
|
expressions rather than SQl strings (which LanceDB supports)
|
||||||
|
3. The Polars streaming engine is not supported because it does not
|
||||||
|
currently implement Python PyArrow dataset scans. Use the default
|
||||||
|
engine when collecting this LazyFrame.
|
||||||
|
|
||||||
Returns
|
Returns
|
||||||
-------
|
-------
|
||||||
@@ -2557,8 +2591,12 @@ class LanceTable(Table):
|
|||||||
from lancedb.integrations.pyarrow import PyarrowDatasetAdapter
|
from lancedb.integrations.pyarrow import PyarrowDatasetAdapter
|
||||||
|
|
||||||
dataset = PyarrowDatasetAdapter(self)
|
dataset = PyarrowDatasetAdapter(self)
|
||||||
return pl.scan_pyarrow_dataset(
|
# Polars 1.32's non-PyArrow callback path passes batch_size twice. Keep
|
||||||
dataset, allow_pyarrow_filter=False, batch_size=batch_size
|
# the compatible PyArrow path, but block predicates because this adapter
|
||||||
|
# cannot translate PyArrow expressions into LanceDB filters.
|
||||||
|
return pl.scan_pyarrow_dataset(dataset, batch_size=batch_size).map_batches(
|
||||||
|
_polars_predicate_pushdown_barrier,
|
||||||
|
predicate_pushdown=False,
|
||||||
)
|
)
|
||||||
|
|
||||||
# New unified API overload
|
# New unified API overload
|
||||||
@@ -2783,6 +2821,34 @@ class LanceTable(Table):
|
|||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def create_index_async(
|
||||||
|
self,
|
||||||
|
column: str,
|
||||||
|
*,
|
||||||
|
config: IndexConfigType,
|
||||||
|
replace: Optional[bool] = None,
|
||||||
|
wait_timeout: Optional[timedelta] = None,
|
||||||
|
name: Optional[str] = None,
|
||||||
|
train: bool = True,
|
||||||
|
) -> Job:
|
||||||
|
"""Create an index, returning a handle to the indexing job.
|
||||||
|
|
||||||
|
The job may already be complete when returned; callers must not assume
|
||||||
|
the index exists until :meth:`Job.wait` returns.
|
||||||
|
"""
|
||||||
|
return Job(
|
||||||
|
LOOP.run(
|
||||||
|
self._table.create_index_async(
|
||||||
|
column,
|
||||||
|
replace=replace,
|
||||||
|
config=config,
|
||||||
|
wait_timeout=wait_timeout,
|
||||||
|
name=name,
|
||||||
|
train=train,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
def _is_legacy_create_index_call(
|
def _is_legacy_create_index_call(
|
||||||
self,
|
self,
|
||||||
first_arg: str,
|
first_arg: str,
|
||||||
@@ -3387,8 +3453,8 @@ class LanceTable(Table):
|
|||||||
fts_columns: Optional[Union[str, List[str]]] = None,
|
fts_columns: Optional[Union[str, List[str]]] = None,
|
||||||
) -> LanceQueryBuilder:
|
) -> LanceQueryBuilder:
|
||||||
"""Create a search query to find the nearest neighbors
|
"""Create a search query to find the nearest neighbors
|
||||||
of the given query vector. We currently support [vector search][search]
|
of the given query vector. We currently support [vector search](https://lancedb.com/docs/search/vector-search/)
|
||||||
and [full-text search][search].
|
and [full-text search](https://lancedb.com/docs/search/full-text-search/).
|
||||||
|
|
||||||
Examples
|
Examples
|
||||||
--------
|
--------
|
||||||
@@ -3418,8 +3484,9 @@ class LanceTable(Table):
|
|||||||
- *default None*.
|
- *default None*.
|
||||||
Acceptable types are: list, np.ndarray, PIL.Image.Image
|
Acceptable types are: list, np.ndarray, PIL.Image.Image
|
||||||
|
|
||||||
- If None then the select/[where][sql]/limit clauses are applied
|
- If None then the
|
||||||
to filter the table
|
select/[where][lancedb.query.LanceQueryBuilder.where]/limit clauses
|
||||||
|
are applied to filter the table
|
||||||
vector_column_name: str, optional
|
vector_column_name: str, optional
|
||||||
The name of the vector column to search.
|
The name of the vector column to search.
|
||||||
|
|
||||||
@@ -3813,6 +3880,8 @@ class LanceTable(Table):
|
|||||||
retrain: bool, default False
|
retrain: bool, default False
|
||||||
This parameter is no longer used and is deprecated.
|
This parameter is no longer used and is deprecated.
|
||||||
|
|
||||||
|
Notes
|
||||||
|
-----
|
||||||
The frequency an application should call optimize is based on the frequency of
|
The frequency an application should call optimize is based on the frequency of
|
||||||
data modifications. If data is frequently added, deleted, or updated then
|
data modifications. If data is frequently added, deleted, or updated then
|
||||||
optimize should be run frequently. A good rule of thumb is to run optimize if
|
optimize should be run frequently. A good rule of thumb is to run optimize if
|
||||||
@@ -3907,6 +3976,28 @@ class LanceTable(Table):
|
|||||||
[`AsyncTable.get_lsm_write_spec`][lancedb.AsyncTable.get_lsm_write_spec]."""
|
[`AsyncTable.get_lsm_write_spec`][lancedb.AsyncTable.get_lsm_write_spec]."""
|
||||||
return LOOP.run(self._table.get_lsm_write_spec())
|
return LOOP.run(self._table.get_lsm_write_spec())
|
||||||
|
|
||||||
|
def checkpoint_lsm(self) -> None:
|
||||||
|
"""Synchronous version of
|
||||||
|
[`AsyncTable.checkpoint_lsm`][lancedb.AsyncTable.checkpoint_lsm]."""
|
||||||
|
return LOOP.run(self._table.checkpoint_lsm())
|
||||||
|
|
||||||
|
def flush_lsm(self) -> None:
|
||||||
|
"""Synchronous version of
|
||||||
|
[`AsyncTable.flush_lsm`][lancedb.AsyncTable.flush_lsm]."""
|
||||||
|
return LOOP.run(self._table.flush_lsm())
|
||||||
|
|
||||||
|
def compact_lsm(self) -> None:
|
||||||
|
"""Synchronous version of
|
||||||
|
[`AsyncTable.compact_lsm`][lancedb.AsyncTable.compact_lsm]."""
|
||||||
|
return LOOP.run(self._table.compact_lsm())
|
||||||
|
|
||||||
|
def get_lsm_stats(self, *, include_generation_rows: bool = False) -> Optional[dict]:
|
||||||
|
"""Synchronous version of
|
||||||
|
[`AsyncTable.get_lsm_stats`][lancedb.AsyncTable.get_lsm_stats]."""
|
||||||
|
return LOOP.run(
|
||||||
|
self._table.get_lsm_stats(include_generation_rows=include_generation_rows)
|
||||||
|
)
|
||||||
|
|
||||||
def close_lsm_writers(self) -> None:
|
def close_lsm_writers(self) -> None:
|
||||||
"""Close cached MemWAL shard writers. See
|
"""Close cached MemWAL shard writers. See
|
||||||
[`AsyncTable.close_lsm_writers`][lancedb.AsyncTable.close_lsm_writers]."""
|
[`AsyncTable.close_lsm_writers`][lancedb.AsyncTable.close_lsm_writers]."""
|
||||||
@@ -4585,6 +4676,13 @@ class AsyncTable:
|
|||||||
via [`set_unenforced_primary_key`]; bucket sharding additionally
|
via [`set_unenforced_primary_key`]; bucket sharding additionally
|
||||||
requires it to be the single column being bucketed.
|
requires it to be the single column being bucketed.
|
||||||
|
|
||||||
|
By default the MemWAL maintains every index on the table, resolved
|
||||||
|
here — a snapshot, so an index created afterwards needs the spec unset
|
||||||
|
and set again. This fails if one cannot be maintained; name the set
|
||||||
|
with ``with_maintained_indexes`` to install anyway. That pins an exact
|
||||||
|
set (a still-building index is rejected, not omitted); ``[]`` maintains
|
||||||
|
none.
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
spec : LsmWriteSpec
|
spec : LsmWriteSpec
|
||||||
@@ -4611,12 +4709,73 @@ class AsyncTable:
|
|||||||
|
|
||||||
Returns ``None`` when the MemWAL LSM write path is not enabled (no
|
Returns ``None`` when the MemWAL LSM write path is not enabled (no
|
||||||
spec has been set, or it was removed with `unset_lsm_write_spec`).
|
spec has been set, or it was removed with `unset_lsm_write_spec`).
|
||||||
The returned spec — including its ``maintained_indexes`` and
|
The returned spec mirrors what was passed to `set_lsm_write_spec`,
|
||||||
``writer_config_defaults`` — mirrors what was passed to
|
except that ``maintained_indexes`` always reports the concrete list
|
||||||
`set_lsm_write_spec`.
|
resolved when the spec was set — ``None`` never round-trips.
|
||||||
"""
|
"""
|
||||||
return await self._inner.get_lsm_write_spec()
|
return await self._inner.get_lsm_write_spec()
|
||||||
|
|
||||||
|
async def checkpoint_lsm(self) -> None:
|
||||||
|
"""Converge this table's LSM write path into its base table.
|
||||||
|
|
||||||
|
One flush, sealing every memtable into L0, then compaction triggers
|
||||||
|
until every generation that existed at that moment has reached base.
|
||||||
|
The loop runs client-side, reading progress from ``get_lsm_stats``.
|
||||||
|
|
||||||
|
Best-effort: generations created *while* it runs are deliberately not
|
||||||
|
waited on, which is what lets it terminate on a table taking writes.
|
||||||
|
Idempotent and safe on a cadence.
|
||||||
|
|
||||||
|
There is no deadline, and the caller owns that. It returns when the
|
||||||
|
target generations are gone, raises on a terminal server fault, and
|
||||||
|
otherwise waits however long the server takes. A slow table and a
|
||||||
|
stuck one are the same picture from the client: the compactor pool is
|
||||||
|
shared across every table on the node, so a checkpoint queued behind
|
||||||
|
unrelated work looks exactly like one that is merging. Wrap this in
|
||||||
|
``asyncio.wait_for`` for a wall-clock bound; abandoning it partway
|
||||||
|
costs nothing.
|
||||||
|
"""
|
||||||
|
return await self._inner.checkpoint_lsm()
|
||||||
|
|
||||||
|
async def flush_lsm(self) -> None:
|
||||||
|
"""Seal every bucket's active memtable into L0.
|
||||||
|
|
||||||
|
Does not touch the base table — moving L0 into base is
|
||||||
|
`compact_lsm`. On a node that has not claimed this table, this claims
|
||||||
|
it and replays its WAL log first.
|
||||||
|
"""
|
||||||
|
return await self._inner.flush_lsm()
|
||||||
|
|
||||||
|
async def compact_lsm(self) -> None:
|
||||||
|
"""Trigger a background L0 to base compaction pass per bucket.
|
||||||
|
|
||||||
|
Returns once the passes are dispatched, not once they finish: watch
|
||||||
|
``get_lsm_stats`` for progress, or use ``checkpoint_lsm`` to loop
|
||||||
|
until the current L0 has reached base.
|
||||||
|
"""
|
||||||
|
return await self._inner.compact_lsm()
|
||||||
|
|
||||||
|
async def get_lsm_stats(
|
||||||
|
self, *, include_generation_rows: bool = False
|
||||||
|
) -> Optional[dict]:
|
||||||
|
"""Read live per-bucket LSM state.
|
||||||
|
|
||||||
|
Answers "how far behind is my fresh tier", "which bucket is hot", and
|
||||||
|
"why is my fresh-tier vector search brute-force". Mutates no table
|
||||||
|
state, though on a node that has not claimed this table it claims it,
|
||||||
|
exactly as a read would.
|
||||||
|
|
||||||
|
Returns ``None`` only when the LSM write path is not enabled.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
include_generation_rows
|
||||||
|
Report a row count per L0 generation. Off by default: each count
|
||||||
|
opens an uncached Lance dataset, and ``checkpoint_lsm`` polls this
|
||||||
|
needing only generation numbers.
|
||||||
|
"""
|
||||||
|
return await self._inner.get_lsm_stats(include_generation_rows)
|
||||||
|
|
||||||
async def close_lsm_writers(self) -> None:
|
async def close_lsm_writers(self) -> None:
|
||||||
"""Drain and close any cached MemWAL shard writers for this table.
|
"""Drain and close any cached MemWAL shard writers for this table.
|
||||||
|
|
||||||
@@ -4691,7 +4850,7 @@ class AsyncTable:
|
|||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
**kwargs
|
**kwargs
|
||||||
Forwarded to [`lance.dataset`][lance.dataset].
|
Forwarded to `lance.dataset`.
|
||||||
|
|
||||||
Returns
|
Returns
|
||||||
-------
|
-------
|
||||||
@@ -4867,6 +5026,46 @@ class AsyncTable:
|
|||||||
)
|
)
|
||||||
raise e
|
raise e
|
||||||
|
|
||||||
|
async def create_index_async(
|
||||||
|
self,
|
||||||
|
column: str,
|
||||||
|
*,
|
||||||
|
replace: Optional[bool] = None,
|
||||||
|
config: Optional[
|
||||||
|
Union[
|
||||||
|
IvfFlat,
|
||||||
|
IvfPq,
|
||||||
|
IvfRq,
|
||||||
|
HnswPq,
|
||||||
|
HnswSq,
|
||||||
|
HnswFlat,
|
||||||
|
BTree,
|
||||||
|
Bitmap,
|
||||||
|
LabelList,
|
||||||
|
Fm,
|
||||||
|
FTS,
|
||||||
|
]
|
||||||
|
] = None,
|
||||||
|
wait_timeout: Optional[timedelta] = None,
|
||||||
|
name: Optional[str] = None,
|
||||||
|
train: bool = True,
|
||||||
|
) -> AsyncJob:
|
||||||
|
"""Create an index, returning a handle to the indexing job.
|
||||||
|
|
||||||
|
Takes the same arguments as :meth:`create_index`. The job may already
|
||||||
|
be complete when returned; callers must not assume the index exists
|
||||||
|
until :meth:`AsyncJob.wait` resolves.
|
||||||
|
"""
|
||||||
|
job = await self._inner.create_index_async(
|
||||||
|
column,
|
||||||
|
index=config,
|
||||||
|
replace=replace,
|
||||||
|
wait_timeout=wait_timeout,
|
||||||
|
name=name,
|
||||||
|
train=train,
|
||||||
|
)
|
||||||
|
return AsyncJob(job)
|
||||||
|
|
||||||
async def drop_index(self, name: str) -> None:
|
async def drop_index(self, name: str) -> None:
|
||||||
"""
|
"""
|
||||||
Drop an index from the table.
|
Drop an index from the table.
|
||||||
@@ -5010,7 +5209,7 @@ class AsyncTable:
|
|||||||
progress: Optional[Union[bool, Callable, Any]] = None,
|
progress: Optional[Union[bool, Callable, Any]] = None,
|
||||||
write_parallelism: Optional[int] = None,
|
write_parallelism: Optional[int] = None,
|
||||||
) -> AddResult:
|
) -> AddResult:
|
||||||
"""Add more data to the [Table](Table).
|
"""Add more data to the [AsyncTable][lancedb.table.AsyncTable].
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
@@ -5212,8 +5411,8 @@ class AsyncTable:
|
|||||||
fts_columns: Optional[Union[str, List[str]]] = None,
|
fts_columns: Optional[Union[str, List[str]]] = None,
|
||||||
) -> Union[AsyncHybridQuery, AsyncFTSQuery, AsyncVectorQuery]:
|
) -> Union[AsyncHybridQuery, AsyncFTSQuery, AsyncVectorQuery]:
|
||||||
"""Create a search query to find the nearest neighbors
|
"""Create a search query to find the nearest neighbors
|
||||||
of the given query vector. We currently support [vector search][search]
|
of the given query vector. We currently support [vector search](https://lancedb.com/docs/search/vector-search/)
|
||||||
and [full-text search][experimental-full-text-search].
|
and [full-text search](https://lancedb.com/docs/search/full-text-search/).
|
||||||
|
|
||||||
All query options are defined in [AsyncQuery][lancedb.query.AsyncQuery].
|
All query options are defined in [AsyncQuery][lancedb.query.AsyncQuery].
|
||||||
|
|
||||||
@@ -5774,15 +5973,14 @@ class AsyncTable:
|
|||||||
change permanent you can use the `[Self::restore]` method.
|
change permanent you can use the `[Self::restore]` method.
|
||||||
|
|
||||||
Any operation that modifies the table will fail while the table is in a checked
|
Any operation that modifies the table will fail while the table is in a checked
|
||||||
out state.
|
out state. To return the table to a normal state use
|
||||||
|
`[Self::checkout_latest]`.
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
version: int | str,
|
version: int | str,
|
||||||
The version to check out. A version number (`int`) or a tag
|
The version to check out. A version number (`int`) or a tag
|
||||||
(`str`) can be provided.
|
(`str`) can be provided.
|
||||||
|
|
||||||
To return the table to a normal state use `[Self::checkout_latest]`
|
|
||||||
"""
|
"""
|
||||||
try:
|
try:
|
||||||
await self._inner.checkout(version)
|
await self._inner.checkout(version)
|
||||||
@@ -5966,6 +6164,8 @@ class AsyncTable:
|
|||||||
retrain: bool, default False
|
retrain: bool, default False
|
||||||
This parameter is no longer used and is deprecated.
|
This parameter is no longer used and is deprecated.
|
||||||
|
|
||||||
|
Notes
|
||||||
|
-----
|
||||||
The frequency an application should call optimize is based on the frequency of
|
The frequency an application should call optimize is based on the frequency of
|
||||||
data modifications. If data is frequently added, deleted, or updated then
|
data modifications. If data is frequently added, deleted, or updated then
|
||||||
optimize should be run frequently. A good rule of thumb is to run optimize if
|
optimize should be run frequently. A good rule of thumb is to run optimize if
|
||||||
@@ -6141,7 +6341,9 @@ class TableStatistics:
|
|||||||
Attributes
|
Attributes
|
||||||
----------
|
----------
|
||||||
total_bytes: int
|
total_bytes: int
|
||||||
The total number of bytes in the table.
|
The total size, in bytes, of the table's data files, index files, and
|
||||||
|
overlay files. Read from the manifest, so this excludes deletion files
|
||||||
|
and manifests.
|
||||||
num_rows: int
|
num_rows: int
|
||||||
The total number of rows in the table.
|
The total number of rows in the table.
|
||||||
num_indices: int
|
num_indices: int
|
||||||
@@ -6346,6 +6548,8 @@ class Branches:
|
|||||||
dry_run: bool, default False
|
dry_run: bool, default False
|
||||||
When True, only preview. When False, attempt the merge.
|
When True, only preview. When False, attempt the merge.
|
||||||
|
|
||||||
|
Notes
|
||||||
|
-----
|
||||||
A rejected merge returns ``status="rejected"`` instead of raising.
|
A rejected merge returns ``status="rejected"`` instead of raising.
|
||||||
"""
|
"""
|
||||||
return LOOP.run(self._table.branches.merge(from_branch, dry_run))
|
return LOOP.run(self._table.branches.merge(from_branch, dry_run))
|
||||||
|
|||||||
@@ -395,6 +395,11 @@ def _(value: dict):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@value_to_sql.register(pa.Scalar)
|
||||||
|
def _(value: pa.Scalar):
|
||||||
|
return value_to_sql(value.as_py())
|
||||||
|
|
||||||
|
|
||||||
@value_to_sql.register(np.ndarray)
|
@value_to_sql.register(np.ndarray)
|
||||||
def _(value: np.ndarray):
|
def _(value: np.ndarray):
|
||||||
return value_to_sql(value.tolist())
|
return value_to_sql(value.tolist())
|
||||||
|
|||||||
@@ -226,13 +226,13 @@ def test_fetch_blob_ranges_validates_requests():
|
|||||||
table = _blob_table("range_validation", [{"id": 1, "image": b"abc"}])
|
table = _blob_table("range_validation", [{"id": 1, "image": b"abc"}])
|
||||||
row_id = _row_ids_by_id(table)[1]
|
row_id = _row_ids_by_id(table)[1]
|
||||||
|
|
||||||
with pytest.raises(RuntimeError, match="exceeds blob size"):
|
with pytest.raises(ValueError, match="exceeds blob size"):
|
||||||
table.fetch_blob_ranges("image", [(row_id, 2, 2)])
|
table.fetch_blob_ranges("image", [(row_id, 2, 2)])
|
||||||
|
|
||||||
with pytest.raises(RuntimeError, match="offset \\+ length overflowed"):
|
with pytest.raises(ValueError, match="offset \\+ length overflowed"):
|
||||||
table.fetch_blob_ranges("image", [(row_id, 2**64 - 1, 1)])
|
table.fetch_blob_ranges("image", [(row_id, 2**64 - 1, 1)])
|
||||||
|
|
||||||
with pytest.raises(ValueError, match="row ids"):
|
with pytest.raises(ValueError, match="row IDs"):
|
||||||
table.fetch_blob_ranges("image", [(2**64 - 1, 0, 1)])
|
table.fetch_blob_ranges("image", [(2**64 - 1, 0, 1)])
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -2,9 +2,11 @@
|
|||||||
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
|
||||||
|
import inspect
|
||||||
import re
|
import re
|
||||||
import sys
|
import sys
|
||||||
from datetime import timedelta
|
from datetime import timedelta
|
||||||
|
from importlib import resources
|
||||||
import os
|
import os
|
||||||
from types import SimpleNamespace
|
from types import SimpleNamespace
|
||||||
|
|
||||||
@@ -17,6 +19,10 @@ from lance_namespace.errors import NamespaceNotEmptyError, TableNotFoundError
|
|||||||
from lancedb.pydantic import LanceModel, Vector
|
from lancedb.pydantic import LanceModel, Vector
|
||||||
|
|
||||||
|
|
||||||
|
def test_package_includes_pep_561_marker():
|
||||||
|
assert resources.files(lancedb).joinpath("py.typed").is_file()
|
||||||
|
|
||||||
|
|
||||||
def test_basic(tmp_path):
|
def test_basic(tmp_path):
|
||||||
db = lancedb.connect(tmp_path)
|
db = lancedb.connect(tmp_path)
|
||||||
|
|
||||||
@@ -62,6 +68,44 @@ def test_basic(tmp_path):
|
|||||||
assert db.open_table("test").name == db["test"].name
|
assert db.open_table("test").name == db["test"].name
|
||||||
|
|
||||||
|
|
||||||
|
def test_sync_debugger_inspection_does_not_use_background_loop(tmp_path, monkeypatch):
|
||||||
|
from lancedb.background_loop import LOOP
|
||||||
|
|
||||||
|
db = lancedb.connect(tmp_path)
|
||||||
|
table = db.create_table("test", data=[{"id": 1}])
|
||||||
|
|
||||||
|
def fail_run(*args, **kwargs):
|
||||||
|
raise AssertionError("debugger inspection should not use the background loop")
|
||||||
|
|
||||||
|
monkeypatch.setattr(LOOP, "run", fail_run)
|
||||||
|
|
||||||
|
# Debuggers enumerate and evaluate every exposed attribute when expanding a
|
||||||
|
# variable. This must remain safe while their breakpoint suspends LOOP's thread.
|
||||||
|
members = dict(inspect.getmembers(db))
|
||||||
|
|
||||||
|
assert members["uri"] == str(tmp_path)
|
||||||
|
assert members["read_consistency_interval"] is None
|
||||||
|
assert repr(db) == f"LanceDBConnection(uri={str(tmp_path)!r})"
|
||||||
|
assert repr(table) == f"LanceTable(name='test', _conn={db!r})"
|
||||||
|
|
||||||
|
|
||||||
|
def test_read_consistency_interval_does_not_use_background_loop(tmp_path, monkeypatch):
|
||||||
|
from lancedb.background_loop import LOOP
|
||||||
|
from lancedb.db import LanceDBConnection
|
||||||
|
|
||||||
|
consistency_interval = timedelta(seconds=5)
|
||||||
|
db = lancedb.connect(tmp_path, read_consistency_interval=consistency_interval)
|
||||||
|
db_from_inner = LanceDBConnection.from_inner(db._inner, consistency_interval)
|
||||||
|
|
||||||
|
def fail_run(*args, **kwargs):
|
||||||
|
raise AssertionError("properties should not use the Python background loop")
|
||||||
|
|
||||||
|
monkeypatch.setattr(LOOP, "run", fail_run)
|
||||||
|
|
||||||
|
assert db.read_consistency_interval == consistency_interval
|
||||||
|
assert db_from_inner.read_consistency_interval == consistency_interval
|
||||||
|
|
||||||
|
|
||||||
def test_ingest_pd(tmp_path):
|
def test_ingest_pd(tmp_path):
|
||||||
db = lancedb.connect(tmp_path)
|
db = lancedb.connect(tmp_path)
|
||||||
|
|
||||||
|
|||||||
@@ -64,6 +64,23 @@ def test_embedding_function(tmp_path):
|
|||||||
assert np.allclose(actual, expected)
|
assert np.allclose(actual, expected)
|
||||||
|
|
||||||
|
|
||||||
|
def test_instructor_ndims_uses_instruction():
|
||||||
|
instructor = get_registry().get("instructor").create()
|
||||||
|
model = MagicMock()
|
||||||
|
model.encode.return_value = np.zeros((1, 384))
|
||||||
|
|
||||||
|
with patch.object(type(instructor), "get_model", return_value=model):
|
||||||
|
assert instructor.ndims() == 384
|
||||||
|
|
||||||
|
model.encode.assert_called_once_with(
|
||||||
|
[[instructor.source_instruction, "foo"]],
|
||||||
|
batch_size=instructor.batch_size,
|
||||||
|
show_progress_bar=instructor.show_progress_bar,
|
||||||
|
normalize_embeddings=instructor.normalize_embeddings,
|
||||||
|
device=instructor.device,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def test_embedding_function_variables():
|
def test_embedding_function_variables():
|
||||||
@register("variable-testing")
|
@register("variable-testing")
|
||||||
class VariableTestingFunction(TextEmbeddingFunction):
|
class VariableTestingFunction(TextEmbeddingFunction):
|
||||||
@@ -115,34 +132,16 @@ def test_embedding_function_variables():
|
|||||||
assert func.safe_model_dump()["secret_key"] == "$var:secret"
|
assert func.safe_model_dump()["secret_key"] == "$var:secret"
|
||||||
|
|
||||||
|
|
||||||
def test_parse_functions_with_variables():
|
def test_openai_variables_survive_metadata_round_trip():
|
||||||
@register("variable-parsing-test")
|
|
||||||
class VariableParsingFunction(TextEmbeddingFunction):
|
|
||||||
api_key: str
|
|
||||||
base_url: Optional[str] = None
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def sensitive_keys():
|
|
||||||
return ["api_key"]
|
|
||||||
|
|
||||||
def ndims(self):
|
|
||||||
return 10
|
|
||||||
|
|
||||||
def generate_embeddings(self, texts):
|
|
||||||
# Mock implementation that just returns random embeddings
|
|
||||||
# In real usage, this would use the api_key to call an API
|
|
||||||
return [np.random.rand(self.ndims()).tolist() for _ in texts]
|
|
||||||
|
|
||||||
registry = EmbeddingFunctionRegistry.get_instance()
|
registry = EmbeddingFunctionRegistry.get_instance()
|
||||||
|
|
||||||
registry.set_var("test_api_key", "sk-test-key-12345")
|
registry.set_var("test_api_key", "sk-test-key-12345")
|
||||||
registry.set_var("test_base_url", "https://api.example.com")
|
|
||||||
|
|
||||||
conf = EmbeddingFunctionConfig(
|
conf = EmbeddingFunctionConfig(
|
||||||
source_column="text",
|
source_column="text",
|
||||||
vector_column="vector",
|
vector_column="vector",
|
||||||
function=registry.get("variable-parsing-test").create(
|
function=registry.get("openai").create(
|
||||||
api_key="$var:test_api_key", base_url="$var:test_base_url"
|
api_key="$var:test_api_key", base_url="https://api.example.com"
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -150,7 +149,10 @@ def test_parse_functions_with_variables():
|
|||||||
|
|
||||||
# Create a mock arrow table with the metadata
|
# Create a mock arrow table with the metadata
|
||||||
schema = pa.schema(
|
schema = pa.schema(
|
||||||
[pa.field("text", pa.string()), pa.field("vector", pa.list_(pa.float32(), 10))]
|
[
|
||||||
|
pa.field("text", pa.string()),
|
||||||
|
pa.field("vector", pa.list_(pa.float32(), 1536)),
|
||||||
|
]
|
||||||
)
|
)
|
||||||
table = pa.table({"text": [], "vector": []}, schema=schema)
|
table = pa.table({"text": [], "vector": []}, schema=schema)
|
||||||
table = table.replace_schema_metadata(metadata)
|
table = table.replace_schema_metadata(metadata)
|
||||||
@@ -164,13 +166,15 @@ def test_parse_functions_with_variables():
|
|||||||
|
|
||||||
assert parsed_func.api_key == "sk-test-key-12345"
|
assert parsed_func.api_key == "sk-test-key-12345"
|
||||||
assert parsed_func.base_url == "https://api.example.com"
|
assert parsed_func.base_url == "https://api.example.com"
|
||||||
|
|
||||||
embeddings = parsed_func.generate_embeddings(["test text"])
|
|
||||||
assert len(embeddings) == 1
|
|
||||||
assert len(embeddings[0]) == 10
|
|
||||||
|
|
||||||
assert parsed_func.safe_model_dump()["api_key"] == "$var:test_api_key"
|
assert parsed_func.safe_model_dump()["api_key"] == "$var:test_api_key"
|
||||||
|
|
||||||
|
with patch("lancedb.embeddings.openai.attempt_import_or_raise") as import_openai:
|
||||||
|
parsed_func._openai_client
|
||||||
|
|
||||||
|
import_openai.return_value.OpenAI.assert_called_once_with(
|
||||||
|
api_key="sk-test-key-12345", base_url="https://api.example.com"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def test_embedding_with_bad_results(tmp_path):
|
def test_embedding_with_bad_results(tmp_path):
|
||||||
@register("null-embedding")
|
@register("null-embedding")
|
||||||
@@ -627,3 +631,23 @@ def test_url_retrieve_downloads_image():
|
|||||||
image_bytes = url_retrieve(image_url)
|
image_bytes = url_retrieve(image_url)
|
||||||
img = Image.open(io.BytesIO(image_bytes))
|
img = Image.open(io.BytesIO(image_bytes))
|
||||||
assert img.size[0] > 0 and img.size[1] > 0
|
assert img.size[0] > 0 and img.size[1] > 0
|
||||||
|
|
||||||
|
|
||||||
|
def test_jina_generate_image_input_dict_local_path(tmp_path):
|
||||||
|
"""
|
||||||
|
JinaEmbeddings._generate_image_input_dict must accept a local image path
|
||||||
|
(str or Path), not just bytes. Previously it crashed with
|
||||||
|
`AttributeError: 'function' object has no attribute 'urlparse'` on any
|
||||||
|
str/Path input because it called `urlparse.urlparse(image)` instead of
|
||||||
|
`urlparse(image)` (urlparse was imported as a function, not a module).
|
||||||
|
"""
|
||||||
|
Image = pytest.importorskip("PIL.Image")
|
||||||
|
from lancedb.embeddings.jinaai import JinaEmbeddings
|
||||||
|
|
||||||
|
image_path = tmp_path / "test.png"
|
||||||
|
Image.new("RGB", (4, 4), color="red").save(image_path, format="PNG")
|
||||||
|
|
||||||
|
for image in (str(image_path), image_path):
|
||||||
|
image_dict = JinaEmbeddings._generate_image_input_dict(image)
|
||||||
|
assert "image" in image_dict
|
||||||
|
assert isinstance(image_dict["image"], str) and len(image_dict["image"]) > 0
|
||||||
|
|||||||
@@ -12,7 +12,7 @@ import pyarrow.compute as pc
|
|||||||
import pytest
|
import pytest
|
||||||
import pytest_asyncio
|
import pytest_asyncio
|
||||||
|
|
||||||
from lancedb.index import FTS
|
from lancedb.index import BTree, FTS, IvfPq
|
||||||
from lancedb.table import AsyncTable, Table
|
from lancedb.table import AsyncTable, Table
|
||||||
|
|
||||||
|
|
||||||
@@ -99,6 +99,86 @@ async def test_async_hybrid_query_filters(table: AsyncTable):
|
|||||||
assert result["text"].to_pylist() == ["cat", "b"]
|
assert result["text"].to_pylist() == ["cat", "b"]
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_hybrid_query_with_stale_fixed_size_binary_prefilter(
|
||||||
|
tmpdir_factory,
|
||||||
|
):
|
||||||
|
tmp_path = str(tmpdir_factory.mktemp("stale_scalar_prefilter"))
|
||||||
|
db = await lancedb.connect_async(tmp_path)
|
||||||
|
|
||||||
|
def fixed_size_binary(value: int) -> bytes:
|
||||||
|
return value.to_bytes(16, byteorder="big")
|
||||||
|
|
||||||
|
num_rows = 1000
|
||||||
|
data = pa.table(
|
||||||
|
{
|
||||||
|
"space_id": pa.array(
|
||||||
|
[fixed_size_binary(i) for i in range(num_rows)],
|
||||||
|
type=pa.binary(16),
|
||||||
|
),
|
||||||
|
"text": ["book"] * num_rows,
|
||||||
|
"vector": pa.array(
|
||||||
|
[[float(i), float(i)] for i in range(num_rows)],
|
||||||
|
type=pa.list_(pa.float32(), 2),
|
||||||
|
),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
table = await db.create_table("test", data)
|
||||||
|
await table.create_index(
|
||||||
|
"vector", config=IvfPq(num_partitions=4, num_sub_vectors=2)
|
||||||
|
)
|
||||||
|
await table.create_index("space_id", config=BTree())
|
||||||
|
await table.create_index("text", config=FTS(with_position=False))
|
||||||
|
|
||||||
|
# Advance the search indices without advancing the scalar index. This is the
|
||||||
|
# state that previously let hybrid search use an incomplete scalar prefilter.
|
||||||
|
await table.add(data)
|
||||||
|
lance_dataset = await table.to_lance()
|
||||||
|
lance_dataset.optimize.optimize_indices(index_names=["vector_idx", "text_idx"])
|
||||||
|
await table.checkout_latest()
|
||||||
|
|
||||||
|
scalar_stats = await table.index_stats("space_id_idx")
|
||||||
|
assert scalar_stats is not None
|
||||||
|
assert scalar_stats.num_indexed_rows == num_rows
|
||||||
|
assert scalar_stats.num_unindexed_rows == num_rows
|
||||||
|
|
||||||
|
for index_name in ["vector_idx", "text_idx"]:
|
||||||
|
search_stats = await table.index_stats(index_name)
|
||||||
|
assert search_stats is not None
|
||||||
|
assert search_stats.num_indexed_rows == num_rows * 2
|
||||||
|
assert search_stats.num_unindexed_rows == 0
|
||||||
|
|
||||||
|
matching_ids = [5, 10, 15, 20, 25, 30]
|
||||||
|
literals = [
|
||||||
|
f"arrow_cast(0x{fixed_size_binary(i).hex()}, 'FixedSizeBinary(16)')"
|
||||||
|
for i in matching_ids
|
||||||
|
]
|
||||||
|
predicate = f"space_id IN ({', '.join(literals)})"
|
||||||
|
expected_ids = sorted(fixed_size_binary(i) for i in matching_ids for _ in range(2))
|
||||||
|
|
||||||
|
vector_query = (
|
||||||
|
table.query().where(predicate).nearest_to([5.0, 5.0]).limit(num_rows * 2)
|
||||||
|
)
|
||||||
|
vector_results = await vector_query.to_arrow()
|
||||||
|
assert sorted(vector_results["space_id"].to_pylist()) == expected_ids
|
||||||
|
|
||||||
|
fts_query = (
|
||||||
|
table.query().where(predicate).nearest_to_text("book").limit(num_rows * 2)
|
||||||
|
)
|
||||||
|
fts_results = await fts_query.to_arrow()
|
||||||
|
assert sorted(fts_results["space_id"].to_pylist()) == expected_ids
|
||||||
|
|
||||||
|
hybrid_results = await (
|
||||||
|
table.query()
|
||||||
|
.where(predicate)
|
||||||
|
.nearest_to([5.0, 5.0])
|
||||||
|
.nearest_to_text("book")
|
||||||
|
.limit(num_rows * 2)
|
||||||
|
.to_arrow()
|
||||||
|
)
|
||||||
|
assert sorted(hybrid_results["space_id"].to_pylist()) == expected_ids
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_async_hybrid_query_default_limit(table: AsyncTable):
|
async def test_async_hybrid_query_default_limit(table: AsyncTable):
|
||||||
# add 10 new rows
|
# add 10 new rows
|
||||||
@@ -123,6 +203,19 @@ async def test_async_hybrid_query_default_limit(table: AsyncTable):
|
|||||||
assert texts.count("a") == 1
|
assert texts.count("a") == 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_hybrid_query_minimum_nprobes_zero_raises(sync_table: Table):
|
||||||
|
# minimum_nprobes(0) must raise the same validation error a plain vector
|
||||||
|
# query raises, not silently no-op because 0 is falsy.
|
||||||
|
with pytest.raises(ValueError, match="minimum_nprobes must be greater than 0"):
|
||||||
|
(
|
||||||
|
sync_table.search(query_type="hybrid")
|
||||||
|
.vector([0.0, 0.4])
|
||||||
|
.text("dog")
|
||||||
|
.minimum_nprobes(0)
|
||||||
|
.to_arrow()
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def test_hybrid_query_distance_range(sync_table: Table):
|
def test_hybrid_query_distance_range(sync_table: Table):
|
||||||
reranker = RRFReranker(return_score="all")
|
reranker = RRFReranker(return_score="all")
|
||||||
result = (
|
result = (
|
||||||
|
|||||||
@@ -0,0 +1,33 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
import re
|
||||||
|
import shutil
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import lancedb._lancedb as _lancedb
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.skipif(sys.platform != "linux", reason="ldd is Linux-specific")
|
||||||
|
def test_native_extension_does_not_link_openssl():
|
||||||
|
"""OpenSSL-linked wheels abort when imported on RHEL hosts in FIPS mode."""
|
||||||
|
ldd = shutil.which("ldd")
|
||||||
|
if ldd is None:
|
||||||
|
pytest.skip("ldd is not installed")
|
||||||
|
|
||||||
|
result = subprocess.run(
|
||||||
|
[ldd, _lancedb.__file__],
|
||||||
|
check=True,
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
)
|
||||||
|
openssl_libraries = re.findall(
|
||||||
|
r"^\s*(lib(?:crypto|ssl)\S*)\s+=>", result.stdout, flags=re.MULTILINE
|
||||||
|
)
|
||||||
|
|
||||||
|
assert not openssl_libraries, (
|
||||||
|
"the LanceDB native extension must use rustls instead of linking OpenSSL: "
|
||||||
|
f"{openssl_libraries}"
|
||||||
|
)
|
||||||
@@ -84,6 +84,15 @@ async def binary_table(db_async):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_create_index_async_returns_done_job(some_table: AsyncTable):
|
||||||
|
job = await some_table.create_index_async("id", config=BTree())
|
||||||
|
assert job.id is None
|
||||||
|
await job.wait()
|
||||||
|
assert len(await some_table.list_indices()) == 1
|
||||||
|
await job.cancel()
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_create_scalar_index(some_table: AsyncTable):
|
async def test_create_scalar_index(some_table: AsyncTable):
|
||||||
# Can create
|
# Can create
|
||||||
@@ -363,6 +372,31 @@ async def test_create_vector_index(some_table: AsyncTable):
|
|||||||
assert stats.num_indices == 1
|
assert stats.num_indices == 1
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_create_ivf_index_reports_unsplittable_partitions(db_async):
|
||||||
|
dim = 8
|
||||||
|
num_partitions = 300 # More than 256 selects hierarchical k-means.
|
||||||
|
base_vectors = [[float(row == column) for column in range(dim)] for row in range(5)]
|
||||||
|
vectors = pa.array(base_vectors * 200, pa.list_(pa.float32(), dim))
|
||||||
|
table = await db_async.create_table(
|
||||||
|
"unsplittable_partitions",
|
||||||
|
pa.table({"vector": vectors}),
|
||||||
|
)
|
||||||
|
|
||||||
|
error_pattern = (
|
||||||
|
rf"Cannot create {num_partitions} IVF partitions: k-means could only form"
|
||||||
|
)
|
||||||
|
with pytest.raises(RuntimeError, match=error_pattern):
|
||||||
|
await table.create_index(
|
||||||
|
"vector",
|
||||||
|
config=IvfFlat(
|
||||||
|
distance_type="dot",
|
||||||
|
num_partitions=num_partitions,
|
||||||
|
max_iterations=10,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_create_4bit_ivfpq_index(some_table: AsyncTable):
|
async def test_create_4bit_ivfpq_index(some_table: AsyncTable):
|
||||||
# Can create
|
# Can create
|
||||||
|
|||||||
@@ -83,7 +83,9 @@ def test_lsm_write_spec_repr():
|
|||||||
assert s.spec_type == "bucket"
|
assert s.spec_type == "bucket"
|
||||||
assert s.column == "id"
|
assert s.column == "id"
|
||||||
assert s.num_buckets == 4
|
assert s.num_buckets == 4
|
||||||
assert s.maintained_indexes == []
|
# A fresh spec defers its maintained set to install time.
|
||||||
|
assert s.maintained_indexes is None
|
||||||
|
assert s.with_maintained_indexes([]).maintained_indexes == []
|
||||||
assert "bucket" in repr(s)
|
assert "bucket" in repr(s)
|
||||||
assert "id" in repr(s)
|
assert "id" in repr(s)
|
||||||
assert "4" in repr(s)
|
assert "4" in repr(s)
|
||||||
@@ -169,18 +171,23 @@ def test_get_lsm_write_spec(tmp_path):
|
|||||||
table.unset_lsm_write_spec()
|
table.unset_lsm_write_spec()
|
||||||
assert table.get_lsm_write_spec() is None
|
assert table.get_lsm_write_spec() is None
|
||||||
|
|
||||||
# Identity round-trips (column recovered from the schema).
|
# Identity round-trips (column recovered from the schema). Leaving the
|
||||||
|
# maintained set to be inferred picks up the index on the table, so the
|
||||||
|
# spec reads back naming it rather than as "infer".
|
||||||
table.set_lsm_write_spec(LsmWriteSpec.identity("id"))
|
table.set_lsm_write_spec(LsmWriteSpec.identity("id"))
|
||||||
spec = table.get_lsm_write_spec()
|
spec = table.get_lsm_write_spec()
|
||||||
assert spec.spec_type == "identity"
|
assert spec.spec_type == "identity"
|
||||||
assert spec.column == "id"
|
assert spec.column == "id"
|
||||||
|
assert spec.maintained_indexes == [idx_name]
|
||||||
table.unset_lsm_write_spec()
|
table.unset_lsm_write_spec()
|
||||||
|
|
||||||
# Unsharded round-trips (no routing column).
|
# Unsharded round-trips (no routing column). Opting out is distinct from
|
||||||
table.set_lsm_write_spec(LsmWriteSpec.unsharded())
|
# the inferred default.
|
||||||
|
table.set_lsm_write_spec(LsmWriteSpec.unsharded().with_maintained_indexes([]))
|
||||||
spec = table.get_lsm_write_spec()
|
spec = table.get_lsm_write_spec()
|
||||||
assert spec.spec_type == "unsharded"
|
assert spec.spec_type == "unsharded"
|
||||||
assert spec.column is None
|
assert spec.column is None
|
||||||
|
assert spec.maintained_indexes == []
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
|
|||||||
@@ -544,7 +544,7 @@ def test_lsm_read_fts_unmaintained_index_errors(tmp_path):
|
|||||||
table.create_index("text", config=FTS())
|
table.create_index("text", config=FTS())
|
||||||
# No maintained indexes: the active memtable FTS arm cannot serve un-compacted
|
# No maintained indexes: the active memtable FTS arm cannot serve un-compacted
|
||||||
# docs, so the search would silently omit them — reject instead.
|
# docs, so the search would silently omit them — reject instead.
|
||||||
table.set_lsm_write_spec(LsmWriteSpec.unsharded())
|
table.set_lsm_write_spec(LsmWriteSpec.unsharded().with_maintained_indexes([]))
|
||||||
with pytest.raises(Exception, match="maintained"):
|
with pytest.raises(Exception, match="maintained"):
|
||||||
table.search("fox", query_type="fts", fts_columns="text").to_arrow()
|
table.search("fox", query_type="fts", fts_columns="text").to_arrow()
|
||||||
|
|
||||||
@@ -631,7 +631,7 @@ def test_lsm_read_vector_unmaintained_index_errors(tmp_path):
|
|||||||
)
|
)
|
||||||
# Spec with NO maintained indexes: the base vector index's catch-up is untracked,
|
# Spec with NO maintained indexes: the base vector index's catch-up is untracked,
|
||||||
# so the scanner rejects rather than risk dropping compacted-but-unindexed rows.
|
# so the scanner rejects rather than risk dropping compacted-but-unindexed rows.
|
||||||
table.set_lsm_write_spec(LsmWriteSpec.unsharded())
|
table.set_lsm_write_spec(LsmWriteSpec.unsharded().with_maintained_indexes([]))
|
||||||
with pytest.raises(Exception, match="maintained"):
|
with pytest.raises(Exception, match="maintained"):
|
||||||
table.search([1.0] * VECTOR_DIM).to_arrow()
|
table.search([1.0] * VECTOR_DIM).to_arrow()
|
||||||
|
|
||||||
|
|||||||
@@ -18,6 +18,7 @@ Tests verify:
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
import copy
|
import copy
|
||||||
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
import sys
|
import sys
|
||||||
import tempfile
|
import tempfile
|
||||||
@@ -239,7 +240,7 @@ def create_tracking_namespace(
|
|||||||
|
|
||||||
dir_props = {f"storage.{k}": v for k, v in storage_options_with_refresh.items()}
|
dir_props = {f"storage.{k}": v for k, v in storage_options_with_refresh.items()}
|
||||||
|
|
||||||
if bucket_name.startswith("/") or bucket_name.startswith("file://"):
|
if os.path.isabs(bucket_name) or bucket_name.startswith("file://"):
|
||||||
dir_props["root"] = f"{bucket_name}/namespace_root"
|
dir_props["root"] = f"{bucket_name}/namespace_root"
|
||||||
else:
|
else:
|
||||||
dir_props["root"] = f"s3://{bucket_name}/namespace_root"
|
dir_props["root"] = f"s3://{bucket_name}/namespace_root"
|
||||||
@@ -767,3 +768,70 @@ def test_namespace_with_schema_only(s3_bucket: str, use_custom: bool):
|
|||||||
|
|
||||||
# Verify data was added
|
# Verify data was added
|
||||||
assert table.count_rows() == 2
|
assert table.count_rows() == 2
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("use_custom", [False, True], ids=["DirectoryNS", "CustomNS"])
|
||||||
|
def test_namespace_exists(use_custom: bool):
|
||||||
|
"""
|
||||||
|
Test namespace_exists returns True for existing and False for non-existent.
|
||||||
|
"""
|
||||||
|
temp_dir = tempfile.mkdtemp()
|
||||||
|
try:
|
||||||
|
ns_client, _ = create_tracking_namespace(
|
||||||
|
bucket_name=temp_dir,
|
||||||
|
storage_options={},
|
||||||
|
credential_expires_in_seconds=3600,
|
||||||
|
use_custom=use_custom,
|
||||||
|
)
|
||||||
|
db = LanceNamespaceDBConnection(ns_client)
|
||||||
|
|
||||||
|
namespace_name = f"test_ns_{uuid.uuid4().hex[:8]}"
|
||||||
|
db.create_namespace([namespace_name])
|
||||||
|
|
||||||
|
# Existing namespace should return True
|
||||||
|
assert db.namespace_exists(namespace_id=[namespace_name]) is True
|
||||||
|
|
||||||
|
# Non-existent namespace should return False
|
||||||
|
assert db.namespace_exists(namespace_id=["nonexistent_ns"]) is False
|
||||||
|
finally:
|
||||||
|
shutil.rmtree(temp_dir, ignore_errors=True)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("use_custom", [False, True], ids=["DirectoryNS", "CustomNS"])
|
||||||
|
def test_table_exists(use_custom: bool):
|
||||||
|
"""
|
||||||
|
Test table_exists returns True for existing table and False for non-existent.
|
||||||
|
"""
|
||||||
|
temp_dir = tempfile.mkdtemp()
|
||||||
|
try:
|
||||||
|
ns_client, _ = create_tracking_namespace(
|
||||||
|
bucket_name=temp_dir,
|
||||||
|
storage_options={},
|
||||||
|
credential_expires_in_seconds=3600,
|
||||||
|
use_custom=use_custom,
|
||||||
|
)
|
||||||
|
db = LanceNamespaceDBConnection(ns_client)
|
||||||
|
|
||||||
|
namespace_name = f"test_ns_{uuid.uuid4().hex[:8]}"
|
||||||
|
db.create_namespace([namespace_name])
|
||||||
|
|
||||||
|
table_name = f"test_table_{uuid.uuid4().hex}"
|
||||||
|
namespace_path = [namespace_name]
|
||||||
|
schema = pa.schema(
|
||||||
|
[
|
||||||
|
pa.field("id", pa.int64()),
|
||||||
|
pa.field("vector", pa.list_(pa.float32(), 2)),
|
||||||
|
pa.field("text", pa.string()),
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
db.create_table(table_name, schema=schema, namespace_path=namespace_path)
|
||||||
|
|
||||||
|
# Existing table should return True
|
||||||
|
table_id = namespace_path + [table_name]
|
||||||
|
assert db.table_exists(table_id=table_id) is True
|
||||||
|
|
||||||
|
# Non-existent table should return False
|
||||||
|
assert db.table_exists(table_id=namespace_path + ["nonexistent_table"]) is False
|
||||||
|
finally:
|
||||||
|
shutil.rmtree(temp_dir, ignore_errors=True)
|
||||||
|
|||||||
@@ -0,0 +1,42 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
import importlib
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
|
||||||
|
def test_pyo3_abi_matches_minimum_supported_python():
|
||||||
|
project_dir = Path(__file__).parents[2]
|
||||||
|
pyproject = (project_dir / "pyproject.toml").read_text()
|
||||||
|
cargo_manifest = (project_dir / "Cargo.toml").read_text()
|
||||||
|
|
||||||
|
minimum_python = re.search(
|
||||||
|
r'^requires-python\s*=\s*">=(\d+)\.(\d+)"$', pyproject, re.MULTILINE
|
||||||
|
)
|
||||||
|
assert minimum_python is not None
|
||||||
|
|
||||||
|
major, minor = minimum_python.groups()
|
||||||
|
expected_abi = f"abi3-py{major}{minor}"
|
||||||
|
configured_abis = re.findall(r'"(abi3-py\d+)"', cargo_manifest)
|
||||||
|
|
||||||
|
assert configured_abis == [expected_abi, expected_abi], (
|
||||||
|
"the pyo3 runtime and build ABI features must both match requires-python"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.skipif(sys.platform != "win32", reason="Windows wheel regression test")
|
||||||
|
def test_windows_wheel_tag_and_native_import():
|
||||||
|
project_dir = Path(__file__).parents[2]
|
||||||
|
wheels = list((project_dir.parent / "target" / "wheels").glob("lancedb-*.whl"))
|
||||||
|
if not wheels:
|
||||||
|
pytest.skip("no wheel artifact is available in this development environment")
|
||||||
|
|
||||||
|
assert len(wheels) == 1
|
||||||
|
assert wheels[0].name.endswith("-cp310-abi3-win_amd64.whl")
|
||||||
|
|
||||||
|
native_module = importlib.import_module("lancedb._lancedb")
|
||||||
|
assert Path(native_module.__file__).suffix == ".pyd"
|
||||||
@@ -6,6 +6,7 @@ import math
|
|||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from lancedb import DBConnection, Table, connect
|
from lancedb import DBConnection, Table, connect
|
||||||
|
from lancedb.background_loop import LOOP
|
||||||
from lancedb.permutation import Permutation, Permutations, permutation_builder
|
from lancedb.permutation import Permutation, Permutations, permutation_builder
|
||||||
|
|
||||||
|
|
||||||
@@ -31,6 +32,25 @@ def test_split_random_ratios(mem_db):
|
|||||||
assert 65 <= split_1_count <= 75 # ~70% ± tolerance
|
assert 65 <= split_1_count <= 75 # ~70% ± tolerance
|
||||||
|
|
||||||
|
|
||||||
|
def test_execute_does_not_reenter_background_loop(tmp_path, monkeypatch):
|
||||||
|
import threading
|
||||||
|
|
||||||
|
db = connect(tmp_path)
|
||||||
|
tbl = db.create_table("test_table", pa.table({"x": range(10)}))
|
||||||
|
original_run = LOOP.run
|
||||||
|
|
||||||
|
def fail_on_reentry(future):
|
||||||
|
assert threading.current_thread() is not LOOP.thread
|
||||||
|
return original_run(future)
|
||||||
|
|
||||||
|
monkeypatch.setattr(LOOP, "run", fail_on_reentry)
|
||||||
|
|
||||||
|
permutation_tbl = permutation_builder(tbl).execute()
|
||||||
|
|
||||||
|
assert permutation_tbl.count_rows() == 10
|
||||||
|
assert permutation_tbl._conn.read_consistency_interval is None
|
||||||
|
|
||||||
|
|
||||||
def test_split_random_counts(mem_db):
|
def test_split_random_counts(mem_db):
|
||||||
"""Test random splitting with absolute counts."""
|
"""Test random splitting with absolute counts."""
|
||||||
tbl = mem_db.create_table(
|
tbl = mem_db.create_table(
|
||||||
|
|||||||
@@ -415,6 +415,17 @@ def test_nullable_vector():
|
|||||||
assert schema == pa.schema([pa.field("vec", pa.list_(pa.float32(), 16), True)])
|
assert schema == pa.schema([pa.field("vec", pa.list_(pa.float32(), 16), True)])
|
||||||
|
|
||||||
|
|
||||||
|
def test_bare_vector_raises_clear_error():
|
||||||
|
namespace = {
|
||||||
|
"__name__": "test_model_without_pyarrow",
|
||||||
|
"LanceModel": LanceModel,
|
||||||
|
"Vector": Vector,
|
||||||
|
}
|
||||||
|
|
||||||
|
with pytest.raises(TypeError, match=r"Vector must be parameterized.*Vector\(128\)"):
|
||||||
|
exec("class TestModel(LanceModel):\n vector: Vector", namespace)
|
||||||
|
|
||||||
|
|
||||||
def test_fixed_size_list_field():
|
def test_fixed_size_list_field():
|
||||||
class TestModel(pydantic.BaseModel):
|
class TestModel(pydantic.BaseModel):
|
||||||
vec: Vector(16)
|
vec: Vector(16)
|
||||||
|
|||||||
@@ -570,6 +570,15 @@ def test_query_builder(table):
|
|||||||
assert all(np.array(rs[0]["vector"]) == [1, 2])
|
assert all(np.array(rs[0]["vector"]) == [1, 2])
|
||||||
|
|
||||||
|
|
||||||
|
def test_query_multiple_vectors(table):
|
||||||
|
results = table.search([np.array([1, 2]), np.array([4, 5])]).limit(1).to_list()
|
||||||
|
|
||||||
|
assert len(results) == 2
|
||||||
|
results_by_query = {result["query_index"]: result for result in results}
|
||||||
|
assert results_by_query[0]["id"] == 1
|
||||||
|
assert results_by_query[1]["id"] == 2
|
||||||
|
|
||||||
|
|
||||||
def test_with_row_id(table: lancedb.table.Table):
|
def test_with_row_id(table: lancedb.table.Table):
|
||||||
rs = table.search().with_row_id(True).to_arrow()
|
rs = table.search().with_row_id(True).to_arrow()
|
||||||
assert "_rowid" in rs.column_names
|
assert "_rowid" in rs.column_names
|
||||||
|
|||||||
@@ -35,6 +35,12 @@ def make_mock_http_handler(handler):
|
|||||||
return MockLanceDBHandler
|
return MockLanceDBHandler
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("db_name", ["a" * 64, "invalid..database"])
|
||||||
|
def test_connect_rejects_invalid_cloud_dns_hostname(db_name):
|
||||||
|
with pytest.raises(ValueError, match="DNS labels must contain 1 to 63 bytes"):
|
||||||
|
lancedb.connect(f"db://{db_name}", api_key="fake")
|
||||||
|
|
||||||
|
|
||||||
@contextlib.contextmanager
|
@contextlib.contextmanager
|
||||||
def mock_lancedb_connection(handler):
|
def mock_lancedb_connection(handler):
|
||||||
with http.server.HTTPServer(
|
with http.server.HTTPServer(
|
||||||
@@ -812,6 +818,121 @@ def test_table_create_indices():
|
|||||||
table.drop_index("custom_fts_idx")
|
table.drop_index("custom_fts_idx")
|
||||||
|
|
||||||
|
|
||||||
|
def test_remote_create_index_async_returns_job():
|
||||||
|
from lancedb.index import BTree
|
||||||
|
|
||||||
|
describe_calls = []
|
||||||
|
|
||||||
|
def handler(request):
|
||||||
|
content_len = int(request.headers.get("Content-Length", 0))
|
||||||
|
body = request.rfile.read(content_len) if content_len > 0 else b""
|
||||||
|
if request.path == "/v1/table/test/create_index/":
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/json")
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(b'{"job_id": "job-1"}')
|
||||||
|
elif request.path == "/v1/jobs/describe":
|
||||||
|
assert json.loads(body)["job_id"] == "job-1"
|
||||||
|
describe_calls.append(1)
|
||||||
|
state = "IN_PROGRESS" if len(describe_calls) == 1 else "DONE"
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/json")
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(
|
||||||
|
json.dumps(dict(job_id="job-1", job_state=state)).encode()
|
||||||
|
)
|
||||||
|
elif request.path == "/v1/jobs/cancel":
|
||||||
|
assert json.loads(body)["job_id"] == "job-1"
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/json")
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(b"{}")
|
||||||
|
elif request.path == "/v1/table/test/create/?mode=create":
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/json")
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(b"{}")
|
||||||
|
elif request.path == "/v1/table/test/describe/":
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/json")
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(
|
||||||
|
json.dumps(
|
||||||
|
dict(
|
||||||
|
version=1,
|
||||||
|
schema=dict(
|
||||||
|
fields=[
|
||||||
|
dict(name="id", type={"type": "int64"}, nullable=False),
|
||||||
|
]
|
||||||
|
),
|
||||||
|
)
|
||||||
|
).encode()
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
request.send_response(404)
|
||||||
|
request.end_headers()
|
||||||
|
|
||||||
|
with mock_lancedb_connection(handler) as db:
|
||||||
|
table = db.create_table("test", [{"id": 1}])
|
||||||
|
job = table.create_index_async("id", config=BTree())
|
||||||
|
assert job.id == "job-1"
|
||||||
|
job.wait(timeout=timedelta(seconds=30))
|
||||||
|
assert len(describe_calls) == 2
|
||||||
|
job.cancel()
|
||||||
|
|
||||||
|
|
||||||
|
def test_remote_job_wait_raises_on_failure():
|
||||||
|
from lancedb.exceptions import JobFailedError
|
||||||
|
from lancedb.index import BTree
|
||||||
|
|
||||||
|
def handler(request):
|
||||||
|
content_len = int(request.headers.get("Content-Length", 0))
|
||||||
|
body = request.rfile.read(content_len) if content_len > 0 else b""
|
||||||
|
if request.path == "/v1/table/test/create_index/":
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/json")
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(b'{"job_id": "job-2"}')
|
||||||
|
elif request.path == "/v1/jobs/describe":
|
||||||
|
assert json.loads(body)["job_id"] == "job-2"
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/json")
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(
|
||||||
|
json.dumps(dict(job_id="job-2", job_state="FAILED")).encode()
|
||||||
|
)
|
||||||
|
elif request.path == "/v1/table/test/create/?mode=create":
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/json")
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(b"{}")
|
||||||
|
elif request.path == "/v1/table/test/describe/":
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/json")
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(
|
||||||
|
json.dumps(
|
||||||
|
dict(
|
||||||
|
version=1,
|
||||||
|
schema=dict(
|
||||||
|
fields=[
|
||||||
|
dict(name="id", type={"type": "int64"}, nullable=False),
|
||||||
|
]
|
||||||
|
),
|
||||||
|
)
|
||||||
|
).encode()
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
request.send_response(404)
|
||||||
|
request.end_headers()
|
||||||
|
|
||||||
|
with mock_lancedb_connection(handler) as db:
|
||||||
|
table = db.create_table("test", [{"id": 1}])
|
||||||
|
job = table.create_index_async("id", config=BTree())
|
||||||
|
with pytest.raises(JobFailedError, match="job-2"):
|
||||||
|
job.wait()
|
||||||
|
|
||||||
|
|
||||||
def test_remote_create_index_new_api():
|
def test_remote_create_index_new_api():
|
||||||
received_requests = []
|
received_requests = []
|
||||||
|
|
||||||
@@ -1020,7 +1141,7 @@ def query_test_table(query_handler, *, server_version=Version("0.1.0")):
|
|||||||
request.send_header("Content-Type", "application/json")
|
request.send_header("Content-Type", "application/json")
|
||||||
request.send_header("phalanx-version", str(server_version))
|
request.send_header("phalanx-version", str(server_version))
|
||||||
request.end_headers()
|
request.end_headers()
|
||||||
request.wfile.write(b"{}")
|
request.wfile.write(b'{"version": 1, "schema": {"fields": []}}')
|
||||||
elif request.path == "/v1/table/test/query/":
|
elif request.path == "/v1/table/test/query/":
|
||||||
content_len = int(request.headers.get("Content-Length"))
|
content_len = int(request.headers.get("Content-Length"))
|
||||||
body = request.rfile.read(content_len)
|
body = request.rfile.read(content_len)
|
||||||
@@ -1858,3 +1979,330 @@ def test_inherited_remote_table_reopens_after_fork():
|
|||||||
finally:
|
finally:
|
||||||
server.shutdown()
|
server.shutdown()
|
||||||
server_thread.join()
|
server_thread.join()
|
||||||
|
|
||||||
|
|
||||||
|
BLOB_DESCRIBE_RESPONSE = {
|
||||||
|
"table": "test",
|
||||||
|
"version": 1,
|
||||||
|
"schema": {
|
||||||
|
"fields": [
|
||||||
|
{"name": "id", "type": {"type": "int64"}, "nullable": False},
|
||||||
|
{
|
||||||
|
"name": "image",
|
||||||
|
"type": {
|
||||||
|
"type": "struct",
|
||||||
|
"fields": [
|
||||||
|
{
|
||||||
|
"name": "data",
|
||||||
|
"type": {"type": "large_binary"},
|
||||||
|
"nullable": True,
|
||||||
|
},
|
||||||
|
{"name": "uri", "type": {"type": "string"}, "nullable": True},
|
||||||
|
],
|
||||||
|
},
|
||||||
|
"nullable": True,
|
||||||
|
"metadata": {
|
||||||
|
"ARROW:extension:name": "lance.blob.v2",
|
||||||
|
"ARROW:extension:metadata": "",
|
||||||
|
},
|
||||||
|
},
|
||||||
|
]
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def blob_query_response_table():
|
||||||
|
image_field = pa.field(
|
||||||
|
"image",
|
||||||
|
pa.struct(
|
||||||
|
[
|
||||||
|
pa.field("kind", pa.uint8(), nullable=False),
|
||||||
|
pa.field("position", pa.uint64(), nullable=False),
|
||||||
|
pa.field("size", pa.uint64(), nullable=False),
|
||||||
|
pa.field("blob_id", pa.uint32(), nullable=False),
|
||||||
|
pa.field("blob_uri", pa.string(), nullable=False),
|
||||||
|
]
|
||||||
|
),
|
||||||
|
metadata={"lance-encoding:blob": "true"},
|
||||||
|
)
|
||||||
|
images = pa.StructArray.from_arrays(
|
||||||
|
[
|
||||||
|
pa.array([1, 0, 0], type=pa.uint8()),
|
||||||
|
pa.array([0, 0, 0], type=pa.uint64()),
|
||||||
|
pa.array([5, 0, 5], type=pa.uint64()),
|
||||||
|
pa.array([1, 0, 2], type=pa.uint32()),
|
||||||
|
pa.array(["", "", ""], type=pa.string()),
|
||||||
|
],
|
||||||
|
fields=image_field.type,
|
||||||
|
mask=pa.array([False, True, False]),
|
||||||
|
)
|
||||||
|
return pa.Table.from_arrays(
|
||||||
|
[
|
||||||
|
pa.array([1, 2, 3], type=pa.int64()),
|
||||||
|
images,
|
||||||
|
pa.array([10, 20, 30], type=pa.uint64()),
|
||||||
|
],
|
||||||
|
schema=pa.schema(
|
||||||
|
[
|
||||||
|
pa.field("id", pa.int64(), nullable=False),
|
||||||
|
image_field,
|
||||||
|
pa.field("_rowid", pa.uint64()),
|
||||||
|
]
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@contextlib.contextmanager
|
||||||
|
def blob_remote_table(*, server_version=Version("0.5.0")):
|
||||||
|
def handler(request):
|
||||||
|
if request.path == "/v1/table/test/describe/":
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/json")
|
||||||
|
request.send_header("phalanx-version", str(server_version))
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(json.dumps(BLOB_DESCRIBE_RESPONSE).encode())
|
||||||
|
elif request.path.startswith("/v1/table/test/blob/image/"):
|
||||||
|
path = request.path.partition("?")[0]
|
||||||
|
row_id = int(path.split("/")[-2])
|
||||||
|
payload = {10: b"alpha", 20: None, 30: b"gamma"}[row_id]
|
||||||
|
if payload is None:
|
||||||
|
request.send_response(204)
|
||||||
|
request.end_headers()
|
||||||
|
return
|
||||||
|
byte_range = request.headers["Range"].removeprefix("bytes=")
|
||||||
|
start_text, end_text = byte_range.split("-", maxsplit=1)
|
||||||
|
start = int(start_text)
|
||||||
|
end = int(end_text) if end_text else len(payload) - 1
|
||||||
|
chunk = payload[start : end + 1]
|
||||||
|
request.send_response(206)
|
||||||
|
request.send_header("Content-Range", f"bytes {start}-{end}/{len(payload)}")
|
||||||
|
request.send_header("Content-Length", str(len(chunk)))
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(chunk)
|
||||||
|
elif request.path == "/v1/table/test/query/":
|
||||||
|
content_len = int(request.headers.get("Content-Length", 0))
|
||||||
|
body = json.loads(request.rfile.read(content_len))
|
||||||
|
assert body["columns"] == ["id", "image"]
|
||||||
|
assert body["with_row_id"] is True
|
||||||
|
response_table = blob_query_response_table()
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/vnd.apache.arrow.file")
|
||||||
|
request.end_headers()
|
||||||
|
with pa.ipc.new_file(request.wfile, response_table.schema) as writer:
|
||||||
|
writer.write_table(response_table)
|
||||||
|
elif request.path == "/v1/table/test/fetch_blobs/":
|
||||||
|
content_len = int(request.headers.get("Content-Length", 0))
|
||||||
|
body = json.loads(request.rfile.read(content_len))
|
||||||
|
assert body["column"] == "image"
|
||||||
|
assert body["row_ids"] == [10, 20, 30]
|
||||||
|
response_table = pa.table(
|
||||||
|
{"image": pa.array([b"alpha", None, b"gamma"], type=pa.large_binary())}
|
||||||
|
)
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/vnd.apache.arrow.stream")
|
||||||
|
request.end_headers()
|
||||||
|
with pa.ipc.new_stream(request.wfile, response_table.schema) as writer:
|
||||||
|
writer.write_table(response_table)
|
||||||
|
else:
|
||||||
|
request.send_response(404)
|
||||||
|
request.end_headers()
|
||||||
|
|
||||||
|
with mock_lancedb_connection(handler) as db:
|
||||||
|
yield db.open_table("test")
|
||||||
|
|
||||||
|
|
||||||
|
def test_remote_blob_columns_and_fetch():
|
||||||
|
with blob_remote_table() as table:
|
||||||
|
assert table.blob_columns() == ["image"]
|
||||||
|
blobs = table.fetch_blobs("image", [10, 20, 30])
|
||||||
|
assert blobs.to_pylist() == [b"alpha", None, b"gamma"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_remote_blob_files_are_lazy_seekable_handles():
|
||||||
|
with blob_remote_table() as table:
|
||||||
|
files = table.fetch_blob_files("image", [10, 20, 30])
|
||||||
|
|
||||||
|
assert len(files) == 3
|
||||||
|
alpha, null_row, gamma = files
|
||||||
|
assert null_row is None
|
||||||
|
assert alpha is not None
|
||||||
|
assert gamma is not None
|
||||||
|
assert alpha.size() == 5
|
||||||
|
assert alpha.read_range(1, 3) == b"lph"
|
||||||
|
gamma.seek(2)
|
||||||
|
assert gamma.read() == b"mma"
|
||||||
|
|
||||||
|
|
||||||
|
def test_remote_blob_fetch_accepts_query_table():
|
||||||
|
hits = pa.table({"_rowid": pa.array([10, 20, 30], type=pa.uint64())})
|
||||||
|
|
||||||
|
with blob_remote_table() as table:
|
||||||
|
blobs = table.fetch_blobs("image", hits)
|
||||||
|
|
||||||
|
assert blobs.to_pylist() == [b"alpha", None, b"gamma"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_remote_blob_query_stashes_row_ids_for_fetch():
|
||||||
|
with blob_remote_table() as table:
|
||||||
|
hits = table.search().select(["id", "image"]).limit(3).to_arrow()
|
||||||
|
assert "_rowid" not in hits.column_names
|
||||||
|
assert "_lance_row_id" in hits.schema.field("image").type.names
|
||||||
|
blobs = table.fetch_blobs("image", hits)
|
||||||
|
|
||||||
|
assert blobs.to_pylist() == [b"alpha", None, b"gamma"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_remote_blob_query_survives_a_server_that_ignores_the_row_id_request():
|
||||||
|
def handler(request):
|
||||||
|
if request.path == "/v1/table/test/describe/":
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/json")
|
||||||
|
request.send_header("phalanx-version", "0.5.0")
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(json.dumps(BLOB_DESCRIBE_RESPONSE).encode())
|
||||||
|
elif request.path == "/v1/table/test/query/":
|
||||||
|
content_len = int(request.headers.get("Content-Length", 0))
|
||||||
|
assert json.loads(request.rfile.read(content_len))["with_row_id"] is True
|
||||||
|
response_table = blob_query_response_table().drop_columns(["_rowid"])
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/vnd.apache.arrow.file")
|
||||||
|
request.end_headers()
|
||||||
|
with pa.ipc.new_file(request.wfile, response_table.schema) as writer:
|
||||||
|
writer.write_table(response_table)
|
||||||
|
else:
|
||||||
|
request.send_response(404)
|
||||||
|
request.end_headers()
|
||||||
|
|
||||||
|
with mock_lancedb_connection(handler) as db:
|
||||||
|
table = db.open_table("test")
|
||||||
|
hits = table.search().select(["id", "image"]).limit(3).to_arrow()
|
||||||
|
|
||||||
|
assert hits.column_names == ["id", "image"]
|
||||||
|
assert "_lance_row_id" not in hits.schema.field("image").type.names
|
||||||
|
with pytest.raises(ValueError, match="pass a list of row ids"):
|
||||||
|
table.fetch_blobs("image", hits)
|
||||||
|
|
||||||
|
|
||||||
|
def test_remote_blob_byte_apis_not_supported_on_old_server():
|
||||||
|
with blob_remote_table(server_version=Version("0.1.0")) as table:
|
||||||
|
assert table.blob_columns() == ["image"]
|
||||||
|
with pytest.raises(NotImplementedError, match="not supported"):
|
||||||
|
table.fetch_blobs("image", [1])
|
||||||
|
with pytest.raises(NotImplementedError, match="not supported"):
|
||||||
|
table.fetch_blob_files("image", [1])
|
||||||
|
|
||||||
|
|
||||||
|
def test_remote_connection_jobs_surface():
|
||||||
|
from lancedb.exceptions import JobFailedError
|
||||||
|
|
||||||
|
schema = pa.schema([("state", pa.string())])
|
||||||
|
batch = pa.record_batch([pa.array(["created", "done"])], schema=schema)
|
||||||
|
sink = pa.BufferOutputStream()
|
||||||
|
with pa.ipc.new_stream(sink, schema) as writer:
|
||||||
|
writer.write_batch(batch)
|
||||||
|
events_body = sink.getvalue().to_pybytes()
|
||||||
|
|
||||||
|
def handler(request):
|
||||||
|
content_len = int(request.headers.get("Content-Length", 0))
|
||||||
|
body = request.rfile.read(content_len) if content_len > 0 else b""
|
||||||
|
payload = json.loads(body) if body else {}
|
||||||
|
if request.path == "/v1/jobs/list":
|
||||||
|
if payload.get("page_token") is None:
|
||||||
|
rsp = dict(
|
||||||
|
jobs=[
|
||||||
|
dict(
|
||||||
|
job_id="job-1",
|
||||||
|
table="t1",
|
||||||
|
job_type="create_index",
|
||||||
|
state="in_progress",
|
||||||
|
created_at_millis=1000,
|
||||||
|
)
|
||||||
|
],
|
||||||
|
page_token="next",
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
assert payload["page_token"] == "next"
|
||||||
|
rsp = dict(
|
||||||
|
jobs=[
|
||||||
|
dict(
|
||||||
|
job_id="job-2",
|
||||||
|
table="t2",
|
||||||
|
job_type="create_index",
|
||||||
|
state="succeeded",
|
||||||
|
created_at_millis=2000,
|
||||||
|
)
|
||||||
|
]
|
||||||
|
)
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/json")
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(json.dumps(rsp).encode())
|
||||||
|
elif request.path == "/v1/jobs/describe":
|
||||||
|
if payload["job_id"] != "job-1":
|
||||||
|
request.send_response(404)
|
||||||
|
request.end_headers()
|
||||||
|
return
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/json")
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(
|
||||||
|
json.dumps(
|
||||||
|
dict(
|
||||||
|
job_id="job-1",
|
||||||
|
job_type="create_index",
|
||||||
|
job_state="FAILED",
|
||||||
|
creation_ms=1000,
|
||||||
|
spec=dict(column="vec"),
|
||||||
|
failure=dict(
|
||||||
|
phase="execute", message="worker died", retryable=True
|
||||||
|
),
|
||||||
|
)
|
||||||
|
).encode()
|
||||||
|
)
|
||||||
|
elif request.path == "/v1/jobs/cancel":
|
||||||
|
if payload["job_id"] != "job-1":
|
||||||
|
request.send_response(404)
|
||||||
|
request.end_headers()
|
||||||
|
return
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/json")
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(b'{"job_id": "job-1"}')
|
||||||
|
elif request.path == "/v1/jobs/query_events":
|
||||||
|
assert payload["job_id"] == "job-1"
|
||||||
|
request.send_response(200)
|
||||||
|
request.send_header("Content-Type", "application/vnd.apache.arrow.stream")
|
||||||
|
request.end_headers()
|
||||||
|
request.wfile.write(events_body)
|
||||||
|
else:
|
||||||
|
request.send_response(404)
|
||||||
|
request.end_headers()
|
||||||
|
|
||||||
|
with mock_lancedb_connection(handler) as db:
|
||||||
|
jobs = db.list_jobs()
|
||||||
|
assert [job.job_id for job in jobs] == ["job-1", "job-2"]
|
||||||
|
assert jobs[0].state == "running"
|
||||||
|
assert jobs[0].table == "t1"
|
||||||
|
assert jobs[1].state == "finished"
|
||||||
|
|
||||||
|
description = db.get_job("job-1")
|
||||||
|
assert description.job_type == "create_index"
|
||||||
|
assert description.state == "failed"
|
||||||
|
assert json.loads(description.spec_json) == {"column": "vec"}
|
||||||
|
assert description.failure.message == "worker died"
|
||||||
|
assert description.failure.retryable is True
|
||||||
|
assert db.get_job("missing") is None
|
||||||
|
|
||||||
|
assert db.cancel_job("job-1") is True
|
||||||
|
assert db.cancel_job("missing") is False
|
||||||
|
|
||||||
|
batches = db.job_history("job-1")
|
||||||
|
assert len(batches) == 1
|
||||||
|
assert batches[0].num_rows == 2
|
||||||
|
assert batches[0].column("state").to_pylist() == ["created", "done"]
|
||||||
|
|
||||||
|
job = db.job("job-1")
|
||||||
|
assert job.id == "job-1"
|
||||||
|
assert job.status() == "failed"
|
||||||
|
with pytest.raises(JobFailedError, match="worker died"):
|
||||||
|
job.wait(timeout=timedelta(seconds=5))
|
||||||
|
|||||||
@@ -2,10 +2,14 @@
|
|||||||
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
|
||||||
|
import ctypes
|
||||||
|
import gc
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
import threading
|
import threading
|
||||||
import warnings
|
import warnings
|
||||||
|
import weakref
|
||||||
|
from concurrent.futures import ThreadPoolExecutor
|
||||||
from datetime import date, datetime, timedelta
|
from datetime import date, datetime, timedelta
|
||||||
from time import sleep
|
from time import sleep
|
||||||
from typing import List
|
from typing import List
|
||||||
@@ -98,6 +102,30 @@ def test_basic(mem_db: DBConnection):
|
|||||||
assert table.to_arrow() == expected_data
|
assert table.to_arrow() == expected_data
|
||||||
|
|
||||||
|
|
||||||
|
def test_search_preserves_nulls_from_sliced_arrow_table(mem_db: DBConnection):
|
||||||
|
data = pa.table(
|
||||||
|
{
|
||||||
|
"id": [0, 1, 2, 3, 4],
|
||||||
|
"score_cn": [None, 22, None, 5, 8],
|
||||||
|
"score_mt": [None, 42, None, 5, 8],
|
||||||
|
"vector": [
|
||||||
|
[20, 19, -1, -1],
|
||||||
|
[41, 38, 22, 42],
|
||||||
|
[10, 10, -1, -1],
|
||||||
|
[5, 5, 5, 5],
|
||||||
|
[8, 8, 8, 8],
|
||||||
|
],
|
||||||
|
}
|
||||||
|
).slice(1)
|
||||||
|
|
||||||
|
table = mem_db.create_table("sliced_nullable", data=data)
|
||||||
|
result = table.search([41, 38, 22, 42]).limit(1).to_arrow()
|
||||||
|
|
||||||
|
assert result["id"].to_pylist() == [1]
|
||||||
|
assert result["score_cn"].to_pylist() == [22]
|
||||||
|
assert result["score_mt"].to_pylist() == [42]
|
||||||
|
|
||||||
|
|
||||||
def test_table_to_pandas_default_matches_arrow(tmp_db: DBConnection):
|
def test_table_to_pandas_default_matches_arrow(tmp_db: DBConnection):
|
||||||
pd = pytest.importorskip("pandas")
|
pd = pytest.importorskip("pandas")
|
||||||
data = pa.table({"id": [1, 2], "text": ["one", "two"]})
|
data = pa.table({"id": [1, 2], "text": ["one", "two"]})
|
||||||
@@ -434,6 +462,38 @@ def test_add(mem_db: DBConnection):
|
|||||||
_add(table, schema)
|
_add(table, schema)
|
||||||
|
|
||||||
|
|
||||||
|
def test_add_releases_arrow_buffers_without_gc(mem_db: DBConnection):
|
||||||
|
"""Regression test for https://github.com/lancedb/lancedb/issues/2512."""
|
||||||
|
schema = pa.schema([pa.field("x", pa.int64())])
|
||||||
|
table = mem_db.create_table("test_add_releases_arrow_buffers", schema=schema)
|
||||||
|
|
||||||
|
class BufferOwner:
|
||||||
|
def __init__(self, size: int):
|
||||||
|
self.memory = ctypes.create_string_buffer(size)
|
||||||
|
|
||||||
|
owner_refs = []
|
||||||
|
gc_was_enabled = gc.isenabled()
|
||||||
|
gc.disable()
|
||||||
|
try:
|
||||||
|
for _ in range(3):
|
||||||
|
size = 8 * 1024
|
||||||
|
owner = BufferOwner(size)
|
||||||
|
arrow_buffer = pa.foreign_buffer(
|
||||||
|
ctypes.addressof(owner.memory), size, owner
|
||||||
|
)
|
||||||
|
array = pa.Array.from_buffers(pa.int64(), 1024, [None, arrow_buffer])
|
||||||
|
batch = pa.RecordBatch.from_arrays([array], schema=schema)
|
||||||
|
owner_refs.append(weakref.ref(owner))
|
||||||
|
|
||||||
|
table.add(batch)
|
||||||
|
del batch, array, arrow_buffer, owner
|
||||||
|
|
||||||
|
assert all(owner_ref() is None for owner_ref in owner_refs)
|
||||||
|
finally:
|
||||||
|
if gc_was_enabled:
|
||||||
|
gc.enable()
|
||||||
|
|
||||||
|
|
||||||
def test_add_write_parallelism(mem_db: DBConnection):
|
def test_add_write_parallelism(mem_db: DBConnection):
|
||||||
schema = pa.schema([pa.field("id", pa.int64())])
|
schema = pa.schema([pa.field("id", pa.int64())])
|
||||||
table = mem_db.create_table("test", schema=schema)
|
table = mem_db.create_table("test", schema=schema)
|
||||||
@@ -869,6 +929,7 @@ def test_polars(mem_db: DBConnection):
|
|||||||
|
|
||||||
# enter table to polars dataframe
|
# enter table to polars dataframe
|
||||||
result = table.to_polars()
|
result = table.to_polars()
|
||||||
|
assert isinstance(result, pl.LazyFrame)
|
||||||
assert np.allclose(result.collect()["vector"].to_list(), data["vector"])
|
assert np.allclose(result.collect()["vector"].to_list(), data["vector"])
|
||||||
|
|
||||||
# make sure filtering isn't broken
|
# make sure filtering isn't broken
|
||||||
@@ -1402,6 +1463,15 @@ async def test_async_open_table_with_branch_version(tmp_path):
|
|||||||
assert await pinned.count_rows() == 4 # writable again
|
assert await pinned.count_rows() == 4 # writable again
|
||||||
|
|
||||||
|
|
||||||
|
def test_create_index_async_returns_done_job(mem_db: DBConnection):
|
||||||
|
table = mem_db.create_table("job_test", [{"id": i} for i in range(10)])
|
||||||
|
job = table.create_index_async("id", config=BTree())
|
||||||
|
assert job.id is None
|
||||||
|
job.wait()
|
||||||
|
assert len(table.list_indices()) == 1
|
||||||
|
job.cancel()
|
||||||
|
|
||||||
|
|
||||||
@patch("lancedb.table.AsyncTable.create_index")
|
@patch("lancedb.table.AsyncTable.create_index")
|
||||||
def test_create_index_method(mock_create_index, mem_db: DBConnection):
|
def test_create_index_method(mock_create_index, mem_db: DBConnection):
|
||||||
table = mem_db.create_table(
|
table = mem_db.create_table(
|
||||||
@@ -1776,6 +1846,27 @@ def test_add_with_empty_fixed_size_list_drops_bad_rows(mem_db: DBConnection):
|
|||||||
assert np.allclose(data["embedding"].to_pylist()[0], np.array([0.1] * 16))
|
assert np.allclose(data["embedding"].to_pylist()[0], np.array([0.1] * 16))
|
||||||
|
|
||||||
|
|
||||||
|
def test_add_nullable_fixed_size_list_with_none(mem_db: DBConnection):
|
||||||
|
"""Regression test for issue #2340."""
|
||||||
|
table = mem_db.create_table(
|
||||||
|
"test_nullable_fixed_size_list",
|
||||||
|
schema=pa.schema(
|
||||||
|
[
|
||||||
|
pa.field("id", pa.string()),
|
||||||
|
pa.field("feature", pa.list_(pa.float32(), 256)),
|
||||||
|
pa.field("tags", pa.list_(pa.string())),
|
||||||
|
]
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
table.add([{"id": "1", "feature": None, "tags": ["tag1", "tag2"]}])
|
||||||
|
|
||||||
|
result = table.to_arrow()
|
||||||
|
assert result.to_pylist() == [
|
||||||
|
{"id": "1", "feature": None, "tags": ["tag1", "tag2"]}
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
def test_add_nullable_struct_with_none(mem_db: DBConnection):
|
def test_add_nullable_struct_with_none(mem_db: DBConnection):
|
||||||
"""Regression test for issue #2654: a nullable struct column whose
|
"""Regression test for issue #2654: a nullable struct column whose
|
||||||
first batch contains only None values must not crash in
|
first batch contains only None values must not crash in
|
||||||
@@ -1815,6 +1906,33 @@ def test_add_nullable_struct_with_none(mem_db: DBConnection):
|
|||||||
assert result.column("data").to_pylist() == [{"x": 1.0}, None]
|
assert result.column("data").to_pylist() == [{"x": 1.0}, None]
|
||||||
|
|
||||||
|
|
||||||
|
def test_read_mostly_null_list_v2_2_page_boundary(tmp_path):
|
||||||
|
# Regression test for #3194. This row/value count crosses a v2.2 structural
|
||||||
|
# encoding page boundary where Lance 3.0.0 sliced repetition/definition
|
||||||
|
# levels by row offset and decoded child arrays at different lengths.
|
||||||
|
num_rows = 64_885
|
||||||
|
num_values = 217
|
||||||
|
list_type = pa.list_(pa.float32())
|
||||||
|
source = pa.table(
|
||||||
|
{
|
||||||
|
"id": np.arange(num_rows, dtype=np.int64),
|
||||||
|
"coords": pa.array(
|
||||||
|
[[1.0, 2.0, 3.0, 4.0]] * num_values + [None] * (num_rows - num_values),
|
||||||
|
type=list_type,
|
||||||
|
),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
db = lancedb.connect(
|
||||||
|
tmp_path,
|
||||||
|
storage_options={"new_table_data_storage_version": "2.2"},
|
||||||
|
)
|
||||||
|
table = db.create_table("test_sparse_nullable_list", data=source)
|
||||||
|
|
||||||
|
result = table.search().select(["id", "coords"]).limit(num_rows).to_arrow()
|
||||||
|
|
||||||
|
assert result.equals(source)
|
||||||
|
|
||||||
|
|
||||||
def test_add_with_integer_embeddings_preserves_casting(mem_db: DBConnection):
|
def test_add_with_integer_embeddings_preserves_casting(mem_db: DBConnection):
|
||||||
class Schema(LanceModel):
|
class Schema(LanceModel):
|
||||||
text: str
|
text: str
|
||||||
@@ -2100,6 +2218,45 @@ def test_merge(tmp_db: DBConnection, tmp_path):
|
|||||||
table.merge(other_dataset, left_on="id")
|
table.merge(other_dataset, left_on="id")
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("storage_version", ["legacy", "stable"])
|
||||||
|
def test_search_after_merge(tmp_path, storage_version):
|
||||||
|
pytest.importorskip("lance")
|
||||||
|
pd = pytest.importorskip("pandas")
|
||||||
|
|
||||||
|
db = lancedb.connect(
|
||||||
|
tmp_path,
|
||||||
|
storage_options={"new_table_data_storage_version": storage_version},
|
||||||
|
)
|
||||||
|
rng = np.random.default_rng(42)
|
||||||
|
row_count = 512
|
||||||
|
vectors = rng.standard_normal((row_count, 8)).astype(np.float32)
|
||||||
|
table = db.create_table(
|
||||||
|
"search_after_merge",
|
||||||
|
data=pd.DataFrame(
|
||||||
|
{
|
||||||
|
"id": [str(i) for i in range(row_count)],
|
||||||
|
"vector": list(vectors),
|
||||||
|
}
|
||||||
|
),
|
||||||
|
)
|
||||||
|
table.create_index("vector", config=IvfPq(num_partitions=1, num_sub_vectors=2))
|
||||||
|
|
||||||
|
links = pd.DataFrame(
|
||||||
|
{
|
||||||
|
"id": [str(i) for i in range(row_count // 2)],
|
||||||
|
"link": [f"https://example.com/{i}" for i in range(row_count // 2)],
|
||||||
|
}
|
||||||
|
)
|
||||||
|
table.merge(links, left_on="id")
|
||||||
|
|
||||||
|
query = table.search(vectors[-1]).refine_factor(50).limit(10)
|
||||||
|
assert "ANN" in query.explain_plan(verbose=True)
|
||||||
|
|
||||||
|
result = query.to_arrow()
|
||||||
|
links_by_id = dict(zip(result["id"].to_pylist(), result["link"].to_pylist()))
|
||||||
|
assert links_by_id[str(row_count - 1)] is None
|
||||||
|
|
||||||
|
|
||||||
def test_delete(mem_db: DBConnection):
|
def test_delete(mem_db: DBConnection):
|
||||||
table = mem_db.create_table(
|
table = mem_db.create_table(
|
||||||
"my_table",
|
"my_table",
|
||||||
@@ -2115,6 +2272,27 @@ def test_delete(mem_db: DBConnection):
|
|||||||
assert table.to_arrow()["id"].to_pylist() == [1]
|
assert table.to_arrow()["id"].to_pylist() == [1]
|
||||||
|
|
||||||
|
|
||||||
|
def test_concurrent_deletes_are_thread_safe(mem_db: DBConnection):
|
||||||
|
num_workers = 8
|
||||||
|
table = mem_db.create_table(
|
||||||
|
"my_table", data=[{"id": row_id} for row_id in range(num_workers)]
|
||||||
|
)
|
||||||
|
barrier = threading.Barrier(num_workers)
|
||||||
|
|
||||||
|
def delete(row_id: int):
|
||||||
|
barrier.wait()
|
||||||
|
return table.delete(f"id = {row_id}")
|
||||||
|
|
||||||
|
with ThreadPoolExecutor(max_workers=num_workers) as pool:
|
||||||
|
results = list(pool.map(delete, range(num_workers)))
|
||||||
|
|
||||||
|
assert all(result.num_deleted_rows == 1 for result in results)
|
||||||
|
assert sorted(result.version for result in results) == list(
|
||||||
|
range(2, num_workers + 2)
|
||||||
|
)
|
||||||
|
assert table.count_rows() == 0
|
||||||
|
|
||||||
|
|
||||||
def test_delete_expr(mem_db: DBConnection):
|
def test_delete_expr(mem_db: DBConnection):
|
||||||
table = mem_db.create_table(
|
table = mem_db.create_table(
|
||||||
"my_table",
|
"my_table",
|
||||||
@@ -2165,6 +2343,20 @@ def test_update(mem_db: DBConnection):
|
|||||||
assert np.allclose(v, np.array([[1.2, 1.9], [1.1, 1.1]]))
|
assert np.allclose(v, np.array([[1.2, 1.9], [1.1, 1.1]]))
|
||||||
|
|
||||||
|
|
||||||
|
def test_update_with_arrow_scalar(mem_db: DBConnection):
|
||||||
|
schema = pa.schema({"id": pa.int64(), "vector": pa.list_(pa.float32(), 4)})
|
||||||
|
table = mem_db.create_table("my_table", schema=schema)
|
||||||
|
table.add([{"id": 1, "vector": [1.0, 2.0, 3.0, 4.0]}])
|
||||||
|
|
||||||
|
value = table.search().select(["vector"]).limit(1).to_arrow()["vector"][0]
|
||||||
|
assert isinstance(value, pa.FixedSizeListScalar)
|
||||||
|
|
||||||
|
result = table.update(where="id == 1", values={"vector": value})
|
||||||
|
|
||||||
|
assert result.rows_updated == 1
|
||||||
|
assert table.to_arrow()["vector"].to_pylist() == [[1.0, 2.0, 3.0, 4.0]]
|
||||||
|
|
||||||
|
|
||||||
def test_update_types(mem_db: DBConnection):
|
def test_update_types(mem_db: DBConnection):
|
||||||
table = mem_db.create_table(
|
table = mem_db.create_table(
|
||||||
"my_table",
|
"my_table",
|
||||||
@@ -2332,6 +2524,55 @@ def test_merge_insert(mem_db: DBConnection):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_merge_insert_nullable_pandas_into_pydantic_schema(mem_db: DBConnection):
|
||||||
|
# Regression test for https://github.com/lancedb/lancedb/issues/2366
|
||||||
|
pd = pytest.importorskip("pandas")
|
||||||
|
|
||||||
|
class Document(LanceModel):
|
||||||
|
id: int
|
||||||
|
title: str
|
||||||
|
content: str
|
||||||
|
|
||||||
|
table = mem_db.create_table("documents", schema=Document)
|
||||||
|
table.add(
|
||||||
|
pd.DataFrame(
|
||||||
|
{
|
||||||
|
"title": ["Old title", "Unchanged"],
|
||||||
|
"id": [2, 3],
|
||||||
|
"content": ["Old content", "Keep this"],
|
||||||
|
}
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
# Pandas produces nullable Arrow fields, in an order that differs from the
|
||||||
|
# non-nullable Pydantic schema. This is valid as long as the data has no nulls.
|
||||||
|
new_data = pd.DataFrame(
|
||||||
|
{
|
||||||
|
"title": ["Inserted", "Updated"],
|
||||||
|
"id": [1, 2],
|
||||||
|
"content": ["New row", "New content"],
|
||||||
|
}
|
||||||
|
)
|
||||||
|
result = (
|
||||||
|
table.merge_insert("id")
|
||||||
|
.when_matched_update_all()
|
||||||
|
.when_not_matched_insert_all()
|
||||||
|
.execute(new_data)
|
||||||
|
)
|
||||||
|
|
||||||
|
assert result.num_inserted_rows == 1
|
||||||
|
assert result.num_updated_rows == 1
|
||||||
|
expected = pa.Table.from_pylist(
|
||||||
|
[
|
||||||
|
{"id": 1, "title": "Inserted", "content": "New row"},
|
||||||
|
{"id": 2, "title": "Updated", "content": "New content"},
|
||||||
|
{"id": 3, "title": "Unchanged", "content": "Keep this"},
|
||||||
|
],
|
||||||
|
schema=Document.to_arrow_schema(),
|
||||||
|
)
|
||||||
|
assert table.to_arrow().sort_by("id") == expected
|
||||||
|
|
||||||
|
|
||||||
def test_merge_insert_by_source_delete_expr(mem_db: DBConnection):
|
def test_merge_insert_by_source_delete_expr(mem_db: DBConnection):
|
||||||
table = mem_db.create_table(
|
table = mem_db.create_table(
|
||||||
"my_table",
|
"my_table",
|
||||||
@@ -2355,6 +2596,29 @@ def test_merge_insert_by_source_delete_expr(mem_db: DBConnection):
|
|||||||
assert table.to_arrow().sort_by("a") == expected
|
assert table.to_arrow().sort_by("a") == expected
|
||||||
|
|
||||||
|
|
||||||
|
def test_merge_insert_by_source_delete_reconfigure(mem_db: DBConnection):
|
||||||
|
# Calling when_not_matched_by_source_delete() again with no condition must
|
||||||
|
# widen the delete to unconditional, not keep the earlier condition around.
|
||||||
|
table = mem_db.create_table(
|
||||||
|
"my_table",
|
||||||
|
data=pa.table({"a": [1, 2, 3], "b": ["a", "b", "c"]}),
|
||||||
|
)
|
||||||
|
new_data = pa.table({"a": [2, 4], "b": ["x", "z"]})
|
||||||
|
|
||||||
|
merge_insert_res = (
|
||||||
|
table.merge_insert("a")
|
||||||
|
.when_matched_update_all()
|
||||||
|
.when_not_matched_insert_all()
|
||||||
|
.when_not_matched_by_source_delete("a > 2")
|
||||||
|
.when_not_matched_by_source_delete()
|
||||||
|
.execute(new_data)
|
||||||
|
)
|
||||||
|
assert merge_insert_res.num_deleted_rows == 2
|
||||||
|
|
||||||
|
expected = pa.table({"a": [2, 4], "b": ["x", "z"]})
|
||||||
|
assert table.to_arrow().sort_by("a") == expected
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_merge_insert_by_source_delete_expr_async(
|
async def test_merge_insert_by_source_delete_expr_async(
|
||||||
mem_db_async: AsyncConnection,
|
mem_db_async: AsyncConnection,
|
||||||
@@ -2409,6 +2673,36 @@ def test_merge_insert_subschema(mem_db: DBConnection, data_format):
|
|||||||
assert table.to_arrow().sort_by("id") == expected
|
assert table.to_arrow().sort_by("id") == expected
|
||||||
|
|
||||||
|
|
||||||
|
def test_repeated_partial_merge_insert_with_scalar_index(mem_db: DBConnection):
|
||||||
|
def make_batch(start: int) -> pa.Table:
|
||||||
|
return pa.table(
|
||||||
|
{
|
||||||
|
"id": [f"id-{i:04}" for i in range(start, start + 100)],
|
||||||
|
"category": ["A"] * 100,
|
||||||
|
"value_a": [float(i) for i in range(start, start + 100)],
|
||||||
|
"value_b": [float(i) / 10 for i in range(100)],
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
table = mem_db.create_table("my_table", data=make_batch(0))
|
||||||
|
table.add(make_batch(100))
|
||||||
|
table.add(make_batch(200))
|
||||||
|
table.create_index("id", config=BTree())
|
||||||
|
|
||||||
|
ids = [f"id-{i:04}" for i in range(100, 200)]
|
||||||
|
for value in (999.0, 888.0):
|
||||||
|
result = (
|
||||||
|
table.merge_insert("id")
|
||||||
|
.when_matched_update_all()
|
||||||
|
.execute(pa.table({"id": ids, "value_a": [value] * 100}))
|
||||||
|
)
|
||||||
|
assert result.num_updated_rows == 100
|
||||||
|
|
||||||
|
actual = table.to_arrow().sort_by("id")
|
||||||
|
assert actual.num_rows == 300
|
||||||
|
assert actual["value_a"].to_pylist()[100:200] == [888.0] * 100
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_merge_insert_async(mem_db_async: AsyncConnection):
|
async def test_merge_insert_async(mem_db_async: AsyncConnection):
|
||||||
data = pa.table({"a": [1, 2, 3], "b": ["a", "b", "c"]})
|
data = pa.table({"a": [1, 2, 3], "b": ["a", "b", "c"]})
|
||||||
@@ -2505,15 +2799,40 @@ def test_create_with_embedding_function(mem_db: DBConnection):
|
|||||||
assert actual == expected
|
assert actual == expected
|
||||||
|
|
||||||
|
|
||||||
|
def test_create_f16_table_from_arrow_data(mem_db: DBConnection):
|
||||||
|
dimension = 32
|
||||||
|
num_rows = 512
|
||||||
|
values = pa.array(
|
||||||
|
np.random.default_rng(42)
|
||||||
|
.standard_normal(num_rows * dimension)
|
||||||
|
.astype(np.float16)
|
||||||
|
)
|
||||||
|
df = pa.table(
|
||||||
|
{
|
||||||
|
"text": [f"s-{i}" for i in range(num_rows)],
|
||||||
|
"vector": pa.FixedSizeListArray.from_arrays(values, dimension),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
table = mem_db.create_table("f16_tbl", data=df)
|
||||||
|
assert table.schema.field("vector").type == pa.list_(pa.float16(), dimension)
|
||||||
|
table.create_index(num_partitions=2, num_sub_vectors=2)
|
||||||
|
|
||||||
|
query = df["vector"][2].as_py()
|
||||||
|
expected = table.search(query).limit(2).to_arrow()
|
||||||
|
|
||||||
|
assert "s-2" in expected["text"].to_pylist()
|
||||||
|
|
||||||
|
|
||||||
def test_create_f16_table(mem_db: DBConnection):
|
def test_create_f16_table(mem_db: DBConnection):
|
||||||
class MyTable(LanceModel):
|
class MyTable(LanceModel):
|
||||||
text: str
|
text: str
|
||||||
vector: Vector(32, value_type=pa.float16())
|
vector: Vector(32, value_type=pa.float16())
|
||||||
|
|
||||||
|
rng = np.random.default_rng(42)
|
||||||
df = pa.table(
|
df = pa.table(
|
||||||
{
|
{
|
||||||
"text": [f"s-{i}" for i in range(512)],
|
"text": [f"s-{i}" for i in range(512)],
|
||||||
"vector": [np.random.randn(32).astype(np.float16) for _ in range(512)],
|
"vector": [rng.standard_normal(32).astype(np.float16) for _ in range(512)],
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
table = mem_db.create_table(
|
table = mem_db.create_table(
|
||||||
@@ -3087,9 +3406,6 @@ def test_consistency(tmp_path, consistency_interval):
|
|||||||
|
|
||||||
db2 = lancedb.connect(tmp_path, read_consistency_interval=consistency_interval)
|
db2 = lancedb.connect(tmp_path, read_consistency_interval=consistency_interval)
|
||||||
table2 = db2.open_table("my_table")
|
table2 = db2.open_table("my_table")
|
||||||
if consistency_interval is not None:
|
|
||||||
assert "read_consistency_interval=datetime.timedelta(" in repr(db2)
|
|
||||||
assert "read_consistency_interval=datetime.timedelta(" in repr(table2)
|
|
||||||
assert table2.version == table.version
|
assert table2.version == table.version
|
||||||
|
|
||||||
table.add([{"id": 1}])
|
table.add([{"id": 1}])
|
||||||
@@ -3397,7 +3713,8 @@ def test_stats(mem_db: DBConnection):
|
|||||||
stats = table.stats()
|
stats = table.stats()
|
||||||
print(f"{stats=}")
|
print(f"{stats=}")
|
||||||
assert stats == {
|
assert stats == {
|
||||||
"total_bytes": 60,
|
# Full on-disk size of the data file, footer and metadata included.
|
||||||
|
"total_bytes": 633,
|
||||||
"num_rows": 2,
|
"num_rows": 2,
|
||||||
"num_indices": 0,
|
"num_indices": 0,
|
||||||
"fragment_stats": {
|
"fragment_stats": {
|
||||||
@@ -3415,6 +3732,13 @@ def test_stats(mem_db: DBConnection):
|
|||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# Index files count toward total_bytes too (only deletion files and
|
||||||
|
# manifests are excluded).
|
||||||
|
table.create_index("id", config=BTree())
|
||||||
|
stats_with_index = table.stats()
|
||||||
|
assert stats_with_index["num_indices"] == 1
|
||||||
|
assert stats_with_index["total_bytes"] > stats["total_bytes"]
|
||||||
|
|
||||||
|
|
||||||
def test_create_table_empty_list_with_schema(mem_db: DBConnection):
|
def test_create_table_empty_list_with_schema(mem_db: DBConnection):
|
||||||
"""Test creating table with empty list data and schema
|
"""Test creating table with empty list data and schema
|
||||||
@@ -3438,8 +3762,8 @@ def test_create_table_empty_list_no_schema_error(mem_db: DBConnection):
|
|||||||
mem_db.create_table("test_empty_no_schema", data=[])
|
mem_db.create_table("test_empty_no_schema", data=[])
|
||||||
|
|
||||||
|
|
||||||
def test_add_table_with_empty_embeddings(tmp_path):
|
def test_create_table_without_data_with_vector_schema(tmp_path):
|
||||||
"""Test exact scenario from issue #1968
|
"""Test exact scenario from issue #1968.
|
||||||
|
|
||||||
Regression test for issue #1968:
|
Regression test for issue #1968:
|
||||||
https://github.com/lancedb/lancedb/issues/1968
|
https://github.com/lancedb/lancedb/issues/1968
|
||||||
@@ -3451,6 +3775,9 @@ def test_add_table_with_empty_embeddings(tmp_path):
|
|||||||
embedding: Vector(16)
|
embedding: Vector(16)
|
||||||
|
|
||||||
table = db.create_table("test", schema=MySchema)
|
table = db.create_table("test", schema=MySchema)
|
||||||
|
assert table.count_rows() == 0
|
||||||
|
assert table.schema == MySchema.to_arrow_schema()
|
||||||
|
|
||||||
table.add(
|
table.add(
|
||||||
[{"text": "bar", "embedding": [0.1] * 16}],
|
[{"text": "bar", "embedding": [0.1] * 16}],
|
||||||
on_bad_vectors="drop",
|
on_bad_vectors="drop",
|
||||||
|
|||||||
@@ -75,6 +75,22 @@ class TestVoyageAIModelRegistration:
|
|||||||
with pytest.raises(ValueError, match="not supported"):
|
with pytest.raises(ValueError, match="not supported"):
|
||||||
func.ndims()
|
func.ndims()
|
||||||
|
|
||||||
|
def test_voyage3_source_embeddings_use_text_api(self, mock_voyageai_client):
|
||||||
|
"""Regression test for text table data being sent to the multimodal API."""
|
||||||
|
mock_voyageai_client.tokenize.return_value = [["hello", "world"]]
|
||||||
|
mock_voyageai_client.embed.return_value.embeddings = [[0.1] * 1024]
|
||||||
|
|
||||||
|
registry = get_registry()
|
||||||
|
func = registry.get("voyageai").create(name="voyage-3")
|
||||||
|
|
||||||
|
embeddings = func.compute_source_embeddings("hello world")
|
||||||
|
|
||||||
|
assert embeddings == [[0.1] * 1024]
|
||||||
|
mock_voyageai_client.embed.assert_called_once_with(
|
||||||
|
texts=["hello world"], model="voyage-3", input_type="document"
|
||||||
|
)
|
||||||
|
mock_voyageai_client.multimodal_embed.assert_not_called()
|
||||||
|
|
||||||
@pytest.mark.parametrize(
|
@pytest.mark.parametrize(
|
||||||
"model_name",
|
"model_name",
|
||||||
[
|
[
|
||||||
|
|||||||
@@ -0,0 +1,15 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
from typing import assert_type
|
||||||
|
|
||||||
|
import lancedb
|
||||||
|
from lancedb import AsyncConnection, DBConnection
|
||||||
|
|
||||||
|
|
||||||
|
def check_connect_type() -> None:
|
||||||
|
assert_type(lancedb.connect("memory://"), DBConnection)
|
||||||
|
|
||||||
|
|
||||||
|
async def check_connect_async_type() -> None:
|
||||||
|
assert_type(await lancedb.connect_async("memory://"), AsyncConnection)
|
||||||
@@ -13,7 +13,11 @@ use crate::{
|
|||||||
runtime::future_into_py,
|
runtime::future_into_py,
|
||||||
table::Table,
|
table::Table,
|
||||||
};
|
};
|
||||||
use arrow::{datatypes::Schema, ffi_stream::ArrowArrayStreamReader, pyarrow::FromPyArrow};
|
use arrow::{
|
||||||
|
datatypes::Schema,
|
||||||
|
ffi_stream::ArrowArrayStreamReader,
|
||||||
|
pyarrow::{FromPyArrow, ToPyArrow},
|
||||||
|
};
|
||||||
use lancedb::{
|
use lancedb::{
|
||||||
connection::Connection as LanceConnection,
|
connection::Connection as LanceConnection,
|
||||||
connection::NamespaceClientPushdownOperation,
|
connection::NamespaceClientPushdownOperation,
|
||||||
@@ -24,7 +28,7 @@ use pyo3::{
|
|||||||
Bound, FromPyObject, Py, PyAny, PyRef, PyResult, Python,
|
Bound, FromPyObject, Py, PyAny, PyRef, PyResult, Python,
|
||||||
exceptions::{PyRuntimeError, PyValueError},
|
exceptions::{PyRuntimeError, PyValueError},
|
||||||
pyclass, pyfunction, pymethods,
|
pyclass, pyfunction, pymethods,
|
||||||
types::{PyDict, PyDictMethods},
|
types::{PyDict, PyDictMethods, PyList, PyListMethods},
|
||||||
};
|
};
|
||||||
|
|
||||||
#[pyclass]
|
#[pyclass]
|
||||||
@@ -536,6 +540,55 @@ impl Connection {
|
|||||||
})
|
})
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn job(&self, job_id: String) -> PyResult<crate::job::Job> {
|
||||||
|
let inner = self.get_inner()?.clone();
|
||||||
|
Ok(crate::job::Job::new(inner.job(job_id).infer_error()?))
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn list_jobs(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
|
||||||
|
let inner = self_.get_inner()?.clone();
|
||||||
|
future_into_py(self_.py(), async move {
|
||||||
|
let jobs = inner.list_jobs().await.infer_error()?;
|
||||||
|
Ok(jobs
|
||||||
|
.into_iter()
|
||||||
|
.map(crate::job::JobInfo::from)
|
||||||
|
.collect::<Vec<_>>())
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn get_job(self_: PyRef<'_, Self>, job_id: String) -> PyResult<Bound<'_, PyAny>> {
|
||||||
|
let inner = self_.get_inner()?.clone();
|
||||||
|
future_into_py(self_.py(), async move {
|
||||||
|
let description = inner.get_job(&job_id).await.infer_error()?;
|
||||||
|
Ok(description.map(crate::job::JobDescription::from))
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn cancel_job(self_: PyRef<'_, Self>, job_id: String) -> PyResult<Bound<'_, PyAny>> {
|
||||||
|
let inner = self_.get_inner()?.clone();
|
||||||
|
future_into_py(self_.py(), async move {
|
||||||
|
inner.cancel_job(&job_id).await.infer_error()
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
#[pyo3(signature = (job_id=None))]
|
||||||
|
pub fn job_history(
|
||||||
|
self_: PyRef<'_, Self>,
|
||||||
|
job_id: Option<String>,
|
||||||
|
) -> PyResult<Bound<'_, PyAny>> {
|
||||||
|
let inner = self_.get_inner()?.clone();
|
||||||
|
future_into_py(self_.py(), async move {
|
||||||
|
let batches = inner.job_history(job_id.as_deref()).await.infer_error()?;
|
||||||
|
Python::attach(|py| {
|
||||||
|
let list = PyList::empty(py);
|
||||||
|
for batch in batches {
|
||||||
|
list.append(batch.to_pyarrow(py)?)?;
|
||||||
|
}
|
||||||
|
Ok(list.unbind())
|
||||||
|
})
|
||||||
|
})
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[pyfunction]
|
#[pyfunction]
|
||||||
|
|||||||
@@ -102,6 +102,18 @@ impl<T> PythonErrorExt<T> for std::result::Result<T, LanceError> {
|
|||||||
err.setattr(intern!(py, "__cause__"), cause_err)?;
|
err.setattr(intern!(py, "__cause__"), cause_err)?;
|
||||||
Err(PyErr::from_value(err))
|
Err(PyErr::from_value(err))
|
||||||
}),
|
}),
|
||||||
|
LanceError::JobFailed { .. } => Python::attach(|py| {
|
||||||
|
let cls = py
|
||||||
|
.import(intern!(py, "lancedb.exceptions"))?
|
||||||
|
.getattr(intern!(py, "JobFailedError"))?;
|
||||||
|
Err(PyErr::from_value(cls.call1((err.to_string(),))?))
|
||||||
|
}),
|
||||||
|
LanceError::JobCancelled { .. } => Python::attach(|py| {
|
||||||
|
let cls = py
|
||||||
|
.import(intern!(py, "lancedb.exceptions"))?
|
||||||
|
.getattr(intern!(py, "JobCancelledError"))?;
|
||||||
|
Err(PyErr::from_value(cls.call1((err.to_string(),))?))
|
||||||
|
}),
|
||||||
_ => self.runtime_error(),
|
_ => self.runtime_error(),
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,145 @@
|
|||||||
|
// SPDX-License-Identifier: Apache-2.0
|
||||||
|
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
use crate::runtime::future_into_py;
|
||||||
|
use pyo3::{Bound, PyAny, PyRef, PyResult, pyclass, pymethods};
|
||||||
|
|
||||||
|
use crate::error::PythonErrorExt;
|
||||||
|
|
||||||
|
#[pyclass]
|
||||||
|
pub struct Job {
|
||||||
|
inner: Arc<lancedb::Job>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Job {
|
||||||
|
pub(crate) fn new(inner: lancedb::Job) -> Self {
|
||||||
|
Self {
|
||||||
|
inner: Arc::new(inner),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[pymethods]
|
||||||
|
impl Job {
|
||||||
|
#[getter]
|
||||||
|
pub fn id(&self) -> Option<String> {
|
||||||
|
self.inner.id().map(str::to_string)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn status(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
|
||||||
|
let inner = self_.inner.clone();
|
||||||
|
future_into_py(
|
||||||
|
self_.py(),
|
||||||
|
async move { inner.status().await.infer_error() },
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn wait(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
|
||||||
|
let inner = self_.inner.clone();
|
||||||
|
future_into_py(self_.py(), async move {
|
||||||
|
inner.wait().await.infer_error()?;
|
||||||
|
Ok(())
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn cancel(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
|
||||||
|
let inner = self_.inner.clone();
|
||||||
|
future_into_py(self_.py(), async move {
|
||||||
|
inner.cancel().await.infer_error()?;
|
||||||
|
Ok(())
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A row from `Connection.list_jobs`: one server-side job.
|
||||||
|
#[pyclass(get_all, skip_from_py_object)]
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct JobInfo {
|
||||||
|
job_id: String,
|
||||||
|
table: String,
|
||||||
|
job_type: String,
|
||||||
|
state: String,
|
||||||
|
created_at_millis: i64,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[pymethods]
|
||||||
|
impl JobInfo {
|
||||||
|
fn __repr__(&self) -> String {
|
||||||
|
format!(
|
||||||
|
"JobInfo(job_id={:?}, table={:?}, job_type={:?}, state={:?}, created_at_millis={})",
|
||||||
|
self.job_id, self.table, self.job_type, self.state, self.created_at_millis
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<lancedb::database::JobInfo> for JobInfo {
|
||||||
|
fn from(info: lancedb::database::JobInfo) -> Self {
|
||||||
|
Self {
|
||||||
|
job_id: info.job_id,
|
||||||
|
table: info.table,
|
||||||
|
job_type: info.job_type,
|
||||||
|
state: info.state,
|
||||||
|
created_at_millis: info.created_at_millis,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The server's account of why a job failed.
|
||||||
|
#[pyclass(get_all, skip_from_py_object)]
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct JobFailureInfo {
|
||||||
|
phase: Option<String>,
|
||||||
|
message: Option<String>,
|
||||||
|
retryable: Option<bool>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[pymethods]
|
||||||
|
impl JobFailureInfo {
|
||||||
|
fn __repr__(&self) -> String {
|
||||||
|
format!(
|
||||||
|
"JobFailureInfo(phase={:?}, message={:?}, retryable={:?})",
|
||||||
|
self.phase, self.message, self.retryable
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A described job from `Connection.get_job`.
|
||||||
|
#[pyclass(get_all, skip_from_py_object)]
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct JobDescription {
|
||||||
|
job_id: String,
|
||||||
|
job_type: String,
|
||||||
|
state: String,
|
||||||
|
creation_ms: i64,
|
||||||
|
spec_json: Option<String>,
|
||||||
|
failure: Option<JobFailureInfo>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[pymethods]
|
||||||
|
impl JobDescription {
|
||||||
|
fn __repr__(&self) -> String {
|
||||||
|
format!(
|
||||||
|
"JobDescription(job_id={:?}, job_type={:?}, state={:?}, creation_ms={})",
|
||||||
|
self.job_id, self.job_type, self.state, self.creation_ms
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<lancedb::database::JobDescription> for JobDescription {
|
||||||
|
fn from(description: lancedb::database::JobDescription) -> Self {
|
||||||
|
Self {
|
||||||
|
job_id: description.job_id,
|
||||||
|
job_type: description.job_type,
|
||||||
|
state: description.state,
|
||||||
|
creation_ms: description.creation_ms,
|
||||||
|
spec_json: (!description.spec.is_null()).then(|| description.spec.to_string()),
|
||||||
|
failure: description.failure.map(|failure| JobFailureInfo {
|
||||||
|
phase: failure.phase,
|
||||||
|
message: failure.message,
|
||||||
|
retryable: failure.retryable,
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -25,6 +25,7 @@ pub mod error;
|
|||||||
pub mod expr;
|
pub mod expr;
|
||||||
pub mod header;
|
pub mod header;
|
||||||
pub mod index;
|
pub mod index;
|
||||||
|
pub mod job;
|
||||||
pub mod namespace;
|
pub mod namespace;
|
||||||
pub mod oauth;
|
pub mod oauth;
|
||||||
pub mod otel;
|
pub mod otel;
|
||||||
@@ -44,6 +45,10 @@ pub fn _lancedb(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> {
|
|||||||
m.add_class::<Connection>()?;
|
m.add_class::<Connection>()?;
|
||||||
m.add_class::<Session>()?;
|
m.add_class::<Session>()?;
|
||||||
m.add_class::<Table>()?;
|
m.add_class::<Table>()?;
|
||||||
|
m.add_class::<crate::job::Job>()?;
|
||||||
|
m.add_class::<crate::job::JobInfo>()?;
|
||||||
|
m.add_class::<crate::job::JobDescription>()?;
|
||||||
|
m.add_class::<crate::job::JobFailureInfo>()?;
|
||||||
m.add_class::<PyBlobFile>()?;
|
m.add_class::<PyBlobFile>()?;
|
||||||
m.add_class::<IndexConfig>()?;
|
m.add_class::<IndexConfig>()?;
|
||||||
m.add_class::<Query>()?;
|
m.add_class::<Query>()?;
|
||||||
|
|||||||
+209
-29
@@ -28,11 +28,72 @@ use pyo3::{
|
|||||||
Bound, FromPyObject, Py, PyAny, PyRef, PyResult, Python,
|
Bound, FromPyObject, Py, PyAny, PyRef, PyResult, Python,
|
||||||
exceptions::{PyRuntimeError, PyValueError},
|
exceptions::{PyRuntimeError, PyValueError},
|
||||||
pyclass, pyfunction, pymethods,
|
pyclass, pyfunction, pymethods,
|
||||||
types::{IntoPyDict, PyAnyMethods, PyBytes, PyDict, PyDictMethods},
|
types::{IntoPyDict, PyAnyMethods, PyBytes, PyDict, PyDictMethods, PyList, PyListMethods},
|
||||||
};
|
};
|
||||||
|
|
||||||
mod scannable;
|
mod scannable;
|
||||||
|
|
||||||
|
/// Convert `LsmStats` to a Python dict, preserving the per-bucket list.
|
||||||
|
///
|
||||||
|
/// Deliberately not flattened to a table-level summary: a table is N
|
||||||
|
/// buckets on one node, and the per-bucket detail is the reason the
|
||||||
|
/// endpoint exists — flattening hides the single hot bucket someone opened
|
||||||
|
/// it to find.
|
||||||
|
fn lsm_stats_to_py(py: Python<'_>, stats: &lancedb::table::LsmStats) -> PyResult<Py<PyDict>> {
|
||||||
|
let out = PyDict::new(py);
|
||||||
|
let buckets = PyList::empty(py);
|
||||||
|
for b in &stats.buckets {
|
||||||
|
let e = PyDict::new(py);
|
||||||
|
e.set_item("shard_id", &b.shard_id)?;
|
||||||
|
e.set_item("status", &b.status)?;
|
||||||
|
e.set_item("writer_epoch", b.writer_epoch)?;
|
||||||
|
e.set_item("manifest_version", b.manifest_version)?;
|
||||||
|
e.set_item("current_generation", b.current_generation)?;
|
||||||
|
e.set_item(
|
||||||
|
"replay_after_wal_entry_position",
|
||||||
|
b.replay_after_wal_entry_position,
|
||||||
|
)?;
|
||||||
|
e.set_item(
|
||||||
|
"wal_entry_position_last_seen",
|
||||||
|
b.wal_entry_position_last_seen,
|
||||||
|
)?;
|
||||||
|
|
||||||
|
let generations = PyList::empty(py);
|
||||||
|
for g in &b.generations {
|
||||||
|
let ge = PyDict::new(py);
|
||||||
|
ge.set_item("generation", g.generation)?;
|
||||||
|
ge.set_item("bytes", g.bytes)?;
|
||||||
|
ge.set_item("rows", g.rows)?;
|
||||||
|
generations.append(ge)?;
|
||||||
|
}
|
||||||
|
e.set_item("generations", generations)?;
|
||||||
|
e.set_item("compacting", b.compacting)?;
|
||||||
|
|
||||||
|
e.set_item(
|
||||||
|
"memtables",
|
||||||
|
b.memtables
|
||||||
|
.as_ref()
|
||||||
|
.map(|ms| {
|
||||||
|
let l = PyList::empty(py);
|
||||||
|
for m in ms {
|
||||||
|
let d = PyDict::new(py);
|
||||||
|
d.set_item("generation", m.generation)?;
|
||||||
|
d.set_item("rows", m.rows)?;
|
||||||
|
d.set_item("bytes", m.bytes)?;
|
||||||
|
d.set_item("batches", m.batches)?;
|
||||||
|
d.set_item("indexes", m.indexes.clone())?;
|
||||||
|
l.append(d)?;
|
||||||
|
}
|
||||||
|
PyResult::Ok(l.unbind())
|
||||||
|
})
|
||||||
|
.transpose()?,
|
||||||
|
)?;
|
||||||
|
buckets.append(e)?;
|
||||||
|
}
|
||||||
|
out.set_item("buckets", buckets)?;
|
||||||
|
Ok(out.unbind())
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(FromPyObject)]
|
#[derive(FromPyObject)]
|
||||||
enum PredicateArg {
|
enum PredicateArg {
|
||||||
Expr(PyExpr),
|
Expr(PyExpr),
|
||||||
@@ -185,12 +246,22 @@ impl From<lancedb::table::MergeResult> for MergeResult {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Render for `__repr__`, so the default reads as Python's `None` rather than
|
||||||
|
/// Rust's `Some([..])`.
|
||||||
|
fn fmt_maintained(maintained: &Option<Vec<String>>) -> String {
|
||||||
|
match maintained {
|
||||||
|
Some(names) => format!("{:?}", names),
|
||||||
|
None => "None".to_string(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Specification selecting Lance's MemWAL LSM-style write path for
|
/// Specification selecting Lance's MemWAL LSM-style write path for
|
||||||
/// `merge_insert`.
|
/// `merge_insert`.
|
||||||
///
|
///
|
||||||
/// Constructed via the `bucket(...)`, `identity(...)`, or `unsharded()`
|
/// Constructed via the `bucket(...)`, `identity(...)`, or `unsharded()`
|
||||||
/// classmethods, then optionally chain `with_maintained_indexes(...)` and
|
/// classmethods, then optionally chain `with_maintained_indexes(...)` and
|
||||||
/// `with_writer_config_defaults(...)`.
|
/// `with_writer_config_defaults(...)`. A fresh spec maintains every index the
|
||||||
|
/// MemWAL supports, resolved on install.
|
||||||
#[pyclass(from_py_object)]
|
#[pyclass(from_py_object)]
|
||||||
#[derive(Clone, Debug)]
|
#[derive(Clone, Debug)]
|
||||||
pub struct LsmWriteSpec {
|
pub struct LsmWriteSpec {
|
||||||
@@ -230,11 +301,11 @@ impl LsmWriteSpec {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Replace the list of indexes the MemWAL should keep up to date as
|
/// Set which indexes the MemWAL maintains. `None` (the default)
|
||||||
/// rows are appended. Each name must reference an index that
|
/// resolves every supported index on install; a list is verbatim,
|
||||||
/// already exists on the table at the time `set_lsm_write_spec`
|
/// and an empty list maintains nothing.
|
||||||
/// is called.
|
#[pyo3(signature = (indexes))]
|
||||||
pub fn with_maintained_indexes(&self, indexes: Vec<String>) -> Self {
|
pub fn with_maintained_indexes(&self, indexes: Option<Vec<String>>) -> Self {
|
||||||
Self {
|
Self {
|
||||||
inner: self.inner.clone().with_maintained_indexes(indexes),
|
inner: self.inner.clone().with_maintained_indexes(indexes),
|
||||||
}
|
}
|
||||||
@@ -256,23 +327,29 @@ impl LsmWriteSpec {
|
|||||||
maintained_indexes,
|
maintained_indexes,
|
||||||
writer_config_defaults,
|
writer_config_defaults,
|
||||||
} => format!(
|
} => format!(
|
||||||
"LsmWriteSpec.bucket(column={:?}, num_buckets={}, maintained_indexes={:?}, writer_config_defaults={:?})",
|
"LsmWriteSpec.bucket(column={:?}, num_buckets={}, maintained_indexes={}, writer_config_defaults={:?})",
|
||||||
column, num_buckets, maintained_indexes, writer_config_defaults,
|
column,
|
||||||
|
num_buckets,
|
||||||
|
fmt_maintained(maintained_indexes),
|
||||||
|
writer_config_defaults,
|
||||||
),
|
),
|
||||||
lancedb::table::LsmWriteSpec::Identity {
|
lancedb::table::LsmWriteSpec::Identity {
|
||||||
column,
|
column,
|
||||||
maintained_indexes,
|
maintained_indexes,
|
||||||
writer_config_defaults,
|
writer_config_defaults,
|
||||||
} => format!(
|
} => format!(
|
||||||
"LsmWriteSpec.identity(column={:?}, maintained_indexes={:?}, writer_config_defaults={:?})",
|
"LsmWriteSpec.identity(column={:?}, maintained_indexes={}, writer_config_defaults={:?})",
|
||||||
column, maintained_indexes, writer_config_defaults,
|
column,
|
||||||
|
fmt_maintained(maintained_indexes),
|
||||||
|
writer_config_defaults,
|
||||||
),
|
),
|
||||||
lancedb::table::LsmWriteSpec::Unsharded {
|
lancedb::table::LsmWriteSpec::Unsharded {
|
||||||
maintained_indexes,
|
maintained_indexes,
|
||||||
writer_config_defaults,
|
writer_config_defaults,
|
||||||
} => format!(
|
} => format!(
|
||||||
"LsmWriteSpec.unsharded(maintained_indexes={:?}, writer_config_defaults={:?})",
|
"LsmWriteSpec.unsharded(maintained_indexes={}, writer_config_defaults={:?})",
|
||||||
maintained_indexes, writer_config_defaults,
|
fmt_maintained(maintained_indexes),
|
||||||
|
writer_config_defaults,
|
||||||
),
|
),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -307,10 +384,10 @@ impl LsmWriteSpec {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Names of indexes the MemWAL should keep up to date during writes.
|
/// Indexes the MemWAL keeps up to date, or `None` for every supported one.
|
||||||
#[getter]
|
#[getter]
|
||||||
pub fn maintained_indexes(&self) -> Vec<String> {
|
pub fn maintained_indexes(&self) -> Option<Vec<String>> {
|
||||||
self.inner.maintained_indexes().to_vec()
|
self.inner.maintained_indexes().map(<[String]>::to_vec)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Default `ShardWriter` configuration recorded by this spec.
|
/// Default `ShardWriter` configuration recorded by this spec.
|
||||||
@@ -426,9 +503,11 @@ pub struct PyBlobFile {
|
|||||||
impl PyBlobFile {
|
impl PyBlobFile {
|
||||||
fn read_bytes(self_: PyRef<'_, Self>) -> PyResult<Py<PyBytes>> {
|
fn read_bytes(self_: PyRef<'_, Self>) -> PyResult<Py<PyBytes>> {
|
||||||
let inner = self_.inner.clone();
|
let inner = self_.inner.clone();
|
||||||
let bytes = block_on(async move { inner.read().await })
|
let py = self_.py();
|
||||||
|
let bytes = py
|
||||||
|
.detach(move || block_on(async move { inner.read().await }))
|
||||||
.map_err(|e| PyRuntimeError::new_err(format!("blob read failed: {e}")))?;
|
.map_err(|e| PyRuntimeError::new_err(format!("blob read failed: {e}")))?;
|
||||||
Ok(PyBytes::new(self_.py(), bytes.as_ref()).unbind())
|
Ok(PyBytes::new(py, bytes.as_ref()).unbind())
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn read(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
|
pub fn read(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
|
||||||
@@ -444,24 +523,32 @@ impl PyBlobFile {
|
|||||||
|
|
||||||
fn close(self_: PyRef<'_, Self>) -> PyResult<()> {
|
fn close(self_: PyRef<'_, Self>) -> PyResult<()> {
|
||||||
let inner = self_.inner.clone();
|
let inner = self_.inner.clone();
|
||||||
block_on(async move { inner.close().await })
|
self_
|
||||||
|
.py()
|
||||||
|
.detach(move || block_on(async move { inner.close().await }))
|
||||||
.map_err(|e| PyRuntimeError::new_err(format!("blob close failed: {e}")))
|
.map_err(|e| PyRuntimeError::new_err(format!("blob close failed: {e}")))
|
||||||
}
|
}
|
||||||
|
|
||||||
fn is_closed(self_: PyRef<'_, Self>) -> bool {
|
fn is_closed(self_: PyRef<'_, Self>) -> bool {
|
||||||
let inner = self_.inner.clone();
|
let inner = self_.inner.clone();
|
||||||
block_on(async move { inner.is_closed().await })
|
self_
|
||||||
|
.py()
|
||||||
|
.detach(move || block_on(async move { inner.is_closed().await }))
|
||||||
}
|
}
|
||||||
|
|
||||||
fn seek(self_: PyRef<'_, Self>, position: u64) -> PyResult<()> {
|
fn seek(self_: PyRef<'_, Self>, position: u64) -> PyResult<()> {
|
||||||
let inner = self_.inner.clone();
|
let inner = self_.inner.clone();
|
||||||
block_on(async move { inner.seek(position).await })
|
self_
|
||||||
|
.py()
|
||||||
|
.detach(move || block_on(async move { inner.seek(position).await }))
|
||||||
.map_err(|e| PyRuntimeError::new_err(format!("blob seek failed: {e}")))
|
.map_err(|e| PyRuntimeError::new_err(format!("blob seek failed: {e}")))
|
||||||
}
|
}
|
||||||
|
|
||||||
fn tell(self_: PyRef<'_, Self>) -> PyResult<u64> {
|
fn tell(self_: PyRef<'_, Self>) -> PyResult<u64> {
|
||||||
let inner = self_.inner.clone();
|
let inner = self_.inner.clone();
|
||||||
block_on(async move { inner.tell().await })
|
self_
|
||||||
|
.py()
|
||||||
|
.detach(move || block_on(async move { inner.tell().await }))
|
||||||
.map_err(|e| PyRuntimeError::new_err(format!("blob tell failed: {e}")))
|
.map_err(|e| PyRuntimeError::new_err(format!("blob tell failed: {e}")))
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -475,16 +562,20 @@ impl PyBlobFile {
|
|||||||
.checked_add(length as u64)
|
.checked_add(length as u64)
|
||||||
.ok_or_else(|| PyValueError::new_err("offset + length overflowed"))?;
|
.ok_or_else(|| PyValueError::new_err("offset + length overflowed"))?;
|
||||||
let inner = self_.inner.clone();
|
let inner = self_.inner.clone();
|
||||||
let bytes = block_on(async move { inner.read_range(offset..end).await })
|
let py = self_.py();
|
||||||
|
let bytes = py
|
||||||
|
.detach(move || block_on(async move { inner.read_range(offset..end).await }))
|
||||||
.map_err(|e| PyRuntimeError::new_err(format!("blob read_range failed: {e}")))?;
|
.map_err(|e| PyRuntimeError::new_err(format!("blob read_range failed: {e}")))?;
|
||||||
Ok(PyBytes::new(self_.py(), bytes.as_ref()).unbind())
|
Ok(PyBytes::new(py, bytes.as_ref()).unbind())
|
||||||
}
|
}
|
||||||
|
|
||||||
fn read_up_to(self_: PyRef<'_, Self>, length: usize) -> PyResult<Py<PyBytes>> {
|
fn read_up_to(self_: PyRef<'_, Self>, length: usize) -> PyResult<Py<PyBytes>> {
|
||||||
let inner = self_.inner.clone();
|
let inner = self_.inner.clone();
|
||||||
let bytes = block_on(async move { inner.read_up_to(length).await })
|
let py = self_.py();
|
||||||
.map_err(|e| PyRuntimeError::new_err(format!("blob read failed: {e}")))?;
|
let bytes = py
|
||||||
Ok(PyBytes::new(self_.py(), bytes.as_ref()).unbind())
|
.detach(move || block_on(async move { inner.read_up_to(length).await }))
|
||||||
|
.map_err(|e| PyRuntimeError::new_err(format!("blob read_up_to failed: {e}")))?;
|
||||||
|
Ok(PyBytes::new(py, bytes.as_ref()).unbind())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -731,6 +822,9 @@ impl Table {
|
|||||||
|
|
||||||
#[allow(private_interfaces)]
|
#[allow(private_interfaces)]
|
||||||
pub fn delete(self_: PyRef<'_, Self>, condition: PredicateArg) -> PyResult<Bound<'_, PyAny>> {
|
pub fn delete(self_: PyRef<'_, Self>, condition: PredicateArg) -> PyResult<Bound<'_, PyAny>> {
|
||||||
|
// Do not hold the Python borrow across the await. The cloned Rust table
|
||||||
|
// handle is thread-safe and allows deletes on the same Python table to
|
||||||
|
// run concurrently without PyO3 reporting "Already borrowed".
|
||||||
let inner = self_.inner_ref()?.clone();
|
let inner = self_.inner_ref()?.clone();
|
||||||
future_into_py(self_.py(), async move {
|
future_into_py(self_.py(), async move {
|
||||||
let result = match &condition {
|
let result = match &condition {
|
||||||
@@ -805,6 +899,37 @@ impl Table {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[pyo3(signature = (column, index=None, replace=None, wait_timeout=None, *, name=None, train=None))]
|
||||||
|
pub fn create_index_async<'a>(
|
||||||
|
self_: PyRef<'a, Self>,
|
||||||
|
column: String,
|
||||||
|
index: Option<Bound<'_, PyAny>>,
|
||||||
|
replace: Option<bool>,
|
||||||
|
wait_timeout: Option<Bound<'_, PyAny>>,
|
||||||
|
name: Option<String>,
|
||||||
|
train: Option<bool>,
|
||||||
|
) -> PyResult<Bound<'a, PyAny>> {
|
||||||
|
let index = extract_index_params(&index)?;
|
||||||
|
let timeout = wait_timeout.map(|t| t.extract::<std::time::Duration>().unwrap());
|
||||||
|
let mut op = self_
|
||||||
|
.inner_ref()?
|
||||||
|
.create_index_with_timeout(&[column], index, timeout);
|
||||||
|
if let Some(replace) = replace {
|
||||||
|
op = op.replace(replace);
|
||||||
|
}
|
||||||
|
if let Some(name) = name {
|
||||||
|
op = op.name(name);
|
||||||
|
}
|
||||||
|
if let Some(train) = train {
|
||||||
|
op = op.train(train);
|
||||||
|
}
|
||||||
|
|
||||||
|
future_into_py(self_.py(), async move {
|
||||||
|
let job = op.execute_async().await.infer_error()?;
|
||||||
|
Ok(crate::job::Job::new(job))
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
pub fn drop_index(self_: PyRef<'_, Self>, index_name: String) -> PyResult<Bound<'_, PyAny>> {
|
pub fn drop_index(self_: PyRef<'_, Self>, index_name: String) -> PyResult<Bound<'_, PyAny>> {
|
||||||
let inner = self_.inner_ref()?.clone();
|
let inner = self_.inner_ref()?.clone();
|
||||||
future_into_py(self_.py(), async move {
|
future_into_py(self_.py(), async move {
|
||||||
@@ -1291,6 +1416,51 @@ impl Table {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Converge the table's LSM write path into its base table.
|
||||||
|
///
|
||||||
|
/// Best-effort: with writes flowing, new rows may land after the last
|
||||||
|
/// pass. Errors if the table stops making progress.
|
||||||
|
pub fn checkpoint_lsm(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
|
||||||
|
let inner = self_.inner_ref()?.clone();
|
||||||
|
future_into_py(self_.py(), async move {
|
||||||
|
inner.checkpoint_lsm().await.infer_error()
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Seal every bucket's active memtable into L0.
|
||||||
|
pub fn flush_lsm(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
|
||||||
|
let inner = self_.inner_ref()?.clone();
|
||||||
|
future_into_py(
|
||||||
|
self_.py(),
|
||||||
|
async move { inner.flush_lsm().await.infer_error() },
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Trigger a background L0 → base pass per bucket. Returns once the
|
||||||
|
/// passes are dispatched, not once they finish — watch `get_lsm_stats`.
|
||||||
|
pub fn compact_lsm(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
|
||||||
|
let inner = self_.inner_ref()?.clone();
|
||||||
|
future_into_py(self_.py(), async move {
|
||||||
|
inner.compact_lsm().await.infer_error()
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Live LSM state, or `None` when the LSM write path is not enabled.
|
||||||
|
#[pyo3(signature = (include_generation_rows=false))]
|
||||||
|
pub fn get_lsm_stats(
|
||||||
|
self_: PyRef<'_, Self>,
|
||||||
|
include_generation_rows: bool,
|
||||||
|
) -> PyResult<Bound<'_, PyAny>> {
|
||||||
|
let inner = self_.inner_ref()?.clone();
|
||||||
|
future_into_py(self_.py(), async move {
|
||||||
|
let stats = inner
|
||||||
|
.get_lsm_stats(include_generation_rows)
|
||||||
|
.await
|
||||||
|
.infer_error()?;
|
||||||
|
Python::attach(|py| stats.map(|s| lsm_stats_to_py(py, &s)).transpose())
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
pub fn close_lsm_writers(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
|
pub fn close_lsm_writers(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
|
||||||
let inner = self_.inner_ref()?.clone();
|
let inner = self_.inner_ref()?.clone();
|
||||||
future_into_py(self_.py(), async move {
|
future_into_py(self_.py(), async move {
|
||||||
@@ -1330,7 +1500,12 @@ impl Table {
|
|||||||
|
|
||||||
let inner = self_.inner_ref()?.clone();
|
let inner = self_.inner_ref()?.clone();
|
||||||
future_into_py(self_.py(), async move {
|
future_into_py(self_.py(), async move {
|
||||||
let result = inner.add_columns(definitions, None).await.infer_error()?;
|
let result = inner
|
||||||
|
.add_columns()
|
||||||
|
.transform(definitions)
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.infer_error()?;
|
||||||
Ok(AddColumnsResult::from(result))
|
Ok(AddColumnsResult::from(result))
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
@@ -1344,7 +1519,12 @@ impl Table {
|
|||||||
|
|
||||||
let inner = self_.inner_ref()?.clone();
|
let inner = self_.inner_ref()?.clone();
|
||||||
future_into_py(self_.py(), async move {
|
future_into_py(self_.py(), async move {
|
||||||
let result = inner.add_columns(transform, None).await.infer_error()?;
|
let result = inner
|
||||||
|
.add_columns()
|
||||||
|
.transform(transform)
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.infer_error()?;
|
||||||
Ok(AddColumnsResult::from(result))
|
Ok(AddColumnsResult::from(result))
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|||||||
Generated
+1169
-1067
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user