mirror of
https://github.com/lancedb/lancedb.git
synced 2026-08-27 16:38:31 +00:00
Compare commits
112 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 9b065c1cd8 | |||
| cf27f6902e | |||
| f76ee304b8 | |||
| a588208de6 | |||
| 685cb01d6d | |||
| 4ba2421254 | |||
| 09843410ec | |||
| 7adcffc2b4 | |||
| c1331e5083 | |||
| 426684cf1b | |||
| e517ba5205 | |||
| 5c1b44020a | |||
| 061a3da8b9 | |||
| 4e042af12f | |||
| 27cea03b7d | |||
| f1c4967eeb | |||
| 11c1d81638 | |||
| f6efdc9e9f | |||
| cdebea118d | |||
| 76942306b7 | |||
| d742b174c4 | |||
| a075aa62f8 | |||
| 040a4120c8 | |||
| 928c3dde2d | |||
| 980818df26 | |||
| c429863122 | |||
| fc0d917d32 | |||
| def869bb78 | |||
| 9e4d8bd1c7 | |||
| 4148dfef72 | |||
| 0ac70a8b9f | |||
| 91c5f344d2 | |||
| ffd35c1a8f | |||
| 790d0c684c | |||
| 251f194696 | |||
| 4b7325bd74 | |||
| 1d75638dea | |||
| 031c3585a8 | |||
| 6fb976cf89 | |||
| a615306f39 | |||
| 920fc0e455 | |||
| 5acce6782e | |||
| 12405a4077 | |||
| 36054be576 | |||
| 77a93fee76 | |||
| 7bb501839a | |||
| 5b347afd99 | |||
| 706a9c327f | |||
| be290447d9 | |||
| 79ba076429 | |||
| ec21e37040 | |||
| 6ba80a960c | |||
| 11f24b1df4 | |||
| 2ba7407dc3 | |||
| 607e556927 | |||
| 564e5d0d56 | |||
| dd5cb4d805 | |||
| dbc3687c7b | |||
| ec80acb668 | |||
| fc44535cee | |||
| 4048150fdd | |||
| 2922c171f7 | |||
| c5f9efefe9 | |||
| f4c668e244 | |||
| b1cfe6edb1 | |||
| 001237c7a4 | |||
| 369b10a377 | |||
| 1c3cd1d918 | |||
| 9707966943 | |||
| 62fe413a52 | |||
| 1493ece3de | |||
| e6444ecc05 | |||
| cc0139c136 | |||
| b20696ef9c | |||
| 772bdeced8 | |||
| c1a3fa7f51 | |||
| 0ba82873c5 | |||
| 3af51541a0 | |||
| 2c06a48bd8 | |||
| ac8b28c010 | |||
| 173f889d2a | |||
| 03b52e5877 | |||
| 798e5364fb | |||
| f1f34dfdd3 | |||
| 123c921c4f | |||
| 99a68db78c | |||
| 9e73d440a3 | |||
| 3956d9dbfa | |||
| 16e1967efc | |||
| 27dd92c67e | |||
| 9e2e711c7a | |||
| c3176a47ce | |||
| 7357d63e87 | |||
| 624a75edf7 | |||
| c7ea91f3ea | |||
| 8e24dd3828 | |||
| f79dc017c4 | |||
| e6ae93f52a | |||
| 3dd9c598e9 | |||
| 9e26bf3fba | |||
| 93354baf34 | |||
| 05602ec7d5 | |||
| e3b472c212 | |||
| a6418b6cb9 | |||
| dd2b11eda2 | |||
| 5a1015ba72 | |||
| 48945d0658 | |||
| 77208fd464 | |||
| b505dc1315 | |||
| 7dfdfe6401 | |||
| 4dc2d9a0f2 | |||
| 1ad6ce3a4e |
@@ -5,7 +5,3 @@ This directory contains repo-scoped code agent skills for the LanceDB project.
|
|||||||
Each skill is a folder that contains a required `SKILL.md` and optional bundled resources.
|
Each skill is a folder that contains a required `SKILL.md` and optional bundled resources.
|
||||||
|
|
||||||
Codex discovers skills from `.agents/skills` in the current working directory and parent directories.
|
Codex discovers skills from `.agents/skills` in the current working directory and parent directories.
|
||||||
|
|
||||||
The `lancedb` skill lives in the `plugins/lancedb` plugin (see `plugins/lancedb/skills/lancedb`)
|
|
||||||
so it can be installed via the plugin marketplaces (`.claude-plugin/marketplace.json` and
|
|
||||||
`.agents/plugins/marketplace.json`); the `lancedb` entry here is a symlink into that plugin.
|
|
||||||
|
|||||||
@@ -1 +0,0 @@
|
|||||||
../../plugins/lancedb/skills/lancedb
|
|
||||||
+1
-1
@@ -1,5 +1,5 @@
|
|||||||
[tool.bumpversion]
|
[tool.bumpversion]
|
||||||
current_version = "0.37.1-beta.0"
|
current_version = "0.38.0-beta.3"
|
||||||
parse = """(?x)
|
parse = """(?x)
|
||||||
(?P<major>0|[1-9]\\d*)\\.
|
(?P<major>0|[1-9]\\d*)\\.
|
||||||
(?P<minor>0|[1-9]\\d*)\\.
|
(?P<minor>0|[1-9]\\d*)\\.
|
||||||
|
|||||||
@@ -9,6 +9,18 @@ debug = true
|
|||||||
codegen-units = 16
|
codegen-units = 16
|
||||||
lto = "thin"
|
lto = "thin"
|
||||||
|
|
||||||
|
[profile.release-no-lto]
|
||||||
|
inherits = "release"
|
||||||
|
debug = true
|
||||||
|
lto = false
|
||||||
|
# Prioritize compile time when LTO is not relevant to the measurement.
|
||||||
|
codegen-units = 16
|
||||||
|
|
||||||
|
[profile.bench]
|
||||||
|
inherits = "release"
|
||||||
|
lto = "thin"
|
||||||
|
codegen-units = 16
|
||||||
|
|
||||||
[target.'cfg(all())']
|
[target.'cfg(all())']
|
||||||
rustflags = [
|
rustflags = [
|
||||||
"-Wclippy::all",
|
"-Wclippy::all",
|
||||||
|
|||||||
@@ -0,0 +1,30 @@
|
|||||||
|
name: CI scripts
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches:
|
||||||
|
- main
|
||||||
|
paths:
|
||||||
|
- ci/set_lance_version.py
|
||||||
|
- ci/tests/**
|
||||||
|
- .github/workflows/ci-scripts.yml
|
||||||
|
pull_request:
|
||||||
|
paths:
|
||||||
|
- ci/set_lance_version.py
|
||||||
|
- ci/tests/**
|
||||||
|
- .github/workflows/ci-scripts.yml
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
test:
|
||||||
|
name: Test CI scripts
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v6
|
||||||
|
- uses: actions/setup-python@v6
|
||||||
|
with:
|
||||||
|
python-version: "3.13"
|
||||||
|
- name: Run tests
|
||||||
|
run: python -m unittest discover -s ci/tests -v
|
||||||
@@ -4,14 +4,14 @@ on:
|
|||||||
workflow_call:
|
workflow_call:
|
||||||
inputs:
|
inputs:
|
||||||
tag:
|
tag:
|
||||||
description: "Tag name from Lance. If omitted, the skill will use the latest Lance release that needs an update."
|
description: "Tag name from Lance (e.g. `v7.2.0-beta.1`). If omitted, the newest release is resolved automatically — stable releases are preferred over pre-releases — and the run is skipped if it is not newer than the version currently pinned in Cargo.toml."
|
||||||
required: false
|
required: false
|
||||||
default: ""
|
default: ""
|
||||||
type: string
|
type: string
|
||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
inputs:
|
inputs:
|
||||||
tag:
|
tag:
|
||||||
description: "Tag name from Lance. Leave empty to use the latest Lance release that needs an update."
|
description: "Tag name from Lance (e.g. `v7.2.0-beta.1`). Leave empty to resolve the newest release automatically — stable releases are preferred over pre-releases — and skip the run if it is not newer than the version currently pinned in Cargo.toml."
|
||||||
required: false
|
required: false
|
||||||
default: ""
|
default: ""
|
||||||
type: string
|
type: string
|
||||||
|
|||||||
@@ -0,0 +1,243 @@
|
|||||||
|
name: Check doc links
|
||||||
|
|
||||||
|
# Checking external links is inherently noisy: third-party sites rate-limit
|
||||||
|
# automated clients, reject non-browser user agents, and go down temporarily.
|
||||||
|
# Blocking pull requests on that trades a lot of false failures for very little
|
||||||
|
# signal, so this runs on a schedule and reports findings in a single tracking
|
||||||
|
# issue instead of failing anyone's build.
|
||||||
|
on:
|
||||||
|
schedule:
|
||||||
|
- cron: "0 7 * * *"
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
# The report lives in one repository-global issue, so runs must not overlap: a
|
||||||
|
# lookup racing a create produces duplicate issues, and a healthy run closing
|
||||||
|
# the issue while a failing run only rewrites its body would leave a broken
|
||||||
|
# report closed. The group is deliberately ref-independent so that a manual
|
||||||
|
# dispatch serializes against the scheduled run.
|
||||||
|
concurrency:
|
||||||
|
group: docs-link-check
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
|
permissions: {}
|
||||||
|
|
||||||
|
env:
|
||||||
|
REPORT_TITLE: "Docs link checker report"
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
scan:
|
||||||
|
name: Scan links
|
||||||
|
runs-on: ubuntu-24.04
|
||||||
|
# lychee-action is pinned by SHA, but its wrapper downloads the lychee
|
||||||
|
# release tarball at run time without verifying a digest, and hands the
|
||||||
|
# resulting binary a GitHub token. Release assets remain replaceable, so
|
||||||
|
# that binary is confined to a job whose token can only read public
|
||||||
|
# content; everything that writes runs in the report job below.
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
outputs:
|
||||||
|
checker_outcome: ${{ steps.lychee.outcome }}
|
||||||
|
exit_code: ${{ steps.lychee.outputs.exit_code }}
|
||||||
|
status: ${{ steps.validate.outputs.status }}
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v6
|
||||||
|
with:
|
||||||
|
# workflow_dispatch can run from any ref, but the report is
|
||||||
|
# repository-global. Always measure the default branch so a manual
|
||||||
|
# run from a topic branch cannot close a report that main warrants,
|
||||||
|
# or overwrite it with branch-only findings.
|
||||||
|
ref: ${{ github.event.repository.default_branch }}
|
||||||
|
persist-credentials: false
|
||||||
|
|
||||||
|
- name: Check links
|
||||||
|
id: lychee
|
||||||
|
continue-on-error: true
|
||||||
|
uses: lycheeverse/lychee-action@e7477775783ea5526144ba13e8db5eec57747ce8 # v2.9.0
|
||||||
|
with:
|
||||||
|
# Restricted to http(s) on purpose. Much of docs/src is generated
|
||||||
|
# API reference (the js/ tree comes from `npm run docs` in nodejs)
|
||||||
|
# and the hand-written pages use mkdocstrings cross-references and
|
||||||
|
# nav-relative paths that only resolve in the site mkdocs builds,
|
||||||
|
# not in this checkout, so relative links would be reported as
|
||||||
|
# broken on every run.
|
||||||
|
args: >-
|
||||||
|
--scheme https
|
||||||
|
--scheme http
|
||||||
|
--no-progress
|
||||||
|
--max-retries 3
|
||||||
|
--timeout 20
|
||||||
|
'docs/src/**/*.md'
|
||||||
|
format: json
|
||||||
|
output: ./lychee/out.json
|
||||||
|
jobSummary: false
|
||||||
|
# The report issue, not a red workflow run, is the signal for link
|
||||||
|
# findings and checker failures alike.
|
||||||
|
fail: false
|
||||||
|
|
||||||
|
- name: Validate report
|
||||||
|
id: validate
|
||||||
|
# lychee does not reserve exit code 2 for broken links: its CLI
|
||||||
|
# parser also exits 2 on an invalid option, before any link was
|
||||||
|
# checked or any report written. Only a parseable report whose
|
||||||
|
# counts agree with a completed exit code (0 or 2) counts as a link
|
||||||
|
# verdict. Everything else becomes a checker-error report instead of
|
||||||
|
# failing the workflow. Exit 2 covers timeouts as well as errors, and a
|
||||||
|
# timed-out host is exactly the transient unavailability this report
|
||||||
|
# exists to surface, so both count as findings. Requiring total > 0
|
||||||
|
# also catches a glob that silently stopped matching any file.
|
||||||
|
if: always()
|
||||||
|
env:
|
||||||
|
CHECKER_OUTCOME: ${{ steps.lychee.outcome }}
|
||||||
|
EXIT_CODE: ${{ steps.lychee.outputs.exit_code }}
|
||||||
|
run: |
|
||||||
|
status=checker-error
|
||||||
|
if [[ "$CHECKER_OUTCOME" == success ]] &&
|
||||||
|
[[ "$EXIT_CODE" == 0 || "$EXIT_CODE" == 2 ]] &&
|
||||||
|
jq -e --argjson code "$EXIT_CODE" '
|
||||||
|
(.total > 0) and
|
||||||
|
(if $code == 0
|
||||||
|
then .errors == 0 and .timeouts == 0
|
||||||
|
and (.error_map | length == 0) and (.timeout_map | length == 0)
|
||||||
|
else (.errors + .timeouts) > 0
|
||||||
|
and ((.error_map | length) + (.timeout_map | length)) > 0
|
||||||
|
end)
|
||||||
|
' ./lychee/out.json
|
||||||
|
then
|
||||||
|
if [[ "$EXIT_CODE" == 0 ]]; then
|
||||||
|
status=healthy
|
||||||
|
else
|
||||||
|
status=findings
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
echo "status=$status" >> "$GITHUB_OUTPUT"
|
||||||
|
echo "Validated link check as $status"
|
||||||
|
|
||||||
|
- name: Upload report
|
||||||
|
if: steps.validate.outputs.status == 'findings'
|
||||||
|
uses: actions/upload-artifact@v7
|
||||||
|
with:
|
||||||
|
name: link-report
|
||||||
|
path: ./lychee/out.json
|
||||||
|
retention-days: 7
|
||||||
|
|
||||||
|
report:
|
||||||
|
name: Update report issue
|
||||||
|
needs: scan
|
||||||
|
runs-on: ubuntu-24.04
|
||||||
|
# Deliberately no checkout: this job needs the report artifact and the
|
||||||
|
# issues API, not the repository contents.
|
||||||
|
permissions:
|
||||||
|
issues: write
|
||||||
|
env:
|
||||||
|
CHECKER_OUTCOME: ${{ needs.scan.outputs.checker_outcome }}
|
||||||
|
EXIT_CODE: ${{ needs.scan.outputs.exit_code }}
|
||||||
|
STATUS: ${{ needs.scan.outputs.status }}
|
||||||
|
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
steps:
|
||||||
|
- name: Find existing report issue
|
||||||
|
id: report
|
||||||
|
# Matched on title alone, and through search rather than a listing:
|
||||||
|
# the issue action applies labels in a separate call after creating the
|
||||||
|
# issue, so a label filter misses a half-created report, and this
|
||||||
|
# repository has far more open issues than one listing page holds.
|
||||||
|
# Closed issues are included because a healthy run closes the report:
|
||||||
|
# an open-only lookup would forget that identity and the next failing
|
||||||
|
# run would open a duplicate. The oldest match stays the canonical
|
||||||
|
# report and is reopened below when a problem recurs.
|
||||||
|
run: |
|
||||||
|
match=$(gh issue list --repo "$GITHUB_REPOSITORY" --state all \
|
||||||
|
--search "in:title \"$REPORT_TITLE\" author:app/github-actions" \
|
||||||
|
--limit 50 --json number,title,state \
|
||||||
|
--jq "[.[] | select(.title == \"$REPORT_TITLE\")] | sort_by(.number) | first // empty")
|
||||||
|
echo "number=$(jq -r '.number // empty' <<<"$match")" >> "$GITHUB_OUTPUT"
|
||||||
|
echo "state=$(jq -r '.state // empty' <<<"$match")" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
|
- name: Download report
|
||||||
|
if: env.STATUS == 'findings'
|
||||||
|
uses: actions/download-artifact@v8
|
||||||
|
with:
|
||||||
|
name: link-report
|
||||||
|
path: ./lychee
|
||||||
|
|
||||||
|
- name: Compose report
|
||||||
|
if: env.STATUS == 'findings'
|
||||||
|
run: |
|
||||||
|
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
|
||||||
|
{
|
||||||
|
echo "Broken documentation links found by [\`$GITHUB_WORKFLOW\`]($run_url)."
|
||||||
|
echo
|
||||||
|
echo "This issue is rewritten by every scheduled run and closed automatically once all links resolve."
|
||||||
|
echo
|
||||||
|
echo "Entries can be false positives: some sites rate-limit or block automated clients while working fine in a browser. Confirm before editing the docs, and add persistent offenders to \`--exclude\` in \`.github/workflows/docs-link-check.yml\`."
|
||||||
|
echo
|
||||||
|
# Timeouts are reported alongside errors: entries land in
|
||||||
|
# timeout_map with a status text instead of an HTTP code.
|
||||||
|
jq -r '
|
||||||
|
"\(.errors) of \(.total) links failed, \(.timeouts) timed out.",
|
||||||
|
"",
|
||||||
|
([(.error_map | to_entries[]), (.timeout_map | to_entries[])]
|
||||||
|
| group_by(.key)[] |
|
||||||
|
"### Errors in \(.[0].key)",
|
||||||
|
"",
|
||||||
|
(map(.value[])[] | "* [\(.status.code // .status.text // "ERR")] <\(.url)> — \(.status.details // .status.text // "unknown error")"),
|
||||||
|
"")
|
||||||
|
' ./lychee/out.json
|
||||||
|
} > ./lychee/issue.md
|
||||||
|
|
||||||
|
- name: Compose checker error report
|
||||||
|
if: env.STATUS == 'checker-error'
|
||||||
|
run: |
|
||||||
|
mkdir -p ./lychee
|
||||||
|
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
|
||||||
|
{
|
||||||
|
echo "The documentation link check did not complete in [the latest run]($run_url)."
|
||||||
|
echo
|
||||||
|
echo "This issue is rewritten by every scheduled run and closed automatically once a trustworthy run finds that all links resolve."
|
||||||
|
echo
|
||||||
|
echo "The checker did not produce a trustworthy link verdict. Treat the previous result, if any, as stale until a later run completes."
|
||||||
|
echo
|
||||||
|
echo "* Action outcome: \`$CHECKER_OUTCOME\`"
|
||||||
|
echo "* Exit code: \`${EXIT_CODE:-not reported}\`"
|
||||||
|
echo "* Verdict validation: \`failed\`"
|
||||||
|
} > ./lychee/issue.md
|
||||||
|
|
||||||
|
- name: Reopen report issue
|
||||||
|
# A healthy run closes the report, and the issue action below only
|
||||||
|
# rewrites the body of whatever number it is given. Without an
|
||||||
|
# explicit reopen, a later finding or checker error would rewrite a
|
||||||
|
# closed issue. A CLOSED state implies the lookup found a canonical
|
||||||
|
# issue, so no separate emptiness check.
|
||||||
|
if: >-
|
||||||
|
env.STATUS != 'healthy' &&
|
||||||
|
steps.report.outputs.state == 'CLOSED'
|
||||||
|
env:
|
||||||
|
ISSUE_NUMBER: ${{ steps.report.outputs.number }}
|
||||||
|
run: |
|
||||||
|
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
|
||||||
|
gh issue reopen "$ISSUE_NUMBER" --repo "$GITHUB_REPOSITORY" \
|
||||||
|
--comment "The documentation link checker reported a problem again in [the latest run]($run_url)."
|
||||||
|
|
||||||
|
- name: Report link-check problem
|
||||||
|
if: env.STATUS != 'healthy'
|
||||||
|
uses: peter-evans/create-issue-from-file@fca9117c27cdc29c6c4db3b86c48e4115a786710 # v6.0.0
|
||||||
|
with:
|
||||||
|
# Empty on the first failing run, which creates the issue; afterwards
|
||||||
|
# the same issue is updated in place.
|
||||||
|
issue-number: ${{ steps.report.outputs.number }}
|
||||||
|
title: ${{ env.REPORT_TITLE }}
|
||||||
|
content-filepath: ./lychee/issue.md
|
||||||
|
labels: documentation
|
||||||
|
|
||||||
|
- name: Close report issue once links are healthy
|
||||||
|
# An OPEN state implies the lookup found a canonical issue; a report
|
||||||
|
# that is already closed needs nothing.
|
||||||
|
if: >-
|
||||||
|
env.STATUS == 'healthy' &&
|
||||||
|
steps.report.outputs.state == 'OPEN'
|
||||||
|
env:
|
||||||
|
ISSUE_NUMBER: ${{ steps.report.outputs.number }}
|
||||||
|
run: |
|
||||||
|
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
|
||||||
|
gh issue close "$ISSUE_NUMBER" --repo "$GITHUB_REPOSITORY" \
|
||||||
|
--comment "All documentation links resolved in [the latest run]($run_url)."
|
||||||
@@ -69,6 +69,16 @@ jobs:
|
|||||||
uses: actions/setup-python@v6
|
uses: actions/setup-python@v6
|
||||||
with:
|
with:
|
||||||
python-version: "3.10"
|
python-version: "3.10"
|
||||||
|
- name: Add swap for Arm fat LTO
|
||||||
|
if: matrix.config.platform == 'aarch64'
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
swap_file="$RUNNER_TEMP/lancedb-swap"
|
||||||
|
sudo fallocate --length 16G "$swap_file"
|
||||||
|
sudo chmod 600 "$swap_file"
|
||||||
|
sudo mkswap "$swap_file"
|
||||||
|
sudo swapon "$swap_file"
|
||||||
|
free -h
|
||||||
- uses: ./.github/workflows/build_linux_wheel
|
- uses: ./.github/workflows/build_linux_wheel
|
||||||
with:
|
with:
|
||||||
python-minor-version: 10
|
python-minor-version: 10
|
||||||
|
|||||||
@@ -229,7 +229,8 @@ jobs:
|
|||||||
# Make sure wheels are not included in the Rust cache
|
# Make sure wheels are not included in the Rust cache
|
||||||
- name: Delete wheels
|
- name: Delete wheels
|
||||||
run: rm -rf target/wheels
|
run: rm -rf target/wheels
|
||||||
pydantic1x:
|
min-deps:
|
||||||
|
name: "Minimum dependencies"
|
||||||
timeout-minutes: 60
|
timeout-minutes: 60
|
||||||
runs-on: "ubuntu-24.04"
|
runs-on: "ubuntu-24.04"
|
||||||
defaults:
|
defaults:
|
||||||
@@ -259,8 +260,7 @@ jobs:
|
|||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
save-if: ${{ github.ref == 'refs/heads/main' }}
|
||||||
- name: Install lancedb
|
- name: Install lancedb
|
||||||
run: |
|
run: |
|
||||||
pip install "pydantic<2"
|
pip install "pydantic==2.7.4" "pyarrow==16"
|
||||||
pip install pyarrow==16
|
|
||||||
pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests]
|
pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests]
|
||||||
- name: Run tests
|
- name: Run tests
|
||||||
run: pytest -m "not slow and not s3_test" -x -v --durations=30 python/tests
|
run: pytest -m "not slow and not s3_test" -x -v --durations=30 python/tests
|
||||||
|
|||||||
+41
-11
@@ -121,7 +121,6 @@ jobs:
|
|||||||
# Need up-to-date compilers for kernels
|
# Need up-to-date compilers for kernels
|
||||||
CC: clang-18
|
CC: clang-18
|
||||||
CXX: clang++-18
|
CXX: clang++-18
|
||||||
GH_TOKEN: ${{ secrets.SOPHON_READ_TOKEN }}
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v6
|
||||||
with:
|
with:
|
||||||
@@ -165,11 +164,40 @@ jobs:
|
|||||||
- name: Run feature tests
|
- name: Run feature tests
|
||||||
run: CARGO_ARGS="--profile ci" make -C ./lancedb feature-tests
|
run: CARGO_ARGS="--profile ci" make -C ./lancedb feature-tests
|
||||||
- name: Run examples
|
- name: Run examples
|
||||||
run: cargo run --profile ci --example simple --locked
|
run: cargo run --profile ci --all-features --example simple --locked
|
||||||
|
|
||||||
|
remote:
|
||||||
|
timeout-minutes: 30
|
||||||
|
# Running this requires access to secrets, so skip if this is a PR from a
|
||||||
|
# fork. Keep it separate from the all-features build so Cargo does not
|
||||||
|
# retain both dependency graphs in one target directory.
|
||||||
|
if: github.event_name != 'pull_request' || !github.event.pull_request.head.repo.fork
|
||||||
|
runs-on: ubuntu-2404-4x-x64
|
||||||
|
defaults:
|
||||||
|
run:
|
||||||
|
shell: bash
|
||||||
|
working-directory: rust
|
||||||
|
env:
|
||||||
|
CC: clang-18
|
||||||
|
CXX: clang++-18
|
||||||
|
GH_TOKEN: ${{ secrets.SOPHON_READ_TOKEN }}
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v6
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
lfs: true
|
||||||
|
- uses: Swatinem/rust-cache@v2
|
||||||
|
with:
|
||||||
|
# Remote tests use a different feature graph from the main Linux
|
||||||
|
# job. Cache downloads, but build into a fresh target directory.
|
||||||
|
cache-targets: false
|
||||||
|
save-if: ${{ github.ref == 'refs/heads/main' }}
|
||||||
|
- name: Install dependencies
|
||||||
|
run: |
|
||||||
|
sudo apt update
|
||||||
|
sudo apt install -y protobuf-compiler libssl-dev
|
||||||
|
- uses: rui314/setup-mold@v1
|
||||||
- name: Run remote tests
|
- name: Run remote tests
|
||||||
# Running this requires access to secrets, so skip if this is
|
|
||||||
# a PR from a fork.
|
|
||||||
if: github.event_name != 'pull_request' || !github.event.pull_request.head.repo.fork
|
|
||||||
run: CARGO_ARGS="--profile ci" make -C ./lancedb remote-tests
|
run: CARGO_ARGS="--profile ci" make -C ./lancedb remote-tests
|
||||||
|
|
||||||
macos:
|
macos:
|
||||||
@@ -296,16 +324,18 @@ jobs:
|
|||||||
cargo update -p aws-types --precise 1.3.9
|
cargo update -p aws-types --precise 1.3.9
|
||||||
cargo update -p aws-sigv4 --precise 1.3.5
|
cargo update -p aws-sigv4 --precise 1.3.5
|
||||||
cargo update -p aws-credential-types --precise 1.2.8
|
cargo update -p aws-credential-types --precise 1.2.8
|
||||||
cargo update -p aws-smithy-checksums --precise 0.63.9
|
# aws-smithy-checksums must stay at or above 0.63.13: OpenDAL's S3
|
||||||
|
# service needs crc-fast ~1.9, and older releases pin it to ~1.3.
|
||||||
|
cargo update -p aws-smithy-checksums --precise 0.63.13
|
||||||
cargo update -p aws-smithy-runtime --precise 1.9.3
|
cargo update -p aws-smithy-runtime --precise 1.9.3
|
||||||
cargo update -p aws-smithy-http --precise 0.62.4
|
cargo update -p aws-smithy-http --precise 0.62.6
|
||||||
cargo update -p aws-smithy-eventstream --precise 0.60.12
|
cargo update -p aws-smithy-eventstream --precise 0.60.14
|
||||||
cargo update -p aws-smithy-http-client --precise 1.1.3
|
cargo update -p aws-smithy-http-client --precise 1.1.3
|
||||||
cargo update -p aws-smithy-observability --precise 0.1.4
|
cargo update -p aws-smithy-observability --precise 0.1.4
|
||||||
cargo update -p aws-smithy-query --precise 0.60.8
|
cargo update -p aws-smithy-query --precise 0.60.8
|
||||||
cargo update -p aws-smithy-runtime-api --precise 1.9.1
|
cargo update -p aws-smithy-runtime-api --precise 1.9.3
|
||||||
cargo update -p aws-smithy-async --precise 1.2.6
|
cargo update -p aws-smithy-async --precise 1.2.7
|
||||||
cargo update -p aws-smithy-types --precise 1.3.5
|
cargo update -p aws-smithy-types --precise 1.3.6
|
||||||
cargo update -p aws-smithy-xml --precise 0.60.11
|
cargo update -p aws-smithy-xml --precise 0.60.11
|
||||||
cargo update -p home --precise 0.5.9
|
cargo update -p home --precise 0.5.9
|
||||||
- name: cargo +${{ matrix.msrv }} check
|
- name: cargo +${{ matrix.msrv }} check
|
||||||
|
|||||||
@@ -18,6 +18,9 @@ Common commands:
|
|||||||
* Run specific test: `cargo test --quiet --features remote -p <package_name> --test <test_name>`
|
* Run specific test: `cargo test --quiet --features remote -p <package_name> --test <test_name>`
|
||||||
* Lint: `cargo clippy --quiet --features remote --tests --examples`
|
* Lint: `cargo clippy --quiet --features remote --tests --examples`
|
||||||
* Format Rust: `cargo fmt --all`
|
* Format Rust: `cargo fmt --all`
|
||||||
|
* Use repository-defined Cargo profiles instead of ad hoc LTO overrides.
|
||||||
|
* Use `release-with-debug` for benchmarks and profiling so optimized builds keep debug symbols without a rebuild.
|
||||||
|
* Use `release-no-lto` only for local debugging, IO-bound benchmarks, or compile-time-sensitive performance investigation where LTO would not affect the measured bottleneck.
|
||||||
* Format Python: `ruff format .`
|
* Format Python: `ruff format .`
|
||||||
* Lint Python: `ruff check .`
|
* Lint Python: `ruff check .`
|
||||||
* Bootstrap Python dev env: `cd python && uv run --extra tests --extra dev maturin develop --extras tests,dev`
|
* Bootstrap Python dev env: `cd python && uv run --extra tests --extra dev maturin develop --extras tests,dev`
|
||||||
@@ -92,6 +95,8 @@ Python bindings changes:
|
|||||||
* Should use `LOOP.run()` to call the corresponding `AsyncTable` method.
|
* Should use `LOOP.run()` to call the corresponding `AsyncTable` method.
|
||||||
6. Add concrete sync method to `RemoteTable` class in `python/python/lancedb/remote/table.py`.
|
6. Add concrete sync method to `RemoteTable` class in `python/python/lancedb/remote/table.py`.
|
||||||
7. Add unit test in `python/tests/test_table.py`.
|
7. Add unit test in `python/tests/test_table.py`.
|
||||||
|
8. If you added a new public class or module-level function (not just a method on an
|
||||||
|
existing class), expose it in the API reference. See "Python API reference" below.
|
||||||
|
|
||||||
TypeScript bindings changes:
|
TypeScript bindings changes:
|
||||||
|
|
||||||
@@ -103,6 +108,33 @@ TypeScript bindings changes:
|
|||||||
5. Add test in `nodejs/__test__/table.test.ts`.
|
5. Add test in `nodejs/__test__/table.test.ts`.
|
||||||
6. Run `npm run docs` to generate TypeScript documentation.
|
6. Run `npm run docs` to generate TypeScript documentation.
|
||||||
|
|
||||||
|
## Python API reference
|
||||||
|
|
||||||
|
`docs/src/python/python.md` is the entire Python API reference. It is maintained by
|
||||||
|
hand, and anything not listed there is not rendered at all, so new public classes and
|
||||||
|
module-level functions have to be added explicitly. How depends on the module:
|
||||||
|
|
||||||
|
* `lancedb.index`, `lancedb.embeddings`, `lancedb.remote`, and `lancedb.rerankers` are
|
||||||
|
rendered by a single directive each, driven by the module's `__all__`. Add the new
|
||||||
|
name to `__all__` and it appears; forget, and it is silently omitted.
|
||||||
|
* Everything else (`lancedb`, `lancedb.table`, `lancedb.query`, `lancedb.db`, ...) is
|
||||||
|
listed symbol by symbol. Add a `::: lancedb.<module>.<Name>` line to the matching
|
||||||
|
section, and remember that the page separates synchronous and asynchronous APIs.
|
||||||
|
|
||||||
|
Deliberately undocumented: concrete implementations reached through an abstract base
|
||||||
|
(`LanceTable`, `LanceDBConnection`, `RemoteDBConnection`), query base classes already
|
||||||
|
covered by `inherited_members`, and internal helpers.
|
||||||
|
|
||||||
|
Cross-references in docstrings use mkdocstrings syntax, `[text][lancedb.table.Table]`.
|
||||||
|
Plain relative links such as `[Table](Table)` do not resolve. To check your work:
|
||||||
|
|
||||||
|
```shell
|
||||||
|
pip install -r docs/requirements.txt
|
||||||
|
cd docs && PYTHONPATH=. mkdocs build
|
||||||
|
```
|
||||||
|
|
||||||
|
The docs site only builds on pushes to `main`, so this is not covered by PR CI.
|
||||||
|
|
||||||
## Review Guidelines
|
## Review Guidelines
|
||||||
|
|
||||||
Please consider the following when reviewing code contributions.
|
Please consider the following when reviewing code contributions.
|
||||||
|
|||||||
Generated
+338
-331
File diff suppressed because it is too large
Load Diff
+23
-16
@@ -13,20 +13,21 @@ categories = ["database-implementations"]
|
|||||||
rust-version = "1.91.0"
|
rust-version = "1.91.0"
|
||||||
|
|
||||||
[workspace.dependencies]
|
[workspace.dependencies]
|
||||||
lance = { "version" = "=10.0.0-beta.5", default-features = false, "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance = { "version" = "=11.0.0-beta.16", default-features = false, "tag" = "v11.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-core = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-core = { "version" = "=11.0.0-beta.16", "tag" = "v11.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-datagen = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-datagen = { "version" = "=11.0.0-beta.16", "tag" = "v11.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-file = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-file = { "version" = "=11.0.0-beta.16", "tag" = "v11.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-io = { "version" = "=10.0.0-beta.5", default-features = false, "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-io = { "version" = "=11.0.0-beta.16", default-features = false, "tag" = "v11.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-index = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-index = { "version" = "=11.0.0-beta.16", "tag" = "v11.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-linalg = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-linalg = { "version" = "=11.0.0-beta.16", "tag" = "v11.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-namespace = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-namespace = { "version" = "=11.0.0-beta.16", "tag" = "v11.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-namespace-impls = { "version" = "=10.0.0-beta.5", default-features = false, "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-namespace-impls = { "version" = "=11.0.0-beta.16", default-features = false, "tag" = "v11.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-table = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-table = { "version" = "=11.0.0-beta.16", "tag" = "v11.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-testing = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-testing = { "version" = "=11.0.0-beta.16", "tag" = "v11.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-datafusion = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-datafusion = { "version" = "=11.0.0-beta.16", "tag" = "v11.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-encoding = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-encoding = { "version" = "=11.0.0-beta.16", "tag" = "v11.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-arrow = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-arrow = { "version" = "=11.0.0-beta.16", "tag" = "v11.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
|
lancedb = { path = "rust/lancedb", default-features = false }
|
||||||
ahash = "0.8"
|
ahash = "0.8"
|
||||||
# Note that this one does not include pyarrow
|
# Note that this one does not include pyarrow
|
||||||
arrow = { version = "58.0.0", optional = false }
|
arrow = { version = "58.0.0", optional = false }
|
||||||
@@ -39,6 +40,7 @@ arrow-schema = "58.0.0"
|
|||||||
arrow-select = "58.0.0"
|
arrow-select = "58.0.0"
|
||||||
arrow-cast = "58.0.0"
|
arrow-cast = "58.0.0"
|
||||||
async-trait = "0"
|
async-trait = "0"
|
||||||
|
bytes = "1"
|
||||||
datafusion = { version = "54.0.0", default-features = false }
|
datafusion = { version = "54.0.0", default-features = false }
|
||||||
datafusion-catalog = "54.0.0"
|
datafusion-catalog = "54.0.0"
|
||||||
datafusion-common = { version = "54.0.0", default-features = false }
|
datafusion-common = { version = "54.0.0", default-features = false }
|
||||||
@@ -52,7 +54,7 @@ env_logger = "0.11"
|
|||||||
half = { "version" = "2.7.1", default-features = false, features = [
|
half = { "version" = "2.7.1", default-features = false, features = [
|
||||||
"num-traits",
|
"num-traits",
|
||||||
] }
|
] }
|
||||||
futures = "0"
|
futures = "0.3"
|
||||||
log = "0.4"
|
log = "0.4"
|
||||||
metrics = "0.24"
|
metrics = "0.24"
|
||||||
metrics-util = "0.19"
|
metrics-util = "0.19"
|
||||||
@@ -65,7 +67,12 @@ url = "2"
|
|||||||
num-traits = "0.2"
|
num-traits = "0.2"
|
||||||
regex = "1.10"
|
regex = "1.10"
|
||||||
semver = "1.0.25"
|
semver = "1.0.25"
|
||||||
chrono = "0.4"
|
serde = "1"
|
||||||
|
serde_json = "1"
|
||||||
|
tempfile = "3.5.0"
|
||||||
|
tokio = { version = "1.23", features = ["rt-multi-thread", "sync"] }
|
||||||
|
uuid = { version = "1.7.0", features = ["v4"] }
|
||||||
|
chrono = { version = "0.4", default-features = false, features = ["clock"] }
|
||||||
|
|
||||||
[profile.ci]
|
[profile.ci]
|
||||||
debug = "line-tables-only"
|
debug = "line-tables-only"
|
||||||
|
|||||||
@@ -2,6 +2,7 @@
|
|||||||
Check whether there are any breaking changes in the PRs between the base and head commits.
|
Check whether there are any breaking changes in the PRs between the base and head commits.
|
||||||
If there are, assert that we have incremented the minor version.
|
If there are, assert that we have incremented the minor version.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
import os
|
import os
|
||||||
from packaging.version import parse
|
from packaging.version import parse
|
||||||
@@ -27,7 +28,7 @@ if __name__ == "__main__":
|
|||||||
else:
|
else:
|
||||||
print("No breaking changes found.")
|
print("No breaking changes found.")
|
||||||
exit(0)
|
exit(0)
|
||||||
|
|
||||||
last_stable_version = parse(args.last_stable_version)
|
last_stable_version = parse(args.last_stable_version)
|
||||||
current_version = parse(args.current_version)
|
current_version = parse(args.current_version)
|
||||||
if current_version.minor <= last_stable_version.minor:
|
if current_version.minor <= last_stable_version.minor:
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""Determine whether a newer Lance tag exists and expose results for CI."""
|
"""Determine whether a newer Lance tag exists and expose results for CI."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
@@ -36,8 +37,16 @@ class SemVer:
|
|||||||
prerelease: Tuple[Union[int, str], ...]
|
prerelease: Tuple[Union[int, str], ...]
|
||||||
|
|
||||||
def __lt__(self, other: "SemVer") -> bool: # pragma: no cover - simple comparison
|
def __lt__(self, other: "SemVer") -> bool: # pragma: no cover - simple comparison
|
||||||
if (self.major, self.minor, self.patch) != (other.major, other.minor, other.patch):
|
if (self.major, self.minor, self.patch) != (
|
||||||
return (self.major, self.minor, self.patch) < (other.major, other.minor, other.patch)
|
other.major,
|
||||||
|
other.minor,
|
||||||
|
other.patch,
|
||||||
|
):
|
||||||
|
return (self.major, self.minor, self.patch) < (
|
||||||
|
other.major,
|
||||||
|
other.minor,
|
||||||
|
other.patch,
|
||||||
|
)
|
||||||
if self.prerelease == other.prerelease:
|
if self.prerelease == other.prerelease:
|
||||||
return False
|
return False
|
||||||
if not self.prerelease:
|
if not self.prerelease:
|
||||||
@@ -142,7 +151,9 @@ def read_current_version(repo_root: Path) -> str:
|
|||||||
deps = data["workspace"]["dependencies"]
|
deps = data["workspace"]["dependencies"]
|
||||||
entry = deps["lance"]
|
entry = deps["lance"]
|
||||||
except KeyError as exc: # pragma: no cover - configuration guard
|
except KeyError as exc: # pragma: no cover - configuration guard
|
||||||
raise RuntimeError("Failed to locate workspace.dependencies.lance in Cargo.toml") from exc
|
raise RuntimeError(
|
||||||
|
"Failed to locate workspace.dependencies.lance in Cargo.toml"
|
||||||
|
) from exc
|
||||||
|
|
||||||
if isinstance(entry, str):
|
if isinstance(entry, str):
|
||||||
raw_version = entry
|
raw_version = entry
|
||||||
|
|||||||
+9
-6
@@ -1,6 +1,7 @@
|
|||||||
# SPDX-License-Identifier: Apache-2.0
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
"""A zero-dependency mock OpenAI embeddings API endpoint for testing purposes."""
|
"""A zero-dependency mock OpenAI embeddings API endpoint for testing purposes."""
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
import json
|
import json
|
||||||
import http.server
|
import http.server
|
||||||
@@ -22,11 +23,13 @@ class MockOpenAIRequestHandler(http.server.BaseHTTPRequestHandler):
|
|||||||
|
|
||||||
data = []
|
data = []
|
||||||
for i in range(num_inputs):
|
for i in range(num_inputs):
|
||||||
data.append({
|
data.append(
|
||||||
"object": "embedding",
|
{
|
||||||
"embedding": [0.1] * 1536,
|
"object": "embedding",
|
||||||
"index": i,
|
"embedding": [0.1] * 1536,
|
||||||
})
|
"index": i,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
response = {
|
response = {
|
||||||
"object": "list",
|
"object": "list",
|
||||||
@@ -35,7 +38,7 @@ class MockOpenAIRequestHandler(http.server.BaseHTTPRequestHandler):
|
|||||||
"usage": {
|
"usage": {
|
||||||
"prompt_tokens": 0,
|
"prompt_tokens": 0,
|
||||||
"total_tokens": 0,
|
"total_tokens": 0,
|
||||||
}
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
self.send_response(200)
|
self.send_response(200)
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ from packaging.version import parse, InvalidVersion
|
|||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
import argparse
|
import argparse
|
||||||
|
|
||||||
parser = argparse.ArgumentParser()
|
parser = argparse.ArgumentParser()
|
||||||
parser.add_argument("prefix", default="v")
|
parser.add_argument("prefix", default="v")
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|||||||
@@ -22,7 +22,7 @@ def run_command(command: str) -> str:
|
|||||||
def get_latest_stable_version() -> str:
|
def get_latest_stable_version() -> str:
|
||||||
version_line = run_command("cargo info lance | grep '^version:'")
|
version_line = run_command("cargo info lance | grep '^version:'")
|
||||||
# Example output: "version: 0.35.0 (latest 0.37.0)"
|
# Example output: "version: 0.35.0 (latest 0.37.0)"
|
||||||
match = re.search(r'\(latest ([0-9.]+)\)', version_line)
|
match = re.search(r"\(latest ([0-9.]+)\)", version_line)
|
||||||
if match:
|
if match:
|
||||||
return match.group(1)
|
return match.group(1)
|
||||||
# Fallback: use the first version after 'version:'
|
# Fallback: use the first version after 'version:'
|
||||||
@@ -69,7 +69,7 @@ def extract_default_features(line: str) -> bool:
|
|||||||
"""
|
"""
|
||||||
import re
|
import re
|
||||||
|
|
||||||
match = re.search(r'default-features\s*=\s*false', line)
|
match = re.search(r"default-features\s*=\s*false", line)
|
||||||
return match is not None
|
return match is not None
|
||||||
|
|
||||||
|
|
||||||
@@ -104,7 +104,7 @@ def dict_to_toml_line(package_name: str, config: dict) -> str:
|
|||||||
# This shouldn't happen with our current usage
|
# This shouldn't happen with our current usage
|
||||||
parts.append(f'"{key}" = {json.dumps(value)}')
|
parts.append(f'"{key}" = {json.dumps(value)}')
|
||||||
|
|
||||||
return f'{package_name} = {{ {", ".join(parts)} }}\n'
|
return f"{package_name} = {{ {', '.join(parts)} }}\n"
|
||||||
|
|
||||||
|
|
||||||
def update_cargo_toml(line_updater):
|
def update_cargo_toml(line_updater):
|
||||||
@@ -119,7 +119,7 @@ def update_cargo_toml(line_updater):
|
|||||||
lance_line = ""
|
lance_line = ""
|
||||||
is_parsing_lance_line = False
|
is_parsing_lance_line = False
|
||||||
for line in lines:
|
for line in lines:
|
||||||
if line.startswith("lance"):
|
if re.match(r"^lance(?:\s|[-_])", line):
|
||||||
# Check if this is a single-line or multi-line entry
|
# Check if this is a single-line or multi-line entry
|
||||||
# Single-line entries either:
|
# Single-line entries either:
|
||||||
# 1. End with } (complete inline table)
|
# 1. End with } (complete inline table)
|
||||||
|
|||||||
@@ -0,0 +1,185 @@
|
|||||||
|
import os
|
||||||
|
import stat
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
import textwrap
|
||||||
|
import unittest
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||||
|
SCRIPT = REPO_ROOT / "ci" / "set_lance_version.py"
|
||||||
|
LANCE_GIT_URL = "https://github.com/lance-format/lance.git"
|
||||||
|
|
||||||
|
CARGO_TOML = """\
|
||||||
|
[workspace.dependencies]
|
||||||
|
lance = { "version" = "=1.0.0", default-features = false, "features" = ["dynamodb"] }
|
||||||
|
lance-core = "1.0.0"
|
||||||
|
lance_datafusion = {
|
||||||
|
"version" = "=1.0.0",
|
||||||
|
"features" = ["substrait"]
|
||||||
|
}
|
||||||
|
lancedb = { path = "rust/lancedb", default-features = false }
|
||||||
|
lancedb-common = { path = "rust/lancedb-common" }
|
||||||
|
lancewood = "1.0.0"
|
||||||
|
my-lance = "1.0.0"
|
||||||
|
"""
|
||||||
|
|
||||||
|
UNTOUCHED_DEPENDENCIES = """\
|
||||||
|
lancedb = { path = "rust/lancedb", default-features = false }
|
||||||
|
lancedb-common = { path = "rust/lancedb-common" }
|
||||||
|
lancewood = "1.0.0"
|
||||||
|
my-lance = "1.0.0"
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
class SetLanceVersionTest(unittest.TestCase):
|
||||||
|
def test_supported_update_modes_only_rewrite_lance_dependencies(self):
|
||||||
|
cases = {
|
||||||
|
"stable": (
|
||||||
|
"""\
|
||||||
|
lance = { "version" = "=9.9.9", default-features = false, "features" = ["dynamodb"] }
|
||||||
|
lance-core = "=9.9.9"
|
||||||
|
lance_datafusion = { "version" = "=9.9.9", "features" = ["substrait"] }
|
||||||
|
""",
|
||||||
|
["cargo info lance", "cargo metadata"],
|
||||||
|
),
|
||||||
|
"preview": (
|
||||||
|
f"""\
|
||||||
|
lance = {{ "version" = "=10.0.0-beta.3", default-features = false, "features" = ["dynamodb"], "tag" = "v10.0.0-beta.3", "git" = "{LANCE_GIT_URL}" }}
|
||||||
|
lance-core = {{ "version" = "=10.0.0-beta.3", "tag" = "v10.0.0-beta.3", "git" = "{LANCE_GIT_URL}" }}
|
||||||
|
lance_datafusion = {{ "version" = "=10.0.0-beta.3", "features" = ["substrait"], "tag" = "v10.0.0-beta.3", "git" = "{LANCE_GIT_URL}" }}
|
||||||
|
""",
|
||||||
|
["git ls-remote --tags", "cargo metadata"],
|
||||||
|
),
|
||||||
|
"local": (
|
||||||
|
"""\
|
||||||
|
lance = { "path" = "../lance/rust/lance", default-features = false, "features" = ["dynamodb"] }
|
||||||
|
lance-core = { "path" = "../lance/rust/lance-core" }
|
||||||
|
lance_datafusion = { "path" = "../lance/rust/lance_datafusion", "features" = ["substrait"] }
|
||||||
|
""",
|
||||||
|
["cargo metadata"],
|
||||||
|
),
|
||||||
|
"v8.1.2": (
|
||||||
|
"""\
|
||||||
|
lance = { "version" = "=8.1.2", default-features = false, "features" = ["dynamodb"] }
|
||||||
|
lance-core = "=8.1.2"
|
||||||
|
lance_datafusion = { "version" = "=8.1.2", "features" = ["substrait"] }
|
||||||
|
""",
|
||||||
|
["cargo metadata"],
|
||||||
|
),
|
||||||
|
"v8.2.0-beta.4": (
|
||||||
|
f"""\
|
||||||
|
lance = {{ "version" = "=8.2.0-beta.4", default-features = false, "features" = ["dynamodb"], "tag" = "v8.2.0-beta.4", "git" = "{LANCE_GIT_URL}" }}
|
||||||
|
lance-core = {{ "version" = "=8.2.0-beta.4", "tag" = "v8.2.0-beta.4", "git" = "{LANCE_GIT_URL}" }}
|
||||||
|
lance_datafusion = {{ "version" = "=8.2.0-beta.4", "features" = ["substrait"], "tag" = "v8.2.0-beta.4", "git" = "{LANCE_GIT_URL}" }}
|
||||||
|
""",
|
||||||
|
["cargo metadata"],
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
for version, (updated_dependencies, expected_commands) in cases.items():
|
||||||
|
with self.subTest(version=version), tempfile.TemporaryDirectory() as tmp:
|
||||||
|
workdir = Path(tmp)
|
||||||
|
(workdir / "Cargo.toml").write_text(CARGO_TOML)
|
||||||
|
command_log = workdir / "commands.log"
|
||||||
|
fake_bin = workdir / "bin"
|
||||||
|
fake_bin.mkdir()
|
||||||
|
self._write_fake_executables(fake_bin)
|
||||||
|
self._write_fake_python_dependencies(workdir)
|
||||||
|
|
||||||
|
env = os.environ.copy()
|
||||||
|
env["PATH"] = os.pathsep.join([str(fake_bin), env["PATH"]])
|
||||||
|
env["FAKE_COMMAND_LOG"] = str(command_log)
|
||||||
|
env["PYTHONPATH"] = os.pathsep.join(
|
||||||
|
filter(None, [str(workdir), env.get("PYTHONPATH")])
|
||||||
|
)
|
||||||
|
result = subprocess.run(
|
||||||
|
[sys.executable, str(SCRIPT), version],
|
||||||
|
cwd=workdir,
|
||||||
|
env=env,
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
timeout=10,
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(result.returncode, 0, result.stderr)
|
||||||
|
self.assertEqual(
|
||||||
|
(workdir / "Cargo.toml").read_text(),
|
||||||
|
"[workspace.dependencies]\n"
|
||||||
|
+ updated_dependencies
|
||||||
|
+ UNTOUCHED_DEPENDENCIES,
|
||||||
|
)
|
||||||
|
commands = command_log.read_text().splitlines()
|
||||||
|
for command in expected_commands:
|
||||||
|
self.assertTrue(
|
||||||
|
any(line.startswith(command) for line in commands),
|
||||||
|
f"{command!r} not found in {commands!r}",
|
||||||
|
)
|
||||||
|
|
||||||
|
def _write_fake_executables(self, fake_bin: Path) -> None:
|
||||||
|
cargo = fake_bin / "cargo"
|
||||||
|
cargo.write_text(
|
||||||
|
textwrap.dedent(
|
||||||
|
"""\
|
||||||
|
#!/bin/sh
|
||||||
|
printf 'cargo %s\\n' "$*" >> "$FAKE_COMMAND_LOG"
|
||||||
|
case "$1" in
|
||||||
|
info)
|
||||||
|
printf '%s\\n' 'version: 8.8.8 (latest 9.9.9)'
|
||||||
|
;;
|
||||||
|
metadata)
|
||||||
|
;;
|
||||||
|
*)
|
||||||
|
exit 2
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
)
|
||||||
|
cargo.chmod(cargo.stat().st_mode | stat.S_IXUSR)
|
||||||
|
|
||||||
|
git = fake_bin / "git"
|
||||||
|
git.write_text(
|
||||||
|
textwrap.dedent(
|
||||||
|
"""\
|
||||||
|
#!/bin/sh
|
||||||
|
printf 'git %s\\n' "$*" >> "$FAKE_COMMAND_LOG"
|
||||||
|
if [ "$1" != "ls-remote" ]; then
|
||||||
|
exit 2
|
||||||
|
fi
|
||||||
|
printf '%s\\n' \\
|
||||||
|
'111111 refs/tags/v9.9.9' \\
|
||||||
|
'222222 refs/tags/v10.0.0-beta.1' \\
|
||||||
|
'333333 refs/tags/v10.0.0-beta.3'
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
)
|
||||||
|
git.chmod(git.stat().st_mode | stat.S_IXUSR)
|
||||||
|
|
||||||
|
def _write_fake_python_dependencies(self, workdir: Path) -> None:
|
||||||
|
packaging = workdir / "packaging"
|
||||||
|
packaging.mkdir()
|
||||||
|
(packaging / "__init__.py").write_text("")
|
||||||
|
(packaging / "version.py").write_text(
|
||||||
|
textwrap.dedent(
|
||||||
|
"""\
|
||||||
|
class Version:
|
||||||
|
def __init__(self, value):
|
||||||
|
release, _, prerelease = value.partition("-beta.")
|
||||||
|
self._key = (
|
||||||
|
tuple(int(part) for part in release.split(".")),
|
||||||
|
not prerelease,
|
||||||
|
int(prerelease or 0),
|
||||||
|
)
|
||||||
|
|
||||||
|
def __lt__(self, other):
|
||||||
|
return self._key < other._key
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -12,7 +12,7 @@ with open("Cargo.toml", "rb") as f:
|
|||||||
elif isinstance(dep, dict):
|
elif isinstance(dep, dict):
|
||||||
# Version doesn't have the beta tag in it, so we instead look
|
# Version doesn't have the beta tag in it, so we instead look
|
||||||
# at the git tag.
|
# at the git tag.
|
||||||
version = dep.get('tag', dep.get('version'))
|
version = dep.get("tag", dep.get("version"))
|
||||||
else:
|
else:
|
||||||
raise ValueError("Unexpected type for dependency: " + str(dep))
|
raise ValueError("Unexpected type for dependency: " + str(dep))
|
||||||
|
|
||||||
|
|||||||
@@ -101,6 +101,19 @@ ignore = [
|
|||||||
# https://rustsec.org/advisories/RUSTSEC-2026-0195
|
# https://rustsec.org/advisories/RUSTSEC-2026-0195
|
||||||
{ id = "RUSTSEC-2026-0194", reason = "transitive via inferno/lance/opendal; XML from trusted cloud endpoints, not attacker-controlled" },
|
{ id = "RUSTSEC-2026-0194", reason = "transitive via inferno/lance/opendal; XML from trusted cloud endpoints, not attacker-controlled" },
|
||||||
{ id = "RUSTSEC-2026-0195", reason = "transitive via inferno/lance/opendal; XML from trusted cloud endpoints, not attacker-controlled" },
|
{ id = "RUSTSEC-2026-0195", reason = "transitive via inferno/lance/opendal; XML from trusted cloud endpoints, not attacker-controlled" },
|
||||||
|
# smartstring: unmaintained — the repository was archived by its author on
|
||||||
|
# 2026-05-03. Not a vulnerability. Reached only transitively through polars
|
||||||
|
# (polars-core/-io/-ops/-time/-utils); nothing in LanceDB depends on it directly.
|
||||||
|
# The advisory states no safe upgrade is available: upstream recommends
|
||||||
|
# compact_str/smol_str, so clearing this requires polars to migrate.
|
||||||
|
# https://rustsec.org/advisories/RUSTSEC-2026-0249
|
||||||
|
{ id = "RUSTSEC-2026-0249", reason = "smartstring unmaintained via polars; no fixed upstream release" },
|
||||||
|
|
||||||
|
# h2 0.3: empty DATA frames can be queued without limit. The patched
|
||||||
|
# h2 0.4 line is locked to 0.4.16, but no patched 0.3 release exists.
|
||||||
|
# The old copy is pulled in by aws-smithy's legacy hyper 0.14 client.
|
||||||
|
# https://rustsec.org/advisories/RUSTSEC-2026-0258
|
||||||
|
{ id = "RUSTSEC-2026-0258", reason = "h2 0.3 via legacy aws-smithy/hyper 0.14; no patched 0.3 release" },
|
||||||
]
|
]
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -164,6 +177,11 @@ multiple-versions = "warn"
|
|||||||
# Wildcard version requirements (`foo = "*"`) are a footgun — they let any
|
# Wildcard version requirements (`foo = "*"`) are a footgun — they let any
|
||||||
# future release in without review. Ban them outright.
|
# future release in without review. Ban them outright.
|
||||||
wildcards = "deny"
|
wildcards = "deny"
|
||||||
|
# Lint every dependency declared by a workspace member against the shared
|
||||||
|
# `[workspace.dependencies]` table: any crate used by more than one member must
|
||||||
|
# go through `workspace = true`, and entries nothing uses are an error. This
|
||||||
|
# keeps versions from drifting between the core crate and the bindings.
|
||||||
|
workspace-dependencies = { duplicates = "deny", unused = "deny" }
|
||||||
# Internal workspace crates reference each other via `path = "..."`, which
|
# Internal workspace crates reference each other via `path = "..."`, which
|
||||||
# cargo-deny sees as a wildcard version. That's fine for private workspace
|
# cargo-deny sees as a wildcard version. That's fine for private workspace
|
||||||
# members (not published to crates.io), so allow it specifically for paths.
|
# members (not published to crates.io), so allow it specifically for paths.
|
||||||
|
|||||||
@@ -51,6 +51,11 @@ plugins:
|
|||||||
paths: [../python/python]
|
paths: [../python/python]
|
||||||
options:
|
options:
|
||||||
docstring_style: numpy
|
docstring_style: numpy
|
||||||
|
docstring_options:
|
||||||
|
# Attributes documented in a `Parameters` section, and pydantic
|
||||||
|
# dataclasses whose `__init__` griffe cannot see statically, both
|
||||||
|
# trip this check. It reports nothing actionable here.
|
||||||
|
warn_unknown_params: false
|
||||||
heading_level: 3
|
heading_level: 3
|
||||||
show_signature_annotations: true
|
show_signature_annotations: true
|
||||||
show_root_heading: true
|
show_root_heading: true
|
||||||
|
|||||||
@@ -5,5 +5,5 @@ mkdocs-autorefs>=0.5,<=1.0
|
|||||||
mkdocstrings[python]>=0.24,<1.0
|
mkdocstrings[python]>=0.24,<1.0
|
||||||
griffe>=0.40,<1.0
|
griffe>=0.40,<1.0
|
||||||
mkdocs-render-swagger-plugin>=0.1.0
|
mkdocs-render-swagger-plugin>=0.1.0
|
||||||
pydantic>=2.0,<3.0
|
pydantic>=2.7.4,<3
|
||||||
mkdocs-redirects>=1.2.0
|
mkdocs-redirects>=1.2.0
|
||||||
|
|||||||
+33
-1
@@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`:
|
|||||||
<dependency>
|
<dependency>
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-core</artifactId>
|
<artifactId>lancedb-core</artifactId>
|
||||||
<version>0.37.1-beta.0</version>
|
<version>0.38.0-beta.3</version>
|
||||||
</dependency>
|
</dependency>
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -55,6 +55,38 @@ LanceNamespace namespaceClient = LanceDbNamespaceClientBuilder.newBuilder()
|
|||||||
| `region(String)` | AWS region (default: "us-east-1") | No |
|
| `region(String)` | AWS region (default: "us-east-1") | No |
|
||||||
| `config(String, String)` | Additional configuration parameters | No |
|
| `config(String, String)` | Additional configuration parameters | No |
|
||||||
|
|
||||||
|
### Opening a Table with Vended Credentials
|
||||||
|
|
||||||
|
When the catalog vends temporary object store credentials, open the table through the
|
||||||
|
namespace client. The Lance dataset builder fetches the table location and storage options
|
||||||
|
from the catalog and refreshes the credentials when they expire.
|
||||||
|
|
||||||
|
```java
|
||||||
|
import com.lancedb.LanceDbNamespaceClientBuilder;
|
||||||
|
import org.lance.Dataset;
|
||||||
|
import org.lance.namespace.LanceNamespace;
|
||||||
|
|
||||||
|
import java.util.Arrays;
|
||||||
|
|
||||||
|
LanceNamespace namespaceClient = LanceDbNamespaceClientBuilder.newBuilder()
|
||||||
|
.apiKey(System.getenv("LANCEDB_API_KEY"))
|
||||||
|
.database(System.getenv("LANCEDB_DATABASE"))
|
||||||
|
// Set the endpoint for a LanceDB Enterprise deployment.
|
||||||
|
// .endpoint("https://your-enterprise-endpoint")
|
||||||
|
.build();
|
||||||
|
|
||||||
|
try (Dataset dataset = Dataset.open()
|
||||||
|
.namespaceClient(namespaceClient)
|
||||||
|
.tableId(Arrays.asList("my_namespace", "my_table"))
|
||||||
|
.build()) {
|
||||||
|
System.out.println("Rows: " + dataset.countRows());
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Do not call `describeTable()` and then open the returned location with `Dataset.open(uri)`.
|
||||||
|
Opening through `namespaceClient()` is what applies the vended storage options and enables
|
||||||
|
automatic credential refresh. No object store credentials need to be passed by the application.
|
||||||
|
|
||||||
## Metadata Operations
|
## Metadata Operations
|
||||||
|
|
||||||
### Creating a Namespace Path
|
### Creating a Namespace Path
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# Contributing to LanceDB Typescript
|
# Contributing to LanceDB Typescript
|
||||||
|
|
||||||
This document outlines the process for contributing to LanceDB Typescript.
|
This document outlines the process for contributing to LanceDB Typescript.
|
||||||
For general contribution guidelines, see [CONTRIBUTING.md](../CONTRIBUTING.md).
|
For general contribution guidelines, see [CONTRIBUTING.md](https://github.com/lancedb/lancedb/blob/main/CONTRIBUTING.md).
|
||||||
|
|
||||||
## Project layout
|
## Project layout
|
||||||
|
|
||||||
|
|||||||
@@ -25,6 +25,27 @@ the underlying connection has been closed.
|
|||||||
|
|
||||||
## Methods
|
## Methods
|
||||||
|
|
||||||
|
### cancelJob()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract cancelJob(jobId): Promise<boolean>
|
||||||
|
```
|
||||||
|
|
||||||
|
Request cancellation of a server-side job by id.
|
||||||
|
|
||||||
|
Resolves to true if the server accepted the cancellation, false if no
|
||||||
|
such job exists. Cancelling an already-terminal job is a no-op success.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **jobId**: `string`
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`boolean`>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### cloneTable()
|
### cloneTable()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -365,6 +386,49 @@ Drop an existing table.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
### dropTableAsync()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract dropTableAsync(name, namespacePath?): Promise<Job>
|
||||||
|
```
|
||||||
|
|
||||||
|
Start dropping a table and return its cleanup job.
|
||||||
|
|
||||||
|
The table may become unavailable before its data files are removed. Wait
|
||||||
|
on the returned job to know when cleanup has finished.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **name**: `string`
|
||||||
|
|
||||||
|
* **namespacePath?**: `string`[]
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<[`Job`](Job.md)>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### getJob()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract getJob(jobId): Promise<null | JobDescription>
|
||||||
|
```
|
||||||
|
|
||||||
|
Describe a single server-side job by id.
|
||||||
|
|
||||||
|
Resolves to `null` when the server has no such job.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **jobId**: `string`
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`null` \| [`JobDescription`](../interfaces/JobDescription.md)>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### isOpen()
|
### isOpen()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -379,6 +443,62 @@ Return true if the connection has not been closed
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
### job()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract job(jobId): Job
|
||||||
|
```
|
||||||
|
|
||||||
|
A [Job](Job.md) handle for a server-side job by id.
|
||||||
|
|
||||||
|
The handle is constructed without a server round trip; an unknown id
|
||||||
|
surfaces when the handle is used. Dropping the handle has no effect on
|
||||||
|
the job itself.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **jobId**: `string`
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
[`Job`](Job.md)
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### jobHistory()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract jobHistory(jobId?): Promise<Table<any>>
|
||||||
|
```
|
||||||
|
|
||||||
|
The lifecycle event history of a server-side job, as an Arrow table.
|
||||||
|
|
||||||
|
Lists history across all jobs when `jobId` is omitted.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **jobId?**: `string`
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`Table`<`any`>>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### listJobs()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract listJobs(): Promise<JobInfo[]>
|
||||||
|
```
|
||||||
|
|
||||||
|
List server-side jobs across the database's tables.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<[`JobInfo`](../interfaces/JobInfo.md)[]>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### listNamespaces()
|
### listNamespaces()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
|
|||||||
@@ -0,0 +1,83 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / Job
|
||||||
|
|
||||||
|
# Class: Job
|
||||||
|
|
||||||
|
A handle to an operation that may still be running.
|
||||||
|
|
||||||
|
## Constructors
|
||||||
|
|
||||||
|
### new Job()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
new Job(): Job
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
[`Job`](Job.md)
|
||||||
|
|
||||||
|
## Accessors
|
||||||
|
|
||||||
|
### id
|
||||||
|
|
||||||
|
```ts
|
||||||
|
get id(): null | string
|
||||||
|
```
|
||||||
|
|
||||||
|
Identifies the operation on the server that is running it. Operations
|
||||||
|
that run in this process have no server id. The value is opaque.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`null` \| `string`
|
||||||
|
|
||||||
|
## Methods
|
||||||
|
|
||||||
|
### cancel()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
cancel(): Promise<void>
|
||||||
|
```
|
||||||
|
|
||||||
|
Request cancellation. Cancelling a finished operation is a no-op.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`void`>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### status()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
status(): Promise<string>
|
||||||
|
```
|
||||||
|
|
||||||
|
The operation's current lifecycle state: "running", "finished",
|
||||||
|
"failed", or "cancelled".
|
||||||
|
|
||||||
|
A point snapshot; unlike [Job.wait](Job.md#wait) it does not block or reject
|
||||||
|
on a terminal failure state. States a newer server reports that this
|
||||||
|
client version does not know pass through as-is.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`string`>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### wait()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
wait(): Promise<void>
|
||||||
|
```
|
||||||
|
|
||||||
|
Wait until the operation reaches a terminal state.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`void`>
|
||||||
@@ -69,14 +69,34 @@ abstract addColumns(newColumnTransforms): Promise<AddColumnsResult>
|
|||||||
|
|
||||||
Add new columns with defined values.
|
Add new columns with defined values.
|
||||||
|
|
||||||
|
The `{ computed }` form stores the expression rather than evaluating it
|
||||||
|
now: the column is committed with no values, and rows get them from
|
||||||
|
[Table#refreshColumn](Table.md#refreshcolumn). Declaring one therefore costs the same on a
|
||||||
|
large table as on an empty one.
|
||||||
|
|
||||||
|
A refresh does not revisit rows it has already filled, so mutating an
|
||||||
|
input leaves the value computed at fill time; recomputing means dropping
|
||||||
|
the column and declaring it again. While a declaration reads a column,
|
||||||
|
that column cannot be renamed, retyped or dropped.
|
||||||
|
|
||||||
|
On LanceDB Cloud and Enterprise the expression is planned by the
|
||||||
|
server, and the refresh runs as a server job -- see
|
||||||
|
[Table#refreshColumnAsync](Table.md#refreshcolumnasync).
|
||||||
|
|
||||||
#### Parameters
|
#### Parameters
|
||||||
|
|
||||||
* **newColumnTransforms**: `Field`<`any`> \| `Field`<`any`>[] \| `Schema`<`any`> \| [`AddColumnsSql`](../interfaces/AddColumnsSql.md)[]
|
* **newColumnTransforms**:
|
||||||
|
\| `Field`<`any`>
|
||||||
|
\| `Field`<`any`>[]
|
||||||
|
\| `Schema`<`any`>
|
||||||
|
\| [`AddColumnsSql`](../interfaces/AddColumnsSql.md)[]
|
||||||
|
\| `object`
|
||||||
Either:
|
Either:
|
||||||
- An array of objects with column names and SQL expressions to calculate values
|
- An array of objects with column names and SQL expressions to calculate values
|
||||||
- A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
- A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
||||||
- An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
- An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
||||||
- An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
- An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
||||||
|
- `{ computed }`, declaring columns defined by a SQL expression whose type and inputs are derived from it
|
||||||
|
|
||||||
#### Returns
|
#### Returns
|
||||||
|
|
||||||
@@ -85,6 +105,13 @@ Add new columns with defined values.
|
|||||||
A promise that resolves to an object
|
A promise that resolves to an object
|
||||||
containing the new version number of the table after adding the columns.
|
containing the new version number of the table after adding the columns.
|
||||||
|
|
||||||
|
#### Example
|
||||||
|
|
||||||
|
```ts
|
||||||
|
await table.addColumns({ computed: [{ name: "doubled", valueSql: "x * 2" }] });
|
||||||
|
const { rowsFilled } = await table.refreshColumn("doubled");
|
||||||
|
```
|
||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### alterColumns()
|
### alterColumns()
|
||||||
@@ -186,6 +213,39 @@ version of the table.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
### checkpointLsm()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract checkpointLsm(): Promise<void>
|
||||||
|
```
|
||||||
|
|
||||||
|
Converge this table's LSM write path into its base table.
|
||||||
|
|
||||||
|
Seals once, then triggers compaction and polls until the L0 that existed
|
||||||
|
at the start is gone. The target set is fixed at the start, so
|
||||||
|
generations created *during* the checkpoint are ignored — that is what
|
||||||
|
lets it terminate under write load, and what makes it best-effort: it
|
||||||
|
converges the fresh tier as of some instant. Idempotent, abandonable at
|
||||||
|
any point, and safe to run on a cadence.
|
||||||
|
|
||||||
|
There is no liveness bound — the compactor pool is shared across tables,
|
||||||
|
so a checkpoint queued behind unrelated work looks exactly like one that
|
||||||
|
is merging. The caller owns the deadline.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`void`>
|
||||||
|
|
||||||
|
#### Example
|
||||||
|
|
||||||
|
```ts
|
||||||
|
const before = await table.getLsmStats();
|
||||||
|
await table.checkpointLsm();
|
||||||
|
const after = await table.getLsmStats();
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### close()
|
### close()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -223,6 +283,24 @@ It is a no-op when no writers are cached.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
### compactLsm()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract compactLsm(): Promise<void>
|
||||||
|
```
|
||||||
|
|
||||||
|
Trigger a background L0 → base compaction pass per bucket.
|
||||||
|
|
||||||
|
Returns once the passes are *dispatched*, not once they finish — watch
|
||||||
|
[Table#getLsmStats](Table.md#getlsmstats) for progress, or use
|
||||||
|
[Table#checkpointLsm](Table.md#checkpointlsm) to wait for convergence.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`void`>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### countRows()
|
### countRows()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -295,6 +373,29 @@ await table.createIndex("my_float_col");
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
### createIndexAsync()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract createIndexAsync(column, options?): Promise<Job>
|
||||||
|
```
|
||||||
|
|
||||||
|
Create an index, returning a handle to the indexing job.
|
||||||
|
|
||||||
|
The job may already be complete when returned; callers must not assume
|
||||||
|
the index exists until [Job.wait](Job.md#wait) resolves.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **column**: `string`
|
||||||
|
|
||||||
|
* **options?**: `Partial`<[`IndexOptions`](../interfaces/IndexOptions.md)>
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<[`Job`](Job.md)>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### currentBranch()
|
### currentBranch()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -398,6 +499,48 @@ Drop an index from the table.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
### flushLsm()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract flushLsm(): Promise<void>
|
||||||
|
```
|
||||||
|
|
||||||
|
Seal every bucket's active memtable into a new L0 generation.
|
||||||
|
|
||||||
|
Returns once the seal is committed. Sealing an empty memtable is a no-op,
|
||||||
|
so this is safe to call repeatedly.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`void`>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### getLsmStats()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract getLsmStats(includeGenerationRows?): Promise<undefined | LsmStats>
|
||||||
|
```
|
||||||
|
|
||||||
|
Read live per-bucket LSM state.
|
||||||
|
|
||||||
|
Answers "how far behind is my fresh tier", "which bucket is hot", and
|
||||||
|
"why is my fresh-tier vector search brute-force". Mutates no table state.
|
||||||
|
|
||||||
|
Resolves to `undefined` only when the LSM write path is not enabled.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **includeGenerationRows?**: `boolean`
|
||||||
|
Also count rows per L0 generation.
|
||||||
|
Off by default because each count opens an uncached Lance dataset.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<`undefined` \| [`LsmStats`](../interfaces/LsmStats.md)>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### getLsmWriteSpec()
|
### getLsmWriteSpec()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -408,9 +551,10 @@ Read the [LsmWriteSpec](../interfaces/LsmWriteSpec.md) currently installed on th
|
|||||||
|
|
||||||
Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
|
Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
|
||||||
spec has been set, or it was removed with [Table#unsetLsmWriteSpec](Table.md#unsetlsmwritespec)).
|
spec has been set, or it was removed with [Table#unsetLsmWriteSpec](Table.md#unsetlsmwritespec)).
|
||||||
The returned spec — including its `maintainedIndexes` and
|
The returned spec mirrors what was passed to
|
||||||
`writerConfigDefaults` — mirrors what was passed to
|
[Table#setLsmWriteSpec](Table.md#setlsmwritespec), except that `maintainedIndexes` always
|
||||||
[Table#setLsmWriteSpec](Table.md#setlsmwritespec).
|
reports the concrete list resolved when the spec was set — `undefined`
|
||||||
|
never round-trips.
|
||||||
|
|
||||||
#### Returns
|
#### Returns
|
||||||
|
|
||||||
@@ -694,6 +838,67 @@ for await (const batch of table.query()) {
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
### refreshColumn()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract refreshColumn(column): Promise<RefreshColumnResult>
|
||||||
|
```
|
||||||
|
|
||||||
|
Fill the rows of a computed column that hold no value yet.
|
||||||
|
|
||||||
|
Rows appended since the last refresh are filled by the next one; rows
|
||||||
|
already filled are left as they are, so the call is idempotent and does
|
||||||
|
not observe a mutated input. Local tables only: a remote refresh runs
|
||||||
|
as a server job, through [Table#refreshColumnAsync](Table.md#refreshcolumnasync).
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **column**: `string`
|
||||||
|
The name of the computed column to fill.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<[`RefreshColumnResult`](../interfaces/RefreshColumnResult.md)>
|
||||||
|
|
||||||
|
A promise that resolves to the
|
||||||
|
number of rows filled and the new version number of the table.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### refreshColumnAsync()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract refreshColumnAsync(column): Promise<Job>
|
||||||
|
```
|
||||||
|
|
||||||
|
Like [Table#refreshColumn](Table.md#refreshcolumn), but returns a handle to the refresh
|
||||||
|
job instead of blocking until it completes.
|
||||||
|
|
||||||
|
The job may already be complete when returned; callers must not assume
|
||||||
|
the column is filled until [Job.wait](Job.md#wait) resolves. Invalid input --
|
||||||
|
an unknown column, or one that is not computed -- rejects here rather
|
||||||
|
than failing the job. On local tables the job runs in-process; on
|
||||||
|
LanceDB Cloud and Enterprise it is the server's backfill job.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **column**: `string`
|
||||||
|
The name of the computed column to fill.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<[`Job`](Job.md)>
|
||||||
|
|
||||||
|
#### Example
|
||||||
|
|
||||||
|
```ts
|
||||||
|
const job = await table.refreshColumnAsync("doubled");
|
||||||
|
await job.wait();
|
||||||
|
console.log(await job.status()); // "finished"
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### restore()
|
### restore()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -783,6 +988,11 @@ All variants require the table to have an unenforced primary key
|
|||||||
([Table#setUnenforcedPrimaryKey](Table.md#setunenforcedprimarykey)); bucket sharding additionally
|
([Table#setUnenforcedPrimaryKey](Table.md#setunenforcedprimarykey)); bucket sharding additionally
|
||||||
requires it to be the single column being bucketed.
|
requires it to be the single column being bucketed.
|
||||||
|
|
||||||
|
Omitting `maintainedIndexes` maintains every index on the table, resolved
|
||||||
|
here, failing if one cannot be maintained — name them to install anyway.
|
||||||
|
Naming them pins an exact set, and a still-building index is rejected
|
||||||
|
rather than quietly omitted.
|
||||||
|
|
||||||
#### Parameters
|
#### Parameters
|
||||||
|
|
||||||
* **spec**: [`LsmWriteSpec`](../interfaces/LsmWriteSpec.md)
|
* **spec**: [`LsmWriteSpec`](../interfaces/LsmWriteSpec.md)
|
||||||
|
|||||||
@@ -25,6 +25,7 @@
|
|||||||
- [Connection](classes/Connection.md)
|
- [Connection](classes/Connection.md)
|
||||||
- [HeaderProvider](classes/HeaderProvider.md)
|
- [HeaderProvider](classes/HeaderProvider.md)
|
||||||
- [Index](classes/Index.md)
|
- [Index](classes/Index.md)
|
||||||
|
- [Job](classes/Job.md)
|
||||||
- [MakeArrowTableOptions](classes/MakeArrowTableOptions.md)
|
- [MakeArrowTableOptions](classes/MakeArrowTableOptions.md)
|
||||||
- [MatchQuery](classes/MatchQuery.md)
|
- [MatchQuery](classes/MatchQuery.md)
|
||||||
- [MergeInsertBuilder](classes/MergeInsertBuilder.md)
|
- [MergeInsertBuilder](classes/MergeInsertBuilder.md)
|
||||||
@@ -57,6 +58,7 @@
|
|||||||
- [BranchDiff](interfaces/BranchDiff.md)
|
- [BranchDiff](interfaces/BranchDiff.md)
|
||||||
- [BranchIndexSummary](interfaces/BranchIndexSummary.md)
|
- [BranchIndexSummary](interfaces/BranchIndexSummary.md)
|
||||||
- [BranchRowCountSummary](interfaces/BranchRowCountSummary.md)
|
- [BranchRowCountSummary](interfaces/BranchRowCountSummary.md)
|
||||||
|
- [BucketStats](interfaces/BucketStats.md)
|
||||||
- [ClientConfig](interfaces/ClientConfig.md)
|
- [ClientConfig](interfaces/ClientConfig.md)
|
||||||
- [ColumnAlteration](interfaces/ColumnAlteration.md)
|
- [ColumnAlteration](interfaces/ColumnAlteration.md)
|
||||||
- [ColumnOrdering](interfaces/ColumnOrdering.md)
|
- [ColumnOrdering](interfaces/ColumnOrdering.md)
|
||||||
@@ -80,6 +82,7 @@
|
|||||||
- [FtsToken](interfaces/FtsToken.md)
|
- [FtsToken](interfaces/FtsToken.md)
|
||||||
- [FullTextQuery](interfaces/FullTextQuery.md)
|
- [FullTextQuery](interfaces/FullTextQuery.md)
|
||||||
- [FullTextSearchOptions](interfaces/FullTextSearchOptions.md)
|
- [FullTextSearchOptions](interfaces/FullTextSearchOptions.md)
|
||||||
|
- [GenerationStats](interfaces/GenerationStats.md)
|
||||||
- [HnswPqOptions](interfaces/HnswPqOptions.md)
|
- [HnswPqOptions](interfaces/HnswPqOptions.md)
|
||||||
- [HnswSqOptions](interfaces/HnswSqOptions.md)
|
- [HnswSqOptions](interfaces/HnswSqOptions.md)
|
||||||
- [IndexConfig](interfaces/IndexConfig.md)
|
- [IndexConfig](interfaces/IndexConfig.md)
|
||||||
@@ -88,9 +91,14 @@
|
|||||||
- [IvfFlatOptions](interfaces/IvfFlatOptions.md)
|
- [IvfFlatOptions](interfaces/IvfFlatOptions.md)
|
||||||
- [IvfPqOptions](interfaces/IvfPqOptions.md)
|
- [IvfPqOptions](interfaces/IvfPqOptions.md)
|
||||||
- [IvfRqOptions](interfaces/IvfRqOptions.md)
|
- [IvfRqOptions](interfaces/IvfRqOptions.md)
|
||||||
|
- [JobDescription](interfaces/JobDescription.md)
|
||||||
|
- [JobFailureInfo](interfaces/JobFailureInfo.md)
|
||||||
|
- [JobInfo](interfaces/JobInfo.md)
|
||||||
- [ListNamespacesOptions](interfaces/ListNamespacesOptions.md)
|
- [ListNamespacesOptions](interfaces/ListNamespacesOptions.md)
|
||||||
- [ListNamespacesResponse](interfaces/ListNamespacesResponse.md)
|
- [ListNamespacesResponse](interfaces/ListNamespacesResponse.md)
|
||||||
|
- [LsmStats](interfaces/LsmStats.md)
|
||||||
- [LsmWriteSpec](interfaces/LsmWriteSpec.md)
|
- [LsmWriteSpec](interfaces/LsmWriteSpec.md)
|
||||||
|
- [MemtableStats](interfaces/MemtableStats.md)
|
||||||
- [MergeBlocker](interfaces/MergeBlocker.md)
|
- [MergeBlocker](interfaces/MergeBlocker.md)
|
||||||
- [MergeBranchResult](interfaces/MergeBranchResult.md)
|
- [MergeBranchResult](interfaces/MergeBranchResult.md)
|
||||||
- [MergePreview](interfaces/MergePreview.md)
|
- [MergePreview](interfaces/MergePreview.md)
|
||||||
@@ -101,6 +109,7 @@
|
|||||||
- [OptimizeOptions](interfaces/OptimizeOptions.md)
|
- [OptimizeOptions](interfaces/OptimizeOptions.md)
|
||||||
- [OptimizeStats](interfaces/OptimizeStats.md)
|
- [OptimizeStats](interfaces/OptimizeStats.md)
|
||||||
- [QueryExecutionOptions](interfaces/QueryExecutionOptions.md)
|
- [QueryExecutionOptions](interfaces/QueryExecutionOptions.md)
|
||||||
|
- [RefreshColumnResult](interfaces/RefreshColumnResult.md)
|
||||||
- [RemovalStats](interfaces/RemovalStats.md)
|
- [RemovalStats](interfaces/RemovalStats.md)
|
||||||
- [RenameTableOptions](interfaces/RenameTableOptions.md)
|
- [RenameTableOptions](interfaces/RenameTableOptions.md)
|
||||||
- [RestNamespaceConfig](interfaces/RestNamespaceConfig.md)
|
- [RestNamespaceConfig](interfaces/RestNamespaceConfig.md)
|
||||||
|
|||||||
@@ -0,0 +1,116 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / BucketStats
|
||||||
|
|
||||||
|
# Interface: BucketStats
|
||||||
|
|
||||||
|
Live state of one bucket. A table is N buckets on one node; flattening to a
|
||||||
|
single number hides the one hot bucket that is usually why someone opened
|
||||||
|
this endpoint.
|
||||||
|
|
||||||
|
## Properties
|
||||||
|
|
||||||
|
### compacting
|
||||||
|
|
||||||
|
```ts
|
||||||
|
compacting: boolean;
|
||||||
|
```
|
||||||
|
|
||||||
|
Whether a pass owns this bucket's compaction latch right now. Says *a*
|
||||||
|
driver is running, not *whose*, and the latch is held from dispatch —
|
||||||
|
including while the pass queues for a pod-wide compactor permit. Read it
|
||||||
|
as "do not pile on", never as "mine is progressing".
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### currentGeneration
|
||||||
|
|
||||||
|
```ts
|
||||||
|
currentGeneration: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
The generation the active memtable will become.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### generations
|
||||||
|
|
||||||
|
```ts
|
||||||
|
generations: GenerationStats[];
|
||||||
|
```
|
||||||
|
|
||||||
|
Flushed L0 generations not yet merged into the base table.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### manifestVersion
|
||||||
|
|
||||||
|
```ts
|
||||||
|
manifestVersion: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
Version of the shard manifest these numbers were read from.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### memtables?
|
||||||
|
|
||||||
|
```ts
|
||||||
|
optional memtables: MemtableStats[];
|
||||||
|
```
|
||||||
|
|
||||||
|
Oldest first, active last. Absent for a `"Sealed"` bucket, whose
|
||||||
|
in-memory state is torn down.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### replayAfterWalEntryPosition
|
||||||
|
|
||||||
|
```ts
|
||||||
|
replayAfterWalEntryPosition: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
WAL position replay resumes from.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### shardId
|
||||||
|
|
||||||
|
```ts
|
||||||
|
shardId: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
The shard this bucket writes.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### status
|
||||||
|
|
||||||
|
```ts
|
||||||
|
status: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
`"Active"` or `"Sealed"` (drop-table 2PC in flight).
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### walEntryPositionLastSeen
|
||||||
|
|
||||||
|
```ts
|
||||||
|
walEntryPositionLastSeen: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
Highest WAL position the writer has seen. The difference against
|
||||||
|
`replayAfterWalEntryPosition` is the WAL lag.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### writerEpoch
|
||||||
|
|
||||||
|
```ts
|
||||||
|
writerEpoch: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
Epoch of the writer that currently owns the shard.
|
||||||
@@ -0,0 +1,40 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / GenerationStats
|
||||||
|
|
||||||
|
# Interface: GenerationStats
|
||||||
|
|
||||||
|
One flushed L0 generation.
|
||||||
|
|
||||||
|
## Properties
|
||||||
|
|
||||||
|
### bytes
|
||||||
|
|
||||||
|
```ts
|
||||||
|
bytes: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
On-disk size of the generation.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### generation
|
||||||
|
|
||||||
|
```ts
|
||||||
|
generation: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
The generation number. Increases as memtables are sealed into L0.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### rows?
|
||||||
|
|
||||||
|
```ts
|
||||||
|
optional rows: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
Present only when `includeGenerationRows` was requested. Off by default
|
||||||
|
because each count opens an uncached Lance dataset.
|
||||||
@@ -0,0 +1,66 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / JobDescription
|
||||||
|
|
||||||
|
# Interface: JobDescription
|
||||||
|
|
||||||
|
A described job from `Connection.getJob`.
|
||||||
|
|
||||||
|
## Properties
|
||||||
|
|
||||||
|
### creationMs
|
||||||
|
|
||||||
|
```ts
|
||||||
|
creationMs: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
When the job was created, in milliseconds since the epoch.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### failure?
|
||||||
|
|
||||||
|
```ts
|
||||||
|
optional failure: JobFailureInfo;
|
||||||
|
```
|
||||||
|
|
||||||
|
Why the job failed, when the job is failed and the server reports a
|
||||||
|
reason.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### jobId
|
||||||
|
|
||||||
|
```ts
|
||||||
|
jobId: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### jobType
|
||||||
|
|
||||||
|
```ts
|
||||||
|
jobType: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### specJson?
|
||||||
|
|
||||||
|
```ts
|
||||||
|
optional specJson: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
The job-type-specific specification as a JSON string, when present.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### state
|
||||||
|
|
||||||
|
```ts
|
||||||
|
state: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
Lifecycle state: "running", "finished", "failed", or "cancelled".
|
||||||
@@ -0,0 +1,33 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / JobFailureInfo
|
||||||
|
|
||||||
|
# Interface: JobFailureInfo
|
||||||
|
|
||||||
|
The server's account of why a job failed.
|
||||||
|
|
||||||
|
## Properties
|
||||||
|
|
||||||
|
### message?
|
||||||
|
|
||||||
|
```ts
|
||||||
|
optional message: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### phase?
|
||||||
|
|
||||||
|
```ts
|
||||||
|
optional phase: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### retryable?
|
||||||
|
|
||||||
|
```ts
|
||||||
|
optional retryable: boolean;
|
||||||
|
```
|
||||||
@@ -0,0 +1,58 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / JobInfo
|
||||||
|
|
||||||
|
# Interface: JobInfo
|
||||||
|
|
||||||
|
A row from `Connection.listJobs`: one server-side job.
|
||||||
|
|
||||||
|
## Properties
|
||||||
|
|
||||||
|
### createdAtMillis
|
||||||
|
|
||||||
|
```ts
|
||||||
|
createdAtMillis: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
When the job was created, in milliseconds since the epoch.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### jobId
|
||||||
|
|
||||||
|
```ts
|
||||||
|
jobId: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
The job id -- what `Connection.getJob` and `Connection.cancelJob`
|
||||||
|
accept.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### jobType
|
||||||
|
|
||||||
|
```ts
|
||||||
|
jobType: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### state
|
||||||
|
|
||||||
|
```ts
|
||||||
|
state: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
Lifecycle state: "running", "finished", "failed", or "cancelled".
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### table
|
||||||
|
|
||||||
|
```ts
|
||||||
|
table: string;
|
||||||
|
```
|
||||||
|
|
||||||
|
The table the job runs against, without URI or namespace.
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / LsmStats
|
||||||
|
|
||||||
|
# Interface: LsmStats
|
||||||
|
|
||||||
|
Live per-bucket LSM state, as returned by `Table#getLsmStats`.
|
||||||
|
|
||||||
|
Nothing here is derived: sums and differences (total L0 bytes, WAL lag) are
|
||||||
|
the caller's to compute.
|
||||||
|
|
||||||
|
## Properties
|
||||||
|
|
||||||
|
### buckets
|
||||||
|
|
||||||
|
```ts
|
||||||
|
buckets: BucketStats[];
|
||||||
|
```
|
||||||
|
|
||||||
|
One entry per bucket backing this table.
|
||||||
@@ -34,7 +34,9 @@ Bucket and identity variants: the sharding column.
|
|||||||
optional maintainedIndexes: string[];
|
optional maintainedIndexes: string[];
|
||||||
```
|
```
|
||||||
|
|
||||||
Names of indexes the MemWAL should keep up to date during writes.
|
Indexes the MemWAL keeps up to date. Omit to maintain every supported
|
||||||
|
index, resolved on install — a snapshot, so indexes created later are not
|
||||||
|
maintained. Pass `[]` for none.
|
||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,60 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / MemtableStats
|
||||||
|
|
||||||
|
# Interface: MemtableStats
|
||||||
|
|
||||||
|
One in-memory memtable.
|
||||||
|
|
||||||
|
## Properties
|
||||||
|
|
||||||
|
### batches
|
||||||
|
|
||||||
|
```ts
|
||||||
|
batches: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
Record batches currently buffered.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### bytes
|
||||||
|
|
||||||
|
```ts
|
||||||
|
bytes: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
Estimated in-memory size.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### generation
|
||||||
|
|
||||||
|
```ts
|
||||||
|
generation: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
The generation this memtable will become once sealed.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### indexes
|
||||||
|
|
||||||
|
```ts
|
||||||
|
indexes: string[];
|
||||||
|
```
|
||||||
|
|
||||||
|
Names of the indexes this memtable carries. An absent name is the whole
|
||||||
|
answer to "why is my fresh-tier search on that column brute-force".
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### rows
|
||||||
|
|
||||||
|
```ts
|
||||||
|
rows: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
Rows currently buffered.
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / RefreshColumnResult
|
||||||
|
|
||||||
|
# Interface: RefreshColumnResult
|
||||||
|
|
||||||
|
## Properties
|
||||||
|
|
||||||
|
### rowsFilled
|
||||||
|
|
||||||
|
```ts
|
||||||
|
rowsFilled: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### version
|
||||||
|
|
||||||
|
```ts
|
||||||
|
version: number;
|
||||||
|
```
|
||||||
@@ -44,4 +44,7 @@ The number of rows in the table
|
|||||||
totalBytes: number;
|
totalBytes: number;
|
||||||
```
|
```
|
||||||
|
|
||||||
The total number of bytes in the table
|
The total size, in bytes, of the table's data files, index files, and
|
||||||
|
overlay files
|
||||||
|
|
||||||
|
Read from the manifest, so this excludes deletion files and manifests.
|
||||||
|
|||||||
+167
-51
@@ -26,6 +26,18 @@ is also an [asynchronous API client](#connections-asynchronous).
|
|||||||
|
|
||||||
::: lancedb.db.DBConnection
|
::: lancedb.db.DBConnection
|
||||||
|
|
||||||
|
::: lancedb.Session
|
||||||
|
|
||||||
|
## Namespaces (Synchronous)
|
||||||
|
|
||||||
|
A namespace-backed connection resolves tables through a
|
||||||
|
[Lance namespace](https://lance-format.github.io/lance-namespace/) service instead of
|
||||||
|
listing a storage directory.
|
||||||
|
|
||||||
|
::: lancedb.connect_namespace
|
||||||
|
|
||||||
|
::: lancedb.namespace.LanceNamespaceDBConnection
|
||||||
|
|
||||||
## Tables (Synchronous)
|
## Tables (Synchronous)
|
||||||
|
|
||||||
::: lancedb.table.Table
|
::: lancedb.table.Table
|
||||||
@@ -34,8 +46,62 @@ is also an [asynchronous API client](#connections-asynchronous).
|
|||||||
|
|
||||||
::: lancedb.table.FragmentSummaryStats
|
::: lancedb.table.FragmentSummaryStats
|
||||||
|
|
||||||
|
::: lancedb.table.TableStatistics
|
||||||
|
|
||||||
::: lancedb.table.Tags
|
::: lancedb.table.Tags
|
||||||
|
|
||||||
|
::: lancedb.table.Branches
|
||||||
|
|
||||||
|
::: lancedb.LsmWriteSpec
|
||||||
|
|
||||||
|
## Functions and Jobs
|
||||||
|
|
||||||
|
::: lancedb.functions.FunctionArtifact
|
||||||
|
|
||||||
|
::: lancedb.functions.FunctionParameter
|
||||||
|
|
||||||
|
::: lancedb.functions.FunctionResultField
|
||||||
|
|
||||||
|
::: lancedb.functions.FunctionOutput
|
||||||
|
|
||||||
|
::: lancedb.functions.FunctionSignature
|
||||||
|
|
||||||
|
::: lancedb.functions.PythonEnvironmentSpec
|
||||||
|
|
||||||
|
::: lancedb.functions.udf
|
||||||
|
|
||||||
|
::: lancedb.functions.UdfDefinition
|
||||||
|
|
||||||
|
::: lancedb.functions.FunctionRegistrationRequest
|
||||||
|
|
||||||
|
::: lancedb.functions.FunctionArtifactRequest
|
||||||
|
|
||||||
|
::: lancedb.functions.FunctionArtifactContent
|
||||||
|
|
||||||
|
::: lancedb.functions.PythonAdapterSpec
|
||||||
|
|
||||||
|
::: lancedb.functions.FunctionVersion
|
||||||
|
|
||||||
|
::: lancedb.functions.PythonRuntimeSpec
|
||||||
|
|
||||||
|
::: lancedb.functions.FunctionVersionRef
|
||||||
|
|
||||||
|
::: lancedb.functions.ApplicationInput
|
||||||
|
|
||||||
|
::: lancedb.functions.FunctionApplication
|
||||||
|
|
||||||
|
::: lancedb.functions.InputBinding
|
||||||
|
|
||||||
|
::: lancedb.functions.OutputMapping
|
||||||
|
|
||||||
|
::: lancedb.functions.FunctionBinding
|
||||||
|
|
||||||
|
::: lancedb.functions.RefreshColumnResult
|
||||||
|
|
||||||
|
::: lancedb.job.Job
|
||||||
|
|
||||||
|
::: lancedb.job.AsyncJob
|
||||||
|
|
||||||
## Expressions
|
## Expressions
|
||||||
|
|
||||||
Type-safe expression builder for filters and projections. Use these instead
|
Type-safe expression builder for filters and projections. Use these instead
|
||||||
@@ -62,29 +128,46 @@ of raw SQL strings with [where][lancedb.query.LanceQueryBuilder.where] and
|
|||||||
|
|
||||||
::: lancedb.query.LanceHybridQueryBuilder
|
::: lancedb.query.LanceHybridQueryBuilder
|
||||||
|
|
||||||
|
::: lancedb.query.LanceEmptyQueryBuilder
|
||||||
|
|
||||||
|
::: lancedb.query.LanceTakeQueryBuilder
|
||||||
|
|
||||||
|
## Full text queries
|
||||||
|
|
||||||
|
Structured full text queries can be passed to
|
||||||
|
[Table.search][lancedb.table.Table.search] or
|
||||||
|
[AsyncTable.search][lancedb.table.AsyncTable.search] in place of a query string,
|
||||||
|
and combined with [BooleanQuery][lancedb.query.BooleanQuery].
|
||||||
|
|
||||||
|
::: lancedb.query.FullTextQuery
|
||||||
|
|
||||||
|
::: lancedb.query.MatchQuery
|
||||||
|
|
||||||
|
::: lancedb.query.PhraseQuery
|
||||||
|
|
||||||
|
::: lancedb.query.BoostQuery
|
||||||
|
|
||||||
|
::: lancedb.query.MultiMatchQuery
|
||||||
|
|
||||||
|
::: lancedb.query.BooleanQuery
|
||||||
|
|
||||||
|
::: lancedb.query.FullTextOperator
|
||||||
|
|
||||||
|
::: lancedb.query.Occur
|
||||||
|
|
||||||
## Embeddings
|
## Embeddings
|
||||||
|
|
||||||
::: lancedb.embeddings.registry.EmbeddingFunctionRegistry
|
::: lancedb.embeddings
|
||||||
|
options:
|
||||||
::: lancedb.embeddings.base.EmbeddingFunctionConfig
|
show_root_heading: false
|
||||||
|
show_root_toc_entry: false
|
||||||
::: lancedb.embeddings.base.EmbeddingFunction
|
|
||||||
|
|
||||||
::: lancedb.embeddings.base.TextEmbeddingFunction
|
|
||||||
|
|
||||||
::: lancedb.embeddings.sentence_transformers.SentenceTransformerEmbeddings
|
|
||||||
|
|
||||||
::: lancedb.embeddings.openai.OpenAIEmbeddings
|
|
||||||
|
|
||||||
::: lancedb.embeddings.open_clip.OpenClipEmbeddings
|
|
||||||
|
|
||||||
## Remote configuration
|
## Remote configuration
|
||||||
|
|
||||||
::: lancedb.remote.ClientConfig
|
::: lancedb.remote
|
||||||
|
options:
|
||||||
::: lancedb.remote.TimeoutConfig
|
show_root_heading: false
|
||||||
|
show_root_toc_entry: false
|
||||||
::: lancedb.remote.RetryConfig
|
|
||||||
|
|
||||||
## Context
|
## Context
|
||||||
|
|
||||||
@@ -118,11 +201,27 @@ The same option is available on `lancedb.tokenize(...)` and the deprecated
|
|||||||
```python
|
```python
|
||||||
import lancedb
|
import lancedb
|
||||||
|
|
||||||
tokens = list(lancedb.tokenize("acme makes searchable data",
|
tokens = list(
|
||||||
custom_stop_words=["acme"]))
|
lancedb.tokenize("acme makes searchable data", custom_stop_words=["acme"])
|
||||||
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
::: lancedb.index.FTS
|
::: lancedb.tokenize
|
||||||
|
|
||||||
|
::: lancedb.FtsToken
|
||||||
|
|
||||||
|
## Blobs
|
||||||
|
|
||||||
|
Blob columns store large binary values out of line so they can be read lazily
|
||||||
|
instead of being materialized with the rest of the row.
|
||||||
|
|
||||||
|
::: lancedb.blob
|
||||||
|
|
||||||
|
::: lancedb.BlobType
|
||||||
|
|
||||||
|
::: lancedb._blob.BlobFile
|
||||||
|
options:
|
||||||
|
show_root_full_path: false
|
||||||
|
|
||||||
## Utilities
|
## Utilities
|
||||||
|
|
||||||
@@ -130,6 +229,14 @@ tokens = list(lancedb.tokenize("acme makes searchable data",
|
|||||||
|
|
||||||
::: lancedb.merge.LanceMergeInsertBuilder
|
::: lancedb.merge.LanceMergeInsertBuilder
|
||||||
|
|
||||||
|
::: lancedb.otel.instrument_lancedb_metrics
|
||||||
|
|
||||||
|
## Exceptions
|
||||||
|
|
||||||
|
::: lancedb.exceptions.MissingValueError
|
||||||
|
|
||||||
|
::: lancedb.exceptions.MissingColumnError
|
||||||
|
|
||||||
## Integrations
|
## Integrations
|
||||||
|
|
||||||
## Pydantic
|
## Pydantic
|
||||||
@@ -138,19 +245,30 @@ tokens = list(lancedb.tokenize("acme makes searchable data",
|
|||||||
|
|
||||||
::: lancedb.pydantic.vector
|
::: lancedb.pydantic.vector
|
||||||
|
|
||||||
|
::: lancedb.pydantic.Vector
|
||||||
|
|
||||||
|
::: lancedb.pydantic.MultiVector
|
||||||
|
|
||||||
::: lancedb.pydantic.LanceModel
|
::: lancedb.pydantic.LanceModel
|
||||||
|
|
||||||
|
## PyTorch
|
||||||
|
|
||||||
|
::: lancedb.streaming.StreamingDataset
|
||||||
|
|
||||||
|
::: lancedb.permutation.permutation_builder
|
||||||
|
|
||||||
|
::: lancedb.permutation.PermutationBuilder
|
||||||
|
|
||||||
|
::: lancedb.permutation.Permutation
|
||||||
|
|
||||||
|
::: lancedb.permutation.Transforms
|
||||||
|
|
||||||
## Reranking
|
## Reranking
|
||||||
|
|
||||||
::: lancedb.rerankers.linear_combination.LinearCombinationReranker
|
::: lancedb.rerankers
|
||||||
|
options:
|
||||||
::: lancedb.rerankers.cohere.CohereReranker
|
show_root_heading: false
|
||||||
|
show_root_toc_entry: false
|
||||||
::: lancedb.rerankers.colbert.ColbertReranker
|
|
||||||
|
|
||||||
::: lancedb.rerankers.cross_encoder.CrossEncoderReranker
|
|
||||||
|
|
||||||
::: lancedb.rerankers.openai.OpenaiReranker
|
|
||||||
|
|
||||||
## Connections (Asynchronous)
|
## Connections (Asynchronous)
|
||||||
|
|
||||||
@@ -161,6 +279,12 @@ can be used to create, list, or open tables.
|
|||||||
|
|
||||||
::: lancedb.db.AsyncConnection
|
::: lancedb.db.AsyncConnection
|
||||||
|
|
||||||
|
## Namespaces (Asynchronous)
|
||||||
|
|
||||||
|
::: lancedb.connect_namespace_async
|
||||||
|
|
||||||
|
::: lancedb.namespace.AsyncLanceNamespaceDBConnection
|
||||||
|
|
||||||
## Tables (Asynchronous)
|
## Tables (Asynchronous)
|
||||||
|
|
||||||
Table hold your actual data as a collection of records / rows.
|
Table hold your actual data as a collection of records / rows.
|
||||||
@@ -169,32 +293,20 @@ Table hold your actual data as a collection of records / rows.
|
|||||||
|
|
||||||
::: lancedb.table.AsyncTags
|
::: lancedb.table.AsyncTags
|
||||||
|
|
||||||
|
::: lancedb.table.AsyncBranches
|
||||||
|
|
||||||
## Indices (Asynchronous)
|
## Indices (Asynchronous)
|
||||||
|
|
||||||
Indices can be created on a table to speed up queries. This section
|
Indices can be created on a table to speed up queries. This section
|
||||||
lists the indices that LanceDb supports.
|
lists the indices that LanceDb supports.
|
||||||
|
|
||||||
::: lancedb.index.BTree
|
::: lancedb.index
|
||||||
|
options:
|
||||||
::: lancedb.index.Bitmap
|
show_root_heading: false
|
||||||
|
show_root_toc_entry: false
|
||||||
::: lancedb.index.LabelList
|
# `lang_mapping` is defined in the module rather than imported, so it is
|
||||||
|
# picked up despite not being in `__all__`. It is an internal lookup table.
|
||||||
::: lancedb.index.FTS
|
filters: ["!^_", "!^lang_mapping$"]
|
||||||
|
|
||||||
::: lancedb.index.IvfPq
|
|
||||||
|
|
||||||
::: lancedb.index.HnswPq
|
|
||||||
|
|
||||||
::: lancedb.index.HnswSq
|
|
||||||
|
|
||||||
::: lancedb.index.IvfFlat
|
|
||||||
|
|
||||||
::: lancedb.index.IvfSq
|
|
||||||
|
|
||||||
::: lancedb.index.IvfRq
|
|
||||||
|
|
||||||
::: lancedb.index.HnswFlat
|
|
||||||
|
|
||||||
::: lancedb.table.IndexStatistics
|
::: lancedb.table.IndexStatistics
|
||||||
|
|
||||||
@@ -222,3 +334,7 @@ rows nearest to a query vector and can be created with the
|
|||||||
::: lancedb.query.AsyncHybridQuery
|
::: lancedb.query.AsyncHybridQuery
|
||||||
options:
|
options:
|
||||||
inherited_members: true
|
inherited_members: true
|
||||||
|
|
||||||
|
::: lancedb.query.AsyncTakeQuery
|
||||||
|
options:
|
||||||
|
inherited_members: true
|
||||||
|
|||||||
@@ -29,6 +29,48 @@ LanceNamespace namespaceClient = LanceDbNamespaceClientBuilder.newBuilder()
|
|||||||
.build();
|
.build();
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## MemWAL LSM write path
|
||||||
|
|
||||||
|
Most table operations reach LanceDB through the `LanceNamespace` above, which is
|
||||||
|
generated from the Lance Namespace specification. The MemWAL LSM routes are not part
|
||||||
|
of that specification, so they are issued through a separate client:
|
||||||
|
|
||||||
|
```java
|
||||||
|
import com.lancedb.LanceDbRestClient;
|
||||||
|
import com.lancedb.LanceDbTableLsm;
|
||||||
|
import com.lancedb.LsmWriteSpec;
|
||||||
|
|
||||||
|
LanceDbRestClient client = LanceDbNamespaceClientBuilder.newBuilder()
|
||||||
|
.apiKey("your_lancedb_cloud_api_key")
|
||||||
|
.database("your_database_name")
|
||||||
|
.buildRestClient();
|
||||||
|
|
||||||
|
LanceDbTableLsm lsm = new LanceDbTableLsm(client, "my_table");
|
||||||
|
|
||||||
|
// Route future merge_insert upserts through the MemWAL, hash-bucketed by `id`.
|
||||||
|
lsm.setLsmWriteSpec(LsmWriteSpec.bucket("id", 16));
|
||||||
|
|
||||||
|
// ... merge_insert traffic ...
|
||||||
|
|
||||||
|
// Converge the fresh tier into the base table.
|
||||||
|
lsm.checkpointLsm();
|
||||||
|
|
||||||
|
// Inspect live per-bucket state.
|
||||||
|
lsm.getLsmStats().ifPresent(stats -> stats.buckets().forEach(bucket ->
|
||||||
|
System.out.println(bucket.shardId() + ": " + bucket.generations().size() + " L0 generations")));
|
||||||
|
|
||||||
|
client.close();
|
||||||
|
```
|
||||||
|
|
||||||
|
`maintainedIndexes` is tri-state, and the null default is the opposite of what a Java
|
||||||
|
reader usually expects:
|
||||||
|
|
||||||
|
| Value | Meaning |
|
||||||
|
| --- | --- |
|
||||||
|
| unset (null) | Maintain **every** index the MemWAL can, resolved on install |
|
||||||
|
| `Collections.emptyList()` | Maintain **none** |
|
||||||
|
| `Arrays.asList("id_idx")` | Maintain exactly those |
|
||||||
|
|
||||||
## Development
|
## Development
|
||||||
|
|
||||||
Build:
|
Build:
|
||||||
|
|||||||
@@ -8,7 +8,7 @@
|
|||||||
<parent>
|
<parent>
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-parent</artifactId>
|
<artifactId>lancedb-parent</artifactId>
|
||||||
<version>0.37.1-beta.0</version>
|
<version>0.38.0-beta.3</version>
|
||||||
<relativePath>../pom.xml</relativePath>
|
<relativePath>../pom.xml</relativePath>
|
||||||
</parent>
|
</parent>
|
||||||
|
|
||||||
@@ -33,6 +33,20 @@
|
|||||||
<artifactId>arrow-memory-netty</artifactId>
|
<artifactId>arrow-memory-netty</artifactId>
|
||||||
</dependency>
|
</dependency>
|
||||||
|
|
||||||
|
<!-- Transport for the LanceDB routes outside the Lance Namespace spec.
|
||||||
|
Versions match what lance-namespace-apache-client resolves to. -->
|
||||||
|
<dependency>
|
||||||
|
<groupId>org.apache.httpcomponents.client5</groupId>
|
||||||
|
<artifactId>httpclient5</artifactId>
|
||||||
|
<version>5.2.1</version>
|
||||||
|
</dependency>
|
||||||
|
|
||||||
|
<dependency>
|
||||||
|
<groupId>com.fasterxml.jackson.core</groupId>
|
||||||
|
<artifactId>jackson-databind</artifactId>
|
||||||
|
<version>2.17.1</version>
|
||||||
|
</dependency>
|
||||||
|
|
||||||
<dependency>
|
<dependency>
|
||||||
<groupId>org.junit.jupiter</groupId>
|
<groupId>org.junit.jupiter</groupId>
|
||||||
<artifactId>junit-jupiter</artifactId>
|
<artifactId>junit-jupiter</artifactId>
|
||||||
|
|||||||
@@ -0,0 +1,194 @@
|
|||||||
|
/*
|
||||||
|
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
* you may not use this file except in compliance with the License.
|
||||||
|
* You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
package com.lancedb;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.Collections;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Optional;
|
||||||
|
import java.util.OptionalLong;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Live state of one bucket. A table is N buckets on one node; flattening to a single number hides
|
||||||
|
* the one hot bucket that is usually why someone opened this endpoint.
|
||||||
|
*/
|
||||||
|
public class BucketStats {
|
||||||
|
private static final String CONTEXT = "bucket stats";
|
||||||
|
|
||||||
|
private final String shardId;
|
||||||
|
private final String status;
|
||||||
|
private final long writerEpoch;
|
||||||
|
private final long manifestVersion;
|
||||||
|
private final long currentGeneration;
|
||||||
|
private final long replayAfterWalEntryPosition;
|
||||||
|
private final long walEntryPositionLastSeen;
|
||||||
|
private final List<GenerationStats> generations;
|
||||||
|
private final boolean compacting;
|
||||||
|
private final List<MemtableStats> memtables;
|
||||||
|
|
||||||
|
BucketStats(
|
||||||
|
String shardId,
|
||||||
|
String status,
|
||||||
|
long writerEpoch,
|
||||||
|
long manifestVersion,
|
||||||
|
long currentGeneration,
|
||||||
|
long replayAfterWalEntryPosition,
|
||||||
|
long walEntryPositionLastSeen,
|
||||||
|
List<GenerationStats> generations,
|
||||||
|
boolean compacting,
|
||||||
|
List<MemtableStats> memtables) {
|
||||||
|
this.shardId = shardId;
|
||||||
|
this.status = status;
|
||||||
|
this.writerEpoch = writerEpoch;
|
||||||
|
this.manifestVersion = manifestVersion;
|
||||||
|
this.currentGeneration = currentGeneration;
|
||||||
|
this.replayAfterWalEntryPosition = replayAfterWalEntryPosition;
|
||||||
|
this.walEntryPositionLastSeen = walEntryPositionLastSeen;
|
||||||
|
this.generations = Collections.unmodifiableList(generations);
|
||||||
|
this.compacting = compacting;
|
||||||
|
this.memtables = memtables == null ? null : Collections.unmodifiableList(memtables);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The shard this bucket writes. */
|
||||||
|
public String shardId() {
|
||||||
|
return shardId;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** {@code "Active"} or {@code "Sealed"} (drop-table 2PC in flight). */
|
||||||
|
public String status() {
|
||||||
|
return status;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Epoch of the writer that currently owns the shard. */
|
||||||
|
public long writerEpoch() {
|
||||||
|
return writerEpoch;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Version of the shard manifest these numbers were read from. */
|
||||||
|
public long manifestVersion() {
|
||||||
|
return manifestVersion;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The generation the active memtable will become. */
|
||||||
|
public long currentGeneration() {
|
||||||
|
return currentGeneration;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** WAL position replay resumes from. */
|
||||||
|
public long replayAfterWalEntryPosition() {
|
||||||
|
return replayAfterWalEntryPosition;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Highest WAL position the writer has seen. The difference against {@link
|
||||||
|
* #replayAfterWalEntryPosition()} is the WAL lag.
|
||||||
|
*/
|
||||||
|
public long walEntryPositionLastSeen() {
|
||||||
|
return walEntryPositionLastSeen;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Flushed L0 generations not yet merged into the base table. */
|
||||||
|
public List<GenerationStats> generations() {
|
||||||
|
return generations;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Whether a pass owns this bucket's compaction latch right now. Says <em>a</em> driver is
|
||||||
|
* running, not <em>whose</em>, and the latch is held from dispatch — including while the pass
|
||||||
|
* queues for a pod-wide compactor permit. Read it as "do not pile on", never as "mine is
|
||||||
|
* progressing".
|
||||||
|
*/
|
||||||
|
public boolean compacting() {
|
||||||
|
return compacting;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Oldest first, active last. Empty for a {@code "Sealed"} bucket, whose state is torn down. */
|
||||||
|
public Optional<List<MemtableStats>> memtables() {
|
||||||
|
return Optional.ofNullable(memtables);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The newest flushed generation, or empty when L0 is empty. */
|
||||||
|
OptionalLong newestGeneration() {
|
||||||
|
OptionalLong newest = OptionalLong.empty();
|
||||||
|
for (GenerationStats generation : generations) {
|
||||||
|
if (!newest.isPresent() || generation.generation() > newest.getAsLong()) {
|
||||||
|
newest = OptionalLong.of(generation.generation());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return newest;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* How many generations at or below {@code target} are still in L0.
|
||||||
|
*
|
||||||
|
* <p>A count, not a boolean: one pass drains a bounded prefix rather than the whole target set,
|
||||||
|
* so a boolean would read as "no progress" for every pass but the last. Compaction drains
|
||||||
|
* oldest-first, so this decreases monotonically.
|
||||||
|
*/
|
||||||
|
long outstandingGenerations(long target) {
|
||||||
|
long count = 0;
|
||||||
|
for (GenerationStats generation : generations) {
|
||||||
|
if (generation.generation() <= target) {
|
||||||
|
count++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
|
||||||
|
static BucketStats fromJson(JsonNode node) {
|
||||||
|
JsonFields.requiredObject(node, CONTEXT);
|
||||||
|
List<GenerationStats> generations = new ArrayList<GenerationStats>();
|
||||||
|
for (JsonNode generation : JsonFields.requiredArray(node, "generations", CONTEXT)) {
|
||||||
|
generations.add(GenerationStats.fromJson(generation));
|
||||||
|
}
|
||||||
|
|
||||||
|
JsonNode memtablesNode = JsonFields.optionalArray(node, "memtables", CONTEXT);
|
||||||
|
List<MemtableStats> memtables = null;
|
||||||
|
if (memtablesNode != null) {
|
||||||
|
memtables = new ArrayList<MemtableStats>();
|
||||||
|
for (JsonNode memtable : memtablesNode) {
|
||||||
|
memtables.add(MemtableStats.fromJson(memtable));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return new BucketStats(
|
||||||
|
JsonFields.requiredText(node, "shard_id", CONTEXT),
|
||||||
|
JsonFields.requiredText(node, "status", CONTEXT),
|
||||||
|
JsonFields.requiredLong(node, "writer_epoch", CONTEXT),
|
||||||
|
JsonFields.requiredLong(node, "manifest_version", CONTEXT),
|
||||||
|
JsonFields.requiredLong(node, "current_generation", CONTEXT),
|
||||||
|
JsonFields.requiredLong(node, "replay_after_wal_entry_position", CONTEXT),
|
||||||
|
JsonFields.requiredLong(node, "wal_entry_position_last_seen", CONTEXT),
|
||||||
|
generations,
|
||||||
|
JsonFields.requiredBoolean(node, "compacting", CONTEXT),
|
||||||
|
memtables);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public String toString() {
|
||||||
|
return "BucketStats{shardId="
|
||||||
|
+ shardId
|
||||||
|
+ ", status="
|
||||||
|
+ status
|
||||||
|
+ ", currentGeneration="
|
||||||
|
+ currentGeneration
|
||||||
|
+ ", generations="
|
||||||
|
+ generations
|
||||||
|
+ ", compacting="
|
||||||
|
+ compacting
|
||||||
|
+ "}";
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,64 @@
|
|||||||
|
/*
|
||||||
|
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
* you may not use this file except in compliance with the License.
|
||||||
|
* You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
package com.lancedb;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
|
||||||
|
import java.util.OptionalLong;
|
||||||
|
|
||||||
|
/** One flushed L0 generation. */
|
||||||
|
public class GenerationStats {
|
||||||
|
private static final String CONTEXT = "generation stats";
|
||||||
|
|
||||||
|
private final long generation;
|
||||||
|
private final long bytes;
|
||||||
|
private final Long rows;
|
||||||
|
|
||||||
|
GenerationStats(long generation, long bytes, Long rows) {
|
||||||
|
this.generation = generation;
|
||||||
|
this.bytes = bytes;
|
||||||
|
this.rows = rows;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The generation number. Increases as memtables are sealed into L0. */
|
||||||
|
public long generation() {
|
||||||
|
return generation;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** On-disk size of the generation. */
|
||||||
|
public long bytes() {
|
||||||
|
return bytes;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Rows in this generation, present only when {@code includeGenerationRows} was requested. Off by
|
||||||
|
* default because each count opens an uncached Lance dataset.
|
||||||
|
*/
|
||||||
|
public OptionalLong rows() {
|
||||||
|
return rows == null ? OptionalLong.empty() : OptionalLong.of(rows);
|
||||||
|
}
|
||||||
|
|
||||||
|
static GenerationStats fromJson(JsonNode node) {
|
||||||
|
JsonFields.requiredObject(node, CONTEXT);
|
||||||
|
return new GenerationStats(
|
||||||
|
JsonFields.requiredLong(node, "generation", CONTEXT),
|
||||||
|
JsonFields.requiredLong(node, "bytes", CONTEXT),
|
||||||
|
JsonFields.optionalLong(node, "rows", CONTEXT));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public String toString() {
|
||||||
|
return "GenerationStats{generation=" + generation + ", bytes=" + bytes + ", rows=" + rows + "}";
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,109 @@
|
|||||||
|
/*
|
||||||
|
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
* you may not use this file except in compliance with the License.
|
||||||
|
* You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
package com.lancedb;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Strict readers for decoding LanceDB JSON responses.
|
||||||
|
*
|
||||||
|
* <p>Every reader fails closed: a missing, null, or wrong-typed field throws rather than
|
||||||
|
* defaulting. That mirrors the serde decoding the Rust client applies to the same payloads in
|
||||||
|
* {@code rust/lancedb/src/table/lsm_stats.rs}, where a required field has no default and a
|
||||||
|
* malformed response is an error rather than a zero.
|
||||||
|
*
|
||||||
|
* <p>The alternative — Jackson's {@code path()}, which yields a missing node that reads as an empty
|
||||||
|
* array or a zero — is unsafe here because {@link LanceDbTableLsm#checkpointLsm()} decides
|
||||||
|
* convergence from these numbers. A defaulted {@code generations} array is indistinguishable from a
|
||||||
|
* drained one, so a malformed response would report a checkpoint that never happened.
|
||||||
|
*/
|
||||||
|
final class JsonFields {
|
||||||
|
private JsonFields() {}
|
||||||
|
|
||||||
|
/** The node itself, once confirmed to be a JSON object. */
|
||||||
|
static JsonNode requiredObject(JsonNode node, String context) {
|
||||||
|
if (node == null || !node.isObject()) {
|
||||||
|
throw new IllegalStateException(context + " is not a JSON object: " + node);
|
||||||
|
}
|
||||||
|
return node;
|
||||||
|
}
|
||||||
|
|
||||||
|
static String requiredText(JsonNode owner, String field, String context) {
|
||||||
|
JsonNode value = required(owner, field, context);
|
||||||
|
if (!value.isTextual()) {
|
||||||
|
throw new IllegalStateException(fieldIs(context, field, "a string", value));
|
||||||
|
}
|
||||||
|
return value.asText();
|
||||||
|
}
|
||||||
|
|
||||||
|
static long requiredLong(JsonNode owner, String field, String context) {
|
||||||
|
JsonNode value = required(owner, field, context);
|
||||||
|
if (!value.isIntegralNumber()) {
|
||||||
|
throw new IllegalStateException(fieldIs(context, field, "an integer", value));
|
||||||
|
}
|
||||||
|
return value.asLong();
|
||||||
|
}
|
||||||
|
|
||||||
|
static boolean requiredBoolean(JsonNode owner, String field, String context) {
|
||||||
|
JsonNode value = required(owner, field, context);
|
||||||
|
if (!value.isBoolean()) {
|
||||||
|
throw new IllegalStateException(fieldIs(context, field, "a boolean", value));
|
||||||
|
}
|
||||||
|
return value.asBoolean();
|
||||||
|
}
|
||||||
|
|
||||||
|
static JsonNode requiredArray(JsonNode owner, String field, String context) {
|
||||||
|
JsonNode value = required(owner, field, context);
|
||||||
|
if (!value.isArray()) {
|
||||||
|
throw new IllegalStateException(fieldIs(context, field, "an array", value));
|
||||||
|
}
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Null when the field is absent or JSON null, mirroring a serde {@code Option}. */
|
||||||
|
static Long optionalLong(JsonNode owner, String field, String context) {
|
||||||
|
JsonNode value = owner.get(field);
|
||||||
|
if (value == null || value.isNull()) {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
if (!value.isIntegralNumber()) {
|
||||||
|
throw new IllegalStateException(fieldIs(context, field, "an integer", value));
|
||||||
|
}
|
||||||
|
return value.asLong();
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Null when the field is absent or JSON null, mirroring a serde {@code Option}. */
|
||||||
|
static JsonNode optionalArray(JsonNode owner, String field, String context) {
|
||||||
|
JsonNode value = owner.get(field);
|
||||||
|
if (value == null || value.isNull()) {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
if (!value.isArray()) {
|
||||||
|
throw new IllegalStateException(fieldIs(context, field, "an array", value));
|
||||||
|
}
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
private static JsonNode required(JsonNode owner, String field, String context) {
|
||||||
|
JsonNode value = owner.get(field);
|
||||||
|
if (value == null || value.isNull()) {
|
||||||
|
throw new IllegalStateException(context + " is missing required field '" + field + "'");
|
||||||
|
}
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
private static String fieldIs(String context, String field, String expected, JsonNode value) {
|
||||||
|
return context + " field '" + field + "' is not " + expected + ": " + value;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -136,29 +136,48 @@ public class LanceDbNamespaceClientBuilder {
|
|||||||
* @throws IllegalStateException if required parameters are missing
|
* @throws IllegalStateException if required parameters are missing
|
||||||
*/
|
*/
|
||||||
public LanceNamespace build() {
|
public LanceNamespace build() {
|
||||||
// Validate required fields
|
validate();
|
||||||
|
|
||||||
|
// Build configuration map
|
||||||
|
Map<String, String> config = new HashMap<>(additionalConfig);
|
||||||
|
config.put("header.x-lancedb-database", database);
|
||||||
|
config.put("header.x-api-key", apiKey);
|
||||||
|
config.put("uri", resolveUri());
|
||||||
|
|
||||||
|
return LanceNamespace.connect("rest", config, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Build a {@link LanceDbRestClient} for the same endpoint.
|
||||||
|
*
|
||||||
|
* <p>Needed only for LanceDB routes that the Lance Namespace specification does not cover — the
|
||||||
|
* MemWAL LSM write path, reached through {@link LanceDbTableLsm}. Every other table operation
|
||||||
|
* belongs on the {@link LanceNamespace} from {@link #build()}.
|
||||||
|
*
|
||||||
|
* <p>The returned client owns an HTTP connection pool; close it when you are done with it.
|
||||||
|
*
|
||||||
|
* @return A configured LanceDbRestClient
|
||||||
|
* @throws IllegalStateException if required parameters are missing
|
||||||
|
*/
|
||||||
|
public LanceDbRestClient buildRestClient() {
|
||||||
|
validate();
|
||||||
|
return new LanceDbRestClient(resolveUri(), apiKey, database);
|
||||||
|
}
|
||||||
|
|
||||||
|
private void validate() {
|
||||||
if (apiKey == null) {
|
if (apiKey == null) {
|
||||||
throw new IllegalStateException("API key is required");
|
throw new IllegalStateException("API key is required");
|
||||||
}
|
}
|
||||||
if (database == null) {
|
if (database == null) {
|
||||||
throw new IllegalStateException("Database is required");
|
throw new IllegalStateException("Database is required");
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Build configuration map
|
/** The custom endpoint when set, else the LanceDB Cloud URL for this database and region. */
|
||||||
Map<String, String> config = new HashMap<>(additionalConfig);
|
private String resolveUri() {
|
||||||
config.put("header.x-lancedb-database", database);
|
|
||||||
config.put("header.x-api-key", apiKey);
|
|
||||||
|
|
||||||
// Determine base URL
|
|
||||||
String uri;
|
|
||||||
if (endpoint.isPresent()) {
|
if (endpoint.isPresent()) {
|
||||||
uri = endpoint.get();
|
return endpoint.get();
|
||||||
} else {
|
|
||||||
String effectiveRegion = region.orElse(DEFAULT_REGION);
|
|
||||||
uri = String.format(CLOUD_URL_PATTERN, database, effectiveRegion);
|
|
||||||
}
|
}
|
||||||
config.put("uri", uri);
|
return String.format(CLOUD_URL_PATTERN, database, region.orElse(DEFAULT_REGION));
|
||||||
|
|
||||||
return LanceNamespace.connect("rest", config, null);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,119 @@
|
|||||||
|
/*
|
||||||
|
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
* you may not use this file except in compliance with the License.
|
||||||
|
* You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
package com.lancedb;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
|
import org.apache.hc.client5.http.classic.methods.HttpPost;
|
||||||
|
import org.apache.hc.client5.http.impl.classic.CloseableHttpClient;
|
||||||
|
import org.apache.hc.client5.http.impl.classic.HttpClients;
|
||||||
|
import org.apache.hc.core5.http.ContentType;
|
||||||
|
import org.apache.hc.core5.http.io.entity.EntityUtils;
|
||||||
|
import org.apache.hc.core5.http.io.entity.StringEntity;
|
||||||
|
|
||||||
|
import java.io.Closeable;
|
||||||
|
import java.io.IOException;
|
||||||
|
import java.io.UncheckedIOException;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Minimal HTTP client for LanceDB Cloud and Enterprise routes that the Lance Namespace
|
||||||
|
* specification does not cover.
|
||||||
|
*
|
||||||
|
* <p>Most table operations reach LanceDB through {@link org.lance.namespace.LanceNamespace}, which
|
||||||
|
* is generated from the namespace spec. A handful of routes — the MemWAL LSM write path in
|
||||||
|
* particular — are served by the same endpoint but are not part of that spec, so they are issued
|
||||||
|
* directly here. See {@link LanceDbTableLsm}.
|
||||||
|
*
|
||||||
|
* <p>Obtain one from {@link LanceDbNamespaceClientBuilder#buildRestClient()}.
|
||||||
|
*/
|
||||||
|
public class LanceDbRestClient implements Closeable {
|
||||||
|
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||||
|
|
||||||
|
private final String baseUri;
|
||||||
|
private final String apiKey;
|
||||||
|
private final String database;
|
||||||
|
private final CloseableHttpClient http;
|
||||||
|
|
||||||
|
LanceDbRestClient(String baseUri, String apiKey, String database) {
|
||||||
|
this.baseUri = baseUri.endsWith("/") ? baseUri.substring(0, baseUri.length() - 1) : baseUri;
|
||||||
|
this.apiKey = apiKey;
|
||||||
|
this.database = database;
|
||||||
|
// Automatic retries off, deliberately. The default strategy retries 429 and 503 —
|
||||||
|
// exactly the two statuses LanceDbTableLsm.checkpointLsm() acts on — which would
|
||||||
|
// silently double its explicit retry budget and would also retry compact_lsm in
|
||||||
|
// place, where the loop is designed to fall through to a fresh stats poll instead.
|
||||||
|
// The checkpoint loop owns the 421/429/503 transitions; the transport must not.
|
||||||
|
this.http = HttpClients.custom().disableAutomaticRetries().build();
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* POST {@code path}, sending {@code body} as JSON when it is non-null.
|
||||||
|
*
|
||||||
|
* @param path Absolute request path, beginning with {@code /}.
|
||||||
|
* @param body Object to serialize as the request body, or null to send no body.
|
||||||
|
* @return The parsed response body, or null when the response carried no content.
|
||||||
|
* @throws HttpException if the server returned a non-2xx status.
|
||||||
|
*/
|
||||||
|
public JsonNode post(String path, Object body) {
|
||||||
|
HttpPost request = new HttpPost(baseUri + path);
|
||||||
|
request.setHeader("x-api-key", apiKey);
|
||||||
|
request.setHeader("x-lancedb-database", database);
|
||||||
|
try {
|
||||||
|
if (body != null) {
|
||||||
|
request.setEntity(
|
||||||
|
new StringEntity(MAPPER.writeValueAsString(body), ContentType.APPLICATION_JSON));
|
||||||
|
}
|
||||||
|
return http.execute(
|
||||||
|
request,
|
||||||
|
response -> {
|
||||||
|
String text =
|
||||||
|
response.getEntity() == null ? "" : EntityUtils.toString(response.getEntity());
|
||||||
|
int status = response.getCode();
|
||||||
|
if (status < 200 || status >= 300) {
|
||||||
|
throw new HttpException(status, "LanceDB request to " + path + " failed: " + text);
|
||||||
|
}
|
||||||
|
return text.isEmpty() ? null : MAPPER.readTree(text);
|
||||||
|
});
|
||||||
|
} catch (IOException e) {
|
||||||
|
throw new UncheckedIOException("LanceDB request to " + path + " failed", e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() throws IOException {
|
||||||
|
http.close();
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A non-2xx response.
|
||||||
|
*
|
||||||
|
* <p>The status is exposed because callers act on it: {@link LanceDbTableLsm#checkpointLsm()}
|
||||||
|
* treats 429 and 503 as retryable and 421 as a lost node claim.
|
||||||
|
*/
|
||||||
|
public static class HttpException extends RuntimeException {
|
||||||
|
private static final long serialVersionUID = 1L;
|
||||||
|
|
||||||
|
private final int statusCode;
|
||||||
|
|
||||||
|
public HttpException(int statusCode, String message) {
|
||||||
|
super(message);
|
||||||
|
this.statusCode = statusCode;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The HTTP status the failed response carried. */
|
||||||
|
public int statusCode() {
|
||||||
|
return statusCode;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,394 @@
|
|||||||
|
/*
|
||||||
|
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
* you may not use this file except in compliance with the License.
|
||||||
|
* You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
package com.lancedb;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
|
||||||
|
import java.util.HashMap;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Optional;
|
||||||
|
import java.util.OptionalLong;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The MemWAL LSM write path for one LanceDB Cloud or Enterprise table.
|
||||||
|
*
|
||||||
|
* <p>Installing an {@link LsmWriteSpec} routes {@code mergeInsert} upserts through Lance's MemWAL —
|
||||||
|
* an LSM-style append — instead of the standard merge path. Rows land in an in-memory memtable,
|
||||||
|
* seal into L0 generations, and are merged into the base table by compaction.
|
||||||
|
*
|
||||||
|
* <p>These routes are not part of the Lance Namespace specification, so they are issued directly
|
||||||
|
* rather than through {@link org.lance.namespace.LanceNamespace}.
|
||||||
|
*
|
||||||
|
* <pre>{@code
|
||||||
|
* LanceDbRestClient client = LanceDbNamespaceClientBuilder.newBuilder()
|
||||||
|
* .apiKey("your_lancedb_cloud_api_key")
|
||||||
|
* .database("your_database_name")
|
||||||
|
* .buildRestClient();
|
||||||
|
*
|
||||||
|
* LanceDbTableLsm lsm = new LanceDbTableLsm(client, "my_table");
|
||||||
|
* lsm.setLsmWriteSpec(LsmWriteSpec.bucket("id", 16));
|
||||||
|
* // ... merge_insert traffic ...
|
||||||
|
* lsm.checkpointLsm();
|
||||||
|
* }</pre>
|
||||||
|
*/
|
||||||
|
public class LanceDbTableLsm {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Interval between {@code get_lsm_stats} polls during a checkpoint. One interval is roughly one
|
||||||
|
* compaction pass, the granularity at which the answer can change.
|
||||||
|
*/
|
||||||
|
private static final long POLL_INTERVAL_MS = 5_000L;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Cap on re-issues from {@code flushLsm} after a 421, so a crash-looping node cannot turn flush →
|
||||||
|
* compact → 421 → flush into a spin.
|
||||||
|
*
|
||||||
|
* <p>Deliberately not shared with {@link #MAX_RETRIES}: a claim that keeps evaporating is a
|
||||||
|
* broken node, while contention is routine and wants a real budget.
|
||||||
|
*/
|
||||||
|
private static final int MAX_REISSUES = 3;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Retryable faults tolerated on a <em>single</em> request, reset on every success — scattered
|
||||||
|
* contention across a long checkpoint must not accumulate toward a cap.
|
||||||
|
*/
|
||||||
|
private static final int MAX_RETRIES = 8;
|
||||||
|
|
||||||
|
private static final long RETRY_BACKOFF_BASE_MS = 100L;
|
||||||
|
private static final long RETRY_BACKOFF_MAX_MS = 5_000L;
|
||||||
|
|
||||||
|
private final LanceDbRestClient client;
|
||||||
|
private final String tableIdentifier;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Bind the LSM routes for one table.
|
||||||
|
*
|
||||||
|
* @param client Transport for the LanceDB endpoint.
|
||||||
|
* @param tableIdentifier The table's full identifier, {@code $}-delimited when it sits inside a
|
||||||
|
* namespace, such as {@code analytics$events}.
|
||||||
|
*/
|
||||||
|
public LanceDbTableLsm(LanceDbRestClient client, String tableIdentifier) {
|
||||||
|
if (client == null) {
|
||||||
|
throw new IllegalArgumentException("Client cannot be null");
|
||||||
|
}
|
||||||
|
if (tableIdentifier == null || tableIdentifier.trim().isEmpty()) {
|
||||||
|
throw new IllegalArgumentException("Table identifier cannot be null or empty");
|
||||||
|
}
|
||||||
|
this.client = client;
|
||||||
|
this.tableIdentifier = tableIdentifier;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Install an {@link LsmWriteSpec} on this table, selecting the MemWAL LSM write path for future
|
||||||
|
* {@code mergeInsert} calls.
|
||||||
|
*
|
||||||
|
* <p>All variants require the table to have an unenforced primary key; bucket sharding
|
||||||
|
* additionally requires it to be the single column being bucketed.
|
||||||
|
*/
|
||||||
|
public void setLsmWriteSpec(LsmWriteSpec spec) {
|
||||||
|
if (spec == null) {
|
||||||
|
throw new IllegalArgumentException("Spec cannot be null");
|
||||||
|
}
|
||||||
|
client.post(route("set_lsm_write_spec"), spec.toRequestBody());
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Remove the {@link LsmWriteSpec} from this table, reverting to the standard {@code mergeInsert}
|
||||||
|
* write path.
|
||||||
|
*
|
||||||
|
* <p>Errors if no spec is currently set.
|
||||||
|
*/
|
||||||
|
public void unsetLsmWriteSpec() {
|
||||||
|
client.post(route("unset_lsm_write_spec"), null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Read the {@link LsmWriteSpec} currently installed on this table.
|
||||||
|
*
|
||||||
|
* <p>Empty when the LSM write path is not enabled. The returned spec mirrors what was installed,
|
||||||
|
* except that {@link LsmWriteSpec#maintainedIndexes()} always reports the concrete list resolved
|
||||||
|
* when the spec was set — a null selection never round-trips.
|
||||||
|
*/
|
||||||
|
public Optional<LsmWriteSpec> getLsmWriteSpec() {
|
||||||
|
JsonNode response = client.post(route("get_lsm_write_spec"), null);
|
||||||
|
if (response == null || !response.hasNonNull("lsm_write_spec")) {
|
||||||
|
return Optional.empty();
|
||||||
|
}
|
||||||
|
return Optional.of(LsmWriteSpec.fromJson(response.get("lsm_write_spec")));
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Seal every bucket's active memtable into a new L0 generation.
|
||||||
|
*
|
||||||
|
* <p>Returns once the seal is committed. Sealing an empty memtable is a no-op, so this is safe to
|
||||||
|
* call repeatedly.
|
||||||
|
*/
|
||||||
|
public void flushLsm() {
|
||||||
|
client.post(route("flush_lsm"), null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Trigger a background L0 → base compaction pass per bucket.
|
||||||
|
*
|
||||||
|
* <p>Returns once the passes are <em>dispatched</em>, not once they finish — watch {@link
|
||||||
|
* #getLsmStats}, or use {@link #checkpointLsm} to wait for convergence.
|
||||||
|
*/
|
||||||
|
public void compactLsm() {
|
||||||
|
client.post(route("compact_lsm"), null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Read live per-bucket LSM state.
|
||||||
|
*
|
||||||
|
* <p>Answers "how far behind is my fresh tier", "which bucket is hot", and "why is my fresh-tier
|
||||||
|
* vector search brute-force". Mutates no table state.
|
||||||
|
*
|
||||||
|
* <p>Empty only when the LSM write path is not enabled — that is, when the server sends an absent
|
||||||
|
* or null {@code lsm_stats}. A stats object that is present is decoded strictly, and a malformed
|
||||||
|
* one throws rather than decoding to something empty, because {@link #checkpointLsm} reads
|
||||||
|
* convergence out of these numbers and cannot tell a defaulted array from a drained one.
|
||||||
|
*
|
||||||
|
* @param includeGenerationRows Also count rows per L0 generation. Off by default because each
|
||||||
|
* count opens an uncached Lance dataset.
|
||||||
|
* @throws IllegalStateException if the response is absent or does not decode.
|
||||||
|
*/
|
||||||
|
public Optional<LsmStats> getLsmStats(boolean includeGenerationRows) {
|
||||||
|
Map<String, Object> body = new LinkedHashMap<String, Object>();
|
||||||
|
body.put("include_generation_rows", includeGenerationRows);
|
||||||
|
JsonNode response = client.post(route("get_lsm_stats"), body);
|
||||||
|
if (response == null) {
|
||||||
|
throw new IllegalStateException("get_lsm_stats returned an empty response body");
|
||||||
|
}
|
||||||
|
JsonNode stats = response.get("lsm_stats");
|
||||||
|
if (stats == null || stats.isNull()) {
|
||||||
|
return Optional.empty();
|
||||||
|
}
|
||||||
|
return Optional.of(LsmStats.fromJson(stats));
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Equivalent to {@code getLsmStats(false)}. */
|
||||||
|
public Optional<LsmStats> getLsmStats() {
|
||||||
|
return getLsmStats(false);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Converge this table's LSM write path into its base table.
|
||||||
|
*
|
||||||
|
* <p>Seals once, fixes a target watermark from the resulting L0, then triggers compaction and
|
||||||
|
* polls until that L0 is gone. The target set is fixed at the start, so generations created
|
||||||
|
* <em>during</em> the checkpoint are ignored — that is what lets it terminate under write load,
|
||||||
|
* and what makes it best-effort: it converges the fresh tier as of some instant. Idempotent,
|
||||||
|
* abandonable at any point, safe on a cadence.
|
||||||
|
*
|
||||||
|
* <p>The loop runs here, not on the server: {@link #compactLsm} dispatches a pass and returns, so
|
||||||
|
* nothing holds a socket and a client can vanish mid-operation with nothing to reconcile.
|
||||||
|
* Completion is read from generation numbers in the shard manifest — durable state, unlike a
|
||||||
|
* count in a compact response, which a concurrent write invalidates.
|
||||||
|
*
|
||||||
|
* <p>No liveness bound — the caller owns the deadline. The compactor pool is shared across
|
||||||
|
* tables, so a checkpoint queued behind unrelated work looks exactly like one that is merging.
|
||||||
|
*/
|
||||||
|
public void checkpointLsm() {
|
||||||
|
for (int reissue = 0; reissue <= MAX_REISSUES; reissue++) {
|
||||||
|
// The seal turns everything written before this call into a generation, so the
|
||||||
|
// watermark has to be read after it. Idempotent: sealing an empty memtable is a
|
||||||
|
// no-op, so a re-issue does not churn empty generations.
|
||||||
|
if (issueVoid(this::flushLsm)) {
|
||||||
|
backoff(reissue);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
Attempt<Optional<LsmStats>> stats = issue(() -> getLsmStats(false));
|
||||||
|
if (stats.lostClaim) {
|
||||||
|
backoff(reissue);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (!stats.value.isPresent()) {
|
||||||
|
// Not WAL-backed; flushLsm would have errored first but for a race.
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
Map<String, Long> targets = newestGenerations(stats.value.get());
|
||||||
|
if (targets.isEmpty()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (drainToTargets(targets)) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
backoff(reissue);
|
||||||
|
}
|
||||||
|
throw new IllegalStateException(
|
||||||
|
"checkpointLsm: the owning node kept losing its claim; re-issued from flush the maximum "
|
||||||
|
+ "number of times");
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Trigger and poll until no bucket holds a generation at or below its target.
|
||||||
|
*
|
||||||
|
* @return true when the drain finished, false when the table needs re-claiming from flush.
|
||||||
|
*/
|
||||||
|
private boolean drainToTargets(Map<String, Long> targets) {
|
||||||
|
while (true) {
|
||||||
|
Attempt<Optional<LsmStats>> stats = issue(() -> getLsmStats(false));
|
||||||
|
if (stats.lostClaim) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (!stats.value.isPresent()) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// `compacting` is the bucket's compaction latch, held from dispatch until the pass
|
||||||
|
// ends — including while it waits on a pod-wide permit. So it answers one question
|
||||||
|
// only: do not pile on. Buckets with nothing outstanding are skipped, not counted
|
||||||
|
// as idle.
|
||||||
|
long outstanding = 0;
|
||||||
|
boolean allCompacting = true;
|
||||||
|
for (BucketStats bucket : stats.value.get().buckets()) {
|
||||||
|
Long target = targets.get(bucket.shardId());
|
||||||
|
if (target == null) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
long remaining = bucket.outstandingGenerations(target);
|
||||||
|
if (remaining > 0) {
|
||||||
|
outstanding += remaining;
|
||||||
|
allCompacting &= bucket.compacting();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (outstanding == 0) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!allCompacting) {
|
||||||
|
try {
|
||||||
|
compactLsm();
|
||||||
|
} catch (LanceDbRestClient.HttpException e) {
|
||||||
|
if (isLostClaim(e)) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (!isRetryable(e)) {
|
||||||
|
throw e;
|
||||||
|
}
|
||||||
|
// A 429 here means the server could latch no bucket at all, which the poll
|
||||||
|
// above already handles. Not retried in place: the latch it would contend for
|
||||||
|
// is the one doing the work, so fall through and re-read — POLL_INTERVAL_MS is
|
||||||
|
// the backoff.
|
||||||
|
}
|
||||||
|
}
|
||||||
|
sleep(POLL_INTERVAL_MS);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The newest generation held by each bucket, skipping buckets holding none. */
|
||||||
|
private static Map<String, Long> newestGenerations(LsmStats stats) {
|
||||||
|
Map<String, Long> targets = new HashMap<String, Long>();
|
||||||
|
for (BucketStats bucket : stats.buckets()) {
|
||||||
|
OptionalLong newest = bucket.newestGeneration();
|
||||||
|
if (newest.isPresent()) {
|
||||||
|
targets.put(bucket.shardId(), newest.getAsLong());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return targets;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 429 (latch held, pool saturated, or the pod replaying its WAL) and 503 (a draining node, or a
|
||||||
|
* proxy between here and it).
|
||||||
|
*/
|
||||||
|
private static boolean isRetryable(LanceDbRestClient.HttpException e) {
|
||||||
|
return e.statusCode() == 429 || e.statusCode() == 503;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 421: the owning node holds no claim. Only {@code flush} re-claims and replays, so this cannot
|
||||||
|
* be retried in place — the caller has to start over.
|
||||||
|
*/
|
||||||
|
private static boolean isLostClaim(LanceDbRestClient.HttpException e) {
|
||||||
|
return e.statusCode() == 421;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Issue one LSM request, retrying in place while the fault is retryable.
|
||||||
|
*
|
||||||
|
* <p>The two recoverable faults have separate budgets: contention clears on its own and retries
|
||||||
|
* here against {@link #MAX_RETRIES}, while a 421 needs {@code flush} to re-claim, which only the
|
||||||
|
* caller can drive.
|
||||||
|
*
|
||||||
|
* <p>An exhausted budget propagates the last error as itself rather than a synthesized one — "429
|
||||||
|
* after nine tries" beats "checkpoint failed".
|
||||||
|
*/
|
||||||
|
private static <T> Attempt<T> issue(Call<T> call) {
|
||||||
|
int retries = 0;
|
||||||
|
while (true) {
|
||||||
|
try {
|
||||||
|
return new Attempt<T>(call.run(), false);
|
||||||
|
} catch (LanceDbRestClient.HttpException e) {
|
||||||
|
if (isLostClaim(e)) {
|
||||||
|
return new Attempt<T>(null, true);
|
||||||
|
}
|
||||||
|
if (!isRetryable(e) || retries >= MAX_RETRIES) {
|
||||||
|
throw e;
|
||||||
|
}
|
||||||
|
backoff(retries);
|
||||||
|
retries++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** {@link #issue} for a call with no return value. Returns true when the claim was lost. */
|
||||||
|
private static boolean issueVoid(Runnable call) {
|
||||||
|
return issue(
|
||||||
|
() -> {
|
||||||
|
call.run();
|
||||||
|
return Boolean.TRUE;
|
||||||
|
})
|
||||||
|
.lostClaim;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Sleep before re-issuing a retryable request. Doubles up to {@link #RETRY_BACKOFF_MAX_MS}. */
|
||||||
|
private static void backoff(int attempt) {
|
||||||
|
long delay = RETRY_BACKOFF_BASE_MS << Math.min(attempt, 8);
|
||||||
|
sleep(Math.min(delay, RETRY_BACKOFF_MAX_MS));
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void sleep(long millis) {
|
||||||
|
try {
|
||||||
|
Thread.sleep(millis);
|
||||||
|
} catch (InterruptedException e) {
|
||||||
|
Thread.currentThread().interrupt();
|
||||||
|
throw new IllegalStateException("Interrupted while waiting on the LSM checkpoint", e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private String route(String operation) {
|
||||||
|
return "/v1/table/" + tableIdentifier + "/" + operation + "/";
|
||||||
|
}
|
||||||
|
|
||||||
|
/** What one LSM request produced: its value, or word that the owning node holds no claim. */
|
||||||
|
private static final class Attempt<T> {
|
||||||
|
private final T value;
|
||||||
|
private final boolean lostClaim;
|
||||||
|
|
||||||
|
private Attempt(T value, boolean lostClaim) {
|
||||||
|
this.value = value;
|
||||||
|
this.lostClaim = lostClaim;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@FunctionalInterface
|
||||||
|
private interface Call<T> {
|
||||||
|
T run();
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,56 @@
|
|||||||
|
/*
|
||||||
|
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
* you may not use this file except in compliance with the License.
|
||||||
|
* You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
package com.lancedb;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.Collections;
|
||||||
|
import java.util.List;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Live per-bucket LSM state, as returned by {@link LanceDbTableLsm#getLsmStats()}.
|
||||||
|
*
|
||||||
|
* <p>Nothing here is derived: sums and differences (total L0 bytes, WAL lag) are the caller's to
|
||||||
|
* compute. There is no "LSM is off" shape — that case is an empty {@link java.util.Optional},
|
||||||
|
* because a stats object of zeros would read as measurements.
|
||||||
|
*/
|
||||||
|
public class LsmStats {
|
||||||
|
private static final String CONTEXT = "lsm stats";
|
||||||
|
|
||||||
|
private final List<BucketStats> buckets;
|
||||||
|
|
||||||
|
LsmStats(List<BucketStats> buckets) {
|
||||||
|
this.buckets = Collections.unmodifiableList(buckets);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** One entry per bucket. */
|
||||||
|
public List<BucketStats> buckets() {
|
||||||
|
return buckets;
|
||||||
|
}
|
||||||
|
|
||||||
|
static LsmStats fromJson(JsonNode node) {
|
||||||
|
JsonFields.requiredObject(node, CONTEXT);
|
||||||
|
List<BucketStats> buckets = new ArrayList<BucketStats>();
|
||||||
|
for (JsonNode bucket : JsonFields.requiredArray(node, "buckets", CONTEXT)) {
|
||||||
|
buckets.add(BucketStats.fromJson(bucket));
|
||||||
|
}
|
||||||
|
return new LsmStats(buckets);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public String toString() {
|
||||||
|
return "LsmStats{buckets=" + buckets + "}";
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,260 @@
|
|||||||
|
/*
|
||||||
|
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
* you may not use this file except in compliance with the License.
|
||||||
|
* You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
package com.lancedb;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.Collections;
|
||||||
|
import java.util.HashMap;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Specification selecting Lance's MemWAL LSM-style write path for {@code mergeInsert}.
|
||||||
|
*
|
||||||
|
* <p>Construct via {@link #bucket}, {@link #identity}, or {@link #unsharded}, then optionally chain
|
||||||
|
* {@link #withMaintainedIndexes} and {@link #withWriterConfigDefaults}. Install it with {@link
|
||||||
|
* LanceDbTableLsm#setLsmWriteSpec} and remove it with {@link LanceDbTableLsm#unsetLsmWriteSpec}.
|
||||||
|
*
|
||||||
|
* <p>This is deliberately not {@code org.lance.memwal.InitializeMemWalParams}. That type is Lance's
|
||||||
|
* own, and its maintained-index default is the opposite of this one: it defaults to maintaining
|
||||||
|
* <em>nothing</em>, while a fresh spec here maintains <em>every</em> index. It also cannot express
|
||||||
|
* the null that asks the server to resolve the set.
|
||||||
|
*/
|
||||||
|
public class LsmWriteSpec {
|
||||||
|
|
||||||
|
/** How writes are routed to MemWAL shards. */
|
||||||
|
public enum Sharding {
|
||||||
|
/** Hash-bucket writes by a scalar column. */
|
||||||
|
BUCKET("bucket"),
|
||||||
|
/** Shard by the raw value of a scalar column. */
|
||||||
|
IDENTITY("identity"),
|
||||||
|
/** Route every write to a single shard. */
|
||||||
|
UNSHARDED("unsharded");
|
||||||
|
|
||||||
|
private final String wireName;
|
||||||
|
|
||||||
|
Sharding(String wireName) {
|
||||||
|
this.wireName = wireName;
|
||||||
|
}
|
||||||
|
|
||||||
|
String wireName() {
|
||||||
|
return wireName;
|
||||||
|
}
|
||||||
|
|
||||||
|
static Sharding fromWireName(String name) {
|
||||||
|
for (Sharding s : values()) {
|
||||||
|
if (s.wireName.equals(name)) {
|
||||||
|
return s;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
throw new IllegalArgumentException("Unknown sharding mode: " + name);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private final Sharding sharding;
|
||||||
|
private final String column;
|
||||||
|
private final Integer numBuckets;
|
||||||
|
private final List<String> maintainedIndexes;
|
||||||
|
private final Map<String, String> writerConfigDefaults;
|
||||||
|
|
||||||
|
private LsmWriteSpec(
|
||||||
|
Sharding sharding,
|
||||||
|
String column,
|
||||||
|
Integer numBuckets,
|
||||||
|
List<String> maintainedIndexes,
|
||||||
|
Map<String, String> writerConfigDefaults) {
|
||||||
|
this.sharding = sharding;
|
||||||
|
this.column = column;
|
||||||
|
this.numBuckets = numBuckets;
|
||||||
|
this.maintainedIndexes = maintainedIndexes;
|
||||||
|
this.writerConfigDefaults = writerConfigDefaults;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Hash-bucket sharding by a scalar column, maintaining every index on the table.
|
||||||
|
*
|
||||||
|
* <p>Iceberg-compatible Murmur3-x86-32 (seed 0) is used, so each row's {@code bucket(column,
|
||||||
|
* numBuckets)} value is stable across processes.
|
||||||
|
*
|
||||||
|
* @param column A non-nested column with a supported scalar type.
|
||||||
|
* @param numBuckets The number of buckets, in {@code [1, 1024]}.
|
||||||
|
*/
|
||||||
|
public static LsmWriteSpec bucket(String column, int numBuckets) {
|
||||||
|
if (column == null || column.trim().isEmpty()) {
|
||||||
|
throw new IllegalArgumentException("Column cannot be null or empty");
|
||||||
|
}
|
||||||
|
return new LsmWriteSpec(
|
||||||
|
Sharding.BUCKET, column, numBuckets, null, new HashMap<String, String>());
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Identity sharding — shard by the raw value of {@code column} — maintaining every index on the
|
||||||
|
* table.
|
||||||
|
*
|
||||||
|
* <p>{@code column} must be a deterministic function of the unenforced primary key: every row
|
||||||
|
* with a given primary key must always produce the same {@code column} value, or upserts of that
|
||||||
|
* key can land in different shards and a stale version can win.
|
||||||
|
*/
|
||||||
|
public static LsmWriteSpec identity(String column) {
|
||||||
|
if (column == null || column.trim().isEmpty()) {
|
||||||
|
throw new IllegalArgumentException("Column cannot be null or empty");
|
||||||
|
}
|
||||||
|
return new LsmWriteSpec(Sharding.IDENTITY, column, null, null, new HashMap<String, String>());
|
||||||
|
}
|
||||||
|
|
||||||
|
/** No sharding — every write goes to a single MemWAL shard — maintaining every index. */
|
||||||
|
public static LsmWriteSpec unsharded() {
|
||||||
|
return new LsmWriteSpec(Sharding.UNSHARDED, null, null, null, new HashMap<String, String>());
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Set the indexes the MemWAL keeps up to date as rows are appended.
|
||||||
|
*
|
||||||
|
* <p>Pass {@code null} — the default for a fresh spec — to maintain every index the MemWAL can,
|
||||||
|
* resolved when the spec is installed. That is a snapshot: indexes created later are not
|
||||||
|
* maintained until the spec is unset and set again. Pass an empty list to maintain none.
|
||||||
|
*
|
||||||
|
* <p>Note that {@code null} and the empty list mean opposite things here.
|
||||||
|
*/
|
||||||
|
public LsmWriteSpec withMaintainedIndexes(List<String> maintainedIndexes) {
|
||||||
|
return new LsmWriteSpec(
|
||||||
|
sharding,
|
||||||
|
column,
|
||||||
|
numBuckets,
|
||||||
|
maintainedIndexes == null ? null : new ArrayList<String>(maintainedIndexes),
|
||||||
|
writerConfigDefaults);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Set default {@code ShardWriter} configuration recorded in the MemWAL index.
|
||||||
|
*
|
||||||
|
* <p>A sparse override map — only the keys you set are recorded. Recognized keys include {@code
|
||||||
|
* durable_write}, {@code max_wal_buffer_size}, {@code max_memtable_size}, {@code
|
||||||
|
* max_memtable_rows}, {@code max_memtable_batches}, {@code manifest_scan_batch_size}, {@code
|
||||||
|
* max_unflushed_memtable_bytes}, and {@code enable_memtable}. Duration knobs carry an {@code _ms}
|
||||||
|
* suffix, such as {@code max_wal_flush_interval_ms}.
|
||||||
|
*/
|
||||||
|
public LsmWriteSpec withWriterConfigDefaults(Map<String, String> writerConfigDefaults) {
|
||||||
|
if (writerConfigDefaults == null) {
|
||||||
|
throw new IllegalArgumentException("writerConfigDefaults cannot be null");
|
||||||
|
}
|
||||||
|
return new LsmWriteSpec(
|
||||||
|
sharding,
|
||||||
|
column,
|
||||||
|
numBuckets,
|
||||||
|
maintainedIndexes,
|
||||||
|
new HashMap<String, String>(writerConfigDefaults));
|
||||||
|
}
|
||||||
|
|
||||||
|
/** How writes are routed to shards. */
|
||||||
|
public Sharding sharding() {
|
||||||
|
return sharding;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The sharding column for {@link Sharding#BUCKET} and {@link Sharding#IDENTITY}, else null. */
|
||||||
|
public String column() {
|
||||||
|
return column;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The bucket count for {@link Sharding#BUCKET}, else null. */
|
||||||
|
public Integer numBuckets() {
|
||||||
|
return numBuckets;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The indexes the MemWAL maintains, or null to have the server resolve every maintainable index
|
||||||
|
* on install. An empty list means none.
|
||||||
|
*/
|
||||||
|
public List<String> maintainedIndexes() {
|
||||||
|
return maintainedIndexes == null ? null : Collections.unmodifiableList(maintainedIndexes);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Default {@code ShardWriter} configuration recorded in the MemWAL index. */
|
||||||
|
public Map<String, String> writerConfigDefaults() {
|
||||||
|
return Collections.unmodifiableMap(writerConfigDefaults);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Render this spec as the {@code set_lsm_write_spec} request body. */
|
||||||
|
Map<String, Object> toRequestBody() {
|
||||||
|
Map<String, Object> shardingBody = new LinkedHashMap<String, Object>();
|
||||||
|
shardingBody.put("mode", sharding.wireName());
|
||||||
|
if (column != null) {
|
||||||
|
shardingBody.put("column", column);
|
||||||
|
}
|
||||||
|
if (numBuckets != null) {
|
||||||
|
shardingBody.put("num_buckets", numBuckets);
|
||||||
|
}
|
||||||
|
|
||||||
|
Map<String, Object> body = new LinkedHashMap<String, Object>();
|
||||||
|
body.put("sharding", shardingBody);
|
||||||
|
// Null is meaningful: it asks the server to resolve every maintainable index.
|
||||||
|
body.put("maintained_indexes", maintainedIndexes);
|
||||||
|
body.put("writer_config_defaults", writerConfigDefaults);
|
||||||
|
return body;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Rebuild a spec from a {@code get_lsm_write_spec} response body.
|
||||||
|
*
|
||||||
|
* <p>The server always reports a concrete maintained-index list, so a null selection never
|
||||||
|
* round-trips.
|
||||||
|
*/
|
||||||
|
static LsmWriteSpec fromJson(JsonNode node) {
|
||||||
|
JsonNode shardingNode = node.get("sharding");
|
||||||
|
if (shardingNode == null || shardingNode.get("mode") == null) {
|
||||||
|
throw new IllegalStateException("get_lsm_write_spec response has no sharding mode");
|
||||||
|
}
|
||||||
|
Sharding sharding = Sharding.fromWireName(shardingNode.get("mode").asText());
|
||||||
|
|
||||||
|
String column = shardingNode.hasNonNull("column") ? shardingNode.get("column").asText() : null;
|
||||||
|
Integer numBuckets =
|
||||||
|
shardingNode.hasNonNull("num_buckets") ? shardingNode.get("num_buckets").asInt() : null;
|
||||||
|
|
||||||
|
List<String> maintainedIndexes = new ArrayList<String>();
|
||||||
|
JsonNode indexesNode = node.get("maintained_indexes");
|
||||||
|
if (indexesNode != null && indexesNode.isArray()) {
|
||||||
|
for (JsonNode index : indexesNode) {
|
||||||
|
maintainedIndexes.add(index.asText());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Map<String, String> defaults = new HashMap<String, String>();
|
||||||
|
JsonNode defaultsNode = node.get("writer_config_defaults");
|
||||||
|
if (defaultsNode != null && defaultsNode.isObject()) {
|
||||||
|
defaultsNode
|
||||||
|
.fieldNames()
|
||||||
|
.forEachRemaining(name -> defaults.put(name, defaultsNode.get(name).asText()));
|
||||||
|
}
|
||||||
|
|
||||||
|
return new LsmWriteSpec(sharding, column, numBuckets, maintainedIndexes, defaults);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public String toString() {
|
||||||
|
return "LsmWriteSpec{sharding="
|
||||||
|
+ sharding
|
||||||
|
+ ", column="
|
||||||
|
+ column
|
||||||
|
+ ", numBuckets="
|
||||||
|
+ numBuckets
|
||||||
|
+ ", maintainedIndexes="
|
||||||
|
+ maintainedIndexes
|
||||||
|
+ ", writerConfigDefaults="
|
||||||
|
+ writerConfigDefaults
|
||||||
|
+ "}";
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,99 @@
|
|||||||
|
/*
|
||||||
|
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
* you may not use this file except in compliance with the License.
|
||||||
|
* You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
package com.lancedb;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.Collections;
|
||||||
|
import java.util.List;
|
||||||
|
|
||||||
|
/** One in-memory memtable. */
|
||||||
|
public class MemtableStats {
|
||||||
|
private static final String CONTEXT = "memtable stats";
|
||||||
|
|
||||||
|
private final long generation;
|
||||||
|
private final long rows;
|
||||||
|
private final long bytes;
|
||||||
|
private final long batches;
|
||||||
|
private final List<String> indexes;
|
||||||
|
|
||||||
|
MemtableStats(long generation, long rows, long bytes, long batches, List<String> indexes) {
|
||||||
|
this.generation = generation;
|
||||||
|
this.rows = rows;
|
||||||
|
this.bytes = bytes;
|
||||||
|
this.batches = batches;
|
||||||
|
this.indexes = Collections.unmodifiableList(indexes);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The generation this memtable will become once sealed. */
|
||||||
|
public long generation() {
|
||||||
|
return generation;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Rows currently buffered. */
|
||||||
|
public long rows() {
|
||||||
|
return rows;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Estimated in-memory size. */
|
||||||
|
public long bytes() {
|
||||||
|
return bytes;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Record batches currently buffered. */
|
||||||
|
public long batches() {
|
||||||
|
return batches;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Names of the indexes this memtable carries. An absent name is the whole answer to "why is my
|
||||||
|
* fresh-tier search on that column brute-force".
|
||||||
|
*/
|
||||||
|
public List<String> indexes() {
|
||||||
|
return indexes;
|
||||||
|
}
|
||||||
|
|
||||||
|
static MemtableStats fromJson(JsonNode node) {
|
||||||
|
JsonFields.requiredObject(node, CONTEXT);
|
||||||
|
List<String> indexes = new ArrayList<String>();
|
||||||
|
for (JsonNode index : JsonFields.requiredArray(node, "indexes", CONTEXT)) {
|
||||||
|
if (!index.isTextual()) {
|
||||||
|
throw new IllegalStateException(CONTEXT + " has a non-string index name: " + index);
|
||||||
|
}
|
||||||
|
indexes.add(index.asText());
|
||||||
|
}
|
||||||
|
return new MemtableStats(
|
||||||
|
JsonFields.requiredLong(node, "generation", CONTEXT),
|
||||||
|
JsonFields.requiredLong(node, "rows", CONTEXT),
|
||||||
|
JsonFields.requiredLong(node, "bytes", CONTEXT),
|
||||||
|
JsonFields.requiredLong(node, "batches", CONTEXT),
|
||||||
|
indexes);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public String toString() {
|
||||||
|
return "MemtableStats{generation="
|
||||||
|
+ generation
|
||||||
|
+ ", rows="
|
||||||
|
+ rows
|
||||||
|
+ ", bytes="
|
||||||
|
+ bytes
|
||||||
|
+ ", batches="
|
||||||
|
+ batches
|
||||||
|
+ ", indexes="
|
||||||
|
+ indexes
|
||||||
|
+ "}";
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,570 @@
|
|||||||
|
/*
|
||||||
|
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
* you may not use this file except in compliance with the License.
|
||||||
|
* You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
package com.lancedb;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
|
import com.sun.net.httpserver.HttpServer;
|
||||||
|
import org.junit.jupiter.api.AfterEach;
|
||||||
|
import org.junit.jupiter.api.BeforeEach;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.io.ByteArrayOutputStream;
|
||||||
|
import java.io.IOException;
|
||||||
|
import java.io.InputStream;
|
||||||
|
import java.io.UncheckedIOException;
|
||||||
|
import java.net.InetSocketAddress;
|
||||||
|
import java.nio.charset.StandardCharsets;
|
||||||
|
import java.util.ArrayDeque;
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.Arrays;
|
||||||
|
import java.util.Collections;
|
||||||
|
import java.util.Deque;
|
||||||
|
import java.util.HashMap;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Optional;
|
||||||
|
import java.util.concurrent.ConcurrentHashMap;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.*;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Unit tests for the MemWAL LSM routes, run against a scripted local HTTP server.
|
||||||
|
*
|
||||||
|
* <p>The wire assertions mirror the Rust mocked-endpoint tests in {@code
|
||||||
|
* rust/lancedb/src/remote/table.rs}, which are the contract these routes have to match.
|
||||||
|
*/
|
||||||
|
public class LanceDbTableLsmTest {
|
||||||
|
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||||
|
|
||||||
|
private HttpServer server;
|
||||||
|
private LanceDbRestClient client;
|
||||||
|
private LanceDbTableLsm lsm;
|
||||||
|
|
||||||
|
private final List<String> requestPaths = Collections.synchronizedList(new ArrayList<String>());
|
||||||
|
private final List<String> requestBodies = Collections.synchronizedList(new ArrayList<String>());
|
||||||
|
private final Map<String, Deque<Reply>> replies = new ConcurrentHashMap<String, Deque<Reply>>();
|
||||||
|
|
||||||
|
@BeforeEach
|
||||||
|
public void setUp() throws IOException {
|
||||||
|
start();
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Tear down and restart the scripted server, for a test that scripts several exchanges. */
|
||||||
|
private void setUpFresh() {
|
||||||
|
try {
|
||||||
|
client.close();
|
||||||
|
server.stop(0);
|
||||||
|
requestPaths.clear();
|
||||||
|
requestBodies.clear();
|
||||||
|
replies.clear();
|
||||||
|
start();
|
||||||
|
} catch (IOException e) {
|
||||||
|
throw new UncheckedIOException(e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private void start() throws IOException {
|
||||||
|
server = HttpServer.create(new InetSocketAddress("127.0.0.1", 0), 0);
|
||||||
|
server.createContext(
|
||||||
|
"/",
|
||||||
|
exchange -> {
|
||||||
|
String path = exchange.getRequestURI().getPath();
|
||||||
|
requestPaths.add(path);
|
||||||
|
requestBodies.add(readAll(exchange.getRequestBody()));
|
||||||
|
|
||||||
|
Reply reply = nextReply(path);
|
||||||
|
byte[] out = reply.body.getBytes(StandardCharsets.UTF_8);
|
||||||
|
exchange.sendResponseHeaders(reply.status, out.length == 0 ? -1 : out.length);
|
||||||
|
if (out.length > 0) {
|
||||||
|
exchange.getResponseBody().write(out);
|
||||||
|
}
|
||||||
|
exchange.close();
|
||||||
|
});
|
||||||
|
server.start();
|
||||||
|
|
||||||
|
client =
|
||||||
|
LanceDbNamespaceClientBuilder.newBuilder()
|
||||||
|
.apiKey("test-key")
|
||||||
|
.database("test-db")
|
||||||
|
.endpoint("http://127.0.0.1:" + server.getAddress().getPort())
|
||||||
|
.buildRestClient();
|
||||||
|
lsm = new LanceDbTableLsm(client, "my_table");
|
||||||
|
}
|
||||||
|
|
||||||
|
@AfterEach
|
||||||
|
public void tearDown() throws IOException {
|
||||||
|
client.close();
|
||||||
|
server.stop(0);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ===========================================================================
|
||||||
|
// set / unset / get spec
|
||||||
|
// ===========================================================================
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testSetLsmWriteSpecUnsharded() throws Exception {
|
||||||
|
enqueue("set_lsm_write_spec", 200, "");
|
||||||
|
|
||||||
|
lsm.setLsmWriteSpec(LsmWriteSpec.unsharded());
|
||||||
|
|
||||||
|
assertEquals("/v1/table/my_table/set_lsm_write_spec/", requestPaths.get(0));
|
||||||
|
JsonNode body = MAPPER.readTree(requestBodies.get(0));
|
||||||
|
assertEquals("unsharded", body.get("sharding").get("mode").asText());
|
||||||
|
assertFalse(body.get("sharding").has("column"));
|
||||||
|
assertFalse(body.get("sharding").has("num_buckets"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testSetLsmWriteSpecBucket() throws Exception {
|
||||||
|
enqueue("set_lsm_write_spec", 200, "");
|
||||||
|
|
||||||
|
lsm.setLsmWriteSpec(
|
||||||
|
LsmWriteSpec.bucket("id", 16).withMaintainedIndexes(Arrays.asList("id_idx")));
|
||||||
|
|
||||||
|
JsonNode body = MAPPER.readTree(requestBodies.get(0));
|
||||||
|
assertEquals("bucket", body.get("sharding").get("mode").asText());
|
||||||
|
assertEquals("id", body.get("sharding").get("column").asText());
|
||||||
|
assertEquals(16, body.get("sharding").get("num_buckets").asInt());
|
||||||
|
assertEquals(1, body.get("maintained_indexes").size());
|
||||||
|
assertEquals("id_idx", body.get("maintained_indexes").get(0).asText());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testSetLsmWriteSpecIdentity() throws Exception {
|
||||||
|
enqueue("set_lsm_write_spec", 200, "");
|
||||||
|
|
||||||
|
lsm.setLsmWriteSpec(LsmWriteSpec.identity("tenant"));
|
||||||
|
|
||||||
|
JsonNode body = MAPPER.readTree(requestBodies.get(0));
|
||||||
|
assertEquals("identity", body.get("sharding").get("mode").asText());
|
||||||
|
assertEquals("tenant", body.get("sharding").get("column").asText());
|
||||||
|
assertFalse(body.get("sharding").has("num_buckets"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The tri-state that motivated a LanceDB-owned spec type: a null selection asks the server to
|
||||||
|
* resolve every maintainable index, while an empty list asks for none. They must not collapse.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
public void testMaintainedIndexesNullAndEmptyAreDistinctOnTheWire() throws Exception {
|
||||||
|
enqueue("set_lsm_write_spec", 200, "");
|
||||||
|
|
||||||
|
lsm.setLsmWriteSpec(LsmWriteSpec.unsharded());
|
||||||
|
JsonNode fresh = MAPPER.readTree(requestBodies.get(0));
|
||||||
|
assertTrue(fresh.has("maintained_indexes"), "the key must be present");
|
||||||
|
assertTrue(fresh.get("maintained_indexes").isNull(), "a fresh spec sends null, not []");
|
||||||
|
|
||||||
|
lsm.setLsmWriteSpec(
|
||||||
|
LsmWriteSpec.unsharded().withMaintainedIndexes(Collections.<String>emptyList()));
|
||||||
|
JsonNode none = MAPPER.readTree(requestBodies.get(1));
|
||||||
|
assertTrue(none.get("maintained_indexes").isArray());
|
||||||
|
assertEquals(0, none.get("maintained_indexes").size());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testSetLsmWriteSpecWriterConfigDefaults() throws Exception {
|
||||||
|
enqueue("set_lsm_write_spec", 200, "");
|
||||||
|
|
||||||
|
Map<String, String> defaults = new HashMap<String, String>();
|
||||||
|
defaults.put("max_memtable_rows", "50000");
|
||||||
|
lsm.setLsmWriteSpec(LsmWriteSpec.unsharded().withWriterConfigDefaults(defaults));
|
||||||
|
|
||||||
|
JsonNode body = MAPPER.readTree(requestBodies.get(0));
|
||||||
|
assertEquals("50000", body.get("writer_config_defaults").get("max_memtable_rows").asText());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testUnsetLsmWriteSpec() {
|
||||||
|
enqueue("unset_lsm_write_spec", 200, "");
|
||||||
|
|
||||||
|
lsm.unsetLsmWriteSpec();
|
||||||
|
|
||||||
|
assertEquals("/v1/table/my_table/unset_lsm_write_spec/", requestPaths.get(0));
|
||||||
|
assertEquals("", requestBodies.get(0));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testGetLsmWriteSpec() {
|
||||||
|
enqueue(
|
||||||
|
"get_lsm_write_spec",
|
||||||
|
200,
|
||||||
|
"{\"lsm_write_spec\":{\"sharding\":{\"mode\":\"bucket\",\"column\":\"id\","
|
||||||
|
+ "\"num_buckets\":16},\"maintained_indexes\":[\"id_idx\"],"
|
||||||
|
+ "\"writer_config_defaults\":{\"durable_write\":\"true\"}}}");
|
||||||
|
|
||||||
|
Optional<LsmWriteSpec> spec = lsm.getLsmWriteSpec();
|
||||||
|
|
||||||
|
assertTrue(spec.isPresent());
|
||||||
|
assertEquals(LsmWriteSpec.Sharding.BUCKET, spec.get().sharding());
|
||||||
|
assertEquals("id", spec.get().column());
|
||||||
|
assertEquals(Integer.valueOf(16), spec.get().numBuckets());
|
||||||
|
assertEquals(Arrays.asList("id_idx"), spec.get().maintainedIndexes());
|
||||||
|
assertEquals("true", spec.get().writerConfigDefaults().get("durable_write"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testGetLsmWriteSpecAbsent() {
|
||||||
|
enqueue("get_lsm_write_spec", 200, "{\"lsm_write_spec\":null}");
|
||||||
|
|
||||||
|
assertFalse(lsm.getLsmWriteSpec().isPresent());
|
||||||
|
}
|
||||||
|
|
||||||
|
// ===========================================================================
|
||||||
|
// stats
|
||||||
|
// ===========================================================================
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testGetLsmStats() throws Exception {
|
||||||
|
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 7L, 8L)));
|
||||||
|
|
||||||
|
Optional<LsmStats> got = lsm.getLsmStats(true);
|
||||||
|
|
||||||
|
assertEquals("/v1/table/my_table/get_lsm_stats/", requestPaths.get(0));
|
||||||
|
assertTrue(MAPPER.readTree(requestBodies.get(0)).get("include_generation_rows").asBoolean());
|
||||||
|
assertTrue(got.isPresent());
|
||||||
|
BucketStats decoded = got.get().buckets().get(0);
|
||||||
|
assertEquals("shard-0", decoded.shardId());
|
||||||
|
assertEquals("Active", decoded.status());
|
||||||
|
assertEquals(1, decoded.writerEpoch());
|
||||||
|
assertEquals(2, decoded.manifestVersion());
|
||||||
|
assertEquals(9, decoded.currentGeneration());
|
||||||
|
assertFalse(decoded.compacting());
|
||||||
|
assertEquals(Arrays.asList(7L, 8L), generationNumbers(decoded));
|
||||||
|
assertEquals(1024, decoded.generations().get(0).bytes());
|
||||||
|
assertFalse(decoded.generations().get(0).rows().isPresent(), "rows absent unless requested");
|
||||||
|
assertFalse(decoded.memtables().isPresent(), "absent memtables stay absent");
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The optional fields decode when the server does send them. */
|
||||||
|
@Test
|
||||||
|
public void testGetLsmStatsDecodesOptionalFields() {
|
||||||
|
enqueue(
|
||||||
|
"get_lsm_stats",
|
||||||
|
200,
|
||||||
|
"{\"lsm_stats\":{\"buckets\":[{\"shard_id\":\"shard-0\",\"status\":\"Active\","
|
||||||
|
+ "\"writer_epoch\":1,\"manifest_version\":2,\"current_generation\":9,"
|
||||||
|
+ "\"replay_after_wal_entry_position\":3,\"wal_entry_position_last_seen\":11,"
|
||||||
|
+ "\"generations\":[{\"generation\":7,\"bytes\":1024,\"rows\":42}],"
|
||||||
|
+ "\"compacting\":true,\"memtables\":[{\"generation\":8,\"rows\":5,"
|
||||||
|
+ "\"bytes\":64,\"batches\":2,\"indexes\":[\"id_idx\"]}]}]}}");
|
||||||
|
|
||||||
|
BucketStats decoded = lsm.getLsmStats(true).get().buckets().get(0);
|
||||||
|
|
||||||
|
assertEquals(3, decoded.replayAfterWalEntryPosition());
|
||||||
|
assertEquals(11, decoded.walEntryPositionLastSeen());
|
||||||
|
assertTrue(decoded.compacting());
|
||||||
|
assertEquals(42, decoded.generations().get(0).rows().getAsLong());
|
||||||
|
assertTrue(decoded.memtables().isPresent());
|
||||||
|
MemtableStats memtable = decoded.memtables().get().get(0);
|
||||||
|
assertEquals(8, memtable.generation());
|
||||||
|
assertEquals(5, memtable.rows());
|
||||||
|
assertEquals(64, memtable.bytes());
|
||||||
|
assertEquals(2, memtable.batches());
|
||||||
|
assertEquals(Arrays.asList("id_idx"), memtable.indexes());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testGetLsmStatsAbsentWhenLsmDisabled() {
|
||||||
|
enqueue("get_lsm_stats", 200, "{\"lsm_stats\":null}");
|
||||||
|
|
||||||
|
assertFalse(lsm.getLsmStats().isPresent());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testGetLsmStatsDefaultsToExcludingGenerationRows() throws Exception {
|
||||||
|
enqueue("get_lsm_stats", 200, stats());
|
||||||
|
|
||||||
|
lsm.getLsmStats();
|
||||||
|
|
||||||
|
assertFalse(MAPPER.readTree(requestBodies.get(0)).get("include_generation_rows").asBoolean());
|
||||||
|
}
|
||||||
|
|
||||||
|
// ===========================================================================
|
||||||
|
// flush / compact
|
||||||
|
// ===========================================================================
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testFlushAndCompactRoutes() {
|
||||||
|
enqueue("flush_lsm", 200, "");
|
||||||
|
enqueue("compact_lsm", 200, "");
|
||||||
|
|
||||||
|
lsm.flushLsm();
|
||||||
|
lsm.compactLsm();
|
||||||
|
|
||||||
|
assertEquals("/v1/table/my_table/flush_lsm/", requestPaths.get(0));
|
||||||
|
assertEquals("/v1/table/my_table/compact_lsm/", requestPaths.get(1));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testHttpErrorCarriesStatus() {
|
||||||
|
enqueue("flush_lsm", 404, "no such table");
|
||||||
|
|
||||||
|
LanceDbRestClient.HttpException e =
|
||||||
|
assertThrows(LanceDbRestClient.HttpException.class, () -> lsm.flushLsm());
|
||||||
|
assertEquals(404, e.statusCode());
|
||||||
|
}
|
||||||
|
|
||||||
|
// ===========================================================================
|
||||||
|
// checkpoint
|
||||||
|
// ===========================================================================
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testCheckpointReturnsWhenLsmDisabled() {
|
||||||
|
enqueue("flush_lsm", 200, "");
|
||||||
|
enqueue("get_lsm_stats", 200, "{\"lsm_stats\":null}");
|
||||||
|
|
||||||
|
lsm.checkpointLsm();
|
||||||
|
|
||||||
|
assertEquals(0, countCalls("compact_lsm"), "nothing to compact when the LSM path is off");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testCheckpointReturnsWhenNoGenerationsOutstanding() {
|
||||||
|
enqueue("flush_lsm", 200, "");
|
||||||
|
// A bucket with no L0 generations yields no target, so the drain never starts.
|
||||||
|
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false)));
|
||||||
|
|
||||||
|
lsm.checkpointLsm();
|
||||||
|
|
||||||
|
assertEquals(0, countCalls("compact_lsm"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testCheckpointConvergesOnceTargetGenerationsAreGone() {
|
||||||
|
enqueue("flush_lsm", 200, "");
|
||||||
|
// Watermark read: shard-0 holds generations 7 and 8, so target = 8.
|
||||||
|
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 7L, 8L)));
|
||||||
|
// First drain poll: both still outstanding, nothing compacting -> dispatch a pass.
|
||||||
|
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 7L, 8L)));
|
||||||
|
// Second drain poll: drained past the target -> done.
|
||||||
|
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 9L)));
|
||||||
|
enqueue("compact_lsm", 200, "");
|
||||||
|
|
||||||
|
lsm.checkpointLsm();
|
||||||
|
|
||||||
|
assertEquals(1, countCalls("compact_lsm"), "one pass dispatched");
|
||||||
|
assertEquals(3, countCalls("get_lsm_stats"), "watermark read plus two drain polls");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testCheckpointDoesNotPileOnWhileEveryTargetBucketIsCompacting() {
|
||||||
|
enqueue("flush_lsm", 200, "");
|
||||||
|
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", true, 4L)));
|
||||||
|
// Still compacting on the first poll, so no pass is dispatched; then it drains.
|
||||||
|
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", true, 4L)));
|
||||||
|
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 5L)));
|
||||||
|
|
||||||
|
lsm.checkpointLsm();
|
||||||
|
|
||||||
|
assertEquals(0, countCalls("compact_lsm"), "a latched bucket is left alone");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testCheckpointRetriesFromFlushAfterLostClaim() {
|
||||||
|
// 421 on the watermark read: the node lost its claim, so the whole thing restarts
|
||||||
|
// from flush rather than retrying the read in place.
|
||||||
|
enqueue("flush_lsm", 200, "");
|
||||||
|
enqueue("get_lsm_stats", 421, "no claim");
|
||||||
|
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false)));
|
||||||
|
|
||||||
|
lsm.checkpointLsm();
|
||||||
|
|
||||||
|
assertEquals(2, countCalls("flush_lsm"), "re-issued from flush");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testCheckpointRetriesRetryableStatusInPlace() {
|
||||||
|
enqueue("flush_lsm", 429, "latch held");
|
||||||
|
enqueue("flush_lsm", 200, "");
|
||||||
|
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false)));
|
||||||
|
|
||||||
|
lsm.checkpointLsm();
|
||||||
|
|
||||||
|
assertEquals(2, countCalls("flush_lsm"), "429 retried in place, not re-issued");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testCheckpointPropagatesTerminalStatus() {
|
||||||
|
enqueue("flush_lsm", 400, "bad request");
|
||||||
|
|
||||||
|
LanceDbRestClient.HttpException e =
|
||||||
|
assertThrows(LanceDbRestClient.HttpException.class, () -> lsm.checkpointLsm());
|
||||||
|
assertEquals(400, e.statusCode());
|
||||||
|
assertEquals(1, countCalls("flush_lsm"), "a terminal status is not retried");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
public void testCheckpointGivesUpAfterRepeatedLostClaims() {
|
||||||
|
enqueue("flush_lsm", 421, "no claim");
|
||||||
|
|
||||||
|
IllegalStateException e = assertThrows(IllegalStateException.class, () -> lsm.checkpointLsm());
|
||||||
|
assertTrue(e.getMessage().contains("kept losing its claim"), e.getMessage());
|
||||||
|
assertEquals(4, countCalls("flush_lsm"), "the initial attempt plus MAX_REISSUES");
|
||||||
|
}
|
||||||
|
|
||||||
|
// ===========================================================================
|
||||||
|
// strict decoding
|
||||||
|
// ===========================================================================
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A stats payload that does not decode must fail closed. Every one of these bodies used to be
|
||||||
|
* read as "no buckets", which is indistinguishable from a drained table, so {@code checkpointLsm}
|
||||||
|
* reported convergence for a checkpoint that never ran.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
public void testCheckpointRejectsMalformedStats() {
|
||||||
|
Map<String, String> malformed = new LinkedHashMap<String, String>();
|
||||||
|
malformed.put("no response body at all", "");
|
||||||
|
malformed.put("stats object with no buckets", "{\"lsm_stats\":{}}");
|
||||||
|
malformed.put("bucket missing its required fields", "{\"lsm_stats\":{\"buckets\":[{}]}}");
|
||||||
|
malformed.put(
|
||||||
|
"bucket missing generations",
|
||||||
|
"{\"lsm_stats\":{\"buckets\":[{\"shard_id\":\"shard-0\",\"status\":\"Active\","
|
||||||
|
+ "\"writer_epoch\":1,\"manifest_version\":2,\"current_generation\":9,"
|
||||||
|
+ "\"replay_after_wal_entry_position\":0,\"wal_entry_position_last_seen\":0,"
|
||||||
|
+ "\"compacting\":false}]}}");
|
||||||
|
malformed.put(
|
||||||
|
"generation with a non-numeric generation number",
|
||||||
|
"{\"lsm_stats\":{\"buckets\":[{\"shard_id\":\"shard-0\",\"status\":\"Active\","
|
||||||
|
+ "\"writer_epoch\":1,\"manifest_version\":2,\"current_generation\":9,"
|
||||||
|
+ "\"replay_after_wal_entry_position\":0,\"wal_entry_position_last_seen\":0,"
|
||||||
|
+ "\"generations\":[{\"generation\":\"7\",\"bytes\":1024}],"
|
||||||
|
+ "\"compacting\":false}]}}");
|
||||||
|
|
||||||
|
for (Map.Entry<String, String> each : malformed.entrySet()) {
|
||||||
|
setUpFresh();
|
||||||
|
enqueue("flush_lsm", 200, "");
|
||||||
|
enqueue("get_lsm_stats", 200, each.getValue());
|
||||||
|
|
||||||
|
assertThrows(
|
||||||
|
IllegalStateException.class,
|
||||||
|
() -> lsm.checkpointLsm(),
|
||||||
|
each.getKey() + " must not report convergence");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The one shape that legitimately means "this table has no LSM write path". */
|
||||||
|
@Test
|
||||||
|
public void testCheckpointTreatsNullStatsAsNotWalBacked() {
|
||||||
|
enqueue("flush_lsm", 200, "");
|
||||||
|
enqueue("get_lsm_stats", 200, "{\"lsm_stats\":null}");
|
||||||
|
|
||||||
|
lsm.checkpointLsm();
|
||||||
|
|
||||||
|
assertEquals(1, countCalls("get_lsm_stats"));
|
||||||
|
}
|
||||||
|
|
||||||
|
// ===========================================================================
|
||||||
|
// retry budget
|
||||||
|
// ===========================================================================
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The transport must not retry on the checkpoint loop's behalf. Apache HttpClient's default
|
||||||
|
* strategy retries exactly 429 and 503 — the two statuses {@code isRetryable} owns — which
|
||||||
|
* doubled every budget here and also retried {@code compact_lsm} in place, where the loop is
|
||||||
|
* built to fall through to a fresh stats poll instead.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
public void testCheckpointRetryBudgetIsNotDoubledByTheTransport() {
|
||||||
|
enqueue("flush_lsm", 429, "latch held");
|
||||||
|
|
||||||
|
LanceDbRestClient.HttpException e =
|
||||||
|
assertThrows(LanceDbRestClient.HttpException.class, () -> lsm.checkpointLsm());
|
||||||
|
|
||||||
|
assertEquals(429, e.statusCode(), "the exhausted budget propagates the last error as itself");
|
||||||
|
assertEquals(9, countCalls("flush_lsm"), "the initial request plus MAX_RETRIES, and no more");
|
||||||
|
}
|
||||||
|
|
||||||
|
// ===========================================================================
|
||||||
|
// harness
|
||||||
|
// ===========================================================================
|
||||||
|
|
||||||
|
private static List<Long> generationNumbers(BucketStats bucket) {
|
||||||
|
List<Long> numbers = new ArrayList<Long>();
|
||||||
|
for (GenerationStats generation : bucket.generations()) {
|
||||||
|
numbers.add(generation.generation());
|
||||||
|
}
|
||||||
|
return numbers;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Build an {@code lsm_stats} response body from bucket fragments. */
|
||||||
|
private static String stats(String... buckets) {
|
||||||
|
return "{\"lsm_stats\":{\"buckets\":[" + String.join(",", buckets) + "]}}";
|
||||||
|
}
|
||||||
|
|
||||||
|
private static String bucket(String shardId, boolean compacting, Long... generations) {
|
||||||
|
StringBuilder gens = new StringBuilder();
|
||||||
|
for (Long generation : generations) {
|
||||||
|
if (gens.length() > 0) {
|
||||||
|
gens.append(",");
|
||||||
|
}
|
||||||
|
gens.append("{\"generation\":").append(generation).append(",\"bytes\":1024}");
|
||||||
|
}
|
||||||
|
return "{\"shard_id\":\""
|
||||||
|
+ shardId
|
||||||
|
+ "\",\"status\":\"Active\",\"writer_epoch\":1,\"manifest_version\":2,"
|
||||||
|
+ "\"current_generation\":9,\"replay_after_wal_entry_position\":0,"
|
||||||
|
+ "\"wal_entry_position_last_seen\":0,\"generations\":["
|
||||||
|
+ gens
|
||||||
|
+ "],\"compacting\":"
|
||||||
|
+ compacting
|
||||||
|
+ "}";
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Queue a reply for an operation. The last queued reply repeats once the queue drains. */
|
||||||
|
private void enqueue(String operation, int status, String body) {
|
||||||
|
replies.computeIfAbsent(operation, key -> new ArrayDeque<Reply>()).add(new Reply(status, body));
|
||||||
|
}
|
||||||
|
|
||||||
|
private Reply nextReply(String path) {
|
||||||
|
String operation = operationOf(path);
|
||||||
|
Deque<Reply> queued = replies.get(operation);
|
||||||
|
if (queued == null || queued.isEmpty()) {
|
||||||
|
return new Reply(200, "");
|
||||||
|
}
|
||||||
|
return queued.size() > 1 ? queued.poll() : queued.peek();
|
||||||
|
}
|
||||||
|
|
||||||
|
private long countCalls(String operation) {
|
||||||
|
return requestPaths.stream().filter(path -> operationOf(path).equals(operation)).count();
|
||||||
|
}
|
||||||
|
|
||||||
|
/** {@code /v1/table/my_table/flush_lsm/} -> {@code flush_lsm}. */
|
||||||
|
private static String operationOf(String path) {
|
||||||
|
String[] segments = path.split("/");
|
||||||
|
return segments.length == 0 ? "" : segments[segments.length - 1];
|
||||||
|
}
|
||||||
|
|
||||||
|
private static String readAll(InputStream in) throws IOException {
|
||||||
|
ByteArrayOutputStream out = new ByteArrayOutputStream();
|
||||||
|
byte[] buffer = new byte[4096];
|
||||||
|
int read;
|
||||||
|
while ((read = in.read(buffer)) != -1) {
|
||||||
|
out.write(buffer, 0, read);
|
||||||
|
}
|
||||||
|
return new String(out.toByteArray(), StandardCharsets.UTF_8);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final class Reply {
|
||||||
|
private final int status;
|
||||||
|
private final String body;
|
||||||
|
|
||||||
|
private Reply(int status, String body) {
|
||||||
|
this.status = status;
|
||||||
|
this.body = body;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
+2
-2
@@ -6,7 +6,7 @@
|
|||||||
|
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-parent</artifactId>
|
<artifactId>lancedb-parent</artifactId>
|
||||||
<version>0.37.1-beta.0</version>
|
<version>0.38.0-beta.3</version>
|
||||||
<packaging>pom</packaging>
|
<packaging>pom</packaging>
|
||||||
<name>${project.artifactId}</name>
|
<name>${project.artifactId}</name>
|
||||||
<description>LanceDB Java SDK Parent POM</description>
|
<description>LanceDB Java SDK Parent POM</description>
|
||||||
@@ -28,7 +28,7 @@
|
|||||||
<properties>
|
<properties>
|
||||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||||
<arrow.version>15.0.0</arrow.version>
|
<arrow.version>15.0.0</arrow.version>
|
||||||
<lance-core.version>10.0.0-beta.5</lance-core.version>
|
<lance-core.version>11.0.0-beta.16</lance-core.version>
|
||||||
<spotless.skip>false</spotless.skip>
|
<spotless.skip>false</spotless.skip>
|
||||||
<spotless.version>2.30.0</spotless.version>
|
<spotless.version>2.30.0</spotless.version>
|
||||||
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
|
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# Contributing to LanceDB Typescript
|
# Contributing to LanceDB Typescript
|
||||||
|
|
||||||
This document outlines the process for contributing to LanceDB Typescript.
|
This document outlines the process for contributing to LanceDB Typescript.
|
||||||
For general contribution guidelines, see [CONTRIBUTING.md](../CONTRIBUTING.md).
|
For general contribution guidelines, see [CONTRIBUTING.md](https://github.com/lancedb/lancedb/blob/main/CONTRIBUTING.md).
|
||||||
|
|
||||||
## Project layout
|
## Project layout
|
||||||
|
|
||||||
|
|||||||
+5
-5
@@ -1,7 +1,7 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "lancedb-nodejs"
|
name = "lancedb-nodejs"
|
||||||
edition.workspace = true
|
edition.workspace = true
|
||||||
version = "0.37.1-beta.0"
|
version = "0.38.0-beta.3"
|
||||||
publish = false
|
publish = false
|
||||||
license.workspace = true
|
license.workspace = true
|
||||||
description.workspace = true
|
description.workspace = true
|
||||||
@@ -16,12 +16,12 @@ crate-type = ["cdylib"]
|
|||||||
async-trait.workspace = true
|
async-trait.workspace = true
|
||||||
arrow-ipc.workspace = true
|
arrow-ipc.workspace = true
|
||||||
arrow-array.workspace = true
|
arrow-array.workspace = true
|
||||||
arrow-buffer = "58.0.0"
|
arrow-buffer.workspace = true
|
||||||
half.workspace = true
|
half.workspace = true
|
||||||
arrow-schema.workspace = true
|
arrow-schema.workspace = true
|
||||||
env_logger.workspace = true
|
env_logger.workspace = true
|
||||||
futures.workspace = true
|
futures.workspace = true
|
||||||
lancedb = { path = "../rust/lancedb", default-features = false }
|
lancedb.workspace = true
|
||||||
lance-namespace.workspace = true
|
lance-namespace.workspace = true
|
||||||
napi = { version = "3.8.3", default-features = false, features = [
|
napi = { version = "3.8.3", default-features = false, features = [
|
||||||
"napi9",
|
"napi9",
|
||||||
@@ -29,8 +29,8 @@ napi = { version = "3.8.3", default-features = false, features = [
|
|||||||
"chrono_date",
|
"chrono_date",
|
||||||
"serde-json",
|
"serde-json",
|
||||||
] }
|
] }
|
||||||
chrono = { version = "0.4", default-features = false, features = ["clock"] }
|
chrono.workspace = true
|
||||||
serde_json = "1"
|
serde_json.workspace = true
|
||||||
napi-derive = "3.5.2"
|
napi-derive = "3.5.2"
|
||||||
# Prevent dynamic linking of lzma, which comes from datafusion
|
# Prevent dynamic linking of lzma, which comes from datafusion
|
||||||
lzma-sys = { version = "0.1", features = ["static"] }
|
lzma-sys = { version = "0.1", features = ["static"] }
|
||||||
|
|||||||
@@ -6,7 +6,9 @@ import * as arrow17 from "apache-arrow-17";
|
|||||||
import * as arrow18 from "apache-arrow-18";
|
import * as arrow18 from "apache-arrow-18";
|
||||||
|
|
||||||
import {
|
import {
|
||||||
|
Vector as CurrentVector,
|
||||||
convertToTable,
|
convertToTable,
|
||||||
|
tableFromIPC as currentTableFromIPC,
|
||||||
fromBufferToRecordBatch,
|
fromBufferToRecordBatch,
|
||||||
fromDataToBuffer,
|
fromDataToBuffer,
|
||||||
fromRecordBatchToBuffer,
|
fromRecordBatchToBuffer,
|
||||||
@@ -19,6 +21,7 @@ import {
|
|||||||
FunctionOptions,
|
FunctionOptions,
|
||||||
} from "../lancedb/embedding/embedding_function";
|
} from "../lancedb/embedding/embedding_function";
|
||||||
import { EmbeddingFunctionConfig } from "../lancedb/embedding/registry";
|
import { EmbeddingFunctionConfig } from "../lancedb/embedding/registry";
|
||||||
|
import { sanitizeTable } from "../lancedb/sanitize";
|
||||||
|
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: skip
|
// biome-ignore lint/suspicious/noExplicitAny: skip
|
||||||
function sampleRecords(): Array<Record<string, any>> {
|
function sampleRecords(): Array<Record<string, any>> {
|
||||||
@@ -64,7 +67,11 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
tableFromIPC,
|
tableFromIPC,
|
||||||
DataType,
|
DataType,
|
||||||
Dictionary,
|
Dictionary,
|
||||||
|
RecordBatch: ArrowRecordBatch,
|
||||||
|
Table: ArrowTable,
|
||||||
Uint8: ArrowUint8,
|
Uint8: ArrowUint8,
|
||||||
|
makeData: arrowMakeData,
|
||||||
|
vectorFromArray,
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: <explanation>
|
// biome-ignore lint/suspicious/noExplicitAny: <explanation>
|
||||||
} = <any>arrow;
|
} = <any>arrow;
|
||||||
type Schema = ApacheArrow["Schema"];
|
type Schema = ApacheArrow["Schema"];
|
||||||
@@ -197,6 +204,35 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
expect(table.getChild("d")?.toJSON()).toEqual([9n, 10n, null]);
|
expect(table.getChild("d")?.toJSON()).toEqual([9n, 10n, null]);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("will use a provided FixedSizeList schema with typed array values", function () {
|
||||||
|
const schema = new Schema([
|
||||||
|
new Field("text", new Utf8(), false),
|
||||||
|
new Field(
|
||||||
|
"vector",
|
||||||
|
new FixedSizeList(3, new Field("item", new Float32(), false)),
|
||||||
|
false,
|
||||||
|
),
|
||||||
|
]);
|
||||||
|
|
||||||
|
const table = makeArrowTable(
|
||||||
|
[
|
||||||
|
{
|
||||||
|
text: "foo",
|
||||||
|
vector: new Float32Array([1, 2, 3]),
|
||||||
|
},
|
||||||
|
],
|
||||||
|
{ schema },
|
||||||
|
);
|
||||||
|
|
||||||
|
expect(table.getChild("text")?.toJSON()).toEqual(["foo"]);
|
||||||
|
expect(
|
||||||
|
table
|
||||||
|
.getChild("vector")
|
||||||
|
?.toJSON()
|
||||||
|
.map((value) => value.toJSON()),
|
||||||
|
).toEqual([[1, 2, 3]]);
|
||||||
|
});
|
||||||
|
|
||||||
it("will assume the column `vector` is FixedSizeList<Float32> by default", async function () {
|
it("will assume the column `vector` is FixedSizeList<Float32> by default", async function () {
|
||||||
const schema = new Schema([
|
const schema = new Schema([
|
||||||
new Field("a", new Float(Precision.DOUBLE), true),
|
new Field("a", new Float(Precision.DOUBLE), true),
|
||||||
@@ -1025,6 +1061,114 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
});
|
});
|
||||||
|
|
||||||
describe("when using two versions of arrow", function () {
|
describe("when using two versions of arrow", function () {
|
||||||
|
it("preserves a dictionary shared by multiple fields", async function () {
|
||||||
|
const values = ["alpha", "beta", "alpha"];
|
||||||
|
const dictionaryVector = vectorFromArray(values);
|
||||||
|
const batch = new ArrowRecordBatch({
|
||||||
|
first: dictionaryVector.data[0],
|
||||||
|
second: dictionaryVector.data[0],
|
||||||
|
});
|
||||||
|
const table = new ArrowTable([batch]);
|
||||||
|
|
||||||
|
const sanitized = sanitizeTable(table);
|
||||||
|
expect([...sanitized.getChild("first")!]).toEqual(values);
|
||||||
|
expect([...sanitized.getChild("second")!]).toEqual(values);
|
||||||
|
const firstType = sanitized.schema.fields[0].type as {
|
||||||
|
dictionary: unknown;
|
||||||
|
};
|
||||||
|
const secondType = sanitized.schema.fields[1].type as {
|
||||||
|
dictionary: unknown;
|
||||||
|
};
|
||||||
|
expect(secondType.dictionary).toBe(firstType.dictionary);
|
||||||
|
expect(sanitized.batches[0].data.children[1].dictionary).toBe(
|
||||||
|
sanitized.batches[0].data.children[0].dictionary,
|
||||||
|
);
|
||||||
|
|
||||||
|
const buf = await fromDataToBuffer(table);
|
||||||
|
const actual = currentTableFromIPC(buf);
|
||||||
|
expect([...actual.getChild("first")!]).toEqual(values);
|
||||||
|
expect([...actual.getChild("second")!]).toEqual(values);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("preserves shared dictionary data from another Arrow version", async function () {
|
||||||
|
const values = ["alpha", "beta", "alpha"];
|
||||||
|
const dictionaryVector = vectorFromArray(values);
|
||||||
|
const firstBatch = new ArrowRecordBatch({
|
||||||
|
label: dictionaryVector.slice(0, 2).data[0],
|
||||||
|
});
|
||||||
|
const secondBatch = new ArrowRecordBatch({
|
||||||
|
label: dictionaryVector.slice(2).data[0],
|
||||||
|
});
|
||||||
|
const table = new ArrowTable([firstBatch, secondBatch]);
|
||||||
|
|
||||||
|
const sanitized = sanitizeTable(table);
|
||||||
|
expect([...sanitized.getChild("label")!]).toEqual(values);
|
||||||
|
|
||||||
|
const dictionaries = sanitized.batches.map(
|
||||||
|
(batch) => batch.data.children[0].dictionary,
|
||||||
|
);
|
||||||
|
expect(dictionaries[0]).toBeInstanceOf(CurrentVector);
|
||||||
|
expect(dictionaries[1]).toBe(dictionaries[0]);
|
||||||
|
|
||||||
|
const buf = await fromDataToBuffer(table);
|
||||||
|
const actual = currentTableFromIPC(buf);
|
||||||
|
expect([...actual.getChild("label")!]).toEqual(values);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("preserves shared chunks in growing dictionaries", async function () {
|
||||||
|
const type = new Dictionary(new Utf8(), new Int32(), 42, false);
|
||||||
|
const firstDictionary = vectorFromArray(["alpha", "beta"], new Utf8());
|
||||||
|
const secondDictionary = firstDictionary.concat(
|
||||||
|
vectorFromArray(["gamma"], new Utf8()),
|
||||||
|
);
|
||||||
|
const firstData = arrowMakeData({
|
||||||
|
type,
|
||||||
|
data: Int32Array.from([0, 1]),
|
||||||
|
dictionary: firstDictionary,
|
||||||
|
});
|
||||||
|
const secondData = arrowMakeData({
|
||||||
|
type,
|
||||||
|
data: Int32Array.from([2]),
|
||||||
|
dictionary: secondDictionary,
|
||||||
|
});
|
||||||
|
const table = new ArrowTable([
|
||||||
|
new ArrowRecordBatch({ label: firstData }),
|
||||||
|
new ArrowRecordBatch({ label: secondData }),
|
||||||
|
]);
|
||||||
|
|
||||||
|
const sanitized = sanitizeTable(table);
|
||||||
|
const expected = ["alpha", "beta", "gamma"];
|
||||||
|
expect([...sanitized.getChild("label")!]).toEqual(expected);
|
||||||
|
const firstLocalDictionary =
|
||||||
|
sanitized.batches[0].data.children[0].dictionary!;
|
||||||
|
const secondLocalDictionary =
|
||||||
|
sanitized.batches[1].data.children[0].dictionary!;
|
||||||
|
expect(secondLocalDictionary.data[0]).toBe(
|
||||||
|
firstLocalDictionary.data[0],
|
||||||
|
);
|
||||||
|
|
||||||
|
const buf = await fromTableToBuffer(sanitized);
|
||||||
|
const actual = currentTableFromIPC(buf);
|
||||||
|
expect([...actual.getChild("label")!]).toEqual(expected);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("can serialize list data from another Arrow version", async function () {
|
||||||
|
const values = [["anime", "action"], [], null];
|
||||||
|
const vector = vectorFromArray(
|
||||||
|
values,
|
||||||
|
new List(new Field("item", new Utf8(), true)),
|
||||||
|
);
|
||||||
|
const table = new ArrowTable({ tags: vector });
|
||||||
|
|
||||||
|
const buf = await fromDataToBuffer(table);
|
||||||
|
const actual = currentTableFromIPC(buf);
|
||||||
|
const actualTags = actual.getChild("tags");
|
||||||
|
|
||||||
|
expect(actualTags?.get(0)?.toJSON()).toEqual(values[0]);
|
||||||
|
expect(actualTags?.get(1)?.toJSON()).toEqual(values[1]);
|
||||||
|
expect(actualTags?.get(2)).toBeNull();
|
||||||
|
});
|
||||||
|
|
||||||
it("can still import data", async function () {
|
it("can still import data", async function () {
|
||||||
const schema = new arrow15.Schema([
|
const schema = new arrow15.Schema([
|
||||||
new arrow15.Field("id", new arrow15.Int32()),
|
new arrow15.Field("id", new arrow15.Int32()),
|
||||||
|
|||||||
@@ -89,6 +89,16 @@ describe("given a connection", () => {
|
|||||||
await db.createTable("test4", [{ id: 1 }, { id: 2 }]);
|
await db.createTable("test4", [{ id: 1 }, { id: 2 }]);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("should return a completed job when dropping a local table", async () => {
|
||||||
|
await db.createTable("async-drop", [{ id: 1 }]);
|
||||||
|
|
||||||
|
const job = await db.dropTableAsync("async-drop");
|
||||||
|
expect(job.id).toBeNull();
|
||||||
|
await expect(job.status()).resolves.toBe("finished");
|
||||||
|
await job.wait();
|
||||||
|
await expect(db.tableNames()).resolves.toEqual([]);
|
||||||
|
});
|
||||||
|
|
||||||
it("should fail if creating table twice, unless overwrite is true", async () => {
|
it("should fail if creating table twice, unless overwrite is true", async () => {
|
||||||
let tbl = await db.createTable("test", [{ id: 1 }, { id: 2 }]);
|
let tbl = await db.createTable("test", [{ id: 1 }, { id: 2 }]);
|
||||||
await expect(tbl.countRows()).resolves.toBe(2);
|
await expect(tbl.countRows()).resolves.toBe(2);
|
||||||
|
|||||||
@@ -11,8 +11,11 @@ import {
|
|||||||
Float16,
|
Float16,
|
||||||
Float32,
|
Float32,
|
||||||
Float64,
|
Float64,
|
||||||
|
Int32,
|
||||||
Schema,
|
Schema,
|
||||||
Utf8,
|
Utf8,
|
||||||
|
fromDataToBuffer,
|
||||||
|
tableFromIPC,
|
||||||
} from "../lancedb/arrow";
|
} from "../lancedb/arrow";
|
||||||
import { EmbeddingFunction, LanceSchema } from "../lancedb/embedding";
|
import { EmbeddingFunction, LanceSchema } from "../lancedb/embedding";
|
||||||
import { getRegistry, register } from "../lancedb/embedding/registry";
|
import { getRegistry, register } from "../lancedb/embedding/registry";
|
||||||
@@ -184,6 +187,63 @@ describe("embedding functions", () => {
|
|||||||
const vector0 = JSON.parse(JSON.stringify(arr[0].vector));
|
const vector0 = JSON.parse(JSON.stringify(arr[0].vector));
|
||||||
expect(vector0).toEqual([1, 2, 3]);
|
expect(vector0).toEqual([1, 2, 3]);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("should append generated vectors to a non-nullable schema", async () => {
|
||||||
|
@register("non_nullable_schema_test")
|
||||||
|
class MockEmbeddingFunction extends EmbeddingFunction<string> {
|
||||||
|
ndims() {
|
||||||
|
return 3;
|
||||||
|
}
|
||||||
|
embeddingDataType(): Float {
|
||||||
|
return new Float64();
|
||||||
|
}
|
||||||
|
async computeSourceEmbeddings(data: string[]) {
|
||||||
|
return data.map(() => [1, 2, 3]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const schema = new Schema([
|
||||||
|
new Field("id", new Int32()),
|
||||||
|
new Field("text", new Utf8()),
|
||||||
|
new Field("type", new Utf8()),
|
||||||
|
new Field(
|
||||||
|
"vector",
|
||||||
|
new FixedSizeList(3, new Field("item", new Float64())),
|
||||||
|
),
|
||||||
|
]);
|
||||||
|
const func = new MockEmbeddingFunction();
|
||||||
|
const db = await connect(tmpDir.name);
|
||||||
|
const table = await db.createEmptyTable("test_non_nullable", schema, {
|
||||||
|
embeddingFunction: {
|
||||||
|
function: func,
|
||||||
|
sourceColumn: "text",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
const data = [
|
||||||
|
{ id: 1, text: "Carrot", type: "vegetable" },
|
||||||
|
{ id: 2, text: "Apple", type: "fruit" },
|
||||||
|
];
|
||||||
|
const buffer = await fromDataToBuffer(
|
||||||
|
data,
|
||||||
|
undefined,
|
||||||
|
await table.schema(),
|
||||||
|
);
|
||||||
|
const generatedTable = tableFromIPC(buffer);
|
||||||
|
const vectorField = generatedTable.schema.fields.find(
|
||||||
|
(field) => field.name === "vector",
|
||||||
|
);
|
||||||
|
expect(vectorField?.nullable).toBe(false);
|
||||||
|
|
||||||
|
await table.add(data);
|
||||||
|
|
||||||
|
const rows = await table.query().toArray();
|
||||||
|
expect(rows).toHaveLength(2);
|
||||||
|
for (const row of rows) {
|
||||||
|
expect([...row.vector]).toEqual([1, 2, 3]);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
it("should error when appending to a table with an unregistered embedding function", async () => {
|
it("should error when appending to a table with an unregistered embedding function", async () => {
|
||||||
@register("mock")
|
@register("mock")
|
||||||
class MockEmbeddingFunction extends EmbeddingFunction<string> {
|
class MockEmbeddingFunction extends EmbeddingFunction<string> {
|
||||||
|
|||||||
@@ -0,0 +1,14 @@
|
|||||||
|
// SPDX-License-Identifier: Apache-2.0
|
||||||
|
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
import packageJson = require("../package.json");
|
||||||
|
|
||||||
|
describe("package metadata", () => {
|
||||||
|
it("requires Node.js type declarations compatible with the runtime", () => {
|
||||||
|
expect(packageJson.engines.node).toBe(">= 18");
|
||||||
|
expect(packageJson.peerDependencies["@types/node"]).toBe(">=18");
|
||||||
|
expect(packageJson.peerDependenciesMeta["@types/node"]).toEqual({
|
||||||
|
optional: true,
|
||||||
|
});
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -110,6 +110,81 @@ describe("Query outputSchema", () => {
|
|||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
describe("Search pagination", () => {
|
||||||
|
let tmpDir: tmp.DirResult;
|
||||||
|
let table: Table;
|
||||||
|
|
||||||
|
beforeEach(async () => {
|
||||||
|
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
||||||
|
const db = await connect(tmpDir.name);
|
||||||
|
const schema = new Schema([
|
||||||
|
new Field("id", new Int64(), false),
|
||||||
|
new Field("text", new Utf8(), false),
|
||||||
|
new Field(
|
||||||
|
"vector",
|
||||||
|
new FixedSizeList(2, new Field("item", new Float32())),
|
||||||
|
false,
|
||||||
|
),
|
||||||
|
]);
|
||||||
|
const data = makeArrowTable(
|
||||||
|
[
|
||||||
|
{ id: 1n, text: "common", vector: [0, 0] },
|
||||||
|
{ id: 2n, text: "common common", vector: [1, 1] },
|
||||||
|
{ id: 3n, text: "common common common", vector: [2, 2] },
|
||||||
|
{ id: 4n, text: "common common common common", vector: [3, 3] },
|
||||||
|
],
|
||||||
|
{ schema },
|
||||||
|
);
|
||||||
|
table = await db.createTable("test", data);
|
||||||
|
});
|
||||||
|
|
||||||
|
afterEach(() => {
|
||||||
|
tmpDir.removeCallback();
|
||||||
|
});
|
||||||
|
|
||||||
|
it("applies offset after the vector search limit", async () => {
|
||||||
|
const allResults = await table
|
||||||
|
.vectorSearch([0, 0])
|
||||||
|
.select(["id"])
|
||||||
|
.limit(4)
|
||||||
|
.toArray();
|
||||||
|
const secondPage = await table
|
||||||
|
.vectorSearch([0, 0])
|
||||||
|
.select(["id"])
|
||||||
|
.limit(2)
|
||||||
|
.offset(2)
|
||||||
|
.toArray();
|
||||||
|
|
||||||
|
expect(allResults).toHaveLength(4);
|
||||||
|
expect(secondPage).toHaveLength(2);
|
||||||
|
expect(secondPage.map((row) => row.id)).toEqual(
|
||||||
|
allResults.slice(2, 4).map((row) => row.id),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("applies offset after the full-text search limit", async () => {
|
||||||
|
await table.createIndex("text", { config: Index.fts() });
|
||||||
|
|
||||||
|
const allResults = await table
|
||||||
|
.search("common", "fts")
|
||||||
|
.select(["id"])
|
||||||
|
.limit(4)
|
||||||
|
.toArray();
|
||||||
|
const secondPage = await table
|
||||||
|
.search("common", "fts")
|
||||||
|
.select(["id"])
|
||||||
|
.limit(2)
|
||||||
|
.offset(2)
|
||||||
|
.toArray();
|
||||||
|
|
||||||
|
expect(allResults).toHaveLength(4);
|
||||||
|
expect(secondPage).toHaveLength(2);
|
||||||
|
expect(secondPage.map((row) => row.id)).toEqual(
|
||||||
|
allResults.slice(2, 4).map((row) => row.id),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
describe("Query orderBy", () => {
|
describe("Query orderBy", () => {
|
||||||
let tmpDir: tmp.DirResult;
|
let tmpDir: tmp.DirResult;
|
||||||
let table: Table;
|
let table: Table;
|
||||||
|
|||||||
@@ -170,6 +170,38 @@ describe("remote connection", () => {
|
|||||||
);
|
);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("surfaces JSON server errors from remote table operations", async () => {
|
||||||
|
await withMockDatabase(
|
||||||
|
(req, res) => {
|
||||||
|
const path = req.url ?? "";
|
||||||
|
if (path.endsWith("/describe/")) {
|
||||||
|
res.writeHead(200, { "Content-Type": "application/json" }).end(
|
||||||
|
JSON.stringify({
|
||||||
|
name: "broken_table",
|
||||||
|
version: 1,
|
||||||
|
schema: { fields: [] },
|
||||||
|
}),
|
||||||
|
);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (path.endsWith("/count_rows/")) {
|
||||||
|
res
|
||||||
|
.writeHead(400, { "Content-Type": "application/json" })
|
||||||
|
.end(JSON.stringify({ error: "count rows failed" }));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
res.writeHead(404).end();
|
||||||
|
},
|
||||||
|
async (db) => {
|
||||||
|
const table = await db.openTable("broken_table");
|
||||||
|
|
||||||
|
await expect(table.countRows()).rejects.toThrow("count rows failed");
|
||||||
|
},
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
it("should pass on requested extra headers", async () => {
|
it("should pass on requested extra headers", async () => {
|
||||||
await withMockDatabase(
|
await withMockDatabase(
|
||||||
(req, res) => {
|
(req, res) => {
|
||||||
@@ -877,3 +909,96 @@ describe("remote connection", () => {
|
|||||||
});
|
});
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
describe("remote connection jobs surface", () => {
|
||||||
|
it("lists, describes, cancels, and reads history", async () => {
|
||||||
|
const { tableFromArrays, tableToIPC } = await import("apache-arrow");
|
||||||
|
const eventsTable = tableFromArrays({ state: ["created", "succeeded"] });
|
||||||
|
const eventsBody = Buffer.from(tableToIPC(eventsTable, "stream"));
|
||||||
|
|
||||||
|
await withMockDatabase(
|
||||||
|
(req, res) => {
|
||||||
|
let body = "";
|
||||||
|
req.on("data", (chunk) => {
|
||||||
|
body += chunk;
|
||||||
|
});
|
||||||
|
req.on("end", () => {
|
||||||
|
const payload = body.length > 0 ? JSON.parse(body) : {};
|
||||||
|
if (req.url === "/v1/jobs/list") {
|
||||||
|
if (payload["page_token"] === undefined) {
|
||||||
|
res
|
||||||
|
.writeHead(200, { "Content-Type": "application/json" })
|
||||||
|
.end(
|
||||||
|
'{"jobs": [{"job_id": "job-1", "table": "t1", ' +
|
||||||
|
'"job_type": "create_index", "state": "in_progress", ' +
|
||||||
|
'"created_at_millis": 1000}], "page_token": "next"}',
|
||||||
|
);
|
||||||
|
} else {
|
||||||
|
res
|
||||||
|
.writeHead(200, { "Content-Type": "application/json" })
|
||||||
|
.end(
|
||||||
|
'{"jobs": [{"job_id": "job-2", "table": "t2", ' +
|
||||||
|
'"job_type": "create_index", "state": "succeeded", ' +
|
||||||
|
'"created_at_millis": 2000}]}',
|
||||||
|
);
|
||||||
|
}
|
||||||
|
} else if (req.url === "/v1/jobs/describe") {
|
||||||
|
if (payload["job_id"] !== "job-1") {
|
||||||
|
res.writeHead(404).end("no such job");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
res
|
||||||
|
.writeHead(200, { "Content-Type": "application/json" })
|
||||||
|
.end(
|
||||||
|
'{"job_id": "job-1", "job_type": "create_index", ' +
|
||||||
|
'"job_state": "FAILED", "creation_ms": 1000, ' +
|
||||||
|
'"spec": {"column": "vec"}, "failure": {"phase": "execute", ' +
|
||||||
|
'"message": "worker died", "retryable": true}}',
|
||||||
|
);
|
||||||
|
} else if (req.url === "/v1/jobs/cancel") {
|
||||||
|
if (payload["job_id"] !== "job-1") {
|
||||||
|
res.writeHead(404).end("no such job");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
res
|
||||||
|
.writeHead(200, { "Content-Type": "application/json" })
|
||||||
|
.end('{"job_id": "job-1"}');
|
||||||
|
} else if (req.url === "/v1/jobs/query_events") {
|
||||||
|
res
|
||||||
|
.writeHead(200, {
|
||||||
|
"Content-Type": "application/vnd.apache.arrow.stream",
|
||||||
|
})
|
||||||
|
.end(eventsBody);
|
||||||
|
} else {
|
||||||
|
res.writeHead(404).end();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
},
|
||||||
|
async (db) => {
|
||||||
|
const jobs = await db.listJobs();
|
||||||
|
expect(jobs.map((job) => job.jobId)).toEqual(["job-1", "job-2"]);
|
||||||
|
expect(jobs[0].state).toEqual("running");
|
||||||
|
expect(jobs[1].state).toEqual("finished");
|
||||||
|
|
||||||
|
const description = await db.getJob("job-1");
|
||||||
|
expect(description?.state).toEqual("failed");
|
||||||
|
expect(JSON.parse(description?.specJson ?? "")).toEqual({
|
||||||
|
column: "vec",
|
||||||
|
});
|
||||||
|
expect(description?.failure?.message).toEqual("worker died");
|
||||||
|
expect(await db.getJob("missing")).toBeNull();
|
||||||
|
|
||||||
|
expect(await db.cancelJob("job-1")).toBe(true);
|
||||||
|
expect(await db.cancelJob("missing")).toBe(false);
|
||||||
|
|
||||||
|
const history = await db.jobHistory("job-1");
|
||||||
|
expect(history.numRows).toEqual(2);
|
||||||
|
|
||||||
|
const job = db.job("job-1");
|
||||||
|
expect(job.id).toEqual("job-1");
|
||||||
|
expect(await job.status()).toEqual("failed");
|
||||||
|
await expect(job.wait()).rejects.toThrow("worker died");
|
||||||
|
},
|
||||||
|
);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|||||||
@@ -86,6 +86,44 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
await expect(table.countRows()).resolves.toBe(3);
|
await expect(table.countRows()).resolves.toBe(3);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("should support a foreign Float64 vector schema end to end", async () => {
|
||||||
|
const conn = await connect(tmpDir.name);
|
||||||
|
const schema = new arrow.Schema([
|
||||||
|
new arrow.Field("resource_id", new arrow.Int32(), false),
|
||||||
|
new arrow.Field(
|
||||||
|
"vector",
|
||||||
|
new arrow.FixedSizeList(
|
||||||
|
3,
|
||||||
|
new arrow.Field("value", new arrow.Float64(), true),
|
||||||
|
),
|
||||||
|
false,
|
||||||
|
),
|
||||||
|
]);
|
||||||
|
const data = [
|
||||||
|
{
|
||||||
|
// biome-ignore lint/style/useNamingConvention: matches the reported schema
|
||||||
|
resource_id: 0,
|
||||||
|
vector: [0.1, 0.1, 0.1],
|
||||||
|
},
|
||||||
|
];
|
||||||
|
|
||||||
|
const resources = await conn.createTable("resources", data, { schema });
|
||||||
|
|
||||||
|
const existing = await resources
|
||||||
|
.query()
|
||||||
|
.where("resource_id = 0")
|
||||||
|
.limit(1)
|
||||||
|
.toArray();
|
||||||
|
expect(existing).toHaveLength(1);
|
||||||
|
|
||||||
|
const matched = await resources
|
||||||
|
.search(Float64Array.from(data[0].vector))
|
||||||
|
.limit(1)
|
||||||
|
.toArray();
|
||||||
|
expect(matched).toHaveLength(1);
|
||||||
|
expect(matched[0]["resource_id"]).toBe(0);
|
||||||
|
});
|
||||||
|
|
||||||
it("should support branches", async () => {
|
it("should support branches", async () => {
|
||||||
await table.add([{ id: 1 }]);
|
await table.add([{ id: 1 }]);
|
||||||
expect(await table.countRows()).toBe(1);
|
expect(await table.countRows()).toBe(1);
|
||||||
@@ -239,8 +277,16 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
},
|
},
|
||||||
numIndices: 0,
|
numIndices: 0,
|
||||||
numRows: 3,
|
numRows: 3,
|
||||||
totalBytes: 44,
|
// Full on-disk size of the two data files, footers and metadata included.
|
||||||
|
totalBytes: 684,
|
||||||
});
|
});
|
||||||
|
|
||||||
|
// Index files count toward totalBytes too (only deletion files and
|
||||||
|
// manifests are excluded).
|
||||||
|
await table.createIndex("id", { config: Index.btree() });
|
||||||
|
const statsWithIndex = await table.stats();
|
||||||
|
expect(statsWithIndex.numIndices).toBe(1);
|
||||||
|
expect(statsWithIndex.totalBytes).toBeGreaterThan(684);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("should overwrite data if asked", async () => {
|
it("should overwrite data if asked", async () => {
|
||||||
@@ -851,7 +897,11 @@ describe("When creating an index", () => {
|
|||||||
afterEach(() => tmpDir.removeCallback());
|
afterEach(() => tmpDir.removeCallback());
|
||||||
|
|
||||||
it("should create a vector index on vector columns", async () => {
|
it("should create a vector index on vector columns", async () => {
|
||||||
await tbl.createIndex("vec");
|
const job = await tbl.createIndexAsync("vec");
|
||||||
|
expect(job.id).toBeNull();
|
||||||
|
await job.wait();
|
||||||
|
// Cancelling a job that already finished succeeds and does nothing.
|
||||||
|
await job.cancel();
|
||||||
|
|
||||||
// check index directory
|
// check index directory
|
||||||
const indexDir = path.join(tmpDir.name, "test.lance", "_indices");
|
const indexDir = path.join(tmpDir.name, "test.lance", "_indices");
|
||||||
@@ -3290,3 +3340,120 @@ describe("LSM merge insert", () => {
|
|||||||
await expect(table.query().useLsm(true).toArray()).rejects.toThrow();
|
await expect(table.query().useLsm(true).toArray()).rejects.toThrow();
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
describe("LSM convergence and stats", () => {
|
||||||
|
let tmpDir: tmp.DirResult;
|
||||||
|
|
||||||
|
beforeEach(() => {
|
||||||
|
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
||||||
|
});
|
||||||
|
afterEach(() => tmpDir.removeCallback());
|
||||||
|
|
||||||
|
async function lsmTable(conn: Connection): Promise<Table> {
|
||||||
|
const table = await conn.createEmptyTable(
|
||||||
|
"t",
|
||||||
|
new arrow.Schema([new arrow.Field("id", new arrow.Utf8(), false)]),
|
||||||
|
);
|
||||||
|
await table.setUnenforcedPrimaryKey("id");
|
||||||
|
await table.setLsmWriteSpec({ specType: "unsharded" });
|
||||||
|
return table;
|
||||||
|
}
|
||||||
|
|
||||||
|
// These four route through the server that owns the MemWAL, so a local table
|
||||||
|
// rejects them rather than answering. What is asserted here is that the
|
||||||
|
// bindings reach the core at all; the behavior against a real endpoint is
|
||||||
|
// covered by the mocked endpoint tests in rust/lancedb/src/remote/table.rs.
|
||||||
|
it("rejects flushLsm on a local table", async () => {
|
||||||
|
const conn = await connect(tmpDir.name);
|
||||||
|
const table = await lsmTable(conn);
|
||||||
|
|
||||||
|
await expect(table.flushLsm()).rejects.toThrow(/not supported/i);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects compactLsm on a local table", async () => {
|
||||||
|
const conn = await connect(tmpDir.name);
|
||||||
|
const table = await lsmTable(conn);
|
||||||
|
|
||||||
|
await expect(table.compactLsm()).rejects.toThrow(/not supported/i);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects getLsmStats on a local table", async () => {
|
||||||
|
const conn = await connect(tmpDir.name);
|
||||||
|
const table = await lsmTable(conn);
|
||||||
|
|
||||||
|
await expect(table.getLsmStats()).rejects.toThrow(/not supported/i);
|
||||||
|
await expect(table.getLsmStats(true)).rejects.toThrow(/not supported/i);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects checkpointLsm on a local table", async () => {
|
||||||
|
const conn = await connect(tmpDir.name);
|
||||||
|
const table = await lsmTable(conn);
|
||||||
|
|
||||||
|
// checkpointLsm seals first, so it surfaces flushLsm's rejection.
|
||||||
|
await expect(table.checkpointLsm()).rejects.toThrow(/not supported/i);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("computed columns", () => {
|
||||||
|
let tmpDir: tmp.DirResult;
|
||||||
|
beforeEach(() => {
|
||||||
|
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
||||||
|
});
|
||||||
|
afterEach(() => tmpDir.removeCallback());
|
||||||
|
|
||||||
|
it("declares a column and fills it on refresh", async () => {
|
||||||
|
const db = await connect(tmpDir.name);
|
||||||
|
const table = await db.createTable("computed", [{ x: 1 }, { x: 2 }]);
|
||||||
|
|
||||||
|
await table.addColumns({
|
||||||
|
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
||||||
|
});
|
||||||
|
let rows = await table.query().toArray();
|
||||||
|
expect(rows.map((r) => r.doubled)).toEqual([null, null]);
|
||||||
|
|
||||||
|
const result = await table.refreshColumn("doubled");
|
||||||
|
expect(result.rowsFilled).toBe(2);
|
||||||
|
|
||||||
|
rows = await table.query().toArray();
|
||||||
|
expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("returns a job handle from refreshColumnAsync", async () => {
|
||||||
|
const db = await connect(tmpDir.name);
|
||||||
|
const table = await db.createTable("computed_job", [{ x: 1 }, { x: 2 }]);
|
||||||
|
|
||||||
|
await table.addColumns({
|
||||||
|
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
||||||
|
});
|
||||||
|
|
||||||
|
const job = await table.refreshColumnAsync("doubled");
|
||||||
|
expect(job.id).toBeNull();
|
||||||
|
await job.wait();
|
||||||
|
expect(await job.status()).toBe("finished");
|
||||||
|
|
||||||
|
const rows = await table.query().toArray();
|
||||||
|
expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]);
|
||||||
|
|
||||||
|
// Bad input rejects at the call, not through the job.
|
||||||
|
await expect(table.refreshColumnAsync("x")).rejects.toThrow(
|
||||||
|
"not a computed column",
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("fills rows added since the last refresh", async () => {
|
||||||
|
const db = await connect(tmpDir.name);
|
||||||
|
const table = await db.createTable("computed_append", [{ x: 1 }]);
|
||||||
|
|
||||||
|
await table.addColumns({
|
||||||
|
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
||||||
|
});
|
||||||
|
await table.refreshColumn("doubled");
|
||||||
|
await table.add([{ x: 5 }]);
|
||||||
|
|
||||||
|
const result = await table.refreshColumn("doubled");
|
||||||
|
expect(result.rowsFilled).toBe(1);
|
||||||
|
|
||||||
|
const rows = await table.query().toArray();
|
||||||
|
expect(rows.map((r) => r.doubled).sort()).toEqual([10, 2]);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
// SPDX-License-Identifier: Apache-2.0
|
// SPDX-License-Identifier: Apache-2.0
|
||||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
import { tableFromIPC } from "apache-arrow";
|
||||||
import {
|
import {
|
||||||
Data,
|
Data,
|
||||||
SchemaLike,
|
SchemaLike,
|
||||||
@@ -20,6 +21,9 @@ import type {
|
|||||||
CreateNamespaceResponse,
|
CreateNamespaceResponse,
|
||||||
DescribeNamespaceResponse,
|
DescribeNamespaceResponse,
|
||||||
DropNamespaceResponse,
|
DropNamespaceResponse,
|
||||||
|
Job,
|
||||||
|
JobDescription,
|
||||||
|
JobInfo,
|
||||||
ListNamespacesResponse,
|
ListNamespacesResponse,
|
||||||
} from "./native";
|
} from "./native";
|
||||||
export type {
|
export type {
|
||||||
@@ -323,6 +327,14 @@ export abstract class Connection {
|
|||||||
*/
|
*/
|
||||||
abstract dropTable(name: string, namespacePath?: string[]): Promise<void>;
|
abstract dropTable(name: string, namespacePath?: string[]): Promise<void>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Start dropping a table and return its cleanup job.
|
||||||
|
*
|
||||||
|
* The table may become unavailable before its data files are removed. Wait
|
||||||
|
* on the returned job to know when cleanup has finished.
|
||||||
|
*/
|
||||||
|
abstract dropTableAsync(name: string, namespacePath?: string[]): Promise<Job>;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Drop all tables in the database.
|
* Drop all tables in the database.
|
||||||
* @param {string[]} namespacePath The namespace path to drop tables from (defaults to root namespace).
|
* @param {string[]} namespacePath The namespace path to drop tables from (defaults to root namespace).
|
||||||
@@ -436,6 +448,40 @@ export abstract class Connection {
|
|||||||
newName: string,
|
newName: string,
|
||||||
options?: RenameTableOptions,
|
options?: RenameTableOptions,
|
||||||
): Promise<void>;
|
): Promise<void>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A {@link Job} handle for a server-side job by id.
|
||||||
|
*
|
||||||
|
* The handle is constructed without a server round trip; an unknown id
|
||||||
|
* surfaces when the handle is used. Dropping the handle has no effect on
|
||||||
|
* the job itself.
|
||||||
|
*/
|
||||||
|
abstract job(jobId: string): Job;
|
||||||
|
|
||||||
|
/** List server-side jobs across the database's tables. */
|
||||||
|
abstract listJobs(): Promise<JobInfo[]>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Describe a single server-side job by id.
|
||||||
|
*
|
||||||
|
* Resolves to `null` when the server has no such job.
|
||||||
|
*/
|
||||||
|
abstract getJob(jobId: string): Promise<JobDescription | null>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Request cancellation of a server-side job by id.
|
||||||
|
*
|
||||||
|
* Resolves to true if the server accepted the cancellation, false if no
|
||||||
|
* such job exists. Cancelling an already-terminal job is a no-op success.
|
||||||
|
*/
|
||||||
|
abstract cancelJob(jobId: string): Promise<boolean>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The lifecycle event history of a server-side job, as an Arrow table.
|
||||||
|
*
|
||||||
|
* Lists history across all jobs when `jobId` is omitted.
|
||||||
|
*/
|
||||||
|
abstract jobHistory(jobId?: string): Promise<ArrowTable>;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** @hideconstructor */
|
/** @hideconstructor */
|
||||||
@@ -667,6 +713,10 @@ export class LocalConnection extends Connection {
|
|||||||
return this.inner.dropTable(name, namespacePath ?? []);
|
return this.inner.dropTable(name, namespacePath ?? []);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async dropTableAsync(name: string, namespacePath?: string[]): Promise<Job> {
|
||||||
|
return this.inner.dropTableAsync(name, namespacePath ?? []);
|
||||||
|
}
|
||||||
|
|
||||||
async dropAllTables(namespacePath?: string[]): Promise<void> {
|
async dropAllTables(namespacePath?: string[]): Promise<void> {
|
||||||
return this.inner.dropAllTables(namespacePath ?? []);
|
return this.inner.dropAllTables(namespacePath ?? []);
|
||||||
}
|
}
|
||||||
@@ -722,6 +772,30 @@ export class LocalConnection extends Connection {
|
|||||||
options?.newNamespacePath,
|
options?.newNamespacePath,
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
job(jobId: string): Job {
|
||||||
|
return this.inner.job(jobId);
|
||||||
|
}
|
||||||
|
|
||||||
|
async listJobs(): Promise<JobInfo[]> {
|
||||||
|
return this.inner.listJobs();
|
||||||
|
}
|
||||||
|
|
||||||
|
async getJob(jobId: string): Promise<JobDescription | null> {
|
||||||
|
return this.inner.getJob(jobId);
|
||||||
|
}
|
||||||
|
|
||||||
|
async cancelJob(jobId: string): Promise<boolean> {
|
||||||
|
return this.inner.cancelJob(jobId);
|
||||||
|
}
|
||||||
|
|
||||||
|
async jobHistory(jobId?: string): Promise<ArrowTable> {
|
||||||
|
const buf = await this.inner.jobHistory(jobId);
|
||||||
|
if (buf.length === 0) {
|
||||||
|
return new ArrowTable();
|
||||||
|
}
|
||||||
|
return tableFromIPC(buf);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
+12
-1
@@ -50,6 +50,7 @@ export {
|
|||||||
MergeResult,
|
MergeResult,
|
||||||
AddResult,
|
AddResult,
|
||||||
AddColumnsResult,
|
AddColumnsResult,
|
||||||
|
RefreshColumnResult,
|
||||||
AlterColumnsResult,
|
AlterColumnsResult,
|
||||||
UpdateFieldMetadataResult,
|
UpdateFieldMetadataResult,
|
||||||
DeleteResult,
|
DeleteResult,
|
||||||
@@ -85,7 +86,13 @@ export {
|
|||||||
RenameTableOptions,
|
RenameTableOptions,
|
||||||
} from "./connection";
|
} from "./connection";
|
||||||
|
|
||||||
export { Session } from "./native.js";
|
export {
|
||||||
|
Job,
|
||||||
|
JobDescription,
|
||||||
|
JobFailureInfo,
|
||||||
|
JobInfo,
|
||||||
|
Session,
|
||||||
|
} from "./native.js";
|
||||||
|
|
||||||
export {
|
export {
|
||||||
ExecutableQuery,
|
ExecutableQuery,
|
||||||
@@ -140,6 +147,10 @@ export {
|
|||||||
FtsToken,
|
FtsToken,
|
||||||
TokenizeTableOptions,
|
TokenizeTableOptions,
|
||||||
LsmWriteSpec,
|
LsmWriteSpec,
|
||||||
|
LsmStats,
|
||||||
|
BucketStats,
|
||||||
|
GenerationStats,
|
||||||
|
MemtableStats,
|
||||||
ColumnAlteration,
|
ColumnAlteration,
|
||||||
FieldMetadataUpdate,
|
FieldMetadataUpdate,
|
||||||
} from "./table";
|
} from "./table";
|
||||||
|
|||||||
+174
-29
@@ -9,7 +9,7 @@
|
|||||||
// comes from the exact same library instance. This is not always the case
|
// comes from the exact same library instance. This is not always the case
|
||||||
// and so we must sanitize the input to ensure that it is compatible.
|
// and so we must sanitize the input to ensure that it is compatible.
|
||||||
|
|
||||||
import { BufferType, Data } from "apache-arrow";
|
import { BufferType, Data, Vector } from "apache-arrow";
|
||||||
import type { IntBitWidth, TKeys, TimeBitWidth } from "apache-arrow/type";
|
import type { IntBitWidth, TKeys, TimeBitWidth } from "apache-arrow/type";
|
||||||
import {
|
import {
|
||||||
Binary,
|
Binary,
|
||||||
@@ -74,6 +74,20 @@ import {
|
|||||||
Utf8,
|
Utf8,
|
||||||
} from "./arrow";
|
} from "./arrow";
|
||||||
|
|
||||||
|
type SanitizationContext = {
|
||||||
|
types: WeakMap<object, DataType>;
|
||||||
|
vectors: WeakMap<object, Vector>;
|
||||||
|
data: WeakMap<object, Data<DataType>>;
|
||||||
|
};
|
||||||
|
|
||||||
|
function createSanitizationContext(): SanitizationContext {
|
||||||
|
return {
|
||||||
|
types: new WeakMap(),
|
||||||
|
vectors: new WeakMap(),
|
||||||
|
data: new WeakMap(),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
export function sanitizeMetadata(
|
export function sanitizeMetadata(
|
||||||
metadataLike?: unknown,
|
metadataLike?: unknown,
|
||||||
): Map<string, string> | undefined {
|
): Map<string, string> | undefined {
|
||||||
@@ -186,6 +200,13 @@ export function sanitizeInterval(typeLike: object) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeList(typeLike: object) {
|
export function sanitizeList(typeLike: object) {
|
||||||
|
return sanitizeListWithContext(typeLike, createSanitizationContext());
|
||||||
|
}
|
||||||
|
|
||||||
|
function sanitizeListWithContext(
|
||||||
|
typeLike: object,
|
||||||
|
context: SanitizationContext,
|
||||||
|
) {
|
||||||
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
|
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
|
||||||
throw Error(
|
throw Error(
|
||||||
"Expected a List type to have an array-like `children` property",
|
"Expected a List type to have an array-like `children` property",
|
||||||
@@ -194,19 +215,35 @@ export function sanitizeList(typeLike: object) {
|
|||||||
if (typeLike.children.length !== 1) {
|
if (typeLike.children.length !== 1) {
|
||||||
throw Error("Expected a List type to have exactly one child");
|
throw Error("Expected a List type to have exactly one child");
|
||||||
}
|
}
|
||||||
return new List(sanitizeField(typeLike.children[0]));
|
return new List(sanitizeFieldWithContext(typeLike.children[0], context));
|
||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeStruct(typeLike: object) {
|
export function sanitizeStruct(typeLike: object) {
|
||||||
|
return sanitizeStructWithContext(typeLike, createSanitizationContext());
|
||||||
|
}
|
||||||
|
|
||||||
|
function sanitizeStructWithContext(
|
||||||
|
typeLike: object,
|
||||||
|
context: SanitizationContext,
|
||||||
|
) {
|
||||||
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
|
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
|
||||||
throw Error(
|
throw Error(
|
||||||
"Expected a Struct type to have an array-like `children` property",
|
"Expected a Struct type to have an array-like `children` property",
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
return new Struct(typeLike.children.map((child) => sanitizeField(child)));
|
return new Struct(
|
||||||
|
typeLike.children.map((child) => sanitizeFieldWithContext(child, context)),
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeUnion(typeLike: object) {
|
export function sanitizeUnion(typeLike: object) {
|
||||||
|
return sanitizeUnionWithContext(typeLike, createSanitizationContext());
|
||||||
|
}
|
||||||
|
|
||||||
|
function sanitizeUnionWithContext(
|
||||||
|
typeLike: object,
|
||||||
|
context: SanitizationContext,
|
||||||
|
) {
|
||||||
if (
|
if (
|
||||||
!("typeIds" in typeLike) ||
|
!("typeIds" in typeLike) ||
|
||||||
!("mode" in typeLike) ||
|
!("mode" in typeLike) ||
|
||||||
@@ -226,7 +263,7 @@ export function sanitizeUnion(typeLike: object) {
|
|||||||
typeLike.mode,
|
typeLike.mode,
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: skip
|
// biome-ignore lint/suspicious/noExplicitAny: skip
|
||||||
typeLike.typeIds as any,
|
typeLike.typeIds as any,
|
||||||
typeLike.children.map((child) => sanitizeField(child)),
|
typeLike.children.map((child) => sanitizeFieldWithContext(child, context)),
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -234,6 +271,19 @@ export function sanitizeTypedUnion(
|
|||||||
typeLike: object,
|
typeLike: object,
|
||||||
// eslint-disable-next-line @typescript-eslint/naming-convention
|
// eslint-disable-next-line @typescript-eslint/naming-convention
|
||||||
UnionType: typeof DenseUnion | typeof SparseUnion,
|
UnionType: typeof DenseUnion | typeof SparseUnion,
|
||||||
|
) {
|
||||||
|
return sanitizeTypedUnionWithContext(
|
||||||
|
typeLike,
|
||||||
|
UnionType,
|
||||||
|
createSanitizationContext(),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
function sanitizeTypedUnionWithContext(
|
||||||
|
typeLike: object,
|
||||||
|
// eslint-disable-next-line @typescript-eslint/naming-convention
|
||||||
|
UnionType: typeof DenseUnion | typeof SparseUnion,
|
||||||
|
context: SanitizationContext,
|
||||||
) {
|
) {
|
||||||
if (!("typeIds" in typeLike)) {
|
if (!("typeIds" in typeLike)) {
|
||||||
throw Error(
|
throw Error(
|
||||||
@@ -248,7 +298,7 @@ export function sanitizeTypedUnion(
|
|||||||
|
|
||||||
return new UnionType(
|
return new UnionType(
|
||||||
typeLike.typeIds as Int32Array | number[],
|
typeLike.typeIds as Int32Array | number[],
|
||||||
typeLike.children.map((child) => sanitizeField(child)),
|
typeLike.children.map((child) => sanitizeFieldWithContext(child, context)),
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -262,6 +312,16 @@ export function sanitizeFixedSizeBinary(typeLike: object) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeFixedSizeList(typeLike: object) {
|
export function sanitizeFixedSizeList(typeLike: object) {
|
||||||
|
return sanitizeFixedSizeListWithContext(
|
||||||
|
typeLike,
|
||||||
|
createSanitizationContext(),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
function sanitizeFixedSizeListWithContext(
|
||||||
|
typeLike: object,
|
||||||
|
context: SanitizationContext,
|
||||||
|
) {
|
||||||
if (!("listSize" in typeLike) || typeof typeLike.listSize !== "number") {
|
if (!("listSize" in typeLike) || typeof typeLike.listSize !== "number") {
|
||||||
throw Error("Expected a FixedSizeList type to have a `listSize` property");
|
throw Error("Expected a FixedSizeList type to have a `listSize` property");
|
||||||
}
|
}
|
||||||
@@ -275,11 +335,18 @@ export function sanitizeFixedSizeList(typeLike: object) {
|
|||||||
}
|
}
|
||||||
return new FixedSizeList(
|
return new FixedSizeList(
|
||||||
typeLike.listSize,
|
typeLike.listSize,
|
||||||
sanitizeField(typeLike.children[0]),
|
sanitizeFieldWithContext(typeLike.children[0], context),
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeMap(typeLike: object) {
|
export function sanitizeMap(typeLike: object) {
|
||||||
|
return sanitizeMapWithContext(typeLike, createSanitizationContext());
|
||||||
|
}
|
||||||
|
|
||||||
|
function sanitizeMapWithContext(
|
||||||
|
typeLike: object,
|
||||||
|
context: SanitizationContext,
|
||||||
|
) {
|
||||||
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
|
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
|
||||||
throw Error(
|
throw Error(
|
||||||
"Expected a Map type to have an array-like `children` property",
|
"Expected a Map type to have an array-like `children` property",
|
||||||
@@ -292,7 +359,10 @@ export function sanitizeMap(typeLike: object) {
|
|||||||
throw Error("Expected a Map type to have exactly one child");
|
throw Error("Expected a Map type to have exactly one child");
|
||||||
}
|
}
|
||||||
|
|
||||||
return new Map_(sanitizeField(typeLike.children[0]), typeLike.keysSorted);
|
return new Map_(
|
||||||
|
sanitizeFieldWithContext(typeLike.children[0], context),
|
||||||
|
typeLike.keysSorted,
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeDuration(typeLike: object) {
|
export function sanitizeDuration(typeLike: object) {
|
||||||
@@ -303,6 +373,13 @@ export function sanitizeDuration(typeLike: object) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeDictionary(typeLike: object) {
|
export function sanitizeDictionary(typeLike: object) {
|
||||||
|
return sanitizeDictionaryWithContext(typeLike, createSanitizationContext());
|
||||||
|
}
|
||||||
|
|
||||||
|
function sanitizeDictionaryWithContext(
|
||||||
|
typeLike: object,
|
||||||
|
context: SanitizationContext,
|
||||||
|
) {
|
||||||
if (!("id" in typeLike) || typeof typeLike.id !== "number") {
|
if (!("id" in typeLike) || typeof typeLike.id !== "number") {
|
||||||
throw Error("Expected a Dictionary type to have an `id` property");
|
throw Error("Expected a Dictionary type to have an `id` property");
|
||||||
}
|
}
|
||||||
@@ -316,8 +393,8 @@ export function sanitizeDictionary(typeLike: object) {
|
|||||||
throw Error("Expected a Dictionary type to have an `isOrdered` property");
|
throw Error("Expected a Dictionary type to have an `isOrdered` property");
|
||||||
}
|
}
|
||||||
return new Dictionary(
|
return new Dictionary(
|
||||||
sanitizeType(typeLike.dictionary),
|
sanitizeTypeWithContext(typeLike.dictionary, context),
|
||||||
sanitizeType(typeLike.indices) as TKeys,
|
sanitizeTypeWithContext(typeLike.indices, context) as TKeys,
|
||||||
typeLike.id,
|
typeLike.id,
|
||||||
typeLike.isOrdered,
|
typeLike.isOrdered,
|
||||||
);
|
);
|
||||||
@@ -325,12 +402,23 @@ export function sanitizeDictionary(typeLike: object) {
|
|||||||
|
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: skip
|
// biome-ignore lint/suspicious/noExplicitAny: skip
|
||||||
export function sanitizeType(typeLike: unknown): DataType<any> {
|
export function sanitizeType(typeLike: unknown): DataType<any> {
|
||||||
|
return sanitizeTypeWithContext(typeLike, createSanitizationContext());
|
||||||
|
}
|
||||||
|
|
||||||
|
function sanitizeTypeWithContext(
|
||||||
|
typeLike: unknown,
|
||||||
|
context: SanitizationContext,
|
||||||
|
): DataType {
|
||||||
if (typeof typeLike === "string") {
|
if (typeof typeLike === "string") {
|
||||||
return dataTypeFromName(typeLike);
|
return dataTypeFromName(typeLike);
|
||||||
}
|
}
|
||||||
if (typeof typeLike !== "object" || typeLike === null) {
|
if (typeof typeLike !== "object" || typeLike === null) {
|
||||||
throw Error("Expected a Type but object was null/undefined");
|
throw Error("Expected a Type but object was null/undefined");
|
||||||
}
|
}
|
||||||
|
const cached = context.types.get(typeLike);
|
||||||
|
if (cached !== undefined) {
|
||||||
|
return cached;
|
||||||
|
}
|
||||||
if (
|
if (
|
||||||
!("typeId" in typeLike) ||
|
!("typeId" in typeLike) ||
|
||||||
!(
|
!(
|
||||||
@@ -349,6 +437,16 @@ export function sanitizeType(typeLike: unknown): DataType<any> {
|
|||||||
throw Error("Type's typeId property was not a function or number");
|
throw Error("Type's typeId property was not a function or number");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const type = sanitizeTypeById(typeLike, typeId, context);
|
||||||
|
context.types.set(typeLike, type);
|
||||||
|
return type;
|
||||||
|
}
|
||||||
|
|
||||||
|
function sanitizeTypeById(
|
||||||
|
typeLike: object,
|
||||||
|
typeId: Type,
|
||||||
|
context: SanitizationContext,
|
||||||
|
): DataType {
|
||||||
switch (typeId) {
|
switch (typeId) {
|
||||||
case Type.NONE:
|
case Type.NONE:
|
||||||
throw Error("Received a Type with a typeId of NONE");
|
throw Error("Received a Type with a typeId of NONE");
|
||||||
@@ -375,21 +473,21 @@ export function sanitizeType(typeLike: unknown): DataType<any> {
|
|||||||
case Type.Interval:
|
case Type.Interval:
|
||||||
return sanitizeInterval(typeLike);
|
return sanitizeInterval(typeLike);
|
||||||
case Type.List:
|
case Type.List:
|
||||||
return sanitizeList(typeLike);
|
return sanitizeListWithContext(typeLike, context);
|
||||||
case Type.Struct:
|
case Type.Struct:
|
||||||
return sanitizeStruct(typeLike);
|
return sanitizeStructWithContext(typeLike, context);
|
||||||
case Type.Union:
|
case Type.Union:
|
||||||
return sanitizeUnion(typeLike);
|
return sanitizeUnionWithContext(typeLike, context);
|
||||||
case Type.FixedSizeBinary:
|
case Type.FixedSizeBinary:
|
||||||
return sanitizeFixedSizeBinary(typeLike);
|
return sanitizeFixedSizeBinary(typeLike);
|
||||||
case Type.FixedSizeList:
|
case Type.FixedSizeList:
|
||||||
return sanitizeFixedSizeList(typeLike);
|
return sanitizeFixedSizeListWithContext(typeLike, context);
|
||||||
case Type.Map:
|
case Type.Map:
|
||||||
return sanitizeMap(typeLike);
|
return sanitizeMapWithContext(typeLike, context);
|
||||||
case Type.Duration:
|
case Type.Duration:
|
||||||
return sanitizeDuration(typeLike);
|
return sanitizeDuration(typeLike);
|
||||||
case Type.Dictionary:
|
case Type.Dictionary:
|
||||||
return sanitizeDictionary(typeLike);
|
return sanitizeDictionaryWithContext(typeLike, context);
|
||||||
case Type.Int8:
|
case Type.Int8:
|
||||||
return new Int8();
|
return new Int8();
|
||||||
case Type.Int16:
|
case Type.Int16:
|
||||||
@@ -433,9 +531,9 @@ export function sanitizeType(typeLike: unknown): DataType<any> {
|
|||||||
case Type.TimestampSecond:
|
case Type.TimestampSecond:
|
||||||
return sanitizeTypedTimestamp(typeLike, TimestampSecond);
|
return sanitizeTypedTimestamp(typeLike, TimestampSecond);
|
||||||
case Type.DenseUnion:
|
case Type.DenseUnion:
|
||||||
return sanitizeTypedUnion(typeLike, DenseUnion);
|
return sanitizeTypedUnionWithContext(typeLike, DenseUnion, context);
|
||||||
case Type.SparseUnion:
|
case Type.SparseUnion:
|
||||||
return sanitizeTypedUnion(typeLike, SparseUnion);
|
return sanitizeTypedUnionWithContext(typeLike, SparseUnion, context);
|
||||||
case Type.IntervalDayTime:
|
case Type.IntervalDayTime:
|
||||||
return new IntervalDayTime();
|
return new IntervalDayTime();
|
||||||
case Type.IntervalYearMonth:
|
case Type.IntervalYearMonth:
|
||||||
@@ -454,6 +552,13 @@ export function sanitizeType(typeLike: unknown): DataType<any> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeField(fieldLike: unknown): Field {
|
export function sanitizeField(fieldLike: unknown): Field {
|
||||||
|
return sanitizeFieldWithContext(fieldLike, createSanitizationContext());
|
||||||
|
}
|
||||||
|
|
||||||
|
function sanitizeFieldWithContext(
|
||||||
|
fieldLike: unknown,
|
||||||
|
context: SanitizationContext,
|
||||||
|
): Field {
|
||||||
if (fieldLike instanceof Field) {
|
if (fieldLike instanceof Field) {
|
||||||
return fieldLike;
|
return fieldLike;
|
||||||
}
|
}
|
||||||
@@ -471,7 +576,7 @@ export function sanitizeField(fieldLike: unknown): Field {
|
|||||||
}
|
}
|
||||||
let type: DataType;
|
let type: DataType;
|
||||||
try {
|
try {
|
||||||
type = sanitizeType(fieldLike.type);
|
type = sanitizeTypeWithContext(fieldLike.type, context);
|
||||||
} catch (error: unknown) {
|
} catch (error: unknown) {
|
||||||
throw Error(
|
throw Error(
|
||||||
`Unable to sanitize type for field: ${fieldLike.name} due to error: ${error}`,
|
`Unable to sanitize type for field: ${fieldLike.name} due to error: ${error}`,
|
||||||
@@ -501,6 +606,13 @@ export function sanitizeField(fieldLike: unknown): Field {
|
|||||||
* than lancedb is using.
|
* than lancedb is using.
|
||||||
*/
|
*/
|
||||||
export function sanitizeSchema(schemaLike: SchemaLike): Schema {
|
export function sanitizeSchema(schemaLike: SchemaLike): Schema {
|
||||||
|
return sanitizeSchemaWithContext(schemaLike, createSanitizationContext());
|
||||||
|
}
|
||||||
|
|
||||||
|
function sanitizeSchemaWithContext(
|
||||||
|
schemaLike: SchemaLike,
|
||||||
|
context: SanitizationContext,
|
||||||
|
): Schema {
|
||||||
if (schemaLike instanceof Schema) {
|
if (schemaLike instanceof Schema) {
|
||||||
return schemaLike;
|
return schemaLike;
|
||||||
}
|
}
|
||||||
@@ -522,7 +634,7 @@ export function sanitizeSchema(schemaLike: SchemaLike): Schema {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
const sanitizedFields = schemaLike.fields.map((field) =>
|
const sanitizedFields = schemaLike.fields.map((field) =>
|
||||||
sanitizeField(field),
|
sanitizeFieldWithContext(field, context),
|
||||||
);
|
);
|
||||||
return new Schema(sanitizedFields, metadata);
|
return new Schema(sanitizedFields, metadata);
|
||||||
}
|
}
|
||||||
@@ -544,13 +656,18 @@ export function sanitizeTable(tableLike: TableLike): Table {
|
|||||||
"The table passed in does not appear to be a table (no 'columns' property)",
|
"The table passed in does not appear to be a table (no 'columns' property)",
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
const schema = sanitizeSchema(tableLike.schema);
|
const context = createSanitizationContext();
|
||||||
|
const schema = sanitizeSchemaWithContext(tableLike.schema, context);
|
||||||
const batches = tableLike.batches.map(sanitizeRecordBatch);
|
const batches = tableLike.batches.map((batch) =>
|
||||||
|
sanitizeRecordBatch(batch, context),
|
||||||
|
);
|
||||||
return new Table(schema, batches);
|
return new Table(schema, batches);
|
||||||
}
|
}
|
||||||
|
|
||||||
function sanitizeRecordBatch(batchLike: RecordBatchLike): RecordBatch {
|
function sanitizeRecordBatch(
|
||||||
|
batchLike: RecordBatchLike,
|
||||||
|
context: SanitizationContext,
|
||||||
|
): RecordBatch {
|
||||||
if (batchLike instanceof RecordBatch) {
|
if (batchLike instanceof RecordBatch) {
|
||||||
return batchLike;
|
return batchLike;
|
||||||
}
|
}
|
||||||
@@ -567,19 +684,43 @@ function sanitizeRecordBatch(batchLike: RecordBatchLike): RecordBatch {
|
|||||||
"The record batch passed in does not appear to be a record batch (no 'data' property)",
|
"The record batch passed in does not appear to be a record batch (no 'data' property)",
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
const schema = sanitizeSchema(batchLike.schema);
|
const schema = sanitizeSchemaWithContext(batchLike.schema, context);
|
||||||
const data = sanitizeData(batchLike.data);
|
const data = sanitizeData(batchLike.data, context) as Data<Struct>;
|
||||||
return new RecordBatch(schema, data);
|
return new RecordBatch(schema, data);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
type DictionaryVectorLike = {
|
||||||
|
data: readonly DataLike[];
|
||||||
|
};
|
||||||
|
|
||||||
|
type DictionaryDataLike = DataLike & {
|
||||||
|
dictionary?: DictionaryVectorLike;
|
||||||
|
};
|
||||||
|
|
||||||
function sanitizeData(
|
function sanitizeData(
|
||||||
dataLike: DataLike,
|
dataLike: DataLike,
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: <explanation>
|
context: SanitizationContext,
|
||||||
): import("apache-arrow").Data<Struct<any>> {
|
): Data<DataType> {
|
||||||
if (dataLike instanceof Data) {
|
if (dataLike instanceof Data) {
|
||||||
return dataLike;
|
return dataLike;
|
||||||
}
|
}
|
||||||
return new Data(
|
const cachedData = context.data.get(dataLike);
|
||||||
dataLike.type,
|
if (cachedData !== undefined) {
|
||||||
|
return cachedData;
|
||||||
|
}
|
||||||
|
const dictionaryLike = (dataLike as DictionaryDataLike).dictionary;
|
||||||
|
let dictionary: Vector | undefined;
|
||||||
|
if (dictionaryLike !== undefined) {
|
||||||
|
dictionary = context.vectors.get(dictionaryLike);
|
||||||
|
if (dictionary === undefined) {
|
||||||
|
dictionary = new Vector(
|
||||||
|
dictionaryLike.data.map((data) => sanitizeData(data, context)),
|
||||||
|
);
|
||||||
|
context.vectors.set(dictionaryLike, dictionary);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const data = new Data(
|
||||||
|
sanitizeTypeWithContext(dataLike.type, context),
|
||||||
dataLike.offset,
|
dataLike.offset,
|
||||||
dataLike.length,
|
dataLike.length,
|
||||||
dataLike.nullCount,
|
dataLike.nullCount,
|
||||||
@@ -589,7 +730,11 @@ function sanitizeData(
|
|||||||
[BufferType.VALIDITY]: dataLike.nullBitmap,
|
[BufferType.VALIDITY]: dataLike.nullBitmap,
|
||||||
[BufferType.TYPE]: dataLike.typeIds,
|
[BufferType.TYPE]: dataLike.typeIds,
|
||||||
},
|
},
|
||||||
|
dataLike.children.map((child) => sanitizeData(child, context)),
|
||||||
|
dictionary,
|
||||||
);
|
);
|
||||||
|
context.data.set(dataLike, data);
|
||||||
|
return data;
|
||||||
}
|
}
|
||||||
|
|
||||||
const constructorsByTypeName = {
|
const constructorsByTypeName = {
|
||||||
|
|||||||
+202
-6
@@ -30,8 +30,11 @@ import {
|
|||||||
DropColumnsResult,
|
DropColumnsResult,
|
||||||
IndexConfig,
|
IndexConfig,
|
||||||
IndexStatistics,
|
IndexStatistics,
|
||||||
|
Job,
|
||||||
|
LsmStats,
|
||||||
Branches as NativeBranches,
|
Branches as NativeBranches,
|
||||||
OptimizeStats,
|
OptimizeStats,
|
||||||
|
RefreshColumnResult,
|
||||||
TableStatistics,
|
TableStatistics,
|
||||||
Tags,
|
Tags,
|
||||||
UpdateFieldMetadataResult,
|
UpdateFieldMetadataResult,
|
||||||
@@ -48,6 +51,12 @@ import {
|
|||||||
import { sanitizeType } from "./sanitize";
|
import { sanitizeType } from "./sanitize";
|
||||||
import { IntoSql, toSQL } from "./util";
|
import { IntoSql, toSQL } from "./util";
|
||||||
export { IndexConfig } from "./native";
|
export { IndexConfig } from "./native";
|
||||||
|
export {
|
||||||
|
BucketStats,
|
||||||
|
GenerationStats,
|
||||||
|
LsmStats,
|
||||||
|
MemtableStats,
|
||||||
|
} from "./native";
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Progress snapshot for a write operation, delivered to the `progress`
|
* Progress snapshot for a write operation, delivered to the `progress`
|
||||||
@@ -196,7 +205,11 @@ export interface LsmWriteSpec {
|
|||||||
column?: string;
|
column?: string;
|
||||||
/** Bucket variant: the number of buckets, in `[1, 1024]`. */
|
/** Bucket variant: the number of buckets, in `[1, 1024]`. */
|
||||||
numBuckets?: number;
|
numBuckets?: number;
|
||||||
/** Names of indexes the MemWAL should keep up to date during writes. */
|
/**
|
||||||
|
* Indexes the MemWAL keeps up to date. Omit to maintain every supported
|
||||||
|
* index, resolved on install — a snapshot, so indexes created later are not
|
||||||
|
* maintained. Pass `[]` for none.
|
||||||
|
*/
|
||||||
maintainedIndexes?: string[];
|
maintainedIndexes?: string[];
|
||||||
/** Default `ShardWriter` configuration recorded in the MemWAL index. */
|
/** Default `ShardWriter` configuration recorded in the MemWAL index. */
|
||||||
writerConfigDefaults?: Record<string, string>;
|
writerConfigDefaults?: Record<string, string>;
|
||||||
@@ -358,6 +371,17 @@ export abstract class Table {
|
|||||||
options?: Partial<IndexOptions>,
|
options?: Partial<IndexOptions>,
|
||||||
): Promise<void>;
|
): Promise<void>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Create an index, returning a handle to the indexing job.
|
||||||
|
*
|
||||||
|
* The job may already be complete when returned; callers must not assume
|
||||||
|
* the index exists until {@link Job.wait} resolves.
|
||||||
|
*/
|
||||||
|
abstract createIndexAsync(
|
||||||
|
column: string,
|
||||||
|
options?: Partial<IndexOptions>,
|
||||||
|
): Promise<Job>;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Drop an index from the table.
|
* Drop an index from the table.
|
||||||
*
|
*
|
||||||
@@ -509,18 +533,75 @@ export abstract class Table {
|
|||||||
abstract vectorSearch(vector: IntoVector | MultiVector): VectorQuery;
|
abstract vectorSearch(vector: IntoVector | MultiVector): VectorQuery;
|
||||||
/**
|
/**
|
||||||
* Add new columns with defined values.
|
* Add new columns with defined values.
|
||||||
|
*
|
||||||
|
* The `{ computed }` form stores the expression rather than evaluating it
|
||||||
|
* now: the column is committed with no values, and rows get them from
|
||||||
|
* {@link Table#refreshColumn}. Declaring one therefore costs the same on a
|
||||||
|
* large table as on an empty one.
|
||||||
|
*
|
||||||
|
* A refresh does not revisit rows it has already filled, so mutating an
|
||||||
|
* input leaves the value computed at fill time; recomputing means dropping
|
||||||
|
* the column and declaring it again. While a declaration reads a column,
|
||||||
|
* that column cannot be renamed, retyped or dropped.
|
||||||
|
*
|
||||||
|
* On LanceDB Cloud and Enterprise the expression is planned by the
|
||||||
|
* server, and the refresh runs as a server job -- see
|
||||||
|
* {@link Table#refreshColumnAsync}.
|
||||||
* @param {AddColumnsSql[] | Field | Field[] | Schema} newColumnTransforms Either:
|
* @param {AddColumnsSql[] | Field | Field[] | Schema} newColumnTransforms Either:
|
||||||
* - An array of objects with column names and SQL expressions to calculate values
|
* - An array of objects with column names and SQL expressions to calculate values
|
||||||
* - A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
* - A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
||||||
* - An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
* - An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
||||||
* - An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
* - An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
||||||
|
* - `{ computed }`, declaring columns defined by a SQL expression whose type and inputs are derived from it
|
||||||
* @returns {Promise<AddColumnsResult>} A promise that resolves to an object
|
* @returns {Promise<AddColumnsResult>} A promise that resolves to an object
|
||||||
* containing the new version number of the table after adding the columns.
|
* containing the new version number of the table after adding the columns.
|
||||||
|
* @example
|
||||||
|
* ```ts
|
||||||
|
* await table.addColumns({ computed: [{ name: "doubled", valueSql: "x * 2" }] });
|
||||||
|
* const { rowsFilled } = await table.refreshColumn("doubled");
|
||||||
|
* ```
|
||||||
*/
|
*/
|
||||||
abstract addColumns(
|
abstract addColumns(
|
||||||
newColumnTransforms: AddColumnsSql[] | Field | Field[] | Schema,
|
newColumnTransforms:
|
||||||
|
| AddColumnsSql[]
|
||||||
|
| Field
|
||||||
|
| Field[]
|
||||||
|
| Schema
|
||||||
|
| { computed: AddColumnsSql[] },
|
||||||
): Promise<AddColumnsResult>;
|
): Promise<AddColumnsResult>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Fill the rows of a computed column that hold no value yet.
|
||||||
|
*
|
||||||
|
* Rows appended since the last refresh are filled by the next one; rows
|
||||||
|
* already filled are left as they are, so the call is idempotent and does
|
||||||
|
* not observe a mutated input. Local tables only: a remote refresh runs
|
||||||
|
* as a server job, through {@link Table#refreshColumnAsync}.
|
||||||
|
* @param {string} column The name of the computed column to fill.
|
||||||
|
* @returns {Promise<RefreshColumnResult>} A promise that resolves to the
|
||||||
|
* number of rows filled and the new version number of the table.
|
||||||
|
*/
|
||||||
|
abstract refreshColumn(column: string): Promise<RefreshColumnResult>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Like {@link Table#refreshColumn}, but returns a handle to the refresh
|
||||||
|
* job instead of blocking until it completes.
|
||||||
|
*
|
||||||
|
* The job may already be complete when returned; callers must not assume
|
||||||
|
* the column is filled until {@link Job.wait} resolves. Invalid input --
|
||||||
|
* an unknown column, or one that is not computed -- rejects here rather
|
||||||
|
* than failing the job. On local tables the job runs in-process; on
|
||||||
|
* LanceDB Cloud and Enterprise it is the server's backfill job.
|
||||||
|
* @param {string} column The name of the computed column to fill.
|
||||||
|
* @example
|
||||||
|
* ```ts
|
||||||
|
* const job = await table.refreshColumnAsync("doubled");
|
||||||
|
* await job.wait();
|
||||||
|
* console.log(await job.status()); // "finished"
|
||||||
|
* ```
|
||||||
|
*/
|
||||||
|
abstract refreshColumnAsync(column: string): Promise<Job>;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Alter the name or nullability of columns.
|
* Alter the name or nullability of columns.
|
||||||
* @param {ColumnAlteration[]} columnAlterations One or more alterations to
|
* @param {ColumnAlteration[]} columnAlterations One or more alterations to
|
||||||
@@ -583,6 +664,11 @@ export abstract class Table {
|
|||||||
* All variants require the table to have an unenforced primary key
|
* All variants require the table to have an unenforced primary key
|
||||||
* ({@link Table#setUnenforcedPrimaryKey}); bucket sharding additionally
|
* ({@link Table#setUnenforcedPrimaryKey}); bucket sharding additionally
|
||||||
* requires it to be the single column being bucketed.
|
* requires it to be the single column being bucketed.
|
||||||
|
*
|
||||||
|
* Omitting `maintainedIndexes` maintains every index on the table, resolved
|
||||||
|
* here, failing if one cannot be maintained — name them to install anyway.
|
||||||
|
* Naming them pins an exact set, and a still-building index is rejected
|
||||||
|
* rather than quietly omitted.
|
||||||
* @param {LsmWriteSpec} spec The sharding spec to install.
|
* @param {LsmWriteSpec} spec The sharding spec to install.
|
||||||
* @returns {Promise<void>}
|
* @returns {Promise<void>}
|
||||||
* @example
|
* @example
|
||||||
@@ -610,9 +696,10 @@ export abstract class Table {
|
|||||||
*
|
*
|
||||||
* Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
|
* Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
|
||||||
* spec has been set, or it was removed with {@link Table#unsetLsmWriteSpec}).
|
* spec has been set, or it was removed with {@link Table#unsetLsmWriteSpec}).
|
||||||
* The returned spec — including its `maintainedIndexes` and
|
* The returned spec mirrors what was passed to
|
||||||
* `writerConfigDefaults` — mirrors what was passed to
|
* {@link Table#setLsmWriteSpec}, except that `maintainedIndexes` always
|
||||||
* {@link Table#setLsmWriteSpec}.
|
* reports the concrete list resolved when the spec was set — `undefined`
|
||||||
|
* never round-trips.
|
||||||
* @returns {Promise<LsmWriteSpec | undefined>}
|
* @returns {Promise<LsmWriteSpec | undefined>}
|
||||||
*/
|
*/
|
||||||
abstract getLsmWriteSpec(): Promise<LsmWriteSpec | undefined>;
|
abstract getLsmWriteSpec(): Promise<LsmWriteSpec | undefined>;
|
||||||
@@ -626,6 +713,59 @@ export abstract class Table {
|
|||||||
* @returns {Promise<void>}
|
* @returns {Promise<void>}
|
||||||
*/
|
*/
|
||||||
abstract closeLsmWriters(): Promise<void>;
|
abstract closeLsmWriters(): Promise<void>;
|
||||||
|
/**
|
||||||
|
* Seal every bucket's active memtable into a new L0 generation.
|
||||||
|
*
|
||||||
|
* Returns once the seal is committed. Sealing an empty memtable is a no-op,
|
||||||
|
* so this is safe to call repeatedly.
|
||||||
|
* @returns {Promise<void>}
|
||||||
|
*/
|
||||||
|
abstract flushLsm(): Promise<void>;
|
||||||
|
/**
|
||||||
|
* Trigger a background L0 → base compaction pass per bucket.
|
||||||
|
*
|
||||||
|
* Returns once the passes are *dispatched*, not once they finish — watch
|
||||||
|
* {@link Table#getLsmStats} for progress, or use
|
||||||
|
* {@link Table#checkpointLsm} to wait for convergence.
|
||||||
|
* @returns {Promise<void>}
|
||||||
|
*/
|
||||||
|
abstract compactLsm(): Promise<void>;
|
||||||
|
/**
|
||||||
|
* Converge this table's LSM write path into its base table.
|
||||||
|
*
|
||||||
|
* Seals once, then triggers compaction and polls until the L0 that existed
|
||||||
|
* at the start is gone. The target set is fixed at the start, so
|
||||||
|
* generations created *during* the checkpoint are ignored — that is what
|
||||||
|
* lets it terminate under write load, and what makes it best-effort: it
|
||||||
|
* converges the fresh tier as of some instant. Idempotent, abandonable at
|
||||||
|
* any point, and safe to run on a cadence.
|
||||||
|
*
|
||||||
|
* There is no liveness bound — the compactor pool is shared across tables,
|
||||||
|
* so a checkpoint queued behind unrelated work looks exactly like one that
|
||||||
|
* is merging. The caller owns the deadline.
|
||||||
|
* @returns {Promise<void>}
|
||||||
|
* @example
|
||||||
|
* ```ts
|
||||||
|
* const before = await table.getLsmStats();
|
||||||
|
* await table.checkpointLsm();
|
||||||
|
* const after = await table.getLsmStats();
|
||||||
|
* ```
|
||||||
|
*/
|
||||||
|
abstract checkpointLsm(): Promise<void>;
|
||||||
|
/**
|
||||||
|
* Read live per-bucket LSM state.
|
||||||
|
*
|
||||||
|
* Answers "how far behind is my fresh tier", "which bucket is hot", and
|
||||||
|
* "why is my fresh-tier vector search brute-force". Mutates no table state.
|
||||||
|
*
|
||||||
|
* Resolves to `undefined` only when the LSM write path is not enabled.
|
||||||
|
* @param {boolean} includeGenerationRows Also count rows per L0 generation.
|
||||||
|
* Off by default because each count opens an uncached Lance dataset.
|
||||||
|
* @returns {Promise<LsmStats | undefined>}
|
||||||
|
*/
|
||||||
|
abstract getLsmStats(
|
||||||
|
includeGenerationRows?: boolean,
|
||||||
|
): Promise<LsmStats | undefined>;
|
||||||
/** Retrieve the version of the table */
|
/** Retrieve the version of the table */
|
||||||
|
|
||||||
abstract version(): Promise<number>;
|
abstract version(): Promise<number>;
|
||||||
@@ -940,6 +1080,22 @@ export class LocalTable extends Table {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async createIndexAsync(
|
||||||
|
column: string,
|
||||||
|
options?: Partial<IndexOptions>,
|
||||||
|
): Promise<Job> {
|
||||||
|
// biome-ignore lint/suspicious/noExplicitAny: skip
|
||||||
|
const nativeIndex = (options?.config as any)?.inner;
|
||||||
|
return await this.inner.createIndexAsync(
|
||||||
|
nativeIndex,
|
||||||
|
column,
|
||||||
|
options?.replace,
|
||||||
|
options?.waitTimeoutSeconds,
|
||||||
|
options?.name,
|
||||||
|
options?.train,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
async dropIndex(name: string): Promise<void> {
|
async dropIndex(name: string): Promise<void> {
|
||||||
await this.inner.dropIndex(name);
|
await this.inner.dropIndex(name);
|
||||||
}
|
}
|
||||||
@@ -1050,8 +1206,22 @@ export class LocalTable extends Table {
|
|||||||
// TODO: Support BatchUDF
|
// TODO: Support BatchUDF
|
||||||
|
|
||||||
async addColumns(
|
async addColumns(
|
||||||
newColumnTransforms: AddColumnsSql[] | Field | Field[] | Schema,
|
newColumnTransforms:
|
||||||
|
| AddColumnsSql[]
|
||||||
|
| Field
|
||||||
|
| Field[]
|
||||||
|
| Schema
|
||||||
|
| { computed: AddColumnsSql[] },
|
||||||
): Promise<AddColumnsResult> {
|
): Promise<AddColumnsResult> {
|
||||||
|
// Columns defined by an expression are declared, not materialized here.
|
||||||
|
if (
|
||||||
|
typeof newColumnTransforms === "object" &&
|
||||||
|
!Array.isArray(newColumnTransforms) &&
|
||||||
|
"computed" in newColumnTransforms
|
||||||
|
) {
|
||||||
|
return await this.inner.addComputedColumns(newColumnTransforms.computed);
|
||||||
|
}
|
||||||
|
|
||||||
// Handle single Field -> convert to array of Fields
|
// Handle single Field -> convert to array of Fields
|
||||||
if (newColumnTransforms instanceof Field) {
|
if (newColumnTransforms instanceof Field) {
|
||||||
newColumnTransforms = [newColumnTransforms];
|
newColumnTransforms = [newColumnTransforms];
|
||||||
@@ -1086,6 +1256,14 @@ export class LocalTable extends Table {
|
|||||||
throw new Error("Invalid input type for addColumns");
|
throw new Error("Invalid input type for addColumns");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async refreshColumn(column: string): Promise<RefreshColumnResult> {
|
||||||
|
return await this.inner.refreshColumn(column);
|
||||||
|
}
|
||||||
|
|
||||||
|
async refreshColumnAsync(column: string): Promise<Job> {
|
||||||
|
return await this.inner.refreshColumnAsync(column);
|
||||||
|
}
|
||||||
|
|
||||||
async alterColumns(
|
async alterColumns(
|
||||||
columnAlterations: ColumnAlteration[],
|
columnAlterations: ColumnAlteration[],
|
||||||
): Promise<AlterColumnsResult> {
|
): Promise<AlterColumnsResult> {
|
||||||
@@ -1148,6 +1326,24 @@ export class LocalTable extends Table {
|
|||||||
return await this.inner.closeLsmWriters();
|
return await this.inner.closeLsmWriters();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async flushLsm(): Promise<void> {
|
||||||
|
return await this.inner.flushLsm();
|
||||||
|
}
|
||||||
|
|
||||||
|
async compactLsm(): Promise<void> {
|
||||||
|
return await this.inner.compactLsm();
|
||||||
|
}
|
||||||
|
|
||||||
|
async checkpointLsm(): Promise<void> {
|
||||||
|
return await this.inner.checkpointLsm();
|
||||||
|
}
|
||||||
|
|
||||||
|
async getLsmStats(
|
||||||
|
includeGenerationRows: boolean = false,
|
||||||
|
): Promise<LsmStats | undefined> {
|
||||||
|
return (await this.inner.getLsmStats(includeGenerationRows)) ?? undefined;
|
||||||
|
}
|
||||||
|
|
||||||
async version(): Promise<number> {
|
async version(): Promise<number> {
|
||||||
return await this.inner.version();
|
return await this.inner.version();
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-darwin-arm64",
|
"name": "@lancedb/lancedb-darwin-arm64",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.38.0-beta.3",
|
||||||
"os": ["darwin"],
|
"os": ["darwin"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.darwin-arm64.node",
|
"main": "lancedb.darwin-arm64.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.38.0-beta.3",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.linux-arm64-gnu.node",
|
"main": "lancedb.linux-arm64-gnu.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-arm64-musl",
|
"name": "@lancedb/lancedb-linux-arm64-musl",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.38.0-beta.3",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.linux-arm64-musl.node",
|
"main": "lancedb.linux-arm64-musl.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-x64-gnu",
|
"name": "@lancedb/lancedb-linux-x64-gnu",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.38.0-beta.3",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["x64"],
|
"cpu": ["x64"],
|
||||||
"main": "lancedb.linux-x64-gnu.node",
|
"main": "lancedb.linux-x64-gnu.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-x64-musl",
|
"name": "@lancedb/lancedb-linux-x64-musl",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.38.0-beta.3",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["x64"],
|
"cpu": ["x64"],
|
||||||
"main": "lancedb.linux-x64-musl.node",
|
"main": "lancedb.linux-x64-musl.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-win32-arm64-msvc",
|
"name": "@lancedb/lancedb-win32-arm64-msvc",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.38.0-beta.3",
|
||||||
"os": [
|
"os": [
|
||||||
"win32"
|
"win32"
|
||||||
],
|
],
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-win32-x64-msvc",
|
"name": "@lancedb/lancedb-win32-x64-msvc",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.38.0-beta.3",
|
||||||
"os": ["win32"],
|
"os": ["win32"],
|
||||||
"cpu": ["x64"],
|
"cpu": ["x64"],
|
||||||
"main": "lancedb.win32-x64-msvc.node",
|
"main": "lancedb.win32-x64-msvc.node",
|
||||||
|
|||||||
Generated
+8
-2
@@ -1,12 +1,12 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb",
|
"name": "@lancedb/lancedb",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.38.0-beta.2",
|
||||||
"lockfileVersion": 3,
|
"lockfileVersion": 3,
|
||||||
"requires": true,
|
"requires": true,
|
||||||
"packages": {
|
"packages": {
|
||||||
"": {
|
"": {
|
||||||
"name": "@lancedb/lancedb",
|
"name": "@lancedb/lancedb",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.38.0-beta.2",
|
||||||
"cpu": [
|
"cpu": [
|
||||||
"x64",
|
"x64",
|
||||||
"arm64"
|
"arm64"
|
||||||
@@ -55,7 +55,13 @@
|
|||||||
"openai": "4.29.2"
|
"openai": "4.29.2"
|
||||||
},
|
},
|
||||||
"peerDependencies": {
|
"peerDependencies": {
|
||||||
|
"@types/node": ">=18",
|
||||||
"apache-arrow": ">=15.0.0 <=18.1.0"
|
"apache-arrow": ">=15.0.0 <=18.1.0"
|
||||||
|
},
|
||||||
|
"peerDependenciesMeta": {
|
||||||
|
"@types/node": {
|
||||||
|
"optional": true
|
||||||
|
}
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
"node_modules/@aws-crypto/crc32": {
|
"node_modules/@aws-crypto/crc32": {
|
||||||
|
|||||||
+7
-1
@@ -11,7 +11,7 @@
|
|||||||
"ann"
|
"ann"
|
||||||
],
|
],
|
||||||
"private": false,
|
"private": false,
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.38.0-beta.3",
|
||||||
"main": "dist/index.js",
|
"main": "dist/index.js",
|
||||||
"exports": {
|
"exports": {
|
||||||
".": "./dist/index.js",
|
".": "./dist/index.js",
|
||||||
@@ -101,6 +101,12 @@
|
|||||||
"openai": "4.29.2"
|
"openai": "4.29.2"
|
||||||
},
|
},
|
||||||
"peerDependencies": {
|
"peerDependencies": {
|
||||||
|
"@types/node": ">=18",
|
||||||
"apache-arrow": ">=15.0.0 <=18.1.0"
|
"apache-arrow": ">=15.0.0 <=18.1.0"
|
||||||
|
},
|
||||||
|
"peerDependenciesMeta": {
|
||||||
|
"@types/node": {
|
||||||
|
"optional": true
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -334,12 +334,91 @@ impl Connection {
|
|||||||
.default_error()
|
.default_error()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Start dropping a table and return its cleanup job.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn drop_table_async(
|
||||||
|
&self,
|
||||||
|
name: String,
|
||||||
|
namespace_path: Option<Vec<String>>,
|
||||||
|
) -> napi::Result<crate::job::Job> {
|
||||||
|
let ns = namespace_path.unwrap_or_default();
|
||||||
|
let job = self
|
||||||
|
.get_inner()?
|
||||||
|
.drop_table_async(&name, &ns)
|
||||||
|
.await
|
||||||
|
.default_error()?;
|
||||||
|
Ok(crate::job::Job::new(job))
|
||||||
|
}
|
||||||
|
|
||||||
#[napi(catch_unwind)]
|
#[napi(catch_unwind)]
|
||||||
pub async fn drop_all_tables(&self, namespace_path: Option<Vec<String>>) -> napi::Result<()> {
|
pub async fn drop_all_tables(&self, namespace_path: Option<Vec<String>>) -> napi::Result<()> {
|
||||||
let ns = namespace_path.unwrap_or_default();
|
let ns = namespace_path.unwrap_or_default();
|
||||||
self.get_inner()?.drop_all_tables(&ns).await.default_error()
|
self.get_inner()?.drop_all_tables(&ns).await.default_error()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A `Job` handle for a server-side job by id.
|
||||||
|
///
|
||||||
|
/// The handle is constructed without a server round trip; an unknown id
|
||||||
|
/// surfaces when the handle is used.
|
||||||
|
#[napi]
|
||||||
|
pub fn job(&self, job_id: String) -> napi::Result<crate::job::Job> {
|
||||||
|
let job = self.get_inner()?.job(job_id).default_error()?;
|
||||||
|
Ok(crate::job::Job::new(job))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// List server-side jobs across the database's tables.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn list_jobs(&self) -> napi::Result<Vec<crate::job::JobInfo>> {
|
||||||
|
let jobs = self.get_inner()?.list_jobs().await.default_error()?;
|
||||||
|
Ok(jobs.into_iter().map(Into::into).collect())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Describe a single server-side job by id. `null` when the server has
|
||||||
|
/// no such job.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn get_job(
|
||||||
|
&self,
|
||||||
|
job_id: String,
|
||||||
|
) -> napi::Result<Option<crate::job::JobDescription>> {
|
||||||
|
let description = self.get_inner()?.get_job(&job_id).await.default_error()?;
|
||||||
|
Ok(description.map(Into::into))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Request cancellation of a server-side job by id. Returns true if the
|
||||||
|
/// server accepted the cancellation, false if no such job exists.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn cancel_job(&self, job_id: String) -> napi::Result<bool> {
|
||||||
|
self.get_inner()?.cancel_job(&job_id).await.default_error()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The lifecycle event history of a server-side job (all jobs when
|
||||||
|
/// `job_id` is null), as an Arrow IPC stream buffer. Empty when there is
|
||||||
|
/// no history.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn job_history(&self, job_id: Option<String>) -> napi::Result<Buffer> {
|
||||||
|
let batches = self
|
||||||
|
.get_inner()?
|
||||||
|
.job_history(job_id.as_deref())
|
||||||
|
.await
|
||||||
|
.default_error()?;
|
||||||
|
let Some(first) = batches.first() else {
|
||||||
|
return Ok(Buffer::from(Vec::<u8>::new()));
|
||||||
|
};
|
||||||
|
let mut out = Vec::new();
|
||||||
|
let mut writer = arrow_ipc::writer::StreamWriter::try_new(&mut out, &first.schema())
|
||||||
|
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
|
||||||
|
for batch in &batches {
|
||||||
|
writer
|
||||||
|
.write(batch)
|
||||||
|
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
|
||||||
|
}
|
||||||
|
writer
|
||||||
|
.finish()
|
||||||
|
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
|
||||||
|
drop(writer);
|
||||||
|
Ok(Buffer::from(out))
|
||||||
|
}
|
||||||
|
|
||||||
#[napi(catch_unwind)]
|
#[napi(catch_unwind)]
|
||||||
/// Describe a namespace and return its properties.
|
/// Describe a namespace and return its properties.
|
||||||
pub async fn describe_namespace(
|
pub async fn describe_namespace(
|
||||||
|
|||||||
@@ -0,0 +1,123 @@
|
|||||||
|
// SPDX-License-Identifier: Apache-2.0
|
||||||
|
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
use napi_derive::napi;
|
||||||
|
|
||||||
|
use crate::error::NapiErrorExt;
|
||||||
|
|
||||||
|
/// A handle to an operation that may still be running.
|
||||||
|
#[napi]
|
||||||
|
pub struct Job {
|
||||||
|
inner: Arc<lancedb::Job>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Job {
|
||||||
|
pub(crate) fn new(inner: lancedb::Job) -> Self {
|
||||||
|
Self {
|
||||||
|
inner: Arc::new(inner),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[napi]
|
||||||
|
impl Job {
|
||||||
|
/// Identifies the operation on the server that is running it. Operations
|
||||||
|
/// that run in this process have no server id. The value is opaque.
|
||||||
|
#[napi(getter)]
|
||||||
|
pub fn id(&self) -> Option<String> {
|
||||||
|
self.inner.id().map(str::to_string)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The operation's current lifecycle state: "running", "finished",
|
||||||
|
/// "failed", or "cancelled".
|
||||||
|
///
|
||||||
|
/// A point snapshot; unlike {@link Job.wait} it does not block or reject
|
||||||
|
/// on a terminal failure state. States a newer server reports that this
|
||||||
|
/// client version does not know pass through as-is.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn status(&self) -> napi::Result<String> {
|
||||||
|
self.inner.status().await.default_error()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Wait until the operation reaches a terminal state.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn wait(&self) -> napi::Result<()> {
|
||||||
|
self.inner.wait().await.default_error()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Request cancellation. Cancelling a finished operation is a no-op.
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn cancel(&self) -> napi::Result<()> {
|
||||||
|
self.inner.cancel().await.default_error()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A row from `Connection.listJobs`: one server-side job.
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct JobInfo {
|
||||||
|
/// The job id -- what `Connection.getJob` and `Connection.cancelJob`
|
||||||
|
/// accept.
|
||||||
|
pub job_id: String,
|
||||||
|
/// The table the job runs against, without URI or namespace.
|
||||||
|
pub table: String,
|
||||||
|
pub job_type: String,
|
||||||
|
/// Lifecycle state: "running", "finished", "failed", or "cancelled".
|
||||||
|
pub state: String,
|
||||||
|
/// When the job was created, in milliseconds since the epoch.
|
||||||
|
pub created_at_millis: i64,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<lancedb::database::JobInfo> for JobInfo {
|
||||||
|
fn from(info: lancedb::database::JobInfo) -> Self {
|
||||||
|
Self {
|
||||||
|
job_id: info.job_id,
|
||||||
|
table: info.table,
|
||||||
|
job_type: info.job_type,
|
||||||
|
state: info.state,
|
||||||
|
created_at_millis: info.created_at_millis,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The server's account of why a job failed.
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct JobFailureInfo {
|
||||||
|
pub phase: Option<String>,
|
||||||
|
pub message: Option<String>,
|
||||||
|
pub retryable: Option<bool>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A described job from `Connection.getJob`.
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct JobDescription {
|
||||||
|
pub job_id: String,
|
||||||
|
pub job_type: String,
|
||||||
|
/// Lifecycle state: "running", "finished", "failed", or "cancelled".
|
||||||
|
pub state: String,
|
||||||
|
/// When the job was created, in milliseconds since the epoch.
|
||||||
|
pub creation_ms: i64,
|
||||||
|
/// The job-type-specific specification as a JSON string, when present.
|
||||||
|
pub spec_json: Option<String>,
|
||||||
|
/// Why the job failed, when the job is failed and the server reports a
|
||||||
|
/// reason.
|
||||||
|
pub failure: Option<JobFailureInfo>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<lancedb::database::JobDescription> for JobDescription {
|
||||||
|
fn from(description: lancedb::database::JobDescription) -> Self {
|
||||||
|
Self {
|
||||||
|
job_id: description.job_id,
|
||||||
|
job_type: description.job_type,
|
||||||
|
state: description.state,
|
||||||
|
creation_ms: description.creation_ms,
|
||||||
|
spec_json: (!description.spec.is_null()).then(|| description.spec.to_string()),
|
||||||
|
failure: description.failure.map(|failure| JobFailureInfo {
|
||||||
|
phase: failure.phase,
|
||||||
|
message: failure.message,
|
||||||
|
retryable: failure.retryable,
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -11,6 +11,7 @@ mod error;
|
|||||||
mod header;
|
mod header;
|
||||||
mod index;
|
mod index;
|
||||||
mod iterator;
|
mod iterator;
|
||||||
|
mod job;
|
||||||
pub mod merge;
|
pub mod merge;
|
||||||
pub mod otel;
|
pub mod otel;
|
||||||
pub mod permutation;
|
pub mod permutation;
|
||||||
|
|||||||
+249
-9
@@ -168,6 +168,39 @@ impl Table {
|
|||||||
builder.execute().await.default_error()
|
builder.execute().await.default_error()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn create_index_async(
|
||||||
|
&self,
|
||||||
|
index: Option<&Index>,
|
||||||
|
column: String,
|
||||||
|
replace: Option<bool>,
|
||||||
|
wait_timeout_s: Option<i64>,
|
||||||
|
name: Option<String>,
|
||||||
|
train: Option<bool>,
|
||||||
|
) -> napi::Result<crate::job::Job> {
|
||||||
|
let lancedb_index = if let Some(index) = index {
|
||||||
|
index.consume()?
|
||||||
|
} else {
|
||||||
|
lancedb::index::Index::Auto
|
||||||
|
};
|
||||||
|
let mut builder = self.inner_ref()?.create_index(&[column], lancedb_index);
|
||||||
|
if let Some(replace) = replace {
|
||||||
|
builder = builder.replace(replace);
|
||||||
|
}
|
||||||
|
if let Some(timeout) = wait_timeout_s {
|
||||||
|
builder =
|
||||||
|
builder.wait_timeout(std::time::Duration::from_secs(timeout.try_into().unwrap()));
|
||||||
|
}
|
||||||
|
if let Some(name) = name {
|
||||||
|
builder = builder.name(name);
|
||||||
|
}
|
||||||
|
if let Some(train) = train {
|
||||||
|
builder = builder.train(train);
|
||||||
|
}
|
||||||
|
let job = builder.execute_async().await.default_error()?;
|
||||||
|
Ok(crate::job::Job::new(job))
|
||||||
|
}
|
||||||
|
|
||||||
#[napi(catch_unwind)]
|
#[napi(catch_unwind)]
|
||||||
pub async fn drop_index(&self, index_name: String) -> napi::Result<()> {
|
pub async fn drop_index(&self, index_name: String) -> napi::Result<()> {
|
||||||
self.inner_ref()?
|
self.inner_ref()?
|
||||||
@@ -306,12 +339,48 @@ impl Table {
|
|||||||
let transforms = NewColumnTransform::SqlExpressions(transforms);
|
let transforms = NewColumnTransform::SqlExpressions(transforms);
|
||||||
let res = self
|
let res = self
|
||||||
.inner_ref()?
|
.inner_ref()?
|
||||||
.add_columns(transforms, None)
|
.add_columns()
|
||||||
|
.transform(transforms)
|
||||||
|
.execute()
|
||||||
.await
|
.await
|
||||||
.default_error()?;
|
.default_error()?;
|
||||||
Ok(res.into())
|
Ok(res.into())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn add_computed_columns(
|
||||||
|
&self,
|
||||||
|
columns: Vec<AddColumnsSql>,
|
||||||
|
) -> napi::Result<AddColumnsResult> {
|
||||||
|
let table = self.inner_ref()?;
|
||||||
|
let mut builder = table.add_columns();
|
||||||
|
for column in columns {
|
||||||
|
builder = builder.computed(column.name, column.value_sql);
|
||||||
|
}
|
||||||
|
let res = builder.execute().await.default_error()?;
|
||||||
|
Ok(res.into())
|
||||||
|
}
|
||||||
|
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn refresh_column(&self, column: String) -> napi::Result<RefreshColumnResult> {
|
||||||
|
let res = self
|
||||||
|
.inner_ref()?
|
||||||
|
.refresh_column(column)
|
||||||
|
.await
|
||||||
|
.default_error()?;
|
||||||
|
Ok(res.into())
|
||||||
|
}
|
||||||
|
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn refresh_column_async(&self, column: String) -> napi::Result<crate::job::Job> {
|
||||||
|
let job = self
|
||||||
|
.inner_ref()?
|
||||||
|
.refresh_column_async(column)
|
||||||
|
.await
|
||||||
|
.default_error()?;
|
||||||
|
Ok(crate::job::Job::new(job))
|
||||||
|
}
|
||||||
|
|
||||||
#[napi(catch_unwind)]
|
#[napi(catch_unwind)]
|
||||||
pub async fn add_columns_with_schema(
|
pub async fn add_columns_with_schema(
|
||||||
&self,
|
&self,
|
||||||
@@ -323,7 +392,9 @@ impl Table {
|
|||||||
let transforms = NewColumnTransform::AllNulls(schema);
|
let transforms = NewColumnTransform::AllNulls(schema);
|
||||||
let res = self
|
let res = self
|
||||||
.inner_ref()?
|
.inner_ref()?
|
||||||
.add_columns(transforms, None)
|
.add_columns()
|
||||||
|
.transform(transforms)
|
||||||
|
.execute()
|
||||||
.await
|
.await
|
||||||
.default_error()?;
|
.default_error()?;
|
||||||
Ok(res.into())
|
Ok(res.into())
|
||||||
@@ -426,6 +497,34 @@ impl Table {
|
|||||||
self.inner_ref()?.close_lsm_writers().await.default_error()
|
self.inner_ref()?.close_lsm_writers().await.default_error()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn flush_lsm(&self) -> napi::Result<()> {
|
||||||
|
self.inner_ref()?.flush_lsm().await.default_error()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn compact_lsm(&self) -> napi::Result<()> {
|
||||||
|
self.inner_ref()?.compact_lsm().await.default_error()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn checkpoint_lsm(&self) -> napi::Result<()> {
|
||||||
|
self.inner_ref()?.checkpoint_lsm().await.default_error()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn get_lsm_stats(
|
||||||
|
&self,
|
||||||
|
include_generation_rows: bool,
|
||||||
|
) -> napi::Result<Option<LsmStats>> {
|
||||||
|
let stats = self
|
||||||
|
.inner_ref()?
|
||||||
|
.get_lsm_stats(include_generation_rows)
|
||||||
|
.await
|
||||||
|
.default_error()?;
|
||||||
|
Ok(stats.map(LsmStats::from))
|
||||||
|
}
|
||||||
|
|
||||||
#[napi(catch_unwind)]
|
#[napi(catch_unwind)]
|
||||||
pub async fn version(&self) -> napi::Result<i64> {
|
pub async fn version(&self) -> napi::Result<i64> {
|
||||||
self.inner_ref()?
|
self.inner_ref()?
|
||||||
@@ -735,7 +834,8 @@ pub struct LsmWriteSpec {
|
|||||||
pub column: Option<String>,
|
pub column: Option<String>,
|
||||||
/// Bucket variant: the number of buckets, in `[1, 1024]`.
|
/// Bucket variant: the number of buckets, in `[1, 1024]`.
|
||||||
pub num_buckets: Option<u32>,
|
pub num_buckets: Option<u32>,
|
||||||
/// Names of indexes the MemWAL should keep up to date during writes.
|
/// Indexes the MemWAL keeps up to date. Omitted resolves every
|
||||||
|
/// maintainable index on install; an empty array means none.
|
||||||
pub maintained_indexes: Option<Vec<String>>,
|
pub maintained_indexes: Option<Vec<String>>,
|
||||||
/// Default `ShardWriter` configuration recorded in the MemWAL index.
|
/// Default `ShardWriter` configuration recorded in the MemWAL index.
|
||||||
pub writer_config_defaults: Option<HashMap<String, String>>,
|
pub writer_config_defaults: Option<HashMap<String, String>>,
|
||||||
@@ -745,7 +845,6 @@ impl TryFrom<LsmWriteSpec> for lancedb::table::LsmWriteSpec {
|
|||||||
type Error = napi::Error;
|
type Error = napi::Error;
|
||||||
|
|
||||||
fn try_from(value: LsmWriteSpec) -> napi::Result<Self> {
|
fn try_from(value: LsmWriteSpec) -> napi::Result<Self> {
|
||||||
let maintained = value.maintained_indexes.unwrap_or_default();
|
|
||||||
let writer_config_defaults = value.writer_config_defaults.unwrap_or_default();
|
let writer_config_defaults = value.writer_config_defaults.unwrap_or_default();
|
||||||
let spec = match value.spec_type.as_str() {
|
let spec = match value.spec_type.as_str() {
|
||||||
"bucket" => {
|
"bucket" => {
|
||||||
@@ -772,7 +871,7 @@ impl TryFrom<LsmWriteSpec> for lancedb::table::LsmWriteSpec {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
Ok(spec
|
Ok(spec
|
||||||
.with_maintained_indexes(maintained)
|
.with_maintained_indexes(value.maintained_indexes)
|
||||||
.with_writer_config_defaults(writer_config_defaults))
|
.with_writer_config_defaults(writer_config_defaults))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -790,7 +889,7 @@ impl From<lancedb::table::LsmWriteSpec> for LsmWriteSpec {
|
|||||||
spec_type: "bucket".to_string(),
|
spec_type: "bucket".to_string(),
|
||||||
column: Some(column),
|
column: Some(column),
|
||||||
num_buckets: Some(num_buckets),
|
num_buckets: Some(num_buckets),
|
||||||
maintained_indexes: Some(maintained_indexes),
|
maintained_indexes,
|
||||||
writer_config_defaults: Some(writer_config_defaults),
|
writer_config_defaults: Some(writer_config_defaults),
|
||||||
},
|
},
|
||||||
Native::Identity {
|
Native::Identity {
|
||||||
@@ -801,7 +900,7 @@ impl From<lancedb::table::LsmWriteSpec> for LsmWriteSpec {
|
|||||||
spec_type: "identity".to_string(),
|
spec_type: "identity".to_string(),
|
||||||
column: Some(column),
|
column: Some(column),
|
||||||
num_buckets: None,
|
num_buckets: None,
|
||||||
maintained_indexes: Some(maintained_indexes),
|
maintained_indexes,
|
||||||
writer_config_defaults: Some(writer_config_defaults),
|
writer_config_defaults: Some(writer_config_defaults),
|
||||||
},
|
},
|
||||||
Native::Unsharded {
|
Native::Unsharded {
|
||||||
@@ -811,13 +910,136 @@ impl From<lancedb::table::LsmWriteSpec> for LsmWriteSpec {
|
|||||||
spec_type: "unsharded".to_string(),
|
spec_type: "unsharded".to_string(),
|
||||||
column: None,
|
column: None,
|
||||||
num_buckets: None,
|
num_buckets: None,
|
||||||
maintained_indexes: Some(maintained_indexes),
|
maintained_indexes,
|
||||||
writer_config_defaults: Some(writer_config_defaults),
|
writer_config_defaults: Some(writer_config_defaults),
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// One flushed L0 generation.
|
||||||
|
#[napi(object)]
|
||||||
|
#[derive(Clone, Debug)]
|
||||||
|
pub struct GenerationStats {
|
||||||
|
/// The generation number. Increases as memtables are sealed into L0.
|
||||||
|
pub generation: i64,
|
||||||
|
/// On-disk size of the generation.
|
||||||
|
pub bytes: i64,
|
||||||
|
/// Present only when `includeGenerationRows` was requested. Off by default
|
||||||
|
/// because each count opens an uncached Lance dataset.
|
||||||
|
pub rows: Option<i64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<lancedb::table::GenerationStats> for GenerationStats {
|
||||||
|
fn from(g: lancedb::table::GenerationStats) -> Self {
|
||||||
|
Self {
|
||||||
|
generation: g.generation as i64,
|
||||||
|
bytes: g.bytes as i64,
|
||||||
|
rows: g.rows.map(|r| r as i64),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One in-memory memtable.
|
||||||
|
#[napi(object)]
|
||||||
|
#[derive(Clone, Debug)]
|
||||||
|
pub struct MemtableStats {
|
||||||
|
/// The generation this memtable will become once sealed.
|
||||||
|
pub generation: i64,
|
||||||
|
/// Rows currently buffered.
|
||||||
|
pub rows: i64,
|
||||||
|
/// Estimated in-memory size.
|
||||||
|
pub bytes: i64,
|
||||||
|
/// Record batches currently buffered.
|
||||||
|
pub batches: i64,
|
||||||
|
/// Names of the indexes this memtable carries. An absent name is the whole
|
||||||
|
/// answer to "why is my fresh-tier search on that column brute-force".
|
||||||
|
pub indexes: Vec<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<lancedb::table::MemtableStats> for MemtableStats {
|
||||||
|
fn from(m: lancedb::table::MemtableStats) -> Self {
|
||||||
|
Self {
|
||||||
|
generation: m.generation as i64,
|
||||||
|
rows: m.rows as i64,
|
||||||
|
bytes: m.bytes as i64,
|
||||||
|
batches: m.batches as i64,
|
||||||
|
indexes: m.indexes,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Live state of one bucket. A table is N buckets on one node; flattening to a
|
||||||
|
/// single number hides the one hot bucket that is usually why someone opened
|
||||||
|
/// this endpoint.
|
||||||
|
#[napi(object)]
|
||||||
|
#[derive(Clone, Debug)]
|
||||||
|
pub struct BucketStats {
|
||||||
|
/// The shard this bucket writes.
|
||||||
|
pub shard_id: String,
|
||||||
|
/// `"Active"` or `"Sealed"` (drop-table 2PC in flight).
|
||||||
|
pub status: String,
|
||||||
|
/// Epoch of the writer that currently owns the shard.
|
||||||
|
pub writer_epoch: i64,
|
||||||
|
/// Version of the shard manifest these numbers were read from.
|
||||||
|
pub manifest_version: i64,
|
||||||
|
/// The generation the active memtable will become.
|
||||||
|
pub current_generation: i64,
|
||||||
|
/// WAL position replay resumes from.
|
||||||
|
pub replay_after_wal_entry_position: i64,
|
||||||
|
/// Highest WAL position the writer has seen. The difference against
|
||||||
|
/// `replayAfterWalEntryPosition` is the WAL lag.
|
||||||
|
pub wal_entry_position_last_seen: i64,
|
||||||
|
/// Flushed L0 generations not yet merged into the base table.
|
||||||
|
pub generations: Vec<GenerationStats>,
|
||||||
|
/// Whether a pass owns this bucket's compaction latch right now. Says *a*
|
||||||
|
/// driver is running, not *whose*, and the latch is held from dispatch —
|
||||||
|
/// including while the pass queues for a pod-wide compactor permit. Read it
|
||||||
|
/// as "do not pile on", never as "mine is progressing".
|
||||||
|
pub compacting: bool,
|
||||||
|
/// Oldest first, active last. Absent for a `"Sealed"` bucket, whose
|
||||||
|
/// in-memory state is torn down.
|
||||||
|
pub memtables: Option<Vec<MemtableStats>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<lancedb::table::BucketStats> for BucketStats {
|
||||||
|
fn from(b: lancedb::table::BucketStats) -> Self {
|
||||||
|
Self {
|
||||||
|
shard_id: b.shard_id,
|
||||||
|
status: b.status,
|
||||||
|
writer_epoch: b.writer_epoch as i64,
|
||||||
|
manifest_version: b.manifest_version as i64,
|
||||||
|
current_generation: b.current_generation as i64,
|
||||||
|
replay_after_wal_entry_position: b.replay_after_wal_entry_position as i64,
|
||||||
|
wal_entry_position_last_seen: b.wal_entry_position_last_seen as i64,
|
||||||
|
generations: b.generations.into_iter().map(Into::into).collect(),
|
||||||
|
compacting: b.compacting,
|
||||||
|
memtables: b
|
||||||
|
.memtables
|
||||||
|
.map(|ms| ms.into_iter().map(Into::into).collect()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Live per-bucket LSM state, as returned by `Table#getLsmStats`.
|
||||||
|
///
|
||||||
|
/// Nothing here is derived: sums and differences (total L0 bytes, WAL lag) are
|
||||||
|
/// the caller's to compute.
|
||||||
|
#[napi(object)]
|
||||||
|
#[derive(Clone, Debug)]
|
||||||
|
pub struct LsmStats {
|
||||||
|
/// One entry per bucket backing this table.
|
||||||
|
pub buckets: Vec<BucketStats>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<lancedb::table::LsmStats> for LsmStats {
|
||||||
|
fn from(stats: lancedb::table::LsmStats) -> Self {
|
||||||
|
Self {
|
||||||
|
buckets: stats.buckets.into_iter().map(Into::into).collect(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Statistics about a compaction operation.
|
/// Statistics about a compaction operation.
|
||||||
#[napi(object)]
|
#[napi(object)]
|
||||||
#[derive(Clone, Debug)]
|
#[derive(Clone, Debug)]
|
||||||
@@ -1006,7 +1228,10 @@ impl From<lancedb::index::IndexStatistics> for IndexStatistics {
|
|||||||
|
|
||||||
#[napi(object)]
|
#[napi(object)]
|
||||||
pub struct TableStatistics {
|
pub struct TableStatistics {
|
||||||
/// The total number of bytes in the table
|
/// The total size, in bytes, of the table's data files, index files, and
|
||||||
|
/// overlay files
|
||||||
|
///
|
||||||
|
/// Read from the manifest, so this excludes deletion files and manifests.
|
||||||
pub total_bytes: i64,
|
pub total_bytes: i64,
|
||||||
|
|
||||||
/// The number of rows in the table
|
/// The number of rows in the table
|
||||||
@@ -1156,6 +1381,21 @@ pub struct AddColumnsResult {
|
|||||||
pub version: i64,
|
pub version: i64,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct RefreshColumnResult {
|
||||||
|
pub rows_filled: i64,
|
||||||
|
pub version: i64,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<lancedb::table::RefreshColumnResult> for RefreshColumnResult {
|
||||||
|
fn from(value: lancedb::table::RefreshColumnResult) -> Self {
|
||||||
|
Self {
|
||||||
|
rows_filled: value.rows_filled as i64,
|
||||||
|
version: value.version as i64,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
impl From<lancedb::table::AddColumnsResult> for AddColumnsResult {
|
impl From<lancedb::table::AddColumnsResult> for AddColumnsResult {
|
||||||
fn from(value: lancedb::table::AddColumnsResult) -> Self {
|
fn from(value: lancedb::table::AddColumnsResult) -> Self {
|
||||||
Self {
|
Self {
|
||||||
|
|||||||
@@ -20,18 +20,16 @@ Do NOT assume local-only table helpers exist on remote tables. If the user asks
|
|||||||
|
|
||||||
1. Identify the SDK: Python, TypeScript, or both.
|
1. Identify the SDK: Python, TypeScript, or both.
|
||||||
2. Identify the table mode: local/embedded OSS, remote Enterprise/Cloud, or portable across both. If the user says "LanceDB Enterprise", choose the remote table path. If the task involves jobs in any way (listing, inspecting, creating, or canceling jobs), it is always the remote path and requires a remote server connection — see "Connecting to the LanceDB remote server" below before doing anything else.
|
2. Identify the table mode: local/embedded OSS, remote Enterprise/Cloud, or portable across both. If the user says "LanceDB Enterprise", choose the remote table path. If the task involves jobs in any way (listing, inspecting, creating, or canceling jobs), it is always the remote path and requires a remote server connection — see "Connecting to the LanceDB remote server" below before doing anything else.
|
||||||
3. Read the matching language branch before writing or changing code:
|
3. Read the matching topic reference before writing or changing code:
|
||||||
- Python patterns: `references/python/patterns.md`
|
|
||||||
- Python API quick reference: `references/python/api_reference.md`
|
|
||||||
- Python performance guidance: `references/python/performance.md`
|
|
||||||
- TypeScript patterns: `references/typescript/patterns.md`
|
|
||||||
- TypeScript API quick reference: `references/typescript/api_reference.md`
|
|
||||||
- TypeScript performance guidance: `references/typescript/performance.md`
|
|
||||||
- Column metadata authoring (both SDKs): `references/column_metadata.md`
|
- Column metadata authoring (both SDKs): `references/column_metadata.md`
|
||||||
- Branch operations (both SDKs): `references/branch_ops.md`
|
- Branch operations (both SDKs): `references/branch_ops.md`
|
||||||
- Remote server connection resolution (jobs, raw REST): `references/remote_connect.md`
|
- Remote server connection resolution (jobs, raw REST): `references/remote_connect.md`
|
||||||
- Job operations REST API (list/describe/cancel/query_events): `references/remote_jobs.md`
|
- Job operations REST API (list/describe/cancel/query_events): `references/remote_jobs.md`
|
||||||
4. Start with `patterns.md` for the selected SDK. Read `api_reference.md` when choosing method names or return collectors. Read `performance.md` when the task involves ingestion, indexing, filtering, query tuning, diagnostics, or large datasets. Read `column_metadata.md` when the task is documenting, tagging, classifying, or grouping table columns (field descriptions, `lancedb:tag:*` tags, logical column families). Read `branch_ops.md` when the task involves branch lifecycle (list/create/delete), writing to a non-main branch, or verifying a change stayed off main. Read `remote_connect.md` when the task involves jobs or direct REST access to an Enterprise deployment, and `remote_jobs.md` for the job REST methods themselves (list, describe, cancel, query_events).
|
|
||||||
|
There is no bundled per-language guide. For exact method names, signatures, and options, look them up in the canonical sources instead of relying on memory:
|
||||||
|
- Python: `docs/src/python/python.md` (the hand-maintained API reference) and the source under `python/python/lancedb/` when working inside the LanceDB repo; otherwise <https://lancedb.github.io/lancedb/python/python/>.
|
||||||
|
- TypeScript: the generated typedoc under `docs/src/js/` and the source under `nodejs/lancedb/` when working inside the LanceDB repo; otherwise <https://lancedb.github.io/lancedb/js/globals/>.
|
||||||
|
4. Apply the SDK invariants in "Per-SDK Invariants" below. Read `column_metadata.md` when the task is documenting, tagging, classifying, or grouping table columns (field descriptions, `lancedb:tag:*` tags, logical column families). Read `branch_ops.md` when the task involves branch lifecycle (list/create/delete), writing to a non-main branch, or verifying a change stayed off main. Read `remote_connect.md` when the task involves jobs or direct REST access to an Enterprise deployment, and `remote_jobs.md` for the job REST methods themselves (list, describe, cancel, query_events).
|
||||||
5. For Python schemas, favor Pydantic models and validate records before writing. Use PyArrow schemas when Arrow-native, streaming, or highly dynamic data makes them materially better suited.
|
5. For Python schemas, favor Pydantic models and validate records before writing. Use PyArrow schemas when Arrow-native, streaming, or highly dynamic data makes them materially better suited.
|
||||||
6. Prefer `search()` or `query()` builders with explicit `select()` and `limit()` for reads.
|
6. Prefer `search()` or `query()` builders with explicit `select()` and `limit()` for reads.
|
||||||
7. Avoid table-level full materialization in remote or portable code. This is the main local-vs-remote read pitfall.
|
7. Avoid table-level full materialization in remote or portable code. This is the main local-vs-remote read pitfall.
|
||||||
@@ -54,6 +52,23 @@ The unsafe pattern is table-level or unbounded collection, plus local-only datas
|
|||||||
- Python: `table.to_pandas()`, `table.to_arrow()`, `table.to_polars()`; `table.to_lance()` is local/OSS-only dataset access, not materialization
|
- Python: `table.to_pandas()`, `table.to_arrow()`, `table.to_polars()`; `table.to_lance()` is local/OSS-only dataset access, not materialization
|
||||||
- TypeScript: `await table.toArrow()`, `await table.query().toArray()` without `limit()`
|
- TypeScript: `await table.toArrow()`, `await table.query().toArray()` without `limit()`
|
||||||
|
|
||||||
|
## Per-SDK Invariants
|
||||||
|
|
||||||
|
Python:
|
||||||
|
|
||||||
|
- Result collectors: default to `.to_list()` (plain dicts, no extra dependency) or `.to_arrow()` (PyArrow ships with LanceDB). Use `.to_pandas()` / `.to_polars()` only when the project already declares that dependency — do not assume pandas or polars is installed.
|
||||||
|
- Plain scans differ by client: the sync client has no `.query()` method — use `table.search()` with no argument; the async client uses `await async_table.query()`.
|
||||||
|
|
||||||
|
TypeScript:
|
||||||
|
|
||||||
|
- Collect bounded results with `.toArray()` (objects) or `.toArrow()` (Arrow) after `select()` and `limit()`.
|
||||||
|
- For large reads, stream batches instead of collecting: `for await (const batch of table.query().where(...).select(...).limit(...)) { ... }`.
|
||||||
|
|
||||||
|
Both SDKs:
|
||||||
|
|
||||||
|
- Ingest in bulk or in batches of thousands of rows; never write per-row in a loop — each write creates a version and fragment, slowing ingestion and later queries.
|
||||||
|
- Build a vector index once brute-force search is too slow (rule of thumb: beyond roughly 100K vectors locally), and scalar indexes for filtered columns and merge/upsert keys. Use index defaults unless the task states recall/latency requirements.
|
||||||
|
|
||||||
## Enterprise: never drop-then-reuse the same table name
|
## Enterprise: never drop-then-reuse the same table name
|
||||||
|
|
||||||
LanceDB Enterprise/Cloud splits a **control plane** (DDL: create/drop/rename) from a **data plane** (query nodes that serve reads). Query nodes cache the resolved dataset for a table name for up to `table_cache_ttl` — **default 300 seconds (5 minutes)**. After you drop or overwrite a table, the control plane updates immediately but the data plane keeps serving the *old* dataset until that cache entry expires. During the window the two planes disagree.
|
LanceDB Enterprise/Cloud splits a **control plane** (DDL: create/drop/rename) from a **data plane** (query nodes that serve reads). Query nodes cache the resolved dataset for a table name for up to `table_cache_ttl` — **default 300 seconds (5 minutes)**. After you drop or overwrite a table, the control plane updates immediately but the data plane keeps serving the *old* dataset until that cache entry expires. During the window the two planes disagree.
|
||||||
|
|||||||
@@ -1,138 +0,0 @@
|
|||||||
# Python API Reference
|
|
||||||
|
|
||||||
Quick method reference for Python LanceDB code. Cross-check source for non-trivial claims.
|
|
||||||
|
|
||||||
## Connect
|
|
||||||
|
|
||||||
If you're connecting to a remote database, use this:
|
|
||||||
```python
|
|
||||||
import lancedb
|
|
||||||
|
|
||||||
db = lancedb.connect("db://my-db", api_key=api_key, host_override=host_override) # remote
|
|
||||||
```
|
|
||||||
(values may be found in LANCEDB_API_KEY and LANCEDB_HOST_OVERRIDE, either in env vars or a .env file)
|
|
||||||
|
|
||||||
If you're connecting to a local table using OSS LanceDB, use this:
|
|
||||||
```python
|
|
||||||
db = lancedb.connect("./camelot-db") # local/OSS
|
|
||||||
```
|
|
||||||
If you're not sure which, or if you can't find the api_key or host_override params, ask the user.
|
|
||||||
|
|
||||||
**Place the local database directory next to the script/entrypoint that opens it** (i.e. resolve the path relative to the script, `Path(__file__).parent / "camelot-db"`), not buried under a shared `data/` folder. The Lance dataset is the database, not a data file — keeping it beside its code makes ownership obvious and paths stable regardless of the working directory the script is launched from.
|
|
||||||
|
|
||||||
**Do not name the directory `lancedb`** (e.g. `./lancedb`, `./data/lancedb`). It collides with the imported `lancedb` package name, which is confusing to read and easy to shadow in scripts. Give it a name derived from the repo or dataset with a clear prefix/suffix — for example `./<dataset>-db`, `./<repo>_lancedb`, or `./vectordb`.
|
|
||||||
|
|
||||||
Async:
|
|
||||||
|
|
||||||
```python
|
|
||||||
db = await lancedb.connect_async("./camelot-db")
|
|
||||||
```
|
|
||||||
|
|
||||||
## Table Reads
|
|
||||||
|
|
||||||
| Task | Preferred API |
|
|
||||||
| --- | --- |
|
|
||||||
| Vector search | `table.search(query_vector).limit(k)` |
|
|
||||||
| Full scan with filters/projection (sync) | `table.search().where(...).select(...).limit(...)` |
|
|
||||||
| Full scan with filters/projection (async) | `table.query().where(...).select(...).limit(...)` |
|
|
||||||
| Filter | `.where("col > 10")` |
|
|
||||||
| Projection | `.select(["id", "text"])` |
|
|
||||||
| Bound result count | `.limit(20)` |
|
|
||||||
| Collect bounded result as Python objects (default, no extra deps) | `.to_list()` on query/search result |
|
|
||||||
| Collect bounded result as Arrow (default, `pyarrow` always available) | `.to_arrow()` on query/search result |
|
|
||||||
| Collect bounded result as pandas (only if project uses pandas) | `.to_pandas()` on query/search result |
|
|
||||||
| Collect bounded result as Polars (only if project uses polars) | `.to_polars()` on query/search result |
|
|
||||||
|
|
||||||
## Sync vs Async Scan API
|
|
||||||
|
|
||||||
The plain-scan entry point differs between the sync and async clients. **Verified against `lancedb` 0.34.0** — re-check if the pinned version changes:
|
|
||||||
|
|
||||||
- **Sync** (`lancedb.connect(...)`): the table has **no `.query()` method**. Use `.search()` with no argument for a plain scan; it returns a query builder that supports `.where()`, `.select()`, `.limit()`, and the `.to_list()` / `.to_arrow()` / `.to_pandas()` / `.to_polars()` collectors.
|
|
||||||
```python
|
|
||||||
rows = table.search().where("status = 'ready'").select(["id", "text"]).limit(20).to_list()
|
|
||||||
```
|
|
||||||
- **Async** (`lancedb.connect_async(...)`): the table has **both** `.query()` and `.search()`. Use `.query()` for a plain scan.
|
|
||||||
```python
|
|
||||||
rows = await async_table.query().where("status = 'ready'").select(["id", "text"]).limit(20).to_list()
|
|
||||||
```
|
|
||||||
|
|
||||||
Do not call `table.query()` on a sync table — it raises `AttributeError`.
|
|
||||||
|
|
||||||
## Local vs Remote Table Methods
|
|
||||||
|
|
||||||
| API | Local table | Remote table | Agent guidance |
|
|
||||||
| --- | --- | --- | --- |
|
|
||||||
| `table.search(...)` | Yes | Yes | Preferred read path (sync + async) |
|
|
||||||
| `table.query()` | Async only | Async only | Sync scan path is `table.search()`; `.query()` is the async scan builder |
|
|
||||||
| `table.to_pandas()` | Yes | No / unsafe for portability | Avoid in portable code |
|
|
||||||
| `table.to_arrow()` | Yes | No / unsafe for portability | Avoid in portable code |
|
|
||||||
| `table.to_polars()` | Yes | No / unsafe for portability | Avoid in portable code |
|
|
||||||
| `table.to_lance()` | Yes | No | Local/OSS escape hatch only |
|
|
||||||
|
|
||||||
## Indexes
|
|
||||||
|
|
||||||
Use `create_index(...)` for vector indexes and modern index configs. Use scalar indexes for filtered or merge keys.
|
|
||||||
|
|
||||||
Common calls:
|
|
||||||
|
|
||||||
```python
|
|
||||||
table.create_index("vector")
|
|
||||||
table.create_scalar_index("status")
|
|
||||||
table.create_fts_index("text")
|
|
||||||
```
|
|
||||||
|
|
||||||
Check source docs before specifying advanced index config names or parameters.
|
|
||||||
|
|
||||||
## Filtering And Recall Knobs
|
|
||||||
|
|
||||||
```python
|
|
||||||
table.search(query_vector).where("status = 'ready'") # pre-filter by default
|
|
||||||
table.search(query_vector).where("status = 'ready'", prefilter=False)
|
|
||||||
table.search(query_vector).limit(10).refine_factor(20)
|
|
||||||
table.search(query_vector).limit(10).nprobes(50)
|
|
||||||
```
|
|
||||||
|
|
||||||
Use post-filtering only when fewer than `limit` results are acceptable.
|
|
||||||
|
|
||||||
## Diagnostics
|
|
||||||
|
|
||||||
```python
|
|
||||||
print(table.search(query_vector).where("year > 2000").limit(10).analyze_plan())
|
|
||||||
print(table.index_stats("vector_idx"))
|
|
||||||
```
|
|
||||||
|
|
||||||
Use these before changing indexes or search tuning.
|
|
||||||
|
|
||||||
## Column (Field) Metadata
|
|
||||||
|
|
||||||
```python
|
|
||||||
schema = table.schema # sync property; async: await table.schema()
|
|
||||||
meta = schema.field("category").metadata # dict[bytes, bytes] — Arrow metadata is bytes-keyed
|
|
||||||
res = table.update_field_metadata( # varargs: one dict per field; works local + remote
|
|
||||||
{"path": "category", "metadata": {"lancedb:description": "...", "lancedb:tag:field_type": "label"}}
|
|
||||||
)
|
|
||||||
res.version # new table version
|
|
||||||
```
|
|
||||||
|
|
||||||
Merges by default; a `None` value deletes that key; `"replace": True` swaps the whole map. Nested fields use dot-paths (`"a.b.c"`). `replace_field_metadata` is deprecated. See `references/column_metadata.md` for key conventions (`lancedb:description`, `lancedb:tag:<name>`, `lancedb:logical-column`) and the authoring workflow.
|
|
||||||
|
|
||||||
## Branches
|
|
||||||
|
|
||||||
```python
|
|
||||||
table.branches.list() # non-main branches; {} = only main
|
|
||||||
exp = table.branches.create("exp") # fork off main -> handle scoped to the branch
|
|
||||||
wip = table.branches.checkout("wip") # existing branch -> scoped handle (version= pins read-only)
|
|
||||||
wip = db.open_table("t", branch="wip") # or open scoped directly
|
|
||||||
table.branches.delete("stale") # removes only the branch pointer
|
|
||||||
table.current_branch() # None = main
|
|
||||||
```
|
|
||||||
|
|
||||||
There is no global switch — scoping is per table handle: any read/write on a branch handle lands on that branch; the original handle keeps targeting main. See `references/branch_ops.md` for the model and isolation checks.
|
|
||||||
|
|
||||||
## Maintenance
|
|
||||||
|
|
||||||
```python
|
|
||||||
table.optimize()
|
|
||||||
```
|
|
||||||
|
|
||||||
Call this after every successful local/OSS ingestion. It handles compaction, cleanup of old versions according to retention, and index optimization. Do not add this for LanceDB Enterprise/Cloud remote tables; Enterprise handles compaction and cleanup automatically from cluster configuration.
|
|
||||||
@@ -1,173 +0,0 @@
|
|||||||
# Python Patterns
|
|
||||||
|
|
||||||
Use these patterns when writing Python code with `lancedb`.
|
|
||||||
|
|
||||||
## Before Writing Code
|
|
||||||
|
|
||||||
Choose the output type from what the project actually depends on. **Do not assume `pandas` or `polars` is installed** — they are heavy dependencies that many LanceDB projects do not use. `pyarrow`, by contrast, ships as a LanceDB dependency and is always available, so it is a safe default to lean on.
|
|
||||||
|
|
||||||
Default output (after applying `select()` and `limit()`):
|
|
||||||
|
|
||||||
- **Python objects**: `.to_list()` — a list of dicts, no extra dependencies. Prefer this for scripts, examples, and agent-generated code unless there is a reason to do otherwise.
|
|
||||||
- **PyArrow**: `.to_arrow()` — a `pyarrow.Table`, when the surrounding code is Arrow-native or you need columnar/zero-copy handoff.
|
|
||||||
|
|
||||||
Only reach for a DataFrame when the project *already* declares that dependency:
|
|
||||||
|
|
||||||
- Pandas projects (pandas in `pyproject.toml`/requirements): `.to_pandas()`.
|
|
||||||
- Polars projects (polars declared): `.to_polars()`.
|
|
||||||
|
|
||||||
If unsure, check the dependency manifest or the imports in surrounding files. When in doubt, use `.to_list()` or `.to_arrow()`.
|
|
||||||
|
|
||||||
## Schema Design and Validation
|
|
||||||
|
|
||||||
Favor `LanceModel` and Pydantic validation for Python schemas. They keep field
|
|
||||||
types readable, validate source records before a write, and map directly to a
|
|
||||||
LanceDB schema. Use `Vector(dimension)` for fixed-size vectors:
|
|
||||||
|
|
||||||
```python
|
|
||||||
from lancedb.pydantic import LanceModel, Vector
|
|
||||||
|
|
||||||
class Document(LanceModel):
|
|
||||||
id: int
|
|
||||||
text: str
|
|
||||||
vector: Vector(384, nullable=False)
|
|
||||||
|
|
||||||
rows = [Document.model_validate(row) for row in source_rows]
|
|
||||||
table = db.create_table("documents", schema=Document)
|
|
||||||
table.add(rows)
|
|
||||||
```
|
|
||||||
|
|
||||||
Use PyArrow schemas instead when the pipeline is already Arrow-native, needs
|
|
||||||
record-batch streaming, or has runtime schema requirements that would make a
|
|
||||||
Pydantic model harder to understand. Declare Pydantic as a direct project
|
|
||||||
dependency when application code imports it, even if LanceDB also depends on it.
|
|
||||||
|
|
||||||
## Recommended Patterns
|
|
||||||
|
|
||||||
### Bounded search or query
|
|
||||||
|
|
||||||
Use this for application reads, examples, notebooks, and agent-generated scripts:
|
|
||||||
|
|
||||||
```python
|
|
||||||
results = (
|
|
||||||
table.search(query_vector)
|
|
||||||
.where("status = 'ready'")
|
|
||||||
.select(["id", "text"])
|
|
||||||
.limit(20)
|
|
||||||
.to_list() # or .to_arrow(); .to_pandas()/.to_polars() only if the project uses them
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
Why: `search()` works across local and remote tables and on both the sync and async clients. `select()` avoids fetching unused columns. `limit()` prevents accidental full-table reads. `.to_list()` and `.to_arrow()` avoid assuming pandas/polars is installed (see "Before Writing Code").
|
|
||||||
|
|
||||||
For a **plain scan** (no query vector), the entry point differs by client:
|
|
||||||
|
|
||||||
```python
|
|
||||||
# Sync client: no .query() method — use .search() with no argument.
|
|
||||||
rows = table.search().where("status = 'ready'").select(["id", "text"]).limit(20).to_list()
|
|
||||||
|
|
||||||
# Async client: use .query().
|
|
||||||
rows = await async_table.query().where("status = 'ready'").select(["id", "text"]).limit(20).to_list()
|
|
||||||
```
|
|
||||||
|
|
||||||
`table.query()` on a sync table raises `AttributeError` (verified on `lancedb` 0.34.0). See the "Sync vs Async Scan API" section in `api_reference.md`.
|
|
||||||
|
|
||||||
### Bounded query result conversion
|
|
||||||
|
|
||||||
It is fine to collect bounded query/search results:
|
|
||||||
|
|
||||||
```python
|
|
||||||
arrow_table = table.search().select(["id"]).limit(100).to_arrow() # sync plain scan
|
|
||||||
rows = table.search(query_vector).limit(10).to_list()
|
|
||||||
df = table.search(query_vector).limit(10).to_pandas() # only if pandas is a project dep
|
|
||||||
```
|
|
||||||
|
|
||||||
### Local-only Lance dataset API
|
|
||||||
|
|
||||||
`table.to_lance()` does not itself materialize the full dataset. It returns the underlying `lance.LanceDataset`, making the table accessible through the PyLance dataset API. Use it when the task is explicitly local/OSS and needs Lance dataset methods not exposed by LanceDB:
|
|
||||||
|
|
||||||
```python
|
|
||||||
# Local/OSS only: RemoteTable does not expose table.to_lance().
|
|
||||||
ds = table.to_lance()
|
|
||||||
for batch in ds.to_batches(columns=["id", "text"], batch_size=10_000):
|
|
||||||
process(batch)
|
|
||||||
```
|
|
||||||
|
|
||||||
### Async Python
|
|
||||||
|
|
||||||
Keep the same shape and bound the result before collecting:
|
|
||||||
|
|
||||||
```python
|
|
||||||
results = await (
|
|
||||||
async_table.query()
|
|
||||||
.where("status = 'ready'")
|
|
||||||
.select(["id", "text"])
|
|
||||||
.limit(20)
|
|
||||||
.to_list() # or .to_arrow()
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
## Anti-Patterns
|
|
||||||
|
|
||||||
**Avoid the following anti-patterns in your code.**
|
|
||||||
|
|
||||||
### Table-level full materialization
|
|
||||||
|
|
||||||
Avoid whole-table collectors in portable or large-table code:
|
|
||||||
|
|
||||||
```python
|
|
||||||
df = table.to_pandas()
|
|
||||||
arrow_table = table.to_arrow()
|
|
||||||
polars_df = table.to_polars()
|
|
||||||
```
|
|
||||||
|
|
||||||
Why: local tables expose these whole-table collectors, but remote tables intentionally do not — a remote production table can be far larger than a local development table, so it is easy to accidentally pull the entire table into memory.
|
|
||||||
|
|
||||||
`table.to_lance()` is different: it is not a full materialization call, but it is still local/OSS-only and should not appear in code meant to run against remote Enterprise tables.
|
|
||||||
|
|
||||||
### Unbounded result collection
|
|
||||||
|
|
||||||
Avoid query/search collection without a meaningful limit:
|
|
||||||
|
|
||||||
```python
|
|
||||||
rows = table.search().to_list() # unbounded plain scan
|
|
||||||
rows = table.search(query_vector).to_list() # unbounded vector search
|
|
||||||
```
|
|
||||||
|
|
||||||
Prefer `select(...).limit(...)` before collecting; for large reads, stream in batches instead.
|
|
||||||
|
|
||||||
### Per-row writes
|
|
||||||
|
|
||||||
Avoid loops that write one row per call:
|
|
||||||
|
|
||||||
```python
|
|
||||||
for row in rows:
|
|
||||||
table.add([row]) # one commit + fragment per row
|
|
||||||
```
|
|
||||||
|
|
||||||
Each `add()` creates a new version and fragment. Pass the whole batch in a single call, or chunk very large inputs:
|
|
||||||
|
|
||||||
```python
|
|
||||||
table.add(rows) # single commit
|
|
||||||
# for very large inputs, add batches of several thousand rows
|
|
||||||
```
|
|
||||||
|
|
||||||
After the final successful write to an embedded OSS table, call
|
|
||||||
`table.optimize()`. Skip this for Enterprise/Cloud tables because their
|
|
||||||
maintenance is automatic.
|
|
||||||
|
|
||||||
### Drop-then-reuse the same table name (Enterprise/Cloud)
|
|
||||||
|
|
||||||
Avoid dropping or overwriting a remote table and then reusing that name right away:
|
|
||||||
|
|
||||||
```python
|
|
||||||
db.drop_table("my_table")
|
|
||||||
table = db.create_table("my_table", data=rows) # reads 500 for ~5 min
|
|
||||||
table = db.create_table("my_table", data=rows, mode="overwrite") # same problem
|
|
||||||
```
|
|
||||||
|
|
||||||
Why: Enterprise/Cloud splits DDL (control plane) from query serving (data plane). The data plane caches the dataset behind a table name for up to `table_cache_ttl` (default 300s / 5 min), so after a drop/overwrite the DDL succeeds but queries against the reused name return `500 Internal Server Error` until the cache expires — and a fresh `describe` may still show the old schema. Instead, write to a **fresh name**, use `list_tables()` and fail if it already exists, then `rename_table(fresh, final)` onto the final name only after the old table's drop has propagated (~5 min). See the "Enterprise: never drop-then-reuse the same table name" section in `SKILL.md`. Local/OSS tables have no separate data plane — overwrite freely there.
|
|
||||||
|
|
||||||
### Guessing performance fixes
|
|
||||||
|
|
||||||
Avoid changing `nprobes`, `refine_factor`, or index types before checking the query plan and index stats. Diagnose first, then tune one knob at a time.
|
|
||||||
@@ -1,131 +0,0 @@
|
|||||||
# Python Performance Guidance
|
|
||||||
|
|
||||||
Use this when writing Python code that ingests data, queries large tables, builds indexes, or investigates latency.
|
|
||||||
|
|
||||||
## Ingestion
|
|
||||||
|
|
||||||
### Recommended: validate schemas and records with Pydantic
|
|
||||||
|
|
||||||
Favor `LanceModel` for readable Python schema definitions and validate source
|
|
||||||
records before writing. Use PyArrow directly for Arrow-native or streaming
|
|
||||||
pipelines where it is the clearer representation.
|
|
||||||
|
|
||||||
```python
|
|
||||||
from lancedb.pydantic import LanceModel, Vector
|
|
||||||
|
|
||||||
class Document(LanceModel):
|
|
||||||
id: int
|
|
||||||
text: str
|
|
||||||
vector: Vector(384, nullable=False)
|
|
||||||
|
|
||||||
rows = [Document.model_validate(row) for row in source_rows]
|
|
||||||
table = db.create_table("documents", schema=Document)
|
|
||||||
table.add(rows)
|
|
||||||
```
|
|
||||||
|
|
||||||
### Recommended: bulk ingestion for materialized data
|
|
||||||
|
|
||||||
```python
|
|
||||||
table.add(arrow_table)
|
|
||||||
table.add(df)
|
|
||||||
table.add(pa.dataset("data/", format="parquet"))
|
|
||||||
```
|
|
||||||
|
|
||||||
For very large initial loads, create the table empty first, then call `add(...)`. Passing data directly to `create_table(name, data)` can skip the auto-parallel write path.
|
|
||||||
|
|
||||||
### Recommended: iterator ingestion for generated or streamed data
|
|
||||||
|
|
||||||
```python
|
|
||||||
def batches():
|
|
||||||
for raw in source:
|
|
||||||
vectors = model.encode(raw["text"])
|
|
||||||
yield pa.RecordBatch.from_pydict({**raw, "vector": vectors})
|
|
||||||
|
|
||||||
table.add(batches())
|
|
||||||
```
|
|
||||||
|
|
||||||
Use chunks of several thousand rows or more when practical. Tiny batches and per-row writes create many small fragments.
|
|
||||||
|
|
||||||
### Anti-pattern: per-row `add()`
|
|
||||||
|
|
||||||
```python
|
|
||||||
for row in rows:
|
|
||||||
table.add([row])
|
|
||||||
```
|
|
||||||
|
|
||||||
Each call creates a version and fragment. This slows ingestion and later queries.
|
|
||||||
|
|
||||||
## Indexing
|
|
||||||
|
|
||||||
- Build a vector index once brute-force vector search becomes too slow. As a rule of thumb, local brute force is fine below roughly 100K vectors; beyond that, build an index.
|
|
||||||
- Use `IVF_PQ` as the general-purpose default. Enterprise builds this automatically.
|
|
||||||
- Use scalar indexes for filtered columns and merge/upsert keys.
|
|
||||||
- Use `BTREE` for mostly distinct numeric/string/temporal columns, `BITMAP` for booleans and low-cardinality columns, and `LABEL_LIST` for list membership queries.
|
|
||||||
- Keep full-text defaults unless phrase queries require position data.
|
|
||||||
|
|
||||||
## Querying
|
|
||||||
|
|
||||||
Always be explicit:
|
|
||||||
|
|
||||||
```python
|
|
||||||
table.search(query_vector).select(["id", "title"]).limit(20)
|
|
||||||
```
|
|
||||||
|
|
||||||
- `select()` reduces bytes read and transferred.
|
|
||||||
- `limit()` prevents accidental full-table materialization.
|
|
||||||
- Pre-filtering is the default and guarantees returned rows satisfy the predicate.
|
|
||||||
- Use post-filtering only when fewer than `limit` results are acceptable.
|
|
||||||
|
|
||||||
## Recall Tuning
|
|
||||||
|
|
||||||
Tune one knob at a time:
|
|
||||||
|
|
||||||
- Quantized indexes: raise `refine_factor` to rescore more candidates on full vectors.
|
|
||||||
- HNSW-backed indexes: raise `ef`; start around `1.5 * k`, increase toward `10 * k` if recall is short.
|
|
||||||
- IVF candidate breadth: `nprobes` is auto-tuned; override only when a selective pre-filter leaves too few neighbors.
|
|
||||||
|
|
||||||
## Maintenance
|
|
||||||
|
|
||||||
After every successful embedded OSS/local ingestion, call `table.optimize()`.
|
|
||||||
Do not add this to LanceDB Enterprise/Cloud remote table code; remote compaction
|
|
||||||
and cleanup are handled automatically based on the Enterprise cluster
|
|
||||||
configuration.
|
|
||||||
|
|
||||||
Why local maintenance is needed:
|
|
||||||
|
|
||||||
- Frequent writes can create many small fragments. Queries then need to scan across more files, which can increase latency.
|
|
||||||
- Updates, deletes, and appends create new table versions. Old versions are retained for time travel and rollback, which can grow disk usage.
|
|
||||||
- Indexes may have newly added rows that are not yet fully optimized into the index structure.
|
|
||||||
|
|
||||||
For local/OSS tables, run `optimize()` after the final successful ingestion
|
|
||||||
write. Also run it after later batches of update/delete operations or on a
|
|
||||||
regular maintenance schedule:
|
|
||||||
|
|
||||||
```python
|
|
||||||
table.optimize()
|
|
||||||
```
|
|
||||||
|
|
||||||
If the user wants more aggressive local disk cleanup, pass a shorter cleanup retention window:
|
|
||||||
|
|
||||||
```python
|
|
||||||
from datetime import timedelta
|
|
||||||
|
|
||||||
table.optimize(cleanup_older_than=timedelta(days=1))
|
|
||||||
```
|
|
||||||
|
|
||||||
Do not use very short cleanup windows when the application depends on time travel, rollback, or old versions.
|
|
||||||
|
|
||||||
## Diagnostics
|
|
||||||
|
|
||||||
Before changing code or indexes, inspect:
|
|
||||||
|
|
||||||
```python
|
|
||||||
print(table.search(query_vector).where("year > 2000").limit(10).analyze_plan())
|
|
||||||
print(table.index_stats("vector_idx"))
|
|
||||||
```
|
|
||||||
|
|
||||||
Look for high scan bytes, missing indexes, fragmented data, and unindexed rows.
|
|
||||||
|
|
||||||
## Python Multiprocessing
|
|
||||||
|
|
||||||
When using multiprocessing, use `spawn` rather than `fork`. LanceDB is multi-threaded internally, and `fork` plus a multi-threaded process is unsafe.
|
|
||||||
@@ -1,105 +0,0 @@
|
|||||||
# TypeScript API Reference
|
|
||||||
|
|
||||||
Quick method reference for TypeScript LanceDB code. Cross-check source for non-trivial claims.
|
|
||||||
|
|
||||||
## Connect
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
import * as lancedb from "@lancedb/lancedb";
|
|
||||||
|
|
||||||
const db = await lancedb.connect("./camelot-db");
|
|
||||||
```
|
|
||||||
|
|
||||||
**Place the local database directory next to the script/entrypoint that opens it** (resolve the path relative to the module, e.g. via `import.meta.dirname` / `__dirname`), not buried under a shared `data/` folder. The Lance dataset is the database, not a data file — keeping it beside its code makes ownership obvious and paths stable regardless of the working directory the script is launched from.
|
|
||||||
|
|
||||||
**Do not name the directory `lancedb`** (e.g. `./lancedb`, `./data/lancedb`). It collides with the imported `lancedb` package/namespace, which is confusing to read. Give it a name derived from the repo or dataset with a clear prefix/suffix — for example `./<dataset>-db`, `./<repo>_lancedb`, or `./vectordb`.
|
|
||||||
|
|
||||||
Remote connections use `db://...` plus Enterprise/Cloud credentials and deployment settings. Check current source/docs for exact connection options.
|
|
||||||
|
|
||||||
## Table Reads
|
|
||||||
|
|
||||||
| Task | Preferred API |
|
|
||||||
| --- | --- |
|
|
||||||
| Vector search | `table.search(queryVector).limit(k)` |
|
|
||||||
| Full scan with filters/projection | `table.query().where(...).select(...).limit(...)` |
|
|
||||||
| Filter | `.where("col > 10")` |
|
|
||||||
| Projection | `.select(["id", "text"])` |
|
|
||||||
| Bound result count | `.limit(20)` |
|
|
||||||
| Collect bounded result as objects | `.toArray()` on query/search result |
|
|
||||||
| Collect bounded result as Arrow | `.toArrow()` on query/search result |
|
|
||||||
| Stream result batches | `for await (const batch of table.query()...)` |
|
|
||||||
|
|
||||||
## Local vs Remote Safety
|
|
||||||
|
|
||||||
| API | Agent guidance |
|
|
||||||
| --- | --- |
|
|
||||||
| `table.search(...)` | Preferred read path |
|
|
||||||
| `table.query()` | Preferred scan/filter path |
|
|
||||||
| `await table.toArrow()` | Avoid in portable or large-table code |
|
|
||||||
| `await table.query().toArray()` with no `limit()` | Avoid; unbounded collection |
|
|
||||||
| `await table.query().toArrow()` with no `limit()` | Avoid; unbounded collection |
|
|
||||||
|
|
||||||
## Indexes
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
await table.createIndex("vector");
|
|
||||||
await table.createIndex("status");
|
|
||||||
```
|
|
||||||
|
|
||||||
Use vector indexes for large vector search workloads and scalar indexes for filtered columns or merge/upsert keys. Check source/docs before specifying advanced index options.
|
|
||||||
|
|
||||||
## Filtering And Recall Knobs
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
await table.search(queryVector).where("status = 'ready'").limit(10).toArray();
|
|
||||||
await table.search(queryVector).limit(10).refineFactor(20).toArray();
|
|
||||||
await table.search(queryVector).limit(10).nprobes(50).toArray();
|
|
||||||
await table.search(queryVector).limit(10).ef(100).toArray();
|
|
||||||
await table.search(queryVector).where("status = 'ready'").postfilter().limit(10).toArray();
|
|
||||||
```
|
|
||||||
|
|
||||||
Use `postfilter()` only when fewer than `limit` results are acceptable.
|
|
||||||
|
|
||||||
## Diagnostics
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
console.log(await table.search(queryVector).where("year > 2000").limit(10).analyzePlan());
|
|
||||||
console.log(await table.indexStats("vector_idx"));
|
|
||||||
```
|
|
||||||
|
|
||||||
Use these before changing indexes or search tuning.
|
|
||||||
|
|
||||||
## Column (Field) Metadata
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
const schema = await table.schema();
|
|
||||||
const meta = schema.fields.find((f) => f.name === "category")?.metadata; // Map<string, string>
|
|
||||||
const res = await table.updateFieldMetadata([
|
|
||||||
{ path: "category", metadata: { "lancedb:description": "...", "lancedb:tag:field_type": "label" } },
|
|
||||||
]);
|
|
||||||
res.version; // new table version
|
|
||||||
```
|
|
||||||
|
|
||||||
Merges by default; a `null` value deletes that key; `replace: true` swaps the whole map. Nested fields use dot-paths (`"a.b.c"`). See `references/column_metadata.md` for key conventions (`lancedb:description`, `lancedb:tag:<name>`, `lancedb:logical-column`) and the authoring workflow.
|
|
||||||
|
|
||||||
## Branches
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
const branches = await table.branches(); // async manager
|
|
||||||
await branches.list(); // non-main branches; {} = only main
|
|
||||||
const exp = await branches.create("exp"); // fork off main -> Table scoped to the branch
|
|
||||||
const wip = await branches.checkout("wip"); // existing branch -> scoped Table (version arg pins read-only)
|
|
||||||
const wip2 = await db.openTable("t", { branch: "wip" }); // or open scoped directly
|
|
||||||
await branches.delete("stale"); // removes only the branch pointer
|
|
||||||
table.currentBranch(); // null = main
|
|
||||||
```
|
|
||||||
|
|
||||||
There is no global switch — scoping is per table handle: any read/write on a branch handle lands on that branch; the original handle keeps targeting main. See `references/branch_ops.md` for the model and isolation checks.
|
|
||||||
|
|
||||||
## Maintenance
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
await table.optimize();
|
|
||||||
```
|
|
||||||
|
|
||||||
Call this after every successful local/OSS ingestion. It handles compaction, cleanup of old versions according to retention, and index optimization. Do not add this for LanceDB Enterprise/Cloud remote tables; Enterprise handles compaction and cleanup automatically from cluster configuration.
|
|
||||||
@@ -1,100 +0,0 @@
|
|||||||
# TypeScript Patterns
|
|
||||||
|
|
||||||
Use these patterns when writing TypeScript code with `@lancedb/lancedb`.
|
|
||||||
|
|
||||||
## Recommended Patterns
|
|
||||||
|
|
||||||
### Bounded query
|
|
||||||
|
|
||||||
Use this for application reads, scripts, and examples:
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
const rows = await table
|
|
||||||
.query()
|
|
||||||
.where("status = 'ready'")
|
|
||||||
.select(["id", "text"])
|
|
||||||
.limit(20)
|
|
||||||
.toArray();
|
|
||||||
```
|
|
||||||
|
|
||||||
### Bounded vector search
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
const rows = await table
|
|
||||||
.search(queryVector)
|
|
||||||
.select(["id", "text"])
|
|
||||||
.limit(20)
|
|
||||||
.toArray();
|
|
||||||
```
|
|
||||||
|
|
||||||
### Batch streaming for larger reads
|
|
||||||
|
|
||||||
When the task needs many rows, avoid collecting everything at once:
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
for await (const batch of table
|
|
||||||
.query()
|
|
||||||
.where("status = 'ready'")
|
|
||||||
.select(["id", "text"])
|
|
||||||
.limit(10_000)) {
|
|
||||||
process(batch);
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
## Anti-Patterns
|
|
||||||
|
|
||||||
**Avoid the following anti-patterns in your code.**
|
|
||||||
|
|
||||||
### Table-level full materialization
|
|
||||||
|
|
||||||
Avoid whole-table collectors in portable or large-table code:
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
const tableArrow = await table.toArrow();
|
|
||||||
```
|
|
||||||
|
|
||||||
Why: local tables expose these whole-table collectors, but remote tables intentionally do not — a remote production table can be far larger than a local development table, so it is easy to accidentally pull the entire table into memory.
|
|
||||||
|
|
||||||
### Unbounded result collection
|
|
||||||
|
|
||||||
Avoid query/search collection without a meaningful limit:
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
const rows = await table.query().toArray(); // unbounded plain scan
|
|
||||||
const rows = await table.search(queryVector).toArray(); // unbounded vector search
|
|
||||||
```
|
|
||||||
|
|
||||||
Prefer `select(...).limit(...)` before collecting; for large reads, stream in batches instead.
|
|
||||||
|
|
||||||
### Per-row writes
|
|
||||||
|
|
||||||
Avoid loops that write one row per call:
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
for (const row of rows) {
|
|
||||||
await table.add([row]); // one commit + fragment per row
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
Each `add()` creates a new version and fragment. Pass the whole batch in a single call, or chunk very large inputs:
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
await table.add(rows); // single commit
|
|
||||||
// for very large inputs, add in chunks of several thousand rows
|
|
||||||
```
|
|
||||||
|
|
||||||
### Drop-then-reuse the same table name (Enterprise/Cloud)
|
|
||||||
|
|
||||||
Avoid dropping or overwriting a remote table and then reusing that name right away:
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
await db.dropTable("my_table");
|
|
||||||
const table = await db.createTable("my_table", rows); // reads 500 for ~5 min
|
|
||||||
const table = await db.createTable("my_table", rows, { mode: "overwrite" }); // same problem
|
|
||||||
```
|
|
||||||
|
|
||||||
Why: Enterprise/Cloud splits DDL (control plane) from query serving (data plane). The data plane caches the dataset behind a table name for up to `table_cache_ttl` (default 300s / 5 min), so after a drop/overwrite the DDL succeeds but queries against the reused name return `500 Internal Server Error` until the cache expires — and a fresh `describe` may still show the old schema. Instead, write to a **fresh name**, use `tableNames()` and fail if it already exists, then `renameTable(fresh, final)` onto the final name only after the old table's drop has propagated (~5 min). See the "Enterprise: never drop-then-reuse the same table name" section in `SKILL.md`. Local/OSS tables have no separate data plane — overwrite freely there.
|
|
||||||
|
|
||||||
### Guessing performance fixes
|
|
||||||
|
|
||||||
Avoid changing `nprobes`, `refineFactor`, `ef`, or index settings before checking `analyzePlan()` and `indexStats(...)`. Diagnose first, then tune one knob at a time.
|
|
||||||
@@ -1,78 +0,0 @@
|
|||||||
# TypeScript Performance Guidance
|
|
||||||
|
|
||||||
Use this when writing TypeScript code that ingests data, queries large tables, builds indexes, or investigates latency.
|
|
||||||
|
|
||||||
## Ingestion
|
|
||||||
|
|
||||||
- Prefer bulk or batched writes.
|
|
||||||
- Avoid per-row write loops; they create many small commits/fragments.
|
|
||||||
- For generated data, accumulate reasonable batches before adding.
|
|
||||||
- For file-backed data, prefer APIs that stream from Arrow/Parquet-style inputs when available.
|
|
||||||
|
|
||||||
## Indexing
|
|
||||||
|
|
||||||
- Build a vector index once brute-force vector search becomes too slow. As a rule of thumb, local brute force is fine below roughly 100K vectors; beyond that, build an index.
|
|
||||||
- Use the general-purpose vector index defaults unless the task has explicit recall/latency requirements.
|
|
||||||
- Build scalar indexes for filtered columns and merge/upsert keys.
|
|
||||||
- Use full-text index phrase options only when phrase queries require them.
|
|
||||||
|
|
||||||
## Querying
|
|
||||||
|
|
||||||
Always be explicit:
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
await table.search(queryVector).select(["id", "title"]).limit(20).toArray();
|
|
||||||
```
|
|
||||||
|
|
||||||
- `select()` reduces bytes read and transferred.
|
|
||||||
- `limit()` prevents accidental full-table collection.
|
|
||||||
- Pre-filtering is the default behavior. Use `postfilter()` only when fewer than `limit` results are acceptable.
|
|
||||||
|
|
||||||
## Recall Tuning
|
|
||||||
|
|
||||||
Tune one knob at a time:
|
|
||||||
|
|
||||||
- Quantized indexes: raise `refineFactor(...)` to rescore more candidates on full vectors.
|
|
||||||
- HNSW-backed indexes: raise `ef(...)`; start around `1.5 * k`, increase toward `10 * k` if recall is short.
|
|
||||||
- IVF candidate breadth: `nprobes(...)` is usually auto-tuned; override only when a selective pre-filter leaves too few neighbors.
|
|
||||||
|
|
||||||
## Maintenance
|
|
||||||
|
|
||||||
After every successful embedded OSS/local ingestion, call `table.optimize()`.
|
|
||||||
Do not add this to LanceDB Enterprise/Cloud remote table code; remote compaction
|
|
||||||
and cleanup are handled automatically based on the Enterprise cluster
|
|
||||||
configuration.
|
|
||||||
|
|
||||||
Why local maintenance is needed:
|
|
||||||
|
|
||||||
- Frequent writes can create many small fragments. Queries then need to scan across more files, which can increase latency.
|
|
||||||
- Updates, deletes, and appends create new table versions. Old versions are retained for time travel and rollback, which can grow disk usage.
|
|
||||||
- Indexes may have newly added rows that are not yet fully optimized into the index structure.
|
|
||||||
|
|
||||||
For local/OSS tables, run `optimize()` after the final successful ingestion
|
|
||||||
write. Also run it after later batches of update/delete operations or on a
|
|
||||||
regular maintenance schedule:
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
await table.optimize();
|
|
||||||
```
|
|
||||||
|
|
||||||
If the user wants more aggressive local disk cleanup, pass a shorter cleanup retention window:
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
const olderThan = new Date(Date.now() - 24 * 60 * 60 * 1000);
|
|
||||||
await table.optimize({ cleanupOlderThan: olderThan });
|
|
||||||
```
|
|
||||||
|
|
||||||
Do not use very short cleanup windows when the application depends on time travel, rollback, or old versions.
|
|
||||||
|
|
||||||
## Diagnostics
|
|
||||||
|
|
||||||
Before changing code or indexes, inspect:
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
console.log(await table.search(queryVector).where("year > 2000").limit(10).analyzePlan());
|
|
||||||
console.log(await table.indexStats("vector_idx"));
|
|
||||||
```
|
|
||||||
|
|
||||||
Look for high scan cost, missing indexes, fragmented data, and unindexed rows.
|
|
||||||
@@ -64,7 +64,9 @@ def scan_python(path: Path, text: str) -> list[Finding]:
|
|||||||
|
|
||||||
def statement_around(text: str, start: int, end: int) -> str:
|
def statement_around(text: str, start: int, end: int) -> str:
|
||||||
before = max(text.rfind(";", 0, start), text.rfind("\n\n", 0, start))
|
before = max(text.rfind(";", 0, start), text.rfind("\n\n", 0, start))
|
||||||
after_candidates = [pos for pos in (text.find(";", end), text.find("\n\n", end)) if pos != -1]
|
after_candidates = [
|
||||||
|
pos for pos in (text.find(";", end), text.find("\n\n", end)) if pos != -1
|
||||||
|
]
|
||||||
after = min(after_candidates) if after_candidates else len(text)
|
after = min(after_candidates) if after_candidates else len(text)
|
||||||
return text[before + 1 : after].strip()
|
return text[before + 1 : after].strip()
|
||||||
|
|
||||||
|
|||||||
+12
-12
@@ -1,6 +1,6 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "lancedb-python"
|
name = "lancedb-python"
|
||||||
version = "0.37.1-beta.0"
|
version = "0.38.0-beta.3"
|
||||||
publish = false
|
publish = false
|
||||||
edition.workspace = true
|
edition.workspace = true
|
||||||
description = "Python bindings for LanceDB"
|
description = "Python bindings for LanceDB"
|
||||||
@@ -15,10 +15,10 @@ name = "_lancedb"
|
|||||||
crate-type = ["cdylib"]
|
crate-type = ["cdylib"]
|
||||||
|
|
||||||
[dependencies]
|
[dependencies]
|
||||||
arrow = { version = "58.0.0", features = ["pyarrow"] }
|
arrow = { workspace = true, features = ["pyarrow"] }
|
||||||
async-trait = "0.1"
|
async-trait.workspace = true
|
||||||
bytes = "1"
|
bytes.workspace = true
|
||||||
lancedb = { path = "../rust/lancedb", default-features = false }
|
lancedb.workspace = true
|
||||||
datafusion-common.workspace = true
|
datafusion-common.workspace = true
|
||||||
lance-core.workspace = true
|
lance-core.workspace = true
|
||||||
lance-namespace.workspace = true
|
lance-namespace.workspace = true
|
||||||
@@ -26,24 +26,24 @@ lance-namespace-impls.workspace = true
|
|||||||
lance-io.workspace = true
|
lance-io.workspace = true
|
||||||
env_logger.workspace = true
|
env_logger.workspace = true
|
||||||
log.workspace = true
|
log.workspace = true
|
||||||
pyo3 = { version = "0.28", features = ["extension-module", "abi3-py39", "chrono"] }
|
pyo3 = { version = "0.28", features = ["extension-module", "abi3-py310", "chrono"] }
|
||||||
chrono = { version = "0.4", default-features = false, features = ["clock"] }
|
chrono.workspace = true
|
||||||
pyo3-async-runtimes = { version = "0.28", features = [
|
pyo3-async-runtimes = { version = "0.28", features = [
|
||||||
"attributes",
|
"attributes",
|
||||||
"tokio-runtime",
|
"tokio-runtime",
|
||||||
] }
|
] }
|
||||||
pin-project = "1.1.5"
|
pin-project.workspace = true
|
||||||
futures.workspace = true
|
futures.workspace = true
|
||||||
serde = "1"
|
serde.workspace = true
|
||||||
serde_json = "1"
|
serde_json.workspace = true
|
||||||
snafu.workspace = true
|
snafu.workspace = true
|
||||||
tokio = { version = "1.40", features = ["sync", "rt-multi-thread"] }
|
tokio.workspace = true
|
||||||
libc = "0.2"
|
libc = "0.2"
|
||||||
|
|
||||||
[build-dependencies]
|
[build-dependencies]
|
||||||
pyo3-build-config = { version = "0.28", features = [
|
pyo3-build-config = { version = "0.28", features = [
|
||||||
"extension-module",
|
"extension-module",
|
||||||
"abi3-py39",
|
"abi3-py310",
|
||||||
] }
|
] }
|
||||||
|
|
||||||
[features]
|
[features]
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ dependencies = [
|
|||||||
"overrides>=0.7; python_version<'3.12'",
|
"overrides>=0.7; python_version<'3.12'",
|
||||||
"packaging>=23.0",
|
"packaging>=23.0",
|
||||||
"pyarrow>=16",
|
"pyarrow>=16",
|
||||||
"pydantic>=1.10",
|
"pydantic>=2.7.4,<3",
|
||||||
"tqdm>=4.27.0",
|
"tqdm>=4.27.0",
|
||||||
"lance-namespace>=0.3.2"
|
"lance-namespace>=0.3.2"
|
||||||
]
|
]
|
||||||
@@ -60,7 +60,7 @@ tests = [
|
|||||||
"pytest-asyncio>=0.21",
|
"pytest-asyncio>=0.21",
|
||||||
"duckdb>=0.9.0",
|
"duckdb>=0.9.0",
|
||||||
"pytz>=2023.3",
|
"pytz>=2023.3",
|
||||||
"polars>=0.19, <=1.3.0",
|
"polars>=0.19, <=1.32.3",
|
||||||
"pyarrow<25",
|
"pyarrow<25",
|
||||||
"pyarrow-stubs>=16.0",
|
"pyarrow-stubs>=16.0",
|
||||||
"pylance==9.0.0rc1",
|
"pylance==9.0.0rc1",
|
||||||
@@ -140,6 +140,7 @@ include = [
|
|||||||
"python/lancedb/remote/errors.py",
|
"python/lancedb/remote/errors.py",
|
||||||
"python/lancedb/embeddings/__init__.py",
|
"python/lancedb/embeddings/__init__.py",
|
||||||
"python/lancedb/_lancedb.pyi",
|
"python/lancedb/_lancedb.pyi",
|
||||||
|
"python/type_tests/connect.py",
|
||||||
]
|
]
|
||||||
exclude = ["python/tests/"]
|
exclude = ["python/tests/"]
|
||||||
pythonVersion = "3.13"
|
pythonVersion = "3.13"
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ __version__ = importlib.metadata.version("lancedb")
|
|||||||
|
|
||||||
from ._lancedb import connect as lancedb_connect
|
from ._lancedb import connect as lancedb_connect
|
||||||
from ._lancedb import FtsToken
|
from ._lancedb import FtsToken
|
||||||
|
from ._lancedb import LsmWriteSpec
|
||||||
from ._lancedb import tokenize as _tokenize
|
from ._lancedb import tokenize as _tokenize
|
||||||
from .common import URI, sanitize_uri
|
from .common import URI, sanitize_uri
|
||||||
from urllib.parse import urlparse
|
from urllib.parse import urlparse
|
||||||
@@ -20,6 +21,17 @@ from .remote import ClientConfig
|
|||||||
from .remote.db import RemoteDBConnection
|
from .remote.db import RemoteDBConnection
|
||||||
from .expr import Expr, col, lit, func
|
from .expr import Expr, col, lit, func
|
||||||
from .schema import blob, vector, BlobType
|
from .schema import blob, vector, BlobType
|
||||||
|
from .job import AsyncJob, Job
|
||||||
|
from .functions import (
|
||||||
|
FunctionArtifactRequest as FunctionArtifactRequest,
|
||||||
|
FunctionApplication as FunctionApplication,
|
||||||
|
FunctionBinding as FunctionBinding,
|
||||||
|
FunctionRegistrationRequest as FunctionRegistrationRequest,
|
||||||
|
FunctionVersion as FunctionVersion,
|
||||||
|
PythonRuntimeSpec as PythonRuntimeSpec,
|
||||||
|
UdfDefinition as UdfDefinition,
|
||||||
|
udf as udf,
|
||||||
|
)
|
||||||
from .table import AsyncTable, Table
|
from .table import AsyncTable, Table
|
||||||
from .types import BaseTokenizerType
|
from .types import BaseTokenizerType
|
||||||
from ._lancedb import Session
|
from ._lancedb import Session
|
||||||
@@ -500,6 +512,7 @@ __all__ = [
|
|||||||
"connect_namespace",
|
"connect_namespace",
|
||||||
"connect_namespace_async",
|
"connect_namespace_async",
|
||||||
"AsyncConnection",
|
"AsyncConnection",
|
||||||
|
"AsyncJob",
|
||||||
"AsyncLanceNamespaceDBConnection",
|
"AsyncLanceNamespaceDBConnection",
|
||||||
"AsyncTable",
|
"AsyncTable",
|
||||||
"FtsToken",
|
"FtsToken",
|
||||||
@@ -513,8 +526,10 @@ __all__ = [
|
|||||||
"BlobType",
|
"BlobType",
|
||||||
"vector",
|
"vector",
|
||||||
"DBConnection",
|
"DBConnection",
|
||||||
|
"Job",
|
||||||
"LanceDBConnection",
|
"LanceDBConnection",
|
||||||
"LanceNamespaceDBConnection",
|
"LanceNamespaceDBConnection",
|
||||||
|
"LsmWriteSpec",
|
||||||
"RemoteDBConnection",
|
"RemoteDBConnection",
|
||||||
"Session",
|
"Session",
|
||||||
"Table",
|
"Table",
|
||||||
|
|||||||
@@ -14,14 +14,10 @@ import pyarrow as pa
|
|||||||
from .expr import Expr
|
from .expr import Expr
|
||||||
from .schema import blob_v2_column_paths
|
from .schema import blob_v2_column_paths
|
||||||
from .types import BlobMode, QueryProjection, QueryProjectionSpec
|
from .types import BlobMode, QueryProjection, QueryProjectionSpec
|
||||||
from .util import get_uri_scheme
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
from _typeshed import WriteableBuffer
|
from _typeshed import WriteableBuffer
|
||||||
|
|
||||||
from .remote.table import RemoteTable
|
|
||||||
from .table import AsyncTable, Table
|
|
||||||
|
|
||||||
BLOB_MODE_TO_HANDLING = {
|
BLOB_MODE_TO_HANDLING = {
|
||||||
"lazy": "blobs_descriptions",
|
"lazy": "blobs_descriptions",
|
||||||
"bytes": "all_binary",
|
"bytes": "all_binary",
|
||||||
@@ -104,22 +100,6 @@ def validate_blob_mode(blob_mode: BlobMode) -> None:
|
|||||||
raise ValueError(f"blob_mode must be one of {modes}, got {blob_mode!r}")
|
raise ValueError(f"blob_mode must be one of {modes}, got {blob_mode!r}")
|
||||||
|
|
||||||
|
|
||||||
def supports_blob_auto_row_id(table: Table | AsyncTable | RemoteTable) -> bool:
|
|
||||||
"""Blob auto row-id applies to native tables, not LanceDB Cloud."""
|
|
||||||
from .remote.table import RemoteTable
|
|
||||||
|
|
||||||
if isinstance(table, RemoteTable):
|
|
||||||
return False
|
|
||||||
|
|
||||||
inner = getattr(table, "_inner", None)
|
|
||||||
if inner is not None:
|
|
||||||
uri = inner.database().uri
|
|
||||||
if isinstance(uri, str) and get_uri_scheme(uri) == "db":
|
|
||||||
return False
|
|
||||||
|
|
||||||
return True
|
|
||||||
|
|
||||||
|
|
||||||
def projection_includes_blob_column(
|
def projection_includes_blob_column(
|
||||||
projection: QueryProjection,
|
projection: QueryProjection,
|
||||||
blob_columns: Iterable[str],
|
blob_columns: Iterable[str],
|
||||||
@@ -164,16 +144,14 @@ def v2_projection_needs_row_id(
|
|||||||
|
|
||||||
|
|
||||||
def blob_auto_row_id_for_scan(
|
def blob_auto_row_id_for_scan(
|
||||||
table: Table | AsyncTable | RemoteTable,
|
|
||||||
schema: pa.Schema,
|
schema: pa.Schema,
|
||||||
projection: QueryProjection,
|
projection: QueryProjection,
|
||||||
*,
|
*,
|
||||||
with_row_id: bool | None,
|
with_row_id: bool | None,
|
||||||
) -> bool:
|
) -> bool:
|
||||||
|
"""Auto row-id only applies when the caller said nothing about row ids."""
|
||||||
if with_row_id is not None:
|
if with_row_id is not None:
|
||||||
return False
|
return False
|
||||||
if not supports_blob_auto_row_id(table):
|
|
||||||
return False
|
|
||||||
return v2_projection_needs_row_id(schema, projection, with_row_id=False)
|
return v2_projection_needs_row_id(schema, projection, with_row_id=False)
|
||||||
|
|
||||||
|
|
||||||
@@ -186,6 +164,11 @@ def finalize_blob_query_table(
|
|||||||
) -> pa.Table:
|
) -> pa.Table:
|
||||||
if user_requested_row_id or not blob_auto_row_id:
|
if user_requested_row_id or not blob_auto_row_id:
|
||||||
return tbl
|
return tbl
|
||||||
|
if "_rowid" not in tbl.column_names:
|
||||||
|
# A backend that ignores the row-id request leaves nothing to stash. Hand
|
||||||
|
# back the projection as-is so fetch_blobs raises the error that names the
|
||||||
|
# ways to supply row ids, rather than failing here about a hidden column.
|
||||||
|
return tbl
|
||||||
return stash_auto_row_ids(tbl, blob_paths)
|
return stash_auto_row_ids(tbl, blob_paths)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -146,6 +146,15 @@ class Connection(object):
|
|||||||
start_after: Optional[str],
|
start_after: Optional[str],
|
||||||
limit: Optional[int],
|
limit: Optional[int],
|
||||||
) -> list[str]: ... # Deprecated: Use list_tables instead
|
) -> list[str]: ... # Deprecated: Use list_tables instead
|
||||||
|
def job(self, job_id: str) -> Job: ...
|
||||||
|
async def create_function_async(self, request_json: str) -> FunctionJob: ...
|
||||||
|
async def get_function(self, name: str, version: str) -> str: ...
|
||||||
|
async def list_jobs(self) -> List[JobInfo]: ...
|
||||||
|
async def get_job(self, job_id: str) -> Optional[JobDescription]: ...
|
||||||
|
async def cancel_job(self, job_id: str) -> bool: ...
|
||||||
|
async def job_history(
|
||||||
|
self, job_id: Optional[str] = None
|
||||||
|
) -> List[pa.RecordBatch]: ...
|
||||||
async def create_table(
|
async def create_table(
|
||||||
self,
|
self,
|
||||||
name: str,
|
name: str,
|
||||||
@@ -191,6 +200,9 @@ class Connection(object):
|
|||||||
async def drop_table(
|
async def drop_table(
|
||||||
self, name: str, namespace_path: Optional[List[str]] = None
|
self, name: str, namespace_path: Optional[List[str]] = None
|
||||||
) -> None: ...
|
) -> None: ...
|
||||||
|
async def drop_table_async(
|
||||||
|
self, name: str, namespace_path: Optional[List[str]] = None
|
||||||
|
) -> Job: ...
|
||||||
async def drop_all_tables(
|
async def drop_all_tables(
|
||||||
self, namespace_path: Optional[List[str]] = None
|
self, namespace_path: Optional[List[str]] = None
|
||||||
) -> None: ...
|
) -> None: ...
|
||||||
@@ -209,6 +221,54 @@ class BlobFile:
|
|||||||
def read_range(self, offset: int, length: int) -> bytes: ...
|
def read_range(self, offset: int, length: int) -> bytes: ...
|
||||||
def read_up_to(self, length: int) -> bytes: ...
|
def read_up_to(self, length: int) -> bytes: ...
|
||||||
|
|
||||||
|
class Job:
|
||||||
|
@property
|
||||||
|
def id(self) -> Optional[str]: ...
|
||||||
|
async def status(self) -> str: ...
|
||||||
|
async def wait(self) -> None: ...
|
||||||
|
async def cancel(self) -> None: ...
|
||||||
|
|
||||||
|
class FunctionJob:
|
||||||
|
@property
|
||||||
|
def id(self) -> Optional[str]: ...
|
||||||
|
async def status(self) -> str: ...
|
||||||
|
async def wait(self) -> str: ...
|
||||||
|
async def cancel(self) -> None: ...
|
||||||
|
|
||||||
|
class JobInfo:
|
||||||
|
@property
|
||||||
|
def job_id(self) -> str: ...
|
||||||
|
@property
|
||||||
|
def table(self) -> str: ...
|
||||||
|
@property
|
||||||
|
def job_type(self) -> str: ...
|
||||||
|
@property
|
||||||
|
def state(self) -> str: ...
|
||||||
|
@property
|
||||||
|
def created_at_millis(self) -> int: ...
|
||||||
|
|
||||||
|
class JobFailureInfo:
|
||||||
|
@property
|
||||||
|
def phase(self) -> Optional[str]: ...
|
||||||
|
@property
|
||||||
|
def message(self) -> Optional[str]: ...
|
||||||
|
@property
|
||||||
|
def retryable(self) -> Optional[bool]: ...
|
||||||
|
|
||||||
|
class JobDescription:
|
||||||
|
@property
|
||||||
|
def job_id(self) -> str: ...
|
||||||
|
@property
|
||||||
|
def job_type(self) -> str: ...
|
||||||
|
@property
|
||||||
|
def state(self) -> str: ...
|
||||||
|
@property
|
||||||
|
def creation_ms(self) -> int: ...
|
||||||
|
@property
|
||||||
|
def spec_json(self) -> Optional[str]: ...
|
||||||
|
@property
|
||||||
|
def failure(self) -> Optional[JobFailureInfo]: ...
|
||||||
|
|
||||||
class Table:
|
class Table:
|
||||||
def name(self) -> str: ...
|
def name(self) -> str: ...
|
||||||
def __repr__(self) -> str: ...
|
def __repr__(self) -> str: ...
|
||||||
@@ -248,6 +308,28 @@ class Table:
|
|||||||
name: Optional[str],
|
name: Optional[str],
|
||||||
train: Optional[bool],
|
train: Optional[bool],
|
||||||
): ...
|
): ...
|
||||||
|
async def create_index_async(
|
||||||
|
self,
|
||||||
|
column: str,
|
||||||
|
index: Union[
|
||||||
|
IvfFlat,
|
||||||
|
IvfSq,
|
||||||
|
IvfPq,
|
||||||
|
HnswPq,
|
||||||
|
HnswSq,
|
||||||
|
HnswFlat,
|
||||||
|
BTree,
|
||||||
|
Bitmap,
|
||||||
|
LabelList,
|
||||||
|
Fm,
|
||||||
|
FTS,
|
||||||
|
],
|
||||||
|
replace: Optional[bool],
|
||||||
|
wait_timeout: Optional[object],
|
||||||
|
*,
|
||||||
|
name: Optional[str],
|
||||||
|
train: Optional[bool],
|
||||||
|
) -> Job: ...
|
||||||
async def list_versions(self) -> List[Dict[str, Any]]: ...
|
async def list_versions(self) -> List[Dict[str, Any]]: ...
|
||||||
async def version(self) -> int: ...
|
async def version(self) -> int: ...
|
||||||
async def checkout(self, version: Union[int, str]): ...
|
async def checkout(self, version: Union[int, str]): ...
|
||||||
@@ -265,6 +347,14 @@ class Table:
|
|||||||
) -> list[FtsToken]: ...
|
) -> list[FtsToken]: ...
|
||||||
async def delete(self, filter: Union[str, PyExpr]) -> DeleteResult: ...
|
async def delete(self, filter: Union[str, PyExpr]) -> DeleteResult: ...
|
||||||
async def add_columns(self, columns: list[tuple[str, str]]) -> AddColumnsResult: ...
|
async def add_columns(self, columns: list[tuple[str, str]]) -> AddColumnsResult: ...
|
||||||
|
async def add_computed_columns(
|
||||||
|
self, columns: list[tuple[str, str]]
|
||||||
|
) -> AddColumnsResult: ...
|
||||||
|
async def add_function_columns(
|
||||||
|
self, application_json: str, output_name: Optional[str]
|
||||||
|
) -> AddColumnsResult: ...
|
||||||
|
async def refresh_column(self, column: str) -> RefreshColumnResult: ...
|
||||||
|
async def refresh_column_async(self, column: str) -> Job: ...
|
||||||
async def add_columns_with_schema(self, schema: pa.Schema) -> AddColumnsResult: ...
|
async def add_columns_with_schema(self, schema: pa.Schema) -> AddColumnsResult: ...
|
||||||
async def alter_columns(
|
async def alter_columns(
|
||||||
self, columns: list[dict[str, Any]]
|
self, columns: list[dict[str, Any]]
|
||||||
@@ -285,6 +375,10 @@ class Table:
|
|||||||
async def set_lsm_write_spec(self, spec: LsmWriteSpec) -> None: ...
|
async def set_lsm_write_spec(self, spec: LsmWriteSpec) -> None: ...
|
||||||
async def unset_lsm_write_spec(self) -> None: ...
|
async def unset_lsm_write_spec(self) -> None: ...
|
||||||
async def get_lsm_write_spec(self) -> Optional[LsmWriteSpec]: ...
|
async def get_lsm_write_spec(self) -> Optional[LsmWriteSpec]: ...
|
||||||
|
async def checkpoint_lsm(self) -> None: ...
|
||||||
|
async def flush_lsm(self) -> None: ...
|
||||||
|
async def compact_lsm(self) -> None: ...
|
||||||
|
async def get_lsm_stats(self, include_generation_rows: bool) -> Optional[dict]: ...
|
||||||
async def close_lsm_writers(self) -> None: ...
|
async def close_lsm_writers(self) -> None: ...
|
||||||
@property
|
@property
|
||||||
def tags(self) -> Tags: ...
|
def tags(self) -> Tags: ...
|
||||||
@@ -579,9 +673,10 @@ class LsmWriteSpec:
|
|||||||
def identity(column: str) -> "LsmWriteSpec": ...
|
def identity(column: str) -> "LsmWriteSpec": ...
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def unsharded() -> "LsmWriteSpec": ...
|
def unsharded() -> "LsmWriteSpec": ...
|
||||||
def with_maintained_indexes(self, indexes: List[str]) -> "LsmWriteSpec":
|
def with_maintained_indexes(self, indexes: Optional[List[str]]) -> "LsmWriteSpec":
|
||||||
"""Return a copy of this spec asking the MemWAL to keep the named
|
"""Set which indexes the MemWAL keeps up to date. None resolves every
|
||||||
indexes up to date as rows are appended."""
|
index on the table at install, failing if one cannot be maintained;
|
||||||
|
a list is verbatim, empty means none."""
|
||||||
...
|
...
|
||||||
def with_writer_config_defaults(self, defaults: Dict[str, str]) -> "LsmWriteSpec":
|
def with_writer_config_defaults(self, defaults: Dict[str, str]) -> "LsmWriteSpec":
|
||||||
"""Return a copy of this spec recording the given default
|
"""Return a copy of this spec recording the given default
|
||||||
@@ -596,13 +691,19 @@ class LsmWriteSpec:
|
|||||||
@property
|
@property
|
||||||
def num_buckets(self) -> Optional[int]: ...
|
def num_buckets(self) -> Optional[int]: ...
|
||||||
@property
|
@property
|
||||||
def maintained_indexes(self) -> List[str]: ...
|
def maintained_indexes(self) -> Optional[List[str]]:
|
||||||
|
"""Indexes the MemWAL keeps up to date, or None for every supported one."""
|
||||||
|
...
|
||||||
@property
|
@property
|
||||||
def writer_config_defaults(self) -> Dict[str, str]: ...
|
def writer_config_defaults(self) -> Dict[str, str]: ...
|
||||||
|
|
||||||
class AddColumnsResult:
|
class AddColumnsResult:
|
||||||
version: int
|
version: int
|
||||||
|
|
||||||
|
class RefreshColumnResult:
|
||||||
|
rows_filled: int
|
||||||
|
version: int
|
||||||
|
|
||||||
class AlterColumnsResult:
|
class AlterColumnsResult:
|
||||||
version: int
|
version: int
|
||||||
|
|
||||||
|
|||||||
+276
-10
@@ -45,6 +45,8 @@ from lance_namespace.errors import NamespaceNotEmptyError, TableNotFoundError
|
|||||||
|
|
||||||
from . import __version__
|
from . import __version__
|
||||||
from ._lancedb import connect as lancedb_connect # type: ignore
|
from ._lancedb import connect as lancedb_connect # type: ignore
|
||||||
|
from .functions import FunctionVersion, UdfDefinition
|
||||||
|
from .job import AsyncJob, Job, _function_job
|
||||||
from .table import (
|
from .table import (
|
||||||
AsyncTable,
|
AsyncTable,
|
||||||
LanceTable,
|
LanceTable,
|
||||||
@@ -63,6 +65,7 @@ if TYPE_CHECKING:
|
|||||||
from .pydantic import LanceModel
|
from .pydantic import LanceModel
|
||||||
|
|
||||||
from ._lancedb import Connection as LanceDbConnection
|
from ._lancedb import Connection as LanceDbConnection
|
||||||
|
from ._lancedb import JobDescription, JobInfo
|
||||||
from .common import DATA, URI
|
from .common import DATA, URI
|
||||||
from .embeddings import EmbeddingFunctionConfig
|
from .embeddings import EmbeddingFunctionConfig
|
||||||
from ._lancedb import Session
|
from ._lancedb import Session
|
||||||
@@ -178,6 +181,51 @@ class DBConnection(EnforceOverrides):
|
|||||||
"Namespace operations are not supported for this connection type"
|
"Namespace operations are not supported for this connection type"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def namespace_exists(self, namespace_id: List[str]) -> bool:
|
||||||
|
"""Check if a namespace exists.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
namespace_id: List[str]
|
||||||
|
The namespace identifier to check.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
bool
|
||||||
|
True if the namespace exists, False otherwise.
|
||||||
|
|
||||||
|
Raises
|
||||||
|
------
|
||||||
|
NotImplementedError
|
||||||
|
If the connection type does not support namespace operations.
|
||||||
|
"""
|
||||||
|
raise NotImplementedError(
|
||||||
|
"Namespace operations are not supported for this connection type"
|
||||||
|
)
|
||||||
|
|
||||||
|
def table_exists(self, table_id: List[str]) -> bool:
|
||||||
|
"""Check if a table exists.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
table_id: List[str]
|
||||||
|
The table identifier to check (full path including namespace
|
||||||
|
segments and table name).
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
bool
|
||||||
|
True if the table exists, False otherwise.
|
||||||
|
|
||||||
|
Raises
|
||||||
|
------
|
||||||
|
NotImplementedError
|
||||||
|
If the connection type does not support namespace operations.
|
||||||
|
"""
|
||||||
|
raise NotImplementedError(
|
||||||
|
"Namespace operations are not supported for this connection type"
|
||||||
|
)
|
||||||
|
|
||||||
def list_tables(
|
def list_tables(
|
||||||
self,
|
self,
|
||||||
namespace_path: Optional[List[str]] = None,
|
namespace_path: Optional[List[str]] = None,
|
||||||
@@ -359,7 +407,7 @@ class DBConnection(EnforceOverrides):
|
|||||||
|
|
||||||
Data is converted to Arrow before being written to disk. For maximum
|
Data is converted to Arrow before being written to disk. For maximum
|
||||||
control over how data is saved, either provide the PyArrow schema to
|
control over how data is saved, either provide the PyArrow schema to
|
||||||
convert to or else provide a [PyArrow Table](pyarrow.Table) directly.
|
convert to or else provide a [PyArrow Table][pyarrow.Table] directly.
|
||||||
|
|
||||||
>>> import pyarrow as pa
|
>>> import pyarrow as pa
|
||||||
>>> custom_schema = pa.schema([
|
>>> custom_schema = pa.schema([
|
||||||
@@ -477,6 +525,12 @@ class DBConnection(EnforceOverrides):
|
|||||||
namespace_path = []
|
namespace_path = []
|
||||||
raise NotImplementedError
|
raise NotImplementedError
|
||||||
|
|
||||||
|
def drop_table_async(
|
||||||
|
self, name: str, namespace_path: Optional[List[str]] = None
|
||||||
|
) -> Job:
|
||||||
|
"""Start dropping a table and return its cleanup job."""
|
||||||
|
raise NotImplementedError
|
||||||
|
|
||||||
def rename_table(
|
def rename_table(
|
||||||
self,
|
self,
|
||||||
cur_name: str,
|
cur_name: str,
|
||||||
@@ -563,6 +617,71 @@ class DBConnection(EnforceOverrides):
|
|||||||
"""
|
"""
|
||||||
raise NotImplementedError("serialize is not supported for this connection type")
|
raise NotImplementedError("serialize is not supported for this connection type")
|
||||||
|
|
||||||
|
def create_function(self, definition: UdfDefinition) -> FunctionVersion:
|
||||||
|
"""Register a scalar Python UDF and wait for its immutable version.
|
||||||
|
|
||||||
|
This is the blocking counterpart of :meth:`create_function_async`.
|
||||||
|
Local connections raise ``NotImplementedError``.
|
||||||
|
"""
|
||||||
|
return self.create_function_async(definition).wait()
|
||||||
|
|
||||||
|
def create_function_async(self, definition: UdfDefinition) -> Job[FunctionVersion]:
|
||||||
|
"""Register a scalar Python UDF through the remote Function catalog.
|
||||||
|
|
||||||
|
Submission returns a typed job. The immutable Function version becomes
|
||||||
|
available only when :meth:`Job.wait` succeeds. Local connections raise
|
||||||
|
``NotImplementedError``.
|
||||||
|
"""
|
||||||
|
raise NotImplementedError(
|
||||||
|
"Function catalog operations are not supported for this connection type"
|
||||||
|
)
|
||||||
|
|
||||||
|
def get_function(self, name: str, *, version: str) -> FunctionVersion:
|
||||||
|
"""Open one exact immutable Function version from the remote catalog."""
|
||||||
|
raise NotImplementedError(
|
||||||
|
"Function catalog operations are not supported for this connection type"
|
||||||
|
)
|
||||||
|
|
||||||
|
def job(self, job_id: str) -> Job:
|
||||||
|
"""A [Job][lancedb.job.Job] handle for a server-side job by id.
|
||||||
|
|
||||||
|
The handle is constructed without a server round trip; an unknown id
|
||||||
|
surfaces when the handle is used. Dropping the handle has no effect
|
||||||
|
on the job itself.
|
||||||
|
"""
|
||||||
|
raise NotImplementedError("job is not supported for this connection type")
|
||||||
|
|
||||||
|
def list_jobs(self) -> List[JobInfo]:
|
||||||
|
"""List server-side jobs across the database's tables."""
|
||||||
|
raise NotImplementedError("list_jobs is not supported for this connection type")
|
||||||
|
|
||||||
|
def get_job(self, job_id: str) -> Optional[JobDescription]:
|
||||||
|
"""Describe a single server-side job by id.
|
||||||
|
|
||||||
|
Returns None when the server has no such job.
|
||||||
|
"""
|
||||||
|
raise NotImplementedError("get_job is not supported for this connection type")
|
||||||
|
|
||||||
|
def cancel_job(self, job_id: str) -> bool:
|
||||||
|
"""Request cancellation of a server-side job by id.
|
||||||
|
|
||||||
|
Returns True if the server accepted the cancellation, False if no
|
||||||
|
such job exists. Cancelling an already-terminal job is a no-op
|
||||||
|
success.
|
||||||
|
"""
|
||||||
|
raise NotImplementedError(
|
||||||
|
"cancel_job is not supported for this connection type"
|
||||||
|
)
|
||||||
|
|
||||||
|
def job_history(self, job_id: Optional[str] = None) -> List[pa.RecordBatch]:
|
||||||
|
"""The lifecycle event history of a server-side job, as Arrow batches.
|
||||||
|
|
||||||
|
Lists history across all jobs when `job_id` is None.
|
||||||
|
"""
|
||||||
|
raise NotImplementedError(
|
||||||
|
"job_history is not supported for this connection type"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
class LanceDBConnection(DBConnection):
|
class LanceDBConnection(DBConnection):
|
||||||
"""
|
"""
|
||||||
@@ -620,6 +739,9 @@ class LanceDBConnection(DBConnection):
|
|||||||
self._namespace_client_properties = namespace_client_properties
|
self._namespace_client_properties = namespace_client_properties
|
||||||
if _inner is not None:
|
if _inner is not None:
|
||||||
self._conn = _inner
|
self._conn = _inner
|
||||||
|
# Native-derived wrappers resolve this in their async reconstruction
|
||||||
|
# path so construction never synchronously re-enters LOOP.
|
||||||
|
self._read_consistency_interval = read_consistency_interval
|
||||||
self._cached_namespace_client = None
|
self._cached_namespace_client = None
|
||||||
return
|
return
|
||||||
|
|
||||||
@@ -669,11 +791,14 @@ class LanceDBConnection(DBConnection):
|
|||||||
# storage_options. Also, this class really shouldn't be holding any state
|
# storage_options. Also, this class really shouldn't be holding any state
|
||||||
# beyond _conn.
|
# beyond _conn.
|
||||||
self._conn = AsyncConnection(LOOP.run(do_connect()))
|
self._conn = AsyncConnection(LOOP.run(do_connect()))
|
||||||
|
# Keep property access synchronous so debugger introspection cannot wait on
|
||||||
|
# the background loop while that thread is suspended at a breakpoint.
|
||||||
|
self._read_consistency_interval = read_consistency_interval
|
||||||
self._cached_namespace_client: Optional[LanceNamespace] = None
|
self._cached_namespace_client: Optional[LanceNamespace] = None
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def read_consistency_interval(self) -> Optional[timedelta]:
|
def read_consistency_interval(self) -> Optional[timedelta]:
|
||||||
return LOOP.run(self._conn.get_read_consistency_interval())
|
return self._read_consistency_interval
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def session(self) -> Optional[Session]:
|
def session(self) -> Optional[Session]:
|
||||||
@@ -684,15 +809,19 @@ class LanceDBConnection(DBConnection):
|
|||||||
return self._conn.uri
|
return self._conn.uri
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def from_inner(cls, inner: LanceDbConnection):
|
def from_inner(
|
||||||
return cls(None, _inner=inner)
|
cls,
|
||||||
|
inner: LanceDbConnection,
|
||||||
|
read_consistency_interval: Optional[timedelta],
|
||||||
|
):
|
||||||
|
return cls(
|
||||||
|
None,
|
||||||
|
read_consistency_interval=read_consistency_interval,
|
||||||
|
_inner=inner,
|
||||||
|
)
|
||||||
|
|
||||||
def __repr__(self) -> str:
|
def __repr__(self) -> str:
|
||||||
val = f"{self.__class__.__name__}(uri={self._conn.uri!r}"
|
return f"{self.__class__.__name__}(uri={self._conn.uri!r})"
|
||||||
if self.read_consistency_interval is not None:
|
|
||||||
val += f", read_consistency_interval={repr(self.read_consistency_interval)}"
|
|
||||||
val += ")"
|
|
||||||
return val
|
|
||||||
|
|
||||||
@override
|
@override
|
||||||
def serialize(self) -> str:
|
def serialize(self) -> str:
|
||||||
@@ -1089,6 +1218,20 @@ class LanceDBConnection(DBConnection):
|
|||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|
||||||
|
@override
|
||||||
|
def drop_table_async(
|
||||||
|
self, name: str, namespace_path: Optional[List[str]] = None
|
||||||
|
) -> Job:
|
||||||
|
"""Start dropping a table and return its cleanup job.
|
||||||
|
|
||||||
|
The table may become unavailable before its data files are removed.
|
||||||
|
Call :meth:`Job.wait` to wait for cleanup to finish.
|
||||||
|
"""
|
||||||
|
if namespace_path is None:
|
||||||
|
namespace_path = []
|
||||||
|
job = LOOP.run(self._conn.drop_table_async(name, namespace_path=namespace_path))
|
||||||
|
return Job(job if isinstance(job, AsyncJob) else AsyncJob(job))
|
||||||
|
|
||||||
@override
|
@override
|
||||||
def drop_all_tables(self, namespace_path: Optional[List[str]] = None):
|
def drop_all_tables(self, namespace_path: Optional[List[str]] = None):
|
||||||
if namespace_path is None:
|
if namespace_path is None:
|
||||||
@@ -1129,6 +1272,56 @@ class LanceDBConnection(DBConnection):
|
|||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|
||||||
|
@override
|
||||||
|
def job(self, job_id: str) -> Job:
|
||||||
|
"""A [Job][lancedb.job.Job] handle for a server-side job by id.
|
||||||
|
|
||||||
|
The handle is constructed without a server round trip; an unknown id
|
||||||
|
surfaces when the handle is used. Dropping the handle has no effect
|
||||||
|
on the job itself.
|
||||||
|
"""
|
||||||
|
return Job(self._conn.job(job_id))
|
||||||
|
|
||||||
|
@override
|
||||||
|
def create_function_async(self, definition: UdfDefinition) -> Job[FunctionVersion]:
|
||||||
|
job = LOOP.run(self._conn.create_function_async(definition))
|
||||||
|
return Job(job)
|
||||||
|
|
||||||
|
@override
|
||||||
|
def get_function(self, name: str, *, version: str) -> FunctionVersion:
|
||||||
|
return LOOP.run(self._conn.get_function(name, version=version))
|
||||||
|
|
||||||
|
@override
|
||||||
|
def list_jobs(self) -> List[JobInfo]:
|
||||||
|
"""List server-side jobs across the database's tables."""
|
||||||
|
return LOOP.run(self._conn.list_jobs())
|
||||||
|
|
||||||
|
@override
|
||||||
|
def get_job(self, job_id: str) -> Optional[JobDescription]:
|
||||||
|
"""Describe a single server-side job by id.
|
||||||
|
|
||||||
|
Returns None when the server has no such job.
|
||||||
|
"""
|
||||||
|
return LOOP.run(self._conn.get_job(job_id))
|
||||||
|
|
||||||
|
@override
|
||||||
|
def cancel_job(self, job_id: str) -> bool:
|
||||||
|
"""Request cancellation of a server-side job by id.
|
||||||
|
|
||||||
|
Returns True if the server accepted the cancellation, False if no
|
||||||
|
such job exists. Cancelling an already-terminal job is a no-op
|
||||||
|
success.
|
||||||
|
"""
|
||||||
|
return LOOP.run(self._conn.cancel_job(job_id))
|
||||||
|
|
||||||
|
@override
|
||||||
|
def job_history(self, job_id: Optional[str] = None) -> List[pa.RecordBatch]:
|
||||||
|
"""The lifecycle event history of a server-side job, as Arrow batches.
|
||||||
|
|
||||||
|
Lists history across all jobs when `job_id` is None.
|
||||||
|
"""
|
||||||
|
return LOOP.run(self._conn.job_history(job_id))
|
||||||
|
|
||||||
@override
|
@override
|
||||||
def namespace_client(self) -> LanceNamespace:
|
def namespace_client(self) -> LanceNamespace:
|
||||||
"""Get the equivalent namespace client for this connection.
|
"""Get the equivalent namespace client for this connection.
|
||||||
@@ -1529,7 +1722,7 @@ class AsyncConnection(object):
|
|||||||
|
|
||||||
Data is converted to Arrow before being written to disk. For maximum
|
Data is converted to Arrow before being written to disk. For maximum
|
||||||
control over how data is saved, either provide the PyArrow schema to
|
control over how data is saved, either provide the PyArrow schema to
|
||||||
convert to or else provide a [PyArrow Table](pyarrow.Table) directly.
|
convert to or else provide a [PyArrow Table][pyarrow.Table] directly.
|
||||||
|
|
||||||
>>> import pyarrow as pa
|
>>> import pyarrow as pa
|
||||||
>>> custom_schema = pa.schema([
|
>>> custom_schema = pa.schema([
|
||||||
@@ -1825,6 +2018,23 @@ class AsyncConnection(object):
|
|||||||
if f"Table '{name}' was not found" not in str(e):
|
if f"Table '{name}' was not found" not in str(e):
|
||||||
raise e
|
raise e
|
||||||
|
|
||||||
|
async def drop_table_async(
|
||||||
|
self,
|
||||||
|
name: str,
|
||||||
|
*,
|
||||||
|
namespace_path: Optional[List[str]] = None,
|
||||||
|
) -> AsyncJob:
|
||||||
|
"""Start dropping a table and return its cleanup job.
|
||||||
|
|
||||||
|
The table may become unavailable before its data files are removed.
|
||||||
|
Await :meth:`AsyncJob.wait` to wait for cleanup to finish.
|
||||||
|
"""
|
||||||
|
if namespace_path is None:
|
||||||
|
namespace_path = []
|
||||||
|
return AsyncJob(
|
||||||
|
await self._inner.drop_table_async(name, namespace_path=namespace_path)
|
||||||
|
)
|
||||||
|
|
||||||
async def drop_all_tables(self, namespace_path: Optional[List[str]] = None):
|
async def drop_all_tables(self, namespace_path: Optional[List[str]] = None):
|
||||||
"""Drop all tables from the database.
|
"""Drop all tables from the database.
|
||||||
|
|
||||||
@@ -1838,6 +2048,62 @@ class AsyncConnection(object):
|
|||||||
namespace_path = []
|
namespace_path = []
|
||||||
await self._inner.drop_all_tables(namespace_path=namespace_path)
|
await self._inner.drop_all_tables(namespace_path=namespace_path)
|
||||||
|
|
||||||
|
def job(self, job_id: str) -> AsyncJob:
|
||||||
|
"""An [AsyncJob][lancedb.job.AsyncJob] handle for a server-side job
|
||||||
|
by id.
|
||||||
|
|
||||||
|
The handle is constructed without a server round trip; an unknown id
|
||||||
|
surfaces when the handle is used. Dropping the handle has no effect
|
||||||
|
on the job itself.
|
||||||
|
"""
|
||||||
|
return AsyncJob(self._inner.job(job_id))
|
||||||
|
|
||||||
|
async def create_function_async(
|
||||||
|
self, definition: UdfDefinition
|
||||||
|
) -> AsyncJob[FunctionVersion]:
|
||||||
|
"""Register a scalar Python UDF through the remote Function catalog.
|
||||||
|
|
||||||
|
The returned typed job resolves to the immutable Function version.
|
||||||
|
Local connections raise ``NotImplementedError``.
|
||||||
|
"""
|
||||||
|
if not isinstance(definition, UdfDefinition):
|
||||||
|
raise TypeError("create_function_async requires a @udf definition")
|
||||||
|
inner = await self._inner.create_function_async(
|
||||||
|
definition.registration_request.to_canonical_json()
|
||||||
|
)
|
||||||
|
return _function_job(inner)
|
||||||
|
|
||||||
|
async def get_function(self, name: str, *, version: str) -> FunctionVersion:
|
||||||
|
"""Open one exact immutable Function version from the remote catalog."""
|
||||||
|
return FunctionVersion.from_json(await self._inner.get_function(name, version))
|
||||||
|
|
||||||
|
async def list_jobs(self) -> List[JobInfo]:
|
||||||
|
"""List server-side jobs across the database's tables."""
|
||||||
|
return await self._inner.list_jobs()
|
||||||
|
|
||||||
|
async def get_job(self, job_id: str) -> Optional[JobDescription]:
|
||||||
|
"""Describe a single server-side job by id.
|
||||||
|
|
||||||
|
Returns None when the server has no such job.
|
||||||
|
"""
|
||||||
|
return await self._inner.get_job(job_id)
|
||||||
|
|
||||||
|
async def cancel_job(self, job_id: str) -> bool:
|
||||||
|
"""Request cancellation of a server-side job by id.
|
||||||
|
|
||||||
|
Returns True if the server accepted the cancellation, False if no
|
||||||
|
such job exists. Cancelling an already-terminal job is a no-op
|
||||||
|
success.
|
||||||
|
"""
|
||||||
|
return await self._inner.cancel_job(job_id)
|
||||||
|
|
||||||
|
async def job_history(self, job_id: Optional[str] = None) -> List[pa.RecordBatch]:
|
||||||
|
"""The lifecycle event history of a server-side job, as Arrow batches.
|
||||||
|
|
||||||
|
Lists history across all jobs when `job_id` is None.
|
||||||
|
"""
|
||||||
|
return await self._inner.job_history(job_id)
|
||||||
|
|
||||||
async def namespace_client(self) -> LanceNamespace:
|
async def namespace_client(self) -> LanceNamespace:
|
||||||
"""Get the equivalent namespace client for this connection.
|
"""Get the equivalent namespace client for this connection.
|
||||||
|
|
||||||
|
|||||||
@@ -21,3 +21,32 @@ from .watsonx import WatsonxEmbeddings
|
|||||||
from .voyageai import VoyageAIEmbeddingFunction
|
from .voyageai import VoyageAIEmbeddingFunction
|
||||||
from .colpali import ColPaliEmbeddings
|
from .colpali import ColPaliEmbeddings
|
||||||
from .siglip import SigLipEmbeddings
|
from .siglip import SigLipEmbeddings
|
||||||
|
|
||||||
|
# The API reference renders this package with a single mkdocstrings directive,
|
||||||
|
# which only picks up names listed here. New embedding functions must be added
|
||||||
|
# to both the imports above and this list, or they will silently go undocumented.
|
||||||
|
__all__ = [
|
||||||
|
"EmbeddingFunction",
|
||||||
|
"EmbeddingFunctionConfig",
|
||||||
|
"TextEmbeddingFunction",
|
||||||
|
"EmbeddingFunctionRegistry",
|
||||||
|
"get_registry",
|
||||||
|
"register",
|
||||||
|
"SentenceTransformerEmbeddings",
|
||||||
|
"OpenAIEmbeddings",
|
||||||
|
"OpenClipEmbeddings",
|
||||||
|
"BedRockText",
|
||||||
|
"CohereEmbeddingFunction",
|
||||||
|
"GeminiText",
|
||||||
|
"GteEmbeddings",
|
||||||
|
"InstructorEmbeddingFunction",
|
||||||
|
"JinaEmbeddings",
|
||||||
|
"OllamaEmbeddings",
|
||||||
|
"TransformersEmbeddingFunction",
|
||||||
|
"ColbertEmbeddings",
|
||||||
|
"VoyageAIEmbeddingFunction",
|
||||||
|
"WatsonxEmbeddings",
|
||||||
|
"ColPaliEmbeddings",
|
||||||
|
"ImageBindEmbeddings",
|
||||||
|
"SigLipEmbeddings",
|
||||||
|
]
|
||||||
|
|||||||
@@ -26,7 +26,6 @@ class EmbeddingFunction(BaseModel, ABC):
|
|||||||
3. ndims() which returns the number of dimensions of the vector column
|
3. ndims() which returns the number of dimensions of the vector column
|
||||||
"""
|
"""
|
||||||
|
|
||||||
__slots__ = ("__weakref__",) # pydantic 1.x compatibility
|
|
||||||
max_retries: int = (
|
max_retries: int = (
|
||||||
7 # Setting 0 disables retires. Maybe this should not be enabled by default,
|
7 # Setting 0 disables retires. Maybe this should not be enabled by default,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -7,8 +7,7 @@ from functools import cached_property
|
|||||||
from typing import List, Union
|
from typing import List, Union
|
||||||
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
|
from pydantic import ConfigDict
|
||||||
from lancedb.pydantic import PYDANTIC_VERSION
|
|
||||||
|
|
||||||
from ..util import attempt_import_or_raise
|
from ..util import attempt_import_or_raise
|
||||||
from .base import TextEmbeddingFunction
|
from .base import TextEmbeddingFunction
|
||||||
@@ -21,20 +20,20 @@ class BedRockText(TextEmbeddingFunction):
|
|||||||
"""
|
"""
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
name: str, default "amazon.titan-embed-text-v1"
|
name : str, default "amazon.titan-embed-text-v1"
|
||||||
The model ID of the bedrock model to use. Supported models for are:
|
The model ID of the bedrock model to use. Supported models for are:
|
||||||
- amazon.titan-embed-text-v1
|
- amazon.titan-embed-text-v1
|
||||||
- cohere.embed-english-v3
|
- cohere.embed-english-v3
|
||||||
- cohere.embed-multilingual-v3
|
- cohere.embed-multilingual-v3
|
||||||
region: str, default "us-east-1"
|
region : str, default "us-east-1"
|
||||||
Optional name of the AWS Region in which the service should be called.
|
Optional name of the AWS Region in which the service should be called.
|
||||||
profile_name: str, default None
|
profile_name : str, default None
|
||||||
Optional name of the AWS profile to use for calling the Bedrock service.
|
Optional name of the AWS profile to use for calling the Bedrock service.
|
||||||
If not specified, the default profile will be used.
|
If not specified, the default profile will be used.
|
||||||
assumed_role: str, default None
|
assumed_role : str, default None
|
||||||
Optional ARN of an AWS IAM role to assume for calling the Bedrock service.
|
Optional ARN of an AWS IAM role to assume for calling the Bedrock service.
|
||||||
If not specified, the current active credentials will be used.
|
If not specified, the current active credentials will be used.
|
||||||
role_session_name: str, default "lancedb-embeddings"
|
role_session_name : str, default "lancedb-embeddings"
|
||||||
Optional name of the AWS IAM role session to use for calling the Bedrock
|
Optional name of the AWS IAM role session to use for calling the Bedrock
|
||||||
service. If not specified, "lancedb-embeddings" name will be used.
|
service. If not specified, "lancedb-embeddings" name will be used.
|
||||||
|
|
||||||
@@ -67,13 +66,7 @@ class BedRockText(TextEmbeddingFunction):
|
|||||||
source_input_type: str = "search_document"
|
source_input_type: str = "search_document"
|
||||||
query_input_type: str = "search_query"
|
query_input_type: str = "search_query"
|
||||||
|
|
||||||
if PYDANTIC_VERSION.major < 2: # Pydantic 1.x compat
|
model_config = ConfigDict(ignored_types=(cached_property,))
|
||||||
|
|
||||||
class Config:
|
|
||||||
keep_untouched = (cached_property,)
|
|
||||||
else:
|
|
||||||
model_config = dict()
|
|
||||||
model_config["ignored_types"] = (cached_property,)
|
|
||||||
|
|
||||||
def ndims(self):
|
def ndims(self):
|
||||||
# return len(self._generate_embedding("test"))
|
# return len(self._generate_embedding("test"))
|
||||||
|
|||||||
@@ -22,7 +22,7 @@ class CohereEmbeddingFunction(TextEmbeddingFunction):
|
|||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
name: str, default "embed-multilingual-v2.0"
|
name : str, default "embed-multilingual-v2.0"
|
||||||
The name of the model to use. List of acceptable models:
|
The name of the model to use. List of acceptable models:
|
||||||
|
|
||||||
* embed-english-v3.0
|
* embed-english-v3.0
|
||||||
@@ -33,12 +33,14 @@ class CohereEmbeddingFunction(TextEmbeddingFunction):
|
|||||||
* embed-english-light-v2.0
|
* embed-english-light-v2.0
|
||||||
* embed-multilingual-v2.0
|
* embed-multilingual-v2.0
|
||||||
|
|
||||||
source_input_type: str, default "search_document"
|
source_input_type : str, default "search_document"
|
||||||
The input type for the source column in the database
|
The input type for the source column in the database
|
||||||
|
|
||||||
query_input_type: str, default "search_query"
|
query_input_type : str, default "search_query"
|
||||||
The input type for the query column in the database
|
The input type for the query column in the database
|
||||||
|
|
||||||
|
Notes
|
||||||
|
-----
|
||||||
Cohere supports following input types:
|
Cohere supports following input types:
|
||||||
|
|
||||||
| Input Type | Description |
|
| Input Type | Description |
|
||||||
|
|||||||
@@ -44,7 +44,7 @@ class ColPaliEmbeddings(EmbeddingFunction):
|
|||||||
The token pooling strategy to use, by default "hierarchical".
|
The token pooling strategy to use, by default "hierarchical".
|
||||||
- "hierarchical": Progressively pools tokens to reduce sequence length.
|
- "hierarchical": Progressively pools tokens to reduce sequence length.
|
||||||
- "lambda": A simpler pooling that uses a custom `pooling_func`.
|
- "lambda": A simpler pooling that uses a custom `pooling_func`.
|
||||||
pooling_func: typing.Callable, optional
|
pooling_func : typing.Callable, optional
|
||||||
A function to use for pooling when `pooling_strategy` is "lambda".
|
A function to use for pooling when `pooling_strategy` is "lambda".
|
||||||
pool_factor : int
|
pool_factor : int
|
||||||
Factor to reduce sequence length if token pooling is enabled (default 2).
|
Factor to reduce sequence length if token pooling is enabled (default 2).
|
||||||
@@ -52,7 +52,7 @@ class ColPaliEmbeddings(EmbeddingFunction):
|
|||||||
Quantization configuration for the model. (default None, bitsandbytes needed)
|
Quantization configuration for the model. (default None, bitsandbytes needed)
|
||||||
batch_size : int
|
batch_size : int
|
||||||
Batch size for processing inputs (default 2).
|
Batch size for processing inputs (default 2).
|
||||||
offload_folder: str, optional
|
offload_folder : str, optional
|
||||||
Folder to offload model weights if using CPU offloading (default None). This is
|
Folder to offload model weights if using CPU offloading (default None). This is
|
||||||
useful for large models that do not fit in memory.
|
useful for large models that do not fit in memory.
|
||||||
"""
|
"""
|
||||||
|
|||||||
@@ -7,8 +7,7 @@ from functools import cached_property
|
|||||||
from typing import List, Optional, Union
|
from typing import List, Optional, Union
|
||||||
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
|
from pydantic import ConfigDict
|
||||||
from lancedb.pydantic import PYDANTIC_VERSION
|
|
||||||
|
|
||||||
from ..util import attempt_import_or_raise
|
from ..util import attempt_import_or_raise
|
||||||
from .base import TextEmbeddingFunction
|
from .base import TextEmbeddingFunction
|
||||||
@@ -48,16 +47,16 @@ class GeminiText(TextEmbeddingFunction):
|
|||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
name: str, default "gemini-embedding-001"
|
name : str, default "gemini-embedding-001"
|
||||||
The name of the model to use. Supported models include:
|
The name of the model to use. Supported models include:
|
||||||
- "gemini-embedding-001" (768 dimensions)
|
- "gemini-embedding-001" (768 dimensions)
|
||||||
|
|
||||||
Note: The legacy "models/embedding-001" format is also supported but
|
Note: The legacy "models/embedding-001" format is also supported but
|
||||||
"gemini-embedding-001" is recommended.
|
"gemini-embedding-001" is recommended.
|
||||||
|
|
||||||
query_task_type: str, default "retrieval_query"
|
query_task_type : str, default "retrieval_query"
|
||||||
Sets the task type for the queries.
|
Sets the task type for the queries.
|
||||||
source_task_type: str, default "retrieval_document"
|
source_task_type : str, default "retrieval_document"
|
||||||
Sets the task type for ingestion.
|
Sets the task type for ingestion.
|
||||||
|
|
||||||
Examples
|
Examples
|
||||||
@@ -87,13 +86,7 @@ class GeminiText(TextEmbeddingFunction):
|
|||||||
query_task_type: str = "retrieval_query"
|
query_task_type: str = "retrieval_query"
|
||||||
source_task_type: str = "retrieval_document"
|
source_task_type: str = "retrieval_document"
|
||||||
|
|
||||||
if PYDANTIC_VERSION.major < 2: # Pydantic 1.x compat
|
model_config = ConfigDict(ignored_types=(cached_property,))
|
||||||
|
|
||||||
class Config:
|
|
||||||
keep_untouched = (cached_property,)
|
|
||||||
else:
|
|
||||||
model_config = dict()
|
|
||||||
model_config["ignored_types"] = (cached_property,)
|
|
||||||
|
|
||||||
def ndims(self):
|
def ndims(self):
|
||||||
if self.dim:
|
if self.dim:
|
||||||
|
|||||||
@@ -26,13 +26,13 @@ class GteEmbeddings(TextEmbeddingFunction):
|
|||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
name: str, default "thenlper/gte-large"
|
name : str, default "thenlper/gte-large"
|
||||||
The name of the model to use.
|
The name of the model to use.
|
||||||
device: str, default "cpu"
|
device : str, default "cpu"
|
||||||
Sets the device type for the model.
|
Sets the device type for the model.
|
||||||
normalize: str, default "True"
|
normalize : str, default "True"
|
||||||
Controls normalize param in encode function for the transformer.
|
Controls normalize param in encode function for the transformer.
|
||||||
mlx: bool, default False
|
mlx : bool, default False
|
||||||
Controls which model to use. False for gte-large,True for the mlx version.
|
Controls which model to use. False for gte-large,True for the mlx version.
|
||||||
|
|
||||||
Examples
|
Examples
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user