mirror of
https://github.com/lancedb/lancedb.git
synced 2026-08-27 16:38:31 +00:00
Compare commits
4 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 011def461c | |||
| ed6be12ad6 | |||
| ac2b689cdb | |||
| 4fc8114871 |
@@ -5,3 +5,7 @@ This directory contains repo-scoped code agent skills for the LanceDB project.
|
|||||||
Each skill is a folder that contains a required `SKILL.md` and optional bundled resources.
|
Each skill is a folder that contains a required `SKILL.md` and optional bundled resources.
|
||||||
|
|
||||||
Codex discovers skills from `.agents/skills` in the current working directory and parent directories.
|
Codex discovers skills from `.agents/skills` in the current working directory and parent directories.
|
||||||
|
|
||||||
|
The `lancedb` skill lives in the `plugins/lancedb` plugin (see `plugins/lancedb/skills/lancedb`)
|
||||||
|
so it can be installed via the plugin marketplaces (`.claude-plugin/marketplace.json` and
|
||||||
|
`.agents/plugins/marketplace.json`); the `lancedb` entry here is a symlink into that plugin.
|
||||||
|
|||||||
Symlink
+1
@@ -0,0 +1 @@
|
|||||||
|
../../plugins/lancedb/skills/lancedb
|
||||||
+1
-1
@@ -1,5 +1,5 @@
|
|||||||
[tool.bumpversion]
|
[tool.bumpversion]
|
||||||
current_version = "0.38.0-beta.11"
|
current_version = "0.37.1-beta.0"
|
||||||
parse = """(?x)
|
parse = """(?x)
|
||||||
(?P<major>0|[1-9]\\d*)\\.
|
(?P<major>0|[1-9]\\d*)\\.
|
||||||
(?P<minor>0|[1-9]\\d*)\\.
|
(?P<minor>0|[1-9]\\d*)\\.
|
||||||
|
|||||||
@@ -9,18 +9,6 @@ debug = true
|
|||||||
codegen-units = 16
|
codegen-units = 16
|
||||||
lto = "thin"
|
lto = "thin"
|
||||||
|
|
||||||
[profile.release-no-lto]
|
|
||||||
inherits = "release"
|
|
||||||
debug = true
|
|
||||||
lto = false
|
|
||||||
# Prioritize compile time when LTO is not relevant to the measurement.
|
|
||||||
codegen-units = 16
|
|
||||||
|
|
||||||
[profile.bench]
|
|
||||||
inherits = "release"
|
|
||||||
lto = "thin"
|
|
||||||
codegen-units = 16
|
|
||||||
|
|
||||||
[target.'cfg(all())']
|
[target.'cfg(all())']
|
||||||
rustflags = [
|
rustflags = [
|
||||||
"-Wclippy::all",
|
"-Wclippy::all",
|
||||||
|
|||||||
@@ -17,18 +17,6 @@ updates:
|
|||||||
# newer minimum versions.
|
# newer minimum versions.
|
||||||
versioning-strategy: lockfile-only
|
versioning-strategy: lockfile-only
|
||||||
groups:
|
groups:
|
||||||
# The arrow-rs and datafusion crates are released in lockstep and have to
|
|
||||||
# move together, so keep them in one PR instead of one per sub-crate.
|
|
||||||
# Listed first: a dependency joins the first group it matches.
|
|
||||||
arrow-datafusion:
|
|
||||||
patterns:
|
|
||||||
- arrow
|
|
||||||
- arrow-*
|
|
||||||
- parquet
|
|
||||||
- parquet-*
|
|
||||||
- datafusion
|
|
||||||
- datafusion-*
|
|
||||||
- object_store
|
|
||||||
rust-minor-patch:
|
rust-minor-patch:
|
||||||
update-types:
|
update-types:
|
||||||
- minor
|
- minor
|
||||||
|
|||||||
@@ -1,30 +0,0 @@
|
|||||||
name: CI scripts
|
|
||||||
|
|
||||||
on:
|
|
||||||
push:
|
|
||||||
branches:
|
|
||||||
- main
|
|
||||||
paths:
|
|
||||||
- ci/set_lance_version.py
|
|
||||||
- ci/tests/**
|
|
||||||
- .github/workflows/ci-scripts.yml
|
|
||||||
pull_request:
|
|
||||||
paths:
|
|
||||||
- ci/set_lance_version.py
|
|
||||||
- ci/tests/**
|
|
||||||
- .github/workflows/ci-scripts.yml
|
|
||||||
|
|
||||||
permissions:
|
|
||||||
contents: read
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
test:
|
|
||||||
name: Test CI scripts
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v6
|
|
||||||
- uses: actions/setup-python@v6
|
|
||||||
with:
|
|
||||||
python-version: "3.13"
|
|
||||||
- name: Run tests
|
|
||||||
run: python -m unittest discover -s ci/tests -v
|
|
||||||
@@ -4,14 +4,14 @@ on:
|
|||||||
workflow_call:
|
workflow_call:
|
||||||
inputs:
|
inputs:
|
||||||
tag:
|
tag:
|
||||||
description: "Tag name from Lance (e.g. `v7.2.0-beta.1`). If omitted, the newest release is resolved automatically — stable releases are preferred over pre-releases — and the run is skipped if it is not newer than the version currently pinned in Cargo.toml."
|
description: "Tag name from Lance. If omitted, the skill will use the latest Lance release that needs an update."
|
||||||
required: false
|
required: false
|
||||||
default: ""
|
default: ""
|
||||||
type: string
|
type: string
|
||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
inputs:
|
inputs:
|
||||||
tag:
|
tag:
|
||||||
description: "Tag name from Lance (e.g. `v7.2.0-beta.1`). Leave empty to resolve the newest release automatically — stable releases are preferred over pre-releases — and skip the run if it is not newer than the version currently pinned in Cargo.toml."
|
description: "Tag name from Lance. Leave empty to use the latest Lance release that needs an update."
|
||||||
required: false
|
required: false
|
||||||
default: ""
|
default: ""
|
||||||
type: string
|
type: string
|
||||||
|
|||||||
@@ -1,243 +0,0 @@
|
|||||||
name: Check doc links
|
|
||||||
|
|
||||||
# Checking external links is inherently noisy: third-party sites rate-limit
|
|
||||||
# automated clients, reject non-browser user agents, and go down temporarily.
|
|
||||||
# Blocking pull requests on that trades a lot of false failures for very little
|
|
||||||
# signal, so this runs on a schedule and reports findings in a single tracking
|
|
||||||
# issue instead of failing anyone's build.
|
|
||||||
on:
|
|
||||||
schedule:
|
|
||||||
- cron: "0 7 * * *"
|
|
||||||
workflow_dispatch:
|
|
||||||
|
|
||||||
# The report lives in one repository-global issue, so runs must not overlap: a
|
|
||||||
# lookup racing a create produces duplicate issues, and a healthy run closing
|
|
||||||
# the issue while a failing run only rewrites its body would leave a broken
|
|
||||||
# report closed. The group is deliberately ref-independent so that a manual
|
|
||||||
# dispatch serializes against the scheduled run.
|
|
||||||
concurrency:
|
|
||||||
group: docs-link-check
|
|
||||||
cancel-in-progress: false
|
|
||||||
|
|
||||||
permissions: {}
|
|
||||||
|
|
||||||
env:
|
|
||||||
REPORT_TITLE: "Docs link checker report"
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
scan:
|
|
||||||
name: Scan links
|
|
||||||
runs-on: ubuntu-24.04
|
|
||||||
# lychee-action is pinned by SHA, but its wrapper downloads the lychee
|
|
||||||
# release tarball at run time without verifying a digest, and hands the
|
|
||||||
# resulting binary a GitHub token. Release assets remain replaceable, so
|
|
||||||
# that binary is confined to a job whose token can only read public
|
|
||||||
# content; everything that writes runs in the report job below.
|
|
||||||
permissions:
|
|
||||||
contents: read
|
|
||||||
outputs:
|
|
||||||
checker_outcome: ${{ steps.lychee.outcome }}
|
|
||||||
exit_code: ${{ steps.lychee.outputs.exit_code }}
|
|
||||||
status: ${{ steps.validate.outputs.status }}
|
|
||||||
steps:
|
|
||||||
- name: Checkout
|
|
||||||
uses: actions/checkout@v6
|
|
||||||
with:
|
|
||||||
# workflow_dispatch can run from any ref, but the report is
|
|
||||||
# repository-global. Always measure the default branch so a manual
|
|
||||||
# run from a topic branch cannot close a report that main warrants,
|
|
||||||
# or overwrite it with branch-only findings.
|
|
||||||
ref: ${{ github.event.repository.default_branch }}
|
|
||||||
persist-credentials: false
|
|
||||||
|
|
||||||
- name: Check links
|
|
||||||
id: lychee
|
|
||||||
continue-on-error: true
|
|
||||||
uses: lycheeverse/lychee-action@e7477775783ea5526144ba13e8db5eec57747ce8 # v2.9.0
|
|
||||||
with:
|
|
||||||
# Restricted to http(s) on purpose. Much of docs/src is generated
|
|
||||||
# API reference (the js/ tree comes from `npm run docs` in nodejs)
|
|
||||||
# and the hand-written pages use mkdocstrings cross-references and
|
|
||||||
# nav-relative paths that only resolve in the site mkdocs builds,
|
|
||||||
# not in this checkout, so relative links would be reported as
|
|
||||||
# broken on every run.
|
|
||||||
args: >-
|
|
||||||
--scheme https
|
|
||||||
--scheme http
|
|
||||||
--no-progress
|
|
||||||
--max-retries 3
|
|
||||||
--timeout 20
|
|
||||||
'docs/src/**/*.md'
|
|
||||||
format: json
|
|
||||||
output: ./lychee/out.json
|
|
||||||
jobSummary: false
|
|
||||||
# The report issue, not a red workflow run, is the signal for link
|
|
||||||
# findings and checker failures alike.
|
|
||||||
fail: false
|
|
||||||
|
|
||||||
- name: Validate report
|
|
||||||
id: validate
|
|
||||||
# lychee does not reserve exit code 2 for broken links: its CLI
|
|
||||||
# parser also exits 2 on an invalid option, before any link was
|
|
||||||
# checked or any report written. Only a parseable report whose
|
|
||||||
# counts agree with a completed exit code (0 or 2) counts as a link
|
|
||||||
# verdict. Everything else becomes a checker-error report instead of
|
|
||||||
# failing the workflow. Exit 2 covers timeouts as well as errors, and a
|
|
||||||
# timed-out host is exactly the transient unavailability this report
|
|
||||||
# exists to surface, so both count as findings. Requiring total > 0
|
|
||||||
# also catches a glob that silently stopped matching any file.
|
|
||||||
if: always()
|
|
||||||
env:
|
|
||||||
CHECKER_OUTCOME: ${{ steps.lychee.outcome }}
|
|
||||||
EXIT_CODE: ${{ steps.lychee.outputs.exit_code }}
|
|
||||||
run: |
|
|
||||||
status=checker-error
|
|
||||||
if [[ "$CHECKER_OUTCOME" == success ]] &&
|
|
||||||
[[ "$EXIT_CODE" == 0 || "$EXIT_CODE" == 2 ]] &&
|
|
||||||
jq -e --argjson code "$EXIT_CODE" '
|
|
||||||
(.total > 0) and
|
|
||||||
(if $code == 0
|
|
||||||
then .errors == 0 and .timeouts == 0
|
|
||||||
and (.error_map | length == 0) and (.timeout_map | length == 0)
|
|
||||||
else (.errors + .timeouts) > 0
|
|
||||||
and ((.error_map | length) + (.timeout_map | length)) > 0
|
|
||||||
end)
|
|
||||||
' ./lychee/out.json
|
|
||||||
then
|
|
||||||
if [[ "$EXIT_CODE" == 0 ]]; then
|
|
||||||
status=healthy
|
|
||||||
else
|
|
||||||
status=findings
|
|
||||||
fi
|
|
||||||
fi
|
|
||||||
echo "status=$status" >> "$GITHUB_OUTPUT"
|
|
||||||
echo "Validated link check as $status"
|
|
||||||
|
|
||||||
- name: Upload report
|
|
||||||
if: steps.validate.outputs.status == 'findings'
|
|
||||||
uses: actions/upload-artifact@v7
|
|
||||||
with:
|
|
||||||
name: link-report
|
|
||||||
path: ./lychee/out.json
|
|
||||||
retention-days: 7
|
|
||||||
|
|
||||||
report:
|
|
||||||
name: Update report issue
|
|
||||||
needs: scan
|
|
||||||
runs-on: ubuntu-24.04
|
|
||||||
# Deliberately no checkout: this job needs the report artifact and the
|
|
||||||
# issues API, not the repository contents.
|
|
||||||
permissions:
|
|
||||||
issues: write
|
|
||||||
env:
|
|
||||||
CHECKER_OUTCOME: ${{ needs.scan.outputs.checker_outcome }}
|
|
||||||
EXIT_CODE: ${{ needs.scan.outputs.exit_code }}
|
|
||||||
STATUS: ${{ needs.scan.outputs.status }}
|
|
||||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
|
||||||
steps:
|
|
||||||
- name: Find existing report issue
|
|
||||||
id: report
|
|
||||||
# Matched on title alone, and through search rather than a listing:
|
|
||||||
# the issue action applies labels in a separate call after creating the
|
|
||||||
# issue, so a label filter misses a half-created report, and this
|
|
||||||
# repository has far more open issues than one listing page holds.
|
|
||||||
# Closed issues are included because a healthy run closes the report:
|
|
||||||
# an open-only lookup would forget that identity and the next failing
|
|
||||||
# run would open a duplicate. The oldest match stays the canonical
|
|
||||||
# report and is reopened below when a problem recurs.
|
|
||||||
run: |
|
|
||||||
match=$(gh issue list --repo "$GITHUB_REPOSITORY" --state all \
|
|
||||||
--search "in:title \"$REPORT_TITLE\" author:app/github-actions" \
|
|
||||||
--limit 50 --json number,title,state \
|
|
||||||
--jq "[.[] | select(.title == \"$REPORT_TITLE\")] | sort_by(.number) | first // empty")
|
|
||||||
echo "number=$(jq -r '.number // empty' <<<"$match")" >> "$GITHUB_OUTPUT"
|
|
||||||
echo "state=$(jq -r '.state // empty' <<<"$match")" >> "$GITHUB_OUTPUT"
|
|
||||||
|
|
||||||
- name: Download report
|
|
||||||
if: env.STATUS == 'findings'
|
|
||||||
uses: actions/download-artifact@v8
|
|
||||||
with:
|
|
||||||
name: link-report
|
|
||||||
path: ./lychee
|
|
||||||
|
|
||||||
- name: Compose report
|
|
||||||
if: env.STATUS == 'findings'
|
|
||||||
run: |
|
|
||||||
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
|
|
||||||
{
|
|
||||||
echo "Broken documentation links found by [\`$GITHUB_WORKFLOW\`]($run_url)."
|
|
||||||
echo
|
|
||||||
echo "This issue is rewritten by every scheduled run and closed automatically once all links resolve."
|
|
||||||
echo
|
|
||||||
echo "Entries can be false positives: some sites rate-limit or block automated clients while working fine in a browser. Confirm before editing the docs, and add persistent offenders to \`--exclude\` in \`.github/workflows/docs-link-check.yml\`."
|
|
||||||
echo
|
|
||||||
# Timeouts are reported alongside errors: entries land in
|
|
||||||
# timeout_map with a status text instead of an HTTP code.
|
|
||||||
jq -r '
|
|
||||||
"\(.errors) of \(.total) links failed, \(.timeouts) timed out.",
|
|
||||||
"",
|
|
||||||
([(.error_map | to_entries[]), (.timeout_map | to_entries[])]
|
|
||||||
| group_by(.key)[] |
|
|
||||||
"### Errors in \(.[0].key)",
|
|
||||||
"",
|
|
||||||
(map(.value[])[] | "* [\(.status.code // .status.text // "ERR")] <\(.url)> — \(.status.details // .status.text // "unknown error")"),
|
|
||||||
"")
|
|
||||||
' ./lychee/out.json
|
|
||||||
} > ./lychee/issue.md
|
|
||||||
|
|
||||||
- name: Compose checker error report
|
|
||||||
if: env.STATUS == 'checker-error'
|
|
||||||
run: |
|
|
||||||
mkdir -p ./lychee
|
|
||||||
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
|
|
||||||
{
|
|
||||||
echo "The documentation link check did not complete in [the latest run]($run_url)."
|
|
||||||
echo
|
|
||||||
echo "This issue is rewritten by every scheduled run and closed automatically once a trustworthy run finds that all links resolve."
|
|
||||||
echo
|
|
||||||
echo "The checker did not produce a trustworthy link verdict. Treat the previous result, if any, as stale until a later run completes."
|
|
||||||
echo
|
|
||||||
echo "* Action outcome: \`$CHECKER_OUTCOME\`"
|
|
||||||
echo "* Exit code: \`${EXIT_CODE:-not reported}\`"
|
|
||||||
echo "* Verdict validation: \`failed\`"
|
|
||||||
} > ./lychee/issue.md
|
|
||||||
|
|
||||||
- name: Reopen report issue
|
|
||||||
# A healthy run closes the report, and the issue action below only
|
|
||||||
# rewrites the body of whatever number it is given. Without an
|
|
||||||
# explicit reopen, a later finding or checker error would rewrite a
|
|
||||||
# closed issue. A CLOSED state implies the lookup found a canonical
|
|
||||||
# issue, so no separate emptiness check.
|
|
||||||
if: >-
|
|
||||||
env.STATUS != 'healthy' &&
|
|
||||||
steps.report.outputs.state == 'CLOSED'
|
|
||||||
env:
|
|
||||||
ISSUE_NUMBER: ${{ steps.report.outputs.number }}
|
|
||||||
run: |
|
|
||||||
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
|
|
||||||
gh issue reopen "$ISSUE_NUMBER" --repo "$GITHUB_REPOSITORY" \
|
|
||||||
--comment "The documentation link checker reported a problem again in [the latest run]($run_url)."
|
|
||||||
|
|
||||||
- name: Report link-check problem
|
|
||||||
if: env.STATUS != 'healthy'
|
|
||||||
uses: peter-evans/create-issue-from-file@fca9117c27cdc29c6c4db3b86c48e4115a786710 # v6.0.0
|
|
||||||
with:
|
|
||||||
# Empty on the first failing run, which creates the issue; afterwards
|
|
||||||
# the same issue is updated in place.
|
|
||||||
issue-number: ${{ steps.report.outputs.number }}
|
|
||||||
title: ${{ env.REPORT_TITLE }}
|
|
||||||
content-filepath: ./lychee/issue.md
|
|
||||||
labels: documentation
|
|
||||||
|
|
||||||
- name: Close report issue once links are healthy
|
|
||||||
# An OPEN state implies the lookup found a canonical issue; a report
|
|
||||||
# that is already closed needs nothing.
|
|
||||||
if: >-
|
|
||||||
env.STATUS == 'healthy' &&
|
|
||||||
steps.report.outputs.state == 'OPEN'
|
|
||||||
env:
|
|
||||||
ISSUE_NUMBER: ${{ steps.report.outputs.number }}
|
|
||||||
run: |
|
|
||||||
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
|
|
||||||
gh issue close "$ISSUE_NUMBER" --repo "$GITHUB_REPOSITORY" \
|
|
||||||
--comment "All documentation links resolved in [the latest run]($run_url)."
|
|
||||||
@@ -69,16 +69,6 @@ jobs:
|
|||||||
uses: actions/setup-python@v6
|
uses: actions/setup-python@v6
|
||||||
with:
|
with:
|
||||||
python-version: "3.10"
|
python-version: "3.10"
|
||||||
- name: Add swap for Arm fat LTO
|
|
||||||
if: matrix.config.platform == 'aarch64'
|
|
||||||
shell: bash
|
|
||||||
run: |
|
|
||||||
swap_file="$RUNNER_TEMP/lancedb-swap"
|
|
||||||
sudo fallocate --length 16G "$swap_file"
|
|
||||||
sudo chmod 600 "$swap_file"
|
|
||||||
sudo mkswap "$swap_file"
|
|
||||||
sudo swapon "$swap_file"
|
|
||||||
free -h
|
|
||||||
- uses: ./.github/workflows/build_linux_wheel
|
- uses: ./.github/workflows/build_linux_wheel
|
||||||
with:
|
with:
|
||||||
python-minor-version: 10
|
python-minor-version: 10
|
||||||
@@ -129,12 +119,6 @@ jobs:
|
|||||||
# link.exe is single-threaded and the long pole on Windows builds. Use
|
# link.exe is single-threaded and the long pole on Windows builds. Use
|
||||||
# rustc's bundled lld-link instead.
|
# rustc's bundled lld-link instead.
|
||||||
CARGO_TARGET_X86_64_PC_WINDOWS_MSVC_LINKER: rust-lld
|
CARGO_TARGET_X86_64_PC_WINDOWS_MSVC_LINKER: rust-lld
|
||||||
# Fat LTO of the cdylib is single-threaded and the peak-memory step of the
|
|
||||||
# build. ThinLTO parallelizes it across the runner's cores, at some cost
|
|
||||||
# to runtime performance on our least performance-sensitive platform.
|
|
||||||
# Matches what the nodejs Windows builds already do in npm-publish.yml.
|
|
||||||
CARGO_PROFILE_RELEASE_LTO: thin
|
|
||||||
CARGO_PROFILE_RELEASE_CODEGEN_UNITS: 16
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v6
|
||||||
with:
|
with:
|
||||||
|
|||||||
@@ -229,8 +229,7 @@ jobs:
|
|||||||
# Make sure wheels are not included in the Rust cache
|
# Make sure wheels are not included in the Rust cache
|
||||||
- name: Delete wheels
|
- name: Delete wheels
|
||||||
run: rm -rf target/wheels
|
run: rm -rf target/wheels
|
||||||
min-deps:
|
pydantic1x:
|
||||||
name: "Minimum dependencies"
|
|
||||||
timeout-minutes: 60
|
timeout-minutes: 60
|
||||||
runs-on: "ubuntu-24.04"
|
runs-on: "ubuntu-24.04"
|
||||||
defaults:
|
defaults:
|
||||||
@@ -260,7 +259,8 @@ jobs:
|
|||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
save-if: ${{ github.ref == 'refs/heads/main' }}
|
||||||
- name: Install lancedb
|
- name: Install lancedb
|
||||||
run: |
|
run: |
|
||||||
pip install "pydantic==2.7.4" "pyarrow==16"
|
pip install "pydantic<2"
|
||||||
|
pip install pyarrow==16
|
||||||
pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests]
|
pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests]
|
||||||
- name: Run tests
|
- name: Run tests
|
||||||
run: pytest -m "not slow and not s3_test" -x -v --durations=30 python/tests
|
run: pytest -m "not slow and not s3_test" -x -v --durations=30 python/tests
|
||||||
|
|||||||
+11
-41
@@ -121,6 +121,7 @@ jobs:
|
|||||||
# Need up-to-date compilers for kernels
|
# Need up-to-date compilers for kernels
|
||||||
CC: clang-18
|
CC: clang-18
|
||||||
CXX: clang++-18
|
CXX: clang++-18
|
||||||
|
GH_TOKEN: ${{ secrets.SOPHON_READ_TOKEN }}
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v6
|
||||||
with:
|
with:
|
||||||
@@ -164,40 +165,11 @@ jobs:
|
|||||||
- name: Run feature tests
|
- name: Run feature tests
|
||||||
run: CARGO_ARGS="--profile ci" make -C ./lancedb feature-tests
|
run: CARGO_ARGS="--profile ci" make -C ./lancedb feature-tests
|
||||||
- name: Run examples
|
- name: Run examples
|
||||||
run: cargo run --profile ci --all-features --example simple --locked
|
run: cargo run --profile ci --example simple --locked
|
||||||
|
|
||||||
remote:
|
|
||||||
timeout-minutes: 30
|
|
||||||
# Running this requires access to secrets, so skip if this is a PR from a
|
|
||||||
# fork. Keep it separate from the all-features build so Cargo does not
|
|
||||||
# retain both dependency graphs in one target directory.
|
|
||||||
if: github.event_name != 'pull_request' || !github.event.pull_request.head.repo.fork
|
|
||||||
runs-on: ubuntu-2404-4x-x64
|
|
||||||
defaults:
|
|
||||||
run:
|
|
||||||
shell: bash
|
|
||||||
working-directory: rust
|
|
||||||
env:
|
|
||||||
CC: clang-18
|
|
||||||
CXX: clang++-18
|
|
||||||
GH_TOKEN: ${{ secrets.SOPHON_READ_TOKEN }}
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v6
|
|
||||||
with:
|
|
||||||
fetch-depth: 0
|
|
||||||
lfs: true
|
|
||||||
- uses: Swatinem/rust-cache@v2
|
|
||||||
with:
|
|
||||||
# Remote tests use a different feature graph from the main Linux
|
|
||||||
# job. Cache downloads, but build into a fresh target directory.
|
|
||||||
cache-targets: false
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- name: Install dependencies
|
|
||||||
run: |
|
|
||||||
sudo apt update
|
|
||||||
sudo apt install -y protobuf-compiler libssl-dev
|
|
||||||
- uses: rui314/setup-mold@v1
|
|
||||||
- name: Run remote tests
|
- name: Run remote tests
|
||||||
|
# Running this requires access to secrets, so skip if this is
|
||||||
|
# a PR from a fork.
|
||||||
|
if: github.event_name != 'pull_request' || !github.event.pull_request.head.repo.fork
|
||||||
run: CARGO_ARGS="--profile ci" make -C ./lancedb remote-tests
|
run: CARGO_ARGS="--profile ci" make -C ./lancedb remote-tests
|
||||||
|
|
||||||
macos:
|
macos:
|
||||||
@@ -324,18 +296,16 @@ jobs:
|
|||||||
cargo update -p aws-types --precise 1.3.9
|
cargo update -p aws-types --precise 1.3.9
|
||||||
cargo update -p aws-sigv4 --precise 1.3.5
|
cargo update -p aws-sigv4 --precise 1.3.5
|
||||||
cargo update -p aws-credential-types --precise 1.2.8
|
cargo update -p aws-credential-types --precise 1.2.8
|
||||||
# aws-smithy-checksums must stay at or above 0.63.13: OpenDAL's S3
|
cargo update -p aws-smithy-checksums --precise 0.63.9
|
||||||
# service needs crc-fast ~1.9, and older releases pin it to ~1.3.
|
|
||||||
cargo update -p aws-smithy-checksums --precise 0.63.13
|
|
||||||
cargo update -p aws-smithy-runtime --precise 1.9.3
|
cargo update -p aws-smithy-runtime --precise 1.9.3
|
||||||
cargo update -p aws-smithy-http --precise 0.62.6
|
cargo update -p aws-smithy-http --precise 0.62.4
|
||||||
cargo update -p aws-smithy-eventstream --precise 0.60.14
|
cargo update -p aws-smithy-eventstream --precise 0.60.12
|
||||||
cargo update -p aws-smithy-http-client --precise 1.1.3
|
cargo update -p aws-smithy-http-client --precise 1.1.3
|
||||||
cargo update -p aws-smithy-observability --precise 0.1.4
|
cargo update -p aws-smithy-observability --precise 0.1.4
|
||||||
cargo update -p aws-smithy-query --precise 0.60.8
|
cargo update -p aws-smithy-query --precise 0.60.8
|
||||||
cargo update -p aws-smithy-runtime-api --precise 1.9.3
|
cargo update -p aws-smithy-runtime-api --precise 1.9.1
|
||||||
cargo update -p aws-smithy-async --precise 1.2.7
|
cargo update -p aws-smithy-async --precise 1.2.6
|
||||||
cargo update -p aws-smithy-types --precise 1.3.6
|
cargo update -p aws-smithy-types --precise 1.3.5
|
||||||
cargo update -p aws-smithy-xml --precise 0.60.11
|
cargo update -p aws-smithy-xml --precise 0.60.11
|
||||||
cargo update -p home --precise 0.5.9
|
cargo update -p home --precise 0.5.9
|
||||||
- name: cargo +${{ matrix.msrv }} check
|
- name: cargo +${{ matrix.msrv }} check
|
||||||
|
|||||||
@@ -18,9 +18,6 @@ Common commands:
|
|||||||
* Run specific test: `cargo test --quiet --features remote -p <package_name> --test <test_name>`
|
* Run specific test: `cargo test --quiet --features remote -p <package_name> --test <test_name>`
|
||||||
* Lint: `cargo clippy --quiet --features remote --tests --examples`
|
* Lint: `cargo clippy --quiet --features remote --tests --examples`
|
||||||
* Format Rust: `cargo fmt --all`
|
* Format Rust: `cargo fmt --all`
|
||||||
* Use repository-defined Cargo profiles instead of ad hoc LTO overrides.
|
|
||||||
* Use `release-with-debug` for benchmarks and profiling so optimized builds keep debug symbols without a rebuild.
|
|
||||||
* Use `release-no-lto` only for local debugging, IO-bound benchmarks, or compile-time-sensitive performance investigation where LTO would not affect the measured bottleneck.
|
|
||||||
* Format Python: `ruff format .`
|
* Format Python: `ruff format .`
|
||||||
* Lint Python: `ruff check .`
|
* Lint Python: `ruff check .`
|
||||||
* Bootstrap Python dev env: `cd python && uv run --extra tests --extra dev maturin develop --extras tests,dev`
|
* Bootstrap Python dev env: `cd python && uv run --extra tests --extra dev maturin develop --extras tests,dev`
|
||||||
|
|||||||
Generated
+330
-341
File diff suppressed because it is too large
Load Diff
+16
-23
@@ -13,21 +13,20 @@ categories = ["database-implementations"]
|
|||||||
rust-version = "1.91.0"
|
rust-version = "1.91.0"
|
||||||
|
|
||||||
[workspace.dependencies]
|
[workspace.dependencies]
|
||||||
lance = { "version" = "=12.0.0-beta.2", default-features = false, "tag" = "v12.0.0-beta.2", "git" = "https://github.com/lance-format/lance.git" }
|
lance = { "version" = "=10.0.0-beta.5", default-features = false, "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-core = { "version" = "=12.0.0-beta.2", "tag" = "v12.0.0-beta.2", "git" = "https://github.com/lance-format/lance.git" }
|
lance-core = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-datagen = { "version" = "=12.0.0-beta.2", "tag" = "v12.0.0-beta.2", "git" = "https://github.com/lance-format/lance.git" }
|
lance-datagen = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-file = { "version" = "=12.0.0-beta.2", "tag" = "v12.0.0-beta.2", "git" = "https://github.com/lance-format/lance.git" }
|
lance-file = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-io = { "version" = "=12.0.0-beta.2", default-features = false, "tag" = "v12.0.0-beta.2", "git" = "https://github.com/lance-format/lance.git" }
|
lance-io = { "version" = "=10.0.0-beta.5", default-features = false, "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-index = { "version" = "=12.0.0-beta.2", "tag" = "v12.0.0-beta.2", "git" = "https://github.com/lance-format/lance.git" }
|
lance-index = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-linalg = { "version" = "=12.0.0-beta.2", "tag" = "v12.0.0-beta.2", "git" = "https://github.com/lance-format/lance.git" }
|
lance-linalg = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-namespace = { "version" = "=12.0.0-beta.2", "tag" = "v12.0.0-beta.2", "git" = "https://github.com/lance-format/lance.git" }
|
lance-namespace = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-namespace-impls = { "version" = "=12.0.0-beta.2", default-features = false, "tag" = "v12.0.0-beta.2", "git" = "https://github.com/lance-format/lance.git" }
|
lance-namespace-impls = { "version" = "=10.0.0-beta.5", default-features = false, "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-table = { "version" = "=12.0.0-beta.2", "tag" = "v12.0.0-beta.2", "git" = "https://github.com/lance-format/lance.git" }
|
lance-table = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-testing = { "version" = "=12.0.0-beta.2", "tag" = "v12.0.0-beta.2", "git" = "https://github.com/lance-format/lance.git" }
|
lance-testing = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-datafusion = { "version" = "=12.0.0-beta.2", "tag" = "v12.0.0-beta.2", "git" = "https://github.com/lance-format/lance.git" }
|
lance-datafusion = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-encoding = { "version" = "=12.0.0-beta.2", "tag" = "v12.0.0-beta.2", "git" = "https://github.com/lance-format/lance.git" }
|
lance-encoding = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-arrow = { "version" = "=12.0.0-beta.2", "tag" = "v12.0.0-beta.2", "git" = "https://github.com/lance-format/lance.git" }
|
lance-arrow = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lancedb = { path = "rust/lancedb", default-features = false }
|
|
||||||
ahash = "0.8"
|
ahash = "0.8"
|
||||||
# Note that this one does not include pyarrow
|
# Note that this one does not include pyarrow
|
||||||
arrow = { version = "58.0.0", optional = false }
|
arrow = { version = "58.0.0", optional = false }
|
||||||
@@ -40,7 +39,6 @@ arrow-schema = "58.0.0"
|
|||||||
arrow-select = "58.0.0"
|
arrow-select = "58.0.0"
|
||||||
arrow-cast = "58.0.0"
|
arrow-cast = "58.0.0"
|
||||||
async-trait = "0"
|
async-trait = "0"
|
||||||
bytes = "1"
|
|
||||||
datafusion = { version = "54.0.0", default-features = false }
|
datafusion = { version = "54.0.0", default-features = false }
|
||||||
datafusion-catalog = "54.0.0"
|
datafusion-catalog = "54.0.0"
|
||||||
datafusion-common = { version = "54.0.0", default-features = false }
|
datafusion-common = { version = "54.0.0", default-features = false }
|
||||||
@@ -54,7 +52,7 @@ env_logger = "0.11"
|
|||||||
half = { "version" = "2.7.1", default-features = false, features = [
|
half = { "version" = "2.7.1", default-features = false, features = [
|
||||||
"num-traits",
|
"num-traits",
|
||||||
] }
|
] }
|
||||||
futures = "0.3"
|
futures = "0"
|
||||||
log = "0.4"
|
log = "0.4"
|
||||||
metrics = "0.24"
|
metrics = "0.24"
|
||||||
metrics-util = "0.19"
|
metrics-util = "0.19"
|
||||||
@@ -67,12 +65,7 @@ url = "2"
|
|||||||
num-traits = "0.2"
|
num-traits = "0.2"
|
||||||
regex = "1.10"
|
regex = "1.10"
|
||||||
semver = "1.0.25"
|
semver = "1.0.25"
|
||||||
serde = "1"
|
chrono = "0.4"
|
||||||
serde_json = "1"
|
|
||||||
tempfile = "3.5.0"
|
|
||||||
tokio = { version = "1.23", features = ["rt-multi-thread", "sync"] }
|
|
||||||
uuid = { version = "1.7.0", features = ["v4"] }
|
|
||||||
chrono = { version = "0.4", default-features = false, features = ["clock"] }
|
|
||||||
|
|
||||||
[profile.ci]
|
[profile.ci]
|
||||||
debug = "line-tables-only"
|
debug = "line-tables-only"
|
||||||
|
|||||||
@@ -2,7 +2,6 @@
|
|||||||
Check whether there are any breaking changes in the PRs between the base and head commits.
|
Check whether there are any breaking changes in the PRs between the base and head commits.
|
||||||
If there are, assert that we have incremented the minor version.
|
If there are, assert that we have incremented the minor version.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
import os
|
import os
|
||||||
from packaging.version import parse
|
from packaging.version import parse
|
||||||
@@ -28,7 +27,7 @@ if __name__ == "__main__":
|
|||||||
else:
|
else:
|
||||||
print("No breaking changes found.")
|
print("No breaking changes found.")
|
||||||
exit(0)
|
exit(0)
|
||||||
|
|
||||||
last_stable_version = parse(args.last_stable_version)
|
last_stable_version = parse(args.last_stable_version)
|
||||||
current_version = parse(args.current_version)
|
current_version = parse(args.current_version)
|
||||||
if current_version.minor <= last_stable_version.minor:
|
if current_version.minor <= last_stable_version.minor:
|
||||||
|
|||||||
@@ -1,6 +1,5 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""Determine whether a newer Lance tag exists and expose results for CI."""
|
"""Determine whether a newer Lance tag exists and expose results for CI."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
@@ -37,16 +36,8 @@ class SemVer:
|
|||||||
prerelease: Tuple[Union[int, str], ...]
|
prerelease: Tuple[Union[int, str], ...]
|
||||||
|
|
||||||
def __lt__(self, other: "SemVer") -> bool: # pragma: no cover - simple comparison
|
def __lt__(self, other: "SemVer") -> bool: # pragma: no cover - simple comparison
|
||||||
if (self.major, self.minor, self.patch) != (
|
if (self.major, self.minor, self.patch) != (other.major, other.minor, other.patch):
|
||||||
other.major,
|
return (self.major, self.minor, self.patch) < (other.major, other.minor, other.patch)
|
||||||
other.minor,
|
|
||||||
other.patch,
|
|
||||||
):
|
|
||||||
return (self.major, self.minor, self.patch) < (
|
|
||||||
other.major,
|
|
||||||
other.minor,
|
|
||||||
other.patch,
|
|
||||||
)
|
|
||||||
if self.prerelease == other.prerelease:
|
if self.prerelease == other.prerelease:
|
||||||
return False
|
return False
|
||||||
if not self.prerelease:
|
if not self.prerelease:
|
||||||
@@ -151,9 +142,7 @@ def read_current_version(repo_root: Path) -> str:
|
|||||||
deps = data["workspace"]["dependencies"]
|
deps = data["workspace"]["dependencies"]
|
||||||
entry = deps["lance"]
|
entry = deps["lance"]
|
||||||
except KeyError as exc: # pragma: no cover - configuration guard
|
except KeyError as exc: # pragma: no cover - configuration guard
|
||||||
raise RuntimeError(
|
raise RuntimeError("Failed to locate workspace.dependencies.lance in Cargo.toml") from exc
|
||||||
"Failed to locate workspace.dependencies.lance in Cargo.toml"
|
|
||||||
) from exc
|
|
||||||
|
|
||||||
if isinstance(entry, str):
|
if isinstance(entry, str):
|
||||||
raw_version = entry
|
raw_version = entry
|
||||||
|
|||||||
+6
-9
@@ -1,7 +1,6 @@
|
|||||||
# SPDX-License-Identifier: Apache-2.0
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
"""A zero-dependency mock OpenAI embeddings API endpoint for testing purposes."""
|
"""A zero-dependency mock OpenAI embeddings API endpoint for testing purposes."""
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
import json
|
import json
|
||||||
import http.server
|
import http.server
|
||||||
@@ -23,13 +22,11 @@ class MockOpenAIRequestHandler(http.server.BaseHTTPRequestHandler):
|
|||||||
|
|
||||||
data = []
|
data = []
|
||||||
for i in range(num_inputs):
|
for i in range(num_inputs):
|
||||||
data.append(
|
data.append({
|
||||||
{
|
"object": "embedding",
|
||||||
"object": "embedding",
|
"embedding": [0.1] * 1536,
|
||||||
"embedding": [0.1] * 1536,
|
"index": i,
|
||||||
"index": i,
|
})
|
||||||
}
|
|
||||||
)
|
|
||||||
|
|
||||||
response = {
|
response = {
|
||||||
"object": "list",
|
"object": "list",
|
||||||
@@ -38,7 +35,7 @@ class MockOpenAIRequestHandler(http.server.BaseHTTPRequestHandler):
|
|||||||
"usage": {
|
"usage": {
|
||||||
"prompt_tokens": 0,
|
"prompt_tokens": 0,
|
||||||
"total_tokens": 0,
|
"total_tokens": 0,
|
||||||
},
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
self.send_response(200)
|
self.send_response(200)
|
||||||
|
|||||||
@@ -7,7 +7,6 @@ from packaging.version import parse, InvalidVersion
|
|||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
import argparse
|
import argparse
|
||||||
|
|
||||||
parser = argparse.ArgumentParser()
|
parser = argparse.ArgumentParser()
|
||||||
parser.add_argument("prefix", default="v")
|
parser.add_argument("prefix", default="v")
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|||||||
@@ -22,7 +22,7 @@ def run_command(command: str) -> str:
|
|||||||
def get_latest_stable_version() -> str:
|
def get_latest_stable_version() -> str:
|
||||||
version_line = run_command("cargo info lance | grep '^version:'")
|
version_line = run_command("cargo info lance | grep '^version:'")
|
||||||
# Example output: "version: 0.35.0 (latest 0.37.0)"
|
# Example output: "version: 0.35.0 (latest 0.37.0)"
|
||||||
match = re.search(r"\(latest ([0-9.]+)\)", version_line)
|
match = re.search(r'\(latest ([0-9.]+)\)', version_line)
|
||||||
if match:
|
if match:
|
||||||
return match.group(1)
|
return match.group(1)
|
||||||
# Fallback: use the first version after 'version:'
|
# Fallback: use the first version after 'version:'
|
||||||
@@ -69,7 +69,7 @@ def extract_default_features(line: str) -> bool:
|
|||||||
"""
|
"""
|
||||||
import re
|
import re
|
||||||
|
|
||||||
match = re.search(r"default-features\s*=\s*false", line)
|
match = re.search(r'default-features\s*=\s*false', line)
|
||||||
return match is not None
|
return match is not None
|
||||||
|
|
||||||
|
|
||||||
@@ -104,7 +104,7 @@ def dict_to_toml_line(package_name: str, config: dict) -> str:
|
|||||||
# This shouldn't happen with our current usage
|
# This shouldn't happen with our current usage
|
||||||
parts.append(f'"{key}" = {json.dumps(value)}')
|
parts.append(f'"{key}" = {json.dumps(value)}')
|
||||||
|
|
||||||
return f"{package_name} = {{ {', '.join(parts)} }}\n"
|
return f'{package_name} = {{ {", ".join(parts)} }}\n'
|
||||||
|
|
||||||
|
|
||||||
def update_cargo_toml(line_updater):
|
def update_cargo_toml(line_updater):
|
||||||
@@ -119,7 +119,7 @@ def update_cargo_toml(line_updater):
|
|||||||
lance_line = ""
|
lance_line = ""
|
||||||
is_parsing_lance_line = False
|
is_parsing_lance_line = False
|
||||||
for line in lines:
|
for line in lines:
|
||||||
if re.match(r"^lance(?:\s|[-_])", line):
|
if line.startswith("lance"):
|
||||||
# Check if this is a single-line or multi-line entry
|
# Check if this is a single-line or multi-line entry
|
||||||
# Single-line entries either:
|
# Single-line entries either:
|
||||||
# 1. End with } (complete inline table)
|
# 1. End with } (complete inline table)
|
||||||
|
|||||||
@@ -1,185 +0,0 @@
|
|||||||
import os
|
|
||||||
import stat
|
|
||||||
import subprocess
|
|
||||||
import sys
|
|
||||||
import tempfile
|
|
||||||
import textwrap
|
|
||||||
import unittest
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
|
|
||||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
|
||||||
SCRIPT = REPO_ROOT / "ci" / "set_lance_version.py"
|
|
||||||
LANCE_GIT_URL = "https://github.com/lance-format/lance.git"
|
|
||||||
|
|
||||||
CARGO_TOML = """\
|
|
||||||
[workspace.dependencies]
|
|
||||||
lance = { "version" = "=1.0.0", default-features = false, "features" = ["dynamodb"] }
|
|
||||||
lance-core = "1.0.0"
|
|
||||||
lance_datafusion = {
|
|
||||||
"version" = "=1.0.0",
|
|
||||||
"features" = ["substrait"]
|
|
||||||
}
|
|
||||||
lancedb = { path = "rust/lancedb", default-features = false }
|
|
||||||
lancedb-common = { path = "rust/lancedb-common" }
|
|
||||||
lancewood = "1.0.0"
|
|
||||||
my-lance = "1.0.0"
|
|
||||||
"""
|
|
||||||
|
|
||||||
UNTOUCHED_DEPENDENCIES = """\
|
|
||||||
lancedb = { path = "rust/lancedb", default-features = false }
|
|
||||||
lancedb-common = { path = "rust/lancedb-common" }
|
|
||||||
lancewood = "1.0.0"
|
|
||||||
my-lance = "1.0.0"
|
|
||||||
"""
|
|
||||||
|
|
||||||
|
|
||||||
class SetLanceVersionTest(unittest.TestCase):
|
|
||||||
def test_supported_update_modes_only_rewrite_lance_dependencies(self):
|
|
||||||
cases = {
|
|
||||||
"stable": (
|
|
||||||
"""\
|
|
||||||
lance = { "version" = "=9.9.9", default-features = false, "features" = ["dynamodb"] }
|
|
||||||
lance-core = "=9.9.9"
|
|
||||||
lance_datafusion = { "version" = "=9.9.9", "features" = ["substrait"] }
|
|
||||||
""",
|
|
||||||
["cargo info lance", "cargo metadata"],
|
|
||||||
),
|
|
||||||
"preview": (
|
|
||||||
f"""\
|
|
||||||
lance = {{ "version" = "=10.0.0-beta.3", default-features = false, "features" = ["dynamodb"], "tag" = "v10.0.0-beta.3", "git" = "{LANCE_GIT_URL}" }}
|
|
||||||
lance-core = {{ "version" = "=10.0.0-beta.3", "tag" = "v10.0.0-beta.3", "git" = "{LANCE_GIT_URL}" }}
|
|
||||||
lance_datafusion = {{ "version" = "=10.0.0-beta.3", "features" = ["substrait"], "tag" = "v10.0.0-beta.3", "git" = "{LANCE_GIT_URL}" }}
|
|
||||||
""",
|
|
||||||
["git ls-remote --tags", "cargo metadata"],
|
|
||||||
),
|
|
||||||
"local": (
|
|
||||||
"""\
|
|
||||||
lance = { "path" = "../lance/rust/lance", default-features = false, "features" = ["dynamodb"] }
|
|
||||||
lance-core = { "path" = "../lance/rust/lance-core" }
|
|
||||||
lance_datafusion = { "path" = "../lance/rust/lance_datafusion", "features" = ["substrait"] }
|
|
||||||
""",
|
|
||||||
["cargo metadata"],
|
|
||||||
),
|
|
||||||
"v8.1.2": (
|
|
||||||
"""\
|
|
||||||
lance = { "version" = "=8.1.2", default-features = false, "features" = ["dynamodb"] }
|
|
||||||
lance-core = "=8.1.2"
|
|
||||||
lance_datafusion = { "version" = "=8.1.2", "features" = ["substrait"] }
|
|
||||||
""",
|
|
||||||
["cargo metadata"],
|
|
||||||
),
|
|
||||||
"v8.2.0-beta.4": (
|
|
||||||
f"""\
|
|
||||||
lance = {{ "version" = "=8.2.0-beta.4", default-features = false, "features" = ["dynamodb"], "tag" = "v8.2.0-beta.4", "git" = "{LANCE_GIT_URL}" }}
|
|
||||||
lance-core = {{ "version" = "=8.2.0-beta.4", "tag" = "v8.2.0-beta.4", "git" = "{LANCE_GIT_URL}" }}
|
|
||||||
lance_datafusion = {{ "version" = "=8.2.0-beta.4", "features" = ["substrait"], "tag" = "v8.2.0-beta.4", "git" = "{LANCE_GIT_URL}" }}
|
|
||||||
""",
|
|
||||||
["cargo metadata"],
|
|
||||||
),
|
|
||||||
}
|
|
||||||
|
|
||||||
for version, (updated_dependencies, expected_commands) in cases.items():
|
|
||||||
with self.subTest(version=version), tempfile.TemporaryDirectory() as tmp:
|
|
||||||
workdir = Path(tmp)
|
|
||||||
(workdir / "Cargo.toml").write_text(CARGO_TOML)
|
|
||||||
command_log = workdir / "commands.log"
|
|
||||||
fake_bin = workdir / "bin"
|
|
||||||
fake_bin.mkdir()
|
|
||||||
self._write_fake_executables(fake_bin)
|
|
||||||
self._write_fake_python_dependencies(workdir)
|
|
||||||
|
|
||||||
env = os.environ.copy()
|
|
||||||
env["PATH"] = os.pathsep.join([str(fake_bin), env["PATH"]])
|
|
||||||
env["FAKE_COMMAND_LOG"] = str(command_log)
|
|
||||||
env["PYTHONPATH"] = os.pathsep.join(
|
|
||||||
filter(None, [str(workdir), env.get("PYTHONPATH")])
|
|
||||||
)
|
|
||||||
result = subprocess.run(
|
|
||||||
[sys.executable, str(SCRIPT), version],
|
|
||||||
cwd=workdir,
|
|
||||||
env=env,
|
|
||||||
capture_output=True,
|
|
||||||
text=True,
|
|
||||||
timeout=10,
|
|
||||||
)
|
|
||||||
|
|
||||||
self.assertEqual(result.returncode, 0, result.stderr)
|
|
||||||
self.assertEqual(
|
|
||||||
(workdir / "Cargo.toml").read_text(),
|
|
||||||
"[workspace.dependencies]\n"
|
|
||||||
+ updated_dependencies
|
|
||||||
+ UNTOUCHED_DEPENDENCIES,
|
|
||||||
)
|
|
||||||
commands = command_log.read_text().splitlines()
|
|
||||||
for command in expected_commands:
|
|
||||||
self.assertTrue(
|
|
||||||
any(line.startswith(command) for line in commands),
|
|
||||||
f"{command!r} not found in {commands!r}",
|
|
||||||
)
|
|
||||||
|
|
||||||
def _write_fake_executables(self, fake_bin: Path) -> None:
|
|
||||||
cargo = fake_bin / "cargo"
|
|
||||||
cargo.write_text(
|
|
||||||
textwrap.dedent(
|
|
||||||
"""\
|
|
||||||
#!/bin/sh
|
|
||||||
printf 'cargo %s\\n' "$*" >> "$FAKE_COMMAND_LOG"
|
|
||||||
case "$1" in
|
|
||||||
info)
|
|
||||||
printf '%s\\n' 'version: 8.8.8 (latest 9.9.9)'
|
|
||||||
;;
|
|
||||||
metadata)
|
|
||||||
;;
|
|
||||||
*)
|
|
||||||
exit 2
|
|
||||||
;;
|
|
||||||
esac
|
|
||||||
"""
|
|
||||||
)
|
|
||||||
)
|
|
||||||
cargo.chmod(cargo.stat().st_mode | stat.S_IXUSR)
|
|
||||||
|
|
||||||
git = fake_bin / "git"
|
|
||||||
git.write_text(
|
|
||||||
textwrap.dedent(
|
|
||||||
"""\
|
|
||||||
#!/bin/sh
|
|
||||||
printf 'git %s\\n' "$*" >> "$FAKE_COMMAND_LOG"
|
|
||||||
if [ "$1" != "ls-remote" ]; then
|
|
||||||
exit 2
|
|
||||||
fi
|
|
||||||
printf '%s\\n' \\
|
|
||||||
'111111 refs/tags/v9.9.9' \\
|
|
||||||
'222222 refs/tags/v10.0.0-beta.1' \\
|
|
||||||
'333333 refs/tags/v10.0.0-beta.3'
|
|
||||||
"""
|
|
||||||
)
|
|
||||||
)
|
|
||||||
git.chmod(git.stat().st_mode | stat.S_IXUSR)
|
|
||||||
|
|
||||||
def _write_fake_python_dependencies(self, workdir: Path) -> None:
|
|
||||||
packaging = workdir / "packaging"
|
|
||||||
packaging.mkdir()
|
|
||||||
(packaging / "__init__.py").write_text("")
|
|
||||||
(packaging / "version.py").write_text(
|
|
||||||
textwrap.dedent(
|
|
||||||
"""\
|
|
||||||
class Version:
|
|
||||||
def __init__(self, value):
|
|
||||||
release, _, prerelease = value.partition("-beta.")
|
|
||||||
self._key = (
|
|
||||||
tuple(int(part) for part in release.split(".")),
|
|
||||||
not prerelease,
|
|
||||||
int(prerelease or 0),
|
|
||||||
)
|
|
||||||
|
|
||||||
def __lt__(self, other):
|
|
||||||
return self._key < other._key
|
|
||||||
"""
|
|
||||||
)
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
unittest.main()
|
|
||||||
@@ -12,7 +12,7 @@ with open("Cargo.toml", "rb") as f:
|
|||||||
elif isinstance(dep, dict):
|
elif isinstance(dep, dict):
|
||||||
# Version doesn't have the beta tag in it, so we instead look
|
# Version doesn't have the beta tag in it, so we instead look
|
||||||
# at the git tag.
|
# at the git tag.
|
||||||
version = dep.get("tag", dep.get("version"))
|
version = dep.get('tag', dep.get('version'))
|
||||||
else:
|
else:
|
||||||
raise ValueError("Unexpected type for dependency: " + str(dep))
|
raise ValueError("Unexpected type for dependency: " + str(dep))
|
||||||
|
|
||||||
|
|||||||
@@ -101,19 +101,6 @@ ignore = [
|
|||||||
# https://rustsec.org/advisories/RUSTSEC-2026-0195
|
# https://rustsec.org/advisories/RUSTSEC-2026-0195
|
||||||
{ id = "RUSTSEC-2026-0194", reason = "transitive via inferno/lance/opendal; XML from trusted cloud endpoints, not attacker-controlled" },
|
{ id = "RUSTSEC-2026-0194", reason = "transitive via inferno/lance/opendal; XML from trusted cloud endpoints, not attacker-controlled" },
|
||||||
{ id = "RUSTSEC-2026-0195", reason = "transitive via inferno/lance/opendal; XML from trusted cloud endpoints, not attacker-controlled" },
|
{ id = "RUSTSEC-2026-0195", reason = "transitive via inferno/lance/opendal; XML from trusted cloud endpoints, not attacker-controlled" },
|
||||||
# smartstring: unmaintained — the repository was archived by its author on
|
|
||||||
# 2026-05-03. Not a vulnerability. Reached only transitively through polars
|
|
||||||
# (polars-core/-io/-ops/-time/-utils); nothing in LanceDB depends on it directly.
|
|
||||||
# The advisory states no safe upgrade is available: upstream recommends
|
|
||||||
# compact_str/smol_str, so clearing this requires polars to migrate.
|
|
||||||
# https://rustsec.org/advisories/RUSTSEC-2026-0249
|
|
||||||
{ id = "RUSTSEC-2026-0249", reason = "smartstring unmaintained via polars; no fixed upstream release" },
|
|
||||||
|
|
||||||
# h2 0.3: empty DATA frames can be queued without limit. The patched
|
|
||||||
# h2 0.4 line is locked to 0.4.16, but no patched 0.3 release exists.
|
|
||||||
# The old copy is pulled in by aws-smithy's legacy hyper 0.14 client.
|
|
||||||
# https://rustsec.org/advisories/RUSTSEC-2026-0258
|
|
||||||
{ id = "RUSTSEC-2026-0258", reason = "h2 0.3 via legacy aws-smithy/hyper 0.14; no patched 0.3 release" },
|
|
||||||
]
|
]
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -177,11 +164,6 @@ multiple-versions = "warn"
|
|||||||
# Wildcard version requirements (`foo = "*"`) are a footgun — they let any
|
# Wildcard version requirements (`foo = "*"`) are a footgun — they let any
|
||||||
# future release in without review. Ban them outright.
|
# future release in without review. Ban them outright.
|
||||||
wildcards = "deny"
|
wildcards = "deny"
|
||||||
# Lint every dependency declared by a workspace member against the shared
|
|
||||||
# `[workspace.dependencies]` table: any crate used by more than one member must
|
|
||||||
# go through `workspace = true`, and entries nothing uses are an error. This
|
|
||||||
# keeps versions from drifting between the core crate and the bindings.
|
|
||||||
workspace-dependencies = { duplicates = "deny", unused = "deny" }
|
|
||||||
# Internal workspace crates reference each other via `path = "..."`, which
|
# Internal workspace crates reference each other via `path = "..."`, which
|
||||||
# cargo-deny sees as a wildcard version. That's fine for private workspace
|
# cargo-deny sees as a wildcard version. That's fine for private workspace
|
||||||
# members (not published to crates.io), so allow it specifically for paths.
|
# members (not published to crates.io), so allow it specifically for paths.
|
||||||
|
|||||||
@@ -5,5 +5,5 @@ mkdocs-autorefs>=0.5,<=1.0
|
|||||||
mkdocstrings[python]>=0.24,<1.0
|
mkdocstrings[python]>=0.24,<1.0
|
||||||
griffe>=0.40,<1.0
|
griffe>=0.40,<1.0
|
||||||
mkdocs-render-swagger-plugin>=0.1.0
|
mkdocs-render-swagger-plugin>=0.1.0
|
||||||
pydantic>=2.7.4,<3
|
pydantic>=2.0,<3.0
|
||||||
mkdocs-redirects>=1.2.0
|
mkdocs-redirects>=1.2.0
|
||||||
+1
-33
@@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`:
|
|||||||
<dependency>
|
<dependency>
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-core</artifactId>
|
<artifactId>lancedb-core</artifactId>
|
||||||
<version>0.38.0-beta.11</version>
|
<version>0.37.1-beta.0</version>
|
||||||
</dependency>
|
</dependency>
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -55,38 +55,6 @@ LanceNamespace namespaceClient = LanceDbNamespaceClientBuilder.newBuilder()
|
|||||||
| `region(String)` | AWS region (default: "us-east-1") | No |
|
| `region(String)` | AWS region (default: "us-east-1") | No |
|
||||||
| `config(String, String)` | Additional configuration parameters | No |
|
| `config(String, String)` | Additional configuration parameters | No |
|
||||||
|
|
||||||
### Opening a Table with Vended Credentials
|
|
||||||
|
|
||||||
When the catalog vends temporary object store credentials, open the table through the
|
|
||||||
namespace client. The Lance dataset builder fetches the table location and storage options
|
|
||||||
from the catalog and refreshes the credentials when they expire.
|
|
||||||
|
|
||||||
```java
|
|
||||||
import com.lancedb.LanceDbNamespaceClientBuilder;
|
|
||||||
import org.lance.Dataset;
|
|
||||||
import org.lance.namespace.LanceNamespace;
|
|
||||||
|
|
||||||
import java.util.Arrays;
|
|
||||||
|
|
||||||
LanceNamespace namespaceClient = LanceDbNamespaceClientBuilder.newBuilder()
|
|
||||||
.apiKey(System.getenv("LANCEDB_API_KEY"))
|
|
||||||
.database(System.getenv("LANCEDB_DATABASE"))
|
|
||||||
// Set the endpoint for a LanceDB Enterprise deployment.
|
|
||||||
// .endpoint("https://your-enterprise-endpoint")
|
|
||||||
.build();
|
|
||||||
|
|
||||||
try (Dataset dataset = Dataset.open()
|
|
||||||
.namespaceClient(namespaceClient)
|
|
||||||
.tableId(Arrays.asList("my_namespace", "my_table"))
|
|
||||||
.build()) {
|
|
||||||
System.out.println("Rows: " + dataset.countRows());
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
Do not call `describeTable()` and then open the returned location with `Dataset.open(uri)`.
|
|
||||||
Opening through `namespaceClient()` is what applies the vended storage options and enables
|
|
||||||
automatic credential refresh. No object store credentials need to be passed by the application.
|
|
||||||
|
|
||||||
## Metadata Operations
|
## Metadata Operations
|
||||||
|
|
||||||
### Creating a Namespace Path
|
### Creating a Namespace Path
|
||||||
|
|||||||
@@ -1,518 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / AutoQuery
|
|
||||||
|
|
||||||
# Class: AutoQuery
|
|
||||||
|
|
||||||
A builder for automatic string searches.
|
|
||||||
|
|
||||||
Automatic search determines whether to use full-text or vector search from
|
|
||||||
the table revision selected for each execution. This builder exposes the
|
|
||||||
common operations supported by both query families.
|
|
||||||
|
|
||||||
## Extends
|
|
||||||
|
|
||||||
- `StandardQueryBase`<`NativeQuery` \| `NativeVectorQuery`>
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### inner
|
|
||||||
|
|
||||||
```ts
|
|
||||||
protected inner: Query | VectorQuery | Promise<Query | VectorQuery>;
|
|
||||||
```
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.inner`
|
|
||||||
|
|
||||||
## Methods
|
|
||||||
|
|
||||||
### analyzePlan()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
analyzePlan(distributedMetrics?): Promise<string>
|
|
||||||
```
|
|
||||||
|
|
||||||
Executes the query and returns the physical query plan annotated with runtime metrics.
|
|
||||||
|
|
||||||
This is useful for debugging and performance analysis, as it shows how the query was executed
|
|
||||||
and includes metrics such as elapsed time, rows processed, and I/O statistics.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **distributedMetrics?**: [`AnalyzePlanDistributedMetrics`](../type-aliases/AnalyzePlanDistributedMetrics.md)
|
|
||||||
How distributed worker metrics are displayed for remote query plans.
|
|
||||||
Defaults to `"aggregate"`.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`string`>
|
|
||||||
|
|
||||||
A query execution plan with runtime metrics for each step.
|
|
||||||
|
|
||||||
#### Example
|
|
||||||
|
|
||||||
```ts
|
|
||||||
import * as lancedb from "@lancedb/lancedb"
|
|
||||||
|
|
||||||
const db = await lancedb.connect("./.lancedb");
|
|
||||||
const table = await db.createTable("my_table", [
|
|
||||||
{ vector: [1.1, 0.9], id: "1" },
|
|
||||||
]);
|
|
||||||
|
|
||||||
const plan = await table.query().nearestTo([0.5, 0.2]).analyzePlan();
|
|
||||||
|
|
||||||
Example output (with runtime metrics inlined):
|
|
||||||
AnalyzeExec verbose=true, metrics=[]
|
|
||||||
ProjectionExec: expr=[id@3 as id, vector@0 as vector, _distance@2 as _distance], metrics=[output_rows=1, elapsed_compute=3.292µs]
|
|
||||||
Take: columns="vector, _rowid, _distance, (id)", metrics=[output_rows=1, elapsed_compute=66.001µs, batches_processed=1, bytes_read=8, iops=1, requests=1]
|
|
||||||
CoalesceBatchesExec: target_batch_size=1024, metrics=[output_rows=1, elapsed_compute=3.333µs]
|
|
||||||
GlobalLimitExec: skip=0, fetch=10, metrics=[output_rows=1, elapsed_compute=167ns]
|
|
||||||
FilterExec: _distance@2 IS NOT NULL, metrics=[output_rows=1, elapsed_compute=8.542µs]
|
|
||||||
SortExec: TopK(fetch=10), expr=[_distance@2 ASC NULLS LAST], metrics=[output_rows=1, elapsed_compute=63.25µs, row_replacements=1]
|
|
||||||
KNNVectorDistance: metric=l2, metrics=[output_rows=1, elapsed_compute=114.333µs, output_batches=1]
|
|
||||||
LanceScan: uri=/path/to/data, projection=[vector], row_id=true, row_addr=false, ordered=false, metrics=[output_rows=1, elapsed_compute=103.626µs, bytes_read=549, iops=2, requests=2]
|
|
||||||
```
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.analyzePlan`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### execute()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
protected execute(options?): AsyncGenerator<RecordBatch<any>, void, unknown>
|
|
||||||
```
|
|
||||||
|
|
||||||
Execute the query and return the results as an
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **options?**: `Partial`<[`QueryExecutionOptions`](../interfaces/QueryExecutionOptions.md)>
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`AsyncGenerator`<`RecordBatch`<`any`>, `void`, `unknown`>
|
|
||||||
|
|
||||||
#### See
|
|
||||||
|
|
||||||
- AsyncIterator
|
|
||||||
of
|
|
||||||
- RecordBatch.
|
|
||||||
|
|
||||||
By default, LanceDb will use many threads to calculate results and, when
|
|
||||||
the result set is large, multiple batches will be processed at one time.
|
|
||||||
This readahead is limited however and backpressure will be applied if this
|
|
||||||
stream is consumed slowly (this constrains the maximum memory used by a
|
|
||||||
single query)
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.execute`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### explainPlan()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
explainPlan(verbose): Promise<string>
|
|
||||||
```
|
|
||||||
|
|
||||||
Generates an explanation of the query execution plan.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **verbose**: `boolean` = `false`
|
|
||||||
If true, provides a more detailed explanation. Defaults to false.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`string`>
|
|
||||||
|
|
||||||
A Promise that resolves to a string containing the query execution plan explanation.
|
|
||||||
|
|
||||||
#### Example
|
|
||||||
|
|
||||||
```ts
|
|
||||||
import * as lancedb from "@lancedb/lancedb"
|
|
||||||
const db = await lancedb.connect("./.lancedb");
|
|
||||||
const table = await db.createTable("my_table", [
|
|
||||||
{ vector: [1.1, 0.9], id: "1" },
|
|
||||||
]);
|
|
||||||
const plan = await table.query().nearestTo([0.5, 0.2]).explainPlan();
|
|
||||||
```
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.explainPlan`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### fastSearch()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
fastSearch(): this
|
|
||||||
```
|
|
||||||
|
|
||||||
Skip searching un-indexed data. This can make search faster, but will miss
|
|
||||||
any data that is not yet indexed.
|
|
||||||
|
|
||||||
Use [Table#optimize](Table.md#optimize) to index all un-indexed data.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`this`
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.fastSearch`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### ~~filter()~~
|
|
||||||
|
|
||||||
```ts
|
|
||||||
filter(predicate): this
|
|
||||||
```
|
|
||||||
|
|
||||||
A filter statement to be applied to this query.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **predicate**: `string`
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`this`
|
|
||||||
|
|
||||||
#### See
|
|
||||||
|
|
||||||
where
|
|
||||||
|
|
||||||
#### Deprecated
|
|
||||||
|
|
||||||
Use `where` instead
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.filter`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### fullTextSearch()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
fullTextSearch(query, options?): this
|
|
||||||
```
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **query**: `string` \| [`FullTextQuery`](../interfaces/FullTextQuery.md)
|
|
||||||
|
|
||||||
* **options?**: `Partial`<[`FullTextSearchOptions`](../interfaces/FullTextSearchOptions.md)>
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`this`
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.fullTextSearch`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### limit()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
limit(limit): this
|
|
||||||
```
|
|
||||||
|
|
||||||
Set the maximum number of results to return.
|
|
||||||
|
|
||||||
By default, a plain search has no limit. If this method is not
|
|
||||||
called then every valid row from the table will be returned.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **limit**: `number`
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`this`
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.limit`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### offset()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
offset(offset): this
|
|
||||||
```
|
|
||||||
|
|
||||||
Set the number of rows to skip before returning results.
|
|
||||||
|
|
||||||
This is useful for pagination.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **offset**: `number`
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`this`
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.offset`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### orderBy()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
orderBy(ordering): this
|
|
||||||
```
|
|
||||||
|
|
||||||
Sort the results by the specified column(s).
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **ordering**: [`ColumnOrdering`](../interfaces/ColumnOrdering.md) \| [`ColumnOrdering`](../interfaces/ColumnOrdering.md)[]
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`this`
|
|
||||||
|
|
||||||
This query builder.
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.orderBy`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### outputSchema()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
outputSchema(): Promise<Schema<any>>
|
|
||||||
```
|
|
||||||
|
|
||||||
Returns the schema of the output that will be returned by this query.
|
|
||||||
|
|
||||||
This can be used to inspect the types and names of the columns that will be
|
|
||||||
returned by the query before executing it.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`Schema`<`any`>>
|
|
||||||
|
|
||||||
An Arrow Schema describing the output columns.
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.outputSchema`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### select()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
select(columns): this
|
|
||||||
```
|
|
||||||
|
|
||||||
Return only the specified columns.
|
|
||||||
|
|
||||||
By default a query will return all columns from the table. However, this can have
|
|
||||||
a very significant impact on latency. LanceDb stores data in a columnar fashion. This
|
|
||||||
means we can finely tune our I/O to select exactly the columns we need.
|
|
||||||
|
|
||||||
As a best practice you should always limit queries to the columns that you need. If you
|
|
||||||
pass in an array of column names then only those columns will be returned.
|
|
||||||
|
|
||||||
You can also use this method to create new "dynamic" columns based on your existing columns.
|
|
||||||
For example, you may not care about "a" or "b" but instead simply want "a + b". This is often
|
|
||||||
seen in the SELECT clause of an SQL query (e.g. `SELECT a+b FROM my_table`).
|
|
||||||
|
|
||||||
To create dynamic columns you can pass in a Map<string, string>. A column will be returned
|
|
||||||
for each entry in the map. The key provides the name of the column. The value is
|
|
||||||
an SQL string used to specify how the column is calculated.
|
|
||||||
|
|
||||||
For example, an SQL query might state `SELECT a + b AS combined, c`. The equivalent
|
|
||||||
input to this method would be:
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **columns**: `string` \| `string`[] \| `Record`<`string`, `string`> \| `Map`<`string`, `string`>
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`this`
|
|
||||||
|
|
||||||
#### Example
|
|
||||||
|
|
||||||
```ts
|
|
||||||
new Map([["combined", "a + b"], ["c", "c"]])
|
|
||||||
|
|
||||||
Columns will always be returned in the order given, even if that order is different than
|
|
||||||
the order used when adding the data.
|
|
||||||
|
|
||||||
Note that you can pass in a `Record<string, string>` (e.g. an object literal). This method
|
|
||||||
uses `Object.entries` which should preserve the insertion order of the object. However,
|
|
||||||
object insertion order is easy to get wrong and `Map` is more foolproof.
|
|
||||||
```
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.select`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### toArray()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
toArray(options?): Promise<any[]>
|
|
||||||
```
|
|
||||||
|
|
||||||
Collect the results as an array of objects.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **options?**: `Partial`<[`QueryExecutionOptions`](../interfaces/QueryExecutionOptions.md)>
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`any`[]>
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.toArray`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### toArrow()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
toArrow(options?): Promise<Table<any>>
|
|
||||||
```
|
|
||||||
|
|
||||||
Collect the results as an Arrow
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **options?**: `Partial`<[`QueryExecutionOptions`](../interfaces/QueryExecutionOptions.md)>
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`Table`<`any`>>
|
|
||||||
|
|
||||||
#### See
|
|
||||||
|
|
||||||
ArrowTable.
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.toArrow`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### useLsm()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
useLsm(enable): this
|
|
||||||
```
|
|
||||||
|
|
||||||
Control MemWAL read routing for this query.
|
|
||||||
|
|
||||||
By default (unset), when the table carries a MemWAL write spec (see
|
|
||||||
[Table#setLsmWriteSpec](Table.md#setlsmwritespec)), reads are routed through the LSM scanner so
|
|
||||||
they also return data written via the `mergeInsert` LSM path that has not yet
|
|
||||||
been compacted into the base table (the active/frozen in-memory memtables and
|
|
||||||
the flushed generations), deduplicated by primary key; a table without a spec
|
|
||||||
reads the base table.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **enable**: `boolean`
|
|
||||||
`true` forces the LSM scanner and errors if the table has no
|
|
||||||
MemWAL write spec. `false` bypasses the MemWAL and reads the base table only,
|
|
||||||
even when a spec is present.
|
|
||||||
Note: the LSM scanner does not support every query shape (e.g. reranking,
|
|
||||||
hybrid search, `orderBy`). On a MemWAL table those shapes error unless
|
|
||||||
`useLsm(false)` is set, because a base-only read would silently exclude
|
|
||||||
un-compacted MemWAL data.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`this`
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.useLsm`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### where()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
where(predicate): this
|
|
||||||
```
|
|
||||||
|
|
||||||
A filter statement to be applied to this query.
|
|
||||||
|
|
||||||
The filter should be supplied as an SQL query string. For example:
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **predicate**: `string`
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`this`
|
|
||||||
|
|
||||||
#### Example
|
|
||||||
|
|
||||||
```ts
|
|
||||||
x > 10
|
|
||||||
y > 0 AND y < 100
|
|
||||||
x > 5 OR y = 'test'
|
|
||||||
|
|
||||||
Filtering performance can often be improved by creating a scalar index
|
|
||||||
on the filter column(s).
|
|
||||||
|
|
||||||
Calling this multiple times combines the filters with a logical AND rather
|
|
||||||
than replacing the previous filter.
|
|
||||||
```
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.where`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### withRowId()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
withRowId(): this
|
|
||||||
```
|
|
||||||
|
|
||||||
Whether to return the row id in the results.
|
|
||||||
|
|
||||||
This column can be used to match results between different queries. For
|
|
||||||
example, to match results from a full text search and a vector search in
|
|
||||||
order to perform hybrid search.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`this`
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.withRowId`
|
|
||||||
@@ -37,31 +37,6 @@ latest and stays writable.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### cherryPick()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
cherryPick(fromBranch, dryRun): Promise<CherryPickResult>
|
|
||||||
```
|
|
||||||
|
|
||||||
Cherry-pick a branch onto main.
|
|
||||||
|
|
||||||
Set `dryRun` to `true` to preview. A failed cherry-pick resolves
|
|
||||||
with `status: "failed"` instead of throwing.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **fromBranch**: `string`
|
|
||||||
Branch to cherry-pick from.
|
|
||||||
|
|
||||||
* **dryRun**: `boolean` = `false`
|
|
||||||
When true, only preview. Defaults to false.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`CherryPickResult`](../interfaces/CherryPickResult.md)>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### create()
|
### create()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -137,3 +112,28 @@ List all branches, mapping name to branch metadata.
|
|||||||
#### Returns
|
#### Returns
|
||||||
|
|
||||||
`Promise`<`Record`<`string`, [`BranchContents`](BranchContents.md)>>
|
`Promise`<`Record`<`string`, [`BranchContents`](BranchContents.md)>>
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### merge()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
merge(fromBranch, dryRun): Promise<MergeBranchResult>
|
||||||
|
```
|
||||||
|
|
||||||
|
Merge a branch into main.
|
||||||
|
|
||||||
|
Set `dryRun` to `true` to preview the merge. A rejected merge resolves
|
||||||
|
with `status: "rejected"` instead of throwing.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **fromBranch**: `string`
|
||||||
|
Branch to merge from.
|
||||||
|
|
||||||
|
* **dryRun**: `boolean` = `false`
|
||||||
|
When true, only preview the merge. Defaults to false.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<[`MergeBranchResult`](../interfaces/MergeBranchResult.md)>
|
||||||
|
|||||||
@@ -25,27 +25,6 @@ the underlying connection has been closed.
|
|||||||
|
|
||||||
## Methods
|
## Methods
|
||||||
|
|
||||||
### cancelJob()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract cancelJob(jobId): Promise<boolean>
|
|
||||||
```
|
|
||||||
|
|
||||||
Request cancellation of a server-side job by id.
|
|
||||||
|
|
||||||
Resolves to true if the server accepted the cancellation, false if no
|
|
||||||
such job exists. Cancelling an already-terminal job is a no-op success.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **jobId**: `string`
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`boolean`>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### cloneTable()
|
### cloneTable()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -169,45 +148,6 @@ Creates a new empty Table
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### createMaterializedView()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract createMaterializedView(
|
|
||||||
name,
|
|
||||||
source,
|
|
||||||
options?): Promise<MaterializedView>
|
|
||||||
```
|
|
||||||
|
|
||||||
Define a materialized view named `name` over the table `source`.
|
|
||||||
|
|
||||||
The view is created empty, with the query recorded in its schema
|
|
||||||
metadata; `view.refresh()` computes the rows. The view is a normal
|
|
||||||
table: it can be queried, indexed and searched, and it appears in
|
|
||||||
`tableNames`. The source table must have stable row ids (create it with
|
|
||||||
the `newTableEnableStableRowIds` storage option); they keep the view's
|
|
||||||
provenance valid across source compactions and cannot be enabled after
|
|
||||||
a table exists. Local databases only.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **name**: `string`
|
|
||||||
|
|
||||||
* **source**: `string`
|
|
||||||
|
|
||||||
* **options?**
|
|
||||||
|
|
||||||
* **options.limit?**: `number`
|
|
||||||
|
|
||||||
* **options.select?**: [`MaterializedViewSelect`](../type-aliases/MaterializedViewSelect.md)
|
|
||||||
|
|
||||||
* **options.where?**: `string`
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`MaterializedView`](MaterializedView.md)>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### createNamespace()
|
### createNamespace()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -425,49 +365,6 @@ Drop an existing table.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### dropTableAsync()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract dropTableAsync(name, namespacePath?): Promise<Job>
|
|
||||||
```
|
|
||||||
|
|
||||||
Start dropping a table and return its cleanup job.
|
|
||||||
|
|
||||||
The table may become unavailable before its data files are removed. Wait
|
|
||||||
on the returned job to know when cleanup has finished.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **name**: `string`
|
|
||||||
|
|
||||||
* **namespacePath?**: `string`[]
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`Job`](Job.md)>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### getJob()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract getJob(jobId): Promise<null | JobDescription>
|
|
||||||
```
|
|
||||||
|
|
||||||
Describe a single server-side job by id.
|
|
||||||
|
|
||||||
Resolves to `null` when the server has no such job.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **jobId**: `string`
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`null` \| [`JobDescription`](../interfaces/JobDescription.md)>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### isOpen()
|
### isOpen()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -482,78 +379,6 @@ Return true if the connection has not been closed
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### job()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract job(jobId): Job
|
|
||||||
```
|
|
||||||
|
|
||||||
A [Job](Job.md) handle for a server-side job by id.
|
|
||||||
|
|
||||||
The handle is constructed without a server round trip; an unknown id
|
|
||||||
surfaces when the handle is used. Dropping the handle has no effect on
|
|
||||||
the job itself.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **jobId**: `string`
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
[`Job`](Job.md)
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### jobHistory()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract jobHistory(jobId?): Promise<Table<any>>
|
|
||||||
```
|
|
||||||
|
|
||||||
The lifecycle event history of a server-side job, as an Arrow table.
|
|
||||||
|
|
||||||
Lists history across all jobs when `jobId` is omitted.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **jobId?**: `string`
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`Table`<`any`>>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### listJobs()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract listJobs(): Promise<JobInfo[]>
|
|
||||||
```
|
|
||||||
|
|
||||||
List server-side jobs across the database's tables.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`JobInfo`](../interfaces/JobInfo.md)[]>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### listMaterializedViews()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract listMaterializedViews(): Promise<string[]>
|
|
||||||
```
|
|
||||||
|
|
||||||
The names of the materialized views in this database.
|
|
||||||
|
|
||||||
Found by reading every table's schema, so this costs an open per table.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`string`[]>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### listNamespaces()
|
### listNamespaces()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -584,90 +409,6 @@ Child namespace names and
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### listTables()
|
|
||||||
|
|
||||||
#### listTables(options)
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract listTables(options?): Promise<ListTablesResponse>
|
|
||||||
```
|
|
||||||
|
|
||||||
List a page of the tables in this database.
|
|
||||||
|
|
||||||
To retrieve the tables after the page, pass the `pageToken` the response
|
|
||||||
carries back in. A page can be shorter than `limit` without being the last
|
|
||||||
one, so walk until a response carries no page token:
|
|
||||||
|
|
||||||
```ts
|
|
||||||
const names = [];
|
|
||||||
let pageToken = undefined;
|
|
||||||
do {
|
|
||||||
const page = await conn.listTables({ pageToken, limit: 100 });
|
|
||||||
names.push(...page.tables);
|
|
||||||
pageToken = page.pageToken;
|
|
||||||
} while (pageToken);
|
|
||||||
```
|
|
||||||
|
|
||||||
##### Parameters
|
|
||||||
|
|
||||||
* **options?**: `Partial`<[`ListTablesOptions`](../interfaces/ListTablesOptions.md)>
|
|
||||||
Pagination options
|
|
||||||
(`pageToken`, `limit`).
|
|
||||||
|
|
||||||
##### Returns
|
|
||||||
|
|
||||||
`Promise`<[`ListTablesResponse`](../interfaces/ListTablesResponse.md)>
|
|
||||||
|
|
||||||
A page of table names and an
|
|
||||||
optional token for the tables after it.
|
|
||||||
|
|
||||||
#### listTables(namespacePath, options)
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract listTables(namespacePath?, options?): Promise<ListTablesResponse>
|
|
||||||
```
|
|
||||||
|
|
||||||
List a page of the tables in this database.
|
|
||||||
|
|
||||||
##### Parameters
|
|
||||||
|
|
||||||
* **namespacePath?**: `string`[]
|
|
||||||
The namespace path to list tables from
|
|
||||||
(defaults to root namespace)
|
|
||||||
|
|
||||||
* **options?**: `Partial`<[`ListTablesOptions`](../interfaces/ListTablesOptions.md)>
|
|
||||||
Pagination options
|
|
||||||
(`pageToken`, `limit`).
|
|
||||||
|
|
||||||
##### Returns
|
|
||||||
|
|
||||||
`Promise`<[`ListTablesResponse`](../interfaces/ListTablesResponse.md)>
|
|
||||||
|
|
||||||
A page of table names and an
|
|
||||||
optional token for the tables after it.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### openMaterializedView()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract openMaterializedView(name): Promise<MaterializedView>
|
|
||||||
```
|
|
||||||
|
|
||||||
Open the materialized view named `name`.
|
|
||||||
|
|
||||||
Rejects a table that exists but is not a materialized view.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **name**: `string`
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`MaterializedView`](MaterializedView.md)>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### openTable()
|
### openTable()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -677,13 +418,18 @@ abstract openTable(
|
|||||||
options?): Promise<Table>
|
options?): Promise<Table>
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Open a table in the database.
|
||||||
|
|
||||||
#### Parameters
|
#### Parameters
|
||||||
|
|
||||||
* **name**: `string`
|
* **name**: `string`
|
||||||
|
The name of the table
|
||||||
|
|
||||||
* **namespacePath?**: `string`[]
|
* **namespacePath?**: `string`[]
|
||||||
|
The namespace path of the table (defaults to root namespace)
|
||||||
|
|
||||||
* **options?**: `Partial`<[`OpenTableOptions`](../interfaces/OpenTableOptions.md)>
|
* **options?**: `Partial`<[`OpenTableOptions`](../interfaces/OpenTableOptions.md)>
|
||||||
|
Additional options
|
||||||
|
|
||||||
#### Returns
|
#### Returns
|
||||||
|
|
||||||
@@ -724,7 +470,7 @@ a "not supported" error.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### ~~tableNames()~~
|
### tableNames()
|
||||||
|
|
||||||
#### tableNames(options)
|
#### tableNames(options)
|
||||||
|
|
||||||
@@ -746,10 +492,6 @@ Tables will be returned in lexicographical order.
|
|||||||
|
|
||||||
`Promise`<`string`[]>
|
`Promise`<`string`[]>
|
||||||
|
|
||||||
##### Deprecated
|
|
||||||
|
|
||||||
Use [Connection.listTables](Connection.md#listtables) instead.
|
|
||||||
|
|
||||||
#### tableNames(namespacePath, options)
|
#### tableNames(namespacePath, options)
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -772,7 +514,3 @@ Tables will be returned in lexicographical order.
|
|||||||
##### Returns
|
##### Returns
|
||||||
|
|
||||||
`Promise`<`string`[]>
|
`Promise`<`string`[]>
|
||||||
|
|
||||||
##### Deprecated
|
|
||||||
|
|
||||||
Use [Connection.listTables](Connection.md#listtables) instead.
|
|
||||||
|
|||||||
@@ -1,83 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / Job
|
|
||||||
|
|
||||||
# Class: Job
|
|
||||||
|
|
||||||
A handle to an operation that may still be running.
|
|
||||||
|
|
||||||
## Constructors
|
|
||||||
|
|
||||||
### new Job()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
new Job(): Job
|
|
||||||
```
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
[`Job`](Job.md)
|
|
||||||
|
|
||||||
## Accessors
|
|
||||||
|
|
||||||
### id
|
|
||||||
|
|
||||||
```ts
|
|
||||||
get id(): null | string
|
|
||||||
```
|
|
||||||
|
|
||||||
Identifies the operation on the server that is running it. Operations
|
|
||||||
that run in this process have no server id. The value is opaque.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`null` \| `string`
|
|
||||||
|
|
||||||
## Methods
|
|
||||||
|
|
||||||
### cancel()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
cancel(): Promise<void>
|
|
||||||
```
|
|
||||||
|
|
||||||
Request cancellation. Cancelling a finished operation is a no-op.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`void`>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### status()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
status(): Promise<string>
|
|
||||||
```
|
|
||||||
|
|
||||||
The operation's current lifecycle state: "running", "finished",
|
|
||||||
"failed", or "cancelled".
|
|
||||||
|
|
||||||
A point snapshot; unlike [Job.wait](Job.md#wait) it does not block or reject
|
|
||||||
on a terminal failure state. States a newer server reports that this
|
|
||||||
client version does not know pass through as-is.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`string`>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### wait()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
wait(): Promise<void>
|
|
||||||
```
|
|
||||||
|
|
||||||
Wait until the operation reaches a terminal state.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`void`>
|
|
||||||
@@ -1,101 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / MaterializedView
|
|
||||||
|
|
||||||
# Class: MaterializedView
|
|
||||||
|
|
||||||
A handle on a materialized view: its table plus its definition.
|
|
||||||
|
|
||||||
Obtained from [Connection#createMaterializedView](Connection.md#creatematerializedview) or
|
|
||||||
[Connection#openMaterializedView](Connection.md#openmaterializedview). The view is a normal table --
|
|
||||||
queries, indexes and search all apply through [MaterializedView#table](MaterializedView.md#table)
|
|
||||||
-- whose contents are maintained by [MaterializedView#refresh](MaterializedView.md#refresh).
|
|
||||||
|
|
||||||
## Constructors
|
|
||||||
|
|
||||||
### new MaterializedView()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
new MaterializedView(table): MaterializedView
|
|
||||||
```
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **table**: [`Table`](Table.md)
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
[`MaterializedView`](MaterializedView.md)
|
|
||||||
|
|
||||||
## Accessors
|
|
||||||
|
|
||||||
### name
|
|
||||||
|
|
||||||
```ts
|
|
||||||
get name(): string
|
|
||||||
```
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`string`
|
|
||||||
|
|
||||||
## Methods
|
|
||||||
|
|
||||||
### definition()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
definition(): Promise<MaterializedViewDefinition>
|
|
||||||
```
|
|
||||||
|
|
||||||
The query that defines the view, read from its stored schema.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`MaterializedViewDefinition`](../interfaces/MaterializedViewDefinition.md)>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### refresh()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
refresh(options?): Promise<RefreshMaterializedViewResult>
|
|
||||||
```
|
|
||||||
|
|
||||||
Recompute the view from its source.
|
|
||||||
|
|
||||||
The refresh is incremental when the source's changes can be reconciled
|
|
||||||
into the view -- rows added, changed or removed since the last one --
|
|
||||||
and otherwise rebuilds. `full` forces a rebuild; `sourceVersion`
|
|
||||||
refreshes to that source version instead of the latest.
|
|
||||||
|
|
||||||
Concurrent refreshes of one view do not duplicate its rows. Two that
|
|
||||||
plan the same source rows conflict on commit, and the loser throws
|
|
||||||
rather than writing them a second time.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **options?**
|
|
||||||
|
|
||||||
* **options.full?**: `boolean`
|
|
||||||
|
|
||||||
* **options.sourceVersion?**: `number`
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`RefreshMaterializedViewResult`](../interfaces/RefreshMaterializedViewResult.md)>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### table()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
table(): Table
|
|
||||||
```
|
|
||||||
|
|
||||||
The view, as the table it is.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
[`Table`](Table.md)
|
|
||||||
@@ -69,34 +69,14 @@ abstract addColumns(newColumnTransforms): Promise<AddColumnsResult>
|
|||||||
|
|
||||||
Add new columns with defined values.
|
Add new columns with defined values.
|
||||||
|
|
||||||
The `{ computed }` form stores the expression rather than evaluating it
|
|
||||||
now: the column is committed with no values, and rows get them from
|
|
||||||
[Table#refreshColumn](Table.md#refreshcolumn). Declaring one therefore costs the same on a
|
|
||||||
large table as on an empty one.
|
|
||||||
|
|
||||||
A refresh does not revisit rows it has already filled, so mutating an
|
|
||||||
input leaves the value computed at fill time; recomputing means dropping
|
|
||||||
the column and declaring it again. While a declaration reads a column,
|
|
||||||
that column cannot be renamed, retyped or dropped.
|
|
||||||
|
|
||||||
On LanceDB Cloud and Enterprise the expression is planned by the
|
|
||||||
server, and the refresh runs as a server job -- see
|
|
||||||
[Table#refreshColumnAsync](Table.md#refreshcolumnasync).
|
|
||||||
|
|
||||||
#### Parameters
|
#### Parameters
|
||||||
|
|
||||||
* **newColumnTransforms**:
|
* **newColumnTransforms**: `Field`<`any`> \| `Field`<`any`>[] \| `Schema`<`any`> \| [`AddColumnsSql`](../interfaces/AddColumnsSql.md)[]
|
||||||
\| `Field`<`any`>
|
|
||||||
\| `Field`<`any`>[]
|
|
||||||
\| `Schema`<`any`>
|
|
||||||
\| [`AddColumnsSql`](../interfaces/AddColumnsSql.md)[]
|
|
||||||
\| `object`
|
|
||||||
Either:
|
Either:
|
||||||
- An array of objects with column names and SQL expressions to calculate values
|
- An array of objects with column names and SQL expressions to calculate values
|
||||||
- A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
- A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
||||||
- An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
- An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
||||||
- An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
- An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
||||||
- `{ computed }`, declaring columns defined by a SQL expression whose type and inputs are derived from it
|
|
||||||
|
|
||||||
#### Returns
|
#### Returns
|
||||||
|
|
||||||
@@ -105,13 +85,6 @@ server, and the refresh runs as a server job -- see
|
|||||||
A promise that resolves to an object
|
A promise that resolves to an object
|
||||||
containing the new version number of the table after adding the columns.
|
containing the new version number of the table after adding the columns.
|
||||||
|
|
||||||
#### Example
|
|
||||||
|
|
||||||
```ts
|
|
||||||
await table.addColumns({ computed: [{ name: "doubled", valueSql: "x * 2" }] });
|
|
||||||
const { rowsFilled } = await table.refreshColumn("doubled");
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### alterColumns()
|
### alterColumns()
|
||||||
@@ -213,39 +186,6 @@ version of the table.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### checkpointLsm()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract checkpointLsm(): Promise<void>
|
|
||||||
```
|
|
||||||
|
|
||||||
Converge this table's LSM write path into its base table.
|
|
||||||
|
|
||||||
Seals once, then triggers compaction and polls until the L0 that existed
|
|
||||||
at the start is gone. The target set is fixed at the start, so
|
|
||||||
generations created *during* the checkpoint are ignored — that is what
|
|
||||||
lets it terminate under write load, and what makes it best-effort: it
|
|
||||||
converges the fresh tier as of some instant. Idempotent, abandonable at
|
|
||||||
any point, and safe to run on a cadence.
|
|
||||||
|
|
||||||
There is no liveness bound — the compactor pool is shared across tables,
|
|
||||||
so a checkpoint queued behind unrelated work looks exactly like one that
|
|
||||||
is merging. The caller owns the deadline.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`void`>
|
|
||||||
|
|
||||||
#### Example
|
|
||||||
|
|
||||||
```ts
|
|
||||||
const before = await table.getLsmStats();
|
|
||||||
await table.checkpointLsm();
|
|
||||||
const after = await table.getLsmStats();
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### close()
|
### close()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -283,24 +223,6 @@ It is a no-op when no writers are cached.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### compactLsm()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract compactLsm(): Promise<void>
|
|
||||||
```
|
|
||||||
|
|
||||||
Trigger a background L0 → base compaction pass per bucket.
|
|
||||||
|
|
||||||
Returns once the passes are *dispatched*, not once they finish — watch
|
|
||||||
[Table#getLsmStats](Table.md#getlsmstats) for progress, or use
|
|
||||||
[Table#checkpointLsm](Table.md#checkpointlsm) to wait for convergence.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`void`>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### countRows()
|
### countRows()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -373,29 +295,6 @@ await table.createIndex("my_float_col");
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### createIndexAsync()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract createIndexAsync(column, options?): Promise<Job>
|
|
||||||
```
|
|
||||||
|
|
||||||
Create an index, returning a handle to the indexing job.
|
|
||||||
|
|
||||||
The job may already be complete when returned; callers must not assume
|
|
||||||
the index exists until [Job.wait](Job.md#wait) resolves.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **column**: `string`
|
|
||||||
|
|
||||||
* **options?**: `Partial`<[`IndexOptions`](../interfaces/IndexOptions.md)>
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`Job`](Job.md)>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### currentBranch()
|
### currentBranch()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -499,48 +398,6 @@ Drop an index from the table.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### flushLsm()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract flushLsm(): Promise<void>
|
|
||||||
```
|
|
||||||
|
|
||||||
Seal every bucket's active memtable into a new L0 generation.
|
|
||||||
|
|
||||||
Returns once the seal is committed. Sealing an empty memtable is a no-op,
|
|
||||||
so this is safe to call repeatedly.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`void`>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### getLsmStats()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract getLsmStats(includeGenerationRows?): Promise<undefined | LsmStats>
|
|
||||||
```
|
|
||||||
|
|
||||||
Read live per-bucket LSM state.
|
|
||||||
|
|
||||||
Answers "how far behind is my fresh tier", "which bucket is hot", and
|
|
||||||
"why is my fresh-tier vector search brute-force". Mutates no table state.
|
|
||||||
|
|
||||||
Resolves to `undefined` only when the LSM write path is not enabled.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **includeGenerationRows?**: `boolean`
|
|
||||||
Also count rows per L0 generation.
|
|
||||||
Off by default because each count opens an uncached Lance dataset.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`undefined` \| [`LsmStats`](../interfaces/LsmStats.md)>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### getLsmWriteSpec()
|
### getLsmWriteSpec()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -551,10 +408,9 @@ Read the [LsmWriteSpec](../interfaces/LsmWriteSpec.md) currently installed on th
|
|||||||
|
|
||||||
Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
|
Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
|
||||||
spec has been set, or it was removed with [Table#unsetLsmWriteSpec](Table.md#unsetlsmwritespec)).
|
spec has been set, or it was removed with [Table#unsetLsmWriteSpec](Table.md#unsetlsmwritespec)).
|
||||||
The returned spec mirrors what was passed to
|
The returned spec — including its `maintainedIndexes` and
|
||||||
[Table#setLsmWriteSpec](Table.md#setlsmwritespec), except that `maintainedIndexes` always
|
`writerConfigDefaults` — mirrors what was passed to
|
||||||
reports the concrete list resolved when the spec was set — `undefined`
|
[Table#setLsmWriteSpec](Table.md#setlsmwritespec).
|
||||||
never round-trips.
|
|
||||||
|
|
||||||
#### Returns
|
#### Returns
|
||||||
|
|
||||||
@@ -838,67 +694,6 @@ for await (const batch of table.query()) {
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### refreshColumn()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract refreshColumn(column): Promise<RefreshColumnResult>
|
|
||||||
```
|
|
||||||
|
|
||||||
Fill the rows of a computed column that hold no value yet.
|
|
||||||
|
|
||||||
Rows appended since the last refresh are filled by the next one; rows
|
|
||||||
already filled are left as they are, so the call is idempotent and does
|
|
||||||
not observe a mutated input. Local tables only: a remote refresh runs
|
|
||||||
as a server job, through [Table#refreshColumnAsync](Table.md#refreshcolumnasync).
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **column**: `string`
|
|
||||||
The name of the computed column to fill.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`RefreshColumnResult`](../interfaces/RefreshColumnResult.md)>
|
|
||||||
|
|
||||||
A promise that resolves to the
|
|
||||||
number of rows filled and the new version number of the table.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### refreshColumnAsync()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract refreshColumnAsync(column): Promise<Job>
|
|
||||||
```
|
|
||||||
|
|
||||||
Like [Table#refreshColumn](Table.md#refreshcolumn), but returns a handle to the refresh
|
|
||||||
job instead of blocking until it completes.
|
|
||||||
|
|
||||||
The job may already be complete when returned; callers must not assume
|
|
||||||
the column is filled until [Job.wait](Job.md#wait) resolves. Invalid input --
|
|
||||||
an unknown column, or one that is not computed -- rejects here rather
|
|
||||||
than failing the job. On local tables the job runs in-process; on
|
|
||||||
LanceDB Cloud and Enterprise it is the server's backfill job.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **column**: `string`
|
|
||||||
The name of the computed column to fill.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`Job`](Job.md)>
|
|
||||||
|
|
||||||
#### Example
|
|
||||||
|
|
||||||
```ts
|
|
||||||
const job = await table.refreshColumnAsync("doubled");
|
|
||||||
await job.wait();
|
|
||||||
console.log(await job.status()); // "finished"
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### restore()
|
### restore()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -942,7 +737,7 @@ Get the schema of the table.
|
|||||||
abstract search(
|
abstract search(
|
||||||
query,
|
query,
|
||||||
queryType?,
|
queryType?,
|
||||||
ftsColumns?): Query | VectorQuery | AutoQuery
|
ftsColumns?): Query | VectorQuery
|
||||||
```
|
```
|
||||||
|
|
||||||
Create a search query to find the nearest neighbors
|
Create a search query to find the nearest neighbors
|
||||||
@@ -964,7 +759,7 @@ of the given query
|
|||||||
|
|
||||||
#### Returns
|
#### Returns
|
||||||
|
|
||||||
[`Query`](Query.md) \| [`VectorQuery`](VectorQuery.md) \| [`AutoQuery`](AutoQuery.md)
|
[`Query`](Query.md) \| [`VectorQuery`](VectorQuery.md)
|
||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
@@ -988,11 +783,6 @@ All variants require the table to have an unenforced primary key
|
|||||||
([Table#setUnenforcedPrimaryKey](Table.md#setunenforcedprimarykey)); bucket sharding additionally
|
([Table#setUnenforcedPrimaryKey](Table.md#setunenforcedprimarykey)); bucket sharding additionally
|
||||||
requires it to be the single column being bucketed.
|
requires it to be the single column being bucketed.
|
||||||
|
|
||||||
Omitting `maintainedIndexes` maintains every index on the table, resolved
|
|
||||||
here, failing if one cannot be maintained — name them to install anyway.
|
|
||||||
Naming them pins an exact set, and a still-building index is rejected
|
|
||||||
rather than quietly omitted.
|
|
||||||
|
|
||||||
#### Parameters
|
#### Parameters
|
||||||
|
|
||||||
* **spec**: [`LsmWriteSpec`](../interfaces/LsmWriteSpec.md)
|
* **spec**: [`LsmWriteSpec`](../interfaces/LsmWriteSpec.md)
|
||||||
@@ -1292,18 +1082,6 @@ abstract updateFieldMetadata(updates): Promise<UpdateFieldMetadataResult>
|
|||||||
|
|
||||||
Update per-field (column) metadata.
|
Update per-field (column) metadata.
|
||||||
|
|
||||||
The following keys are treated specially, by convention, and should be
|
|
||||||
used when appropriate:
|
|
||||||
|
|
||||||
- `lancedb:description`: for a human-readable description of a field.
|
|
||||||
- `lancedb:tag:<name>`: for a user-defined key-value tag, where the suffix
|
|
||||||
names the tag category; e.g. `lancedb:tag:model: "clip"`.
|
|
||||||
- `lancedb:logical-column`: for a column grouping; e.g. `feature_v1` and
|
|
||||||
`feature_v2` might be in the same logical column.
|
|
||||||
- `lancedb:status`: for status options (`production`, `candidate`,
|
|
||||||
`deprecated`, `archived`) to designate the current life cycle state of
|
|
||||||
this column.
|
|
||||||
|
|
||||||
#### Parameters
|
#### Parameters
|
||||||
|
|
||||||
* **updates**: [`FieldMetadataUpdate`](../interfaces/FieldMetadataUpdate.md)[]
|
* **updates**: [`FieldMetadataUpdate`](../interfaces/FieldMetadataUpdate.md)[]
|
||||||
|
|||||||
+3
-19
@@ -18,7 +18,6 @@
|
|||||||
|
|
||||||
## Classes
|
## Classes
|
||||||
|
|
||||||
- [AutoQuery](classes/AutoQuery.md)
|
|
||||||
- [BooleanQuery](classes/BooleanQuery.md)
|
- [BooleanQuery](classes/BooleanQuery.md)
|
||||||
- [BoostQuery](classes/BoostQuery.md)
|
- [BoostQuery](classes/BoostQuery.md)
|
||||||
- [BranchContents](classes/BranchContents.md)
|
- [BranchContents](classes/BranchContents.md)
|
||||||
@@ -26,10 +25,8 @@
|
|||||||
- [Connection](classes/Connection.md)
|
- [Connection](classes/Connection.md)
|
||||||
- [HeaderProvider](classes/HeaderProvider.md)
|
- [HeaderProvider](classes/HeaderProvider.md)
|
||||||
- [Index](classes/Index.md)
|
- [Index](classes/Index.md)
|
||||||
- [Job](classes/Job.md)
|
|
||||||
- [MakeArrowTableOptions](classes/MakeArrowTableOptions.md)
|
- [MakeArrowTableOptions](classes/MakeArrowTableOptions.md)
|
||||||
- [MatchQuery](classes/MatchQuery.md)
|
- [MatchQuery](classes/MatchQuery.md)
|
||||||
- [MaterializedView](classes/MaterializedView.md)
|
|
||||||
- [MergeInsertBuilder](classes/MergeInsertBuilder.md)
|
- [MergeInsertBuilder](classes/MergeInsertBuilder.md)
|
||||||
- [MultiMatchQuery](classes/MultiMatchQuery.md)
|
- [MultiMatchQuery](classes/MultiMatchQuery.md)
|
||||||
- [NativeJsHeaderProvider](classes/NativeJsHeaderProvider.md)
|
- [NativeJsHeaderProvider](classes/NativeJsHeaderProvider.md)
|
||||||
@@ -60,10 +57,6 @@
|
|||||||
- [BranchDiff](interfaces/BranchDiff.md)
|
- [BranchDiff](interfaces/BranchDiff.md)
|
||||||
- [BranchIndexSummary](interfaces/BranchIndexSummary.md)
|
- [BranchIndexSummary](interfaces/BranchIndexSummary.md)
|
||||||
- [BranchRowCountSummary](interfaces/BranchRowCountSummary.md)
|
- [BranchRowCountSummary](interfaces/BranchRowCountSummary.md)
|
||||||
- [BucketStats](interfaces/BucketStats.md)
|
|
||||||
- [CherryPickError](interfaces/CherryPickError.md)
|
|
||||||
- [CherryPickPreview](interfaces/CherryPickPreview.md)
|
|
||||||
- [CherryPickResult](interfaces/CherryPickResult.md)
|
|
||||||
- [ClientConfig](interfaces/ClientConfig.md)
|
- [ClientConfig](interfaces/ClientConfig.md)
|
||||||
- [ColumnAlteration](interfaces/ColumnAlteration.md)
|
- [ColumnAlteration](interfaces/ColumnAlteration.md)
|
||||||
- [ColumnOrdering](interfaces/ColumnOrdering.md)
|
- [ColumnOrdering](interfaces/ColumnOrdering.md)
|
||||||
@@ -87,7 +80,6 @@
|
|||||||
- [FtsToken](interfaces/FtsToken.md)
|
- [FtsToken](interfaces/FtsToken.md)
|
||||||
- [FullTextQuery](interfaces/FullTextQuery.md)
|
- [FullTextQuery](interfaces/FullTextQuery.md)
|
||||||
- [FullTextSearchOptions](interfaces/FullTextSearchOptions.md)
|
- [FullTextSearchOptions](interfaces/FullTextSearchOptions.md)
|
||||||
- [GenerationStats](interfaces/GenerationStats.md)
|
|
||||||
- [HnswPqOptions](interfaces/HnswPqOptions.md)
|
- [HnswPqOptions](interfaces/HnswPqOptions.md)
|
||||||
- [HnswSqOptions](interfaces/HnswSqOptions.md)
|
- [HnswSqOptions](interfaces/HnswSqOptions.md)
|
||||||
- [IndexConfig](interfaces/IndexConfig.md)
|
- [IndexConfig](interfaces/IndexConfig.md)
|
||||||
@@ -96,17 +88,12 @@
|
|||||||
- [IvfFlatOptions](interfaces/IvfFlatOptions.md)
|
- [IvfFlatOptions](interfaces/IvfFlatOptions.md)
|
||||||
- [IvfPqOptions](interfaces/IvfPqOptions.md)
|
- [IvfPqOptions](interfaces/IvfPqOptions.md)
|
||||||
- [IvfRqOptions](interfaces/IvfRqOptions.md)
|
- [IvfRqOptions](interfaces/IvfRqOptions.md)
|
||||||
- [JobDescription](interfaces/JobDescription.md)
|
|
||||||
- [JobFailureInfo](interfaces/JobFailureInfo.md)
|
|
||||||
- [JobInfo](interfaces/JobInfo.md)
|
|
||||||
- [ListNamespacesOptions](interfaces/ListNamespacesOptions.md)
|
- [ListNamespacesOptions](interfaces/ListNamespacesOptions.md)
|
||||||
- [ListNamespacesResponse](interfaces/ListNamespacesResponse.md)
|
- [ListNamespacesResponse](interfaces/ListNamespacesResponse.md)
|
||||||
- [ListTablesOptions](interfaces/ListTablesOptions.md)
|
|
||||||
- [ListTablesResponse](interfaces/ListTablesResponse.md)
|
|
||||||
- [LsmStats](interfaces/LsmStats.md)
|
|
||||||
- [LsmWriteSpec](interfaces/LsmWriteSpec.md)
|
- [LsmWriteSpec](interfaces/LsmWriteSpec.md)
|
||||||
- [MaterializedViewDefinition](interfaces/MaterializedViewDefinition.md)
|
- [MergeBlocker](interfaces/MergeBlocker.md)
|
||||||
- [MemtableStats](interfaces/MemtableStats.md)
|
- [MergeBranchResult](interfaces/MergeBranchResult.md)
|
||||||
|
- [MergePreview](interfaces/MergePreview.md)
|
||||||
- [MergeResult](interfaces/MergeResult.md)
|
- [MergeResult](interfaces/MergeResult.md)
|
||||||
- [NativeOAuthConfig](interfaces/NativeOAuthConfig.md)
|
- [NativeOAuthConfig](interfaces/NativeOAuthConfig.md)
|
||||||
- [OAuthConfig](interfaces/OAuthConfig.md)
|
- [OAuthConfig](interfaces/OAuthConfig.md)
|
||||||
@@ -114,8 +101,6 @@
|
|||||||
- [OptimizeOptions](interfaces/OptimizeOptions.md)
|
- [OptimizeOptions](interfaces/OptimizeOptions.md)
|
||||||
- [OptimizeStats](interfaces/OptimizeStats.md)
|
- [OptimizeStats](interfaces/OptimizeStats.md)
|
||||||
- [QueryExecutionOptions](interfaces/QueryExecutionOptions.md)
|
- [QueryExecutionOptions](interfaces/QueryExecutionOptions.md)
|
||||||
- [RefreshColumnResult](interfaces/RefreshColumnResult.md)
|
|
||||||
- [RefreshMaterializedViewResult](interfaces/RefreshMaterializedViewResult.md)
|
|
||||||
- [RemovalStats](interfaces/RemovalStats.md)
|
- [RemovalStats](interfaces/RemovalStats.md)
|
||||||
- [RenameTableOptions](interfaces/RenameTableOptions.md)
|
- [RenameTableOptions](interfaces/RenameTableOptions.md)
|
||||||
- [RestNamespaceConfig](interfaces/RestNamespaceConfig.md)
|
- [RestNamespaceConfig](interfaces/RestNamespaceConfig.md)
|
||||||
@@ -148,7 +133,6 @@
|
|||||||
- [FieldLike](type-aliases/FieldLike.md)
|
- [FieldLike](type-aliases/FieldLike.md)
|
||||||
- [IntoSql](type-aliases/IntoSql.md)
|
- [IntoSql](type-aliases/IntoSql.md)
|
||||||
- [IntoVector](type-aliases/IntoVector.md)
|
- [IntoVector](type-aliases/IntoVector.md)
|
||||||
- [MaterializedViewSelect](type-aliases/MaterializedViewSelect.md)
|
|
||||||
- [MultiVector](type-aliases/MultiVector.md)
|
- [MultiVector](type-aliases/MultiVector.md)
|
||||||
- [RecordBatchLike](type-aliases/RecordBatchLike.md)
|
- [RecordBatchLike](type-aliases/RecordBatchLike.md)
|
||||||
- [SchemaLike](type-aliases/SchemaLike.md)
|
- [SchemaLike](type-aliases/SchemaLike.md)
|
||||||
|
|||||||
@@ -50,14 +50,6 @@ changedColumns: BranchColumnChange[];
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### errors
|
|
||||||
|
|
||||||
```ts
|
|
||||||
errors: CherryPickError[];
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### fromBranch
|
### fromBranch
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -74,6 +66,22 @@ mainVersion: number;
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
### mergeBlockers
|
||||||
|
|
||||||
|
```ts
|
||||||
|
mergeBlockers: MergeBlocker[];
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### mergeable
|
||||||
|
|
||||||
|
```ts
|
||||||
|
mergeable: boolean;
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### parentVersion
|
### parentVersion
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
|
|||||||
@@ -1,116 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / BucketStats
|
|
||||||
|
|
||||||
# Interface: BucketStats
|
|
||||||
|
|
||||||
Live state of one bucket. A table is N buckets on one node; flattening to a
|
|
||||||
single number hides the one hot bucket that is usually why someone opened
|
|
||||||
this endpoint.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### compacting
|
|
||||||
|
|
||||||
```ts
|
|
||||||
compacting: boolean;
|
|
||||||
```
|
|
||||||
|
|
||||||
Whether a pass owns this bucket's compaction latch right now. Says *a*
|
|
||||||
driver is running, not *whose*, and the latch is held from dispatch —
|
|
||||||
including while the pass queues for a pod-wide compactor permit. Read it
|
|
||||||
as "do not pile on", never as "mine is progressing".
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### currentGeneration
|
|
||||||
|
|
||||||
```ts
|
|
||||||
currentGeneration: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
The generation the active memtable will become.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### generations
|
|
||||||
|
|
||||||
```ts
|
|
||||||
generations: GenerationStats[];
|
|
||||||
```
|
|
||||||
|
|
||||||
Flushed L0 generations not yet merged into the base table.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### manifestVersion
|
|
||||||
|
|
||||||
```ts
|
|
||||||
manifestVersion: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
Version of the shard manifest these numbers were read from.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### memtables?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional memtables: MemtableStats[];
|
|
||||||
```
|
|
||||||
|
|
||||||
Oldest first, active last. Absent for a `"Sealed"` bucket, whose
|
|
||||||
in-memory state is torn down.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### replayAfterWalEntryPosition
|
|
||||||
|
|
||||||
```ts
|
|
||||||
replayAfterWalEntryPosition: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
WAL position replay resumes from.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### shardId
|
|
||||||
|
|
||||||
```ts
|
|
||||||
shardId: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
The shard this bucket writes.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### status
|
|
||||||
|
|
||||||
```ts
|
|
||||||
status: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
`"Active"` or `"Sealed"` (drop-table 2PC in flight).
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### walEntryPositionLastSeen
|
|
||||||
|
|
||||||
```ts
|
|
||||||
walEntryPositionLastSeen: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
Highest WAL position the writer has seen. The difference against
|
|
||||||
`replayAfterWalEntryPosition` is the WAL lag.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### writerEpoch
|
|
||||||
|
|
||||||
```ts
|
|
||||||
writerEpoch: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
Epoch of the writer that currently owns the shard.
|
|
||||||
@@ -1,17 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / CherryPickPreview
|
|
||||||
|
|
||||||
# Interface: CherryPickPreview
|
|
||||||
|
|
||||||
Changes that would be, or were, promoted by a cherry-pick.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### promotedColumns
|
|
||||||
|
|
||||||
```ts
|
|
||||||
promotedColumns: string[];
|
|
||||||
```
|
|
||||||
@@ -17,8 +17,7 @@ metadata: Record<string, null | string>;
|
|||||||
```
|
```
|
||||||
|
|
||||||
Metadata key/value pairs. Merged into the field's existing metadata by
|
Metadata key/value pairs. Merged into the field's existing metadata by
|
||||||
default; a value of `null` deletes that key. See
|
default; a value of `null` deletes that key.
|
||||||
[Table.updateFieldMetadata](../classes/Table.md#updatefieldmetadata) for the conventional `lancedb:*` keys.
|
|
||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
|||||||
@@ -1,40 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / GenerationStats
|
|
||||||
|
|
||||||
# Interface: GenerationStats
|
|
||||||
|
|
||||||
One flushed L0 generation.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### bytes
|
|
||||||
|
|
||||||
```ts
|
|
||||||
bytes: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
On-disk size of the generation.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### generation
|
|
||||||
|
|
||||||
```ts
|
|
||||||
generation: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
The generation number. Increases as memtables are sealed into L0.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### rows?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional rows: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
Present only when `includeGenerationRows` was requested. Off by default
|
|
||||||
because each count opens an uncached Lance dataset.
|
|
||||||
@@ -1,66 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / JobDescription
|
|
||||||
|
|
||||||
# Interface: JobDescription
|
|
||||||
|
|
||||||
A described job from `Connection.getJob`.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### creationMs
|
|
||||||
|
|
||||||
```ts
|
|
||||||
creationMs: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
When the job was created, in milliseconds since the epoch.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### failure?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional failure: JobFailureInfo;
|
|
||||||
```
|
|
||||||
|
|
||||||
Why the job failed, when the job is failed and the server reports a
|
|
||||||
reason.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### jobId
|
|
||||||
|
|
||||||
```ts
|
|
||||||
jobId: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### jobType
|
|
||||||
|
|
||||||
```ts
|
|
||||||
jobType: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### specJson?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional specJson: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
The job-type-specific specification as a JSON string, when present.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### state
|
|
||||||
|
|
||||||
```ts
|
|
||||||
state: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Lifecycle state: "running", "finished", "failed", or "cancelled".
|
|
||||||
@@ -1,33 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / JobFailureInfo
|
|
||||||
|
|
||||||
# Interface: JobFailureInfo
|
|
||||||
|
|
||||||
The server's account of why a job failed.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### message?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional message: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### phase?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional phase: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### retryable?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional retryable: boolean;
|
|
||||||
```
|
|
||||||
@@ -1,58 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / JobInfo
|
|
||||||
|
|
||||||
# Interface: JobInfo
|
|
||||||
|
|
||||||
A row from `Connection.listJobs`: one server-side job.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### createdAtMillis
|
|
||||||
|
|
||||||
```ts
|
|
||||||
createdAtMillis: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
When the job was created, in milliseconds since the epoch.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### jobId
|
|
||||||
|
|
||||||
```ts
|
|
||||||
jobId: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
The job id -- what `Connection.getJob` and `Connection.cancelJob`
|
|
||||||
accept.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### jobType
|
|
||||||
|
|
||||||
```ts
|
|
||||||
jobType: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### state
|
|
||||||
|
|
||||||
```ts
|
|
||||||
state: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Lifecycle state: "running", "finished", "failed", or "cancelled".
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### table
|
|
||||||
|
|
||||||
```ts
|
|
||||||
table: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
The table the job runs against, without URI or namespace.
|
|
||||||
@@ -1,34 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / ListTablesOptions
|
|
||||||
|
|
||||||
# Interface: ListTablesOptions
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### limit?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional limit: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
An upper bound on how many tables to return.
|
|
||||||
|
|
||||||
A page may hold fewer than this and still not be the last one, so keep
|
|
||||||
going while the response carries a page token rather than while pages are
|
|
||||||
full.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### pageToken?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional pageToken: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Token from a previous response, to resume listing where it left off.
|
|
||||||
|
|
||||||
The token is opaque: it carries whatever the database needs to resume, and
|
|
||||||
callers should not construct or interpret one.
|
|
||||||
@@ -1,23 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / ListTablesResponse
|
|
||||||
|
|
||||||
# Interface: ListTablesResponse
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### pageToken?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional pageToken: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### tables
|
|
||||||
|
|
||||||
```ts
|
|
||||||
tables: string[];
|
|
||||||
```
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / LsmStats
|
|
||||||
|
|
||||||
# Interface: LsmStats
|
|
||||||
|
|
||||||
Live per-bucket LSM state, as returned by `Table#getLsmStats`.
|
|
||||||
|
|
||||||
Nothing here is derived: sums and differences (total L0 bytes, WAL lag) are
|
|
||||||
the caller's to compute.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### buckets
|
|
||||||
|
|
||||||
```ts
|
|
||||||
buckets: BucketStats[];
|
|
||||||
```
|
|
||||||
|
|
||||||
One entry per bucket backing this table.
|
|
||||||
@@ -34,9 +34,7 @@ Bucket and identity variants: the sharding column.
|
|||||||
optional maintainedIndexes: string[];
|
optional maintainedIndexes: string[];
|
||||||
```
|
```
|
||||||
|
|
||||||
Indexes the MemWAL keeps up to date. Omit to maintain every supported
|
Names of indexes the MemWAL should keep up to date during writes.
|
||||||
index, resolved on install — a snapshot, so indexes created later are not
|
|
||||||
maintained. Pass `[]` for none.
|
|
||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
|||||||
@@ -1,59 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / MaterializedViewDefinition
|
|
||||||
|
|
||||||
# Interface: MaterializedViewDefinition
|
|
||||||
|
|
||||||
The query that defines a materialized view.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### filter?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional filter: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
SQL predicate selecting the source rows the view holds.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### inputs
|
|
||||||
|
|
||||||
```ts
|
|
||||||
inputs: string[];
|
|
||||||
```
|
|
||||||
|
|
||||||
Source columns the projections and filter read.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### limit?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional limit: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
Cap on the number of rows the view holds.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### projections
|
|
||||||
|
|
||||||
```ts
|
|
||||||
projections: [string, string][];
|
|
||||||
```
|
|
||||||
|
|
||||||
`[output column, SQL expression]` pairs, in view schema order.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### sourceTable
|
|
||||||
|
|
||||||
```ts
|
|
||||||
sourceTable: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Name of the source table, in the same database as the view.
|
|
||||||
@@ -1,60 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / MemtableStats
|
|
||||||
|
|
||||||
# Interface: MemtableStats
|
|
||||||
|
|
||||||
One in-memory memtable.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### batches
|
|
||||||
|
|
||||||
```ts
|
|
||||||
batches: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
Record batches currently buffered.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### bytes
|
|
||||||
|
|
||||||
```ts
|
|
||||||
bytes: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
Estimated in-memory size.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### generation
|
|
||||||
|
|
||||||
```ts
|
|
||||||
generation: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
The generation this memtable will become once sealed.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### indexes
|
|
||||||
|
|
||||||
```ts
|
|
||||||
indexes: string[];
|
|
||||||
```
|
|
||||||
|
|
||||||
Names of the indexes this memtable carries. An absent name is the whole
|
|
||||||
answer to "why is my fresh-tier search on that column brute-force".
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### rows
|
|
||||||
|
|
||||||
```ts
|
|
||||||
rows: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
Rows currently buffered.
|
|
||||||
@@ -2,11 +2,11 @@
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / CherryPickError
|
[@lancedb/lancedb](../globals.md) / MergeBlocker
|
||||||
|
|
||||||
# Interface: CherryPickError
|
# Interface: MergeBlocker
|
||||||
|
|
||||||
A reason why a cherry-pick cannot currently land.
|
A reason why a branch cannot currently be merged.
|
||||||
|
|
||||||
## Properties
|
## Properties
|
||||||
|
|
||||||
+6
-6
@@ -2,11 +2,11 @@
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / CherryPickResult
|
[@lancedb/lancedb](../globals.md) / MergeBranchResult
|
||||||
|
|
||||||
# Interface: CherryPickResult
|
# Interface: MergeBranchResult
|
||||||
|
|
||||||
Result of previewing or attempting a cherry-pick.
|
Result of previewing or attempting a branch merge.
|
||||||
|
|
||||||
## Properties
|
## Properties
|
||||||
|
|
||||||
@@ -29,7 +29,7 @@ optional mainVersionAfter: number;
|
|||||||
### preview
|
### preview
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
preview: CherryPickPreview;
|
preview: MergePreview;
|
||||||
```
|
```
|
||||||
|
|
||||||
***
|
***
|
||||||
@@ -38,9 +38,9 @@ preview: CherryPickPreview;
|
|||||||
|
|
||||||
```ts
|
```ts
|
||||||
status:
|
status:
|
||||||
| "failed"
|
|
||||||
| "unknown"
|
| "unknown"
|
||||||
|
| "rejected"
|
||||||
| "ready"
|
| "ready"
|
||||||
| "notImplemented"
|
| "notImplemented"
|
||||||
| "cherryPicked";
|
| "merged";
|
||||||
```
|
```
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / MergePreview
|
||||||
|
|
||||||
|
# Interface: MergePreview
|
||||||
|
|
||||||
|
Changes that would be, or were, promoted by a branch merge.
|
||||||
|
|
||||||
|
## Properties
|
||||||
|
|
||||||
|
### promotedColumns
|
||||||
|
|
||||||
|
```ts
|
||||||
|
promotedColumns: string[];
|
||||||
|
```
|
||||||
@@ -1,23 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / RefreshColumnResult
|
|
||||||
|
|
||||||
# Interface: RefreshColumnResult
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### rowsFilled
|
|
||||||
|
|
||||||
```ts
|
|
||||||
rowsFilled: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### version
|
|
||||||
|
|
||||||
```ts
|
|
||||||
version: number;
|
|
||||||
```
|
|
||||||
@@ -1,41 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / RefreshMaterializedViewResult
|
|
||||||
|
|
||||||
# Interface: RefreshMaterializedViewResult
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### mode
|
|
||||||
|
|
||||||
```ts
|
|
||||||
mode: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
How the view was brought up to date: "rebuild", "incremental" or "no_op".
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### rowsWritten
|
|
||||||
|
|
||||||
```ts
|
|
||||||
rowsWritten: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### sourceVersion
|
|
||||||
|
|
||||||
```ts
|
|
||||||
sourceVersion: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### version
|
|
||||||
|
|
||||||
```ts
|
|
||||||
version: number;
|
|
||||||
```
|
|
||||||
@@ -4,16 +4,11 @@
|
|||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / TableNamesOptions
|
[@lancedb/lancedb](../globals.md) / TableNamesOptions
|
||||||
|
|
||||||
# Interface: ~~TableNamesOptions~~
|
# Interface: TableNamesOptions
|
||||||
|
|
||||||
## Deprecated
|
|
||||||
|
|
||||||
Use [ListTablesOptions](ListTablesOptions.md) with [Connection.listTables](../classes/Connection.md#listtables)
|
|
||||||
instead.
|
|
||||||
|
|
||||||
## Properties
|
## Properties
|
||||||
|
|
||||||
### ~~limit?~~
|
### limit?
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
optional limit: number;
|
optional limit: number;
|
||||||
@@ -23,7 +18,7 @@ An optional limit to the number of results to return.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### ~~startAfter?~~
|
### startAfter?
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
optional startAfter: string;
|
optional startAfter: string;
|
||||||
|
|||||||
@@ -44,7 +44,4 @@ The number of rows in the table
|
|||||||
totalBytes: number;
|
totalBytes: number;
|
||||||
```
|
```
|
||||||
|
|
||||||
The total size, in bytes, of the table's data files, index files, and
|
The total number of bytes in the table
|
||||||
overlay files
|
|
||||||
|
|
||||||
Read from the manifest, so this excludes deletion files and manifests.
|
|
||||||
|
|||||||
@@ -25,12 +25,9 @@
|
|||||||
### Type Aliases
|
### Type Aliases
|
||||||
|
|
||||||
- [CreateReturnType](type-aliases/CreateReturnType.md)
|
- [CreateReturnType](type-aliases/CreateReturnType.md)
|
||||||
- [EmbeddingMetadataEntry](type-aliases/EmbeddingMetadataEntry.md)
|
|
||||||
- [ResolvedEmbeddingFunctionConfig](type-aliases/ResolvedEmbeddingFunctionConfig.md)
|
|
||||||
|
|
||||||
### Functions
|
### Functions
|
||||||
|
|
||||||
- [LanceSchema](functions/LanceSchema.md)
|
- [LanceSchema](functions/LanceSchema.md)
|
||||||
- [getRegistry](functions/getRegistry.md)
|
- [getRegistry](functions/getRegistry.md)
|
||||||
- [parseEmbeddingMetadata](functions/parseEmbeddingMetadata.md)
|
|
||||||
- [register](functions/register.md)
|
- [register](functions/register.md)
|
||||||
|
|||||||
@@ -10,12 +10,16 @@
|
|||||||
function getRegistry(): EmbeddingFunctionRegistry
|
function getRegistry(): EmbeddingFunctionRegistry
|
||||||
```
|
```
|
||||||
|
|
||||||
Get the global embedding function registry.
|
Utility function to get the global instance of the registry
|
||||||
|
|
||||||
LanceDB built-in providers are initialized when this public API is first
|
|
||||||
used, so importing the root package does not change automatic search
|
|
||||||
selection for tables without embedding metadata.
|
|
||||||
|
|
||||||
## Returns
|
## Returns
|
||||||
|
|
||||||
[`EmbeddingFunctionRegistry`](../classes/EmbeddingFunctionRegistry.md)
|
[`EmbeddingFunctionRegistry`](../classes/EmbeddingFunctionRegistry.md)
|
||||||
|
|
||||||
|
`EmbeddingFunctionRegistry` The global instance of the registry
|
||||||
|
|
||||||
|
## Example
|
||||||
|
|
||||||
|
```ts
|
||||||
|
const registry = getRegistry();
|
||||||
|
const openai = registry.get("openai").create();
|
||||||
|
|||||||
@@ -1,22 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../../../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../../../globals.md) / [embedding](../README.md) / parseEmbeddingMetadata
|
|
||||||
|
|
||||||
# Function: parseEmbeddingMetadata()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
function parseEmbeddingMetadata(json): EmbeddingMetadataEntry[]
|
|
||||||
```
|
|
||||||
|
|
||||||
The single parser for `embedding_functions` schema metadata: every reader
|
|
||||||
goes through here, so the wire contract cannot fork between them.
|
|
||||||
|
|
||||||
## Parameters
|
|
||||||
|
|
||||||
* **json**: `string`
|
|
||||||
|
|
||||||
## Returns
|
|
||||||
|
|
||||||
[`EmbeddingMetadataEntry`](../type-aliases/EmbeddingMetadataEntry.md)[]
|
|
||||||
@@ -1,40 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../../../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../../../globals.md) / [embedding](../README.md) / EmbeddingMetadataEntry
|
|
||||||
|
|
||||||
# Type Alias: EmbeddingMetadataEntry
|
|
||||||
|
|
||||||
```ts
|
|
||||||
type EmbeddingMetadataEntry: object;
|
|
||||||
```
|
|
||||||
|
|
||||||
One entry of the `embedding_functions` schema metadata, with the column
|
|
||||||
keys normalized across the bindings' spellings.
|
|
||||||
|
|
||||||
## Type declaration
|
|
||||||
|
|
||||||
### model
|
|
||||||
|
|
||||||
```ts
|
|
||||||
model: EmbeddingFunction["TOptions"];
|
|
||||||
```
|
|
||||||
|
|
||||||
### name
|
|
||||||
|
|
||||||
```ts
|
|
||||||
name: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
### sourceColumn
|
|
||||||
|
|
||||||
```ts
|
|
||||||
sourceColumn: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
### vectorColumn
|
|
||||||
|
|
||||||
```ts
|
|
||||||
vectorColumn: string;
|
|
||||||
```
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../../../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../../../globals.md) / [embedding](../README.md) / ResolvedEmbeddingFunctionConfig
|
|
||||||
|
|
||||||
# Type Alias: ResolvedEmbeddingFunctionConfig
|
|
||||||
|
|
||||||
```ts
|
|
||||||
type ResolvedEmbeddingFunctionConfig: EmbeddingFunctionConfig & object;
|
|
||||||
```
|
|
||||||
|
|
||||||
An [EmbeddingFunctionConfig] read back from table metadata, where the
|
|
||||||
vector column is always recorded.
|
|
||||||
|
|
||||||
## Type declaration
|
|
||||||
|
|
||||||
### vectorColumn
|
|
||||||
|
|
||||||
```ts
|
|
||||||
vectorColumn: string;
|
|
||||||
```
|
|
||||||
@@ -1,14 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / MaterializedViewSelect
|
|
||||||
|
|
||||||
# Type Alias: MaterializedViewSelect
|
|
||||||
|
|
||||||
```ts
|
|
||||||
type MaterializedViewSelect: (string | [string, string])[] | Record<string, string>;
|
|
||||||
```
|
|
||||||
|
|
||||||
The view's columns: column names, `[alias, SQL expression]` pairs, or a
|
|
||||||
record of the same. A bare name projects itself.
|
|
||||||
@@ -31,7 +31,7 @@ is also an [asynchronous API client](#connections-asynchronous).
|
|||||||
## Namespaces (Synchronous)
|
## Namespaces (Synchronous)
|
||||||
|
|
||||||
A namespace-backed connection resolves tables through a
|
A namespace-backed connection resolves tables through a
|
||||||
[Lance namespace](https://lance-format.github.io/lance-namespace/) service instead of
|
[Lance namespace](https://lancedb.github.io/lance-namespace/) service instead of
|
||||||
listing a storage directory.
|
listing a storage directory.
|
||||||
|
|
||||||
::: lancedb.connect_namespace
|
::: lancedb.connect_namespace
|
||||||
@@ -42,8 +42,6 @@ listing a storage directory.
|
|||||||
|
|
||||||
::: lancedb.table.Table
|
::: lancedb.table.Table
|
||||||
|
|
||||||
::: lancedb.table.CompactionOptions
|
|
||||||
|
|
||||||
::: lancedb.table.FragmentStatistics
|
::: lancedb.table.FragmentStatistics
|
||||||
|
|
||||||
::: lancedb.table.FragmentSummaryStats
|
::: lancedb.table.FragmentSummaryStats
|
||||||
@@ -54,62 +52,6 @@ listing a storage directory.
|
|||||||
|
|
||||||
::: lancedb.table.Branches
|
::: lancedb.table.Branches
|
||||||
|
|
||||||
::: lancedb.LsmWriteSpec
|
|
||||||
|
|
||||||
## Functions and Jobs
|
|
||||||
|
|
||||||
::: lancedb.functions.FunctionArtifact
|
|
||||||
|
|
||||||
::: lancedb.functions.FunctionParameter
|
|
||||||
|
|
||||||
::: lancedb.functions.FunctionResultField
|
|
||||||
|
|
||||||
::: lancedb.functions.FunctionOutput
|
|
||||||
|
|
||||||
::: lancedb.functions.FunctionSignature
|
|
||||||
|
|
||||||
::: lancedb.functions.PythonEnvironmentSpec
|
|
||||||
|
|
||||||
::: lancedb.functions.udf
|
|
||||||
|
|
||||||
::: lancedb.functions.UdfDefinition
|
|
||||||
|
|
||||||
::: lancedb.functions.FunctionRegistrationRequest
|
|
||||||
|
|
||||||
::: lancedb.functions.FunctionArtifactRequest
|
|
||||||
|
|
||||||
::: lancedb.functions.FunctionArtifactContent
|
|
||||||
|
|
||||||
::: lancedb.functions.PythonAdapterSpec
|
|
||||||
|
|
||||||
::: lancedb.functions.FunctionVersion
|
|
||||||
|
|
||||||
::: lancedb.functions.PythonRuntimeSpec
|
|
||||||
|
|
||||||
::: lancedb.functions.FunctionVersionRef
|
|
||||||
|
|
||||||
::: lancedb.functions.ApplicationInput
|
|
||||||
|
|
||||||
::: lancedb.functions.FunctionApplication
|
|
||||||
|
|
||||||
::: lancedb.functions.InputBinding
|
|
||||||
|
|
||||||
::: lancedb.functions.OutputMapping
|
|
||||||
|
|
||||||
::: lancedb.functions.FunctionBinding
|
|
||||||
|
|
||||||
::: lancedb.functions.RefreshColumnResult
|
|
||||||
|
|
||||||
::: lancedb.job.Job
|
|
||||||
|
|
||||||
::: lancedb.job.AsyncJob
|
|
||||||
|
|
||||||
## Materialized Views (Synchronous)
|
|
||||||
|
|
||||||
::: lancedb.materialized_view.MaterializedView
|
|
||||||
|
|
||||||
::: lancedb.materialized_view.MaterializedViewDefinition
|
|
||||||
|
|
||||||
## Expressions
|
## Expressions
|
||||||
|
|
||||||
Type-safe expression builder for filters and projections. Use these instead
|
Type-safe expression builder for filters and projections. Use these instead
|
||||||
@@ -161,8 +103,6 @@ and combined with [BooleanQuery][lancedb.query.BooleanQuery].
|
|||||||
|
|
||||||
::: lancedb.query.FullTextOperator
|
::: lancedb.query.FullTextOperator
|
||||||
|
|
||||||
::: lancedb.query.DocumentGranularity
|
|
||||||
|
|
||||||
::: lancedb.query.Occur
|
::: lancedb.query.Occur
|
||||||
|
|
||||||
## Embeddings
|
## Embeddings
|
||||||
@@ -211,9 +151,8 @@ The same option is available on `lancedb.tokenize(...)` and the deprecated
|
|||||||
```python
|
```python
|
||||||
import lancedb
|
import lancedb
|
||||||
|
|
||||||
tokens = list(
|
tokens = list(lancedb.tokenize("acme makes searchable data",
|
||||||
lancedb.tokenize("acme makes searchable data", custom_stop_words=["acme"])
|
custom_stop_words=["acme"]))
|
||||||
)
|
|
||||||
```
|
```
|
||||||
|
|
||||||
::: lancedb.tokenize
|
::: lancedb.tokenize
|
||||||
@@ -265,8 +204,6 @@ instead of being materialized with the rest of the row.
|
|||||||
|
|
||||||
::: lancedb.streaming.StreamingDataset
|
::: lancedb.streaming.StreamingDataset
|
||||||
|
|
||||||
::: lancedb.streaming.StreamingDataLoader
|
|
||||||
|
|
||||||
::: lancedb.permutation.permutation_builder
|
::: lancedb.permutation.permutation_builder
|
||||||
|
|
||||||
::: lancedb.permutation.PermutationBuilder
|
::: lancedb.permutation.PermutationBuilder
|
||||||
@@ -307,10 +244,6 @@ Table hold your actual data as a collection of records / rows.
|
|||||||
|
|
||||||
::: lancedb.table.AsyncBranches
|
::: lancedb.table.AsyncBranches
|
||||||
|
|
||||||
## Materialized Views (Asynchronous)
|
|
||||||
|
|
||||||
::: lancedb.materialized_view.AsyncMaterializedView
|
|
||||||
|
|
||||||
## Indices (Asynchronous)
|
## Indices (Asynchronous)
|
||||||
|
|
||||||
Indices can be created on a table to speed up queries. This section
|
Indices can be created on a table to speed up queries. This section
|
||||||
|
|||||||
@@ -29,48 +29,6 @@ LanceNamespace namespaceClient = LanceDbNamespaceClientBuilder.newBuilder()
|
|||||||
.build();
|
.build();
|
||||||
```
|
```
|
||||||
|
|
||||||
## MemWAL LSM write path
|
|
||||||
|
|
||||||
Most table operations reach LanceDB through the `LanceNamespace` above, which is
|
|
||||||
generated from the Lance Namespace specification. The MemWAL LSM routes are not part
|
|
||||||
of that specification, so they are issued through a separate client:
|
|
||||||
|
|
||||||
```java
|
|
||||||
import com.lancedb.LanceDbRestClient;
|
|
||||||
import com.lancedb.LanceDbTableLsm;
|
|
||||||
import com.lancedb.LsmWriteSpec;
|
|
||||||
|
|
||||||
LanceDbRestClient client = LanceDbNamespaceClientBuilder.newBuilder()
|
|
||||||
.apiKey("your_lancedb_cloud_api_key")
|
|
||||||
.database("your_database_name")
|
|
||||||
.buildRestClient();
|
|
||||||
|
|
||||||
LanceDbTableLsm lsm = new LanceDbTableLsm(client, "my_table");
|
|
||||||
|
|
||||||
// Route future merge_insert upserts through the MemWAL, hash-bucketed by `id`.
|
|
||||||
lsm.setLsmWriteSpec(LsmWriteSpec.bucket("id", 16));
|
|
||||||
|
|
||||||
// ... merge_insert traffic ...
|
|
||||||
|
|
||||||
// Converge the fresh tier into the base table.
|
|
||||||
lsm.checkpointLsm();
|
|
||||||
|
|
||||||
// Inspect live per-bucket state.
|
|
||||||
lsm.getLsmStats().ifPresent(stats -> stats.buckets().forEach(bucket ->
|
|
||||||
System.out.println(bucket.shardId() + ": " + bucket.generations().size() + " L0 generations")));
|
|
||||||
|
|
||||||
client.close();
|
|
||||||
```
|
|
||||||
|
|
||||||
`maintainedIndexes` is tri-state, and the null default is the opposite of what a Java
|
|
||||||
reader usually expects:
|
|
||||||
|
|
||||||
| Value | Meaning |
|
|
||||||
| --- | --- |
|
|
||||||
| unset (null) | Maintain **every** index the MemWAL can, resolved on install |
|
|
||||||
| `Collections.emptyList()` | Maintain **none** |
|
|
||||||
| `Arrays.asList("id_idx")` | Maintain exactly those |
|
|
||||||
|
|
||||||
## Development
|
## Development
|
||||||
|
|
||||||
Build:
|
Build:
|
||||||
|
|||||||
@@ -8,7 +8,7 @@
|
|||||||
<parent>
|
<parent>
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-parent</artifactId>
|
<artifactId>lancedb-parent</artifactId>
|
||||||
<version>0.38.0-beta.11</version>
|
<version>0.37.1-beta.0</version>
|
||||||
<relativePath>../pom.xml</relativePath>
|
<relativePath>../pom.xml</relativePath>
|
||||||
</parent>
|
</parent>
|
||||||
|
|
||||||
@@ -33,20 +33,6 @@
|
|||||||
<artifactId>arrow-memory-netty</artifactId>
|
<artifactId>arrow-memory-netty</artifactId>
|
||||||
</dependency>
|
</dependency>
|
||||||
|
|
||||||
<!-- Transport for the LanceDB routes outside the Lance Namespace spec.
|
|
||||||
Versions match what lance-namespace-apache-client resolves to. -->
|
|
||||||
<dependency>
|
|
||||||
<groupId>org.apache.httpcomponents.client5</groupId>
|
|
||||||
<artifactId>httpclient5</artifactId>
|
|
||||||
<version>5.2.1</version>
|
|
||||||
</dependency>
|
|
||||||
|
|
||||||
<dependency>
|
|
||||||
<groupId>com.fasterxml.jackson.core</groupId>
|
|
||||||
<artifactId>jackson-databind</artifactId>
|
|
||||||
<version>2.17.1</version>
|
|
||||||
</dependency>
|
|
||||||
|
|
||||||
<dependency>
|
<dependency>
|
||||||
<groupId>org.junit.jupiter</groupId>
|
<groupId>org.junit.jupiter</groupId>
|
||||||
<artifactId>junit-jupiter</artifactId>
|
<artifactId>junit-jupiter</artifactId>
|
||||||
|
|||||||
@@ -1,194 +0,0 @@
|
|||||||
/*
|
|
||||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
* you may not use this file except in compliance with the License.
|
|
||||||
* You may obtain a copy of the License at
|
|
||||||
*
|
|
||||||
* http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
*
|
|
||||||
* Unless required by applicable law or agreed to in writing, software
|
|
||||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
* See the License for the specific language governing permissions and
|
|
||||||
* limitations under the License.
|
|
||||||
*/
|
|
||||||
package com.lancedb;
|
|
||||||
|
|
||||||
import com.fasterxml.jackson.databind.JsonNode;
|
|
||||||
|
|
||||||
import java.util.ArrayList;
|
|
||||||
import java.util.Collections;
|
|
||||||
import java.util.List;
|
|
||||||
import java.util.Optional;
|
|
||||||
import java.util.OptionalLong;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Live state of one bucket. A table is N buckets on one node; flattening to a single number hides
|
|
||||||
* the one hot bucket that is usually why someone opened this endpoint.
|
|
||||||
*/
|
|
||||||
public class BucketStats {
|
|
||||||
private static final String CONTEXT = "bucket stats";
|
|
||||||
|
|
||||||
private final String shardId;
|
|
||||||
private final String status;
|
|
||||||
private final long writerEpoch;
|
|
||||||
private final long manifestVersion;
|
|
||||||
private final long currentGeneration;
|
|
||||||
private final long replayAfterWalEntryPosition;
|
|
||||||
private final long walEntryPositionLastSeen;
|
|
||||||
private final List<GenerationStats> generations;
|
|
||||||
private final boolean compacting;
|
|
||||||
private final List<MemtableStats> memtables;
|
|
||||||
|
|
||||||
BucketStats(
|
|
||||||
String shardId,
|
|
||||||
String status,
|
|
||||||
long writerEpoch,
|
|
||||||
long manifestVersion,
|
|
||||||
long currentGeneration,
|
|
||||||
long replayAfterWalEntryPosition,
|
|
||||||
long walEntryPositionLastSeen,
|
|
||||||
List<GenerationStats> generations,
|
|
||||||
boolean compacting,
|
|
||||||
List<MemtableStats> memtables) {
|
|
||||||
this.shardId = shardId;
|
|
||||||
this.status = status;
|
|
||||||
this.writerEpoch = writerEpoch;
|
|
||||||
this.manifestVersion = manifestVersion;
|
|
||||||
this.currentGeneration = currentGeneration;
|
|
||||||
this.replayAfterWalEntryPosition = replayAfterWalEntryPosition;
|
|
||||||
this.walEntryPositionLastSeen = walEntryPositionLastSeen;
|
|
||||||
this.generations = Collections.unmodifiableList(generations);
|
|
||||||
this.compacting = compacting;
|
|
||||||
this.memtables = memtables == null ? null : Collections.unmodifiableList(memtables);
|
|
||||||
}
|
|
||||||
|
|
||||||
/** The shard this bucket writes. */
|
|
||||||
public String shardId() {
|
|
||||||
return shardId;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** {@code "Active"} or {@code "Sealed"} (drop-table 2PC in flight). */
|
|
||||||
public String status() {
|
|
||||||
return status;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Epoch of the writer that currently owns the shard. */
|
|
||||||
public long writerEpoch() {
|
|
||||||
return writerEpoch;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Version of the shard manifest these numbers were read from. */
|
|
||||||
public long manifestVersion() {
|
|
||||||
return manifestVersion;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** The generation the active memtable will become. */
|
|
||||||
public long currentGeneration() {
|
|
||||||
return currentGeneration;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** WAL position replay resumes from. */
|
|
||||||
public long replayAfterWalEntryPosition() {
|
|
||||||
return replayAfterWalEntryPosition;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Highest WAL position the writer has seen. The difference against {@link
|
|
||||||
* #replayAfterWalEntryPosition()} is the WAL lag.
|
|
||||||
*/
|
|
||||||
public long walEntryPositionLastSeen() {
|
|
||||||
return walEntryPositionLastSeen;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Flushed L0 generations not yet merged into the base table. */
|
|
||||||
public List<GenerationStats> generations() {
|
|
||||||
return generations;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Whether a pass owns this bucket's compaction latch right now. Says <em>a</em> driver is
|
|
||||||
* running, not <em>whose</em>, and the latch is held from dispatch — including while the pass
|
|
||||||
* queues for a pod-wide compactor permit. Read it as "do not pile on", never as "mine is
|
|
||||||
* progressing".
|
|
||||||
*/
|
|
||||||
public boolean compacting() {
|
|
||||||
return compacting;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Oldest first, active last. Empty for a {@code "Sealed"} bucket, whose state is torn down. */
|
|
||||||
public Optional<List<MemtableStats>> memtables() {
|
|
||||||
return Optional.ofNullable(memtables);
|
|
||||||
}
|
|
||||||
|
|
||||||
/** The newest flushed generation, or empty when L0 is empty. */
|
|
||||||
OptionalLong newestGeneration() {
|
|
||||||
OptionalLong newest = OptionalLong.empty();
|
|
||||||
for (GenerationStats generation : generations) {
|
|
||||||
if (!newest.isPresent() || generation.generation() > newest.getAsLong()) {
|
|
||||||
newest = OptionalLong.of(generation.generation());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return newest;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* How many generations at or below {@code target} are still in L0.
|
|
||||||
*
|
|
||||||
* <p>A count, not a boolean: one pass drains a bounded prefix rather than the whole target set,
|
|
||||||
* so a boolean would read as "no progress" for every pass but the last. Compaction drains
|
|
||||||
* oldest-first, so this decreases monotonically.
|
|
||||||
*/
|
|
||||||
long outstandingGenerations(long target) {
|
|
||||||
long count = 0;
|
|
||||||
for (GenerationStats generation : generations) {
|
|
||||||
if (generation.generation() <= target) {
|
|
||||||
count++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return count;
|
|
||||||
}
|
|
||||||
|
|
||||||
static BucketStats fromJson(JsonNode node) {
|
|
||||||
JsonFields.requiredObject(node, CONTEXT);
|
|
||||||
List<GenerationStats> generations = new ArrayList<GenerationStats>();
|
|
||||||
for (JsonNode generation : JsonFields.requiredArray(node, "generations", CONTEXT)) {
|
|
||||||
generations.add(GenerationStats.fromJson(generation));
|
|
||||||
}
|
|
||||||
|
|
||||||
JsonNode memtablesNode = JsonFields.optionalArray(node, "memtables", CONTEXT);
|
|
||||||
List<MemtableStats> memtables = null;
|
|
||||||
if (memtablesNode != null) {
|
|
||||||
memtables = new ArrayList<MemtableStats>();
|
|
||||||
for (JsonNode memtable : memtablesNode) {
|
|
||||||
memtables.add(MemtableStats.fromJson(memtable));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return new BucketStats(
|
|
||||||
JsonFields.requiredText(node, "shard_id", CONTEXT),
|
|
||||||
JsonFields.requiredText(node, "status", CONTEXT),
|
|
||||||
JsonFields.requiredLong(node, "writer_epoch", CONTEXT),
|
|
||||||
JsonFields.requiredLong(node, "manifest_version", CONTEXT),
|
|
||||||
JsonFields.requiredLong(node, "current_generation", CONTEXT),
|
|
||||||
JsonFields.requiredLong(node, "replay_after_wal_entry_position", CONTEXT),
|
|
||||||
JsonFields.requiredLong(node, "wal_entry_position_last_seen", CONTEXT),
|
|
||||||
generations,
|
|
||||||
JsonFields.requiredBoolean(node, "compacting", CONTEXT),
|
|
||||||
memtables);
|
|
||||||
}
|
|
||||||
|
|
||||||
@Override
|
|
||||||
public String toString() {
|
|
||||||
return "BucketStats{shardId="
|
|
||||||
+ shardId
|
|
||||||
+ ", status="
|
|
||||||
+ status
|
|
||||||
+ ", currentGeneration="
|
|
||||||
+ currentGeneration
|
|
||||||
+ ", generations="
|
|
||||||
+ generations
|
|
||||||
+ ", compacting="
|
|
||||||
+ compacting
|
|
||||||
+ "}";
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,64 +0,0 @@
|
|||||||
/*
|
|
||||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
* you may not use this file except in compliance with the License.
|
|
||||||
* You may obtain a copy of the License at
|
|
||||||
*
|
|
||||||
* http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
*
|
|
||||||
* Unless required by applicable law or agreed to in writing, software
|
|
||||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
* See the License for the specific language governing permissions and
|
|
||||||
* limitations under the License.
|
|
||||||
*/
|
|
||||||
package com.lancedb;
|
|
||||||
|
|
||||||
import com.fasterxml.jackson.databind.JsonNode;
|
|
||||||
|
|
||||||
import java.util.OptionalLong;
|
|
||||||
|
|
||||||
/** One flushed L0 generation. */
|
|
||||||
public class GenerationStats {
|
|
||||||
private static final String CONTEXT = "generation stats";
|
|
||||||
|
|
||||||
private final long generation;
|
|
||||||
private final long bytes;
|
|
||||||
private final Long rows;
|
|
||||||
|
|
||||||
GenerationStats(long generation, long bytes, Long rows) {
|
|
||||||
this.generation = generation;
|
|
||||||
this.bytes = bytes;
|
|
||||||
this.rows = rows;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** The generation number. Increases as memtables are sealed into L0. */
|
|
||||||
public long generation() {
|
|
||||||
return generation;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** On-disk size of the generation. */
|
|
||||||
public long bytes() {
|
|
||||||
return bytes;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Rows in this generation, present only when {@code includeGenerationRows} was requested. Off by
|
|
||||||
* default because each count opens an uncached Lance dataset.
|
|
||||||
*/
|
|
||||||
public OptionalLong rows() {
|
|
||||||
return rows == null ? OptionalLong.empty() : OptionalLong.of(rows);
|
|
||||||
}
|
|
||||||
|
|
||||||
static GenerationStats fromJson(JsonNode node) {
|
|
||||||
JsonFields.requiredObject(node, CONTEXT);
|
|
||||||
return new GenerationStats(
|
|
||||||
JsonFields.requiredLong(node, "generation", CONTEXT),
|
|
||||||
JsonFields.requiredLong(node, "bytes", CONTEXT),
|
|
||||||
JsonFields.optionalLong(node, "rows", CONTEXT));
|
|
||||||
}
|
|
||||||
|
|
||||||
@Override
|
|
||||||
public String toString() {
|
|
||||||
return "GenerationStats{generation=" + generation + ", bytes=" + bytes + ", rows=" + rows + "}";
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,109 +0,0 @@
|
|||||||
/*
|
|
||||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
* you may not use this file except in compliance with the License.
|
|
||||||
* You may obtain a copy of the License at
|
|
||||||
*
|
|
||||||
* http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
*
|
|
||||||
* Unless required by applicable law or agreed to in writing, software
|
|
||||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
* See the License for the specific language governing permissions and
|
|
||||||
* limitations under the License.
|
|
||||||
*/
|
|
||||||
package com.lancedb;
|
|
||||||
|
|
||||||
import com.fasterxml.jackson.databind.JsonNode;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Strict readers for decoding LanceDB JSON responses.
|
|
||||||
*
|
|
||||||
* <p>Every reader fails closed: a missing, null, or wrong-typed field throws rather than
|
|
||||||
* defaulting. That mirrors the serde decoding the Rust client applies to the same payloads in
|
|
||||||
* {@code rust/lancedb/src/table/lsm_stats.rs}, where a required field has no default and a
|
|
||||||
* malformed response is an error rather than a zero.
|
|
||||||
*
|
|
||||||
* <p>The alternative — Jackson's {@code path()}, which yields a missing node that reads as an empty
|
|
||||||
* array or a zero — is unsafe here because {@link LanceDbTableLsm#checkpointLsm()} decides
|
|
||||||
* convergence from these numbers. A defaulted {@code generations} array is indistinguishable from a
|
|
||||||
* drained one, so a malformed response would report a checkpoint that never happened.
|
|
||||||
*/
|
|
||||||
final class JsonFields {
|
|
||||||
private JsonFields() {}
|
|
||||||
|
|
||||||
/** The node itself, once confirmed to be a JSON object. */
|
|
||||||
static JsonNode requiredObject(JsonNode node, String context) {
|
|
||||||
if (node == null || !node.isObject()) {
|
|
||||||
throw new IllegalStateException(context + " is not a JSON object: " + node);
|
|
||||||
}
|
|
||||||
return node;
|
|
||||||
}
|
|
||||||
|
|
||||||
static String requiredText(JsonNode owner, String field, String context) {
|
|
||||||
JsonNode value = required(owner, field, context);
|
|
||||||
if (!value.isTextual()) {
|
|
||||||
throw new IllegalStateException(fieldIs(context, field, "a string", value));
|
|
||||||
}
|
|
||||||
return value.asText();
|
|
||||||
}
|
|
||||||
|
|
||||||
static long requiredLong(JsonNode owner, String field, String context) {
|
|
||||||
JsonNode value = required(owner, field, context);
|
|
||||||
if (!value.isIntegralNumber()) {
|
|
||||||
throw new IllegalStateException(fieldIs(context, field, "an integer", value));
|
|
||||||
}
|
|
||||||
return value.asLong();
|
|
||||||
}
|
|
||||||
|
|
||||||
static boolean requiredBoolean(JsonNode owner, String field, String context) {
|
|
||||||
JsonNode value = required(owner, field, context);
|
|
||||||
if (!value.isBoolean()) {
|
|
||||||
throw new IllegalStateException(fieldIs(context, field, "a boolean", value));
|
|
||||||
}
|
|
||||||
return value.asBoolean();
|
|
||||||
}
|
|
||||||
|
|
||||||
static JsonNode requiredArray(JsonNode owner, String field, String context) {
|
|
||||||
JsonNode value = required(owner, field, context);
|
|
||||||
if (!value.isArray()) {
|
|
||||||
throw new IllegalStateException(fieldIs(context, field, "an array", value));
|
|
||||||
}
|
|
||||||
return value;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Null when the field is absent or JSON null, mirroring a serde {@code Option}. */
|
|
||||||
static Long optionalLong(JsonNode owner, String field, String context) {
|
|
||||||
JsonNode value = owner.get(field);
|
|
||||||
if (value == null || value.isNull()) {
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
if (!value.isIntegralNumber()) {
|
|
||||||
throw new IllegalStateException(fieldIs(context, field, "an integer", value));
|
|
||||||
}
|
|
||||||
return value.asLong();
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Null when the field is absent or JSON null, mirroring a serde {@code Option}. */
|
|
||||||
static JsonNode optionalArray(JsonNode owner, String field, String context) {
|
|
||||||
JsonNode value = owner.get(field);
|
|
||||||
if (value == null || value.isNull()) {
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
if (!value.isArray()) {
|
|
||||||
throw new IllegalStateException(fieldIs(context, field, "an array", value));
|
|
||||||
}
|
|
||||||
return value;
|
|
||||||
}
|
|
||||||
|
|
||||||
private static JsonNode required(JsonNode owner, String field, String context) {
|
|
||||||
JsonNode value = owner.get(field);
|
|
||||||
if (value == null || value.isNull()) {
|
|
||||||
throw new IllegalStateException(context + " is missing required field '" + field + "'");
|
|
||||||
}
|
|
||||||
return value;
|
|
||||||
}
|
|
||||||
|
|
||||||
private static String fieldIs(String context, String field, String expected, JsonNode value) {
|
|
||||||
return context + " field '" + field + "' is not " + expected + ": " + value;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -136,48 +136,29 @@ public class LanceDbNamespaceClientBuilder {
|
|||||||
* @throws IllegalStateException if required parameters are missing
|
* @throws IllegalStateException if required parameters are missing
|
||||||
*/
|
*/
|
||||||
public LanceNamespace build() {
|
public LanceNamespace build() {
|
||||||
validate();
|
// Validate required fields
|
||||||
|
|
||||||
// Build configuration map
|
|
||||||
Map<String, String> config = new HashMap<>(additionalConfig);
|
|
||||||
config.put("header.x-lancedb-database", database);
|
|
||||||
config.put("header.x-api-key", apiKey);
|
|
||||||
config.put("uri", resolveUri());
|
|
||||||
|
|
||||||
return LanceNamespace.connect("rest", config, null);
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Build a {@link LanceDbRestClient} for the same endpoint.
|
|
||||||
*
|
|
||||||
* <p>Needed only for LanceDB routes that the Lance Namespace specification does not cover — the
|
|
||||||
* MemWAL LSM write path, reached through {@link LanceDbTableLsm}. Every other table operation
|
|
||||||
* belongs on the {@link LanceNamespace} from {@link #build()}.
|
|
||||||
*
|
|
||||||
* <p>The returned client owns an HTTP connection pool; close it when you are done with it.
|
|
||||||
*
|
|
||||||
* @return A configured LanceDbRestClient
|
|
||||||
* @throws IllegalStateException if required parameters are missing
|
|
||||||
*/
|
|
||||||
public LanceDbRestClient buildRestClient() {
|
|
||||||
validate();
|
|
||||||
return new LanceDbRestClient(resolveUri(), apiKey, database);
|
|
||||||
}
|
|
||||||
|
|
||||||
private void validate() {
|
|
||||||
if (apiKey == null) {
|
if (apiKey == null) {
|
||||||
throw new IllegalStateException("API key is required");
|
throw new IllegalStateException("API key is required");
|
||||||
}
|
}
|
||||||
if (database == null) {
|
if (database == null) {
|
||||||
throw new IllegalStateException("Database is required");
|
throw new IllegalStateException("Database is required");
|
||||||
}
|
}
|
||||||
}
|
|
||||||
|
|
||||||
/** The custom endpoint when set, else the LanceDB Cloud URL for this database and region. */
|
// Build configuration map
|
||||||
private String resolveUri() {
|
Map<String, String> config = new HashMap<>(additionalConfig);
|
||||||
|
config.put("header.x-lancedb-database", database);
|
||||||
|
config.put("header.x-api-key", apiKey);
|
||||||
|
|
||||||
|
// Determine base URL
|
||||||
|
String uri;
|
||||||
if (endpoint.isPresent()) {
|
if (endpoint.isPresent()) {
|
||||||
return endpoint.get();
|
uri = endpoint.get();
|
||||||
|
} else {
|
||||||
|
String effectiveRegion = region.orElse(DEFAULT_REGION);
|
||||||
|
uri = String.format(CLOUD_URL_PATTERN, database, effectiveRegion);
|
||||||
}
|
}
|
||||||
return String.format(CLOUD_URL_PATTERN, database, region.orElse(DEFAULT_REGION));
|
config.put("uri", uri);
|
||||||
|
|
||||||
|
return LanceNamespace.connect("rest", config, null);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,119 +0,0 @@
|
|||||||
/*
|
|
||||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
* you may not use this file except in compliance with the License.
|
|
||||||
* You may obtain a copy of the License at
|
|
||||||
*
|
|
||||||
* http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
*
|
|
||||||
* Unless required by applicable law or agreed to in writing, software
|
|
||||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
* See the License for the specific language governing permissions and
|
|
||||||
* limitations under the License.
|
|
||||||
*/
|
|
||||||
package com.lancedb;
|
|
||||||
|
|
||||||
import com.fasterxml.jackson.databind.JsonNode;
|
|
||||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
|
||||||
import org.apache.hc.client5.http.classic.methods.HttpPost;
|
|
||||||
import org.apache.hc.client5.http.impl.classic.CloseableHttpClient;
|
|
||||||
import org.apache.hc.client5.http.impl.classic.HttpClients;
|
|
||||||
import org.apache.hc.core5.http.ContentType;
|
|
||||||
import org.apache.hc.core5.http.io.entity.EntityUtils;
|
|
||||||
import org.apache.hc.core5.http.io.entity.StringEntity;
|
|
||||||
|
|
||||||
import java.io.Closeable;
|
|
||||||
import java.io.IOException;
|
|
||||||
import java.io.UncheckedIOException;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Minimal HTTP client for LanceDB Cloud and Enterprise routes that the Lance Namespace
|
|
||||||
* specification does not cover.
|
|
||||||
*
|
|
||||||
* <p>Most table operations reach LanceDB through {@link org.lance.namespace.LanceNamespace}, which
|
|
||||||
* is generated from the namespace spec. A handful of routes — the MemWAL LSM write path in
|
|
||||||
* particular — are served by the same endpoint but are not part of that spec, so they are issued
|
|
||||||
* directly here. See {@link LanceDbTableLsm}.
|
|
||||||
*
|
|
||||||
* <p>Obtain one from {@link LanceDbNamespaceClientBuilder#buildRestClient()}.
|
|
||||||
*/
|
|
||||||
public class LanceDbRestClient implements Closeable {
|
|
||||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
|
||||||
|
|
||||||
private final String baseUri;
|
|
||||||
private final String apiKey;
|
|
||||||
private final String database;
|
|
||||||
private final CloseableHttpClient http;
|
|
||||||
|
|
||||||
LanceDbRestClient(String baseUri, String apiKey, String database) {
|
|
||||||
this.baseUri = baseUri.endsWith("/") ? baseUri.substring(0, baseUri.length() - 1) : baseUri;
|
|
||||||
this.apiKey = apiKey;
|
|
||||||
this.database = database;
|
|
||||||
// Automatic retries off, deliberately. The default strategy retries 429 and 503 —
|
|
||||||
// exactly the two statuses LanceDbTableLsm.checkpointLsm() acts on — which would
|
|
||||||
// silently double its explicit retry budget and would also retry compact_lsm in
|
|
||||||
// place, where the loop is designed to fall through to a fresh stats poll instead.
|
|
||||||
// The checkpoint loop owns the 421/429/503 transitions; the transport must not.
|
|
||||||
this.http = HttpClients.custom().disableAutomaticRetries().build();
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* POST {@code path}, sending {@code body} as JSON when it is non-null.
|
|
||||||
*
|
|
||||||
* @param path Absolute request path, beginning with {@code /}.
|
|
||||||
* @param body Object to serialize as the request body, or null to send no body.
|
|
||||||
* @return The parsed response body, or null when the response carried no content.
|
|
||||||
* @throws HttpException if the server returned a non-2xx status.
|
|
||||||
*/
|
|
||||||
public JsonNode post(String path, Object body) {
|
|
||||||
HttpPost request = new HttpPost(baseUri + path);
|
|
||||||
request.setHeader("x-api-key", apiKey);
|
|
||||||
request.setHeader("x-lancedb-database", database);
|
|
||||||
try {
|
|
||||||
if (body != null) {
|
|
||||||
request.setEntity(
|
|
||||||
new StringEntity(MAPPER.writeValueAsString(body), ContentType.APPLICATION_JSON));
|
|
||||||
}
|
|
||||||
return http.execute(
|
|
||||||
request,
|
|
||||||
response -> {
|
|
||||||
String text =
|
|
||||||
response.getEntity() == null ? "" : EntityUtils.toString(response.getEntity());
|
|
||||||
int status = response.getCode();
|
|
||||||
if (status < 200 || status >= 300) {
|
|
||||||
throw new HttpException(status, "LanceDB request to " + path + " failed: " + text);
|
|
||||||
}
|
|
||||||
return text.isEmpty() ? null : MAPPER.readTree(text);
|
|
||||||
});
|
|
||||||
} catch (IOException e) {
|
|
||||||
throw new UncheckedIOException("LanceDB request to " + path + " failed", e);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
@Override
|
|
||||||
public void close() throws IOException {
|
|
||||||
http.close();
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* A non-2xx response.
|
|
||||||
*
|
|
||||||
* <p>The status is exposed because callers act on it: {@link LanceDbTableLsm#checkpointLsm()}
|
|
||||||
* treats 429 and 503 as retryable and 421 as a lost node claim.
|
|
||||||
*/
|
|
||||||
public static class HttpException extends RuntimeException {
|
|
||||||
private static final long serialVersionUID = 1L;
|
|
||||||
|
|
||||||
private final int statusCode;
|
|
||||||
|
|
||||||
public HttpException(int statusCode, String message) {
|
|
||||||
super(message);
|
|
||||||
this.statusCode = statusCode;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** The HTTP status the failed response carried. */
|
|
||||||
public int statusCode() {
|
|
||||||
return statusCode;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,394 +0,0 @@
|
|||||||
/*
|
|
||||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
* you may not use this file except in compliance with the License.
|
|
||||||
* You may obtain a copy of the License at
|
|
||||||
*
|
|
||||||
* http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
*
|
|
||||||
* Unless required by applicable law or agreed to in writing, software
|
|
||||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
* See the License for the specific language governing permissions and
|
|
||||||
* limitations under the License.
|
|
||||||
*/
|
|
||||||
package com.lancedb;
|
|
||||||
|
|
||||||
import com.fasterxml.jackson.databind.JsonNode;
|
|
||||||
|
|
||||||
import java.util.HashMap;
|
|
||||||
import java.util.LinkedHashMap;
|
|
||||||
import java.util.Map;
|
|
||||||
import java.util.Optional;
|
|
||||||
import java.util.OptionalLong;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* The MemWAL LSM write path for one LanceDB Cloud or Enterprise table.
|
|
||||||
*
|
|
||||||
* <p>Installing an {@link LsmWriteSpec} routes {@code mergeInsert} upserts through Lance's MemWAL —
|
|
||||||
* an LSM-style append — instead of the standard merge path. Rows land in an in-memory memtable,
|
|
||||||
* seal into L0 generations, and are merged into the base table by compaction.
|
|
||||||
*
|
|
||||||
* <p>These routes are not part of the Lance Namespace specification, so they are issued directly
|
|
||||||
* rather than through {@link org.lance.namespace.LanceNamespace}.
|
|
||||||
*
|
|
||||||
* <pre>{@code
|
|
||||||
* LanceDbRestClient client = LanceDbNamespaceClientBuilder.newBuilder()
|
|
||||||
* .apiKey("your_lancedb_cloud_api_key")
|
|
||||||
* .database("your_database_name")
|
|
||||||
* .buildRestClient();
|
|
||||||
*
|
|
||||||
* LanceDbTableLsm lsm = new LanceDbTableLsm(client, "my_table");
|
|
||||||
* lsm.setLsmWriteSpec(LsmWriteSpec.bucket("id", 16));
|
|
||||||
* // ... merge_insert traffic ...
|
|
||||||
* lsm.checkpointLsm();
|
|
||||||
* }</pre>
|
|
||||||
*/
|
|
||||||
public class LanceDbTableLsm {
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Interval between {@code get_lsm_stats} polls during a checkpoint. One interval is roughly one
|
|
||||||
* compaction pass, the granularity at which the answer can change.
|
|
||||||
*/
|
|
||||||
private static final long POLL_INTERVAL_MS = 5_000L;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Cap on re-issues from {@code flushLsm} after a 421, so a crash-looping node cannot turn flush →
|
|
||||||
* compact → 421 → flush into a spin.
|
|
||||||
*
|
|
||||||
* <p>Deliberately not shared with {@link #MAX_RETRIES}: a claim that keeps evaporating is a
|
|
||||||
* broken node, while contention is routine and wants a real budget.
|
|
||||||
*/
|
|
||||||
private static final int MAX_REISSUES = 3;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Retryable faults tolerated on a <em>single</em> request, reset on every success — scattered
|
|
||||||
* contention across a long checkpoint must not accumulate toward a cap.
|
|
||||||
*/
|
|
||||||
private static final int MAX_RETRIES = 8;
|
|
||||||
|
|
||||||
private static final long RETRY_BACKOFF_BASE_MS = 100L;
|
|
||||||
private static final long RETRY_BACKOFF_MAX_MS = 5_000L;
|
|
||||||
|
|
||||||
private final LanceDbRestClient client;
|
|
||||||
private final String tableIdentifier;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Bind the LSM routes for one table.
|
|
||||||
*
|
|
||||||
* @param client Transport for the LanceDB endpoint.
|
|
||||||
* @param tableIdentifier The table's full identifier, {@code $}-delimited when it sits inside a
|
|
||||||
* namespace, such as {@code analytics$events}.
|
|
||||||
*/
|
|
||||||
public LanceDbTableLsm(LanceDbRestClient client, String tableIdentifier) {
|
|
||||||
if (client == null) {
|
|
||||||
throw new IllegalArgumentException("Client cannot be null");
|
|
||||||
}
|
|
||||||
if (tableIdentifier == null || tableIdentifier.trim().isEmpty()) {
|
|
||||||
throw new IllegalArgumentException("Table identifier cannot be null or empty");
|
|
||||||
}
|
|
||||||
this.client = client;
|
|
||||||
this.tableIdentifier = tableIdentifier;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Install an {@link LsmWriteSpec} on this table, selecting the MemWAL LSM write path for future
|
|
||||||
* {@code mergeInsert} calls.
|
|
||||||
*
|
|
||||||
* <p>All variants require the table to have an unenforced primary key; bucket sharding
|
|
||||||
* additionally requires it to be the single column being bucketed.
|
|
||||||
*/
|
|
||||||
public void setLsmWriteSpec(LsmWriteSpec spec) {
|
|
||||||
if (spec == null) {
|
|
||||||
throw new IllegalArgumentException("Spec cannot be null");
|
|
||||||
}
|
|
||||||
client.post(route("set_lsm_write_spec"), spec.toRequestBody());
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Remove the {@link LsmWriteSpec} from this table, reverting to the standard {@code mergeInsert}
|
|
||||||
* write path.
|
|
||||||
*
|
|
||||||
* <p>Errors if no spec is currently set.
|
|
||||||
*/
|
|
||||||
public void unsetLsmWriteSpec() {
|
|
||||||
client.post(route("unset_lsm_write_spec"), null);
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Read the {@link LsmWriteSpec} currently installed on this table.
|
|
||||||
*
|
|
||||||
* <p>Empty when the LSM write path is not enabled. The returned spec mirrors what was installed,
|
|
||||||
* except that {@link LsmWriteSpec#maintainedIndexes()} always reports the concrete list resolved
|
|
||||||
* when the spec was set — a null selection never round-trips.
|
|
||||||
*/
|
|
||||||
public Optional<LsmWriteSpec> getLsmWriteSpec() {
|
|
||||||
JsonNode response = client.post(route("get_lsm_write_spec"), null);
|
|
||||||
if (response == null || !response.hasNonNull("lsm_write_spec")) {
|
|
||||||
return Optional.empty();
|
|
||||||
}
|
|
||||||
return Optional.of(LsmWriteSpec.fromJson(response.get("lsm_write_spec")));
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Seal every bucket's active memtable into a new L0 generation.
|
|
||||||
*
|
|
||||||
* <p>Returns once the seal is committed. Sealing an empty memtable is a no-op, so this is safe to
|
|
||||||
* call repeatedly.
|
|
||||||
*/
|
|
||||||
public void flushLsm() {
|
|
||||||
client.post(route("flush_lsm"), null);
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Trigger a background L0 → base compaction pass per bucket.
|
|
||||||
*
|
|
||||||
* <p>Returns once the passes are <em>dispatched</em>, not once they finish — watch {@link
|
|
||||||
* #getLsmStats}, or use {@link #checkpointLsm} to wait for convergence.
|
|
||||||
*/
|
|
||||||
public void compactLsm() {
|
|
||||||
client.post(route("compact_lsm"), null);
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Read live per-bucket LSM state.
|
|
||||||
*
|
|
||||||
* <p>Answers "how far behind is my fresh tier", "which bucket is hot", and "why is my fresh-tier
|
|
||||||
* vector search brute-force". Mutates no table state.
|
|
||||||
*
|
|
||||||
* <p>Empty only when the LSM write path is not enabled — that is, when the server sends an absent
|
|
||||||
* or null {@code lsm_stats}. A stats object that is present is decoded strictly, and a malformed
|
|
||||||
* one throws rather than decoding to something empty, because {@link #checkpointLsm} reads
|
|
||||||
* convergence out of these numbers and cannot tell a defaulted array from a drained one.
|
|
||||||
*
|
|
||||||
* @param includeGenerationRows Also count rows per L0 generation. Off by default because each
|
|
||||||
* count opens an uncached Lance dataset.
|
|
||||||
* @throws IllegalStateException if the response is absent or does not decode.
|
|
||||||
*/
|
|
||||||
public Optional<LsmStats> getLsmStats(boolean includeGenerationRows) {
|
|
||||||
Map<String, Object> body = new LinkedHashMap<String, Object>();
|
|
||||||
body.put("include_generation_rows", includeGenerationRows);
|
|
||||||
JsonNode response = client.post(route("get_lsm_stats"), body);
|
|
||||||
if (response == null) {
|
|
||||||
throw new IllegalStateException("get_lsm_stats returned an empty response body");
|
|
||||||
}
|
|
||||||
JsonNode stats = response.get("lsm_stats");
|
|
||||||
if (stats == null || stats.isNull()) {
|
|
||||||
return Optional.empty();
|
|
||||||
}
|
|
||||||
return Optional.of(LsmStats.fromJson(stats));
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Equivalent to {@code getLsmStats(false)}. */
|
|
||||||
public Optional<LsmStats> getLsmStats() {
|
|
||||||
return getLsmStats(false);
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Converge this table's LSM write path into its base table.
|
|
||||||
*
|
|
||||||
* <p>Seals once, fixes a target watermark from the resulting L0, then triggers compaction and
|
|
||||||
* polls until that L0 is gone. The target set is fixed at the start, so generations created
|
|
||||||
* <em>during</em> the checkpoint are ignored — that is what lets it terminate under write load,
|
|
||||||
* and what makes it best-effort: it converges the fresh tier as of some instant. Idempotent,
|
|
||||||
* abandonable at any point, safe on a cadence.
|
|
||||||
*
|
|
||||||
* <p>The loop runs here, not on the server: {@link #compactLsm} dispatches a pass and returns, so
|
|
||||||
* nothing holds a socket and a client can vanish mid-operation with nothing to reconcile.
|
|
||||||
* Completion is read from generation numbers in the shard manifest — durable state, unlike a
|
|
||||||
* count in a compact response, which a concurrent write invalidates.
|
|
||||||
*
|
|
||||||
* <p>No liveness bound — the caller owns the deadline. The compactor pool is shared across
|
|
||||||
* tables, so a checkpoint queued behind unrelated work looks exactly like one that is merging.
|
|
||||||
*/
|
|
||||||
public void checkpointLsm() {
|
|
||||||
for (int reissue = 0; reissue <= MAX_REISSUES; reissue++) {
|
|
||||||
// The seal turns everything written before this call into a generation, so the
|
|
||||||
// watermark has to be read after it. Idempotent: sealing an empty memtable is a
|
|
||||||
// no-op, so a re-issue does not churn empty generations.
|
|
||||||
if (issueVoid(this::flushLsm)) {
|
|
||||||
backoff(reissue);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
Attempt<Optional<LsmStats>> stats = issue(() -> getLsmStats(false));
|
|
||||||
if (stats.lostClaim) {
|
|
||||||
backoff(reissue);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
if (!stats.value.isPresent()) {
|
|
||||||
// Not WAL-backed; flushLsm would have errored first but for a race.
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
Map<String, Long> targets = newestGenerations(stats.value.get());
|
|
||||||
if (targets.isEmpty()) {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (drainToTargets(targets)) {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
backoff(reissue);
|
|
||||||
}
|
|
||||||
throw new IllegalStateException(
|
|
||||||
"checkpointLsm: the owning node kept losing its claim; re-issued from flush the maximum "
|
|
||||||
+ "number of times");
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Trigger and poll until no bucket holds a generation at or below its target.
|
|
||||||
*
|
|
||||||
* @return true when the drain finished, false when the table needs re-claiming from flush.
|
|
||||||
*/
|
|
||||||
private boolean drainToTargets(Map<String, Long> targets) {
|
|
||||||
while (true) {
|
|
||||||
Attempt<Optional<LsmStats>> stats = issue(() -> getLsmStats(false));
|
|
||||||
if (stats.lostClaim) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
if (!stats.value.isPresent()) {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
// `compacting` is the bucket's compaction latch, held from dispatch until the pass
|
|
||||||
// ends — including while it waits on a pod-wide permit. So it answers one question
|
|
||||||
// only: do not pile on. Buckets with nothing outstanding are skipped, not counted
|
|
||||||
// as idle.
|
|
||||||
long outstanding = 0;
|
|
||||||
boolean allCompacting = true;
|
|
||||||
for (BucketStats bucket : stats.value.get().buckets()) {
|
|
||||||
Long target = targets.get(bucket.shardId());
|
|
||||||
if (target == null) {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
long remaining = bucket.outstandingGenerations(target);
|
|
||||||
if (remaining > 0) {
|
|
||||||
outstanding += remaining;
|
|
||||||
allCompacting &= bucket.compacting();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (outstanding == 0) {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (!allCompacting) {
|
|
||||||
try {
|
|
||||||
compactLsm();
|
|
||||||
} catch (LanceDbRestClient.HttpException e) {
|
|
||||||
if (isLostClaim(e)) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
if (!isRetryable(e)) {
|
|
||||||
throw e;
|
|
||||||
}
|
|
||||||
// A 429 here means the server could latch no bucket at all, which the poll
|
|
||||||
// above already handles. Not retried in place: the latch it would contend for
|
|
||||||
// is the one doing the work, so fall through and re-read — POLL_INTERVAL_MS is
|
|
||||||
// the backoff.
|
|
||||||
}
|
|
||||||
}
|
|
||||||
sleep(POLL_INTERVAL_MS);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/** The newest generation held by each bucket, skipping buckets holding none. */
|
|
||||||
private static Map<String, Long> newestGenerations(LsmStats stats) {
|
|
||||||
Map<String, Long> targets = new HashMap<String, Long>();
|
|
||||||
for (BucketStats bucket : stats.buckets()) {
|
|
||||||
OptionalLong newest = bucket.newestGeneration();
|
|
||||||
if (newest.isPresent()) {
|
|
||||||
targets.put(bucket.shardId(), newest.getAsLong());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return targets;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* 429 (latch held, pool saturated, or the pod replaying its WAL) and 503 (a draining node, or a
|
|
||||||
* proxy between here and it).
|
|
||||||
*/
|
|
||||||
private static boolean isRetryable(LanceDbRestClient.HttpException e) {
|
|
||||||
return e.statusCode() == 429 || e.statusCode() == 503;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* 421: the owning node holds no claim. Only {@code flush} re-claims and replays, so this cannot
|
|
||||||
* be retried in place — the caller has to start over.
|
|
||||||
*/
|
|
||||||
private static boolean isLostClaim(LanceDbRestClient.HttpException e) {
|
|
||||||
return e.statusCode() == 421;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Issue one LSM request, retrying in place while the fault is retryable.
|
|
||||||
*
|
|
||||||
* <p>The two recoverable faults have separate budgets: contention clears on its own and retries
|
|
||||||
* here against {@link #MAX_RETRIES}, while a 421 needs {@code flush} to re-claim, which only the
|
|
||||||
* caller can drive.
|
|
||||||
*
|
|
||||||
* <p>An exhausted budget propagates the last error as itself rather than a synthesized one — "429
|
|
||||||
* after nine tries" beats "checkpoint failed".
|
|
||||||
*/
|
|
||||||
private static <T> Attempt<T> issue(Call<T> call) {
|
|
||||||
int retries = 0;
|
|
||||||
while (true) {
|
|
||||||
try {
|
|
||||||
return new Attempt<T>(call.run(), false);
|
|
||||||
} catch (LanceDbRestClient.HttpException e) {
|
|
||||||
if (isLostClaim(e)) {
|
|
||||||
return new Attempt<T>(null, true);
|
|
||||||
}
|
|
||||||
if (!isRetryable(e) || retries >= MAX_RETRIES) {
|
|
||||||
throw e;
|
|
||||||
}
|
|
||||||
backoff(retries);
|
|
||||||
retries++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/** {@link #issue} for a call with no return value. Returns true when the claim was lost. */
|
|
||||||
private static boolean issueVoid(Runnable call) {
|
|
||||||
return issue(
|
|
||||||
() -> {
|
|
||||||
call.run();
|
|
||||||
return Boolean.TRUE;
|
|
||||||
})
|
|
||||||
.lostClaim;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Sleep before re-issuing a retryable request. Doubles up to {@link #RETRY_BACKOFF_MAX_MS}. */
|
|
||||||
private static void backoff(int attempt) {
|
|
||||||
long delay = RETRY_BACKOFF_BASE_MS << Math.min(attempt, 8);
|
|
||||||
sleep(Math.min(delay, RETRY_BACKOFF_MAX_MS));
|
|
||||||
}
|
|
||||||
|
|
||||||
private static void sleep(long millis) {
|
|
||||||
try {
|
|
||||||
Thread.sleep(millis);
|
|
||||||
} catch (InterruptedException e) {
|
|
||||||
Thread.currentThread().interrupt();
|
|
||||||
throw new IllegalStateException("Interrupted while waiting on the LSM checkpoint", e);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
private String route(String operation) {
|
|
||||||
return "/v1/table/" + tableIdentifier + "/" + operation + "/";
|
|
||||||
}
|
|
||||||
|
|
||||||
/** What one LSM request produced: its value, or word that the owning node holds no claim. */
|
|
||||||
private static final class Attempt<T> {
|
|
||||||
private final T value;
|
|
||||||
private final boolean lostClaim;
|
|
||||||
|
|
||||||
private Attempt(T value, boolean lostClaim) {
|
|
||||||
this.value = value;
|
|
||||||
this.lostClaim = lostClaim;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
@FunctionalInterface
|
|
||||||
private interface Call<T> {
|
|
||||||
T run();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,56 +0,0 @@
|
|||||||
/*
|
|
||||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
* you may not use this file except in compliance with the License.
|
|
||||||
* You may obtain a copy of the License at
|
|
||||||
*
|
|
||||||
* http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
*
|
|
||||||
* Unless required by applicable law or agreed to in writing, software
|
|
||||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
* See the License for the specific language governing permissions and
|
|
||||||
* limitations under the License.
|
|
||||||
*/
|
|
||||||
package com.lancedb;
|
|
||||||
|
|
||||||
import com.fasterxml.jackson.databind.JsonNode;
|
|
||||||
|
|
||||||
import java.util.ArrayList;
|
|
||||||
import java.util.Collections;
|
|
||||||
import java.util.List;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Live per-bucket LSM state, as returned by {@link LanceDbTableLsm#getLsmStats()}.
|
|
||||||
*
|
|
||||||
* <p>Nothing here is derived: sums and differences (total L0 bytes, WAL lag) are the caller's to
|
|
||||||
* compute. There is no "LSM is off" shape — that case is an empty {@link java.util.Optional},
|
|
||||||
* because a stats object of zeros would read as measurements.
|
|
||||||
*/
|
|
||||||
public class LsmStats {
|
|
||||||
private static final String CONTEXT = "lsm stats";
|
|
||||||
|
|
||||||
private final List<BucketStats> buckets;
|
|
||||||
|
|
||||||
LsmStats(List<BucketStats> buckets) {
|
|
||||||
this.buckets = Collections.unmodifiableList(buckets);
|
|
||||||
}
|
|
||||||
|
|
||||||
/** One entry per bucket. */
|
|
||||||
public List<BucketStats> buckets() {
|
|
||||||
return buckets;
|
|
||||||
}
|
|
||||||
|
|
||||||
static LsmStats fromJson(JsonNode node) {
|
|
||||||
JsonFields.requiredObject(node, CONTEXT);
|
|
||||||
List<BucketStats> buckets = new ArrayList<BucketStats>();
|
|
||||||
for (JsonNode bucket : JsonFields.requiredArray(node, "buckets", CONTEXT)) {
|
|
||||||
buckets.add(BucketStats.fromJson(bucket));
|
|
||||||
}
|
|
||||||
return new LsmStats(buckets);
|
|
||||||
}
|
|
||||||
|
|
||||||
@Override
|
|
||||||
public String toString() {
|
|
||||||
return "LsmStats{buckets=" + buckets + "}";
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,260 +0,0 @@
|
|||||||
/*
|
|
||||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
* you may not use this file except in compliance with the License.
|
|
||||||
* You may obtain a copy of the License at
|
|
||||||
*
|
|
||||||
* http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
*
|
|
||||||
* Unless required by applicable law or agreed to in writing, software
|
|
||||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
* See the License for the specific language governing permissions and
|
|
||||||
* limitations under the License.
|
|
||||||
*/
|
|
||||||
package com.lancedb;
|
|
||||||
|
|
||||||
import com.fasterxml.jackson.databind.JsonNode;
|
|
||||||
|
|
||||||
import java.util.ArrayList;
|
|
||||||
import java.util.Collections;
|
|
||||||
import java.util.HashMap;
|
|
||||||
import java.util.LinkedHashMap;
|
|
||||||
import java.util.List;
|
|
||||||
import java.util.Map;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Specification selecting Lance's MemWAL LSM-style write path for {@code mergeInsert}.
|
|
||||||
*
|
|
||||||
* <p>Construct via {@link #bucket}, {@link #identity}, or {@link #unsharded}, then optionally chain
|
|
||||||
* {@link #withMaintainedIndexes} and {@link #withWriterConfigDefaults}. Install it with {@link
|
|
||||||
* LanceDbTableLsm#setLsmWriteSpec} and remove it with {@link LanceDbTableLsm#unsetLsmWriteSpec}.
|
|
||||||
*
|
|
||||||
* <p>This is deliberately not {@code org.lance.memwal.InitializeMemWalParams}. That type is Lance's
|
|
||||||
* own, and its maintained-index default is the opposite of this one: it defaults to maintaining
|
|
||||||
* <em>nothing</em>, while a fresh spec here maintains <em>every</em> index. It also cannot express
|
|
||||||
* the null that asks the server to resolve the set.
|
|
||||||
*/
|
|
||||||
public class LsmWriteSpec {
|
|
||||||
|
|
||||||
/** How writes are routed to MemWAL shards. */
|
|
||||||
public enum Sharding {
|
|
||||||
/** Hash-bucket writes by a scalar column. */
|
|
||||||
BUCKET("bucket"),
|
|
||||||
/** Shard by the raw value of a scalar column. */
|
|
||||||
IDENTITY("identity"),
|
|
||||||
/** Route every write to a single shard. */
|
|
||||||
UNSHARDED("unsharded");
|
|
||||||
|
|
||||||
private final String wireName;
|
|
||||||
|
|
||||||
Sharding(String wireName) {
|
|
||||||
this.wireName = wireName;
|
|
||||||
}
|
|
||||||
|
|
||||||
String wireName() {
|
|
||||||
return wireName;
|
|
||||||
}
|
|
||||||
|
|
||||||
static Sharding fromWireName(String name) {
|
|
||||||
for (Sharding s : values()) {
|
|
||||||
if (s.wireName.equals(name)) {
|
|
||||||
return s;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
throw new IllegalArgumentException("Unknown sharding mode: " + name);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
private final Sharding sharding;
|
|
||||||
private final String column;
|
|
||||||
private final Integer numBuckets;
|
|
||||||
private final List<String> maintainedIndexes;
|
|
||||||
private final Map<String, String> writerConfigDefaults;
|
|
||||||
|
|
||||||
private LsmWriteSpec(
|
|
||||||
Sharding sharding,
|
|
||||||
String column,
|
|
||||||
Integer numBuckets,
|
|
||||||
List<String> maintainedIndexes,
|
|
||||||
Map<String, String> writerConfigDefaults) {
|
|
||||||
this.sharding = sharding;
|
|
||||||
this.column = column;
|
|
||||||
this.numBuckets = numBuckets;
|
|
||||||
this.maintainedIndexes = maintainedIndexes;
|
|
||||||
this.writerConfigDefaults = writerConfigDefaults;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Hash-bucket sharding by a scalar column, maintaining every index on the table.
|
|
||||||
*
|
|
||||||
* <p>Iceberg-compatible Murmur3-x86-32 (seed 0) is used, so each row's {@code bucket(column,
|
|
||||||
* numBuckets)} value is stable across processes.
|
|
||||||
*
|
|
||||||
* @param column A non-nested column with a supported scalar type.
|
|
||||||
* @param numBuckets The number of buckets, in {@code [1, 1024]}.
|
|
||||||
*/
|
|
||||||
public static LsmWriteSpec bucket(String column, int numBuckets) {
|
|
||||||
if (column == null || column.trim().isEmpty()) {
|
|
||||||
throw new IllegalArgumentException("Column cannot be null or empty");
|
|
||||||
}
|
|
||||||
return new LsmWriteSpec(
|
|
||||||
Sharding.BUCKET, column, numBuckets, null, new HashMap<String, String>());
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Identity sharding — shard by the raw value of {@code column} — maintaining every index on the
|
|
||||||
* table.
|
|
||||||
*
|
|
||||||
* <p>{@code column} must be a deterministic function of the unenforced primary key: every row
|
|
||||||
* with a given primary key must always produce the same {@code column} value, or upserts of that
|
|
||||||
* key can land in different shards and a stale version can win.
|
|
||||||
*/
|
|
||||||
public static LsmWriteSpec identity(String column) {
|
|
||||||
if (column == null || column.trim().isEmpty()) {
|
|
||||||
throw new IllegalArgumentException("Column cannot be null or empty");
|
|
||||||
}
|
|
||||||
return new LsmWriteSpec(Sharding.IDENTITY, column, null, null, new HashMap<String, String>());
|
|
||||||
}
|
|
||||||
|
|
||||||
/** No sharding — every write goes to a single MemWAL shard — maintaining every index. */
|
|
||||||
public static LsmWriteSpec unsharded() {
|
|
||||||
return new LsmWriteSpec(Sharding.UNSHARDED, null, null, null, new HashMap<String, String>());
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Set the indexes the MemWAL keeps up to date as rows are appended.
|
|
||||||
*
|
|
||||||
* <p>Pass {@code null} — the default for a fresh spec — to maintain every index the MemWAL can,
|
|
||||||
* resolved when the spec is installed. That is a snapshot: indexes created later are not
|
|
||||||
* maintained until the spec is unset and set again. Pass an empty list to maintain none.
|
|
||||||
*
|
|
||||||
* <p>Note that {@code null} and the empty list mean opposite things here.
|
|
||||||
*/
|
|
||||||
public LsmWriteSpec withMaintainedIndexes(List<String> maintainedIndexes) {
|
|
||||||
return new LsmWriteSpec(
|
|
||||||
sharding,
|
|
||||||
column,
|
|
||||||
numBuckets,
|
|
||||||
maintainedIndexes == null ? null : new ArrayList<String>(maintainedIndexes),
|
|
||||||
writerConfigDefaults);
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Set default {@code ShardWriter} configuration recorded in the MemWAL index.
|
|
||||||
*
|
|
||||||
* <p>A sparse override map — only the keys you set are recorded. Recognized keys include {@code
|
|
||||||
* durable_write}, {@code max_wal_buffer_size}, {@code max_memtable_size}, {@code
|
|
||||||
* max_memtable_rows}, {@code max_memtable_batches}, {@code manifest_scan_batch_size}, {@code
|
|
||||||
* max_unflushed_memtable_bytes}, and {@code enable_memtable}. Duration knobs carry an {@code _ms}
|
|
||||||
* suffix, such as {@code max_wal_flush_interval_ms}.
|
|
||||||
*/
|
|
||||||
public LsmWriteSpec withWriterConfigDefaults(Map<String, String> writerConfigDefaults) {
|
|
||||||
if (writerConfigDefaults == null) {
|
|
||||||
throw new IllegalArgumentException("writerConfigDefaults cannot be null");
|
|
||||||
}
|
|
||||||
return new LsmWriteSpec(
|
|
||||||
sharding,
|
|
||||||
column,
|
|
||||||
numBuckets,
|
|
||||||
maintainedIndexes,
|
|
||||||
new HashMap<String, String>(writerConfigDefaults));
|
|
||||||
}
|
|
||||||
|
|
||||||
/** How writes are routed to shards. */
|
|
||||||
public Sharding sharding() {
|
|
||||||
return sharding;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** The sharding column for {@link Sharding#BUCKET} and {@link Sharding#IDENTITY}, else null. */
|
|
||||||
public String column() {
|
|
||||||
return column;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** The bucket count for {@link Sharding#BUCKET}, else null. */
|
|
||||||
public Integer numBuckets() {
|
|
||||||
return numBuckets;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* The indexes the MemWAL maintains, or null to have the server resolve every maintainable index
|
|
||||||
* on install. An empty list means none.
|
|
||||||
*/
|
|
||||||
public List<String> maintainedIndexes() {
|
|
||||||
return maintainedIndexes == null ? null : Collections.unmodifiableList(maintainedIndexes);
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Default {@code ShardWriter} configuration recorded in the MemWAL index. */
|
|
||||||
public Map<String, String> writerConfigDefaults() {
|
|
||||||
return Collections.unmodifiableMap(writerConfigDefaults);
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Render this spec as the {@code set_lsm_write_spec} request body. */
|
|
||||||
Map<String, Object> toRequestBody() {
|
|
||||||
Map<String, Object> shardingBody = new LinkedHashMap<String, Object>();
|
|
||||||
shardingBody.put("mode", sharding.wireName());
|
|
||||||
if (column != null) {
|
|
||||||
shardingBody.put("column", column);
|
|
||||||
}
|
|
||||||
if (numBuckets != null) {
|
|
||||||
shardingBody.put("num_buckets", numBuckets);
|
|
||||||
}
|
|
||||||
|
|
||||||
Map<String, Object> body = new LinkedHashMap<String, Object>();
|
|
||||||
body.put("sharding", shardingBody);
|
|
||||||
// Null is meaningful: it asks the server to resolve every maintainable index.
|
|
||||||
body.put("maintained_indexes", maintainedIndexes);
|
|
||||||
body.put("writer_config_defaults", writerConfigDefaults);
|
|
||||||
return body;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Rebuild a spec from a {@code get_lsm_write_spec} response body.
|
|
||||||
*
|
|
||||||
* <p>The server always reports a concrete maintained-index list, so a null selection never
|
|
||||||
* round-trips.
|
|
||||||
*/
|
|
||||||
static LsmWriteSpec fromJson(JsonNode node) {
|
|
||||||
JsonNode shardingNode = node.get("sharding");
|
|
||||||
if (shardingNode == null || shardingNode.get("mode") == null) {
|
|
||||||
throw new IllegalStateException("get_lsm_write_spec response has no sharding mode");
|
|
||||||
}
|
|
||||||
Sharding sharding = Sharding.fromWireName(shardingNode.get("mode").asText());
|
|
||||||
|
|
||||||
String column = shardingNode.hasNonNull("column") ? shardingNode.get("column").asText() : null;
|
|
||||||
Integer numBuckets =
|
|
||||||
shardingNode.hasNonNull("num_buckets") ? shardingNode.get("num_buckets").asInt() : null;
|
|
||||||
|
|
||||||
List<String> maintainedIndexes = new ArrayList<String>();
|
|
||||||
JsonNode indexesNode = node.get("maintained_indexes");
|
|
||||||
if (indexesNode != null && indexesNode.isArray()) {
|
|
||||||
for (JsonNode index : indexesNode) {
|
|
||||||
maintainedIndexes.add(index.asText());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
Map<String, String> defaults = new HashMap<String, String>();
|
|
||||||
JsonNode defaultsNode = node.get("writer_config_defaults");
|
|
||||||
if (defaultsNode != null && defaultsNode.isObject()) {
|
|
||||||
defaultsNode
|
|
||||||
.fieldNames()
|
|
||||||
.forEachRemaining(name -> defaults.put(name, defaultsNode.get(name).asText()));
|
|
||||||
}
|
|
||||||
|
|
||||||
return new LsmWriteSpec(sharding, column, numBuckets, maintainedIndexes, defaults);
|
|
||||||
}
|
|
||||||
|
|
||||||
@Override
|
|
||||||
public String toString() {
|
|
||||||
return "LsmWriteSpec{sharding="
|
|
||||||
+ sharding
|
|
||||||
+ ", column="
|
|
||||||
+ column
|
|
||||||
+ ", numBuckets="
|
|
||||||
+ numBuckets
|
|
||||||
+ ", maintainedIndexes="
|
|
||||||
+ maintainedIndexes
|
|
||||||
+ ", writerConfigDefaults="
|
|
||||||
+ writerConfigDefaults
|
|
||||||
+ "}";
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,99 +0,0 @@
|
|||||||
/*
|
|
||||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
* you may not use this file except in compliance with the License.
|
|
||||||
* You may obtain a copy of the License at
|
|
||||||
*
|
|
||||||
* http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
*
|
|
||||||
* Unless required by applicable law or agreed to in writing, software
|
|
||||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
* See the License for the specific language governing permissions and
|
|
||||||
* limitations under the License.
|
|
||||||
*/
|
|
||||||
package com.lancedb;
|
|
||||||
|
|
||||||
import com.fasterxml.jackson.databind.JsonNode;
|
|
||||||
|
|
||||||
import java.util.ArrayList;
|
|
||||||
import java.util.Collections;
|
|
||||||
import java.util.List;
|
|
||||||
|
|
||||||
/** One in-memory memtable. */
|
|
||||||
public class MemtableStats {
|
|
||||||
private static final String CONTEXT = "memtable stats";
|
|
||||||
|
|
||||||
private final long generation;
|
|
||||||
private final long rows;
|
|
||||||
private final long bytes;
|
|
||||||
private final long batches;
|
|
||||||
private final List<String> indexes;
|
|
||||||
|
|
||||||
MemtableStats(long generation, long rows, long bytes, long batches, List<String> indexes) {
|
|
||||||
this.generation = generation;
|
|
||||||
this.rows = rows;
|
|
||||||
this.bytes = bytes;
|
|
||||||
this.batches = batches;
|
|
||||||
this.indexes = Collections.unmodifiableList(indexes);
|
|
||||||
}
|
|
||||||
|
|
||||||
/** The generation this memtable will become once sealed. */
|
|
||||||
public long generation() {
|
|
||||||
return generation;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Rows currently buffered. */
|
|
||||||
public long rows() {
|
|
||||||
return rows;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Estimated in-memory size. */
|
|
||||||
public long bytes() {
|
|
||||||
return bytes;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Record batches currently buffered. */
|
|
||||||
public long batches() {
|
|
||||||
return batches;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Names of the indexes this memtable carries. An absent name is the whole answer to "why is my
|
|
||||||
* fresh-tier search on that column brute-force".
|
|
||||||
*/
|
|
||||||
public List<String> indexes() {
|
|
||||||
return indexes;
|
|
||||||
}
|
|
||||||
|
|
||||||
static MemtableStats fromJson(JsonNode node) {
|
|
||||||
JsonFields.requiredObject(node, CONTEXT);
|
|
||||||
List<String> indexes = new ArrayList<String>();
|
|
||||||
for (JsonNode index : JsonFields.requiredArray(node, "indexes", CONTEXT)) {
|
|
||||||
if (!index.isTextual()) {
|
|
||||||
throw new IllegalStateException(CONTEXT + " has a non-string index name: " + index);
|
|
||||||
}
|
|
||||||
indexes.add(index.asText());
|
|
||||||
}
|
|
||||||
return new MemtableStats(
|
|
||||||
JsonFields.requiredLong(node, "generation", CONTEXT),
|
|
||||||
JsonFields.requiredLong(node, "rows", CONTEXT),
|
|
||||||
JsonFields.requiredLong(node, "bytes", CONTEXT),
|
|
||||||
JsonFields.requiredLong(node, "batches", CONTEXT),
|
|
||||||
indexes);
|
|
||||||
}
|
|
||||||
|
|
||||||
@Override
|
|
||||||
public String toString() {
|
|
||||||
return "MemtableStats{generation="
|
|
||||||
+ generation
|
|
||||||
+ ", rows="
|
|
||||||
+ rows
|
|
||||||
+ ", bytes="
|
|
||||||
+ bytes
|
|
||||||
+ ", batches="
|
|
||||||
+ batches
|
|
||||||
+ ", indexes="
|
|
||||||
+ indexes
|
|
||||||
+ "}";
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,570 +0,0 @@
|
|||||||
/*
|
|
||||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
* you may not use this file except in compliance with the License.
|
|
||||||
* You may obtain a copy of the License at
|
|
||||||
*
|
|
||||||
* http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
*
|
|
||||||
* Unless required by applicable law or agreed to in writing, software
|
|
||||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
* See the License for the specific language governing permissions and
|
|
||||||
* limitations under the License.
|
|
||||||
*/
|
|
||||||
package com.lancedb;
|
|
||||||
|
|
||||||
import com.fasterxml.jackson.databind.JsonNode;
|
|
||||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
|
||||||
import com.sun.net.httpserver.HttpServer;
|
|
||||||
import org.junit.jupiter.api.AfterEach;
|
|
||||||
import org.junit.jupiter.api.BeforeEach;
|
|
||||||
import org.junit.jupiter.api.Test;
|
|
||||||
|
|
||||||
import java.io.ByteArrayOutputStream;
|
|
||||||
import java.io.IOException;
|
|
||||||
import java.io.InputStream;
|
|
||||||
import java.io.UncheckedIOException;
|
|
||||||
import java.net.InetSocketAddress;
|
|
||||||
import java.nio.charset.StandardCharsets;
|
|
||||||
import java.util.ArrayDeque;
|
|
||||||
import java.util.ArrayList;
|
|
||||||
import java.util.Arrays;
|
|
||||||
import java.util.Collections;
|
|
||||||
import java.util.Deque;
|
|
||||||
import java.util.HashMap;
|
|
||||||
import java.util.LinkedHashMap;
|
|
||||||
import java.util.List;
|
|
||||||
import java.util.Map;
|
|
||||||
import java.util.Optional;
|
|
||||||
import java.util.concurrent.ConcurrentHashMap;
|
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.*;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Unit tests for the MemWAL LSM routes, run against a scripted local HTTP server.
|
|
||||||
*
|
|
||||||
* <p>The wire assertions mirror the Rust mocked-endpoint tests in {@code
|
|
||||||
* rust/lancedb/src/remote/table.rs}, which are the contract these routes have to match.
|
|
||||||
*/
|
|
||||||
public class LanceDbTableLsmTest {
|
|
||||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
|
||||||
|
|
||||||
private HttpServer server;
|
|
||||||
private LanceDbRestClient client;
|
|
||||||
private LanceDbTableLsm lsm;
|
|
||||||
|
|
||||||
private final List<String> requestPaths = Collections.synchronizedList(new ArrayList<String>());
|
|
||||||
private final List<String> requestBodies = Collections.synchronizedList(new ArrayList<String>());
|
|
||||||
private final Map<String, Deque<Reply>> replies = new ConcurrentHashMap<String, Deque<Reply>>();
|
|
||||||
|
|
||||||
@BeforeEach
|
|
||||||
public void setUp() throws IOException {
|
|
||||||
start();
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Tear down and restart the scripted server, for a test that scripts several exchanges. */
|
|
||||||
private void setUpFresh() {
|
|
||||||
try {
|
|
||||||
client.close();
|
|
||||||
server.stop(0);
|
|
||||||
requestPaths.clear();
|
|
||||||
requestBodies.clear();
|
|
||||||
replies.clear();
|
|
||||||
start();
|
|
||||||
} catch (IOException e) {
|
|
||||||
throw new UncheckedIOException(e);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
private void start() throws IOException {
|
|
||||||
server = HttpServer.create(new InetSocketAddress("127.0.0.1", 0), 0);
|
|
||||||
server.createContext(
|
|
||||||
"/",
|
|
||||||
exchange -> {
|
|
||||||
String path = exchange.getRequestURI().getPath();
|
|
||||||
requestPaths.add(path);
|
|
||||||
requestBodies.add(readAll(exchange.getRequestBody()));
|
|
||||||
|
|
||||||
Reply reply = nextReply(path);
|
|
||||||
byte[] out = reply.body.getBytes(StandardCharsets.UTF_8);
|
|
||||||
exchange.sendResponseHeaders(reply.status, out.length == 0 ? -1 : out.length);
|
|
||||||
if (out.length > 0) {
|
|
||||||
exchange.getResponseBody().write(out);
|
|
||||||
}
|
|
||||||
exchange.close();
|
|
||||||
});
|
|
||||||
server.start();
|
|
||||||
|
|
||||||
client =
|
|
||||||
LanceDbNamespaceClientBuilder.newBuilder()
|
|
||||||
.apiKey("test-key")
|
|
||||||
.database("test-db")
|
|
||||||
.endpoint("http://127.0.0.1:" + server.getAddress().getPort())
|
|
||||||
.buildRestClient();
|
|
||||||
lsm = new LanceDbTableLsm(client, "my_table");
|
|
||||||
}
|
|
||||||
|
|
||||||
@AfterEach
|
|
||||||
public void tearDown() throws IOException {
|
|
||||||
client.close();
|
|
||||||
server.stop(0);
|
|
||||||
}
|
|
||||||
|
|
||||||
// ===========================================================================
|
|
||||||
// set / unset / get spec
|
|
||||||
// ===========================================================================
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testSetLsmWriteSpecUnsharded() throws Exception {
|
|
||||||
enqueue("set_lsm_write_spec", 200, "");
|
|
||||||
|
|
||||||
lsm.setLsmWriteSpec(LsmWriteSpec.unsharded());
|
|
||||||
|
|
||||||
assertEquals("/v1/table/my_table/set_lsm_write_spec/", requestPaths.get(0));
|
|
||||||
JsonNode body = MAPPER.readTree(requestBodies.get(0));
|
|
||||||
assertEquals("unsharded", body.get("sharding").get("mode").asText());
|
|
||||||
assertFalse(body.get("sharding").has("column"));
|
|
||||||
assertFalse(body.get("sharding").has("num_buckets"));
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testSetLsmWriteSpecBucket() throws Exception {
|
|
||||||
enqueue("set_lsm_write_spec", 200, "");
|
|
||||||
|
|
||||||
lsm.setLsmWriteSpec(
|
|
||||||
LsmWriteSpec.bucket("id", 16).withMaintainedIndexes(Arrays.asList("id_idx")));
|
|
||||||
|
|
||||||
JsonNode body = MAPPER.readTree(requestBodies.get(0));
|
|
||||||
assertEquals("bucket", body.get("sharding").get("mode").asText());
|
|
||||||
assertEquals("id", body.get("sharding").get("column").asText());
|
|
||||||
assertEquals(16, body.get("sharding").get("num_buckets").asInt());
|
|
||||||
assertEquals(1, body.get("maintained_indexes").size());
|
|
||||||
assertEquals("id_idx", body.get("maintained_indexes").get(0).asText());
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testSetLsmWriteSpecIdentity() throws Exception {
|
|
||||||
enqueue("set_lsm_write_spec", 200, "");
|
|
||||||
|
|
||||||
lsm.setLsmWriteSpec(LsmWriteSpec.identity("tenant"));
|
|
||||||
|
|
||||||
JsonNode body = MAPPER.readTree(requestBodies.get(0));
|
|
||||||
assertEquals("identity", body.get("sharding").get("mode").asText());
|
|
||||||
assertEquals("tenant", body.get("sharding").get("column").asText());
|
|
||||||
assertFalse(body.get("sharding").has("num_buckets"));
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* The tri-state that motivated a LanceDB-owned spec type: a null selection asks the server to
|
|
||||||
* resolve every maintainable index, while an empty list asks for none. They must not collapse.
|
|
||||||
*/
|
|
||||||
@Test
|
|
||||||
public void testMaintainedIndexesNullAndEmptyAreDistinctOnTheWire() throws Exception {
|
|
||||||
enqueue("set_lsm_write_spec", 200, "");
|
|
||||||
|
|
||||||
lsm.setLsmWriteSpec(LsmWriteSpec.unsharded());
|
|
||||||
JsonNode fresh = MAPPER.readTree(requestBodies.get(0));
|
|
||||||
assertTrue(fresh.has("maintained_indexes"), "the key must be present");
|
|
||||||
assertTrue(fresh.get("maintained_indexes").isNull(), "a fresh spec sends null, not []");
|
|
||||||
|
|
||||||
lsm.setLsmWriteSpec(
|
|
||||||
LsmWriteSpec.unsharded().withMaintainedIndexes(Collections.<String>emptyList()));
|
|
||||||
JsonNode none = MAPPER.readTree(requestBodies.get(1));
|
|
||||||
assertTrue(none.get("maintained_indexes").isArray());
|
|
||||||
assertEquals(0, none.get("maintained_indexes").size());
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testSetLsmWriteSpecWriterConfigDefaults() throws Exception {
|
|
||||||
enqueue("set_lsm_write_spec", 200, "");
|
|
||||||
|
|
||||||
Map<String, String> defaults = new HashMap<String, String>();
|
|
||||||
defaults.put("max_memtable_rows", "50000");
|
|
||||||
lsm.setLsmWriteSpec(LsmWriteSpec.unsharded().withWriterConfigDefaults(defaults));
|
|
||||||
|
|
||||||
JsonNode body = MAPPER.readTree(requestBodies.get(0));
|
|
||||||
assertEquals("50000", body.get("writer_config_defaults").get("max_memtable_rows").asText());
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testUnsetLsmWriteSpec() {
|
|
||||||
enqueue("unset_lsm_write_spec", 200, "");
|
|
||||||
|
|
||||||
lsm.unsetLsmWriteSpec();
|
|
||||||
|
|
||||||
assertEquals("/v1/table/my_table/unset_lsm_write_spec/", requestPaths.get(0));
|
|
||||||
assertEquals("", requestBodies.get(0));
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testGetLsmWriteSpec() {
|
|
||||||
enqueue(
|
|
||||||
"get_lsm_write_spec",
|
|
||||||
200,
|
|
||||||
"{\"lsm_write_spec\":{\"sharding\":{\"mode\":\"bucket\",\"column\":\"id\","
|
|
||||||
+ "\"num_buckets\":16},\"maintained_indexes\":[\"id_idx\"],"
|
|
||||||
+ "\"writer_config_defaults\":{\"durable_write\":\"true\"}}}");
|
|
||||||
|
|
||||||
Optional<LsmWriteSpec> spec = lsm.getLsmWriteSpec();
|
|
||||||
|
|
||||||
assertTrue(spec.isPresent());
|
|
||||||
assertEquals(LsmWriteSpec.Sharding.BUCKET, spec.get().sharding());
|
|
||||||
assertEquals("id", spec.get().column());
|
|
||||||
assertEquals(Integer.valueOf(16), spec.get().numBuckets());
|
|
||||||
assertEquals(Arrays.asList("id_idx"), spec.get().maintainedIndexes());
|
|
||||||
assertEquals("true", spec.get().writerConfigDefaults().get("durable_write"));
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testGetLsmWriteSpecAbsent() {
|
|
||||||
enqueue("get_lsm_write_spec", 200, "{\"lsm_write_spec\":null}");
|
|
||||||
|
|
||||||
assertFalse(lsm.getLsmWriteSpec().isPresent());
|
|
||||||
}
|
|
||||||
|
|
||||||
// ===========================================================================
|
|
||||||
// stats
|
|
||||||
// ===========================================================================
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testGetLsmStats() throws Exception {
|
|
||||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 7L, 8L)));
|
|
||||||
|
|
||||||
Optional<LsmStats> got = lsm.getLsmStats(true);
|
|
||||||
|
|
||||||
assertEquals("/v1/table/my_table/get_lsm_stats/", requestPaths.get(0));
|
|
||||||
assertTrue(MAPPER.readTree(requestBodies.get(0)).get("include_generation_rows").asBoolean());
|
|
||||||
assertTrue(got.isPresent());
|
|
||||||
BucketStats decoded = got.get().buckets().get(0);
|
|
||||||
assertEquals("shard-0", decoded.shardId());
|
|
||||||
assertEquals("Active", decoded.status());
|
|
||||||
assertEquals(1, decoded.writerEpoch());
|
|
||||||
assertEquals(2, decoded.manifestVersion());
|
|
||||||
assertEquals(9, decoded.currentGeneration());
|
|
||||||
assertFalse(decoded.compacting());
|
|
||||||
assertEquals(Arrays.asList(7L, 8L), generationNumbers(decoded));
|
|
||||||
assertEquals(1024, decoded.generations().get(0).bytes());
|
|
||||||
assertFalse(decoded.generations().get(0).rows().isPresent(), "rows absent unless requested");
|
|
||||||
assertFalse(decoded.memtables().isPresent(), "absent memtables stay absent");
|
|
||||||
}
|
|
||||||
|
|
||||||
/** The optional fields decode when the server does send them. */
|
|
||||||
@Test
|
|
||||||
public void testGetLsmStatsDecodesOptionalFields() {
|
|
||||||
enqueue(
|
|
||||||
"get_lsm_stats",
|
|
||||||
200,
|
|
||||||
"{\"lsm_stats\":{\"buckets\":[{\"shard_id\":\"shard-0\",\"status\":\"Active\","
|
|
||||||
+ "\"writer_epoch\":1,\"manifest_version\":2,\"current_generation\":9,"
|
|
||||||
+ "\"replay_after_wal_entry_position\":3,\"wal_entry_position_last_seen\":11,"
|
|
||||||
+ "\"generations\":[{\"generation\":7,\"bytes\":1024,\"rows\":42}],"
|
|
||||||
+ "\"compacting\":true,\"memtables\":[{\"generation\":8,\"rows\":5,"
|
|
||||||
+ "\"bytes\":64,\"batches\":2,\"indexes\":[\"id_idx\"]}]}]}}");
|
|
||||||
|
|
||||||
BucketStats decoded = lsm.getLsmStats(true).get().buckets().get(0);
|
|
||||||
|
|
||||||
assertEquals(3, decoded.replayAfterWalEntryPosition());
|
|
||||||
assertEquals(11, decoded.walEntryPositionLastSeen());
|
|
||||||
assertTrue(decoded.compacting());
|
|
||||||
assertEquals(42, decoded.generations().get(0).rows().getAsLong());
|
|
||||||
assertTrue(decoded.memtables().isPresent());
|
|
||||||
MemtableStats memtable = decoded.memtables().get().get(0);
|
|
||||||
assertEquals(8, memtable.generation());
|
|
||||||
assertEquals(5, memtable.rows());
|
|
||||||
assertEquals(64, memtable.bytes());
|
|
||||||
assertEquals(2, memtable.batches());
|
|
||||||
assertEquals(Arrays.asList("id_idx"), memtable.indexes());
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testGetLsmStatsAbsentWhenLsmDisabled() {
|
|
||||||
enqueue("get_lsm_stats", 200, "{\"lsm_stats\":null}");
|
|
||||||
|
|
||||||
assertFalse(lsm.getLsmStats().isPresent());
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testGetLsmStatsDefaultsToExcludingGenerationRows() throws Exception {
|
|
||||||
enqueue("get_lsm_stats", 200, stats());
|
|
||||||
|
|
||||||
lsm.getLsmStats();
|
|
||||||
|
|
||||||
assertFalse(MAPPER.readTree(requestBodies.get(0)).get("include_generation_rows").asBoolean());
|
|
||||||
}
|
|
||||||
|
|
||||||
// ===========================================================================
|
|
||||||
// flush / compact
|
|
||||||
// ===========================================================================
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testFlushAndCompactRoutes() {
|
|
||||||
enqueue("flush_lsm", 200, "");
|
|
||||||
enqueue("compact_lsm", 200, "");
|
|
||||||
|
|
||||||
lsm.flushLsm();
|
|
||||||
lsm.compactLsm();
|
|
||||||
|
|
||||||
assertEquals("/v1/table/my_table/flush_lsm/", requestPaths.get(0));
|
|
||||||
assertEquals("/v1/table/my_table/compact_lsm/", requestPaths.get(1));
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testHttpErrorCarriesStatus() {
|
|
||||||
enqueue("flush_lsm", 404, "no such table");
|
|
||||||
|
|
||||||
LanceDbRestClient.HttpException e =
|
|
||||||
assertThrows(LanceDbRestClient.HttpException.class, () -> lsm.flushLsm());
|
|
||||||
assertEquals(404, e.statusCode());
|
|
||||||
}
|
|
||||||
|
|
||||||
// ===========================================================================
|
|
||||||
// checkpoint
|
|
||||||
// ===========================================================================
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testCheckpointReturnsWhenLsmDisabled() {
|
|
||||||
enqueue("flush_lsm", 200, "");
|
|
||||||
enqueue("get_lsm_stats", 200, "{\"lsm_stats\":null}");
|
|
||||||
|
|
||||||
lsm.checkpointLsm();
|
|
||||||
|
|
||||||
assertEquals(0, countCalls("compact_lsm"), "nothing to compact when the LSM path is off");
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testCheckpointReturnsWhenNoGenerationsOutstanding() {
|
|
||||||
enqueue("flush_lsm", 200, "");
|
|
||||||
// A bucket with no L0 generations yields no target, so the drain never starts.
|
|
||||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false)));
|
|
||||||
|
|
||||||
lsm.checkpointLsm();
|
|
||||||
|
|
||||||
assertEquals(0, countCalls("compact_lsm"));
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testCheckpointConvergesOnceTargetGenerationsAreGone() {
|
|
||||||
enqueue("flush_lsm", 200, "");
|
|
||||||
// Watermark read: shard-0 holds generations 7 and 8, so target = 8.
|
|
||||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 7L, 8L)));
|
|
||||||
// First drain poll: both still outstanding, nothing compacting -> dispatch a pass.
|
|
||||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 7L, 8L)));
|
|
||||||
// Second drain poll: drained past the target -> done.
|
|
||||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 9L)));
|
|
||||||
enqueue("compact_lsm", 200, "");
|
|
||||||
|
|
||||||
lsm.checkpointLsm();
|
|
||||||
|
|
||||||
assertEquals(1, countCalls("compact_lsm"), "one pass dispatched");
|
|
||||||
assertEquals(3, countCalls("get_lsm_stats"), "watermark read plus two drain polls");
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testCheckpointDoesNotPileOnWhileEveryTargetBucketIsCompacting() {
|
|
||||||
enqueue("flush_lsm", 200, "");
|
|
||||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", true, 4L)));
|
|
||||||
// Still compacting on the first poll, so no pass is dispatched; then it drains.
|
|
||||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", true, 4L)));
|
|
||||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 5L)));
|
|
||||||
|
|
||||||
lsm.checkpointLsm();
|
|
||||||
|
|
||||||
assertEquals(0, countCalls("compact_lsm"), "a latched bucket is left alone");
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testCheckpointRetriesFromFlushAfterLostClaim() {
|
|
||||||
// 421 on the watermark read: the node lost its claim, so the whole thing restarts
|
|
||||||
// from flush rather than retrying the read in place.
|
|
||||||
enqueue("flush_lsm", 200, "");
|
|
||||||
enqueue("get_lsm_stats", 421, "no claim");
|
|
||||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false)));
|
|
||||||
|
|
||||||
lsm.checkpointLsm();
|
|
||||||
|
|
||||||
assertEquals(2, countCalls("flush_lsm"), "re-issued from flush");
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testCheckpointRetriesRetryableStatusInPlace() {
|
|
||||||
enqueue("flush_lsm", 429, "latch held");
|
|
||||||
enqueue("flush_lsm", 200, "");
|
|
||||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false)));
|
|
||||||
|
|
||||||
lsm.checkpointLsm();
|
|
||||||
|
|
||||||
assertEquals(2, countCalls("flush_lsm"), "429 retried in place, not re-issued");
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testCheckpointPropagatesTerminalStatus() {
|
|
||||||
enqueue("flush_lsm", 400, "bad request");
|
|
||||||
|
|
||||||
LanceDbRestClient.HttpException e =
|
|
||||||
assertThrows(LanceDbRestClient.HttpException.class, () -> lsm.checkpointLsm());
|
|
||||||
assertEquals(400, e.statusCode());
|
|
||||||
assertEquals(1, countCalls("flush_lsm"), "a terminal status is not retried");
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
public void testCheckpointGivesUpAfterRepeatedLostClaims() {
|
|
||||||
enqueue("flush_lsm", 421, "no claim");
|
|
||||||
|
|
||||||
IllegalStateException e = assertThrows(IllegalStateException.class, () -> lsm.checkpointLsm());
|
|
||||||
assertTrue(e.getMessage().contains("kept losing its claim"), e.getMessage());
|
|
||||||
assertEquals(4, countCalls("flush_lsm"), "the initial attempt plus MAX_REISSUES");
|
|
||||||
}
|
|
||||||
|
|
||||||
// ===========================================================================
|
|
||||||
// strict decoding
|
|
||||||
// ===========================================================================
|
|
||||||
|
|
||||||
/**
|
|
||||||
* A stats payload that does not decode must fail closed. Every one of these bodies used to be
|
|
||||||
* read as "no buckets", which is indistinguishable from a drained table, so {@code checkpointLsm}
|
|
||||||
* reported convergence for a checkpoint that never ran.
|
|
||||||
*/
|
|
||||||
@Test
|
|
||||||
public void testCheckpointRejectsMalformedStats() {
|
|
||||||
Map<String, String> malformed = new LinkedHashMap<String, String>();
|
|
||||||
malformed.put("no response body at all", "");
|
|
||||||
malformed.put("stats object with no buckets", "{\"lsm_stats\":{}}");
|
|
||||||
malformed.put("bucket missing its required fields", "{\"lsm_stats\":{\"buckets\":[{}]}}");
|
|
||||||
malformed.put(
|
|
||||||
"bucket missing generations",
|
|
||||||
"{\"lsm_stats\":{\"buckets\":[{\"shard_id\":\"shard-0\",\"status\":\"Active\","
|
|
||||||
+ "\"writer_epoch\":1,\"manifest_version\":2,\"current_generation\":9,"
|
|
||||||
+ "\"replay_after_wal_entry_position\":0,\"wal_entry_position_last_seen\":0,"
|
|
||||||
+ "\"compacting\":false}]}}");
|
|
||||||
malformed.put(
|
|
||||||
"generation with a non-numeric generation number",
|
|
||||||
"{\"lsm_stats\":{\"buckets\":[{\"shard_id\":\"shard-0\",\"status\":\"Active\","
|
|
||||||
+ "\"writer_epoch\":1,\"manifest_version\":2,\"current_generation\":9,"
|
|
||||||
+ "\"replay_after_wal_entry_position\":0,\"wal_entry_position_last_seen\":0,"
|
|
||||||
+ "\"generations\":[{\"generation\":\"7\",\"bytes\":1024}],"
|
|
||||||
+ "\"compacting\":false}]}}");
|
|
||||||
|
|
||||||
for (Map.Entry<String, String> each : malformed.entrySet()) {
|
|
||||||
setUpFresh();
|
|
||||||
enqueue("flush_lsm", 200, "");
|
|
||||||
enqueue("get_lsm_stats", 200, each.getValue());
|
|
||||||
|
|
||||||
assertThrows(
|
|
||||||
IllegalStateException.class,
|
|
||||||
() -> lsm.checkpointLsm(),
|
|
||||||
each.getKey() + " must not report convergence");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/** The one shape that legitimately means "this table has no LSM write path". */
|
|
||||||
@Test
|
|
||||||
public void testCheckpointTreatsNullStatsAsNotWalBacked() {
|
|
||||||
enqueue("flush_lsm", 200, "");
|
|
||||||
enqueue("get_lsm_stats", 200, "{\"lsm_stats\":null}");
|
|
||||||
|
|
||||||
lsm.checkpointLsm();
|
|
||||||
|
|
||||||
assertEquals(1, countCalls("get_lsm_stats"));
|
|
||||||
}
|
|
||||||
|
|
||||||
// ===========================================================================
|
|
||||||
// retry budget
|
|
||||||
// ===========================================================================
|
|
||||||
|
|
||||||
/**
|
|
||||||
* The transport must not retry on the checkpoint loop's behalf. Apache HttpClient's default
|
|
||||||
* strategy retries exactly 429 and 503 — the two statuses {@code isRetryable} owns — which
|
|
||||||
* doubled every budget here and also retried {@code compact_lsm} in place, where the loop is
|
|
||||||
* built to fall through to a fresh stats poll instead.
|
|
||||||
*/
|
|
||||||
@Test
|
|
||||||
public void testCheckpointRetryBudgetIsNotDoubledByTheTransport() {
|
|
||||||
enqueue("flush_lsm", 429, "latch held");
|
|
||||||
|
|
||||||
LanceDbRestClient.HttpException e =
|
|
||||||
assertThrows(LanceDbRestClient.HttpException.class, () -> lsm.checkpointLsm());
|
|
||||||
|
|
||||||
assertEquals(429, e.statusCode(), "the exhausted budget propagates the last error as itself");
|
|
||||||
assertEquals(9, countCalls("flush_lsm"), "the initial request plus MAX_RETRIES, and no more");
|
|
||||||
}
|
|
||||||
|
|
||||||
// ===========================================================================
|
|
||||||
// harness
|
|
||||||
// ===========================================================================
|
|
||||||
|
|
||||||
private static List<Long> generationNumbers(BucketStats bucket) {
|
|
||||||
List<Long> numbers = new ArrayList<Long>();
|
|
||||||
for (GenerationStats generation : bucket.generations()) {
|
|
||||||
numbers.add(generation.generation());
|
|
||||||
}
|
|
||||||
return numbers;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Build an {@code lsm_stats} response body from bucket fragments. */
|
|
||||||
private static String stats(String... buckets) {
|
|
||||||
return "{\"lsm_stats\":{\"buckets\":[" + String.join(",", buckets) + "]}}";
|
|
||||||
}
|
|
||||||
|
|
||||||
private static String bucket(String shardId, boolean compacting, Long... generations) {
|
|
||||||
StringBuilder gens = new StringBuilder();
|
|
||||||
for (Long generation : generations) {
|
|
||||||
if (gens.length() > 0) {
|
|
||||||
gens.append(",");
|
|
||||||
}
|
|
||||||
gens.append("{\"generation\":").append(generation).append(",\"bytes\":1024}");
|
|
||||||
}
|
|
||||||
return "{\"shard_id\":\""
|
|
||||||
+ shardId
|
|
||||||
+ "\",\"status\":\"Active\",\"writer_epoch\":1,\"manifest_version\":2,"
|
|
||||||
+ "\"current_generation\":9,\"replay_after_wal_entry_position\":0,"
|
|
||||||
+ "\"wal_entry_position_last_seen\":0,\"generations\":["
|
|
||||||
+ gens
|
|
||||||
+ "],\"compacting\":"
|
|
||||||
+ compacting
|
|
||||||
+ "}";
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Queue a reply for an operation. The last queued reply repeats once the queue drains. */
|
|
||||||
private void enqueue(String operation, int status, String body) {
|
|
||||||
replies.computeIfAbsent(operation, key -> new ArrayDeque<Reply>()).add(new Reply(status, body));
|
|
||||||
}
|
|
||||||
|
|
||||||
private Reply nextReply(String path) {
|
|
||||||
String operation = operationOf(path);
|
|
||||||
Deque<Reply> queued = replies.get(operation);
|
|
||||||
if (queued == null || queued.isEmpty()) {
|
|
||||||
return new Reply(200, "");
|
|
||||||
}
|
|
||||||
return queued.size() > 1 ? queued.poll() : queued.peek();
|
|
||||||
}
|
|
||||||
|
|
||||||
private long countCalls(String operation) {
|
|
||||||
return requestPaths.stream().filter(path -> operationOf(path).equals(operation)).count();
|
|
||||||
}
|
|
||||||
|
|
||||||
/** {@code /v1/table/my_table/flush_lsm/} -> {@code flush_lsm}. */
|
|
||||||
private static String operationOf(String path) {
|
|
||||||
String[] segments = path.split("/");
|
|
||||||
return segments.length == 0 ? "" : segments[segments.length - 1];
|
|
||||||
}
|
|
||||||
|
|
||||||
private static String readAll(InputStream in) throws IOException {
|
|
||||||
ByteArrayOutputStream out = new ByteArrayOutputStream();
|
|
||||||
byte[] buffer = new byte[4096];
|
|
||||||
int read;
|
|
||||||
while ((read = in.read(buffer)) != -1) {
|
|
||||||
out.write(buffer, 0, read);
|
|
||||||
}
|
|
||||||
return new String(out.toByteArray(), StandardCharsets.UTF_8);
|
|
||||||
}
|
|
||||||
|
|
||||||
private static final class Reply {
|
|
||||||
private final int status;
|
|
||||||
private final String body;
|
|
||||||
|
|
||||||
private Reply(int status, String body) {
|
|
||||||
this.status = status;
|
|
||||||
this.body = body;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
+2
-2
@@ -6,7 +6,7 @@
|
|||||||
|
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-parent</artifactId>
|
<artifactId>lancedb-parent</artifactId>
|
||||||
<version>0.38.0-beta.11</version>
|
<version>0.37.1-beta.0</version>
|
||||||
<packaging>pom</packaging>
|
<packaging>pom</packaging>
|
||||||
<name>${project.artifactId}</name>
|
<name>${project.artifactId}</name>
|
||||||
<description>LanceDB Java SDK Parent POM</description>
|
<description>LanceDB Java SDK Parent POM</description>
|
||||||
@@ -28,7 +28,7 @@
|
|||||||
<properties>
|
<properties>
|
||||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||||
<arrow.version>15.0.0</arrow.version>
|
<arrow.version>15.0.0</arrow.version>
|
||||||
<lance-core.version>12.0.0-beta.2</lance-core.version>
|
<lance-core.version>10.0.0-beta.5</lance-core.version>
|
||||||
<spotless.skip>false</spotless.skip>
|
<spotless.skip>false</spotless.skip>
|
||||||
<spotless.version>2.30.0</spotless.version>
|
<spotless.version>2.30.0</spotless.version>
|
||||||
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
|
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
|
||||||
|
|||||||
+5
-5
@@ -1,7 +1,7 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "lancedb-nodejs"
|
name = "lancedb-nodejs"
|
||||||
edition.workspace = true
|
edition.workspace = true
|
||||||
version = "0.38.0-beta.11"
|
version = "0.37.1-beta.0"
|
||||||
publish = false
|
publish = false
|
||||||
license.workspace = true
|
license.workspace = true
|
||||||
description.workspace = true
|
description.workspace = true
|
||||||
@@ -16,12 +16,12 @@ crate-type = ["cdylib"]
|
|||||||
async-trait.workspace = true
|
async-trait.workspace = true
|
||||||
arrow-ipc.workspace = true
|
arrow-ipc.workspace = true
|
||||||
arrow-array.workspace = true
|
arrow-array.workspace = true
|
||||||
arrow-buffer.workspace = true
|
arrow-buffer = "58.0.0"
|
||||||
half.workspace = true
|
half.workspace = true
|
||||||
arrow-schema.workspace = true
|
arrow-schema.workspace = true
|
||||||
env_logger.workspace = true
|
env_logger.workspace = true
|
||||||
futures.workspace = true
|
futures.workspace = true
|
||||||
lancedb.workspace = true
|
lancedb = { path = "../rust/lancedb", default-features = false }
|
||||||
lance-namespace.workspace = true
|
lance-namespace.workspace = true
|
||||||
napi = { version = "3.8.3", default-features = false, features = [
|
napi = { version = "3.8.3", default-features = false, features = [
|
||||||
"napi9",
|
"napi9",
|
||||||
@@ -29,8 +29,8 @@ napi = { version = "3.8.3", default-features = false, features = [
|
|||||||
"chrono_date",
|
"chrono_date",
|
||||||
"serde-json",
|
"serde-json",
|
||||||
] }
|
] }
|
||||||
chrono.workspace = true
|
chrono = { version = "0.4", default-features = false, features = ["clock"] }
|
||||||
serde_json.workspace = true
|
serde_json = "1"
|
||||||
napi-derive = "3.5.2"
|
napi-derive = "3.5.2"
|
||||||
# Prevent dynamic linking of lzma, which comes from datafusion
|
# Prevent dynamic linking of lzma, which comes from datafusion
|
||||||
lzma-sys = { version = "0.1", features = ["static"] }
|
lzma-sys = { version = "0.1", features = ["static"] }
|
||||||
|
|||||||
@@ -6,12 +6,7 @@ import * as arrow17 from "apache-arrow-17";
|
|||||||
import * as arrow18 from "apache-arrow-18";
|
import * as arrow18 from "apache-arrow-18";
|
||||||
|
|
||||||
import {
|
import {
|
||||||
Field as CurrentField,
|
|
||||||
LargeBinary as CurrentLargeBinary,
|
|
||||||
Schema as CurrentSchema,
|
|
||||||
Vector as CurrentVector,
|
|
||||||
convertToTable,
|
convertToTable,
|
||||||
tableFromIPC as currentTableFromIPC,
|
|
||||||
fromBufferToRecordBatch,
|
fromBufferToRecordBatch,
|
||||||
fromDataToBuffer,
|
fromDataToBuffer,
|
||||||
fromRecordBatchToBuffer,
|
fromRecordBatchToBuffer,
|
||||||
@@ -24,7 +19,6 @@ import {
|
|||||||
FunctionOptions,
|
FunctionOptions,
|
||||||
} from "../lancedb/embedding/embedding_function";
|
} from "../lancedb/embedding/embedding_function";
|
||||||
import { EmbeddingFunctionConfig } from "../lancedb/embedding/registry";
|
import { EmbeddingFunctionConfig } from "../lancedb/embedding/registry";
|
||||||
import { sanitizeTable } from "../lancedb/sanitize";
|
|
||||||
|
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: skip
|
// biome-ignore lint/suspicious/noExplicitAny: skip
|
||||||
function sampleRecords(): Array<Record<string, any>> {
|
function sampleRecords(): Array<Record<string, any>> {
|
||||||
@@ -39,24 +33,6 @@ function sampleRecords(): Array<Record<string, any>> {
|
|||||||
},
|
},
|
||||||
];
|
];
|
||||||
}
|
}
|
||||||
|
|
||||||
it("preserves field metadata from a provided schema", async function () {
|
|
||||||
const jsonMetadata = new Map([["ARROW:extension:name", "lance.json"]]);
|
|
||||||
const schema = new CurrentSchema([
|
|
||||||
new CurrentField("meta", new CurrentLargeBinary(), true, jsonMetadata),
|
|
||||||
]);
|
|
||||||
|
|
||||||
const table = makeArrowTable(
|
|
||||||
[{ meta: Buffer.from(JSON.stringify({ source: "test" })) }],
|
|
||||||
{ schema },
|
|
||||||
);
|
|
||||||
|
|
||||||
expect(table.schema.fields[0].metadata).toEqual(jsonMetadata);
|
|
||||||
|
|
||||||
const roundTripped = currentTableFromIPC(await fromTableToBuffer(table));
|
|
||||||
expect(roundTripped.schema.fields[0].metadata).toEqual(jsonMetadata);
|
|
||||||
});
|
|
||||||
|
|
||||||
describe.each([arrow15, arrow16, arrow17, arrow18])(
|
describe.each([arrow15, arrow16, arrow17, arrow18])(
|
||||||
"Arrow",
|
"Arrow",
|
||||||
(
|
(
|
||||||
@@ -88,11 +64,7 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
tableFromIPC,
|
tableFromIPC,
|
||||||
DataType,
|
DataType,
|
||||||
Dictionary,
|
Dictionary,
|
||||||
RecordBatch: ArrowRecordBatch,
|
|
||||||
Table: ArrowTable,
|
|
||||||
Uint8: ArrowUint8,
|
Uint8: ArrowUint8,
|
||||||
makeData: arrowMakeData,
|
|
||||||
vectorFromArray,
|
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: <explanation>
|
// biome-ignore lint/suspicious/noExplicitAny: <explanation>
|
||||||
} = <any>arrow;
|
} = <any>arrow;
|
||||||
type Schema = ApacheArrow["Schema"];
|
type Schema = ApacheArrow["Schema"];
|
||||||
@@ -194,36 +166,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
}
|
}
|
||||||
|
|
||||||
describe("The function makeArrowTable", function () {
|
describe("The function makeArrowTable", function () {
|
||||||
it("accepts snake_case embedding metadata like camelCase", function () {
|
|
||||||
const spellings = [
|
|
||||||
// biome-ignore lint/style/useNamingConvention: the Python wire spelling
|
|
||||||
{ source_column: "text", vector_column: "vector" },
|
|
||||||
{ sourceColumn: "text", vectorColumn: "vector" },
|
|
||||||
];
|
|
||||||
for (const columns of spellings) {
|
|
||||||
const schema = new Schema(
|
|
||||||
[
|
|
||||||
new Field("text", new Utf8(), false),
|
|
||||||
new Field(
|
|
||||||
"vector",
|
|
||||||
new FixedSizeList(3, new Field("item", new Float32(), true)),
|
|
||||||
false,
|
|
||||||
),
|
|
||||||
],
|
|
||||||
new Map([
|
|
||||||
[
|
|
||||||
"embedding_functions",
|
|
||||||
JSON.stringify([{ name: "mock", model: {}, ...columns }]),
|
|
||||||
],
|
|
||||||
]),
|
|
||||||
);
|
|
||||||
// The vector field is non-nullable and absent from the data; only a
|
|
||||||
// recognized embedding config makes that acceptable.
|
|
||||||
const table = makeArrowTable([{ text: "hello" }], { schema });
|
|
||||||
expect(table.numRows).toBe(1);
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
it("will use data types from a provided schema instead of inference", async function () {
|
it("will use data types from a provided schema instead of inference", async function () {
|
||||||
const schema = new Schema([
|
const schema = new Schema([
|
||||||
new Field("a", new Int32(), false),
|
new Field("a", new Int32(), false),
|
||||||
@@ -255,35 +197,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
expect(table.getChild("d")?.toJSON()).toEqual([9n, 10n, null]);
|
expect(table.getChild("d")?.toJSON()).toEqual([9n, 10n, null]);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("will use a provided FixedSizeList schema with typed array values", function () {
|
|
||||||
const schema = new Schema([
|
|
||||||
new Field("text", new Utf8(), false),
|
|
||||||
new Field(
|
|
||||||
"vector",
|
|
||||||
new FixedSizeList(3, new Field("item", new Float32(), false)),
|
|
||||||
false,
|
|
||||||
),
|
|
||||||
]);
|
|
||||||
|
|
||||||
const table = makeArrowTable(
|
|
||||||
[
|
|
||||||
{
|
|
||||||
text: "foo",
|
|
||||||
vector: new Float32Array([1, 2, 3]),
|
|
||||||
},
|
|
||||||
],
|
|
||||||
{ schema },
|
|
||||||
);
|
|
||||||
|
|
||||||
expect(table.getChild("text")?.toJSON()).toEqual(["foo"]);
|
|
||||||
expect(
|
|
||||||
table
|
|
||||||
.getChild("vector")
|
|
||||||
?.toJSON()
|
|
||||||
.map((value) => value.toJSON()),
|
|
||||||
).toEqual([[1, 2, 3]]);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("will assume the column `vector` is FixedSizeList<Float32> by default", async function () {
|
it("will assume the column `vector` is FixedSizeList<Float32> by default", async function () {
|
||||||
const schema = new Schema([
|
const schema = new Schema([
|
||||||
new Field("a", new Float(Precision.DOUBLE), true),
|
new Field("a", new Float(Precision.DOUBLE), true),
|
||||||
@@ -536,137 +449,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
);
|
);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("will allow matching inferred types across records", function () {
|
|
||||||
expect(() =>
|
|
||||||
makeArrowTable([{ value: 1 }, { value: 2 }]),
|
|
||||||
).not.toThrow();
|
|
||||||
});
|
|
||||||
|
|
||||||
it("will reject mismatched inferred types across records", function () {
|
|
||||||
expect(() => makeArrowTable([{ value: 1 }, { value: "two" }])).toThrow(
|
|
||||||
"Failed to infer schema for data. Previously inferred type Float64 but found Utf8 for field value at row 1. Consider providing an explicit schema.",
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("will ignore generated dictionary IDs when comparing inferred types", function () {
|
|
||||||
const table = makeArrowTable([{ str: "a" }, { str: "b" }], {
|
|
||||||
dictionaryEncodeStrings: true,
|
|
||||||
});
|
|
||||||
|
|
||||||
expect(table.getChild("str")?.toJSON()).toEqual(["a", "b"]);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("will preserve null values without treating them as type mismatches", function () {
|
|
||||||
for (const records of [
|
|
||||||
[{ vector: [1, 2, 3] }, { vector: null }],
|
|
||||||
[{ vector: null }, { vector: [1, 2, 3] }],
|
|
||||||
]) {
|
|
||||||
const table = makeArrowTable(records);
|
|
||||||
|
|
||||||
expect(table.numRows).toBe(2);
|
|
||||||
expect(table.getChild("vector")?.nullCount).toBe(1);
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
it("will preserve empty variable-size lists", function () {
|
|
||||||
for (const records of [
|
|
||||||
[{ items: [1] }, { items: [] }],
|
|
||||||
[{ items: [] }, { items: [1] }],
|
|
||||||
]) {
|
|
||||||
const table = makeArrowTable(records);
|
|
||||||
expect(
|
|
||||||
table
|
|
||||||
.getChild("items")
|
|
||||||
?.toJSON()
|
|
||||||
.map((value) => value.toJSON()),
|
|
||||||
).toEqual(records.map((record) => record.items));
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
it("will propagate deferred evidence through nested lists", function () {
|
|
||||||
for (const records of [
|
|
||||||
[{ items: [1] }, { items: [null] }],
|
|
||||||
[{ items: [null] }, { items: [1] }],
|
|
||||||
[{ items: [null, 1] }, { items: [2, null] }],
|
|
||||||
]) {
|
|
||||||
const table = makeArrowTable(records);
|
|
||||||
expect(
|
|
||||||
table
|
|
||||||
.getChild("items")
|
|
||||||
?.toJSON()
|
|
||||||
.map((value) => value.toJSON()),
|
|
||||||
).toEqual(records.map((record) => record.items));
|
|
||||||
}
|
|
||||||
|
|
||||||
const nestedRecords = [{ items: [[1]] }, { items: [[null]] }];
|
|
||||||
const nestedTable = makeArrowTable(nestedRecords);
|
|
||||||
expect(
|
|
||||||
nestedTable
|
|
||||||
.getChild("items")
|
|
||||||
?.toJSON()
|
|
||||||
.map((value) =>
|
|
||||||
value
|
|
||||||
.toJSON()
|
|
||||||
.map((nestedValue: { toJSON: () => unknown[] }) =>
|
|
||||||
nestedValue.toJSON(),
|
|
||||||
),
|
|
||||||
),
|
|
||||||
).toEqual(nestedRecords.map((record) => record.items));
|
|
||||||
});
|
|
||||||
|
|
||||||
it("will reject incompatible deferred evidence within a list", function () {
|
|
||||||
for (const items of [
|
|
||||||
[[], 1],
|
|
||||||
[1, []],
|
|
||||||
[[null], 1],
|
|
||||||
[1, [null]],
|
|
||||||
]) {
|
|
||||||
expect(() => makeArrowTable([{ items }])).toThrow(
|
|
||||||
"Failed to infer data type for field items at row 0.",
|
|
||||||
);
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
it("will reject empty fixed-size lists", function () {
|
|
||||||
expect(() =>
|
|
||||||
makeArrowTable([{ vector: [1, 2, 3] }, { vector: [] }]),
|
|
||||||
).toThrow(
|
|
||||||
"Failed to infer schema for data. Previously inferred type FixedSizeList[3]<Float32> but found List[0] for field vector at row 1.",
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("will reject inferred leaf and branch shape changes", function () {
|
|
||||||
expect(() =>
|
|
||||||
makeArrowTable([{ value: 1 }, { value: { nested: 2 } }]),
|
|
||||||
).toThrow(
|
|
||||||
"Failed to infer schema for data. Previously inferred type Float64 but found Struct for field value at row 1.",
|
|
||||||
);
|
|
||||||
expect(() =>
|
|
||||||
makeArrowTable([{ value: { nested: 1 } }, { value: 2 }]),
|
|
||||||
).toThrow(
|
|
||||||
"Failed to infer schema for data. Previously inferred type Struct but found Float64 for field value at row 1.",
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("will allow null values around inferred struct values", function () {
|
|
||||||
for (const { records, nullIndex } of [
|
|
||||||
{
|
|
||||||
records: [{ value: null }, { value: { nested: 2 } }],
|
|
||||||
nullIndex: 0,
|
|
||||||
},
|
|
||||||
{
|
|
||||||
records: [{ value: { nested: 1 } }, { value: null }],
|
|
||||||
nullIndex: 1,
|
|
||||||
},
|
|
||||||
]) {
|
|
||||||
const table = makeArrowTable(records);
|
|
||||||
const values = table.getChild("value");
|
|
||||||
|
|
||||||
expect(values?.nullCount).toBe(1);
|
|
||||||
expect(values?.get(nullIndex)).toBeNull();
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
it("will allow a schema to be provided", async function () {
|
it("will allow a schema to be provided", async function () {
|
||||||
await checkTableCreation(
|
await checkTableCreation(
|
||||||
async (records, _, schema) =>
|
async (records, _, schema) =>
|
||||||
@@ -1243,114 +1025,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
});
|
});
|
||||||
|
|
||||||
describe("when using two versions of arrow", function () {
|
describe("when using two versions of arrow", function () {
|
||||||
it("preserves a dictionary shared by multiple fields", async function () {
|
|
||||||
const values = ["alpha", "beta", "alpha"];
|
|
||||||
const dictionaryVector = vectorFromArray(values);
|
|
||||||
const batch = new ArrowRecordBatch({
|
|
||||||
first: dictionaryVector.data[0],
|
|
||||||
second: dictionaryVector.data[0],
|
|
||||||
});
|
|
||||||
const table = new ArrowTable([batch]);
|
|
||||||
|
|
||||||
const sanitized = sanitizeTable(table);
|
|
||||||
expect([...sanitized.getChild("first")!]).toEqual(values);
|
|
||||||
expect([...sanitized.getChild("second")!]).toEqual(values);
|
|
||||||
const firstType = sanitized.schema.fields[0].type as {
|
|
||||||
dictionary: unknown;
|
|
||||||
};
|
|
||||||
const secondType = sanitized.schema.fields[1].type as {
|
|
||||||
dictionary: unknown;
|
|
||||||
};
|
|
||||||
expect(secondType.dictionary).toBe(firstType.dictionary);
|
|
||||||
expect(sanitized.batches[0].data.children[1].dictionary).toBe(
|
|
||||||
sanitized.batches[0].data.children[0].dictionary,
|
|
||||||
);
|
|
||||||
|
|
||||||
const buf = await fromDataToBuffer(table);
|
|
||||||
const actual = currentTableFromIPC(buf);
|
|
||||||
expect([...actual.getChild("first")!]).toEqual(values);
|
|
||||||
expect([...actual.getChild("second")!]).toEqual(values);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("preserves shared dictionary data from another Arrow version", async function () {
|
|
||||||
const values = ["alpha", "beta", "alpha"];
|
|
||||||
const dictionaryVector = vectorFromArray(values);
|
|
||||||
const firstBatch = new ArrowRecordBatch({
|
|
||||||
label: dictionaryVector.slice(0, 2).data[0],
|
|
||||||
});
|
|
||||||
const secondBatch = new ArrowRecordBatch({
|
|
||||||
label: dictionaryVector.slice(2).data[0],
|
|
||||||
});
|
|
||||||
const table = new ArrowTable([firstBatch, secondBatch]);
|
|
||||||
|
|
||||||
const sanitized = sanitizeTable(table);
|
|
||||||
expect([...sanitized.getChild("label")!]).toEqual(values);
|
|
||||||
|
|
||||||
const dictionaries = sanitized.batches.map(
|
|
||||||
(batch) => batch.data.children[0].dictionary,
|
|
||||||
);
|
|
||||||
expect(dictionaries[0]).toBeInstanceOf(CurrentVector);
|
|
||||||
expect(dictionaries[1]).toBe(dictionaries[0]);
|
|
||||||
|
|
||||||
const buf = await fromDataToBuffer(table);
|
|
||||||
const actual = currentTableFromIPC(buf);
|
|
||||||
expect([...actual.getChild("label")!]).toEqual(values);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("preserves shared chunks in growing dictionaries", async function () {
|
|
||||||
const type = new Dictionary(new Utf8(), new Int32(), 42, false);
|
|
||||||
const firstDictionary = vectorFromArray(["alpha", "beta"], new Utf8());
|
|
||||||
const secondDictionary = firstDictionary.concat(
|
|
||||||
vectorFromArray(["gamma"], new Utf8()),
|
|
||||||
);
|
|
||||||
const firstData = arrowMakeData({
|
|
||||||
type,
|
|
||||||
data: Int32Array.from([0, 1]),
|
|
||||||
dictionary: firstDictionary,
|
|
||||||
});
|
|
||||||
const secondData = arrowMakeData({
|
|
||||||
type,
|
|
||||||
data: Int32Array.from([2]),
|
|
||||||
dictionary: secondDictionary,
|
|
||||||
});
|
|
||||||
const table = new ArrowTable([
|
|
||||||
new ArrowRecordBatch({ label: firstData }),
|
|
||||||
new ArrowRecordBatch({ label: secondData }),
|
|
||||||
]);
|
|
||||||
|
|
||||||
const sanitized = sanitizeTable(table);
|
|
||||||
const expected = ["alpha", "beta", "gamma"];
|
|
||||||
expect([...sanitized.getChild("label")!]).toEqual(expected);
|
|
||||||
const firstLocalDictionary =
|
|
||||||
sanitized.batches[0].data.children[0].dictionary!;
|
|
||||||
const secondLocalDictionary =
|
|
||||||
sanitized.batches[1].data.children[0].dictionary!;
|
|
||||||
expect(secondLocalDictionary.data[0]).toBe(
|
|
||||||
firstLocalDictionary.data[0],
|
|
||||||
);
|
|
||||||
|
|
||||||
const buf = await fromTableToBuffer(sanitized);
|
|
||||||
const actual = currentTableFromIPC(buf);
|
|
||||||
expect([...actual.getChild("label")!]).toEqual(expected);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("can serialize list data from another Arrow version", async function () {
|
|
||||||
const values = [["anime", "action"], [], null];
|
|
||||||
const vector = vectorFromArray(
|
|
||||||
values,
|
|
||||||
new List(new Field("item", new Utf8(), true)),
|
|
||||||
);
|
|
||||||
const table = new ArrowTable({ tags: vector });
|
|
||||||
|
|
||||||
const buf = await fromDataToBuffer(table);
|
|
||||||
const actual = currentTableFromIPC(buf);
|
|
||||||
const actualTags = actual.getChild("tags");
|
|
||||||
|
|
||||||
expect(actualTags?.get(0)?.toJSON()).toEqual(values[0]);
|
|
||||||
expect(actualTags?.get(1)?.toJSON()).toEqual(values[1]);
|
|
||||||
expect(actualTags?.get(2)).toBeNull();
|
|
||||||
});
|
|
||||||
|
|
||||||
it("can still import data", async function () {
|
it("can still import data", async function () {
|
||||||
const schema = new arrow15.Schema([
|
const schema = new arrow15.Schema([
|
||||||
new arrow15.Field("id", new arrow15.Int32()),
|
new arrow15.Field("id", new arrow15.Int32()),
|
||||||
|
|||||||
@@ -4,13 +4,7 @@
|
|||||||
import { readdirSync } from "fs";
|
import { readdirSync } from "fs";
|
||||||
import { Field, Float64, Schema } from "apache-arrow";
|
import { Field, Float64, Schema } from "apache-arrow";
|
||||||
import * as tmp from "tmp";
|
import * as tmp from "tmp";
|
||||||
import {
|
import { Connection, Table, connect, connectNamespace } from "../lancedb";
|
||||||
Connection,
|
|
||||||
ListTablesResponse,
|
|
||||||
Table,
|
|
||||||
connect,
|
|
||||||
connectNamespace,
|
|
||||||
} from "../lancedb";
|
|
||||||
import { LocalTable } from "../lancedb/table";
|
import { LocalTable } from "../lancedb/table";
|
||||||
|
|
||||||
describe("when connecting", () => {
|
describe("when connecting", () => {
|
||||||
@@ -53,7 +47,6 @@ describe("given a connection", () => {
|
|||||||
await db.close();
|
await db.close();
|
||||||
expect(db.isOpen()).toBe(false);
|
expect(db.isOpen()).toBe(false);
|
||||||
await expect(db.tableNames()).rejects.toThrow("Connection is closed");
|
await expect(db.tableNames()).rejects.toThrow("Connection is closed");
|
||||||
await expect(db.listTables()).rejects.toThrow("Connection is closed");
|
|
||||||
await expect(db.renameTable("a", "b")).rejects.toThrow(
|
await expect(db.renameTable("a", "b")).rejects.toThrow(
|
||||||
"Connection is closed",
|
"Connection is closed",
|
||||||
);
|
);
|
||||||
@@ -96,16 +89,6 @@ describe("given a connection", () => {
|
|||||||
await db.createTable("test4", [{ id: 1 }, { id: 2 }]);
|
await db.createTable("test4", [{ id: 1 }, { id: 2 }]);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("should return a completed job when dropping a local table", async () => {
|
|
||||||
await db.createTable("async-drop", [{ id: 1 }]);
|
|
||||||
|
|
||||||
const job = await db.dropTableAsync("async-drop");
|
|
||||||
expect(job.id).toBeNull();
|
|
||||||
await expect(job.status()).resolves.toBe("finished");
|
|
||||||
await job.wait();
|
|
||||||
await expect(db.tableNames()).resolves.toEqual([]);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("should fail if creating table twice, unless overwrite is true", async () => {
|
it("should fail if creating table twice, unless overwrite is true", async () => {
|
||||||
let tbl = await db.createTable("test", [{ id: 1 }, { id: 2 }]);
|
let tbl = await db.createTable("test", [{ id: 1 }, { id: 2 }]);
|
||||||
await expect(tbl.countRows()).resolves.toBe(2);
|
await expect(tbl.countRows()).resolves.toBe(2);
|
||||||
@@ -136,66 +119,6 @@ describe("given a connection", () => {
|
|||||||
expect(tables).toEqual(["b", "c"]);
|
expect(tables).toEqual(["b", "c"]);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("should respect limit and page token when listing tables", async () => {
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
|
|
||||||
await db.createTable("b", [{ id: 1 }]);
|
|
||||||
await db.createTable("a", [{ id: 1 }]);
|
|
||||||
await db.createTable("c", [{ id: 1 }]);
|
|
||||||
|
|
||||||
const all = await db.listTables();
|
|
||||||
expect(all.tables).toEqual(["a", "b", "c"]);
|
|
||||||
expect(all.pageToken).toBeUndefined();
|
|
||||||
|
|
||||||
const first = await db.listTables({ limit: 1 });
|
|
||||||
expect(first.tables).toEqual(["a"]);
|
|
||||||
expect(first.pageToken).toBeDefined();
|
|
||||||
|
|
||||||
const second = await db.listTables({
|
|
||||||
limit: 1,
|
|
||||||
pageToken: first.pageToken,
|
|
||||||
});
|
|
||||||
expect(second.tables).toEqual(["b"]);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("should visit every table exactly once when walking pages", async () => {
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
|
|
||||||
const created = ["a", "b", "c", "d", "e"];
|
|
||||||
for (const name of created) {
|
|
||||||
await db.createTable(name, [{ id: 1 }]);
|
|
||||||
}
|
|
||||||
|
|
||||||
const seen: string[] = [];
|
|
||||||
let pageToken: string | undefined = undefined;
|
|
||||||
do {
|
|
||||||
const page: ListTablesResponse = await db.listTables({
|
|
||||||
limit: 2,
|
|
||||||
pageToken,
|
|
||||||
});
|
|
||||||
seen.push(...page.tables);
|
|
||||||
pageToken = page.pageToken;
|
|
||||||
} while (pageToken);
|
|
||||||
|
|
||||||
expect(seen).toEqual(created);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("should list tables in a namespace", async () => {
|
|
||||||
const db = await connect(tmpDir.name, {
|
|
||||||
// biome-ignore lint/style/useNamingConvention: opaque backend property key, must match Rust
|
|
||||||
namespaceClientProperties: { manifest_enabled: "true" },
|
|
||||||
});
|
|
||||||
await db.createNamespace(["child"]);
|
|
||||||
await db.createTable("nested", [{ id: 1 }], ["child"]);
|
|
||||||
|
|
||||||
await expect(db.listTables(["child"])).resolves.toEqual(
|
|
||||||
expect.objectContaining({ tables: ["nested"] }),
|
|
||||||
);
|
|
||||||
await expect(db.listTables()).resolves.toEqual(
|
|
||||||
expect.objectContaining({ tables: [] }),
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("should create tables in v2 mode", async () => {
|
it("should create tables in v2 mode", async () => {
|
||||||
const db = await connect(tmpDir.name);
|
const db = await connect(tmpDir.name);
|
||||||
const data = [...Array(10000).keys()].map((i) => ({ id: i }));
|
const data = [...Array(10000).keys()].map((i) => ({ id: i }));
|
||||||
|
|||||||
@@ -11,11 +11,8 @@ import {
|
|||||||
Float16,
|
Float16,
|
||||||
Float32,
|
Float32,
|
||||||
Float64,
|
Float64,
|
||||||
Int32,
|
|
||||||
Schema,
|
Schema,
|
||||||
Utf8,
|
Utf8,
|
||||||
fromDataToBuffer,
|
|
||||||
tableFromIPC,
|
|
||||||
} from "../lancedb/arrow";
|
} from "../lancedb/arrow";
|
||||||
import { EmbeddingFunction, LanceSchema } from "../lancedb/embedding";
|
import { EmbeddingFunction, LanceSchema } from "../lancedb/embedding";
|
||||||
import { getRegistry, register } from "../lancedb/embedding/registry";
|
import { getRegistry, register } from "../lancedb/embedding/registry";
|
||||||
@@ -187,115 +184,6 @@ describe("embedding functions", () => {
|
|||||||
const vector0 = JSON.parse(JSON.stringify(arr[0].vector));
|
const vector0 = JSON.parse(JSON.stringify(arr[0].vector));
|
||||||
expect(vector0).toEqual([1, 2, 3]);
|
expect(vector0).toEqual([1, 2, 3]);
|
||||||
});
|
});
|
||||||
it("should append multiple Python embeddings with the same alias", async () => {
|
|
||||||
@register("python-mock")
|
|
||||||
// biome-ignore lint/correctness/noUnusedVariables: the decorator registers this class
|
|
||||||
class MockEmbeddingFunction extends EmbeddingFunction<string> {
|
|
||||||
ndims() {
|
|
||||||
return 3;
|
|
||||||
}
|
|
||||||
embeddingDataType(): Float {
|
|
||||||
return new Float32();
|
|
||||||
}
|
|
||||||
async computeQueryEmbeddings(_data: string) {
|
|
||||||
return [1, 2, 3];
|
|
||||||
}
|
|
||||||
async computeSourceEmbeddings(data: string[]) {
|
|
||||||
return data.map((value) =>
|
|
||||||
value === "hello world" ? [1, 2, 3] : [4, 5, 6],
|
|
||||||
);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
const metadata = new Map([
|
|
||||||
[
|
|
||||||
"embedding_functions",
|
|
||||||
'[{"source_column":"text1","vector_column":"vector1","name":"python-mock","model":{}},{"source_column":"text2","vector_column":"vector2","name":"python-mock","model":{}}]',
|
|
||||||
],
|
|
||||||
]);
|
|
||||||
const schema = new Schema(
|
|
||||||
[
|
|
||||||
new Field("text1", new Utf8(), true),
|
|
||||||
new Field("text2", new Utf8(), true),
|
|
||||||
new Field(
|
|
||||||
"vector1",
|
|
||||||
new FixedSizeList(3, new Field("item", new Float32(), true)),
|
|
||||||
true,
|
|
||||||
),
|
|
||||||
new Field(
|
|
||||||
"vector2",
|
|
||||||
new FixedSizeList(3, new Field("item", new Float32(), true)),
|
|
||||||
true,
|
|
||||||
),
|
|
||||||
],
|
|
||||||
metadata,
|
|
||||||
);
|
|
||||||
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
const table = await db.createEmptyTable("test", schema);
|
|
||||||
await table.add([{ text1: "hello world", text2: "goodbye world" }]);
|
|
||||||
|
|
||||||
const rows = await table.query().toArray();
|
|
||||||
expect(JSON.parse(JSON.stringify(rows[0].vector1))).toEqual([1, 2, 3]);
|
|
||||||
expect(JSON.parse(JSON.stringify(rows[0].vector2))).toEqual([4, 5, 6]);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("should append generated vectors to a non-nullable schema", async () => {
|
|
||||||
@register("non_nullable_schema_test")
|
|
||||||
class MockEmbeddingFunction extends EmbeddingFunction<string> {
|
|
||||||
ndims() {
|
|
||||||
return 3;
|
|
||||||
}
|
|
||||||
embeddingDataType(): Float {
|
|
||||||
return new Float64();
|
|
||||||
}
|
|
||||||
async computeSourceEmbeddings(data: string[]) {
|
|
||||||
return data.map(() => [1, 2, 3]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
const schema = new Schema([
|
|
||||||
new Field("id", new Int32()),
|
|
||||||
new Field("text", new Utf8()),
|
|
||||||
new Field("type", new Utf8()),
|
|
||||||
new Field(
|
|
||||||
"vector",
|
|
||||||
new FixedSizeList(3, new Field("item", new Float64())),
|
|
||||||
),
|
|
||||||
]);
|
|
||||||
const func = new MockEmbeddingFunction();
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
const table = await db.createEmptyTable("test_non_nullable", schema, {
|
|
||||||
embeddingFunction: {
|
|
||||||
function: func,
|
|
||||||
sourceColumn: "text",
|
|
||||||
},
|
|
||||||
});
|
|
||||||
|
|
||||||
const data = [
|
|
||||||
{ id: 1, text: "Carrot", type: "vegetable" },
|
|
||||||
{ id: 2, text: "Apple", type: "fruit" },
|
|
||||||
];
|
|
||||||
const buffer = await fromDataToBuffer(
|
|
||||||
data,
|
|
||||||
undefined,
|
|
||||||
await table.schema(),
|
|
||||||
);
|
|
||||||
const generatedTable = tableFromIPC(buffer);
|
|
||||||
const vectorField = generatedTable.schema.fields.find(
|
|
||||||
(field) => field.name === "vector",
|
|
||||||
);
|
|
||||||
expect(vectorField?.nullable).toBe(false);
|
|
||||||
|
|
||||||
await table.add(data);
|
|
||||||
|
|
||||||
const rows = await table.query().toArray();
|
|
||||||
expect(rows).toHaveLength(2);
|
|
||||||
for (const row of rows) {
|
|
||||||
expect([...row.vector]).toEqual([1, 2, 3]);
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
it("should error when appending to a table with an unregistered embedding function", async () => {
|
it("should error when appending to a table with an unregistered embedding function", async () => {
|
||||||
@register("mock")
|
@register("mock")
|
||||||
class MockEmbeddingFunction extends EmbeddingFunction<string> {
|
class MockEmbeddingFunction extends EmbeddingFunction<string> {
|
||||||
@@ -539,52 +427,4 @@ describe("embedding functions", () => {
|
|||||||
expect(stringSchema3).toEqual(stringExpectedSchema);
|
expect(stringSchema3).toEqual(stringExpectedSchema);
|
||||||
},
|
},
|
||||||
);
|
);
|
||||||
test("parses one function writing several vector columns", async () => {
|
|
||||||
class MockEmbeddingFunction extends EmbeddingFunction<string> {
|
|
||||||
ndims() {
|
|
||||||
return 3;
|
|
||||||
}
|
|
||||||
embeddingDataType(): Float {
|
|
||||||
return new Float32();
|
|
||||||
}
|
|
||||||
async computeQueryEmbeddings(_data: string) {
|
|
||||||
return [1, 2, 3];
|
|
||||||
}
|
|
||||||
async computeSourceEmbeddings(data: string[]) {
|
|
||||||
return Array.from({ length: data.length }).fill([
|
|
||||||
1, 2, 3,
|
|
||||||
]) as number[][];
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const registry = getRegistry();
|
|
||||||
registry.register("multi_output_mock")(MockEmbeddingFunction);
|
|
||||||
|
|
||||||
// A materialized view can project one source vector column under two
|
|
||||||
// names, so a table's configuration names the same function twice.
|
|
||||||
const parsed = await registry.parseFunctions(
|
|
||||||
new Map([
|
|
||||||
[
|
|
||||||
"embedding_functions",
|
|
||||||
JSON.stringify([
|
|
||||||
{
|
|
||||||
name: "multi_output_mock",
|
|
||||||
sourceColumn: "text",
|
|
||||||
vectorColumn: "vector_a",
|
|
||||||
model: {},
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "multi_output_mock",
|
|
||||||
sourceColumn: "text",
|
|
||||||
vectorColumn: "vector_b",
|
|
||||||
model: {},
|
|
||||||
},
|
|
||||||
]),
|
|
||||||
],
|
|
||||||
]),
|
|
||||||
);
|
|
||||||
|
|
||||||
expect(
|
|
||||||
[...parsed.values()].map(({ vectorColumn }) => vectorColumn).sort(),
|
|
||||||
).toEqual(["vector_a", "vector_b"]);
|
|
||||||
});
|
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -1,95 +0,0 @@
|
|||||||
// SPDX-License-Identifier: Apache-2.0
|
|
||||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
|
||||||
|
|
||||||
import { execFileSync } from "node:child_process";
|
|
||||||
import { resolve } from "node:path";
|
|
||||||
|
|
||||||
import type { OpenAIEmbeddingFunction } from "../lancedb/embedding/openai";
|
|
||||||
import type { EmbeddingFunctionRegistry } from "../lancedb/embedding/registry";
|
|
||||||
|
|
||||||
type EmbeddingModule = typeof import("../lancedb/embedding");
|
|
||||||
type OpenAIModule = typeof import("../lancedb/embedding/openai");
|
|
||||||
type RegistryModule = typeof import("../lancedb/embedding/registry");
|
|
||||||
|
|
||||||
describe("embedding function registry", () => {
|
|
||||||
const registries: EmbeddingFunctionRegistry[] = [];
|
|
||||||
|
|
||||||
afterEach(() => {
|
|
||||||
for (const registry of registries) {
|
|
||||||
registry.reset();
|
|
||||||
}
|
|
||||||
registries.length = 0;
|
|
||||||
});
|
|
||||||
|
|
||||||
it("defers built-in providers until the public registry API is used", () => {
|
|
||||||
jest.isolateModules(() => {
|
|
||||||
const embedding = require("../lancedb/embedding") as EmbeddingModule;
|
|
||||||
const { getRegistry: getInternalRegistry } =
|
|
||||||
require("../lancedb/embedding/registry") as RegistryModule;
|
|
||||||
const registry = getInternalRegistry();
|
|
||||||
registries.push(registry);
|
|
||||||
|
|
||||||
expect(registry.length()).toBe(0);
|
|
||||||
expect(embedding.getRegistry()).toBe(registry);
|
|
||||||
expect(registry.get("openai")).toBeDefined();
|
|
||||||
expect(registry.get("huggingface")).toBeDefined();
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
it("preserves automatic FTS search in a fresh process", () => {
|
|
||||||
execFileSync(
|
|
||||||
process.execPath,
|
|
||||||
[resolve(__dirname, "fixtures", "auto_fts_search.cjs")],
|
|
||||||
{ stdio: "pipe" },
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("shares registrations across duplicated provider module graphs", () => {
|
|
||||||
let registeringRegistry: EmbeddingFunctionRegistry | undefined;
|
|
||||||
let latestOpenAIConstructor: typeof OpenAIEmbeddingFunction | undefined;
|
|
||||||
|
|
||||||
jest.isolateModules(() => {
|
|
||||||
require("../lancedb/embedding/openai");
|
|
||||||
const { getRegistry } =
|
|
||||||
require("../lancedb/embedding/registry") as RegistryModule;
|
|
||||||
registeringRegistry = getRegistry();
|
|
||||||
registries.push(registeringRegistry);
|
|
||||||
expect(registeringRegistry.get("openai")).toBeDefined();
|
|
||||||
});
|
|
||||||
|
|
||||||
expect(() => {
|
|
||||||
jest.isolateModules(() => {
|
|
||||||
const { OpenAIEmbeddingFunction } =
|
|
||||||
require("../lancedb/embedding/openai") as OpenAIModule;
|
|
||||||
latestOpenAIConstructor = OpenAIEmbeddingFunction;
|
|
||||||
const { getRegistry } =
|
|
||||||
require("../lancedb/embedding/registry") as RegistryModule;
|
|
||||||
registries.push(getRegistry());
|
|
||||||
});
|
|
||||||
}).not.toThrow();
|
|
||||||
|
|
||||||
const previousApiKey = process.env.OPENAI_API_KEY;
|
|
||||||
process.env.OPENAI_API_KEY = "test";
|
|
||||||
try {
|
|
||||||
const latestOpenAI = registeringRegistry!
|
|
||||||
.get<OpenAIEmbeddingFunction>("openai")!
|
|
||||||
.create();
|
|
||||||
expect(latestOpenAI).toBeInstanceOf(latestOpenAIConstructor!);
|
|
||||||
} finally {
|
|
||||||
if (previousApiKey === undefined) {
|
|
||||||
delete process.env.OPENAI_API_KEY;
|
|
||||||
} else {
|
|
||||||
process.env.OPENAI_API_KEY = previousApiKey;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
jest.isolateModules(() => {
|
|
||||||
const { getRegistry } =
|
|
||||||
require("../lancedb/embedding") as EmbeddingModule;
|
|
||||||
const publicRegistry = getRegistry();
|
|
||||||
registries.push(publicRegistry);
|
|
||||||
expect(publicRegistry).toBe(registeringRegistry);
|
|
||||||
expect(publicRegistry.get("openai")).toBeDefined();
|
|
||||||
});
|
|
||||||
});
|
|
||||||
});
|
|
||||||
@@ -1,33 +0,0 @@
|
|||||||
// SPDX-License-Identifier: Apache-2.0
|
|
||||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
|
||||||
|
|
||||||
const assert = require("node:assert/strict");
|
|
||||||
const tmp = require("tmp");
|
|
||||||
const { connect, embedding, Index } = require("../../dist");
|
|
||||||
const { getRegistry } = require("../../dist/embedding/registry");
|
|
||||||
|
|
||||||
async function main() {
|
|
||||||
assert.equal(typeof embedding.getRegistry, "function");
|
|
||||||
assert.equal(getRegistry().length(), 0);
|
|
||||||
assert.equal(embedding.getRegistry(), getRegistry());
|
|
||||||
assert.equal(getRegistry().length(), 2);
|
|
||||||
|
|
||||||
const dir = tmp.dirSync({ unsafeCleanup: true });
|
|
||||||
let db;
|
|
||||||
try {
|
|
||||||
db = await connect(dir.name);
|
|
||||||
const table = await db.createTable("docs", [{ text: "hello world" }]);
|
|
||||||
await table.createIndex("text", { config: Index.fts() });
|
|
||||||
|
|
||||||
const rows = await table.search("hello").toArray();
|
|
||||||
assert.equal(rows[0].text, "hello world");
|
|
||||||
} finally {
|
|
||||||
db?.close();
|
|
||||||
dir.removeCallback();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
main().catch((error) => {
|
|
||||||
console.error(error);
|
|
||||||
process.exitCode = 1;
|
|
||||||
});
|
|
||||||
@@ -1,147 +0,0 @@
|
|||||||
// SPDX-License-Identifier: Apache-2.0
|
|
||||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
|
||||||
|
|
||||||
import * as tmp from "tmp";
|
|
||||||
|
|
||||||
import { Connection, connect } from "../lancedb";
|
|
||||||
import {
|
|
||||||
DEFINITION_META_KEY,
|
|
||||||
definitionFromMetadata,
|
|
||||||
} from "../lancedb/materialized_view";
|
|
||||||
|
|
||||||
describe("materialized views", () => {
|
|
||||||
let tmpDir: tmp.DirResult;
|
|
||||||
let db: Connection;
|
|
||||||
|
|
||||||
beforeEach(async () => {
|
|
||||||
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
|
||||||
db = await connect(tmpDir.name);
|
|
||||||
await db.createTable(
|
|
||||||
"people",
|
|
||||||
[
|
|
||||||
{ name: "ada", age: 36 },
|
|
||||||
{ name: "kid", age: 7 },
|
|
||||||
{ name: "grace", age: 85 },
|
|
||||||
],
|
|
||||||
{ storageOptions: { newTableEnableStableRowIds: "true" } },
|
|
||||||
);
|
|
||||||
});
|
|
||||||
afterEach(() => tmpDir.removeCallback());
|
|
||||||
|
|
||||||
it("rejects a stored limit a number cannot carry", () => {
|
|
||||||
const big = new Map([
|
|
||||||
[
|
|
||||||
DEFINITION_META_KEY,
|
|
||||||
'{"kind":"select","source_table":"people","limit":9007199254740993}',
|
|
||||||
],
|
|
||||||
]);
|
|
||||||
expect(() => definitionFromMetadata(big, "v")).toThrow(
|
|
||||||
/too large to represent exactly/,
|
|
||||||
);
|
|
||||||
|
|
||||||
const safe = new Map([
|
|
||||||
[
|
|
||||||
DEFINITION_META_KEY,
|
|
||||||
'{"kind":"select","source_table":"people","limit":42}',
|
|
||||||
],
|
|
||||||
]);
|
|
||||||
expect(definitionFromMetadata(safe, "v").limit).toBe(42);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("creates, refreshes and queries a view", async () => {
|
|
||||||
const view = await db.createMaterializedView("adults", "people", {
|
|
||||||
select: ["name", ["shout", "upper(name)"]],
|
|
||||||
where: "age >= 18",
|
|
||||||
});
|
|
||||||
expect(view.name).toBe("adults");
|
|
||||||
expect(await view.table().countRows()).toBe(0);
|
|
||||||
|
|
||||||
const result = await view.refresh();
|
|
||||||
expect(result.mode).toBe("rebuild");
|
|
||||||
expect(Number(result.rowsWritten)).toBe(2);
|
|
||||||
|
|
||||||
const rows = await view.table().query().toArray();
|
|
||||||
expect(rows.map((r) => r.shout).sort()).toEqual(["ADA", "GRACE"]);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("round-trips the definition", async () => {
|
|
||||||
await db.createMaterializedView("adults", "people", {
|
|
||||||
where: "age >= 18",
|
|
||||||
});
|
|
||||||
const view = await db.openMaterializedView("adults");
|
|
||||||
const definition = await view.definition();
|
|
||||||
expect(definition.sourceTable).toBe("people");
|
|
||||||
expect(definition.filter).toBe("age >= 18");
|
|
||||||
expect(definition.projections).toEqual([
|
|
||||||
["name", "`name`"],
|
|
||||||
["age", "`age`"],
|
|
||||||
]);
|
|
||||||
expect(definition.inputs).toEqual(["age", "name"]);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("refreshes incrementally after an append", async () => {
|
|
||||||
const view = await db.createMaterializedView("copy", "people");
|
|
||||||
await view.refresh();
|
|
||||||
|
|
||||||
const people = await db.openTable("people");
|
|
||||||
await people.add([{ name: "alan", age: 41 }]);
|
|
||||||
const result = await view.refresh();
|
|
||||||
expect(result.mode).toBe("incremental");
|
|
||||||
expect(Number(result.rowsWritten)).toBe(1);
|
|
||||||
expect(await view.table().countRows()).toBe(4);
|
|
||||||
|
|
||||||
expect((await view.refresh()).mode).toBe("no_op");
|
|
||||||
});
|
|
||||||
|
|
||||||
it("lists views and rejects non-views", async () => {
|
|
||||||
await db.createMaterializedView("adults", "people", {
|
|
||||||
where: "age >= 18",
|
|
||||||
});
|
|
||||||
expect(await db.listMaterializedViews()).toEqual(["adults"]);
|
|
||||||
await expect(db.openMaterializedView("people")).rejects.toThrow(
|
|
||||||
"not a materialized view",
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("rejects an invalid expression at create time", async () => {
|
|
||||||
await expect(
|
|
||||||
db.createMaterializedView("bad", "people", {
|
|
||||||
select: [["x", "missing + 1"]],
|
|
||||||
}),
|
|
||||||
).rejects.toThrow("missing");
|
|
||||||
});
|
|
||||||
|
|
||||||
it("rejects invalid numeric options before creating anything", async () => {
|
|
||||||
for (const limit of [-5, 1.5, Infinity, NaN]) {
|
|
||||||
await expect(
|
|
||||||
db.createMaterializedView("bad", "people", { limit }),
|
|
||||||
).rejects.toThrow("non-negative integer");
|
|
||||||
}
|
|
||||||
expect(await db.listMaterializedViews()).toEqual([]);
|
|
||||||
|
|
||||||
const view = await db.createMaterializedView("copy", "people");
|
|
||||||
for (const sourceVersion of [-1, 1.5, Infinity, NaN]) {
|
|
||||||
await expect(view.refresh({ sourceVersion })).rejects.toThrow(
|
|
||||||
"non-negative integer",
|
|
||||||
);
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
it("quotes bare select names", async () => {
|
|
||||||
await db.createTable("odd_names", [{ "order item": "widget" }], {
|
|
||||||
storageOptions: { newTableEnableStableRowIds: "true" },
|
|
||||||
});
|
|
||||||
const view = await db.createMaterializedView("quoted", "odd_names", {
|
|
||||||
select: ["order item"],
|
|
||||||
});
|
|
||||||
const result = await view.refresh();
|
|
||||||
expect(Number(result.rowsWritten)).toBe(1);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("requires stable row ids on the source", async () => {
|
|
||||||
await db.createTable("plain", [{ x: 1 }]);
|
|
||||||
await expect(db.createMaterializedView("v", "plain")).rejects.toThrow(
|
|
||||||
"stable row ids",
|
|
||||||
);
|
|
||||||
});
|
|
||||||
});
|
|
||||||
@@ -1,14 +0,0 @@
|
|||||||
// SPDX-License-Identifier: Apache-2.0
|
|
||||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
|
||||||
|
|
||||||
import packageJson = require("../package.json");
|
|
||||||
|
|
||||||
describe("package metadata", () => {
|
|
||||||
it("requires Node.js type declarations compatible with the runtime", () => {
|
|
||||||
expect(packageJson.engines.node).toBe(">= 18");
|
|
||||||
expect(packageJson.peerDependencies["@types/node"]).toBe(">=18");
|
|
||||||
expect(packageJson.peerDependenciesMeta["@types/node"]).toEqual({
|
|
||||||
optional: true,
|
|
||||||
});
|
|
||||||
});
|
|
||||||
});
|
|
||||||
@@ -110,81 +110,6 @@ describe("Query outputSchema", () => {
|
|||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
describe("Search pagination", () => {
|
|
||||||
let tmpDir: tmp.DirResult;
|
|
||||||
let table: Table;
|
|
||||||
|
|
||||||
beforeEach(async () => {
|
|
||||||
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
const schema = new Schema([
|
|
||||||
new Field("id", new Int64(), false),
|
|
||||||
new Field("text", new Utf8(), false),
|
|
||||||
new Field(
|
|
||||||
"vector",
|
|
||||||
new FixedSizeList(2, new Field("item", new Float32())),
|
|
||||||
false,
|
|
||||||
),
|
|
||||||
]);
|
|
||||||
const data = makeArrowTable(
|
|
||||||
[
|
|
||||||
{ id: 1n, text: "common", vector: [0, 0] },
|
|
||||||
{ id: 2n, text: "common common", vector: [1, 1] },
|
|
||||||
{ id: 3n, text: "common common common", vector: [2, 2] },
|
|
||||||
{ id: 4n, text: "common common common common", vector: [3, 3] },
|
|
||||||
],
|
|
||||||
{ schema },
|
|
||||||
);
|
|
||||||
table = await db.createTable("test", data);
|
|
||||||
});
|
|
||||||
|
|
||||||
afterEach(() => {
|
|
||||||
tmpDir.removeCallback();
|
|
||||||
});
|
|
||||||
|
|
||||||
it("applies offset after the vector search limit", async () => {
|
|
||||||
const allResults = await table
|
|
||||||
.vectorSearch([0, 0])
|
|
||||||
.select(["id"])
|
|
||||||
.limit(4)
|
|
||||||
.toArray();
|
|
||||||
const secondPage = await table
|
|
||||||
.vectorSearch([0, 0])
|
|
||||||
.select(["id"])
|
|
||||||
.limit(2)
|
|
||||||
.offset(2)
|
|
||||||
.toArray();
|
|
||||||
|
|
||||||
expect(allResults).toHaveLength(4);
|
|
||||||
expect(secondPage).toHaveLength(2);
|
|
||||||
expect(secondPage.map((row) => row.id)).toEqual(
|
|
||||||
allResults.slice(2, 4).map((row) => row.id),
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("applies offset after the full-text search limit", async () => {
|
|
||||||
await table.createIndex("text", { config: Index.fts() });
|
|
||||||
|
|
||||||
const allResults = await table
|
|
||||||
.search("common", "fts")
|
|
||||||
.select(["id"])
|
|
||||||
.limit(4)
|
|
||||||
.toArray();
|
|
||||||
const secondPage = await table
|
|
||||||
.search("common", "fts")
|
|
||||||
.select(["id"])
|
|
||||||
.limit(2)
|
|
||||||
.offset(2)
|
|
||||||
.toArray();
|
|
||||||
|
|
||||||
expect(allResults).toHaveLength(4);
|
|
||||||
expect(secondPage).toHaveLength(2);
|
|
||||||
expect(secondPage.map((row) => row.id)).toEqual(
|
|
||||||
allResults.slice(2, 4).map((row) => row.id),
|
|
||||||
);
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
describe("Query orderBy", () => {
|
describe("Query orderBy", () => {
|
||||||
let tmpDir: tmp.DirResult;
|
let tmpDir: tmp.DirResult;
|
||||||
let table: Table;
|
let table: Table;
|
||||||
|
|||||||
@@ -106,77 +106,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])("Registry", (arrow) => {
|
|||||||
'Embedding function with alias "mock-embedding" already exists',
|
'Embedding function with alias "mock-embedding" already exists',
|
||||||
);
|
);
|
||||||
});
|
});
|
||||||
test("parseFunctions keeps entries sharing a function name", async () => {
|
|
||||||
class MockEmbeddingFunction extends EmbeddingFunction<string> {
|
|
||||||
ndims() {
|
|
||||||
return 3;
|
|
||||||
}
|
|
||||||
embeddingDataType() {
|
|
||||||
return new arrow.Float32() as apiArrow.Float;
|
|
||||||
}
|
|
||||||
async computeSourceEmbeddings(data: string[]) {
|
|
||||||
return data.map(() => [1, 2, 3]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
register("mock-embedding")(MockEmbeddingFunction);
|
|
||||||
const parsed = await getRegistry().parseFunctions(
|
|
||||||
new Map([
|
|
||||||
[
|
|
||||||
"embedding_functions",
|
|
||||||
JSON.stringify([
|
|
||||||
{
|
|
||||||
name: "mock-embedding",
|
|
||||||
sourceColumn: "text",
|
|
||||||
vectorColumn: "vector_a",
|
|
||||||
model: {},
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "mock-embedding",
|
|
||||||
sourceColumn: "text",
|
|
||||||
vectorColumn: "vector_b",
|
|
||||||
model: {},
|
|
||||||
},
|
|
||||||
]),
|
|
||||||
],
|
|
||||||
]),
|
|
||||||
);
|
|
||||||
expect([...parsed.values()].map((f) => f.vectorColumn)).toEqual([
|
|
||||||
"vector_a",
|
|
||||||
"vector_b",
|
|
||||||
]);
|
|
||||||
|
|
||||||
// The Python bindings write snake_case keys.
|
|
||||||
const snake = await getRegistry().parseFunctions(
|
|
||||||
new Map([
|
|
||||||
[
|
|
||||||
"embedding_functions",
|
|
||||||
JSON.stringify([
|
|
||||||
{
|
|
||||||
name: "mock-embedding",
|
|
||||||
// biome-ignore lint/style/useNamingConvention: the Python wire spelling
|
|
||||||
source_column: "text",
|
|
||||||
// biome-ignore lint/style/useNamingConvention: the Python wire spelling
|
|
||||||
vector_column: "vector_a",
|
|
||||||
model: {},
|
|
||||||
},
|
|
||||||
{
|
|
||||||
name: "mock-embedding",
|
|
||||||
// biome-ignore lint/style/useNamingConvention: the Python wire spelling
|
|
||||||
source_column: "text",
|
|
||||||
// biome-ignore lint/style/useNamingConvention: the Python wire spelling
|
|
||||||
vector_column: "vector_b",
|
|
||||||
model: {},
|
|
||||||
},
|
|
||||||
]),
|
|
||||||
],
|
|
||||||
]),
|
|
||||||
);
|
|
||||||
expect([...snake.keys()]).toEqual(["vector_a", "vector_b"]);
|
|
||||||
expect([...snake.values()].map((f) => f.sourceColumn)).toEqual([
|
|
||||||
"text",
|
|
||||||
"text",
|
|
||||||
]);
|
|
||||||
});
|
|
||||||
test("schema should contain correct metadata", async () => {
|
test("schema should contain correct metadata", async () => {
|
||||||
class MockEmbeddingFunction extends EmbeddingFunction<string> {
|
class MockEmbeddingFunction extends EmbeddingFunction<string> {
|
||||||
constructor(args: FunctionOptions = {}) {
|
constructor(args: FunctionOptions = {}) {
|
||||||
|
|||||||
+14
-156
@@ -75,25 +75,6 @@ async function withMockDatabase(
|
|||||||
}
|
}
|
||||||
|
|
||||||
describe("remote connection", () => {
|
describe("remote connection", () => {
|
||||||
it("refuses materialized views before issuing any request", async () => {
|
|
||||||
const paths: string[] = [];
|
|
||||||
await withMockDatabase(
|
|
||||||
(req, res) => {
|
|
||||||
paths.push(req.url ?? "");
|
|
||||||
res.writeHead(404).end();
|
|
||||||
},
|
|
||||||
async (db) => {
|
|
||||||
await expect(db.openMaterializedView("secret_table")).rejects.toThrow(
|
|
||||||
/only on local databases/,
|
|
||||||
);
|
|
||||||
await expect(db.listMaterializedViews()).rejects.toThrow(
|
|
||||||
/only on local databases/,
|
|
||||||
);
|
|
||||||
expect(paths).toEqual([]);
|
|
||||||
},
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("should accept partial connection options", async () => {
|
it("should accept partial connection options", async () => {
|
||||||
await connect("db://test", {
|
await connect("db://test", {
|
||||||
apiKey: "fake",
|
apiKey: "fake",
|
||||||
@@ -189,38 +170,6 @@ describe("remote connection", () => {
|
|||||||
);
|
);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("surfaces JSON server errors from remote table operations", async () => {
|
|
||||||
await withMockDatabase(
|
|
||||||
(req, res) => {
|
|
||||||
const path = req.url ?? "";
|
|
||||||
if (path.endsWith("/describe/")) {
|
|
||||||
res.writeHead(200, { "Content-Type": "application/json" }).end(
|
|
||||||
JSON.stringify({
|
|
||||||
name: "broken_table",
|
|
||||||
version: 1,
|
|
||||||
schema: { fields: [] },
|
|
||||||
}),
|
|
||||||
);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (path.endsWith("/count_rows/")) {
|
|
||||||
res
|
|
||||||
.writeHead(400, { "Content-Type": "application/json" })
|
|
||||||
.end(JSON.stringify({ error: "count rows failed" }));
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
res.writeHead(404).end();
|
|
||||||
},
|
|
||||||
async (db) => {
|
|
||||||
const table = await db.openTable("broken_table");
|
|
||||||
|
|
||||||
await expect(table.countRows()).rejects.toThrow("count rows failed");
|
|
||||||
},
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("should pass on requested extra headers", async () => {
|
it("should pass on requested extra headers", async () => {
|
||||||
await withMockDatabase(
|
await withMockDatabase(
|
||||||
(req, res) => {
|
(req, res) => {
|
||||||
@@ -330,7 +279,7 @@ describe("remote connection", () => {
|
|||||||
expect(createIndexBody?.["custom_stop_words"]).toEqual(["the"]);
|
expect(createIndexBody?.["custom_stop_words"]).toEqual(["the"]);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("diffs and cherry-picks remote branches", async () => {
|
it("diffs and merges remote branches", async () => {
|
||||||
const sampleDiff = {
|
const sampleDiff = {
|
||||||
fromBranch: "exp",
|
fromBranch: "exp",
|
||||||
parentVersion: 1,
|
parentVersion: 1,
|
||||||
@@ -352,9 +301,10 @@ describe("remote connection", () => {
|
|||||||
changedColumns: [],
|
changedColumns: [],
|
||||||
addedIndexes: [],
|
addedIndexes: [],
|
||||||
removedIndexes: [],
|
removedIndexes: [],
|
||||||
errors: [],
|
mergeable: true,
|
||||||
|
mergeBlockers: [],
|
||||||
};
|
};
|
||||||
const cherryPickBodies: Record<string, unknown>[] = [];
|
const mergeBodies: Record<string, unknown>[] = [];
|
||||||
|
|
||||||
await withMockDatabase(
|
await withMockDatabase(
|
||||||
(req, res) => {
|
(req, res) => {
|
||||||
@@ -384,16 +334,17 @@ describe("remote connection", () => {
|
|||||||
.end(JSON.stringify(sampleDiff));
|
.end(JSON.stringify(sampleDiff));
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if (path.endsWith("/branches/cherry_pick/")) {
|
if (path.endsWith("/branches/merge/")) {
|
||||||
cherryPickBodies.push(body);
|
mergeBodies.push(body);
|
||||||
const dryRun = body["dry_run"] === true;
|
const dryRun = body["dry_run"] === true;
|
||||||
const response = {
|
const response = {
|
||||||
status: dryRun ? "ready" : "failed",
|
status: dryRun ? "ready" : "rejected",
|
||||||
diff: dryRun
|
diff: dryRun
|
||||||
? sampleDiff
|
? sampleDiff
|
||||||
: {
|
: {
|
||||||
...sampleDiff,
|
...sampleDiff,
|
||||||
errors: [
|
mergeable: false,
|
||||||
|
mergeBlockers: [
|
||||||
{ code: "baseMoved", message: "main has advanced" },
|
{ code: "baseMoved", message: "main has advanced" },
|
||||||
],
|
],
|
||||||
},
|
},
|
||||||
@@ -415,19 +366,19 @@ describe("remote connection", () => {
|
|||||||
|
|
||||||
await expect(branches.diff("exp")).resolves.toEqual(sampleDiff);
|
await expect(branches.diff("exp")).resolves.toEqual(sampleDiff);
|
||||||
|
|
||||||
const failed = await branches.cherryPick("exp");
|
const rejected = await branches.merge("exp");
|
||||||
expect(failed.status).toBe("failed");
|
expect(rejected.status).toBe("rejected");
|
||||||
expect(failed.diff.errors).toEqual([
|
expect(rejected.diff.mergeBlockers).toEqual([
|
||||||
{ code: "baseMoved", message: "main has advanced" },
|
{ code: "baseMoved", message: "main has advanced" },
|
||||||
]);
|
]);
|
||||||
|
|
||||||
const preview = await branches.cherryPick("exp", true);
|
const preview = await branches.merge("exp", true);
|
||||||
expect(preview.status).toBe("ready");
|
expect(preview.status).toBe("ready");
|
||||||
expect(preview.preview.promotedColumns).toEqual(["tag"]);
|
expect(preview.preview.promotedColumns).toEqual(["tag"]);
|
||||||
},
|
},
|
||||||
);
|
);
|
||||||
|
|
||||||
expect(cherryPickBodies).toEqual([
|
expect(mergeBodies).toEqual([
|
||||||
// biome-ignore lint/style/useNamingConvention: snake_case mandated by the server wire format
|
// biome-ignore lint/style/useNamingConvention: snake_case mandated by the server wire format
|
||||||
{ from_branch: "exp", dry_run: false },
|
{ from_branch: "exp", dry_run: false },
|
||||||
// biome-ignore lint/style/useNamingConvention: snake_case mandated by the server wire format
|
// biome-ignore lint/style/useNamingConvention: snake_case mandated by the server wire format
|
||||||
@@ -926,96 +877,3 @@ describe("remote connection", () => {
|
|||||||
});
|
});
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
describe("remote connection jobs surface", () => {
|
|
||||||
it("lists, describes, cancels, and reads history", async () => {
|
|
||||||
const { tableFromArrays, tableToIPC } = await import("apache-arrow");
|
|
||||||
const eventsTable = tableFromArrays({ state: ["created", "succeeded"] });
|
|
||||||
const eventsBody = Buffer.from(tableToIPC(eventsTable, "stream"));
|
|
||||||
|
|
||||||
await withMockDatabase(
|
|
||||||
(req, res) => {
|
|
||||||
let body = "";
|
|
||||||
req.on("data", (chunk) => {
|
|
||||||
body += chunk;
|
|
||||||
});
|
|
||||||
req.on("end", () => {
|
|
||||||
const payload = body.length > 0 ? JSON.parse(body) : {};
|
|
||||||
if (req.url === "/v1/jobs/list") {
|
|
||||||
if (payload["page_token"] === undefined) {
|
|
||||||
res
|
|
||||||
.writeHead(200, { "Content-Type": "application/json" })
|
|
||||||
.end(
|
|
||||||
'{"jobs": [{"job_id": "job-1", "table": "t1", ' +
|
|
||||||
'"job_type": "create_index", "state": "in_progress", ' +
|
|
||||||
'"created_at_millis": 1000}], "page_token": "next"}',
|
|
||||||
);
|
|
||||||
} else {
|
|
||||||
res
|
|
||||||
.writeHead(200, { "Content-Type": "application/json" })
|
|
||||||
.end(
|
|
||||||
'{"jobs": [{"job_id": "job-2", "table": "t2", ' +
|
|
||||||
'"job_type": "create_index", "state": "succeeded", ' +
|
|
||||||
'"created_at_millis": 2000}]}',
|
|
||||||
);
|
|
||||||
}
|
|
||||||
} else if (req.url === "/v1/jobs/describe") {
|
|
||||||
if (payload["job_id"] !== "job-1") {
|
|
||||||
res.writeHead(404).end("no such job");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
res
|
|
||||||
.writeHead(200, { "Content-Type": "application/json" })
|
|
||||||
.end(
|
|
||||||
'{"job_id": "job-1", "job_type": "create_index", ' +
|
|
||||||
'"job_state": "FAILED", "creation_ms": 1000, ' +
|
|
||||||
'"spec": {"column": "vec"}, "failure": {"phase": "execute", ' +
|
|
||||||
'"message": "worker died", "retryable": true}}',
|
|
||||||
);
|
|
||||||
} else if (req.url === "/v1/jobs/cancel") {
|
|
||||||
if (payload["job_id"] !== "job-1") {
|
|
||||||
res.writeHead(404).end("no such job");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
res
|
|
||||||
.writeHead(200, { "Content-Type": "application/json" })
|
|
||||||
.end('{"job_id": "job-1"}');
|
|
||||||
} else if (req.url === "/v1/jobs/query_events") {
|
|
||||||
res
|
|
||||||
.writeHead(200, {
|
|
||||||
"Content-Type": "application/vnd.apache.arrow.stream",
|
|
||||||
})
|
|
||||||
.end(eventsBody);
|
|
||||||
} else {
|
|
||||||
res.writeHead(404).end();
|
|
||||||
}
|
|
||||||
});
|
|
||||||
},
|
|
||||||
async (db) => {
|
|
||||||
const jobs = await db.listJobs();
|
|
||||||
expect(jobs.map((job) => job.jobId)).toEqual(["job-1", "job-2"]);
|
|
||||||
expect(jobs[0].state).toEqual("running");
|
|
||||||
expect(jobs[1].state).toEqual("finished");
|
|
||||||
|
|
||||||
const description = await db.getJob("job-1");
|
|
||||||
expect(description?.state).toEqual("failed");
|
|
||||||
expect(JSON.parse(description?.specJson ?? "")).toEqual({
|
|
||||||
column: "vec",
|
|
||||||
});
|
|
||||||
expect(description?.failure?.message).toEqual("worker died");
|
|
||||||
expect(await db.getJob("missing")).toBeNull();
|
|
||||||
|
|
||||||
expect(await db.cancelJob("job-1")).toBe(true);
|
|
||||||
expect(await db.cancelJob("missing")).toBe(false);
|
|
||||||
|
|
||||||
const history = await db.jobHistory("job-1");
|
|
||||||
expect(history.numRows).toEqual(2);
|
|
||||||
|
|
||||||
const job = db.job("job-1");
|
|
||||||
expect(job.id).toEqual("job-1");
|
|
||||||
expect(await job.status()).toEqual("failed");
|
|
||||||
await expect(job.wait()).rejects.toThrow("worker died");
|
|
||||||
},
|
|
||||||
);
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|||||||
@@ -11,13 +11,10 @@ import * as arrow17 from "apache-arrow-17";
|
|||||||
import * as arrow18 from "apache-arrow-18";
|
import * as arrow18 from "apache-arrow-18";
|
||||||
|
|
||||||
import {
|
import {
|
||||||
AutoQuery,
|
|
||||||
Connection,
|
Connection,
|
||||||
MatchQuery,
|
MatchQuery,
|
||||||
PhraseQuery,
|
PhraseQuery,
|
||||||
Query,
|
|
||||||
Table,
|
Table,
|
||||||
VectorQuery,
|
|
||||||
connect,
|
connect,
|
||||||
tokenize,
|
tokenize,
|
||||||
} from "../lancedb";
|
} from "../lancedb";
|
||||||
@@ -89,44 +86,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
await expect(table.countRows()).resolves.toBe(3);
|
await expect(table.countRows()).resolves.toBe(3);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("should support a foreign Float64 vector schema end to end", async () => {
|
|
||||||
const conn = await connect(tmpDir.name);
|
|
||||||
const schema = new arrow.Schema([
|
|
||||||
new arrow.Field("resource_id", new arrow.Int32(), false),
|
|
||||||
new arrow.Field(
|
|
||||||
"vector",
|
|
||||||
new arrow.FixedSizeList(
|
|
||||||
3,
|
|
||||||
new arrow.Field("value", new arrow.Float64(), true),
|
|
||||||
),
|
|
||||||
false,
|
|
||||||
),
|
|
||||||
]);
|
|
||||||
const data = [
|
|
||||||
{
|
|
||||||
// biome-ignore lint/style/useNamingConvention: matches the reported schema
|
|
||||||
resource_id: 0,
|
|
||||||
vector: [0.1, 0.1, 0.1],
|
|
||||||
},
|
|
||||||
];
|
|
||||||
|
|
||||||
const resources = await conn.createTable("resources", data, { schema });
|
|
||||||
|
|
||||||
const existing = await resources
|
|
||||||
.query()
|
|
||||||
.where("resource_id = 0")
|
|
||||||
.limit(1)
|
|
||||||
.toArray();
|
|
||||||
expect(existing).toHaveLength(1);
|
|
||||||
|
|
||||||
const matched = await resources
|
|
||||||
.search(Float64Array.from(data[0].vector))
|
|
||||||
.limit(1)
|
|
||||||
.toArray();
|
|
||||||
expect(matched).toHaveLength(1);
|
|
||||||
expect(matched[0]["resource_id"]).toBe(0);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("should support branches", async () => {
|
it("should support branches", async () => {
|
||||||
await table.add([{ id: 1 }]);
|
await table.add([{ id: 1 }]);
|
||||||
expect(await table.countRows()).toBe(1);
|
expect(await table.countRows()).toBe(1);
|
||||||
@@ -280,16 +239,8 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
},
|
},
|
||||||
numIndices: 0,
|
numIndices: 0,
|
||||||
numRows: 3,
|
numRows: 3,
|
||||||
// Full on-disk size of the two data files, footers and metadata included.
|
totalBytes: 44,
|
||||||
totalBytes: 684,
|
|
||||||
});
|
});
|
||||||
|
|
||||||
// Index files count toward totalBytes too (only deletion files and
|
|
||||||
// manifests are excluded).
|
|
||||||
await table.createIndex("id", { config: Index.btree() });
|
|
||||||
const statsWithIndex = await table.stats();
|
|
||||||
expect(statsWithIndex.numIndices).toBe(1);
|
|
||||||
expect(statsWithIndex.totalBytes).toBeGreaterThan(684);
|
|
||||||
});
|
});
|
||||||
|
|
||||||
it("should overwrite data if asked", async () => {
|
it("should overwrite data if asked", async () => {
|
||||||
@@ -685,56 +636,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
},
|
},
|
||||||
);
|
);
|
||||||
|
|
||||||
// https://github.com/lancedb/lancedb/issues/1963
|
|
||||||
it("should query documents with LangChain PDF metadata", async () => {
|
|
||||||
const tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
|
||||||
try {
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
const documents = [
|
|
||||||
{
|
|
||||||
text: "first page",
|
|
||||||
vector: [1, 0],
|
|
||||||
source: "first.pdf",
|
|
||||||
loc: { pageNumber: 1, lines: { from: 1, to: 12 } },
|
|
||||||
pdf: {
|
|
||||||
version: "1.10.100",
|
|
||||||
info: {
|
|
||||||
format: "PDF 1.7",
|
|
||||||
producer: "pdf.js",
|
|
||||||
creator: "Writer",
|
|
||||||
},
|
|
||||||
totalPages: 2,
|
|
||||||
},
|
|
||||||
},
|
|
||||||
{
|
|
||||||
text: "second page",
|
|
||||||
vector: [0, 1],
|
|
||||||
source: "second.pdf",
|
|
||||||
loc: { pageNumber: 2, lines: { from: 13, to: 24 } },
|
|
||||||
pdf: {
|
|
||||||
version: "1.10.100",
|
|
||||||
info: {
|
|
||||||
format: "PDF 1.7",
|
|
||||||
producer: "pdf.js",
|
|
||||||
creator: "Writer",
|
|
||||||
},
|
|
||||||
totalPages: 2,
|
|
||||||
},
|
|
||||||
},
|
|
||||||
];
|
|
||||||
const documentsTable = await db.createTable("documents", documents);
|
|
||||||
|
|
||||||
const results = await documentsTable.query().toArray();
|
|
||||||
|
|
||||||
expect(results).toHaveLength(2);
|
|
||||||
expect(results[0].source).toBe("first.pdf");
|
|
||||||
expect(results[0].pdf.info.producer).toBe("pdf.js");
|
|
||||||
expect(results[1].loc.pageNumber).toBe(2);
|
|
||||||
} finally {
|
|
||||||
tmpDir.removeCallback();
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
describe("merge insert", () => {
|
describe("merge insert", () => {
|
||||||
let tmpDir: tmp.DirResult;
|
let tmpDir: tmp.DirResult;
|
||||||
let table: Table;
|
let table: Table;
|
||||||
@@ -950,11 +851,7 @@ describe("When creating an index", () => {
|
|||||||
afterEach(() => tmpDir.removeCallback());
|
afterEach(() => tmpDir.removeCallback());
|
||||||
|
|
||||||
it("should create a vector index on vector columns", async () => {
|
it("should create a vector index on vector columns", async () => {
|
||||||
const job = await tbl.createIndexAsync("vec");
|
await tbl.createIndex("vec");
|
||||||
expect(job.id).toBeNull();
|
|
||||||
await job.wait();
|
|
||||||
// Cancelling a job that already finished succeeds and does nothing.
|
|
||||||
await job.cancel();
|
|
||||||
|
|
||||||
// check index directory
|
// check index directory
|
||||||
const indexDir = path.join(tmpDir.name, "test.lance", "_indices");
|
const indexDir = path.join(tmpDir.name, "test.lance", "_indices");
|
||||||
@@ -1830,194 +1727,6 @@ describe("Read consistency interval", () => {
|
|||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
describe("automatic search schema consistency", () => {
|
|
||||||
let tmpDir: tmp.DirResult;
|
|
||||||
|
|
||||||
class SchemaRefreshEmbedding extends EmbeddingFunction<string> {
|
|
||||||
ndims() {
|
|
||||||
return 2;
|
|
||||||
}
|
|
||||||
|
|
||||||
embeddingDataType() {
|
|
||||||
return new Float32();
|
|
||||||
}
|
|
||||||
|
|
||||||
async computeSourceEmbeddings(data: string[]) {
|
|
||||||
return data.map((value) => [value.length, 1]);
|
|
||||||
}
|
|
||||||
|
|
||||||
async computeQueryEmbeddings(value: string) {
|
|
||||||
return [value.length, 1];
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
function embeddingSchema() {
|
|
||||||
const func = new SchemaRefreshEmbedding();
|
|
||||||
return LanceSchema({
|
|
||||||
text: func.sourceField(new Utf8()),
|
|
||||||
vector: func.vectorField(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
beforeEach(() => {
|
|
||||||
getRegistry().reset();
|
|
||||||
register("schema-refresh")(SchemaRefreshEmbedding);
|
|
||||||
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
|
||||||
});
|
|
||||||
|
|
||||||
afterEach(() => {
|
|
||||||
getRegistry().reset();
|
|
||||||
tmpDir.removeCallback();
|
|
||||||
});
|
|
||||||
|
|
||||||
it("uses the schema refreshed from another connection", async () => {
|
|
||||||
const first = await connect(tmpDir.name, { readConsistencyInterval: 0 });
|
|
||||||
const second = await connect(tmpDir.name, { readConsistencyInterval: 0 });
|
|
||||||
|
|
||||||
try {
|
|
||||||
const stale = await first.createTable("docs", [{ text: "before" }], {
|
|
||||||
schema: embeddingSchema(),
|
|
||||||
});
|
|
||||||
const replacement = await second.createTable(
|
|
||||||
"docs",
|
|
||||||
[{ text: "after hello" }],
|
|
||||||
{ mode: "overwrite" },
|
|
||||||
);
|
|
||||||
await replacement.createIndex("text", { config: Index.fts() });
|
|
||||||
|
|
||||||
const search = stale.search("hello");
|
|
||||||
expect(search).toBeInstanceOf(AutoQuery);
|
|
||||||
expect(search).not.toBeInstanceOf(Query);
|
|
||||||
expect(search).not.toBeInstanceOf(VectorQuery);
|
|
||||||
expect("nprobes" in search).toBe(false);
|
|
||||||
|
|
||||||
const rows = await search.toArray();
|
|
||||||
expect(rows[0].text).toBe("after hello");
|
|
||||||
expect((await stale.schema()).metadata.has("embedding_functions")).toBe(
|
|
||||||
false,
|
|
||||||
);
|
|
||||||
} finally {
|
|
||||||
first.close();
|
|
||||||
second.close();
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
it("tracks embedding metadata across checkout and restore", async () => {
|
|
||||||
const first = await connect(tmpDir.name, { readConsistencyInterval: 0 });
|
|
||||||
const second = await connect(tmpDir.name, { readConsistencyInterval: 0 });
|
|
||||||
|
|
||||||
try {
|
|
||||||
await first.createTable("docs", [{ text: "before" }], {
|
|
||||||
schema: embeddingSchema(),
|
|
||||||
});
|
|
||||||
const table = await second.createTable(
|
|
||||||
"docs",
|
|
||||||
[{ text: "after hello" }],
|
|
||||||
{ mode: "overwrite" },
|
|
||||||
);
|
|
||||||
await table.createIndex("text", { config: Index.fts() });
|
|
||||||
|
|
||||||
await table.checkout(1);
|
|
||||||
expect((await table.search("before").toArray())[0].text).toBe("before");
|
|
||||||
|
|
||||||
await table.checkoutLatest();
|
|
||||||
expect((await table.search("hello").toArray())[0].text).toBe(
|
|
||||||
"after hello",
|
|
||||||
);
|
|
||||||
|
|
||||||
await table.checkout(1);
|
|
||||||
await table.restore();
|
|
||||||
expect((await table.search("before").toArray())[0].text).toBe("before");
|
|
||||||
} finally {
|
|
||||||
first.close();
|
|
||||||
second.close();
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
it("pins automatic search while computing an embedding", async () => {
|
|
||||||
let markStarted!: () => void;
|
|
||||||
let releaseEmbedding!: () => void;
|
|
||||||
const started = new Promise<void>((resolve) => {
|
|
||||||
markStarted = resolve;
|
|
||||||
});
|
|
||||||
const released = new Promise<void>((resolve) => {
|
|
||||||
releaseEmbedding = resolve;
|
|
||||||
});
|
|
||||||
|
|
||||||
class BlockingEmbedding extends SchemaRefreshEmbedding {
|
|
||||||
async computeQueryEmbeddings(value: string) {
|
|
||||||
markStarted();
|
|
||||||
await released;
|
|
||||||
return [value.length, 1];
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
register("schema-refresh-blocking")(BlockingEmbedding);
|
|
||||||
const func = new BlockingEmbedding();
|
|
||||||
const schema = LanceSchema({
|
|
||||||
text: func.sourceField(new Utf8()),
|
|
||||||
vector: func.vectorField(),
|
|
||||||
});
|
|
||||||
const first = await connect(tmpDir.name, { readConsistencyInterval: 0 });
|
|
||||||
const second = await connect(tmpDir.name, { readConsistencyInterval: 0 });
|
|
||||||
|
|
||||||
try {
|
|
||||||
const table = await first.createTable(
|
|
||||||
"docs",
|
|
||||||
[{ text: "hello before" }],
|
|
||||||
{ schema },
|
|
||||||
);
|
|
||||||
const pending = table.search("hello").toArray();
|
|
||||||
await started;
|
|
||||||
|
|
||||||
const replacement = await second.createTable(
|
|
||||||
"docs",
|
|
||||||
[{ text: "hello after" }],
|
|
||||||
{ mode: "overwrite" },
|
|
||||||
);
|
|
||||||
await replacement.createIndex("text", { config: Index.fts() });
|
|
||||||
releaseEmbedding();
|
|
||||||
|
|
||||||
expect((await pending)[0].text).toBe("hello before");
|
|
||||||
} finally {
|
|
||||||
releaseEmbedding();
|
|
||||||
first.close();
|
|
||||||
second.close();
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
it("refreshes a reused automatic search for every execution", async () => {
|
|
||||||
const first = await connect(tmpDir.name, { readConsistencyInterval: 0 });
|
|
||||||
const second = await connect(tmpDir.name, { readConsistencyInterval: 0 });
|
|
||||||
|
|
||||||
try {
|
|
||||||
const table = await first.createTable("docs", [
|
|
||||||
{ text: "hello before", marker: "before" },
|
|
||||||
]);
|
|
||||||
await table.createIndex("text", { config: Index.fts() });
|
|
||||||
const search = table.search("hello").select(["text"]);
|
|
||||||
|
|
||||||
const before = (await search.toArray())[0];
|
|
||||||
expect(before.text).toBe("hello before");
|
|
||||||
expect(before.marker).toBeUndefined();
|
|
||||||
|
|
||||||
const replacement = await second.createTable(
|
|
||||||
"docs",
|
|
||||||
[{ text: "hello after", marker: "after" }],
|
|
||||||
{ mode: "overwrite" },
|
|
||||||
);
|
|
||||||
await replacement.createIndex("text", { config: Index.fts() });
|
|
||||||
|
|
||||||
const after = (await search.toArray())[0];
|
|
||||||
expect(after.text).toBe("hello after");
|
|
||||||
expect(after.marker).toBeUndefined();
|
|
||||||
} finally {
|
|
||||||
first.close();
|
|
||||||
second.close();
|
|
||||||
}
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
describe("schema evolution", function () {
|
describe("schema evolution", function () {
|
||||||
let tmpDir: tmp.DirResult;
|
let tmpDir: tmp.DirResult;
|
||||||
beforeEach(() => {
|
beforeEach(() => {
|
||||||
@@ -2585,24 +2294,7 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
);
|
);
|
||||||
});
|
});
|
||||||
|
|
||||||
test("full text search if only an unrelated embedding function is registered", async () => {
|
test("full text search if no embedding function provided", async () => {
|
||||||
register("unused")(
|
|
||||||
class extends EmbeddingFunction<string> {
|
|
||||||
ndims() {
|
|
||||||
return 3;
|
|
||||||
}
|
|
||||||
embeddingDataType() {
|
|
||||||
return new Float32();
|
|
||||||
}
|
|
||||||
async computeQueryEmbeddings(_data: string) {
|
|
||||||
return [1, 2, 3];
|
|
||||||
}
|
|
||||||
async computeSourceEmbeddings(data: string[]) {
|
|
||||||
return data.map(() => [1, 2, 3]);
|
|
||||||
}
|
|
||||||
},
|
|
||||||
);
|
|
||||||
|
|
||||||
const db = await connect(tmpDir.name);
|
const db = await connect(tmpDir.name);
|
||||||
const data = [
|
const data = [
|
||||||
{ text: "hello world", vector: [0.1, 0.2, 0.3] },
|
{ text: "hello world", vector: [0.1, 0.2, 0.3] },
|
||||||
@@ -2624,306 +2316,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
expect(results2[0].text).toBe(data[1].text);
|
expect(results2[0].text).toBe(data[1].text);
|
||||||
});
|
});
|
||||||
|
|
||||||
test("auto search stays consistent with the active revision", async () => {
|
|
||||||
let initCalls = 0;
|
|
||||||
let queryCalls = 0;
|
|
||||||
let markStarted!: () => void;
|
|
||||||
const started = new Promise<void>((resolve) => {
|
|
||||||
markStarted = resolve;
|
|
||||||
});
|
|
||||||
let releaseEmbedding!: () => void;
|
|
||||||
const embeddingReleased = new Promise<void>((resolve) => {
|
|
||||||
releaseEmbedding = resolve;
|
|
||||||
});
|
|
||||||
|
|
||||||
@register("refresh-test")
|
|
||||||
class TestEmbedding extends EmbeddingFunction<string> {
|
|
||||||
async init() {
|
|
||||||
initCalls += 1;
|
|
||||||
}
|
|
||||||
ndims() {
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
embeddingDataType() {
|
|
||||||
return new arrow.Float32();
|
|
||||||
}
|
|
||||||
async computeQueryEmbeddings(value: string) {
|
|
||||||
queryCalls += 1;
|
|
||||||
if (value === "blocked") {
|
|
||||||
markStarted();
|
|
||||||
await embeddingReleased;
|
|
||||||
}
|
|
||||||
return value === "greetings" ? [0.1] : [0.2];
|
|
||||||
}
|
|
||||||
async computeSourceEmbeddings(values: string[]) {
|
|
||||||
return values.map((value) =>
|
|
||||||
value === "hello world" ? [0.1] : [0.2],
|
|
||||||
);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
const writer = await connect(tmpDir.name);
|
|
||||||
await writer.createTable("test", [{ text: "plain", vector: [0.0] }]);
|
|
||||||
const reader = await connect(tmpDir.name, {
|
|
||||||
readConsistencyInterval: 0,
|
|
||||||
});
|
|
||||||
const tracked = await reader.openTable("test");
|
|
||||||
type SnapshotCountingNative = {
|
|
||||||
querySnapshot: () => Promise<unknown>;
|
|
||||||
};
|
|
||||||
const native = (tracked as unknown as { inner: SnapshotCountingNative })
|
|
||||||
.inner;
|
|
||||||
const querySnapshot = native.querySnapshot.bind(native);
|
|
||||||
let snapshotCalls = 0;
|
|
||||||
native.querySnapshot = async () => {
|
|
||||||
snapshotCalls += 1;
|
|
||||||
return await querySnapshot();
|
|
||||||
};
|
|
||||||
const autoQuery = tracked.search("greetings").select(["text"]).limit(1);
|
|
||||||
|
|
||||||
const func = new TestEmbedding();
|
|
||||||
const schema = LanceSchema({
|
|
||||||
text: func.sourceField(new arrow.Utf8()),
|
|
||||||
vector: func.vectorField(),
|
|
||||||
});
|
|
||||||
const data = [{ text: "hello world" }, { text: "goodbye world" }];
|
|
||||||
await writer.createTable("test", data, { mode: "overwrite", schema });
|
|
||||||
const baselineInitCalls = initCalls;
|
|
||||||
|
|
||||||
expect(
|
|
||||||
(await tracked.schema()).metadata.get("embedding_functions"),
|
|
||||||
).toBeDefined();
|
|
||||||
const results = await autoQuery.toArray();
|
|
||||||
expect(results[0].text).toBe(data[0].text);
|
|
||||||
expect(initCalls).toBe(baselineInitCalls + 1);
|
|
||||||
expect(queryCalls).toBe(1);
|
|
||||||
expect(snapshotCalls).toBe(1);
|
|
||||||
|
|
||||||
const repeatedResults = await autoQuery.toArray();
|
|
||||||
expect(repeatedResults[0].text).toBe(data[0].text);
|
|
||||||
expect(initCalls).toBe(baselineInitCalls + 1);
|
|
||||||
expect(queryCalls).toBe(1);
|
|
||||||
expect(snapshotCalls).toBe(2);
|
|
||||||
|
|
||||||
const pending = tracked
|
|
||||||
.search("blocked")
|
|
||||||
.select(["text"])
|
|
||||||
.limit(1)
|
|
||||||
.toArray();
|
|
||||||
await started;
|
|
||||||
|
|
||||||
const ftsData = [
|
|
||||||
{ text: "greetings from full text", vector: [0.0] },
|
|
||||||
{ text: "blocked from full text", vector: [0.0] },
|
|
||||||
];
|
|
||||||
const ftsTable = await writer.createTable("test", ftsData, {
|
|
||||||
mode: "overwrite",
|
|
||||||
});
|
|
||||||
await ftsTable.createIndex("text", { config: Index.fts() });
|
|
||||||
releaseEmbedding();
|
|
||||||
|
|
||||||
const pendingResults = await pending;
|
|
||||||
expect(pendingResults[0].text).toBe(data[1].text);
|
|
||||||
|
|
||||||
expect(
|
|
||||||
(await tracked.schema()).metadata.get("embedding_functions"),
|
|
||||||
).toBeUndefined();
|
|
||||||
const ftsResults = await autoQuery.toArray();
|
|
||||||
expect(ftsResults[0].text).toBe(ftsData[0].text);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("auto search keeps newer preparation during a revision race", async () => {
|
|
||||||
let aCalls = 0;
|
|
||||||
let bCalls = 0;
|
|
||||||
let markAStarted!: () => void;
|
|
||||||
const aStarted = new Promise<void>((resolve) => {
|
|
||||||
markAStarted = resolve;
|
|
||||||
});
|
|
||||||
let releaseA!: () => void;
|
|
||||||
const aReleased = new Promise<void>((resolve) => {
|
|
||||||
releaseA = resolve;
|
|
||||||
});
|
|
||||||
let markBStarted!: () => void;
|
|
||||||
const bStarted = new Promise<void>((resolve) => {
|
|
||||||
markBStarted = resolve;
|
|
||||||
});
|
|
||||||
let releaseB!: () => void;
|
|
||||||
const bReleased = new Promise<void>((resolve) => {
|
|
||||||
releaseB = resolve;
|
|
||||||
});
|
|
||||||
|
|
||||||
@register("race-a")
|
|
||||||
class EmbeddingA extends EmbeddingFunction<string> {
|
|
||||||
ndims() {
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
embeddingDataType() {
|
|
||||||
return new arrow.Float32();
|
|
||||||
}
|
|
||||||
async computeQueryEmbeddings() {
|
|
||||||
aCalls += 1;
|
|
||||||
markAStarted();
|
|
||||||
await aReleased;
|
|
||||||
return [0.1];
|
|
||||||
}
|
|
||||||
async computeSourceEmbeddings(values: string[]) {
|
|
||||||
return values.map(() => [0.1]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
@register("race-b")
|
|
||||||
class EmbeddingB extends EmbeddingFunction<string> {
|
|
||||||
ndims() {
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
embeddingDataType() {
|
|
||||||
return new arrow.Float32();
|
|
||||||
}
|
|
||||||
async computeQueryEmbeddings() {
|
|
||||||
bCalls += 1;
|
|
||||||
markBStarted();
|
|
||||||
await bReleased;
|
|
||||||
return [0.2];
|
|
||||||
}
|
|
||||||
async computeSourceEmbeddings(values: string[]) {
|
|
||||||
return values.map(() => [0.2]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
const writer = await connect(tmpDir.name);
|
|
||||||
const embeddingA = new EmbeddingA();
|
|
||||||
const schemaA = LanceSchema({
|
|
||||||
text: embeddingA.sourceField(new arrow.Utf8()),
|
|
||||||
vector: embeddingA.vectorField(),
|
|
||||||
});
|
|
||||||
await writer.createTable("race", [{ text: "revision a" }], {
|
|
||||||
schema: schemaA,
|
|
||||||
});
|
|
||||||
const reader = await connect(tmpDir.name, {
|
|
||||||
readConsistencyInterval: 0,
|
|
||||||
});
|
|
||||||
const tracked = await reader.openTable("race");
|
|
||||||
const query = tracked.search("query");
|
|
||||||
|
|
||||||
const first = query.toArray();
|
|
||||||
await aStarted;
|
|
||||||
|
|
||||||
const embeddingB = new EmbeddingB();
|
|
||||||
const schemaB = LanceSchema({
|
|
||||||
text: embeddingB.sourceField(new arrow.Utf8()),
|
|
||||||
vector: embeddingB.vectorField(),
|
|
||||||
});
|
|
||||||
await writer.createTable("race", [{ text: "revision b" }], {
|
|
||||||
mode: "overwrite",
|
|
||||||
schema: schemaB,
|
|
||||||
});
|
|
||||||
const second = query.toArray();
|
|
||||||
await bStarted;
|
|
||||||
|
|
||||||
releaseA();
|
|
||||||
releaseB();
|
|
||||||
await Promise.all([first, second]);
|
|
||||||
expect(aCalls).toBe(1);
|
|
||||||
expect(bCalls).toBe(1);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("stale FTS routing keeps newer vector preparation", async () => {
|
|
||||||
let vectorCalls = 0;
|
|
||||||
let markVectorStarted!: () => void;
|
|
||||||
const vectorStarted = new Promise<void>((resolve) => {
|
|
||||||
markVectorStarted = resolve;
|
|
||||||
});
|
|
||||||
let releaseVector!: () => void;
|
|
||||||
const vectorReleased = new Promise<void>((resolve) => {
|
|
||||||
releaseVector = resolve;
|
|
||||||
});
|
|
||||||
|
|
||||||
@register("stale-fts-race")
|
|
||||||
class RaceEmbedding extends EmbeddingFunction<string> {
|
|
||||||
ndims() {
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
embeddingDataType() {
|
|
||||||
return new arrow.Float32();
|
|
||||||
}
|
|
||||||
async computeQueryEmbeddings() {
|
|
||||||
vectorCalls += 1;
|
|
||||||
markVectorStarted();
|
|
||||||
await vectorReleased;
|
|
||||||
return [0.1];
|
|
||||||
}
|
|
||||||
async computeSourceEmbeddings(values: string[]) {
|
|
||||||
return values.map(() => [0.1]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
const writer = await connect(tmpDir.name);
|
|
||||||
const ftsTable = await writer.createTable("stale_fts", [
|
|
||||||
{ text: "hello", vector: [0.0] },
|
|
||||||
]);
|
|
||||||
await ftsTable.createIndex("text", { config: Index.fts() });
|
|
||||||
|
|
||||||
const reader = await connect(tmpDir.name, {
|
|
||||||
readConsistencyInterval: 0,
|
|
||||||
});
|
|
||||||
const tracked = await reader.openTable("stale_fts");
|
|
||||||
type Snapshot = {
|
|
||||||
schema: () => Promise<Buffer>;
|
|
||||||
};
|
|
||||||
type NativeWithSnapshot = {
|
|
||||||
querySnapshot: () => Promise<Snapshot>;
|
|
||||||
};
|
|
||||||
const native = (tracked as unknown as { inner: NativeWithSnapshot })
|
|
||||||
.inner;
|
|
||||||
const querySnapshot = native.querySnapshot.bind(native);
|
|
||||||
let snapshotCalls = 0;
|
|
||||||
let markStaleSchemaStarted!: () => void;
|
|
||||||
const staleSchemaStarted = new Promise<void>((resolve) => {
|
|
||||||
markStaleSchemaStarted = resolve;
|
|
||||||
});
|
|
||||||
let releaseStaleSchema!: () => void;
|
|
||||||
const staleSchemaReleased = new Promise<void>((resolve) => {
|
|
||||||
releaseStaleSchema = resolve;
|
|
||||||
});
|
|
||||||
native.querySnapshot = async () => {
|
|
||||||
const snapshot = await querySnapshot();
|
|
||||||
snapshotCalls += 1;
|
|
||||||
if (snapshotCalls === 1) {
|
|
||||||
const schema = snapshot.schema.bind(snapshot);
|
|
||||||
snapshot.schema = async () => {
|
|
||||||
markStaleSchemaStarted();
|
|
||||||
await staleSchemaReleased;
|
|
||||||
return await schema();
|
|
||||||
};
|
|
||||||
}
|
|
||||||
return snapshot;
|
|
||||||
};
|
|
||||||
|
|
||||||
const query = tracked.search("hello");
|
|
||||||
const staleFtsExecution = query.toArray();
|
|
||||||
await staleSchemaStarted;
|
|
||||||
|
|
||||||
const embedding = new RaceEmbedding();
|
|
||||||
const vectorSchema = LanceSchema({
|
|
||||||
text: embedding.sourceField(new arrow.Utf8()),
|
|
||||||
vector: embedding.vectorField(),
|
|
||||||
});
|
|
||||||
await writer.createTable("stale_fts", [{ text: "hello" }], {
|
|
||||||
mode: "overwrite",
|
|
||||||
schema: vectorSchema,
|
|
||||||
});
|
|
||||||
|
|
||||||
const vectorExecution = query.toArray();
|
|
||||||
await vectorStarted;
|
|
||||||
releaseStaleSchema();
|
|
||||||
await staleFtsExecution;
|
|
||||||
releaseVector();
|
|
||||||
await vectorExecution;
|
|
||||||
|
|
||||||
await query.toArray();
|
|
||||||
expect(vectorCalls).toBe(1);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("tokenizes FTS queries by column or index name", async () => {
|
test("tokenizes FTS queries by column or index name", async () => {
|
||||||
const db = await connect(tmpDir.name);
|
const db = await connect(tmpDir.name);
|
||||||
const data = [
|
const data = [
|
||||||
@@ -3474,30 +2866,6 @@ describe("column name options", () => {
|
|||||||
expect(results[1].query_index).toBe(1);
|
expect(results[1].query_index).toBe(1);
|
||||||
});
|
});
|
||||||
|
|
||||||
test("observes promised additional vectors while the query is pending", async () => {
|
|
||||||
const initialVector = new Promise<number[]>(() => undefined);
|
|
||||||
const query = table.query().nearestTo(initialVector);
|
|
||||||
const unhandled: unknown[] = [];
|
|
||||||
const onUnhandled = (reason: unknown) => unhandled.push(reason);
|
|
||||||
process.on("unhandledRejection", onUnhandled);
|
|
||||||
|
|
||||||
try {
|
|
||||||
query.addQueryVector(Promise.reject(new Error("extra vector failed")));
|
|
||||||
await new Promise<void>((resolve) => setImmediate(resolve));
|
|
||||||
expect(unhandled).toEqual([]);
|
|
||||||
|
|
||||||
const rejectedQuery = table
|
|
||||||
.query()
|
|
||||||
.nearestTo([0.1, 0.2])
|
|
||||||
.addQueryVector(Promise.reject(new Error("consumed vector failed")));
|
|
||||||
await expect(rejectedQuery.toArray()).rejects.toThrow(
|
|
||||||
"consumed vector failed",
|
|
||||||
);
|
|
||||||
} finally {
|
|
||||||
process.off("unhandledRejection", onUnhandled);
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
test("index and search multivectors", async () => {
|
test("index and search multivectors", async () => {
|
||||||
const db = await connect(tmpDir.name);
|
const db = await connect(tmpDir.name);
|
||||||
const data = [];
|
const data = [];
|
||||||
@@ -3535,7 +2903,7 @@ describe("column name options", () => {
|
|||||||
.limit(10)
|
.limit(10)
|
||||||
.toArray();
|
.toArray();
|
||||||
expect(results2.length).toBe(10);
|
expect(results2.length).toBe(10);
|
||||||
}, 30_000);
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
describe("when creating an empty table", () => {
|
describe("when creating an empty table", () => {
|
||||||
@@ -3561,27 +2929,6 @@ describe("when creating an empty table", () => {
|
|||||||
expect((actualSchema.fields[1].type as Float64).precision).toBe(2);
|
expect((actualSchema.fields[1].type as Float64).precision).toBe(2);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("can add and query JSON data", async () => {
|
|
||||||
const schema = new Schema([
|
|
||||||
new Field("id", new Int32(), true),
|
|
||||||
new Field(
|
|
||||||
"meta",
|
|
||||||
new Utf8(),
|
|
||||||
true,
|
|
||||||
new Map([["ARROW:extension:name", "arrow.json"]]),
|
|
||||||
),
|
|
||||||
]);
|
|
||||||
const table = await con.createEmptyTable("json", schema);
|
|
||||||
const meta = JSON.stringify({ x: 1 });
|
|
||||||
|
|
||||||
await table.add([{ id: 1, meta }]);
|
|
||||||
|
|
||||||
const rows = await table.query().toArray();
|
|
||||||
expect(rows).toHaveLength(1);
|
|
||||||
expect(rows[0].id).toBe(1);
|
|
||||||
expect(rows[0].meta).toBe(meta);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("can create an empty table from schema that specifies field types by name", async () => {
|
it("can create an empty table from schema that specifies field types by name", async () => {
|
||||||
const schemaLike = {
|
const schemaLike = {
|
||||||
fields: [
|
fields: [
|
||||||
@@ -3943,120 +3290,3 @@ describe("LSM merge insert", () => {
|
|||||||
await expect(table.query().useLsm(true).toArray()).rejects.toThrow();
|
await expect(table.query().useLsm(true).toArray()).rejects.toThrow();
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
describe("LSM convergence and stats", () => {
|
|
||||||
let tmpDir: tmp.DirResult;
|
|
||||||
|
|
||||||
beforeEach(() => {
|
|
||||||
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
|
||||||
});
|
|
||||||
afterEach(() => tmpDir.removeCallback());
|
|
||||||
|
|
||||||
async function lsmTable(conn: Connection): Promise<Table> {
|
|
||||||
const table = await conn.createEmptyTable(
|
|
||||||
"t",
|
|
||||||
new arrow.Schema([new arrow.Field("id", new arrow.Utf8(), false)]),
|
|
||||||
);
|
|
||||||
await table.setUnenforcedPrimaryKey("id");
|
|
||||||
await table.setLsmWriteSpec({ specType: "unsharded" });
|
|
||||||
return table;
|
|
||||||
}
|
|
||||||
|
|
||||||
// These four route through the server that owns the MemWAL, so a local table
|
|
||||||
// rejects them rather than answering. What is asserted here is that the
|
|
||||||
// bindings reach the core at all; the behavior against a real endpoint is
|
|
||||||
// covered by the mocked endpoint tests in rust/lancedb/src/remote/table.rs.
|
|
||||||
it("rejects flushLsm on a local table", async () => {
|
|
||||||
const conn = await connect(tmpDir.name);
|
|
||||||
const table = await lsmTable(conn);
|
|
||||||
|
|
||||||
await expect(table.flushLsm()).rejects.toThrow(/not supported/i);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("rejects compactLsm on a local table", async () => {
|
|
||||||
const conn = await connect(tmpDir.name);
|
|
||||||
const table = await lsmTable(conn);
|
|
||||||
|
|
||||||
await expect(table.compactLsm()).rejects.toThrow(/not supported/i);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("rejects getLsmStats on a local table", async () => {
|
|
||||||
const conn = await connect(tmpDir.name);
|
|
||||||
const table = await lsmTable(conn);
|
|
||||||
|
|
||||||
await expect(table.getLsmStats()).rejects.toThrow(/not supported/i);
|
|
||||||
await expect(table.getLsmStats(true)).rejects.toThrow(/not supported/i);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("rejects checkpointLsm on a local table", async () => {
|
|
||||||
const conn = await connect(tmpDir.name);
|
|
||||||
const table = await lsmTable(conn);
|
|
||||||
|
|
||||||
// checkpointLsm seals first, so it surfaces flushLsm's rejection.
|
|
||||||
await expect(table.checkpointLsm()).rejects.toThrow(/not supported/i);
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
describe("computed columns", () => {
|
|
||||||
let tmpDir: tmp.DirResult;
|
|
||||||
beforeEach(() => {
|
|
||||||
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
|
||||||
});
|
|
||||||
afterEach(() => tmpDir.removeCallback());
|
|
||||||
|
|
||||||
it("declares a column and fills it on refresh", async () => {
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
const table = await db.createTable("computed", [{ x: 1 }, { x: 2 }]);
|
|
||||||
|
|
||||||
await table.addColumns({
|
|
||||||
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
|
||||||
});
|
|
||||||
let rows = await table.query().toArray();
|
|
||||||
expect(rows.map((r) => r.doubled)).toEqual([null, null]);
|
|
||||||
|
|
||||||
const result = await table.refreshColumn("doubled");
|
|
||||||
expect(result.rowsFilled).toBe(2);
|
|
||||||
|
|
||||||
rows = await table.query().toArray();
|
|
||||||
expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("returns a job handle from refreshColumnAsync", async () => {
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
const table = await db.createTable("computed_job", [{ x: 1 }, { x: 2 }]);
|
|
||||||
|
|
||||||
await table.addColumns({
|
|
||||||
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
|
||||||
});
|
|
||||||
|
|
||||||
const job = await table.refreshColumnAsync("doubled");
|
|
||||||
expect(job.id).toBeNull();
|
|
||||||
await job.wait();
|
|
||||||
expect(await job.status()).toBe("finished");
|
|
||||||
|
|
||||||
const rows = await table.query().toArray();
|
|
||||||
expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]);
|
|
||||||
|
|
||||||
// Bad input rejects at the call, not through the job.
|
|
||||||
await expect(table.refreshColumnAsync("x")).rejects.toThrow(
|
|
||||||
"not a computed column",
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("fills rows added since the last refresh", async () => {
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
const table = await db.createTable("computed_append", [{ x: 1 }]);
|
|
||||||
|
|
||||||
await table.addColumns({
|
|
||||||
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
|
||||||
});
|
|
||||||
await table.refreshColumn("doubled");
|
|
||||||
await table.add([{ x: 5 }]);
|
|
||||||
|
|
||||||
const result = await table.refreshColumn("doubled");
|
|
||||||
expect(result.rowsFilled).toBe(1);
|
|
||||||
|
|
||||||
const rows = await table.query().toArray();
|
|
||||||
expect(rows.map((r) => r.doubled).sort()).toEqual([10, 2]);
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|||||||
@@ -170,7 +170,7 @@ test("basic table examples", async () => {
|
|||||||
// --8<-- [end:create_index]
|
// --8<-- [end:create_index]
|
||||||
|
|
||||||
// --8<-- [start:delete_rows]
|
// --8<-- [start:delete_rows]
|
||||||
await tbl.delete("item = 'fizz'");
|
await tbl.delete('item = "fizz"');
|
||||||
// --8<-- [end:delete_rows]
|
// --8<-- [end:delete_rows]
|
||||||
|
|
||||||
// --8<-- [start:drop_table]
|
// --8<-- [start:drop_table]
|
||||||
|
|||||||
+312
-41
@@ -5,6 +5,7 @@ import {
|
|||||||
Data as ArrowData,
|
Data as ArrowData,
|
||||||
Table as ArrowTable,
|
Table as ArrowTable,
|
||||||
Binary,
|
Binary,
|
||||||
|
Bool,
|
||||||
BufferType,
|
BufferType,
|
||||||
DataType,
|
DataType,
|
||||||
DateUnit,
|
DateUnit,
|
||||||
@@ -17,7 +18,12 @@ import {
|
|||||||
FixedSizeList,
|
FixedSizeList,
|
||||||
Float,
|
Float,
|
||||||
Float32,
|
Float32,
|
||||||
|
Float64,
|
||||||
Int,
|
Int,
|
||||||
|
Int8,
|
||||||
|
Int16,
|
||||||
|
Int32,
|
||||||
|
Int64,
|
||||||
LargeBinary,
|
LargeBinary,
|
||||||
List,
|
List,
|
||||||
Null,
|
Null,
|
||||||
@@ -30,29 +36,33 @@ import {
|
|||||||
Struct,
|
Struct,
|
||||||
Timestamp,
|
Timestamp,
|
||||||
Type,
|
Type,
|
||||||
|
Uint8,
|
||||||
|
Uint16,
|
||||||
|
Uint32,
|
||||||
Utf8,
|
Utf8,
|
||||||
Vector,
|
Vector,
|
||||||
makeVector as arrowMakeVector,
|
makeVector as arrowMakeVector,
|
||||||
util as arrowUtil,
|
|
||||||
vectorFromArray as badVectorFromArray,
|
vectorFromArray as badVectorFromArray,
|
||||||
makeBuilder,
|
makeBuilder,
|
||||||
makeData,
|
makeData,
|
||||||
} from "apache-arrow";
|
} from "apache-arrow";
|
||||||
import { Buffers } from "apache-arrow/data";
|
import { Buffers } from "apache-arrow/data";
|
||||||
import { typedArrayToArrowType } from "./arrow_type";
|
|
||||||
import { type EmbeddingFunction } from "./embedding/embedding_function";
|
import { type EmbeddingFunction } from "./embedding/embedding_function";
|
||||||
import {
|
import { EmbeddingFunctionConfig, getRegistry } from "./embedding/registry";
|
||||||
EmbeddingFunctionConfig,
|
|
||||||
getRegistry,
|
|
||||||
parseEmbeddingMetadata,
|
|
||||||
} from "./embedding/registry";
|
|
||||||
import {
|
import {
|
||||||
sanitizeField,
|
sanitizeField,
|
||||||
sanitizeSchema,
|
sanitizeSchema,
|
||||||
sanitizeTable,
|
sanitizeTable,
|
||||||
sanitizeType,
|
sanitizeType,
|
||||||
} from "./sanitize";
|
} from "./sanitize";
|
||||||
import { inferSchema } from "./schema";
|
|
||||||
|
/**
|
||||||
|
* Check if a field name indicates a vector column.
|
||||||
|
*/
|
||||||
|
function nameSuggestsVectorColumn(fieldName: string): boolean {
|
||||||
|
const nameLower = fieldName.toLowerCase();
|
||||||
|
return nameLower.includes("vector") || nameLower.includes("embedding");
|
||||||
|
}
|
||||||
|
|
||||||
export * from "apache-arrow";
|
export * from "apache-arrow";
|
||||||
export type SchemaLike =
|
export type SchemaLike =
|
||||||
@@ -445,6 +455,110 @@ export function makeArrowTable(
|
|||||||
return new ArrowTable(inferredSchema, finalColumns);
|
return new ArrowTable(inferredSchema, finalColumns);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function inferSchema(
|
||||||
|
data: Array<Record<string, unknown>>,
|
||||||
|
schema: Schema | undefined,
|
||||||
|
opts: MakeArrowTableOptions,
|
||||||
|
): Schema {
|
||||||
|
// We will collect all fields we see in the data.
|
||||||
|
const pathTree = new PathTree<DataType>();
|
||||||
|
|
||||||
|
for (const [rowI, row] of data.entries()) {
|
||||||
|
for (const [path, value] of rowPathsAndValues(row)) {
|
||||||
|
if (!pathTree.has(path)) {
|
||||||
|
// First time seeing this field.
|
||||||
|
if (schema !== undefined) {
|
||||||
|
const field = getFieldForPath(schema, path);
|
||||||
|
if (field === undefined) {
|
||||||
|
throw new Error(
|
||||||
|
`Found field not in schema: ${path.join(".")} at row ${rowI}`,
|
||||||
|
);
|
||||||
|
} else {
|
||||||
|
pathTree.set(path, field.type);
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
const inferredType = inferType(value, path, opts);
|
||||||
|
if (inferredType === undefined) {
|
||||||
|
throw new Error(`Failed to infer data type for field ${path.join(
|
||||||
|
".",
|
||||||
|
)} at row ${rowI}. \
|
||||||
|
Consider providing an explicit schema.`);
|
||||||
|
}
|
||||||
|
pathTree.set(path, inferredType);
|
||||||
|
}
|
||||||
|
} else if (schema === undefined) {
|
||||||
|
const currentType = pathTree.get(path);
|
||||||
|
const newType = inferType(value, path, opts);
|
||||||
|
if (currentType !== newType) {
|
||||||
|
new Error(`Failed to infer schema for data. Previously inferred type \
|
||||||
|
${currentType} but found ${newType} at row ${rowI}. Consider \
|
||||||
|
providing an explicit schema.`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (schema === undefined) {
|
||||||
|
function fieldsFromPathTree(pathTree: PathTree<DataType>): Field[] {
|
||||||
|
const fields = [];
|
||||||
|
for (const [name, value] of pathTree.map.entries()) {
|
||||||
|
if (value instanceof PathTree) {
|
||||||
|
const children = fieldsFromPathTree(value);
|
||||||
|
fields.push(new Field(name, new Struct(children), true));
|
||||||
|
} else {
|
||||||
|
fields.push(new Field(name, value, true));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return fields;
|
||||||
|
}
|
||||||
|
const fields = fieldsFromPathTree(pathTree);
|
||||||
|
return new Schema(fields);
|
||||||
|
} else {
|
||||||
|
function takeMatchingFields(
|
||||||
|
fields: Field[],
|
||||||
|
pathTree: PathTree<DataType>,
|
||||||
|
): Field[] {
|
||||||
|
const outFields = [];
|
||||||
|
for (const field of fields) {
|
||||||
|
if (pathTree.map.has(field.name)) {
|
||||||
|
const value = pathTree.get([field.name]);
|
||||||
|
if (value instanceof PathTree) {
|
||||||
|
const struct = field.type as Struct;
|
||||||
|
const children = takeMatchingFields(struct.children, value);
|
||||||
|
outFields.push(
|
||||||
|
new Field(field.name, new Struct(children), field.nullable),
|
||||||
|
);
|
||||||
|
} else {
|
||||||
|
outFields.push(
|
||||||
|
new Field(field.name, value as DataType, field.nullable),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return outFields;
|
||||||
|
}
|
||||||
|
const fields = takeMatchingFields(schema.fields, pathTree);
|
||||||
|
return new Schema(fields);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function* rowPathsAndValues(
|
||||||
|
row: Record<string, unknown>,
|
||||||
|
basePath: string[] = [],
|
||||||
|
): Generator<[string[], unknown]> {
|
||||||
|
for (const [key, value] of Object.entries(row)) {
|
||||||
|
if (isObject(value)) {
|
||||||
|
yield* rowPathsAndValues(value, [...basePath, key]);
|
||||||
|
} else {
|
||||||
|
// Skip undefined values - they should be treated the same as missing fields
|
||||||
|
// for embedding function purposes
|
||||||
|
if (value !== undefined) {
|
||||||
|
yield [[...basePath, key], value];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
function isObject(value: unknown): value is Record<string, unknown> {
|
function isObject(value: unknown): value is Record<string, unknown> {
|
||||||
return (
|
return (
|
||||||
typeof value === "object" &&
|
typeof value === "object" &&
|
||||||
@@ -459,19 +573,146 @@ function isObject(value: unknown): value is Record<string, unknown> {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
function valueAtPath(datum: Record<string, unknown>, path: string[]): unknown {
|
function getFieldForPath(schema: Schema, path: string[]): Field | undefined {
|
||||||
let current: unknown = datum;
|
let current: Field | Schema = schema;
|
||||||
for (const key of path) {
|
for (const key of path) {
|
||||||
if (current == null) {
|
if (current instanceof Schema) {
|
||||||
return null;
|
const field: Field | undefined = current.fields.find(
|
||||||
}
|
(f) => f.name === key,
|
||||||
if (isObject(current) && (Object.hasOwn(current, key) || key in current)) {
|
);
|
||||||
current = current[key];
|
if (field === undefined) {
|
||||||
|
return undefined;
|
||||||
|
}
|
||||||
|
current = field;
|
||||||
|
} else if (current instanceof Field && DataType.isStruct(current.type)) {
|
||||||
|
const struct: Struct = current.type;
|
||||||
|
const field = struct.children.find((f) => f.name === key);
|
||||||
|
if (field === undefined) {
|
||||||
|
return undefined;
|
||||||
|
}
|
||||||
|
current = field;
|
||||||
} else {
|
} else {
|
||||||
return undefined;
|
return undefined;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return current;
|
if (current instanceof Field) {
|
||||||
|
return current;
|
||||||
|
} else {
|
||||||
|
return undefined;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Try to infer which Arrow type to use for a given value.
|
||||||
|
*
|
||||||
|
* May return undefined if the type cannot be inferred.
|
||||||
|
*/
|
||||||
|
function inferType(
|
||||||
|
value: unknown,
|
||||||
|
path: string[],
|
||||||
|
opts: MakeArrowTableOptions,
|
||||||
|
): DataType | undefined {
|
||||||
|
if (typeof value === "bigint") {
|
||||||
|
return new Int64();
|
||||||
|
} else if (typeof value === "number") {
|
||||||
|
// Even if it's an integer, it's safer to assume Float64. Users can
|
||||||
|
// always provide an explicit schema or use BigInt if they mean integer.
|
||||||
|
return new Float64();
|
||||||
|
} else if (typeof value === "string") {
|
||||||
|
if (opts.dictionaryEncodeStrings) {
|
||||||
|
return new Dictionary(new Utf8(), new Int32());
|
||||||
|
} else {
|
||||||
|
return new Utf8();
|
||||||
|
}
|
||||||
|
} else if (typeof value === "boolean") {
|
||||||
|
return new Bool();
|
||||||
|
} else if (value instanceof Buffer) {
|
||||||
|
return new Binary();
|
||||||
|
} else if (ArrayBuffer.isView(value) && !(value instanceof DataView)) {
|
||||||
|
const info = typedArrayToArrowType(value);
|
||||||
|
if (info !== undefined) {
|
||||||
|
const child = new Field("item", info.elementType, true);
|
||||||
|
return new FixedSizeList(info.length, child);
|
||||||
|
}
|
||||||
|
return undefined;
|
||||||
|
} else if (Array.isArray(value)) {
|
||||||
|
if (value.length === 0) {
|
||||||
|
return undefined; // Without any values we can't infer the type
|
||||||
|
}
|
||||||
|
if (path.length === 1 && Object.hasOwn(opts.vectorColumns, path[0])) {
|
||||||
|
const floatType = sanitizeType(opts.vectorColumns[path[0]].type);
|
||||||
|
return new FixedSizeList(
|
||||||
|
value.length,
|
||||||
|
new Field("item", floatType, true),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
const valueType = inferType(value[0], path, opts);
|
||||||
|
if (valueType === undefined) {
|
||||||
|
return undefined;
|
||||||
|
}
|
||||||
|
// Try to automatically detect embedding columns.
|
||||||
|
if (nameSuggestsVectorColumn(path[path.length - 1])) {
|
||||||
|
// Check if value is a Uint8Array for integer vector type determination
|
||||||
|
if (value instanceof Uint8Array) {
|
||||||
|
// For integer vectors, we default to Uint8 (matching Python implementation)
|
||||||
|
const child = new Field("item", new Uint8(), true);
|
||||||
|
return new FixedSizeList(value.length, child);
|
||||||
|
} else {
|
||||||
|
// For float vectors, we default to Float32
|
||||||
|
const child = new Field("item", new Float32(), true);
|
||||||
|
return new FixedSizeList(value.length, child);
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
const child = new Field("item", valueType, true);
|
||||||
|
return new List(child);
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
// TODO: timestamp
|
||||||
|
return undefined;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
class PathTree<V> {
|
||||||
|
map: Map<string, V | PathTree<V>>;
|
||||||
|
|
||||||
|
constructor(entries?: [string[], V][]) {
|
||||||
|
this.map = new Map();
|
||||||
|
if (entries !== undefined) {
|
||||||
|
for (const [path, value] of entries) {
|
||||||
|
this.set(path, value);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
has(path: string[]): boolean {
|
||||||
|
let ref: PathTree<V> = this;
|
||||||
|
for (const part of path) {
|
||||||
|
if (!(ref instanceof PathTree) || !ref.map.has(part)) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
ref = ref.map.get(part) as PathTree<V>;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
get(path: string[]): V | undefined {
|
||||||
|
let ref: PathTree<V> = this;
|
||||||
|
for (const part of path) {
|
||||||
|
if (!(ref instanceof PathTree) || !ref.map.has(part)) {
|
||||||
|
return undefined;
|
||||||
|
}
|
||||||
|
ref = ref.map.get(part) as PathTree<V>;
|
||||||
|
}
|
||||||
|
return ref as V;
|
||||||
|
}
|
||||||
|
set(path: string[], value: V): void {
|
||||||
|
let ref: PathTree<V> = this;
|
||||||
|
for (const part of path.slice(0, path.length - 1)) {
|
||||||
|
if (!ref.map.has(part)) {
|
||||||
|
ref.map.set(part, new PathTree<V>());
|
||||||
|
}
|
||||||
|
ref = ref.map.get(part) as PathTree<V>;
|
||||||
|
}
|
||||||
|
ref.map.set(path[path.length - 1], value);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
function transposeData(
|
function transposeData(
|
||||||
@@ -479,26 +720,37 @@ function transposeData(
|
|||||||
field: Field,
|
field: Field,
|
||||||
path: string[] = [],
|
path: string[] = [],
|
||||||
): Vector {
|
): Vector {
|
||||||
const valuesPath = [...path, field.name];
|
|
||||||
const values = data.map((datum) => valueAtPath(datum, valuesPath));
|
|
||||||
if (field.type instanceof Struct) {
|
if (field.type instanceof Struct) {
|
||||||
const childFields = field.type.children;
|
const childFields = field.type.children;
|
||||||
|
const fullPath = [...path, field.name];
|
||||||
const childVectors = childFields.map((child) => {
|
const childVectors = childFields.map((child) => {
|
||||||
return transposeData(data, child, valuesPath);
|
return transposeData(data, child, fullPath);
|
||||||
});
|
});
|
||||||
const nullCount = values.filter((value) => value === null).length;
|
|
||||||
const structData = makeData({
|
const structData = makeData({
|
||||||
type: field.type,
|
type: field.type,
|
||||||
length: values.length,
|
|
||||||
nullCount,
|
|
||||||
nullBitmap:
|
|
||||||
nullCount > 0
|
|
||||||
? arrowUtil.packBools(values.map((value) => value !== null))
|
|
||||||
: undefined,
|
|
||||||
children: childVectors as unknown as ArrowData<DataType>[],
|
children: childVectors as unknown as ArrowData<DataType>[],
|
||||||
});
|
});
|
||||||
return arrowMakeVector(structData);
|
return arrowMakeVector(structData);
|
||||||
} else {
|
} else {
|
||||||
|
const valuesPath = [...path, field.name];
|
||||||
|
const values = data.map((datum) => {
|
||||||
|
let current: unknown = datum;
|
||||||
|
for (const key of valuesPath) {
|
||||||
|
if (current == null) {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (
|
||||||
|
isObject(current) &&
|
||||||
|
(Object.hasOwn(current, key) || key in current)
|
||||||
|
) {
|
||||||
|
current = current[key];
|
||||||
|
} else {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return current;
|
||||||
|
});
|
||||||
return makeVector(values, field.type, undefined, field.nullable);
|
return makeVector(values, field.type, undefined, field.nullable);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -541,6 +793,32 @@ function makeListVector(lists: unknown[][]): Vector<unknown> {
|
|||||||
return listBuilder.finish().toVector();
|
return listBuilder.finish().toVector();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Map a JS TypedArray instance to the corresponding Arrow element DataType
|
||||||
|
* and its length. Returns undefined if the value is not a recognized TypedArray.
|
||||||
|
*/
|
||||||
|
function typedArrayToArrowType(
|
||||||
|
value: ArrayBufferView,
|
||||||
|
): { elementType: DataType; length: number } | undefined {
|
||||||
|
if (value instanceof Float32Array)
|
||||||
|
return { elementType: new Float32(), length: value.length };
|
||||||
|
if (value instanceof Float64Array)
|
||||||
|
return { elementType: new Float64(), length: value.length };
|
||||||
|
if (value instanceof Uint8Array)
|
||||||
|
return { elementType: new Uint8(), length: value.length };
|
||||||
|
if (value instanceof Uint16Array)
|
||||||
|
return { elementType: new Uint16(), length: value.length };
|
||||||
|
if (value instanceof Uint32Array)
|
||||||
|
return { elementType: new Uint32(), length: value.length };
|
||||||
|
if (value instanceof Int8Array)
|
||||||
|
return { elementType: new Int8(), length: value.length };
|
||||||
|
if (value instanceof Int16Array)
|
||||||
|
return { elementType: new Int16(), length: value.length };
|
||||||
|
if (value instanceof Int32Array)
|
||||||
|
return { elementType: new Int32(), length: value.length };
|
||||||
|
return undefined;
|
||||||
|
}
|
||||||
|
|
||||||
/** Helper function to convert an Array of JS values to an Arrow Vector */
|
/** Helper function to convert an Array of JS values to an Arrow Vector */
|
||||||
function makeVector(
|
function makeVector(
|
||||||
values: unknown[],
|
values: unknown[],
|
||||||
@@ -655,7 +933,7 @@ async function applyEmbeddingsFromMetadata(
|
|||||||
|
|
||||||
for (const functionEntry of functions.values()) {
|
for (const functionEntry of functions.values()) {
|
||||||
const sourceColumn = columns[functionEntry.sourceColumn];
|
const sourceColumn = columns[functionEntry.sourceColumn];
|
||||||
const destColumn = functionEntry.vectorColumn;
|
const destColumn = functionEntry.vectorColumn ?? "vector";
|
||||||
if (sourceColumn === undefined) {
|
if (sourceColumn === undefined) {
|
||||||
throw new Error(
|
throw new Error(
|
||||||
`Cannot apply embedding function because the source column '${functionEntry.sourceColumn}' was not present in the data`,
|
`Cannot apply embedding function because the source column '${functionEntry.sourceColumn}' was not present in the data`,
|
||||||
@@ -1107,10 +1385,11 @@ function validateSchemaEmbeddings(
|
|||||||
|
|
||||||
// Check schema metadata for embedding functions
|
// Check schema metadata for embedding functions
|
||||||
if (schema.metadata.has("embedding_functions")) {
|
if (schema.metadata.has("embedding_functions")) {
|
||||||
const entries = parseEmbeddingMetadata(
|
const embeddings = JSON.parse(
|
||||||
schema.metadata.get("embedding_functions")!,
|
schema.metadata.get("embedding_functions")!,
|
||||||
);
|
);
|
||||||
if (entries.some((f) => f.vectorColumn === field.name)) {
|
// biome-ignore lint/suspicious/noExplicitAny: we don't know the type of `f`
|
||||||
|
if (embeddings.find((f: any) => f["vectorColumn"] === field.name)) {
|
||||||
hasEmbeddingFunction = true;
|
hasEmbeddingFunction = true;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1180,12 +1459,8 @@ export function ensureNestedFieldsExist(
|
|||||||
completeRow[field.name] = row[field.name];
|
completeRow[field.name] = row[field.name];
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
// Keep a missing struct valid while filling each of its children with
|
// Field is missing from the data - set to null
|
||||||
// null. This is distinct from an explicitly null struct value.
|
completeRow[field.name] = null;
|
||||||
completeRow[field.name] =
|
|
||||||
field.type.constructor.name === "Struct"
|
|
||||||
? ensureStructFieldsExist({}, field.type as Struct)
|
|
||||||
: null;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1220,12 +1495,8 @@ function ensureStructFieldsExist(
|
|||||||
completeStruct[childField.name] = data[childField.name];
|
completeStruct[childField.name] = data[childField.name];
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
// Keep a missing struct valid while filling each of its children with
|
// Field is missing - set to null
|
||||||
// null. This is distinct from an explicitly null struct value.
|
completeStruct[childField.name] = null;
|
||||||
completeStruct[childField.name] =
|
|
||||||
childField.type.constructor.name === "Struct"
|
|
||||||
? ensureStructFieldsExist({}, childField.type as Struct)
|
|
||||||
: null;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -1,40 +0,0 @@
|
|||||||
// SPDX-License-Identifier: Apache-2.0
|
|
||||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
|
||||||
|
|
||||||
import {
|
|
||||||
type DataType,
|
|
||||||
Float32,
|
|
||||||
Float64,
|
|
||||||
Int8,
|
|
||||||
Int16,
|
|
||||||
Int32,
|
|
||||||
Uint8,
|
|
||||||
Uint16,
|
|
||||||
Uint32,
|
|
||||||
} from "apache-arrow";
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Map a JS TypedArray instance to the corresponding Arrow element type and
|
|
||||||
* length. Returns undefined when the view is not a supported TypedArray.
|
|
||||||
*/
|
|
||||||
export function typedArrayToArrowType(
|
|
||||||
value: ArrayBufferView,
|
|
||||||
): { elementType: DataType; length: number } | undefined {
|
|
||||||
if (value instanceof Float32Array)
|
|
||||||
return { elementType: new Float32(), length: value.length };
|
|
||||||
if (value instanceof Float64Array)
|
|
||||||
return { elementType: new Float64(), length: value.length };
|
|
||||||
if (value instanceof Uint8Array)
|
|
||||||
return { elementType: new Uint8(), length: value.length };
|
|
||||||
if (value instanceof Uint16Array)
|
|
||||||
return { elementType: new Uint16(), length: value.length };
|
|
||||||
if (value instanceof Uint32Array)
|
|
||||||
return { elementType: new Uint32(), length: value.length };
|
|
||||||
if (value instanceof Int8Array)
|
|
||||||
return { elementType: new Int8(), length: value.length };
|
|
||||||
if (value instanceof Int16Array)
|
|
||||||
return { elementType: new Int16(), length: value.length };
|
|
||||||
if (value instanceof Int32Array)
|
|
||||||
return { elementType: new Int32(), length: value.length };
|
|
||||||
return undefined;
|
|
||||||
}
|
|
||||||
@@ -1,7 +1,6 @@
|
|||||||
// SPDX-License-Identifier: Apache-2.0
|
// SPDX-License-Identifier: Apache-2.0
|
||||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
import { tableFromIPC } from "apache-arrow";
|
|
||||||
import {
|
import {
|
||||||
Data,
|
Data,
|
||||||
SchemaLike,
|
SchemaLike,
|
||||||
@@ -16,29 +15,18 @@ import {
|
|||||||
makeEmptyTable,
|
makeEmptyTable,
|
||||||
} from "./arrow";
|
} from "./arrow";
|
||||||
import { EmbeddingFunctionConfig, getRegistry } from "./embedding/registry";
|
import { EmbeddingFunctionConfig, getRegistry } from "./embedding/registry";
|
||||||
import {
|
|
||||||
MaterializedView,
|
|
||||||
MaterializedViewSelect,
|
|
||||||
normalizeSelect,
|
|
||||||
validateNonNegativeInteger,
|
|
||||||
} from "./materialized_view";
|
|
||||||
import { Connection as LanceDbConnection } from "./native";
|
import { Connection as LanceDbConnection } from "./native";
|
||||||
import type {
|
import type {
|
||||||
CreateNamespaceResponse,
|
CreateNamespaceResponse,
|
||||||
DescribeNamespaceResponse,
|
DescribeNamespaceResponse,
|
||||||
DropNamespaceResponse,
|
DropNamespaceResponse,
|
||||||
Job,
|
|
||||||
JobDescription,
|
|
||||||
JobInfo,
|
|
||||||
ListNamespacesResponse,
|
ListNamespacesResponse,
|
||||||
ListTablesResponse,
|
|
||||||
} from "./native";
|
} from "./native";
|
||||||
export type {
|
export type {
|
||||||
CreateNamespaceResponse,
|
CreateNamespaceResponse,
|
||||||
DescribeNamespaceResponse,
|
DescribeNamespaceResponse,
|
||||||
DropNamespaceResponse,
|
DropNamespaceResponse,
|
||||||
ListNamespacesResponse,
|
ListNamespacesResponse,
|
||||||
ListTablesResponse,
|
|
||||||
};
|
};
|
||||||
import { sanitizeTable } from "./sanitize";
|
import { sanitizeTable } from "./sanitize";
|
||||||
import { LocalTable, Table } from "./table";
|
import { LocalTable, Table } from "./table";
|
||||||
@@ -136,10 +124,6 @@ export interface OpenTableOptions {
|
|||||||
indexCacheSize?: number;
|
indexCacheSize?: number;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* @deprecated Use {@link ListTablesOptions} with {@link Connection.listTables}
|
|
||||||
* instead.
|
|
||||||
*/
|
|
||||||
export interface TableNamesOptions {
|
export interface TableNamesOptions {
|
||||||
/**
|
/**
|
||||||
* If present, only return names that come lexicographically after the
|
* If present, only return names that come lexicographically after the
|
||||||
@@ -153,24 +137,6 @@ export interface TableNamesOptions {
|
|||||||
limit?: number;
|
limit?: number;
|
||||||
}
|
}
|
||||||
|
|
||||||
export interface ListTablesOptions {
|
|
||||||
/**
|
|
||||||
* Token from a previous response, to resume listing where it left off.
|
|
||||||
*
|
|
||||||
* The token is opaque: it carries whatever the database needs to resume, and
|
|
||||||
* callers should not construct or interpret one.
|
|
||||||
*/
|
|
||||||
pageToken?: string;
|
|
||||||
/**
|
|
||||||
* An upper bound on how many tables to return.
|
|
||||||
*
|
|
||||||
* A page may hold fewer than this and still not be the last one, so keep
|
|
||||||
* going while the response carries a page token rather than while pages are
|
|
||||||
* full.
|
|
||||||
*/
|
|
||||||
limit?: number;
|
|
||||||
}
|
|
||||||
|
|
||||||
export interface ListNamespacesOptions {
|
export interface ListNamespacesOptions {
|
||||||
/** Token from a previous response for pagination. */
|
/** Token from a previous response for pagination. */
|
||||||
pageToken?: string;
|
pageToken?: string;
|
||||||
@@ -255,7 +221,6 @@ export abstract class Connection {
|
|||||||
* @param {Partial<TableNamesOptions>} options - options to control the
|
* @param {Partial<TableNamesOptions>} options - options to control the
|
||||||
* paging / start point (backwards compatibility)
|
* paging / start point (backwards compatibility)
|
||||||
*
|
*
|
||||||
* @deprecated Use {@link Connection.listTables} instead.
|
|
||||||
*/
|
*/
|
||||||
abstract tableNames(options?: Partial<TableNamesOptions>): Promise<string[]>;
|
abstract tableNames(options?: Partial<TableNamesOptions>): Promise<string[]>;
|
||||||
/**
|
/**
|
||||||
@@ -266,94 +231,18 @@ export abstract class Connection {
|
|||||||
* @param {Partial<TableNamesOptions>} options - options to control the
|
* @param {Partial<TableNamesOptions>} options - options to control the
|
||||||
* paging / start point
|
* paging / start point
|
||||||
*
|
*
|
||||||
* @deprecated Use {@link Connection.listTables} instead.
|
|
||||||
*/
|
*/
|
||||||
abstract tableNames(
|
abstract tableNames(
|
||||||
namespacePath?: string[],
|
namespacePath?: string[],
|
||||||
options?: Partial<TableNamesOptions>,
|
options?: Partial<TableNamesOptions>,
|
||||||
): Promise<string[]>;
|
): Promise<string[]>;
|
||||||
|
|
||||||
/**
|
|
||||||
* List a page of the tables in this database.
|
|
||||||
*
|
|
||||||
* To retrieve the tables after the page, pass the `pageToken` the response
|
|
||||||
* carries back in. A page can be shorter than `limit` without being the last
|
|
||||||
* one, so walk until a response carries no page token:
|
|
||||||
*
|
|
||||||
* ```ts
|
|
||||||
* const names = [];
|
|
||||||
* let pageToken = undefined;
|
|
||||||
* do {
|
|
||||||
* const page = await conn.listTables({ pageToken, limit: 100 });
|
|
||||||
* names.push(...page.tables);
|
|
||||||
* pageToken = page.pageToken;
|
|
||||||
* } while (pageToken);
|
|
||||||
* ```
|
|
||||||
*
|
|
||||||
* @param {Partial<ListTablesOptions>} options - Pagination options
|
|
||||||
* (`pageToken`, `limit`).
|
|
||||||
* @returns {Promise<ListTablesResponse>} A page of table names and an
|
|
||||||
* optional token for the tables after it.
|
|
||||||
*/
|
|
||||||
abstract listTables(
|
|
||||||
options?: Partial<ListTablesOptions>,
|
|
||||||
): Promise<ListTablesResponse>;
|
|
||||||
/**
|
|
||||||
* List a page of the tables in this database.
|
|
||||||
*
|
|
||||||
* @param {string[]} namespacePath - The namespace path to list tables from
|
|
||||||
* (defaults to root namespace)
|
|
||||||
* @param {Partial<ListTablesOptions>} options - Pagination options
|
|
||||||
* (`pageToken`, `limit`).
|
|
||||||
* @returns {Promise<ListTablesResponse>} A page of table names and an
|
|
||||||
* optional token for the tables after it.
|
|
||||||
*/
|
|
||||||
abstract listTables(
|
|
||||||
namespacePath?: string[],
|
|
||||||
options?: Partial<ListTablesOptions>,
|
|
||||||
): Promise<ListTablesResponse>;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Open a table in the database.
|
* Open a table in the database.
|
||||||
* @param {string} name - The name of the table
|
* @param {string} name - The name of the table
|
||||||
* @param {string[]} namespacePath - The namespace path of the table (defaults to root namespace)
|
* @param {string[]} namespacePath - The namespace path of the table (defaults to root namespace)
|
||||||
* @param {Partial<OpenTableOptions>} options - Additional options
|
* @param {Partial<OpenTableOptions>} options - Additional options
|
||||||
*/
|
*/
|
||||||
/**
|
|
||||||
* Define a materialized view named `name` over the table `source`.
|
|
||||||
*
|
|
||||||
* The view is created empty, with the query recorded in its schema
|
|
||||||
* metadata; `view.refresh()` computes the rows. The view is a normal
|
|
||||||
* table: it can be queried, indexed and searched, and it appears in
|
|
||||||
* `tableNames`. The source table must have stable row ids (create it with
|
|
||||||
* the `newTableEnableStableRowIds` storage option); they keep the view's
|
|
||||||
* provenance valid across source compactions and cannot be enabled after
|
|
||||||
* a table exists. Local databases only.
|
|
||||||
*/
|
|
||||||
abstract createMaterializedView(
|
|
||||||
name: string,
|
|
||||||
source: string,
|
|
||||||
options?: {
|
|
||||||
select?: MaterializedViewSelect;
|
|
||||||
where?: string;
|
|
||||||
limit?: number;
|
|
||||||
},
|
|
||||||
): Promise<MaterializedView>;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Open the materialized view named `name`.
|
|
||||||
*
|
|
||||||
* Rejects a table that exists but is not a materialized view.
|
|
||||||
*/
|
|
||||||
abstract openMaterializedView(name: string): Promise<MaterializedView>;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* The names of the materialized views in this database.
|
|
||||||
*
|
|
||||||
* Found by reading every table's schema, so this costs an open per table.
|
|
||||||
*/
|
|
||||||
abstract listMaterializedViews(): Promise<string[]>;
|
|
||||||
|
|
||||||
abstract openTable(
|
abstract openTable(
|
||||||
name: string,
|
name: string,
|
||||||
namespacePath?: string[],
|
namespacePath?: string[],
|
||||||
@@ -434,14 +323,6 @@ export abstract class Connection {
|
|||||||
*/
|
*/
|
||||||
abstract dropTable(name: string, namespacePath?: string[]): Promise<void>;
|
abstract dropTable(name: string, namespacePath?: string[]): Promise<void>;
|
||||||
|
|
||||||
/**
|
|
||||||
* Start dropping a table and return its cleanup job.
|
|
||||||
*
|
|
||||||
* The table may become unavailable before its data files are removed. Wait
|
|
||||||
* on the returned job to know when cleanup has finished.
|
|
||||||
*/
|
|
||||||
abstract dropTableAsync(name: string, namespacePath?: string[]): Promise<Job>;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Drop all tables in the database.
|
* Drop all tables in the database.
|
||||||
* @param {string[]} namespacePath The namespace path to drop tables from (defaults to root namespace).
|
* @param {string[]} namespacePath The namespace path to drop tables from (defaults to root namespace).
|
||||||
@@ -555,40 +436,6 @@ export abstract class Connection {
|
|||||||
newName: string,
|
newName: string,
|
||||||
options?: RenameTableOptions,
|
options?: RenameTableOptions,
|
||||||
): Promise<void>;
|
): Promise<void>;
|
||||||
|
|
||||||
/**
|
|
||||||
* A {@link Job} handle for a server-side job by id.
|
|
||||||
*
|
|
||||||
* The handle is constructed without a server round trip; an unknown id
|
|
||||||
* surfaces when the handle is used. Dropping the handle has no effect on
|
|
||||||
* the job itself.
|
|
||||||
*/
|
|
||||||
abstract job(jobId: string): Job;
|
|
||||||
|
|
||||||
/** List server-side jobs across the database's tables. */
|
|
||||||
abstract listJobs(): Promise<JobInfo[]>;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Describe a single server-side job by id.
|
|
||||||
*
|
|
||||||
* Resolves to `null` when the server has no such job.
|
|
||||||
*/
|
|
||||||
abstract getJob(jobId: string): Promise<JobDescription | null>;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Request cancellation of a server-side job by id.
|
|
||||||
*
|
|
||||||
* Resolves to true if the server accepted the cancellation, false if no
|
|
||||||
* such job exists. Cancelling an already-terminal job is a no-op success.
|
|
||||||
*/
|
|
||||||
abstract cancelJob(jobId: string): Promise<boolean>;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* The lifecycle event history of a server-side job, as an Arrow table.
|
|
||||||
*
|
|
||||||
* Lists history across all jobs when `jobId` is omitted.
|
|
||||||
*/
|
|
||||||
abstract jobHistory(jobId?: string): Promise<ArrowTable>;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/** @hideconstructor */
|
/** @hideconstructor */
|
||||||
@@ -638,54 +485,6 @@ export class LocalConnection extends Connection {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
async createMaterializedView(
|
|
||||||
name: string,
|
|
||||||
source: string,
|
|
||||||
options?: {
|
|
||||||
select?: MaterializedViewSelect;
|
|
||||||
where?: string;
|
|
||||||
limit?: number;
|
|
||||||
},
|
|
||||||
): Promise<MaterializedView> {
|
|
||||||
validateNonNegativeInteger(options?.limit, "limit");
|
|
||||||
const innerTable = await this.inner.createMaterializedView(
|
|
||||||
name,
|
|
||||||
source,
|
|
||||||
normalizeSelect(options?.select),
|
|
||||||
options?.where,
|
|
||||||
options?.limit,
|
|
||||||
);
|
|
||||||
return new MaterializedView(new LocalTable(innerTable));
|
|
||||||
}
|
|
||||||
|
|
||||||
async openMaterializedView(name: string): Promise<MaterializedView> {
|
|
||||||
const innerTable = await this.inner.openMaterializedView(name);
|
|
||||||
return new MaterializedView(new LocalTable(innerTable));
|
|
||||||
}
|
|
||||||
|
|
||||||
async listMaterializedViews(): Promise<string[]> {
|
|
||||||
return await this.inner.listMaterializedViews();
|
|
||||||
}
|
|
||||||
|
|
||||||
async listTables(
|
|
||||||
namespacePathOrOptions?: string[] | Partial<ListTablesOptions>,
|
|
||||||
options?: Partial<ListTablesOptions>,
|
|
||||||
): Promise<ListTablesResponse> {
|
|
||||||
// Detect if first argument is namespacePath array or options object
|
|
||||||
const namespacePath = Array.isArray(namespacePathOrOptions)
|
|
||||||
? namespacePathOrOptions
|
|
||||||
: undefined;
|
|
||||||
const listTablesOptions = Array.isArray(namespacePathOrOptions)
|
|
||||||
? options
|
|
||||||
: namespacePathOrOptions;
|
|
||||||
|
|
||||||
return this.inner.listTables(
|
|
||||||
namespacePath ?? [],
|
|
||||||
listTablesOptions?.pageToken,
|
|
||||||
listTablesOptions?.limit,
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
async openTable(
|
async openTable(
|
||||||
name: string,
|
name: string,
|
||||||
namespacePath?: string[],
|
namespacePath?: string[],
|
||||||
@@ -868,10 +667,6 @@ export class LocalConnection extends Connection {
|
|||||||
return this.inner.dropTable(name, namespacePath ?? []);
|
return this.inner.dropTable(name, namespacePath ?? []);
|
||||||
}
|
}
|
||||||
|
|
||||||
async dropTableAsync(name: string, namespacePath?: string[]): Promise<Job> {
|
|
||||||
return this.inner.dropTableAsync(name, namespacePath ?? []);
|
|
||||||
}
|
|
||||||
|
|
||||||
async dropAllTables(namespacePath?: string[]): Promise<void> {
|
async dropAllTables(namespacePath?: string[]): Promise<void> {
|
||||||
return this.inner.dropAllTables(namespacePath ?? []);
|
return this.inner.dropAllTables(namespacePath ?? []);
|
||||||
}
|
}
|
||||||
@@ -927,30 +722,6 @@ export class LocalConnection extends Connection {
|
|||||||
options?.newNamespacePath,
|
options?.newNamespacePath,
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
job(jobId: string): Job {
|
|
||||||
return this.inner.job(jobId);
|
|
||||||
}
|
|
||||||
|
|
||||||
async listJobs(): Promise<JobInfo[]> {
|
|
||||||
return this.inner.listJobs();
|
|
||||||
}
|
|
||||||
|
|
||||||
async getJob(jobId: string): Promise<JobDescription | null> {
|
|
||||||
return this.inner.getJob(jobId);
|
|
||||||
}
|
|
||||||
|
|
||||||
async cancelJob(jobId: string): Promise<boolean> {
|
|
||||||
return this.inner.cancelJob(jobId);
|
|
||||||
}
|
|
||||||
|
|
||||||
async jobHistory(jobId?: string): Promise<ArrowTable> {
|
|
||||||
const buf = await this.inner.jobHistory(jobId);
|
|
||||||
if (buf.length === 0) {
|
|
||||||
return new ArrowTable();
|
|
||||||
}
|
|
||||||
return tableFromIPC(buf);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
@@ -4,15 +4,7 @@
|
|||||||
import { Field, Schema } from "../arrow";
|
import { Field, Schema } from "../arrow";
|
||||||
import { sanitizeType } from "../sanitize";
|
import { sanitizeType } from "../sanitize";
|
||||||
import { EmbeddingFunction } from "./embedding_function";
|
import { EmbeddingFunction } from "./embedding_function";
|
||||||
import {
|
import { EmbeddingFunctionConfig, getRegistry } from "./registry";
|
||||||
EmbeddingFunctionConfig,
|
|
||||||
EmbeddingFunctionRegistry,
|
|
||||||
getRegistry as getGlobalRegistry,
|
|
||||||
registerBuiltIn,
|
|
||||||
} from "./registry";
|
|
||||||
|
|
||||||
type OpenAIModule = typeof import("./openai");
|
|
||||||
type TransformersModule = typeof import("./transformers");
|
|
||||||
|
|
||||||
export {
|
export {
|
||||||
FieldOptions,
|
FieldOptions,
|
||||||
@@ -22,39 +14,7 @@ export {
|
|||||||
EmbeddingFunctionConstructor,
|
EmbeddingFunctionConstructor,
|
||||||
} from "./embedding_function";
|
} from "./embedding_function";
|
||||||
|
|
||||||
export {
|
export * from "./registry";
|
||||||
EmbeddingFunctionRegistry,
|
|
||||||
parseEmbeddingMetadata,
|
|
||||||
register,
|
|
||||||
} from "./registry";
|
|
||||||
export type {
|
|
||||||
CreateReturnType,
|
|
||||||
EmbeddingFunctionConfig,
|
|
||||||
EmbeddingFunctionCreate,
|
|
||||||
EmbeddingMetadataEntry,
|
|
||||||
ResolvedEmbeddingFunctionConfig,
|
|
||||||
} from "./registry";
|
|
||||||
|
|
||||||
function initializeBuiltInProviders() {
|
|
||||||
const { OpenAIEmbeddingFunction } = require("./openai") as OpenAIModule;
|
|
||||||
const { TransformersEmbeddingFunction } =
|
|
||||||
require("./transformers") as TransformersModule;
|
|
||||||
|
|
||||||
registerBuiltIn("openai", OpenAIEmbeddingFunction);
|
|
||||||
registerBuiltIn("huggingface", TransformersEmbeddingFunction);
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Get the global embedding function registry.
|
|
||||||
*
|
|
||||||
* LanceDB built-in providers are initialized when this public API is first
|
|
||||||
* used, so importing the root package does not change automatic search
|
|
||||||
* selection for tables without embedding metadata.
|
|
||||||
*/
|
|
||||||
export function getRegistry(): EmbeddingFunctionRegistry {
|
|
||||||
initializeBuiltInProviders();
|
|
||||||
return getGlobalRegistry();
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Create a schema with embedding functions.
|
* Create a schema with embedding functions.
|
||||||
|
|||||||
@@ -5,13 +5,14 @@ import type OpenAI from "openai";
|
|||||||
import type { EmbeddingCreateParams } from "openai/resources/index";
|
import type { EmbeddingCreateParams } from "openai/resources/index";
|
||||||
import { Float, Float32 } from "../arrow";
|
import { Float, Float32 } from "../arrow";
|
||||||
import { EmbeddingFunction } from "./embedding_function";
|
import { EmbeddingFunction } from "./embedding_function";
|
||||||
import { registerBuiltIn } from "./registry";
|
import { register } from "./registry";
|
||||||
|
|
||||||
export type OpenAIOptions = {
|
export type OpenAIOptions = {
|
||||||
apiKey: string;
|
apiKey: string;
|
||||||
model: EmbeddingCreateParams["model"];
|
model: EmbeddingCreateParams["model"];
|
||||||
};
|
};
|
||||||
|
|
||||||
|
@register("openai")
|
||||||
export class OpenAIEmbeddingFunction extends EmbeddingFunction<
|
export class OpenAIEmbeddingFunction extends EmbeddingFunction<
|
||||||
string,
|
string,
|
||||||
Partial<OpenAIOptions>
|
Partial<OpenAIOptions>
|
||||||
@@ -99,5 +100,3 @@ export class OpenAIEmbeddingFunction extends EmbeddingFunction<
|
|||||||
return response.data[0].embedding;
|
return response.data[0].embedding;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
registerBuiltIn("openai", OpenAIEmbeddingFunction);
|
|
||||||
|
|||||||
@@ -7,10 +7,6 @@ import {
|
|||||||
} from "./embedding_function";
|
} from "./embedding_function";
|
||||||
import "reflect-metadata";
|
import "reflect-metadata";
|
||||||
|
|
||||||
const builtInFunctionsKey = Symbol.for(
|
|
||||||
"@lancedb/lancedb::embedding-built-in-functions::v1",
|
|
||||||
);
|
|
||||||
|
|
||||||
export type CreateReturnType<T> = T extends { init: () => Promise<void> }
|
export type CreateReturnType<T> = T extends { init: () => Promise<void> }
|
||||||
? Promise<T>
|
? Promise<T>
|
||||||
: T;
|
: T;
|
||||||
@@ -63,15 +59,6 @@ export class EmbeddingFunctionRegistry {
|
|||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
/** @ignore */
|
|
||||||
setBuiltIn<
|
|
||||||
T extends EmbeddingFunctionConstructor = EmbeddingFunctionConstructor,
|
|
||||||
>(name: string, ctor: T): T {
|
|
||||||
this.#functions.set(name, ctor);
|
|
||||||
Reflect.defineMetadata("lancedb::embedding::name", name, ctor);
|
|
||||||
return ctor;
|
|
||||||
}
|
|
||||||
|
|
||||||
get<T extends EmbeddingFunction<unknown>>(
|
get<T extends EmbeddingFunction<unknown>>(
|
||||||
name: string,
|
name: string,
|
||||||
): EmbeddingFunctionCreate<T> | undefined;
|
): EmbeddingFunctionCreate<T> | undefined;
|
||||||
@@ -109,7 +96,6 @@ export class EmbeddingFunctionRegistry {
|
|||||||
*/
|
*/
|
||||||
reset(this: EmbeddingFunctionRegistry) {
|
reset(this: EmbeddingFunctionRegistry) {
|
||||||
this.#functions.clear();
|
this.#functions.clear();
|
||||||
getBuiltInFunctions(this).clear();
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -118,29 +104,41 @@ export class EmbeddingFunctionRegistry {
|
|||||||
async parseFunctions(
|
async parseFunctions(
|
||||||
this: EmbeddingFunctionRegistry,
|
this: EmbeddingFunctionRegistry,
|
||||||
metadata: Map<string, string>,
|
metadata: Map<string, string>,
|
||||||
): Promise<Map<string, ResolvedEmbeddingFunctionConfig>> {
|
): Promise<Map<string, EmbeddingFunctionConfig>> {
|
||||||
if (!metadata.has("embedding_functions")) {
|
if (!metadata.has("embedding_functions")) {
|
||||||
return new Map();
|
return new Map();
|
||||||
|
} else {
|
||||||
|
type FunctionConfig = {
|
||||||
|
name: string;
|
||||||
|
sourceColumn: string;
|
||||||
|
vectorColumn: string;
|
||||||
|
model: EmbeddingFunction["TOptions"];
|
||||||
|
};
|
||||||
|
|
||||||
|
const functions = <FunctionConfig[]>(
|
||||||
|
JSON.parse(metadata.get("embedding_functions")!)
|
||||||
|
);
|
||||||
|
|
||||||
|
const items: [string, EmbeddingFunctionConfig][] = await Promise.all(
|
||||||
|
functions.map(async (f) => {
|
||||||
|
const fn = this.get(f.name);
|
||||||
|
if (!fn) {
|
||||||
|
throw new Error(`Function "${f.name}" not found in registry`);
|
||||||
|
}
|
||||||
|
const func = await this.get(f.name)!.create(f.model);
|
||||||
|
return [
|
||||||
|
f.name,
|
||||||
|
{
|
||||||
|
sourceColumn: f.sourceColumn,
|
||||||
|
vectorColumn: f.vectorColumn,
|
||||||
|
function: func,
|
||||||
|
},
|
||||||
|
];
|
||||||
|
}),
|
||||||
|
);
|
||||||
|
|
||||||
|
return new Map(items);
|
||||||
}
|
}
|
||||||
const entries = parseEmbeddingMetadata(
|
|
||||||
metadata.get("embedding_functions")!,
|
|
||||||
);
|
|
||||||
const items = await Promise.all(
|
|
||||||
entries.map(async (f): Promise<ResolvedEmbeddingFunctionConfig> => {
|
|
||||||
const fn = this.get(f.name);
|
|
||||||
if (!fn) {
|
|
||||||
throw new Error(`Function "${f.name}" not found in registry`);
|
|
||||||
}
|
|
||||||
const func = await fn.create(f.model);
|
|
||||||
return {
|
|
||||||
sourceColumn: f.sourceColumn,
|
|
||||||
vectorColumn: f.vectorColumn,
|
|
||||||
function: func,
|
|
||||||
};
|
|
||||||
}),
|
|
||||||
);
|
|
||||||
// Keyed by output column: one function may serve several columns.
|
|
||||||
return new Map(items.map((config) => [config.vectorColumn, config]));
|
|
||||||
}
|
}
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: <explanation>
|
// biome-ignore lint/suspicious/noExplicitAny: <explanation>
|
||||||
functionToMetadata(conf: EmbeddingFunctionConfig): Record<string, any> {
|
functionToMetadata(conf: EmbeddingFunctionConfig): Record<string, any> {
|
||||||
@@ -197,56 +195,12 @@ export class EmbeddingFunctionRegistry {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
function getBuiltInFunctions(registry: EmbeddingFunctionRegistry): Set<string> {
|
const _REGISTRY = new EmbeddingFunctionRegistry();
|
||||||
const registryWithBuiltIns = registry as EmbeddingFunctionRegistry & {
|
|
||||||
[key: symbol]: Set<string> | undefined;
|
|
||||||
};
|
|
||||||
let builtInFunctions = registryWithBuiltIns[builtInFunctionsKey];
|
|
||||||
if (builtInFunctions === undefined) {
|
|
||||||
builtInFunctions = new Set<string>();
|
|
||||||
registryWithBuiltIns[builtInFunctionsKey] = builtInFunctions;
|
|
||||||
}
|
|
||||||
return builtInFunctions;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Server bundlers can load the side-effect embedding entry points and the public
|
|
||||||
// embedding API from separate module graphs. Keep their registry shared.
|
|
||||||
const registryKey = Symbol.for(
|
|
||||||
"@lancedb/lancedb::embedding-function-registry::v1",
|
|
||||||
);
|
|
||||||
const registryGlobal = globalThis as typeof globalThis & {
|
|
||||||
[key: symbol]: EmbeddingFunctionRegistry | undefined;
|
|
||||||
};
|
|
||||||
|
|
||||||
function getGlobalRegistry(): EmbeddingFunctionRegistry {
|
|
||||||
const existingRegistry = registryGlobal[registryKey];
|
|
||||||
if (existingRegistry !== undefined) {
|
|
||||||
return existingRegistry;
|
|
||||||
}
|
|
||||||
const registry = new EmbeddingFunctionRegistry();
|
|
||||||
registryGlobal[registryKey] = registry;
|
|
||||||
return registry;
|
|
||||||
}
|
|
||||||
|
|
||||||
const _REGISTRY = getGlobalRegistry();
|
|
||||||
|
|
||||||
export function register(name?: string) {
|
export function register(name?: string) {
|
||||||
return _REGISTRY.register(name);
|
return _REGISTRY.register(name);
|
||||||
}
|
}
|
||||||
|
|
||||||
/** @ignore */
|
|
||||||
export function registerBuiltIn<
|
|
||||||
T extends EmbeddingFunctionConstructor = EmbeddingFunctionConstructor,
|
|
||||||
>(name: string, ctor: T): T {
|
|
||||||
const builtInFunctions = getBuiltInFunctions(_REGISTRY);
|
|
||||||
if (builtInFunctions.has(name)) {
|
|
||||||
return _REGISTRY.setBuiltIn(name, ctor);
|
|
||||||
}
|
|
||||||
_REGISTRY.register(name)(ctor);
|
|
||||||
builtInFunctions.add(name);
|
|
||||||
return ctor;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Utility function to get the global instance of the registry
|
* Utility function to get the global instance of the registry
|
||||||
* @returns `EmbeddingFunctionRegistry` The global instance of the registry
|
* @returns `EmbeddingFunctionRegistry` The global instance of the registry
|
||||||
@@ -264,52 +218,3 @@ export interface EmbeddingFunctionConfig {
|
|||||||
vectorColumn?: string;
|
vectorColumn?: string;
|
||||||
function: EmbeddingFunction;
|
function: EmbeddingFunction;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** An [EmbeddingFunctionConfig] read back from table metadata, where the
|
|
||||||
* vector column is always recorded. */
|
|
||||||
export type ResolvedEmbeddingFunctionConfig = EmbeddingFunctionConfig & {
|
|
||||||
vectorColumn: string;
|
|
||||||
};
|
|
||||||
|
|
||||||
/** One entry of the `embedding_functions` schema metadata, with the column
|
|
||||||
* keys normalized across the bindings' spellings. */
|
|
||||||
export type EmbeddingMetadataEntry = {
|
|
||||||
name: string;
|
|
||||||
sourceColumn: string;
|
|
||||||
vectorColumn: string;
|
|
||||||
model: EmbeddingFunction["TOptions"];
|
|
||||||
};
|
|
||||||
|
|
||||||
/** The single parser for `embedding_functions` schema metadata: every reader
|
|
||||||
* goes through here, so the wire contract cannot fork between them. */
|
|
||||||
export function parseEmbeddingMetadata(json: string): EmbeddingMetadataEntry[] {
|
|
||||||
// The wire format, honestly: the Python bindings write snake_case keys.
|
|
||||||
type Raw = {
|
|
||||||
name: string;
|
|
||||||
sourceColumn?: string;
|
|
||||||
// biome-ignore lint/style/useNamingConvention: the Python wire spelling
|
|
||||||
source_column?: string;
|
|
||||||
vectorColumn?: string;
|
|
||||||
// biome-ignore lint/style/useNamingConvention: the Python wire spelling
|
|
||||||
vector_column?: string;
|
|
||||||
model: EmbeddingFunction["TOptions"];
|
|
||||||
};
|
|
||||||
const entries = <Raw[]>JSON.parse(json);
|
|
||||||
const seen = new Set<string>();
|
|
||||||
return entries.map((f) => {
|
|
||||||
const sourceColumn = f.sourceColumn ?? f.source_column;
|
|
||||||
const vectorColumn = f.vectorColumn ?? f.vector_column;
|
|
||||||
if (sourceColumn === undefined || vectorColumn === undefined) {
|
|
||||||
throw new Error(
|
|
||||||
`Embedding function "${f.name}" metadata names no source or vector column`,
|
|
||||||
);
|
|
||||||
}
|
|
||||||
if (seen.has(vectorColumn)) {
|
|
||||||
throw new Error(
|
|
||||||
`Multiple embedding configs claim vector column "${vectorColumn}"`,
|
|
||||||
);
|
|
||||||
}
|
|
||||||
seen.add(vectorColumn);
|
|
||||||
return { name: f.name, sourceColumn, vectorColumn, model: f.model };
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -3,7 +3,7 @@
|
|||||||
|
|
||||||
import { Float, Float32 } from "../arrow";
|
import { Float, Float32 } from "../arrow";
|
||||||
import { EmbeddingFunction } from "./embedding_function";
|
import { EmbeddingFunction } from "./embedding_function";
|
||||||
import { registerBuiltIn } from "./registry";
|
import { register } from "./registry";
|
||||||
|
|
||||||
export type XenovaTransformerOptions = {
|
export type XenovaTransformerOptions = {
|
||||||
/** The wasm compatible model to use */
|
/** The wasm compatible model to use */
|
||||||
@@ -31,6 +31,7 @@ export type XenovaTransformerOptions = {
|
|||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
|
@register("huggingface")
|
||||||
export class TransformersEmbeddingFunction extends EmbeddingFunction<
|
export class TransformersEmbeddingFunction extends EmbeddingFunction<
|
||||||
string,
|
string,
|
||||||
Partial<XenovaTransformerOptions>
|
Partial<XenovaTransformerOptions>
|
||||||
@@ -157,8 +158,6 @@ export class TransformersEmbeddingFunction extends EmbeddingFunction<
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
registerBuiltIn("huggingface", TransformersEmbeddingFunction);
|
|
||||||
|
|
||||||
const tensorDiv = (
|
const tensorDiv = (
|
||||||
src: import("@huggingface/transformers").Tensor,
|
src: import("@huggingface/transformers").Tensor,
|
||||||
divBy: number,
|
divBy: number,
|
||||||
|
|||||||
+4
-24
@@ -21,11 +21,6 @@ import type { BaseTokenizer } from "./indices";
|
|||||||
import type { FtsToken } from "./table";
|
import type { FtsToken } from "./table";
|
||||||
|
|
||||||
// Re-export native header provider for use with connectWithHeaderProvider
|
// Re-export native header provider for use with connectWithHeaderProvider
|
||||||
export {
|
|
||||||
MaterializedView,
|
|
||||||
MaterializedViewDefinition,
|
|
||||||
MaterializedViewSelect,
|
|
||||||
} from "./materialized_view";
|
|
||||||
export { JsHeaderProvider as NativeJsHeaderProvider } from "./native.js";
|
export { JsHeaderProvider as NativeJsHeaderProvider } from "./native.js";
|
||||||
|
|
||||||
// OpenTelemetry metrics bridge. Only the high-level entry point is public; the
|
// OpenTelemetry metrics bridge. Only the high-level entry point is public; the
|
||||||
@@ -55,8 +50,6 @@ export {
|
|||||||
MergeResult,
|
MergeResult,
|
||||||
AddResult,
|
AddResult,
|
||||||
AddColumnsResult,
|
AddColumnsResult,
|
||||||
RefreshColumnResult,
|
|
||||||
RefreshMaterializedViewResult,
|
|
||||||
AlterColumnsResult,
|
AlterColumnsResult,
|
||||||
UpdateFieldMetadataResult,
|
UpdateFieldMetadataResult,
|
||||||
DeleteResult,
|
DeleteResult,
|
||||||
@@ -81,29 +74,20 @@ export {
|
|||||||
Connection,
|
Connection,
|
||||||
CreateTableOptions,
|
CreateTableOptions,
|
||||||
TableNamesOptions,
|
TableNamesOptions,
|
||||||
ListTablesOptions,
|
|
||||||
OpenTableOptions,
|
OpenTableOptions,
|
||||||
ListNamespacesOptions,
|
ListNamespacesOptions,
|
||||||
CreateNamespaceOptions,
|
CreateNamespaceOptions,
|
||||||
DropNamespaceOptions,
|
DropNamespaceOptions,
|
||||||
ListNamespacesResponse,
|
ListNamespacesResponse,
|
||||||
ListTablesResponse,
|
|
||||||
CreateNamespaceResponse,
|
CreateNamespaceResponse,
|
||||||
DropNamespaceResponse,
|
DropNamespaceResponse,
|
||||||
DescribeNamespaceResponse,
|
DescribeNamespaceResponse,
|
||||||
RenameTableOptions,
|
RenameTableOptions,
|
||||||
} from "./connection";
|
} from "./connection";
|
||||||
|
|
||||||
export {
|
export { Session } from "./native.js";
|
||||||
Job,
|
|
||||||
JobDescription,
|
|
||||||
JobFailureInfo,
|
|
||||||
JobInfo,
|
|
||||||
Session,
|
|
||||||
} from "./native.js";
|
|
||||||
|
|
||||||
export {
|
export {
|
||||||
AutoQuery,
|
|
||||||
ExecutableQuery,
|
ExecutableQuery,
|
||||||
Query,
|
Query,
|
||||||
QueryBase,
|
QueryBase,
|
||||||
@@ -144,10 +128,10 @@ export {
|
|||||||
BranchColumnChange,
|
BranchColumnChange,
|
||||||
BranchIndexSummary,
|
BranchIndexSummary,
|
||||||
BranchRowCountSummary,
|
BranchRowCountSummary,
|
||||||
CherryPickError,
|
MergeBlocker,
|
||||||
BranchDiff,
|
BranchDiff,
|
||||||
CherryPickPreview,
|
MergePreview,
|
||||||
CherryPickResult,
|
MergeBranchResult,
|
||||||
AddDataOptions,
|
AddDataOptions,
|
||||||
UpdateOptions,
|
UpdateOptions,
|
||||||
OptimizeOptions,
|
OptimizeOptions,
|
||||||
@@ -156,10 +140,6 @@ export {
|
|||||||
FtsToken,
|
FtsToken,
|
||||||
TokenizeTableOptions,
|
TokenizeTableOptions,
|
||||||
LsmWriteSpec,
|
LsmWriteSpec,
|
||||||
LsmStats,
|
|
||||||
BucketStats,
|
|
||||||
GenerationStats,
|
|
||||||
MemtableStats,
|
|
||||||
ColumnAlteration,
|
ColumnAlteration,
|
||||||
FieldMetadataUpdate,
|
FieldMetadataUpdate,
|
||||||
} from "./table";
|
} from "./table";
|
||||||
|
|||||||
@@ -1,161 +0,0 @@
|
|||||||
// SPDX-License-Identifier: Apache-2.0
|
|
||||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
|
||||||
|
|
||||||
import { RefreshMaterializedViewResult } from "./native";
|
|
||||||
import { Table } from "./table";
|
|
||||||
|
|
||||||
/** Schema metadata key holding a materialized view's definition. */
|
|
||||||
export const DEFINITION_META_KEY = "mv.definition";
|
|
||||||
|
|
||||||
/** The query that defines a materialized view. */
|
|
||||||
export interface MaterializedViewDefinition {
|
|
||||||
/** Name of the source table, in the same database as the view. */
|
|
||||||
sourceTable: string;
|
|
||||||
/** `[output column, SQL expression]` pairs, in view schema order. */
|
|
||||||
projections: [string, string][];
|
|
||||||
/** SQL predicate selecting the source rows the view holds. */
|
|
||||||
filter?: string;
|
|
||||||
/** Cap on the number of rows the view holds. */
|
|
||||||
limit?: number;
|
|
||||||
/** Source columns the projections and filter read. */
|
|
||||||
inputs: string[];
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* The view's columns: column names, `[alias, SQL expression]` pairs, or a
|
|
||||||
* record of the same. A bare name projects itself.
|
|
||||||
*/
|
|
||||||
export type MaterializedViewSelect =
|
|
||||||
| (string | [string, string])[]
|
|
||||||
| Record<string, string>;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* @internal Reject a numeric option N-API would otherwise silently coerce:
|
|
||||||
* `Infinity` reaches Rust as 0, `1.5` as 1.
|
|
||||||
*/
|
|
||||||
export function validateNonNegativeInteger(
|
|
||||||
value: number | undefined,
|
|
||||||
name: string,
|
|
||||||
): void {
|
|
||||||
if (value !== undefined && !(Number.isSafeInteger(value) && value >= 0)) {
|
|
||||||
throw new Error(`${name} must be a non-negative integer`);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/** @internal Quote a column name as a Lance SQL identifier (backticks). */
|
|
||||||
function quoteIdentifier(name: string): string {
|
|
||||||
return "`" + name.replace(/`/g, "``") + "`";
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* @internal Normalize a select argument into `[alias, expression]` pairs.
|
|
||||||
* A bare name projects itself and is quoted, so any valid column name works;
|
|
||||||
* pair and record entries are kept verbatim because their right side is an
|
|
||||||
* expression.
|
|
||||||
*/
|
|
||||||
export function normalizeSelect(
|
|
||||||
select?: MaterializedViewSelect,
|
|
||||||
): [string, string][] | undefined {
|
|
||||||
if (select === undefined) {
|
|
||||||
return undefined;
|
|
||||||
}
|
|
||||||
if (Array.isArray(select)) {
|
|
||||||
return select.map((item) =>
|
|
||||||
typeof item === "string" ? [item, quoteIdentifier(item)] : item,
|
|
||||||
);
|
|
||||||
}
|
|
||||||
return Object.entries(select);
|
|
||||||
}
|
|
||||||
|
|
||||||
/** @internal Parse a definition off a table's stored schema metadata. */
|
|
||||||
export function definitionFromMetadata(
|
|
||||||
metadata: Map<string, string>,
|
|
||||||
name: string,
|
|
||||||
): MaterializedViewDefinition {
|
|
||||||
const raw = metadata.get(DEFINITION_META_KEY);
|
|
||||||
if (raw === undefined) {
|
|
||||||
throw new Error(`Table '${name}' is not a materialized view`);
|
|
||||||
}
|
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: raw JSON
|
|
||||||
const value: any = JSON.parse(raw);
|
|
||||||
if (value.kind !== "select") {
|
|
||||||
throw new Error(
|
|
||||||
`materialized view '${name}' is defined by '${value.kind}', which this ` +
|
|
||||||
"version of lancedb cannot refresh",
|
|
||||||
);
|
|
||||||
}
|
|
||||||
const limit = value.limit ?? undefined;
|
|
||||||
// JSON.parse rounds integers past 2^53; every exact u64 parses to a safe
|
|
||||||
// integer and every rounded one does not, so this rejects precisely the
|
|
||||||
// values a number cannot carry.
|
|
||||||
if (limit !== undefined && !Number.isSafeInteger(limit)) {
|
|
||||||
throw new Error(
|
|
||||||
`materialized view '${name}' has a stored limit too large to represent exactly`,
|
|
||||||
);
|
|
||||||
}
|
|
||||||
return {
|
|
||||||
sourceTable: value.source_table,
|
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: raw JSON
|
|
||||||
projections: (value.projections ?? []).map((p: any) => [
|
|
||||||
p.output,
|
|
||||||
p.expression,
|
|
||||||
]),
|
|
||||||
filter: value.filter ?? undefined,
|
|
||||||
limit,
|
|
||||||
inputs: value.inputs ?? [],
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* A handle on a materialized view: its table plus its definition.
|
|
||||||
*
|
|
||||||
* Obtained from {@link Connection#createMaterializedView} or
|
|
||||||
* {@link Connection#openMaterializedView}. The view is a normal table --
|
|
||||||
* queries, indexes and search all apply through {@link MaterializedView#table}
|
|
||||||
* -- whose contents are maintained by {@link MaterializedView#refresh}.
|
|
||||||
*/
|
|
||||||
export class MaterializedView {
|
|
||||||
private readonly inner: Table;
|
|
||||||
|
|
||||||
constructor(table: Table) {
|
|
||||||
this.inner = table;
|
|
||||||
}
|
|
||||||
|
|
||||||
get name(): string {
|
|
||||||
return this.inner.name;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** The view, as the table it is. */
|
|
||||||
table(): Table {
|
|
||||||
return this.inner;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** The query that defines the view, read from its stored schema. */
|
|
||||||
async definition(): Promise<MaterializedViewDefinition> {
|
|
||||||
const schema = await this.inner.schema();
|
|
||||||
return definitionFromMetadata(schema.metadata, this.name);
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Recompute the view from its source.
|
|
||||||
*
|
|
||||||
* The refresh is incremental when the source's changes can be reconciled
|
|
||||||
* into the view -- rows added, changed or removed since the last one --
|
|
||||||
* and otherwise rebuilds. `full` forces a rebuild; `sourceVersion`
|
|
||||||
* refreshes to that source version instead of the latest.
|
|
||||||
*
|
|
||||||
* Concurrent refreshes of one view do not duplicate its rows. Two that
|
|
||||||
* plan the same source rows conflict on commit, and the loser throws
|
|
||||||
* rather than writing them a second time.
|
|
||||||
*/
|
|
||||||
async refresh(options?: {
|
|
||||||
full?: boolean;
|
|
||||||
sourceVersion?: number;
|
|
||||||
}): Promise<RefreshMaterializedViewResult> {
|
|
||||||
validateNonNegativeInteger(options?.sourceVersion, "sourceVersion");
|
|
||||||
return await this.inner.refreshMaterializedView(
|
|
||||||
options?.full,
|
|
||||||
options?.sourceVersion,
|
|
||||||
);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
+106
-205
@@ -100,29 +100,6 @@ export interface FullTextSearchOptions {
|
|||||||
columns?: string | string[];
|
columns?: string | string[];
|
||||||
}
|
}
|
||||||
|
|
||||||
function nearestToNative(
|
|
||||||
inner: NativeQuery,
|
|
||||||
vector: Awaited<IntoVector>,
|
|
||||||
): NativeVectorQuery {
|
|
||||||
const raw = Array.isArray(vector) ? null : extractVectorBuffer(vector);
|
|
||||||
if (raw) {
|
|
||||||
return inner.nearestToRaw(raw.data, raw.dtype);
|
|
||||||
}
|
|
||||||
return inner.nearestTo(Float32Array.from(vector as number[]));
|
|
||||||
}
|
|
||||||
|
|
||||||
function addQueryVectorToNative(
|
|
||||||
inner: NativeVectorQuery,
|
|
||||||
vector: Awaited<IntoVector>,
|
|
||||||
) {
|
|
||||||
const raw = Array.isArray(vector) ? null : extractVectorBuffer(vector);
|
|
||||||
if (raw) {
|
|
||||||
inner.addQueryVectorRaw(raw.data, raw.dtype);
|
|
||||||
} else {
|
|
||||||
inner.addQueryVector(Float32Array.from(vector as number[]));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Common methods supported by all query types
|
/** Common methods supported by all query types
|
||||||
*
|
*
|
||||||
* @see {@link Query}
|
* @see {@link Query}
|
||||||
@@ -134,15 +111,13 @@ export class QueryBase<
|
|||||||
NativeQueryType extends NativeQuery | NativeVectorQuery | NativeTakeQuery,
|
NativeQueryType extends NativeQuery | NativeVectorQuery | NativeTakeQuery,
|
||||||
> implements AsyncIterable<RecordBatch>
|
> implements AsyncIterable<RecordBatch>
|
||||||
{
|
{
|
||||||
protected inner!: NativeQueryType | Promise<NativeQueryType>;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @hidden
|
* @hidden
|
||||||
*/
|
*/
|
||||||
protected constructor(inner?: NativeQueryType | Promise<NativeQueryType>) {
|
protected constructor(
|
||||||
if (inner !== undefined) {
|
protected inner: NativeQueryType | Promise<NativeQueryType>,
|
||||||
this.inner = inner;
|
) {
|
||||||
}
|
// intentionally empty
|
||||||
}
|
}
|
||||||
|
|
||||||
// call a function on the inner (either a promise or the actual object)
|
// call a function on the inner (either a promise or the actual object)
|
||||||
@@ -160,15 +135,6 @@ export class QueryBase<
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Return the native query used by the next terminal operation.
|
|
||||||
*
|
|
||||||
* @hidden
|
|
||||||
*/
|
|
||||||
protected async getInner(): Promise<NativeQueryType> {
|
|
||||||
return this.inner;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Return only the specified columns.
|
* Return only the specified columns.
|
||||||
*
|
*
|
||||||
@@ -241,11 +207,16 @@ export class QueryBase<
|
|||||||
/**
|
/**
|
||||||
* @hidden
|
* @hidden
|
||||||
*/
|
*/
|
||||||
protected async nativeExecute(
|
protected nativeExecute(
|
||||||
options?: Partial<QueryExecutionOptions>,
|
options?: Partial<QueryExecutionOptions>,
|
||||||
): Promise<NativeBatchIterator> {
|
): Promise<NativeBatchIterator> {
|
||||||
const inner = await this.getInner();
|
if (this.inner instanceof Promise) {
|
||||||
return inner.execute(options?.maxBatchLength, options?.timeoutMs);
|
return this.inner.then((inner) =>
|
||||||
|
inner.execute(options?.maxBatchLength, options?.timeoutMs),
|
||||||
|
);
|
||||||
|
} else {
|
||||||
|
return this.inner.execute(options?.maxBatchLength, options?.timeoutMs);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -274,7 +245,12 @@ export class QueryBase<
|
|||||||
/** Collect the results as an Arrow @see {@link ArrowTable}. */
|
/** Collect the results as an Arrow @see {@link ArrowTable}. */
|
||||||
async toArrow(options?: Partial<QueryExecutionOptions>): Promise<ArrowTable> {
|
async toArrow(options?: Partial<QueryExecutionOptions>): Promise<ArrowTable> {
|
||||||
const batches = [];
|
const batches = [];
|
||||||
const inner = await this.getInner();
|
let inner;
|
||||||
|
if (this.inner instanceof Promise) {
|
||||||
|
inner = await this.inner;
|
||||||
|
} else {
|
||||||
|
inner = this.inner;
|
||||||
|
}
|
||||||
for await (const batch of new RecordBatchIterable(inner, options)) {
|
for await (const batch of new RecordBatchIterable(inner, options)) {
|
||||||
batches.push(batch);
|
batches.push(batch);
|
||||||
}
|
}
|
||||||
@@ -303,8 +279,11 @@ export class QueryBase<
|
|||||||
* @returns A Promise that resolves to a string containing the query execution plan explanation.
|
* @returns A Promise that resolves to a string containing the query execution plan explanation.
|
||||||
*/
|
*/
|
||||||
async explainPlan(verbose = false): Promise<string> {
|
async explainPlan(verbose = false): Promise<string> {
|
||||||
const inner = await this.getInner();
|
if (this.inner instanceof Promise) {
|
||||||
return inner.explainPlan(verbose);
|
return this.inner.then((inner) => inner.explainPlan(verbose));
|
||||||
|
} else {
|
||||||
|
return this.inner.explainPlan(verbose);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -342,8 +321,13 @@ export class QueryBase<
|
|||||||
distributedMetrics?: AnalyzePlanDistributedMetrics,
|
distributedMetrics?: AnalyzePlanDistributedMetrics,
|
||||||
): Promise<string> {
|
): Promise<string> {
|
||||||
const distributedMetricsMode = distributedMetrics ?? "aggregate";
|
const distributedMetricsMode = distributedMetrics ?? "aggregate";
|
||||||
const inner = await this.getInner();
|
if (this.inner instanceof Promise) {
|
||||||
return inner.analyzePlan(distributedMetricsMode);
|
return this.inner.then((inner) =>
|
||||||
|
inner.analyzePlan(distributedMetricsMode),
|
||||||
|
);
|
||||||
|
} else {
|
||||||
|
return this.inner.analyzePlan(distributedMetricsMode);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -355,8 +339,12 @@ export class QueryBase<
|
|||||||
* @returns An Arrow Schema describing the output columns.
|
* @returns An Arrow Schema describing the output columns.
|
||||||
*/
|
*/
|
||||||
async outputSchema(): Promise<import("./arrow").Schema> {
|
async outputSchema(): Promise<import("./arrow").Schema> {
|
||||||
const inner = await this.getInner();
|
let schemaBuffer: Buffer;
|
||||||
const schemaBuffer = await inner.outputSchema();
|
if (this.inner instanceof Promise) {
|
||||||
|
schemaBuffer = await this.inner.then((inner) => inner.outputSchema());
|
||||||
|
} else {
|
||||||
|
schemaBuffer = await this.inner.outputSchema();
|
||||||
|
}
|
||||||
const schema = tableFromIPC(schemaBuffer).schema;
|
const schema = tableFromIPC(schemaBuffer).schema;
|
||||||
return schema;
|
return schema;
|
||||||
}
|
}
|
||||||
@@ -368,7 +356,7 @@ export class StandardQueryBase<
|
|||||||
extends QueryBase<NativeQueryType>
|
extends QueryBase<NativeQueryType>
|
||||||
implements ExecutableQuery
|
implements ExecutableQuery
|
||||||
{
|
{
|
||||||
constructor(inner?: NativeQueryType | Promise<NativeQueryType>) {
|
constructor(inner: NativeQueryType | Promise<NativeQueryType>) {
|
||||||
super(inner);
|
super(inner);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -522,13 +510,6 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
|
|||||||
super(inner);
|
super(inner);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* @hidden
|
|
||||||
*/
|
|
||||||
protected doVectorCall(fn: (inner: NativeVectorQuery) => void) {
|
|
||||||
super.doCall(fn);
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Set the number of partitions to search (probe)
|
* Set the number of partitions to search (probe)
|
||||||
*
|
*
|
||||||
@@ -556,7 +537,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
|
|||||||
* the minimum and maximum to the same value.
|
* the minimum and maximum to the same value.
|
||||||
*/
|
*/
|
||||||
nprobes(nprobes: number): VectorQuery {
|
nprobes(nprobes: number): VectorQuery {
|
||||||
this.doVectorCall((inner) => inner.nprobes(nprobes));
|
super.doCall((inner) => inner.nprobes(nprobes));
|
||||||
|
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
@@ -570,7 +551,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
|
|||||||
* but will also increase latency.
|
* but will also increase latency.
|
||||||
*/
|
*/
|
||||||
minimumNprobes(minimumNprobes: number): VectorQuery {
|
minimumNprobes(minimumNprobes: number): VectorQuery {
|
||||||
this.doVectorCall((inner) => inner.minimumNprobes(minimumNprobes));
|
super.doCall((inner) => inner.minimumNprobes(minimumNprobes));
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -584,7 +565,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
|
|||||||
* potential false negatives.
|
* potential false negatives.
|
||||||
*/
|
*/
|
||||||
maximumNprobes(maximumNprobes: number): VectorQuery {
|
maximumNprobes(maximumNprobes: number): VectorQuery {
|
||||||
this.doVectorCall((inner) => inner.maximumNprobes(maximumNprobes));
|
super.doCall((inner) => inner.maximumNprobes(maximumNprobes));
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -597,7 +578,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
|
|||||||
* `undefined` means no lower or upper bound.
|
* `undefined` means no lower or upper bound.
|
||||||
*/
|
*/
|
||||||
distanceRange(lowerBound?: number, upperBound?: number): VectorQuery {
|
distanceRange(lowerBound?: number, upperBound?: number): VectorQuery {
|
||||||
this.doVectorCall((inner) => inner.distanceRange(lowerBound, upperBound));
|
super.doCall((inner) => inner.distanceRange(lowerBound, upperBound));
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -611,7 +592,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
|
|||||||
* also increase the latency of your query. The default value is 1.5*limit.
|
* also increase the latency of your query. The default value is 1.5*limit.
|
||||||
*/
|
*/
|
||||||
ef(ef: number): VectorQuery {
|
ef(ef: number): VectorQuery {
|
||||||
this.doVectorCall((inner) => inner.ef(ef));
|
super.doCall((inner) => inner.ef(ef));
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -625,7 +606,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
|
|||||||
* whose data type is a fixed-size-list of floats.
|
* whose data type is a fixed-size-list of floats.
|
||||||
*/
|
*/
|
||||||
column(column: string): VectorQuery {
|
column(column: string): VectorQuery {
|
||||||
this.doVectorCall((inner) => inner.column(column));
|
super.doCall((inner) => inner.column(column));
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -646,7 +627,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
|
|||||||
distanceType(
|
distanceType(
|
||||||
distanceType: Required<IvfPqOptions>["distanceType"],
|
distanceType: Required<IvfPqOptions>["distanceType"],
|
||||||
): VectorQuery {
|
): VectorQuery {
|
||||||
this.doVectorCall((inner) => inner.distanceType(distanceType));
|
super.doCall((inner) => inner.distanceType(distanceType));
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -680,7 +661,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
|
|||||||
* distance between the query vector and the actual uncompressed vector.
|
* distance between the query vector and the actual uncompressed vector.
|
||||||
*/
|
*/
|
||||||
refineFactor(refineFactor: number): VectorQuery {
|
refineFactor(refineFactor: number): VectorQuery {
|
||||||
this.doVectorCall((inner) => inner.refineFactor(refineFactor));
|
super.doCall((inner) => inner.refineFactor(refineFactor));
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -705,7 +686,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
|
|||||||
* factor can often help restore some of the results lost by post filtering.
|
* factor can often help restore some of the results lost by post filtering.
|
||||||
*/
|
*/
|
||||||
postfilter(): VectorQuery {
|
postfilter(): VectorQuery {
|
||||||
this.doVectorCall((inner) => inner.postfilter());
|
super.doCall((inner) => inner.postfilter());
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -719,7 +700,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
|
|||||||
* calculate your recall to select an appropriate value for nprobes.
|
* calculate your recall to select an appropriate value for nprobes.
|
||||||
*/
|
*/
|
||||||
bypassVectorIndex(): VectorQuery {
|
bypassVectorIndex(): VectorQuery {
|
||||||
this.doVectorCall((inner) => inner.bypassVectorIndex());
|
super.doCall((inner) => inner.bypassVectorIndex());
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -727,39 +708,43 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
|
|||||||
* Add a query vector to the search
|
* Add a query vector to the search
|
||||||
*
|
*
|
||||||
* This method can be called multiple times to add multiple query vectors
|
* This method can be called multiple times to add multiple query vectors
|
||||||
* to the search. A column called `query_index` will be added to indicate the index
|
* to the search. If multiple query vectors are added, then they will be searched
|
||||||
* of the query vector that produced the result. Flat searches share one table scan
|
* in parallel, and the results will be concatenated. A column called `query_index`
|
||||||
* across the query vectors, avoiding the scan and memory amplification of running
|
* will be added to indicate the index of the query vector that produced the result.
|
||||||
* multiple queries concurrently. Indexed searches may still perform per-vector
|
*
|
||||||
* index work.
|
* Performance wise, this is equivalent to running multiple queries concurrently.
|
||||||
*/
|
*/
|
||||||
addQueryVector(vector: IntoVector): VectorQuery {
|
addQueryVector(vector: IntoVector): VectorQuery {
|
||||||
if (vector instanceof Promise) {
|
if (vector instanceof Promise) {
|
||||||
// Observe the promise as soon as it is accepted. The existing native
|
|
||||||
// query may still be pending, and delaying observation until it resolves
|
|
||||||
// can otherwise surface a fast rejection as unhandled.
|
|
||||||
const settledVector = vector.then(
|
|
||||||
(value) => ({ status: "fulfilled" as const, value }),
|
|
||||||
(reason) => ({ status: "rejected" as const, reason }),
|
|
||||||
);
|
|
||||||
const res = (async () => {
|
const res = (async () => {
|
||||||
const inner = await this.getInner();
|
try {
|
||||||
const outcome = await settledVector;
|
const v = await vector;
|
||||||
if (outcome.status === "rejected") {
|
// biome-ignore lint/suspicious/noExplicitAny: we need to get the `inner`, but js has no package scoping
|
||||||
throw outcome.reason;
|
const value: any = this.addQueryVector(v);
|
||||||
|
const inner = value.inner as
|
||||||
|
| NativeVectorQuery
|
||||||
|
| Promise<NativeVectorQuery>;
|
||||||
|
return inner;
|
||||||
|
} catch (e) {
|
||||||
|
return Promise.reject(e);
|
||||||
}
|
}
|
||||||
addQueryVectorToNative(inner, outcome.value);
|
|
||||||
return inner;
|
|
||||||
})();
|
})();
|
||||||
return new VectorQuery(res);
|
return new VectorQuery(res);
|
||||||
} else {
|
} else {
|
||||||
this.doVectorCall((inner) => addQueryVectorToNative(inner, vector));
|
super.doCall((inner) => {
|
||||||
|
const raw = Array.isArray(vector) ? null : extractVectorBuffer(vector);
|
||||||
|
if (raw) {
|
||||||
|
inner.addQueryVectorRaw(raw.data, raw.dtype);
|
||||||
|
} else {
|
||||||
|
inner.addQueryVector(Float32Array.from(vector as number[]));
|
||||||
|
}
|
||||||
|
});
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
rerank(reranker: Reranker): VectorQuery {
|
rerank(reranker: Reranker): VectorQuery {
|
||||||
this.doVectorCall((inner) =>
|
super.doCall((inner) =>
|
||||||
inner.rerank(async (args) => {
|
inner.rerank(async (args) => {
|
||||||
const vecResults = await fromBufferToRecordBatch(args.vecResults);
|
const vecResults = await fromBufferToRecordBatch(args.vecResults);
|
||||||
const ftsResults = await fromBufferToRecordBatch(args.ftsResults);
|
const ftsResults = await fromBufferToRecordBatch(args.ftsResults);
|
||||||
@@ -778,71 +763,6 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Create a string query whose vector/FTS routing is resolved against the active
|
|
||||||
* table schema when the query executes.
|
|
||||||
*
|
|
||||||
* @hidden
|
|
||||||
*/
|
|
||||||
export function createAutoQuery(
|
|
||||||
table: NativeTable,
|
|
||||||
query: string,
|
|
||||||
columns: string[] | null,
|
|
||||||
getVector: (metadata: string) => Promise<Awaited<IntoVector>>,
|
|
||||||
): AutoQuery {
|
|
||||||
type RouteSnapshot = {
|
|
||||||
table: NativeTable;
|
|
||||||
embeddingMetadata: string | undefined;
|
|
||||||
};
|
|
||||||
type CachedPreparation = {
|
|
||||||
metadata: string;
|
|
||||||
vector: Promise<Awaited<IntoVector>>;
|
|
||||||
};
|
|
||||||
|
|
||||||
let cachedPreparation: CachedPreparation | undefined;
|
|
||||||
|
|
||||||
const snapshotRoute = async (): Promise<RouteSnapshot> => {
|
|
||||||
const snapshot = await table.querySnapshot();
|
|
||||||
const schema = tableFromIPC(await snapshot.schema()).schema;
|
|
||||||
return {
|
|
||||||
table: snapshot,
|
|
||||||
embeddingMetadata: schema.metadata.get("embedding_functions"),
|
|
||||||
};
|
|
||||||
};
|
|
||||||
|
|
||||||
const createInner = async (): Promise<NativeQuery | NativeVectorQuery> => {
|
|
||||||
const route = await snapshotRoute();
|
|
||||||
if (route.embeddingMetadata === undefined) {
|
|
||||||
const inner = route.table.query();
|
|
||||||
inner.fullTextSearch({ query, columns });
|
|
||||||
return inner;
|
|
||||||
}
|
|
||||||
|
|
||||||
const metadata = route.embeddingMetadata;
|
|
||||||
if (cachedPreparation?.metadata !== metadata) {
|
|
||||||
cachedPreparation = {
|
|
||||||
metadata,
|
|
||||||
vector: Promise.resolve().then(() => getVector(metadata)),
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
const preparation = cachedPreparation;
|
|
||||||
let vector: Awaited<IntoVector>;
|
|
||||||
try {
|
|
||||||
vector = await preparation.vector;
|
|
||||||
} catch (error) {
|
|
||||||
if (cachedPreparation === preparation) {
|
|
||||||
cachedPreparation = undefined;
|
|
||||||
}
|
|
||||||
throw error;
|
|
||||||
}
|
|
||||||
|
|
||||||
return nearestToNative(route.table.query(), vector);
|
|
||||||
};
|
|
||||||
|
|
||||||
return new AutoQuery(createInner);
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* A query that returns a subset of the rows in the table.
|
* A query that returns a subset of the rows in the table.
|
||||||
*
|
*
|
||||||
@@ -868,51 +788,6 @@ export class TakeQuery extends QueryBase<NativeTakeQuery> {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* A builder for automatic string searches.
|
|
||||||
*
|
|
||||||
* Automatic search determines whether to use full-text or vector search from
|
|
||||||
* the table revision selected for each execution. This builder exposes the
|
|
||||||
* common operations supported by both query families.
|
|
||||||
*
|
|
||||||
* @hideconstructor
|
|
||||||
*/
|
|
||||||
export class AutoQuery extends StandardQueryBase<
|
|
||||||
NativeQuery | NativeVectorQuery
|
|
||||||
> {
|
|
||||||
private readonly calls: Array<
|
|
||||||
(inner: NativeQuery | NativeVectorQuery) => void
|
|
||||||
> = [];
|
|
||||||
|
|
||||||
/** @hidden */
|
|
||||||
constructor(
|
|
||||||
private readonly createInner: () => Promise<
|
|
||||||
NativeQuery | NativeVectorQuery
|
|
||||||
>,
|
|
||||||
) {
|
|
||||||
super();
|
|
||||||
}
|
|
||||||
|
|
||||||
/** @hidden */
|
|
||||||
protected override doCall(
|
|
||||||
fn: (inner: NativeQuery | NativeVectorQuery) => void,
|
|
||||||
) {
|
|
||||||
this.calls.push(fn);
|
|
||||||
}
|
|
||||||
|
|
||||||
/** @hidden */
|
|
||||||
protected override async getInner(): Promise<
|
|
||||||
NativeQuery | NativeVectorQuery
|
|
||||||
> {
|
|
||||||
const calls = [...this.calls];
|
|
||||||
const inner = await this.createInner();
|
|
||||||
for (const call of calls) {
|
|
||||||
call(inner);
|
|
||||||
}
|
|
||||||
return inner;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/** A builder for LanceDB queries.
|
/** A builder for LanceDB queries.
|
||||||
*
|
*
|
||||||
* @see {@link Table#query}, {@link Table#search}
|
* @see {@link Table#query}, {@link Table#search}
|
||||||
@@ -965,19 +840,45 @@ export class Query extends StandardQueryBase<NativeQuery> {
|
|||||||
* a default `limit` of 10 will be used. @see {@link Query#limit}
|
* a default `limit` of 10 will be used. @see {@link Query#limit}
|
||||||
*/
|
*/
|
||||||
nearestTo(vector: IntoVector): VectorQuery {
|
nearestTo(vector: IntoVector): VectorQuery {
|
||||||
const inner = this.inner;
|
const callNearestTo = (
|
||||||
if (inner instanceof Promise) {
|
inner: NativeQuery,
|
||||||
const nativeQuery = inner.then(async (resolvedInner) =>
|
resolved: Float32Array | Float64Array | Uint8Array | number[],
|
||||||
nearestToNative(resolvedInner, await vector),
|
): NativeVectorQuery => {
|
||||||
);
|
const raw = Array.isArray(resolved)
|
||||||
|
? null
|
||||||
|
: extractVectorBuffer(resolved);
|
||||||
|
if (raw) {
|
||||||
|
return inner.nearestToRaw(raw.data, raw.dtype);
|
||||||
|
}
|
||||||
|
return inner.nearestTo(Float32Array.from(resolved as number[]));
|
||||||
|
};
|
||||||
|
|
||||||
|
if (this.inner instanceof Promise) {
|
||||||
|
const nativeQuery = this.inner.then(async (inner) => {
|
||||||
|
const resolved = vector instanceof Promise ? await vector : vector;
|
||||||
|
return callNearestTo(inner, resolved);
|
||||||
|
});
|
||||||
return new VectorQuery(nativeQuery);
|
return new VectorQuery(nativeQuery);
|
||||||
}
|
}
|
||||||
if (vector instanceof Promise) {
|
if (vector instanceof Promise) {
|
||||||
return new VectorQuery(
|
const res = (async () => {
|
||||||
vector.then((resolvedVector) => nearestToNative(inner, resolvedVector)),
|
try {
|
||||||
);
|
const v = await vector;
|
||||||
|
// biome-ignore lint/suspicious/noExplicitAny: we need to get the `inner`, but js has no package scoping
|
||||||
|
const value: any = this.nearestTo(v);
|
||||||
|
const inner = value.inner as
|
||||||
|
| NativeVectorQuery
|
||||||
|
| Promise<NativeVectorQuery>;
|
||||||
|
return inner;
|
||||||
|
} catch (e) {
|
||||||
|
return Promise.reject(e);
|
||||||
|
}
|
||||||
|
})();
|
||||||
|
return new VectorQuery(res);
|
||||||
|
} else {
|
||||||
|
const vectorQuery = callNearestTo(this.inner, vector);
|
||||||
|
return new VectorQuery(vectorQuery);
|
||||||
}
|
}
|
||||||
return new VectorQuery(nearestToNative(inner, vector));
|
|
||||||
}
|
}
|
||||||
|
|
||||||
nearestToText(query: string | FullTextQuery, columns?: string[]): Query {
|
nearestToText(query: string | FullTextQuery, columns?: string[]): Query {
|
||||||
|
|||||||
+29
-174
@@ -9,7 +9,7 @@
|
|||||||
// comes from the exact same library instance. This is not always the case
|
// comes from the exact same library instance. This is not always the case
|
||||||
// and so we must sanitize the input to ensure that it is compatible.
|
// and so we must sanitize the input to ensure that it is compatible.
|
||||||
|
|
||||||
import { BufferType, Data, Vector } from "apache-arrow";
|
import { BufferType, Data } from "apache-arrow";
|
||||||
import type { IntBitWidth, TKeys, TimeBitWidth } from "apache-arrow/type";
|
import type { IntBitWidth, TKeys, TimeBitWidth } from "apache-arrow/type";
|
||||||
import {
|
import {
|
||||||
Binary,
|
Binary,
|
||||||
@@ -74,20 +74,6 @@ import {
|
|||||||
Utf8,
|
Utf8,
|
||||||
} from "./arrow";
|
} from "./arrow";
|
||||||
|
|
||||||
type SanitizationContext = {
|
|
||||||
types: WeakMap<object, DataType>;
|
|
||||||
vectors: WeakMap<object, Vector>;
|
|
||||||
data: WeakMap<object, Data<DataType>>;
|
|
||||||
};
|
|
||||||
|
|
||||||
function createSanitizationContext(): SanitizationContext {
|
|
||||||
return {
|
|
||||||
types: new WeakMap(),
|
|
||||||
vectors: new WeakMap(),
|
|
||||||
data: new WeakMap(),
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
export function sanitizeMetadata(
|
export function sanitizeMetadata(
|
||||||
metadataLike?: unknown,
|
metadataLike?: unknown,
|
||||||
): Map<string, string> | undefined {
|
): Map<string, string> | undefined {
|
||||||
@@ -200,13 +186,6 @@ export function sanitizeInterval(typeLike: object) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeList(typeLike: object) {
|
export function sanitizeList(typeLike: object) {
|
||||||
return sanitizeListWithContext(typeLike, createSanitizationContext());
|
|
||||||
}
|
|
||||||
|
|
||||||
function sanitizeListWithContext(
|
|
||||||
typeLike: object,
|
|
||||||
context: SanitizationContext,
|
|
||||||
) {
|
|
||||||
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
|
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
|
||||||
throw Error(
|
throw Error(
|
||||||
"Expected a List type to have an array-like `children` property",
|
"Expected a List type to have an array-like `children` property",
|
||||||
@@ -215,35 +194,19 @@ function sanitizeListWithContext(
|
|||||||
if (typeLike.children.length !== 1) {
|
if (typeLike.children.length !== 1) {
|
||||||
throw Error("Expected a List type to have exactly one child");
|
throw Error("Expected a List type to have exactly one child");
|
||||||
}
|
}
|
||||||
return new List(sanitizeFieldWithContext(typeLike.children[0], context));
|
return new List(sanitizeField(typeLike.children[0]));
|
||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeStruct(typeLike: object) {
|
export function sanitizeStruct(typeLike: object) {
|
||||||
return sanitizeStructWithContext(typeLike, createSanitizationContext());
|
|
||||||
}
|
|
||||||
|
|
||||||
function sanitizeStructWithContext(
|
|
||||||
typeLike: object,
|
|
||||||
context: SanitizationContext,
|
|
||||||
) {
|
|
||||||
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
|
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
|
||||||
throw Error(
|
throw Error(
|
||||||
"Expected a Struct type to have an array-like `children` property",
|
"Expected a Struct type to have an array-like `children` property",
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
return new Struct(
|
return new Struct(typeLike.children.map((child) => sanitizeField(child)));
|
||||||
typeLike.children.map((child) => sanitizeFieldWithContext(child, context)),
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeUnion(typeLike: object) {
|
export function sanitizeUnion(typeLike: object) {
|
||||||
return sanitizeUnionWithContext(typeLike, createSanitizationContext());
|
|
||||||
}
|
|
||||||
|
|
||||||
function sanitizeUnionWithContext(
|
|
||||||
typeLike: object,
|
|
||||||
context: SanitizationContext,
|
|
||||||
) {
|
|
||||||
if (
|
if (
|
||||||
!("typeIds" in typeLike) ||
|
!("typeIds" in typeLike) ||
|
||||||
!("mode" in typeLike) ||
|
!("mode" in typeLike) ||
|
||||||
@@ -263,7 +226,7 @@ function sanitizeUnionWithContext(
|
|||||||
typeLike.mode,
|
typeLike.mode,
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: skip
|
// biome-ignore lint/suspicious/noExplicitAny: skip
|
||||||
typeLike.typeIds as any,
|
typeLike.typeIds as any,
|
||||||
typeLike.children.map((child) => sanitizeFieldWithContext(child, context)),
|
typeLike.children.map((child) => sanitizeField(child)),
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -271,19 +234,6 @@ export function sanitizeTypedUnion(
|
|||||||
typeLike: object,
|
typeLike: object,
|
||||||
// eslint-disable-next-line @typescript-eslint/naming-convention
|
// eslint-disable-next-line @typescript-eslint/naming-convention
|
||||||
UnionType: typeof DenseUnion | typeof SparseUnion,
|
UnionType: typeof DenseUnion | typeof SparseUnion,
|
||||||
) {
|
|
||||||
return sanitizeTypedUnionWithContext(
|
|
||||||
typeLike,
|
|
||||||
UnionType,
|
|
||||||
createSanitizationContext(),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
function sanitizeTypedUnionWithContext(
|
|
||||||
typeLike: object,
|
|
||||||
// eslint-disable-next-line @typescript-eslint/naming-convention
|
|
||||||
UnionType: typeof DenseUnion | typeof SparseUnion,
|
|
||||||
context: SanitizationContext,
|
|
||||||
) {
|
) {
|
||||||
if (!("typeIds" in typeLike)) {
|
if (!("typeIds" in typeLike)) {
|
||||||
throw Error(
|
throw Error(
|
||||||
@@ -298,7 +248,7 @@ function sanitizeTypedUnionWithContext(
|
|||||||
|
|
||||||
return new UnionType(
|
return new UnionType(
|
||||||
typeLike.typeIds as Int32Array | number[],
|
typeLike.typeIds as Int32Array | number[],
|
||||||
typeLike.children.map((child) => sanitizeFieldWithContext(child, context)),
|
typeLike.children.map((child) => sanitizeField(child)),
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -312,16 +262,6 @@ export function sanitizeFixedSizeBinary(typeLike: object) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeFixedSizeList(typeLike: object) {
|
export function sanitizeFixedSizeList(typeLike: object) {
|
||||||
return sanitizeFixedSizeListWithContext(
|
|
||||||
typeLike,
|
|
||||||
createSanitizationContext(),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
function sanitizeFixedSizeListWithContext(
|
|
||||||
typeLike: object,
|
|
||||||
context: SanitizationContext,
|
|
||||||
) {
|
|
||||||
if (!("listSize" in typeLike) || typeof typeLike.listSize !== "number") {
|
if (!("listSize" in typeLike) || typeof typeLike.listSize !== "number") {
|
||||||
throw Error("Expected a FixedSizeList type to have a `listSize` property");
|
throw Error("Expected a FixedSizeList type to have a `listSize` property");
|
||||||
}
|
}
|
||||||
@@ -335,18 +275,11 @@ function sanitizeFixedSizeListWithContext(
|
|||||||
}
|
}
|
||||||
return new FixedSizeList(
|
return new FixedSizeList(
|
||||||
typeLike.listSize,
|
typeLike.listSize,
|
||||||
sanitizeFieldWithContext(typeLike.children[0], context),
|
sanitizeField(typeLike.children[0]),
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeMap(typeLike: object) {
|
export function sanitizeMap(typeLike: object) {
|
||||||
return sanitizeMapWithContext(typeLike, createSanitizationContext());
|
|
||||||
}
|
|
||||||
|
|
||||||
function sanitizeMapWithContext(
|
|
||||||
typeLike: object,
|
|
||||||
context: SanitizationContext,
|
|
||||||
) {
|
|
||||||
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
|
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
|
||||||
throw Error(
|
throw Error(
|
||||||
"Expected a Map type to have an array-like `children` property",
|
"Expected a Map type to have an array-like `children` property",
|
||||||
@@ -359,10 +292,7 @@ function sanitizeMapWithContext(
|
|||||||
throw Error("Expected a Map type to have exactly one child");
|
throw Error("Expected a Map type to have exactly one child");
|
||||||
}
|
}
|
||||||
|
|
||||||
return new Map_(
|
return new Map_(sanitizeField(typeLike.children[0]), typeLike.keysSorted);
|
||||||
sanitizeFieldWithContext(typeLike.children[0], context),
|
|
||||||
typeLike.keysSorted,
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeDuration(typeLike: object) {
|
export function sanitizeDuration(typeLike: object) {
|
||||||
@@ -373,13 +303,6 @@ export function sanitizeDuration(typeLike: object) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeDictionary(typeLike: object) {
|
export function sanitizeDictionary(typeLike: object) {
|
||||||
return sanitizeDictionaryWithContext(typeLike, createSanitizationContext());
|
|
||||||
}
|
|
||||||
|
|
||||||
function sanitizeDictionaryWithContext(
|
|
||||||
typeLike: object,
|
|
||||||
context: SanitizationContext,
|
|
||||||
) {
|
|
||||||
if (!("id" in typeLike) || typeof typeLike.id !== "number") {
|
if (!("id" in typeLike) || typeof typeLike.id !== "number") {
|
||||||
throw Error("Expected a Dictionary type to have an `id` property");
|
throw Error("Expected a Dictionary type to have an `id` property");
|
||||||
}
|
}
|
||||||
@@ -393,8 +316,8 @@ function sanitizeDictionaryWithContext(
|
|||||||
throw Error("Expected a Dictionary type to have an `isOrdered` property");
|
throw Error("Expected a Dictionary type to have an `isOrdered` property");
|
||||||
}
|
}
|
||||||
return new Dictionary(
|
return new Dictionary(
|
||||||
sanitizeTypeWithContext(typeLike.dictionary, context),
|
sanitizeType(typeLike.dictionary),
|
||||||
sanitizeTypeWithContext(typeLike.indices, context) as TKeys,
|
sanitizeType(typeLike.indices) as TKeys,
|
||||||
typeLike.id,
|
typeLike.id,
|
||||||
typeLike.isOrdered,
|
typeLike.isOrdered,
|
||||||
);
|
);
|
||||||
@@ -402,23 +325,12 @@ function sanitizeDictionaryWithContext(
|
|||||||
|
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: skip
|
// biome-ignore lint/suspicious/noExplicitAny: skip
|
||||||
export function sanitizeType(typeLike: unknown): DataType<any> {
|
export function sanitizeType(typeLike: unknown): DataType<any> {
|
||||||
return sanitizeTypeWithContext(typeLike, createSanitizationContext());
|
|
||||||
}
|
|
||||||
|
|
||||||
function sanitizeTypeWithContext(
|
|
||||||
typeLike: unknown,
|
|
||||||
context: SanitizationContext,
|
|
||||||
): DataType {
|
|
||||||
if (typeof typeLike === "string") {
|
if (typeof typeLike === "string") {
|
||||||
return dataTypeFromName(typeLike);
|
return dataTypeFromName(typeLike);
|
||||||
}
|
}
|
||||||
if (typeof typeLike !== "object" || typeLike === null) {
|
if (typeof typeLike !== "object" || typeLike === null) {
|
||||||
throw Error("Expected a Type but object was null/undefined");
|
throw Error("Expected a Type but object was null/undefined");
|
||||||
}
|
}
|
||||||
const cached = context.types.get(typeLike);
|
|
||||||
if (cached !== undefined) {
|
|
||||||
return cached;
|
|
||||||
}
|
|
||||||
if (
|
if (
|
||||||
!("typeId" in typeLike) ||
|
!("typeId" in typeLike) ||
|
||||||
!(
|
!(
|
||||||
@@ -437,16 +349,6 @@ function sanitizeTypeWithContext(
|
|||||||
throw Error("Type's typeId property was not a function or number");
|
throw Error("Type's typeId property was not a function or number");
|
||||||
}
|
}
|
||||||
|
|
||||||
const type = sanitizeTypeById(typeLike, typeId, context);
|
|
||||||
context.types.set(typeLike, type);
|
|
||||||
return type;
|
|
||||||
}
|
|
||||||
|
|
||||||
function sanitizeTypeById(
|
|
||||||
typeLike: object,
|
|
||||||
typeId: Type,
|
|
||||||
context: SanitizationContext,
|
|
||||||
): DataType {
|
|
||||||
switch (typeId) {
|
switch (typeId) {
|
||||||
case Type.NONE:
|
case Type.NONE:
|
||||||
throw Error("Received a Type with a typeId of NONE");
|
throw Error("Received a Type with a typeId of NONE");
|
||||||
@@ -473,21 +375,21 @@ function sanitizeTypeById(
|
|||||||
case Type.Interval:
|
case Type.Interval:
|
||||||
return sanitizeInterval(typeLike);
|
return sanitizeInterval(typeLike);
|
||||||
case Type.List:
|
case Type.List:
|
||||||
return sanitizeListWithContext(typeLike, context);
|
return sanitizeList(typeLike);
|
||||||
case Type.Struct:
|
case Type.Struct:
|
||||||
return sanitizeStructWithContext(typeLike, context);
|
return sanitizeStruct(typeLike);
|
||||||
case Type.Union:
|
case Type.Union:
|
||||||
return sanitizeUnionWithContext(typeLike, context);
|
return sanitizeUnion(typeLike);
|
||||||
case Type.FixedSizeBinary:
|
case Type.FixedSizeBinary:
|
||||||
return sanitizeFixedSizeBinary(typeLike);
|
return sanitizeFixedSizeBinary(typeLike);
|
||||||
case Type.FixedSizeList:
|
case Type.FixedSizeList:
|
||||||
return sanitizeFixedSizeListWithContext(typeLike, context);
|
return sanitizeFixedSizeList(typeLike);
|
||||||
case Type.Map:
|
case Type.Map:
|
||||||
return sanitizeMapWithContext(typeLike, context);
|
return sanitizeMap(typeLike);
|
||||||
case Type.Duration:
|
case Type.Duration:
|
||||||
return sanitizeDuration(typeLike);
|
return sanitizeDuration(typeLike);
|
||||||
case Type.Dictionary:
|
case Type.Dictionary:
|
||||||
return sanitizeDictionaryWithContext(typeLike, context);
|
return sanitizeDictionary(typeLike);
|
||||||
case Type.Int8:
|
case Type.Int8:
|
||||||
return new Int8();
|
return new Int8();
|
||||||
case Type.Int16:
|
case Type.Int16:
|
||||||
@@ -531,9 +433,9 @@ function sanitizeTypeById(
|
|||||||
case Type.TimestampSecond:
|
case Type.TimestampSecond:
|
||||||
return sanitizeTypedTimestamp(typeLike, TimestampSecond);
|
return sanitizeTypedTimestamp(typeLike, TimestampSecond);
|
||||||
case Type.DenseUnion:
|
case Type.DenseUnion:
|
||||||
return sanitizeTypedUnionWithContext(typeLike, DenseUnion, context);
|
return sanitizeTypedUnion(typeLike, DenseUnion);
|
||||||
case Type.SparseUnion:
|
case Type.SparseUnion:
|
||||||
return sanitizeTypedUnionWithContext(typeLike, SparseUnion, context);
|
return sanitizeTypedUnion(typeLike, SparseUnion);
|
||||||
case Type.IntervalDayTime:
|
case Type.IntervalDayTime:
|
||||||
return new IntervalDayTime();
|
return new IntervalDayTime();
|
||||||
case Type.IntervalYearMonth:
|
case Type.IntervalYearMonth:
|
||||||
@@ -552,13 +454,6 @@ function sanitizeTypeById(
|
|||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeField(fieldLike: unknown): Field {
|
export function sanitizeField(fieldLike: unknown): Field {
|
||||||
return sanitizeFieldWithContext(fieldLike, createSanitizationContext());
|
|
||||||
}
|
|
||||||
|
|
||||||
function sanitizeFieldWithContext(
|
|
||||||
fieldLike: unknown,
|
|
||||||
context: SanitizationContext,
|
|
||||||
): Field {
|
|
||||||
if (fieldLike instanceof Field) {
|
if (fieldLike instanceof Field) {
|
||||||
return fieldLike;
|
return fieldLike;
|
||||||
}
|
}
|
||||||
@@ -576,7 +471,7 @@ function sanitizeFieldWithContext(
|
|||||||
}
|
}
|
||||||
let type: DataType;
|
let type: DataType;
|
||||||
try {
|
try {
|
||||||
type = sanitizeTypeWithContext(fieldLike.type, context);
|
type = sanitizeType(fieldLike.type);
|
||||||
} catch (error: unknown) {
|
} catch (error: unknown) {
|
||||||
throw Error(
|
throw Error(
|
||||||
`Unable to sanitize type for field: ${fieldLike.name} due to error: ${error}`,
|
`Unable to sanitize type for field: ${fieldLike.name} due to error: ${error}`,
|
||||||
@@ -606,13 +501,6 @@ function sanitizeFieldWithContext(
|
|||||||
* than lancedb is using.
|
* than lancedb is using.
|
||||||
*/
|
*/
|
||||||
export function sanitizeSchema(schemaLike: SchemaLike): Schema {
|
export function sanitizeSchema(schemaLike: SchemaLike): Schema {
|
||||||
return sanitizeSchemaWithContext(schemaLike, createSanitizationContext());
|
|
||||||
}
|
|
||||||
|
|
||||||
function sanitizeSchemaWithContext(
|
|
||||||
schemaLike: SchemaLike,
|
|
||||||
context: SanitizationContext,
|
|
||||||
): Schema {
|
|
||||||
if (schemaLike instanceof Schema) {
|
if (schemaLike instanceof Schema) {
|
||||||
return schemaLike;
|
return schemaLike;
|
||||||
}
|
}
|
||||||
@@ -634,7 +522,7 @@ function sanitizeSchemaWithContext(
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
const sanitizedFields = schemaLike.fields.map((field) =>
|
const sanitizedFields = schemaLike.fields.map((field) =>
|
||||||
sanitizeFieldWithContext(field, context),
|
sanitizeField(field),
|
||||||
);
|
);
|
||||||
return new Schema(sanitizedFields, metadata);
|
return new Schema(sanitizedFields, metadata);
|
||||||
}
|
}
|
||||||
@@ -656,18 +544,13 @@ export function sanitizeTable(tableLike: TableLike): Table {
|
|||||||
"The table passed in does not appear to be a table (no 'columns' property)",
|
"The table passed in does not appear to be a table (no 'columns' property)",
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
const context = createSanitizationContext();
|
const schema = sanitizeSchema(tableLike.schema);
|
||||||
const schema = sanitizeSchemaWithContext(tableLike.schema, context);
|
|
||||||
const batches = tableLike.batches.map((batch) =>
|
const batches = tableLike.batches.map(sanitizeRecordBatch);
|
||||||
sanitizeRecordBatch(batch, context),
|
|
||||||
);
|
|
||||||
return new Table(schema, batches);
|
return new Table(schema, batches);
|
||||||
}
|
}
|
||||||
|
|
||||||
function sanitizeRecordBatch(
|
function sanitizeRecordBatch(batchLike: RecordBatchLike): RecordBatch {
|
||||||
batchLike: RecordBatchLike,
|
|
||||||
context: SanitizationContext,
|
|
||||||
): RecordBatch {
|
|
||||||
if (batchLike instanceof RecordBatch) {
|
if (batchLike instanceof RecordBatch) {
|
||||||
return batchLike;
|
return batchLike;
|
||||||
}
|
}
|
||||||
@@ -684,43 +567,19 @@ function sanitizeRecordBatch(
|
|||||||
"The record batch passed in does not appear to be a record batch (no 'data' property)",
|
"The record batch passed in does not appear to be a record batch (no 'data' property)",
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
const schema = sanitizeSchemaWithContext(batchLike.schema, context);
|
const schema = sanitizeSchema(batchLike.schema);
|
||||||
const data = sanitizeData(batchLike.data, context) as Data<Struct>;
|
const data = sanitizeData(batchLike.data);
|
||||||
return new RecordBatch(schema, data);
|
return new RecordBatch(schema, data);
|
||||||
}
|
}
|
||||||
|
|
||||||
type DictionaryVectorLike = {
|
|
||||||
data: readonly DataLike[];
|
|
||||||
};
|
|
||||||
|
|
||||||
type DictionaryDataLike = DataLike & {
|
|
||||||
dictionary?: DictionaryVectorLike;
|
|
||||||
};
|
|
||||||
|
|
||||||
function sanitizeData(
|
function sanitizeData(
|
||||||
dataLike: DataLike,
|
dataLike: DataLike,
|
||||||
context: SanitizationContext,
|
// biome-ignore lint/suspicious/noExplicitAny: <explanation>
|
||||||
): Data<DataType> {
|
): import("apache-arrow").Data<Struct<any>> {
|
||||||
if (dataLike instanceof Data) {
|
if (dataLike instanceof Data) {
|
||||||
return dataLike;
|
return dataLike;
|
||||||
}
|
}
|
||||||
const cachedData = context.data.get(dataLike);
|
return new Data(
|
||||||
if (cachedData !== undefined) {
|
dataLike.type,
|
||||||
return cachedData;
|
|
||||||
}
|
|
||||||
const dictionaryLike = (dataLike as DictionaryDataLike).dictionary;
|
|
||||||
let dictionary: Vector | undefined;
|
|
||||||
if (dictionaryLike !== undefined) {
|
|
||||||
dictionary = context.vectors.get(dictionaryLike);
|
|
||||||
if (dictionary === undefined) {
|
|
||||||
dictionary = new Vector(
|
|
||||||
dictionaryLike.data.map((data) => sanitizeData(data, context)),
|
|
||||||
);
|
|
||||||
context.vectors.set(dictionaryLike, dictionary);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const data = new Data(
|
|
||||||
sanitizeTypeWithContext(dataLike.type, context),
|
|
||||||
dataLike.offset,
|
dataLike.offset,
|
||||||
dataLike.length,
|
dataLike.length,
|
||||||
dataLike.nullCount,
|
dataLike.nullCount,
|
||||||
@@ -730,11 +589,7 @@ function sanitizeData(
|
|||||||
[BufferType.VALIDITY]: dataLike.nullBitmap,
|
[BufferType.VALIDITY]: dataLike.nullBitmap,
|
||||||
[BufferType.TYPE]: dataLike.typeIds,
|
[BufferType.TYPE]: dataLike.typeIds,
|
||||||
},
|
},
|
||||||
dataLike.children.map((child) => sanitizeData(child, context)),
|
|
||||||
dictionary,
|
|
||||||
);
|
);
|
||||||
context.data.set(dataLike, data);
|
|
||||||
return data;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
const constructorsByTypeName = {
|
const constructorsByTypeName = {
|
||||||
|
|||||||
@@ -1,567 +0,0 @@
|
|||||||
// SPDX-License-Identifier: Apache-2.0
|
|
||||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
|
||||||
|
|
||||||
import {
|
|
||||||
Binary,
|
|
||||||
Bool,
|
|
||||||
DataType,
|
|
||||||
Dictionary,
|
|
||||||
Field,
|
|
||||||
FixedSizeList,
|
|
||||||
Float32,
|
|
||||||
Float64,
|
|
||||||
Int32,
|
|
||||||
Int64,
|
|
||||||
List,
|
|
||||||
Schema,
|
|
||||||
Struct,
|
|
||||||
Utf8,
|
|
||||||
util as arrowUtil,
|
|
||||||
} from "apache-arrow";
|
|
||||||
import { typedArrayToArrowType } from "./arrow_type";
|
|
||||||
import { sanitizeType } from "./sanitize";
|
|
||||||
|
|
||||||
type InferenceOptions = {
|
|
||||||
dictionaryEncodeStrings: boolean;
|
|
||||||
vectorColumns: Record<string, { type: unknown }>;
|
|
||||||
};
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Infer the Arrow schema represented by a set of records.
|
|
||||||
*
|
|
||||||
* This is the intentionally small interface to schema inference. The stateful
|
|
||||||
* details of combining partial type evidence are encapsulated below so callers
|
|
||||||
* only need to provide records, an optional schema, and inference options.
|
|
||||||
*/
|
|
||||||
export function inferSchema(
|
|
||||||
data: Array<Record<string, unknown>>,
|
|
||||||
schema: Schema | undefined,
|
|
||||||
options: InferenceOptions,
|
|
||||||
): Schema {
|
|
||||||
return new SchemaInferrer(schema, options).infer(data);
|
|
||||||
}
|
|
||||||
|
|
||||||
class SchemaInferrer {
|
|
||||||
private readonly fields = new FieldTree();
|
|
||||||
|
|
||||||
constructor(
|
|
||||||
private readonly providedSchema: Schema | undefined,
|
|
||||||
private readonly options: InferenceOptions,
|
|
||||||
) {}
|
|
||||||
|
|
||||||
infer(data: Array<Record<string, unknown>>): Schema {
|
|
||||||
for (const [row, record] of data.entries()) {
|
|
||||||
for (const [path, value] of recordPathsAndValues(record)) {
|
|
||||||
this.observe(path, value, row);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return this.providedSchema === undefined
|
|
||||||
? new Schema(fieldsFromTree(this.fields))
|
|
||||||
: new Schema(matchingFields(this.providedSchema.fields, this.fields));
|
|
||||||
}
|
|
||||||
|
|
||||||
private observe(path: string[], value: unknown, row: number): void {
|
|
||||||
const current = this.fields.get(path);
|
|
||||||
if (current === undefined) {
|
|
||||||
this.addField(path, value, row);
|
|
||||||
} else if (this.providedSchema === undefined) {
|
|
||||||
this.updateInferredField(path, value, row, current);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
private addField(path: string[], value: unknown, row: number): void {
|
|
||||||
if (this.providedSchema !== undefined) {
|
|
||||||
this.addSchemaField(this.providedSchema, path, row);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
const evidence =
|
|
||||||
this.inferType(value, path) ?? DeferredTypeEvidence.from(value, row);
|
|
||||||
if (evidence === undefined) {
|
|
||||||
throw typeInferenceError(path, row);
|
|
||||||
}
|
|
||||||
|
|
||||||
const conflict = this.fields.set(
|
|
||||||
path,
|
|
||||||
evidence,
|
|
||||||
(existing) =>
|
|
||||||
existing instanceof DeferredTypeEvidence && existing.isOnlyNulls(),
|
|
||||||
);
|
|
||||||
if (conflict !== undefined) {
|
|
||||||
throw branchConflictError(conflict, row, "Struct");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
private addSchemaField(schema: Schema, path: string[], row: number): void {
|
|
||||||
const field = fieldAtPath(schema, path);
|
|
||||||
if (field === undefined) {
|
|
||||||
throw new Error(
|
|
||||||
`Found field not in schema: ${path.join(".")} at row ${row}`,
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
const conflict = this.fields.set(path, field.type);
|
|
||||||
if (conflict !== undefined) {
|
|
||||||
throw branchConflictError(conflict, row, "Struct");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
private updateInferredField(
|
|
||||||
path: string[],
|
|
||||||
value: unknown,
|
|
||||||
row: number,
|
|
||||||
current: FieldNode,
|
|
||||||
): void {
|
|
||||||
const newType = this.inferType(value, path);
|
|
||||||
const deferred = DeferredTypeEvidence.from(value, row);
|
|
||||||
|
|
||||||
if (current instanceof FieldTree) {
|
|
||||||
if (deferred?.isOnlyNulls()) {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
throw schemaInferenceError(
|
|
||||||
path,
|
|
||||||
row,
|
|
||||||
"Struct",
|
|
||||||
describeEvidence(newType ?? deferred),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
if (current instanceof DeferredTypeEvidence) {
|
|
||||||
this.resolveDeferredField(path, row, current, newType, deferred);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (newType !== undefined) {
|
|
||||||
if (!inferredTypesEqual(current, newType)) {
|
|
||||||
throw schemaInferenceError(
|
|
||||||
path,
|
|
||||||
row,
|
|
||||||
describeEvidence(current),
|
|
||||||
describeEvidence(newType),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (deferred === undefined || !deferred.matches(current)) {
|
|
||||||
throw schemaInferenceError(
|
|
||||||
path,
|
|
||||||
row,
|
|
||||||
describeEvidence(current),
|
|
||||||
describeEvidence(deferred),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
private resolveDeferredField(
|
|
||||||
path: string[],
|
|
||||||
row: number,
|
|
||||||
current: DeferredTypeEvidence,
|
|
||||||
newType: DataType | undefined,
|
|
||||||
deferred: DeferredTypeEvidence | undefined,
|
|
||||||
): void {
|
|
||||||
if (newType !== undefined) {
|
|
||||||
if (!current.matches(newType)) {
|
|
||||||
throw schemaInferenceError(
|
|
||||||
path,
|
|
||||||
row,
|
|
||||||
current.describe(),
|
|
||||||
describeEvidence(newType),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
this.fields.set(path, newType);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (deferred !== undefined) {
|
|
||||||
this.fields.set(path, current.merge(deferred));
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
throw schemaInferenceError(
|
|
||||||
path,
|
|
||||||
row,
|
|
||||||
current.describe(),
|
|
||||||
describeEvidence(newType),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
private inferType(value: unknown, path: string[]): DataType | undefined {
|
|
||||||
if (typeof value === "bigint") {
|
|
||||||
return new Int64();
|
|
||||||
}
|
|
||||||
if (typeof value === "number") {
|
|
||||||
return new Float64();
|
|
||||||
}
|
|
||||||
if (typeof value === "string") {
|
|
||||||
return this.options.dictionaryEncodeStrings
|
|
||||||
? new Dictionary(new Utf8(), new Int32())
|
|
||||||
: new Utf8();
|
|
||||||
}
|
|
||||||
if (typeof value === "boolean") {
|
|
||||||
return new Bool();
|
|
||||||
}
|
|
||||||
if (value instanceof Buffer) {
|
|
||||||
return new Binary();
|
|
||||||
}
|
|
||||||
if (ArrayBuffer.isView(value) && !(value instanceof DataView)) {
|
|
||||||
const typedArray = typedArrayToArrowType(value);
|
|
||||||
return typedArray === undefined
|
|
||||||
? undefined
|
|
||||||
: new FixedSizeList(
|
|
||||||
typedArray.length,
|
|
||||||
new Field("item", typedArray.elementType, true),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
if (!Array.isArray(value) || value.length === 0) {
|
|
||||||
return undefined;
|
|
||||||
}
|
|
||||||
|
|
||||||
const configuredVector =
|
|
||||||
path.length === 1 ? this.options.vectorColumns[path[0]] : undefined;
|
|
||||||
if (configuredVector !== undefined) {
|
|
||||||
return new FixedSizeList(
|
|
||||||
value.length,
|
|
||||||
new Field("item", sanitizeType(configuredVector.type), true),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
const itemType = this.inferArrayItemType(value, path);
|
|
||||||
if (itemType === undefined) {
|
|
||||||
return undefined;
|
|
||||||
}
|
|
||||||
|
|
||||||
return nameSuggestsVectorColumn(path[path.length - 1])
|
|
||||||
? new FixedSizeList(value.length, new Field("item", new Float32(), true))
|
|
||||||
: new List(new Field("item", itemType, true));
|
|
||||||
}
|
|
||||||
|
|
||||||
private inferArrayItemType(
|
|
||||||
values: unknown[],
|
|
||||||
path: string[],
|
|
||||||
): DataType | undefined {
|
|
||||||
let itemType: DataType | undefined;
|
|
||||||
const deferredItems: unknown[] = [];
|
|
||||||
|
|
||||||
for (const value of values) {
|
|
||||||
const candidate = this.inferType(value, path);
|
|
||||||
if (candidate === undefined) {
|
|
||||||
if (!isDeferredValue(value)) {
|
|
||||||
return undefined;
|
|
||||||
}
|
|
||||||
deferredItems.push(value);
|
|
||||||
} else if (itemType === undefined) {
|
|
||||||
itemType = candidate;
|
|
||||||
} else if (!inferredTypesEqual(itemType, candidate)) {
|
|
||||||
return undefined;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (itemType === undefined) {
|
|
||||||
return undefined;
|
|
||||||
}
|
|
||||||
return deferredItems.every((value) =>
|
|
||||||
deferredValueMatchesType(value, itemType),
|
|
||||||
)
|
|
||||||
? itemType
|
|
||||||
: undefined;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Nulls and empty/all-null lists that do not determine a type by themselves. */
|
|
||||||
class DeferredTypeEvidence {
|
|
||||||
private constructor(
|
|
||||||
private readonly values: Array<{ value: unknown; row: number }>,
|
|
||||||
) {}
|
|
||||||
|
|
||||||
static from(value: unknown, row: number): DeferredTypeEvidence | undefined {
|
|
||||||
return isDeferredValue(value)
|
|
||||||
? new DeferredTypeEvidence([{ value, row }])
|
|
||||||
: undefined;
|
|
||||||
}
|
|
||||||
|
|
||||||
isOnlyNulls(): boolean {
|
|
||||||
return this.values.every(({ value }) => value == null);
|
|
||||||
}
|
|
||||||
|
|
||||||
matches(type: DataType): boolean {
|
|
||||||
return this.values.every(({ value }) =>
|
|
||||||
deferredValueMatchesType(value, type),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
merge(other: DeferredTypeEvidence): DeferredTypeEvidence {
|
|
||||||
return new DeferredTypeEvidence([...this.values, ...other.values]);
|
|
||||||
}
|
|
||||||
|
|
||||||
describe(): string {
|
|
||||||
const list = this.values.find(({ value }) => Array.isArray(value));
|
|
||||||
return list === undefined
|
|
||||||
? "null"
|
|
||||||
: `List[${(list.value as unknown[]).length}]`;
|
|
||||||
}
|
|
||||||
|
|
||||||
firstRow(): number {
|
|
||||||
return this.values[0].row;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
type FieldNode = DataType | DeferredTypeEvidence | FieldTree;
|
|
||||||
type LeafNode = Exclude<FieldNode, FieldTree>;
|
|
||||||
type FieldConflict = { path: string[]; value: FieldNode };
|
|
||||||
|
|
||||||
/** Nested field state, kept separate from Arrow's eventual Struct types. */
|
|
||||||
class FieldTree {
|
|
||||||
private readonly children = new Map<string, FieldNode>();
|
|
||||||
|
|
||||||
get(path: string[]): FieldNode | undefined {
|
|
||||||
let current: FieldNode = this;
|
|
||||||
for (const part of path) {
|
|
||||||
if (!(current instanceof FieldTree)) {
|
|
||||||
return undefined;
|
|
||||||
}
|
|
||||||
const child = current.children.get(part);
|
|
||||||
if (child === undefined) {
|
|
||||||
return undefined;
|
|
||||||
}
|
|
||||||
current = child;
|
|
||||||
}
|
|
||||||
return current;
|
|
||||||
}
|
|
||||||
|
|
||||||
set(
|
|
||||||
path: string[],
|
|
||||||
value: LeafNode,
|
|
||||||
canReplaceLeaf: (value: LeafNode) => boolean = () => false,
|
|
||||||
): FieldConflict | undefined {
|
|
||||||
let branch: FieldTree = this;
|
|
||||||
for (const [index, part] of path.slice(0, -1).entries()) {
|
|
||||||
const child = branch.children.get(part);
|
|
||||||
if (child === undefined || (isLeaf(child) && canReplaceLeaf(child))) {
|
|
||||||
const nextBranch = new FieldTree();
|
|
||||||
branch.children.set(part, nextBranch);
|
|
||||||
branch = nextBranch;
|
|
||||||
} else if (child instanceof FieldTree) {
|
|
||||||
branch = child;
|
|
||||||
} else {
|
|
||||||
return { path: path.slice(0, index + 1), value: child };
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
const name = path[path.length - 1];
|
|
||||||
const current = branch.children.get(name);
|
|
||||||
if (current instanceof FieldTree) {
|
|
||||||
return { path, value: current };
|
|
||||||
}
|
|
||||||
branch.children.set(name, value);
|
|
||||||
return undefined;
|
|
||||||
}
|
|
||||||
|
|
||||||
entries(): IterableIterator<[string, FieldNode]> {
|
|
||||||
return this.children.entries();
|
|
||||||
}
|
|
||||||
|
|
||||||
has(name: string): boolean {
|
|
||||||
return this.children.has(name);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
function isLeaf(value: FieldNode): value is LeafNode {
|
|
||||||
return !(value instanceof FieldTree);
|
|
||||||
}
|
|
||||||
|
|
||||||
function fieldsFromTree(tree: FieldTree, path: string[] = []): Field[] {
|
|
||||||
const fields: Field[] = [];
|
|
||||||
for (const [name, value] of tree.entries()) {
|
|
||||||
if (value instanceof FieldTree) {
|
|
||||||
fields.push(
|
|
||||||
new Field(
|
|
||||||
name,
|
|
||||||
new Struct(fieldsFromTree(value, [...path, name])),
|
|
||||||
true,
|
|
||||||
),
|
|
||||||
);
|
|
||||||
} else if (value instanceof DeferredTypeEvidence) {
|
|
||||||
throw typeInferenceError([...path, name], value.firstRow());
|
|
||||||
} else {
|
|
||||||
fields.push(new Field(name, value, true));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return fields;
|
|
||||||
}
|
|
||||||
|
|
||||||
function matchingFields(fields: Field[], tree: FieldTree): Field[] {
|
|
||||||
const matches: Field[] = [];
|
|
||||||
for (const field of fields) {
|
|
||||||
if (!tree.has(field.name)) {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
const value = tree.get([field.name]);
|
|
||||||
if (value instanceof FieldTree) {
|
|
||||||
const struct = field.type as Struct;
|
|
||||||
matches.push(
|
|
||||||
new Field(
|
|
||||||
field.name,
|
|
||||||
new Struct(matchingFields(struct.children, value)),
|
|
||||||
field.nullable,
|
|
||||||
field.metadata,
|
|
||||||
),
|
|
||||||
);
|
|
||||||
} else {
|
|
||||||
matches.push(field);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return matches;
|
|
||||||
}
|
|
||||||
|
|
||||||
function* recordPathsAndValues(
|
|
||||||
record: Record<string, unknown>,
|
|
||||||
path: string[] = [],
|
|
||||||
): Generator<[string[], unknown]> {
|
|
||||||
for (const [name, value] of Object.entries(record)) {
|
|
||||||
if (isRecord(value)) {
|
|
||||||
yield* recordPathsAndValues(value, [...path, name]);
|
|
||||||
} else if (value !== undefined) {
|
|
||||||
yield [[...path, name], value];
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
||||||
return (
|
|
||||||
typeof value === "object" &&
|
|
||||||
value !== null &&
|
|
||||||
!Array.isArray(value) &&
|
|
||||||
!(value instanceof RegExp) &&
|
|
||||||
!(value instanceof Date) &&
|
|
||||||
!(value instanceof Set) &&
|
|
||||||
!(value instanceof Map) &&
|
|
||||||
!(value instanceof Buffer) &&
|
|
||||||
!ArrayBuffer.isView(value)
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
function fieldAtPath(schema: Schema, path: string[]): Field | undefined {
|
|
||||||
let fields = schema.fields;
|
|
||||||
let field: Field | undefined;
|
|
||||||
for (const [index, name] of path.entries()) {
|
|
||||||
field = fields.find((candidate) => candidate.name === name);
|
|
||||||
if (field === undefined || index === path.length - 1) {
|
|
||||||
return field;
|
|
||||||
}
|
|
||||||
if (!DataType.isStruct(field.type)) {
|
|
||||||
return undefined;
|
|
||||||
}
|
|
||||||
fields = field.type.children;
|
|
||||||
}
|
|
||||||
return field;
|
|
||||||
}
|
|
||||||
|
|
||||||
function isDeferredValue(value: unknown): boolean {
|
|
||||||
return (
|
|
||||||
value == null || (Array.isArray(value) && value.every(isDeferredValue))
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
function deferredValueMatchesType(value: unknown, type: DataType): boolean {
|
|
||||||
if (value == null) {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
if (!Array.isArray(value)) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
if (DataType.isList(type)) {
|
|
||||||
return value.every((item) =>
|
|
||||||
deferredValueMatchesType(item, type.valueType),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
if (DataType.isFixedSizeList(type)) {
|
|
||||||
return (
|
|
||||||
value.length === type.listSize &&
|
|
||||||
value.every((item) => deferredValueMatchesType(item, type.valueType))
|
|
||||||
);
|
|
||||||
}
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
function inferredTypesEqual(current: DataType, candidate: DataType): boolean {
|
|
||||||
if (DataType.isDictionary(current)) {
|
|
||||||
return (
|
|
||||||
DataType.isDictionary(candidate) &&
|
|
||||||
current.isOrdered === candidate.isOrdered &&
|
|
||||||
inferredTypesEqual(current.indices, candidate.indices) &&
|
|
||||||
inferredTypesEqual(current.dictionary, candidate.dictionary)
|
|
||||||
);
|
|
||||||
}
|
|
||||||
if (DataType.isList(current)) {
|
|
||||||
return (
|
|
||||||
DataType.isList(candidate) &&
|
|
||||||
current.valueField.name === candidate.valueField.name &&
|
|
||||||
current.valueField.nullable === candidate.valueField.nullable &&
|
|
||||||
inferredTypesEqual(current.valueType, candidate.valueType)
|
|
||||||
);
|
|
||||||
}
|
|
||||||
if (DataType.isFixedSizeList(current)) {
|
|
||||||
return (
|
|
||||||
DataType.isFixedSizeList(candidate) &&
|
|
||||||
current.listSize === candidate.listSize &&
|
|
||||||
current.valueField.name === candidate.valueField.name &&
|
|
||||||
current.valueField.nullable === candidate.valueField.nullable &&
|
|
||||||
inferredTypesEqual(current.valueType, candidate.valueType)
|
|
||||||
);
|
|
||||||
}
|
|
||||||
return arrowUtil.compareTypes(current, candidate);
|
|
||||||
}
|
|
||||||
|
|
||||||
function describeEvidence(
|
|
||||||
evidence: DataType | DeferredTypeEvidence | undefined,
|
|
||||||
): string {
|
|
||||||
if (evidence === undefined) {
|
|
||||||
return "an unsupported value";
|
|
||||||
}
|
|
||||||
return evidence instanceof DeferredTypeEvidence
|
|
||||||
? evidence.describe()
|
|
||||||
: evidence.toString();
|
|
||||||
}
|
|
||||||
|
|
||||||
function branchConflictError(
|
|
||||||
conflict: FieldConflict,
|
|
||||||
row: number,
|
|
||||||
candidate: string,
|
|
||||||
): Error {
|
|
||||||
return schemaInferenceError(
|
|
||||||
conflict.path,
|
|
||||||
row,
|
|
||||||
conflict.value instanceof FieldTree
|
|
||||||
? "Struct"
|
|
||||||
: describeEvidence(conflict.value),
|
|
||||||
candidate,
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
function schemaInferenceError(
|
|
||||||
path: string[],
|
|
||||||
row: number,
|
|
||||||
currentType: string,
|
|
||||||
newType: string,
|
|
||||||
): Error {
|
|
||||||
return new Error(
|
|
||||||
`Failed to infer schema for data. Previously inferred type ${currentType} ` +
|
|
||||||
`but found ${newType} for field ${path.join(".")} at row ${row}. ` +
|
|
||||||
"Consider providing an explicit schema.",
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
function typeInferenceError(path: string[], row: number): Error {
|
|
||||||
return new Error(
|
|
||||||
`Failed to infer data type for field ${path.join(".")} at row ${row}. ` +
|
|
||||||
"Consider providing an explicit schema.",
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
function nameSuggestsVectorColumn(name: string): boolean {
|
|
||||||
const normalized = name.toLowerCase();
|
|
||||||
return normalized.includes("vector") || normalized.includes("embedding");
|
|
||||||
}
|
|
||||||
+40
-285
@@ -30,12 +30,8 @@ import {
|
|||||||
DropColumnsResult,
|
DropColumnsResult,
|
||||||
IndexConfig,
|
IndexConfig,
|
||||||
IndexStatistics,
|
IndexStatistics,
|
||||||
Job,
|
|
||||||
LsmStats,
|
|
||||||
Branches as NativeBranches,
|
Branches as NativeBranches,
|
||||||
OptimizeStats,
|
OptimizeStats,
|
||||||
RefreshColumnResult,
|
|
||||||
RefreshMaterializedViewResult,
|
|
||||||
TableStatistics,
|
TableStatistics,
|
||||||
Tags,
|
Tags,
|
||||||
UpdateFieldMetadataResult,
|
UpdateFieldMetadataResult,
|
||||||
@@ -43,23 +39,15 @@ import {
|
|||||||
Table as _NativeTable,
|
Table as _NativeTable,
|
||||||
} from "./native";
|
} from "./native";
|
||||||
import {
|
import {
|
||||||
AutoQuery,
|
|
||||||
FullTextQuery,
|
FullTextQuery,
|
||||||
Query,
|
Query,
|
||||||
TakeQuery,
|
TakeQuery,
|
||||||
VectorQuery,
|
VectorQuery,
|
||||||
createAutoQuery,
|
|
||||||
instanceOfFullTextQuery,
|
instanceOfFullTextQuery,
|
||||||
} from "./query";
|
} from "./query";
|
||||||
import { sanitizeType } from "./sanitize";
|
import { sanitizeType } from "./sanitize";
|
||||||
import { IntoSql, toSQL } from "./util";
|
import { IntoSql, toSQL } from "./util";
|
||||||
export { IndexConfig } from "./native";
|
export { IndexConfig } from "./native";
|
||||||
export {
|
|
||||||
BucketStats,
|
|
||||||
GenerationStats,
|
|
||||||
LsmStats,
|
|
||||||
MemtableStats,
|
|
||||||
} from "./native";
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Progress snapshot for a write operation, delivered to the `progress`
|
* Progress snapshot for a write operation, delivered to the `progress`
|
||||||
@@ -208,11 +196,7 @@ export interface LsmWriteSpec {
|
|||||||
column?: string;
|
column?: string;
|
||||||
/** Bucket variant: the number of buckets, in `[1, 1024]`. */
|
/** Bucket variant: the number of buckets, in `[1, 1024]`. */
|
||||||
numBuckets?: number;
|
numBuckets?: number;
|
||||||
/**
|
/** Names of indexes the MemWAL should keep up to date during writes. */
|
||||||
* Indexes the MemWAL keeps up to date. Omit to maintain every supported
|
|
||||||
* index, resolved on install — a snapshot, so indexes created later are not
|
|
||||||
* maintained. Pass `[]` for none.
|
|
||||||
*/
|
|
||||||
maintainedIndexes?: string[];
|
maintainedIndexes?: string[];
|
||||||
/** Default `ShardWriter` configuration recorded in the MemWAL index. */
|
/** Default `ShardWriter` configuration recorded in the MemWAL index. */
|
||||||
writerConfigDefaults?: Record<string, string>;
|
writerConfigDefaults?: Record<string, string>;
|
||||||
@@ -374,17 +358,6 @@ export abstract class Table {
|
|||||||
options?: Partial<IndexOptions>,
|
options?: Partial<IndexOptions>,
|
||||||
): Promise<void>;
|
): Promise<void>;
|
||||||
|
|
||||||
/**
|
|
||||||
* Create an index, returning a handle to the indexing job.
|
|
||||||
*
|
|
||||||
* The job may already be complete when returned; callers must not assume
|
|
||||||
* the index exists until {@link Job.wait} resolves.
|
|
||||||
*/
|
|
||||||
abstract createIndexAsync(
|
|
||||||
column: string,
|
|
||||||
options?: Partial<IndexOptions>,
|
|
||||||
): Promise<Job>;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Drop an index from the table.
|
* Drop an index from the table.
|
||||||
*
|
*
|
||||||
@@ -525,7 +498,7 @@ export abstract class Table {
|
|||||||
query: string | IntoVector | MultiVector | FullTextQuery,
|
query: string | IntoVector | MultiVector | FullTextQuery,
|
||||||
queryType?: string,
|
queryType?: string,
|
||||||
ftsColumns?: string | string[],
|
ftsColumns?: string | string[],
|
||||||
): VectorQuery | Query | AutoQuery;
|
): VectorQuery | Query;
|
||||||
/**
|
/**
|
||||||
* Search the table with a given query vector.
|
* Search the table with a given query vector.
|
||||||
*
|
*
|
||||||
@@ -536,87 +509,18 @@ export abstract class Table {
|
|||||||
abstract vectorSearch(vector: IntoVector | MultiVector): VectorQuery;
|
abstract vectorSearch(vector: IntoVector | MultiVector): VectorQuery;
|
||||||
/**
|
/**
|
||||||
* Add new columns with defined values.
|
* Add new columns with defined values.
|
||||||
*
|
|
||||||
* The `{ computed }` form stores the expression rather than evaluating it
|
|
||||||
* now: the column is committed with no values, and rows get them from
|
|
||||||
* {@link Table#refreshColumn}. Declaring one therefore costs the same on a
|
|
||||||
* large table as on an empty one.
|
|
||||||
*
|
|
||||||
* A refresh does not revisit rows it has already filled, so mutating an
|
|
||||||
* input leaves the value computed at fill time; recomputing means dropping
|
|
||||||
* the column and declaring it again. While a declaration reads a column,
|
|
||||||
* that column cannot be renamed, retyped or dropped.
|
|
||||||
*
|
|
||||||
* On LanceDB Cloud and Enterprise the expression is planned by the
|
|
||||||
* server, and the refresh runs as a server job -- see
|
|
||||||
* {@link Table#refreshColumnAsync}.
|
|
||||||
* @param {AddColumnsSql[] | Field | Field[] | Schema} newColumnTransforms Either:
|
* @param {AddColumnsSql[] | Field | Field[] | Schema} newColumnTransforms Either:
|
||||||
* - An array of objects with column names and SQL expressions to calculate values
|
* - An array of objects with column names and SQL expressions to calculate values
|
||||||
* - A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
* - A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
||||||
* - An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
* - An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
||||||
* - An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
* - An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
||||||
* - `{ computed }`, declaring columns defined by a SQL expression whose type and inputs are derived from it
|
|
||||||
* @returns {Promise<AddColumnsResult>} A promise that resolves to an object
|
* @returns {Promise<AddColumnsResult>} A promise that resolves to an object
|
||||||
* containing the new version number of the table after adding the columns.
|
* containing the new version number of the table after adding the columns.
|
||||||
* @example
|
|
||||||
* ```ts
|
|
||||||
* await table.addColumns({ computed: [{ name: "doubled", valueSql: "x * 2" }] });
|
|
||||||
* const { rowsFilled } = await table.refreshColumn("doubled");
|
|
||||||
* ```
|
|
||||||
*/
|
*/
|
||||||
abstract addColumns(
|
abstract addColumns(
|
||||||
newColumnTransforms:
|
newColumnTransforms: AddColumnsSql[] | Field | Field[] | Schema,
|
||||||
| AddColumnsSql[]
|
|
||||||
| Field
|
|
||||||
| Field[]
|
|
||||||
| Schema
|
|
||||||
| { computed: AddColumnsSql[] },
|
|
||||||
): Promise<AddColumnsResult>;
|
): Promise<AddColumnsResult>;
|
||||||
|
|
||||||
/**
|
|
||||||
* Fill the rows of a computed column that hold no value yet.
|
|
||||||
*
|
|
||||||
* Rows appended since the last refresh are filled by the next one; rows
|
|
||||||
* already filled are left as they are, so the call is idempotent and does
|
|
||||||
* not observe a mutated input. Local tables only: a remote refresh runs
|
|
||||||
* as a server job, through {@link Table#refreshColumnAsync}.
|
|
||||||
* @param {string} column The name of the computed column to fill.
|
|
||||||
* @returns {Promise<RefreshColumnResult>} A promise that resolves to the
|
|
||||||
* number of rows filled and the new version number of the table.
|
|
||||||
*/
|
|
||||||
abstract refreshColumn(column: string): Promise<RefreshColumnResult>;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Like {@link Table#refreshColumn}, but returns a handle to the refresh
|
|
||||||
* job instead of blocking until it completes.
|
|
||||||
*
|
|
||||||
* The job may already be complete when returned; callers must not assume
|
|
||||||
* the column is filled until {@link Job.wait} resolves. Invalid input --
|
|
||||||
* an unknown column, or one that is not computed -- rejects here rather
|
|
||||||
* than failing the job. On local tables the job runs in-process; on
|
|
||||||
* LanceDB Cloud and Enterprise it is the server's backfill job.
|
|
||||||
* @param {string} column The name of the computed column to fill.
|
|
||||||
* @example
|
|
||||||
* ```ts
|
|
||||||
* const job = await table.refreshColumnAsync("doubled");
|
|
||||||
* await job.wait();
|
|
||||||
* console.log(await job.status()); // "finished"
|
|
||||||
* ```
|
|
||||||
*/
|
|
||||||
abstract refreshColumnAsync(column: string): Promise<Job>;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Recompute this table's contents from its materialized-view definition.
|
|
||||||
*
|
|
||||||
* Plumbing for {@link MaterializedView.refresh}, which is the way to call
|
|
||||||
* it: rejects tables that carry no view definition. Local tables only.
|
|
||||||
* @ignore
|
|
||||||
*/
|
|
||||||
abstract refreshMaterializedView(
|
|
||||||
full?: boolean,
|
|
||||||
sourceVersion?: number,
|
|
||||||
): Promise<RefreshMaterializedViewResult>;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Alter the name or nullability of columns.
|
* Alter the name or nullability of columns.
|
||||||
* @param {ColumnAlteration[]} columnAlterations One or more alterations to
|
* @param {ColumnAlteration[]} columnAlterations One or more alterations to
|
||||||
@@ -630,18 +534,6 @@ export abstract class Table {
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* Update per-field (column) metadata.
|
* Update per-field (column) metadata.
|
||||||
*
|
|
||||||
* The following keys are treated specially, by convention, and should be
|
|
||||||
* used when appropriate:
|
|
||||||
*
|
|
||||||
* - `lancedb:description`: for a human-readable description of a field.
|
|
||||||
* - `lancedb:tag:<name>`: for a user-defined key-value tag, where the suffix
|
|
||||||
* names the tag category; e.g. `lancedb:tag:model: "clip"`.
|
|
||||||
* - `lancedb:logical-column`: for a column grouping; e.g. `feature_v1` and
|
|
||||||
* `feature_v2` might be in the same logical column.
|
|
||||||
* - `lancedb:status`: for status options (`production`, `candidate`,
|
|
||||||
* `deprecated`, `archived`) to designate the current life cycle state of
|
|
||||||
* this column.
|
|
||||||
* @param {FieldMetadataUpdate[]} updates One or more per-field updates. Each
|
* @param {FieldMetadataUpdate[]} updates One or more per-field updates. Each
|
||||||
* update's metadata is merged into the field's existing metadata by default;
|
* update's metadata is merged into the field's existing metadata by default;
|
||||||
* a value of `null` deletes that key, and `replace: true` swaps the whole map.
|
* a value of `null` deletes that key, and `replace: true` swaps the whole map.
|
||||||
@@ -691,11 +583,6 @@ export abstract class Table {
|
|||||||
* All variants require the table to have an unenforced primary key
|
* All variants require the table to have an unenforced primary key
|
||||||
* ({@link Table#setUnenforcedPrimaryKey}); bucket sharding additionally
|
* ({@link Table#setUnenforcedPrimaryKey}); bucket sharding additionally
|
||||||
* requires it to be the single column being bucketed.
|
* requires it to be the single column being bucketed.
|
||||||
*
|
|
||||||
* Omitting `maintainedIndexes` maintains every index on the table, resolved
|
|
||||||
* here, failing if one cannot be maintained — name them to install anyway.
|
|
||||||
* Naming them pins an exact set, and a still-building index is rejected
|
|
||||||
* rather than quietly omitted.
|
|
||||||
* @param {LsmWriteSpec} spec The sharding spec to install.
|
* @param {LsmWriteSpec} spec The sharding spec to install.
|
||||||
* @returns {Promise<void>}
|
* @returns {Promise<void>}
|
||||||
* @example
|
* @example
|
||||||
@@ -723,10 +610,9 @@ export abstract class Table {
|
|||||||
*
|
*
|
||||||
* Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
|
* Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
|
||||||
* spec has been set, or it was removed with {@link Table#unsetLsmWriteSpec}).
|
* spec has been set, or it was removed with {@link Table#unsetLsmWriteSpec}).
|
||||||
* The returned spec mirrors what was passed to
|
* The returned spec — including its `maintainedIndexes` and
|
||||||
* {@link Table#setLsmWriteSpec}, except that `maintainedIndexes` always
|
* `writerConfigDefaults` — mirrors what was passed to
|
||||||
* reports the concrete list resolved when the spec was set — `undefined`
|
* {@link Table#setLsmWriteSpec}.
|
||||||
* never round-trips.
|
|
||||||
* @returns {Promise<LsmWriteSpec | undefined>}
|
* @returns {Promise<LsmWriteSpec | undefined>}
|
||||||
*/
|
*/
|
||||||
abstract getLsmWriteSpec(): Promise<LsmWriteSpec | undefined>;
|
abstract getLsmWriteSpec(): Promise<LsmWriteSpec | undefined>;
|
||||||
@@ -740,59 +626,6 @@ export abstract class Table {
|
|||||||
* @returns {Promise<void>}
|
* @returns {Promise<void>}
|
||||||
*/
|
*/
|
||||||
abstract closeLsmWriters(): Promise<void>;
|
abstract closeLsmWriters(): Promise<void>;
|
||||||
/**
|
|
||||||
* Seal every bucket's active memtable into a new L0 generation.
|
|
||||||
*
|
|
||||||
* Returns once the seal is committed. Sealing an empty memtable is a no-op,
|
|
||||||
* so this is safe to call repeatedly.
|
|
||||||
* @returns {Promise<void>}
|
|
||||||
*/
|
|
||||||
abstract flushLsm(): Promise<void>;
|
|
||||||
/**
|
|
||||||
* Trigger a background L0 → base compaction pass per bucket.
|
|
||||||
*
|
|
||||||
* Returns once the passes are *dispatched*, not once they finish — watch
|
|
||||||
* {@link Table#getLsmStats} for progress, or use
|
|
||||||
* {@link Table#checkpointLsm} to wait for convergence.
|
|
||||||
* @returns {Promise<void>}
|
|
||||||
*/
|
|
||||||
abstract compactLsm(): Promise<void>;
|
|
||||||
/**
|
|
||||||
* Converge this table's LSM write path into its base table.
|
|
||||||
*
|
|
||||||
* Seals once, then triggers compaction and polls until the L0 that existed
|
|
||||||
* at the start is gone. The target set is fixed at the start, so
|
|
||||||
* generations created *during* the checkpoint are ignored — that is what
|
|
||||||
* lets it terminate under write load, and what makes it best-effort: it
|
|
||||||
* converges the fresh tier as of some instant. Idempotent, abandonable at
|
|
||||||
* any point, and safe to run on a cadence.
|
|
||||||
*
|
|
||||||
* There is no liveness bound — the compactor pool is shared across tables,
|
|
||||||
* so a checkpoint queued behind unrelated work looks exactly like one that
|
|
||||||
* is merging. The caller owns the deadline.
|
|
||||||
* @returns {Promise<void>}
|
|
||||||
* @example
|
|
||||||
* ```ts
|
|
||||||
* const before = await table.getLsmStats();
|
|
||||||
* await table.checkpointLsm();
|
|
||||||
* const after = await table.getLsmStats();
|
|
||||||
* ```
|
|
||||||
*/
|
|
||||||
abstract checkpointLsm(): Promise<void>;
|
|
||||||
/**
|
|
||||||
* Read live per-bucket LSM state.
|
|
||||||
*
|
|
||||||
* Answers "how far behind is my fresh tier", "which bucket is hot", and
|
|
||||||
* "why is my fresh-tier vector search brute-force". Mutates no table state.
|
|
||||||
*
|
|
||||||
* Resolves to `undefined` only when the LSM write path is not enabled.
|
|
||||||
* @param {boolean} includeGenerationRows Also count rows per L0 generation.
|
|
||||||
* Off by default because each count opens an uncached Lance dataset.
|
|
||||||
* @returns {Promise<LsmStats | undefined>}
|
|
||||||
*/
|
|
||||||
abstract getLsmStats(
|
|
||||||
includeGenerationRows?: boolean,
|
|
||||||
): Promise<LsmStats | undefined>;
|
|
||||||
/** Retrieve the version of the table */
|
/** Retrieve the version of the table */
|
||||||
|
|
||||||
abstract version(): Promise<number>;
|
abstract version(): Promise<number>;
|
||||||
@@ -989,11 +822,10 @@ export class LocalTable extends Table {
|
|||||||
return this.inner.display();
|
return this.inner.display();
|
||||||
}
|
}
|
||||||
|
|
||||||
private async getEmbeddingFunctions(
|
private async getEmbeddingFunctions(): Promise<
|
||||||
inner: _NativeTable = this.inner,
|
Map<string, EmbeddingFunctionConfig>
|
||||||
): Promise<Map<string, EmbeddingFunctionConfig>> {
|
> {
|
||||||
const schemaBuf = await inner.schema();
|
const schema = await this.schema();
|
||||||
const schema = tableFromIPC(schemaBuf).schema;
|
|
||||||
const registry = getRegistry();
|
const registry = getRegistry();
|
||||||
return registry.parseFunctions(schema.metadata);
|
return registry.parseFunctions(schema.metadata);
|
||||||
}
|
}
|
||||||
@@ -1108,22 +940,6 @@ export class LocalTable extends Table {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
async createIndexAsync(
|
|
||||||
column: string,
|
|
||||||
options?: Partial<IndexOptions>,
|
|
||||||
): Promise<Job> {
|
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: skip
|
|
||||||
const nativeIndex = (options?.config as any)?.inner;
|
|
||||||
return await this.inner.createIndexAsync(
|
|
||||||
nativeIndex,
|
|
||||||
column,
|
|
||||||
options?.replace,
|
|
||||||
options?.waitTimeoutSeconds,
|
|
||||||
options?.name,
|
|
||||||
options?.train,
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
async dropIndex(name: string): Promise<void> {
|
async dropIndex(name: string): Promise<void> {
|
||||||
await this.inner.dropIndex(name);
|
await this.inner.dropIndex(name);
|
||||||
}
|
}
|
||||||
@@ -1175,7 +991,7 @@ export class LocalTable extends Table {
|
|||||||
query: string | IntoVector | MultiVector | FullTextQuery,
|
query: string | IntoVector | MultiVector | FullTextQuery,
|
||||||
queryType: string = "auto",
|
queryType: string = "auto",
|
||||||
ftsColumns?: string | string[],
|
ftsColumns?: string | string[],
|
||||||
): VectorQuery | Query | AutoQuery {
|
): VectorQuery | Query {
|
||||||
if (typeof query !== "string" && !instanceOfFullTextQuery(query)) {
|
if (typeof query !== "string" && !instanceOfFullTextQuery(query)) {
|
||||||
if (queryType === "fts") {
|
if (queryType === "fts") {
|
||||||
throw new Error("Cannot perform full text search on a vector query");
|
throw new Error("Cannot perform full text search on a vector query");
|
||||||
@@ -1190,28 +1006,14 @@ export class LocalTable extends Table {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
if (queryType === "auto") {
|
// The query type is auto or vector
|
||||||
if (instanceOfFullTextQuery(query)) {
|
// fall back to full text search if no embedding functions are defined and the query is a string
|
||||||
return this.query().fullTextSearch(query, {
|
if (
|
||||||
columns: ftsColumns,
|
queryType === "auto" &&
|
||||||
});
|
(getRegistry().length() === 0 || instanceOfFullTextQuery(query))
|
||||||
}
|
) {
|
||||||
|
return this.query().fullTextSearch(query, {
|
||||||
const columns =
|
columns: ftsColumns,
|
||||||
typeof ftsColumns === "string" ? [ftsColumns] : (ftsColumns ?? null);
|
|
||||||
return createAutoQuery(this.inner, query, columns, async (metadata) => {
|
|
||||||
const functions = await getRegistry().parseFunctions(
|
|
||||||
new Map([["embedding_functions", metadata]]),
|
|
||||||
);
|
|
||||||
// TODO: Support multiple embedding functions
|
|
||||||
const embeddingFunc: EmbeddingFunctionConfig | undefined = functions
|
|
||||||
.values()
|
|
||||||
.next().value;
|
|
||||||
// The route only calls this callback when embedding metadata exists.
|
|
||||||
// parseFunctions either yields a provider or reports malformed metadata.
|
|
||||||
if (!embeddingFunc)
|
|
||||||
throw new Error("Invalid embedding function metadata");
|
|
||||||
return await embeddingFunc.function.computeQueryEmbeddings(query);
|
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1248,22 +1050,8 @@ export class LocalTable extends Table {
|
|||||||
// TODO: Support BatchUDF
|
// TODO: Support BatchUDF
|
||||||
|
|
||||||
async addColumns(
|
async addColumns(
|
||||||
newColumnTransforms:
|
newColumnTransforms: AddColumnsSql[] | Field | Field[] | Schema,
|
||||||
| AddColumnsSql[]
|
|
||||||
| Field
|
|
||||||
| Field[]
|
|
||||||
| Schema
|
|
||||||
| { computed: AddColumnsSql[] },
|
|
||||||
): Promise<AddColumnsResult> {
|
): Promise<AddColumnsResult> {
|
||||||
// Columns defined by an expression are declared, not materialized here.
|
|
||||||
if (
|
|
||||||
typeof newColumnTransforms === "object" &&
|
|
||||||
!Array.isArray(newColumnTransforms) &&
|
|
||||||
"computed" in newColumnTransforms
|
|
||||||
) {
|
|
||||||
return await this.inner.addComputedColumns(newColumnTransforms.computed);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Handle single Field -> convert to array of Fields
|
// Handle single Field -> convert to array of Fields
|
||||||
if (newColumnTransforms instanceof Field) {
|
if (newColumnTransforms instanceof Field) {
|
||||||
newColumnTransforms = [newColumnTransforms];
|
newColumnTransforms = [newColumnTransforms];
|
||||||
@@ -1298,21 +1086,6 @@ export class LocalTable extends Table {
|
|||||||
throw new Error("Invalid input type for addColumns");
|
throw new Error("Invalid input type for addColumns");
|
||||||
}
|
}
|
||||||
|
|
||||||
async refreshColumn(column: string): Promise<RefreshColumnResult> {
|
|
||||||
return await this.inner.refreshColumn(column);
|
|
||||||
}
|
|
||||||
|
|
||||||
async refreshColumnAsync(column: string): Promise<Job> {
|
|
||||||
return await this.inner.refreshColumnAsync(column);
|
|
||||||
}
|
|
||||||
|
|
||||||
async refreshMaterializedView(
|
|
||||||
full?: boolean,
|
|
||||||
sourceVersion?: number,
|
|
||||||
): Promise<RefreshMaterializedViewResult> {
|
|
||||||
return await this.inner.refreshMaterializedView(full, sourceVersion);
|
|
||||||
}
|
|
||||||
|
|
||||||
async alterColumns(
|
async alterColumns(
|
||||||
columnAlterations: ColumnAlteration[],
|
columnAlterations: ColumnAlteration[],
|
||||||
): Promise<AlterColumnsResult> {
|
): Promise<AlterColumnsResult> {
|
||||||
@@ -1375,24 +1148,6 @@ export class LocalTable extends Table {
|
|||||||
return await this.inner.closeLsmWriters();
|
return await this.inner.closeLsmWriters();
|
||||||
}
|
}
|
||||||
|
|
||||||
async flushLsm(): Promise<void> {
|
|
||||||
return await this.inner.flushLsm();
|
|
||||||
}
|
|
||||||
|
|
||||||
async compactLsm(): Promise<void> {
|
|
||||||
return await this.inner.compactLsm();
|
|
||||||
}
|
|
||||||
|
|
||||||
async checkpointLsm(): Promise<void> {
|
|
||||||
return await this.inner.checkpointLsm();
|
|
||||||
}
|
|
||||||
|
|
||||||
async getLsmStats(
|
|
||||||
includeGenerationRows: boolean = false,
|
|
||||||
): Promise<LsmStats | undefined> {
|
|
||||||
return (await this.inner.getLsmStats(includeGenerationRows)) ?? undefined;
|
|
||||||
}
|
|
||||||
|
|
||||||
async version(): Promise<number> {
|
async version(): Promise<number> {
|
||||||
return await this.inner.version();
|
return await this.inner.version();
|
||||||
}
|
}
|
||||||
@@ -1567,8 +1322,7 @@ export interface FieldMetadataUpdate {
|
|||||||
path: string;
|
path: string;
|
||||||
/**
|
/**
|
||||||
* Metadata key/value pairs. Merged into the field's existing metadata by
|
* Metadata key/value pairs. Merged into the field's existing metadata by
|
||||||
* default; a value of `null` deletes that key. See
|
* default; a value of `null` deletes that key.
|
||||||
* {@link Table.updateFieldMetadata} for the conventional `lancedb:*` keys.
|
|
||||||
*/
|
*/
|
||||||
metadata: Record<string, string | null>;
|
metadata: Record<string, string | null>;
|
||||||
/** If true, replace the field's entire metadata map instead of merging. */
|
/** If true, replace the field's entire metadata map instead of merging. */
|
||||||
@@ -1607,8 +1361,8 @@ export interface BranchRowCountSummary {
|
|||||||
deltaAvailable: boolean;
|
deltaAvailable: boolean;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** A reason why a cherry-pick cannot currently land. */
|
/** A reason why a branch cannot currently be merged. */
|
||||||
export interface CherryPickError {
|
export interface MergeBlocker {
|
||||||
code: string;
|
code: string;
|
||||||
message: string;
|
message: string;
|
||||||
}
|
}
|
||||||
@@ -1628,19 +1382,20 @@ export interface BranchDiff {
|
|||||||
changedColumns: BranchColumnChange[];
|
changedColumns: BranchColumnChange[];
|
||||||
addedIndexes: BranchIndexSummary[];
|
addedIndexes: BranchIndexSummary[];
|
||||||
removedIndexes: BranchIndexSummary[];
|
removedIndexes: BranchIndexSummary[];
|
||||||
errors: CherryPickError[];
|
mergeable: boolean;
|
||||||
|
mergeBlockers: MergeBlocker[];
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Changes that would be, or were, promoted by a cherry-pick. */
|
/** Changes that would be, or were, promoted by a branch merge. */
|
||||||
export interface CherryPickPreview {
|
export interface MergePreview {
|
||||||
promotedColumns: string[];
|
promotedColumns: string[];
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Result of previewing or attempting a cherry-pick. */
|
/** Result of previewing or attempting a branch merge. */
|
||||||
export interface CherryPickResult {
|
export interface MergeBranchResult {
|
||||||
status: "ready" | "failed" | "notImplemented" | "cherryPicked" | "unknown";
|
status: "ready" | "rejected" | "notImplemented" | "merged" | "unknown";
|
||||||
diff: BranchDiff;
|
diff: BranchDiff;
|
||||||
preview: CherryPickPreview;
|
preview: MergePreview;
|
||||||
mainVersionAfter?: number;
|
mainVersionAfter?: number;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1703,21 +1458,21 @@ export class Branches {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Cherry-pick a branch onto main.
|
* Merge a branch into main.
|
||||||
*
|
*
|
||||||
* Set `dryRun` to `true` to preview. A failed cherry-pick resolves
|
* Set `dryRun` to `true` to preview the merge. A rejected merge resolves
|
||||||
* with `status: "failed"` instead of throwing.
|
* with `status: "rejected"` instead of throwing.
|
||||||
*
|
*
|
||||||
* @param fromBranch Branch to cherry-pick from.
|
* @param fromBranch Branch to merge from.
|
||||||
* @param dryRun When true, only preview. Defaults to false.
|
* @param dryRun When true, only preview the merge. Defaults to false.
|
||||||
*/
|
*/
|
||||||
async cherryPick(
|
async merge(
|
||||||
fromBranch: string,
|
fromBranch: string,
|
||||||
dryRun: boolean = false,
|
dryRun: boolean = false,
|
||||||
): Promise<CherryPickResult> {
|
): Promise<MergeBranchResult> {
|
||||||
return (await this.#inner.cherryPick(
|
return (await this.#inner.merge(
|
||||||
fromBranch,
|
fromBranch,
|
||||||
dryRun,
|
dryRun,
|
||||||
)) as unknown as CherryPickResult;
|
)) as unknown as MergeBranchResult;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-darwin-arm64",
|
"name": "@lancedb/lancedb-darwin-arm64",
|
||||||
"version": "0.38.0-beta.11",
|
"version": "0.37.1-beta.0",
|
||||||
"os": ["darwin"],
|
"os": ["darwin"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.darwin-arm64.node",
|
"main": "lancedb.darwin-arm64.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
||||||
"version": "0.38.0-beta.11",
|
"version": "0.37.1-beta.0",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.linux-arm64-gnu.node",
|
"main": "lancedb.linux-arm64-gnu.node",
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user