Compare commits

..

6 Commits

Author SHA1 Message Date
Lance Release cbd5b4e288 Bump version: 0.36.0-beta.1 → 0.36.0 2026-07-28 21:31:43 +00:00
Lance Release 897c492531 Bump version: 0.36.0-beta.0 → 0.36.0-beta.1 2026-07-28 21:31:42 +00:00
Lance Release 4f0cf10a9b Bump version: 0.32.0-beta.2 → 0.33.0-beta.0 2026-07-24 19:33:56 +00:00
Lance Release f1626012df Bump version: 0.35.0-beta.2 → 0.36.0-beta.0 2026-07-24 19:33:04 +00:00
Will Jones de5ca64037 chore: pin pylance test dependency to 9.0.0
The tests extra pinned 9.0.0rc1 because 9.0.0 had not been released yet.
It now resolves from PyPI instead of the preview index.
2026-07-24 12:26:31 -07:00
Will Jones 9b42f0a8bc chore: update lance dependency to v9.0.0 2026-07-24 11:59:37 -07:00
185 changed files with 3012 additions and 22801 deletions
+1 -8
View File
@@ -1,5 +1,5 @@
[tool.bumpversion] [tool.bumpversion]
current_version = "0.38.0-beta.0" current_version = "0.33.0-beta.0"
parse = """(?x) parse = """(?x)
(?P<major>0|[1-9]\\d*)\\. (?P<major>0|[1-9]\\d*)\\.
(?P<minor>0|[1-9]\\d*)\\. (?P<minor>0|[1-9]\\d*)\\.
@@ -75,13 +75,6 @@ filename = "nodejs/Cargo.toml"
replace = "\nversion = \"{new_version}\"" replace = "\nversion = \"{new_version}\""
search = "\nversion = \"{current_version}\"" search = "\nversion = \"{current_version}\""
# The Python package takes its version from here (pyproject.toml declares
# `dynamic = ["version"]`, so maturin reads it out of the crate manifest).
[[tool.bumpversion.files]]
filename = "python/Cargo.toml"
replace = "\nversion = \"{new_version}\""
search = "\nversion = \"{current_version}\""
# Java documentation # Java documentation
[[tool.bumpversion.files]] [[tool.bumpversion.files]]
filename = "docs/src/java/java.md" filename = "docs/src/java/java.md"
@@ -27,31 +27,19 @@ runs:
# Extract failed job names # Extract failed job names
FAILED_JOBS=$(echo "$JOB_RESULTS" | jq -r 'to_entries | map(select(.value.result == "failure")) | map(.key) | join(", ")') FAILED_JOBS=$(echo "$JOB_RESULTS" | jq -r 'to_entries | map(select(.value.result == "failure")) | map(.key) | join(", ")')
TITLE="$WORKFLOW_NAME Failed ($FAILED_JOBS)" # Create issue with workflow name, failed jobs, and run URL
gh issue create \
# This action now also runs on nightly schedules, so a breakage that --title "$WORKFLOW_NAME Failed ($FAILED_JOBS)" \
# persists for a few days would otherwise file one issue per night. --body "The workflow **$WORKFLOW_NAME** failed during execution.
# Comment on the open report instead when one already exists.
EXISTING=$(gh issue list --state open --label ci --limit 100 --json number,title \
| jq -r --arg title "$TITLE" 'map(select(.title == $title)) | .[0].number // empty')
if [ -n "$EXISTING" ]; then
gh issue comment "$EXISTING" --body "Failed again: $RUN_URL"
echo "Commented on existing issue #$EXISTING"
else
gh issue create \
--title "$TITLE" \
--body "The workflow **$WORKFLOW_NAME** failed during execution.
**Failed jobs:** $FAILED_JOBS **Failed jobs:** $FAILED_JOBS
**Run URL:** $RUN_URL **Run URL:** $RUN_URL
Please investigate the failed jobs and address any issues." \ Please investigate the failed jobs and address any issues." \
--label "ci" --label "ci"
echo "Issue created successfully" echo "Issue created successfully"
fi
else else
echo "No job failures detected, skipping issue creation" echo "No job failures detected, skipping issue creation"
fi fi
+1
View File
@@ -6,6 +6,7 @@ on:
# We don't publish pre-releases for Rust. Crates.io is just a source # We don't publish pre-releases for Rust. Crates.io is just a source
# distribution, so we don't need to publish pre-releases. # distribution, so we don't need to publish pre-releases.
- "v*-beta*" - "v*-beta*"
- "*-v*" # for example, python-vX.Y.Z
env: env:
# This env var is used by Swatinem/rust-cache@v2 for the cache # This env var is used by Swatinem/rust-cache@v2 for the cache
@@ -4,14 +4,14 @@ on:
workflow_call: workflow_call:
inputs: inputs:
tag: tag:
description: "Tag name from Lance (e.g. `v7.2.0-beta.1`). If omitted, the newest release is resolved automatically — stable releases are preferred over pre-releases — and the run is skipped if it is not newer than the version currently pinned in Cargo.toml." description: "Tag name from Lance. If omitted, the skill will use the latest Lance release that needs an update."
required: false required: false
default: "" default: ""
type: string type: string
workflow_dispatch: workflow_dispatch:
inputs: inputs:
tag: tag:
description: "Tag name from Lance (e.g. `v7.2.0-beta.1`). Leave empty to resolve the newest release automatically — stable releases are preferred over pre-releases — and skip the run if it is not newer than the version currently pinned in Cargo.toml." description: "Tag name from Lance. Leave empty to use the latest Lance release that needs an update."
required: false required: false
default: "" default: ""
type: string type: string
-243
View File
@@ -1,243 +0,0 @@
name: Check doc links
# Checking external links is inherently noisy: third-party sites rate-limit
# automated clients, reject non-browser user agents, and go down temporarily.
# Blocking pull requests on that trades a lot of false failures for very little
# signal, so this runs on a schedule and reports findings in a single tracking
# issue instead of failing anyone's build.
on:
schedule:
- cron: "0 7 * * *"
workflow_dispatch:
# The report lives in one repository-global issue, so runs must not overlap: a
# lookup racing a create produces duplicate issues, and a healthy run closing
# the issue while a failing run only rewrites its body would leave a broken
# report closed. The group is deliberately ref-independent so that a manual
# dispatch serializes against the scheduled run.
concurrency:
group: docs-link-check
cancel-in-progress: false
permissions: {}
env:
REPORT_TITLE: "Docs link checker report"
jobs:
scan:
name: Scan links
runs-on: ubuntu-24.04
# lychee-action is pinned by SHA, but its wrapper downloads the lychee
# release tarball at run time without verifying a digest, and hands the
# resulting binary a GitHub token. Release assets remain replaceable, so
# that binary is confined to a job whose token can only read public
# content; everything that writes runs in the report job below.
permissions:
contents: read
outputs:
checker_outcome: ${{ steps.lychee.outcome }}
exit_code: ${{ steps.lychee.outputs.exit_code }}
status: ${{ steps.validate.outputs.status }}
steps:
- name: Checkout
uses: actions/checkout@v6
with:
# workflow_dispatch can run from any ref, but the report is
# repository-global. Always measure the default branch so a manual
# run from a topic branch cannot close a report that main warrants,
# or overwrite it with branch-only findings.
ref: ${{ github.event.repository.default_branch }}
persist-credentials: false
- name: Check links
id: lychee
continue-on-error: true
uses: lycheeverse/lychee-action@e7477775783ea5526144ba13e8db5eec57747ce8 # v2.9.0
with:
# Restricted to http(s) on purpose. Much of docs/src is generated
# API reference (the js/ tree comes from `npm run docs` in nodejs)
# and the hand-written pages use mkdocstrings cross-references and
# nav-relative paths that only resolve in the site mkdocs builds,
# not in this checkout, so relative links would be reported as
# broken on every run.
args: >-
--scheme https
--scheme http
--no-progress
--max-retries 3
--timeout 20
'docs/src/**/*.md'
format: json
output: ./lychee/out.json
jobSummary: false
# The report issue, not a red workflow run, is the signal for link
# findings and checker failures alike.
fail: false
- name: Validate report
id: validate
# lychee does not reserve exit code 2 for broken links: its CLI
# parser also exits 2 on an invalid option, before any link was
# checked or any report written. Only a parseable report whose
# counts agree with a completed exit code (0 or 2) counts as a link
# verdict. Everything else becomes a checker-error report instead of
# failing the workflow. Exit 2 covers timeouts as well as errors, and a
# timed-out host is exactly the transient unavailability this report
# exists to surface, so both count as findings. Requiring total > 0
# also catches a glob that silently stopped matching any file.
if: always()
env:
CHECKER_OUTCOME: ${{ steps.lychee.outcome }}
EXIT_CODE: ${{ steps.lychee.outputs.exit_code }}
run: |
status=checker-error
if [[ "$CHECKER_OUTCOME" == success ]] &&
[[ "$EXIT_CODE" == 0 || "$EXIT_CODE" == 2 ]] &&
jq -e --argjson code "$EXIT_CODE" '
(.total > 0) and
(if $code == 0
then .errors == 0 and .timeouts == 0
and (.error_map | length == 0) and (.timeout_map | length == 0)
else (.errors + .timeouts) > 0
and ((.error_map | length) + (.timeout_map | length)) > 0
end)
' ./lychee/out.json
then
if [[ "$EXIT_CODE" == 0 ]]; then
status=healthy
else
status=findings
fi
fi
echo "status=$status" >> "$GITHUB_OUTPUT"
echo "Validated link check as $status"
- name: Upload report
if: steps.validate.outputs.status == 'findings'
uses: actions/upload-artifact@v7
with:
name: link-report
path: ./lychee/out.json
retention-days: 7
report:
name: Update report issue
needs: scan
runs-on: ubuntu-24.04
# Deliberately no checkout: this job needs the report artifact and the
# issues API, not the repository contents.
permissions:
issues: write
env:
CHECKER_OUTCOME: ${{ needs.scan.outputs.checker_outcome }}
EXIT_CODE: ${{ needs.scan.outputs.exit_code }}
STATUS: ${{ needs.scan.outputs.status }}
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
steps:
- name: Find existing report issue
id: report
# Matched on title alone, and through search rather than a listing:
# the issue action applies labels in a separate call after creating the
# issue, so a label filter misses a half-created report, and this
# repository has far more open issues than one listing page holds.
# Closed issues are included because a healthy run closes the report:
# an open-only lookup would forget that identity and the next failing
# run would open a duplicate. The oldest match stays the canonical
# report and is reopened below when a problem recurs.
run: |
match=$(gh issue list --repo "$GITHUB_REPOSITORY" --state all \
--search "in:title \"$REPORT_TITLE\" author:app/github-actions" \
--limit 50 --json number,title,state \
--jq "[.[] | select(.title == \"$REPORT_TITLE\")] | sort_by(.number) | first // empty")
echo "number=$(jq -r '.number // empty' <<<"$match")" >> "$GITHUB_OUTPUT"
echo "state=$(jq -r '.state // empty' <<<"$match")" >> "$GITHUB_OUTPUT"
- name: Download report
if: env.STATUS == 'findings'
uses: actions/download-artifact@v8
with:
name: link-report
path: ./lychee
- name: Compose report
if: env.STATUS == 'findings'
run: |
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
{
echo "Broken documentation links found by [\`$GITHUB_WORKFLOW\`]($run_url)."
echo
echo "This issue is rewritten by every scheduled run and closed automatically once all links resolve."
echo
echo "Entries can be false positives: some sites rate-limit or block automated clients while working fine in a browser. Confirm before editing the docs, and add persistent offenders to \`--exclude\` in \`.github/workflows/docs-link-check.yml\`."
echo
# Timeouts are reported alongside errors: entries land in
# timeout_map with a status text instead of an HTTP code.
jq -r '
"\(.errors) of \(.total) links failed, \(.timeouts) timed out.",
"",
([(.error_map | to_entries[]), (.timeout_map | to_entries[])]
| group_by(.key)[] |
"### Errors in \(.[0].key)",
"",
(map(.value[])[] | "* [\(.status.code // .status.text // "ERR")] <\(.url)> — \(.status.details // .status.text // "unknown error")"),
"")
' ./lychee/out.json
} > ./lychee/issue.md
- name: Compose checker error report
if: env.STATUS == 'checker-error'
run: |
mkdir -p ./lychee
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
{
echo "The documentation link check did not complete in [the latest run]($run_url)."
echo
echo "This issue is rewritten by every scheduled run and closed automatically once a trustworthy run finds that all links resolve."
echo
echo "The checker did not produce a trustworthy link verdict. Treat the previous result, if any, as stale until a later run completes."
echo
echo "* Action outcome: \`$CHECKER_OUTCOME\`"
echo "* Exit code: \`${EXIT_CODE:-not reported}\`"
echo "* Verdict validation: \`failed\`"
} > ./lychee/issue.md
- name: Reopen report issue
# A healthy run closes the report, and the issue action below only
# rewrites the body of whatever number it is given. Without an
# explicit reopen, a later finding or checker error would rewrite a
# closed issue. A CLOSED state implies the lookup found a canonical
# issue, so no separate emptiness check.
if: >-
env.STATUS != 'healthy' &&
steps.report.outputs.state == 'CLOSED'
env:
ISSUE_NUMBER: ${{ steps.report.outputs.number }}
run: |
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
gh issue reopen "$ISSUE_NUMBER" --repo "$GITHUB_REPOSITORY" \
--comment "The documentation link checker reported a problem again in [the latest run]($run_url)."
- name: Report link-check problem
if: env.STATUS != 'healthy'
uses: peter-evans/create-issue-from-file@fca9117c27cdc29c6c4db3b86c48e4115a786710 # v6.0.0
with:
# Empty on the first failing run, which creates the issue; afterwards
# the same issue is updated in place.
issue-number: ${{ steps.report.outputs.number }}
title: ${{ env.REPORT_TITLE }}
content-filepath: ./lychee/issue.md
labels: documentation
- name: Close report issue once links are healthy
# An OPEN state implies the lookup found a canonical issue; a report
# that is already closed needs nothing.
if: >-
env.STATUS == 'healthy' &&
steps.report.outputs.state == 'OPEN'
env:
ISSUE_NUMBER: ${{ steps.report.outputs.number }}
run: |
run_url="$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
gh issue close "$ISSUE_NUMBER" --repo "$GITHUB_REPOSITORY" \
--comment "All documentation links resolved in [the latest run]($run_url)."
-85
View File
@@ -1,85 +0,0 @@
name: GitHub Release
# All SDKs share one version, so a single `vX.Y.Z` tag produces a single GitHub
# release covering all of them. The per-package publish workflows (PyPI, NPM,
# Cargo, Maven) trigger off the same tag independently.
on:
push:
tags:
- "v*"
permissions:
contents: read
jobs:
gh-release:
runs-on: ubuntu-latest
permissions:
contents: write
steps:
- uses: actions/checkout@v6
with:
fetch-depth: 0
lfs: true
- name: Extract version
id: extract_version
env:
GITHUB_REF: ${{ github.ref }}
run: |
set -e
echo "Extracting tag and version from $GITHUB_REF"
if [[ $GITHUB_REF =~ refs/tags/v(.*) ]]; then
VERSION=${BASH_REMATCH[1]}
TAG=v$VERSION
echo "tag=$TAG" >> $GITHUB_OUTPUT
echo "version=$VERSION" >> $GITHUB_OUTPUT
else
echo "Failed to extract version from $GITHUB_REF"
exit 1
fi
echo "Extracted version $VERSION from $GITHUB_REF"
if [[ $VERSION =~ beta ]]; then
echo "This is a beta release"
echo "prerelease=true" >> $GITHUB_OUTPUT
# Get last release (that is not this one)
FROM_TAG=$(git tag --sort='version:refname' \
| grep ^v \
| grep -vF "$TAG" \
| python ci/semver_sort.py v \
| tail -n 1)
else
echo "This is a stable release"
echo "prerelease=false" >> $GITHUB_OUTPUT
# Get last stable tag (ignore betas)
FROM_TAG=$(git tag --sort='version:refname' \
| grep ^v \
| grep -vF "$TAG" \
| grep -v beta \
| python ci/semver_sort.py v \
| tail -n 1)
fi
echo "Found from tag $FROM_TAG"
echo "from_tag=$FROM_TAG" >> $GITHUB_OUTPUT
- name: Create Release Notes
id: release_notes
uses: mikepenz/release-changelog-builder-action@v4
with:
configuration: .github/release_notes.json
toTag: ${{ steps.extract_version.outputs.tag }}
fromTag: ${{ steps.extract_version.outputs.from_tag }}
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
- name: Create GH release
uses: softprops/action-gh-release@v2
with:
# Marking betas as pre-releases keeps them from taking the "Latest"
# badge on the releases page.
prerelease: ${{ steps.extract_version.outputs.prerelease }}
make_latest: ${{ steps.extract_version.outputs.prerelease == 'false' }}
tag_name: ${{ steps.extract_version.outputs.tag }}
token: ${{ secrets.GITHUB_TOKEN }}
generate_release_notes: false
name: LanceDB v${{ steps.extract_version.outputs.version }}
body: ${{ steps.release_notes.outputs.changelog }}
+29 -7
View File
@@ -1,14 +1,13 @@
name: Create release commit name: Create release commit
# This workflow increments the version, tags it, and pushes it. All SDKs share # This workflow increments versions, tags the version, and pushes it.
# a single version, so one tag releases all of them.
# When a tag is pushed, another workflow is triggered that creates a GH release # When a tag is pushed, another workflow is triggered that creates a GH release
# and uploads the binaries. This workflow is only for creating the tag. # and uploads the binaries. This workflow is only for creating the tag.
# This script will enforce that a minor version is incremented if there are any # This script will enforce that a minor version is incremented if there are any
# breaking changes since the last minor increment. A breaking change in any SDK # breaking changes since the last minor increment. However, it isn't able to
# bumps the minor version for all of them. If you wish to bypass this check, you # differentiate between breaking changes in Node versus Python. If you wish to
# can manually increment the version and push the tag. # bypass this check, you can manually increment the version and push the tag.
on: on:
workflow_dispatch: workflow_dispatch:
inputs: inputs:
@@ -25,6 +24,16 @@ on:
options: options:
- preview - preview
- stable - stable
python:
description: 'Make a Python release'
required: true
default: true
type: boolean
other:
description: 'Make a Node/Rust/Java release'
required: true
default: true
type: boolean
bump-minor: bump-minor:
description: 'Bump minor version' description: 'Bump minor version'
required: true required: true
@@ -56,12 +65,25 @@ jobs:
run: | run: |
git config user.name 'Lance Release' git config user.name 'Lance Release'
git config user.email 'lance-dev@lancedb.com' git config user.email 'lance-dev@lancedb.com'
- name: Bump version - name: Bump Python version
if: ${{ inputs.python }}
working-directory: python
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
# Need to get the commit before bumping the version, so we can
# determine if there are breaking changes in the next step as well.
echo "COMMIT_BEFORE_BUMP=$(git rev-parse HEAD)" >> $GITHUB_ENV
pip install bump-my-version PyGithub packaging
bash ../ci/bump_version.sh ${{ inputs.type }} ${{ inputs.bump-minor }} python-v
- name: Bump Node/Rust version
if: ${{ inputs.other }}
env: env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: | run: |
pip install bump-my-version PyGithub packaging pip install bump-my-version PyGithub packaging
bash ci/bump_version.sh ${{ inputs.type }} ${{ inputs.bump-minor }} bash ci/bump_version.sh ${{ inputs.type }} ${{ inputs.bump-minor }} v $COMMIT_BEFORE_BUMP
bash ci/update_lockfiles.sh --amend bash ci/update_lockfiles.sh --amend
- name: Push new version tag - name: Push new version tag
if: ${{ !inputs.dry_run }} if: ${{ !inputs.dry_run }}
-15
View File
@@ -61,11 +61,6 @@ jobs:
sudo apt update sudo apt update
sudo apt install -y protobuf-compiler libssl-dev sudo apt install -y protobuf-compiler libssl-dev
- uses: Swatinem/rust-cache@v2 - uses: Swatinem/rust-cache@v2
with:
# Restore everywhere, but only save from main. Per-PR saves are
# unreadable outside their own branch anyway, since GitHub scopes
# caches to the creating ref.
save-if: ${{ github.ref == 'refs/heads/main' }}
- name: Format Rust - name: Format Rust
run: cargo fmt --all -- --check run: cargo fmt --all -- --check
- name: Lint Rust - name: Lint Rust
@@ -108,11 +103,6 @@ jobs:
cache: 'pnpm' cache: 'pnpm'
cache-dependency-path: nodejs/pnpm-lock.yaml cache-dependency-path: nodejs/pnpm-lock.yaml
- uses: Swatinem/rust-cache@v2 - uses: Swatinem/rust-cache@v2
with:
# Restore everywhere, but only save from main. Per-PR saves are
# unreadable outside their own branch anyway, since GitHub scopes
# caches to the creating ref.
save-if: ${{ github.ref == 'refs/heads/main' }}
- name: Install dependencies - name: Install dependencies
run: | run: |
sudo apt update sudo apt update
@@ -192,11 +182,6 @@ jobs:
cache-dependency-path: nodejs/pnpm-lock.yaml cache-dependency-path: nodejs/pnpm-lock.yaml
- uses: dtolnay/rust-toolchain@stable - uses: dtolnay/rust-toolchain@stable
- uses: Swatinem/rust-cache@v2 - uses: Swatinem/rust-cache@v2
with:
# Restore everywhere, but only save from main. Per-PR saves are
# unreadable outside their own branch anyway, since GitHub scopes
# caches to the creating ref.
save-if: ${{ github.ref == 'refs/heads/main' }}
- name: Install dependencies - name: Install dependencies
run: | run: |
brew install protobuf brew install protobuf
+89 -95
View File
@@ -10,16 +10,10 @@ permissions:
on: on:
push: push:
branches:
- main
tags: tags:
- "v*" - "v*"
# The cross-compiled targets (musl especially) break from toolchain and
# dependency changes that nothing else in CI catches, and discovering that
# mid-release is expensive. A nightly run keeps that signal while dropping
# the full 8-target release matrix from all ~90 pushes to main each month.
# `report-failure` files an issue when a nightly breaks.
schedule:
- cron: "0 8 * * *"
workflow_dispatch:
pull_request: pull_request:
# This should trigger a dry run (we skip the final publish step) # This should trigger a dry run (we skip the final publish step)
paths: paths:
@@ -32,6 +26,73 @@ concurrency:
cancel-in-progress: true cancel-in-progress: true
jobs: jobs:
gh-release:
if: startsWith(github.ref, 'refs/tags/v')
runs-on: ubuntu-latest
permissions:
contents: write
steps:
- uses: actions/checkout@v6
with:
fetch-depth: 0
lfs: true
- name: Extract version
id: extract_version
env:
GITHUB_REF: ${{ github.ref }}
run: |
set -e
echo "Extracting tag and version from $GITHUB_REF"
if [[ $GITHUB_REF =~ refs/tags/v(.*) ]]; then
VERSION=${BASH_REMATCH[1]}
TAG=v$VERSION
echo "tag=$TAG" >> $GITHUB_OUTPUT
echo "version=$VERSION" >> $GITHUB_OUTPUT
else
echo "Failed to extract version from $GITHUB_REF"
exit 1
fi
echo "Extracted version $VERSION from $GITHUB_REF"
if [[ $VERSION =~ beta ]]; then
echo "This is a beta release"
# Get last release (that is not this one)
FROM_TAG=$(git tag --sort='version:refname' \
| grep ^v \
| grep -vF "$TAG" \
| python ci/semver_sort.py v \
| tail -n 1)
else
echo "This is a stable release"
# Get last stable tag (ignore betas)
FROM_TAG=$(git tag --sort='version:refname' \
| grep ^v \
| grep -vF "$TAG" \
| grep -v beta \
| python ci/semver_sort.py v \
| tail -n 1)
fi
echo "Found from tag $FROM_TAG"
echo "from_tag=$FROM_TAG" >> $GITHUB_OUTPUT
- name: Create Release Notes
id: release_notes
uses: mikepenz/release-changelog-builder-action@v4
with:
configuration: .github/release_notes.json
toTag: ${{ steps.extract_version.outputs.tag }}
fromTag: ${{ steps.extract_version.outputs.from_tag }}
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
- name: Create GH release
uses: softprops/action-gh-release@v2
with:
prerelease: ${{ contains('beta', github.ref) }}
tag_name: ${{ steps.extract_version.outputs.tag }}
token: ${{ secrets.GITHUB_TOKEN }}
generate_release_notes: false
name: Node/Rust LanceDB v${{ steps.extract_version.outputs.version }}
body: ${{ steps.release_notes.outputs.changelog }}
build-lancedb: build-lancedb:
strategy: strategy:
fail-fast: false fail-fast: false
@@ -40,18 +101,9 @@ jobs:
- target: aarch64-apple-darwin - target: aarch64-apple-darwin
host: macos-latest host: macos-latest
features: fp16kernels features: fp16kernels
pre_build: |- pre_build: brew install protobuf
brew install protobuf
# Fat LTO (the workspace default in .cargo/config.toml) is
# single-threaded and is the peak-memory step of the build. On
# this runner it accounted for ~111 of the job's ~113 minutes,
# making it the critical path of the entire publish pipeline.
# ThinLTO parallelizes it across the runner's cores, for a few
# percent of runtime performance.
export CARGO_PROFILE_RELEASE_LTO=thin
export CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16
- target: x86_64-pc-windows-msvc - target: x86_64-pc-windows-msvc
host: windows-2025 host: windows-2025-8x-x64
features: "," features: ","
pre_build: |- pre_build: |-
choco install --no-progress protoc ninja nasm choco install --no-progress protoc ninja nasm
@@ -59,19 +111,19 @@ jobs:
# There is an issue where choco doesn't add nasm to the path # There is an issue where choco doesn't add nasm to the path
export PATH="$PATH:/c/Program Files/NASM" export PATH="$PATH:/c/Program Files/NASM"
nasm -v nasm -v
# See the ThinLTO note on aarch64-apple-darwin above. Keeping # Fat LTO of the cdylib is single-threaded and the peak-memory
# peak memory down is also what lets this run on the standard # step of the build, and had started hitting rustc-LLVM OOM on the
# 4-core runner: the 8-core larger runner was only needed to # Windows runners. ThinLTO parallelizes it across the runner's
# stop fat LTO from OOMing rustc-LLVM. # cores and keeps peak memory well under the limit.
export CARGO_PROFILE_RELEASE_LTO=thin export CARGO_PROFILE_RELEASE_LTO=thin
export CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16 export CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16
- target: aarch64-pc-windows-msvc - target: aarch64-pc-windows-msvc
host: windows-2025 host: windows-2025-8x-x64
features: "," features: ","
pre_build: |- pre_build: |-
choco install --no-progress protoc choco install --no-progress protoc
rustup target add aarch64-pc-windows-msvc rustup target add aarch64-pc-windows-msvc
# See the ThinLTO note on aarch64-apple-darwin above. # See ThinLTO note on the x86_64-pc-windows-msvc target above.
export CARGO_PROFILE_RELEASE_LTO=thin export CARGO_PROFILE_RELEASE_LTO=thin
export CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16 export CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16
- target: x86_64-unknown-linux-gnu - target: x86_64-unknown-linux-gnu
@@ -146,49 +198,16 @@ jobs:
with: with:
toolchain: stable toolchain: stable
targets: ${{ matrix.settings.target }} targets: ${{ matrix.settings.target }}
# These builds were entirely uncached: the old key was static, so - name: Cache cargo
# `actions/cache` (which only writes on a miss) could never refresh it, uses: actions/cache@v5
# and the multi-GB whole-`target/` copy it tried to store never fit the
# repo's cache budget, so no entry was ever saved. rust-cache prunes
# `target/` to dependency artifacts and keys on Cargo.lock plus the rustc
# version, which both fixes the key and keeps entries a sane size.
#
# This caches dependency *compilation* only. The LTO link of the cdylib
# re-runs regardless, since the local crate changes every time, so the
# win is larger on the non-LTO jobs than here.
- name: Cache cargo (native builds)
uses: Swatinem/rust-cache@v2
if: ${{ !matrix.settings.docker }}
with: with:
# The release profile and per-target dirs differ from what the test path: |
# workflows cache, so these need to be separate entries. ~/.cargo/registry/index/
key: release-${{ matrix.settings.target }} ~/.cargo/registry/cache/
# Only the nightly run on main writes, so tag and PR runs restore a ~/.cargo/git/db/
# warm entry without every dependabot PR writing its own (which would .cargo-cache
# be unreadable elsewhere anyway, since GitHub scopes caches to the target/
# creating ref). The nightly cadence also keeps entries inside key: nodejs-${{ matrix.settings.target }}-cargo-${{ matrix.settings.host }}
# GitHub's 7-day eviction window, which a tag-only trigger would not.
save-if: ${{ github.ref == 'refs/heads/main' }}
# Docker builds can use rust-cache too. `target/` already lives on the
# host because the whole workspace is bind-mounted into the container, and
# rust-cache's prune and save run host-side, so they can manage it -- which
# is what keeps the entry to dependency artifacts rather than a multi-GB
# copy of everything.
#
# Two differences from the native builds. The container's CARGO_HOME is
# bind-mounted from `.cargo-cache` rather than the host's ~/.cargo, so that
# has to be cached explicitly. And the key is derived from the *host* rustc
# version, which is not the compiler that produced these artifacts; that is
# safe because cargo fingerprints the real compiler and rebuilds on a
# mismatch, it just means a base-image toolchain bump costs one cold build
# instead of invalidating the key.
- name: Cache cargo (docker builds)
uses: Swatinem/rust-cache@v2
if: ${{ matrix.settings.docker }}
with:
key: docker-${{ matrix.settings.target }}
cache-directories: .cargo-cache
save-if: ${{ github.ref == 'refs/heads/main' }}
- name: Install dependencies - name: Install dependencies
run: pnpm install --frozen-lockfile run: pnpm install --frozen-lockfile
- name: Install Zig - name: Install Zig
@@ -206,13 +225,9 @@ jobs:
if: ${{ matrix.settings.docker }} if: ${{ matrix.settings.docker }}
with: with:
image: ${{ matrix.settings.docker }} image: ${{ matrix.settings.docker }}
# All three mounts must live under `.cargo-cache`, which is what the
# cache step above saves. Previously the registry mounts pointed at
# `.cargo/...`, a path nothing cached, so the container re-downloaded
# the whole crate registry on every run.
options: "--user 0:0 -v ${{ github.workspace }}/.cargo-cache/git/db:/usr/local/cargo/git/db \ options: "--user 0:0 -v ${{ github.workspace }}/.cargo-cache/git/db:/usr/local/cargo/git/db \
-v ${{ github.workspace }}/.cargo-cache/registry/cache:/usr/local/cargo/registry/cache \ -v ${{ github.workspace }}/.cargo/registry/cache:/usr/local/cargo/registry/cache \
-v ${{ github.workspace }}/.cargo-cache/registry/index:/usr/local/cargo/registry/index \ -v ${{ github.workspace }}/.cargo/registry/index:/usr/local/cargo/registry/index \
-v ${{ github.workspace }}:/build -w /build/nodejs" -v ${{ github.workspace }}:/build -w /build/nodejs"
run: | run: |
set -e set -e
@@ -224,16 +239,6 @@ jobs:
--js ../lancedb/native.js \ --js ../lancedb/native.js \
--strip \ --strip \
--output-dir dist/ --output-dir dist/
# The container runs as root (`--user 0:0`), so everything it wrote to the
# mounted cache dirs is root-owned. rust-cache's post step runs as the
# runner user and has to both read these and delete from them while
# pruning, so hand them back before it runs.
- name: Take ownership of docker build output
if: ${{ matrix.settings.docker }}
run: |
sudo chown -R "$(id -u):$(id -g)" \
"${{ github.workspace }}/.cargo-cache" \
"${{ github.workspace }}/target"
- name: Build - name: Build
run: | run: |
${{ matrix.settings.pre_build }} ${{ matrix.settings.pre_build }}
@@ -247,15 +252,6 @@ jobs:
--output-dir dist/ --output-dir dist/
if: ${{ !matrix.settings.docker }} if: ${{ !matrix.settings.docker }}
shell: bash shell: bash
# The standard Windows runners have ~14 GB free, and a release `target/`
# for this workspace is a large fraction of that. Report the remaining
# headroom so a build that only just fits is visible before a dependency
# bump turns it into a failed release. `always()` so the numbers are
# still there when the build is what ran out of space.
- name: Report disk headroom
if: always()
run: df -h
shell: bash
- name: Upload artifact - name: Upload artifact
uses: actions/upload-artifact@v7 uses: actions/upload-artifact@v7
with: with:
@@ -406,9 +402,7 @@ jobs:
name: Report Workflow Failure name: Report Workflow Failure
runs-on: ubuntu-latest runs-on: ubuntu-latest
needs: [build-lancedb, test-lancedb, publish] needs: [build-lancedb, test-lancedb, publish]
# Nightly runs are the only thing watching the cross-compiled targets now, if: always() && failure() && startsWith(github.ref, 'refs/tags/v')
# so they have to report failures too or the signal is silently lost.
if: always() && failure() && (startsWith(github.ref, 'refs/tags/v') || github.event_name == 'schedule')
permissions: permissions:
contents: read contents: read
issues: write issues: write
+72 -29
View File
@@ -3,7 +3,7 @@ name: PyPI Publish
on: on:
push: push:
tags: tags:
- 'v*' - 'python-v*'
pull_request: pull_request:
# This should trigger a dry run (we skip the final publish step) # This should trigger a dry run (we skip the final publish step)
paths: paths:
@@ -20,12 +20,6 @@ env:
permissions: permissions:
contents: read contents: read
# Without this, a force-push to a PR leaves the previous run going -- including
# a ~74 minute Windows job and a billed arm64 wheel build.
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs: jobs:
linux: linux:
name: Python ${{ matrix.config.package_name }} ${{ matrix.config.platform }} manylinux${{ matrix.config.manylinux }} name: Python ${{ matrix.config.package_name }} ${{ matrix.config.platform }} manylinux${{ matrix.config.manylinux }}
@@ -69,16 +63,6 @@ jobs:
uses: actions/setup-python@v6 uses: actions/setup-python@v6
with: with:
python-version: "3.10" python-version: "3.10"
- name: Add swap for Arm fat LTO
if: matrix.config.platform == 'aarch64'
shell: bash
run: |
swap_file="$RUNNER_TEMP/lancedb-swap"
sudo fallocate --length 16G "$swap_file"
sudo chmod 600 "$swap_file"
sudo mkswap "$swap_file"
sudo swapon "$swap_file"
free -h
- uses: ./.github/workflows/build_linux_wheel - uses: ./.github/workflows/build_linux_wheel
with: with:
python-minor-version: 10 python-minor-version: 10
@@ -88,7 +72,7 @@ jobs:
package-name: ${{ matrix.config.package_name }} package-name: ${{ matrix.config.package_name }}
rustflags: ${{ matrix.config.rustflags }} rustflags: ${{ matrix.config.rustflags }}
- uses: actions/upload-artifact@v7 - uses: actions/upload-artifact@v7
if: startsWith(github.ref, 'refs/tags/v') if: startsWith(github.ref, 'refs/tags/python-v')
with: with:
name: wheels-linux-${{ matrix.config.package_name }}-${{ matrix.config.platform }}-${{ matrix.config.manylinux }} name: wheels-linux-${{ matrix.config.package_name }}-${{ matrix.config.platform }}-${{ matrix.config.manylinux }}
path: target/wheels/*.whl path: target/wheels/*.whl
@@ -117,7 +101,7 @@ jobs:
python-minor-version: 10 python-minor-version: 10
args: "--release --strip --target ${{ matrix.config.target }} --features fp16kernels" args: "--release --strip --target ${{ matrix.config.target }} --features fp16kernels"
- uses: actions/upload-artifact@v7 - uses: actions/upload-artifact@v7
if: startsWith(github.ref, 'refs/tags/v') if: startsWith(github.ref, 'refs/tags/python-v')
with: with:
name: wheels-mac-${{ matrix.config.target }} name: wheels-mac-${{ matrix.config.target }}
path: target/wheels/lancedb-*.whl path: target/wheels/lancedb-*.whl
@@ -138,26 +122,19 @@ jobs:
uses: actions/setup-python@v6 uses: actions/setup-python@v6
with: with:
python-version: "3.13" python-version: "3.13"
# NOTE: caching cargo here would be a no-op. This workflow only runs on
# tags and PRs, and GitHub only lets a run restore caches from its own ref
# or the default branch -- so with no run on main there is nothing that
# can populate an entry the release build would be allowed to read. Fixing
# this needs a main/nightly trigger (which would also catch wheel-build
# breakage before a release); the ~74 minutes here is otherwise dominated
# by the fat-LTO link, which no cache avoids.
- uses: ./.github/workflows/build_windows_wheel - uses: ./.github/workflows/build_windows_wheel
with: with:
python-minor-version: 10 python-minor-version: 10
args: "--release --strip" args: "--release --strip"
- uses: actions/upload-artifact@v7 - uses: actions/upload-artifact@v7
if: startsWith(github.ref, 'refs/tags/v') if: startsWith(github.ref, 'refs/tags/python-v')
with: with:
name: wheels-windows name: wheels-windows
path: target/wheels/lancedb-*.whl path: target/wheels/lancedb-*.whl
if-no-files-found: error if-no-files-found: error
publish: publish:
name: Publish wheels name: Publish wheels
if: startsWith(github.ref, 'refs/tags/v') if: startsWith(github.ref, 'refs/tags/python-v')
needs: [linux, mac, windows] needs: [linux, mac, windows]
runs-on: ubuntu-latest runs-on: ubuntu-latest
permissions: permissions:
@@ -206,6 +183,72 @@ jobs:
uses: pypa/gh-action-pypi-publish@release/v1 uses: pypa/gh-action-pypi-publish@release/v1
with: with:
packages-dir: target/wheels/ packages-dir: target/wheels/
gh-release:
if: startsWith(github.ref, 'refs/tags/python-v')
runs-on: ubuntu-latest
permissions:
contents: write
steps:
- uses: actions/checkout@v6
with:
fetch-depth: 0
lfs: true
- name: Extract version
id: extract_version
env:
GITHUB_REF: ${{ github.ref }}
run: |
set -e
echo "Extracting tag and version from $GITHUB_REF"
if [[ $GITHUB_REF =~ refs/tags/python-v(.*) ]]; then
VERSION=${BASH_REMATCH[1]}
TAG=python-v$VERSION
echo "tag=$TAG" >> $GITHUB_OUTPUT
echo "version=$VERSION" >> $GITHUB_OUTPUT
else
echo "Failed to extract version from $GITHUB_REF"
exit 1
fi
echo "Extracted version $VERSION from $GITHUB_REF"
if [[ $VERSION =~ beta ]]; then
echo "This is a beta release"
# Get last release (that is not this one)
FROM_TAG=$(git tag --sort='version:refname' \
| grep ^python-v \
| grep -vF "$TAG" \
| python ci/semver_sort.py python-v \
| tail -n 1)
else
echo "This is a stable release"
# Get last stable tag (ignore betas)
FROM_TAG=$(git tag --sort='version:refname' \
| grep ^python-v \
| grep -vF "$TAG" \
| grep -v beta \
| python ci/semver_sort.py python-v \
| tail -n 1)
fi
echo "Found from tag $FROM_TAG"
echo "from_tag=$FROM_TAG" >> $GITHUB_OUTPUT
- name: Create Python Release Notes
id: python_release_notes
uses: mikepenz/release-changelog-builder-action@v4
with:
configuration: .github/release_notes.json
toTag: ${{ steps.extract_version.outputs.tag }}
fromTag: ${{ steps.extract_version.outputs.from_tag }}
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
- name: Create Python GH release
uses: softprops/action-gh-release@v2
with:
prerelease: ${{ contains('beta', github.ref) }}
tag_name: ${{ steps.extract_version.outputs.tag }}
token: ${{ secrets.GITHUB_TOKEN }}
generate_release_notes: false
name: Python LanceDB v${{ steps.extract_version.outputs.version }}
body: ${{ steps.python_release_notes.outputs.changelog }}
report-failure: report-failure:
name: Report Workflow Failure name: Report Workflow Failure
runs-on: ubuntu-latest runs-on: ubuntu-latest
@@ -213,7 +256,7 @@ jobs:
permissions: permissions:
contents: read contents: read
issues: write issues: write
if: always() && failure() && startsWith(github.ref, 'refs/tags/v') if: always() && failure() && startsWith(github.ref, 'refs/tags/python-v')
steps: steps:
- uses: actions/checkout@v6 - uses: actions/checkout@v6
- uses: ./.github/actions/create-failure-issue - uses: ./.github/actions/create-failure-issue
-33
View File
@@ -108,15 +108,6 @@ jobs:
run: | run: |
sudo apt update sudo apt update
sudo apt install -y protobuf-compiler sudo apt install -y protobuf-compiler
# `pip install -e .` builds the extension with maturin, which is most of
# this job's ~33 minutes. It had no Rust cache, so every dependency was
# recompiled from scratch on every run.
- uses: Swatinem/rust-cache@v2
with:
# Restore everywhere, but only save from main. Per-PR saves are
# unreadable outside their own branch anyway, since GitHub scopes
# caches to the creating ref.
save-if: ${{ github.ref == 'refs/heads/main' }}
- name: Install - name: Install
run: | run: |
pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests,dev,embeddings] pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests,dev,embeddings]
@@ -177,14 +168,6 @@ jobs:
uses: actions/setup-python@v6 uses: actions/setup-python@v6
with: with:
python-version: "3.13" python-version: "3.13"
# maturin runs cargo natively on macOS (docker is Linux-only), so the host
# target dir is cacheable. This job had no Rust cache.
- uses: Swatinem/rust-cache@v2
with:
# Restore everywhere, but only save from main. Per-PR saves are
# unreadable outside their own branch anyway, since GitHub scopes
# caches to the creating ref.
save-if: ${{ github.ref == 'refs/heads/main' }}
- uses: ./.github/workflows/build_mac_wheel - uses: ./.github/workflows/build_mac_wheel
with: with:
args: --profile ci args: --profile ci
@@ -214,14 +197,6 @@ jobs:
uses: actions/setup-python@v6 uses: actions/setup-python@v6
with: with:
python-version: "3.13" python-version: "3.13"
# maturin runs cargo natively on Windows (docker is Linux-only), so the
# host target dir is cacheable. This job had no Rust cache at all and so
# rebuilt every dependency from scratch on every run.
- uses: Swatinem/rust-cache@v2
with:
# Restore everywhere, but only save from main. The repo sits at
# GitHub's cache cap, so per-PR saves just evict main's entries.
save-if: ${{ github.ref == 'refs/heads/main' }}
- uses: ./.github/workflows/build_windows_wheel - uses: ./.github/workflows/build_windows_wheel
with: with:
args: --profile ci args: --profile ci
@@ -249,14 +224,6 @@ jobs:
uses: actions/setup-python@v6 uses: actions/setup-python@v6
with: with:
python-version: "3.10" python-version: "3.10"
# As with Doctest, `pip install -e .` compiles the extension and this job
# had no Rust cache, which is most of its ~37 minutes.
- uses: Swatinem/rust-cache@v2
with:
# Restore everywhere, but only save from main. Per-PR saves are
# unreadable outside their own branch anyway, since GitHub scopes
# caches to the creating ref.
save-if: ${{ github.ref == 'refs/heads/main' }}
- name: Install lancedb - name: Install lancedb
run: | run: |
pip install "pydantic<2" pip install "pydantic<2"
+13 -53
View File
@@ -48,11 +48,6 @@ jobs:
with: with:
components: rustfmt, clippy components: rustfmt, clippy
- uses: Swatinem/rust-cache@v2 - uses: Swatinem/rust-cache@v2
with:
# Restore everywhere, but only save from main. Per-PR saves are
# unreadable outside their own branch anyway, since GitHub scopes
# caches to the creating ref.
save-if: ${{ github.ref == 'refs/heads/main' }}
- name: Install dependencies - name: Install dependencies
run: | run: |
sudo apt update sudo apt update
@@ -94,11 +89,6 @@ jobs:
run: rm -f Cargo.lock run: rm -f Cargo.lock
- uses: rui314/setup-mold@v1 - uses: rui314/setup-mold@v1
- uses: Swatinem/rust-cache@v2 - uses: Swatinem/rust-cache@v2
with:
# Restore everywhere, but only save from main. Per-PR saves are
# unreadable outside their own branch anyway, since GitHub scopes
# caches to the creating ref.
save-if: ${{ github.ref == 'refs/heads/main' }}
- name: Install dependencies - name: Install dependencies
run: | run: |
sudo apt update sudo apt update
@@ -128,11 +118,6 @@ jobs:
fetch-depth: 0 fetch-depth: 0
lfs: true lfs: true
- uses: Swatinem/rust-cache@v2 - uses: Swatinem/rust-cache@v2
with:
# Restore everywhere, but only save from main. Per-PR saves are
# unreadable outside their own branch anyway, since GitHub scopes
# caches to the creating ref.
save-if: ${{ github.ref == 'refs/heads/main' }}
- name: Install dependencies - name: Install dependencies
run: | run: |
sudo apt update sudo apt update
@@ -190,11 +175,6 @@ jobs:
- name: CPU features - name: CPU features
run: sysctl -a | grep cpu run: sysctl -a | grep cpu
- uses: Swatinem/rust-cache@v2 - uses: Swatinem/rust-cache@v2
with:
# Restore everywhere, but only save from main. Per-PR saves are
# unreadable outside their own branch anyway, since GitHub scopes
# caches to the creating ref.
save-if: ${{ github.ref == 'refs/heads/main' }}
- name: Install dependencies - name: Install dependencies
run: brew install protobuf run: brew install protobuf
- name: Run tests - name: Run tests
@@ -207,19 +187,12 @@ jobs:
cargo test --profile ci --features $ALL_FEATURES --locked cargo test --profile ci --features $ALL_FEATURES --locked
windows: windows:
runs-on: windows-2022
strategy: strategy:
fail-fast: false
matrix: matrix:
include: target:
- target: x86_64-pc-windows-msvc - x86_64-pc-windows-msvc
runner: windows-2022 - aarch64-pc-windows-msvc
# windows-11-arm is a standard runner, so it is free on public repos.
# Running natively lets the aarch64 tests actually execute -- this
# job used to cross-compile them and then skip the test step, paying
# full codegen and link cost for a compile check.
- target: aarch64-pc-windows-msvc
runner: windows-11-arm
runs-on: ${{ matrix.runner }}
defaults: defaults:
run: run:
working-directory: rust/lancedb working-directory: rust/lancedb
@@ -228,11 +201,6 @@ jobs:
- name: Set target - name: Set target
run: rustup target add ${{ matrix.target }} run: rustup target add ${{ matrix.target }}
- uses: Swatinem/rust-cache@v2 - uses: Swatinem/rust-cache@v2
with:
# Restore everywhere, but only save from main. Per-PR saves are
# unreadable outside their own branch anyway, since GitHub scopes
# caches to the creating ref.
save-if: ${{ github.ref == 'refs/heads/main' }}
- name: Install Protoc v21.12 - name: Install Protoc v21.12
run: choco install --no-progress protoc run: choco install --no-progress protoc
- name: Build - name: Build
@@ -240,12 +208,11 @@ jobs:
$env:VCPKG_ROOT = $env:VCPKG_INSTALLATION_ROOT $env:VCPKG_ROOT = $env:VCPKG_INSTALLATION_ROOT
cargo build --profile ci --features aws,remote --tests --locked --target ${{ matrix.target }} cargo build --profile ci --features aws,remote --tests --locked --target ${{ matrix.target }}
- name: Run tests - name: Run tests
# Can only run tests when target matches host
if: ${{ matrix.target == 'x86_64-pc-windows-msvc' }}
run: | run: |
$env:VCPKG_ROOT = $env:VCPKG_INSTALLATION_ROOT $env:VCPKG_ROOT = $env:VCPKG_INSTALLATION_ROOT
# `--target` has to match the build step above. Without it cargo uses cargo test --profile ci --features aws,remote --locked
# target/ci/ rather than target/<triple>/ci/ and rebuilds the entire
# dependency graph a second time.
cargo test --profile ci --features aws,remote --locked --target ${{ matrix.target }}
msrv: msrv:
# Check the minimum supported Rust version # Check the minimum supported Rust version
@@ -271,11 +238,6 @@ jobs:
with: with:
toolchain: ${{ matrix.msrv }} toolchain: ${{ matrix.msrv }}
- uses: Swatinem/rust-cache@v2 - uses: Swatinem/rust-cache@v2
with:
# Restore everywhere, but only save from main. Per-PR saves are
# unreadable outside their own branch anyway, since GitHub scopes
# caches to the creating ref.
save-if: ${{ github.ref == 'refs/heads/main' }}
- name: Downgrade dependencies - name: Downgrade dependencies
# These packages have newer requirements for MSRV # These packages have newer requirements for MSRV
run: | run: |
@@ -296,18 +258,16 @@ jobs:
cargo update -p aws-types --precise 1.3.9 cargo update -p aws-types --precise 1.3.9
cargo update -p aws-sigv4 --precise 1.3.5 cargo update -p aws-sigv4 --precise 1.3.5
cargo update -p aws-credential-types --precise 1.2.8 cargo update -p aws-credential-types --precise 1.2.8
# aws-smithy-checksums must stay at or above 0.63.13: OpenDAL's S3 cargo update -p aws-smithy-checksums --precise 0.63.9
# service needs crc-fast ~1.9, and older releases pin it to ~1.3.
cargo update -p aws-smithy-checksums --precise 0.63.13
cargo update -p aws-smithy-runtime --precise 1.9.3 cargo update -p aws-smithy-runtime --precise 1.9.3
cargo update -p aws-smithy-http --precise 0.62.6 cargo update -p aws-smithy-http --precise 0.62.4
cargo update -p aws-smithy-eventstream --precise 0.60.14 cargo update -p aws-smithy-eventstream --precise 0.60.12
cargo update -p aws-smithy-http-client --precise 1.1.3 cargo update -p aws-smithy-http-client --precise 1.1.3
cargo update -p aws-smithy-observability --precise 0.1.4 cargo update -p aws-smithy-observability --precise 0.1.4
cargo update -p aws-smithy-query --precise 0.60.8 cargo update -p aws-smithy-query --precise 0.60.8
cargo update -p aws-smithy-runtime-api --precise 1.9.3 cargo update -p aws-smithy-runtime-api --precise 1.9.1
cargo update -p aws-smithy-async --precise 1.2.7 cargo update -p aws-smithy-async --precise 1.2.6
cargo update -p aws-smithy-types --precise 1.3.6 cargo update -p aws-smithy-types --precise 1.3.5
cargo update -p aws-smithy-xml --precise 0.60.11 cargo update -p aws-smithy-xml --precise 0.60.11
cargo update -p home --precise 0.5.9 cargo update -p home --precise 0.5.9
- name: cargo +${{ matrix.msrv }} check - name: cargo +${{ matrix.msrv }} check
-29
View File
@@ -92,8 +92,6 @@ Python bindings changes:
* Should use `LOOP.run()` to call the corresponding `AsyncTable` method. * Should use `LOOP.run()` to call the corresponding `AsyncTable` method.
6. Add concrete sync method to `RemoteTable` class in `python/python/lancedb/remote/table.py`. 6. Add concrete sync method to `RemoteTable` class in `python/python/lancedb/remote/table.py`.
7. Add unit test in `python/tests/test_table.py`. 7. Add unit test in `python/tests/test_table.py`.
8. If you added a new public class or module-level function (not just a method on an
existing class), expose it in the API reference. See "Python API reference" below.
TypeScript bindings changes: TypeScript bindings changes:
@@ -105,33 +103,6 @@ TypeScript bindings changes:
5. Add test in `nodejs/__test__/table.test.ts`. 5. Add test in `nodejs/__test__/table.test.ts`.
6. Run `npm run docs` to generate TypeScript documentation. 6. Run `npm run docs` to generate TypeScript documentation.
## Python API reference
`docs/src/python/python.md` is the entire Python API reference. It is maintained by
hand, and anything not listed there is not rendered at all, so new public classes and
module-level functions have to be added explicitly. How depends on the module:
* `lancedb.index`, `lancedb.embeddings`, `lancedb.remote`, and `lancedb.rerankers` are
rendered by a single directive each, driven by the module's `__all__`. Add the new
name to `__all__` and it appears; forget, and it is silently omitted.
* Everything else (`lancedb`, `lancedb.table`, `lancedb.query`, `lancedb.db`, ...) is
listed symbol by symbol. Add a `::: lancedb.<module>.<Name>` line to the matching
section, and remember that the page separates synchronous and asynchronous APIs.
Deliberately undocumented: concrete implementations reached through an abstract base
(`LanceTable`, `LanceDBConnection`, `RemoteDBConnection`), query base classes already
covered by `inherited_members`, and internal helpers.
Cross-references in docstrings use mkdocstrings syntax, `[text][lancedb.table.Table]`.
Plain relative links such as `[Table](Table)` do not resolve. To check your work:
```shell
pip install -r docs/requirements.txt
cd docs && PYTHONPATH=. mkdocs build
```
The docs site only builds on pushes to `main`, so this is not covered by PR CI.
## Review Guidelines ## Review Guidelines
Please consider the following when reviewing code contributions. Please consider the following when reviewing code contributions.
Generated
+375 -373
View File
File diff suppressed because it is too large Load Diff
+15 -15
View File
@@ -13,20 +13,20 @@ categories = ["database-implementations"]
rust-version = "1.91.0" rust-version = "1.91.0"
[workspace.dependencies] [workspace.dependencies]
lance = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } lance = { "version" = "=9.0.0", default-features = false }
lance-core = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } lance-core = "=9.0.0"
lance-datagen = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } lance-datagen = "=9.0.0"
lance-file = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } lance-file = "=9.0.0"
lance-io = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } lance-io = { "version" = "=9.0.0", default-features = false }
lance-index = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } lance-index = "=9.0.0"
lance-linalg = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } lance-linalg = "=9.0.0"
lance-namespace = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } lance-namespace = "=9.0.0"
lance-namespace-impls = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } lance-namespace-impls = { "version" = "=9.0.0", default-features = false }
lance-table = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } lance-table = "=9.0.0"
lance-testing = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } lance-testing = "=9.0.0"
lance-datafusion = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } lance-datafusion = "=9.0.0"
lance-encoding = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } lance-encoding = "=9.0.0"
lance-arrow = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } lance-arrow = "=9.0.0"
ahash = "0.8" ahash = "0.8"
# Note that this one does not include pyarrow # Note that this one does not include pyarrow
arrow = { version = "58.0.0", optional = false } arrow = { version = "58.0.0", optional = false }
@@ -52,7 +52,7 @@ env_logger = "0.11"
half = { "version" = "2.7.1", default-features = false, features = [ half = { "version" = "2.7.1", default-features = false, features = [
"num-traits", "num-traits",
] } ] }
futures = "0.3" futures = "0"
log = "0.4" log = "0.4"
metrics = "0.24" metrics = "0.24"
metrics-util = "0.19" metrics-util = "0.19"
+3 -3
View File
@@ -2,9 +2,9 @@ set -e
RELEASE_TYPE=${1:-"stable"} RELEASE_TYPE=${1:-"stable"}
BUMP_MINOR=${2:-false} BUMP_MINOR=${2:-false}
HEAD_SHA=$(git rev-parse HEAD) TAG_PREFIX=${3:-"v"} # Such as "python-v"
HEAD_SHA=${4:-$(git rev-parse HEAD)}
readonly TAG_PREFIX="v"
readonly SELF_DIR=$(cd "$( dirname "${BASH_SOURCE[0]}" )" && pwd ) readonly SELF_DIR=$(cd "$( dirname "${BASH_SOURCE[0]}" )" && pwd )
PREV_TAG=$(git tag --sort='version:refname' | grep ^$TAG_PREFIX | python $SELF_DIR/semver_sort.py $TAG_PREFIX | tail -n 1) PREV_TAG=$(git tag --sort='version:refname' | grep ^$TAG_PREFIX | python $SELF_DIR/semver_sort.py $TAG_PREFIX | tail -n 1)
@@ -12,7 +12,7 @@ echo "Found previous tag $PREV_TAG"
# Initially, we don't want to tag if we are doing stable, because we will bump # Initially, we don't want to tag if we are doing stable, because we will bump
# again later. See comment at end for why. # again later. See comment at end for why.
if [[ "$RELEASE_TYPE" == 'stable' ]]; then if [[ "$RELEASE_TYPE" == 'stable' ]]; then
BUMP_ARGS="--no-tag" BUMP_ARGS="--no-tag"
fi fi
-7
View File
@@ -101,13 +101,6 @@ ignore = [
# https://rustsec.org/advisories/RUSTSEC-2026-0195 # https://rustsec.org/advisories/RUSTSEC-2026-0195
{ id = "RUSTSEC-2026-0194", reason = "transitive via inferno/lance/opendal; XML from trusted cloud endpoints, not attacker-controlled" }, { id = "RUSTSEC-2026-0194", reason = "transitive via inferno/lance/opendal; XML from trusted cloud endpoints, not attacker-controlled" },
{ id = "RUSTSEC-2026-0195", reason = "transitive via inferno/lance/opendal; XML from trusted cloud endpoints, not attacker-controlled" }, { id = "RUSTSEC-2026-0195", reason = "transitive via inferno/lance/opendal; XML from trusted cloud endpoints, not attacker-controlled" },
# smartstring: unmaintained — the repository was archived by its author on
# 2026-05-03. Not a vulnerability. Reached only transitively through polars
# (polars-core/-io/-ops/-time/-utils); nothing in LanceDB depends on it directly.
# The advisory states no safe upgrade is available: upstream recommends
# compact_str/smol_str, so clearing this requires polars to migrate.
# https://rustsec.org/advisories/RUSTSEC-2026-0249
{ id = "RUSTSEC-2026-0249", reason = "smartstring unmaintained via polars; no fixed upstream release" },
] ]
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
-5
View File
@@ -51,11 +51,6 @@ plugins:
paths: [../python/python] paths: [../python/python]
options: options:
docstring_style: numpy docstring_style: numpy
docstring_options:
# Attributes documented in a `Parameters` section, and pydantic
# dataclasses whose `__init__` griffe cannot see statically, both
# trip this check. It reports nothing actionable here.
warn_unknown_params: false
heading_level: 3 heading_level: 3
show_signature_annotations: true show_signature_annotations: true
show_root_heading: true show_root_heading: true
+1 -11
View File
@@ -453,16 +453,6 @@ paths:
The metric type to use for the index. l2, Cosine, Dot are supported. The metric type to use for the index. l2, Cosine, Dot are supported.
index_type: index_type:
type: string type: string
custom_stop_words:
type: [array, "null"]
items:
type: string
description: |
The custom stop-word list for an FTS index. A non-null
array replaces the language's built-in stop-word list and is only
applied when remove_stop_words is enabled. Null uses the built-in
language list, while an empty array explicitly replaces it with no
stop words.
responses: responses:
"200": "200":
description: Index successfully created description: Index successfully created
@@ -520,4 +510,4 @@ paths:
"401": "401":
$ref: "#/components/responses/unauthorized" $ref: "#/components/responses/unauthorized"
"404": "404":
$ref: "#/components/responses/not_found" $ref: "#/components/responses/not_found"
+1 -1
View File
@@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`:
<dependency> <dependency>
<groupId>com.lancedb</groupId> <groupId>com.lancedb</groupId>
<artifactId>lancedb-core</artifactId> <artifactId>lancedb-core</artifactId>
<version>0.38.0-beta.0</version> <version>0.33.0-beta.0</version>
</dependency> </dependency>
``` ```
+1 -1
View File
@@ -1,7 +1,7 @@
# Contributing to LanceDB Typescript # Contributing to LanceDB Typescript
This document outlines the process for contributing to LanceDB Typescript. This document outlines the process for contributing to LanceDB Typescript.
For general contribution guidelines, see [CONTRIBUTING.md](https://github.com/lancedb/lancedb/blob/main/CONTRIBUTING.md). For general contribution guidelines, see [CONTRIBUTING.md](../CONTRIBUTING.md).
## Project layout ## Project layout
-120
View File
@@ -25,27 +25,6 @@ the underlying connection has been closed.
## Methods ## Methods
### cancelJob()
```ts
abstract cancelJob(jobId): Promise<boolean>
```
Request cancellation of a server-side job by id.
Resolves to true if the server accepted the cancellation, false if no
such job exists. Cancelling an already-terminal job is a no-op success.
#### Parameters
* **jobId**: `string`
#### Returns
`Promise`&lt;`boolean`&gt;
***
### cloneTable() ### cloneTable()
```ts ```ts
@@ -386,49 +365,6 @@ Drop an existing table.
*** ***
### dropTableAsync()
```ts
abstract dropTableAsync(name, namespacePath?): Promise<Job>
```
Start dropping a table and return its cleanup job.
The table may become unavailable before its data files are removed. Wait
on the returned job to know when cleanup has finished.
#### Parameters
* **name**: `string`
* **namespacePath?**: `string`[]
#### Returns
`Promise`&lt;[`Job`](Job.md)&gt;
***
### getJob()
```ts
abstract getJob(jobId): Promise<null | JobDescription>
```
Describe a single server-side job by id.
Resolves to `null` when the server has no such job.
#### Parameters
* **jobId**: `string`
#### Returns
`Promise`&lt;`null` \| [`JobDescription`](../interfaces/JobDescription.md)&gt;
***
### isOpen() ### isOpen()
```ts ```ts
@@ -443,62 +379,6 @@ Return true if the connection has not been closed
*** ***
### job()
```ts
abstract job(jobId): Job
```
A [Job](Job.md) handle for a server-side job by id.
The handle is constructed without a server round trip; an unknown id
surfaces when the handle is used. Dropping the handle has no effect on
the job itself.
#### Parameters
* **jobId**: `string`
#### Returns
[`Job`](Job.md)
***
### jobHistory()
```ts
abstract jobHistory(jobId?): Promise<Table<any>>
```
The lifecycle event history of a server-side job, as an Arrow table.
Lists history across all jobs when `jobId` is omitted.
#### Parameters
* **jobId?**: `string`
#### Returns
`Promise`&lt;`Table`&lt;`any`&gt;&gt;
***
### listJobs()
```ts
abstract listJobs(): Promise<JobInfo[]>
```
List server-side jobs across the database's tables.
#### Returns
`Promise`&lt;[`JobInfo`](../interfaces/JobInfo.md)[]&gt;
***
### listNamespaces() ### listNamespaces()
```ts ```ts
-83
View File
@@ -1,83 +0,0 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / Job
# Class: Job
A handle to an operation that may still be running.
## Constructors
### new Job()
```ts
new Job(): Job
```
#### Returns
[`Job`](Job.md)
## Accessors
### id
```ts
get id(): null | string
```
Identifies the operation on the server that is running it. Operations
that run in this process have no server id. The value is opaque.
#### Returns
`null` \| `string`
## Methods
### cancel()
```ts
cancel(): Promise<void>
```
Request cancellation. Cancelling a finished operation is a no-op.
#### Returns
`Promise`&lt;`void`&gt;
***
### status()
```ts
status(): Promise<string>
```
The operation's current lifecycle state: "running", "finished",
"failed", or "cancelled".
A point snapshot; unlike [Job.wait](Job.md#wait) it does not block or reject
on a terminal failure state. States a newer server reports that this
client version does not know pass through as-is.
#### Returns
`Promise`&lt;`string`&gt;
***
### wait()
```ts
wait(): Promise<void>
```
Wait until the operation reaches a terminal state.
#### Returns
`Promise`&lt;`void`&gt;
+9 -8
View File
@@ -76,23 +76,24 @@ the query optimizer chooses a suboptimal path.
*** ***
### useLsm() ### useLsmWrite()
```ts ```ts
useLsm(enable): MergeInsertBuilder useLsmWrite(useLsmWrite): MergeInsertBuilder
``` ```
Control MemWAL routing for this merge. Controls whether the merge uses the MemWAL LSM write path.
By default (unset), a `mergeInsert` on a table with an LSM write spec is By default (unset), a `mergeInsert` on a table with an LSM write spec is
routed through Lance's MemWAL shard writer, and a table without one uses the routed through Lance's MemWAL shard writer, and a table without one uses
standard path. the standard path. Pass `false` to force the standard path even when a
spec is set. Pass `true` to require a spec — `mergeInsert` rejects if none
is installed.
#### Parameters #### Parameters
* **enable**: `boolean` * **useLsmWrite**: `boolean`
`true` forces MemWAL routing and errors if the table has no Whether to use the LSM write path.
LSM write spec. `false` forces the standard write path even when a spec is set.
#### Returns #### Returns
-36
View File
@@ -497,42 +497,6 @@ ArrowTable.
*** ***
### useLsm()
```ts
useLsm(enable): this
```
Control MemWAL read routing for this query.
By default (unset), when the table carries a MemWAL write spec (see
[Table#setLsmWriteSpec](Table.md#setlsmwritespec)), reads are routed through the LSM scanner so
they also return data written via the `mergeInsert` LSM path that has not yet
been compacted into the base table (the active/frozen in-memory memtables and
the flushed generations), deduplicated by primary key; a table without a spec
reads the base table.
#### Parameters
* **enable**: `boolean`
`true` forces the LSM scanner and errors if the table has no
MemWAL write spec. `false` bypasses the MemWAL and reads the base table only,
even when a spec is present.
Note: the LSM scanner does not support every query shape (e.g. reranking,
hybrid search, `orderBy`). On a MemWAL table those shapes error unless
`useLsm(false)` is set, because a base-only read would silently exclude
un-compacted MemWAL data.
#### Returns
`this`
#### Inherited from
`StandardQueryBase.useLsm`
***
### where() ### where()
```ts ```ts
+4 -121
View File
@@ -69,34 +69,14 @@ abstract addColumns(newColumnTransforms): Promise<AddColumnsResult>
Add new columns with defined values. Add new columns with defined values.
The `{ computed }` form stores the expression rather than evaluating it
now: the column is committed with no values, and rows get them from
[Table#refreshColumn](Table.md#refreshcolumn). Declaring one therefore costs the same on a
large table as on an empty one.
A refresh does not revisit rows it has already filled, so mutating an
input leaves the value computed at fill time; recomputing means dropping
the column and declaring it again. While a declaration reads a column,
that column cannot be renamed, retyped or dropped.
On LanceDB Cloud and Enterprise the expression is planned by the
server, and the refresh runs as a server job -- see
[Table#refreshColumnAsync](Table.md#refreshcolumnasync).
#### Parameters #### Parameters
* **newColumnTransforms**: * **newColumnTransforms**: `Field`&lt;`any`&gt; \| `Field`&lt;`any`&gt;[] \| `Schema`&lt;`any`&gt; \| [`AddColumnsSql`](../interfaces/AddColumnsSql.md)[]
\| `Field`&lt;`any`&gt;
\| `Field`&lt;`any`&gt;[]
\| `Schema`&lt;`any`&gt;
\| [`AddColumnsSql`](../interfaces/AddColumnsSql.md)[]
\| `object`
Either: Either:
- An array of objects with column names and SQL expressions to calculate values - An array of objects with column names and SQL expressions to calculate values
- A single Arrow Field defining one column with its data type (column will be initialized with null values) - A single Arrow Field defining one column with its data type (column will be initialized with null values)
- An array of Arrow Fields defining columns with their data types (columns will be initialized with null values) - An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
- An Arrow Schema defining columns with their data types (columns will be initialized with null values) - An Arrow Schema defining columns with their data types (columns will be initialized with null values)
- `{ computed }`, declaring columns defined by a SQL expression whose type and inputs are derived from it
#### Returns #### Returns
@@ -105,13 +85,6 @@ server, and the refresh runs as a server job -- see
A promise that resolves to an object A promise that resolves to an object
containing the new version number of the table after adding the columns. containing the new version number of the table after adding the columns.
#### Example
```ts
await table.addColumns({ computed: [{ name: "doubled", valueSql: "x * 2" }] });
const { rowsFilled } = await table.refreshColumn("doubled");
```
*** ***
### alterColumns() ### alterColumns()
@@ -322,29 +295,6 @@ await table.createIndex("my_float_col");
*** ***
### createIndexAsync()
```ts
abstract createIndexAsync(column, options?): Promise<Job>
```
Create an index, returning a handle to the indexing job.
The job may already be complete when returned; callers must not assume
the index exists until [Job.wait](Job.md#wait) resolves.
#### Parameters
* **column**: `string`
* **options?**: `Partial`&lt;[`IndexOptions`](../interfaces/IndexOptions.md)&gt;
#### Returns
`Promise`&lt;[`Job`](Job.md)&gt;
***
### currentBranch() ### currentBranch()
```ts ```ts
@@ -458,10 +408,9 @@ Read the [LsmWriteSpec](../interfaces/LsmWriteSpec.md) currently installed on th
Resolves to `undefined` when the MemWAL LSM write path is not enabled (no Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
spec has been set, or it was removed with [Table#unsetLsmWriteSpec](Table.md#unsetlsmwritespec)). spec has been set, or it was removed with [Table#unsetLsmWriteSpec](Table.md#unsetlsmwritespec)).
The returned spec mirrors what was passed to The returned spec — including its `maintainedIndexes` and
[Table#setLsmWriteSpec](Table.md#setlsmwritespec), except that `maintainedIndexes` always `writerConfigDefaults` — mirrors what was passed to
reports the concrete list resolved when the spec was set — `undefined` [Table#setLsmWriteSpec](Table.md#setlsmwritespec).
never round-trips.
#### Returns #### Returns
@@ -745,67 +694,6 @@ for await (const batch of table.query()) {
*** ***
### refreshColumn()
```ts
abstract refreshColumn(column): Promise<RefreshColumnResult>
```
Fill the rows of a computed column that hold no value yet.
Rows appended since the last refresh are filled by the next one; rows
already filled are left as they are, so the call is idempotent and does
not observe a mutated input. Local tables only: a remote refresh runs
as a server job, through [Table#refreshColumnAsync](Table.md#refreshcolumnasync).
#### Parameters
* **column**: `string`
The name of the computed column to fill.
#### Returns
`Promise`&lt;[`RefreshColumnResult`](../interfaces/RefreshColumnResult.md)&gt;
A promise that resolves to the
number of rows filled and the new version number of the table.
***
### refreshColumnAsync()
```ts
abstract refreshColumnAsync(column): Promise<Job>
```
Like [Table#refreshColumn](Table.md#refreshcolumn), but returns a handle to the refresh
job instead of blocking until it completes.
The job may already be complete when returned; callers must not assume
the column is filled until [Job.wait](Job.md#wait) resolves. Invalid input --
an unknown column, or one that is not computed -- rejects here rather
than failing the job. On local tables the job runs in-process; on
LanceDB Cloud and Enterprise it is the server's backfill job.
#### Parameters
* **column**: `string`
The name of the computed column to fill.
#### Returns
`Promise`&lt;[`Job`](Job.md)&gt;
#### Example
```ts
const job = await table.refreshColumnAsync("doubled");
await job.wait();
console.log(await job.status()); // "finished"
```
***
### restore() ### restore()
```ts ```ts
@@ -895,11 +783,6 @@ All variants require the table to have an unenforced primary key
([Table#setUnenforcedPrimaryKey](Table.md#setunenforcedprimarykey)); bucket sharding additionally ([Table#setUnenforcedPrimaryKey](Table.md#setunenforcedprimarykey)); bucket sharding additionally
requires it to be the single column being bucketed. requires it to be the single column being bucketed.
Omitting `maintainedIndexes` maintains every index on the table, resolved
here, failing if one cannot be maintained — name them to install anyway.
Naming them pins an exact set, and a still-building index is rejected
rather than quietly omitted.
#### Parameters #### Parameters
* **spec**: [`LsmWriteSpec`](../interfaces/LsmWriteSpec.md) * **spec**: [`LsmWriteSpec`](../interfaces/LsmWriteSpec.md)
-23
View File
@@ -273,29 +273,6 @@ ArrowTable.
*** ***
### useLsm()
```ts
useLsm(enable): this
```
Control MemWAL read routing for this take query.
`false` bypasses the MemWAL and reads the base table only — the escape hatch,
since take-by-row-id/offset is not supported on the LSM scanner and, on a
MemWAL table, auto-routes to it and errors otherwise.
#### Parameters
* **enable**: `boolean`
`false` reads the base table only.
#### Returns
`this`
***
### withRowId() ### withRowId()
```ts ```ts
-36
View File
@@ -746,42 +746,6 @@ ArrowTable.
*** ***
### useLsm()
```ts
useLsm(enable): this
```
Control MemWAL read routing for this query.
By default (unset), when the table carries a MemWAL write spec (see
[Table#setLsmWriteSpec](Table.md#setlsmwritespec)), reads are routed through the LSM scanner so
they also return data written via the `mergeInsert` LSM path that has not yet
been compacted into the base table (the active/frozen in-memory memtables and
the flushed generations), deduplicated by primary key; a table without a spec
reads the base table.
#### Parameters
* **enable**: `boolean`
`true` forces the LSM scanner and errors if the table has no
MemWAL write spec. `false` bypasses the MemWAL and reads the base table only,
even when a spec is present.
Note: the LSM scanner does not support every query shape (e.g. reranking,
hybrid search, `orderBy`). On a MemWAL table those shapes error unless
`useLsm(false)` is set, because a base-only read would silently exclude
un-compacted MemWAL data.
#### Returns
`this`
#### Inherited from
`StandardQueryBase.useLsm`
***
### where() ### where()
```ts ```ts
-5
View File
@@ -25,7 +25,6 @@
- [Connection](classes/Connection.md) - [Connection](classes/Connection.md)
- [HeaderProvider](classes/HeaderProvider.md) - [HeaderProvider](classes/HeaderProvider.md)
- [Index](classes/Index.md) - [Index](classes/Index.md)
- [Job](classes/Job.md)
- [MakeArrowTableOptions](classes/MakeArrowTableOptions.md) - [MakeArrowTableOptions](classes/MakeArrowTableOptions.md)
- [MatchQuery](classes/MatchQuery.md) - [MatchQuery](classes/MatchQuery.md)
- [MergeInsertBuilder](classes/MergeInsertBuilder.md) - [MergeInsertBuilder](classes/MergeInsertBuilder.md)
@@ -89,9 +88,6 @@
- [IvfFlatOptions](interfaces/IvfFlatOptions.md) - [IvfFlatOptions](interfaces/IvfFlatOptions.md)
- [IvfPqOptions](interfaces/IvfPqOptions.md) - [IvfPqOptions](interfaces/IvfPqOptions.md)
- [IvfRqOptions](interfaces/IvfRqOptions.md) - [IvfRqOptions](interfaces/IvfRqOptions.md)
- [JobDescription](interfaces/JobDescription.md)
- [JobFailureInfo](interfaces/JobFailureInfo.md)
- [JobInfo](interfaces/JobInfo.md)
- [ListNamespacesOptions](interfaces/ListNamespacesOptions.md) - [ListNamespacesOptions](interfaces/ListNamespacesOptions.md)
- [ListNamespacesResponse](interfaces/ListNamespacesResponse.md) - [ListNamespacesResponse](interfaces/ListNamespacesResponse.md)
- [LsmWriteSpec](interfaces/LsmWriteSpec.md) - [LsmWriteSpec](interfaces/LsmWriteSpec.md)
@@ -105,7 +101,6 @@
- [OptimizeOptions](interfaces/OptimizeOptions.md) - [OptimizeOptions](interfaces/OptimizeOptions.md)
- [OptimizeStats](interfaces/OptimizeStats.md) - [OptimizeStats](interfaces/OptimizeStats.md)
- [QueryExecutionOptions](interfaces/QueryExecutionOptions.md) - [QueryExecutionOptions](interfaces/QueryExecutionOptions.md)
- [RefreshColumnResult](interfaces/RefreshColumnResult.md)
- [RemovalStats](interfaces/RemovalStats.md) - [RemovalStats](interfaces/RemovalStats.md)
- [RenameTableOptions](interfaces/RenameTableOptions.md) - [RenameTableOptions](interfaces/RenameTableOptions.md)
- [RestNamespaceConfig](interfaces/RestNamespaceConfig.md) - [RestNamespaceConfig](interfaces/RestNamespaceConfig.md)
-28
View File
@@ -43,34 +43,6 @@ The following tokenizers are available:
*** ***
### blockSize?
```ts
optional blockSize: 128 | 256;
```
Number of documents per compressed posting block.
The default is 128. Supported values are 128 and 256. A value of 256 uses
the experimental FTS V3 format and may introduce breaking changes.
***
### customStopWords?
```ts
optional customStopWords: string[];
```
Custom stop words that replace the built-in list for `language`.
This option only affects tokenization when `removeStopWords` is true.
`undefined` keeps the built-in language list. An empty array explicitly
replaces it with no stop words.
***
### language? ### language?
```ts ```ts
-66
View File
@@ -1,66 +0,0 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / JobDescription
# Interface: JobDescription
A described job from `Connection.getJob`.
## Properties
### creationMs
```ts
creationMs: number;
```
When the job was created, in milliseconds since the epoch.
***
### failure?
```ts
optional failure: JobFailureInfo;
```
Why the job failed, when the job is failed and the server reports a
reason.
***
### jobId
```ts
jobId: string;
```
***
### jobType
```ts
jobType: string;
```
***
### specJson?
```ts
optional specJson: string;
```
The job-type-specific specification as a JSON string, when present.
***
### state
```ts
state: string;
```
Lifecycle state: "running", "finished", "failed", or "cancelled".
-33
View File
@@ -1,33 +0,0 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / JobFailureInfo
# Interface: JobFailureInfo
The server's account of why a job failed.
## Properties
### message?
```ts
optional message: string;
```
***
### phase?
```ts
optional phase: string;
```
***
### retryable?
```ts
optional retryable: boolean;
```
-58
View File
@@ -1,58 +0,0 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / JobInfo
# Interface: JobInfo
A row from `Connection.listJobs`: one server-side job.
## Properties
### createdAtMillis
```ts
createdAtMillis: number;
```
When the job was created, in milliseconds since the epoch.
***
### jobId
```ts
jobId: string;
```
The job id -- what `Connection.getJob` and `Connection.cancelJob`
accept.
***
### jobType
```ts
jobType: string;
```
***
### state
```ts
state: string;
```
Lifecycle state: "running", "finished", "failed", or "cancelled".
***
### table
```ts
table: string;
```
The table the job runs against, without URI or namespace.
+1 -3
View File
@@ -34,9 +34,7 @@ Bucket and identity variants: the sharding column.
optional maintainedIndexes: string[]; optional maintainedIndexes: string[];
``` ```
Indexes the MemWAL keeps up to date. Omit to maintain every supported Names of indexes the MemWAL should keep up to date during writes.
index, resolved on install — a snapshot, so indexes created later are not
maintained. Pass `[]` for none.
*** ***
@@ -1,23 +0,0 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / RefreshColumnResult
# Interface: RefreshColumnResult
## Properties
### rowsFilled
```ts
rowsFilled: number;
```
***
### version
```ts
version: number;
```
+1 -4
View File
@@ -44,7 +44,4 @@ The number of rows in the table
totalBytes: number; totalBytes: number;
``` ```
The total size, in bytes, of the table's data files, index files, and The total number of bytes in the table
overlay files
Read from the manifest, so this excludes deletion files and manifests.
-15
View File
@@ -30,21 +30,6 @@ The tokenizer to use. The default is "simple".
*** ***
### customStopWords?
```ts
optional customStopWords: string[];
```
Custom stop words that replace the built-in list for `language`.
This option only affects tokenization when `removeStopWords` is true.
`undefined` keeps the built-in language list. An empty array explicitly
replaces it with no stop words.
***
### language? ### language?
```ts ```ts
+52 -141
View File
@@ -26,18 +26,6 @@ is also an [asynchronous API client](#connections-asynchronous).
::: lancedb.db.DBConnection ::: lancedb.db.DBConnection
::: lancedb.Session
## Namespaces (Synchronous)
A namespace-backed connection resolves tables through a
[Lance namespace](https://lance-format.github.io/lance-namespace/) service instead of
listing a storage directory.
::: lancedb.connect_namespace
::: lancedb.namespace.LanceNamespaceDBConnection
## Tables (Synchronous) ## Tables (Synchronous)
::: lancedb.table.Table ::: lancedb.table.Table
@@ -46,12 +34,8 @@ listing a storage directory.
::: lancedb.table.FragmentSummaryStats ::: lancedb.table.FragmentSummaryStats
::: lancedb.table.TableStatistics
::: lancedb.table.Tags ::: lancedb.table.Tags
::: lancedb.table.Branches
## Expressions ## Expressions
Type-safe expression builder for filters and projections. Use these instead Type-safe expression builder for filters and projections. Use these instead
@@ -78,46 +62,29 @@ of raw SQL strings with [where][lancedb.query.LanceQueryBuilder.where] and
::: lancedb.query.LanceHybridQueryBuilder ::: lancedb.query.LanceHybridQueryBuilder
::: lancedb.query.LanceEmptyQueryBuilder
::: lancedb.query.LanceTakeQueryBuilder
## Full text queries
Structured full text queries can be passed to
[Table.search][lancedb.table.Table.search] or
[AsyncTable.search][lancedb.table.AsyncTable.search] in place of a query string,
and combined with [BooleanQuery][lancedb.query.BooleanQuery].
::: lancedb.query.FullTextQuery
::: lancedb.query.MatchQuery
::: lancedb.query.PhraseQuery
::: lancedb.query.BoostQuery
::: lancedb.query.MultiMatchQuery
::: lancedb.query.BooleanQuery
::: lancedb.query.FullTextOperator
::: lancedb.query.Occur
## Embeddings ## Embeddings
::: lancedb.embeddings ::: lancedb.embeddings.registry.EmbeddingFunctionRegistry
options:
show_root_heading: false ::: lancedb.embeddings.base.EmbeddingFunctionConfig
show_root_toc_entry: false
::: lancedb.embeddings.base.EmbeddingFunction
::: lancedb.embeddings.base.TextEmbeddingFunction
::: lancedb.embeddings.sentence_transformers.SentenceTransformerEmbeddings
::: lancedb.embeddings.openai.OpenAIEmbeddings
::: lancedb.embeddings.open_clip.OpenClipEmbeddings
## Remote configuration ## Remote configuration
::: lancedb.remote ::: lancedb.remote.ClientConfig
options:
show_root_heading: false ::: lancedb.remote.TimeoutConfig
show_root_toc_entry: false
::: lancedb.remote.RetryConfig
## Context ## Context
@@ -127,50 +94,11 @@ and combined with [BooleanQuery][lancedb.query.BooleanQuery].
## Full text search ## Full text search
Pass `custom_stop_words` to [lancedb.index.FTS][]: Use [lancedb.table.Table.create_fts_index][] for the synchronous API or
[lancedb.table.AsyncTable.create_index][] with [lancedb.index.FTS][] for the
asynchronous API.
```python ::: lancedb.index.FTS
from lancedb.index import FTS
table.create_index(
"text",
config=FTS(remove_stop_words=True, custom_stop_words=["acme", "internal"]),
)
```
The list replaces the built-in stop words and is used only when
`remove_stop_words=True`:
- `custom_stop_words=None` uses the built-in list for `language`.
- `custom_stop_words=[]` removes no words.
- Values are passed through without trimming, lowercasing, or other rewriting.
The same option is available on `lancedb.tokenize(...)` and the deprecated
[lancedb.table.Table.create_fts_index][] compatibility helper:
```python
import lancedb
tokens = list(lancedb.tokenize("acme makes searchable data",
custom_stop_words=["acme"]))
```
::: lancedb.tokenize
::: lancedb.FtsToken
## Blobs
Blob columns store large binary values out of line so they can be read lazily
instead of being materialized with the rest of the row.
::: lancedb.blob
::: lancedb.BlobType
::: lancedb._blob.BlobFile
options:
show_root_full_path: false
## Utilities ## Utilities
@@ -178,14 +106,6 @@ instead of being materialized with the rest of the row.
::: lancedb.merge.LanceMergeInsertBuilder ::: lancedb.merge.LanceMergeInsertBuilder
::: lancedb.otel.instrument_lancedb_metrics
## Exceptions
::: lancedb.exceptions.MissingValueError
::: lancedb.exceptions.MissingColumnError
## Integrations ## Integrations
## Pydantic ## Pydantic
@@ -194,30 +114,19 @@ instead of being materialized with the rest of the row.
::: lancedb.pydantic.vector ::: lancedb.pydantic.vector
::: lancedb.pydantic.Vector
::: lancedb.pydantic.MultiVector
::: lancedb.pydantic.LanceModel ::: lancedb.pydantic.LanceModel
## PyTorch
::: lancedb.streaming.StreamingDataset
::: lancedb.permutation.permutation_builder
::: lancedb.permutation.PermutationBuilder
::: lancedb.permutation.Permutation
::: lancedb.permutation.Transforms
## Reranking ## Reranking
::: lancedb.rerankers ::: lancedb.rerankers.linear_combination.LinearCombinationReranker
options:
show_root_heading: false ::: lancedb.rerankers.cohere.CohereReranker
show_root_toc_entry: false
::: lancedb.rerankers.colbert.ColbertReranker
::: lancedb.rerankers.cross_encoder.CrossEncoderReranker
::: lancedb.rerankers.openai.OpenaiReranker
## Connections (Asynchronous) ## Connections (Asynchronous)
@@ -228,12 +137,6 @@ can be used to create, list, or open tables.
::: lancedb.db.AsyncConnection ::: lancedb.db.AsyncConnection
## Namespaces (Asynchronous)
::: lancedb.connect_namespace_async
::: lancedb.namespace.AsyncLanceNamespaceDBConnection
## Tables (Asynchronous) ## Tables (Asynchronous)
Table hold your actual data as a collection of records / rows. Table hold your actual data as a collection of records / rows.
@@ -242,20 +145,32 @@ Table hold your actual data as a collection of records / rows.
::: lancedb.table.AsyncTags ::: lancedb.table.AsyncTags
::: lancedb.table.AsyncBranches
## Indices (Asynchronous) ## Indices (Asynchronous)
Indices can be created on a table to speed up queries. This section Indices can be created on a table to speed up queries. This section
lists the indices that LanceDb supports. lists the indices that LanceDb supports.
::: lancedb.index ::: lancedb.index.BTree
options:
show_root_heading: false ::: lancedb.index.Bitmap
show_root_toc_entry: false
# `lang_mapping` is defined in the module rather than imported, so it is ::: lancedb.index.LabelList
# picked up despite not being in `__all__`. It is an internal lookup table.
filters: ["!^_", "!^lang_mapping$"] ::: lancedb.index.FTS
::: lancedb.index.IvfPq
::: lancedb.index.HnswPq
::: lancedb.index.HnswSq
::: lancedb.index.IvfFlat
::: lancedb.index.IvfSq
::: lancedb.index.IvfRq
::: lancedb.index.HnswFlat
::: lancedb.table.IndexStatistics ::: lancedb.table.IndexStatistics
@@ -283,7 +198,3 @@ rows nearest to a query vector and can be created with the
::: lancedb.query.AsyncHybridQuery ::: lancedb.query.AsyncHybridQuery
options: options:
inherited_members: true inherited_members: true
::: lancedb.query.AsyncTakeQuery
options:
inherited_members: true
+1 -1
View File
@@ -8,7 +8,7 @@
<parent> <parent>
<groupId>com.lancedb</groupId> <groupId>com.lancedb</groupId>
<artifactId>lancedb-parent</artifactId> <artifactId>lancedb-parent</artifactId>
<version>0.38.0-beta.0</version> <version>0.33.0-beta.0</version>
<relativePath>../pom.xml</relativePath> <relativePath>../pom.xml</relativePath>
</parent> </parent>
+2 -2
View File
@@ -6,7 +6,7 @@
<groupId>com.lancedb</groupId> <groupId>com.lancedb</groupId>
<artifactId>lancedb-parent</artifactId> <artifactId>lancedb-parent</artifactId>
<version>0.38.0-beta.0</version> <version>0.33.0-beta.0</version>
<packaging>pom</packaging> <packaging>pom</packaging>
<name>${project.artifactId}</name> <name>${project.artifactId}</name>
<description>LanceDB Java SDK Parent POM</description> <description>LanceDB Java SDK Parent POM</description>
@@ -28,7 +28,7 @@
<properties> <properties>
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding> <project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
<arrow.version>15.0.0</arrow.version> <arrow.version>15.0.0</arrow.version>
<lance-core.version>11.0.0-beta.13</lance-core.version> <lance-core.version>9.0.0</lance-core.version>
<spotless.skip>false</spotless.skip> <spotless.skip>false</spotless.skip>
<spotless.version>2.30.0</spotless.version> <spotless.version>2.30.0</spotless.version>
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version> <spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
+1 -1
View File
@@ -1,7 +1,7 @@
# Contributing to LanceDB Typescript # Contributing to LanceDB Typescript
This document outlines the process for contributing to LanceDB Typescript. This document outlines the process for contributing to LanceDB Typescript.
For general contribution guidelines, see [CONTRIBUTING.md](https://github.com/lancedb/lancedb/blob/main/CONTRIBUTING.md). For general contribution guidelines, see [CONTRIBUTING.md](../CONTRIBUTING.md).
## Project layout ## Project layout
+1 -1
View File
@@ -1,7 +1,7 @@
[package] [package]
name = "lancedb-nodejs" name = "lancedb-nodejs"
edition.workspace = true edition.workspace = true
version = "0.38.0-beta.0" version = "0.33.0-beta.0"
publish = false publish = false
license.workspace = true license.workspace = true
description.workspace = true description.workspace = true
-175
View File
@@ -6,9 +6,7 @@ import * as arrow17 from "apache-arrow-17";
import * as arrow18 from "apache-arrow-18"; import * as arrow18 from "apache-arrow-18";
import { import {
Vector as CurrentVector,
convertToTable, convertToTable,
tableFromIPC as currentTableFromIPC,
fromBufferToRecordBatch, fromBufferToRecordBatch,
fromDataToBuffer, fromDataToBuffer,
fromRecordBatchToBuffer, fromRecordBatchToBuffer,
@@ -21,7 +19,6 @@ import {
FunctionOptions, FunctionOptions,
} from "../lancedb/embedding/embedding_function"; } from "../lancedb/embedding/embedding_function";
import { EmbeddingFunctionConfig } from "../lancedb/embedding/registry"; import { EmbeddingFunctionConfig } from "../lancedb/embedding/registry";
import { sanitizeTable } from "../lancedb/sanitize";
// biome-ignore lint/suspicious/noExplicitAny: skip // biome-ignore lint/suspicious/noExplicitAny: skip
function sampleRecords(): Array<Record<string, any>> { function sampleRecords(): Array<Record<string, any>> {
@@ -67,11 +64,7 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
tableFromIPC, tableFromIPC,
DataType, DataType,
Dictionary, Dictionary,
RecordBatch: ArrowRecordBatch,
Table: ArrowTable,
Uint8: ArrowUint8, Uint8: ArrowUint8,
makeData: arrowMakeData,
vectorFromArray,
// biome-ignore lint/suspicious/noExplicitAny: <explanation> // biome-ignore lint/suspicious/noExplicitAny: <explanation>
} = <any>arrow; } = <any>arrow;
type Schema = ApacheArrow["Schema"]; type Schema = ApacheArrow["Schema"];
@@ -204,35 +197,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
expect(table.getChild("d")?.toJSON()).toEqual([9n, 10n, null]); expect(table.getChild("d")?.toJSON()).toEqual([9n, 10n, null]);
}); });
it("will use a provided FixedSizeList schema with typed array values", function () {
const schema = new Schema([
new Field("text", new Utf8(), false),
new Field(
"vector",
new FixedSizeList(3, new Field("item", new Float32(), false)),
false,
),
]);
const table = makeArrowTable(
[
{
text: "foo",
vector: new Float32Array([1, 2, 3]),
},
],
{ schema },
);
expect(table.getChild("text")?.toJSON()).toEqual(["foo"]);
expect(
table
.getChild("vector")
?.toJSON()
.map((value) => value.toJSON()),
).toEqual([[1, 2, 3]]);
});
it("will assume the column `vector` is FixedSizeList<Float32> by default", async function () { it("will assume the column `vector` is FixedSizeList<Float32> by default", async function () {
const schema = new Schema([ const schema = new Schema([
new Field("a", new Float(Precision.DOUBLE), true), new Field("a", new Float(Precision.DOUBLE), true),
@@ -1027,148 +991,9 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
expectValidMapField(roundTripped.schema.fields[0]); expectValidMapField(roundTripped.schema.fields[0]);
}); });
it("preserves string schema metadata", function () {
const metadata = new Map([["source", "fixture"]]);
const schema = new Schema(
[new Field("value", new Int32(), true)],
metadata,
);
expect(makeEmptyTable(schema).schema.metadata.get("source")).toBe(
"fixture",
);
});
it.each([
["non-string keys", new Map<unknown, unknown>([[42, "fixture"]])],
["non-string values", new Map<unknown, unknown>([["source", 42]])],
[
"non-string keys and values",
new Map<unknown, unknown>([[42, false]]),
],
])("rejects schema metadata with %s", function (_, metadataLike) {
const metadata = metadataLike as unknown as Map<string, string>;
const schema = new Schema(
[new Field("value", new Int32(), true)],
metadata,
);
expect(() => makeEmptyTable(schema)).toThrow(
"Expected metadata, if present, to be a Map<string, string> but it had non-string keys or values",
);
});
}); });
describe("when using two versions of arrow", function () { describe("when using two versions of arrow", function () {
it("preserves a dictionary shared by multiple fields", async function () {
const values = ["alpha", "beta", "alpha"];
const dictionaryVector = vectorFromArray(values);
const batch = new ArrowRecordBatch({
first: dictionaryVector.data[0],
second: dictionaryVector.data[0],
});
const table = new ArrowTable([batch]);
const sanitized = sanitizeTable(table);
expect([...sanitized.getChild("first")!]).toEqual(values);
expect([...sanitized.getChild("second")!]).toEqual(values);
const firstType = sanitized.schema.fields[0].type as {
dictionary: unknown;
};
const secondType = sanitized.schema.fields[1].type as {
dictionary: unknown;
};
expect(secondType.dictionary).toBe(firstType.dictionary);
expect(sanitized.batches[0].data.children[1].dictionary).toBe(
sanitized.batches[0].data.children[0].dictionary,
);
const buf = await fromDataToBuffer(table);
const actual = currentTableFromIPC(buf);
expect([...actual.getChild("first")!]).toEqual(values);
expect([...actual.getChild("second")!]).toEqual(values);
});
it("preserves shared dictionary data from another Arrow version", async function () {
const values = ["alpha", "beta", "alpha"];
const dictionaryVector = vectorFromArray(values);
const firstBatch = new ArrowRecordBatch({
label: dictionaryVector.slice(0, 2).data[0],
});
const secondBatch = new ArrowRecordBatch({
label: dictionaryVector.slice(2).data[0],
});
const table = new ArrowTable([firstBatch, secondBatch]);
const sanitized = sanitizeTable(table);
expect([...sanitized.getChild("label")!]).toEqual(values);
const dictionaries = sanitized.batches.map(
(batch) => batch.data.children[0].dictionary,
);
expect(dictionaries[0]).toBeInstanceOf(CurrentVector);
expect(dictionaries[1]).toBe(dictionaries[0]);
const buf = await fromDataToBuffer(table);
const actual = currentTableFromIPC(buf);
expect([...actual.getChild("label")!]).toEqual(values);
});
it("preserves shared chunks in growing dictionaries", async function () {
const type = new Dictionary(new Utf8(), new Int32(), 42, false);
const firstDictionary = vectorFromArray(["alpha", "beta"], new Utf8());
const secondDictionary = firstDictionary.concat(
vectorFromArray(["gamma"], new Utf8()),
);
const firstData = arrowMakeData({
type,
data: Int32Array.from([0, 1]),
dictionary: firstDictionary,
});
const secondData = arrowMakeData({
type,
data: Int32Array.from([2]),
dictionary: secondDictionary,
});
const table = new ArrowTable([
new ArrowRecordBatch({ label: firstData }),
new ArrowRecordBatch({ label: secondData }),
]);
const sanitized = sanitizeTable(table);
const expected = ["alpha", "beta", "gamma"];
expect([...sanitized.getChild("label")!]).toEqual(expected);
const firstLocalDictionary =
sanitized.batches[0].data.children[0].dictionary!;
const secondLocalDictionary =
sanitized.batches[1].data.children[0].dictionary!;
expect(secondLocalDictionary.data[0]).toBe(
firstLocalDictionary.data[0],
);
const buf = await fromTableToBuffer(sanitized);
const actual = currentTableFromIPC(buf);
expect([...actual.getChild("label")!]).toEqual(expected);
});
it("can serialize list data from another Arrow version", async function () {
const values = [["anime", "action"], [], null];
const vector = vectorFromArray(
values,
new List(new Field("item", new Utf8(), true)),
);
const table = new ArrowTable({ tags: vector });
const buf = await fromDataToBuffer(table);
const actual = currentTableFromIPC(buf);
const actualTags = actual.getChild("tags");
expect(actualTags?.get(0)?.toJSON()).toEqual(values[0]);
expect(actualTags?.get(1)?.toJSON()).toEqual(values[1]);
expect(actualTags?.get(2)).toBeNull();
});
it("can still import data", async function () { it("can still import data", async function () {
const schema = new arrow15.Schema([ const schema = new arrow15.Schema([
new arrow15.Field("id", new arrow15.Int32()), new arrow15.Field("id", new arrow15.Int32()),
-10
View File
@@ -89,16 +89,6 @@ describe("given a connection", () => {
await db.createTable("test4", [{ id: 1 }, { id: 2 }]); await db.createTable("test4", [{ id: 1 }, { id: 2 }]);
}); });
it("should return a completed job when dropping a local table", async () => {
await db.createTable("async-drop", [{ id: 1 }]);
const job = await db.dropTableAsync("async-drop");
expect(job.id).toBeNull();
await expect(job.status()).resolves.toBe("finished");
await job.wait();
await expect(db.tableNames()).resolves.toEqual([]);
});
it("should fail if creating table twice, unless overwrite is true", async () => { it("should fail if creating table twice, unless overwrite is true", async () => {
let tbl = await db.createTable("test", [{ id: 1 }, { id: 2 }]); let tbl = await db.createTable("test", [{ id: 1 }, { id: 2 }]);
await expect(tbl.countRows()).resolves.toBe(2); await expect(tbl.countRows()).resolves.toBe(2);
-60
View File
@@ -11,11 +11,8 @@ import {
Float16, Float16,
Float32, Float32,
Float64, Float64,
Int32,
Schema, Schema,
Utf8, Utf8,
fromDataToBuffer,
tableFromIPC,
} from "../lancedb/arrow"; } from "../lancedb/arrow";
import { EmbeddingFunction, LanceSchema } from "../lancedb/embedding"; import { EmbeddingFunction, LanceSchema } from "../lancedb/embedding";
import { getRegistry, register } from "../lancedb/embedding/registry"; import { getRegistry, register } from "../lancedb/embedding/registry";
@@ -187,63 +184,6 @@ describe("embedding functions", () => {
const vector0 = JSON.parse(JSON.stringify(arr[0].vector)); const vector0 = JSON.parse(JSON.stringify(arr[0].vector));
expect(vector0).toEqual([1, 2, 3]); expect(vector0).toEqual([1, 2, 3]);
}); });
it("should append generated vectors to a non-nullable schema", async () => {
@register("non_nullable_schema_test")
class MockEmbeddingFunction extends EmbeddingFunction<string> {
ndims() {
return 3;
}
embeddingDataType(): Float {
return new Float64();
}
async computeSourceEmbeddings(data: string[]) {
return data.map(() => [1, 2, 3]);
}
}
const schema = new Schema([
new Field("id", new Int32()),
new Field("text", new Utf8()),
new Field("type", new Utf8()),
new Field(
"vector",
new FixedSizeList(3, new Field("item", new Float64())),
),
]);
const func = new MockEmbeddingFunction();
const db = await connect(tmpDir.name);
const table = await db.createEmptyTable("test_non_nullable", schema, {
embeddingFunction: {
function: func,
sourceColumn: "text",
},
});
const data = [
{ id: 1, text: "Carrot", type: "vegetable" },
{ id: 2, text: "Apple", type: "fruit" },
];
const buffer = await fromDataToBuffer(
data,
undefined,
await table.schema(),
);
const generatedTable = tableFromIPC(buffer);
const vectorField = generatedTable.schema.fields.find(
(field) => field.name === "vector",
);
expect(vectorField?.nullable).toBe(false);
await table.add(data);
const rows = await table.query().toArray();
expect(rows).toHaveLength(2);
for (const row of rows) {
expect([...row.vector]).toEqual([1, 2, 3]);
}
});
it("should error when appending to a table with an unregistered embedding function", async () => { it("should error when appending to a table with an unregistered embedding function", async () => {
@register("mock") @register("mock")
class MockEmbeddingFunction extends EmbeddingFunction<string> { class MockEmbeddingFunction extends EmbeddingFunction<string> {
-14
View File
@@ -1,14 +0,0 @@
// SPDX-License-Identifier: Apache-2.0
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
import packageJson = require("../package.json");
describe("package metadata", () => {
it("requires Node.js type declarations compatible with the runtime", () => {
expect(packageJson.engines.node).toBe(">= 18");
expect(packageJson.peerDependencies["@types/node"]).toBe(">=18");
expect(packageJson.peerDependenciesMeta["@types/node"]).toEqual({
optional: true,
});
});
});
-75
View File
@@ -110,81 +110,6 @@ describe("Query outputSchema", () => {
}); });
}); });
describe("Search pagination", () => {
let tmpDir: tmp.DirResult;
let table: Table;
beforeEach(async () => {
tmpDir = tmp.dirSync({ unsafeCleanup: true });
const db = await connect(tmpDir.name);
const schema = new Schema([
new Field("id", new Int64(), false),
new Field("text", new Utf8(), false),
new Field(
"vector",
new FixedSizeList(2, new Field("item", new Float32())),
false,
),
]);
const data = makeArrowTable(
[
{ id: 1n, text: "common", vector: [0, 0] },
{ id: 2n, text: "common common", vector: [1, 1] },
{ id: 3n, text: "common common common", vector: [2, 2] },
{ id: 4n, text: "common common common common", vector: [3, 3] },
],
{ schema },
);
table = await db.createTable("test", data);
});
afterEach(() => {
tmpDir.removeCallback();
});
it("applies offset after the vector search limit", async () => {
const allResults = await table
.vectorSearch([0, 0])
.select(["id"])
.limit(4)
.toArray();
const secondPage = await table
.vectorSearch([0, 0])
.select(["id"])
.limit(2)
.offset(2)
.toArray();
expect(allResults).toHaveLength(4);
expect(secondPage).toHaveLength(2);
expect(secondPage.map((row) => row.id)).toEqual(
allResults.slice(2, 4).map((row) => row.id),
);
});
it("applies offset after the full-text search limit", async () => {
await table.createIndex("text", { config: Index.fts() });
const allResults = await table
.search("common", "fts")
.select(["id"])
.limit(4)
.toArray();
const secondPage = await table
.search("common", "fts")
.select(["id"])
.limit(2)
.offset(2)
.toArray();
expect(allResults).toHaveLength(4);
expect(secondPage).toHaveLength(2);
expect(secondPage.map((row) => row.id)).toEqual(
allResults.slice(2, 4).map((row) => row.id),
);
});
});
describe("Query orderBy", () => { describe("Query orderBy", () => {
let tmpDir: tmp.DirResult; let tmpDir: tmp.DirResult;
let table: Table; let table: Table;
-224
View File
@@ -15,7 +15,6 @@ import {
OAuthHeaderProvider, OAuthHeaderProvider,
StaticHeaderProvider, StaticHeaderProvider,
} from "../lancedb/header"; } from "../lancedb/header";
import { Index } from "../lancedb/indices";
// Test-only header providers // Test-only header providers
class CustomProvider extends HeaderProvider { class CustomProvider extends HeaderProvider {
@@ -170,38 +169,6 @@ describe("remote connection", () => {
); );
}); });
it("surfaces JSON server errors from remote table operations", async () => {
await withMockDatabase(
(req, res) => {
const path = req.url ?? "";
if (path.endsWith("/describe/")) {
res.writeHead(200, { "Content-Type": "application/json" }).end(
JSON.stringify({
name: "broken_table",
version: 1,
schema: { fields: [] },
}),
);
return;
}
if (path.endsWith("/count_rows/")) {
res
.writeHead(400, { "Content-Type": "application/json" })
.end(JSON.stringify({ error: "count rows failed" }));
return;
}
res.writeHead(404).end();
},
async (db) => {
const table = await db.openTable("broken_table");
await expect(table.countRows()).rejects.toThrow("count rows failed");
},
);
});
it("should pass on requested extra headers", async () => { it("should pass on requested extra headers", async () => {
await withMockDatabase( await withMockDatabase(
(req, res) => { (req, res) => {
@@ -258,59 +225,6 @@ describe("remote connection", () => {
); );
}); });
it("sends FTS options to remote tables", async () => {
let createIndexBody: Record<string, unknown> | undefined;
await withMockDatabase(
(req, res) => {
const path = req.url ?? "";
if (path.endsWith("/describe/")) {
res.writeHead(200, { "Content-Type": "application/json" }).end(
JSON.stringify({
name: "t",
version: 1,
schema: {
fields: [
{ name: "text", type: { type: "string" }, nullable: false },
],
},
}),
);
return;
}
if (path.endsWith("/create_index/")) {
let raw = "";
req.on("data", (chunk) => {
raw += chunk;
});
req.on("end", () => {
createIndexBody = JSON.parse(raw);
res.writeHead(200).end();
});
return;
}
res.writeHead(404).end();
},
async (db) => {
const table = await db.openTable("t");
await table.createIndex("text", {
config: Index.fts({
blockSize: 256,
removeStopWords: true,
customStopWords: ["the"],
}),
});
},
);
expect(createIndexBody?.["column"]).toBe("text");
expect(createIndexBody?.["index_type"]).toBe("FTS");
expect(createIndexBody?.["block_size"]).toBe(256);
expect(createIndexBody?.["custom_stop_words"]).toEqual(["the"]);
});
it("diffs and merges remote branches", async () => { it("diffs and merges remote branches", async () => {
const sampleDiff = { const sampleDiff = {
fromBranch: "exp", fromBranch: "exp",
@@ -909,141 +823,3 @@ describe("remote connection", () => {
}); });
}); });
}); });
describe("remote connection jobs surface", () => {
it("lists, describes, cancels, and reads history", async () => {
const { tableFromArrays, tableToIPC } = await import("apache-arrow");
const eventsTable = tableFromArrays({ state: ["created", "succeeded"] });
const eventsBody = Buffer.from(tableToIPC(eventsTable, "stream"));
await withMockDatabase(
(req, res) => {
let body = "";
req.on("data", (chunk) => {
body += chunk;
});
req.on("end", () => {
const payload = body.length > 0 ? JSON.parse(body) : {};
if (req.url === "/v1/jobs/list") {
if (payload["page_token"] === undefined) {
res
.writeHead(200, { "Content-Type": "application/json" })
.end(
'{"jobs": [{"job_id": "job-1", "table": "t1", ' +
'"job_type": "create_index", "state": "in_progress", ' +
'"created_at_millis": 1000}], "page_token": "next"}',
);
} else {
res
.writeHead(200, { "Content-Type": "application/json" })
.end(
'{"jobs": [{"job_id": "job-2", "table": "t2", ' +
'"job_type": "create_index", "state": "succeeded", ' +
'"created_at_millis": 2000}]}',
);
}
} else if (req.url === "/v1/jobs/describe") {
if (payload["job_id"] !== "job-1") {
res.writeHead(404).end("no such job");
return;
}
res
.writeHead(200, { "Content-Type": "application/json" })
.end(
'{"job_id": "job-1", "job_type": "create_index", ' +
'"job_state": "FAILED", "creation_ms": 1000, ' +
'"spec": {"column": "vec"}, "failure": {"phase": "execute", ' +
'"message": "worker died", "retryable": true}}',
);
} else if (req.url === "/v1/jobs/cancel") {
if (payload["job_id"] !== "job-1") {
res.writeHead(404).end("no such job");
return;
}
res
.writeHead(200, { "Content-Type": "application/json" })
.end('{"job_id": "job-1"}');
} else if (req.url === "/v1/jobs/query_events") {
res
.writeHead(200, {
"Content-Type": "application/vnd.apache.arrow.stream",
})
.end(eventsBody);
} else {
res.writeHead(404).end();
}
});
},
async (db) => {
const jobs = await db.listJobs();
expect(jobs.map((job) => job.jobId)).toEqual(["job-1", "job-2"]);
expect(jobs[0].state).toEqual("running");
expect(jobs[1].state).toEqual("finished");
const description = await db.getJob("job-1");
expect(description?.state).toEqual("failed");
expect(JSON.parse(description?.specJson ?? "")).toEqual({
column: "vec",
});
expect(description?.failure?.message).toEqual("worker died");
expect(await db.getJob("missing")).toBeNull();
expect(await db.cancelJob("job-1")).toBe(true);
expect(await db.cancelJob("missing")).toBe(false);
const history = await db.jobHistory("job-1");
expect(history.numRows).toEqual(2);
const job = db.job("job-1");
expect(job.id).toEqual("job-1");
expect(await job.status()).toEqual("failed");
await expect(job.wait()).rejects.toThrow("worker died");
},
);
});
it("addBases posts the bases array", async () => {
const postedBodies: unknown[] = [];
await withMockDatabase(
(req, res) => {
const path = req.url ?? "";
if (path.endsWith("/describe/")) {
res.writeHead(200, { "Content-Type": "application/json" }).end(
JSON.stringify({
name: "photos",
version: 1,
schema: { fields: [] },
}),
);
return;
}
if (path.endsWith("/bases/")) {
const chunks: Buffer[] = [];
req.on("data", (chunk) => chunks.push(chunk));
req.on("end", () => {
postedBodies.push(JSON.parse(Buffer.concat(chunks).toString()));
res
.writeHead(200, { "Content-Type": "application/json" })
.end(JSON.stringify({ version: 2 }));
});
return;
}
res.writeHead(404).end();
},
async (db) => {
const table = await db.openTable("photos");
await table.addBases({ path: "s3://bucket/media/" });
},
);
expect(postedBodies).toEqual([
{
bases: [
{
path: "s3://bucket/media/",
isDatasetRoot: false,
},
],
},
]);
});
});
+4 -216
View File
@@ -4,7 +4,6 @@
import * as fs from "fs"; import * as fs from "fs";
import * as path from "path"; import * as path from "path";
import * as tmp from "tmp"; import * as tmp from "tmp";
import { pathToFileURL } from "url";
import * as arrow15 from "apache-arrow-15"; import * as arrow15 from "apache-arrow-15";
import * as arrow16 from "apache-arrow-16"; import * as arrow16 from "apache-arrow-16";
@@ -87,44 +86,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
await expect(table.countRows()).resolves.toBe(3); await expect(table.countRows()).resolves.toBe(3);
}); });
it("should support a foreign Float64 vector schema end to end", async () => {
const conn = await connect(tmpDir.name);
const schema = new arrow.Schema([
new arrow.Field("resource_id", new arrow.Int32(), false),
new arrow.Field(
"vector",
new arrow.FixedSizeList(
3,
new arrow.Field("value", new arrow.Float64(), true),
),
false,
),
]);
const data = [
{
// biome-ignore lint/style/useNamingConvention: matches the reported schema
resource_id: 0,
vector: [0.1, 0.1, 0.1],
},
];
const resources = await conn.createTable("resources", data, { schema });
const existing = await resources
.query()
.where("resource_id = 0")
.limit(1)
.toArray();
expect(existing).toHaveLength(1);
const matched = await resources
.search(Float64Array.from(data[0].vector))
.limit(1)
.toArray();
expect(matched).toHaveLength(1);
expect(matched[0]["resource_id"]).toBe(0);
});
it("should support branches", async () => { it("should support branches", async () => {
await table.add([{ id: 1 }]); await table.add([{ id: 1 }]);
expect(await table.countRows()).toBe(1); expect(await table.countRows()).toBe(1);
@@ -278,16 +239,8 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
}, },
numIndices: 0, numIndices: 0,
numRows: 3, numRows: 3,
// Full on-disk size of the two data files, footers and metadata included. totalBytes: 44,
totalBytes: 684,
}); });
// Index files count toward totalBytes too (only deletion files and
// manifests are excluded).
await table.createIndex("id", { config: Index.btree() });
const statsWithIndex = await table.stats();
expect(statsWithIndex.numIndices).toBe(1);
expect(statsWithIndex.totalBytes).toBeGreaterThan(684);
}); });
it("should overwrite data if asked", async () => { it("should overwrite data if asked", async () => {
@@ -574,14 +527,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
); );
}); });
it("should expose useLsm on takeRowIds as the base-only escape hatch", async () => {
await table.add([{ id: 1 }, { id: 2 }, { id: 3 }]);
// useLsm(false) is reachable on TakeQuery (the escape hatch for MemWAL tables,
// where take-by-row-id auto-routes to the LSM scanner and is rejected).
const res = await table.takeRowIds([0, 2]).useLsm(false).toArray();
expect(res.map((r) => r.id)).toEqual([1, 3]);
});
it("should throw for negative number in takeRowIds", () => { it("should throw for negative number in takeRowIds", () => {
expect(() => table.takeRowIds([-1])).toThrow("Row id cannot be negative"); expect(() => table.takeRowIds([-1])).toThrow("Row id cannot be negative");
expect(() => table.takeRowIds([0, -5, 2])).toThrow( expect(() => table.takeRowIds([0, -5, 2])).toThrow(
@@ -898,11 +843,7 @@ describe("When creating an index", () => {
afterEach(() => tmpDir.removeCallback()); afterEach(() => tmpDir.removeCallback());
it("should create a vector index on vector columns", async () => { it("should create a vector index on vector columns", async () => {
const job = await tbl.createIndexAsync("vec"); await tbl.createIndex("vec");
expect(job.id).toBeNull();
await job.wait();
// Cancelling a job that already finished succeeds and does nothing.
await job.cancel();
// check index directory // check index directory
const indexDir = path.join(tmpDir.name, "test.lance", "_indices"); const indexDir = path.join(tmpDir.name, "test.lance", "_indices");
@@ -2586,35 +2527,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
expect(results3.length).toBe(1); expect(results3.length).toBe(1);
}); });
test("full text search with custom posting block size", async () => {
const db = await connect(tmpDir.name);
const data = [
{ text: "hello world", vector: [0.1, 0.2, 0.3] },
{ text: "goodbye world", vector: [0.4, 0.5, 0.6] },
];
const table = await db.createTable("test", data);
await table.createIndex("text", {
config: Index.fts({ blockSize: 256 }),
});
const index = (await table.listIndices()).find(
(index) => index.indexType === "FTS",
);
expect(index?.indexVersion).toBe(3);
expect(
(index?.indexDetails as Record<string, unknown>)["block_size"],
).toBe(256);
const results = await table.search("hello").toArray();
expect(results[0].text).toBe(data[0].text);
});
test("rejects invalid full text posting block size", () => {
expect(() => Index.fts({ blockSize: 129 as 128 | 256 })).toThrow(
"128 or 256",
);
});
test("full text search without lowercase", async () => { test("full text search without lowercase", async () => {
const db = await connect(tmpDir.name); const db = await connect(tmpDir.name);
const data = [ const data = [
@@ -2820,15 +2732,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
}, },
); );
test("tokenize supports custom stop words", async () => {
const tokens = await tokenize("the lance data", {
stem: false,
removeStopWords: true,
customStopWords: ["lance"],
});
expect(tokens.map((token) => token.text)).toEqual(["the", "data"]);
});
describe("when calling explainPlan", () => { describe("when calling explainPlan", () => {
let tmpDir: tmp.DirResult; let tmpDir: tmp.DirResult;
let table: Table; let table: Table;
@@ -3267,14 +3170,14 @@ describe("LSM merge insert", () => {
await table.closeLsmWriters(); await table.closeLsmWriters();
}); });
it("falls back to the standard path with useLsm(false)", async () => { it("falls back to the standard path with useLsmWrite(false)", async () => {
const conn = await connect(tmpDir.name); const conn = await connect(tmpDir.name);
const table = await bucketTable(conn); const table = await bucketTable(conn);
const res = await table const res = await table
.mergeInsert("id") .mergeInsert("id")
.whenNotMatchedInsertAll() .whenNotMatchedInsertAll()
.useLsm(false) .useLsmWrite(false)
.execute([ .execute([
{ id: "b", value: 9 }, { id: "b", value: 9 },
{ id: "e", value: 5 }, { id: "e", value: 5 },
@@ -3308,119 +3211,4 @@ describe("LSM merge insert", () => {
.execute([{ id: "g", value: 7 }]), .execute([{ id: "g", value: 7 }]),
).rejects.toThrow(); ).rejects.toThrow();
}); });
it("auto-routes reads through the MemWAL scanner", async () => {
const conn = await connect(tmpDir.name);
const table = await bucketTable(conn); // base ids "a", "b"
await table
.mergeInsert("id")
.whenMatchedUpdateAll()
.whenNotMatchedInsertAll()
.execute([{ id: "c", value: 3 }]);
// Default read auto-routes and includes the active memtable row.
const lsm = await table.query().toArray();
expect(lsm.map((r) => r.id).sort()).toEqual(["a", "b", "c"]);
// useLsm(false) bypasses the MemWAL and reads the base table only.
const baseOnly = await table.query().useLsm(false).toArray();
expect(baseOnly.map((r) => r.id).sort()).toEqual(["a", "b"]);
});
it("reads the base table when no LSM spec is installed", async () => {
const conn = await connect(tmpDir.name);
const table = await conn.createEmptyTable(
"plain",
new arrow.Schema([new arrow.Field("id", new arrow.Utf8(), false)]),
);
// No spec: default read and useLsm(false) both succeed against the base table.
await expect(table.query().toArray()).resolves.toBeDefined();
await expect(table.query().useLsm(false).toArray()).resolves.toBeDefined();
// useLsm(true) demands MemWAL routing; without a spec it errors.
await expect(table.query().useLsm(true).toArray()).rejects.toThrow();
});
});
describe("computed columns", () => {
let tmpDir: tmp.DirResult;
beforeEach(() => {
tmpDir = tmp.dirSync({ unsafeCleanup: true });
});
afterEach(() => tmpDir.removeCallback());
it("declares a column and fills it on refresh", async () => {
const db = await connect(tmpDir.name);
const table = await db.createTable("computed", [{ x: 1 }, { x: 2 }]);
await table.addColumns({
computed: [{ name: "doubled", valueSql: "x * 2" }],
});
let rows = await table.query().toArray();
expect(rows.map((r) => r.doubled)).toEqual([null, null]);
const result = await table.refreshColumn("doubled");
expect(result.rowsFilled).toBe(2);
rows = await table.query().toArray();
expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]);
});
it("returns a job handle from refreshColumnAsync", async () => {
const db = await connect(tmpDir.name);
const table = await db.createTable("computed_job", [{ x: 1 }, { x: 2 }]);
await table.addColumns({
computed: [{ name: "doubled", valueSql: "x * 2" }],
});
const job = await table.refreshColumnAsync("doubled");
expect(job.id).toBeNull();
await job.wait();
expect(await job.status()).toBe("finished");
const rows = await table.query().toArray();
expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]);
// Bad input rejects at the call, not through the job.
await expect(table.refreshColumnAsync("x")).rejects.toThrow(
"not a computed column",
);
});
it("fills rows added since the last refresh", async () => {
const db = await connect(tmpDir.name);
const table = await db.createTable("computed_append", [{ x: 1 }]);
await table.addColumns({
computed: [{ name: "doubled", valueSql: "x * 2" }],
});
await table.refreshColumn("doubled");
await table.add([{ x: 5 }]);
const result = await table.refreshColumn("doubled");
expect(result.rowsFilled).toBe(1);
const rows = await table.query().toArray();
expect(rows.map((r) => r.doubled).sort()).toEqual([10, 2]);
});
});
describe("table bases", () => {
let tmpDir: tmp.DirResult;
beforeEach(() => {
tmpDir = tmp.dirSync({ unsafeCleanup: true });
});
afterEach(() => tmpDir.removeCallback());
it("addBases accepts a file uri", async () => {
const conn = await connect(tmpDir.name);
const table = await conn.createEmptyTable(
"photos",
new arrow.Schema([new arrow.Field("id", new arrow.Int64(), false)]),
);
const media = path.join(tmpDir.name, "media");
fs.mkdirSync(media);
await table.addBases(pathToFileURL(media).toString());
});
}); });
+1 -7
View File
@@ -29,14 +29,8 @@ test("full text search", async () => {
const tbl = await db.createTable("myVectors", data, { mode: "overwrite" }); const tbl = await db.createTable("myVectors", data, { mode: "overwrite" });
await tbl.createIndex("doc", { await tbl.createIndex("doc", {
config: lancedb.Index.fts({ config: lancedb.Index.fts(),
stem: false,
removeStopWords: true,
customStopWords: ["banana"],
}),
}); });
const tokens = await tbl.tokenize("apple banana", { column: "doc" });
expect(tokens.map((token) => token.text)).toEqual(["apple"]);
// --8<-- [start:full_text_search] // --8<-- [start:full_text_search]
const result = await tbl const result = await tbl
-74
View File
@@ -1,7 +1,6 @@
// SPDX-License-Identifier: Apache-2.0 // SPDX-License-Identifier: Apache-2.0
// SPDX-FileCopyrightText: Copyright The LanceDB Authors // SPDX-FileCopyrightText: Copyright The LanceDB Authors
import { tableFromIPC } from "apache-arrow";
import { import {
Data, Data,
SchemaLike, SchemaLike,
@@ -21,9 +20,6 @@ import type {
CreateNamespaceResponse, CreateNamespaceResponse,
DescribeNamespaceResponse, DescribeNamespaceResponse,
DropNamespaceResponse, DropNamespaceResponse,
Job,
JobDescription,
JobInfo,
ListNamespacesResponse, ListNamespacesResponse,
} from "./native"; } from "./native";
export type { export type {
@@ -327,14 +323,6 @@ export abstract class Connection {
*/ */
abstract dropTable(name: string, namespacePath?: string[]): Promise<void>; abstract dropTable(name: string, namespacePath?: string[]): Promise<void>;
/**
* Start dropping a table and return its cleanup job.
*
* The table may become unavailable before its data files are removed. Wait
* on the returned job to know when cleanup has finished.
*/
abstract dropTableAsync(name: string, namespacePath?: string[]): Promise<Job>;
/** /**
* Drop all tables in the database. * Drop all tables in the database.
* @param {string[]} namespacePath The namespace path to drop tables from (defaults to root namespace). * @param {string[]} namespacePath The namespace path to drop tables from (defaults to root namespace).
@@ -448,40 +436,6 @@ export abstract class Connection {
newName: string, newName: string,
options?: RenameTableOptions, options?: RenameTableOptions,
): Promise<void>; ): Promise<void>;
/**
* A {@link Job} handle for a server-side job by id.
*
* The handle is constructed without a server round trip; an unknown id
* surfaces when the handle is used. Dropping the handle has no effect on
* the job itself.
*/
abstract job(jobId: string): Job;
/** List server-side jobs across the database's tables. */
abstract listJobs(): Promise<JobInfo[]>;
/**
* Describe a single server-side job by id.
*
* Resolves to `null` when the server has no such job.
*/
abstract getJob(jobId: string): Promise<JobDescription | null>;
/**
* Request cancellation of a server-side job by id.
*
* Resolves to true if the server accepted the cancellation, false if no
* such job exists. Cancelling an already-terminal job is a no-op success.
*/
abstract cancelJob(jobId: string): Promise<boolean>;
/**
* The lifecycle event history of a server-side job, as an Arrow table.
*
* Lists history across all jobs when `jobId` is omitted.
*/
abstract jobHistory(jobId?: string): Promise<ArrowTable>;
} }
/** @hideconstructor */ /** @hideconstructor */
@@ -713,10 +667,6 @@ export class LocalConnection extends Connection {
return this.inner.dropTable(name, namespacePath ?? []); return this.inner.dropTable(name, namespacePath ?? []);
} }
async dropTableAsync(name: string, namespacePath?: string[]): Promise<Job> {
return this.inner.dropTableAsync(name, namespacePath ?? []);
}
async dropAllTables(namespacePath?: string[]): Promise<void> { async dropAllTables(namespacePath?: string[]): Promise<void> {
return this.inner.dropAllTables(namespacePath ?? []); return this.inner.dropAllTables(namespacePath ?? []);
} }
@@ -772,30 +722,6 @@ export class LocalConnection extends Connection {
options?.newNamespacePath, options?.newNamespacePath,
); );
} }
job(jobId: string): Job {
return this.inner.job(jobId);
}
async listJobs(): Promise<JobInfo[]> {
return this.inner.listJobs();
}
async getJob(jobId: string): Promise<JobDescription | null> {
return this.inner.getJob(jobId);
}
async cancelJob(jobId: string): Promise<boolean> {
return this.inner.cancelJob(jobId);
}
async jobHistory(jobId?: string): Promise<ArrowTable> {
const buf = await this.inner.jobHistory(jobId);
if (buf.length === 0) {
return new ArrowTable();
}
return tableFromIPC(buf);
}
} }
/** /**
+1 -20
View File
@@ -50,7 +50,6 @@ export {
MergeResult, MergeResult,
AddResult, AddResult,
AddColumnsResult, AddColumnsResult,
RefreshColumnResult,
AlterColumnsResult, AlterColumnsResult,
UpdateFieldMetadataResult, UpdateFieldMetadataResult,
DeleteResult, DeleteResult,
@@ -86,13 +85,7 @@ export {
RenameTableOptions, RenameTableOptions,
} from "./connection"; } from "./connection";
export { export { Session } from "./native.js";
Job,
JobDescription,
JobFailureInfo,
JobInfo,
Session,
} from "./native.js";
export { export {
ExecutableQuery, ExecutableQuery,
@@ -130,7 +123,6 @@ export {
export { export {
Table, Table,
TableBase,
Branches, Branches,
BranchColumnSummary, BranchColumnSummary,
BranchColumnChange, BranchColumnChange,
@@ -202,16 +194,6 @@ export interface TokenizeOptions {
/** Whether to remove stop words. */ /** Whether to remove stop words. */
removeStopWords?: boolean; removeStopWords?: boolean;
/**
* Custom stop words that replace the built-in list for `language`.
*
* This option only affects tokenization when `removeStopWords` is true.
*
* `undefined` keeps the built-in language list. An empty array explicitly
* replaces it with no stop words.
*/
customStopWords?: string[];
/** Whether to fold ASCII characters. */ /** Whether to fold ASCII characters. */
asciiFolding?: boolean; asciiFolding?: boolean;
@@ -243,7 +225,6 @@ export async function tokenize(
options?.lowercase, options?.lowercase,
options?.stem, options?.stem,
options?.removeStopWords, options?.removeStopWords,
options?.customStopWords,
options?.asciiFolding, options?.asciiFolding,
options?.ngramMinLength, options?.ngramMinLength,
options?.ngramMaxLength, options?.ngramMaxLength,
-20
View File
@@ -553,16 +553,6 @@ export interface FtsOptions {
*/ */
removeStopWords?: boolean; removeStopWords?: boolean;
/**
* Custom stop words that replace the built-in list for `language`.
*
* This option only affects tokenization when `removeStopWords` is true.
*
* `undefined` keeps the built-in language list. An empty array explicitly
* replaces it with no stop words.
*/
customStopWords?: string[];
/** /**
* whether to remove punctuation * whether to remove punctuation
*/ */
@@ -582,14 +572,6 @@ export interface FtsOptions {
* whether to only index the prefix of the token for ngram tokenizer * whether to only index the prefix of the token for ngram tokenizer
*/ */
prefixOnly?: boolean; prefixOnly?: boolean;
/**
* Number of documents per compressed posting block.
*
* The default is 128. Supported values are 128 and 256. A value of 256 uses
* the experimental FTS V3 format and may introduce breaking changes.
*/
blockSize?: 128 | 256;
} }
export class Index { export class Index {
@@ -765,12 +747,10 @@ export class Index {
options?.lowercase, options?.lowercase,
options?.stem, options?.stem,
options?.removeStopWords, options?.removeStopWords,
options?.customStopWords,
options?.asciiFolding, options?.asciiFolding,
options?.ngramMinLength, options?.ngramMinLength,
options?.ngramMaxLength, options?.ngramMaxLength,
options?.prefixOnly, options?.prefixOnly,
options?.blockSize,
), ),
); );
} }
+11 -7
View File
@@ -88,17 +88,21 @@ export class MergeInsertBuilder {
); );
} }
/** /**
* Control MemWAL routing for this merge. * Controls whether the merge uses the MemWAL LSM write path.
* *
* By default (unset), a `mergeInsert` on a table with an LSM write spec is * By default (unset), a `mergeInsert` on a table with an LSM write spec is
* routed through Lance's MemWAL shard writer, and a table without one uses the * routed through Lance's MemWAL shard writer, and a table without one uses
* standard path. * the standard path. Pass `false` to force the standard path even when a
* spec is set. Pass `true` to require a spec — `mergeInsert` rejects if none
* is installed.
* *
* @param enable - `true` forces MemWAL routing and errors if the table has no * @param useLsmWrite - Whether to use the LSM write path.
* LSM write spec. `false` forces the standard write path even when a spec is set.
*/ */
useLsm(enable: boolean): MergeInsertBuilder { useLsmWrite(useLsmWrite: boolean): MergeInsertBuilder {
return new MergeInsertBuilder(this.#native.useLsm(enable), this.#schema); return new MergeInsertBuilder(
this.#native.useLsmWrite(useLsmWrite),
this.#schema,
);
} }
/** /**
* Controls how an LSM merge checks that its input targets a single shard. * Controls how an LSM merge checks that its input targets a single shard.
-38
View File
@@ -460,30 +460,6 @@ export class StandardQueryBase<
this.doCall((inner: NativeQueryType) => inner.fastSearch()); this.doCall((inner: NativeQueryType) => inner.fastSearch());
return this; return this;
} }
/**
* Control MemWAL read routing for this query.
*
* By default (unset), when the table carries a MemWAL write spec (see
* {@link Table#setLsmWriteSpec}), reads are routed through the LSM scanner so
* they also return data written via the `mergeInsert` LSM path that has not yet
* been compacted into the base table (the active/frozen in-memory memtables and
* the flushed generations), deduplicated by primary key; a table without a spec
* reads the base table.
*
* @param enable - `true` forces the LSM scanner and errors if the table has no
* MemWAL write spec. `false` bypasses the MemWAL and reads the base table only,
* even when a spec is present.
*
* Note: the LSM scanner does not support every query shape (e.g. reranking,
* hybrid search, `orderBy`). On a MemWAL table those shapes error unless
* `useLsm(false)` is set, because a base-only read would silently exclude
* un-compacted MemWAL data.
*/
useLsm(enable: boolean): this {
this.doCall((inner: NativeQueryType) => inner.useLsm(enable));
return this;
}
} }
/** /**
@@ -772,20 +748,6 @@ export class TakeQuery extends QueryBase<NativeTakeQuery> {
constructor(inner: NativeTakeQuery) { constructor(inner: NativeTakeQuery) {
super(inner); super(inner);
} }
/**
* Control MemWAL read routing for this take query.
*
* `false` bypasses the MemWAL and reads the base table only — the escape hatch,
* since take-by-row-id/offset is not supported on the LSM scanner and, on a
* MemWAL table, auto-routes to it and errors otherwise.
*
* @param enable - `false` reads the base table only.
*/
useLsm(enable: boolean): this {
this.doCall((inner: NativeTakeQuery) => inner.useLsm(enable));
return this;
}
} }
/** A builder for LanceDB queries. /** A builder for LanceDB queries.
+30 -175
View File
@@ -9,7 +9,7 @@
// comes from the exact same library instance. This is not always the case // comes from the exact same library instance. This is not always the case
// and so we must sanitize the input to ensure that it is compatible. // and so we must sanitize the input to ensure that it is compatible.
import { BufferType, Data, Vector } from "apache-arrow"; import { BufferType, Data } from "apache-arrow";
import type { IntBitWidth, TKeys, TimeBitWidth } from "apache-arrow/type"; import type { IntBitWidth, TKeys, TimeBitWidth } from "apache-arrow/type";
import { import {
Binary, Binary,
@@ -74,20 +74,6 @@ import {
Utf8, Utf8,
} from "./arrow"; } from "./arrow";
type SanitizationContext = {
types: WeakMap<object, DataType>;
vectors: WeakMap<object, Vector>;
data: WeakMap<object, Data<DataType>>;
};
function createSanitizationContext(): SanitizationContext {
return {
types: new WeakMap(),
vectors: new WeakMap(),
data: new WeakMap(),
};
}
export function sanitizeMetadata( export function sanitizeMetadata(
metadataLike?: unknown, metadataLike?: unknown,
): Map<string, string> | undefined { ): Map<string, string> | undefined {
@@ -98,7 +84,7 @@ export function sanitizeMetadata(
throw Error("Expected metadata, if present, to be a Map<string, string>"); throw Error("Expected metadata, if present, to be a Map<string, string>");
} }
for (const item of metadataLike) { for (const item of metadataLike) {
if (typeof item[0] !== "string" || typeof item[1] !== "string") { if (!(typeof item[0] === "string" || !(typeof item[1] === "string"))) {
throw Error( throw Error(
"Expected metadata, if present, to be a Map<string, string> but it had non-string keys or values", "Expected metadata, if present, to be a Map<string, string> but it had non-string keys or values",
); );
@@ -200,13 +186,6 @@ export function sanitizeInterval(typeLike: object) {
} }
export function sanitizeList(typeLike: object) { export function sanitizeList(typeLike: object) {
return sanitizeListWithContext(typeLike, createSanitizationContext());
}
function sanitizeListWithContext(
typeLike: object,
context: SanitizationContext,
) {
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) { if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
throw Error( throw Error(
"Expected a List type to have an array-like `children` property", "Expected a List type to have an array-like `children` property",
@@ -215,35 +194,19 @@ function sanitizeListWithContext(
if (typeLike.children.length !== 1) { if (typeLike.children.length !== 1) {
throw Error("Expected a List type to have exactly one child"); throw Error("Expected a List type to have exactly one child");
} }
return new List(sanitizeFieldWithContext(typeLike.children[0], context)); return new List(sanitizeField(typeLike.children[0]));
} }
export function sanitizeStruct(typeLike: object) { export function sanitizeStruct(typeLike: object) {
return sanitizeStructWithContext(typeLike, createSanitizationContext());
}
function sanitizeStructWithContext(
typeLike: object,
context: SanitizationContext,
) {
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) { if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
throw Error( throw Error(
"Expected a Struct type to have an array-like `children` property", "Expected a Struct type to have an array-like `children` property",
); );
} }
return new Struct( return new Struct(typeLike.children.map((child) => sanitizeField(child)));
typeLike.children.map((child) => sanitizeFieldWithContext(child, context)),
);
} }
export function sanitizeUnion(typeLike: object) { export function sanitizeUnion(typeLike: object) {
return sanitizeUnionWithContext(typeLike, createSanitizationContext());
}
function sanitizeUnionWithContext(
typeLike: object,
context: SanitizationContext,
) {
if ( if (
!("typeIds" in typeLike) || !("typeIds" in typeLike) ||
!("mode" in typeLike) || !("mode" in typeLike) ||
@@ -263,7 +226,7 @@ function sanitizeUnionWithContext(
typeLike.mode, typeLike.mode,
// biome-ignore lint/suspicious/noExplicitAny: skip // biome-ignore lint/suspicious/noExplicitAny: skip
typeLike.typeIds as any, typeLike.typeIds as any,
typeLike.children.map((child) => sanitizeFieldWithContext(child, context)), typeLike.children.map((child) => sanitizeField(child)),
); );
} }
@@ -271,19 +234,6 @@ export function sanitizeTypedUnion(
typeLike: object, typeLike: object,
// eslint-disable-next-line @typescript-eslint/naming-convention // eslint-disable-next-line @typescript-eslint/naming-convention
UnionType: typeof DenseUnion | typeof SparseUnion, UnionType: typeof DenseUnion | typeof SparseUnion,
) {
return sanitizeTypedUnionWithContext(
typeLike,
UnionType,
createSanitizationContext(),
);
}
function sanitizeTypedUnionWithContext(
typeLike: object,
// eslint-disable-next-line @typescript-eslint/naming-convention
UnionType: typeof DenseUnion | typeof SparseUnion,
context: SanitizationContext,
) { ) {
if (!("typeIds" in typeLike)) { if (!("typeIds" in typeLike)) {
throw Error( throw Error(
@@ -298,7 +248,7 @@ function sanitizeTypedUnionWithContext(
return new UnionType( return new UnionType(
typeLike.typeIds as Int32Array | number[], typeLike.typeIds as Int32Array | number[],
typeLike.children.map((child) => sanitizeFieldWithContext(child, context)), typeLike.children.map((child) => sanitizeField(child)),
); );
} }
@@ -312,16 +262,6 @@ export function sanitizeFixedSizeBinary(typeLike: object) {
} }
export function sanitizeFixedSizeList(typeLike: object) { export function sanitizeFixedSizeList(typeLike: object) {
return sanitizeFixedSizeListWithContext(
typeLike,
createSanitizationContext(),
);
}
function sanitizeFixedSizeListWithContext(
typeLike: object,
context: SanitizationContext,
) {
if (!("listSize" in typeLike) || typeof typeLike.listSize !== "number") { if (!("listSize" in typeLike) || typeof typeLike.listSize !== "number") {
throw Error("Expected a FixedSizeList type to have a `listSize` property"); throw Error("Expected a FixedSizeList type to have a `listSize` property");
} }
@@ -335,18 +275,11 @@ function sanitizeFixedSizeListWithContext(
} }
return new FixedSizeList( return new FixedSizeList(
typeLike.listSize, typeLike.listSize,
sanitizeFieldWithContext(typeLike.children[0], context), sanitizeField(typeLike.children[0]),
); );
} }
export function sanitizeMap(typeLike: object) { export function sanitizeMap(typeLike: object) {
return sanitizeMapWithContext(typeLike, createSanitizationContext());
}
function sanitizeMapWithContext(
typeLike: object,
context: SanitizationContext,
) {
if (!("children" in typeLike) || !Array.isArray(typeLike.children)) { if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
throw Error( throw Error(
"Expected a Map type to have an array-like `children` property", "Expected a Map type to have an array-like `children` property",
@@ -359,10 +292,7 @@ function sanitizeMapWithContext(
throw Error("Expected a Map type to have exactly one child"); throw Error("Expected a Map type to have exactly one child");
} }
return new Map_( return new Map_(sanitizeField(typeLike.children[0]), typeLike.keysSorted);
sanitizeFieldWithContext(typeLike.children[0], context),
typeLike.keysSorted,
);
} }
export function sanitizeDuration(typeLike: object) { export function sanitizeDuration(typeLike: object) {
@@ -373,13 +303,6 @@ export function sanitizeDuration(typeLike: object) {
} }
export function sanitizeDictionary(typeLike: object) { export function sanitizeDictionary(typeLike: object) {
return sanitizeDictionaryWithContext(typeLike, createSanitizationContext());
}
function sanitizeDictionaryWithContext(
typeLike: object,
context: SanitizationContext,
) {
if (!("id" in typeLike) || typeof typeLike.id !== "number") { if (!("id" in typeLike) || typeof typeLike.id !== "number") {
throw Error("Expected a Dictionary type to have an `id` property"); throw Error("Expected a Dictionary type to have an `id` property");
} }
@@ -393,8 +316,8 @@ function sanitizeDictionaryWithContext(
throw Error("Expected a Dictionary type to have an `isOrdered` property"); throw Error("Expected a Dictionary type to have an `isOrdered` property");
} }
return new Dictionary( return new Dictionary(
sanitizeTypeWithContext(typeLike.dictionary, context), sanitizeType(typeLike.dictionary),
sanitizeTypeWithContext(typeLike.indices, context) as TKeys, sanitizeType(typeLike.indices) as TKeys,
typeLike.id, typeLike.id,
typeLike.isOrdered, typeLike.isOrdered,
); );
@@ -402,23 +325,12 @@ function sanitizeDictionaryWithContext(
// biome-ignore lint/suspicious/noExplicitAny: skip // biome-ignore lint/suspicious/noExplicitAny: skip
export function sanitizeType(typeLike: unknown): DataType<any> { export function sanitizeType(typeLike: unknown): DataType<any> {
return sanitizeTypeWithContext(typeLike, createSanitizationContext());
}
function sanitizeTypeWithContext(
typeLike: unknown,
context: SanitizationContext,
): DataType {
if (typeof typeLike === "string") { if (typeof typeLike === "string") {
return dataTypeFromName(typeLike); return dataTypeFromName(typeLike);
} }
if (typeof typeLike !== "object" || typeLike === null) { if (typeof typeLike !== "object" || typeLike === null) {
throw Error("Expected a Type but object was null/undefined"); throw Error("Expected a Type but object was null/undefined");
} }
const cached = context.types.get(typeLike);
if (cached !== undefined) {
return cached;
}
if ( if (
!("typeId" in typeLike) || !("typeId" in typeLike) ||
!( !(
@@ -437,16 +349,6 @@ function sanitizeTypeWithContext(
throw Error("Type's typeId property was not a function or number"); throw Error("Type's typeId property was not a function or number");
} }
const type = sanitizeTypeById(typeLike, typeId, context);
context.types.set(typeLike, type);
return type;
}
function sanitizeTypeById(
typeLike: object,
typeId: Type,
context: SanitizationContext,
): DataType {
switch (typeId) { switch (typeId) {
case Type.NONE: case Type.NONE:
throw Error("Received a Type with a typeId of NONE"); throw Error("Received a Type with a typeId of NONE");
@@ -473,21 +375,21 @@ function sanitizeTypeById(
case Type.Interval: case Type.Interval:
return sanitizeInterval(typeLike); return sanitizeInterval(typeLike);
case Type.List: case Type.List:
return sanitizeListWithContext(typeLike, context); return sanitizeList(typeLike);
case Type.Struct: case Type.Struct:
return sanitizeStructWithContext(typeLike, context); return sanitizeStruct(typeLike);
case Type.Union: case Type.Union:
return sanitizeUnionWithContext(typeLike, context); return sanitizeUnion(typeLike);
case Type.FixedSizeBinary: case Type.FixedSizeBinary:
return sanitizeFixedSizeBinary(typeLike); return sanitizeFixedSizeBinary(typeLike);
case Type.FixedSizeList: case Type.FixedSizeList:
return sanitizeFixedSizeListWithContext(typeLike, context); return sanitizeFixedSizeList(typeLike);
case Type.Map: case Type.Map:
return sanitizeMapWithContext(typeLike, context); return sanitizeMap(typeLike);
case Type.Duration: case Type.Duration:
return sanitizeDuration(typeLike); return sanitizeDuration(typeLike);
case Type.Dictionary: case Type.Dictionary:
return sanitizeDictionaryWithContext(typeLike, context); return sanitizeDictionary(typeLike);
case Type.Int8: case Type.Int8:
return new Int8(); return new Int8();
case Type.Int16: case Type.Int16:
@@ -531,9 +433,9 @@ function sanitizeTypeById(
case Type.TimestampSecond: case Type.TimestampSecond:
return sanitizeTypedTimestamp(typeLike, TimestampSecond); return sanitizeTypedTimestamp(typeLike, TimestampSecond);
case Type.DenseUnion: case Type.DenseUnion:
return sanitizeTypedUnionWithContext(typeLike, DenseUnion, context); return sanitizeTypedUnion(typeLike, DenseUnion);
case Type.SparseUnion: case Type.SparseUnion:
return sanitizeTypedUnionWithContext(typeLike, SparseUnion, context); return sanitizeTypedUnion(typeLike, SparseUnion);
case Type.IntervalDayTime: case Type.IntervalDayTime:
return new IntervalDayTime(); return new IntervalDayTime();
case Type.IntervalYearMonth: case Type.IntervalYearMonth:
@@ -552,13 +454,6 @@ function sanitizeTypeById(
} }
export function sanitizeField(fieldLike: unknown): Field { export function sanitizeField(fieldLike: unknown): Field {
return sanitizeFieldWithContext(fieldLike, createSanitizationContext());
}
function sanitizeFieldWithContext(
fieldLike: unknown,
context: SanitizationContext,
): Field {
if (fieldLike instanceof Field) { if (fieldLike instanceof Field) {
return fieldLike; return fieldLike;
} }
@@ -576,7 +471,7 @@ function sanitizeFieldWithContext(
} }
let type: DataType; let type: DataType;
try { try {
type = sanitizeTypeWithContext(fieldLike.type, context); type = sanitizeType(fieldLike.type);
} catch (error: unknown) { } catch (error: unknown) {
throw Error( throw Error(
`Unable to sanitize type for field: ${fieldLike.name} due to error: ${error}`, `Unable to sanitize type for field: ${fieldLike.name} due to error: ${error}`,
@@ -606,13 +501,6 @@ function sanitizeFieldWithContext(
* than lancedb is using. * than lancedb is using.
*/ */
export function sanitizeSchema(schemaLike: SchemaLike): Schema { export function sanitizeSchema(schemaLike: SchemaLike): Schema {
return sanitizeSchemaWithContext(schemaLike, createSanitizationContext());
}
function sanitizeSchemaWithContext(
schemaLike: SchemaLike,
context: SanitizationContext,
): Schema {
if (schemaLike instanceof Schema) { if (schemaLike instanceof Schema) {
return schemaLike; return schemaLike;
} }
@@ -634,7 +522,7 @@ function sanitizeSchemaWithContext(
); );
} }
const sanitizedFields = schemaLike.fields.map((field) => const sanitizedFields = schemaLike.fields.map((field) =>
sanitizeFieldWithContext(field, context), sanitizeField(field),
); );
return new Schema(sanitizedFields, metadata); return new Schema(sanitizedFields, metadata);
} }
@@ -656,18 +544,13 @@ export function sanitizeTable(tableLike: TableLike): Table {
"The table passed in does not appear to be a table (no 'columns' property)", "The table passed in does not appear to be a table (no 'columns' property)",
); );
} }
const context = createSanitizationContext(); const schema = sanitizeSchema(tableLike.schema);
const schema = sanitizeSchemaWithContext(tableLike.schema, context);
const batches = tableLike.batches.map((batch) => const batches = tableLike.batches.map(sanitizeRecordBatch);
sanitizeRecordBatch(batch, context),
);
return new Table(schema, batches); return new Table(schema, batches);
} }
function sanitizeRecordBatch( function sanitizeRecordBatch(batchLike: RecordBatchLike): RecordBatch {
batchLike: RecordBatchLike,
context: SanitizationContext,
): RecordBatch {
if (batchLike instanceof RecordBatch) { if (batchLike instanceof RecordBatch) {
return batchLike; return batchLike;
} }
@@ -684,43 +567,19 @@ function sanitizeRecordBatch(
"The record batch passed in does not appear to be a record batch (no 'data' property)", "The record batch passed in does not appear to be a record batch (no 'data' property)",
); );
} }
const schema = sanitizeSchemaWithContext(batchLike.schema, context); const schema = sanitizeSchema(batchLike.schema);
const data = sanitizeData(batchLike.data, context) as Data<Struct>; const data = sanitizeData(batchLike.data);
return new RecordBatch(schema, data); return new RecordBatch(schema, data);
} }
type DictionaryVectorLike = {
data: readonly DataLike[];
};
type DictionaryDataLike = DataLike & {
dictionary?: DictionaryVectorLike;
};
function sanitizeData( function sanitizeData(
dataLike: DataLike, dataLike: DataLike,
context: SanitizationContext, // biome-ignore lint/suspicious/noExplicitAny: <explanation>
): Data<DataType> { ): import("apache-arrow").Data<Struct<any>> {
if (dataLike instanceof Data) { if (dataLike instanceof Data) {
return dataLike; return dataLike;
} }
const cachedData = context.data.get(dataLike); return new Data(
if (cachedData !== undefined) { dataLike.type,
return cachedData;
}
const dictionaryLike = (dataLike as DictionaryDataLike).dictionary;
let dictionary: Vector | undefined;
if (dictionaryLike !== undefined) {
dictionary = context.vectors.get(dictionaryLike);
if (dictionary === undefined) {
dictionary = new Vector(
dictionaryLike.data.map((data) => sanitizeData(data, context)),
);
context.vectors.set(dictionaryLike, dictionary);
}
}
const data = new Data(
sanitizeTypeWithContext(dataLike.type, context),
dataLike.offset, dataLike.offset,
dataLike.length, dataLike.length,
dataLike.nullCount, dataLike.nullCount,
@@ -730,11 +589,7 @@ function sanitizeData(
[BufferType.VALIDITY]: dataLike.nullBitmap, [BufferType.VALIDITY]: dataLike.nullBitmap,
[BufferType.TYPE]: dataLike.typeIds, [BufferType.TYPE]: dataLike.typeIds,
}, },
dataLike.children.map((child) => sanitizeData(child, context)),
dictionary,
); );
context.data.set(dataLike, data);
return data;
} }
const constructorsByTypeName = { const constructorsByTypeName = {
+6 -173
View File
@@ -30,10 +30,8 @@ import {
DropColumnsResult, DropColumnsResult,
IndexConfig, IndexConfig,
IndexStatistics, IndexStatistics,
Job,
Branches as NativeBranches, Branches as NativeBranches,
OptimizeStats, OptimizeStats,
RefreshColumnResult,
TableStatistics, TableStatistics,
Tags, Tags,
UpdateFieldMetadataResult, UpdateFieldMetadataResult,
@@ -78,25 +76,6 @@ export interface WriteProgress {
done: boolean; done: boolean;
} }
/**
* An extra storage prefix registered on a table.
*
* `path` is an object-store URI. `name` is an optional alias. `isDatasetRoot`
* is true when `path` points to a Lance dataset root. When false, `path`
* points directly to the directory containing the referenced files.
*/
export interface TableBase {
/** Object store URI such as `s3://bucket/media/`. */
path: string;
/** Optional alias. */
name?: string;
/**
* True when `path` is a Lance dataset root. When false, `path` is the
* directory containing the referenced files.
*/
isDatasetRoot?: boolean;
}
/** /**
* Options for adding data to a table. * Options for adding data to a table.
*/ */
@@ -217,11 +196,7 @@ export interface LsmWriteSpec {
column?: string; column?: string;
/** Bucket variant: the number of buckets, in `[1, 1024]`. */ /** Bucket variant: the number of buckets, in `[1, 1024]`. */
numBuckets?: number; numBuckets?: number;
/** /** Names of indexes the MemWAL should keep up to date during writes. */
* Indexes the MemWAL keeps up to date. Omit to maintain every supported
* index, resolved on install — a snapshot, so indexes created later are not
* maintained. Pass `[]` for none.
*/
maintainedIndexes?: string[]; maintainedIndexes?: string[];
/** Default `ShardWriter` configuration recorded in the MemWAL index. */ /** Default `ShardWriter` configuration recorded in the MemWAL index. */
writerConfigDefaults?: Record<string, string>; writerConfigDefaults?: Record<string, string>;
@@ -383,17 +358,6 @@ export abstract class Table {
options?: Partial<IndexOptions>, options?: Partial<IndexOptions>,
): Promise<void>; ): Promise<void>;
/**
* Create an index, returning a handle to the indexing job.
*
* The job may already be complete when returned; callers must not assume
* the index exists until {@link Job.wait} resolves.
*/
abstract createIndexAsync(
column: string,
options?: Partial<IndexOptions>,
): Promise<Job>;
/** /**
* Drop an index from the table. * Drop an index from the table.
* *
@@ -545,84 +509,18 @@ export abstract class Table {
abstract vectorSearch(vector: IntoVector | MultiVector): VectorQuery; abstract vectorSearch(vector: IntoVector | MultiVector): VectorQuery;
/** /**
* Add new columns with defined values. * Add new columns with defined values.
*
* The `{ computed }` form stores the expression rather than evaluating it
* now: the column is committed with no values, and rows get them from
* {@link Table#refreshColumn}. Declaring one therefore costs the same on a
* large table as on an empty one.
*
* A refresh does not revisit rows it has already filled, so mutating an
* input leaves the value computed at fill time; recomputing means dropping
* the column and declaring it again. While a declaration reads a column,
* that column cannot be renamed, retyped or dropped.
*
* On LanceDB Cloud and Enterprise the expression is planned by the
* server, and the refresh runs as a server job -- see
* {@link Table#refreshColumnAsync}.
* @param {AddColumnsSql[] | Field | Field[] | Schema} newColumnTransforms Either: * @param {AddColumnsSql[] | Field | Field[] | Schema} newColumnTransforms Either:
* - An array of objects with column names and SQL expressions to calculate values * - An array of objects with column names and SQL expressions to calculate values
* - A single Arrow Field defining one column with its data type (column will be initialized with null values) * - A single Arrow Field defining one column with its data type (column will be initialized with null values)
* - An array of Arrow Fields defining columns with their data types (columns will be initialized with null values) * - An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
* - An Arrow Schema defining columns with their data types (columns will be initialized with null values) * - An Arrow Schema defining columns with their data types (columns will be initialized with null values)
* - `{ computed }`, declaring columns defined by a SQL expression whose type and inputs are derived from it
* @returns {Promise<AddColumnsResult>} A promise that resolves to an object * @returns {Promise<AddColumnsResult>} A promise that resolves to an object
* containing the new version number of the table after adding the columns. * containing the new version number of the table after adding the columns.
* @example
* ```ts
* await table.addColumns({ computed: [{ name: "doubled", valueSql: "x * 2" }] });
* const { rowsFilled } = await table.refreshColumn("doubled");
* ```
*/ */
abstract addColumns( abstract addColumns(
newColumnTransforms: newColumnTransforms: AddColumnsSql[] | Field | Field[] | Schema,
| AddColumnsSql[]
| Field
| Field[]
| Schema
| { computed: AddColumnsSql[] },
): Promise<AddColumnsResult>; ): Promise<AddColumnsResult>;
/**
* Register additional storage bases for this table.
*
* A URI string is a non-root base with no alias.
*/
abstract addBases(
bases: string | TableBase | Array<string | TableBase>,
): Promise<void>;
/**
* Fill the rows of a computed column that hold no value yet.
*
* Rows appended since the last refresh are filled by the next one; rows
* already filled are left as they are, so the call is idempotent and does
* not observe a mutated input. Local tables only: a remote refresh runs
* as a server job, through {@link Table#refreshColumnAsync}.
* @param {string} column The name of the computed column to fill.
* @returns {Promise<RefreshColumnResult>} A promise that resolves to the
* number of rows filled and the new version number of the table.
*/
abstract refreshColumn(column: string): Promise<RefreshColumnResult>;
/**
* Like {@link Table#refreshColumn}, but returns a handle to the refresh
* job instead of blocking until it completes.
*
* The job may already be complete when returned; callers must not assume
* the column is filled until {@link Job.wait} resolves. Invalid input --
* an unknown column, or one that is not computed -- rejects here rather
* than failing the job. On local tables the job runs in-process; on
* LanceDB Cloud and Enterprise it is the server's backfill job.
* @param {string} column The name of the computed column to fill.
* @example
* ```ts
* const job = await table.refreshColumnAsync("doubled");
* await job.wait();
* console.log(await job.status()); // "finished"
* ```
*/
abstract refreshColumnAsync(column: string): Promise<Job>;
/** /**
* Alter the name or nullability of columns. * Alter the name or nullability of columns.
* @param {ColumnAlteration[]} columnAlterations One or more alterations to * @param {ColumnAlteration[]} columnAlterations One or more alterations to
@@ -685,11 +583,6 @@ export abstract class Table {
* All variants require the table to have an unenforced primary key * All variants require the table to have an unenforced primary key
* ({@link Table#setUnenforcedPrimaryKey}); bucket sharding additionally * ({@link Table#setUnenforcedPrimaryKey}); bucket sharding additionally
* requires it to be the single column being bucketed. * requires it to be the single column being bucketed.
*
* Omitting `maintainedIndexes` maintains every index on the table, resolved
* here, failing if one cannot be maintained — name them to install anyway.
* Naming them pins an exact set, and a still-building index is rejected
* rather than quietly omitted.
* @param {LsmWriteSpec} spec The sharding spec to install. * @param {LsmWriteSpec} spec The sharding spec to install.
* @returns {Promise<void>} * @returns {Promise<void>}
* @example * @example
@@ -717,10 +610,9 @@ export abstract class Table {
* *
* Resolves to `undefined` when the MemWAL LSM write path is not enabled (no * Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
* spec has been set, or it was removed with {@link Table#unsetLsmWriteSpec}). * spec has been set, or it was removed with {@link Table#unsetLsmWriteSpec}).
* The returned spec mirrors what was passed to * The returned spec — including its `maintainedIndexes` and
* {@link Table#setLsmWriteSpec}, except that `maintainedIndexes` always * `writerConfigDefaults` — mirrors what was passed to
* reports the concrete list resolved when the spec was set — `undefined` * {@link Table#setLsmWriteSpec}.
* never round-trips.
* @returns {Promise<LsmWriteSpec | undefined>} * @returns {Promise<LsmWriteSpec | undefined>}
*/ */
abstract getLsmWriteSpec(): Promise<LsmWriteSpec | undefined>; abstract getLsmWriteSpec(): Promise<LsmWriteSpec | undefined>;
@@ -1048,22 +940,6 @@ export class LocalTable extends Table {
); );
} }
async createIndexAsync(
column: string,
options?: Partial<IndexOptions>,
): Promise<Job> {
// biome-ignore lint/suspicious/noExplicitAny: skip
const nativeIndex = (options?.config as any)?.inner;
return await this.inner.createIndexAsync(
nativeIndex,
column,
options?.replace,
options?.waitTimeoutSeconds,
options?.name,
options?.train,
);
}
async dropIndex(name: string): Promise<void> { async dropIndex(name: string): Promise<void> {
await this.inner.dropIndex(name); await this.inner.dropIndex(name);
} }
@@ -1174,22 +1050,8 @@ export class LocalTable extends Table {
// TODO: Support BatchUDF // TODO: Support BatchUDF
async addColumns( async addColumns(
newColumnTransforms: newColumnTransforms: AddColumnsSql[] | Field | Field[] | Schema,
| AddColumnsSql[]
| Field
| Field[]
| Schema
| { computed: AddColumnsSql[] },
): Promise<AddColumnsResult> { ): Promise<AddColumnsResult> {
// Columns defined by an expression are declared, not materialized here.
if (
typeof newColumnTransforms === "object" &&
!Array.isArray(newColumnTransforms) &&
"computed" in newColumnTransforms
) {
return await this.inner.addComputedColumns(newColumnTransforms.computed);
}
// Handle single Field -> convert to array of Fields // Handle single Field -> convert to array of Fields
if (newColumnTransforms instanceof Field) { if (newColumnTransforms instanceof Field) {
newColumnTransforms = [newColumnTransforms]; newColumnTransforms = [newColumnTransforms];
@@ -1224,20 +1086,6 @@ export class LocalTable extends Table {
throw new Error("Invalid input type for addColumns"); throw new Error("Invalid input type for addColumns");
} }
async addBases(
bases: string | TableBase | Array<string | TableBase>,
): Promise<void> {
await this.inner.addBases(normalizeBases(bases));
}
async refreshColumn(column: string): Promise<RefreshColumnResult> {
return await this.inner.refreshColumn(column);
}
async refreshColumnAsync(column: string): Promise<Job> {
return await this.inner.refreshColumnAsync(column);
}
async alterColumns( async alterColumns(
columnAlterations: ColumnAlteration[], columnAlterations: ColumnAlteration[],
): Promise<AlterColumnsResult> { ): Promise<AlterColumnsResult> {
@@ -1430,21 +1278,6 @@ export class LocalTable extends Table {
} }
} }
function normalizeBases(
bases: string | TableBase | Array<string | TableBase>,
): TableBase[] {
const baseInputs = Array.isArray(bases) ? bases : [bases];
return baseInputs.map((base) =>
typeof base === "string"
? { path: base, isDatasetRoot: false }
: {
path: base.path,
name: base.name,
isDatasetRoot: base.isDatasetRoot ?? false,
},
);
}
/** /**
* A definition of a column alteration. The alteration changes the column at * A definition of a column alteration. The alteration changes the column at
* `path` to have the new name `name`, to be nullable if `nullable` is true, * `path` to have the new name `name`, to be nullable if `nullable` is true,
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@lancedb/lancedb-darwin-arm64", "name": "@lancedb/lancedb-darwin-arm64",
"version": "0.38.0-beta.0", "version": "0.33.0-beta.0",
"os": ["darwin"], "os": ["darwin"],
"cpu": ["arm64"], "cpu": ["arm64"],
"main": "lancedb.darwin-arm64.node", "main": "lancedb.darwin-arm64.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@lancedb/lancedb-linux-arm64-gnu", "name": "@lancedb/lancedb-linux-arm64-gnu",
"version": "0.38.0-beta.0", "version": "0.33.0-beta.0",
"os": ["linux"], "os": ["linux"],
"cpu": ["arm64"], "cpu": ["arm64"],
"main": "lancedb.linux-arm64-gnu.node", "main": "lancedb.linux-arm64-gnu.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@lancedb/lancedb-linux-arm64-musl", "name": "@lancedb/lancedb-linux-arm64-musl",
"version": "0.38.0-beta.0", "version": "0.33.0-beta.0",
"os": ["linux"], "os": ["linux"],
"cpu": ["arm64"], "cpu": ["arm64"],
"main": "lancedb.linux-arm64-musl.node", "main": "lancedb.linux-arm64-musl.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@lancedb/lancedb-linux-x64-gnu", "name": "@lancedb/lancedb-linux-x64-gnu",
"version": "0.38.0-beta.0", "version": "0.33.0-beta.0",
"os": ["linux"], "os": ["linux"],
"cpu": ["x64"], "cpu": ["x64"],
"main": "lancedb.linux-x64-gnu.node", "main": "lancedb.linux-x64-gnu.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@lancedb/lancedb-linux-x64-musl", "name": "@lancedb/lancedb-linux-x64-musl",
"version": "0.38.0-beta.0", "version": "0.33.0-beta.0",
"os": ["linux"], "os": ["linux"],
"cpu": ["x64"], "cpu": ["x64"],
"main": "lancedb.linux-x64-musl.node", "main": "lancedb.linux-x64-musl.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@lancedb/lancedb-win32-arm64-msvc", "name": "@lancedb/lancedb-win32-arm64-msvc",
"version": "0.38.0-beta.0", "version": "0.33.0-beta.0",
"os": [ "os": [
"win32" "win32"
], ],
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@lancedb/lancedb-win32-x64-msvc", "name": "@lancedb/lancedb-win32-x64-msvc",
"version": "0.38.0-beta.0", "version": "0.33.0-beta.0",
"os": ["win32"], "os": ["win32"],
"cpu": ["x64"], "cpu": ["x64"],
"main": "lancedb.win32-x64-msvc.node", "main": "lancedb.win32-x64-msvc.node",
+2 -8
View File
@@ -1,12 +1,12 @@
{ {
"name": "@lancedb/lancedb", "name": "@lancedb/lancedb",
"version": "0.38.0-beta.0", "version": "0.33.0-beta.0",
"lockfileVersion": 3, "lockfileVersion": 3,
"requires": true, "requires": true,
"packages": { "packages": {
"": { "": {
"name": "@lancedb/lancedb", "name": "@lancedb/lancedb",
"version": "0.38.0-beta.0", "version": "0.33.0-beta.0",
"cpu": [ "cpu": [
"x64", "x64",
"arm64" "arm64"
@@ -55,13 +55,7 @@
"openai": "4.29.2" "openai": "4.29.2"
}, },
"peerDependencies": { "peerDependencies": {
"@types/node": ">=18",
"apache-arrow": ">=15.0.0 <=18.1.0" "apache-arrow": ">=15.0.0 <=18.1.0"
},
"peerDependenciesMeta": {
"@types/node": {
"optional": true
}
} }
}, },
"node_modules/@aws-crypto/crc32": { "node_modules/@aws-crypto/crc32": {
+1 -7
View File
@@ -11,7 +11,7 @@
"ann" "ann"
], ],
"private": false, "private": false,
"version": "0.38.0-beta.0", "version": "0.33.0-beta.0",
"main": "dist/index.js", "main": "dist/index.js",
"exports": { "exports": {
".": "./dist/index.js", ".": "./dist/index.js",
@@ -101,12 +101,6 @@
"openai": "4.29.2" "openai": "4.29.2"
}, },
"peerDependencies": { "peerDependencies": {
"@types/node": ">=18",
"apache-arrow": ">=15.0.0 <=18.1.0" "apache-arrow": ">=15.0.0 <=18.1.0"
},
"peerDependenciesMeta": {
"@types/node": {
"optional": true
}
} }
} }
-79
View File
@@ -334,91 +334,12 @@ impl Connection {
.default_error() .default_error()
} }
/// Start dropping a table and return its cleanup job.
#[napi(catch_unwind)]
pub async fn drop_table_async(
&self,
name: String,
namespace_path: Option<Vec<String>>,
) -> napi::Result<crate::job::Job> {
let ns = namespace_path.unwrap_or_default();
let job = self
.get_inner()?
.drop_table_async(&name, &ns)
.await
.default_error()?;
Ok(crate::job::Job::new(job))
}
#[napi(catch_unwind)] #[napi(catch_unwind)]
pub async fn drop_all_tables(&self, namespace_path: Option<Vec<String>>) -> napi::Result<()> { pub async fn drop_all_tables(&self, namespace_path: Option<Vec<String>>) -> napi::Result<()> {
let ns = namespace_path.unwrap_or_default(); let ns = namespace_path.unwrap_or_default();
self.get_inner()?.drop_all_tables(&ns).await.default_error() self.get_inner()?.drop_all_tables(&ns).await.default_error()
} }
/// A `Job` handle for a server-side job by id.
///
/// The handle is constructed without a server round trip; an unknown id
/// surfaces when the handle is used.
#[napi]
pub fn job(&self, job_id: String) -> napi::Result<crate::job::Job> {
let job = self.get_inner()?.job(job_id).default_error()?;
Ok(crate::job::Job::new(job))
}
/// List server-side jobs across the database's tables.
#[napi(catch_unwind)]
pub async fn list_jobs(&self) -> napi::Result<Vec<crate::job::JobInfo>> {
let jobs = self.get_inner()?.list_jobs().await.default_error()?;
Ok(jobs.into_iter().map(Into::into).collect())
}
/// Describe a single server-side job by id. `null` when the server has
/// no such job.
#[napi(catch_unwind)]
pub async fn get_job(
&self,
job_id: String,
) -> napi::Result<Option<crate::job::JobDescription>> {
let description = self.get_inner()?.get_job(&job_id).await.default_error()?;
Ok(description.map(Into::into))
}
/// Request cancellation of a server-side job by id. Returns true if the
/// server accepted the cancellation, false if no such job exists.
#[napi(catch_unwind)]
pub async fn cancel_job(&self, job_id: String) -> napi::Result<bool> {
self.get_inner()?.cancel_job(&job_id).await.default_error()
}
/// The lifecycle event history of a server-side job (all jobs when
/// `job_id` is null), as an Arrow IPC stream buffer. Empty when there is
/// no history.
#[napi(catch_unwind)]
pub async fn job_history(&self, job_id: Option<String>) -> napi::Result<Buffer> {
let batches = self
.get_inner()?
.job_history(job_id.as_deref())
.await
.default_error()?;
let Some(first) = batches.first() else {
return Ok(Buffer::from(Vec::<u8>::new()));
};
let mut out = Vec::new();
let mut writer = arrow_ipc::writer::StreamWriter::try_new(&mut out, &first.schema())
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
for batch in &batches {
writer
.write(batch)
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
}
writer
.finish()
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
drop(writer);
Ok(Buffer::from(out))
}
#[napi(catch_unwind)] #[napi(catch_unwind)]
/// Describe a namespace and return its properties. /// Describe a namespace and return its properties.
pub async fn describe_namespace( pub async fn describe_namespace(
+3 -13
View File
@@ -43,7 +43,6 @@ pub fn tokenize(
lower_case: Option<bool>, lower_case: Option<bool>,
stem: Option<bool>, stem: Option<bool>,
remove_stop_words: Option<bool>, remove_stop_words: Option<bool>,
custom_stop_words: Option<Vec<String>>,
ascii_folding: Option<bool>, ascii_folding: Option<bool>,
ngram_min_length: Option<u32>, ngram_min_length: Option<u32>,
ngram_max_length: Option<u32>, ngram_max_length: Option<u32>,
@@ -73,7 +72,6 @@ pub fn tokenize(
if let Some(remove_stop_words) = remove_stop_words { if let Some(remove_stop_words) = remove_stop_words {
opts = opts.remove_stop_words(remove_stop_words); opts = opts.remove_stop_words(remove_stop_words);
} }
opts = opts.custom_stop_words(custom_stop_words);
if let Some(ascii_folding) = ascii_folding { if let Some(ascii_folding) = ascii_folding {
opts = opts.ascii_folding(ascii_folding); opts = opts.ascii_folding(ascii_folding);
} }
@@ -224,13 +222,11 @@ impl Index {
lower_case: Option<bool>, lower_case: Option<bool>,
stem: Option<bool>, stem: Option<bool>,
remove_stop_words: Option<bool>, remove_stop_words: Option<bool>,
custom_stop_words: Option<Vec<String>>,
ascii_folding: Option<bool>, ascii_folding: Option<bool>,
ngram_min_length: Option<u32>, ngram_min_length: Option<u32>,
ngram_max_length: Option<u32>, ngram_max_length: Option<u32>,
prefix_only: Option<bool>, prefix_only: Option<bool>,
block_size: Option<u32>, ) -> Self {
) -> napi::Result<Self> {
let mut opts = FtsIndexBuilder::default(); let mut opts = FtsIndexBuilder::default();
if let Some(with_position) = with_position { if let Some(with_position) = with_position {
opts = opts.with_position(with_position); opts = opts.with_position(with_position);
@@ -253,7 +249,6 @@ impl Index {
if let Some(remove_stop_words) = remove_stop_words { if let Some(remove_stop_words) = remove_stop_words {
opts = opts.remove_stop_words(remove_stop_words); opts = opts.remove_stop_words(remove_stop_words);
} }
opts = opts.custom_stop_words(custom_stop_words);
if let Some(ascii_folding) = ascii_folding { if let Some(ascii_folding) = ascii_folding {
opts = opts.ascii_folding(ascii_folding); opts = opts.ascii_folding(ascii_folding);
} }
@@ -266,15 +261,10 @@ impl Index {
if let Some(prefix_only) = prefix_only { if let Some(prefix_only) = prefix_only {
opts = opts.ngram_prefix_only(prefix_only); opts = opts.ngram_prefix_only(prefix_only);
} }
if let Some(block_size) = block_size {
opts = opts
.block_size(block_size as usize)
.map_err(|err| napi::Error::from_reason(err.to_string()))?;
}
Ok(Self { Self {
inner: Mutex::new(Some(LanceDbIndex::FTS(opts))), inner: Mutex::new(Some(LanceDbIndex::FTS(opts))),
}) }
} }
#[napi(factory)] #[napi(factory)]
-123
View File
@@ -1,123 +0,0 @@
// SPDX-License-Identifier: Apache-2.0
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
use std::sync::Arc;
use napi_derive::napi;
use crate::error::NapiErrorExt;
/// A handle to an operation that may still be running.
#[napi]
pub struct Job {
inner: Arc<lancedb::Job>,
}
impl Job {
pub(crate) fn new(inner: lancedb::Job) -> Self {
Self {
inner: Arc::new(inner),
}
}
}
#[napi]
impl Job {
/// Identifies the operation on the server that is running it. Operations
/// that run in this process have no server id. The value is opaque.
#[napi(getter)]
pub fn id(&self) -> Option<String> {
self.inner.id().map(str::to_string)
}
/// The operation's current lifecycle state: "running", "finished",
/// "failed", or "cancelled".
///
/// A point snapshot; unlike {@link Job.wait} it does not block or reject
/// on a terminal failure state. States a newer server reports that this
/// client version does not know pass through as-is.
#[napi(catch_unwind)]
pub async fn status(&self) -> napi::Result<String> {
self.inner.status().await.default_error()
}
/// Wait until the operation reaches a terminal state.
#[napi(catch_unwind)]
pub async fn wait(&self) -> napi::Result<()> {
self.inner.wait().await.default_error()
}
/// Request cancellation. Cancelling a finished operation is a no-op.
#[napi(catch_unwind)]
pub async fn cancel(&self) -> napi::Result<()> {
self.inner.cancel().await.default_error()
}
}
/// A row from `Connection.listJobs`: one server-side job.
#[napi(object)]
pub struct JobInfo {
/// The job id -- what `Connection.getJob` and `Connection.cancelJob`
/// accept.
pub job_id: String,
/// The table the job runs against, without URI or namespace.
pub table: String,
pub job_type: String,
/// Lifecycle state: "running", "finished", "failed", or "cancelled".
pub state: String,
/// When the job was created, in milliseconds since the epoch.
pub created_at_millis: i64,
}
impl From<lancedb::database::JobInfo> for JobInfo {
fn from(info: lancedb::database::JobInfo) -> Self {
Self {
job_id: info.job_id,
table: info.table,
job_type: info.job_type,
state: info.state,
created_at_millis: info.created_at_millis,
}
}
}
/// The server's account of why a job failed.
#[napi(object)]
pub struct JobFailureInfo {
pub phase: Option<String>,
pub message: Option<String>,
pub retryable: Option<bool>,
}
/// A described job from `Connection.getJob`.
#[napi(object)]
pub struct JobDescription {
pub job_id: String,
pub job_type: String,
/// Lifecycle state: "running", "finished", "failed", or "cancelled".
pub state: String,
/// When the job was created, in milliseconds since the epoch.
pub creation_ms: i64,
/// The job-type-specific specification as a JSON string, when present.
pub spec_json: Option<String>,
/// Why the job failed, when the job is failed and the server reports a
/// reason.
pub failure: Option<JobFailureInfo>,
}
impl From<lancedb::database::JobDescription> for JobDescription {
fn from(description: lancedb::database::JobDescription) -> Self {
Self {
job_id: description.job_id,
job_type: description.job_type,
state: description.state,
creation_ms: description.creation_ms,
spec_json: (!description.spec.is_null()).then(|| description.spec.to_string()),
failure: description.failure.map(|failure| JobFailureInfo {
phase: failure.phase,
message: failure.message,
retryable: failure.retryable,
}),
}
}
}
-1
View File
@@ -11,7 +11,6 @@ mod error;
mod header; mod header;
mod index; mod index;
mod iterator; mod iterator;
mod job;
pub mod merge; pub mod merge;
pub mod otel; pub mod otel;
pub mod permutation; pub mod permutation;
+2 -2
View File
@@ -51,9 +51,9 @@ impl NativeMergeInsertBuilder {
} }
#[napi] #[napi]
pub fn use_lsm(&self, enable: bool) -> Self { pub fn use_lsm_write(&self, use_lsm_write: bool) -> Self {
let mut this = self.clone(); let mut this = self.clone();
this.inner.use_lsm(enable); this.inner.use_lsm_write(use_lsm_write);
this this
} }
-15
View File
@@ -168,11 +168,6 @@ impl Query {
self.inner = self.inner.clone().with_row_id(); self.inner = self.inner.clone().with_row_id();
} }
#[napi]
pub fn use_lsm(&mut self, enable: bool) {
self.inner = self.inner.clone().use_lsm(enable);
}
#[napi] #[napi]
pub fn order_by(&mut self, ordering: Option<Vec<ColumnOrdering>>) -> napi::Result<()> { pub fn order_by(&mut self, ordering: Option<Vec<ColumnOrdering>>) -> napi::Result<()> {
let ordering = ordering.map(|ordering| { let ordering = ordering.map(|ordering| {
@@ -379,11 +374,6 @@ impl VectorQuery {
self.inner = self.inner.clone().with_row_id(); self.inner = self.inner.clone().with_row_id();
} }
#[napi]
pub fn use_lsm(&mut self, enable: bool) {
self.inner = self.inner.clone().use_lsm(enable);
}
#[napi] #[napi]
pub fn rerank( pub fn rerank(
&mut self, &mut self,
@@ -489,11 +479,6 @@ impl TakeQuery {
self.inner = self.inner.clone().with_row_id(); self.inner = self.inner.clone().with_row_id();
} }
#[napi]
pub fn use_lsm(&mut self, enable: bool) {
self.inner = self.inner.clone().use_lsm(enable);
}
#[napi(catch_unwind)] #[napi(catch_unwind)]
pub async fn output_schema(&self) -> napi::Result<Buffer> { pub async fn output_schema(&self) -> napi::Result<Buffer> {
let schema = self.inner.output_schema().await.default_error()?; let schema = self.inner.output_schema().await.default_error()?;
+9 -123
View File
@@ -10,7 +10,6 @@ use lancedb::table::{
AddDataMode, ColumnAlteration as LanceColumnAlteration, Duration, AddDataMode, ColumnAlteration as LanceColumnAlteration, Duration,
FieldMetadataUpdate as LanceFieldMetadataUpdate, FtsToken as LanceDbFtsToken, FieldMetadataUpdate as LanceFieldMetadataUpdate, FtsToken as LanceDbFtsToken,
NewColumnTransform, OptimizeAction, OptimizeOptions, Ref, Table as LanceDbTable, NewColumnTransform, OptimizeAction, OptimizeOptions, Ref, Table as LanceDbTable,
TableBase as LanceTableBase,
}; };
use napi::bindgen_prelude::*; use napi::bindgen_prelude::*;
use napi::threadsafe_function::{ThreadsafeFunction, ThreadsafeFunctionCallMode}; use napi::threadsafe_function::{ThreadsafeFunction, ThreadsafeFunctionCallMode};
@@ -169,39 +168,6 @@ impl Table {
builder.execute().await.default_error() builder.execute().await.default_error()
} }
#[napi(catch_unwind)]
pub async fn create_index_async(
&self,
index: Option<&Index>,
column: String,
replace: Option<bool>,
wait_timeout_s: Option<i64>,
name: Option<String>,
train: Option<bool>,
) -> napi::Result<crate::job::Job> {
let lancedb_index = if let Some(index) = index {
index.consume()?
} else {
lancedb::index::Index::Auto
};
let mut builder = self.inner_ref()?.create_index(&[column], lancedb_index);
if let Some(replace) = replace {
builder = builder.replace(replace);
}
if let Some(timeout) = wait_timeout_s {
builder =
builder.wait_timeout(std::time::Duration::from_secs(timeout.try_into().unwrap()));
}
if let Some(name) = name {
builder = builder.name(name);
}
if let Some(train) = train {
builder = builder.train(train);
}
let job = builder.execute_async().await.default_error()?;
Ok(crate::job::Job::new(job))
}
#[napi(catch_unwind)] #[napi(catch_unwind)]
pub async fn drop_index(&self, index_name: String) -> napi::Result<()> { pub async fn drop_index(&self, index_name: String) -> napi::Result<()> {
self.inner_ref()? self.inner_ref()?
@@ -340,48 +306,12 @@ impl Table {
let transforms = NewColumnTransform::SqlExpressions(transforms); let transforms = NewColumnTransform::SqlExpressions(transforms);
let res = self let res = self
.inner_ref()? .inner_ref()?
.add_columns() .add_columns(transforms, None)
.transform(transforms)
.execute()
.await .await
.default_error()?; .default_error()?;
Ok(res.into()) Ok(res.into())
} }
#[napi(catch_unwind)]
pub async fn add_computed_columns(
&self,
columns: Vec<AddColumnsSql>,
) -> napi::Result<AddColumnsResult> {
let table = self.inner_ref()?;
let mut builder = table.add_columns();
for column in columns {
builder = builder.computed(column.name, column.value_sql);
}
let res = builder.execute().await.default_error()?;
Ok(res.into())
}
#[napi(catch_unwind)]
pub async fn refresh_column(&self, column: String) -> napi::Result<RefreshColumnResult> {
let res = self
.inner_ref()?
.refresh_column(column)
.await
.default_error()?;
Ok(res.into())
}
#[napi(catch_unwind)]
pub async fn refresh_column_async(&self, column: String) -> napi::Result<crate::job::Job> {
let job = self
.inner_ref()?
.refresh_column_async(column)
.await
.default_error()?;
Ok(crate::job::Job::new(job))
}
#[napi(catch_unwind)] #[napi(catch_unwind)]
pub async fn add_columns_with_schema( pub async fn add_columns_with_schema(
&self, &self,
@@ -393,9 +323,7 @@ impl Table {
let transforms = NewColumnTransform::AllNulls(schema); let transforms = NewColumnTransform::AllNulls(schema);
let res = self let res = self
.inner_ref()? .inner_ref()?
.add_columns() .add_columns(transforms, None)
.transform(transforms)
.execute()
.await .await
.default_error()?; .default_error()?;
Ok(res.into()) Ok(res.into())
@@ -447,18 +375,6 @@ impl Table {
Ok(res.into()) Ok(res.into())
} }
#[napi(catch_unwind)]
pub async fn add_bases(&self, bases: Vec<TableBase>) -> napi::Result<()> {
self.inner_ref()?
.add_bases(bases.into_iter().map(|base| LanceTableBase {
path: base.path,
name: base.name,
is_dataset_root: base.is_dataset_root,
}))
.await
.default_error()
}
#[napi(catch_unwind)] #[napi(catch_unwind)]
pub async fn drop_columns(&self, columns: Vec<String>) -> napi::Result<DropColumnsResult> { pub async fn drop_columns(&self, columns: Vec<String>) -> napi::Result<DropColumnsResult> {
let col_refs = columns.iter().map(String::as_str).collect::<Vec<_>>(); let col_refs = columns.iter().map(String::as_str).collect::<Vec<_>>();
@@ -713,18 +629,6 @@ impl Table {
} }
} }
#[napi(object)]
/// An extra storage prefix registered on a table.
pub struct TableBase {
/// Object store URI such as `s3://bucket/media/`.
pub path: String,
/// Optional alias.
pub name: Option<String>,
/// True when `path` is a Lance dataset root. When false, `path` is the
/// directory containing the referenced files.
pub is_dataset_root: bool,
}
#[napi(object)] #[napi(object)]
/// A description of an index currently configured on a column /// A description of an index currently configured on a column
pub struct IndexConfig { pub struct IndexConfig {
@@ -831,8 +735,7 @@ pub struct LsmWriteSpec {
pub column: Option<String>, pub column: Option<String>,
/// Bucket variant: the number of buckets, in `[1, 1024]`. /// Bucket variant: the number of buckets, in `[1, 1024]`.
pub num_buckets: Option<u32>, pub num_buckets: Option<u32>,
/// Indexes the MemWAL keeps up to date. Omitted resolves every /// Names of indexes the MemWAL should keep up to date during writes.
/// maintainable index on install; an empty array means none.
pub maintained_indexes: Option<Vec<String>>, pub maintained_indexes: Option<Vec<String>>,
/// Default `ShardWriter` configuration recorded in the MemWAL index. /// Default `ShardWriter` configuration recorded in the MemWAL index.
pub writer_config_defaults: Option<HashMap<String, String>>, pub writer_config_defaults: Option<HashMap<String, String>>,
@@ -842,6 +745,7 @@ impl TryFrom<LsmWriteSpec> for lancedb::table::LsmWriteSpec {
type Error = napi::Error; type Error = napi::Error;
fn try_from(value: LsmWriteSpec) -> napi::Result<Self> { fn try_from(value: LsmWriteSpec) -> napi::Result<Self> {
let maintained = value.maintained_indexes.unwrap_or_default();
let writer_config_defaults = value.writer_config_defaults.unwrap_or_default(); let writer_config_defaults = value.writer_config_defaults.unwrap_or_default();
let spec = match value.spec_type.as_str() { let spec = match value.spec_type.as_str() {
"bucket" => { "bucket" => {
@@ -868,7 +772,7 @@ impl TryFrom<LsmWriteSpec> for lancedb::table::LsmWriteSpec {
} }
}; };
Ok(spec Ok(spec
.with_maintained_indexes(value.maintained_indexes) .with_maintained_indexes(maintained)
.with_writer_config_defaults(writer_config_defaults)) .with_writer_config_defaults(writer_config_defaults))
} }
} }
@@ -886,7 +790,7 @@ impl From<lancedb::table::LsmWriteSpec> for LsmWriteSpec {
spec_type: "bucket".to_string(), spec_type: "bucket".to_string(),
column: Some(column), column: Some(column),
num_buckets: Some(num_buckets), num_buckets: Some(num_buckets),
maintained_indexes, maintained_indexes: Some(maintained_indexes),
writer_config_defaults: Some(writer_config_defaults), writer_config_defaults: Some(writer_config_defaults),
}, },
Native::Identity { Native::Identity {
@@ -897,7 +801,7 @@ impl From<lancedb::table::LsmWriteSpec> for LsmWriteSpec {
spec_type: "identity".to_string(), spec_type: "identity".to_string(),
column: Some(column), column: Some(column),
num_buckets: None, num_buckets: None,
maintained_indexes, maintained_indexes: Some(maintained_indexes),
writer_config_defaults: Some(writer_config_defaults), writer_config_defaults: Some(writer_config_defaults),
}, },
Native::Unsharded { Native::Unsharded {
@@ -907,7 +811,7 @@ impl From<lancedb::table::LsmWriteSpec> for LsmWriteSpec {
spec_type: "unsharded".to_string(), spec_type: "unsharded".to_string(),
column: None, column: None,
num_buckets: None, num_buckets: None,
maintained_indexes, maintained_indexes: Some(maintained_indexes),
writer_config_defaults: Some(writer_config_defaults), writer_config_defaults: Some(writer_config_defaults),
}, },
} }
@@ -1102,10 +1006,7 @@ impl From<lancedb::index::IndexStatistics> for IndexStatistics {
#[napi(object)] #[napi(object)]
pub struct TableStatistics { pub struct TableStatistics {
/// The total size, in bytes, of the table's data files, index files, and /// The total number of bytes in the table
/// overlay files
///
/// Read from the manifest, so this excludes deletion files and manifests.
pub total_bytes: i64, pub total_bytes: i64,
/// The number of rows in the table /// The number of rows in the table
@@ -1255,21 +1156,6 @@ pub struct AddColumnsResult {
pub version: i64, pub version: i64,
} }
#[napi(object)]
pub struct RefreshColumnResult {
pub rows_filled: i64,
pub version: i64,
}
impl From<lancedb::table::RefreshColumnResult> for RefreshColumnResult {
fn from(value: lancedb::table::RefreshColumnResult) -> Self {
Self {
rows_filled: value.rows_filled as i64,
version: value.version as i64,
}
}
}
impl From<lancedb::table::AddColumnsResult> for AddColumnsResult { impl From<lancedb::table::AddColumnsResult> for AddColumnsResult {
fn from(value: lancedb::table::AddColumnsResult) -> Self { fn from(value: lancedb::table::AddColumnsResult) -> Self {
Self { Self {
+49
View File
@@ -0,0 +1,49 @@
[tool.bumpversion]
current_version = "0.36.0"
parse = """(?x)
(?P<major>0|[1-9]\\d*)\\.
(?P<minor>0|[1-9]\\d*)\\.
(?P<patch>0|[1-9]\\d*)
(?:-(?P<pre_l>[a-zA-Z-]+)\\.(?P<pre_n>0|[1-9]\\d*))?
"""
serialize = [
"{major}.{minor}.{patch}-{pre_l}.{pre_n}",
"{major}.{minor}.{patch}",
]
search = "{current_version}"
replace = "{new_version}"
regex = false
ignore_missing_version = false
ignore_missing_files = false
tag = true
sign_tags = false
tag_name = "python-v{new_version}"
tag_message = "Bump version: {current_version} → {new_version}"
allow_dirty = true
commit = true
message = "Bump version: {current_version} → {new_version}"
commit_args = ""
# bump-my-version >=1.4.0 rejects pre_commit_hooks containing shell syntax unless opted in.
allow_shell_hooks = true
# Update Cargo.lock after version bump
pre_commit_hooks = [
"""
cd python && cargo update -p lancedb-python
if git diff --quiet ../Cargo.lock; then
echo "Cargo.lock unchanged"
else
git add ../Cargo.lock
echo "Updated and staged Cargo.lock"
fi
""",
]
[tool.bumpversion.parts.pre_l]
values = ["beta", "final"]
optional_value = "final"
[[tool.bumpversion.files]]
filename = "Cargo.toml"
search = "\nversion = \"{current_version}\""
replace = "\nversion = \"{new_version}\""
+3 -3
View File
@@ -1,6 +1,6 @@
[package] [package]
name = "lancedb-python" name = "lancedb-python"
version = "0.38.0-beta.0" version = "0.36.0"
publish = false publish = false
edition.workspace = true edition.workspace = true
description = "Python bindings for LanceDB" description = "Python bindings for LanceDB"
@@ -26,7 +26,7 @@ lance-namespace-impls.workspace = true
lance-io.workspace = true lance-io.workspace = true
env_logger.workspace = true env_logger.workspace = true
log.workspace = true log.workspace = true
pyo3 = { version = "0.28", features = ["extension-module", "abi3-py310", "chrono"] } pyo3 = { version = "0.28", features = ["extension-module", "abi3-py39", "chrono"] }
chrono = { version = "0.4", default-features = false, features = ["clock"] } chrono = { version = "0.4", default-features = false, features = ["clock"] }
pyo3-async-runtimes = { version = "0.28", features = [ pyo3-async-runtimes = { version = "0.28", features = [
"attributes", "attributes",
@@ -43,7 +43,7 @@ libc = "0.2"
[build-dependencies] [build-dependencies]
pyo3-build-config = { version = "0.28", features = [ pyo3-build-config = { version = "0.28", features = [
"extension-module", "extension-module",
"abi3-py310", "abi3-py39",
] } ] }
[features] [features]
+2 -3
View File
@@ -60,10 +60,10 @@ tests = [
"pytest-asyncio>=0.21", "pytest-asyncio>=0.21",
"duckdb>=0.9.0", "duckdb>=0.9.0",
"pytz>=2023.3", "pytz>=2023.3",
"polars>=0.19, <=1.32.3", "polars>=0.19, <=1.3.0",
"pyarrow<25", "pyarrow<25",
"pyarrow-stubs>=16.0", "pyarrow-stubs>=16.0",
"pylance==9.0.0rc1", "pylance==9.0.0",
"requests>=2.31.0", "requests>=2.31.0",
"datafusion>=54,<55", "datafusion>=54,<55",
"opentelemetry-sdk>=1.30.0", "opentelemetry-sdk>=1.30.0",
@@ -140,7 +140,6 @@ include = [
"python/lancedb/remote/errors.py", "python/lancedb/remote/errors.py",
"python/lancedb/embeddings/__init__.py", "python/lancedb/embeddings/__init__.py",
"python/lancedb/_lancedb.pyi", "python/lancedb/_lancedb.pyi",
"python/type_tests/connect.py",
] ]
exclude = ["python/tests/"] exclude = ["python/tests/"]
pythonVersion = "3.13" pythonVersion = "3.13"
+3 -10
View File
@@ -20,8 +20,7 @@ from .remote import ClientConfig
from .remote.db import RemoteDBConnection from .remote.db import RemoteDBConnection
from .expr import Expr, col, lit, func from .expr import Expr, col, lit, func
from .schema import blob, vector, BlobType from .schema import blob, vector, BlobType
from .job import AsyncJob, Job from .table import AsyncTable, Table
from .table import AsyncTable, Table, TableBase
from .types import BaseTokenizerType from .types import BaseTokenizerType
from ._lancedb import Session from ._lancedb import Session
from .namespace import ( from .namespace import (
@@ -259,7 +258,6 @@ def tokenize(
lower_case: bool = True, lower_case: bool = True,
stem: bool = True, stem: bool = True,
remove_stop_words: bool = True, remove_stop_words: bool = True,
custom_stop_words: Optional[List[str]] = None,
ascii_folding: bool = True, ascii_folding: bool = True,
ngram_min_length: int = 3, ngram_min_length: int = 3,
ngram_max_length: int = 3, ngram_max_length: int = 3,
@@ -267,10 +265,9 @@ def tokenize(
) -> Iterable[FtsToken]: ) -> Iterable[FtsToken]:
"""Tokenize a full-text search query using an explicit tokenizer. """Tokenize a full-text search query using an explicit tokenizer.
This does not require an FTS index. The tokenizer options match This does not require a table or FTS index. The tokenizer options match
:class:`lancedb.index.FTS`. ``custom_stop_words`` accepts a list of strings. :class:`lancedb.index.FTS`.
""" """
return _tokenize( return _tokenize(
query, query,
base_tokenizer=base_tokenizer, base_tokenizer=base_tokenizer,
@@ -279,7 +276,6 @@ def tokenize(
lower_case=lower_case, lower_case=lower_case,
stem=stem, stem=stem,
remove_stop_words=remove_stop_words, remove_stop_words=remove_stop_words,
custom_stop_words=custom_stop_words,
ascii_folding=ascii_folding, ascii_folding=ascii_folding,
ngram_min_length=ngram_min_length, ngram_min_length=ngram_min_length,
ngram_max_length=ngram_max_length, ngram_max_length=ngram_max_length,
@@ -501,7 +497,6 @@ __all__ = [
"connect_namespace", "connect_namespace",
"connect_namespace_async", "connect_namespace_async",
"AsyncConnection", "AsyncConnection",
"AsyncJob",
"AsyncLanceNamespaceDBConnection", "AsyncLanceNamespaceDBConnection",
"AsyncTable", "AsyncTable",
"FtsToken", "FtsToken",
@@ -515,12 +510,10 @@ __all__ = [
"BlobType", "BlobType",
"vector", "vector",
"DBConnection", "DBConnection",
"Job",
"LanceDBConnection", "LanceDBConnection",
"LanceNamespaceDBConnection", "LanceNamespaceDBConnection",
"RemoteDBConnection", "RemoteDBConnection",
"Session", "Session",
"Table", "Table",
"TableBase",
"__version__", "__version__",
] ]
+23 -6
View File
@@ -14,10 +14,14 @@ import pyarrow as pa
from .expr import Expr from .expr import Expr
from .schema import blob_v2_column_paths from .schema import blob_v2_column_paths
from .types import BlobMode, QueryProjection, QueryProjectionSpec from .types import BlobMode, QueryProjection, QueryProjectionSpec
from .util import get_uri_scheme
if TYPE_CHECKING: if TYPE_CHECKING:
from _typeshed import WriteableBuffer from _typeshed import WriteableBuffer
from .remote.table import RemoteTable
from .table import AsyncTable, Table
BLOB_MODE_TO_HANDLING = { BLOB_MODE_TO_HANDLING = {
"lazy": "blobs_descriptions", "lazy": "blobs_descriptions",
"bytes": "all_binary", "bytes": "all_binary",
@@ -100,6 +104,22 @@ def validate_blob_mode(blob_mode: BlobMode) -> None:
raise ValueError(f"blob_mode must be one of {modes}, got {blob_mode!r}") raise ValueError(f"blob_mode must be one of {modes}, got {blob_mode!r}")
def supports_blob_auto_row_id(table: Table | AsyncTable | RemoteTable) -> bool:
"""Blob auto row-id applies to native tables, not LanceDB Cloud."""
from .remote.table import RemoteTable
if isinstance(table, RemoteTable):
return False
inner = getattr(table, "_inner", None)
if inner is not None:
uri = inner.database().uri
if isinstance(uri, str) and get_uri_scheme(uri) == "db":
return False
return True
def projection_includes_blob_column( def projection_includes_blob_column(
projection: QueryProjection, projection: QueryProjection,
blob_columns: Iterable[str], blob_columns: Iterable[str],
@@ -144,14 +164,16 @@ def v2_projection_needs_row_id(
def blob_auto_row_id_for_scan( def blob_auto_row_id_for_scan(
table: Table | AsyncTable | RemoteTable,
schema: pa.Schema, schema: pa.Schema,
projection: QueryProjection, projection: QueryProjection,
*, *,
with_row_id: bool | None, with_row_id: bool | None,
) -> bool: ) -> bool:
"""Auto row-id only applies when the caller said nothing about row ids."""
if with_row_id is not None: if with_row_id is not None:
return False return False
if not supports_blob_auto_row_id(table):
return False
return v2_projection_needs_row_id(schema, projection, with_row_id=False) return v2_projection_needs_row_id(schema, projection, with_row_id=False)
@@ -164,11 +186,6 @@ def finalize_blob_query_table(
) -> pa.Table: ) -> pa.Table:
if user_requested_row_id or not blob_auto_row_id: if user_requested_row_id or not blob_auto_row_id:
return tbl return tbl
if "_rowid" not in tbl.column_names:
# A backend that ignores the row-id request leaves nothing to stash. Hand
# back the projection as-is so fetch_blobs raises the error that names the
# ways to supply row ids, rather than failing here about a hidden column.
return tbl
return stash_auto_row_ids(tbl, blob_paths) return stash_auto_row_ids(tbl, blob_paths)
+4 -106
View File
@@ -59,7 +59,6 @@ def tokenize(
lower_case: bool = True, lower_case: bool = True,
stem: bool = True, stem: bool = True,
remove_stop_words: bool = True, remove_stop_words: bool = True,
custom_stop_words: Optional[List[str]] = None,
ascii_folding: bool = True, ascii_folding: bool = True,
ngram_min_length: int = 3, ngram_min_length: int = 3,
ngram_max_length: int = 3, ngram_max_length: int = 3,
@@ -146,13 +145,6 @@ class Connection(object):
start_after: Optional[str], start_after: Optional[str],
limit: Optional[int], limit: Optional[int],
) -> list[str]: ... # Deprecated: Use list_tables instead ) -> list[str]: ... # Deprecated: Use list_tables instead
def job(self, job_id: str) -> Job: ...
async def list_jobs(self) -> List[JobInfo]: ...
async def get_job(self, job_id: str) -> Optional[JobDescription]: ...
async def cancel_job(self, job_id: str) -> bool: ...
async def job_history(
self, job_id: Optional[str] = None
) -> List[pa.RecordBatch]: ...
async def create_table( async def create_table(
self, self,
name: str, name: str,
@@ -198,9 +190,6 @@ class Connection(object):
async def drop_table( async def drop_table(
self, name: str, namespace_path: Optional[List[str]] = None self, name: str, namespace_path: Optional[List[str]] = None
) -> None: ... ) -> None: ...
async def drop_table_async(
self, name: str, namespace_path: Optional[List[str]] = None
) -> Job: ...
async def drop_all_tables( async def drop_all_tables(
self, namespace_path: Optional[List[str]] = None self, namespace_path: Optional[List[str]] = None
) -> None: ... ) -> None: ...
@@ -219,47 +208,6 @@ class BlobFile:
def read_range(self, offset: int, length: int) -> bytes: ... def read_range(self, offset: int, length: int) -> bytes: ...
def read_up_to(self, length: int) -> bytes: ... def read_up_to(self, length: int) -> bytes: ...
class Job:
@property
def id(self) -> Optional[str]: ...
async def status(self) -> str: ...
async def wait(self) -> None: ...
async def cancel(self) -> None: ...
class JobInfo:
@property
def job_id(self) -> str: ...
@property
def table(self) -> str: ...
@property
def job_type(self) -> str: ...
@property
def state(self) -> str: ...
@property
def created_at_millis(self) -> int: ...
class JobFailureInfo:
@property
def phase(self) -> Optional[str]: ...
@property
def message(self) -> Optional[str]: ...
@property
def retryable(self) -> Optional[bool]: ...
class JobDescription:
@property
def job_id(self) -> str: ...
@property
def job_type(self) -> str: ...
@property
def state(self) -> str: ...
@property
def creation_ms(self) -> int: ...
@property
def spec_json(self) -> Optional[str]: ...
@property
def failure(self) -> Optional[JobFailureInfo]: ...
class Table: class Table:
def name(self) -> str: ... def name(self) -> str: ...
def __repr__(self) -> str: ... def __repr__(self) -> str: ...
@@ -299,28 +247,6 @@ class Table:
name: Optional[str], name: Optional[str],
train: Optional[bool], train: Optional[bool],
): ... ): ...
async def create_index_async(
self,
column: str,
index: Union[
IvfFlat,
IvfSq,
IvfPq,
HnswPq,
HnswSq,
HnswFlat,
BTree,
Bitmap,
LabelList,
Fm,
FTS,
],
replace: Optional[bool],
wait_timeout: Optional[object],
*,
name: Optional[str],
train: Optional[bool],
) -> Job: ...
async def list_versions(self) -> List[Dict[str, Any]]: ... async def list_versions(self) -> List[Dict[str, Any]]: ...
async def version(self) -> int: ... async def version(self) -> int: ...
async def checkout(self, version: Union[int, str]): ... async def checkout(self, version: Union[int, str]): ...
@@ -338,11 +264,6 @@ class Table:
) -> list[FtsToken]: ... ) -> list[FtsToken]: ...
async def delete(self, filter: Union[str, PyExpr]) -> DeleteResult: ... async def delete(self, filter: Union[str, PyExpr]) -> DeleteResult: ...
async def add_columns(self, columns: list[tuple[str, str]]) -> AddColumnsResult: ... async def add_columns(self, columns: list[tuple[str, str]]) -> AddColumnsResult: ...
async def add_computed_columns(
self, columns: list[tuple[str, str]]
) -> AddColumnsResult: ...
async def refresh_column(self, column: str) -> RefreshColumnResult: ...
async def refresh_column_async(self, column: str) -> Job: ...
async def add_columns_with_schema(self, schema: pa.Schema) -> AddColumnsResult: ... async def add_columns_with_schema(self, schema: pa.Schema) -> AddColumnsResult: ...
async def alter_columns( async def alter_columns(
self, columns: list[dict[str, Any]] self, columns: list[dict[str, Any]]
@@ -363,10 +284,6 @@ class Table:
async def set_lsm_write_spec(self, spec: LsmWriteSpec) -> None: ... async def set_lsm_write_spec(self, spec: LsmWriteSpec) -> None: ...
async def unset_lsm_write_spec(self) -> None: ... async def unset_lsm_write_spec(self) -> None: ...
async def get_lsm_write_spec(self) -> Optional[LsmWriteSpec]: ... async def get_lsm_write_spec(self) -> Optional[LsmWriteSpec]: ...
async def checkpoint_lsm(self) -> None: ...
async def flush_lsm(self) -> None: ...
async def compact_lsm(self) -> None: ...
async def get_lsm_stats(self, include_generation_rows: bool) -> Optional[dict]: ...
async def close_lsm_writers(self) -> None: ... async def close_lsm_writers(self) -> None: ...
@property @property
def tags(self) -> Tags: ... def tags(self) -> Tags: ...
@@ -377,15 +294,9 @@ class Table:
def take_offsets(self, offsets: list[int]) -> TakeQuery: ... def take_offsets(self, offsets: list[int]) -> TakeQuery: ...
def take_row_ids(self, row_ids: list[int]) -> TakeQuery: ... def take_row_ids(self, row_ids: list[int]) -> TakeQuery: ...
async def blob_columns(self) -> list[str]: ... async def blob_columns(self) -> list[str]: ...
async def add_bases(self, bases: list[Any]) -> None: ...
async def fetch_blobs( async def fetch_blobs(
self, column: str, row_ids: list[int] self, column: str, row_ids: list[int]
) -> pa.LargeBinaryArray: ... ) -> pa.LargeBinaryArray: ...
async def fetch_blob_ranges(
self,
column: str,
requests: List[Tuple[int, int, int]],
) -> pa.LargeBinaryArray: ...
async def fetch_blob_files( async def fetch_blob_files(
self, column: str, row_ids: list[int] self, column: str, row_ids: list[int]
) -> list[Optional[BlobFile]]: ... ) -> list[Optional[BlobFile]]: ...
@@ -480,7 +391,6 @@ class Query:
def fast_search(self): ... def fast_search(self): ...
def with_row_id(self): ... def with_row_id(self): ...
def postfilter(self): ... def postfilter(self): ...
def use_lsm(self, enable: bool): ...
def nearest_to(self, query_vec: pa.Array) -> VectorQuery: ... def nearest_to(self, query_vec: pa.Array) -> VectorQuery: ...
def nearest_to_text(self, query: dict) -> FTSQuery: ... def nearest_to_text(self, query: dict) -> FTSQuery: ...
def order_by(self, ordering: Optional[List[ColumnOrdering]]): ... def order_by(self, ordering: Optional[List[ColumnOrdering]]): ...
@@ -497,7 +407,6 @@ class Query:
class TakeQuery: class TakeQuery:
def select(self, columns: List[str]): ... def select(self, columns: List[str]): ...
def with_row_id(self): ... def with_row_id(self): ...
def use_lsm(self, enable: bool): ...
async def output_schema(self) -> pa.Schema: ... async def output_schema(self) -> pa.Schema: ...
async def execute(self) -> RecordBatchStream: ... async def execute(self) -> RecordBatchStream: ...
async def explain_plan(self, verbose: Optional[bool]) -> str: ... async def explain_plan(self, verbose: Optional[bool]) -> str: ...
@@ -516,7 +425,6 @@ class FTSQuery:
def fast_search(self): ... def fast_search(self): ...
def with_row_id(self): ... def with_row_id(self): ...
def postfilter(self): ... def postfilter(self): ...
def use_lsm(self, enable: bool): ...
def get_query(self) -> str: ... def get_query(self) -> str: ...
def add_query_vector(self, query_vec: pa.Array) -> None: ... def add_query_vector(self, query_vec: pa.Array) -> None: ...
def nearest_to(self, query_vec: pa.Array) -> HybridQuery: ... def nearest_to(self, query_vec: pa.Array) -> HybridQuery: ...
@@ -544,7 +452,6 @@ class VectorQuery:
def column(self, column: str): ... def column(self, column: str): ...
def distance_type(self, distance_type: str): ... def distance_type(self, distance_type: str): ...
def postfilter(self): ... def postfilter(self): ...
def use_lsm(self, enable: bool): ...
def refine_factor(self, refine_factor: int): ... def refine_factor(self, refine_factor: int): ...
def nprobes(self, nprobes: int): ... def nprobes(self, nprobes: int): ...
def minimum_nprobes(self, minimum_nprobes: int): ... def minimum_nprobes(self, minimum_nprobes: int): ...
@@ -568,7 +475,6 @@ class HybridQuery:
def fast_search(self): ... def fast_search(self): ...
def with_row_id(self): ... def with_row_id(self): ...
def postfilter(self): ... def postfilter(self): ...
def use_lsm(self, enable: bool): ...
def distance_type(self, distance_type: str): ... def distance_type(self, distance_type: str): ...
def refine_factor(self, refine_factor: int): ... def refine_factor(self, refine_factor: int): ...
def nprobes(self, nprobes: int): ... def nprobes(self, nprobes: int): ...
@@ -593,7 +499,6 @@ class PyQueryRequest:
select: Optional[Union[str, List[str]]] select: Optional[Union[str, List[str]]]
fast_search: Optional[bool] fast_search: Optional[bool]
with_row_id: Optional[bool] with_row_id: Optional[bool]
use_lsm: Optional[bool]
column: Optional[str] column: Optional[str]
query_vector: Optional[List[pa.Array]] query_vector: Optional[List[pa.Array]]
minimum_nprobes: Optional[int] minimum_nprobes: Optional[int]
@@ -662,10 +567,9 @@ class LsmWriteSpec:
def identity(column: str) -> "LsmWriteSpec": ... def identity(column: str) -> "LsmWriteSpec": ...
@staticmethod @staticmethod
def unsharded() -> "LsmWriteSpec": ... def unsharded() -> "LsmWriteSpec": ...
def with_maintained_indexes(self, indexes: Optional[List[str]]) -> "LsmWriteSpec": def with_maintained_indexes(self, indexes: List[str]) -> "LsmWriteSpec":
"""Set which indexes the MemWAL keeps up to date. None resolves every """Return a copy of this spec asking the MemWAL to keep the named
index on the table at install, failing if one cannot be maintained; indexes up to date as rows are appended."""
a list is verbatim, empty means none."""
... ...
def with_writer_config_defaults(self, defaults: Dict[str, str]) -> "LsmWriteSpec": def with_writer_config_defaults(self, defaults: Dict[str, str]) -> "LsmWriteSpec":
"""Return a copy of this spec recording the given default """Return a copy of this spec recording the given default
@@ -680,19 +584,13 @@ class LsmWriteSpec:
@property @property
def num_buckets(self) -> Optional[int]: ... def num_buckets(self) -> Optional[int]: ...
@property @property
def maintained_indexes(self) -> Optional[List[str]]: def maintained_indexes(self) -> List[str]: ...
"""Indexes the MemWAL keeps up to date, or None for every supported one."""
...
@property @property
def writer_config_defaults(self) -> Dict[str, str]: ... def writer_config_defaults(self) -> Dict[str, str]: ...
class AddColumnsResult: class AddColumnsResult:
version: int version: int
class RefreshColumnResult:
rows_filled: int
version: int
class AlterColumnsResult: class AlterColumnsResult:
version: int version: int
+10 -222
View File
@@ -45,7 +45,6 @@ from lance_namespace.errors import NamespaceNotEmptyError, TableNotFoundError
from . import __version__ from . import __version__
from ._lancedb import connect as lancedb_connect # type: ignore from ._lancedb import connect as lancedb_connect # type: ignore
from .job import AsyncJob, Job
from .table import ( from .table import (
AsyncTable, AsyncTable,
LanceTable, LanceTable,
@@ -64,7 +63,6 @@ if TYPE_CHECKING:
from .pydantic import LanceModel from .pydantic import LanceModel
from ._lancedb import Connection as LanceDbConnection from ._lancedb import Connection as LanceDbConnection
from ._lancedb import JobDescription, JobInfo
from .common import DATA, URI from .common import DATA, URI
from .embeddings import EmbeddingFunctionConfig from .embeddings import EmbeddingFunctionConfig
from ._lancedb import Session from ._lancedb import Session
@@ -180,51 +178,6 @@ class DBConnection(EnforceOverrides):
"Namespace operations are not supported for this connection type" "Namespace operations are not supported for this connection type"
) )
def namespace_exists(self, namespace_id: List[str]) -> bool:
"""Check if a namespace exists.
Parameters
----------
namespace_id: List[str]
The namespace identifier to check.
Returns
-------
bool
True if the namespace exists, False otherwise.
Raises
------
NotImplementedError
If the connection type does not support namespace operations.
"""
raise NotImplementedError(
"Namespace operations are not supported for this connection type"
)
def table_exists(self, table_id: List[str]) -> bool:
"""Check if a table exists.
Parameters
----------
table_id: List[str]
The table identifier to check (full path including namespace
segments and table name).
Returns
-------
bool
True if the table exists, False otherwise.
Raises
------
NotImplementedError
If the connection type does not support namespace operations.
"""
raise NotImplementedError(
"Namespace operations are not supported for this connection type"
)
def list_tables( def list_tables(
self, self,
namespace_path: Optional[List[str]] = None, namespace_path: Optional[List[str]] = None,
@@ -406,7 +359,7 @@ class DBConnection(EnforceOverrides):
Data is converted to Arrow before being written to disk. For maximum Data is converted to Arrow before being written to disk. For maximum
control over how data is saved, either provide the PyArrow schema to control over how data is saved, either provide the PyArrow schema to
convert to or else provide a [PyArrow Table][pyarrow.Table] directly. convert to or else provide a [PyArrow Table](pyarrow.Table) directly.
>>> import pyarrow as pa >>> import pyarrow as pa
>>> custom_schema = pa.schema([ >>> custom_schema = pa.schema([
@@ -524,12 +477,6 @@ class DBConnection(EnforceOverrides):
namespace_path = [] namespace_path = []
raise NotImplementedError raise NotImplementedError
def drop_table_async(
self, name: str, namespace_path: Optional[List[str]] = None
) -> Job:
"""Start dropping a table and return its cleanup job."""
raise NotImplementedError
def rename_table( def rename_table(
self, self,
cur_name: str, cur_name: str,
@@ -616,46 +563,6 @@ class DBConnection(EnforceOverrides):
""" """
raise NotImplementedError("serialize is not supported for this connection type") raise NotImplementedError("serialize is not supported for this connection type")
def job(self, job_id: str) -> Job:
"""A [Job][lancedb.job.Job] handle for a server-side job by id.
The handle is constructed without a server round trip; an unknown id
surfaces when the handle is used. Dropping the handle has no effect
on the job itself.
"""
raise NotImplementedError("job is not supported for this connection type")
def list_jobs(self) -> List[JobInfo]:
"""List server-side jobs across the database's tables."""
raise NotImplementedError("list_jobs is not supported for this connection type")
def get_job(self, job_id: str) -> Optional[JobDescription]:
"""Describe a single server-side job by id.
Returns None when the server has no such job.
"""
raise NotImplementedError("get_job is not supported for this connection type")
def cancel_job(self, job_id: str) -> bool:
"""Request cancellation of a server-side job by id.
Returns True if the server accepted the cancellation, False if no
such job exists. Cancelling an already-terminal job is a no-op
success.
"""
raise NotImplementedError(
"cancel_job is not supported for this connection type"
)
def job_history(self, job_id: Optional[str] = None) -> List[pa.RecordBatch]:
"""The lifecycle event history of a server-side job, as Arrow batches.
Lists history across all jobs when `job_id` is None.
"""
raise NotImplementedError(
"job_history is not supported for this connection type"
)
class LanceDBConnection(DBConnection): class LanceDBConnection(DBConnection):
""" """
@@ -713,9 +620,6 @@ class LanceDBConnection(DBConnection):
self._namespace_client_properties = namespace_client_properties self._namespace_client_properties = namespace_client_properties
if _inner is not None: if _inner is not None:
self._conn = _inner self._conn = _inner
# Native-derived wrappers resolve this in their async reconstruction
# path so construction never synchronously re-enters LOOP.
self._read_consistency_interval = read_consistency_interval
self._cached_namespace_client = None self._cached_namespace_client = None
return return
@@ -765,14 +669,11 @@ class LanceDBConnection(DBConnection):
# storage_options. Also, this class really shouldn't be holding any state # storage_options. Also, this class really shouldn't be holding any state
# beyond _conn. # beyond _conn.
self._conn = AsyncConnection(LOOP.run(do_connect())) self._conn = AsyncConnection(LOOP.run(do_connect()))
# Keep property access synchronous so debugger introspection cannot wait on
# the background loop while that thread is suspended at a breakpoint.
self._read_consistency_interval = read_consistency_interval
self._cached_namespace_client: Optional[LanceNamespace] = None self._cached_namespace_client: Optional[LanceNamespace] = None
@property @property
def read_consistency_interval(self) -> Optional[timedelta]: def read_consistency_interval(self) -> Optional[timedelta]:
return self._read_consistency_interval return LOOP.run(self._conn.get_read_consistency_interval())
@property @property
def session(self) -> Optional[Session]: def session(self) -> Optional[Session]:
@@ -783,19 +684,15 @@ class LanceDBConnection(DBConnection):
return self._conn.uri return self._conn.uri
@classmethod @classmethod
def from_inner( def from_inner(cls, inner: LanceDbConnection):
cls, return cls(None, _inner=inner)
inner: LanceDbConnection,
read_consistency_interval: Optional[timedelta],
):
return cls(
None,
read_consistency_interval=read_consistency_interval,
_inner=inner,
)
def __repr__(self) -> str: def __repr__(self) -> str:
return f"{self.__class__.__name__}(uri={self._conn.uri!r})" val = f"{self.__class__.__name__}(uri={self._conn.uri!r}"
if self.read_consistency_interval is not None:
val += f", read_consistency_interval={repr(self.read_consistency_interval)}"
val += ")"
return val
@override @override
def serialize(self) -> str: def serialize(self) -> str:
@@ -1192,20 +1089,6 @@ class LanceDBConnection(DBConnection):
) )
) )
@override
def drop_table_async(
self, name: str, namespace_path: Optional[List[str]] = None
) -> Job:
"""Start dropping a table and return its cleanup job.
The table may become unavailable before its data files are removed.
Call :meth:`Job.wait` to wait for cleanup to finish.
"""
if namespace_path is None:
namespace_path = []
job = LOOP.run(self._conn.drop_table_async(name, namespace_path=namespace_path))
return Job(job if isinstance(job, AsyncJob) else AsyncJob(job))
@override @override
def drop_all_tables(self, namespace_path: Optional[List[str]] = None): def drop_all_tables(self, namespace_path: Optional[List[str]] = None):
if namespace_path is None: if namespace_path is None:
@@ -1246,47 +1129,6 @@ class LanceDBConnection(DBConnection):
) )
) )
@override
def job(self, job_id: str) -> Job:
"""A [Job][lancedb.job.Job] handle for a server-side job by id.
The handle is constructed without a server round trip; an unknown id
surfaces when the handle is used. Dropping the handle has no effect
on the job itself.
"""
return Job(self._conn.job(job_id))
@override
def list_jobs(self) -> List[JobInfo]:
"""List server-side jobs across the database's tables."""
return LOOP.run(self._conn.list_jobs())
@override
def get_job(self, job_id: str) -> Optional[JobDescription]:
"""Describe a single server-side job by id.
Returns None when the server has no such job.
"""
return LOOP.run(self._conn.get_job(job_id))
@override
def cancel_job(self, job_id: str) -> bool:
"""Request cancellation of a server-side job by id.
Returns True if the server accepted the cancellation, False if no
such job exists. Cancelling an already-terminal job is a no-op
success.
"""
return LOOP.run(self._conn.cancel_job(job_id))
@override
def job_history(self, job_id: Optional[str] = None) -> List[pa.RecordBatch]:
"""The lifecycle event history of a server-side job, as Arrow batches.
Lists history across all jobs when `job_id` is None.
"""
return LOOP.run(self._conn.job_history(job_id))
@override @override
def namespace_client(self) -> LanceNamespace: def namespace_client(self) -> LanceNamespace:
"""Get the equivalent namespace client for this connection. """Get the equivalent namespace client for this connection.
@@ -1687,7 +1529,7 @@ class AsyncConnection(object):
Data is converted to Arrow before being written to disk. For maximum Data is converted to Arrow before being written to disk. For maximum
control over how data is saved, either provide the PyArrow schema to control over how data is saved, either provide the PyArrow schema to
convert to or else provide a [PyArrow Table][pyarrow.Table] directly. convert to or else provide a [PyArrow Table](pyarrow.Table) directly.
>>> import pyarrow as pa >>> import pyarrow as pa
>>> custom_schema = pa.schema([ >>> custom_schema = pa.schema([
@@ -1983,23 +1825,6 @@ class AsyncConnection(object):
if f"Table '{name}' was not found" not in str(e): if f"Table '{name}' was not found" not in str(e):
raise e raise e
async def drop_table_async(
self,
name: str,
*,
namespace_path: Optional[List[str]] = None,
) -> AsyncJob:
"""Start dropping a table and return its cleanup job.
The table may become unavailable before its data files are removed.
Await :meth:`AsyncJob.wait` to wait for cleanup to finish.
"""
if namespace_path is None:
namespace_path = []
return AsyncJob(
await self._inner.drop_table_async(name, namespace_path=namespace_path)
)
async def drop_all_tables(self, namespace_path: Optional[List[str]] = None): async def drop_all_tables(self, namespace_path: Optional[List[str]] = None):
"""Drop all tables from the database. """Drop all tables from the database.
@@ -2013,43 +1838,6 @@ class AsyncConnection(object):
namespace_path = [] namespace_path = []
await self._inner.drop_all_tables(namespace_path=namespace_path) await self._inner.drop_all_tables(namespace_path=namespace_path)
def job(self, job_id: str) -> AsyncJob:
"""An [AsyncJob][lancedb.job.AsyncJob] handle for a server-side job
by id.
The handle is constructed without a server round trip; an unknown id
surfaces when the handle is used. Dropping the handle has no effect
on the job itself.
"""
return AsyncJob(self._inner.job(job_id))
async def list_jobs(self) -> List[JobInfo]:
"""List server-side jobs across the database's tables."""
return await self._inner.list_jobs()
async def get_job(self, job_id: str) -> Optional[JobDescription]:
"""Describe a single server-side job by id.
Returns None when the server has no such job.
"""
return await self._inner.get_job(job_id)
async def cancel_job(self, job_id: str) -> bool:
"""Request cancellation of a server-side job by id.
Returns True if the server accepted the cancellation, False if no
such job exists. Cancelling an already-terminal job is a no-op
success.
"""
return await self._inner.cancel_job(job_id)
async def job_history(self, job_id: Optional[str] = None) -> List[pa.RecordBatch]:
"""The lifecycle event history of a server-side job, as Arrow batches.
Lists history across all jobs when `job_id` is None.
"""
return await self._inner.job_history(job_id)
async def namespace_client(self) -> LanceNamespace: async def namespace_client(self) -> LanceNamespace:
"""Get the equivalent namespace client for this connection. """Get the equivalent namespace client for this connection.
@@ -21,32 +21,3 @@ from .watsonx import WatsonxEmbeddings
from .voyageai import VoyageAIEmbeddingFunction from .voyageai import VoyageAIEmbeddingFunction
from .colpali import ColPaliEmbeddings from .colpali import ColPaliEmbeddings
from .siglip import SigLipEmbeddings from .siglip import SigLipEmbeddings
# The API reference renders this package with a single mkdocstrings directive,
# which only picks up names listed here. New embedding functions must be added
# to both the imports above and this list, or they will silently go undocumented.
__all__ = [
"EmbeddingFunction",
"EmbeddingFunctionConfig",
"TextEmbeddingFunction",
"EmbeddingFunctionRegistry",
"get_registry",
"register",
"SentenceTransformerEmbeddings",
"OpenAIEmbeddings",
"OpenClipEmbeddings",
"BedRockText",
"CohereEmbeddingFunction",
"GeminiText",
"GteEmbeddings",
"InstructorEmbeddingFunction",
"JinaEmbeddings",
"OllamaEmbeddings",
"TransformersEmbeddingFunction",
"ColbertEmbeddings",
"VoyageAIEmbeddingFunction",
"WatsonxEmbeddings",
"ColPaliEmbeddings",
"ImageBindEmbeddings",
"SigLipEmbeddings",
]
+5 -5
View File
@@ -21,20 +21,20 @@ class BedRockText(TextEmbeddingFunction):
""" """
Parameters Parameters
---------- ----------
name : str, default "amazon.titan-embed-text-v1" name: str, default "amazon.titan-embed-text-v1"
The model ID of the bedrock model to use. Supported models for are: The model ID of the bedrock model to use. Supported models for are:
- amazon.titan-embed-text-v1 - amazon.titan-embed-text-v1
- cohere.embed-english-v3 - cohere.embed-english-v3
- cohere.embed-multilingual-v3 - cohere.embed-multilingual-v3
region : str, default "us-east-1" region: str, default "us-east-1"
Optional name of the AWS Region in which the service should be called. Optional name of the AWS Region in which the service should be called.
profile_name : str, default None profile_name: str, default None
Optional name of the AWS profile to use for calling the Bedrock service. Optional name of the AWS profile to use for calling the Bedrock service.
If not specified, the default profile will be used. If not specified, the default profile will be used.
assumed_role : str, default None assumed_role: str, default None
Optional ARN of an AWS IAM role to assume for calling the Bedrock service. Optional ARN of an AWS IAM role to assume for calling the Bedrock service.
If not specified, the current active credentials will be used. If not specified, the current active credentials will be used.
role_session_name : str, default "lancedb-embeddings" role_session_name: str, default "lancedb-embeddings"
Optional name of the AWS IAM role session to use for calling the Bedrock Optional name of the AWS IAM role session to use for calling the Bedrock
service. If not specified, "lancedb-embeddings" name will be used. service. If not specified, "lancedb-embeddings" name will be used.
+3 -5
View File
@@ -22,7 +22,7 @@ class CohereEmbeddingFunction(TextEmbeddingFunction):
Parameters Parameters
---------- ----------
name : str, default "embed-multilingual-v2.0" name: str, default "embed-multilingual-v2.0"
The name of the model to use. List of acceptable models: The name of the model to use. List of acceptable models:
* embed-english-v3.0 * embed-english-v3.0
@@ -33,14 +33,12 @@ class CohereEmbeddingFunction(TextEmbeddingFunction):
* embed-english-light-v2.0 * embed-english-light-v2.0
* embed-multilingual-v2.0 * embed-multilingual-v2.0
source_input_type : str, default "search_document" source_input_type: str, default "search_document"
The input type for the source column in the database The input type for the source column in the database
query_input_type : str, default "search_query" query_input_type: str, default "search_query"
The input type for the query column in the database The input type for the query column in the database
Notes
-----
Cohere supports following input types: Cohere supports following input types:
| Input Type | Description | | Input Type | Description |
+2 -2
View File
@@ -44,7 +44,7 @@ class ColPaliEmbeddings(EmbeddingFunction):
The token pooling strategy to use, by default "hierarchical". The token pooling strategy to use, by default "hierarchical".
- "hierarchical": Progressively pools tokens to reduce sequence length. - "hierarchical": Progressively pools tokens to reduce sequence length.
- "lambda": A simpler pooling that uses a custom `pooling_func`. - "lambda": A simpler pooling that uses a custom `pooling_func`.
pooling_func : typing.Callable, optional pooling_func: typing.Callable, optional
A function to use for pooling when `pooling_strategy` is "lambda". A function to use for pooling when `pooling_strategy` is "lambda".
pool_factor : int pool_factor : int
Factor to reduce sequence length if token pooling is enabled (default 2). Factor to reduce sequence length if token pooling is enabled (default 2).
@@ -52,7 +52,7 @@ class ColPaliEmbeddings(EmbeddingFunction):
Quantization configuration for the model. (default None, bitsandbytes needed) Quantization configuration for the model. (default None, bitsandbytes needed)
batch_size : int batch_size : int
Batch size for processing inputs (default 2). Batch size for processing inputs (default 2).
offload_folder : str, optional offload_folder: str, optional
Folder to offload model weights if using CPU offloading (default None). This is Folder to offload model weights if using CPU offloading (default None). This is
useful for large models that do not fit in memory. useful for large models that do not fit in memory.
""" """
@@ -48,16 +48,16 @@ class GeminiText(TextEmbeddingFunction):
Parameters Parameters
---------- ----------
name : str, default "gemini-embedding-001" name: str, default "gemini-embedding-001"
The name of the model to use. Supported models include: The name of the model to use. Supported models include:
- "gemini-embedding-001" (768 dimensions) - "gemini-embedding-001" (768 dimensions)
Note: The legacy "models/embedding-001" format is also supported but Note: The legacy "models/embedding-001" format is also supported but
"gemini-embedding-001" is recommended. "gemini-embedding-001" is recommended.
query_task_type : str, default "retrieval_query" query_task_type: str, default "retrieval_query"
Sets the task type for the queries. Sets the task type for the queries.
source_task_type : str, default "retrieval_document" source_task_type: str, default "retrieval_document"
Sets the task type for ingestion. Sets the task type for ingestion.
Examples Examples
+4 -4
View File
@@ -26,13 +26,13 @@ class GteEmbeddings(TextEmbeddingFunction):
Parameters Parameters
---------- ----------
name : str, default "thenlper/gte-large" name: str, default "thenlper/gte-large"
The name of the model to use. The name of the model to use.
device : str, default "cpu" device: str, default "cpu"
Sets the device type for the model. Sets the device type for the model.
normalize : str, default "True" normalize: str, default "True"
Controls normalize param in encode function for the transformer. Controls normalize param in encode function for the transformer.
mlx : bool, default False mlx: bool, default False
Controls which model to use. False for gte-large,True for the mlx version. Controls which model to use. False for gte-large,True for the mlx version.
Examples Examples
+10 -9
View File
@@ -35,23 +35,23 @@ class InstructorEmbeddingFunction(TextEmbeddingFunction):
Parameters Parameters
---------- ----------
name : str name: str
The name of the model to use. Available models are listed at The name of the model to use. Available models are listed at
https://github.com/xlang-ai/instructor-embedding#model-list; https://github.com/xlang-ai/instructor-embedding#model-list;
The default model is hkunlp/instructor-base The default model is hkunlp/instructor-base
batch_size : int, default 32 batch_size: int, default 32
The batch size to use when generating embeddings The batch size to use when generating embeddings
device : str, default "cpu" device: str, default "cpu"
The device to use when generating embeddings The device to use when generating embeddings
show_progress_bar : bool, default True show_progress_bar: bool, default True
Whether to show a progress bar when generating embeddings Whether to show a progress bar when generating embeddings
normalize_embeddings : bool, default True normalize_embeddings: bool, default True
Whether to normalize the embeddings Whether to normalize the embeddings
quantize : bool, default False quantize: bool, default False
Whether to quantize the model Whether to quantize the model
source_instruction : str, default "represent the document for retrieval" source_instruction: str, default "represent the document for retrieval"
The instruction for the source column The instruction for the source column
query_instruction : str, default "represent the document for retrieving the most query_instruction: str, default "represent the document for retrieving the most
similar documents" similar documents"
The instruction for the query The instruction for the query
@@ -101,7 +101,8 @@ class InstructorEmbeddingFunction(TextEmbeddingFunction):
@weak_lru(maxsize=1) @weak_lru(maxsize=1)
def ndims(self): def ndims(self):
return len(self.generate_embeddings([[self.source_instruction, "foo"]])[0]) model = self.get_model()
return model.encode("foo").shape[0]
def compute_query_embeddings(self, query: str, *args, **kwargs) -> List[np.array]: def compute_query_embeddings(self, query: str, *args, **kwargs) -> List[np.array]:
return self.generate_embeddings([[self.query_instruction, query]]) return self.generate_embeddings([[self.query_instruction, query]])
+5 -6
View File
@@ -40,10 +40,10 @@ class JinaEmbeddings(EmbeddingFunction):
Parameters Parameters
---------- ----------
name : str, default "jina-clip-v1". Note that some models support both image name: str, default "jina-clip-v1". Note that some models support both image
and text embeddings and some just text embedding and text embeddings and some just text embedding
api_key : str, default None api_key: str, default None
The api key to access Jina API. If you pass None, you can set JINA_API_KEY The api key to access Jina API. If you pass None, you can set JINA_API_KEY
environment variable environment variable
@@ -87,13 +87,12 @@ class JinaEmbeddings(EmbeddingFunction):
if isinstance(image, bytes): if isinstance(image, bytes):
image_dict = {"image": base64.b64encode(image).decode("utf-8")} image_dict = {"image": base64.b64encode(image).decode("utf-8")}
elif isinstance(image, (str, Path)): elif isinstance(image, (str, Path)):
parsed = urlparse(str(image)) parsed = urlparse.urlparse(image)
# TODO handle drive letter on windows.
PIL_Image = attempt_import_or_raise("PIL.Image", "pillow") PIL_Image = attempt_import_or_raise("PIL.Image", "pillow")
if parsed.scheme == "file": if parsed.scheme == "file":
pil_image = PIL_Image.open(parsed.path) pil_image = PIL_Image.open(parsed.path)
elif parsed.scheme == "" or (os.name == "nt" and len(parsed.scheme) == 1): elif parsed.scheme == "":
# A Windows drive letter parses as a one-character scheme
# ("C:\\img.png" -> scheme="c"), so treat it as a local path.
pil_image = PIL_Image.open(image if os.name == "nt" else parsed.path) pil_image = PIL_Image.open(image if os.name == "nt" else parsed.path)
elif parsed.scheme.startswith("http"): elif parsed.scheme.startswith("http"):
pil_image = PIL_Image.open(io.BytesIO(url_retrieve(image))) pil_image = PIL_Image.open(io.BytesIO(url_retrieve(image)))
@@ -21,13 +21,13 @@ class SentenceTransformerEmbeddings(TextEmbeddingFunction):
Parameters Parameters
---------- ----------
name : str, default "all-MiniLM-L6-v2" name: str, default "all-MiniLM-L6-v2"
The name of the model to use. The name of the model to use.
device : str, default "cpu" device: str, default "cpu"
The device to use for the model The device to use for the model
normalize : bool, default True normalize: bool, default True
Whether to normalize the embeddings Whether to normalize the embeddings
trust_remote_code : bool, default True trust_remote_code: bool, default True
Whether to trust the remote code Whether to trust the remote code
""" """
+2 -2
View File
@@ -167,7 +167,7 @@ class VoyageAIEmbeddingFunction(EmbeddingFunction):
Parameters Parameters
---------- ----------
name : str name: str
The name of the model to use. List of acceptable models: The name of the model to use. List of acceptable models:
* voyage-4 (1024 dims, general-purpose and multilingual retrieval) * voyage-4 (1024 dims, general-purpose and multilingual retrieval)
@@ -185,7 +185,7 @@ class VoyageAIEmbeddingFunction(EmbeddingFunction):
* voyage-law-2 * voyage-law-2
* voyage-code-2 * voyage-code-2
output_dimension : int, optional output_dimension: int, optional
The output dimension for models that support flexible dimensions. The output dimension for models that support flexible dimensions.
Currently only voyage-multimodal-3.5 supports this feature. Currently only voyage-multimodal-3.5 supports this feature.
Valid options: 256, 512, 1024 (default), 2048. Valid options: 256, 512, 1024 (default), 2048.
-12
View File
@@ -23,15 +23,3 @@ class MissingColumnError(KeyError):
return ( return (
f"Error: Column '{self.column_name}' does not exist in the DataFrame object" f"Error: Column '{self.column_name}' does not exist in the DataFrame object"
) )
class JobFailedError(RuntimeError):
"""Exception raised when an asynchronous job reaches the failed state."""
pass
class JobCancelledError(RuntimeError):
"""Exception raised when an asynchronous job was cancelled."""
pass
+24 -44
View File
@@ -2,7 +2,7 @@
# SPDX-FileCopyrightText: Copyright The LanceDB Authors # SPDX-FileCopyrightText: Copyright The LanceDB Authors
from dataclasses import dataclass from dataclasses import dataclass
from typing import List, Literal, Optional from typing import Literal, Optional
from ._lancedb import ( from ._lancedb import (
IndexConfig, IndexConfig,
@@ -115,12 +115,6 @@ class FTS:
For example, it works with `title`, `description`, `content`, etc. For example, it works with `title`, `description`, `content`, etc.
Examples
--------
Create an index configuration that uses 256-document posting blocks:
>>> config = FTS(block_size=256)
Attributes Attributes
---------- ----------
with_position : bool, default False with_position : bool, default False
@@ -151,18 +145,9 @@ class FTS:
remove_stop_words : bool, default True remove_stop_words : bool, default True
Whether to remove stop words. Stop words are common words that are often Whether to remove stop words. Stop words are common words that are often
removed from text before indexing. For example, in English "the" and "and". removed from text before indexing. For example, in English "the" and "and".
custom_stop_words : list of str, optional
Custom words replace the built-in language stop words
and only take effect when ``remove_stop_words`` is True. ``None`` uses
the built-in language list, while an empty list explicitly uses no
stop words.
ascii_folding : bool, default True ascii_folding : bool, default True
Whether to fold ASCII characters. This converts accented characters to Whether to fold ASCII characters. This converts accented characters to
their ASCII equivalent. For example, "café" would be converted to "cafe". their ASCII equivalent. For example, "café" would be converted to "cafe".
block_size : int, default 128
The number of documents per compressed posting block. Supported values
are 128 and 256. A value of 256 uses the experimental FTS V3 format
and may introduce breaking changes.
Notes Notes
----- -----
@@ -183,8 +168,6 @@ class FTS:
ngram_min_length: int = 3 ngram_min_length: int = 3
ngram_max_length: int = 3 ngram_max_length: int = 3
prefix_only: bool = False prefix_only: bool = False
block_size: int = 128
custom_stop_words: Optional[List[str]] = None
@dataclass @dataclass
@@ -219,7 +202,7 @@ class HnswPq:
distance has a range of (-, ). If the vectors are normalized (i.e. their distance has a range of (-, ). If the vectors are normalized (i.e. their
l2 norm is 1), then dot distance is equivalent to the cosine distance. l2 norm is 1), then dot distance is equivalent to the cosine distance.
num_partitions: int, default sqrt(num_rows) num_partitions, default sqrt(num_rows)
The number of IVF partitions to create. The number of IVF partitions to create.
@@ -228,7 +211,7 @@ class HnswPq:
will require too much memory. Each partition becomes its own HNSW graph, so will require too much memory. Each partition becomes its own HNSW graph, so
setting this value higher reduces the peak memory use of training. setting this value higher reduces the peak memory use of training.
num_sub_vectors: int, default is vector dimension / 16 num_sub_vectors, default is vector dimension / 16
Number of sub-vectors of PQ. Number of sub-vectors of PQ.
@@ -244,13 +227,13 @@ class HnswPq:
If the dimension is not visible by 8 then we use 1 subvector. This is not If the dimension is not visible by 8 then we use 1 subvector. This is not
ideal and will likely result in poor performance. ideal and will likely result in poor performance.
num_bits: int, default 8 num_bits: int, default 8
Number of bits to encode each sub-vector. Number of bits to encode each sub-vector.
This value controls how much the sub-vectors are compressed. The more bits This value controls how much the sub-vectors are compressed. The more bits
the more accurate the index but the slower search. Only 4 and 8 are supported. the more accurate the index but the slower search. Only 4 and 8 are supported.
max_iterations: int, default 50 max_iterations, default 50
Max iterations to train kmeans. Max iterations to train kmeans.
@@ -263,7 +246,7 @@ class HnswPq:
those cases it is unlikely that setting this larger will lead to the index those cases it is unlikely that setting this larger will lead to the index
converging anyways. converging anyways.
sample_rate: int, default 256 sample_rate, default 256
The rate used to calculate the number of training vectors for kmeans. The rate used to calculate the number of training vectors for kmeans.
@@ -279,14 +262,14 @@ class HnswPq:
Increasing this value might improve the quality of the index but in Increasing this value might improve the quality of the index but in
most cases the default should be sufficient. most cases the default should be sufficient.
m: int, default 20 m, default 20
The number of neighbors to select for each vector in the HNSW graph. The number of neighbors to select for each vector in the HNSW graph.
This value controls the tradeoff between search speed and accuracy. This value controls the tradeoff between search speed and accuracy.
The higher the value the more accurate the search but the slower it will be. The higher the value the more accurate the search but the slower it will be.
ef_construction: int, default 300 ef_construction, default 300
The number of candidates to evaluate during the construction of the HNSW graph. The number of candidates to evaluate during the construction of the HNSW graph.
@@ -297,7 +280,7 @@ class HnswPq:
This value should be set to a value that is not less than `ef` in the This value should be set to a value that is not less than `ef` in the
search phase. search phase.
target_partition_size: int, default is 1,048,576 target_partition_size, default is 1,048,576
The target size of each partition. The target size of each partition.
@@ -351,7 +334,7 @@ class HnswSq:
distance has a range of (-, ). If the vectors are normalized (i.e. their distance has a range of (-, ). If the vectors are normalized (i.e. their
l2 norm is 1), then dot distance is equivalent to the cosine distance. l2 norm is 1), then dot distance is equivalent to the cosine distance.
num_partitions: int, default sqrt(num_rows) num_partitions, default sqrt(num_rows)
The number of IVF partitions to create. The number of IVF partitions to create.
@@ -360,7 +343,7 @@ class HnswSq:
will require too much memory. Each partition becomes its own HNSW graph, so will require too much memory. Each partition becomes its own HNSW graph, so
setting this value higher reduces the peak memory use of training. setting this value higher reduces the peak memory use of training.
max_iterations: int, default 50 max_iterations, default 50
Max iterations to train kmeans. Max iterations to train kmeans.
@@ -373,7 +356,7 @@ class HnswSq:
In those cases it is unlikely that setting this larger will lead to In those cases it is unlikely that setting this larger will lead to
the index converging anyways. the index converging anyways.
sample_rate: int, default 256 sample_rate, default 256
The rate used to calculate the number of training vectors for kmeans. The rate used to calculate the number of training vectors for kmeans.
@@ -389,14 +372,14 @@ class HnswSq:
Increasing this value might improve the quality of the index but in Increasing this value might improve the quality of the index but in
most cases the default should be sufficient. most cases the default should be sufficient.
m: int, default 20 m, default 20
The number of neighbors to select for each vector in the HNSW graph. The number of neighbors to select for each vector in the HNSW graph.
This value controls the tradeoff between search speed and accuracy. This value controls the tradeoff between search speed and accuracy.
The higher the value the more accurate the search but the slower it will be. The higher the value the more accurate the search but the slower it will be.
ef_construction: int, default 300 ef_construction, default 300
The number of candidates to evaluate during the construction of the HNSW graph. The number of candidates to evaluate during the construction of the HNSW graph.
@@ -407,7 +390,7 @@ class HnswSq:
This value should be set to a value that is not less than `ef` in the search This value should be set to a value that is not less than `ef` in the search
phase. phase.
target_partition_size: int, default is 1,048,576 target_partition_size, default is 1,048,576
The target size of each partition. The target size of each partition.
@@ -460,7 +443,7 @@ class HnswFlat:
distance has a range of (-, ). If the vectors are normalized (i.e. their distance has a range of (-, ). If the vectors are normalized (i.e. their
l2 norm is 1), then dot distance is equivalent to the cosine distance. l2 norm is 1), then dot distance is equivalent to the cosine distance.
num_partitions: int, default sqrt(num_rows) num_partitions, default sqrt(num_rows)
The number of IVF partitions to create. The number of IVF partitions to create.
@@ -470,18 +453,18 @@ class HnswFlat:
graph, so setting this value higher reduces the peak memory use of graph, so setting this value higher reduces the peak memory use of
training. training.
max_iterations: int, default 50 max_iterations, default 50
Max iterations to train kmeans. Max iterations to train kmeans.
When training an IVF index we use kmeans to calculate the partitions. When training an IVF index we use kmeans to calculate the partitions.
This parameter controls how many iterations of kmeans to run. This parameter controls how many iterations of kmeans to run.
sample_rate: int, default 256 sample_rate, default 256
The rate used to calculate the number of training vectors for kmeans. The rate used to calculate the number of training vectors for kmeans.
m: int, default 20 m, default 20
The number of neighbors to select for each vector in the HNSW graph. The number of neighbors to select for each vector in the HNSW graph.
@@ -489,7 +472,7 @@ class HnswFlat:
The higher the value the more accurate the search but the slower it The higher the value the more accurate the search but the slower it
will be. will be.
ef_construction: int, default 300 ef_construction, default 300
The number of candidates to evaluate during the construction of the HNSW The number of candidates to evaluate during the construction of the HNSW
graph. graph.
@@ -501,7 +484,7 @@ class HnswFlat:
than 500. This value should be set to a value that is not less than `ef` than 500. This value should be set to a value that is not less than `ef`
in the search phase. in the search phase.
target_partition_size: int, default is 1,048,576 target_partition_size, default is 1,048,576
The target size of each partition. The target size of each partition.
""" """
@@ -605,7 +588,7 @@ class IvfFlat:
The default value is 256. The default value is 256.
target_partition_size: int, default is 8192 target_partition_size, default is 8192
The target size of each partition. The target size of each partition.
@@ -769,7 +752,7 @@ class IvfPq:
The default value is 256. The default value is 256.
target_partition_size: int, default is 8192 target_partition_size, default is 8192
The target size of each partition. The target size of each partition.
@@ -830,7 +813,7 @@ class IvfRq:
sample_rate: int, default 256 sample_rate: int, default 256
Controls the number of training vectors: sample_rate * num_partitions. Controls the number of training vectors: sample_rate * num_partitions.
target_partition_size: int, default is 8192 target_partition_size, default is 8192
Target size of each partition. Target size of each partition.
""" """
@@ -845,9 +828,6 @@ class IvfRq:
accelerator: Optional[str] = None accelerator: Optional[str] = None
# The API reference renders this module with a single mkdocstrings directive,
# which only picks up names listed here. New public names must be added to this
# list, or they will silently go undocumented.
__all__ = [ __all__ = [
"BTree", "BTree",
"IvfPq", "IvfPq",
-105
View File
@@ -1,105 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
"""Handles to operations a server may run asynchronously."""
import asyncio
from datetime import timedelta
from typing import Optional
from lancedb.background_loop import LOOP
from . import _lancedb
class AsyncJob:
"""A handle to an operation that may still be running.
The operation may already be complete when the handle is created.
"""
def __init__(self, inner: Optional["_lancedb.Job"]):
self._inner = inner
@property
def id(self) -> Optional[str]:
"""Identifies the operation on the server that is running it.
Returned for correlating with server logs or the jobs API. Operations
that run in this process have no server id and return `None`. The value
is opaque: parsing it or storing it to resume the job later is not
supported.
"""
return self._inner.id if self._inner is not None else None
async def status(self) -> str:
"""The operation's current lifecycle state: "running", "finished",
"failed", or "cancelled".
A point snapshot; unlike `wait` it does not block or raise on a
terminal failure state. States a newer server reports that this
client version does not know pass through as-is.
"""
if self._inner is None:
return "finished"
return await self._inner.status()
async def wait(self, timeout: Optional[timedelta] = None):
"""Wait until the operation reaches a terminal state.
Raises `JobFailedError` if the operation failed, `JobCancelledError`
if it was cancelled, and `TimeoutError` if `timeout` elapses first.
"""
if self._inner is None:
return
if timeout is None:
await self._inner.wait()
else:
await asyncio.wait_for(self._inner.wait(), timeout.total_seconds())
async def cancel(self):
"""Request cancellation. Cancelling a finished operation is a no-op."""
if self._inner is None:
return
await self._inner.cancel()
class Job:
"""Synchronous counterpart of `AsyncJob`."""
def __init__(self, inner: Optional[AsyncJob]):
self._inner = inner
@property
def id(self) -> Optional[str]:
"""Identifies the operation on the server that is running it.
See :attr:`AsyncJob.id`.
"""
return self._inner.id if self._inner is not None else None
def status(self) -> str:
"""The operation's current lifecycle state: "running", "finished",
"failed", or "cancelled".
See :meth:`AsyncJob.status`.
"""
if self._inner is None:
return "finished"
return LOOP.run(self._inner.status())
def wait(self, timeout: Optional[timedelta] = None):
"""Block until the operation reaches a terminal state.
Raises `JobFailedError` if the operation failed, `JobCancelledError`
if it was cancelled, and `TimeoutError` if `timeout` elapses first.
"""
if self._inner is None:
return
LOOP.run(self._inner.wait(timeout))
def cancel(self):
"""Request cancellation. Cancelling a finished operation is a no-op."""
if self._inner is None:
return
LOOP.run(self._inner.cancel())
+12 -14
View File
@@ -37,7 +37,7 @@ class LanceMergeInsertBuilder(object):
self._when_not_matched_by_source_condition_expr = None self._when_not_matched_by_source_condition_expr = None
self._timeout = None self._timeout = None
self._use_index = True self._use_index = True
self._use_lsm = None self._use_lsm_write = None
self._validate_single_shard = None self._validate_single_shard = None
def when_matched_update_all( def when_matched_update_all(
@@ -92,10 +92,8 @@ class LanceMergeInsertBuilder(object):
self._when_not_matched_by_source_delete = True self._when_not_matched_by_source_delete = True
if isinstance(condition, Expr): if isinstance(condition, Expr):
self._when_not_matched_by_source_condition_expr = condition._inner self._when_not_matched_by_source_condition_expr = condition._inner
self._when_not_matched_by_source_condition = None elif condition is not None:
else:
self._when_not_matched_by_source_condition = condition self._when_not_matched_by_source_condition = condition
self._when_not_matched_by_source_condition_expr = None
return self return self
def use_index(self, use_index: bool) -> LanceMergeInsertBuilder: def use_index(self, use_index: bool) -> LanceMergeInsertBuilder:
@@ -115,22 +113,22 @@ class LanceMergeInsertBuilder(object):
self._use_index = use_index self._use_index = use_index
return self return self
def use_lsm(self, enable: bool) -> LanceMergeInsertBuilder: def use_lsm_write(self, use_lsm_write: bool) -> LanceMergeInsertBuilder:
""" """
Control MemWAL routing for this merge. Controls whether the merge uses the MemWAL LSM write path.
By default (unset), a `merge_insert` on a table with an LSM write spec is By default (unset), a `merge_insert` on a table with an LSM write spec
routed through Lance's MemWAL shard writer, and a table without one uses is routed through Lance's MemWAL shard writer, and a table without one
the standard path. uses the standard path. Pass `False` to force the standard path even
when a spec is set. Pass `True` to require a spec `merge_insert`
raises an error if none is installed.
Parameters Parameters
---------- ----------
enable: bool use_lsm_write: bool
``True`` forces MemWAL routing and errors if the table has no LSM Whether to use the LSM write path.
write spec. ``False`` forces the standard write path even when a spec
is set.
""" """
self._use_lsm = enable self._use_lsm_write = use_lsm_write
return self return self
def validate_single_shard( def validate_single_shard(
+1 -116
View File
@@ -38,18 +38,13 @@ from lance_namespace_urllib3_client.models.query_table_request_vector import (
QueryTableRequestVector, QueryTableRequestVector,
) )
from lance_namespace_urllib3_client.models.string_fts_query import StringFtsQuery from lance_namespace_urllib3_client.models.string_fts_query import StringFtsQuery
from lance_namespace.errors import ( from lance_namespace.errors import NamespaceNotEmptyError, TableNotFoundError
NamespaceNotEmptyError,
NamespaceNotFoundError,
TableNotFoundError,
)
from lancedb._lancedb import ( from lancedb._lancedb import (
connect_namespace as _connect_namespace, connect_namespace as _connect_namespace,
connect_namespace_client as _connect_namespace_client, connect_namespace_client as _connect_namespace_client,
) )
from lancedb.background_loop import LOOP from lancedb.background_loop import LOOP
from lancedb.db import AsyncConnection, DBConnection from lancedb.db import AsyncConnection, DBConnection
from lancedb.job import AsyncJob, Job
from lance_namespace import ( from lance_namespace import (
LanceNamespace, LanceNamespace,
connect as namespace_connect, connect as namespace_connect,
@@ -58,8 +53,6 @@ from lance_namespace import (
DropNamespaceResponse, DropNamespaceResponse,
ListNamespacesResponse, ListNamespacesResponse,
ListTablesResponse, ListTablesResponse,
NamespaceExistsRequest,
TableExistsRequest,
) )
from lancedb.table import AsyncTable, LanceTable, Table from lancedb.table import AsyncTable, LanceTable, Table
from lancedb.util import validate_table_name from lancedb.util import validate_table_name
@@ -625,18 +618,6 @@ class LanceNamespaceDBConnection(DBConnection):
namespace_path = [] namespace_path = []
LOOP.run(self._inner.drop_table(name, namespace_path=namespace_path)) LOOP.run(self._inner.drop_table(name, namespace_path=namespace_path))
@override
def drop_table_async(
self, name: str, namespace_path: Optional[List[str]] = None
) -> Job:
"""Start dropping a table and return its cleanup job."""
if namespace_path is None:
namespace_path = []
job = LOOP.run(
self._inner.drop_table_async(name, namespace_path=namespace_path)
)
return Job(job if isinstance(job, AsyncJob) else AsyncJob(job))
@override @override
def rename_table( def rename_table(
self, self,
@@ -799,51 +780,6 @@ class LanceNamespaceDBConnection(DBConnection):
""" """
return LOOP.run(self._inner.describe_namespace(namespace_path)) return LOOP.run(self._inner.describe_namespace(namespace_path))
@override
def namespace_exists(self, namespace_id: List[str]) -> bool:
"""
Check if a namespace exists.
Parameters
----------
namespace_id : List[str]
The namespace identifier to check.
Returns
-------
bool
True if the namespace exists, False otherwise.
"""
request = NamespaceExistsRequest(id=namespace_id)
try:
self._namespace_client.namespace_exists(request)
return True
except NamespaceNotFoundError:
return False
@override
def table_exists(self, table_id: List[str]) -> bool:
"""
Check if a table exists.
Parameters
----------
table_id : List[str]
The table identifier to check (full path including namespace
segments and table name).
Returns
-------
bool
True if the table exists, False otherwise.
"""
request = TableExistsRequest(id=table_id)
try:
self._namespace_client.table_exists(request)
return True
except TableNotFoundError:
return False
@override @override
def list_tables( def list_tables(
self, self,
@@ -1147,14 +1083,6 @@ class AsyncLanceNamespaceDBConnection:
namespace_path = [] namespace_path = []
await self._inner.drop_table(name, namespace_path=namespace_path) await self._inner.drop_table(name, namespace_path=namespace_path)
async def drop_table_async(
self, name: str, namespace_path: Optional[List[str]] = None
) -> AsyncJob:
"""Start dropping a table and return its cleanup job."""
if namespace_path is None:
namespace_path = []
return await self._inner.drop_table_async(name, namespace_path=namespace_path)
async def rename_table( async def rename_table(
self, self,
cur_name: str, cur_name: str,
@@ -1305,49 +1233,6 @@ class AsyncLanceNamespaceDBConnection:
""" """
return await self._inner.describe_namespace(namespace_path) return await self._inner.describe_namespace(namespace_path)
async def namespace_exists(self, namespace_id: List[str]) -> bool:
"""
Check if a namespace exists.
Parameters
----------
namespace_id : List[str]
The namespace identifier to check.
Returns
-------
bool
True if the namespace exists, False otherwise.
"""
request = NamespaceExistsRequest(id=namespace_id)
try:
self._namespace_client.namespace_exists(request)
return True
except NamespaceNotFoundError:
return False
async def table_exists(self, table_id: List[str]) -> bool:
"""
Check if a table exists.
Parameters
----------
table_id : List[str]
The table identifier to check (full path including namespace
segments and table name).
Returns
-------
bool
True if the table exists, False otherwise.
"""
request = TableExistsRequest(id=table_id)
try:
self._namespace_client.table_exists(request)
return True
except TableNotFoundError:
return False
async def list_tables( async def list_tables(
self, self,
namespace_path: Optional[List[str]] = None, namespace_path: Optional[List[str]] = None,
+7 -12
View File
@@ -226,7 +226,7 @@ class PermutationBuilder:
async def do_execute(): async def do_execute():
inner_tbl = await self._async.execute() inner_tbl = await self._async.execute()
return await LanceTable.from_inner(inner_tbl) return LanceTable.from_inner(inner_tbl)
return LOOP.run(do_execute()) return LOOP.run(do_execute())
@@ -438,8 +438,7 @@ class Permutation:
_reader: Optional[PermutationReader] = None, _reader: Optional[PermutationReader] = None,
): ):
""" """
Internal constructor. Use Internal constructor. Use [from_tables](#from_tables) instead.
[from_tables][lancedb.permutation.Permutation.from_tables] instead.
""" """
assert base_table is not None, "base_table is required" assert base_table is not None, "base_table is required"
assert selection is not None, "selection is required" assert selection is not None, "selection is required"
@@ -986,9 +985,8 @@ class Permutation:
types. Conversion of strings, lists, and structs will require creating python types. Conversion of strings, lists, and structs will require creating python
objects and this is not zero-copy. objects and this is not zero-copy.
For custom formatting, use For custom formatting, use [with_transform](#with_transform) which overrides
[with_transform][lancedb.permutation.Permutation.with_transform] which this method.
overrides this method.
""" """
assert format is not None, "format is required" assert format is not None, "format is required"
if format == "python": if format == "python":
@@ -1063,8 +1061,7 @@ class Permutation:
Note: this method returns a new permutation and does not modify `self` Note: this method returns a new permutation and does not modify `self`
It is provided for compatibility with the huggingface Dataset API. It is provided for compatibility with the huggingface Dataset API.
Use [with_skip][lancedb.permutation.Permutation.with_skip] instead to Use [with_skip](#with_skip) instead to avoid confusion.
avoid confusion.
""" """
return self.with_skip(skip) return self.with_skip(skip)
@@ -1087,8 +1084,7 @@ class Permutation:
Note: this method returns a new permutation and does not modify `self` Note: this method returns a new permutation and does not modify `self`
It is provided for compatibility with the huggingface Dataset API. It is provided for compatibility with the huggingface Dataset API.
Use [with_take][lancedb.permutation.Permutation.with_take] instead to Use [with_take](#with_take) instead to avoid confusion.
avoid confusion.
""" """
return self.with_take(limit) return self.with_take(limit)
@@ -1111,8 +1107,7 @@ class Permutation:
Note: this method returns a new permutation and does not modify `self` Note: this method returns a new permutation and does not modify `self`
It is provided for compatibility with the huggingface Dataset API. It is provided for compatibility with the huggingface Dataset API.
Use [with_repeat][lancedb.permutation.Permutation.with_repeat] instead Use [with_repeat](#with_repeat) instead to avoid confusion.
to avoid confusion.
""" """
return self.with_repeat(times) return self.with_repeat(times)
-1
View File
@@ -1 +0,0 @@
-10
View File
@@ -153,16 +153,6 @@ def Vector(
return FixedSizeList return FixedSizeList
def _raise_bare_vector_error(*_args):
raise TypeError("Vector must be parameterized with a dimension, e.g. Vector(128).")
# Pydantic v1 and v2 otherwise treat the bare Vector factory as a field validator
# and inspect its signature, which produces misleading errors about internal types.
setattr(Vector, "__get_validators__", _raise_bare_vector_error)
setattr(Vector, "__get_pydantic_core_schema__", _raise_bare_vector_error)
def MultiVector( def MultiVector(
dim: int, value_type: pa.DataType = pa.float32(), nullable: bool = True dim: int, value_type: pa.DataType = pa.float32(), nullable: bool = True
) -> Type: ) -> Type:
+24 -101
View File
@@ -52,6 +52,7 @@ from ._blob import (
finalize_blob_query_table, finalize_blob_query_table,
replace_v2_blob_columns_with_bytes, replace_v2_blob_columns_with_bytes,
replace_v2_blob_columns_with_bytes_sync, replace_v2_blob_columns_with_bytes_sync,
supports_blob_auto_row_id,
validate_blob_mode, validate_blob_mode,
) )
from .types import BlobMode, QueryProjection from .types import BlobMode, QueryProjection
@@ -650,8 +651,7 @@ class Query(pydantic.BaseModel):
distance_type : Optional[str] distance_type : Optional[str]
the distance type to use for vector search the distance type to use for vector search
This can be l2 (default), cosine and dot. See This can be l2 (default), cosine and dot. See [metric definitions][search] for
[metric definitions](https://lancedb.com/docs/search/vector-search/) for
more details. more details.
If this is not a vector search this will be None. If this is not a vector search this will be None.
@@ -664,9 +664,8 @@ class Query(pydantic.BaseModel):
- A higher number makes search more accurate but also slower. - A higher number makes search more accurate but also slower.
- See discussion in - See discussion in [Querying an ANN Index][querying-an-ann-index] for
[Querying an ANN Index](https://lancedb.com/docs/indexing/) tuning advice.
for tuning advice.
Will be None if this is not a vector search. Will be None if this is not a vector search.
refine_factor : Optional[int] refine_factor : Optional[int]
@@ -674,9 +673,8 @@ class Query(pydantic.BaseModel):
- A higher number makes search more accurate but also slower. - A higher number makes search more accurate but also slower.
- See discussion in - See discussion in [Querying an ANN Index][querying-an-ann-index] for
[Querying an ANN Index](https://lancedb.com/docs/indexing/) tuning advice.
for tuning advice.
Will be None if this is not a vector search. Will be None if this is not a vector search.
lower_bound : Optional[float] lower_bound : Optional[float]
@@ -780,11 +778,6 @@ class Query(pydantic.BaseModel):
# if true, will only search the indexed data # if true, will only search the indexed data
fast_search: Optional[bool] = None fast_search: Optional[bool] = None
# MemWAL LSM read routing: None auto-routes when the table carries a write
# spec, True forces the LSM scanner (errors without a spec), False reads the
# base table only
use_lsm: Optional[bool] = None
# size of the nearest neighbor list maintained during HNSW search # size of the nearest neighbor list maintained during HNSW search
ef: Optional[int] = None ef: Optional[int] = None
@@ -802,9 +795,6 @@ class Query(pydantic.BaseModel):
query.full_text_query = req.full_text_search query.full_text_query = req.full_text_search
query.columns = req.select query.columns = req.select
query.with_row_id = req.with_row_id query.with_row_id = req.with_row_id
# use_lsm is a genuine tri-state (None / True / False); preserve it as-is
# so a round-tripped query keeps an explicit False.
query.use_lsm = req.use_lsm
query.vector_column = req.column query.vector_column = req.column
query.vector = req.query_vector query.vector = req.query_vector
query.distance_type = req.distance_type query.distance_type = req.distance_type
@@ -977,7 +967,6 @@ class LanceQueryBuilder(ABC):
self._with_row_address = None self._with_row_address = None
self._fragments = None self._fragments = None
self._fragment_ids = None self._fragment_ids = None
self._use_lsm = None
self._vector = None self._vector = None
self._text = None self._text = None
self._ef = None self._ef = None
@@ -1279,7 +1268,10 @@ class LanceQueryBuilder(ABC):
return self._with_row_id is True return self._with_row_id is True
def _blob_auto_row_id_enabled(self) -> bool: def _blob_auto_row_id_enabled(self) -> bool:
if not supports_blob_auto_row_id(self._table):
return False
return blob_auto_row_id_for_scan( return blob_auto_row_id_for_scan(
self._table,
self._table.schema, self._table.schema,
self._columns, self._columns,
with_row_id=self._with_row_id, with_row_id=self._with_row_id,
@@ -1334,30 +1326,6 @@ class LanceQueryBuilder(ABC):
self._fragment_ids = fragment_ids self._fragment_ids = fragment_ids
return self return self
def use_lsm(self, enable: bool) -> Self:
"""Control MemWAL LSM read routing for this query.
By default (unset), a query against a table with an LSM write spec is
routed through the LSM scanner so it also returns data written via the
``merge_insert`` LSM path that has not yet been compacted into the base
table (active/frozen memtables + flushed generations); a table without a
spec reads the base table.
Parameters
----------
enable : bool
``True`` forces the LSM scanner and errors if the table has no LSM
write spec. ``False`` bypasses the MemWAL and reads the base table
only, even when a spec is present.
Returns
-------
LanceQueryBuilder
The LanceQueryBuilder object.
"""
self._use_lsm = enable
return self
def explain_plan(self, verbose: Optional[bool] = False) -> str: def explain_plan(self, verbose: Optional[bool] = False) -> str:
"""Return the execution plan for this query. """Return the execution plan for this query.
@@ -1650,8 +1618,8 @@ class LanceVectorQueryBuilder(LanceQueryBuilder):
Higher values will yield better recall (more likely to find vectors if Higher values will yield better recall (more likely to find vectors if
they exist) at the expense of latency. they exist) at the expense of latency.
See discussion in [Querying an ANN Index](https://lancedb.com/docs/indexing/) See discussion in [Querying an ANN Index][querying-an-ann-index] for
for tuning advice. tuning advice.
This method sets both the minimum and maximum number of probes to the same This method sets both the minimum and maximum number of probes to the same
value. See `minimum_nprobes` and `maximum_nprobes` for more fine-grained value. See `minimum_nprobes` and `maximum_nprobes` for more fine-grained
@@ -1751,8 +1719,8 @@ class LanceVectorQueryBuilder(LanceQueryBuilder):
As an example, a refine factor of 2 will sample 2x as many vectors as As an example, a refine factor of 2 will sample 2x as many vectors as
requested, re-ranks them, and returns the top half most relevant results. requested, re-ranks them, and returns the top half most relevant results.
See discussion in [Querying an ANN Index](https://lancedb.com/docs/indexing/) See discussion in [Querying an ANN Index][querying-an-ann-index] for
for tuning advice. tuning advice.
Parameters Parameters
---------- ----------
@@ -1820,7 +1788,6 @@ class LanceVectorQueryBuilder(LanceQueryBuilder):
with_row_address=self._with_row_address, with_row_address=self._with_row_address,
fragments=self._fragments, fragments=self._fragments,
fragment_ids=self._fragment_ids, fragment_ids=self._fragment_ids,
use_lsm=self._use_lsm,
offset=self._offset, offset=self._offset,
fast_search=self._fast_search, fast_search=self._fast_search,
ef=self._ef, ef=self._ef,
@@ -2045,7 +2012,6 @@ class LanceFtsQueryBuilder(LanceQueryBuilder):
with_row_address=self._with_row_address, with_row_address=self._with_row_address,
fragments=self._fragments, fragments=self._fragments,
fragment_ids=self._fragment_ids, fragment_ids=self._fragment_ids,
use_lsm=self._use_lsm,
full_text_query=FullTextSearchQuery( full_text_query=FullTextSearchQuery(
query=self._query_with_phrase_semantics(), columns=self._fts_columns query=self._query_with_phrase_semantics(), columns=self._fts_columns
), ),
@@ -2112,7 +2078,6 @@ class LanceEmptyQueryBuilder(LanceQueryBuilder):
with_row_address=self._with_row_address, with_row_address=self._with_row_address,
fragments=self._fragments, fragments=self._fragments,
fragment_ids=self._fragment_ids, fragment_ids=self._fragment_ids,
use_lsm=self._use_lsm,
offset=self._offset, offset=self._offset,
order_by=self._order_by, order_by=self._order_by,
) )
@@ -2690,14 +2655,11 @@ class LanceHybridQueryBuilder(LanceQueryBuilder):
if self._with_row_id: if self._with_row_id:
self._vector_query.with_row_id(True) self._vector_query.with_row_id(True)
self._fts_query.with_row_id(True) self._fts_query.with_row_id(True)
if self._use_lsm is not None:
self._vector_query.use_lsm(self._use_lsm)
self._fts_query.use_lsm(self._use_lsm)
if self._phrase_query: if self._phrase_query:
self._fts_query.phrase_query(True) self._fts_query.phrase_query(True)
if self._distance_type: if self._distance_type:
self._vector_query.metric(self._distance_type) self._vector_query.metric(self._distance_type)
if self._minimum_nprobes is not None: if self._minimum_nprobes:
self._vector_query.minimum_nprobes(self._minimum_nprobes) self._vector_query.minimum_nprobes(self._minimum_nprobes)
if self._maximum_nprobes is not None: if self._maximum_nprobes is not None:
self._vector_query.maximum_nprobes(self._maximum_nprobes) self._vector_query.maximum_nprobes(self._maximum_nprobes)
@@ -2770,7 +2732,7 @@ class AsyncQueryBase(object):
) )
async def _maybe_add_blob_row_id(self) -> None: async def _maybe_add_blob_row_id(self) -> None:
if self._table is None: if self._table is None or not supports_blob_auto_row_id(self._table):
self._blob_auto_row_id = False self._blob_auto_row_id = False
self._blob_paths = () self._blob_paths = ()
return return
@@ -2778,6 +2740,7 @@ class AsyncQueryBase(object):
req = self._inner.to_query_request() req = self._inner.to_query_request()
schema = await self._table.schema() schema = await self._table.schema()
self._blob_auto_row_id = blob_auto_row_id_for_scan( self._blob_auto_row_id = blob_auto_row_id_for_scan(
self._table,
schema, schema,
req.select, req.select,
with_row_id=self._with_row_id, with_row_id=self._with_row_id,
@@ -3029,6 +2992,7 @@ class AsyncQueryBase(object):
schema = await self._table.schema() schema = await self._table.schema()
blob_auto_row_id = blob_auto_row_id_for_scan( blob_auto_row_id = blob_auto_row_id_for_scan(
self._table,
schema, schema,
query.columns, query.columns,
with_row_id=self._with_row_id, with_row_id=self._with_row_id,
@@ -3038,7 +3002,7 @@ class AsyncQueryBase(object):
if blob_mode == "bytes" if blob_mode == "bytes"
else {} else {}
) )
dataset = await self._table.to_lance() dataset = await self._table._to_lance()
scanner = dataset.scanner( scanner = dataset.scanner(
**_scanner_kwargs_for_query( **_scanner_kwargs_for_query(
query, query,
@@ -3267,27 +3231,6 @@ class AsyncStandardQuery(AsyncQueryBase):
self._inner.fast_search() self._inner.fast_search()
return self return self
def use_lsm(self, enable: bool) -> Self:
"""
Control MemWAL LSM read routing for this query.
By default (unset), a query against a table with an LSM write spec (see
[AsyncTable.set_lsm_write_spec][lancedb.table.AsyncTable.set_lsm_write_spec])
is routed through the LSM scanner so it also returns data written via the
``merge_insert`` LSM path that has not yet been compacted into the base
table (the active/frozen in-memory memtables and the flushed generations),
deduplicated by primary key; a table without a spec reads the base table.
Parameters
----------
enable : bool
``True`` forces the LSM scanner and errors if the table has no LSM
write spec. ``False`` bypasses the MemWAL and reads the base table
only, even when a spec is present.
"""
self._inner.use_lsm(enable)
return self
def postfilter(self) -> Self: def postfilter(self) -> Self:
""" """
If this is called then filtering will happen after the search instead of If this is called then filtering will happen after the search instead of
@@ -3376,9 +3319,8 @@ class AsyncQuery(AsyncStandardQuery):
are various ANN search parameters that will let you fine tune your recall are various ANN search parameters that will let you fine tune your recall
accuracy vs search latency. accuracy vs search latency.
Vector searches always have a Vector searches always have a [limit][]. If `limit` has not been called then
[limit][lancedb.query.AsyncVectorQuery.limit]. If `limit` has not been a default `limit` of 10 will be used.
called then a default `limit` of 10 will be used.
Typically, a single vector is passed in as the query. However, you can also Typically, a single vector is passed in as the query. However, you can also
pass in multiple vectors. When multiple vectors are passed in, if the vector pass in multiple vectors. When multiple vectors are passed in, if the vector
@@ -3509,9 +3451,8 @@ class AsyncFTSQuery(AsyncStandardQuery):
are various ANN search parameters that will let you fine tune your recall are various ANN search parameters that will let you fine tune your recall
accuracy vs search latency. accuracy vs search latency.
Hybrid searches always have a Hybrid searches always have a [limit][]. If `limit` has not been called then
[limit][lancedb.query.AsyncHybridQuery.limit]. If `limit` has not been a default `limit` of 10 will be used.
called then a default `limit` of 10 will be used.
Typically, a single vector is passed in as the query. However, you can also Typically, a single vector is passed in as the query. However, you can also
pass in multiple vectors. This can be useful if you want to find the nearest pass in multiple vectors. This can be useful if you want to find the nearest
@@ -3874,9 +3815,10 @@ class AsyncHybridQuery(AsyncStandardQuery, AsyncVectorQueryBase):
req = fts_query._inner.to_query_request() req = fts_query._inner.to_query_request()
blob_auto_row_id = False blob_auto_row_id = False
blob_paths: tuple[str, ...] = () blob_paths: tuple[str, ...] = ()
if self._table is not None: if self._table is not None and supports_blob_auto_row_id(self._table):
schema = await self._table.schema() schema = await self._table.schema()
blob_auto_row_id = blob_auto_row_id_for_scan( blob_auto_row_id = blob_auto_row_id_for_scan(
self._table,
schema, schema,
req.select, req.select,
with_row_id=self._with_row_id, with_row_id=self._with_row_id,
@@ -4002,15 +3944,6 @@ class AsyncTakeQuery(AsyncQueryBase):
def __init__(self, inner: LanceTakeQuery, table: Optional["AsyncTable"] = None): def __init__(self, inner: LanceTakeQuery, table: Optional["AsyncTable"] = None):
super().__init__(inner, table) super().__init__(inner, table)
def use_lsm(self, enable: bool) -> "AsyncTakeQuery":
"""Control MemWAL LSM read routing for this take query.
``False`` bypasses the MemWAL and reads the base table only the escape
hatch, since take-by-row-id/offset is not supported on the LSM scanner.
"""
self._inner.use_lsm(enable)
return self
async def _plain_scan_to_pandas( async def _plain_scan_to_pandas(
self, self,
blob_mode: BlobMode, blob_mode: BlobMode,
@@ -4069,16 +4002,6 @@ class BaseQueryBuilder(object):
self._inner.with_row_id() self._inner.with_row_id()
return self return self
def use_lsm(self, enable: bool) -> Self:
"""
Control MemWAL LSM read routing for this query.
``False`` bypasses the MemWAL and reads the base table only, the escape
hatch for shapes the LSM scanner cannot honor (e.g. take-by-row-id).
"""
self._inner.use_lsm(enable)
return self
def with_row_address(self, with_row_address: bool = True) -> Self: def with_row_address(self, with_row_address: bool = True) -> Self:
""" """
Include the _rowaddr column in scanner-backed plain query results. Include the _rowaddr column in scanner-backed plain query results.
-3
View File
@@ -11,9 +11,6 @@ from lancedb import __version__
from .header import HeaderProvider from .header import HeaderProvider
from .oauth import OAuthConfig, OAuthFlowType from .oauth import OAuthConfig, OAuthFlowType
# The API reference renders this module with a single mkdocstrings directive,
# which only picks up names listed here. New public names must be added to this
# list, or they will silently go undocumented.
__all__ = [ __all__ = [
"TimeoutConfig", "TimeoutConfig",
"RetryConfig", "RetryConfig",

Some files were not shown because too many files have changed in this diff Show More