Compare commits

..
Author SHA1 Message Date
Lance Release 8f85833407 Bump version: 0.38.0-beta.4 → 0.38.0-beta.5 2026-08-23 08:07:09 +00:00
198 changed files with 14762 additions and 27081 deletions
+1 -1
View File
@@ -1,5 +1,5 @@
[tool.bumpversion] [tool.bumpversion]
current_version = "0.39.0-beta.4" current_version = "0.38.0-beta.5"
parse = """(?x) parse = """(?x)
(?P<major>0|[1-9]\\d*)\\. (?P<major>0|[1-9]\\d*)\\.
(?P<minor>0|[1-9]\\d*)\\. (?P<minor>0|[1-9]\\d*)\\.
-24
View File
@@ -44,27 +44,3 @@ updates:
python-deps: python-deps:
patterns: patterns:
- "*" - "*"
# The npm ecosystem covers pnpm lockfiles. There are two separate installs:
# the bindings themselves and the examples, which have their own lockfile.
# As with cargo and pip above, only bump the lockfile — the version ranges
# in package.json are our consumers' constraints, not ours.
- package-ecosystem: npm
directory: /nodejs
schedule:
interval: weekly
versioning-strategy: lockfile-only
groups:
nodejs-deps:
patterns:
- "*"
- package-ecosystem: npm
directory: /nodejs/examples
schedule:
interval: weekly
versioning-strategy: lockfile-only
groups:
nodejs-examples-deps:
patterns:
- "*"
+4 -10
View File
@@ -29,14 +29,12 @@ jobs:
steps: steps:
- uses: actions/setup-node@v6 - uses: actions/setup-node@v6
with: with:
node-version: "24" node-version: "18"
- uses: pnpm/action-setup@v6
with:
version: 11.1.1
# These rules are disabled because Github will always ensure there # These rules are disabled because Github will always ensure there
# is a blank line between the title and the body and Github will # is a blank line between the title and the body and Github will
# word wrap the description field to ensure a reasonable max line # word wrap the description field to ensure a reasonable max line
# length. # length.
- run: npm install @commitlint/config-conventional
- run: > - run: >
echo 'module.exports = { echo 'module.exports = {
"rules": { "rules": {
@@ -45,11 +43,7 @@ jobs:
"body-leading-blank": [0, "always"] "body-leading-blank": [0, "always"]
} }
}' > .commitlintrc.js }' > .commitlintrc.js
- run: > - run: npx commitlint --extends @commitlint/config-conventional --verbose <<< $COMMIT_MSG
pnpm dlx
--package @commitlint/cli@21.2.2
--package @commitlint/config-conventional@21.2.2
commitlint --extends @commitlint/config-conventional --verbose <<< $COMMIT_MSG
env: env:
COMMIT_MSG: > COMMIT_MSG: >
${{ github.event.pull_request.title }} ${{ github.event.pull_request.title }}
@@ -60,7 +54,7 @@ jobs:
with: with:
script: | script: |
const message = `**ACTION NEEDED** const message = `**ACTION NEEDED**
Lance follows the [Conventional Commits specification](https://www.conventionalcommits.org/en/v1.0.0/) for release automation. Lance follows the [Conventional Commits specification](https://www.conventionalcommits.org/en/v1.0.0/) for release automation.
The PR title and description are used as the merge commit message.\ The PR title and description are used as the merge commit message.\
+1 -1
View File
@@ -56,7 +56,7 @@ jobs:
uses: lycheeverse/lychee-action@e7477775783ea5526144ba13e8db5eec57747ce8 # v2.9.0 uses: lycheeverse/lychee-action@e7477775783ea5526144ba13e8db5eec57747ce8 # v2.9.0
with: with:
# Restricted to http(s) on purpose. Much of docs/src is generated # Restricted to http(s) on purpose. Much of docs/src is generated
# API reference (the js/ tree comes from `pnpm run docs` in nodejs) # API reference (the js/ tree comes from `npm run docs` in nodejs)
# and the hand-written pages use mkdocstrings cross-references and # and the hand-written pages use mkdocstrings cross-references and
# nav-relative paths that only resolve in the site mkdocs builds, # nav-relative paths that only resolve in the site mkdocs builds,
# not in this checkout, so relative links would be reported as # not in this checkout, so relative links would be reported as
+3 -1
View File
@@ -55,7 +55,9 @@ jobs:
- name: Set up node - name: Set up node
uses: actions/setup-node@v6 uses: actions/setup-node@v6
with: with:
node-version: 24 node-version: 20
cache: 'npm'
cache-dependency-path: docs/package-lock.json
- name: Install node dependencies - name: Install node dependencies
working-directory: nodejs working-directory: nodejs
run: | run: |
+13 -11
View File
@@ -47,8 +47,9 @@ jobs:
version: 11.1.1 version: 11.1.1
- uses: actions/setup-node@v6 - uses: actions/setup-node@v6
with: with:
# Build on a supported LTS; the matrix job below covers every # pnpm 11 requires Node >= 22.13; use 24 since 22 hits EOL
# Node version the library claims to support. # in October. The library itself still supports Node >= 18
# (see test matrix below).
node-version: 24 node-version: 24
cache: 'pnpm' cache: 'pnpm'
cache-dependency-path: nodejs/pnpm-lock.yaml cache-dependency-path: nodejs/pnpm-lock.yaml
@@ -83,7 +84,7 @@ jobs:
timeout-minutes: 30 timeout-minutes: 30
strategy: strategy:
matrix: matrix:
node-version: [ "22", "24", "26" ] node-version: [ "18", "20" ]
runs-on: "ubuntu-22.04" runs-on: "ubuntu-22.04"
defaults: defaults:
run: run:
@@ -100,9 +101,9 @@ jobs:
- uses: actions/setup-node@v6 - uses: actions/setup-node@v6
name: Setup Node.js 24 for build name: Setup Node.js 24 for build
with: with:
# Build and install once on a fixed version so the generated docs # pnpm 11 requires Node >= 22.13; use 24 since 22 hits EOL
# are identical across matrix legs; the tests below then run on each # in October. Build/install runs on Node 24; tests run on the
# supported Node version. # matrix version below using direct jest invocation.
node-version: 24 node-version: 24
cache: 'pnpm' cache: 'pnpm'
cache-dependency-path: nodejs/pnpm-lock.yaml cache-dependency-path: nodejs/pnpm-lock.yaml
@@ -151,9 +152,9 @@ jobs:
S3_TEST: "1" S3_TEST: "1"
# Newer @smithy/core uses dynamic ESM imports. # Newer @smithy/core uses dynamic ESM imports.
NODE_OPTIONS: "--experimental-vm-modules" NODE_OPTIONS: "--experimental-vm-modules"
# Invoke the installed jest binary directly; the pnpm shim is set up # Invoke jest directly because pnpm 11 itself requires Node 22+
# against the build-phase Node, not the version selected above. # while the matrix tests on older Node versions.
run: node_modules/.bin/jest --verbose run: npx jest --verbose
- name: Test examples - name: Test examples
working-directory: ./ working-directory: ./
env: env:
@@ -163,7 +164,7 @@ jobs:
run: | run: |
python ci/mock_openai.py & python ci/mock_openai.py &
cd nodejs/examples cd nodejs/examples
node_modules/.bin/jest --testEnvironment jest-environment-node-single-context --verbose npx jest --testEnvironment jest-environment-node-single-context --verbose
macos: macos:
timeout-minutes: 30 timeout-minutes: 30
# macos-15 ships a newer linker; the older macos-14 linker fails to insert # macos-15 ships a newer linker; the older macos-14 linker fails to insert
@@ -184,7 +185,8 @@ jobs:
version: 11.1.1 version: 11.1.1
- uses: actions/setup-node@v6 - uses: actions/setup-node@v6
with: with:
# pnpm 11 requires Node >= 22.13. # pnpm 11 requires Node >= 22.13; use 24 since 22 hits EOL
# in October.
node-version: 24 node-version: 24
cache: 'pnpm' cache: 'pnpm'
cache-dependency-path: nodejs/pnpm-lock.yaml cache-dependency-path: nodejs/pnpm-lock.yaml
+36 -81
View File
@@ -40,31 +40,40 @@ jobs:
- target: aarch64-apple-darwin - target: aarch64-apple-darwin
host: macos-latest host: macos-latest
features: fp16kernels features: fp16kernels
# Fat LTO was ~111 of this job's ~113 minutes.
lto: thin
codegen_units: 16
pre_build: |- pre_build: |-
brew install protobuf brew install protobuf
# Fat LTO (the workspace default in .cargo/config.toml) is
# single-threaded and is the peak-memory step of the build. On
# this runner it accounted for ~111 of the job's ~113 minutes,
# making it the critical path of the entire publish pipeline.
# ThinLTO parallelizes it across the runner's cores, for a few
# percent of runtime performance.
export CARGO_PROFILE_RELEASE_LTO=thin
export CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16
- target: x86_64-pc-windows-msvc - target: x86_64-pc-windows-msvc
host: windows-2025 host: windows-2025
features: "," features: ","
# The lower peak also keeps this on the standard 4-core runner.
lto: thin
codegen_units: 16
pre_build: |- pre_build: |-
choco install --no-progress protoc ninja nasm choco install --no-progress protoc ninja nasm
tail -n 1000 /c/ProgramData/chocolatey/logs/chocolatey.log tail -n 1000 /c/ProgramData/chocolatey/logs/chocolatey.log
# There is an issue where choco doesn't add nasm to the path # There is an issue where choco doesn't add nasm to the path
export PATH="$PATH:/c/Program Files/NASM" export PATH="$PATH:/c/Program Files/NASM"
nasm -v nasm -v
# See the ThinLTO note on aarch64-apple-darwin above. Keeping
# peak memory down is also what lets this run on the standard
# 4-core runner: the 8-core larger runner was only needed to
# stop fat LTO from OOMing rustc-LLVM.
export CARGO_PROFILE_RELEASE_LTO=thin
export CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16
- target: aarch64-pc-windows-msvc - target: aarch64-pc-windows-msvc
host: windows-2025 host: windows-2025
features: "," features: ","
lto: thin
codegen_units: 16
pre_build: |- pre_build: |-
choco install --no-progress protoc choco install --no-progress protoc
rustup target add aarch64-pc-windows-msvc rustup target add aarch64-pc-windows-msvc
# See the ThinLTO note on aarch64-apple-darwin above.
export CARGO_PROFILE_RELEASE_LTO=thin
export CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16
- target: x86_64-unknown-linux-gnu - target: x86_64-unknown-linux-gnu
host: ubuntu-latest host: ubuntu-latest
features: fp16kernels features: fp16kernels
@@ -94,14 +103,6 @@ jobs:
# https://github.com/napi-rs/napi-rs/blob/main/debian-aarch64.Dockerfile # https://github.com/napi-rs/napi-rs/blob/main/debian-aarch64.Dockerfile
docker: ghcr.io/napi-rs/napi-rs/nodejs-rust:lts-debian-aarch64 docker: ghcr.io/napi-rs/napi-rs/nodejs-rust:lts-debian-aarch64
features: "fp16kernels" features: "fp16kernels"
# Fat LTO OOM-killed rustc every nightly; even with lld it peaked
# at 31391 MiB of the runner's 32 GiB.
lto: thin
codegen_units: 16
# arm64 Linux links through GNU `ld` where x86_64 defaults to
# `rust-lld`, which is why only arm64 OOM'd. lld cut the largest
# linker process 7.0 -> 4.0 GiB (lancedb/sophon#7313).
linker: /tmp/aarch64-lld-clang
pre_build: |- pre_build: |-
set -e && set -e &&
apt-get update && apt-get update &&
@@ -111,30 +112,9 @@ jobs:
# AT_HWCAP2 (added in Linux 3.17). Define it for aws-lc-sys. # AT_HWCAP2 (added in Linux 3.17). Define it for aws-lc-sys.
export CFLAGS="$CFLAGS -DAT_HWCAP2=26" && export CFLAGS="$CFLAGS -DAT_HWCAP2=26" &&
rustup target add aarch64-unknown-linux-gnu rustup target add aarch64-unknown-linux-gnu
# Not `&&`-chained: in dash, errexit does not fire for a
# non-final command in an `&&` list, so failures were ignored.
#
# A wrapper rather than `-C link-arg` because the per-target
# rustflags variable does not reach every unit that links, while
# the linker variable does. `clang` because GCC silently ignores
# `-fuse-ld=lld` unless built with lld support. Two echoes
# because printf's newline escape gets rewritten to `;` between
# here and the container.
echo '#!/bin/sh' > /tmp/aarch64-lld-clang
echo 'exec clang --target=aarch64-unknown-linux-gnu --sysroot=/usr/aarch64-unknown-linux-gnu/aarch64-unknown-linux-gnu/sysroot --gcc-toolchain=/usr/aarch64-unknown-linux-gnu -fuse-ld=lld "$@"' >> /tmp/aarch64-lld-clang
chmod 0755 /tmp/aarch64-lld-clang
# Fail now, not at the cdylib link ~30 minutes later. Linking at
# all also proves lld resolved; clang errors out when it cannot.
echo 'int main(void){return 0;}' > /tmp/probe.c
/tmp/aarch64-lld-clang /tmp/probe.c -o /tmp/probe
readelf -h /tmp/probe | grep AArch64
- target: aarch64-unknown-linux-musl - target: aarch64-unknown-linux-musl
host: ubuntu-2404-8x-x64 host: ubuntu-2404-8x-x64
features: "," features: ","
# Fat LTO took the whole runner down. lld cannot help: it died
# inside rustc's LLVM, before any linker was spawned.
lto: thin
codegen_units: 16
pre_build: |- pre_build: |-
set -e && set -e &&
sudo apt-get update && sudo apt-get update &&
@@ -143,19 +123,6 @@ jobs:
export EXTRA_ARGS="-x" export EXTRA_ARGS="-x"
name: build - ${{ matrix.settings.target }} name: build - ${{ matrix.settings.target }}
runs-on: ${{ matrix.settings.host }} runs-on: ${{ matrix.settings.host }}
# On the job, not exported from `pre_build`: `Swatinem/rust-cache` hashes
# `CARGO_*` into its cache key before any step runs, so a step-local export
# leaves the key unchanged while cargo still rebuilds cold. The ThinLTO
# legs had been doing that every run.
#
# Not `RUSTFLAGS`: setting it, even to "", discards every config-file
# rustflag, silently dropping .cargo/config.toml's `target-cpu` and
# `target-feature` from the published binaries.
env:
CARGO_PROFILE_RELEASE_LTO: ${{ matrix.settings.lto || 'fat' }}
CARGO_PROFILE_RELEASE_CODEGEN_UNITS: ${{ matrix.settings.codegen_units || '1' }}
# Empty elsewhere: a per-target variable is only read for that triple.
CARGO_TARGET_AARCH64_UNKNOWN_LINUX_GNU_LINKER: ${{ matrix.settings.linker }}
defaults: defaults:
run: run:
working-directory: nodejs working-directory: nodejs
@@ -168,7 +135,8 @@ jobs:
- name: Setup node - name: Setup node
uses: actions/setup-node@v6 uses: actions/setup-node@v6
with: with:
# pnpm 11 requires Node >= 22.13. # pnpm 11 requires Node >= 22.13; use 24 since 22 hits EOL
# in October.
node-version: 24 node-version: 24
cache: pnpm cache: pnpm
cache-dependency-path: nodejs/pnpm-lock.yaml cache-dependency-path: nodejs/pnpm-lock.yaml
@@ -201,15 +169,19 @@ jobs:
# creating ref). The nightly cadence also keeps entries inside # creating ref). The nightly cadence also keeps entries inside
# GitHub's 7-day eviction window, which a tag-only trigger would not. # GitHub's 7-day eviction window, which a tag-only trigger would not.
save-if: ${{ github.ref == 'refs/heads/main' }} save-if: ${{ github.ref == 'refs/heads/main' }}
# Docker builds can use rust-cache too: the workspace is bind-mounted, so # Docker builds can use rust-cache too. `target/` already lives on the
# `target/` lives on the host and rust-cache's prune keeps the entry # host because the whole workspace is bind-mounted into the container, and
# small. # rust-cache's prune and save run host-side, so they can manage it -- which
# is what keeps the entry to dependency artifacts rather than a multi-GB
# copy of everything.
# #
# Two differences from the native builds. The container's CARGO_HOME is # Two differences from the native builds. The container's CARGO_HOME is
# bind-mounted from `.cargo-cache` rather than ~/.cargo, so that is cached # bind-mounted from `.cargo-cache` rather than the host's ~/.cargo, so that
# explicitly. And the key uses the *host* rustc version, not the compiler # has to be cached explicitly. And the key is derived from the *host* rustc
# that built these artifacts -- safe, since cargo fingerprints the real # version, which is not the compiler that produced these artifacts; that is
# one; a base-image bump just costs one cold build. # safe because cargo fingerprints the real compiler and rebuilds on a
# mismatch, it just means a base-image toolchain bump costs one cold build
# instead of invalidating the key.
- name: Cache cargo (docker builds) - name: Cache cargo (docker builds)
uses: Swatinem/rust-cache@v2 uses: Swatinem/rust-cache@v2
if: ${{ matrix.settings.docker }} if: ${{ matrix.settings.docker }}
@@ -238,19 +210,14 @@ jobs:
# cache step above saves. Previously the registry mounts pointed at # cache step above saves. Previously the registry mounts pointed at
# `.cargo/...`, a path nothing cached, so the container re-downloaded # `.cargo/...`, a path nothing cached, so the container re-downloaded
# the whole crate registry on every run. # the whole crate registry on every run.
#
# `docker run` inherits nothing; `-e NAME` carries the job's `env:` in.
options: "--user 0:0 -v ${{ github.workspace }}/.cargo-cache/git/db:/usr/local/cargo/git/db \ options: "--user 0:0 -v ${{ github.workspace }}/.cargo-cache/git/db:/usr/local/cargo/git/db \
-v ${{ github.workspace }}/.cargo-cache/registry/cache:/usr/local/cargo/registry/cache \ -v ${{ github.workspace }}/.cargo-cache/registry/cache:/usr/local/cargo/registry/cache \
-v ${{ github.workspace }}/.cargo-cache/registry/index:/usr/local/cargo/registry/index \ -v ${{ github.workspace }}/.cargo-cache/registry/index:/usr/local/cargo/registry/index \
-e CARGO_PROFILE_RELEASE_LTO \
-e CARGO_PROFILE_RELEASE_CODEGEN_UNITS \
-e CARGO_TARGET_AARCH64_UNKNOWN_LINUX_GNU_LINKER \
-v ${{ github.workspace }}:/build -w /build/nodejs" -v ${{ github.workspace }}:/build -w /build/nodejs"
run: | run: |
set -e set -e
${{ matrix.settings.pre_build }} ${{ matrix.settings.pre_build }}
node_modules/.bin/napi build --platform --release \ npx napi build --platform --release \
--features ${{ matrix.settings.features }} \ --features ${{ matrix.settings.features }} \
--target ${{ matrix.settings.target }} \ --target ${{ matrix.settings.target }} \
--dts ../lancedb/native.d.ts \ --dts ../lancedb/native.d.ts \
@@ -270,7 +237,7 @@ jobs:
- name: Build - name: Build
run: | run: |
${{ matrix.settings.pre_build }} ${{ matrix.settings.pre_build }}
node_modules/.bin/napi build --platform --release \ npx napi build --platform --release \
--features ${{ matrix.settings.features }} \ --features ${{ matrix.settings.features }} \
--target ${{ matrix.settings.target }} \ --target ${{ matrix.settings.target }} \
--dts ../lancedb/native.d.ts \ --dts ../lancedb/native.d.ts \
@@ -289,18 +256,6 @@ jobs:
if: always() if: always()
run: df -h run: df -h
shell: bash shell: bash
- name: Report peak memory
if: always() && runner.os == 'Linux'
shell: bash
run: |
peak=$(find /sys/fs/cgroup -name memory.peak -readable \
-exec cat {} + 2>/dev/null | sort -n | tail -1)
if [ -n "$peak" ]; then
echo "peak memory: $((peak / 1024 / 1024)) MiB"
else
echo "peak memory: unavailable (no readable cgroup v2 memory.peak)"
fi
free -g || true
- name: Upload artifact - name: Upload artifact
uses: actions/upload-artifact@v7 uses: actions/upload-artifact@v7
with: with:
@@ -338,7 +293,7 @@ jobs:
- target: aarch64-unknown-linux-gnu - target: aarch64-unknown-linux-gnu
host: ubuntu-2404-8x-arm64 host: ubuntu-2404-8x-arm64
node: node:
- '22' - '20'
runs-on: ${{ matrix.settings.host }} runs-on: ${{ matrix.settings.host }}
defaults: defaults:
run: run:
@@ -384,9 +339,9 @@ jobs:
- name: Move built files - name: Move built files
run: cp dist/native.d.ts dist/native.js dist/*.node lancedb/ run: cp dist/native.d.ts dist/native.js dist/*.node lancedb/
- name: Test bindings - name: Test bindings
# Invoke the installed jest binary directly; the pnpm shim is set up # Invoke jest directly because pnpm 11 itself requires Node 22+
# against the install-phase Node, not the version selected above. # while the matrix tests on older Node versions.
run: node_modules/.bin/jest --verbose run: npx jest --verbose
publish: publish:
name: Publish name: Publish
runs-on: ubuntu-latest runs-on: ubuntu-latest
+1 -4
View File
@@ -232,10 +232,7 @@ jobs:
ALL_FEATURES=`cargo metadata --format-version=1 --no-deps \ ALL_FEATURES=`cargo metadata --format-version=1 --no-deps \
| jq -r '.packages[] | .features | keys | .[]' \ | jq -r '.packages[] | .features | keys | .[]' \
| grep -v s3-test | sort | uniq | paste -s -d "," -` | grep -v s3-test | sort | uniq | paste -s -d "," -`
# Run doctests before test binaries fill the runner disk. Examples are cargo test --profile ci --features $ALL_FEATURES --locked
# already built by the Linux job, so avoid retaining them here.
cargo test --profile ci --features $ALL_FEATURES --locked --doc
cargo test --profile ci --features $ALL_FEATURES --locked --lib --tests
windows: windows:
strategy: strategy:
@@ -0,0 +1,22 @@
name: Update package-lock.json
on:
workflow_dispatch:
permissions:
contents: read
jobs:
publish:
runs-on: ubuntu-latest
steps:
- name: Checkout
uses: actions/checkout@v6
with:
ref: main
persist-credentials: false
fetch-depth: 0
lfs: true
- uses: ./.github/workflows/update_package_lock
with:
github_token: ${{ secrets.LANCEDB_RELEASE_TOKEN }}
@@ -0,0 +1,22 @@
name: Update NodeJs package-lock.json
on:
workflow_dispatch:
permissions:
contents: read
jobs:
publish:
runs-on: ubuntu-latest
steps:
- name: Checkout
uses: actions/checkout@v6
with:
ref: main
persist-credentials: false
fetch-depth: 0
lfs: true
- uses: ./.github/workflows/update_package_lock_nodejs
with:
github_token: ${{ secrets.LANCEDB_RELEASE_TOKEN }}
+1 -4
View File
@@ -20,10 +20,7 @@ repos:
hooks: hooks:
- id: local-biome-check - id: local-biome-check
name: biome check name: biome check
# Use the biome from nodejs/package.json rather than a separately entry: npx @biomejs/biome@1.8.3 check --config-path nodejs/biome.json nodejs/
# pinned one: the two drifted apart and disagreed on formatting, so
# this hook rejected code that `pnpm lint` accepted.
entry: nodejs/node_modules/.bin/biome check --config-path nodejs/biome.json nodejs/
language: system language: system
types: [text] types: [text]
files: "nodejs/.*" files: "nodejs/.*"
+3 -3
View File
@@ -38,7 +38,7 @@ Before committing changes, run formatting for every language you touched. At min
* Rust changes: run `cargo fmt --all`. * Rust changes: run `cargo fmt --all`.
* Python changes: run `ruff format .` and `ruff check .` from the repository root, * Python changes: run `ruff format .` and `ruff check .` from the repository root,
and run targeted tests through `cd python && uv run ...`. and run targeted tests through `cd python && uv run ...`.
* TypeScript changes: run the relevant `pnpm` lint, format, build, and docs commands in `nodejs`. * TypeScript changes: run the relevant `npm`/`pnpm` lint, format, build, and docs commands in `nodejs`.
Before creating a PR, the exact value passed to `gh pr create --title` must follow Before creating a PR, the exact value passed to `gh pr create --title` must follow
Conventional Commits, such as `fix: support nested field paths in native index creation` Conventional Commits, such as `fix: support nested field paths in native index creation`
@@ -101,12 +101,12 @@ Python bindings changes:
TypeScript bindings changes: TypeScript bindings changes:
1. Add napi-rs method binding on `Table` in `nodejs/src/table.rs`. 1. Add napi-rs method binding on `Table` in `nodejs/src/table.rs`.
2. Run `pnpm build` to generate TypeScript definitions. 2. Run `npm run build` to generate TypeScript definitions.
3. Add typescript method on abstract class `Table` in `nodejs/src/table.ts`. 3. Add typescript method on abstract class `Table` in `nodejs/src/table.ts`.
4. Add concrete method on `LocalTable` class in `nodejs/src/native_table.ts`. 4. Add concrete method on `LocalTable` class in `nodejs/src/native_table.ts`.
* Note: despite the name, this class is also used for remote tables. * Note: despite the name, this class is also used for remote tables.
5. Add test in `nodejs/__test__/table.test.ts`. 5. Add test in `nodejs/__test__/table.test.ts`.
6. Run `pnpm run docs` to generate TypeScript documentation. 6. Run `npm run docs` to generate TypeScript documentation.
## Python API reference ## Python API reference
Generated
+76 -170
View File
@@ -332,34 +332,6 @@ dependencies = [
"num-traits", "num-traits",
] ]
[[package]]
name = "arrow-flight"
version = "58.4.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b2dbe34824c639e43136af8f106992792ab456540d54b880bc320a3192502d2e"
dependencies = [
"arrow-arith",
"arrow-array",
"arrow-buffer",
"arrow-cast",
"arrow-data",
"arrow-ipc",
"arrow-ord",
"arrow-row",
"arrow-schema",
"arrow-select",
"arrow-string",
"base64 0.22.1",
"bytes",
"futures",
"once_cell",
"paste",
"prost",
"prost-types",
"tonic",
"tonic-prost",
]
[[package]] [[package]]
name = "arrow-ipc" name = "arrow-ipc"
version = "58.4.0" version = "58.4.0"
@@ -563,9 +535,9 @@ dependencies = [
[[package]] [[package]]
name = "async-trait" name = "async-trait"
version = "0.1.92" version = "0.1.91"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "82f6aeea286b8eb4dd3431a1be1b59d290ace00f5bfd8e2a159bc2a05e2c1667" checksum = "ae36dc4177970ef04fde5178d3e2429882def40e57a451f919c098f72baa6cec"
dependencies = [ dependencies = [
"proc-macro2", "proc-macro2",
"quote", "quote",
@@ -1157,7 +1129,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "edca88bc138befd0323b20752846e6587272d3b03b0343c8ea28a6f819e6e71f" checksum = "edca88bc138befd0323b20752846e6587272d3b03b0343c8ea28a6f819e6e71f"
dependencies = [ dependencies = [
"async-trait", "async-trait",
"axum-core 0.4.5", "axum-core",
"bytes", "bytes",
"futures-util", "futures-util",
"http 1.5.0", "http 1.5.0",
@@ -1166,7 +1138,7 @@ dependencies = [
"hyper 1.9.0", "hyper 1.9.0",
"hyper-util", "hyper-util",
"itoa", "itoa",
"matchit 0.7.3", "matchit",
"memchr", "memchr",
"mime", "mime",
"percent-encoding", "percent-encoding",
@@ -1184,31 +1156,6 @@ dependencies = [
"tracing", "tracing",
] ]
[[package]]
name = "axum"
version = "0.8.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "31b698c5f9a010f6573133b09e0de5408834d0c82f8d7475a89fc1867a71cd90"
dependencies = [
"axum-core 0.5.6",
"bytes",
"futures-util",
"http 1.5.0",
"http-body 1.1.0",
"http-body-util",
"itoa",
"matchit 0.8.4",
"memchr",
"mime",
"percent-encoding",
"pin-project-lite",
"serde_core",
"sync_wrapper",
"tower",
"tower-layer",
"tower-service",
]
[[package]] [[package]]
name = "axum-core" name = "axum-core"
version = "0.4.5" version = "0.4.5"
@@ -1230,24 +1177,6 @@ dependencies = [
"tracing", "tracing",
] ]
[[package]]
name = "axum-core"
version = "0.5.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "08c78f31d7b1291f7ee735c1c6780ccde7785daae9a9206026862dab7d8792d1"
dependencies = [
"bytes",
"futures-core",
"http 1.5.0",
"http-body 1.1.0",
"http-body-util",
"mime",
"pin-project-lite",
"sync_wrapper",
"tower-layer",
"tower-service",
]
[[package]] [[package]]
name = "backoff" name = "backoff"
version = "0.4.0" version = "0.4.0"
@@ -1514,9 +1443,9 @@ checksum = "175812e0be2bccb6abe50bb8d566126198344f707e304f45c648fd8f2cc0365e"
[[package]] [[package]]
name = "bytemuck" name = "bytemuck"
version = "1.25.2" version = "1.25.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "95832e849adfb21180ccb6826a99da14e5d266ae5c2e668e1602cf234f153797" checksum = "c8efb64bd706a16a1bdde310ae86b351e4d21550d98d056f22f8a7f7a2183fec"
dependencies = [ dependencies = [
"bytemuck_derive", "bytemuck_derive",
] ]
@@ -1668,9 +1597,9 @@ checksum = "613afe47fcd5fac7ccf1db93babcb082c5994d996f20b8b159f2ad1658eb5724"
[[package]] [[package]]
name = "chacha20" name = "chacha20"
version = "0.10.2" version = "0.10.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "65c35e4b699c7e15ccbe7ee35c005e4fc0a278d22238a2857e6ce2dadeda1b06" checksum = "6f8d983286843e49675a4b7a2d174efe136dc93a18d69130dd18198a6c167601"
dependencies = [ dependencies = [
"cfg-if 1.0.4", "cfg-if 1.0.4",
"cpufeatures 0.3.0", "cpufeatures 0.3.0",
@@ -3526,8 +3455,8 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c"
[[package]] [[package]]
name = "fsst" name = "fsst"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow-array", "arrow-array",
"rand 0.9.5", "rand 0.9.5",
@@ -4886,8 +4815,8 @@ checksum = "e037a2e1d8d5fdbd49b16a4ea09d5d6401c1f29eca5ff29d03d3824dba16256a"
[[package]] [[package]]
name = "lance" name = "lance"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arc-swap", "arc-swap",
"arrow", "arrow",
@@ -4959,8 +4888,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-arrow" name = "lance-arrow"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow-array", "arrow-array",
"arrow-buffer", "arrow-buffer",
@@ -4982,7 +4911,7 @@ dependencies = [
[[package]] [[package]]
name = "lance-arrow-scalar" name = "lance-arrow-scalar"
version = "58.0.0" version = "58.0.0"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow-array", "arrow-array",
"arrow-buffer", "arrow-buffer",
@@ -4996,7 +4925,7 @@ dependencies = [
[[package]] [[package]]
name = "lance-arrow-stats" name = "lance-arrow-stats"
version = "58.0.0" version = "58.0.0"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow-array", "arrow-array",
"arrow-schema", "arrow-schema",
@@ -5005,8 +4934,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-bitpacking" name = "lance-bitpacking"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrayref", "arrayref",
"crunchy", "crunchy",
@@ -5016,8 +4945,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-core" name = "lance-core"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow-array", "arrow-array",
"arrow-buffer", "arrow-buffer",
@@ -5054,8 +4983,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-datafusion" name = "lance-datafusion"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow", "arrow",
"arrow-array", "arrow-array",
@@ -5071,7 +5000,6 @@ dependencies = [
"datafusion-functions", "datafusion-functions",
"datafusion-physical-expr", "datafusion-physical-expr",
"futures", "futures",
"half",
"jsonb", "jsonb",
"lance-arrow", "lance-arrow",
"lance-core", "lance-core",
@@ -5085,8 +5013,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-datagen" name = "lance-datagen"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow", "arrow",
"arrow-array", "arrow-array",
@@ -5103,8 +5031,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-derive" name = "lance-derive"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"proc-macro2", "proc-macro2",
"quote", "quote",
@@ -5113,8 +5041,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-encoding" name = "lance-encoding"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow-arith", "arrow-arith",
"arrow-array", "arrow-array",
@@ -5147,8 +5075,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-file" name = "lance-file"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow-arith", "arrow-arith",
"arrow-array", "arrow-array",
@@ -5179,8 +5107,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-index" name = "lance-index"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arc-swap", "arc-swap",
"arrow", "arrow",
@@ -5244,8 +5172,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-index-core" name = "lance-index-core"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow-array", "arrow-array",
"arrow-schema", "arrow-schema",
@@ -5267,8 +5195,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-io" name = "lance-io"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow", "arrow",
"arrow-array", "arrow-array",
@@ -5294,11 +5222,7 @@ dependencies = [
"pin-project", "pin-project",
"prost", "prost",
"rand 0.9.5", "rand 0.9.5",
"reqsign-core",
"reqsign-file-read-tokio",
"reqsign-google",
"serde", "serde",
"serde_json",
"tempfile", "tempfile",
"tokio", "tokio",
"tracing", "tracing",
@@ -5308,8 +5232,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-linalg" name = "lance-linalg"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow-array", "arrow-array",
"arrow-schema", "arrow-schema",
@@ -5323,29 +5247,27 @@ dependencies = [
[[package]] [[package]]
name = "lance-namespace" name = "lance-namespace"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow", "arrow",
"async-trait", "async-trait",
"bytes", "bytes",
"lance-core", "lance-core",
"lance-namespace-reqwest-client", "lance-namespace-reqwest-client",
"serde",
"serde_json",
"snafu 0.9.0", "snafu 0.9.0",
] ]
[[package]] [[package]]
name = "lance-namespace-impls" name = "lance-namespace-impls"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow", "arrow",
"arrow-ipc", "arrow-ipc",
"arrow-schema", "arrow-schema",
"async-trait", "async-trait",
"axum 0.7.9", "axum",
"base64 0.22.1", "base64 0.22.1",
"bytes", "bytes",
"chrono", "chrono",
@@ -5378,9 +5300,9 @@ dependencies = [
[[package]] [[package]]
name = "lance-namespace-reqwest-client" name = "lance-namespace-reqwest-client"
version = "0.12.0" version = "0.11.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d8d23e54b1634d5bbb434f8dd33dc3c05f6e58d876a9a27b3b4aef58ddbe11af" checksum = "0a030196da1c994b63a96a4f0bf5b0cfa459fe6dadc9e962320246ca328da22a"
dependencies = [ dependencies = [
"reqwest 0.12.28", "reqwest 0.12.28",
"serde", "serde",
@@ -5392,8 +5314,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-select" name = "lance-select"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow-array", "arrow-array",
"arrow-buffer", "arrow-buffer",
@@ -5407,8 +5329,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-table" name = "lance-table"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow", "arrow",
"arrow-array", "arrow-array",
@@ -5448,8 +5370,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-testing" name = "lance-testing"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"arrow-array", "arrow-array",
"arrow-schema", "arrow-schema",
@@ -5462,8 +5384,8 @@ dependencies = [
[[package]] [[package]]
name = "lance-tokenizer" name = "lance-tokenizer"
version = "12.0.0-beta.14" version = "11.0.0-beta.21"
source = "git+https://github.com/lance-format/lance.git?tag=v12.0.0-beta.14#a8a101774a1c9647065cc60137094feadbe55296" source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.21#dd08336cf61b117701a7f5bbf76a7f7080f7e210"
dependencies = [ dependencies = [
"frostem", "frostem",
"icu_segmenter", "icu_segmenter",
@@ -5476,7 +5398,7 @@ dependencies = [
[[package]] [[package]]
name = "lancedb" name = "lancedb"
version = "0.39.0-beta.4" version = "0.38.0-beta.4"
dependencies = [ dependencies = [
"ahash", "ahash",
"anyhow", "anyhow",
@@ -5485,7 +5407,6 @@ dependencies = [
"arrow-buffer", "arrow-buffer",
"arrow-cast", "arrow-cast",
"arrow-data", "arrow-data",
"arrow-flight",
"arrow-ipc", "arrow-ipc",
"arrow-ord", "arrow-ord",
"arrow-schema", "arrow-schema",
@@ -5541,7 +5462,6 @@ dependencies = [
"polars", "polars",
"polars-arrow", "polars-arrow",
"pprof 0.14.1", "pprof 0.14.1",
"prost",
"rand 0.9.5", "rand 0.9.5",
"random_word", "random_word",
"regex", "regex",
@@ -5558,7 +5478,6 @@ dependencies = [
"test-log", "test-log",
"tokenizers", "tokenizers",
"tokio", "tokio",
"tonic",
"url", "url",
"urlencoding", "urlencoding",
"uuid", "uuid",
@@ -5567,7 +5486,7 @@ dependencies = [
[[package]] [[package]]
name = "lancedb-nodejs" name = "lancedb-nodejs"
version = "0.39.0-beta.4" version = "0.38.0-beta.4"
dependencies = [ dependencies = [
"arrow-array", "arrow-array",
"arrow-buffer", "arrow-buffer",
@@ -5592,7 +5511,7 @@ dependencies = [
[[package]] [[package]]
name = "lancedb-python" name = "lancedb-python"
version = "0.39.0-beta.4" version = "0.38.0-beta.4"
dependencies = [ dependencies = [
"arrow", "arrow",
"async-trait", "async-trait",
@@ -5616,7 +5535,6 @@ dependencies = [
"serde_json", "serde_json",
"snafu 0.8.9", "snafu 0.8.9",
"tokio", "tokio",
"uuid",
] ]
[[package]] [[package]]
@@ -5826,9 +5744,9 @@ dependencies = [
[[package]] [[package]]
name = "log" name = "log"
version = "0.4.34" version = "0.4.33"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad"
[[package]] [[package]]
name = "loom" name = "loom"
@@ -5937,12 +5855,6 @@ version = "0.7.3"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0e7465ac9959cc2b1404e8e2367b43684a6d13790fe23056cc8c6c5a6b7bcb94" checksum = "0e7465ac9959cc2b1404e8e2367b43684a6d13790fe23056cc8c6c5a6b7bcb94"
[[package]]
name = "matchit"
version = "0.8.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "47e1ffaa40ddd1f3ed91f717a33c8c0ee23fff369e3aa8772b9605cc1d22f4c3"
[[package]] [[package]]
name = "matrixmultiply" name = "matrixmultiply"
version = "0.3.10" version = "0.3.10"
@@ -6085,9 +5997,9 @@ dependencies = [
[[package]] [[package]]
name = "moka" name = "moka"
version = "0.12.16" version = "0.12.15"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4293f18e7567a1caf3c584855554377025c65e0aa445344d04171f5ad63d19b9" checksum = "957228ad12042ee839f93c8f257b62b4c0ab5eaae1d4fa60de53b27c9d7c5046"
dependencies = [ dependencies = [
"async-lock", "async-lock",
"crossbeam-channel", "crossbeam-channel",
@@ -6181,15 +6093,14 @@ dependencies = [
[[package]] [[package]]
name = "napi" name = "napi"
version = "3.12.2" version = "3.11.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "58c5f4d5375213fdb7be2655e152386e82f026f9a5ba36a75556e11359aafe09" checksum = "de33522036981030a75c231829566bc63414e08101a6f5ff4ac6cef19c8e0941"
dependencies = [ dependencies = [
"bitflags 2.11.1", "bitflags 2.11.1",
"chrono", "chrono",
"ctor 1.0.12", "ctor 1.0.12",
"futures", "futures",
"libc",
"napi-build", "napi-build",
"napi-sys", "napi-sys",
"nohash-hasher", "nohash-hasher",
@@ -6201,15 +6112,15 @@ dependencies = [
[[package]] [[package]]
name = "napi-build" name = "napi-build"
version = "2.4.1" version = "2.4.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "60fdf9b392c50e7c4170fa633bd909490ed7835cea4c046776d1a4dd8d2ae0ab" checksum = "5282704fbe8d49b0cf8b08e3f33233416a528658f205c7e5ace63b582de0b11c"
[[package]] [[package]]
name = "napi-derive" name = "napi-derive"
version = "3.6.3" version = "3.6.1"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0fa55ea69990c90b888e9e77044410e304ce7f35de599dc6d0b5c1923d2e59af" checksum = "4d5c9c02556ea6dc99dffd36c1ce60141411657438501a125b675776d011ce92"
dependencies = [ dependencies = [
"convert_case", "convert_case",
"ctor 1.0.12", "ctor 1.0.12",
@@ -6221,9 +6132,9 @@ dependencies = [
[[package]] [[package]]
name = "napi-derive-backend" name = "napi-derive-backend"
version = "6.1.2" version = "6.1.1"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "df4056ac7c18e4438ccf0edaed4340ca0d269278c8ec19284f7b23cb039fd0ae" checksum = "d60b5d773ad46c698c8cc2cd9fde0b283d39cbb7f71c04bee633c7bdba4423bd"
dependencies = [ dependencies = [
"convert_case", "convert_case",
"proc-macro2", "proc-macro2",
@@ -7818,7 +7729,6 @@ dependencies = [
"pyo3-build-config", "pyo3-build-config",
"pyo3-ffi", "pyo3-ffi",
"pyo3-macros", "pyo3-macros",
"uuid",
] ]
[[package]] [[package]]
@@ -8687,9 +8597,9 @@ dependencies = [
[[package]] [[package]]
name = "roaring" name = "roaring"
version = "0.11.5" version = "0.11.4"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "18bd8a37d17a58532776dcdf6041ce64929adca78e8489d5cacbafe99229d3e1" checksum = "1dedc5658c6ecb3bdb5ef5f3295bb9253f42dcf3fd1402c03f6b1f7659c3c4a9"
dependencies = [ dependencies = [
"bytemuck", "bytemuck",
"byteorder", "byteorder",
@@ -9149,9 +9059,9 @@ dependencies = [
[[package]] [[package]]
name = "serde_with" name = "serde_with"
version = "3.22.0" version = "3.21.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ee78f1fbe43ac4a0e47aadb3dbd357b69eb0d3793e948624cd03dd2750ab1c0a" checksum = "76a5c54c7310e7b8b9577c286d7e399ddd876c3e12b3ed917a8aabc4b96e9e8c"
dependencies = [ dependencies = [
"base64 0.22.1", "base64 0.22.1",
"bs58", "bs58",
@@ -9159,7 +9069,6 @@ dependencies = [
"hex", "hex",
"indexmap 1.9.3", "indexmap 1.9.3",
"indexmap 2.14.0", "indexmap 2.14.0",
"jiff",
"schemars 0.9.0", "schemars 0.9.0",
"schemars 1.2.1", "schemars 1.2.1",
"serde_core", "serde_core",
@@ -9170,9 +9079,9 @@ dependencies = [
[[package]] [[package]]
name = "serde_with_macros" name = "serde_with_macros"
version = "3.22.0" version = "3.21.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8705578779c2b6bd90d84d66eb2e206b708b1a4d7b9f17641b293545bf1c7e46" checksum = "84d57bc0c8b9a17920c178daa6bb924850d54a9c97ab45194bb8c17ad66bb660"
dependencies = [ dependencies = [
"darling 0.23.0", "darling 0.23.0",
"proc-macro2", "proc-macro2",
@@ -10171,7 +10080,6 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ac2a5518c70fa84342385732db33fb3f44bc4cc748936eb5833d2df34d6445ef" checksum = "ac2a5518c70fa84342385732db33fb3f44bc4cc748936eb5833d2df34d6445ef"
dependencies = [ dependencies = [
"async-trait", "async-trait",
"axum 0.8.9",
"base64 0.22.1", "base64 0.22.1",
"bytes", "bytes",
"h2 0.4.16", "h2 0.4.16",
@@ -10183,11 +10091,9 @@ dependencies = [
"hyper-util", "hyper-util",
"percent-encoding", "percent-encoding",
"pin-project", "pin-project",
"rustls-native-certs",
"socket2 0.6.3", "socket2 0.6.3",
"sync_wrapper", "sync_wrapper",
"tokio", "tokio",
"tokio-rustls 0.26.4",
"tokio-stream", "tokio-stream",
"tower", "tower",
"tower-layer", "tower-layer",
@@ -10542,9 +10448,9 @@ checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821"
[[package]] [[package]]
name = "uuid" name = "uuid"
version = "1.26.0" version = "1.24.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b5772d71c9be8a8a6ac2117d949c5b224c1b72241bb611d9a3012edcf8af7812" checksum = "bf3923a6f5c4c6382e0b653c4117f48d631ea17f38ed86e2a828e6f7412f5239"
dependencies = [ dependencies = [
"getrandom 0.4.2", "getrandom 0.4.2",
"js-sys", "js-sys",
+15 -17
View File
@@ -13,20 +13,20 @@ categories = ["database-implementations"]
rust-version = "1.91.0" rust-version = "1.91.0"
[workspace.dependencies] [workspace.dependencies]
lance = { "version" = "=12.0.0-beta.14", default-features = false, "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } lance = { "version" = "=11.0.0-beta.21", default-features = false, "tag" = "v11.0.0-beta.21", "git" = "https://github.com/lance-format/lance.git" }
lance-core = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } lance-core = { "version" = "=11.0.0-beta.21", "tag" = "v11.0.0-beta.21", "git" = "https://github.com/lance-format/lance.git" }
lance-datagen = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } lance-datagen = { "version" = "=11.0.0-beta.21", "tag" = "v11.0.0-beta.21", "git" = "https://github.com/lance-format/lance.git" }
lance-file = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } lance-file = { "version" = "=11.0.0-beta.21", "tag" = "v11.0.0-beta.21", "git" = "https://github.com/lance-format/lance.git" }
lance-io = { "version" = "=12.0.0-beta.14", default-features = false, "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } lance-io = { "version" = "=11.0.0-beta.21", default-features = false, "tag" = "v11.0.0-beta.21", "git" = "https://github.com/lance-format/lance.git" }
lance-index = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } lance-index = { "version" = "=11.0.0-beta.21", "tag" = "v11.0.0-beta.21", "git" = "https://github.com/lance-format/lance.git" }
lance-linalg = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } lance-linalg = { "version" = "=11.0.0-beta.21", "tag" = "v11.0.0-beta.21", "git" = "https://github.com/lance-format/lance.git" }
lance-namespace = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } lance-namespace = { "version" = "=11.0.0-beta.21", "tag" = "v11.0.0-beta.21", "git" = "https://github.com/lance-format/lance.git" }
lance-namespace-impls = { "version" = "=12.0.0-beta.14", default-features = false, "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } lance-namespace-impls = { "version" = "=11.0.0-beta.21", default-features = false, "tag" = "v11.0.0-beta.21", "git" = "https://github.com/lance-format/lance.git" }
lance-table = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } lance-table = { "version" = "=11.0.0-beta.21", "tag" = "v11.0.0-beta.21", "git" = "https://github.com/lance-format/lance.git" }
lance-testing = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } lance-testing = { "version" = "=11.0.0-beta.21", "tag" = "v11.0.0-beta.21", "git" = "https://github.com/lance-format/lance.git" }
lance-datafusion = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } lance-datafusion = { "version" = "=11.0.0-beta.21", "tag" = "v11.0.0-beta.21", "git" = "https://github.com/lance-format/lance.git" }
lance-encoding = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } lance-encoding = { "version" = "=11.0.0-beta.21", "tag" = "v11.0.0-beta.21", "git" = "https://github.com/lance-format/lance.git" }
lance-arrow = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } lance-arrow = { "version" = "=11.0.0-beta.21", "tag" = "v11.0.0-beta.21", "git" = "https://github.com/lance-format/lance.git" }
lancedb = { path = "rust/lancedb", default-features = false } lancedb = { path = "rust/lancedb", default-features = false }
ahash = "0.8" ahash = "0.8"
# Note that this one does not include pyarrow # Note that this one does not include pyarrow
@@ -39,7 +39,6 @@ arrow-ord = "58.0.0"
arrow-schema = "58.0.0" arrow-schema = "58.0.0"
arrow-select = "58.0.0" arrow-select = "58.0.0"
arrow-cast = "58.0.0" arrow-cast = "58.0.0"
arrow-flight = { version = "58.0.0", features = ["flight-sql-experimental"] }
async-trait = "0" async-trait = "0"
bytes = "1" bytes = "1"
datafusion = { version = "54.0.0", default-features = false } datafusion = { version = "54.0.0", default-features = false }
@@ -72,8 +71,7 @@ serde = "1"
serde_json = "1" serde_json = "1"
tempfile = "3.5.0" tempfile = "3.5.0"
tokio = { version = "1.23", features = ["rt-multi-thread", "sync"] } tokio = { version = "1.23", features = ["rt-multi-thread", "sync"] }
tonic = { version = "0.14", features = ["tls-native-roots", "tls-ring"] } uuid = { version = "1.7.0", features = ["v4"] }
uuid = { version = "1.7.0", features = ["v4", "v7"] }
chrono = { version = "0.4", default-features = false, features = ["clock"] } chrono = { version = "0.4", default-features = false, features = ["clock"] }
[profile.ci] [profile.ci]
+1 -1
View File
@@ -5,5 +5,5 @@ licenses:
cd python && cargo about generate ../about.hbs -o RUST_THIRD_PARTY_LICENSES.html -c ../about.toml cd python && cargo about generate ../about.hbs -o RUST_THIRD_PARTY_LICENSES.html -c ../about.toml
cd python && uv sync --all-extras && uv tool run pip-licenses --python .venv/bin/python --format=markdown --with-urls --output-file=PYTHON_THIRD_PARTY_LICENSES.md cd python && uv sync --all-extras && uv tool run pip-licenses --python .venv/bin/python --format=markdown --with-urls --output-file=PYTHON_THIRD_PARTY_LICENSES.md
cd nodejs && cargo about generate ../about.hbs -o RUST_THIRD_PARTY_LICENSES.html -c ../about.toml cd nodejs && cargo about generate ../about.hbs -o RUST_THIRD_PARTY_LICENSES.html -c ../about.toml
cd nodejs && pnpm dlx license-checker@25 --markdown --out NODEJS_THIRD_PARTY_LICENSES.md cd nodejs && npx license-checker --markdown --out NODEJS_THIRD_PARTY_LICENSES.md
cd java && ./mvnw license:aggregate-add-third-party -q cd java && ./mvnw license:aggregate-add-third-party -q
+6 -2
View File
@@ -12,12 +12,16 @@ done
# This updates the lockfile without building # This updates the lockfile without building
cargo metadata --quiet > /dev/null cargo metadata --quiet > /dev/null
pushd nodejs || exit 1
npm install --package-lock-only --silent
popd
if git diff --quiet --exit-code; then if git diff --quiet --exit-code; then
echo "No lockfile changes to commit; skipping amend." echo "No lockfile changes to commit; skipping amend."
elif $AMEND; then elif $AMEND; then
git add Cargo.lock git add Cargo.lock nodejs/package-lock.json
git commit --amend --no-edit git commit --amend --no-edit
else else
git add Cargo.lock git add Cargo.lock nodejs/package-lock.json
git commit -m "Update lockfiles" git commit -m "Update lockfiles"
fi fi
+11 -1
View File
@@ -131,13 +131,18 @@ allow = [
"BSD-3-Clause", "BSD-3-Clause",
"ISC", "ISC",
"Unicode-3.0", "Unicode-3.0",
"Unicode-DFS-2016",
"Zlib", "Zlib",
"CC0-1.0", "CC0-1.0",
"MPL-2.0", "MPL-2.0",
"BSL-1.0", "BSL-1.0",
"OpenSSL",
# 0BSD ("BSD Zero Clause") is effectively public domain — no attribution # 0BSD ("BSD Zero Clause") is effectively public domain — no attribution
# required. Pulled in by `mock_instant`. # required. Pulled in by `mock_instant`.
"0BSD", "0BSD",
# bzip2-1.0.6 is the permissive upstream bzip2 license (BSD-like). Pulled
# in by `libbz2-rs-sys`, the pure-Rust bzip2 implementation.
"bzip2-1.0.6",
# CDLA-Permissive-2.0 is a permissive data license used by `webpki-roots` # CDLA-Permissive-2.0 is a permissive data license used by `webpki-roots`
# for the Mozilla CA root bundle. Data-only, distribution-compatible. # for the Mozilla CA root bundle. Data-only, distribution-compatible.
"CDLA-Permissive-2.0", "CDLA-Permissive-2.0",
@@ -145,7 +150,12 @@ allow = [
confidence-threshold = 0.8 confidence-threshold = 0.8
# Per-crate license exceptions: allow a license for a specific crate only, # Per-crate license exceptions: allow a license for a specific crate only,
# rather than globally via the `allow` list above. # rather than globally via the `allow` list above.
exceptions = [] exceptions = [
# CDDL-1.0 (copyleft) is pulled in only as a dev/profiling dependency via
# `inferno` -> `pprof` -> `lance-testing`; it is a test dependency that we
# do not distribute, so scope the allowance to `inferno` alone.
{ allow = ["CDDL-1.0"], crate = "inferno" },
]
# Crates whose license cannot be determined from Cargo metadata but whose # Crates whose license cannot be determined from Cargo metadata but whose
# license we've manually confirmed from upstream. Keep this list minimal. # license we've manually confirmed from upstream. Keep this list minimal.
[[licenses.clarify]] [[licenses.clarify]]
+8 -11
View File
@@ -47,24 +47,22 @@ pytest -vv python/tests/docs
### Checking typescript examples ### Checking typescript examples
The examples depend on `@lancedb/lancedb` at `file:../dist`, so the package must be The `@lancedb/lancedb` package must be built before running the tests:
built before running the tests. This uses pnpm; see the
[Typescript contributing guide](../nodejs/CONTRIBUTING.md) for the toolchain setup.
```shell ```shell
pushd nodejs pushd nodejs
pnpm install npm ci
pnpm build npm run build
popd popd
``` ```
Then you can run the examples by going to the `nodejs/examples` directory, which is a Then you can run the examples by going to the `nodejs/examples` directory and
separate pnpm package with its own lockfile: running the tests like a normal npm package:
```shell ```shell
pushd nodejs/examples pushd nodejs/examples
pnpm install npm ci
pnpm test npm test
popd popd
``` ```
@@ -86,7 +84,6 @@ The new files should be checked into the repository.
```shell ```shell
pushd nodejs pushd nodejs
# `pnpm docs` would invoke pnpm's built-in `docs` command, not the script. npm run docs
pnpm run docs
popd popd
``` ```
-9
View File
@@ -446,15 +446,6 @@ paths:
properties: properties:
column: column:
type: string type: string
name:
type: string
description: Optional name for the created index.
replace:
type: boolean
default: true
description: |
Whether to replace an existing index with the same resolved
name. Defaults to true.
metric_type: metric_type:
type: string type: string
nullable: false nullable: false
+135
View File
@@ -0,0 +1,135 @@
{
"name": "lancedb-docs-test",
"version": "1.0.0",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "lancedb-docs-test",
"version": "1.0.0",
"license": "Apache 2",
"dependencies": {
"apache-arrow": "file:../node/node_modules/apache-arrow",
"vectordb": "file:../node"
},
"devDependencies": {
"@types/node": "^20.11.8",
"typescript": "^5.3.3"
}
},
"../node": {
"name": "vectordb",
"version": "0.21.2-beta.0",
"cpu": [
"x64",
"arm64"
],
"license": "Apache-2.0",
"os": [
"darwin",
"linux",
"win32"
],
"dependencies": {
"@neon-rs/load": "^0.0.74",
"axios": "^1.4.0"
},
"devDependencies": {
"@neon-rs/cli": "^0.0.160",
"@types/chai": "^4.3.4",
"@types/chai-as-promised": "^7.1.5",
"@types/mocha": "^10.0.1",
"@types/node": "^18.16.2",
"@types/sinon": "^10.0.15",
"@types/temp": "^0.9.1",
"@types/uuid": "^9.0.3",
"@typescript-eslint/eslint-plugin": "^5.59.1",
"apache-arrow-old": "npm:apache-arrow@13.0.0",
"cargo-cp-artifact": "^0.1",
"chai": "^4.3.7",
"chai-as-promised": "^7.1.1",
"eslint": "^8.39.0",
"eslint-config-standard-with-typescript": "^34.0.1",
"eslint-plugin-import": "^2.26.0",
"eslint-plugin-n": "^15.7.0",
"eslint-plugin-promise": "^6.1.1",
"mocha": "^10.2.0",
"openai": "^4.24.1",
"sinon": "^15.1.0",
"temp": "^0.9.4",
"ts-node": "^10.9.1",
"ts-node-dev": "^2.0.0",
"typedoc": "^0.24.7",
"typedoc-plugin-markdown": "^3.15.3",
"typescript": "^5.1.0",
"uuid": "^9.0.0"
},
"optionalDependencies": {
"@lancedb/vectordb-darwin-arm64": "0.21.2-beta.0",
"@lancedb/vectordb-darwin-x64": "0.21.2-beta.0",
"@lancedb/vectordb-linux-arm64-gnu": "0.21.2-beta.0",
"@lancedb/vectordb-linux-x64-gnu": "0.21.2-beta.0",
"@lancedb/vectordb-win32-x64-msvc": "0.21.2-beta.0"
},
"peerDependencies": {
"@apache-arrow/ts": "^14.0.2",
"apache-arrow": "^14.0.2"
}
},
"../node/node_modules/apache-arrow": {
"version": "14.0.2",
"license": "Apache-2.0",
"dependencies": {
"@types/command-line-args": "5.2.0",
"@types/command-line-usage": "5.0.2",
"@types/node": "20.3.0",
"@types/pad-left": "2.1.1",
"command-line-args": "5.2.1",
"command-line-usage": "7.0.1",
"flatbuffers": "23.5.26",
"json-bignum": "^0.0.3",
"pad-left": "^2.1.0",
"tslib": "^2.5.3"
},
"bin": {
"arrow2csv": "bin/arrow2csv.js"
}
},
"node_modules/@types/node": {
"version": "20.11.8",
"resolved": "https://registry.npmjs.org/@types/node/-/node-20.11.8.tgz",
"integrity": "sha512-i7omyekpPTNdv4Jb/Rgqg0RU8YqLcNsI12quKSDkRXNfx7Wxdm6HhK1awT3xTgEkgxPn3bvnSpiEAc7a7Lpyow==",
"dev": true,
"dependencies": {
"undici-types": "~5.26.4"
}
},
"node_modules/apache-arrow": {
"resolved": "../node/node_modules/apache-arrow",
"link": true
},
"node_modules/typescript": {
"version": "5.3.3",
"resolved": "https://registry.npmjs.org/typescript/-/typescript-5.3.3.tgz",
"integrity": "sha512-pXWcraxM0uxAS+tN0AG/BF2TyqmHO014Z070UsJ+pFvYuRSq8KH8DmWpnbXe0pEPDHXZV3FcAbJkijJ5oNEnWw==",
"dev": true,
"bin": {
"tsc": "bin/tsc",
"tsserver": "bin/tsserver"
},
"engines": {
"node": ">=14.17"
}
},
"node_modules/undici-types": {
"version": "5.26.5",
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-5.26.5.tgz",
"integrity": "sha512-JlCMO+ehdEIKqlFxk6IfVoAUVmgz7cU7zD/h9XZ0qzeosSHmUJVOzSQvvYSYWXkFXC+IfLKSIffhv0sVZup6pA==",
"dev": true
},
"node_modules/vectordb": {
"resolved": "../node",
"link": true
}
}
}
+20
View File
@@ -0,0 +1,20 @@
{
"name": "lancedb-docs-test",
"version": "1.0.0",
"description": "auto-generated tests from doc",
"author": "dev@lancedb.com",
"license": "Apache 2",
"dependencies": {
"apache-arrow": "file:../node/node_modules/apache-arrow",
"vectordb": "file:../node"
},
"scripts": {
"build": "tsc -b && cd ../node && npm run build-release",
"example": "npm run build && node",
"test": "npm run build && ls dist/*.js | xargs -n 1 node"
},
"devDependencies": {
"@types/node": "^20.11.8",
"typescript": "^5.3.3"
}
}
+1 -1
View File
@@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`:
<dependency> <dependency>
<groupId>com.lancedb</groupId> <groupId>com.lancedb</groupId>
<artifactId>lancedb-core</artifactId> <artifactId>lancedb-core</artifactId>
<version>0.39.0-beta.4</version> <version>0.38.0-beta.5</version>
</dependency> </dependency>
``` ```
-518
View File
@@ -1,518 +0,0 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / AutoQuery
# Class: AutoQuery
A builder for automatic string searches.
Automatic search determines whether to use full-text or vector search from
the table revision selected for each execution. This builder exposes the
common operations supported by both query families.
## Extends
- `StandardQueryBase`&lt;`NativeQuery` \| `NativeVectorQuery`&gt;
## Properties
### inner
```ts
protected inner: Query | VectorQuery | Promise<Query | VectorQuery>;
```
#### Inherited from
`StandardQueryBase.inner`
## Methods
### analyzePlan()
```ts
analyzePlan(distributedMetrics?): Promise<string>
```
Executes the query and returns the physical query plan annotated with runtime metrics.
This is useful for debugging and performance analysis, as it shows how the query was executed
and includes metrics such as elapsed time, rows processed, and I/O statistics.
#### Parameters
* **distributedMetrics?**: [`AnalyzePlanDistributedMetrics`](../type-aliases/AnalyzePlanDistributedMetrics.md)
How distributed worker metrics are displayed for remote query plans.
Defaults to `"aggregate"`.
#### Returns
`Promise`&lt;`string`&gt;
A query execution plan with runtime metrics for each step.
#### Example
```ts
import * as lancedb from "@lancedb/lancedb"
const db = await lancedb.connect("./.lancedb");
const table = await db.createTable("my_table", [
{ vector: [1.1, 0.9], id: "1" },
]);
const plan = await table.query().nearestTo([0.5, 0.2]).analyzePlan();
Example output (with runtime metrics inlined):
AnalyzeExec verbose=true, metrics=[]
ProjectionExec: expr=[id@3 as id, vector@0 as vector, _distance@2 as _distance], metrics=[output_rows=1, elapsed_compute=3.292µs]
Take: columns="vector, _rowid, _distance, (id)", metrics=[output_rows=1, elapsed_compute=66.001µs, batches_processed=1, bytes_read=8, iops=1, requests=1]
CoalesceBatchesExec: target_batch_size=1024, metrics=[output_rows=1, elapsed_compute=3.333µs]
GlobalLimitExec: skip=0, fetch=10, metrics=[output_rows=1, elapsed_compute=167ns]
FilterExec: _distance@2 IS NOT NULL, metrics=[output_rows=1, elapsed_compute=8.542µs]
SortExec: TopK(fetch=10), expr=[_distance@2 ASC NULLS LAST], metrics=[output_rows=1, elapsed_compute=63.25µs, row_replacements=1]
KNNVectorDistance: metric=l2, metrics=[output_rows=1, elapsed_compute=114.333µs, output_batches=1]
LanceScan: uri=/path/to/data, projection=[vector], row_id=true, row_addr=false, ordered=false, metrics=[output_rows=1, elapsed_compute=103.626µs, bytes_read=549, iops=2, requests=2]
```
#### Inherited from
`StandardQueryBase.analyzePlan`
***
### execute()
```ts
protected execute(options?): AsyncGenerator<RecordBatch<any>, void, unknown>
```
Execute the query and return the results as an
#### Parameters
* **options?**: `Partial`&lt;[`QueryExecutionOptions`](../interfaces/QueryExecutionOptions.md)&gt;
#### Returns
`AsyncGenerator`&lt;`RecordBatch`&lt;`any`&gt;, `void`, `unknown`&gt;
#### See
- AsyncIterator
of
- RecordBatch.
By default, LanceDb will use many threads to calculate results and, when
the result set is large, multiple batches will be processed at one time.
This readahead is limited however and backpressure will be applied if this
stream is consumed slowly (this constrains the maximum memory used by a
single query)
#### Inherited from
`StandardQueryBase.execute`
***
### explainPlan()
```ts
explainPlan(verbose): Promise<string>
```
Generates an explanation of the query execution plan.
#### Parameters
* **verbose**: `boolean` = `false`
If true, provides a more detailed explanation. Defaults to false.
#### Returns
`Promise`&lt;`string`&gt;
A Promise that resolves to a string containing the query execution plan explanation.
#### Example
```ts
import * as lancedb from "@lancedb/lancedb"
const db = await lancedb.connect("./.lancedb");
const table = await db.createTable("my_table", [
{ vector: [1.1, 0.9], id: "1" },
]);
const plan = await table.query().nearestTo([0.5, 0.2]).explainPlan();
```
#### Inherited from
`StandardQueryBase.explainPlan`
***
### fastSearch()
```ts
fastSearch(): this
```
Skip searching un-indexed data. This can make search faster, but will miss
any data that is not yet indexed.
Use [Table#optimize](Table.md#optimize) to index all un-indexed data.
#### Returns
`this`
#### Inherited from
`StandardQueryBase.fastSearch`
***
### ~~filter()~~
```ts
filter(predicate): this
```
A filter statement to be applied to this query.
#### Parameters
* **predicate**: `string`
#### Returns
`this`
#### See
where
#### Deprecated
Use `where` instead
#### Inherited from
`StandardQueryBase.filter`
***
### fullTextSearch()
```ts
fullTextSearch(query, options?): this
```
#### Parameters
* **query**: `string` \| [`FullTextQuery`](../interfaces/FullTextQuery.md)
* **options?**: `Partial`&lt;[`FullTextSearchOptions`](../interfaces/FullTextSearchOptions.md)&gt;
#### Returns
`this`
#### Inherited from
`StandardQueryBase.fullTextSearch`
***
### limit()
```ts
limit(limit): this
```
Set the maximum number of results to return.
By default, a plain search has no limit. If this method is not
called then every valid row from the table will be returned.
#### Parameters
* **limit**: `number`
#### Returns
`this`
#### Inherited from
`StandardQueryBase.limit`
***
### offset()
```ts
offset(offset): this
```
Set the number of rows to skip before returning results.
This is useful for pagination.
#### Parameters
* **offset**: `number`
#### Returns
`this`
#### Inherited from
`StandardQueryBase.offset`
***
### orderBy()
```ts
orderBy(ordering): this
```
Sort the results by the specified column(s).
#### Parameters
* **ordering**: [`ColumnOrdering`](../interfaces/ColumnOrdering.md) \| [`ColumnOrdering`](../interfaces/ColumnOrdering.md)[]
#### Returns
`this`
This query builder.
#### Inherited from
`StandardQueryBase.orderBy`
***
### outputSchema()
```ts
outputSchema(): Promise<Schema<any>>
```
Returns the schema of the output that will be returned by this query.
This can be used to inspect the types and names of the columns that will be
returned by the query before executing it.
#### Returns
`Promise`&lt;`Schema`&lt;`any`&gt;&gt;
An Arrow Schema describing the output columns.
#### Inherited from
`StandardQueryBase.outputSchema`
***
### select()
```ts
select(columns): this
```
Return only the specified columns.
By default a query will return all columns from the table. However, this can have
a very significant impact on latency. LanceDb stores data in a columnar fashion. This
means we can finely tune our I/O to select exactly the columns we need.
As a best practice you should always limit queries to the columns that you need. If you
pass in an array of column names then only those columns will be returned.
You can also use this method to create new "dynamic" columns based on your existing columns.
For example, you may not care about "a" or "b" but instead simply want "a + b". This is often
seen in the SELECT clause of an SQL query (e.g. `SELECT a+b FROM my_table`).
To create dynamic columns you can pass in a Map<string, string>. A column will be returned
for each entry in the map. The key provides the name of the column. The value is
an SQL string used to specify how the column is calculated.
For example, an SQL query might state `SELECT a + b AS combined, c`. The equivalent
input to this method would be:
#### Parameters
* **columns**: `string` \| `string`[] \| `Record`&lt;`string`, `string`&gt; \| `Map`&lt;`string`, `string`&gt;
#### Returns
`this`
#### Example
```ts
new Map([["combined", "a + b"], ["c", "c"]])
Columns will always be returned in the order given, even if that order is different than
the order used when adding the data.
Note that you can pass in a `Record<string, string>` (e.g. an object literal). This method
uses `Object.entries` which should preserve the insertion order of the object. However,
object insertion order is easy to get wrong and `Map` is more foolproof.
```
#### Inherited from
`StandardQueryBase.select`
***
### toArray()
```ts
toArray(options?): Promise<any[]>
```
Collect the results as an array of objects.
#### Parameters
* **options?**: `Partial`&lt;[`QueryExecutionOptions`](../interfaces/QueryExecutionOptions.md)&gt;
#### Returns
`Promise`&lt;`any`[]&gt;
#### Inherited from
`StandardQueryBase.toArray`
***
### toArrow()
```ts
toArrow(options?): Promise<Table<any>>
```
Collect the results as an Arrow
#### Parameters
* **options?**: `Partial`&lt;[`QueryExecutionOptions`](../interfaces/QueryExecutionOptions.md)&gt;
#### Returns
`Promise`&lt;`Table`&lt;`any`&gt;&gt;
#### See
ArrowTable.
#### Inherited from
`StandardQueryBase.toArrow`
***
### useLsm()
```ts
useLsm(enable): this
```
Control MemWAL read routing for this query.
By default (unset), when the table carries a MemWAL write spec (see
[Table#setLsmWriteSpec](Table.md#setlsmwritespec)), reads are routed through the LSM scanner so
they also return data written via the `mergeInsert` LSM path that has not yet
been compacted into the base table (the active/frozen in-memory memtables and
the flushed generations), deduplicated by primary key; a table without a spec
reads the base table.
#### Parameters
* **enable**: `boolean`
`true` forces the LSM scanner and errors if the table has no
MemWAL write spec. `false` bypasses the MemWAL and reads the base table only,
even when a spec is present.
Note: the LSM scanner does not support every query shape (e.g. reranking,
hybrid search, `orderBy`). On a MemWAL table those shapes error unless
`useLsm(false)` is set, because a base-only read would silently exclude
un-compacted MemWAL data.
#### Returns
`this`
#### Inherited from
`StandardQueryBase.useLsm`
***
### where()
```ts
where(predicate): this
```
A filter statement to be applied to this query.
The filter should be supplied as an SQL query string. For example:
#### Parameters
* **predicate**: `string`
#### Returns
`this`
#### Example
```ts
x > 10
y > 0 AND y < 100
x > 5 OR y = 'test'
Filtering performance can often be improved by creating a scalar index
on the filter column(s).
Calling this multiple times combines the filters with a logical AND rather
than replacing the previous filter.
```
#### Inherited from
`StandardQueryBase.where`
***
### withRowId()
```ts
withRowId(): this
```
Whether to return the row id in the results.
This column can be used to match results between different queries. For
example, to match results from a full text search and a vector search in
order to perform hybrid search.
#### Returns
`this`
#### Inherited from
`StandardQueryBase.withRowId`
+63 -97
View File
@@ -448,6 +448,26 @@ on the returned job to know when cleanup has finished.
*** ***
### getJob()
```ts
abstract getJob(jobId): Promise<null | JobDescription>
```
Describe a single server-side job by id.
Resolves to `null` when the server has no such job.
#### Parameters
* **jobId**: `string`
#### Returns
`Promise`&lt;`null` \| [`JobDescription`](../interfaces/JobDescription.md)&gt;
***
### isOpen() ### isOpen()
```ts ```ts
@@ -462,6 +482,48 @@ Return true if the connection has not been closed
*** ***
### job()
```ts
abstract job(jobId): Job
```
A [Job](Job.md) handle for a server-side job by id.
The handle is constructed without a server round trip; an unknown id
surfaces when the handle is used. Dropping the handle has no effect on
the job itself.
#### Parameters
* **jobId**: `string`
#### Returns
[`Job`](Job.md)
***
### jobHistory()
```ts
abstract jobHistory(jobId?): Promise<Table<any>>
```
The lifecycle event history of a server-side job, as an Arrow table.
Lists history across all jobs when `jobId` is omitted.
#### Parameters
* **jobId?**: `string`
#### Returns
`Promise`&lt;`Table`&lt;`any`&gt;&gt;
***
### listJobs() ### listJobs()
```ts ```ts
@@ -522,94 +584,6 @@ Child namespace names and
*** ***
### listTables()
#### listTables(options)
```ts
abstract listTables(options?): Promise<ListTablesResponse>
```
List a page of the tables in this database.
To retrieve the tables after the page, pass the `pageToken` the response
carries back in. A page can be shorter than `limit` without being the last
one, so walk until a response carries no page token:
```ts
const names = [];
let pageToken = undefined;
do {
const page = await conn.listTables({ pageToken, limit: 100 });
names.push(...page.tables);
pageToken = page.pageToken;
} while (pageToken);
```
##### Parameters
* **options?**: `Partial`&lt;[`ListTablesOptions`](../interfaces/ListTablesOptions.md)&gt;
Pagination options
(`pageToken`, `limit`).
##### Returns
`Promise`&lt;[`ListTablesResponse`](../interfaces/ListTablesResponse.md)&gt;
A page of table names and an
optional token for the tables after it.
#### listTables(namespacePath, options)
```ts
abstract listTables(namespacePath?, options?): Promise<ListTablesResponse>
```
List a page of the tables in this database.
##### Parameters
* **namespacePath?**: `string`[]
The namespace path to list tables from
(defaults to root namespace)
* **options?**: `Partial`&lt;[`ListTablesOptions`](../interfaces/ListTablesOptions.md)&gt;
Pagination options
(`pageToken`, `limit`).
##### Returns
`Promise`&lt;[`ListTablesResponse`](../interfaces/ListTablesResponse.md)&gt;
A page of table names and an
optional token for the tables after it.
***
### openJob()
```ts
abstract openJob(jobId): Promise<Job>
```
Open a server-side job by id, returning a handle with its record already
populated. Rejects when the server has no such job, the way
[Connection.openTable](Connection.md#opentable) does for a missing table.
The returned [Job](Job.md) answers for its own state, specification,
result, failure and event history, so there is no separate
connection-level call for any of them.
#### Parameters
* **jobId**: `string`
#### Returns
`Promise`&lt;[`Job`](Job.md)&gt;
***
### openMaterializedView() ### openMaterializedView()
```ts ```ts
@@ -686,7 +660,7 @@ a "not supported" error.
*** ***
### ~~tableNames()~~ ### tableNames()
#### tableNames(options) #### tableNames(options)
@@ -708,10 +682,6 @@ Tables will be returned in lexicographical order.
`Promise`&lt;`string`[]&gt; `Promise`&lt;`string`[]&gt;
##### Deprecated
Use [Connection.listTables](Connection.md#listtables) instead.
#### tableNames(namespacePath, options) #### tableNames(namespacePath, options)
```ts ```ts
@@ -734,7 +704,3 @@ Tables will be returned in lexicographical order.
##### Returns ##### Returns
`Promise`&lt;`string`[]&gt; `Promise`&lt;`string`[]&gt;
##### Deprecated
Use [Connection.listTables](Connection.md#listtables) instead.
+16 -163
View File
@@ -8,116 +8,28 @@
A handle to an operation that may still be running. A handle to an operation that may still be running.
The operation may already be complete when the handle is created. ## Constructors
The detail getters read what the handle last observed. Submitting an ### new Job()
operation returns only a job id, so populating them eagerly would cost an
extra round trip on every call:
- [Job.refresh](Job.md#refresh) and [Job.status](Job.md#status) fetch the whole record. ```ts
- [Job.wait](Job.md#wait) records the terminal state it establishes, but not the new Job(): Job
rest of the record. ```
- Everything is null until one of those runs.
#### Returns
[`Job`](Job.md)
## Accessors ## Accessors
### creationMs
```ts
get creationMs(): null | number
```
When the job was created, in milliseconds since the epoch.
#### Returns
`null` \| `number`
***
### failure
```ts
get failure(): null | JobFailureInfo
```
Why the job failed, when it failed and the server reports a reason.
#### Returns
`null` \| [`JobFailureInfo`](../interfaces/JobFailureInfo.md)
***
### id ### id
```ts ```ts
get id(): null | string get id(): null | string
``` ```
Identifies the operation on the server that is running it. Identifies the operation on the server that is running it. Operations
that run in this process have no server id. The value is opaque.
Operations that run in this process have no server id. The value is
opaque: parsing it or storing it to resume the job later is not supported.
#### Returns
`null` \| `string`
***
### jobType
```ts
get jobType(): null | string
```
The job's type, as the server names it. Null for an in-process job, which
has no server-side record.
#### Returns
`null` \| `string`
***
### result
```ts
get result(): any
```
The job-type-specific terminal result. Null until the job succeeds, so a
job that never terminates reports its progress through [Job.events](Job.md#events)
instead.
#### Returns
`any`
***
### spec
```ts
get spec(): any
```
The job-type-specific specification it was submitted with.
#### Returns
`any`
***
### state
```ts
get state(): null | string
```
The last observed lifecycle state, without contacting the backend.
#### Returns #### Returns
@@ -139,61 +51,18 @@ Request cancellation. Cancelling a finished operation is a no-op.
*** ***
### events()
```ts
events(options?): Promise<Table<any>>
```
This job's recorded lifecycle events.
Where the getters above report a terminal result only once the job reaches
one, events are written as the job runs and outlive the workers that
produced them. A distributed job records a `claim`/`claim_complete` pair
per unit of work, each carrying `rows_processed`, so a job that never
finishes still accounts for what it did.
The server caps results at 1000 rows by default and 10,000 at most, and
truncates without saying so, so pass `limit` for a job that emits an event
per fragment. `filter` is a SQL-like expression over the `state`,
`updated_by`, `emitted_from`, `emitted_by`, and `claim_entity` columns.
#### Parameters
* **options?**: [`JobEventsOptions`](../interfaces/JobEventsOptions.md)
#### Returns
`Promise`&lt;`Table`&lt;`any`&gt;&gt;
***
### refresh()
```ts
refresh(): Promise<void>
```
Ask the backend for this job's current state, and for a server-side job
its full record, then cache it for the getters above.
#### Returns
`Promise`&lt;`void`&gt;
***
### status() ### status()
```ts ```ts
status(): Promise<string> status(): Promise<string>
``` ```
The operation's current lifecycle state: "running", "finished", "failed", The operation's current lifecycle state: "running", "finished",
or "cancelled". "failed", or "cancelled".
A point snapshot; unlike [Job.wait](Job.md#wait) it does not block or reject on a A point snapshot; unlike [Job.wait](Job.md#wait) it does not block or reject
terminal failure state. Also refreshes the getters above. on a terminal failure state. States a newer server reports that this
client version does not know pass through as-is.
#### Returns #### Returns
@@ -201,22 +70,6 @@ terminal failure state. Also refreshes the getters above.
*** ***
### toString()
```ts
toString(): string
```
Every field the handle currently knows, one per line, with the JSON
payloads indented -- a refresh job's spec and result are the point of
printing it.
#### Returns
`string`
***
### wait() ### wait()
```ts ```ts
+2 -22
View File
@@ -676,17 +676,9 @@ List all the versions of the table
abstract mergeInsert(on): MergeInsertBuilder abstract mergeInsert(on): MergeInsertBuilder
``` ```
Create a [MergeInsertBuilder](MergeInsertBuilder.md), which combines new data with the
existing table in a single transaction — inserting, updating and deleting
rows depending on how they match.
#### Parameters #### Parameters
* **on**: `string` \| `string`[] * **on**: `string` \| `string`[]
The column, or columns, to match source rows against target
rows on. Typically a key or id column. Several columns match on the
composite key: a source row updates a target row only when it agrees on
every one of them.
#### Returns #### Returns
@@ -950,7 +942,7 @@ Get the schema of the table.
abstract search( abstract search(
query, query,
queryType?, queryType?,
ftsColumns?): Query | VectorQuery | AutoQuery ftsColumns?): Query | VectorQuery
``` ```
Create a search query to find the nearest neighbors Create a search query to find the nearest neighbors
@@ -972,7 +964,7 @@ of the given query
#### Returns #### Returns
[`Query`](Query.md) \| [`VectorQuery`](VectorQuery.md) \| [`AutoQuery`](AutoQuery.md) [`Query`](Query.md) \| [`VectorQuery`](VectorQuery.md)
*** ***
@@ -1300,18 +1292,6 @@ abstract updateFieldMetadata(updates): Promise<UpdateFieldMetadataResult>
Update per-field (column) metadata. Update per-field (column) metadata.
The following keys are treated specially, by convention, and should be
used when appropriate:
- `lancedb:description`: for a human-readable description of a field.
- `lancedb:tag:<name>`: for a user-defined key-value tag, where the suffix
names the tag category; e.g. `lancedb:tag:model: "clip"`.
- `lancedb:logical-column`: for a column grouping; e.g. `feature_v1` and
`feature_v2` might be in the same logical column.
- `lancedb:status`: for status options (`production`, `candidate`,
`deprecated`, `archived`) to designate the current life cycle state of
this column.
#### Parameters #### Parameters
* **updates**: [`FieldMetadataUpdate`](../interfaces/FieldMetadataUpdate.md)[] * **updates**: [`FieldMetadataUpdate`](../interfaces/FieldMetadataUpdate.md)[]
+1 -4
View File
@@ -18,7 +18,6 @@
## Classes ## Classes
- [AutoQuery](classes/AutoQuery.md)
- [BooleanQuery](classes/BooleanQuery.md) - [BooleanQuery](classes/BooleanQuery.md)
- [BoostQuery](classes/BoostQuery.md) - [BoostQuery](classes/BoostQuery.md)
- [BranchContents](classes/BranchContents.md) - [BranchContents](classes/BranchContents.md)
@@ -96,13 +95,11 @@
- [IvfFlatOptions](interfaces/IvfFlatOptions.md) - [IvfFlatOptions](interfaces/IvfFlatOptions.md)
- [IvfPqOptions](interfaces/IvfPqOptions.md) - [IvfPqOptions](interfaces/IvfPqOptions.md)
- [IvfRqOptions](interfaces/IvfRqOptions.md) - [IvfRqOptions](interfaces/IvfRqOptions.md)
- [JobEventsOptions](interfaces/JobEventsOptions.md) - [JobDescription](interfaces/JobDescription.md)
- [JobFailureInfo](interfaces/JobFailureInfo.md) - [JobFailureInfo](interfaces/JobFailureInfo.md)
- [JobInfo](interfaces/JobInfo.md) - [JobInfo](interfaces/JobInfo.md)
- [ListNamespacesOptions](interfaces/ListNamespacesOptions.md) - [ListNamespacesOptions](interfaces/ListNamespacesOptions.md)
- [ListNamespacesResponse](interfaces/ListNamespacesResponse.md) - [ListNamespacesResponse](interfaces/ListNamespacesResponse.md)
- [ListTablesOptions](interfaces/ListTablesOptions.md)
- [ListTablesResponse](interfaces/ListTablesResponse.md)
- [LsmStats](interfaces/LsmStats.md) - [LsmStats](interfaces/LsmStats.md)
- [LsmWriteSpec](interfaces/LsmWriteSpec.md) - [LsmWriteSpec](interfaces/LsmWriteSpec.md)
- [MaterializedViewDefinition](interfaces/MaterializedViewDefinition.md) - [MaterializedViewDefinition](interfaces/MaterializedViewDefinition.md)
@@ -17,8 +17,7 @@ metadata: Record<string, null | string>;
``` ```
Metadata key/value pairs. Merged into the field's existing metadata by Metadata key/value pairs. Merged into the field's existing metadata by
default; a value of `null` deletes that key. See default; a value of `null` deletes that key.
[Table.updateFieldMetadata](../classes/Table.md#updatefieldmetadata) for the conventional `lancedb:*` keys.
*** ***
+66
View File
@@ -0,0 +1,66 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / JobDescription
# Interface: JobDescription
A described job from `Connection.getJob`.
## Properties
### creationMs
```ts
creationMs: number;
```
When the job was created, in milliseconds since the epoch.
***
### failure?
```ts
optional failure: JobFailureInfo;
```
Why the job failed, when the job is failed and the server reports a
reason.
***
### jobId
```ts
jobId: string;
```
***
### jobType
```ts
jobType: string;
```
***
### specJson?
```ts
optional specJson: string;
```
The job-type-specific specification as a JSON string, when present.
***
### state
```ts
state: string;
```
Lifecycle state: "running", "finished", "failed", or "cancelled".
@@ -1,29 +0,0 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / JobEventsOptions
# Interface: JobEventsOptions
Which of a job's events [Job.events](../classes/Job.md#events) returns.
## Properties
### filter?
```ts
optional filter: string;
```
SQL-like filter over the event columns.
***
### limit?
```ts
optional limit: number;
```
Maximum event rows to return, up to the server maximum of 10,000.
+1 -1
View File
@@ -26,7 +26,7 @@ When the job was created, in milliseconds since the epoch.
jobId: string; jobId: string;
``` ```
The job id -- what `Connection.openJob` and `Connection.cancelJob` The job id -- what `Connection.getJob` and `Connection.cancelJob`
accept. accept.
*** ***
@@ -1,34 +0,0 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / ListTablesOptions
# Interface: ListTablesOptions
## Properties
### limit?
```ts
optional limit: number;
```
An upper bound on how many tables to return.
A page may hold fewer than this and still not be the last one, so keep
going while the response carries a page token rather than while pages are
full.
***
### pageToken?
```ts
optional pageToken: string;
```
Token from a previous response, to resume listing where it left off.
The token is opaque: it carries whatever the database needs to resume, and
callers should not construct or interpret one.
@@ -1,23 +0,0 @@
[**@lancedb/lancedb**](../README.md) • **Docs**
***
[@lancedb/lancedb](../globals.md) / ListTablesResponse
# Interface: ListTablesResponse
## Properties
### pageToken?
```ts
optional pageToken: string;
```
***
### tables
```ts
tables: string[];
```
@@ -50,16 +50,6 @@ projections: [string, string][];
*** ***
### sourceNamespace
```ts
sourceNamespace: string[];
```
Namespace holding the source table; empty is the root namespace.
***
### sourceTable ### sourceTable
```ts ```ts
+3 -8
View File
@@ -4,16 +4,11 @@
[@lancedb/lancedb](../globals.md) / TableNamesOptions [@lancedb/lancedb](../globals.md) / TableNamesOptions
# Interface: ~~TableNamesOptions~~ # Interface: TableNamesOptions
## Deprecated
Use [ListTablesOptions](ListTablesOptions.md) with [Connection.listTables](../classes/Connection.md#listtables)
instead.
## Properties ## Properties
### ~~limit?~~ ### limit?
```ts ```ts
optional limit: number; optional limit: number;
@@ -23,7 +18,7 @@ An optional limit to the number of results to return.
*** ***
### ~~startAfter?~~ ### startAfter?
```ts ```ts
optional startAfter: string; optional startAfter: string;
@@ -10,12 +10,16 @@
function getRegistry(): EmbeddingFunctionRegistry function getRegistry(): EmbeddingFunctionRegistry
``` ```
Get the global embedding function registry. Utility function to get the global instance of the registry
LanceDB built-in providers are initialized when this public API is first
used, so importing the root package does not change automatic search
selection for tables without embedding metadata.
## Returns ## Returns
[`EmbeddingFunctionRegistry`](../classes/EmbeddingFunctionRegistry.md) [`EmbeddingFunctionRegistry`](../classes/EmbeddingFunctionRegistry.md)
`EmbeddingFunctionRegistry` The global instance of the registry
## Example
```ts
const registry = getRegistry();
const openai = registry.get("openai").create();
+2 -87
View File
@@ -28,59 +28,6 @@ is also an [asynchronous API client](#connections-asynchronous).
::: lancedb.Session ::: lancedb.Session
## Remote SQL
Submit SQL against a remote LanceDB database through the connection.
The connected database and `default_namespace_path=["public"]` are used for
unqualified tables. Fully qualified references can still query other databases
and namespaces available to the same deployment. `execute_query` returns a
reader as soon as its initial result stream is available. `execute_query_async`
returns a query handle immediately; use it to inspect progress, open a reader,
or cancel the query. The SQL client is initialized by the first query and
retained for the lifetime of the remote connection. Query ids are random,
connection-scoped references rather than encoded SQL or durable resume tokens:
```python
import lancedb
db = lancedb.connect(
"db://analytics",
api_key="ldb_...",
host_override="https://api.example.com",
sql_host_override="grpc+tls://sql.example.com:10026",
)
reader = db.execute_query(
"""
SELECT events.id, accounts.name
FROM analytics.public.events AS events
JOIN users.public.accounts AS accounts ON events.user_id = accounts.id
""",
default_namespace_path=["public"],
)
for batch in reader:
print(batch.num_rows)
query = db.execute_query_async("SELECT * FROM events")
print(query.id)
print(query.describe().status)
for batch in query.reader():
print(batch.num_rows)
# The async connection exposes the same lifecycle without blocking:
# async_db = await lancedb.connect_async(
# "db://analytics",
# api_key="ldb_...",
# host_override="https://api.example.com",
# sql_host_override="grpc+tls://sql.example.com:10026",
# )
# reader = await async_db.execute_query("SELECT * FROM events")
# query = await async_db.execute_query_async("SELECT * FROM events")
# description = await async_db.describe_query(query.id)
# async for batch in await query.reader():
# print(batch.num_rows)
# await query.cancel()
```
## Namespaces (Synchronous) ## Namespaces (Synchronous)
A namespace-backed connection resolves tables through a A namespace-backed connection resolves tables through a
@@ -125,10 +72,6 @@ listing a storage directory.
::: lancedb.functions.UdfDefinition ::: lancedb.functions.UdfDefinition
::: lancedb.secrets.EnvVarSecret
::: lancedb.secrets.SecretInfo
::: lancedb.functions.FunctionRegistrationRequest ::: lancedb.functions.FunctionRegistrationRequest
::: lancedb.functions.FunctionArtifactRequest ::: lancedb.functions.FunctionArtifactRequest
@@ -151,8 +94,6 @@ listing a storage directory.
::: lancedb.functions.OutputMapping ::: lancedb.functions.OutputMapping
::: lancedb.functions.AssignmentMapping
::: lancedb.functions.FunctionBinding ::: lancedb.functions.FunctionBinding
::: lancedb.functions.RefreshColumnResult ::: lancedb.functions.RefreshColumnResult
@@ -161,18 +102,6 @@ listing a storage directory.
::: lancedb.job.AsyncJob ::: lancedb.job.AsyncJob
::: lancedb.job.JobInfo
::: lancedb.job.JobDescription
::: lancedb.job.JobFailureInfo
::: lancedb.sql.Query
::: lancedb.sql.AsyncQuery
::: lancedb.sql.QueryDescription
## Materialized Views (Synchronous) ## Materialized Views (Synchronous)
::: lancedb.materialized_view.MaterializedView ::: lancedb.materialized_view.MaterializedView
@@ -230,8 +159,6 @@ and combined with [BooleanQuery][lancedb.query.BooleanQuery].
::: lancedb.query.FullTextOperator ::: lancedb.query.FullTextOperator
::: lancedb.query.DocumentGranularity
::: lancedb.query.Occur ::: lancedb.query.Occur
## Embeddings ## Embeddings
@@ -294,14 +221,10 @@ tokens = list(
Blob columns store large binary values out of line so they can be read lazily Blob columns store large binary values out of line so they can be read lazily
instead of being materialized with the rest of the row. instead of being materialized with the rest of the row.
`lancedb.BlobType` is `lance.blob.BlobType` when pylance is installed. Without
pylance, LanceDB uses a matching `lance.blob.v2` extension type so blob columns
still work. Queries return descriptors. Call
[`fetch_blob_files`][lancedb.table.Table.fetch_blob_files] for lazy reads or
[`fetch_blobs`][lancedb.table.Table.fetch_blobs] for eager bytes.
::: lancedb.blob ::: lancedb.blob
::: lancedb.BlobType
::: lancedb._blob.BlobFile ::: lancedb._blob.BlobFile
options: options:
show_root_full_path: false show_root_full_path: false
@@ -320,12 +243,6 @@ still work. Queries return descriptors. Call
::: lancedb.exceptions.MissingColumnError ::: lancedb.exceptions.MissingColumnError
::: lancedb.exceptions.JobNotFoundError
::: lancedb.exceptions.JobFailedError
::: lancedb.exceptions.JobCancelledError
## Integrations ## Integrations
## Pydantic ## Pydantic
@@ -344,8 +261,6 @@ still work. Queries return descriptors. Call
::: lancedb.streaming.StreamingDataset ::: lancedb.streaming.StreamingDataset
::: lancedb.streaming.StreamingDataLoader
::: lancedb.permutation.permutation_builder ::: lancedb.permutation.permutation_builder
::: lancedb.permutation.PermutationBuilder ::: lancedb.permutation.PermutationBuilder
+17
View File
@@ -0,0 +1,17 @@
{
"include": [
"src/*.ts",
],
"compilerOptions": {
"target": "es2022",
"module": "nodenext",
"declaration": true,
"outDir": "./dist",
"strict": true,
"allowJs": true,
"resolveJsonModule": true,
},
"exclude": [
"./dist/*",
]
}
+1 -1
View File
@@ -8,7 +8,7 @@
<parent> <parent>
<groupId>com.lancedb</groupId> <groupId>com.lancedb</groupId>
<artifactId>lancedb-parent</artifactId> <artifactId>lancedb-parent</artifactId>
<version>0.39.0-beta.4</version> <version>0.38.0-beta.5</version>
<relativePath>../pom.xml</relativePath> <relativePath>../pom.xml</relativePath>
</parent> </parent>
+2 -2
View File
@@ -6,7 +6,7 @@
<groupId>com.lancedb</groupId> <groupId>com.lancedb</groupId>
<artifactId>lancedb-parent</artifactId> <artifactId>lancedb-parent</artifactId>
<version>0.39.0-beta.4</version> <version>0.38.0-beta.5</version>
<packaging>pom</packaging> <packaging>pom</packaging>
<name>${project.artifactId}</name> <name>${project.artifactId}</name>
<description>LanceDB Java SDK Parent POM</description> <description>LanceDB Java SDK Parent POM</description>
@@ -28,7 +28,7 @@
<properties> <properties>
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding> <project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
<arrow.version>15.0.0</arrow.version> <arrow.version>15.0.0</arrow.version>
<lance-core.version>12.0.0-beta.14</lance-core.version> <lance-core.version>11.0.0-beta.21</lance-core.version>
<spotless.skip>false</spotless.skip> <spotless.skip>false</spotless.skip>
<spotless.version>2.30.0</spotless.version> <spotless.version>2.30.0</spotless.version>
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version> <spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
+1 -1
View File
@@ -1,7 +1,7 @@
[package] [package]
name = "lancedb-nodejs" name = "lancedb-nodejs"
edition.workspace = true edition.workspace = true
version = "0.39.0-beta.4" version = "0.38.0-beta.5"
publish = false publish = false
license.workspace = true license.workspace = true
description.workspace = true description.workspace = true
-189
View File
@@ -1,16 +1,11 @@
// SPDX-License-Identifier: Apache-2.0 // SPDX-License-Identifier: Apache-2.0
// SPDX-FileCopyrightText: Copyright The LanceDB Authors // SPDX-FileCopyrightText: Copyright The LanceDB Authors
import * as fs from "node:fs";
import * as vm from "node:vm";
import * as arrow15 from "apache-arrow-15"; import * as arrow15 from "apache-arrow-15";
import * as arrow16 from "apache-arrow-16"; import * as arrow16 from "apache-arrow-16";
import * as arrow17 from "apache-arrow-17"; import * as arrow17 from "apache-arrow-17";
import * as arrow18 from "apache-arrow-18"; import * as arrow18 from "apache-arrow-18";
import { import {
Field as CurrentField,
LargeBinary as CurrentLargeBinary,
Schema as CurrentSchema,
Vector as CurrentVector, Vector as CurrentVector,
convertToTable, convertToTable,
tableFromIPC as currentTableFromIPC, tableFromIPC as currentTableFromIPC,
@@ -41,59 +36,6 @@ function sampleRecords(): Array<Record<string, any>> {
}, },
]; ];
} }
it("serializes an Arrow Table created in another JavaScript realm", async () => {
const context = vm.createContext({
TextDecoder,
TextEncoder,
console,
setTimeout,
clearTimeout,
});
vm.runInContext(
fs.readFileSync(
require.resolve("apache-arrow-15/Arrow.es2015.min"),
"utf8",
),
context,
);
const foreignTable: unknown = vm.runInContext(
"Arrow.tableFromArrays({ id: new Int32Array([1, 2, 3]), text: ['foo', 'bar', 'baz'] })",
context,
);
const foreignMetadata = (
foreignTable as { schema: { metadata: Map<string, string> } }
).schema.metadata;
expect(foreignMetadata).not.toBeInstanceOf(Map);
const buf = await fromDataToBuffer(
foreignTable as Parameters<typeof fromDataToBuffer>[0],
);
const actual = currentTableFromIPC(buf);
expect(actual.numRows).toBe(3);
expect(actual.getChild("id")?.toJSON()).toEqual([1, 2, 3]);
expect(actual.getChild("text")?.toJSON()).toEqual(["foo", "bar", "baz"]);
});
it("preserves field metadata from a provided schema", async function () {
const jsonMetadata = new Map([["ARROW:extension:name", "lance.json"]]);
const schema = new CurrentSchema([
new CurrentField("meta", new CurrentLargeBinary(), true, jsonMetadata),
]);
const table = makeArrowTable(
[{ meta: Buffer.from(JSON.stringify({ source: "test" })) }],
{ schema },
);
expect(table.schema.fields[0].metadata).toEqual(jsonMetadata);
const roundTripped = currentTableFromIPC(await fromTableToBuffer(table));
expect(roundTripped.schema.fields[0].metadata).toEqual(jsonMetadata);
});
describe.each([arrow15, arrow16, arrow17, arrow18])( describe.each([arrow15, arrow16, arrow17, arrow18])(
"Arrow", "Arrow",
( (
@@ -573,137 +515,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
); );
}); });
it("will allow matching inferred types across records", function () {
expect(() =>
makeArrowTable([{ value: 1 }, { value: 2 }]),
).not.toThrow();
});
it("will reject mismatched inferred types across records", function () {
expect(() => makeArrowTable([{ value: 1 }, { value: "two" }])).toThrow(
"Failed to infer schema for data. Previously inferred type Float64 but found Utf8 for field value at row 1. Consider providing an explicit schema.",
);
});
it("will ignore generated dictionary IDs when comparing inferred types", function () {
const table = makeArrowTable([{ str: "a" }, { str: "b" }], {
dictionaryEncodeStrings: true,
});
expect(table.getChild("str")?.toJSON()).toEqual(["a", "b"]);
});
it("will preserve null values without treating them as type mismatches", function () {
for (const records of [
[{ vector: [1, 2, 3] }, { vector: null }],
[{ vector: null }, { vector: [1, 2, 3] }],
]) {
const table = makeArrowTable(records);
expect(table.numRows).toBe(2);
expect(table.getChild("vector")?.nullCount).toBe(1);
}
});
it("will preserve empty variable-size lists", function () {
for (const records of [
[{ items: [1] }, { items: [] }],
[{ items: [] }, { items: [1] }],
]) {
const table = makeArrowTable(records);
expect(
table
.getChild("items")
?.toJSON()
.map((value) => value.toJSON()),
).toEqual(records.map((record) => record.items));
}
});
it("will propagate deferred evidence through nested lists", function () {
for (const records of [
[{ items: [1] }, { items: [null] }],
[{ items: [null] }, { items: [1] }],
[{ items: [null, 1] }, { items: [2, null] }],
]) {
const table = makeArrowTable(records);
expect(
table
.getChild("items")
?.toJSON()
.map((value) => value.toJSON()),
).toEqual(records.map((record) => record.items));
}
const nestedRecords = [{ items: [[1]] }, { items: [[null]] }];
const nestedTable = makeArrowTable(nestedRecords);
expect(
nestedTable
.getChild("items")
?.toJSON()
.map((value) =>
value
.toJSON()
.map((nestedValue: { toJSON: () => unknown[] }) =>
nestedValue.toJSON(),
),
),
).toEqual(nestedRecords.map((record) => record.items));
});
it("will reject incompatible deferred evidence within a list", function () {
for (const items of [
[[], 1],
[1, []],
[[null], 1],
[1, [null]],
]) {
expect(() => makeArrowTable([{ items }])).toThrow(
"Failed to infer data type for field items at row 0.",
);
}
});
it("will reject empty fixed-size lists", function () {
expect(() =>
makeArrowTable([{ vector: [1, 2, 3] }, { vector: [] }]),
).toThrow(
"Failed to infer schema for data. Previously inferred type FixedSizeList[3]<Float32> but found List[0] for field vector at row 1.",
);
});
it("will reject inferred leaf and branch shape changes", function () {
expect(() =>
makeArrowTable([{ value: 1 }, { value: { nested: 2 } }]),
).toThrow(
"Failed to infer schema for data. Previously inferred type Float64 but found Struct for field value at row 1.",
);
expect(() =>
makeArrowTable([{ value: { nested: 1 } }, { value: 2 }]),
).toThrow(
"Failed to infer schema for data. Previously inferred type Struct but found Float64 for field value at row 1.",
);
});
it("will allow null values around inferred struct values", function () {
for (const { records, nullIndex } of [
{
records: [{ value: null }, { value: { nested: 2 } }],
nullIndex: 0,
},
{
records: [{ value: { nested: 1 } }, { value: null }],
nullIndex: 1,
},
]) {
const table = makeArrowTable(records);
const values = table.getChild("value");
expect(values?.nullCount).toBe(1);
expect(values?.get(nullIndex)).toBeNull();
}
});
it("will allow a schema to be provided", async function () { it("will allow a schema to be provided", async function () {
await checkTableCreation( await checkTableCreation(
async (records, _, schema) => async (records, _, schema) =>
+1 -68
View File
@@ -4,13 +4,7 @@
import { readdirSync } from "fs"; import { readdirSync } from "fs";
import { Field, Float64, Schema } from "apache-arrow"; import { Field, Float64, Schema } from "apache-arrow";
import * as tmp from "tmp"; import * as tmp from "tmp";
import { import { Connection, Table, connect, connectNamespace } from "../lancedb";
Connection,
ListTablesResponse,
Table,
connect,
connectNamespace,
} from "../lancedb";
import { LocalTable } from "../lancedb/table"; import { LocalTable } from "../lancedb/table";
describe("when connecting", () => { describe("when connecting", () => {
@@ -53,7 +47,6 @@ describe("given a connection", () => {
await db.close(); await db.close();
expect(db.isOpen()).toBe(false); expect(db.isOpen()).toBe(false);
await expect(db.tableNames()).rejects.toThrow("Connection is closed"); await expect(db.tableNames()).rejects.toThrow("Connection is closed");
await expect(db.listTables()).rejects.toThrow("Connection is closed");
await expect(db.renameTable("a", "b")).rejects.toThrow( await expect(db.renameTable("a", "b")).rejects.toThrow(
"Connection is closed", "Connection is closed",
); );
@@ -136,66 +129,6 @@ describe("given a connection", () => {
expect(tables).toEqual(["b", "c"]); expect(tables).toEqual(["b", "c"]);
}); });
it("should respect limit and page token when listing tables", async () => {
const db = await connect(tmpDir.name);
await db.createTable("b", [{ id: 1 }]);
await db.createTable("a", [{ id: 1 }]);
await db.createTable("c", [{ id: 1 }]);
const all = await db.listTables();
expect(all.tables).toEqual(["a", "b", "c"]);
expect(all.pageToken).toBeUndefined();
const first = await db.listTables({ limit: 1 });
expect(first.tables).toEqual(["a"]);
expect(first.pageToken).toBeDefined();
const second = await db.listTables({
limit: 1,
pageToken: first.pageToken,
});
expect(second.tables).toEqual(["b"]);
});
it("should visit every table exactly once when walking pages", async () => {
const db = await connect(tmpDir.name);
const created = ["a", "b", "c", "d", "e"];
for (const name of created) {
await db.createTable(name, [{ id: 1 }]);
}
const seen: string[] = [];
let pageToken: string | undefined = undefined;
do {
const page: ListTablesResponse = await db.listTables({
limit: 2,
pageToken,
});
seen.push(...page.tables);
pageToken = page.pageToken;
} while (pageToken);
expect(seen).toEqual(created);
});
it("should list tables in a namespace", async () => {
const db = await connect(tmpDir.name, {
// biome-ignore lint/style/useNamingConvention: opaque backend property key, must match Rust
namespaceClientProperties: { manifest_enabled: "true" },
});
await db.createNamespace(["child"]);
await db.createTable("nested", [{ id: 1 }], ["child"]);
await expect(db.listTables(["child"])).resolves.toEqual(
expect.objectContaining({ tables: ["nested"] }),
);
await expect(db.listTables()).resolves.toEqual(
expect.objectContaining({ tables: [] }),
);
});
it("should create tables in v2 mode", async () => { it("should create tables in v2 mode", async () => {
const db = await connect(tmpDir.name); const db = await connect(tmpDir.name);
const data = [...Array(10000).keys()].map((i) => ({ id: i })); const data = [...Array(10000).keys()].map((i) => ({ id: i }));
-52
View File
@@ -187,58 +187,6 @@ describe("embedding functions", () => {
const vector0 = JSON.parse(JSON.stringify(arr[0].vector)); const vector0 = JSON.parse(JSON.stringify(arr[0].vector));
expect(vector0).toEqual([1, 2, 3]); expect(vector0).toEqual([1, 2, 3]);
}); });
it("should append multiple Python embeddings with the same alias", async () => {
@register("python-mock")
// biome-ignore lint/correctness/noUnusedVariables: the decorator registers this class
class MockEmbeddingFunction extends EmbeddingFunction<string> {
ndims() {
return 3;
}
embeddingDataType(): Float {
return new Float32();
}
async computeQueryEmbeddings(_data: string) {
return [1, 2, 3];
}
async computeSourceEmbeddings(data: string[]) {
return data.map((value) =>
value === "hello world" ? [1, 2, 3] : [4, 5, 6],
);
}
}
const metadata = new Map([
[
"embedding_functions",
'[{"source_column":"text1","vector_column":"vector1","name":"python-mock","model":{}},{"source_column":"text2","vector_column":"vector2","name":"python-mock","model":{}}]',
],
]);
const schema = new Schema(
[
new Field("text1", new Utf8(), true),
new Field("text2", new Utf8(), true),
new Field(
"vector1",
new FixedSizeList(3, new Field("item", new Float32(), true)),
true,
),
new Field(
"vector2",
new FixedSizeList(3, new Field("item", new Float32(), true)),
true,
),
],
metadata,
);
const db = await connect(tmpDir.name);
const table = await db.createEmptyTable("test", schema);
await table.add([{ text1: "hello world", text2: "goodbye world" }]);
const rows = await table.query().toArray();
expect(JSON.parse(JSON.stringify(rows[0].vector1))).toEqual([1, 2, 3]);
expect(JSON.parse(JSON.stringify(rows[0].vector2))).toEqual([4, 5, 6]);
});
it("should append generated vectors to a non-nullable schema", async () => { it("should append generated vectors to a non-nullable schema", async () => {
@register("non_nullable_schema_test") @register("non_nullable_schema_test")
@@ -1,95 +0,0 @@
// SPDX-License-Identifier: Apache-2.0
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
import { execFileSync } from "node:child_process";
import { resolve } from "node:path";
import type { OpenAIEmbeddingFunction } from "../lancedb/embedding/openai";
import type { EmbeddingFunctionRegistry } from "../lancedb/embedding/registry";
type EmbeddingModule = typeof import("../lancedb/embedding");
type OpenAIModule = typeof import("../lancedb/embedding/openai");
type RegistryModule = typeof import("../lancedb/embedding/registry");
describe("embedding function registry", () => {
const registries: EmbeddingFunctionRegistry[] = [];
afterEach(() => {
for (const registry of registries) {
registry.reset();
}
registries.length = 0;
});
it("defers built-in providers until the public registry API is used", () => {
jest.isolateModules(() => {
const embedding = require("../lancedb/embedding") as EmbeddingModule;
const { getRegistry: getInternalRegistry } =
require("../lancedb/embedding/registry") as RegistryModule;
const registry = getInternalRegistry();
registries.push(registry);
expect(registry.length()).toBe(0);
expect(embedding.getRegistry()).toBe(registry);
expect(registry.get("openai")).toBeDefined();
expect(registry.get("huggingface")).toBeDefined();
});
});
it("preserves automatic FTS search in a fresh process", () => {
execFileSync(
process.execPath,
[resolve(__dirname, "fixtures", "auto_fts_search.cjs")],
{ stdio: "pipe" },
);
});
it("shares registrations across duplicated provider module graphs", () => {
let registeringRegistry: EmbeddingFunctionRegistry | undefined;
let latestOpenAIConstructor: typeof OpenAIEmbeddingFunction | undefined;
jest.isolateModules(() => {
require("../lancedb/embedding/openai");
const { getRegistry } =
require("../lancedb/embedding/registry") as RegistryModule;
registeringRegistry = getRegistry();
registries.push(registeringRegistry);
expect(registeringRegistry.get("openai")).toBeDefined();
});
expect(() => {
jest.isolateModules(() => {
const { OpenAIEmbeddingFunction } =
require("../lancedb/embedding/openai") as OpenAIModule;
latestOpenAIConstructor = OpenAIEmbeddingFunction;
const { getRegistry } =
require("../lancedb/embedding/registry") as RegistryModule;
registries.push(getRegistry());
});
}).not.toThrow();
const previousApiKey = process.env.OPENAI_API_KEY;
process.env.OPENAI_API_KEY = "test";
try {
const latestOpenAI = registeringRegistry!
.get<OpenAIEmbeddingFunction>("openai")!
.create();
expect(latestOpenAI).toBeInstanceOf(latestOpenAIConstructor!);
} finally {
if (previousApiKey === undefined) {
delete process.env.OPENAI_API_KEY;
} else {
process.env.OPENAI_API_KEY = previousApiKey;
}
}
jest.isolateModules(() => {
const { getRegistry } =
require("../lancedb/embedding") as EmbeddingModule;
const publicRegistry = getRegistry();
registries.push(publicRegistry);
expect(publicRegistry).toBe(registeringRegistry);
expect(publicRegistry.get("openai")).toBeDefined();
});
});
});
@@ -1,33 +0,0 @@
// SPDX-License-Identifier: Apache-2.0
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
const assert = require("node:assert/strict");
const tmp = require("tmp");
const { connect, embedding, Index } = require("../../dist");
const { getRegistry } = require("../../dist/embedding/registry");
async function main() {
assert.equal(typeof embedding.getRegistry, "function");
assert.equal(getRegistry().length(), 0);
assert.equal(embedding.getRegistry(), getRegistry());
assert.equal(getRegistry().length(), 2);
const dir = tmp.dirSync({ unsafeCleanup: true });
let db;
try {
db = await connect(dir.name);
const table = await db.createTable("docs", [{ text: "hello world" }]);
await table.createIndex("text", { config: Index.fts() });
const rows = await table.search("hello").toArray();
assert.equal(rows[0].text, "hello world");
} finally {
db?.close();
dir.removeCallback();
}
}
main().catch((error) => {
console.error(error);
process.exitCode = 1;
});
-22
View File
@@ -48,28 +48,6 @@ describe("materialized views", () => {
expect(definitionFromMetadata(safe, "v").limit).toBe(42); expect(definitionFromMetadata(safe, "v").limit).toBe(42);
}); });
it("reads the namespaced select kind and refuses unknown kinds", () => {
// "namespaced_select" is the namespaced form of "select": same shape, a
// separate kind so readers that predate it refuse instead of resolving
// the source at the root.
const namespaced = new Map([
[
DEFINITION_META_KEY,
'{"kind":"namespaced_select","source_table":"people","source_namespace":["ns"]}',
],
]);
const definition = definitionFromMetadata(namespaced, "v");
expect(definition.sourceTable).toBe("people");
expect(definition.sourceNamespace).toEqual(["ns"]);
const unknown = new Map([
[DEFINITION_META_KEY, '{"kind":"select_v3","source_table":"people"}'],
]);
expect(() => definitionFromMetadata(unknown, "v")).toThrow(
/cannot refresh/,
);
});
it("creates, refreshes and queries a view", async () => { it("creates, refreshes and queries a view", async () => {
const view = await db.createMaterializedView("adults", "people", { const view = await db.createMaterializedView("adults", "people", {
select: ["name", ["shout", "upper(name)"]], select: ["name", ["shout", "upper(name)"]],
+2 -2
View File
@@ -5,8 +5,8 @@ import packageJson = require("../package.json");
describe("package metadata", () => { describe("package metadata", () => {
it("requires Node.js type declarations compatible with the runtime", () => { it("requires Node.js type declarations compatible with the runtime", () => {
expect(packageJson.engines.node).toBe(">= 22"); expect(packageJson.engines.node).toBe(">= 18");
expect(packageJson.peerDependencies["@types/node"]).toBe(">=22"); expect(packageJson.peerDependencies["@types/node"]).toBe(">=18");
expect(packageJson.peerDependenciesMeta["@types/node"]).toEqual({ expect(packageJson.peerDependenciesMeta["@types/node"]).toEqual({
optional: true, optional: true,
}); });
+13 -75
View File
@@ -3,7 +3,6 @@
import * as http from "http"; import * as http from "http";
import { RequestListener } from "http"; import { RequestListener } from "http";
import packageJson = require("../package.json");
import { import {
ClientConfig, ClientConfig,
Connection, Connection,
@@ -71,13 +70,7 @@ async function withMockDatabase(
try { try {
await callback(db); await callback(db);
} finally { } finally {
// `close()` alone leaves the port bound until keep-alive sockets drain, so server.close();
// a single failing test would cascade into EADDRINUSE for every test after
// it. Destroy the connections and wait for the port to actually be free.
await new Promise<void>((resolve) => {
server.closeAllConnections();
server.close(() => resolve());
});
} }
} }
@@ -138,7 +131,7 @@ describe("remote connection", () => {
(req, res) => { (req, res) => {
expect(req.headers["x-api-key"]).toEqual("fake"); expect(req.headers["x-api-key"]).toEqual("fake");
expect(req.headers["user-agent"]).toEqual( expect(req.headers["user-agent"]).toEqual(
`LanceDB-Node-Client/${packageJson.version}`, `LanceDB-Node-Client/${process.env.npm_package_version}`,
); );
const body = JSON.stringify({ tables: [] }); const body = JSON.stringify({ tables: [] });
@@ -939,7 +932,6 @@ describe("remote connection jobs surface", () => {
const { tableFromArrays, tableToIPC } = await import("apache-arrow"); const { tableFromArrays, tableToIPC } = await import("apache-arrow");
const eventsTable = tableFromArrays({ state: ["created", "succeeded"] }); const eventsTable = tableFromArrays({ state: ["created", "succeeded"] });
const eventsBody = Buffer.from(tableToIPC(eventsTable, "stream")); const eventsBody = Buffer.from(tableToIPC(eventsTable, "stream"));
const queryEventsPayloads: Record<string, unknown>[] = [];
await withMockDatabase( await withMockDatabase(
(req, res) => { (req, res) => {
@@ -968,16 +960,6 @@ describe("remote connection jobs surface", () => {
); );
} }
} else if (req.url === "/v1/jobs/describe") { } else if (req.url === "/v1/jobs/describe") {
if (payload["job_id"] === "job-2") {
res
.writeHead(200, { "Content-Type": "application/json" })
.end(
'{"job_id": "job-2", "job_type": "refresh_column", ' +
'"job_state": "DONE", "creation_ms": 2000, ' +
'"result": {"rows_assigned": 1000000}}',
);
return;
}
if (payload["job_id"] !== "job-1") { if (payload["job_id"] !== "job-1") {
res.writeHead(404).end("no such job"); res.writeHead(404).end("no such job");
return; return;
@@ -999,7 +981,6 @@ describe("remote connection jobs surface", () => {
.writeHead(200, { "Content-Type": "application/json" }) .writeHead(200, { "Content-Type": "application/json" })
.end('{"job_id": "job-1"}'); .end('{"job_id": "job-1"}');
} else if (req.url === "/v1/jobs/query_events") { } else if (req.url === "/v1/jobs/query_events") {
queryEventsPayloads.push(payload);
res res
.writeHead(200, { .writeHead(200, {
"Content-Type": "application/vnd.apache.arrow.stream", "Content-Type": "application/vnd.apache.arrow.stream",
@@ -1016,65 +997,22 @@ describe("remote connection jobs surface", () => {
expect(jobs[0].state).toEqual("running"); expect(jobs[0].state).toEqual("running");
expect(jobs[1].state).toEqual("finished"); expect(jobs[1].state).toEqual("finished");
const description = await db.getJob("job-1");
expect(description?.state).toEqual("failed");
expect(JSON.parse(description?.specJson ?? "")).toEqual({
column: "vec",
});
expect(description?.failure?.message).toEqual("worker died");
expect(await db.getJob("missing")).toBeNull();
expect(await db.cancelJob("job-1")).toBe(true); expect(await db.cancelJob("job-1")).toBe(true);
expect(await db.cancelJob("missing")).toBe(false); expect(await db.cancelJob("missing")).toBe(false);
// Opening a job hands back a populated handle; a missing one rejects. const history = await db.jobHistory("job-1");
await expect(db.openJob("missing")).rejects.toThrow("not found"); expect(history.numRows).toEqual(2);
const finished = await db.openJob("job-2");
expect(finished.state).toEqual("finished");
expect(finished.result).toEqual({
// biome-ignore lint/style/useNamingConvention: snake_case mandated by the server wire format
rows_assigned: 1000000,
});
const job = await db.openJob("job-1"); const job = db.job("job-1");
expect(job.id).toEqual("job-1"); expect(job.id).toEqual("job-1");
// openJob already populated the handle; refresh() re-reads it.
expect(job.state).toEqual("failed");
await job.refresh();
expect(job.state).toEqual("failed");
expect(job.jobType).toEqual("create_index");
expect(job.creationMs).toEqual(1000);
expect(job.spec).toEqual({ column: "vec" });
expect(job.result).toBeNull();
expect(job.failure?.message).toEqual("worker died");
// The handle reaches its own events, supplying its job id.
const jobEvents = await job.events({
limit: 500,
filter: "state = 'claim_complete'",
});
expect(jobEvents.numRows).toEqual(2);
expect(queryEventsPayloads.pop()).toEqual({
// biome-ignore lint/style/useNamingConvention: snake_case mandated by the server wire format
job_id: "job-1",
limit: 500,
filter: "state = 'claim_complete'",
});
// Printing lays every known field out on its own line, with the JSON
// payloads indented rather than crammed onto one line.
expect(`${job}`).toEqual(
[
"Job(",
' id="job-1",',
' state="failed",',
' jobType="create_index",',
" creationMs=1000,",
" spec={",
' "column": "vec"',
" },",
" failure={",
' "phase": "execute",',
' "message": "worker died",',
' "retryable": true',
" },",
")",
].join("\n"),
);
expect(await job.status()).toEqual("failed"); expect(await job.status()).toEqual("failed");
await expect(job.wait()).rejects.toThrow("worker died"); await expect(job.wait()).rejects.toThrow("worker died");
}, },
+2 -638
View File
@@ -11,13 +11,10 @@ import * as arrow17 from "apache-arrow-17";
import * as arrow18 from "apache-arrow-18"; import * as arrow18 from "apache-arrow-18";
import { import {
AutoQuery,
Connection, Connection,
MatchQuery, MatchQuery,
PhraseQuery, PhraseQuery,
Query,
Table, Table,
VectorQuery,
connect, connect,
tokenize, tokenize,
} from "../lancedb"; } from "../lancedb";
@@ -685,64 +682,13 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
}, },
); );
// https://github.com/lancedb/lancedb/issues/1963
it("should query documents with LangChain PDF metadata", async () => {
const tmpDir = tmp.dirSync({ unsafeCleanup: true });
try {
const db = await connect(tmpDir.name);
const documents = [
{
text: "first page",
vector: [1, 0],
source: "first.pdf",
loc: { pageNumber: 1, lines: { from: 1, to: 12 } },
pdf: {
version: "1.10.100",
info: {
format: "PDF 1.7",
producer: "pdf.js",
creator: "Writer",
},
totalPages: 2,
},
},
{
text: "second page",
vector: [0, 1],
source: "second.pdf",
loc: { pageNumber: 2, lines: { from: 13, to: 24 } },
pdf: {
version: "1.10.100",
info: {
format: "PDF 1.7",
producer: "pdf.js",
creator: "Writer",
},
totalPages: 2,
},
},
];
const documentsTable = await db.createTable("documents", documents);
const results = await documentsTable.query().toArray();
expect(results).toHaveLength(2);
expect(results[0].source).toBe("first.pdf");
expect(results[0].pdf.info.producer).toBe("pdf.js");
expect(results[1].loc.pageNumber).toBe(2);
} finally {
tmpDir.removeCallback();
}
});
describe("merge insert", () => { describe("merge insert", () => {
let tmpDir: tmp.DirResult; let tmpDir: tmp.DirResult;
let conn: Connection;
let table: Table; let table: Table;
beforeEach(async () => { beforeEach(async () => {
tmpDir = tmp.dirSync({ unsafeCleanup: true }); tmpDir = tmp.dirSync({ unsafeCleanup: true });
conn = await connect(tmpDir.name); const conn = await connect(tmpDir.name);
table = await conn.createTable("some_table", [ table = await conn.createTable("some_table", [
{ a: 1, b: "a" }, { a: 1, b: "a" },
@@ -780,38 +726,6 @@ describe("merge insert", () => {
expect(result.map((row) => ({ ...row }))).toEqual(expected); expect(result.map((row) => ({ ...row }))).toEqual(expected);
}); });
test("upsert on a composite key", async () => {
const composite = await conn.createTable("composite", [
{ shard: "a", id: 1, val: "x" },
{ shard: "a", id: 2, val: "y" },
{ shard: "b", id: 1, val: "z" },
]);
// ("a", 1) matches an existing row and updates it. ("b", 2) agrees with an
// existing row on each key column separately but on neither pair, so it is
// an insert.
const mergeInsertRes = await composite
.mergeInsert(["shard", "id"])
.whenMatchedUpdateAll()
.whenNotMatchedInsertAll()
.execute([
{ shard: "a", id: 1, val: "X" },
{ shard: "b", id: 2, val: "W" },
]);
expect(mergeInsertRes.numUpdatedRows).toBe(1);
expect(mergeInsertRes.numInsertedRows).toBe(1);
const result = (await composite.toArrow())
.toArray()
.sort((a, b) => a.shard.localeCompare(b.shard) || a.id - b.id);
expect(result.map((row) => ({ ...row }))).toEqual([
{ shard: "a", id: 1, val: "X" },
{ shard: "a", id: 2, val: "y" },
{ shard: "b", id: 1, val: "z" },
{ shard: "b", id: 2, val: "W" },
]);
});
test("conditional update", async () => { test("conditional update", async () => {
const newData = [ const newData = [
{ a: 2, b: "x" }, { a: 2, b: "x" },
@@ -1863,194 +1777,6 @@ describe("Read consistency interval", () => {
}); });
}); });
describe("automatic search schema consistency", () => {
let tmpDir: tmp.DirResult;
class SchemaRefreshEmbedding extends EmbeddingFunction<string> {
ndims() {
return 2;
}
embeddingDataType() {
return new Float32();
}
async computeSourceEmbeddings(data: string[]) {
return data.map((value) => [value.length, 1]);
}
async computeQueryEmbeddings(value: string) {
return [value.length, 1];
}
}
function embeddingSchema() {
const func = new SchemaRefreshEmbedding();
return LanceSchema({
text: func.sourceField(new Utf8()),
vector: func.vectorField(),
});
}
beforeEach(() => {
getRegistry().reset();
register("schema-refresh")(SchemaRefreshEmbedding);
tmpDir = tmp.dirSync({ unsafeCleanup: true });
});
afterEach(() => {
getRegistry().reset();
tmpDir.removeCallback();
});
it("uses the schema refreshed from another connection", async () => {
const first = await connect(tmpDir.name, { readConsistencyInterval: 0 });
const second = await connect(tmpDir.name, { readConsistencyInterval: 0 });
try {
const stale = await first.createTable("docs", [{ text: "before" }], {
schema: embeddingSchema(),
});
const replacement = await second.createTable(
"docs",
[{ text: "after hello" }],
{ mode: "overwrite" },
);
await replacement.createIndex("text", { config: Index.fts() });
const search = stale.search("hello");
expect(search).toBeInstanceOf(AutoQuery);
expect(search).not.toBeInstanceOf(Query);
expect(search).not.toBeInstanceOf(VectorQuery);
expect("nprobes" in search).toBe(false);
const rows = await search.toArray();
expect(rows[0].text).toBe("after hello");
expect((await stale.schema()).metadata.has("embedding_functions")).toBe(
false,
);
} finally {
first.close();
second.close();
}
});
it("tracks embedding metadata across checkout and restore", async () => {
const first = await connect(tmpDir.name, { readConsistencyInterval: 0 });
const second = await connect(tmpDir.name, { readConsistencyInterval: 0 });
try {
await first.createTable("docs", [{ text: "before" }], {
schema: embeddingSchema(),
});
const table = await second.createTable(
"docs",
[{ text: "after hello" }],
{ mode: "overwrite" },
);
await table.createIndex("text", { config: Index.fts() });
await table.checkout(1);
expect((await table.search("before").toArray())[0].text).toBe("before");
await table.checkoutLatest();
expect((await table.search("hello").toArray())[0].text).toBe(
"after hello",
);
await table.checkout(1);
await table.restore();
expect((await table.search("before").toArray())[0].text).toBe("before");
} finally {
first.close();
second.close();
}
});
it("pins automatic search while computing an embedding", async () => {
let markStarted!: () => void;
let releaseEmbedding!: () => void;
const started = new Promise<void>((resolve) => {
markStarted = resolve;
});
const released = new Promise<void>((resolve) => {
releaseEmbedding = resolve;
});
class BlockingEmbedding extends SchemaRefreshEmbedding {
async computeQueryEmbeddings(value: string) {
markStarted();
await released;
return [value.length, 1];
}
}
register("schema-refresh-blocking")(BlockingEmbedding);
const func = new BlockingEmbedding();
const schema = LanceSchema({
text: func.sourceField(new Utf8()),
vector: func.vectorField(),
});
const first = await connect(tmpDir.name, { readConsistencyInterval: 0 });
const second = await connect(tmpDir.name, { readConsistencyInterval: 0 });
try {
const table = await first.createTable(
"docs",
[{ text: "hello before" }],
{ schema },
);
const pending = table.search("hello").toArray();
await started;
const replacement = await second.createTable(
"docs",
[{ text: "hello after" }],
{ mode: "overwrite" },
);
await replacement.createIndex("text", { config: Index.fts() });
releaseEmbedding();
expect((await pending)[0].text).toBe("hello before");
} finally {
releaseEmbedding();
first.close();
second.close();
}
});
it("refreshes a reused automatic search for every execution", async () => {
const first = await connect(tmpDir.name, { readConsistencyInterval: 0 });
const second = await connect(tmpDir.name, { readConsistencyInterval: 0 });
try {
const table = await first.createTable("docs", [
{ text: "hello before", marker: "before" },
]);
await table.createIndex("text", { config: Index.fts() });
const search = table.search("hello").select(["text"]);
const before = (await search.toArray())[0];
expect(before.text).toBe("hello before");
expect(before.marker).toBeUndefined();
const replacement = await second.createTable(
"docs",
[{ text: "hello after", marker: "after" }],
{ mode: "overwrite" },
);
await replacement.createIndex("text", { config: Index.fts() });
const after = (await search.toArray())[0];
expect(after.text).toBe("hello after");
expect(after.marker).toBeUndefined();
} finally {
first.close();
second.close();
}
});
});
describe("schema evolution", function () { describe("schema evolution", function () {
let tmpDir: tmp.DirResult; let tmpDir: tmp.DirResult;
beforeEach(() => { beforeEach(() => {
@@ -2618,24 +2344,7 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
); );
}); });
test("full text search if only an unrelated embedding function is registered", async () => { test("full text search if no embedding function provided", async () => {
register("unused")(
class extends EmbeddingFunction<string> {
ndims() {
return 3;
}
embeddingDataType() {
return new Float32();
}
async computeQueryEmbeddings(_data: string) {
return [1, 2, 3];
}
async computeSourceEmbeddings(data: string[]) {
return data.map(() => [1, 2, 3]);
}
},
);
const db = await connect(tmpDir.name); const db = await connect(tmpDir.name);
const data = [ const data = [
{ text: "hello world", vector: [0.1, 0.2, 0.3] }, { text: "hello world", vector: [0.1, 0.2, 0.3] },
@@ -2657,306 +2366,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
expect(results2[0].text).toBe(data[1].text); expect(results2[0].text).toBe(data[1].text);
}); });
test("auto search stays consistent with the active revision", async () => {
let initCalls = 0;
let queryCalls = 0;
let markStarted!: () => void;
const started = new Promise<void>((resolve) => {
markStarted = resolve;
});
let releaseEmbedding!: () => void;
const embeddingReleased = new Promise<void>((resolve) => {
releaseEmbedding = resolve;
});
@register("refresh-test")
class TestEmbedding extends EmbeddingFunction<string> {
async init() {
initCalls += 1;
}
ndims() {
return 1;
}
embeddingDataType() {
return new arrow.Float32();
}
async computeQueryEmbeddings(value: string) {
queryCalls += 1;
if (value === "blocked") {
markStarted();
await embeddingReleased;
}
return value === "greetings" ? [0.1] : [0.2];
}
async computeSourceEmbeddings(values: string[]) {
return values.map((value) =>
value === "hello world" ? [0.1] : [0.2],
);
}
}
const writer = await connect(tmpDir.name);
await writer.createTable("test", [{ text: "plain", vector: [0.0] }]);
const reader = await connect(tmpDir.name, {
readConsistencyInterval: 0,
});
const tracked = await reader.openTable("test");
type SnapshotCountingNative = {
querySnapshot: () => Promise<unknown>;
};
const native = (tracked as unknown as { inner: SnapshotCountingNative })
.inner;
const querySnapshot = native.querySnapshot.bind(native);
let snapshotCalls = 0;
native.querySnapshot = async () => {
snapshotCalls += 1;
return await querySnapshot();
};
const autoQuery = tracked.search("greetings").select(["text"]).limit(1);
const func = new TestEmbedding();
const schema = LanceSchema({
text: func.sourceField(new arrow.Utf8()),
vector: func.vectorField(),
});
const data = [{ text: "hello world" }, { text: "goodbye world" }];
await writer.createTable("test", data, { mode: "overwrite", schema });
const baselineInitCalls = initCalls;
expect(
(await tracked.schema()).metadata.get("embedding_functions"),
).toBeDefined();
const results = await autoQuery.toArray();
expect(results[0].text).toBe(data[0].text);
expect(initCalls).toBe(baselineInitCalls + 1);
expect(queryCalls).toBe(1);
expect(snapshotCalls).toBe(1);
const repeatedResults = await autoQuery.toArray();
expect(repeatedResults[0].text).toBe(data[0].text);
expect(initCalls).toBe(baselineInitCalls + 1);
expect(queryCalls).toBe(1);
expect(snapshotCalls).toBe(2);
const pending = tracked
.search("blocked")
.select(["text"])
.limit(1)
.toArray();
await started;
const ftsData = [
{ text: "greetings from full text", vector: [0.0] },
{ text: "blocked from full text", vector: [0.0] },
];
const ftsTable = await writer.createTable("test", ftsData, {
mode: "overwrite",
});
await ftsTable.createIndex("text", { config: Index.fts() });
releaseEmbedding();
const pendingResults = await pending;
expect(pendingResults[0].text).toBe(data[1].text);
expect(
(await tracked.schema()).metadata.get("embedding_functions"),
).toBeUndefined();
const ftsResults = await autoQuery.toArray();
expect(ftsResults[0].text).toBe(ftsData[0].text);
});
test("auto search keeps newer preparation during a revision race", async () => {
let aCalls = 0;
let bCalls = 0;
let markAStarted!: () => void;
const aStarted = new Promise<void>((resolve) => {
markAStarted = resolve;
});
let releaseA!: () => void;
const aReleased = new Promise<void>((resolve) => {
releaseA = resolve;
});
let markBStarted!: () => void;
const bStarted = new Promise<void>((resolve) => {
markBStarted = resolve;
});
let releaseB!: () => void;
const bReleased = new Promise<void>((resolve) => {
releaseB = resolve;
});
@register("race-a")
class EmbeddingA extends EmbeddingFunction<string> {
ndims() {
return 1;
}
embeddingDataType() {
return new arrow.Float32();
}
async computeQueryEmbeddings() {
aCalls += 1;
markAStarted();
await aReleased;
return [0.1];
}
async computeSourceEmbeddings(values: string[]) {
return values.map(() => [0.1]);
}
}
@register("race-b")
class EmbeddingB extends EmbeddingFunction<string> {
ndims() {
return 1;
}
embeddingDataType() {
return new arrow.Float32();
}
async computeQueryEmbeddings() {
bCalls += 1;
markBStarted();
await bReleased;
return [0.2];
}
async computeSourceEmbeddings(values: string[]) {
return values.map(() => [0.2]);
}
}
const writer = await connect(tmpDir.name);
const embeddingA = new EmbeddingA();
const schemaA = LanceSchema({
text: embeddingA.sourceField(new arrow.Utf8()),
vector: embeddingA.vectorField(),
});
await writer.createTable("race", [{ text: "revision a" }], {
schema: schemaA,
});
const reader = await connect(tmpDir.name, {
readConsistencyInterval: 0,
});
const tracked = await reader.openTable("race");
const query = tracked.search("query");
const first = query.toArray();
await aStarted;
const embeddingB = new EmbeddingB();
const schemaB = LanceSchema({
text: embeddingB.sourceField(new arrow.Utf8()),
vector: embeddingB.vectorField(),
});
await writer.createTable("race", [{ text: "revision b" }], {
mode: "overwrite",
schema: schemaB,
});
const second = query.toArray();
await bStarted;
releaseA();
releaseB();
await Promise.all([first, second]);
expect(aCalls).toBe(1);
expect(bCalls).toBe(1);
});
test("stale FTS routing keeps newer vector preparation", async () => {
let vectorCalls = 0;
let markVectorStarted!: () => void;
const vectorStarted = new Promise<void>((resolve) => {
markVectorStarted = resolve;
});
let releaseVector!: () => void;
const vectorReleased = new Promise<void>((resolve) => {
releaseVector = resolve;
});
@register("stale-fts-race")
class RaceEmbedding extends EmbeddingFunction<string> {
ndims() {
return 1;
}
embeddingDataType() {
return new arrow.Float32();
}
async computeQueryEmbeddings() {
vectorCalls += 1;
markVectorStarted();
await vectorReleased;
return [0.1];
}
async computeSourceEmbeddings(values: string[]) {
return values.map(() => [0.1]);
}
}
const writer = await connect(tmpDir.name);
const ftsTable = await writer.createTable("stale_fts", [
{ text: "hello", vector: [0.0] },
]);
await ftsTable.createIndex("text", { config: Index.fts() });
const reader = await connect(tmpDir.name, {
readConsistencyInterval: 0,
});
const tracked = await reader.openTable("stale_fts");
type Snapshot = {
schema: () => Promise<Buffer>;
};
type NativeWithSnapshot = {
querySnapshot: () => Promise<Snapshot>;
};
const native = (tracked as unknown as { inner: NativeWithSnapshot })
.inner;
const querySnapshot = native.querySnapshot.bind(native);
let snapshotCalls = 0;
let markStaleSchemaStarted!: () => void;
const staleSchemaStarted = new Promise<void>((resolve) => {
markStaleSchemaStarted = resolve;
});
let releaseStaleSchema!: () => void;
const staleSchemaReleased = new Promise<void>((resolve) => {
releaseStaleSchema = resolve;
});
native.querySnapshot = async () => {
const snapshot = await querySnapshot();
snapshotCalls += 1;
if (snapshotCalls === 1) {
const schema = snapshot.schema.bind(snapshot);
snapshot.schema = async () => {
markStaleSchemaStarted();
await staleSchemaReleased;
return await schema();
};
}
return snapshot;
};
const query = tracked.search("hello");
const staleFtsExecution = query.toArray();
await staleSchemaStarted;
const embedding = new RaceEmbedding();
const vectorSchema = LanceSchema({
text: embedding.sourceField(new arrow.Utf8()),
vector: embedding.vectorField(),
});
await writer.createTable("stale_fts", [{ text: "hello" }], {
mode: "overwrite",
schema: vectorSchema,
});
const vectorExecution = query.toArray();
await vectorStarted;
releaseStaleSchema();
await staleFtsExecution;
releaseVector();
await vectorExecution;
await query.toArray();
expect(vectorCalls).toBe(1);
});
test("tokenizes FTS queries by column or index name", async () => { test("tokenizes FTS queries by column or index name", async () => {
const db = await connect(tmpDir.name); const db = await connect(tmpDir.name);
const data = [ const data = [
@@ -3507,30 +2916,6 @@ describe("column name options", () => {
expect(results[1].query_index).toBe(1); expect(results[1].query_index).toBe(1);
}); });
test("observes promised additional vectors while the query is pending", async () => {
const initialVector = new Promise<number[]>(() => undefined);
const query = table.query().nearestTo(initialVector);
const unhandled: unknown[] = [];
const onUnhandled = (reason: unknown) => unhandled.push(reason);
process.on("unhandledRejection", onUnhandled);
try {
query.addQueryVector(Promise.reject(new Error("extra vector failed")));
await new Promise<void>((resolve) => setImmediate(resolve));
expect(unhandled).toEqual([]);
const rejectedQuery = table
.query()
.nearestTo([0.1, 0.2])
.addQueryVector(Promise.reject(new Error("consumed vector failed")));
await expect(rejectedQuery.toArray()).rejects.toThrow(
"consumed vector failed",
);
} finally {
process.off("unhandledRejection", onUnhandled);
}
});
test("index and search multivectors", async () => { test("index and search multivectors", async () => {
const db = await connect(tmpDir.name); const db = await connect(tmpDir.name);
const data = []; const data = [];
@@ -3594,27 +2979,6 @@ describe("when creating an empty table", () => {
expect((actualSchema.fields[1].type as Float64).precision).toBe(2); expect((actualSchema.fields[1].type as Float64).precision).toBe(2);
}); });
it("can add and query JSON data", async () => {
const schema = new Schema([
new Field("id", new Int32(), true),
new Field(
"meta",
new Utf8(),
true,
new Map([["ARROW:extension:name", "arrow.json"]]),
),
]);
const table = await con.createEmptyTable("json", schema);
const meta = JSON.stringify({ x: 1 });
await table.add([{ id: 1, meta }]);
const rows = await table.query().toArray();
expect(rows).toHaveLength(1);
expect(rows[0].id).toBe(1);
expect(rows[0].meta).toBe(meta);
});
it("can create an empty table from schema that specifies field types by name", async () => { it("can create an empty table from schema that specifies field types by name", async () => {
const schemaLike = { const schemaLike = {
fields: [ fields: [
+1 -1
View File
@@ -170,7 +170,7 @@ test("basic table examples", async () => {
// --8<-- [end:create_index] // --8<-- [end:create_index]
// --8<-- [start:delete_rows] // --8<-- [start:delete_rows]
await tbl.delete("item = 'fizz'"); await tbl.delete('item = "fizz"');
// --8<-- [end:delete_rows] // --8<-- [end:delete_rows]
// --8<-- [start:drop_table] // --8<-- [start:drop_table]
+1 -2
View File
@@ -8,8 +8,7 @@
"//1": "--experimental-vm-modules is needed to run jest with sentence-transformers", "//1": "--experimental-vm-modules is needed to run jest with sentence-transformers",
"//2": "--testEnvironment is needed to run jest with sentence-transformers", "//2": "--testEnvironment is needed to run jest with sentence-transformers",
"//3": "See: https://github.com/huggingface/transformers.js/issues/57", "//3": "See: https://github.com/huggingface/transformers.js/issues/57",
"//4": "jest is invoked by its JS entry, not node_modules/.bin/jest: under pnpm that path is a shell shim, which `node` cannot execute", "test": "node --experimental-vm-modules node_modules/.bin/jest --testEnvironment jest-environment-node-single-context --verbose",
"test": "node --experimental-vm-modules node_modules/jest/bin/jest.js --testEnvironment jest-environment-node-single-context --verbose",
"lint": "biome check *.ts && biome format *.ts", "lint": "biome check *.ts && biome format *.ts",
"lint-ci": "biome ci .", "lint-ci": "biome ci .",
"lint-fix": "biome check --write *.ts && pnpm format", "lint-fix": "biome check --write *.ts && pnpm format",
+309 -35
View File
@@ -5,6 +5,7 @@ import {
Data as ArrowData, Data as ArrowData,
Table as ArrowTable, Table as ArrowTable,
Binary, Binary,
Bool,
BufferType, BufferType,
DataType, DataType,
DateUnit, DateUnit,
@@ -17,7 +18,12 @@ import {
FixedSizeList, FixedSizeList,
Float, Float,
Float32, Float32,
Float64,
Int, Int,
Int8,
Int16,
Int32,
Int64,
LargeBinary, LargeBinary,
List, List,
Null, Null,
@@ -30,16 +36,17 @@ import {
Struct, Struct,
Timestamp, Timestamp,
Type, Type,
Uint8,
Uint16,
Uint32,
Utf8, Utf8,
Vector, Vector,
makeVector as arrowMakeVector, makeVector as arrowMakeVector,
util as arrowUtil,
vectorFromArray as badVectorFromArray, vectorFromArray as badVectorFromArray,
makeBuilder, makeBuilder,
makeData, makeData,
} from "apache-arrow"; } from "apache-arrow";
import { Buffers } from "apache-arrow/data"; import { Buffers } from "apache-arrow/data";
import { typedArrayToArrowType } from "./arrow_type";
import { type EmbeddingFunction } from "./embedding/embedding_function"; import { type EmbeddingFunction } from "./embedding/embedding_function";
import { import {
EmbeddingFunctionConfig, EmbeddingFunctionConfig,
@@ -52,7 +59,14 @@ import {
sanitizeTable, sanitizeTable,
sanitizeType, sanitizeType,
} from "./sanitize"; } from "./sanitize";
import { inferSchema } from "./schema";
/**
* Check if a field name indicates a vector column.
*/
function nameSuggestsVectorColumn(fieldName: string): boolean {
const nameLower = fieldName.toLowerCase();
return nameLower.includes("vector") || nameLower.includes("embedding");
}
export * from "apache-arrow"; export * from "apache-arrow";
export type SchemaLike = export type SchemaLike =
@@ -72,7 +86,8 @@ export type FieldLike =
}; };
export type DataLike = export type DataLike =
| import("apache-arrow").Data // biome-ignore lint/suspicious/noExplicitAny: <explanation>
| import("apache-arrow").Data<Struct<any>>
| { | {
// biome-ignore lint/suspicious/noExplicitAny: <explanation> // biome-ignore lint/suspicious/noExplicitAny: <explanation>
type: any; type: any;
@@ -81,7 +96,6 @@ export type DataLike =
stride: number; stride: number;
nullable: boolean; nullable: boolean;
children: DataLike[]; children: DataLike[];
dictionary?: { data: readonly DataLike[] };
get nullCount(): number; get nullCount(): number;
// biome-ignore lint/suspicious/noExplicitAny: <explanation> // biome-ignore lint/suspicious/noExplicitAny: <explanation>
values: Buffers<any>[BufferType.DATA]; values: Buffers<any>[BufferType.DATA];
@@ -445,6 +459,110 @@ export function makeArrowTable(
return new ArrowTable(inferredSchema, finalColumns); return new ArrowTable(inferredSchema, finalColumns);
} }
function inferSchema(
data: Array<Record<string, unknown>>,
schema: Schema | undefined,
opts: MakeArrowTableOptions,
): Schema {
// We will collect all fields we see in the data.
const pathTree = new PathTree<DataType>();
for (const [rowI, row] of data.entries()) {
for (const [path, value] of rowPathsAndValues(row)) {
if (!pathTree.has(path)) {
// First time seeing this field.
if (schema !== undefined) {
const field = getFieldForPath(schema, path);
if (field === undefined) {
throw new Error(
`Found field not in schema: ${path.join(".")} at row ${rowI}`,
);
} else {
pathTree.set(path, field.type);
}
} else {
const inferredType = inferType(value, path, opts);
if (inferredType === undefined) {
throw new Error(`Failed to infer data type for field ${path.join(
".",
)} at row ${rowI}. \
Consider providing an explicit schema.`);
}
pathTree.set(path, inferredType);
}
} else if (schema === undefined) {
const currentType = pathTree.get(path);
const newType = inferType(value, path, opts);
if (currentType !== newType) {
new Error(`Failed to infer schema for data. Previously inferred type \
${currentType} but found ${newType} at row ${rowI}. Consider \
providing an explicit schema.`);
}
}
}
}
if (schema === undefined) {
function fieldsFromPathTree(pathTree: PathTree<DataType>): Field[] {
const fields = [];
for (const [name, value] of pathTree.map.entries()) {
if (value instanceof PathTree) {
const children = fieldsFromPathTree(value);
fields.push(new Field(name, new Struct(children), true));
} else {
fields.push(new Field(name, value, true));
}
}
return fields;
}
const fields = fieldsFromPathTree(pathTree);
return new Schema(fields);
} else {
function takeMatchingFields(
fields: Field[],
pathTree: PathTree<DataType>,
): Field[] {
const outFields = [];
for (const field of fields) {
if (pathTree.map.has(field.name)) {
const value = pathTree.get([field.name]);
if (value instanceof PathTree) {
const struct = field.type as Struct;
const children = takeMatchingFields(struct.children, value);
outFields.push(
new Field(field.name, new Struct(children), field.nullable),
);
} else {
outFields.push(
new Field(field.name, value as DataType, field.nullable),
);
}
}
}
return outFields;
}
const fields = takeMatchingFields(schema.fields, pathTree);
return new Schema(fields);
}
}
function* rowPathsAndValues(
row: Record<string, unknown>,
basePath: string[] = [],
): Generator<[string[], unknown]> {
for (const [key, value] of Object.entries(row)) {
if (isObject(value)) {
yield* rowPathsAndValues(value, [...basePath, key]);
} else {
// Skip undefined values - they should be treated the same as missing fields
// for embedding function purposes
if (value !== undefined) {
yield [[...basePath, key], value];
}
}
}
}
function isObject(value: unknown): value is Record<string, unknown> { function isObject(value: unknown): value is Record<string, unknown> {
return ( return (
typeof value === "object" && typeof value === "object" &&
@@ -459,19 +577,146 @@ function isObject(value: unknown): value is Record<string, unknown> {
); );
} }
function valueAtPath(datum: Record<string, unknown>, path: string[]): unknown { function getFieldForPath(schema: Schema, path: string[]): Field | undefined {
let current: unknown = datum; let current: Field | Schema = schema;
for (const key of path) { for (const key of path) {
if (current == null) { if (current instanceof Schema) {
return null; const field: Field | undefined = current.fields.find(
} (f) => f.name === key,
if (isObject(current) && (Object.hasOwn(current, key) || key in current)) { );
current = current[key]; if (field === undefined) {
return undefined;
}
current = field;
} else if (current instanceof Field && DataType.isStruct(current.type)) {
const struct: Struct = current.type;
const field = struct.children.find((f) => f.name === key);
if (field === undefined) {
return undefined;
}
current = field;
} else { } else {
return undefined; return undefined;
} }
} }
return current; if (current instanceof Field) {
return current;
} else {
return undefined;
}
}
/**
* Try to infer which Arrow type to use for a given value.
*
* May return undefined if the type cannot be inferred.
*/
function inferType(
value: unknown,
path: string[],
opts: MakeArrowTableOptions,
): DataType | undefined {
if (typeof value === "bigint") {
return new Int64();
} else if (typeof value === "number") {
// Even if it's an integer, it's safer to assume Float64. Users can
// always provide an explicit schema or use BigInt if they mean integer.
return new Float64();
} else if (typeof value === "string") {
if (opts.dictionaryEncodeStrings) {
return new Dictionary(new Utf8(), new Int32());
} else {
return new Utf8();
}
} else if (typeof value === "boolean") {
return new Bool();
} else if (value instanceof Buffer) {
return new Binary();
} else if (ArrayBuffer.isView(value) && !(value instanceof DataView)) {
const info = typedArrayToArrowType(value);
if (info !== undefined) {
const child = new Field("item", info.elementType, true);
return new FixedSizeList(info.length, child);
}
return undefined;
} else if (Array.isArray(value)) {
if (value.length === 0) {
return undefined; // Without any values we can't infer the type
}
if (path.length === 1 && Object.hasOwn(opts.vectorColumns, path[0])) {
const floatType = sanitizeType(opts.vectorColumns[path[0]].type);
return new FixedSizeList(
value.length,
new Field("item", floatType, true),
);
}
const valueType = inferType(value[0], path, opts);
if (valueType === undefined) {
return undefined;
}
// Try to automatically detect embedding columns.
if (nameSuggestsVectorColumn(path[path.length - 1])) {
// Check if value is a Uint8Array for integer vector type determination
if (value instanceof Uint8Array) {
// For integer vectors, we default to Uint8 (matching Python implementation)
const child = new Field("item", new Uint8(), true);
return new FixedSizeList(value.length, child);
} else {
// For float vectors, we default to Float32
const child = new Field("item", new Float32(), true);
return new FixedSizeList(value.length, child);
}
} else {
const child = new Field("item", valueType, true);
return new List(child);
}
} else {
// TODO: timestamp
return undefined;
}
}
class PathTree<V> {
map: Map<string, V | PathTree<V>>;
constructor(entries?: [string[], V][]) {
this.map = new Map();
if (entries !== undefined) {
for (const [path, value] of entries) {
this.set(path, value);
}
}
}
has(path: string[]): boolean {
let ref: PathTree<V> = this;
for (const part of path) {
if (!(ref instanceof PathTree) || !ref.map.has(part)) {
return false;
}
ref = ref.map.get(part) as PathTree<V>;
}
return true;
}
get(path: string[]): V | undefined {
let ref: PathTree<V> = this;
for (const part of path) {
if (!(ref instanceof PathTree) || !ref.map.has(part)) {
return undefined;
}
ref = ref.map.get(part) as PathTree<V>;
}
return ref as V;
}
set(path: string[], value: V): void {
let ref: PathTree<V> = this;
for (const part of path.slice(0, path.length - 1)) {
if (!ref.map.has(part)) {
ref.map.set(part, new PathTree<V>());
}
ref = ref.map.get(part) as PathTree<V>;
}
ref.map.set(path[path.length - 1], value);
}
} }
function transposeData( function transposeData(
@@ -479,26 +724,37 @@ function transposeData(
field: Field, field: Field,
path: string[] = [], path: string[] = [],
): Vector { ): Vector {
const valuesPath = [...path, field.name];
const values = data.map((datum) => valueAtPath(datum, valuesPath));
if (field.type instanceof Struct) { if (field.type instanceof Struct) {
const childFields = field.type.children; const childFields = field.type.children;
const fullPath = [...path, field.name];
const childVectors = childFields.map((child) => { const childVectors = childFields.map((child) => {
return transposeData(data, child, valuesPath); return transposeData(data, child, fullPath);
}); });
const nullCount = values.filter((value) => value === null).length;
const structData = makeData({ const structData = makeData({
type: field.type, type: field.type,
length: values.length,
nullCount,
nullBitmap:
nullCount > 0
? arrowUtil.packBools(values.map((value) => value !== null))
: undefined,
children: childVectors as unknown as ArrowData<DataType>[], children: childVectors as unknown as ArrowData<DataType>[],
}); });
return arrowMakeVector(structData); return arrowMakeVector(structData);
} else { } else {
const valuesPath = [...path, field.name];
const values = data.map((datum) => {
let current: unknown = datum;
for (const key of valuesPath) {
if (current == null) {
return null;
}
if (
isObject(current) &&
(Object.hasOwn(current, key) || key in current)
) {
current = current[key];
} else {
return null;
}
}
return current;
});
return makeVector(values, field.type, undefined, field.nullable); return makeVector(values, field.type, undefined, field.nullable);
} }
} }
@@ -541,6 +797,32 @@ function makeListVector(lists: unknown[][]): Vector<unknown> {
return listBuilder.finish().toVector(); return listBuilder.finish().toVector();
} }
/**
* Map a JS TypedArray instance to the corresponding Arrow element DataType
* and its length. Returns undefined if the value is not a recognized TypedArray.
*/
function typedArrayToArrowType(
value: ArrayBufferView,
): { elementType: DataType; length: number } | undefined {
if (value instanceof Float32Array)
return { elementType: new Float32(), length: value.length };
if (value instanceof Float64Array)
return { elementType: new Float64(), length: value.length };
if (value instanceof Uint8Array)
return { elementType: new Uint8(), length: value.length };
if (value instanceof Uint16Array)
return { elementType: new Uint16(), length: value.length };
if (value instanceof Uint32Array)
return { elementType: new Uint32(), length: value.length };
if (value instanceof Int8Array)
return { elementType: new Int8(), length: value.length };
if (value instanceof Int16Array)
return { elementType: new Int16(), length: value.length };
if (value instanceof Int32Array)
return { elementType: new Int32(), length: value.length };
return undefined;
}
/** Helper function to convert an Array of JS values to an Arrow Vector */ /** Helper function to convert an Array of JS values to an Arrow Vector */
function makeVector( function makeVector(
values: unknown[], values: unknown[],
@@ -1180,12 +1462,8 @@ export function ensureNestedFieldsExist(
completeRow[field.name] = row[field.name]; completeRow[field.name] = row[field.name];
} }
} else { } else {
// Keep a missing struct valid while filling each of its children with // Field is missing from the data - set to null
// null. This is distinct from an explicitly null struct value. completeRow[field.name] = null;
completeRow[field.name] =
field.type.constructor.name === "Struct"
? ensureStructFieldsExist({}, field.type as Struct)
: null;
} }
} }
@@ -1220,12 +1498,8 @@ function ensureStructFieldsExist(
completeStruct[childField.name] = data[childField.name]; completeStruct[childField.name] = data[childField.name];
} }
} else { } else {
// Keep a missing struct valid while filling each of its children with // Field is missing - set to null
// null. This is distinct from an explicitly null struct value. completeStruct[childField.name] = null;
completeStruct[childField.name] =
childField.type.constructor.name === "Struct"
? ensureStructFieldsExist({}, childField.type as Struct)
: null;
} }
} }
-40
View File
@@ -1,40 +0,0 @@
// SPDX-License-Identifier: Apache-2.0
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
import {
type DataType,
Float32,
Float64,
Int8,
Int16,
Int32,
Uint8,
Uint16,
Uint32,
} from "apache-arrow";
/**
* Map a JS TypedArray instance to the corresponding Arrow element type and
* length. Returns undefined when the view is not a supported TypedArray.
*/
export function typedArrayToArrowType(
value: ArrayBufferView,
): { elementType: DataType; length: number } | undefined {
if (value instanceof Float32Array)
return { elementType: new Float32(), length: value.length };
if (value instanceof Float64Array)
return { elementType: new Float64(), length: value.length };
if (value instanceof Uint8Array)
return { elementType: new Uint8(), length: value.length };
if (value instanceof Uint16Array)
return { elementType: new Uint16(), length: value.length };
if (value instanceof Uint32Array)
return { elementType: new Uint32(), length: value.length };
if (value instanceof Int8Array)
return { elementType: new Int8(), length: value.length };
if (value instanceof Int16Array)
return { elementType: new Int16(), length: value.length };
if (value instanceof Int32Array)
return { elementType: new Int32(), length: value.length };
return undefined;
}
+37 -96
View File
@@ -1,6 +1,7 @@
// SPDX-License-Identifier: Apache-2.0 // SPDX-License-Identifier: Apache-2.0
// SPDX-FileCopyrightText: Copyright The LanceDB Authors // SPDX-FileCopyrightText: Copyright The LanceDB Authors
import { tableFromIPC } from "apache-arrow";
import { import {
Data, Data,
SchemaLike, SchemaLike,
@@ -15,7 +16,6 @@ import {
makeEmptyTable, makeEmptyTable,
} from "./arrow"; } from "./arrow";
import { EmbeddingFunctionConfig, getRegistry } from "./embedding/registry"; import { EmbeddingFunctionConfig, getRegistry } from "./embedding/registry";
import { Job } from "./job";
import { import {
MaterializedView, MaterializedView,
MaterializedViewSelect, MaterializedViewSelect,
@@ -27,16 +27,16 @@ import type {
CreateNamespaceResponse, CreateNamespaceResponse,
DescribeNamespaceResponse, DescribeNamespaceResponse,
DropNamespaceResponse, DropNamespaceResponse,
Job,
JobDescription,
JobInfo, JobInfo,
ListNamespacesResponse, ListNamespacesResponse,
ListTablesResponse,
} from "./native"; } from "./native";
export type { export type {
CreateNamespaceResponse, CreateNamespaceResponse,
DescribeNamespaceResponse, DescribeNamespaceResponse,
DropNamespaceResponse, DropNamespaceResponse,
ListNamespacesResponse, ListNamespacesResponse,
ListTablesResponse,
}; };
import { sanitizeTable } from "./sanitize"; import { sanitizeTable } from "./sanitize";
import { LocalTable, Table } from "./table"; import { LocalTable, Table } from "./table";
@@ -134,10 +134,6 @@ export interface OpenTableOptions {
indexCacheSize?: number; indexCacheSize?: number;
} }
/**
* @deprecated Use {@link ListTablesOptions} with {@link Connection.listTables}
* instead.
*/
export interface TableNamesOptions { export interface TableNamesOptions {
/** /**
* If present, only return names that come lexicographically after the * If present, only return names that come lexicographically after the
@@ -151,24 +147,6 @@ export interface TableNamesOptions {
limit?: number; limit?: number;
} }
export interface ListTablesOptions {
/**
* Token from a previous response, to resume listing where it left off.
*
* The token is opaque: it carries whatever the database needs to resume, and
* callers should not construct or interpret one.
*/
pageToken?: string;
/**
* An upper bound on how many tables to return.
*
* A page may hold fewer than this and still not be the last one, so keep
* going while the response carries a page token rather than while pages are
* full.
*/
limit?: number;
}
export interface ListNamespacesOptions { export interface ListNamespacesOptions {
/** Token from a previous response for pagination. */ /** Token from a previous response for pagination. */
pageToken?: string; pageToken?: string;
@@ -253,7 +231,6 @@ export abstract class Connection {
* @param {Partial<TableNamesOptions>} options - options to control the * @param {Partial<TableNamesOptions>} options - options to control the
* paging / start point (backwards compatibility) * paging / start point (backwards compatibility)
* *
* @deprecated Use {@link Connection.listTables} instead.
*/ */
abstract tableNames(options?: Partial<TableNamesOptions>): Promise<string[]>; abstract tableNames(options?: Partial<TableNamesOptions>): Promise<string[]>;
/** /**
@@ -264,53 +241,12 @@ export abstract class Connection {
* @param {Partial<TableNamesOptions>} options - options to control the * @param {Partial<TableNamesOptions>} options - options to control the
* paging / start point * paging / start point
* *
* @deprecated Use {@link Connection.listTables} instead.
*/ */
abstract tableNames( abstract tableNames(
namespacePath?: string[], namespacePath?: string[],
options?: Partial<TableNamesOptions>, options?: Partial<TableNamesOptions>,
): Promise<string[]>; ): Promise<string[]>;
/**
* List a page of the tables in this database.
*
* To retrieve the tables after the page, pass the `pageToken` the response
* carries back in. A page can be shorter than `limit` without being the last
* one, so walk until a response carries no page token:
*
* ```ts
* const names = [];
* let pageToken = undefined;
* do {
* const page = await conn.listTables({ pageToken, limit: 100 });
* names.push(...page.tables);
* pageToken = page.pageToken;
* } while (pageToken);
* ```
*
* @param {Partial<ListTablesOptions>} options - Pagination options
* (`pageToken`, `limit`).
* @returns {Promise<ListTablesResponse>} A page of table names and an
* optional token for the tables after it.
*/
abstract listTables(
options?: Partial<ListTablesOptions>,
): Promise<ListTablesResponse>;
/**
* List a page of the tables in this database.
*
* @param {string[]} namespacePath - The namespace path to list tables from
* (defaults to root namespace)
* @param {Partial<ListTablesOptions>} options - Pagination options
* (`pageToken`, `limit`).
* @returns {Promise<ListTablesResponse>} A page of table names and an
* optional token for the tables after it.
*/
abstract listTables(
namespacePath?: string[],
options?: Partial<ListTablesOptions>,
): Promise<ListTablesResponse>;
/** /**
* Open a table in the database. * Open a table in the database.
* @param {string} name - The name of the table * @param {string} name - The name of the table
@@ -555,19 +491,24 @@ export abstract class Connection {
): Promise<void>; ): Promise<void>;
/** /**
* Open a server-side job by id, returning a handle with its record already * A {@link Job} handle for a server-side job by id.
* populated. Rejects when the server has no such job, the way
* {@link Connection.openTable} does for a missing table.
* *
* The returned {@link Job} answers for its own state, specification, * The handle is constructed without a server round trip; an unknown id
* result, failure and event history, so there is no separate * surfaces when the handle is used. Dropping the handle has no effect on
* connection-level call for any of them. * the job itself.
*/ */
abstract openJob(jobId: string): Promise<Job>; abstract job(jobId: string): Job;
/** List server-side jobs across the database's tables. */ /** List server-side jobs across the database's tables. */
abstract listJobs(): Promise<JobInfo[]>; abstract listJobs(): Promise<JobInfo[]>;
/**
* Describe a single server-side job by id.
*
* Resolves to `null` when the server has no such job.
*/
abstract getJob(jobId: string): Promise<JobDescription | null>;
/** /**
* Request cancellation of a server-side job by id. * Request cancellation of a server-side job by id.
* *
@@ -575,6 +516,13 @@ export abstract class Connection {
* such job exists. Cancelling an already-terminal job is a no-op success. * such job exists. Cancelling an already-terminal job is a no-op success.
*/ */
abstract cancelJob(jobId: string): Promise<boolean>; abstract cancelJob(jobId: string): Promise<boolean>;
/**
* The lifecycle event history of a server-side job, as an Arrow table.
*
* Lists history across all jobs when `jobId` is omitted.
*/
abstract jobHistory(jobId?: string): Promise<ArrowTable>;
} }
/** @hideconstructor */ /** @hideconstructor */
@@ -653,25 +601,6 @@ export class LocalConnection extends Connection {
return await this.inner.listMaterializedViews(); return await this.inner.listMaterializedViews();
} }
async listTables(
namespacePathOrOptions?: string[] | Partial<ListTablesOptions>,
options?: Partial<ListTablesOptions>,
): Promise<ListTablesResponse> {
// Detect if first argument is namespacePath array or options object
const namespacePath = Array.isArray(namespacePathOrOptions)
? namespacePathOrOptions
: undefined;
const listTablesOptions = Array.isArray(namespacePathOrOptions)
? options
: namespacePathOrOptions;
return this.inner.listTables(
namespacePath ?? [],
listTablesOptions?.pageToken,
listTablesOptions?.limit,
);
}
async openTable( async openTable(
name: string, name: string,
namespacePath?: string[], namespacePath?: string[],
@@ -855,7 +784,7 @@ export class LocalConnection extends Connection {
} }
async dropTableAsync(name: string, namespacePath?: string[]): Promise<Job> { async dropTableAsync(name: string, namespacePath?: string[]): Promise<Job> {
return new Job(await this.inner.dropTableAsync(name, namespacePath ?? [])); return this.inner.dropTableAsync(name, namespacePath ?? []);
} }
async dropAllTables(namespacePath?: string[]): Promise<void> { async dropAllTables(namespacePath?: string[]): Promise<void> {
@@ -914,17 +843,29 @@ export class LocalConnection extends Connection {
); );
} }
async openJob(jobId: string): Promise<Job> { job(jobId: string): Job {
return new Job(await this.inner.openJob(jobId)); return this.inner.job(jobId);
} }
async listJobs(): Promise<JobInfo[]> { async listJobs(): Promise<JobInfo[]> {
return this.inner.listJobs(); return this.inner.listJobs();
} }
async getJob(jobId: string): Promise<JobDescription | null> {
return this.inner.getJob(jobId);
}
async cancelJob(jobId: string): Promise<boolean> { async cancelJob(jobId: string): Promise<boolean> {
return this.inner.cancelJob(jobId); return this.inner.cancelJob(jobId);
} }
async jobHistory(jobId?: string): Promise<ArrowTable> {
const buf = await this.inner.jobHistory(jobId);
if (buf.length === 0) {
return new ArrowTable();
}
return tableFromIPC(buf);
}
} }
/** /**
+2 -42
View File
@@ -4,15 +4,7 @@
import { Field, Schema } from "../arrow"; import { Field, Schema } from "../arrow";
import { sanitizeType } from "../sanitize"; import { sanitizeType } from "../sanitize";
import { EmbeddingFunction } from "./embedding_function"; import { EmbeddingFunction } from "./embedding_function";
import { import { EmbeddingFunctionConfig, getRegistry } from "./registry";
EmbeddingFunctionConfig,
EmbeddingFunctionRegistry,
getRegistry as getGlobalRegistry,
registerBuiltIn,
} from "./registry";
type OpenAIModule = typeof import("./openai");
type TransformersModule = typeof import("./transformers");
export { export {
FieldOptions, FieldOptions,
@@ -22,39 +14,7 @@ export {
EmbeddingFunctionConstructor, EmbeddingFunctionConstructor,
} from "./embedding_function"; } from "./embedding_function";
export { export * from "./registry";
EmbeddingFunctionRegistry,
parseEmbeddingMetadata,
register,
} from "./registry";
export type {
CreateReturnType,
EmbeddingFunctionConfig,
EmbeddingFunctionCreate,
EmbeddingMetadataEntry,
ResolvedEmbeddingFunctionConfig,
} from "./registry";
function initializeBuiltInProviders() {
const { OpenAIEmbeddingFunction } = require("./openai") as OpenAIModule;
const { TransformersEmbeddingFunction } =
require("./transformers") as TransformersModule;
registerBuiltIn("openai", OpenAIEmbeddingFunction);
registerBuiltIn("huggingface", TransformersEmbeddingFunction);
}
/**
* Get the global embedding function registry.
*
* LanceDB built-in providers are initialized when this public API is first
* used, so importing the root package does not change automatic search
* selection for tables without embedding metadata.
*/
export function getRegistry(): EmbeddingFunctionRegistry {
initializeBuiltInProviders();
return getGlobalRegistry();
}
/** /**
* Create a schema with embedding functions. * Create a schema with embedding functions.
+2 -3
View File
@@ -5,13 +5,14 @@ import type OpenAI from "openai";
import type { EmbeddingCreateParams } from "openai/resources/index"; import type { EmbeddingCreateParams } from "openai/resources/index";
import { Float, Float32 } from "../arrow"; import { Float, Float32 } from "../arrow";
import { EmbeddingFunction } from "./embedding_function"; import { EmbeddingFunction } from "./embedding_function";
import { registerBuiltIn } from "./registry"; import { register } from "./registry";
export type OpenAIOptions = { export type OpenAIOptions = {
apiKey: string; apiKey: string;
model: EmbeddingCreateParams["model"]; model: EmbeddingCreateParams["model"];
}; };
@register("openai")
export class OpenAIEmbeddingFunction extends EmbeddingFunction< export class OpenAIEmbeddingFunction extends EmbeddingFunction<
string, string,
Partial<OpenAIOptions> Partial<OpenAIOptions>
@@ -99,5 +100,3 @@ export class OpenAIEmbeddingFunction extends EmbeddingFunction<
return response.data[0].embedding; return response.data[0].embedding;
} }
} }
registerBuiltIn("openai", OpenAIEmbeddingFunction);
+1 -59
View File
@@ -7,10 +7,6 @@ import {
} from "./embedding_function"; } from "./embedding_function";
import "reflect-metadata"; import "reflect-metadata";
const builtInFunctionsKey = Symbol.for(
"@lancedb/lancedb::embedding-built-in-functions::v1",
);
export type CreateReturnType<T> = T extends { init: () => Promise<void> } export type CreateReturnType<T> = T extends { init: () => Promise<void> }
? Promise<T> ? Promise<T>
: T; : T;
@@ -63,15 +59,6 @@ export class EmbeddingFunctionRegistry {
}; };
} }
/** @ignore */
setBuiltIn<
T extends EmbeddingFunctionConstructor = EmbeddingFunctionConstructor,
>(name: string, ctor: T): T {
this.#functions.set(name, ctor);
Reflect.defineMetadata("lancedb::embedding::name", name, ctor);
return ctor;
}
get<T extends EmbeddingFunction<unknown>>( get<T extends EmbeddingFunction<unknown>>(
name: string, name: string,
): EmbeddingFunctionCreate<T> | undefined; ): EmbeddingFunctionCreate<T> | undefined;
@@ -109,7 +96,6 @@ export class EmbeddingFunctionRegistry {
*/ */
reset(this: EmbeddingFunctionRegistry) { reset(this: EmbeddingFunctionRegistry) {
this.#functions.clear(); this.#functions.clear();
getBuiltInFunctions(this).clear();
} }
/** /**
@@ -197,56 +183,12 @@ export class EmbeddingFunctionRegistry {
} }
} }
function getBuiltInFunctions(registry: EmbeddingFunctionRegistry): Set<string> { const _REGISTRY = new EmbeddingFunctionRegistry();
const registryWithBuiltIns = registry as EmbeddingFunctionRegistry & {
[key: symbol]: Set<string> | undefined;
};
let builtInFunctions = registryWithBuiltIns[builtInFunctionsKey];
if (builtInFunctions === undefined) {
builtInFunctions = new Set<string>();
registryWithBuiltIns[builtInFunctionsKey] = builtInFunctions;
}
return builtInFunctions;
}
// Server bundlers can load the side-effect embedding entry points and the public
// embedding API from separate module graphs. Keep their registry shared.
const registryKey = Symbol.for(
"@lancedb/lancedb::embedding-function-registry::v1",
);
const registryGlobal = globalThis as typeof globalThis & {
[key: symbol]: EmbeddingFunctionRegistry | undefined;
};
function getGlobalRegistry(): EmbeddingFunctionRegistry {
const existingRegistry = registryGlobal[registryKey];
if (existingRegistry !== undefined) {
return existingRegistry;
}
const registry = new EmbeddingFunctionRegistry();
registryGlobal[registryKey] = registry;
return registry;
}
const _REGISTRY = getGlobalRegistry();
export function register(name?: string) { export function register(name?: string) {
return _REGISTRY.register(name); return _REGISTRY.register(name);
} }
/** @ignore */
export function registerBuiltIn<
T extends EmbeddingFunctionConstructor = EmbeddingFunctionConstructor,
>(name: string, ctor: T): T {
const builtInFunctions = getBuiltInFunctions(_REGISTRY);
if (builtInFunctions.has(name)) {
return _REGISTRY.setBuiltIn(name, ctor);
}
_REGISTRY.register(name)(ctor);
builtInFunctions.add(name);
return ctor;
}
/** /**
* Utility function to get the global instance of the registry * Utility function to get the global instance of the registry
* @returns `EmbeddingFunctionRegistry` The global instance of the registry * @returns `EmbeddingFunctionRegistry` The global instance of the registry
+2 -3
View File
@@ -3,7 +3,7 @@
import { Float, Float32 } from "../arrow"; import { Float, Float32 } from "../arrow";
import { EmbeddingFunction } from "./embedding_function"; import { EmbeddingFunction } from "./embedding_function";
import { registerBuiltIn } from "./registry"; import { register } from "./registry";
export type XenovaTransformerOptions = { export type XenovaTransformerOptions = {
/** The wasm compatible model to use */ /** The wasm compatible model to use */
@@ -31,6 +31,7 @@ export type XenovaTransformerOptions = {
}; };
}; };
@register("huggingface")
export class TransformersEmbeddingFunction extends EmbeddingFunction< export class TransformersEmbeddingFunction extends EmbeddingFunction<
string, string,
Partial<XenovaTransformerOptions> Partial<XenovaTransformerOptions>
@@ -157,8 +158,6 @@ export class TransformersEmbeddingFunction extends EmbeddingFunction<
} }
} }
registerBuiltIn("huggingface", TransformersEmbeddingFunction);
const tensorDiv = ( const tensorDiv = (
src: import("@huggingface/transformers").Tensor, src: import("@huggingface/transformers").Tensor,
divBy: number, divBy: number,
+7 -6
View File
@@ -81,25 +81,26 @@ export {
Connection, Connection,
CreateTableOptions, CreateTableOptions,
TableNamesOptions, TableNamesOptions,
ListTablesOptions,
OpenTableOptions, OpenTableOptions,
ListNamespacesOptions, ListNamespacesOptions,
CreateNamespaceOptions, CreateNamespaceOptions,
DropNamespaceOptions, DropNamespaceOptions,
ListNamespacesResponse, ListNamespacesResponse,
ListTablesResponse,
CreateNamespaceResponse, CreateNamespaceResponse,
DropNamespaceResponse, DropNamespaceResponse,
DescribeNamespaceResponse, DescribeNamespaceResponse,
RenameTableOptions, RenameTableOptions,
} from "./connection"; } from "./connection";
export { JobFailureInfo, JobInfo, Session } from "./native.js"; export {
Job,
export { Job, JobEventsOptions } from "./job"; JobDescription,
JobFailureInfo,
JobInfo,
Session,
} from "./native.js";
export { export {
AutoQuery,
ExecutableQuery, ExecutableQuery,
Query, Query,
QueryBase, QueryBase,
-188
View File
@@ -1,188 +0,0 @@
// SPDX-License-Identifier: Apache-2.0
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
import { Table as ArrowTable, tableFromIPC } from "apache-arrow";
import { JobFailureInfo, Job as NativeJob } from "./native";
/** Which of a job's events {@link Job.events} returns. */
export interface JobEventsOptions {
/** Maximum event rows to return, up to the server maximum of 10,000. */
limit?: number;
/** SQL-like filter over the event columns. */
filter?: string;
}
/**
* A handle to an operation that may still be running.
*
* The operation may already be complete when the handle is created.
*
* The detail getters read what the handle last observed. Submitting an
* operation returns only a job id, so populating them eagerly would cost an
* extra round trip on every call:
*
* - {@link Job.refresh} and {@link Job.status} fetch the whole record.
* - {@link Job.wait} records the terminal state it establishes, but not the
* rest of the record.
* - Everything is null until one of those runs.
*
* @hideconstructor
*/
export class Job {
private readonly inner: NativeJob;
constructor(inner: NativeJob) {
this.inner = inner;
}
/**
* Identifies the operation on the server that is running it.
*
* Operations that run in this process have no server id. The value is
* opaque: parsing it or storing it to resume the job later is not supported.
*/
get id(): string | null {
return this.inner.id ?? null;
}
/** The last observed lifecycle state, without contacting the backend. */
get state(): string | null {
return this.inner.state ?? null;
}
/**
* The job's type, as the server names it. Null for an in-process job, which
* has no server-side record.
*/
get jobType(): string | null {
return this.inner.jobType ?? null;
}
/** When the job was created, in milliseconds since the epoch. */
get creationMs(): number | null {
return this.inner.creationMs ?? null;
}
/** The job-type-specific specification it was submitted with. */
// biome-ignore lint/suspicious/noExplicitAny: shape varies by job type
get spec(): any | null {
return parseJson(this.inner.specJson);
}
/**
* The job-type-specific terminal result. Null until the job succeeds, so a
* job that never terminates reports its progress through {@link Job.events}
* instead.
*/
// biome-ignore lint/suspicious/noExplicitAny: shape varies by job type
get result(): any | null {
return parseJson(this.inner.resultJson);
}
/** Why the job failed, when it failed and the server reports a reason. */
get failure(): JobFailureInfo | null {
return this.inner.failure ?? null;
}
/**
* The operation's current lifecycle state: "running", "finished", "failed",
* or "cancelled".
*
* A point snapshot; unlike {@link Job.wait} it does not block or reject on a
* terminal failure state. Also refreshes the getters above.
*/
async status(): Promise<string> {
return this.inner.status();
}
/** Wait until the operation reaches a terminal state. */
async wait(): Promise<void> {
return this.inner.wait();
}
/** Request cancellation. Cancelling a finished operation is a no-op. */
async cancel(): Promise<void> {
return this.inner.cancel();
}
/**
* Ask the backend for this job's current state, and for a server-side job
* its full record, then cache it for the getters above.
*/
async refresh(): Promise<void> {
return this.inner.refresh();
}
/**
* This job's recorded lifecycle events.
*
* Where the getters above report a terminal result only once the job reaches
* one, events are written as the job runs and outlive the workers that
* produced them. A distributed job records a `claim`/`claim_complete` pair
* per unit of work, each carrying `rows_processed`, so a job that never
* finishes still accounts for what it did.
*
* The server caps results at 1000 rows by default and 10,000 at most, and
* truncates without saying so, so pass `limit` for a job that emits an event
* per fragment. `filter` is a SQL-like expression over the `state`,
* `updated_by`, `emitted_from`, `emitted_by`, and `claim_entity` columns.
*/
async events(options?: JobEventsOptions): Promise<ArrowTable> {
const buf = await this.inner.events(options?.limit, options?.filter);
if (buf.length === 0) {
return new ArrowTable();
}
return tableFromIPC(buf);
}
/**
* Every field the handle currently knows, one per line, with the JSON
* payloads indented -- a refresh job's spec and result are the point of
* printing it.
*/
toString(): string {
if (this.state === null) {
const known = this.id === null ? "" : `id=${JSON.stringify(this.id)}, `;
return `Job(${known}not refreshed)`;
}
const fields: string[] = [];
if (this.id !== null) {
fields.push(`id=${JSON.stringify(this.id)}`);
}
fields.push(`state=${JSON.stringify(this.state)}`);
if (this.jobType !== null) {
fields.push(`jobType=${JSON.stringify(this.jobType)}`);
}
if (this.creationMs !== null) {
fields.push(`creationMs=${this.creationMs}`);
}
for (const [name, value] of [
["spec", this.spec],
["result", this.result],
] as const) {
if (value !== null) {
fields.push(`${name}=${indentJson(value)}`);
}
}
if (this.failure !== null) {
fields.push(`failure=${indentJson(this.failure)}`);
}
return `Job(${fields.map((field) => `\n${REPR_INDENT}${field},`).join("")}\n)`;
}
[Symbol.for("nodejs.util.inspect.custom")](): string {
return this.toString();
}
}
const REPR_INDENT = " ";
// biome-ignore lint/suspicious/noExplicitAny: shape varies by job type
function indentJson(value: any): string {
return JSON.stringify(value, null, 4).replace(/\n/g, `\n${REPR_INDENT}`);
}
// biome-ignore lint/suspicious/noExplicitAny: shape varies by job type
function parseJson(raw: string | null | undefined): any | null {
return raw === null || raw === undefined ? null : JSON.parse(raw);
}
+1 -5
View File
@@ -19,8 +19,6 @@ export interface MaterializedViewDefinition {
limit?: number; limit?: number;
/** Source columns the projections and filter read. */ /** Source columns the projections and filter read. */
inputs: string[]; inputs: string[];
/** Namespace holding the source table; empty is the root namespace. */
sourceNamespace: string[];
} }
/** /**
@@ -80,8 +78,7 @@ export function definitionFromMetadata(
} }
// biome-ignore lint/suspicious/noExplicitAny: raw JSON // biome-ignore lint/suspicious/noExplicitAny: raw JSON
const value: any = JSON.parse(raw); const value: any = JSON.parse(raw);
// "namespaced_select" keeps older readers from resolving the source at root. if (value.kind !== "select") {
if (value.kind !== "select" && value.kind !== "namespaced_select") {
throw new Error( throw new Error(
`materialized view '${name}' is defined by '${value.kind}', which this ` + `materialized view '${name}' is defined by '${value.kind}', which this ` +
"version of lancedb cannot refresh", "version of lancedb cannot refresh",
@@ -106,7 +103,6 @@ export function definitionFromMetadata(
filter: value.filter ?? undefined, filter: value.filter ?? undefined,
limit, limit,
inputs: value.inputs ?? [], inputs: value.inputs ?? [],
sourceNamespace: value.source_namespace ?? [],
}; };
} }
+106 -205
View File
@@ -100,29 +100,6 @@ export interface FullTextSearchOptions {
columns?: string | string[]; columns?: string | string[];
} }
function nearestToNative(
inner: NativeQuery,
vector: Awaited<IntoVector>,
): NativeVectorQuery {
const raw = Array.isArray(vector) ? null : extractVectorBuffer(vector);
if (raw) {
return inner.nearestToRaw(raw.data, raw.dtype);
}
return inner.nearestTo(Float32Array.from(vector as number[]));
}
function addQueryVectorToNative(
inner: NativeVectorQuery,
vector: Awaited<IntoVector>,
) {
const raw = Array.isArray(vector) ? null : extractVectorBuffer(vector);
if (raw) {
inner.addQueryVectorRaw(raw.data, raw.dtype);
} else {
inner.addQueryVector(Float32Array.from(vector as number[]));
}
}
/** Common methods supported by all query types /** Common methods supported by all query types
* *
* @see {@link Query} * @see {@link Query}
@@ -134,15 +111,13 @@ export class QueryBase<
NativeQueryType extends NativeQuery | NativeVectorQuery | NativeTakeQuery, NativeQueryType extends NativeQuery | NativeVectorQuery | NativeTakeQuery,
> implements AsyncIterable<RecordBatch> > implements AsyncIterable<RecordBatch>
{ {
protected inner!: NativeQueryType | Promise<NativeQueryType>;
/** /**
* @hidden * @hidden
*/ */
protected constructor(inner?: NativeQueryType | Promise<NativeQueryType>) { protected constructor(
if (inner !== undefined) { protected inner: NativeQueryType | Promise<NativeQueryType>,
this.inner = inner; ) {
} // intentionally empty
} }
// call a function on the inner (either a promise or the actual object) // call a function on the inner (either a promise or the actual object)
@@ -160,15 +135,6 @@ export class QueryBase<
} }
} }
/**
* Return the native query used by the next terminal operation.
*
* @hidden
*/
protected async getInner(): Promise<NativeQueryType> {
return this.inner;
}
/** /**
* Return only the specified columns. * Return only the specified columns.
* *
@@ -241,11 +207,16 @@ export class QueryBase<
/** /**
* @hidden * @hidden
*/ */
protected async nativeExecute( protected nativeExecute(
options?: Partial<QueryExecutionOptions>, options?: Partial<QueryExecutionOptions>,
): Promise<NativeBatchIterator> { ): Promise<NativeBatchIterator> {
const inner = await this.getInner(); if (this.inner instanceof Promise) {
return inner.execute(options?.maxBatchLength, options?.timeoutMs); return this.inner.then((inner) =>
inner.execute(options?.maxBatchLength, options?.timeoutMs),
);
} else {
return this.inner.execute(options?.maxBatchLength, options?.timeoutMs);
}
} }
/** /**
@@ -274,7 +245,12 @@ export class QueryBase<
/** Collect the results as an Arrow @see {@link ArrowTable}. */ /** Collect the results as an Arrow @see {@link ArrowTable}. */
async toArrow(options?: Partial<QueryExecutionOptions>): Promise<ArrowTable> { async toArrow(options?: Partial<QueryExecutionOptions>): Promise<ArrowTable> {
const batches = []; const batches = [];
const inner = await this.getInner(); let inner;
if (this.inner instanceof Promise) {
inner = await this.inner;
} else {
inner = this.inner;
}
for await (const batch of new RecordBatchIterable(inner, options)) { for await (const batch of new RecordBatchIterable(inner, options)) {
batches.push(batch); batches.push(batch);
} }
@@ -303,8 +279,11 @@ export class QueryBase<
* @returns A Promise that resolves to a string containing the query execution plan explanation. * @returns A Promise that resolves to a string containing the query execution plan explanation.
*/ */
async explainPlan(verbose = false): Promise<string> { async explainPlan(verbose = false): Promise<string> {
const inner = await this.getInner(); if (this.inner instanceof Promise) {
return inner.explainPlan(verbose); return this.inner.then((inner) => inner.explainPlan(verbose));
} else {
return this.inner.explainPlan(verbose);
}
} }
/** /**
@@ -342,8 +321,13 @@ export class QueryBase<
distributedMetrics?: AnalyzePlanDistributedMetrics, distributedMetrics?: AnalyzePlanDistributedMetrics,
): Promise<string> { ): Promise<string> {
const distributedMetricsMode = distributedMetrics ?? "aggregate"; const distributedMetricsMode = distributedMetrics ?? "aggregate";
const inner = await this.getInner(); if (this.inner instanceof Promise) {
return inner.analyzePlan(distributedMetricsMode); return this.inner.then((inner) =>
inner.analyzePlan(distributedMetricsMode),
);
} else {
return this.inner.analyzePlan(distributedMetricsMode);
}
} }
/** /**
@@ -355,8 +339,12 @@ export class QueryBase<
* @returns An Arrow Schema describing the output columns. * @returns An Arrow Schema describing the output columns.
*/ */
async outputSchema(): Promise<import("./arrow").Schema> { async outputSchema(): Promise<import("./arrow").Schema> {
const inner = await this.getInner(); let schemaBuffer: Buffer;
const schemaBuffer = await inner.outputSchema(); if (this.inner instanceof Promise) {
schemaBuffer = await this.inner.then((inner) => inner.outputSchema());
} else {
schemaBuffer = await this.inner.outputSchema();
}
const schema = tableFromIPC(schemaBuffer).schema; const schema = tableFromIPC(schemaBuffer).schema;
return schema; return schema;
} }
@@ -368,7 +356,7 @@ export class StandardQueryBase<
extends QueryBase<NativeQueryType> extends QueryBase<NativeQueryType>
implements ExecutableQuery implements ExecutableQuery
{ {
constructor(inner?: NativeQueryType | Promise<NativeQueryType>) { constructor(inner: NativeQueryType | Promise<NativeQueryType>) {
super(inner); super(inner);
} }
@@ -522,13 +510,6 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
super(inner); super(inner);
} }
/**
* @hidden
*/
protected doVectorCall(fn: (inner: NativeVectorQuery) => void) {
super.doCall(fn);
}
/** /**
* Set the number of partitions to search (probe) * Set the number of partitions to search (probe)
* *
@@ -556,7 +537,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
* the minimum and maximum to the same value. * the minimum and maximum to the same value.
*/ */
nprobes(nprobes: number): VectorQuery { nprobes(nprobes: number): VectorQuery {
this.doVectorCall((inner) => inner.nprobes(nprobes)); super.doCall((inner) => inner.nprobes(nprobes));
return this; return this;
} }
@@ -570,7 +551,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
* but will also increase latency. * but will also increase latency.
*/ */
minimumNprobes(minimumNprobes: number): VectorQuery { minimumNprobes(minimumNprobes: number): VectorQuery {
this.doVectorCall((inner) => inner.minimumNprobes(minimumNprobes)); super.doCall((inner) => inner.minimumNprobes(minimumNprobes));
return this; return this;
} }
@@ -584,7 +565,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
* potential false negatives. * potential false negatives.
*/ */
maximumNprobes(maximumNprobes: number): VectorQuery { maximumNprobes(maximumNprobes: number): VectorQuery {
this.doVectorCall((inner) => inner.maximumNprobes(maximumNprobes)); super.doCall((inner) => inner.maximumNprobes(maximumNprobes));
return this; return this;
} }
@@ -597,7 +578,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
* `undefined` means no lower or upper bound. * `undefined` means no lower or upper bound.
*/ */
distanceRange(lowerBound?: number, upperBound?: number): VectorQuery { distanceRange(lowerBound?: number, upperBound?: number): VectorQuery {
this.doVectorCall((inner) => inner.distanceRange(lowerBound, upperBound)); super.doCall((inner) => inner.distanceRange(lowerBound, upperBound));
return this; return this;
} }
@@ -611,7 +592,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
* also increase the latency of your query. The default value is 1.5*limit. * also increase the latency of your query. The default value is 1.5*limit.
*/ */
ef(ef: number): VectorQuery { ef(ef: number): VectorQuery {
this.doVectorCall((inner) => inner.ef(ef)); super.doCall((inner) => inner.ef(ef));
return this; return this;
} }
@@ -625,7 +606,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
* whose data type is a fixed-size-list of floats. * whose data type is a fixed-size-list of floats.
*/ */
column(column: string): VectorQuery { column(column: string): VectorQuery {
this.doVectorCall((inner) => inner.column(column)); super.doCall((inner) => inner.column(column));
return this; return this;
} }
@@ -646,7 +627,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
distanceType( distanceType(
distanceType: Required<IvfPqOptions>["distanceType"], distanceType: Required<IvfPqOptions>["distanceType"],
): VectorQuery { ): VectorQuery {
this.doVectorCall((inner) => inner.distanceType(distanceType)); super.doCall((inner) => inner.distanceType(distanceType));
return this; return this;
} }
@@ -680,7 +661,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
* distance between the query vector and the actual uncompressed vector. * distance between the query vector and the actual uncompressed vector.
*/ */
refineFactor(refineFactor: number): VectorQuery { refineFactor(refineFactor: number): VectorQuery {
this.doVectorCall((inner) => inner.refineFactor(refineFactor)); super.doCall((inner) => inner.refineFactor(refineFactor));
return this; return this;
} }
@@ -705,7 +686,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
* factor can often help restore some of the results lost by post filtering. * factor can often help restore some of the results lost by post filtering.
*/ */
postfilter(): VectorQuery { postfilter(): VectorQuery {
this.doVectorCall((inner) => inner.postfilter()); super.doCall((inner) => inner.postfilter());
return this; return this;
} }
@@ -719,7 +700,7 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
* calculate your recall to select an appropriate value for nprobes. * calculate your recall to select an appropriate value for nprobes.
*/ */
bypassVectorIndex(): VectorQuery { bypassVectorIndex(): VectorQuery {
this.doVectorCall((inner) => inner.bypassVectorIndex()); super.doCall((inner) => inner.bypassVectorIndex());
return this; return this;
} }
@@ -727,39 +708,43 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
* Add a query vector to the search * Add a query vector to the search
* *
* This method can be called multiple times to add multiple query vectors * This method can be called multiple times to add multiple query vectors
* to the search. A column called `query_index` will be added to indicate the index * to the search. If multiple query vectors are added, then they will be searched
* of the query vector that produced the result. Flat searches share one table scan * in parallel, and the results will be concatenated. A column called `query_index`
* across the query vectors, avoiding the scan and memory amplification of running * will be added to indicate the index of the query vector that produced the result.
* multiple queries concurrently. Indexed searches may still perform per-vector *
* index work. * Performance wise, this is equivalent to running multiple queries concurrently.
*/ */
addQueryVector(vector: IntoVector): VectorQuery { addQueryVector(vector: IntoVector): VectorQuery {
if (vector instanceof Promise) { if (vector instanceof Promise) {
// Observe the promise as soon as it is accepted. The existing native
// query may still be pending, and delaying observation until it resolves
// can otherwise surface a fast rejection as unhandled.
const settledVector = vector.then(
(value) => ({ status: "fulfilled" as const, value }),
(reason) => ({ status: "rejected" as const, reason }),
);
const res = (async () => { const res = (async () => {
const inner = await this.getInner(); try {
const outcome = await settledVector; const v = await vector;
if (outcome.status === "rejected") { // biome-ignore lint/suspicious/noExplicitAny: we need to get the `inner`, but js has no package scoping
throw outcome.reason; const value: any = this.addQueryVector(v);
const inner = value.inner as
| NativeVectorQuery
| Promise<NativeVectorQuery>;
return inner;
} catch (e) {
return Promise.reject(e);
} }
addQueryVectorToNative(inner, outcome.value);
return inner;
})(); })();
return new VectorQuery(res); return new VectorQuery(res);
} else { } else {
this.doVectorCall((inner) => addQueryVectorToNative(inner, vector)); super.doCall((inner) => {
const raw = Array.isArray(vector) ? null : extractVectorBuffer(vector);
if (raw) {
inner.addQueryVectorRaw(raw.data, raw.dtype);
} else {
inner.addQueryVector(Float32Array.from(vector as number[]));
}
});
return this; return this;
} }
} }
rerank(reranker: Reranker): VectorQuery { rerank(reranker: Reranker): VectorQuery {
this.doVectorCall((inner) => super.doCall((inner) =>
inner.rerank(async (args) => { inner.rerank(async (args) => {
const vecResults = await fromBufferToRecordBatch(args.vecResults); const vecResults = await fromBufferToRecordBatch(args.vecResults);
const ftsResults = await fromBufferToRecordBatch(args.ftsResults); const ftsResults = await fromBufferToRecordBatch(args.ftsResults);
@@ -778,71 +763,6 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
} }
} }
/**
* Create a string query whose vector/FTS routing is resolved against the active
* table schema when the query executes.
*
* @hidden
*/
export function createAutoQuery(
table: NativeTable,
query: string,
columns: string[] | null,
getVector: (metadata: string) => Promise<Awaited<IntoVector>>,
): AutoQuery {
type RouteSnapshot = {
table: NativeTable;
embeddingMetadata: string | undefined;
};
type CachedPreparation = {
metadata: string;
vector: Promise<Awaited<IntoVector>>;
};
let cachedPreparation: CachedPreparation | undefined;
const snapshotRoute = async (): Promise<RouteSnapshot> => {
const snapshot = await table.querySnapshot();
const schema = tableFromIPC(await snapshot.schema()).schema;
return {
table: snapshot,
embeddingMetadata: schema.metadata.get("embedding_functions"),
};
};
const createInner = async (): Promise<NativeQuery | NativeVectorQuery> => {
const route = await snapshotRoute();
if (route.embeddingMetadata === undefined) {
const inner = route.table.query();
inner.fullTextSearch({ query, columns });
return inner;
}
const metadata = route.embeddingMetadata;
if (cachedPreparation?.metadata !== metadata) {
cachedPreparation = {
metadata,
vector: Promise.resolve().then(() => getVector(metadata)),
};
}
const preparation = cachedPreparation;
let vector: Awaited<IntoVector>;
try {
vector = await preparation.vector;
} catch (error) {
if (cachedPreparation === preparation) {
cachedPreparation = undefined;
}
throw error;
}
return nearestToNative(route.table.query(), vector);
};
return new AutoQuery(createInner);
}
/** /**
* A query that returns a subset of the rows in the table. * A query that returns a subset of the rows in the table.
* *
@@ -868,51 +788,6 @@ export class TakeQuery extends QueryBase<NativeTakeQuery> {
} }
} }
/**
* A builder for automatic string searches.
*
* Automatic search determines whether to use full-text or vector search from
* the table revision selected for each execution. This builder exposes the
* common operations supported by both query families.
*
* @hideconstructor
*/
export class AutoQuery extends StandardQueryBase<
NativeQuery | NativeVectorQuery
> {
private readonly calls: Array<
(inner: NativeQuery | NativeVectorQuery) => void
> = [];
/** @hidden */
constructor(
private readonly createInner: () => Promise<
NativeQuery | NativeVectorQuery
>,
) {
super();
}
/** @hidden */
protected override doCall(
fn: (inner: NativeQuery | NativeVectorQuery) => void,
) {
this.calls.push(fn);
}
/** @hidden */
protected override async getInner(): Promise<
NativeQuery | NativeVectorQuery
> {
const calls = [...this.calls];
const inner = await this.createInner();
for (const call of calls) {
call(inner);
}
return inner;
}
}
/** A builder for LanceDB queries. /** A builder for LanceDB queries.
* *
* @see {@link Table#query}, {@link Table#search} * @see {@link Table#query}, {@link Table#search}
@@ -965,19 +840,45 @@ export class Query extends StandardQueryBase<NativeQuery> {
* a default `limit` of 10 will be used. @see {@link Query#limit} * a default `limit` of 10 will be used. @see {@link Query#limit}
*/ */
nearestTo(vector: IntoVector): VectorQuery { nearestTo(vector: IntoVector): VectorQuery {
const inner = this.inner; const callNearestTo = (
if (inner instanceof Promise) { inner: NativeQuery,
const nativeQuery = inner.then(async (resolvedInner) => resolved: Float32Array | Float64Array | Uint8Array | number[],
nearestToNative(resolvedInner, await vector), ): NativeVectorQuery => {
); const raw = Array.isArray(resolved)
? null
: extractVectorBuffer(resolved);
if (raw) {
return inner.nearestToRaw(raw.data, raw.dtype);
}
return inner.nearestTo(Float32Array.from(resolved as number[]));
};
if (this.inner instanceof Promise) {
const nativeQuery = this.inner.then(async (inner) => {
const resolved = vector instanceof Promise ? await vector : vector;
return callNearestTo(inner, resolved);
});
return new VectorQuery(nativeQuery); return new VectorQuery(nativeQuery);
} }
if (vector instanceof Promise) { if (vector instanceof Promise) {
return new VectorQuery( const res = (async () => {
vector.then((resolvedVector) => nearestToNative(inner, resolvedVector)), try {
); const v = await vector;
// biome-ignore lint/suspicious/noExplicitAny: we need to get the `inner`, but js has no package scoping
const value: any = this.nearestTo(v);
const inner = value.inner as
| NativeVectorQuery
| Promise<NativeVectorQuery>;
return inner;
} catch (e) {
return Promise.reject(e);
}
})();
return new VectorQuery(res);
} else {
const vectorQuery = callNearestTo(this.inner, vector);
return new VectorQuery(vectorQuery);
} }
return new VectorQuery(nearestToNative(inner, vector));
} }
nearestToText(query: string | FullTextQuery, columns?: string[]): Query { nearestToText(query: string | FullTextQuery, columns?: string[]): Query {
+4 -11
View File
@@ -94,24 +94,17 @@ export function sanitizeMetadata(
if (metadataLike === undefined || metadataLike === null) { if (metadataLike === undefined || metadataLike === null) {
return undefined; return undefined;
} }
if (!(metadataLike instanceof Map)) {
let entries: IterableIterator<[unknown, unknown]>;
try {
entries = Map.prototype.entries.call(metadataLike);
} catch {
throw Error("Expected metadata, if present, to be a Map<string, string>"); throw Error("Expected metadata, if present, to be a Map<string, string>");
} }
for (const item of metadataLike) {
const metadata = new Map<string, string>(); if (typeof item[0] !== "string" || typeof item[1] !== "string") {
for (const [key, value] of entries) {
if (typeof key !== "string" || typeof value !== "string") {
throw Error( throw Error(
"Expected metadata, if present, to be a Map<string, string> but it had non-string keys or values", "Expected metadata, if present, to be a Map<string, string> but it had non-string keys or values",
); );
} }
metadata.set(key, value);
} }
return metadata; return metadataLike as Map<string, string>;
} }
export function sanitizeInt(typeLike: object) { export function sanitizeInt(typeLike: object) {
-567
View File
@@ -1,567 +0,0 @@
// SPDX-License-Identifier: Apache-2.0
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
import {
Binary,
Bool,
DataType,
Dictionary,
Field,
FixedSizeList,
Float32,
Float64,
Int32,
Int64,
List,
Schema,
Struct,
Utf8,
util as arrowUtil,
} from "apache-arrow";
import { typedArrayToArrowType } from "./arrow_type";
import { sanitizeType } from "./sanitize";
type InferenceOptions = {
dictionaryEncodeStrings: boolean;
vectorColumns: Record<string, { type: unknown }>;
};
/**
* Infer the Arrow schema represented by a set of records.
*
* This is the intentionally small interface to schema inference. The stateful
* details of combining partial type evidence are encapsulated below so callers
* only need to provide records, an optional schema, and inference options.
*/
export function inferSchema(
data: Array<Record<string, unknown>>,
schema: Schema | undefined,
options: InferenceOptions,
): Schema {
return new SchemaInferrer(schema, options).infer(data);
}
class SchemaInferrer {
private readonly fields = new FieldTree();
constructor(
private readonly providedSchema: Schema | undefined,
private readonly options: InferenceOptions,
) {}
infer(data: Array<Record<string, unknown>>): Schema {
for (const [row, record] of data.entries()) {
for (const [path, value] of recordPathsAndValues(record)) {
this.observe(path, value, row);
}
}
return this.providedSchema === undefined
? new Schema(fieldsFromTree(this.fields))
: new Schema(matchingFields(this.providedSchema.fields, this.fields));
}
private observe(path: string[], value: unknown, row: number): void {
const current = this.fields.get(path);
if (current === undefined) {
this.addField(path, value, row);
} else if (this.providedSchema === undefined) {
this.updateInferredField(path, value, row, current);
}
}
private addField(path: string[], value: unknown, row: number): void {
if (this.providedSchema !== undefined) {
this.addSchemaField(this.providedSchema, path, row);
return;
}
const evidence =
this.inferType(value, path) ?? DeferredTypeEvidence.from(value, row);
if (evidence === undefined) {
throw typeInferenceError(path, row);
}
const conflict = this.fields.set(
path,
evidence,
(existing) =>
existing instanceof DeferredTypeEvidence && existing.isOnlyNulls(),
);
if (conflict !== undefined) {
throw branchConflictError(conflict, row, "Struct");
}
}
private addSchemaField(schema: Schema, path: string[], row: number): void {
const field = fieldAtPath(schema, path);
if (field === undefined) {
throw new Error(
`Found field not in schema: ${path.join(".")} at row ${row}`,
);
}
const conflict = this.fields.set(path, field.type);
if (conflict !== undefined) {
throw branchConflictError(conflict, row, "Struct");
}
}
private updateInferredField(
path: string[],
value: unknown,
row: number,
current: FieldNode,
): void {
const newType = this.inferType(value, path);
const deferred = DeferredTypeEvidence.from(value, row);
if (current instanceof FieldTree) {
if (deferred?.isOnlyNulls()) {
return;
}
throw schemaInferenceError(
path,
row,
"Struct",
describeEvidence(newType ?? deferred),
);
}
if (current instanceof DeferredTypeEvidence) {
this.resolveDeferredField(path, row, current, newType, deferred);
return;
}
if (newType !== undefined) {
if (!inferredTypesEqual(current, newType)) {
throw schemaInferenceError(
path,
row,
describeEvidence(current),
describeEvidence(newType),
);
}
return;
}
if (deferred === undefined || !deferred.matches(current)) {
throw schemaInferenceError(
path,
row,
describeEvidence(current),
describeEvidence(deferred),
);
}
}
private resolveDeferredField(
path: string[],
row: number,
current: DeferredTypeEvidence,
newType: DataType | undefined,
deferred: DeferredTypeEvidence | undefined,
): void {
if (newType !== undefined) {
if (!current.matches(newType)) {
throw schemaInferenceError(
path,
row,
current.describe(),
describeEvidence(newType),
);
}
this.fields.set(path, newType);
return;
}
if (deferred !== undefined) {
this.fields.set(path, current.merge(deferred));
return;
}
throw schemaInferenceError(
path,
row,
current.describe(),
describeEvidence(newType),
);
}
private inferType(value: unknown, path: string[]): DataType | undefined {
if (typeof value === "bigint") {
return new Int64();
}
if (typeof value === "number") {
return new Float64();
}
if (typeof value === "string") {
return this.options.dictionaryEncodeStrings
? new Dictionary(new Utf8(), new Int32())
: new Utf8();
}
if (typeof value === "boolean") {
return new Bool();
}
if (value instanceof Buffer) {
return new Binary();
}
if (ArrayBuffer.isView(value) && !(value instanceof DataView)) {
const typedArray = typedArrayToArrowType(value);
return typedArray === undefined
? undefined
: new FixedSizeList(
typedArray.length,
new Field("item", typedArray.elementType, true),
);
}
if (!Array.isArray(value) || value.length === 0) {
return undefined;
}
const configuredVector =
path.length === 1 ? this.options.vectorColumns[path[0]] : undefined;
if (configuredVector !== undefined) {
return new FixedSizeList(
value.length,
new Field("item", sanitizeType(configuredVector.type), true),
);
}
const itemType = this.inferArrayItemType(value, path);
if (itemType === undefined) {
return undefined;
}
return nameSuggestsVectorColumn(path[path.length - 1])
? new FixedSizeList(value.length, new Field("item", new Float32(), true))
: new List(new Field("item", itemType, true));
}
private inferArrayItemType(
values: unknown[],
path: string[],
): DataType | undefined {
let itemType: DataType | undefined;
const deferredItems: unknown[] = [];
for (const value of values) {
const candidate = this.inferType(value, path);
if (candidate === undefined) {
if (!isDeferredValue(value)) {
return undefined;
}
deferredItems.push(value);
} else if (itemType === undefined) {
itemType = candidate;
} else if (!inferredTypesEqual(itemType, candidate)) {
return undefined;
}
}
if (itemType === undefined) {
return undefined;
}
return deferredItems.every((value) =>
deferredValueMatchesType(value, itemType),
)
? itemType
: undefined;
}
}
/** Nulls and empty/all-null lists that do not determine a type by themselves. */
class DeferredTypeEvidence {
private constructor(
private readonly values: Array<{ value: unknown; row: number }>,
) {}
static from(value: unknown, row: number): DeferredTypeEvidence | undefined {
return isDeferredValue(value)
? new DeferredTypeEvidence([{ value, row }])
: undefined;
}
isOnlyNulls(): boolean {
return this.values.every(({ value }) => value == null);
}
matches(type: DataType): boolean {
return this.values.every(({ value }) =>
deferredValueMatchesType(value, type),
);
}
merge(other: DeferredTypeEvidence): DeferredTypeEvidence {
return new DeferredTypeEvidence([...this.values, ...other.values]);
}
describe(): string {
const list = this.values.find(({ value }) => Array.isArray(value));
return list === undefined
? "null"
: `List[${(list.value as unknown[]).length}]`;
}
firstRow(): number {
return this.values[0].row;
}
}
type FieldNode = DataType | DeferredTypeEvidence | FieldTree;
type LeafNode = Exclude<FieldNode, FieldTree>;
type FieldConflict = { path: string[]; value: FieldNode };
/** Nested field state, kept separate from Arrow's eventual Struct types. */
class FieldTree {
private readonly children = new Map<string, FieldNode>();
get(path: string[]): FieldNode | undefined {
let current: FieldNode = this;
for (const part of path) {
if (!(current instanceof FieldTree)) {
return undefined;
}
const child = current.children.get(part);
if (child === undefined) {
return undefined;
}
current = child;
}
return current;
}
set(
path: string[],
value: LeafNode,
canReplaceLeaf: (value: LeafNode) => boolean = () => false,
): FieldConflict | undefined {
let branch: FieldTree = this;
for (const [index, part] of path.slice(0, -1).entries()) {
const child = branch.children.get(part);
if (child === undefined || (isLeaf(child) && canReplaceLeaf(child))) {
const nextBranch = new FieldTree();
branch.children.set(part, nextBranch);
branch = nextBranch;
} else if (child instanceof FieldTree) {
branch = child;
} else {
return { path: path.slice(0, index + 1), value: child };
}
}
const name = path[path.length - 1];
const current = branch.children.get(name);
if (current instanceof FieldTree) {
return { path, value: current };
}
branch.children.set(name, value);
return undefined;
}
entries(): IterableIterator<[string, FieldNode]> {
return this.children.entries();
}
has(name: string): boolean {
return this.children.has(name);
}
}
function isLeaf(value: FieldNode): value is LeafNode {
return !(value instanceof FieldTree);
}
function fieldsFromTree(tree: FieldTree, path: string[] = []): Field[] {
const fields: Field[] = [];
for (const [name, value] of tree.entries()) {
if (value instanceof FieldTree) {
fields.push(
new Field(
name,
new Struct(fieldsFromTree(value, [...path, name])),
true,
),
);
} else if (value instanceof DeferredTypeEvidence) {
throw typeInferenceError([...path, name], value.firstRow());
} else {
fields.push(new Field(name, value, true));
}
}
return fields;
}
function matchingFields(fields: Field[], tree: FieldTree): Field[] {
const matches: Field[] = [];
for (const field of fields) {
if (!tree.has(field.name)) {
continue;
}
const value = tree.get([field.name]);
if (value instanceof FieldTree) {
const struct = field.type as Struct;
matches.push(
new Field(
field.name,
new Struct(matchingFields(struct.children, value)),
field.nullable,
field.metadata,
),
);
} else {
matches.push(field);
}
}
return matches;
}
function* recordPathsAndValues(
record: Record<string, unknown>,
path: string[] = [],
): Generator<[string[], unknown]> {
for (const [name, value] of Object.entries(record)) {
if (isRecord(value)) {
yield* recordPathsAndValues(value, [...path, name]);
} else if (value !== undefined) {
yield [[...path, name], value];
}
}
}
function isRecord(value: unknown): value is Record<string, unknown> {
return (
typeof value === "object" &&
value !== null &&
!Array.isArray(value) &&
!(value instanceof RegExp) &&
!(value instanceof Date) &&
!(value instanceof Set) &&
!(value instanceof Map) &&
!(value instanceof Buffer) &&
!ArrayBuffer.isView(value)
);
}
function fieldAtPath(schema: Schema, path: string[]): Field | undefined {
let fields = schema.fields;
let field: Field | undefined;
for (const [index, name] of path.entries()) {
field = fields.find((candidate) => candidate.name === name);
if (field === undefined || index === path.length - 1) {
return field;
}
if (!DataType.isStruct(field.type)) {
return undefined;
}
fields = field.type.children;
}
return field;
}
function isDeferredValue(value: unknown): boolean {
return (
value == null || (Array.isArray(value) && value.every(isDeferredValue))
);
}
function deferredValueMatchesType(value: unknown, type: DataType): boolean {
if (value == null) {
return true;
}
if (!Array.isArray(value)) {
return false;
}
if (DataType.isList(type)) {
return value.every((item) =>
deferredValueMatchesType(item, type.valueType),
);
}
if (DataType.isFixedSizeList(type)) {
return (
value.length === type.listSize &&
value.every((item) => deferredValueMatchesType(item, type.valueType))
);
}
return false;
}
function inferredTypesEqual(current: DataType, candidate: DataType): boolean {
if (DataType.isDictionary(current)) {
return (
DataType.isDictionary(candidate) &&
current.isOrdered === candidate.isOrdered &&
inferredTypesEqual(current.indices, candidate.indices) &&
inferredTypesEqual(current.dictionary, candidate.dictionary)
);
}
if (DataType.isList(current)) {
return (
DataType.isList(candidate) &&
current.valueField.name === candidate.valueField.name &&
current.valueField.nullable === candidate.valueField.nullable &&
inferredTypesEqual(current.valueType, candidate.valueType)
);
}
if (DataType.isFixedSizeList(current)) {
return (
DataType.isFixedSizeList(candidate) &&
current.listSize === candidate.listSize &&
current.valueField.name === candidate.valueField.name &&
current.valueField.nullable === candidate.valueField.nullable &&
inferredTypesEqual(current.valueType, candidate.valueType)
);
}
return arrowUtil.compareTypes(current, candidate);
}
function describeEvidence(
evidence: DataType | DeferredTypeEvidence | undefined,
): string {
if (evidence === undefined) {
return "an unsupported value";
}
return evidence instanceof DeferredTypeEvidence
? evidence.describe()
: evidence.toString();
}
function branchConflictError(
conflict: FieldConflict,
row: number,
candidate: string,
): Error {
return schemaInferenceError(
conflict.path,
row,
conflict.value instanceof FieldTree
? "Struct"
: describeEvidence(conflict.value),
candidate,
);
}
function schemaInferenceError(
path: string[],
row: number,
currentType: string,
newType: string,
): Error {
return new Error(
`Failed to infer schema for data. Previously inferred type ${currentType} ` +
`but found ${newType} for field ${path.join(".")} at row ${row}. ` +
"Consider providing an explicit schema.",
);
}
function typeInferenceError(path: string[], row: number): Error {
return new Error(
`Failed to infer data type for field ${path.join(".")} at row ${row}. ` +
"Consider providing an explicit schema.",
);
}
function nameSuggestsVectorColumn(name: string): boolean {
const normalized = name.toLowerCase();
return normalized.includes("vector") || normalized.includes("embedding");
}
+24 -66
View File
@@ -19,7 +19,6 @@ import {
import { EmbeddingFunctionConfig, getRegistry } from "./embedding/registry"; import { EmbeddingFunctionConfig, getRegistry } from "./embedding/registry";
import { IndexOptions } from "./indices"; import { IndexOptions } from "./indices";
import { Job } from "./job";
import { MergeInsertBuilder } from "./merge"; import { MergeInsertBuilder } from "./merge";
import { import {
AddColumnsResult, AddColumnsResult,
@@ -31,6 +30,7 @@ import {
DropColumnsResult, DropColumnsResult,
IndexConfig, IndexConfig,
IndexStatistics, IndexStatistics,
Job,
LsmStats, LsmStats,
Branches as NativeBranches, Branches as NativeBranches,
OptimizeStats, OptimizeStats,
@@ -43,12 +43,10 @@ import {
Table as _NativeTable, Table as _NativeTable,
} from "./native"; } from "./native";
import { import {
AutoQuery,
FullTextQuery, FullTextQuery,
Query, Query,
TakeQuery, TakeQuery,
VectorQuery, VectorQuery,
createAutoQuery,
instanceOfFullTextQuery, instanceOfFullTextQuery,
} from "./query"; } from "./query";
import { sanitizeType } from "./sanitize"; import { sanitizeType } from "./sanitize";
@@ -525,7 +523,7 @@ export abstract class Table {
query: string | IntoVector | MultiVector | FullTextQuery, query: string | IntoVector | MultiVector | FullTextQuery,
queryType?: string, queryType?: string,
ftsColumns?: string | string[], ftsColumns?: string | string[],
): VectorQuery | Query | AutoQuery; ): VectorQuery | Query;
/** /**
* Search the table with a given query vector. * Search the table with a given query vector.
* *
@@ -630,18 +628,6 @@ export abstract class Table {
/** /**
* Update per-field (column) metadata. * Update per-field (column) metadata.
*
* The following keys are treated specially, by convention, and should be
* used when appropriate:
*
* - `lancedb:description`: for a human-readable description of a field.
* - `lancedb:tag:<name>`: for a user-defined key-value tag, where the suffix
* names the tag category; e.g. `lancedb:tag:model: "clip"`.
* - `lancedb:logical-column`: for a column grouping; e.g. `feature_v1` and
* `feature_v2` might be in the same logical column.
* - `lancedb:status`: for status options (`production`, `candidate`,
* `deprecated`, `archived`) to designate the current life cycle state of
* this column.
* @param {FieldMetadataUpdate[]} updates One or more per-field updates. Each * @param {FieldMetadataUpdate[]} updates One or more per-field updates. Each
* update's metadata is merged into the field's existing metadata by default; * update's metadata is merged into the field's existing metadata by default;
* a value of `null` deletes that key, and `replace: true` swaps the whole map. * a value of `null` deletes that key, and `replace: true` swaps the whole map.
@@ -919,16 +905,6 @@ export abstract class Table {
/** Return the table as an arrow table */ /** Return the table as an arrow table */
abstract toArrow(): Promise<ArrowTable>; abstract toArrow(): Promise<ArrowTable>;
/**
* Create a {@link MergeInsertBuilder}, which combines new data with the
* existing table in a single transaction — inserting, updating and deleting
* rows depending on how they match.
*
* @param on - The column, or columns, to match source rows against target
* rows on. Typically a key or id column. Several columns match on the
* composite key: a source row updates a target row only when it agrees on
* every one of them.
*/
abstract mergeInsert(on: string | string[]): MergeInsertBuilder; abstract mergeInsert(on: string | string[]): MergeInsertBuilder;
/** List all the stats of a specified index /** List all the stats of a specified index
@@ -999,11 +975,10 @@ export class LocalTable extends Table {
return this.inner.display(); return this.inner.display();
} }
private async getEmbeddingFunctions( private async getEmbeddingFunctions(): Promise<
inner: _NativeTable = this.inner, Map<string, EmbeddingFunctionConfig>
): Promise<Map<string, EmbeddingFunctionConfig>> { > {
const schemaBuf = await inner.schema(); const schema = await this.schema();
const schema = tableFromIPC(schemaBuf).schema;
const registry = getRegistry(); const registry = getRegistry();
return registry.parseFunctions(schema.metadata); return registry.parseFunctions(schema.metadata);
} }
@@ -1124,15 +1099,13 @@ export class LocalTable extends Table {
): Promise<Job> { ): Promise<Job> {
// biome-ignore lint/suspicious/noExplicitAny: skip // biome-ignore lint/suspicious/noExplicitAny: skip
const nativeIndex = (options?.config as any)?.inner; const nativeIndex = (options?.config as any)?.inner;
return new Job( return await this.inner.createIndexAsync(
await this.inner.createIndexAsync( nativeIndex,
nativeIndex, column,
column, options?.replace,
options?.replace, options?.waitTimeoutSeconds,
options?.waitTimeoutSeconds, options?.name,
options?.name, options?.train,
options?.train,
),
); );
} }
@@ -1187,7 +1160,7 @@ export class LocalTable extends Table {
query: string | IntoVector | MultiVector | FullTextQuery, query: string | IntoVector | MultiVector | FullTextQuery,
queryType: string = "auto", queryType: string = "auto",
ftsColumns?: string | string[], ftsColumns?: string | string[],
): VectorQuery | Query | AutoQuery { ): VectorQuery | Query {
if (typeof query !== "string" && !instanceOfFullTextQuery(query)) { if (typeof query !== "string" && !instanceOfFullTextQuery(query)) {
if (queryType === "fts") { if (queryType === "fts") {
throw new Error("Cannot perform full text search on a vector query"); throw new Error("Cannot perform full text search on a vector query");
@@ -1202,28 +1175,14 @@ export class LocalTable extends Table {
}); });
} }
if (queryType === "auto") { // The query type is auto or vector
if (instanceOfFullTextQuery(query)) { // fall back to full text search if no embedding functions are defined and the query is a string
return this.query().fullTextSearch(query, { if (
columns: ftsColumns, queryType === "auto" &&
}); (getRegistry().length() === 0 || instanceOfFullTextQuery(query))
} ) {
return this.query().fullTextSearch(query, {
const columns = columns: ftsColumns,
typeof ftsColumns === "string" ? [ftsColumns] : (ftsColumns ?? null);
return createAutoQuery(this.inner, query, columns, async (metadata) => {
const functions = await getRegistry().parseFunctions(
new Map([["embedding_functions", metadata]]),
);
// TODO: Support multiple embedding functions
const embeddingFunc: EmbeddingFunctionConfig | undefined = functions
.values()
.next().value;
// The route only calls this callback when embedding metadata exists.
// parseFunctions either yields a provider or reports malformed metadata.
if (!embeddingFunc)
throw new Error("Invalid embedding function metadata");
return await embeddingFunc.function.computeQueryEmbeddings(query);
}); });
} }
@@ -1315,7 +1274,7 @@ export class LocalTable extends Table {
} }
async refreshColumnAsync(column: string): Promise<Job> { async refreshColumnAsync(column: string): Promise<Job> {
return new Job(await this.inner.refreshColumnAsync(column)); return await this.inner.refreshColumnAsync(column);
} }
async refreshMaterializedView( async refreshMaterializedView(
@@ -1579,8 +1538,7 @@ export interface FieldMetadataUpdate {
path: string; path: string;
/** /**
* Metadata key/value pairs. Merged into the field's existing metadata by * Metadata key/value pairs. Merged into the field's existing metadata by
* default; a value of `null` deletes that key. See * default; a value of `null` deletes that key.
* {@link Table.updateFieldMetadata} for the conventional `lancedb:*` keys.
*/ */
metadata: Record<string, string | null>; metadata: Record<string, string | null>;
/** If true, replace the field's entire metadata map instead of merging. */ /** If true, replace the field's entire metadata map instead of merging. */
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@lancedb/lancedb-darwin-arm64", "name": "@lancedb/lancedb-darwin-arm64",
"version": "0.39.0-beta.4", "version": "0.38.0-beta.5",
"os": ["darwin"], "os": ["darwin"],
"cpu": ["arm64"], "cpu": ["arm64"],
"main": "lancedb.darwin-arm64.node", "main": "lancedb.darwin-arm64.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@lancedb/lancedb-linux-arm64-gnu", "name": "@lancedb/lancedb-linux-arm64-gnu",
"version": "0.39.0-beta.4", "version": "0.38.0-beta.5",
"os": ["linux"], "os": ["linux"],
"cpu": ["arm64"], "cpu": ["arm64"],
"main": "lancedb.linux-arm64-gnu.node", "main": "lancedb.linux-arm64-gnu.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@lancedb/lancedb-linux-arm64-musl", "name": "@lancedb/lancedb-linux-arm64-musl",
"version": "0.39.0-beta.4", "version": "0.38.0-beta.5",
"os": ["linux"], "os": ["linux"],
"cpu": ["arm64"], "cpu": ["arm64"],
"main": "lancedb.linux-arm64-musl.node", "main": "lancedb.linux-arm64-musl.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@lancedb/lancedb-linux-x64-gnu", "name": "@lancedb/lancedb-linux-x64-gnu",
"version": "0.39.0-beta.4", "version": "0.38.0-beta.5",
"os": ["linux"], "os": ["linux"],
"cpu": ["x64"], "cpu": ["x64"],
"main": "lancedb.linux-x64-gnu.node", "main": "lancedb.linux-x64-gnu.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@lancedb/lancedb-linux-x64-musl", "name": "@lancedb/lancedb-linux-x64-musl",
"version": "0.39.0-beta.4", "version": "0.38.0-beta.5",
"os": ["linux"], "os": ["linux"],
"cpu": ["x64"], "cpu": ["x64"],
"main": "lancedb.linux-x64-musl.node", "main": "lancedb.linux-x64-musl.node",
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@lancedb/lancedb-win32-arm64-msvc", "name": "@lancedb/lancedb-win32-arm64-msvc",
"version": "0.39.0-beta.4", "version": "0.38.0-beta.5",
"os": [ "os": [
"win32" "win32"
], ],
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@lancedb/lancedb-win32-x64-msvc", "name": "@lancedb/lancedb-win32-x64-msvc",
"version": "0.39.0-beta.4", "version": "0.38.0-beta.5",
"os": ["win32"], "os": ["win32"],
"cpu": ["x64"], "cpu": ["x64"],
"main": "lancedb.win32-x64-msvc.node", "main": "lancedb.win32-x64-msvc.node",
+11106
View File
File diff suppressed because it is too large Load Diff
+5 -5
View File
@@ -11,7 +11,7 @@
"ann" "ann"
], ],
"private": false, "private": false,
"version": "0.39.0-beta.4", "version": "0.38.0-beta.5",
"main": "dist/index.js", "main": "dist/index.js",
"exports": { "exports": {
".": "./dist/index.js", ".": "./dist/index.js",
@@ -44,7 +44,7 @@
"@biomejs/biome": "^1.7.3", "@biomejs/biome": "^1.7.3",
"@jest/globals": "^29.7.0", "@jest/globals": "^29.7.0",
"@napi-rs/cli": "3.7.0", "@napi-rs/cli": "3.7.0",
"@opentelemetry/sdk-metrics": "^2.10.0", "@opentelemetry/sdk-metrics": "^1.30.0",
"@types/axios": "^0.14.0", "@types/axios": "^0.14.0",
"@types/jest": "^29.1.2", "@types/jest": "^29.1.2",
"@types/node": "22.7.4", "@types/node": "22.7.4",
@@ -56,7 +56,7 @@
"eslint": "^8.57.0", "eslint": "^8.57.0",
"jest": "^29.7.0", "jest": "^29.7.0",
"shx": "^0.3.4", "shx": "^0.3.4",
"tmp": "^0.2.7", "tmp": "^0.2.3",
"ts-jest": "^29.1.2", "ts-jest": "^29.1.2",
"typedoc": "0.26.4", "typedoc": "0.26.4",
"typedoc-plugin-markdown": "4.2.1", "typedoc-plugin-markdown": "4.2.1",
@@ -67,7 +67,7 @@
"timeout": "3m" "timeout": "3m"
}, },
"engines": { "engines": {
"node": ">= 22" "node": ">= 18"
}, },
"packageManager": "pnpm@11.1.1", "packageManager": "pnpm@11.1.1",
"cpu": ["x64", "arm64"], "cpu": ["x64", "arm64"],
@@ -101,7 +101,7 @@
"openai": "4.29.2" "openai": "4.29.2"
}, },
"peerDependencies": { "peerDependencies": {
"@types/node": ">=22", "@types/node": ">=18",
"apache-arrow": ">=15.0.0 <=18.1.0" "apache-arrow": ">=15.0.0 <=18.1.0"
}, },
"peerDependenciesMeta": { "peerDependenciesMeta": {
+443 -599
View File
File diff suppressed because it is too large Load Diff
-38
View File
@@ -16,41 +16,3 @@ allowBuilds:
onnxruntime-node: true onnxruntime-node: true
protobufjs: true protobufjs: true
sharp: true sharp: true
minimumReleaseAgeExclude:
- protobufjs@7.5.8
- tmp@0.2.6
- form-data@4.0.6
- tar@7.5.16
- markdown-it@14.1.2
- linkify-it@5.0.1
- js-yaml@3.15.0
- js-yaml@4.1.2
- protobufjs@7.6.1
- protobufjs@7.6.3
- '@babel/core@7.29.1'
- axios@1.18.0
- brace-expansion@2.1.2
- brace-expansion@1.1.16
- js-yaml@4.3.0
- tar@7.5.18
- tar@7.5.19
- tar@7.5.17
- protobufjs@7.6.5
- linkify-it@5.0.2
- sharp@0.35.0
- brace-expansion@1.1.17
- brace-expansion@2.1.3
- brace-expansion@2.1.4
- brace-expansion@1.1.18
- js-yaml@3.15.1
- js-yaml@4.3.1
- tar@7.5.21
- '@opentelemetry/core@2.8.0'
# @huggingface/transformers pins sharp ^0.33.5 and no released version has moved
# past ^0.34.5, all of which inherit the libvips CVEs in GHSA-f88m-g3jw-g9cj.
# Force the patched line. sharp is only reached by transformers' image pipeline,
# which LanceDB's text embedding function never uses.
overrides:
sharp: ^0.35.4
+45 -42
View File
@@ -17,7 +17,6 @@ use lancedb::connection::{ConnectBuilder, Connection as LanceDBConnection, conne
use lance_namespace::models::{ use lance_namespace::models::{
CreateNamespaceRequest, DescribeNamespaceRequest, DropNamespaceRequest, ListNamespacesRequest, CreateNamespaceRequest, DescribeNamespaceRequest, DropNamespaceRequest, ListNamespacesRequest,
ListTablesRequest,
}; };
use lancedb::ipc::{ipc_file_to_batches, ipc_file_to_schema}; use lancedb::ipc::{ipc_file_to_batches, ipc_file_to_schema};
@@ -37,12 +36,6 @@ pub struct ListNamespacesResponse {
pub page_token: Option<String>, pub page_token: Option<String>,
} }
#[napi(object)]
pub struct ListTablesResponse {
pub tables: Vec<String>,
pub page_token: Option<String>,
}
#[napi(object)] #[napi(object)]
pub struct CreateNamespaceResponse { pub struct CreateNamespaceResponse {
pub properties: Option<HashMap<String, String>>, pub properties: Option<HashMap<String, String>>,
@@ -213,33 +206,6 @@ impl Connection {
op.execute().await.default_error() op.execute().await.default_error()
} }
/// List a page of tables in the database.
#[napi(catch_unwind)]
pub async fn list_tables(
&self,
namespace_path: Option<Vec<String>>,
page_token: Option<String>,
limit: Option<u32>,
) -> napi::Result<ListTablesResponse> {
let request = ListTablesRequest {
// The root namespace is an empty path, not an absent one: a namespace-backed
// database rejects a request that names no namespace.
id: Some(namespace_path.unwrap_or_default()),
page_token,
limit: limit.map(|limit| i32::try_from(limit).unwrap_or(i32::MAX)),
..Default::default()
};
let response = self
.get_inner()?
.list_tables(request)
.await
.default_error()?;
Ok(ListTablesResponse {
tables: response.tables,
page_token: response.page_token,
})
}
/// Create table from a Apache Arrow IPC (file) buffer. /// Create table from a Apache Arrow IPC (file) buffer.
/// ///
/// Parameters: /// Parameters:
@@ -442,15 +408,13 @@ impl Connection {
self.get_inner()?.drop_all_tables(&ns).await.default_error() self.get_inner()?.drop_all_tables(&ns).await.default_error()
} }
/// Open a server-side job by id, returning a handle with its record /// A `Job` handle for a server-side job by id.
/// already populated. Rejects when the server has no such job.
/// ///
/// The returned handle answers for its own state, specification, result, /// The handle is constructed without a server round trip; an unknown id
/// failure and event history, so there is no separate connection-level /// surfaces when the handle is used.
/// call for any of them. #[napi]
#[napi(catch_unwind)] pub fn job(&self, job_id: String) -> napi::Result<crate::job::Job> {
pub async fn open_job(&self, job_id: String) -> napi::Result<crate::job::Job> { let job = self.get_inner()?.job(job_id).default_error()?;
let job = self.get_inner()?.open_job(&job_id).await.default_error()?;
Ok(crate::job::Job::new(job)) Ok(crate::job::Job::new(job))
} }
@@ -461,6 +425,17 @@ impl Connection {
Ok(jobs.into_iter().map(Into::into).collect()) Ok(jobs.into_iter().map(Into::into).collect())
} }
/// Describe a single server-side job by id. `null` when the server has
/// no such job.
#[napi(catch_unwind)]
pub async fn get_job(
&self,
job_id: String,
) -> napi::Result<Option<crate::job::JobDescription>> {
let description = self.get_inner()?.get_job(&job_id).await.default_error()?;
Ok(description.map(Into::into))
}
/// Request cancellation of a server-side job by id. Returns true if the /// Request cancellation of a server-side job by id. Returns true if the
/// server accepted the cancellation, false if no such job exists. /// server accepted the cancellation, false if no such job exists.
#[napi(catch_unwind)] #[napi(catch_unwind)]
@@ -468,6 +443,34 @@ impl Connection {
self.get_inner()?.cancel_job(&job_id).await.default_error() self.get_inner()?.cancel_job(&job_id).await.default_error()
} }
/// The lifecycle event history of a server-side job (all jobs when
/// `job_id` is null), as an Arrow IPC stream buffer. Empty when there is
/// no history.
#[napi(catch_unwind)]
pub async fn job_history(&self, job_id: Option<String>) -> napi::Result<Buffer> {
let batches = self
.get_inner()?
.job_history(job_id.as_deref())
.await
.default_error()?;
let Some(first) = batches.first() else {
return Ok(Buffer::from(Vec::<u8>::new()));
};
let mut out = Vec::new();
let mut writer = arrow_ipc::writer::StreamWriter::try_new(&mut out, &first.schema())
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
for batch in &batches {
writer
.write(batch)
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
}
writer
.finish()
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
drop(writer);
Ok(Buffer::from(out))
}
#[napi(catch_unwind)] #[napi(catch_unwind)]
/// Describe a namespace and return its properties. /// Describe a namespace and return its properties.
pub async fn describe_namespace( pub async fn describe_namespace(
+36 -95
View File
@@ -3,9 +3,6 @@
use std::sync::Arc; use std::sync::Arc;
use arrow_array::RecordBatch;
use lancedb::job::JobEventsRequest;
use napi::bindgen_prelude::Buffer;
use napi_derive::napi; use napi_derive::napi;
use crate::error::NapiErrorExt; use crate::error::NapiErrorExt;
@@ -17,12 +14,9 @@ pub struct Job {
} }
impl Job { impl Job {
pub(crate) fn new<T>(inner: lancedb::Job<T>) -> Self pub(crate) fn new(inner: lancedb::Job) -> Self {
where
T: Clone + Send + Sync + 'static,
{
Self { Self {
inner: Arc::new(inner.map(|_| ())), inner: Arc::new(inner),
} }
} }
} }
@@ -58,98 +52,12 @@ impl Job {
pub async fn cancel(&self) -> napi::Result<()> { pub async fn cancel(&self) -> napi::Result<()> {
self.inner.cancel().await.default_error() self.inner.cancel().await.default_error()
} }
/// Ask the backend for this job's current state, and for a server-side job
/// its full record, then cache it for the getters below.
///
/// They are all null until this runs, because submitting an operation
/// returns only a job id. {@link Job.status} fetches the whole record too;
/// {@link Job.wait} records only the terminal state it establishes.
#[napi(catch_unwind)]
pub async fn refresh(&self) -> napi::Result<()> {
self.inner.refresh().await.default_error()
}
/// The last observed lifecycle state, without contacting the backend.
#[napi(getter)]
pub fn state(&self) -> Option<String> {
self.inner.state()
}
/// The job's type, as the server names it. Null for an in-process job,
/// which has no server-side record.
#[napi(getter)]
pub fn job_type(&self) -> Option<String> {
self.inner.job_type()
}
/// When the job was created, in milliseconds since the epoch.
#[napi(getter)]
pub fn creation_ms(&self) -> Option<i64> {
self.inner.creation_ms()
}
/// The job-type-specific specification as a JSON string, when present.
#[napi(getter)]
pub fn spec_json(&self) -> Option<String> {
self.inner.spec().map(|spec| spec.to_string())
}
/// The job-type-specific terminal result as a JSON string. Null until the
/// job succeeds, so a job that never terminates reports its progress
/// through {@link Job.events} instead.
#[napi(getter)]
pub fn result_json(&self) -> Option<String> {
self.inner.result().map(|result| result.to_string())
}
/// Why the job failed, when it failed and the server reports a reason.
#[napi(getter)]
pub fn failure(&self) -> Option<JobFailureInfo> {
self.inner.failure().map(|failure| JobFailureInfo {
phase: failure.phase,
message: failure.message,
retryable: failure.retryable,
})
}
/// This job's recorded lifecycle events, as an Arrow IPC stream buffer.
/// The TypeScript wrapper turns it into an Arrow table.
#[napi(catch_unwind)]
pub async fn events(&self, limit: Option<u32>, filter: Option<String>) -> napi::Result<Buffer> {
let batches = self
.inner
.events(JobEventsRequest { limit, filter })
.await
.default_error()?;
batches_to_ipc_buffer(&batches)
}
}
/// Serialise Arrow batches as a single IPC stream for the TypeScript layer.
fn batches_to_ipc_buffer(batches: &[RecordBatch]) -> napi::Result<Buffer> {
let Some(first) = batches.first() else {
return Ok(Buffer::from(Vec::<u8>::new()));
};
let mut out = Vec::new();
let mut writer = arrow_ipc::writer::StreamWriter::try_new(&mut out, &first.schema())
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
for batch in batches {
writer
.write(batch)
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
}
writer
.finish()
.map_err(|e| napi::Error::from_reason(e.to_string()))?;
drop(writer);
Ok(Buffer::from(out))
} }
/// A row from `Connection.listJobs`: one server-side job. /// A row from `Connection.listJobs`: one server-side job.
#[napi(object)] #[napi(object)]
pub struct JobInfo { pub struct JobInfo {
/// The job id -- what `Connection.openJob` and `Connection.cancelJob` /// The job id -- what `Connection.getJob` and `Connection.cancelJob`
/// accept. /// accept.
pub job_id: String, pub job_id: String,
/// The table the job runs against, without URI or namespace. /// The table the job runs against, without URI or namespace.
@@ -180,3 +88,36 @@ pub struct JobFailureInfo {
pub message: Option<String>, pub message: Option<String>,
pub retryable: Option<bool>, pub retryable: Option<bool>,
} }
/// A described job from `Connection.getJob`.
#[napi(object)]
pub struct JobDescription {
pub job_id: String,
pub job_type: String,
/// Lifecycle state: "running", "finished", "failed", or "cancelled".
pub state: String,
/// When the job was created, in milliseconds since the epoch.
pub creation_ms: i64,
/// The job-type-specific specification as a JSON string, when present.
pub spec_json: Option<String>,
/// Why the job failed, when the job is failed and the server reports a
/// reason.
pub failure: Option<JobFailureInfo>,
}
impl From<lancedb::database::JobDescription> for JobDescription {
fn from(description: lancedb::database::JobDescription) -> Self {
Self {
job_id: description.job_id,
job_type: description.job_type,
state: description.state,
creation_ms: description.creation_ms,
spec_json: (!description.spec.is_null()).then(|| description.spec.to_string()),
failure: description.failure.map(|failure| JobFailureInfo {
phase: failure.phase,
message: failure.message,
retryable: failure.retryable,
}),
}
}
}
+1 -5
View File
@@ -664,11 +664,7 @@ impl JsFullTextQuery {
} }
fn parse_fts_query(query: Object) -> napi::Result<FullTextSearchQuery> { fn parse_fts_query(query: Object) -> napi::Result<FullTextSearchQuery> {
// `&JsFullTextQuery` recovers a native class reference through napi's borrow-tracked if let Ok(Some(query)) = query.get::<&JsFullTextQuery>("query") {
// path, which is only usable from generated `#[napi]` argument conversion. This is a
// manual lookup on a nested `Object` property instead, so use `ClassInstance`, which
// unwraps the class without requiring a borrow scope.
if let Ok(Some(query)) = query.get::<ClassInstance<JsFullTextQuery>>("query") {
Ok(FullTextSearchQuery::new_query(query.inner.clone())) Ok(FullTextSearchQuery::new_query(query.inner.clone()))
} else if let Ok(Some(query_text)) = query.get::<String>("query") { } else if let Ok(Some(query_text)) = query.get::<String>("query") {
let mut query_text = query_text; let mut query_text = query_text;
-13
View File
@@ -278,13 +278,6 @@ impl Table {
Ok(Query::new(self.inner_ref()?.query())) Ok(Query::new(self.inner_ref()?.query()))
} }
/// Return a read-only table handle pinned to the current query revision.
#[napi(catch_unwind)]
pub async fn query_snapshot(&self) -> napi::Result<Self> {
let snapshot = self.inner_ref()?.query_snapshot().await.default_error()?;
Ok(Self::new(snapshot))
}
#[napi(catch_unwind)] #[napi(catch_unwind)]
pub fn take_offsets(&self, offsets: Vec<i64>) -> napi::Result<TakeQuery> { pub fn take_offsets(&self, offsets: Vec<i64>) -> napi::Result<TakeQuery> {
Ok(TakeQuery::new( Ok(TakeQuery::new(
@@ -561,12 +554,6 @@ impl Table {
.default_error() .default_error()
} }
#[napi(catch_unwind)]
pub async fn checkout_current(&self) -> napi::Result<Self> {
let table = self.inner_ref()?.checkout_current().await.default_error()?;
Ok(Self::new(table))
}
#[napi(catch_unwind)] #[napi(catch_unwind)]
pub async fn checkout(&self, version: i64) -> napi::Result<()> { pub async fn checkout(&self, version: i64) -> napi::Result<()> {
self.inner_ref()? self.inner_ref()?
+2 -3
View File
@@ -1,6 +1,6 @@
[package] [package]
name = "lancedb-python" name = "lancedb-python"
version = "0.39.0-beta.4" version = "0.38.0-beta.5"
publish = false publish = false
edition.workspace = true edition.workspace = true
description = "Python bindings for LanceDB" description = "Python bindings for LanceDB"
@@ -28,7 +28,7 @@ env_logger.workspace = true
log.workspace = true log.workspace = true
# Maturin enables extension-module mode for Python builds. Keeping it out of # Maturin enables extension-module mode for Python builds. Keeping it out of
# Cargo features lets Rust unit tests link against libpython. # Cargo features lets Rust unit tests link against libpython.
pyo3 = { version = "0.28", features = ["abi3-py310", "chrono", "uuid"] } pyo3 = { version = "0.28", features = ["abi3-py310", "chrono"] }
chrono.workspace = true chrono.workspace = true
pyo3-async-runtimes = { version = "0.28", features = [ pyo3-async-runtimes = { version = "0.28", features = [
"attributes", "attributes",
@@ -40,7 +40,6 @@ serde.workspace = true
serde_json.workspace = true serde_json.workspace = true
snafu.workspace = true snafu.workspace = true
tokio.workspace = true tokio.workspace = true
uuid.workspace = true
libc = "0.2" libc = "0.2"
[build-dependencies] [build-dependencies]
+1 -5
View File
@@ -101,12 +101,9 @@ azure = ["adlfs>=2024.2.0"]
[tool.maturin] [tool.maturin]
python-source = "python" python-source = "python"
module-name = "lancedb._lancedb" module-name = "lancedb._lancedb"
# uv installs the project as an editable package before `uv run`, so keep that
# bootstrap build consistent with `maturin develop`.
editable-profile = "dev"
[build-system] [build-system]
requires = ["maturin>=1.10"] requires = ["maturin>=1.9.4"]
build-backend = "maturin" build-backend = "maturin"
[tool.ruff.lint] [tool.ruff.lint]
@@ -139,7 +136,6 @@ include = [
"python/lancedb/exceptions.py", "python/lancedb/exceptions.py",
"python/lancedb/background_loop.py", "python/lancedb/background_loop.py",
"python/lancedb/schema.py", "python/lancedb/schema.py",
"python/lancedb/sql.py",
"python/lancedb/remote/__init__.py", "python/lancedb/remote/__init__.py",
"python/lancedb/remote/errors.py", "python/lancedb/remote/errors.py",
"python/lancedb/embeddings/__init__.py", "python/lancedb/embeddings/__init__.py",
+2 -57
View File
@@ -6,7 +6,7 @@ import importlib.metadata
import os import os
from concurrent.futures import ThreadPoolExecutor from concurrent.futures import ThreadPoolExecutor
from datetime import timedelta from datetime import timedelta
from typing import Dict, Optional, Union, Any, List, Iterable, TYPE_CHECKING from typing import Dict, Optional, Union, Any, List, Iterable
__version__ = importlib.metadata.version("lancedb") __version__ = importlib.metadata.version("lancedb")
@@ -20,25 +20,18 @@ from .db import AsyncConnection, DBConnection, LanceDBConnection
from .remote import ClientConfig from .remote import ClientConfig
from .remote.db import RemoteDBConnection from .remote.db import RemoteDBConnection
from .expr import Expr, col, lit, func from .expr import Expr, col, lit, func
from .schema import blob, vector from .schema import blob, vector, BlobType
from .job import AsyncJob, Job from .job import AsyncJob, Job
from .sql import AsyncQuery as AsyncSqlQuery
from .sql import Query as SqlQuery
from .sql import QueryDescription
from .functions import ( from .functions import (
AssignmentMapping as AssignmentMapping,
FunctionArtifactRequest as FunctionArtifactRequest, FunctionArtifactRequest as FunctionArtifactRequest,
FunctionApplication as FunctionApplication, FunctionApplication as FunctionApplication,
FunctionBinding as FunctionBinding, FunctionBinding as FunctionBinding,
FunctionRegistrationRequest as FunctionRegistrationRequest, FunctionRegistrationRequest as FunctionRegistrationRequest,
FunctionVersion as FunctionVersion, FunctionVersion as FunctionVersion,
PythonRuntimeSpec as PythonRuntimeSpec, PythonRuntimeSpec as PythonRuntimeSpec,
RefreshColumnResult as RefreshColumnResult,
UdfDefinition as UdfDefinition, UdfDefinition as UdfDefinition,
udf as udf, udf as udf,
) )
from .secrets import EnvVarSecret as EnvVarSecret
from .secrets import SecretInfo as SecretInfo
from .materialized_view import ( from .materialized_view import (
AsyncMaterializedView, AsyncMaterializedView,
MaterializedView, MaterializedView,
@@ -55,19 +48,6 @@ from .namespace import (
) )
if TYPE_CHECKING:
from lance.blob import BlobType as BlobType
def __getattr__(name: str):
if name == "BlobType":
from .schema import BlobType
globals()["BlobType"] = BlobType
return BlobType
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
def _check_s3_bucket_with_dots( def _check_s3_bucket_with_dots(
uri: str, storage_options: Optional[Dict[str, str]] uri: str, storage_options: Optional[Dict[str, str]]
) -> None: ) -> None:
@@ -107,7 +87,6 @@ def connect(
api_key: Optional[str] = None, api_key: Optional[str] = None,
region: str = "us-east-1", region: str = "us-east-1",
host_override: Optional[str] = None, host_override: Optional[str] = None,
sql_host_override: Optional[str] = None,
read_consistency_interval: Optional[timedelta] = None, read_consistency_interval: Optional[timedelta] = None,
request_thread_pool: Optional[Union[int, ThreadPoolExecutor]] = None, request_thread_pool: Optional[Union[int, ThreadPoolExecutor]] = None,
client_config: Union[ClientConfig, Dict[str, Any], None] = None, client_config: Union[ClientConfig, Dict[str, Any], None] = None,
@@ -136,9 +115,6 @@ def connect(
The region to use for LanceDB Cloud. The region to use for LanceDB Cloud.
host_override: str, optional host_override: str, optional
The override url for LanceDB Cloud. The override url for LanceDB Cloud.
sql_host_override: str, optional
The remote SQL service endpoint override. The client connects lazily when SQL
is first executed and retains that connection.
read_consistency_interval: timedelta, default None read_consistency_interval: timedelta, default None
The interval at which to check for updates to the table from other The interval at which to check for updates to the table from other
processes. If None, then consistency is not checked. For performance processes. If None, then consistency is not checked. For performance
@@ -202,18 +178,6 @@ def connect(
... }, ... },
... ) ... )
For Azure Blob Storage, credentials can be passed directly without setting
environment variables:
>>> azure_storage_options = {
... "account_name": "some-account",
... "account_key": "some-key",
... }
>>> db = lancedb.connect( # doctest: +SKIP
... "az://my-container/my-database",
... storage_options=azure_storage_options,
... )
For tests and temporary data, use an in-memory database: For tests and temporary data, use an in-memory database:
>>> db = lancedb.connect("memory://") >>> db = lancedb.connect("memory://")
@@ -280,7 +244,6 @@ def connect(
api_key, api_key,
region, region,
host_override, host_override,
sql_host_override=sql_host_override,
# TODO: remove this (deprecation warning downstream) # TODO: remove this (deprecation warning downstream)
request_thread_pool=request_thread_pool, request_thread_pool=request_thread_pool,
client_config=client_config, client_config=client_config,
@@ -423,7 +386,6 @@ def deserialize_conn(
parsed["api_key"], parsed["api_key"],
parsed.get("region", "us-east-1"), parsed.get("region", "us-east-1"),
host_override=parsed.get("host_override"), host_override=parsed.get("host_override"),
sql_host_override=parsed.get("sql_host_override"),
client_config=parsed.get("client_config"), client_config=parsed.get("client_config"),
storage_options=storage_options, storage_options=storage_options,
) )
@@ -437,7 +399,6 @@ async def connect_async(
api_key: Optional[str] = None, api_key: Optional[str] = None,
region: str = "us-east-1", region: str = "us-east-1",
host_override: Optional[str] = None, host_override: Optional[str] = None,
sql_host_override: Optional[str] = None,
read_consistency_interval: Optional[timedelta] = None, read_consistency_interval: Optional[timedelta] = None,
client_config: Optional[Union[ClientConfig, Dict[str, Any]]] = None, client_config: Optional[Union[ClientConfig, Dict[str, Any]]] = None,
storage_options: Optional[Dict[str, str]] = None, storage_options: Optional[Dict[str, str]] = None,
@@ -460,9 +421,6 @@ async def connect_async(
The region to use for LanceDB Cloud. The region to use for LanceDB Cloud.
host_override: str, optional host_override: str, optional
The override url for LanceDB Cloud. The override url for LanceDB Cloud.
sql_host_override: str, optional
The remote SQL service endpoint override. The client connects lazily when SQL
is first executed and retains that connection.
read_consistency_interval: timedelta, default None read_consistency_interval: timedelta, default None
The interval at which to check for updates to the table from other The interval at which to check for updates to the table from other
processes. If None, then consistency is not checked. For performance processes. If None, then consistency is not checked. For performance
@@ -506,10 +464,6 @@ async def connect_async(
-------- --------
>>> import lancedb >>> import lancedb
>>> azure_storage_options = {
... "account_name": "some-account",
... "account_key": "some-key",
... }
>>> async def doctest_example(): >>> async def doctest_example():
... # For a local directory, provide a path to the database ... # For a local directory, provide a path to the database
... db = await lancedb.connect_async("~/.lancedb") ... db = await lancedb.connect_async("~/.lancedb")
@@ -517,11 +471,6 @@ async def connect_async(
... db = await lancedb.connect_async("s3://my-bucket/lancedb", ... db = await lancedb.connect_async("s3://my-bucket/lancedb",
... storage_options={ ... storage_options={
... "aws_access_key_id": "***"}) ... "aws_access_key_id": "***"})
... # Azure credentials can also be passed directly
... db = await lancedb.connect_async(
... "az://my-container/my-database",
... storage_options=azure_storage_options,
... )
... # For tests and temporary data, use an in-memory database ... # For tests and temporary data, use an in-memory database
... db = await lancedb.connect_async("memory://") ... db = await lancedb.connect_async("memory://")
... # Connect to LanceDB cloud ... # Connect to LanceDB cloud
@@ -550,7 +499,6 @@ async def connect_async(
api_key, api_key,
region, region,
host_override, host_override,
sql_host_override,
read_consistency_interval_secs, read_consistency_interval_secs,
client_config, client_config,
storage_options, storage_options,
@@ -573,7 +521,6 @@ __all__ = [
"connect_namespace_async", "connect_namespace_async",
"AsyncConnection", "AsyncConnection",
"AsyncJob", "AsyncJob",
"AsyncSqlQuery",
"AsyncLanceNamespaceDBConnection", "AsyncLanceNamespaceDBConnection",
"AsyncTable", "AsyncTable",
"FtsToken", "FtsToken",
@@ -588,8 +535,6 @@ __all__ = [
"vector", "vector",
"DBConnection", "DBConnection",
"Job", "Job",
"QueryDescription",
"SqlQuery",
"LanceDBConnection", "LanceDBConnection",
"LanceNamespaceDBConnection", "LanceNamespaceDBConnection",
"LsmWriteSpec", "LsmWriteSpec",
+5 -9
View File
@@ -12,7 +12,7 @@ from typing import TYPE_CHECKING, Optional, Union
import pyarrow as pa import pyarrow as pa
from .expr import Expr from .expr import Expr
from .schema import row_addressable_blob_v2_paths from .schema import blob_v2_column_paths
from .types import BlobMode, QueryProjection, QueryProjectionSpec from .types import BlobMode, QueryProjection, QueryProjectionSpec
if TYPE_CHECKING: if TYPE_CHECKING:
@@ -119,7 +119,7 @@ def blob_v2_projection_sources(
schema: pa.Schema, schema: pa.Schema,
projection: QueryProjection, projection: QueryProjection,
) -> dict[str, str]: ) -> dict[str, str]:
blob_columns = row_addressable_blob_v2_paths(schema) blob_columns = blob_v2_column_paths(schema)
if not blob_columns: if not blob_columns:
return {} return {}
columns = set(blob_columns) columns = set(blob_columns)
@@ -140,9 +140,7 @@ def v2_projection_needs_row_id(
) -> bool: ) -> bool:
if with_row_id: if with_row_id:
return False return False
return projection_includes_blob_column( return projection_includes_blob_column(projection, blob_v2_column_paths(schema))
projection, row_addressable_blob_v2_paths(schema)
)
def blob_auto_row_id_for_scan( def blob_auto_row_id_for_scan(
@@ -272,8 +270,7 @@ def _iter_projection_pairs(
if isinstance(expr, str): if isinstance(expr, str):
yield name, expr yield name, expr
elif isinstance(expr, Expr): elif isinstance(expr, Expr):
source = expr._column_name() yield name, expr.to_sql()
yield name, source if source is not None else expr.to_sql()
return return
for column in projection: for column in projection:
if isinstance(column, str): if isinstance(column, str):
@@ -283,8 +280,7 @@ def _iter_projection_pairs(
if isinstance(expr, str): if isinstance(expr, str):
yield name, expr yield name, expr
elif isinstance(expr, Expr): elif isinstance(expr, Expr):
source = expr._column_name() yield name, expr.to_sql()
yield name, source if source is not None else expr.to_sql()
def _set_blob_column(tbl: pa.Table, output_name: str, blobs: pa.Array) -> pa.Table: def _set_blob_column(tbl: pa.Table, output_name: str, blobs: pa.Array) -> pa.Table:
+15 -58
View File
@@ -1,7 +1,6 @@
from datetime import date, datetime, timedelta from datetime import date, datetime, timedelta
from decimal import Decimal from decimal import Decimal
from typing import Dict, List, Optional, Tuple, Any, TypedDict, Union, Literal from typing import Dict, List, Optional, Tuple, Any, TypedDict, Union, Literal
from uuid import UUID
import pyarrow as pa import pyarrow as pa
@@ -88,7 +87,6 @@ class PyExpr:
def contains(self, substr: "PyExpr") -> "PyExpr": ... def contains(self, substr: "PyExpr") -> "PyExpr": ...
def isin(self, values: List["PyExpr"]) -> "PyExpr": ... def isin(self, values: List["PyExpr"]) -> "PyExpr": ...
def cast(self, data_type: pa.DataType) -> "PyExpr": ... def cast(self, data_type: pa.DataType) -> "PyExpr": ...
def column_name(self) -> Optional[str]: ...
def to_sql(self) -> str: ... def to_sql(self) -> str: ...
def expr_col(name: str) -> PyExpr: ... def expr_col(name: str) -> PyExpr: ...
@@ -148,25 +146,15 @@ class Connection(object):
start_after: Optional[str], start_after: Optional[str],
limit: Optional[int], limit: Optional[int],
) -> list[str]: ... # Deprecated: Use list_tables instead ) -> list[str]: ... # Deprecated: Use list_tables instead
async def open_job(self, job_id: str) -> Job: ... def job(self, job_id: str) -> Job: ...
async def create_function_async(self, request_json: str) -> Job: ... async def create_function_async(self, request_json: str) -> FunctionJob: ...
async def get_function(self, name: str, version: str) -> str: ... async def get_function(self, name: str, version: str) -> str: ...
async def list_functions(self) -> List[str]: ...
async def drop_function(self, name: str, version: str) -> bool: ...
async def create_secret(self, name: str, value: str) -> None: ...
async def alter_secret(self, name: str, value: str) -> None: ...
async def list_secrets(self) -> List[str]: ...
async def drop_secret(self, name: str) -> None: ...
async def describe_secret(self, name: str) -> Dict[str, str]: ...
async def list_jobs(self) -> List[JobInfo]: ... async def list_jobs(self) -> List[JobInfo]: ...
async def get_job(self, job_id: str) -> Optional[JobDescription]: ...
async def cancel_job(self, job_id: str) -> bool: ... async def cancel_job(self, job_id: str) -> bool: ...
async def execute_query_async( async def job_history(
self, self, job_id: Optional[str] = None
query: str, ) -> List[pa.RecordBatch]: ...
*,
default_namespace_path: Optional[List[str]] = None,
) -> SqlQuery: ...
async def describe_query(self, query_id: UUID) -> QueryDescription: ...
async def create_table( async def create_table(
self, self,
name: str, name: str,
@@ -245,20 +233,16 @@ class BlobFile:
class Job: class Job:
@property @property
def id(self) -> Optional[str]: ... def id(self) -> Optional[str]: ...
@property
def _state(self) -> Optional[str]: ...
@property
def _description(self) -> Optional[JobDescription]: ...
async def status(self) -> str: ... async def status(self) -> str: ...
async def wait(self) -> Optional[str]: ... async def wait(self) -> None: ...
async def cancel(self) -> None: ...
class FunctionJob:
@property
def id(self) -> Optional[str]: ...
async def status(self) -> str: ...
async def wait(self) -> str: ...
async def cancel(self) -> None: ... async def cancel(self) -> None: ...
async def refresh(self) -> None: ...
async def events(
self,
*,
limit: Optional[int] = None,
filter: Optional[str] = None,
) -> pa.Table: ...
class JobInfo: class JobInfo:
@property @property
@@ -290,33 +274,10 @@ class JobDescription:
@property @property
def creation_ms(self) -> int: ... def creation_ms(self) -> int: ...
@property @property
def _spec_json(self) -> Optional[str]: ... def spec_json(self) -> Optional[str]: ...
@property
def _result_json(self) -> Optional[str]: ...
@property
def spec(self) -> Optional[Any]: ...
@property
def result(self) -> Optional[Any]: ...
@property @property
def failure(self) -> Optional[JobFailureInfo]: ... def failure(self) -> Optional[JobFailureInfo]: ...
class SqlQuery:
@property
def id(self) -> UUID: ...
async def describe(self) -> QueryDescription: ...
async def reader(self) -> RecordBatchStream: ...
async def cancel(self) -> None: ...
class QueryDescription:
@property
def id(self) -> UUID: ...
@property
def status(self) -> str: ...
@property
def progress(self) -> Optional[float]: ...
@property
def expires_at(self) -> Optional[datetime]: ...
class Table: class Table:
def name(self) -> str: ... def name(self) -> str: ...
def __repr__(self) -> str: ... def __repr__(self) -> str: ...
@@ -329,7 +290,6 @@ class Table:
mode: Literal["append", "overwrite"], mode: Literal["append", "overwrite"],
progress: Optional[Any] = None, progress: Optional[Any] = None,
write_parallelism: Optional[int] = None, write_parallelism: Optional[int] = None,
allow_external_blob_outside_bases: bool = False,
) -> AddResult: ... ) -> AddResult: ...
async def update( async def update(
self, updates: Dict[str, str], where: Optional[str] self, updates: Dict[str, str], where: Optional[str]
@@ -495,7 +455,6 @@ async def connect(
api_key: Optional[str], api_key: Optional[str],
region: Optional[str], region: Optional[str],
host_override: Optional[str], host_override: Optional[str],
sql_host_override: Optional[str],
read_consistency_interval: Optional[float], read_consistency_interval: Optional[float],
client_config: Optional[Union[ClientConfig, Dict[str, Any]]], client_config: Optional[Union[ClientConfig, Dict[str, Any]]],
storage_options: Optional[Dict[str, str]], storage_options: Optional[Dict[str, str]],
@@ -652,11 +611,9 @@ class FullTextQuery:
class PyQueryRequest: class PyQueryRequest:
limit: Optional[int] limit: Optional[int]
offset: Optional[int] offset: Optional[int]
take_offsets: Optional[List[int]]
filter: Optional[Union[str, bytes]] filter: Optional[Union[str, bytes]]
full_text_search: Optional[FullTextQuery] full_text_search: Optional[FullTextQuery]
select: Optional[Union[str, List[str]]] select: Optional[Union[str, List[str]]]
select_source_columns: Optional[Dict[str, str]]
fast_search: Optional[bool] fast_search: Optional[bool]
with_row_id: Optional[bool] with_row_id: Optional[bool]
use_lsm: Optional[bool] use_lsm: Optional[bool]
+74 -306
View File
@@ -17,10 +17,8 @@ from typing import (
List, List,
Literal, Literal,
Optional, Optional,
Sequence,
Union, Union,
) )
from uuid import UUID
if sys.version_info >= (3, 12): if sys.version_info >= (3, 12):
from typing import override from typing import override
@@ -48,17 +46,13 @@ from lance_namespace.errors import NamespaceNotEmptyError, TableNotFoundError
from . import __version__ from . import __version__
from ._lancedb import connect as lancedb_connect # type: ignore from ._lancedb import connect as lancedb_connect # type: ignore
from .functions import FunctionVersion, UdfDefinition from .functions import FunctionVersion, UdfDefinition
from .job import AsyncJob, Job, _typed_job from .job import AsyncJob, Job, _function_job
from .sql import AsyncQuery as AsyncSqlQuery
from .sql import Query as SqlQuery
from .sql import QueryDescription
from .materialized_view import ( from .materialized_view import (
AsyncMaterializedView, AsyncMaterializedView,
MaterializedView, MaterializedView,
SelectArg, SelectArg,
normalize_select, normalize_select,
) )
from .secrets import EnvVarSecret, SecretInfo, validate_secret_name
from .table import ( from .table import (
AsyncTable, AsyncTable,
LanceTable, LanceTable,
@@ -74,11 +68,10 @@ import deprecation
if TYPE_CHECKING: if TYPE_CHECKING:
import pyarrow as pa import pyarrow as pa
from .arrow import AsyncRecordBatchReader
from .pydantic import LanceModel from .pydantic import LanceModel
from ._lancedb import Connection as LanceDbConnection from ._lancedb import Connection as LanceDbConnection
from ._lancedb import JobInfo from ._lancedb import JobDescription, JobInfo
from .common import DATA, URI from .common import DATA, URI
from .embeddings import EmbeddingFunctionConfig from .embeddings import EmbeddingFunctionConfig
from ._lancedb import Session from ._lancedb import Session
@@ -694,47 +687,15 @@ class DBConnection(EnforceOverrides):
""" """
raise NotImplementedError("serialize is not supported for this connection type") raise NotImplementedError("serialize is not supported for this connection type")
def create_function( def create_function(self, definition: UdfDefinition) -> FunctionVersion:
self,
definition: UdfDefinition,
*,
secrets: Optional[Sequence[EnvVarSecret]] = None,
) -> FunctionVersion:
"""Register a scalar Python UDF and wait for its immutable version. """Register a scalar Python UDF and wait for its immutable version.
This is the blocking counterpart of :meth:`create_function_async`. This is the blocking counterpart of :meth:`create_function_async`.
Local connections raise ``NotImplementedError``. Local connections raise ``NotImplementedError``.
Parameters
----------
definition : UdfDefinition
A callable decorated with [udf][lancedb.udf].
secrets : sequence of EnvVarSecret, optional
One [EnvVarSecret][lancedb.secrets.EnvVarSecret] per credential the
Function needs, each naming a Secret and the environment variable
its value arrives in. The Function's source is unchanged by this;
it reads the variable the way it already did.
Examples
--------
```python
db.create_secret("openai-prod", os.environ["OPENAI_API_KEY"])
db.create_function(
analyze_caption,
secrets=[
EnvVarSecret(secret="openai-prod", env_variable="OPENAI_API_KEY")
],
)
```
""" """
return self.create_function_async(definition, secrets=secrets).wait() return self.create_function_async(definition).wait()
def create_function_async( def create_function_async(self, definition: UdfDefinition) -> Job[FunctionVersion]:
self,
definition: UdfDefinition,
*,
secrets: Optional[Sequence[EnvVarSecret]] = None,
) -> Job[FunctionVersion]:
"""Register a scalar Python UDF through the remote Function catalog. """Register a scalar Python UDF through the remote Function catalog.
Submission returns a typed job. The immutable Function version becomes Submission returns a typed job. The immutable Function version becomes
@@ -751,107 +712,26 @@ class DBConnection(EnforceOverrides):
"Function catalog operations are not supported for this connection type" "Function catalog operations are not supported for this connection type"
) )
def list_functions(self) -> List[FunctionVersion]: def job(self, job_id: str) -> Job:
"""List every published immutable Function version. """A [Job][lancedb.job.Job] handle for a server-side job by id.
Results are ordered by Function name then version. Local connections The handle is constructed without a server round trip; an unknown id
raise ``NotImplementedError``. surfaces when the handle is used. Dropping the handle has no effect
on the job itself.
Examples
--------
List the identities available to use in Function-backed columns:
```python
[(function.name, function.version) for function in db.list_functions()]
```
""" """
raise NotImplementedError( raise NotImplementedError("job is not supported for this connection type")
"Function catalog operations are not supported for this connection type"
)
def drop_function(self, name: str, *, version: str) -> bool:
"""Drop one exact immutable Function version from the remote catalog.
Returns True when the version changed to Dropped and False for an
idempotent replay. Local connections raise NotImplementedError.
"""
raise NotImplementedError(
"Function catalog operations are not supported for this connection type"
)
def create_secret(self, name: str, value: str) -> None:
"""Create a named Secret in this database.
Fails if the name is taken, so a create never silently becomes a
rotation. Nothing reads the value back: it is bound to a Function by
name and resolved by the service when that Function runs. Local
connections raise ``NotImplementedError``.
"""
raise NotImplementedError(
"Secret operations are not supported for this connection type"
)
def alter_secret(self, name: str, value: str) -> None:
"""Replace the credential behind an existing Secret.
Fails if it does not exist. Every Function bound to the Secret uses the
new value from its next job, and no new Function version is created --
which is how a rotation reaches columns pinned to a version registered
before it. Local connections raise ``NotImplementedError``.
"""
raise NotImplementedError(
"Secret operations are not supported for this connection type"
)
def list_secrets(self) -> List[str]:
"""The names of every Secret in this database.
Names only. No method returns a stored credential, by construction
rather than by policy. Local connections raise ``NotImplementedError``.
"""
raise NotImplementedError(
"Secret operations are not supported for this connection type"
)
def drop_secret(self, name: str) -> None:
"""Drop a Secret.
Functions bound to it fail at their next job, naming the Secret; that
is the revocation path. The name becomes free to reuse, and a new
Secret under it is picked up by everything still bound to that name.
Local connections raise ``NotImplementedError``.
"""
raise NotImplementedError(
"Secret operations are not supported for this connection type"
)
def describe_secret(self, name: str) -> SecretInfo:
"""What this database records about a Secret: name and timestamps.
Never the value -- there is no code path that could return one. Local
connections raise ``NotImplementedError``.
"""
raise NotImplementedError(
"Secret operations are not supported for this connection type"
)
def open_job(self, job_id: str) -> Job:
"""Open a server-side job by id, returning a handle with its record
already populated.
The returned [Job][lancedb.job.Job] answers for its own state,
specification, result, failure and event history, so there is no
separate connection-level call for any of them.
Raises `JobNotFoundError` when the server has no such job, the way
`open_table` does for a missing table.
"""
raise NotImplementedError("open_job is not supported for this connection type")
def list_jobs(self) -> List[JobInfo]: def list_jobs(self) -> List[JobInfo]:
"""List server-side jobs across the database's tables.""" """List server-side jobs across the database's tables."""
raise NotImplementedError("list_jobs is not supported for this connection type") raise NotImplementedError("list_jobs is not supported for this connection type")
def get_job(self, job_id: str) -> Optional[JobDescription]:
"""Describe a single server-side job by id.
Returns None when the server has no such job.
"""
raise NotImplementedError("get_job is not supported for this connection type")
def cancel_job(self, job_id: str) -> bool: def cancel_job(self, job_id: str) -> bool:
"""Request cancellation of a server-side job by id. """Request cancellation of a server-side job by id.
@@ -863,38 +743,14 @@ class DBConnection(EnforceOverrides):
"cancel_job is not supported for this connection type" "cancel_job is not supported for this connection type"
) )
def execute_query( def job_history(self, job_id: Optional[str] = None) -> List[pa.RecordBatch]:
self, """The lifecycle event history of a server-side job, as Arrow batches.
query: str,
*,
default_namespace_path: Optional[List[str]] = None,
) -> pa.RecordBatchReader:
"""Execute SQL and return a blocking Arrow reader.
This submits through :meth:`execute_query_async` and waits until the Lists history across all jobs when `job_id` is None.
initial result stream is readable. It does not wait for the full query
to finish.
""" """
return self.execute_query_async( raise NotImplementedError(
query, "job_history is not supported for this connection type"
default_namespace_path=default_namespace_path, )
).reader()
def execute_query_async(
self,
query: str,
*,
default_namespace_path: Optional[List[str]] = None,
) -> SqlQuery:
"""Start executing SQL and return its query handle.
Local connections do not support SQL.
"""
raise NotImplementedError("SQL is not supported for this connection type")
def describe_query(self, query_id: UUID) -> QueryDescription:
"""Describe a submitted SQL query by its connection-scoped id."""
raise NotImplementedError("SQL is not supported for this connection type")
class LanceDBConnection(DBConnection): class LanceDBConnection(DBConnection):
@@ -991,7 +847,6 @@ class LanceDBConnection(DBConnection):
None, None,
None, None,
None, None,
None,
read_consistency_interval_secs, read_consistency_interval_secs,
None, None,
storage_options, storage_options,
@@ -1540,59 +1395,37 @@ class LanceDBConnection(DBConnection):
) )
@override @override
def open_job(self, job_id: str) -> Job: def job(self, job_id: str) -> Job:
"""Open a server-side job by id. See """A [Job][lancedb.job.Job] handle for a server-side job by id.
[DBConnection.open_job][lancedb.db.DBConnection.open_job].
The handle is constructed without a server round trip; an unknown id
surfaces when the handle is used. Dropping the handle has no effect
on the job itself.
""" """
return Job(LOOP.run(self._conn.open_job(job_id))) return Job(self._conn.job(job_id))
@override @override
def create_function_async( def create_function_async(self, definition: UdfDefinition) -> Job[FunctionVersion]:
self, job = LOOP.run(self._conn.create_function_async(definition))
definition: UdfDefinition,
*,
secrets: Optional[Sequence[EnvVarSecret]] = None,
) -> Job[FunctionVersion]:
job = LOOP.run(self._conn.create_function_async(definition, secrets=secrets))
return Job(job) return Job(job)
@override @override
def get_function(self, name: str, *, version: str) -> FunctionVersion: def get_function(self, name: str, *, version: str) -> FunctionVersion:
return LOOP.run(self._conn.get_function(name, version=version)) return LOOP.run(self._conn.get_function(name, version=version))
@override
def list_functions(self) -> List[FunctionVersion]:
return LOOP.run(self._conn.list_functions())
@override
def drop_function(self, name: str, *, version: str) -> bool:
return LOOP.run(self._conn.drop_function(name, version=version))
@override
def create_secret(self, name: str, value: str) -> None:
LOOP.run(self._conn.create_secret(name, value))
@override
def alter_secret(self, name: str, value: str) -> None:
LOOP.run(self._conn.alter_secret(name, value))
@override
def list_secrets(self) -> List[str]:
return LOOP.run(self._conn.list_secrets())
@override
def drop_secret(self, name: str) -> None:
LOOP.run(self._conn.drop_secret(name))
@override
def describe_secret(self, name: str) -> SecretInfo:
return LOOP.run(self._conn.describe_secret(name))
@override @override
def list_jobs(self) -> List[JobInfo]: def list_jobs(self) -> List[JobInfo]:
"""List server-side jobs across the database's tables.""" """List server-side jobs across the database's tables."""
return LOOP.run(self._conn.list_jobs()) return LOOP.run(self._conn.list_jobs())
@override
def get_job(self, job_id: str) -> Optional[JobDescription]:
"""Describe a single server-side job by id.
Returns None when the server has no such job.
"""
return LOOP.run(self._conn.get_job(job_id))
@override @override
def cancel_job(self, job_id: str) -> bool: def cancel_job(self, job_id: str) -> bool:
"""Request cancellation of a server-side job by id. """Request cancellation of a server-side job by id.
@@ -1603,6 +1436,14 @@ class LanceDBConnection(DBConnection):
""" """
return LOOP.run(self._conn.cancel_job(job_id)) return LOOP.run(self._conn.cancel_job(job_id))
@override
def job_history(self, job_id: Optional[str] = None) -> List[pa.RecordBatch]:
"""The lifecycle event history of a server-side job, as Arrow batches.
Lists history across all jobs when `job_id` is None.
"""
return LOOP.run(self._conn.job_history(job_id))
@override @override
def namespace_client(self) -> LanceNamespace: def namespace_client(self) -> LanceNamespace:
"""Get the equivalent namespace client for this connection. """Get the equivalent namespace client for this connection.
@@ -2373,85 +2214,46 @@ class AsyncConnection(object):
namespace_path = [] namespace_path = []
await self._inner.drop_all_tables(namespace_path=namespace_path) await self._inner.drop_all_tables(namespace_path=namespace_path)
async def open_job(self, job_id: str) -> AsyncJob: def job(self, job_id: str) -> AsyncJob:
"""Open a server-side job by id. See """An [AsyncJob][lancedb.job.AsyncJob] handle for a server-side job
[DBConnection.open_job][lancedb.db.DBConnection.open_job]. by id.
The handle is constructed without a server round trip; an unknown id
surfaces when the handle is used. Dropping the handle has no effect
on the job itself.
""" """
return AsyncJob(await self._inner.open_job(job_id)) return AsyncJob(self._inner.job(job_id))
async def create_function_async( async def create_function_async(
self, self, definition: UdfDefinition
definition: UdfDefinition,
*,
secrets: Optional[Sequence[EnvVarSecret]] = None,
) -> AsyncJob[FunctionVersion]: ) -> AsyncJob[FunctionVersion]:
"""Register a scalar Python UDF through the remote Function catalog. """Register a scalar Python UDF through the remote Function catalog.
The returned typed job resolves to the immutable Function version. The returned typed job resolves to the immutable Function version.
``secrets`` is a sequence of Local connections raise ``NotImplementedError``.
[EnvVarSecret][lancedb.secrets.EnvVarSecret], each naming a Secret and
the environment variable its value arrives in. Local connections raise
``NotImplementedError``.
""" """
if not isinstance(definition, UdfDefinition): if not isinstance(definition, UdfDefinition):
raise TypeError("create_function_async requires a @udf definition") raise TypeError("create_function_async requires a @udf definition")
request = definition.bind_secrets(secrets) inner = await self._inner.create_function_async(
inner = await self._inner.create_function_async(request.to_canonical_json()) definition.registration_request.to_canonical_json()
return _typed_job(inner, FunctionVersion.from_json) )
return _function_job(inner)
async def get_function(self, name: str, *, version: str) -> FunctionVersion: async def get_function(self, name: str, *, version: str) -> FunctionVersion:
"""Open one exact immutable Function version from the remote catalog.""" """Open one exact immutable Function version from the remote catalog."""
return FunctionVersion.from_json(await self._inner.get_function(name, version)) return FunctionVersion.from_json(await self._inner.get_function(name, version))
async def list_functions(self) -> List[FunctionVersion]:
"""List every published immutable Function version.
Results are ordered by Function name then version. Local connections
raise ``NotImplementedError``.
"""
return [
FunctionVersion.from_json(value)
for value in await self._inner.list_functions()
]
async def drop_function(self, name: str, *, version: str) -> bool:
"""Drop one exact immutable Function version from the remote catalog."""
return await self._inner.drop_function(name, version)
async def create_secret(self, name: str, value: str) -> None:
"""Create a named Secret in this database.
Fails if the name is taken, so a create never silently becomes a
rotation. Nothing reads the value back.
"""
await self._inner.create_secret(validate_secret_name(name), value)
async def alter_secret(self, name: str, value: str) -> None:
"""Replace the credential behind an existing Secret.
Fails if it does not exist. Bound Functions use the new value from
their next job, with no new Function version.
"""
await self._inner.alter_secret(validate_secret_name(name), value)
async def list_secrets(self) -> List[str]:
"""The names of every Secret in this database. Names only."""
return await self._inner.list_secrets()
async def drop_secret(self, name: str) -> None:
"""Drop a Secret. Bound Functions fail at their next job."""
await self._inner.drop_secret(validate_secret_name(name))
async def describe_secret(self, name: str) -> SecretInfo:
"""What this database records about a Secret. Never the value."""
return SecretInfo.from_json(
await self._inner.describe_secret(validate_secret_name(name))
)
async def list_jobs(self) -> List[JobInfo]: async def list_jobs(self) -> List[JobInfo]:
"""List server-side jobs across the database's tables.""" """List server-side jobs across the database's tables."""
return await self._inner.list_jobs() return await self._inner.list_jobs()
async def get_job(self, job_id: str) -> Optional[JobDescription]:
"""Describe a single server-side job by id.
Returns None when the server has no such job.
"""
return await self._inner.get_job(job_id)
async def cancel_job(self, job_id: str) -> bool: async def cancel_job(self, job_id: str) -> bool:
"""Request cancellation of a server-side job by id. """Request cancellation of a server-side job by id.
@@ -2461,46 +2263,12 @@ class AsyncConnection(object):
""" """
return await self._inner.cancel_job(job_id) return await self._inner.cancel_job(job_id)
async def execute_query( async def job_history(self, job_id: Optional[str] = None) -> List[pa.RecordBatch]:
self, """The lifecycle event history of a server-side job, as Arrow batches.
query: str,
*,
default_namespace_path: Optional[List[str]] = None,
) -> AsyncRecordBatchReader:
"""Execute SQL and return an asynchronous Arrow reader.
This submits through :meth:`execute_query_async` and waits until the Lists history across all jobs when `job_id` is None.
initial result stream is readable. It does not wait for the full query
to finish.
""" """
submitted = await self.execute_query_async( return await self._inner.job_history(job_id)
query,
default_namespace_path=default_namespace_path,
)
return await submitted.reader()
async def execute_query_async(
self,
query: str,
*,
default_namespace_path: Optional[List[str]] = None,
) -> AsyncSqlQuery:
"""Start executing SQL and return its query handle.
The database from ``connect_async`` is used for unqualified database
references. The namespace defaults to ``["public"]``. Local
connections raise ``NotImplementedError``.
"""
return AsyncSqlQuery(
await self._inner.execute_query_async(
query,
default_namespace_path=default_namespace_path,
)
)
async def describe_query(self, query_id: UUID) -> QueryDescription:
"""Describe a submitted SQL query by its connection-scoped id."""
return await self._inner.describe_query(query_id)
async def namespace_client(self) -> LanceNamespace: async def namespace_client(self) -> LanceNamespace:
"""Get the equivalent namespace client for this connection. """Get the equivalent namespace client for this connection.
-6
View File
@@ -35,9 +35,3 @@ class JobCancelledError(RuntimeError):
"""Exception raised when an asynchronous job was cancelled.""" """Exception raised when an asynchronous job was cancelled."""
pass pass
class JobNotFoundError(ValueError):
"""Exception raised when opening a job the server does not have."""
pass
+1 -5
View File
@@ -249,10 +249,6 @@ class Expr:
# ── utilities ──────────────────────────────────────────────────────────── # ── utilities ────────────────────────────────────────────────────────────
def _column_name(self) -> str | None:
"""Return the source name when this is a bare column expression."""
return self._inner.column_name()
def to_sql(self) -> str: def to_sql(self) -> str:
"""Render the expression as a SQL string (useful for debugging).""" """Render the expression as a SQL string (useful for debugging)."""
return self._inner.to_sql() return self._inner.to_sql()
@@ -316,7 +312,7 @@ def func(name: str, *args: ExprLike) -> Expr:
-------- --------
>>> from lancedb.expr import col, func >>> from lancedb.expr import col, func
>>> func("lower", col("name")) >>> func("lower", col("name"))
Expr(lower(`name`)) Expr(lower(name))
""" """
inner_args = [_coerce(a)._inner for a in args] inner_args = [_coerce(a)._inner for a in args]
return Expr(expr_func(name, inner_args)) return Expr(expr_func(name, inner_args))
File diff suppressed because it is too large Load Diff
-12
View File
@@ -7,7 +7,6 @@ from typing import List, Literal, Optional
from ._lancedb import ( from ._lancedb import (
IndexConfig, IndexConfig,
) )
from .query import DocumentGranularity
from .types import BaseTokenizerType from .types import BaseTokenizerType
lang_mapping = { lang_mapping = {
@@ -122,11 +121,6 @@ class FTS:
>>> config = FTS(block_size=256) >>> config = FTS(block_size=256)
Create an index that treats each deepest-list element as one document:
>>> from lancedb.query import DocumentGranularity
>>> config = FTS(document_granularity=DocumentGranularity.LIST_ELEMENT)
Attributes Attributes
---------- ----------
with_position : bool, default False with_position : bool, default False
@@ -178,11 +172,6 @@ class FTS:
roughly half of the available CPU cores. The effective value is roughly half of the available CPU cores. The effective value is
limited by the available compute capacity. This build-only setting is limited by the available compute capacity. This build-only setting is
not persisted with the index and does not apply to remote tables. not persisted with the index and does not apply to remote tables.
document_granularity : DocumentGranularity, default ROW
``ROW`` treats the selected text in one table row as one document.
``LIST_ELEMENT`` treats each element of the deepest list on the indexed
field path as one document and returns its physical coordinates in
``_doc_index`` for matching queries.
Notes Notes
----- -----
@@ -207,7 +196,6 @@ class FTS:
custom_stop_words: Optional[List[str]] = None custom_stop_words: Optional[List[str]] = None
memory_limit: Optional[int] = None memory_limit: Optional[int] = None
num_workers: Optional[int] = None num_workers: Optional[int] = None
document_granularity: DocumentGranularity = DocumentGranularity.ROW
@dataclass @dataclass
+23 -245
View File
@@ -4,43 +4,25 @@
"""Handles to operations a server may run asynchronously.""" """Handles to operations a server may run asynchronously."""
import asyncio import asyncio
import json
from datetime import timedelta from datetime import timedelta
from typing import Any, Callable, Generic, Optional, TypeVar, cast from typing import Any, Generic, Optional, TypeVar, cast
import pyarrow as pa
from lancedb.background_loop import LOOP from lancedb.background_loop import LOOP
from . import _lancedb from . import _lancedb
from ._lancedb import JobDescription, JobFailureInfo, JobInfo from .functions import FunctionVersion
T = TypeVar("T") T = TypeVar("T")
__all__ = [
"AsyncJob",
"Job",
"JobDescription",
"JobFailureInfo",
"JobInfo",
]
class AsyncJob(Generic[T]): class AsyncJob(Generic[T]):
"""A handle to an operation that may still be running. """A handle to an operation that may still be running.
The operation may already be complete when the handle is created. ``T`` The operation may already be complete when the handle is created.
is the endpoint's terminal result type; unit-result jobs resolve to
``None``.
""" """
def __init__( def __init__(self, inner: Optional[Any]):
self,
inner: Optional[Any],
result_decoder: Optional[Callable[[Any], T]] = None,
):
self._inner = inner self._inner = inner
self._result_decoder = result_decoder
@property @property
def id(self) -> Optional[str]: def id(self) -> Optional[str]:
@@ -68,21 +50,17 @@ class AsyncJob(Generic[T]):
async def wait(self, timeout: Optional[timedelta] = None) -> T: async def wait(self, timeout: Optional[timedelta] = None) -> T:
"""Wait until the operation reaches a terminal state. """Wait until the operation reaches a terminal state.
Returns the endpoint's typed result, or ``None`` for a unit-result
job.
Raises `JobFailedError` if the operation failed, `JobCancelledError` Raises `JobFailedError` if the operation failed, `JobCancelledError`
if it was cancelled, and `TimeoutError` if `timeout` elapses first. if it was cancelled, and `TimeoutError` if `timeout` elapses first.
""" """
if self._inner is None: if self._inner is None:
return cast(T, None) return cast(T, None)
if timeout is None: if timeout is None:
result = await self._inner.wait() return cast(T, await self._inner.wait())
else: return cast(
result = await asyncio.wait_for(self._inner.wait(), timeout.total_seconds()) T,
if self._result_decoder is not None: await asyncio.wait_for(self._inner.wait(), timeout.total_seconds()),
return self._result_decoder(result) )
return cast(T, result)
async def cancel(self): async def cancel(self):
"""Request cancellation. Cancelling a finished operation is a no-op.""" """Request cancellation. Cancelling a finished operation is a no-op."""
@@ -90,152 +68,9 @@ class AsyncJob(Generic[T]):
return return
await self._inner.cancel() await self._inner.cancel()
async def refresh(self) -> None:
"""Ask the backend for this job's current state, and for a server-side
job its full record, then cache it for the properties below.
The properties are all `None` until this runs, because submitting an
operation returns only a job id. `status` fetches the whole record too;
`wait` records only the terminal state it establishes.
"""
if self._inner is None:
return
await self._inner.refresh()
@property
def state(self) -> Optional[str]:
"""The last observed lifecycle state, without contacting the backend.
`None` until the handle has talked to it. See :meth:`AsyncJob.refresh`.
"""
if self._inner is None:
return "finished"
return self._inner._state
@property
def job_type(self) -> Optional[str]:
"""The job's type, as the server names it.
`None` for an in-process job, which has no server-side record.
"""
return self._field("job_type")
@property
def creation_ms(self) -> Optional[int]:
"""When the job was created, in milliseconds since the epoch."""
return self._field("creation_ms")
@property
def spec(self) -> Optional[Any]:
"""The job-type-specific specification it was submitted with."""
return self._field("spec")
@property
def result(self) -> Optional[Any]:
"""The job-type-specific terminal result, as reported data rather than
the typed model :meth:`AsyncJob.wait` returns.
`None` until the job succeeds, so a job that never terminates reports
its progress through :meth:`AsyncJob.events` instead.
"""
return self._field("result")
@property
def failure(self) -> Optional[JobFailureInfo]:
"""Why the job failed, when it failed and the server reports a reason."""
return self._field("failure")
@property
def _spec_json(self) -> Optional[str]:
return self._field("_spec_json")
@property
def _result_json(self) -> Optional[str]:
return self._field("_result_json")
def _field(self, name: str) -> Optional[Any]:
description = self._inner._description if self._inner is not None else None
return getattr(description, name) if description is not None else None
async def events(
self,
*,
limit: Optional[int] = None,
filter: Optional[str] = None,
) -> "pa.Table":
"""This job's recorded lifecycle events.
Where the properties above report a terminal result only once the job
reaches one, events are written as the job runs and outlive the workers
that produced them. A distributed job records a `claim`/`claim_complete`
pair per unit of work, each carrying `rows_processed`, so a job that
never finishes still accounts for what it did.
Parameters
----------
limit: int, optional
Maximum event rows to return. The server caps results at 1000 by
default and 10,000 at most, and truncates without saying so, so
pass this for a job that emits an event per fragment.
filter: str, optional
SQL-like expression over the `state`, `updated_by`, `emitted_from`,
`emitted_by`, and `claim_entity` columns, such as
``state = 'claim_complete'``.
"""
if self._inner is None:
raise NotImplementedError(
"job event history is only available for server-side jobs"
)
return await self._inner.events(limit=limit, filter=filter)
def __repr__(self) -> str:
return _job_repr("AsyncJob", self)
_REPR_INDENT = " " * 4
def _repr_payload(value: Any) -> str:
"""Render a job payload as indented JSON, aligned under its field."""
try:
rendered = json.dumps(value, indent=4)
except TypeError:
return repr(value)
return rendered.replace("\n", "\n" + _REPR_INDENT)
def _job_repr(kind: str, job: Any) -> str:
"""Render every field the handle currently knows, omitting the rest.
One field per line, with the JSON payloads indented, because a refresh
job's spec and result are the point of printing it.
"""
state = job.state
if state is None:
# Nothing has been fetched yet, so there is nothing to lay out.
known = f"id={job.id!r}, " if job.id is not None else ""
return f"{kind}({known}not refreshed)"
fields = []
if job.id is not None:
fields.append(f"id={job.id!r}")
fields.append(f"state={state!r}")
for name in ("job_type", "creation_ms"):
value = getattr(job, name)
if value is not None:
fields.append(f"{name}={value!r}")
for name in ("spec", "result"):
value = getattr(job, name)
if value is not None:
fields.append(f"{name}={_repr_payload(value)}")
if job.failure is not None:
fields.append(f"failure={job.failure!r}")
body = "".join(f"\n{_REPR_INDENT}{field}," for field in fields)
return f"{kind}({body}\n)"
class Job(Generic[T]): class Job(Generic[T]):
"""Synchronous counterpart of `AsyncJob` with the same result type.""" """Synchronous counterpart of `AsyncJob`."""
def __init__(self, inner: Optional[AsyncJob[T]]): def __init__(self, inner: Optional[AsyncJob[T]]):
self._inner = inner self._inner = inner
@@ -261,9 +96,6 @@ class Job(Generic[T]):
def wait(self, timeout: Optional[timedelta] = None) -> T: def wait(self, timeout: Optional[timedelta] = None) -> T:
"""Block until the operation reaches a terminal state. """Block until the operation reaches a terminal state.
Returns the endpoint's typed result, or ``None`` for a unit-result
job.
Raises `JobFailedError` if the operation failed, `JobCancelledError` Raises `JobFailedError` if the operation failed, `JobCancelledError`
if it was cancelled, and `TimeoutError` if `timeout` elapses first. if it was cancelled, and `TimeoutError` if `timeout` elapses first.
""" """
@@ -277,78 +109,24 @@ class Job(Generic[T]):
return return
LOOP.run(self._inner.cancel()) LOOP.run(self._inner.cancel())
def refresh(self) -> None:
"""Ask the backend for this job's current state and record.
See :meth:`AsyncJob.refresh`. class _FunctionJobAdapter:
""" def __init__(self, inner: "_lancedb.FunctionJob"):
if self._inner is None: self._inner = inner
return
LOOP.run(self._inner.refresh())
@property @property
def state(self) -> Optional[str]: def id(self) -> Optional[str]:
"""The last observed lifecycle state. See :attr:`AsyncJob.state`.""" return self._inner.id
return self._inner.state if self._inner is not None else "finished"
@property async def status(self) -> str:
def job_type(self) -> Optional[str]: return await self._inner.status()
"""The job's type. See :attr:`AsyncJob.job_type`."""
return self._field("job_type")
@property async def wait(self) -> FunctionVersion:
def creation_ms(self) -> Optional[int]: return FunctionVersion.from_json(await self._inner.wait())
"""When the job was created. See :attr:`AsyncJob.creation_ms`."""
return self._field("creation_ms")
@property async def cancel(self):
def spec(self) -> Optional[Any]: await self._inner.cancel()
"""The job's specification. See :attr:`AsyncJob.spec`."""
return self._field("spec")
@property
def result(self) -> Optional[Any]:
"""The job's terminal result. See :attr:`AsyncJob.result`."""
return self._field("result")
@property
def failure(self) -> Optional[JobFailureInfo]:
"""Why the job failed. See :attr:`AsyncJob.failure`."""
return self._field("failure")
@property
def _spec_json(self) -> Optional[str]:
return self._field("_spec_json")
@property
def _result_json(self) -> Optional[str]:
return self._field("_result_json")
def _field(self, name: str) -> Optional[Any]:
return getattr(self._inner, name) if self._inner is not None else None
def events(
self,
*,
limit: Optional[int] = None,
filter: Optional[str] = None,
) -> "pa.Table":
"""This job's recorded lifecycle events.
See :meth:`AsyncJob.events`.
"""
if self._inner is None:
raise NotImplementedError(
"job event history is only available for server-side jobs"
)
return LOOP.run(self._inner.events(limit=limit, filter=filter))
def __repr__(self) -> str:
return _job_repr("Job", self)
def _typed_job( def _function_job(inner: "_lancedb.FunctionJob") -> AsyncJob[FunctionVersion]:
inner: "_lancedb.Job", result_decoder: Callable[[str], T] return AsyncJob(_FunctionJobAdapter(inner))
) -> AsyncJob[T]:
"""Bind an internal JSON-producing job to its public result model."""
return AsyncJob(inner, result_decoder)
+1 -5
View File
@@ -42,8 +42,6 @@ class MaterializedViewDefinition:
"""Cap on the number of rows the view holds.""" """Cap on the number of rows the view holds."""
inputs: List[str] = field(default_factory=list) inputs: List[str] = field(default_factory=list)
"""Source columns the projections and filter read.""" """Source columns the projections and filter read."""
source_namespace: List[str] = field(default_factory=list)
"""Namespace holding the source table; empty is the root namespace."""
def _definition_from_schema( def _definition_from_schema(
@@ -55,8 +53,7 @@ def _definition_from_schema(
raise ValueError(f"Table '{name}' is not a materialized view") raise ValueError(f"Table '{name}' is not a materialized view")
value = json.loads(raw) value = json.loads(raw)
kind = value.get("kind") kind = value.get("kind")
# "namespaced_select" keeps older readers from resolving the source at root. if kind != "select":
if kind not in ("select", "namespaced_select"):
raise NotImplementedError( raise NotImplementedError(
f"materialized view '{name}' is defined by '{kind}', which this " f"materialized view '{name}' is defined by '{kind}', which this "
"version of lancedb cannot refresh" "version of lancedb cannot refresh"
@@ -69,7 +66,6 @@ def _definition_from_schema(
filter=value.get("filter"), filter=value.get("filter"),
limit=value.get("limit"), limit=value.get("limit"),
inputs=value.get("inputs", []), inputs=value.get("inputs", []),
source_namespace=value.get("source_namespace", []),
) )
-35
View File
@@ -12,7 +12,6 @@ from __future__ import annotations
import sys import sys
from typing import TYPE_CHECKING, Any, Dict, Iterable, List, Optional, Union from typing import TYPE_CHECKING, Any, Dict, Iterable, List, Optional, Union
from uuid import UUID
if sys.version_info >= (3, 12): if sys.version_info >= (3, 12):
from typing import override from typing import override
@@ -49,11 +48,8 @@ from lancedb._lancedb import (
connect_namespace_client as _connect_namespace_client, connect_namespace_client as _connect_namespace_client,
) )
from lancedb.background_loop import LOOP from lancedb.background_loop import LOOP
from lancedb.arrow import AsyncRecordBatchReader
from lancedb.db import AsyncConnection, DBConnection from lancedb.db import AsyncConnection, DBConnection
from lancedb.job import AsyncJob, Job from lancedb.job import AsyncJob, Job
from lancedb.sql import AsyncQuery as AsyncSqlQuery
from lancedb.sql import QueryDescription
from lance_namespace import ( from lance_namespace import (
LanceNamespace, LanceNamespace,
connect as namespace_connect, connect as namespace_connect,
@@ -1451,37 +1447,6 @@ class AsyncLanceNamespaceDBConnection:
namespace_path=namespace_path, page_token=page_token, limit=limit namespace_path=namespace_path, page_token=page_token, limit=limit
) )
async def execute_query(
self,
query: str,
*,
default_namespace_path: Optional[List[str]] = None,
) -> AsyncRecordBatchReader:
"""Execute SQL when supported by the underlying connection."""
return await self._inner.execute_query(
query,
default_namespace_path=default_namespace_path,
)
async def execute_query_async(
self,
query: str,
*,
default_namespace_path: Optional[List[str]] = None,
) -> AsyncSqlQuery:
"""Start executing SQL when supported by the underlying connection.
Namespace-backed local connections do not support SQL.
"""
return await self._inner.execute_query_async(
query,
default_namespace_path=default_namespace_path,
)
async def describe_query(self, query_id: UUID) -> QueryDescription:
"""Describe a submitted SQL query when supported."""
return await self._inner.describe_query(query_id)
async def namespace_client(self) -> LanceNamespace: async def namespace_client(self) -> LanceNamespace:
"""Get the namespace client for this connection. """Get the namespace client for this connection.
+5 -21
View File
@@ -391,15 +391,6 @@ def _table_to_pickle_state(table: Table) -> dict[str, Any]:
} }
def _drop_base_version(permutation_data: pa.Table) -> pa.Table:
"""Strip the recorded base version so the reader leaves the base table unpinned."""
metadata = dict(permutation_data.schema.metadata or {})
if metadata.pop(b"base_version", None) is None:
return permutation_data
metadata.pop(b"base_branch", None)
return permutation_data.replace_schema_metadata(metadata)
def _table_from_pickle_state(state: dict[str, Any]) -> Table: def _table_from_pickle_state(state: dict[str, Any]) -> Table:
from . import connect from . import connect
@@ -688,15 +679,11 @@ class Permutation:
from . import connect from . import connect
connection_factory = state["connection_factory"] connection_factory = state["connection_factory"]
rebuilt_base = False
if connection_factory is not None: if connection_factory is not None:
base_table = connection_factory(state["base_table_name"]) base_table = connection_factory(state["base_table_name"])
elif "base_table_state" in state: elif "base_table_state" in state:
base_state = state["base_table_state"] base_table = _table_from_pickle_state(state["base_table_state"])
rebuilt_base = base_state["kind"] == "memory"
base_table = _table_from_pickle_state(base_state)
elif "base_table_data" in state: elif "base_table_data" in state:
rebuilt_base = True
# In-memory base table inlined into the pickle; rebuild the same # In-memory base table inlined into the pickle; rebuild the same
# way we rebuild the in-memory permutation table. # way we rebuild the in-memory permutation table.
mem_db = connect("memory://") mem_db = connect("memory://")
@@ -714,14 +701,11 @@ class Permutation:
) )
permutation_table: Optional[Table] = None permutation_table: Optional[Table] = None
permutation_data = state["permutation_data"] if state["permutation_data"] is not None:
if permutation_data is not None:
if rebuilt_base:
# The base table was materialized from Arrow, so it is a fresh
# single-version dataset and the recorded pin cannot resolve on it.
permutation_data = _drop_base_version(permutation_data)
mem_db = connect("memory://") mem_db = connect("memory://")
permutation_table = mem_db.create_table("permutation", permutation_data) permutation_table = mem_db.create_table(
"permutation", state["permutation_data"]
)
self.base_table = base_table self.base_table = base_table
self.permutation_table = permutation_table self.permutation_table = permutation_table
+9 -47
View File
@@ -109,7 +109,6 @@ def _query_is_plain_scan(query: Query) -> bool:
return ( return (
query.vector is None query.vector is None
and query.full_text_query is None and query.full_text_query is None
and query.take_offsets is None
and not query.postfilter and not query.postfilter
and not query.order_by and not query.order_by
) )
@@ -168,12 +167,6 @@ def _projection_to_scanner_kwargs(columns: QueryProjection) -> Dict[str, Any]:
return {"columns": projection} return {"columns": projection}
def _query_request_projection(req: "PyQueryRequest") -> QueryProjection:
if req.select_source_columns is not None:
return req.select_source_columns
return req.select
def _scanner_kwargs_for_query( def _scanner_kwargs_for_query(
query: Query, query: Query,
blob_mode: BlobMode, blob_mode: BlobMode,
@@ -382,13 +375,6 @@ class FullTextOperator(str, Enum):
OR = "OR" OR = "OR"
class DocumentGranularity(str, Enum):
"""The unit treated as one full-text-search document."""
ROW = "row"
LIST_ELEMENT = "list_element"
class Occur(str, Enum): class Occur(str, Enum):
SHOULD = "SHOULD" SHOULD = "SHOULD"
MUST = "MUST" MUST = "MUST"
@@ -492,10 +478,6 @@ class MatchQuery(FullTextQuery):
prefix_length : int, optional prefix_length : int, optional
The number of beginning characters being unchanged for fuzzy matching. The number of beginning characters being unchanged for fuzzy matching.
This is useful to achieve prefix matching. This is useful to achieve prefix matching.
document_granularity : DocumentGranularity, optional
Explicitly select row or deepest-list-element documents. If omitted,
the indexed granularity is inferred. When both granularities are indexed
for the field, this must be specified. With no index, row granularity is used.
""" """
query: str query: str
@@ -505,9 +487,6 @@ class MatchQuery(FullTextQuery):
max_expansions: int = pydantic.Field(50, kw_only=True) max_expansions: int = pydantic.Field(50, kw_only=True)
operator: FullTextOperator = pydantic.Field(FullTextOperator.OR, kw_only=True) operator: FullTextOperator = pydantic.Field(FullTextOperator.OR, kw_only=True)
prefix_length: int = pydantic.Field(0, kw_only=True) prefix_length: int = pydantic.Field(0, kw_only=True)
document_granularity: Optional[DocumentGranularity] = pydantic.Field(
None, kw_only=True
)
def query_type(self) -> FullTextQueryType: def query_type(self) -> FullTextQueryType:
return FullTextQueryType.MATCH return FullTextQueryType.MATCH
@@ -524,20 +503,11 @@ class PhraseQuery(FullTextQuery):
The query string to match against. The query string to match against.
column : str column : str
The name of the column to match against. The name of the column to match against.
slop : int, default 0
The maximum number of intervening positions permitted in the phrase.
document_granularity : DocumentGranularity, optional
Explicitly select row or deepest-list-element documents. If omitted,
the indexed granularity is inferred. When both granularities are indexed
for the field, this must be specified. With no index, row granularity is used.
""" """
query: str query: str
column: str column: str
slop: int = pydantic.Field(0, kw_only=True) slop: int = pydantic.Field(0, kw_only=True)
document_granularity: Optional[DocumentGranularity] = pydantic.Field(
None, kw_only=True
)
def query_type(self) -> FullTextQueryType: def query_type(self) -> FullTextQueryType:
return FullTextQueryType.MATCH_PHRASE return FullTextQueryType.MATCH_PHRASE
@@ -805,10 +775,6 @@ class Query(pydantic.BaseModel):
# offset to start fetching results from # offset to start fetching results from
offset: Optional[int] = None offset: Optional[int] = None
# Dataset offsets whose duplicate occurrences must be restored after lookup.
# This is populated when a take query is converted to this serializable form.
take_offsets: Optional[List[int]] = None
# if true, will only search the indexed data # if true, will only search the indexed data
fast_search: Optional[bool] = None fast_search: Optional[bool] = None
@@ -830,7 +796,6 @@ class Query(pydantic.BaseModel):
query = cls() query = cls()
query.limit = req.limit query.limit = req.limit
query.offset = req.offset query.offset = req.offset
query.take_offsets = req.take_offsets
query.filter = req.filter query.filter = req.filter
query.full_text_query = req.full_text_search query.full_text_query = req.full_text_search
query.columns = req.select query.columns = req.select
@@ -2811,16 +2776,15 @@ class AsyncQueryBase(object):
req = self._inner.to_query_request() req = self._inner.to_query_request()
schema = await self._table.schema() schema = await self._table.schema()
projection = _query_request_projection(req)
self._blob_auto_row_id = blob_auto_row_id_for_scan( self._blob_auto_row_id = blob_auto_row_id_for_scan(
schema, schema,
projection, req.select,
with_row_id=self._with_row_id, with_row_id=self._with_row_id,
) )
if not self._blob_auto_row_id: if not self._blob_auto_row_id:
self._blob_paths = () self._blob_paths = ()
return return
self._blob_paths = tuple(blob_v2_projection_sources(schema, projection).keys()) self._blob_paths = tuple(blob_v2_projection_sources(schema, req.select).keys())
self._inner.with_row_id() self._inner.with_row_id()
def select(self, columns: Union[List[str], dict[str, str]]) -> Self: def select(self, columns: Union[List[str], dict[str, str]]) -> Self:
@@ -3414,10 +3378,9 @@ class AsyncQuery(AsyncStandardQuery):
pass in multiple vectors. When multiple vectors are passed in, if the vector pass in multiple vectors. When multiple vectors are passed in, if the vector
column is with multivector type, then the vectors will be treated as a single column is with multivector type, then the vectors will be treated as a single
query. Or the vectors will be treated as multiple queries, this can be useful query. Or the vectors will be treated as multiple queries, this can be useful
if you want to find the nearest vectors to multiple query vectors. Flat if you want to find the nearest vectors to multiple query vectors.
searches share one table scan across the query vectors, avoiding the scan This is not expected to be faster than making multiple queries concurrently;
and memory amplification of making multiple queries concurrently. If it is just a convenience method. If multiple vectors are passed in then
multiple vectors are passed in then
an additional column `query_index` will be added to the results. This column an additional column `query_index` will be added to the results. This column
will contain the index of the query vector that the result is nearest to. will contain the index of the query vector that the result is nearest to.
""" """
@@ -3546,8 +3509,8 @@ class AsyncFTSQuery(AsyncStandardQuery):
Typically, a single vector is passed in as the query. However, you can also Typically, a single vector is passed in as the query. However, you can also
pass in multiple vectors. This can be useful if you want to find the nearest pass in multiple vectors. This can be useful if you want to find the nearest
vectors to multiple query vectors. Flat searches share one table scan across vectors to multiple query vectors. This is not expected to be faster than
the query vectors instead of issuing concurrent full scans. making multiple queries concurrently; it is just a convenience method.
If multiple vectors are passed in then an additional column `query_index` If multiple vectors are passed in then an additional column `query_index`
will be added to the results. This column will contain the index of the will be added to the results. This column will contain the index of the
query vector that the result is nearest to. query vector that the result is nearest to.
@@ -3907,15 +3870,14 @@ class AsyncHybridQuery(AsyncStandardQuery, AsyncVectorQueryBase):
blob_paths: tuple[str, ...] = () blob_paths: tuple[str, ...] = ()
if self._table is not None: if self._table is not None:
schema = await self._table.schema() schema = await self._table.schema()
projection = _query_request_projection(req)
blob_auto_row_id = blob_auto_row_id_for_scan( blob_auto_row_id = blob_auto_row_id_for_scan(
schema, schema,
projection, req.select,
with_row_id=self._with_row_id, with_row_id=self._with_row_id,
) )
if blob_auto_row_id: if blob_auto_row_id:
blob_paths = tuple( blob_paths = tuple(
blob_v2_projection_sources(schema, projection).keys() blob_v2_projection_sources(schema, req.select).keys()
) )
self._blob_auto_row_id = blob_auto_row_id self._blob_auto_row_id = blob_auto_row_id
self._blob_paths = blob_paths self._blob_paths = blob_paths
+23 -86
View File
@@ -7,18 +7,8 @@ import json
import logging import logging
from concurrent.futures import ThreadPoolExecutor from concurrent.futures import ThreadPoolExecutor
import sys import sys
from typing import ( from typing import TYPE_CHECKING, Any, Dict, Iterable, List, Optional, Union
TYPE_CHECKING,
Any,
Dict,
Iterable,
List,
Optional,
Sequence,
Union,
)
from urllib.parse import urlparse from urllib.parse import urlparse
from uuid import UUID
import warnings import warnings
if sys.version_info >= (3, 12): if sys.version_info >= (3, 12):
@@ -35,13 +25,10 @@ from ..common import DATA
from ..db import DBConnection, LOOP from ..db import DBConnection, LOOP
from ..functions import FunctionVersion, UdfDefinition from ..functions import FunctionVersion, UdfDefinition
from ..job import AsyncJob, Job from ..job import AsyncJob, Job
from ..sql import Query as SqlQuery
from ..sql import QueryDescription
from ..materialized_view import MaterializedView, SelectArg from ..materialized_view import MaterializedView, SelectArg
from ..secrets import EnvVarSecret, SecretInfo
if TYPE_CHECKING: if TYPE_CHECKING:
from .._lancedb import JobInfo from .._lancedb import JobDescription, JobInfo
from ..embeddings import EmbeddingFunctionConfig from ..embeddings import EmbeddingFunctionConfig
from lance_namespace import ( from lance_namespace import (
LanceNamespace, LanceNamespace,
@@ -129,7 +116,6 @@ class RemoteDBConnection(DBConnection):
read_timeout: Optional[float] = None, read_timeout: Optional[float] = None,
storage_options: Optional[Dict[str, str]] = None, storage_options: Optional[Dict[str, str]] = None,
read_consistency_interval: Optional[timedelta] = None, read_consistency_interval: Optional[timedelta] = None,
sql_host_override: Optional[str] = None,
): ):
"""Connect to a remote LanceDB database.""" """Connect to a remote LanceDB database."""
if isinstance(client_config, dict): if isinstance(client_config, dict):
@@ -175,7 +161,6 @@ class RemoteDBConnection(DBConnection):
self.api_key = api_key self.api_key = api_key
self.region = region self.region = region
self.host_override = host_override self.host_override = host_override
self.sql_host_override = sql_host_override
self.storage_options = storage_options self.storage_options = storage_options
self.db_name = parsed.netloc self.db_name = parsed.netloc
@@ -190,7 +175,6 @@ class RemoteDBConnection(DBConnection):
api_key=api_key, api_key=api_key,
region=region, region=region,
host_override=host_override, host_override=host_override,
sql_host_override=sql_host_override,
client_config=client_config, client_config=client_config,
storage_options=storage_options, storage_options=storage_options,
read_consistency_interval=read_consistency_interval, read_consistency_interval=read_consistency_interval,
@@ -209,7 +193,6 @@ class RemoteDBConnection(DBConnection):
"api_key": self.api_key, "api_key": self.api_key,
"region": self.region, "region": self.region,
"host_override": self.host_override, "host_override": self.host_override,
"sql_host_override": self.sql_host_override,
"client_config": _client_config_to_dict(self.client_config), "client_config": _client_config_to_dict(self.client_config),
"storage_options": self.storage_options, "storage_options": self.storage_options,
} }
@@ -749,59 +732,36 @@ class RemoteDBConnection(DBConnection):
) )
@override @override
def open_job(self, job_id: str) -> Job: def job(self, job_id: str) -> Job:
"""Open a server-side job by id. See """A [Job][lancedb.job.Job] handle for a server-side job by id.
[DBConnection.open_job][lancedb.db.DBConnection.open_job].
The handle is constructed without a server round trip; an unknown id
surfaces when the handle is used. Dropping the handle has no effect
on the job itself.
""" """
return Job(LOOP.run(self._conn.open_job(job_id))) return Job(self._conn.job(job_id))
@override @override
def create_function_async( def create_function_async(self, definition: UdfDefinition) -> Job[FunctionVersion]:
self, return Job(LOOP.run(self._conn.create_function_async(definition)))
definition: UdfDefinition,
*,
secrets: Optional[Sequence[EnvVarSecret]] = None,
) -> Job[FunctionVersion]:
job = LOOP.run(self._conn.create_function_async(definition, secrets=secrets))
return Job(job)
@override @override
def get_function(self, name: str, *, version: str) -> FunctionVersion: def get_function(self, name: str, *, version: str) -> FunctionVersion:
return LOOP.run(self._conn.get_function(name, version=version)) return LOOP.run(self._conn.get_function(name, version=version))
@override
def list_functions(self) -> List[FunctionVersion]:
return LOOP.run(self._conn.list_functions())
@override
def drop_function(self, name: str, *, version: str) -> bool:
return LOOP.run(self._conn.drop_function(name, version=version))
@override
def create_secret(self, name: str, value: str) -> None:
LOOP.run(self._conn.create_secret(name, value))
@override
def alter_secret(self, name: str, value: str) -> None:
LOOP.run(self._conn.alter_secret(name, value))
@override
def describe_secret(self, name: str) -> SecretInfo:
return LOOP.run(self._conn.describe_secret(name))
@override
def list_secrets(self) -> List[str]:
return LOOP.run(self._conn.list_secrets())
@override
def drop_secret(self, name: str) -> None:
LOOP.run(self._conn.drop_secret(name))
@override @override
def list_jobs(self) -> List["JobInfo"]: def list_jobs(self) -> List["JobInfo"]:
"""List server-side jobs across the database's tables.""" """List server-side jobs across the database's tables."""
return LOOP.run(self._conn.list_jobs()) return LOOP.run(self._conn.list_jobs())
@override
def get_job(self, job_id: str) -> Optional["JobDescription"]:
"""Describe a single server-side job by id.
Returns None when the server has no such job.
"""
return LOOP.run(self._conn.get_job(job_id))
@override @override
def cancel_job(self, job_id: str) -> bool: def cancel_job(self, job_id: str) -> bool:
"""Request cancellation of a server-side job by id. """Request cancellation of a server-side job by id.
@@ -813,35 +773,12 @@ class RemoteDBConnection(DBConnection):
return LOOP.run(self._conn.cancel_job(job_id)) return LOOP.run(self._conn.cancel_job(job_id))
@override @override
def execute_query_async( def job_history(self, job_id: Optional[str] = None) -> List[pa.RecordBatch]:
self, """The lifecycle event history of a server-side job, as Arrow batches.
query: str,
*,
default_namespace_path: Optional[List[str]] = None,
) -> SqlQuery:
"""Start executing SQL through this remote connection.
Unqualified tables use this connection's database and the Lists history across all jobs when `job_id` is None.
``["public"]`` namespace by default. Fully qualified table names may
reference other databases available to the same deployment.
""" """
return SqlQuery( return LOOP.run(self._conn.job_history(job_id))
LOOP.run(
self._conn.execute_query_async(
query,
default_namespace_path=default_namespace_path,
)
)
)
@override
def describe_query(self, query_id: UUID) -> QueryDescription:
"""Describe a submitted SQL query by its connection-scoped id."""
return LOOP.run(
self._conn.describe_query(
query_id,
)
)
@override @override
def namespace_client(self) -> LanceNamespace: def namespace_client(self) -> LanceNamespace:
+1 -4
View File
@@ -177,7 +177,4 @@ class OAuthProvider(HeaderProvider):
if not self._current_token: if not self._current_token:
raise RuntimeError("Failed to obtain OAuth token") raise RuntimeError("Failed to obtain OAuth token")
return { return {"Authorization": f"Bearer {self._current_token}"}
"Authorization": f"Bearer {self._current_token}",
"x-lancedb-credential-type": "oidc",
}
+7 -26
View File
@@ -36,7 +36,6 @@ from lancedb._lancedb import (
UpdateResult, UpdateResult,
) )
from lancedb.embeddings.base import EmbeddingFunctionConfig from lancedb.embeddings.base import EmbeddingFunctionConfig
from lancedb.expr import Expr
from lancedb.index import ( from lancedb.index import (
FTS, FTS,
BTree, BTree,
@@ -50,7 +49,7 @@ from lancedb.index import (
LabelList, LabelList,
) )
from lancedb.job import Job from lancedb.job import Job
from lancedb.functions import FunctionApplication, RefreshColumnResult from lancedb.functions import FunctionApplication
from lancedb.remote.db import LOOP from lancedb.remote.db import LOOP
from lancedb.table import IndexConfigType, KNOWN_METRICS from lancedb.table import IndexConfigType, KNOWN_METRICS
import pyarrow as pa import pyarrow as pa
@@ -62,20 +61,11 @@ from lancedb.table import _normalize_progress
from ..query import ( from ..query import (
AnalyzePlanDistributedMetrics, AnalyzePlanDistributedMetrics,
DocumentGranularity,
LanceQueryBuilder, LanceQueryBuilder,
LanceTakeQueryBuilder, LanceTakeQueryBuilder,
LanceVectorQueryBuilder, LanceVectorQueryBuilder,
) )
from ..table import ( from ..table import AsyncTable, BlobMode, Branches, IndexStatistics, Query, Table, Tags
AsyncTable,
BlobMode,
Branches,
IndexStatistics,
Query,
Table,
Tags,
)
from ..types import BaseTokenizerType from ..types import BaseTokenizerType
@@ -359,7 +349,6 @@ class RemoteTable(Table):
ngram_max_length: int = 3, ngram_max_length: int = 3,
prefix_only: bool = False, prefix_only: bool = False,
block_size: int = 128, block_size: int = 128,
document_granularity: DocumentGranularity = DocumentGranularity.ROW,
name: Optional[str] = None, name: Optional[str] = None,
): ):
"""Create a full-text search index on a column. """Create a full-text search index on a column.
@@ -382,7 +371,6 @@ class RemoteTable(Table):
ngram_max_length=ngram_max_length, ngram_max_length=ngram_max_length,
prefix_only=prefix_only, prefix_only=prefix_only,
block_size=block_size, block_size=block_size,
document_granularity=document_granularity,
) )
LOOP.run( LOOP.run(
self._table.create_index( self._table.create_index(
@@ -548,7 +536,6 @@ class RemoteTable(Table):
LOOP.run( LOOP.run(
self._table.create_index( self._table.create_index(
column, column,
replace=replace,
config=config, config=config,
wait_timeout=wait_timeout, wait_timeout=wait_timeout,
name=name, name=name,
@@ -623,7 +610,6 @@ class RemoteTable(Table):
fill_value: float = 0.0, fill_value: float = 0.0,
progress: Optional[Union[bool, Callable, Any]] = None, progress: Optional[Union[bool, Callable, Any]] = None,
write_parallelism: Optional[int] = None, write_parallelism: Optional[int] = None,
allow_external_blob_outside_bases: bool = False,
) -> AddResult: ) -> AddResult:
"""Add more data to the [Table][lancedb.table.Table]. """Add more data to the [Table][lancedb.table.Table].
@@ -656,8 +642,6 @@ class RemoteTable(Table):
data in flight. Defaults to an estimate based on the data size, data in flight. Defaults to an estimate based on the data size,
capped at the number of CPU cores. Lower this if bulk ingestion is capped at the number of CPU cores. Lower this if bulk ingestion is
using too much memory. using too much memory.
allow_external_blob_outside_bases: bool, default False
Not supported on LanceDB Cloud. Setting this raises.
Returns Returns
------- -------
@@ -674,7 +658,6 @@ class RemoteTable(Table):
fill_value=fill_value, fill_value=fill_value,
progress=progress, progress=progress,
write_parallelism=write_parallelism, write_parallelism=write_parallelism,
allow_external_blob_outside_bases=allow_external_blob_outside_bases,
) )
) )
finally: finally:
@@ -873,7 +856,7 @@ class RemoteTable(Table):
def update( def update(
self, self,
where: Optional[Union[str, Expr]] = None, where: Optional[str] = None,
values: Optional[dict] = None, values: Optional[dict] = None,
*, *,
values_sql: Optional[Dict[str, str]] = None, values_sql: Optional[Dict[str, str]] = None,
@@ -884,11 +867,9 @@ class RemoteTable(Table):
Parameters Parameters
---------- ----------
where: str or [Expr][lancedb.expr.Expr], optional where: str, optional
The filter condition. Can be a SQL string or a type-safe The SQL where clause to use when updating rows. For example, 'x = 2'
[Expr][lancedb.expr.Expr] built with [col][lancedb.expr.col] and or 'x IN (1, 2, 3)'. The filter must not be empty, or it will error.
[lit][lancedb.expr.lit]. The filter must not be empty, or it will
error.
values: dict, optional values: dict, optional
The values to update. The keys are the column names and the values The values to update. The keys are the column names and the values
are the values to set. are the values to set.
@@ -991,7 +972,7 @@ class RemoteTable(Table):
def refresh_column(self, column: str): def refresh_column(self, column: str):
return LOOP.run(self._table.refresh_column(column)) return LOOP.run(self._table.refresh_column(column))
def refresh_column_async(self, column: str) -> Job[RefreshColumnResult]: def refresh_column_async(self, column: str) -> Job:
return Job(LOOP.run(self._table.refresh_column_async(column))) return Job(LOOP.run(self._table.refresh_column_async(column)))
def alter_columns( def alter_columns(
+34 -101
View File
@@ -4,34 +4,30 @@
"""Schema helpers for Lance blob columns.""" """Schema helpers for Lance blob columns."""
import importlib
from typing import TYPE_CHECKING
import pyarrow as pa import pyarrow as pa
import pyarrow.ipc
if TYPE_CHECKING:
from lance.blob import BlobType as BlobType
_BLOB_EXTENSION_NAME = "lance.blob.v2" _BLOB_EXTENSION_NAME = "lance.blob.v2"
_BLOB_V1_KEY = "lance-encoding:blob" _BLOB_V1_KEY = "lance-encoding:blob"
_ARROW_EXT_NAME_KEY = "ARROW:extension:name" _ARROW_EXT_NAME_KEY = "ARROW:extension:name"
_BLOB_V2_STORAGE_TYPE = pa.struct(
[
pa.field("data", pa.large_binary(), nullable=True),
pa.field("uri", pa.utf8(), nullable=True),
pa.field("position", pa.uint64(), nullable=True),
pa.field("size", pa.uint64(), nullable=True),
]
)
_resolved_blob_type = None
class _FallbackBlobType(pa.ExtensionType): class BlobType(pa.ExtensionType):
"""lance.blob.v2 extension type used when pylance is not installed.""" """PyArrow extension type for a Lance blob v2 column.
Queries return descriptors; call :meth:`~lancedb.table.Table.fetch_blob_files`
for lazy reads or :meth:`~lancedb.table.Table.fetch_blobs` for eager bytes.
"""
def __init__(self) -> None: def __init__(self) -> None:
pa.ExtensionType.__init__(self, _BLOB_V2_STORAGE_TYPE, _BLOB_EXTENSION_NAME) storage_type = pa.struct(
[
pa.field("data", pa.large_binary(), nullable=True),
pa.field("uri", pa.utf8(), nullable=True),
pa.field("position", pa.uint64(), nullable=True),
pa.field("size", pa.uint64(), nullable=True),
]
)
super().__init__(storage_type, _BLOB_EXTENSION_NAME)
def __arrow_ext_serialize__(self) -> bytes: def __arrow_ext_serialize__(self) -> bytes:
return b"" return b""
@@ -39,16 +35,23 @@ class _FallbackBlobType(pa.ExtensionType):
@classmethod @classmethod
def __arrow_ext_deserialize__( def __arrow_ext_deserialize__(
cls, storage_type: pa.DataType, serialized: bytes cls, storage_type: pa.DataType, serialized: bytes
) -> "_FallbackBlobType": ) -> "BlobType":
return cls() return cls()
def __reduce__(self): def __reduce__(self):
# Ensure pickle round-trips on older pyarrow (apache/arrow#35599).
return type(self).__arrow_ext_deserialize__, ( return type(self).__arrow_ext_deserialize__, (
self.storage_type, self.storage_type,
self.__arrow_ext_serialize__(), self.__arrow_ext_serialize__(),
) )
try:
pa.register_extension_type(BlobType()) # type: ignore[arg-type]
except pa.ArrowKeyError:
pass
def _metadata_value(metadata: dict, key: str): def _metadata_value(metadata: dict, key: str):
return metadata.get(key.encode()) or metadata.get(key) return metadata.get(key.encode()) or metadata.get(key)
@@ -89,105 +92,43 @@ def is_blob_like_field(field: pa.Field) -> bool:
return is_blob_v2_field(field) or _metadata_marks_legacy_blob(field.metadata or {}) return is_blob_v2_field(field) or _metadata_marks_legacy_blob(field.metadata or {})
def _collect_blob_paths(schema: pa.Schema, is_blob) -> list[tuple[str, bool]]: def _collect_blob_paths(schema: pa.Schema, is_blob) -> list[str]:
"""Walk the schema and return (path, has_list_ancestor) for each blob field.""" paths: list[str] = []
paths: list[tuple[str, bool]] = []
def walk(fields, prefix: str, has_list_ancestor: bool) -> None: def walk(fields, prefix: str) -> None:
for field in fields: for field in fields:
path = f"{prefix}.{field.name}" if prefix else field.name path = f"{prefix}.{field.name}" if prefix else field.name
if is_blob(field): if is_blob(field):
paths.append((path, has_list_ancestor)) paths.append(path)
elif pa.types.is_struct(field.type): elif pa.types.is_struct(field.type):
walk(field.type, path, has_list_ancestor) walk(field.type, path)
elif ( elif (
pa.types.is_list(field.type) pa.types.is_list(field.type)
or pa.types.is_large_list(field.type) or pa.types.is_large_list(field.type)
or pa.types.is_fixed_size_list(field.type) or pa.types.is_fixed_size_list(field.type)
): ):
walk([field.type.value_field], path, True) walk([field.type.value_field], path)
walk(schema, "", False) walk(schema, "")
return paths return paths
def blob_column_paths(schema: pa.Schema) -> list[str]: def blob_column_paths(schema: pa.Schema) -> list[str]:
"""Dotted paths of blob-like columns (v2 extension or legacy metadata).""" """Dotted paths of blob-like columns (v2 extension or legacy metadata)."""
return [path for path, _ in _collect_blob_paths(schema, is_blob_like_field)] return _collect_blob_paths(schema, is_blob_like_field)
def blob_v2_column_paths(schema: pa.Schema) -> list[str]: def blob_v2_column_paths(schema: pa.Schema) -> list[str]:
return [path for path, _ in _collect_blob_paths(schema, is_blob_v2_field)] return _collect_blob_paths(schema, is_blob_v2_field)
def row_addressable_blob_v2_paths(schema: pa.Schema) -> list[str]:
"""Blob v2 paths with one blob addressable by table row id.
``fetch_blobs`` and the descriptor row-id ride-along address one blob per
row, so a blob inside a list container has no row-id slot and no fetch
path. Those columns still store and query as raw descriptors.
"""
return [
path
for path, has_list_ancestor in _collect_blob_paths(schema, is_blob_v2_field)
if not has_list_ancestor
]
def schema_has_blob_field(schema: pa.Schema) -> bool: def schema_has_blob_field(schema: pa.Schema) -> bool:
return bool(blob_column_paths(schema)) return bool(blob_column_paths(schema))
def _deserialize_registered_type(extension_type: pa.ExtensionType) -> pa.DataType:
"""Return the type Arrow reconstructs for this extension name."""
schema = pa.schema([pa.field("value", extension_type)])
restored = pa.ipc.read_schema(schema.serialize())
return restored.field("value").type
def _resolve_blob_type():
"""Return the BlobType class this process should use.
pylance's class when it owns the lance.blob.v2 registry entry,
otherwise LanceDB's fallback. A different registered class is an error.
"""
global _resolved_blob_type
if _resolved_blob_type is not None:
return _resolved_blob_type
try:
blob_module = importlib.import_module("lance.blob")
except ModuleNotFoundError as err:
if err.name not in ("lance", "lance.blob"):
raise
else:
blob_type = getattr(blob_module, "BlobType", None)
if blob_type is not None:
registered_type = _deserialize_registered_type(blob_type())
if type(registered_type) is not blob_type:
registered_cls = type(registered_type)
raise ValueError(
"lance.blob.v2 is already registered by "
f"{registered_cls.__module__}.{registered_cls.__qualname__}"
)
_resolved_blob_type = blob_type
return blob_type
try:
pa.register_extension_type(_FallbackBlobType()) # type: ignore[arg-type]
except pa.ArrowKeyError as err:
raise ValueError(
"lance.blob.v2 is already registered by another extension class"
) from err
_resolved_blob_type = _FallbackBlobType
return _resolved_blob_type
def blob(name: str, nullable: bool = True) -> pa.Field: def blob(name: str, nullable: bool = True) -> pa.Field:
"""Create a Lance blob v2 column field. """Create a Lance blob v2 column field."""
return pa.field(name, BlobType(), nullable=nullable)
When pylance is installed this is ``lance.blob.BlobType``.
"""
blob_type = _resolve_blob_type()
return pa.field(name, blob_type(), nullable=nullable)
def vector(dimension: int, value_type: pa.DataType = pa.float32()) -> pa.DataType: def vector(dimension: int, value_type: pa.DataType = pa.float32()) -> pa.DataType:
@@ -214,11 +155,3 @@ def vector(dimension: int, value_type: pa.DataType = pa.float32()) -> pa.DataTyp
... ]) ... ])
""" """
return pa.list_(value_type, dimension) return pa.list_(value_type, dimension)
def __getattr__(name: str):
if name == "BlobType":
blob_type = _resolve_blob_type()
globals()["BlobType"] = blob_type
return blob_type
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")

Some files were not shown because too many files have changed in this diff Show More