mirror of
https://github.com/lancedb/lancedb.git
synced 2026-08-27 16:38:31 +00:00
Compare commits
6 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| c0a9a4d48a | |||
| 32c3d39f2a | |||
| 7a41f4a5eb | |||
| 28fc8f0f26 | |||
| 7677a5279c | |||
| b6c645592a |
@@ -1,20 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "lancedb",
|
|
||||||
"interface": {
|
|
||||||
"displayName": "LanceDB"
|
|
||||||
},
|
|
||||||
"plugins": [
|
|
||||||
{
|
|
||||||
"name": "lancedb",
|
|
||||||
"source": {
|
|
||||||
"source": "local",
|
|
||||||
"path": "./plugins/lancedb"
|
|
||||||
},
|
|
||||||
"policy": {
|
|
||||||
"installation": "AVAILABLE",
|
|
||||||
"authentication": "ON_INSTALL"
|
|
||||||
},
|
|
||||||
"category": "Developer Tools"
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
@@ -1,11 +0,0 @@
|
|||||||
# Agent Skills
|
|
||||||
|
|
||||||
This directory contains repo-scoped code agent skills for the LanceDB project.
|
|
||||||
|
|
||||||
Each skill is a folder that contains a required `SKILL.md` and optional bundled resources.
|
|
||||||
|
|
||||||
Codex discovers skills from `.agents/skills` in the current working directory and parent directories.
|
|
||||||
|
|
||||||
The `lancedb` skill lives in the `plugins/lancedb` plugin (see `plugins/lancedb/skills/lancedb`)
|
|
||||||
so it can be installed via the plugin marketplaces (`.claude-plugin/marketplace.json` and
|
|
||||||
`.agents/plugins/marketplace.json`); the `lancedb` entry here is a symlink into that plugin.
|
|
||||||
@@ -1 +0,0 @@
|
|||||||
../../plugins/lancedb/skills/lancedb
|
|
||||||
@@ -1,98 +0,0 @@
|
|||||||
---
|
|
||||||
name: lancedb-update-lance-dependency
|
|
||||||
description: Update LanceDB to a specific Lance release or tag. Use when bumping Lance dependencies in the lancedb repository, including Rust workspace Lance crates, Java lance-core, validation, branch creation, commit, push, and PR creation when requested.
|
|
||||||
---
|
|
||||||
|
|
||||||
# LanceDB Update Lance Dependency
|
|
||||||
|
|
||||||
## Scope
|
|
||||||
|
|
||||||
Use this skill in the `lancedb/lancedb` repository when updating the Lance dependency to a specific Lance version or tag.
|
|
||||||
|
|
||||||
Inputs can be a version (`7.2.0-beta.1`), a tag (`v7.2.0-beta.1`), a tag ref (`refs/tags/v7.2.0-beta.1`), or `latest`.
|
|
||||||
|
|
||||||
## Workflow
|
|
||||||
|
|
||||||
1. Confirm the worktree status with `git status --short`.
|
|
||||||
2. Resolve the target Lance version:
|
|
||||||
|
|
||||||
- If the input is `latest`, empty, or omitted, run:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
python3 ci/check_lance_release.py
|
|
||||||
```
|
|
||||||
|
|
||||||
Parse the JSON output. If `needs_update` is not `true`, stop without creating a PR. Otherwise use `latest_tag`.
|
|
||||||
|
|
||||||
- If the input is explicit, use it directly.
|
|
||||||
|
|
||||||
3. Compute update metadata without changing files:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
python3 ci/update_lance_dependency.py "$TAG_OR_VERSION" --metadata-only
|
|
||||||
```
|
|
||||||
|
|
||||||
Before making changes, check for an existing open PR with the emitted `pr_title`:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
gh pr list --search "\"$PR_TITLE\" in:title" --state open --limit 1 --json number,url,title
|
|
||||||
```
|
|
||||||
|
|
||||||
If a matching open PR exists, stop and report it instead of creating a duplicate.
|
|
||||||
|
|
||||||
4. Run the deterministic update entrypoint:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
python3 ci/update_lance_dependency.py "$TAG_OR_VERSION"
|
|
||||||
```
|
|
||||||
|
|
||||||
This updates the Rust workspace Lance dependencies through `ci/set_lance_version.py`, updates `java/pom.xml`, refreshes Cargo metadata, and prints JSON metadata containing `branch_name`, `commit_message`, and `pr_title`.
|
|
||||||
|
|
||||||
5. Run validation:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
cargo clippy --quiet --workspace --tests --all-features -- -D warnings
|
|
||||||
cargo fmt --all --quiet
|
|
||||||
```
|
|
||||||
|
|
||||||
Fix real diagnostics and rerun clippy until it succeeds. Do not skip warnings.
|
|
||||||
|
|
||||||
6. Inspect `git status --short` and `git diff` to ensure only the Lance dependency update and required compatibility fixes are present.
|
|
||||||
|
|
||||||
7. If the task only asks to prepare local changes, stop here and report the changed files and validation result.
|
|
||||||
|
|
||||||
8. If the task asks to publish the update, create a branch using the printed `branch_name`, stage all relevant files, and commit using the printed `commit_message`. Do not amend or rewrite existing commits.
|
|
||||||
|
|
||||||
9. Push to `origin`. Before creating the PR, check that the current token has push permission:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
gh api repos/lancedb/lancedb --jq .permissions.push
|
|
||||||
```
|
|
||||||
|
|
||||||
If the remote branch already exists for the same generated branch name, delete the remote ref with `gh api -X DELETE repos/lancedb/lancedb/git/refs/heads/$BRANCH_NAME`, then push. Do not force-push.
|
|
||||||
|
|
||||||
10. Create a PR targeting `main` with the printed `pr_title`. If there is no PR template, keep the body to two or three concise sentences: state the Lance dependency bump, note any required compatibility fixes, and link the triggering Lance tag or release.
|
|
||||||
|
|
||||||
11. Read back the remote PR title after creation. If it is not a Conventional Commit title, fix it immediately.
|
|
||||||
|
|
||||||
12. When running in GitHub Actions after creating the LanceDB PR, trigger the Sophon dependency update:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
gh workflow run codex-bump-lancedb-lance.yml \
|
|
||||||
--repo lancedb/sophon \
|
|
||||||
-f lance_ref="$LANCE_TAG" \
|
|
||||||
-f lancedb_ref="$BRANCH_NAME"
|
|
||||||
gh run list --repo lancedb/sophon --workflow codex-bump-lancedb-lance.yml --limit 1 --json databaseId,url,displayTitle
|
|
||||||
```
|
|
||||||
|
|
||||||
Use the emitted metadata `tag` value as `LANCE_TAG`. Do this only after a new LanceDB PR has been created. If the update was skipped because no update is needed or an open PR already exists, do not trigger Sophon.
|
|
||||||
|
|
||||||
## GitHub Actions
|
|
||||||
|
|
||||||
When this skill is used from GitHub Actions, `TAG`, `GH_TOKEN`, and `GITHUB_TOKEN` may already be set. Resolve `latest` first when `TAG` is empty. Once an explicit tag or version is known, use:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
python3 ci/update_lance_dependency.py "$TAG" --github-output "$GITHUB_OUTPUT"
|
|
||||||
```
|
|
||||||
|
|
||||||
Then use the emitted `branch_name`, `commit_message`, and `pr_title` values for branch, commit, and PR creation.
|
|
||||||
+1
-10
@@ -1,5 +1,5 @@
|
|||||||
[tool.bumpversion]
|
[tool.bumpversion]
|
||||||
current_version = "0.37.1-beta.0"
|
current_version = "0.30.0"
|
||||||
parse = """(?x)
|
parse = """(?x)
|
||||||
(?P<major>0|[1-9]\\d*)\\.
|
(?P<major>0|[1-9]\\d*)\\.
|
||||||
(?P<minor>0|[1-9]\\d*)\\.
|
(?P<minor>0|[1-9]\\d*)\\.
|
||||||
@@ -23,8 +23,6 @@ allow_dirty = true
|
|||||||
commit = true
|
commit = true
|
||||||
message = "Bump version: {current_version} → {new_version}"
|
message = "Bump version: {current_version} → {new_version}"
|
||||||
commit_args = ""
|
commit_args = ""
|
||||||
# bump-my-version >=1.4.0 rejects pre_commit_hooks containing shell syntax unless opted in.
|
|
||||||
allow_shell_hooks = true
|
|
||||||
|
|
||||||
# Java maven files
|
# Java maven files
|
||||||
pre_commit_hooks = [
|
pre_commit_hooks = [
|
||||||
@@ -75,13 +73,6 @@ filename = "nodejs/Cargo.toml"
|
|||||||
replace = "\nversion = \"{new_version}\""
|
replace = "\nversion = \"{new_version}\""
|
||||||
search = "\nversion = \"{current_version}\""
|
search = "\nversion = \"{current_version}\""
|
||||||
|
|
||||||
# The Python package takes its version from here (pyproject.toml declares
|
|
||||||
# `dynamic = ["version"]`, so maturin reads it out of the crate manifest).
|
|
||||||
[[tool.bumpversion.files]]
|
|
||||||
filename = "python/Cargo.toml"
|
|
||||||
replace = "\nversion = \"{new_version}\""
|
|
||||||
search = "\nversion = \"{current_version}\""
|
|
||||||
|
|
||||||
# Java documentation
|
# Java documentation
|
||||||
[[tool.bumpversion.files]]
|
[[tool.bumpversion.files]]
|
||||||
filename = "docs/src/java/java.md"
|
filename = "docs/src/java/java.md"
|
||||||
|
|||||||
@@ -1,19 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "lancedb",
|
|
||||||
"owner": {
|
|
||||||
"name": "LanceDB"
|
|
||||||
},
|
|
||||||
"description": "LanceDB plugins for Claude Code.",
|
|
||||||
"plugins": [
|
|
||||||
{
|
|
||||||
"name": "lancedb",
|
|
||||||
"source": "./plugins/lancedb",
|
|
||||||
"description": "Write, review, debug, and document LanceDB pipelines in Python and TypeScript that work across local LanceDB OSS tables and remote LanceDB Enterprise/Cloud tables.",
|
|
||||||
"version": "0.1.0",
|
|
||||||
"author": {
|
|
||||||
"name": "LanceDB"
|
|
||||||
},
|
|
||||||
"category": "development"
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
@@ -27,31 +27,19 @@ runs:
|
|||||||
# Extract failed job names
|
# Extract failed job names
|
||||||
FAILED_JOBS=$(echo "$JOB_RESULTS" | jq -r 'to_entries | map(select(.value.result == "failure")) | map(.key) | join(", ")')
|
FAILED_JOBS=$(echo "$JOB_RESULTS" | jq -r 'to_entries | map(select(.value.result == "failure")) | map(.key) | join(", ")')
|
||||||
|
|
||||||
TITLE="$WORKFLOW_NAME Failed ($FAILED_JOBS)"
|
# Create issue with workflow name, failed jobs, and run URL
|
||||||
|
gh issue create \
|
||||||
# This action now also runs on nightly schedules, so a breakage that
|
--title "$WORKFLOW_NAME Failed ($FAILED_JOBS)" \
|
||||||
# persists for a few days would otherwise file one issue per night.
|
--body "The workflow **$WORKFLOW_NAME** failed during execution.
|
||||||
# Comment on the open report instead when one already exists.
|
|
||||||
EXISTING=$(gh issue list --state open --label ci --limit 100 --json number,title \
|
|
||||||
| jq -r --arg title "$TITLE" 'map(select(.title == $title)) | .[0].number // empty')
|
|
||||||
|
|
||||||
if [ -n "$EXISTING" ]; then
|
|
||||||
gh issue comment "$EXISTING" --body "Failed again: $RUN_URL"
|
|
||||||
echo "Commented on existing issue #$EXISTING"
|
|
||||||
else
|
|
||||||
gh issue create \
|
|
||||||
--title "$TITLE" \
|
|
||||||
--body "The workflow **$WORKFLOW_NAME** failed during execution.
|
|
||||||
|
|
||||||
**Failed jobs:** $FAILED_JOBS
|
**Failed jobs:** $FAILED_JOBS
|
||||||
|
|
||||||
**Run URL:** $RUN_URL
|
**Run URL:** $RUN_URL
|
||||||
|
|
||||||
Please investigate the failed jobs and address any issues." \
|
Please investigate the failed jobs and address any issues." \
|
||||||
--label "ci"
|
--label "ci"
|
||||||
|
|
||||||
echo "Issue created successfully"
|
echo "Issue created successfully"
|
||||||
fi
|
|
||||||
else
|
else
|
||||||
echo "No job failures detected, skipping issue creation"
|
echo "No job failures detected, skipping issue creation"
|
||||||
fi
|
fi
|
||||||
|
|||||||
@@ -21,14 +21,3 @@ updates:
|
|||||||
update-types:
|
update-types:
|
||||||
- minor
|
- minor
|
||||||
- patch
|
- patch
|
||||||
|
|
||||||
- package-ecosystem: pip
|
|
||||||
directory: /python
|
|
||||||
schedule:
|
|
||||||
interval: weekly
|
|
||||||
# Only update uv.lock, never widen version requirements in pyproject.toml.
|
|
||||||
versioning-strategy: lockfile-only
|
|
||||||
groups:
|
|
||||||
python-deps:
|
|
||||||
patterns:
|
|
||||||
- "*"
|
|
||||||
|
|||||||
@@ -18,14 +18,6 @@ inputs:
|
|||||||
description: "The manylinux version to build for"
|
description: "The manylinux version to build for"
|
||||||
required: false
|
required: false
|
||||||
default: "2_17"
|
default: "2_17"
|
||||||
package-name:
|
|
||||||
description: "Override [project] name in python/pyproject.toml (e.g. 'lancedb-compat'). Default keeps 'lancedb'."
|
|
||||||
required: false
|
|
||||||
default: "lancedb"
|
|
||||||
rustflags:
|
|
||||||
description: "RUSTFLAGS for the build container, as a single whitespace-free token (e.g. '-Ctarget-cpu=x86-64-v2'). Empty leaves RUSTFLAGS unset, keeping the defaults from .cargo/config.toml."
|
|
||||||
required: false
|
|
||||||
default: ""
|
|
||||||
runs:
|
runs:
|
||||||
using: "composite"
|
using: "composite"
|
||||||
steps:
|
steps:
|
||||||
@@ -35,18 +27,6 @@ runs:
|
|||||||
ARM_BUILD: ${{ inputs.arm-build }}
|
ARM_BUILD: ${{ inputs.arm-build }}
|
||||||
run: |
|
run: |
|
||||||
echo "ARM BUILD: $ARM_BUILD"
|
echo "ARM BUILD: $ARM_BUILD"
|
||||||
- name: Patch package name for variant build
|
|
||||||
if: ${{ inputs.package-name != 'lancedb' }}
|
|
||||||
shell: bash
|
|
||||||
env:
|
|
||||||
PACKAGE_NAME: ${{ inputs.package-name }}
|
|
||||||
run: |
|
|
||||||
# Swap the [project] name so this build produces e.g. lancedb-compat
|
|
||||||
# wheels. The package still installs files under the lancedb/
|
|
||||||
# namespace -- import lancedb still works after pip install.
|
|
||||||
sed -i.bak 's/^name = "lancedb"$/name = "'"$PACKAGE_NAME"'"/' python/pyproject.toml
|
|
||||||
rm -f python/pyproject.toml.bak
|
|
||||||
grep '^name = ' python/pyproject.toml
|
|
||||||
- name: Build x86_64 Manylinux wheel
|
- name: Build x86_64 Manylinux wheel
|
||||||
if: ${{ inputs.arm-build == 'false' }}
|
if: ${{ inputs.arm-build == 'false' }}
|
||||||
uses: PyO3/maturin-action@v1
|
uses: PyO3/maturin-action@v1
|
||||||
@@ -54,16 +34,15 @@ runs:
|
|||||||
maturin-version: "1.12.4"
|
maturin-version: "1.12.4"
|
||||||
command: build
|
command: build
|
||||||
working-directory: python
|
working-directory: python
|
||||||
docker-options: "-e PIP_EXTRA_INDEX_URL='https://pypi.fury.io/lance-format/ https://pypi.fury.io/lancedb/' -e PROTOC=/usr/local/bin/protoc ${{ inputs.rustflags != '' && format('-e RUSTFLAGS={0}', inputs.rustflags) || '' }}"
|
docker-options: "-e PIP_EXTRA_INDEX_URL='https://pypi.fury.io/lance-format/ https://pypi.fury.io/lancedb/'"
|
||||||
target: x86_64-unknown-linux-gnu
|
target: x86_64-unknown-linux-gnu
|
||||||
manylinux: ${{ inputs.manylinux }}
|
manylinux: ${{ inputs.manylinux }}
|
||||||
args: ${{ inputs.args }}
|
args: ${{ inputs.args }}
|
||||||
before-script-linux: |
|
before-script-linux: |
|
||||||
set -e
|
set -e
|
||||||
curl -fsSL https://github.com/protocolbuffers/protobuf/releases/download/v24.4/protoc-24.4-linux-x86_64.zip -o /tmp/protoc.zip
|
curl -L https://github.com/protocolbuffers/protobuf/releases/download/v24.4/protoc-24.4-linux-$(uname -m).zip > /tmp/protoc.zip \
|
||||||
unzip /tmp/protoc.zip -d /usr/local
|
&& unzip /tmp/protoc.zip -d /usr/local \
|
||||||
rm /tmp/protoc.zip
|
&& rm /tmp/protoc.zip
|
||||||
/usr/local/bin/protoc --version
|
|
||||||
- name: Build Arm Manylinux Wheel
|
- name: Build Arm Manylinux Wheel
|
||||||
if: ${{ inputs.arm-build == 'true' }}
|
if: ${{ inputs.arm-build == 'true' }}
|
||||||
uses: PyO3/maturin-action@v1
|
uses: PyO3/maturin-action@v1
|
||||||
@@ -71,14 +50,13 @@ runs:
|
|||||||
maturin-version: "1.12.4"
|
maturin-version: "1.12.4"
|
||||||
command: build
|
command: build
|
||||||
working-directory: python
|
working-directory: python
|
||||||
docker-options: "-e PIP_EXTRA_INDEX_URL='https://pypi.fury.io/lance-format/ https://pypi.fury.io/lancedb/' -e PROTOC=/usr/local/bin/protoc ${{ inputs.rustflags != '' && format('-e RUSTFLAGS={0}', inputs.rustflags) || '' }}"
|
docker-options: "-e PIP_EXTRA_INDEX_URL='https://pypi.fury.io/lance-format/ https://pypi.fury.io/lancedb/'"
|
||||||
target: aarch64-unknown-linux-gnu
|
target: aarch64-unknown-linux-gnu
|
||||||
manylinux: ${{ inputs.manylinux }}
|
manylinux: ${{ inputs.manylinux }}
|
||||||
args: ${{ inputs.args }}
|
args: ${{ inputs.args }}
|
||||||
before-script-linux: |
|
before-script-linux: |
|
||||||
set -e
|
set -e
|
||||||
yum install -y clang
|
yum install -y clang \
|
||||||
curl -fsSL https://github.com/protocolbuffers/protobuf/releases/download/v24.4/protoc-24.4-linux-aarch_64.zip -o /tmp/protoc.zip
|
&& curl -L https://github.com/protocolbuffers/protobuf/releases/download/v24.4/protoc-24.4-linux-aarch_64.zip > /tmp/protoc.zip \
|
||||||
unzip /tmp/protoc.zip -d /usr/local
|
&& unzip /tmp/protoc.zip -d /usr/local \
|
||||||
rm /tmp/protoc.zip
|
&& rm /tmp/protoc.zip
|
||||||
/usr/local/bin/protoc --version
|
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ on:
|
|||||||
# We don't publish pre-releases for Rust. Crates.io is just a source
|
# We don't publish pre-releases for Rust. Crates.io is just a source
|
||||||
# distribution, so we don't need to publish pre-releases.
|
# distribution, so we don't need to publish pre-releases.
|
||||||
- "v*-beta*"
|
- "v*-beta*"
|
||||||
|
- "*-v*" # for example, python-vX.Y.Z
|
||||||
|
|
||||||
env:
|
env:
|
||||||
# This env var is used by Swatinem/rust-cache@v2 for the cache
|
# This env var is used by Swatinem/rust-cache@v2 for the cache
|
||||||
@@ -24,7 +25,7 @@ jobs:
|
|||||||
# Only runs on tags that matches the make-release action
|
# Only runs on tags that matches the make-release action
|
||||||
if: startsWith(github.ref, 'refs/tags/v')
|
if: startsWith(github.ref, 'refs/tags/v')
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
with:
|
||||||
workspaces: rust
|
workspaces: rust
|
||||||
@@ -46,7 +47,7 @@ jobs:
|
|||||||
contents: read
|
contents: read
|
||||||
issues: write
|
issues: write
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
- uses: ./.github/actions/create-failure-issue
|
- uses: ./.github/actions/create-failure-issue
|
||||||
with:
|
with:
|
||||||
job-results: ${{ toJSON(needs) }}
|
job-results: ${{ toJSON(needs) }}
|
||||||
|
|||||||
@@ -36,14 +36,14 @@ jobs:
|
|||||||
echo "guidelines = ${{ inputs.guidelines }}"
|
echo "guidelines = ${{ inputs.guidelines }}"
|
||||||
|
|
||||||
- name: Checkout Repo
|
- name: Checkout Repo
|
||||||
uses: actions/checkout@v6
|
uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
ref: ${{ inputs.branch }}
|
ref: ${{ inputs.branch }}
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
persist-credentials: true
|
persist-credentials: true
|
||||||
|
|
||||||
- name: Set up Node.js
|
- name: Set up Node.js
|
||||||
uses: actions/setup-node@v6
|
uses: actions/setup-node@v4
|
||||||
with:
|
with:
|
||||||
# pnpm 11 (used by the nodejs install step below) requires
|
# pnpm 11 (used by the nodejs install step below) requires
|
||||||
# Node >= 22.13; use 24 since 22 hits EOL in October.
|
# Node >= 22.13; use 24 since 22 hits EOL in October.
|
||||||
@@ -82,7 +82,7 @@ jobs:
|
|||||||
cache: maven
|
cache: maven
|
||||||
|
|
||||||
- name: Setup pnpm
|
- name: Setup pnpm
|
||||||
uses: pnpm/action-setup@v6
|
uses: pnpm/action-setup@v4
|
||||||
with:
|
with:
|
||||||
version: 11.1.1
|
version: 11.1.1
|
||||||
- name: Install Node.js dependencies for TypeScript bindings
|
- name: Install Node.js dependencies for TypeScript bindings
|
||||||
|
|||||||
@@ -4,16 +4,14 @@ on:
|
|||||||
workflow_call:
|
workflow_call:
|
||||||
inputs:
|
inputs:
|
||||||
tag:
|
tag:
|
||||||
description: "Tag name from Lance. If omitted, the skill will use the latest Lance release that needs an update."
|
description: "Tag name from Lance"
|
||||||
required: false
|
required: true
|
||||||
default: ""
|
|
||||||
type: string
|
type: string
|
||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
inputs:
|
inputs:
|
||||||
tag:
|
tag:
|
||||||
description: "Tag name from Lance. Leave empty to use the latest Lance release that needs an update."
|
description: "Tag name from Lance"
|
||||||
required: false
|
required: true
|
||||||
default: ""
|
|
||||||
type: string
|
type: string
|
||||||
|
|
||||||
permissions:
|
permissions:
|
||||||
@@ -27,16 +25,16 @@ jobs:
|
|||||||
steps:
|
steps:
|
||||||
- name: Show inputs
|
- name: Show inputs
|
||||||
run: |
|
run: |
|
||||||
echo "tag = ${{ inputs.tag || 'latest' }}"
|
echo "tag = ${{ inputs.tag }}"
|
||||||
|
|
||||||
- name: Checkout Repo LanceDB
|
- name: Checkout Repo LanceDB
|
||||||
uses: actions/checkout@v6
|
uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
persist-credentials: true
|
persist-credentials: true
|
||||||
|
|
||||||
- name: Set up Node.js
|
- name: Set up Node.js
|
||||||
uses: actions/setup-node@v6
|
uses: actions/setup-node@v4
|
||||||
with:
|
with:
|
||||||
node-version: 20
|
node-version: 20
|
||||||
|
|
||||||
@@ -73,21 +71,65 @@ jobs:
|
|||||||
OPENAI_API_KEY: ${{ secrets.CODEX_TOKEN }}
|
OPENAI_API_KEY: ${{ secrets.CODEX_TOKEN }}
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
TARGET_TAG="${TAG:-latest}"
|
VERSION="${TAG#refs/tags/}"
|
||||||
|
VERSION="${VERSION#v}"
|
||||||
|
BRANCH_NAME="codex/update-lance-${VERSION//[^a-zA-Z0-9]/-}"
|
||||||
|
|
||||||
|
# Use "chore" for beta/rc versions, "feat" for stable releases
|
||||||
|
if [[ "${VERSION}" == *beta* ]] || [[ "${VERSION}" == *rc* ]]; then
|
||||||
|
COMMIT_TYPE="chore"
|
||||||
|
else
|
||||||
|
COMMIT_TYPE="feat"
|
||||||
|
fi
|
||||||
|
|
||||||
cat <<EOF >/tmp/codex-prompt.txt
|
cat <<EOF >/tmp/codex-prompt.txt
|
||||||
You are running inside the lancedb repository on a GitHub Actions runner.
|
You are running inside the lancedb repository on a GitHub Actions runner. Update the Lance dependency to version ${VERSION} and prepare a pull request for maintainers to review.
|
||||||
|
|
||||||
Use \$lancedb-update-lance-dependency with target "${TARGET_TAG}".
|
Follow these steps exactly:
|
||||||
|
1. Use script "ci/set_lance_version.py" to update Lance Rust dependencies. The script already refreshes Cargo metadata, so allow it to finish even if it takes time.
|
||||||
|
2. Update the Java lance-core dependency version in "java/pom.xml": change the "<lance-core.version>...</lance-core.version>" property to "${VERSION}".
|
||||||
|
3. Run "cargo clippy --workspace --tests --all-features -- -D warnings". If diagnostics appear, fix them yourself and rerun clippy until it exits cleanly. Do not skip any warnings.
|
||||||
|
4. After clippy succeeds, run "cargo fmt --all" to format the workspace.
|
||||||
|
5. Ensure the repository is clean except for intentional changes. Inspect "git status --short" and "git diff" to confirm the dependency update and any required fixes.
|
||||||
|
6. Create and switch to a new branch named "${BRANCH_NAME}" (replace any duplicated hyphens if necessary).
|
||||||
|
7. Stage all relevant files with "git add -A". Commit using the message "${COMMIT_TYPE}: update lance dependency to v${VERSION}".
|
||||||
|
8. Push the branch to origin. If the remote branch already exists, delete it first with "gh api -X DELETE repos/lancedb/lancedb/git/refs/heads/${BRANCH_NAME}" then push with "git push origin ${BRANCH_NAME}". Do NOT use "git push --force" or "git push -f".
|
||||||
|
9. env "GH_TOKEN" is available, use "gh" tools for github related operations like creating pull request.
|
||||||
|
10. Create a pull request targeting "main" with title "${COMMIT_TYPE}: update lance dependency to v${VERSION}". First, write the PR body to /tmp/pr-body.md using a heredoc (cat <<'EOF' > /tmp/pr-body.md). The body should summarize the dependency bump, clippy/fmt verification, and link the triggering tag (${TAG}). Then run "gh pr create --body-file /tmp/pr-body.md".
|
||||||
|
11. After creating the PR, display the PR URL, "git status --short", and a concise summary of the commands run and their results.
|
||||||
|
|
||||||
Constraints:
|
Constraints:
|
||||||
- Use env "GH_TOKEN" for GitHub operations.
|
- Use bash commands; avoid modifying GitHub workflow files other than through the scripted task above.
|
||||||
- Do not merge the pull request.
|
- Do not merge the PR.
|
||||||
- Do not force-push.
|
- If any command fails, diagnose and fix the issue instead of aborting.
|
||||||
- Do not create a duplicate pull request if an open PR already exists for the target Lance version.
|
|
||||||
- If any command fails, diagnose and fix the root cause instead of aborting.
|
|
||||||
- After creating the PR, display the PR URL, "git status --short", and a concise summary of the commands run and their results.
|
|
||||||
EOF
|
EOF
|
||||||
|
|
||||||
printenv OPENAI_API_KEY | codex login --with-api-key
|
printenv OPENAI_API_KEY | codex login --with-api-key
|
||||||
codex --config shell_environment_policy.ignore_default_excludes=true exec --dangerously-bypass-approvals-and-sandbox "$(cat /tmp/codex-prompt.txt)"
|
codex --config shell_environment_policy.ignore_default_excludes=true exec --dangerously-bypass-approvals-and-sandbox "$(cat /tmp/codex-prompt.txt)"
|
||||||
|
|
||||||
|
- name: Trigger sophon dependency update
|
||||||
|
env:
|
||||||
|
TAG: ${{ inputs.tag }}
|
||||||
|
GH_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
VERSION="${TAG#refs/tags/}"
|
||||||
|
VERSION="${VERSION#v}"
|
||||||
|
LANCEDB_BRANCH="codex/update-lance-${VERSION//[^a-zA-Z0-9]/-}"
|
||||||
|
|
||||||
|
echo "Triggering sophon workflow with:"
|
||||||
|
echo " lance_ref: ${TAG#refs/tags/}"
|
||||||
|
echo " lancedb_ref: ${LANCEDB_BRANCH}"
|
||||||
|
|
||||||
|
gh workflow run codex-bump-lancedb-lance.yml \
|
||||||
|
--repo lancedb/sophon \
|
||||||
|
-f lance_ref="${TAG#refs/tags/}" \
|
||||||
|
-f lancedb_ref="${LANCEDB_BRANCH}"
|
||||||
|
|
||||||
|
- name: Show latest sophon workflow run
|
||||||
|
env:
|
||||||
|
GH_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
echo "Latest sophon workflow run:"
|
||||||
|
gh run list --repo lancedb/sophon --workflow codex-bump-lancedb-lance.yml --limit 1 --json databaseId,url,displayTitle
|
||||||
|
|||||||
@@ -27,7 +27,7 @@ jobs:
|
|||||||
name: Verify PR title / description conforms to semantic-release
|
name: Verify PR title / description conforms to semantic-release
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/setup-node@v6
|
- uses: actions/setup-node@v4
|
||||||
with:
|
with:
|
||||||
node-version: "18"
|
node-version: "18"
|
||||||
# These rules are disabled because Github will always ensure there
|
# These rules are disabled because Github will always ensure there
|
||||||
|
|||||||
@@ -35,7 +35,7 @@ jobs:
|
|||||||
runs-on: ubuntu-24.04
|
runs-on: ubuntu-24.04
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout
|
- name: Checkout
|
||||||
uses: actions/checkout@v6
|
uses: actions/checkout@v4
|
||||||
- name: Install dependencies needed for ubuntu
|
- name: Install dependencies needed for ubuntu
|
||||||
run: |
|
run: |
|
||||||
sudo apt install -y protobuf-compiler libssl-dev
|
sudo apt install -y protobuf-compiler libssl-dev
|
||||||
@@ -53,7 +53,7 @@ jobs:
|
|||||||
python -m pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -e .
|
python -m pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -e .
|
||||||
python -m pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -r ../docs/requirements.txt
|
python -m pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -r ../docs/requirements.txt
|
||||||
- name: Set up node
|
- name: Set up node
|
||||||
uses: actions/setup-node@v6
|
uses: actions/setup-node@v4
|
||||||
with:
|
with:
|
||||||
node-version: 20
|
node-version: 20
|
||||||
cache: 'npm'
|
cache: 'npm'
|
||||||
|
|||||||
@@ -1,85 +0,0 @@
|
|||||||
name: GitHub Release
|
|
||||||
|
|
||||||
# All SDKs share one version, so a single `vX.Y.Z` tag produces a single GitHub
|
|
||||||
# release covering all of them. The per-package publish workflows (PyPI, NPM,
|
|
||||||
# Cargo, Maven) trigger off the same tag independently.
|
|
||||||
|
|
||||||
on:
|
|
||||||
push:
|
|
||||||
tags:
|
|
||||||
- "v*"
|
|
||||||
|
|
||||||
permissions:
|
|
||||||
contents: read
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
gh-release:
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
permissions:
|
|
||||||
contents: write
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v6
|
|
||||||
with:
|
|
||||||
fetch-depth: 0
|
|
||||||
lfs: true
|
|
||||||
- name: Extract version
|
|
||||||
id: extract_version
|
|
||||||
env:
|
|
||||||
GITHUB_REF: ${{ github.ref }}
|
|
||||||
run: |
|
|
||||||
set -e
|
|
||||||
echo "Extracting tag and version from $GITHUB_REF"
|
|
||||||
if [[ $GITHUB_REF =~ refs/tags/v(.*) ]]; then
|
|
||||||
VERSION=${BASH_REMATCH[1]}
|
|
||||||
TAG=v$VERSION
|
|
||||||
echo "tag=$TAG" >> $GITHUB_OUTPUT
|
|
||||||
echo "version=$VERSION" >> $GITHUB_OUTPUT
|
|
||||||
else
|
|
||||||
echo "Failed to extract version from $GITHUB_REF"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
echo "Extracted version $VERSION from $GITHUB_REF"
|
|
||||||
if [[ $VERSION =~ beta ]]; then
|
|
||||||
echo "This is a beta release"
|
|
||||||
echo "prerelease=true" >> $GITHUB_OUTPUT
|
|
||||||
|
|
||||||
# Get last release (that is not this one)
|
|
||||||
FROM_TAG=$(git tag --sort='version:refname' \
|
|
||||||
| grep ^v \
|
|
||||||
| grep -vF "$TAG" \
|
|
||||||
| python ci/semver_sort.py v \
|
|
||||||
| tail -n 1)
|
|
||||||
else
|
|
||||||
echo "This is a stable release"
|
|
||||||
echo "prerelease=false" >> $GITHUB_OUTPUT
|
|
||||||
# Get last stable tag (ignore betas)
|
|
||||||
FROM_TAG=$(git tag --sort='version:refname' \
|
|
||||||
| grep ^v \
|
|
||||||
| grep -vF "$TAG" \
|
|
||||||
| grep -v beta \
|
|
||||||
| python ci/semver_sort.py v \
|
|
||||||
| tail -n 1)
|
|
||||||
fi
|
|
||||||
echo "Found from tag $FROM_TAG"
|
|
||||||
echo "from_tag=$FROM_TAG" >> $GITHUB_OUTPUT
|
|
||||||
- name: Create Release Notes
|
|
||||||
id: release_notes
|
|
||||||
uses: mikepenz/release-changelog-builder-action@v4
|
|
||||||
with:
|
|
||||||
configuration: .github/release_notes.json
|
|
||||||
toTag: ${{ steps.extract_version.outputs.tag }}
|
|
||||||
fromTag: ${{ steps.extract_version.outputs.from_tag }}
|
|
||||||
env:
|
|
||||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
|
||||||
- name: Create GH release
|
|
||||||
uses: softprops/action-gh-release@v2
|
|
||||||
with:
|
|
||||||
# Marking betas as pre-releases keeps them from taking the "Latest"
|
|
||||||
# badge on the releases page.
|
|
||||||
prerelease: ${{ steps.extract_version.outputs.prerelease }}
|
|
||||||
make_latest: ${{ steps.extract_version.outputs.prerelease == 'false' }}
|
|
||||||
tag_name: ${{ steps.extract_version.outputs.tag }}
|
|
||||||
token: ${{ secrets.GITHUB_TOKEN }}
|
|
||||||
generate_release_notes: false
|
|
||||||
name: LanceDB v${{ steps.extract_version.outputs.version }}
|
|
||||||
body: ${{ steps.release_notes.outputs.changelog }}
|
|
||||||
@@ -32,7 +32,7 @@ jobs:
|
|||||||
working-directory: ./java
|
working-directory: ./java
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout repository
|
- name: Checkout repository
|
||||||
uses: actions/checkout@v6
|
uses: actions/checkout@v4
|
||||||
- name: Set up Java 8
|
- name: Set up Java 8
|
||||||
uses: actions/setup-java@v4
|
uses: actions/setup-java@v4
|
||||||
with:
|
with:
|
||||||
@@ -73,7 +73,7 @@ jobs:
|
|||||||
contents: read
|
contents: read
|
||||||
issues: write
|
issues: write
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
- uses: ./.github/actions/create-failure-issue
|
- uses: ./.github/actions/create-failure-issue
|
||||||
with:
|
with:
|
||||||
job-results: ${{ toJSON(needs) }}
|
job-results: ${{ toJSON(needs) }}
|
||||||
|
|||||||
@@ -36,7 +36,7 @@ jobs:
|
|||||||
working-directory: ./java
|
working-directory: ./java
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout repository
|
- name: Checkout repository
|
||||||
uses: actions/checkout@v6
|
uses: actions/checkout@v4
|
||||||
- name: Set up Java 17
|
- name: Set up Java 17
|
||||||
uses: actions/setup-java@v4
|
uses: actions/setup-java@v4
|
||||||
with:
|
with:
|
||||||
|
|||||||
@@ -0,0 +1,62 @@
|
|||||||
|
name: Lance Release Timer
|
||||||
|
|
||||||
|
on:
|
||||||
|
schedule:
|
||||||
|
- cron: "*/10 * * * *"
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
actions: write
|
||||||
|
|
||||||
|
concurrency:
|
||||||
|
group: lance-release-timer
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
trigger-update:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Checkout repository
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Check for new Lance tag
|
||||||
|
id: check
|
||||||
|
env:
|
||||||
|
GH_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
||||||
|
run: |
|
||||||
|
python3 ci/check_lance_release.py --github-output "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
|
- name: Look for existing PR
|
||||||
|
if: steps.check.outputs.needs_update == 'true'
|
||||||
|
id: pr
|
||||||
|
env:
|
||||||
|
GH_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
TITLE="chore: update lance dependency to v${{ steps.check.outputs.latest_version }}"
|
||||||
|
COUNT=$(gh pr list --search "\"$TITLE\" in:title" --state open --limit 1 --json number --jq 'length')
|
||||||
|
if [ "$COUNT" -gt 0 ]; then
|
||||||
|
echo "Open PR already exists for $TITLE"
|
||||||
|
echo "pr_exists=true" >> "$GITHUB_OUTPUT"
|
||||||
|
else
|
||||||
|
echo "No existing PR for $TITLE"
|
||||||
|
echo "pr_exists=false" >> "$GITHUB_OUTPUT"
|
||||||
|
fi
|
||||||
|
|
||||||
|
- name: Trigger codex update workflow
|
||||||
|
if: steps.check.outputs.needs_update == 'true' && steps.pr.outputs.pr_exists != 'true'
|
||||||
|
env:
|
||||||
|
GH_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
TAG=${{ steps.check.outputs.latest_tag }}
|
||||||
|
gh workflow run codex-update-lance-dependency.yml -f tag=refs/tags/$TAG
|
||||||
|
|
||||||
|
- name: Show latest codex workflow run
|
||||||
|
if: steps.check.outputs.needs_update == 'true' && steps.pr.outputs.pr_exists != 'true'
|
||||||
|
env:
|
||||||
|
GH_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
gh run list --workflow codex-update-lance-dependency.yml --limit 1 --json databaseId,url,displayTitle
|
||||||
@@ -19,7 +19,7 @@ jobs:
|
|||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- name: Check out code
|
- name: Check out code
|
||||||
uses: actions/checkout@v6
|
uses: actions/checkout@v4
|
||||||
- name: Install license-header-checker
|
- name: Install license-header-checker
|
||||||
working-directory: /tmp
|
working-directory: /tmp
|
||||||
run: |
|
run: |
|
||||||
|
|||||||
@@ -1,14 +1,13 @@
|
|||||||
name: Create release commit
|
name: Create release commit
|
||||||
|
|
||||||
# This workflow increments the version, tags it, and pushes it. All SDKs share
|
# This workflow increments versions, tags the version, and pushes it.
|
||||||
# a single version, so one tag releases all of them.
|
|
||||||
# When a tag is pushed, another workflow is triggered that creates a GH release
|
# When a tag is pushed, another workflow is triggered that creates a GH release
|
||||||
# and uploads the binaries. This workflow is only for creating the tag.
|
# and uploads the binaries. This workflow is only for creating the tag.
|
||||||
|
|
||||||
# This script will enforce that a minor version is incremented if there are any
|
# This script will enforce that a minor version is incremented if there are any
|
||||||
# breaking changes since the last minor increment. A breaking change in any SDK
|
# breaking changes since the last minor increment. However, it isn't able to
|
||||||
# bumps the minor version for all of them. If you wish to bypass this check, you
|
# differentiate between breaking changes in Node versus Python. If you wish to
|
||||||
# can manually increment the version and push the tag.
|
# bypass this check, you can manually increment the version and push the tag.
|
||||||
on:
|
on:
|
||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
inputs:
|
inputs:
|
||||||
@@ -25,6 +24,16 @@ on:
|
|||||||
options:
|
options:
|
||||||
- preview
|
- preview
|
||||||
- stable
|
- stable
|
||||||
|
python:
|
||||||
|
description: 'Make a Python release'
|
||||||
|
required: true
|
||||||
|
default: true
|
||||||
|
type: boolean
|
||||||
|
other:
|
||||||
|
description: 'Make a Node/Rust/Java release'
|
||||||
|
required: true
|
||||||
|
default: true
|
||||||
|
type: boolean
|
||||||
bump-minor:
|
bump-minor:
|
||||||
description: 'Bump minor version'
|
description: 'Bump minor version'
|
||||||
required: true
|
required: true
|
||||||
@@ -40,7 +49,7 @@ jobs:
|
|||||||
steps:
|
steps:
|
||||||
- name: Output Inputs
|
- name: Output Inputs
|
||||||
run: echo "${{ toJSON(github.event.inputs) }}"
|
run: echo "${{ toJSON(github.event.inputs) }}"
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
@@ -56,16 +65,29 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
git config user.name 'Lance Release'
|
git config user.name 'Lance Release'
|
||||||
git config user.email 'lance-dev@lancedb.com'
|
git config user.email 'lance-dev@lancedb.com'
|
||||||
- name: Bump version
|
- name: Bump Python version
|
||||||
|
if: ${{ inputs.python }}
|
||||||
|
working-directory: python
|
||||||
|
env:
|
||||||
|
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
run: |
|
||||||
|
# Need to get the commit before bumping the version, so we can
|
||||||
|
# determine if there are breaking changes in the next step as well.
|
||||||
|
echo "COMMIT_BEFORE_BUMP=$(git rev-parse HEAD)" >> $GITHUB_ENV
|
||||||
|
|
||||||
|
pip install bump-my-version PyGithub packaging
|
||||||
|
bash ../ci/bump_version.sh ${{ inputs.type }} ${{ inputs.bump-minor }} python-v
|
||||||
|
- name: Bump Node/Rust version
|
||||||
|
if: ${{ inputs.other }}
|
||||||
env:
|
env:
|
||||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
run: |
|
run: |
|
||||||
pip install bump-my-version PyGithub packaging
|
pip install bump-my-version PyGithub packaging
|
||||||
bash ci/bump_version.sh ${{ inputs.type }} ${{ inputs.bump-minor }}
|
bash ci/bump_version.sh ${{ inputs.type }} ${{ inputs.bump-minor }} v $COMMIT_BEFORE_BUMP
|
||||||
bash ci/update_lockfiles.sh --amend
|
bash ci/update_lockfiles.sh --amend
|
||||||
- name: Push new version tag
|
- name: Push new version tag
|
||||||
if: ${{ !inputs.dry_run }}
|
if: ${{ !inputs.dry_run }}
|
||||||
uses: ad-m/github-push-action@881a6320fdb16eb5318c5054f31c218aec2b324c # v1.3.0
|
uses: ad-m/github-push-action@master
|
||||||
with:
|
with:
|
||||||
# Need to use PAT here too to trigger next workflow. See comment above.
|
# Need to use PAT here too to trigger next workflow. See comment above.
|
||||||
github_token: ${{ secrets.LANCEDB_RELEASE_TOKEN }}
|
github_token: ${{ secrets.LANCEDB_RELEASE_TOKEN }}
|
||||||
|
|||||||
@@ -38,14 +38,14 @@ jobs:
|
|||||||
CC: gcc-12
|
CC: gcc-12
|
||||||
CXX: g++-12
|
CXX: g++-12
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
- uses: pnpm/action-setup@v6
|
- uses: pnpm/action-setup@v4
|
||||||
with:
|
with:
|
||||||
version: 11.1.1
|
version: 11.1.1
|
||||||
- uses: actions/setup-node@v6
|
- uses: actions/setup-node@v4
|
||||||
with:
|
with:
|
||||||
# pnpm 11 requires Node >= 22.13; use 24 since 22 hits EOL
|
# pnpm 11 requires Node >= 22.13; use 24 since 22 hits EOL
|
||||||
# in October. The library itself still supports Node >= 18
|
# in October. The library itself still supports Node >= 18
|
||||||
@@ -61,11 +61,6 @@ jobs:
|
|||||||
sudo apt update
|
sudo apt update
|
||||||
sudo apt install -y protobuf-compiler libssl-dev
|
sudo apt install -y protobuf-compiler libssl-dev
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
|
||||||
# Restore everywhere, but only save from main. Per-PR saves are
|
|
||||||
# unreadable outside their own branch anyway, since GitHub scopes
|
|
||||||
# caches to the creating ref.
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- name: Format Rust
|
- name: Format Rust
|
||||||
run: cargo fmt --all -- --check
|
run: cargo fmt --all -- --check
|
||||||
- name: Lint Rust
|
- name: Lint Rust
|
||||||
@@ -91,14 +86,14 @@ jobs:
|
|||||||
shell: bash
|
shell: bash
|
||||||
working-directory: nodejs
|
working-directory: nodejs
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
- uses: pnpm/action-setup@v6
|
- uses: pnpm/action-setup@v4
|
||||||
with:
|
with:
|
||||||
version: 11.1.1
|
version: 11.1.1
|
||||||
- uses: actions/setup-node@v6
|
- uses: actions/setup-node@v4
|
||||||
name: Setup Node.js 24 for build
|
name: Setup Node.js 24 for build
|
||||||
with:
|
with:
|
||||||
# pnpm 11 requires Node >= 22.13; use 24 since 22 hits EOL
|
# pnpm 11 requires Node >= 22.13; use 24 since 22 hits EOL
|
||||||
@@ -108,11 +103,6 @@ jobs:
|
|||||||
cache: 'pnpm'
|
cache: 'pnpm'
|
||||||
cache-dependency-path: nodejs/pnpm-lock.yaml
|
cache-dependency-path: nodejs/pnpm-lock.yaml
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
|
||||||
# Restore everywhere, but only save from main. Per-PR saves are
|
|
||||||
# unreadable outside their own branch anyway, since GitHub scopes
|
|
||||||
# caches to the creating ref.
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: |
|
run: |
|
||||||
sudo apt update
|
sudo apt update
|
||||||
@@ -140,7 +130,7 @@ jobs:
|
|||||||
echo "Run 'pnpm run docs', fix any warnings, and commit the changes."
|
echo "Run 'pnpm run docs', fix any warnings, and commit the changes."
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
- uses: actions/setup-node@v6
|
- uses: actions/setup-node@v4
|
||||||
name: Setup Node.js ${{ matrix.node-version }} for test
|
name: Setup Node.js ${{ matrix.node-version }} for test
|
||||||
with:
|
with:
|
||||||
node-version: ${{ matrix.node-version }}
|
node-version: ${{ matrix.node-version }}
|
||||||
@@ -176,14 +166,14 @@ jobs:
|
|||||||
shell: bash
|
shell: bash
|
||||||
working-directory: nodejs
|
working-directory: nodejs
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
- uses: pnpm/action-setup@v6
|
- uses: pnpm/action-setup@v4
|
||||||
with:
|
with:
|
||||||
version: 11.1.1
|
version: 11.1.1
|
||||||
- uses: actions/setup-node@v6
|
- uses: actions/setup-node@v4
|
||||||
with:
|
with:
|
||||||
# pnpm 11 requires Node >= 22.13; use 24 since 22 hits EOL
|
# pnpm 11 requires Node >= 22.13; use 24 since 22 hits EOL
|
||||||
# in October.
|
# in October.
|
||||||
@@ -192,11 +182,6 @@ jobs:
|
|||||||
cache-dependency-path: nodejs/pnpm-lock.yaml
|
cache-dependency-path: nodejs/pnpm-lock.yaml
|
||||||
- uses: dtolnay/rust-toolchain@stable
|
- uses: dtolnay/rust-toolchain@stable
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
|
||||||
# Restore everywhere, but only save from main. Per-PR saves are
|
|
||||||
# unreadable outside their own branch anyway, since GitHub scopes
|
|
||||||
# caches to the creating ref.
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: |
|
run: |
|
||||||
brew install protobuf
|
brew install protobuf
|
||||||
|
|||||||
+101
-116
@@ -10,16 +10,10 @@ permissions:
|
|||||||
|
|
||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
|
branches:
|
||||||
|
- main
|
||||||
tags:
|
tags:
|
||||||
- "v*"
|
- "v*"
|
||||||
# The cross-compiled targets (musl especially) break from toolchain and
|
|
||||||
# dependency changes that nothing else in CI catches, and discovering that
|
|
||||||
# mid-release is expensive. A nightly run keeps that signal while dropping
|
|
||||||
# the full 8-target release matrix from all ~90 pushes to main each month.
|
|
||||||
# `report-failure` files an issue when a nightly breaks.
|
|
||||||
schedule:
|
|
||||||
- cron: "0 8 * * *"
|
|
||||||
workflow_dispatch:
|
|
||||||
pull_request:
|
pull_request:
|
||||||
# This should trigger a dry run (we skip the final publish step)
|
# This should trigger a dry run (we skip the final publish step)
|
||||||
paths:
|
paths:
|
||||||
@@ -32,6 +26,73 @@ concurrency:
|
|||||||
cancel-in-progress: true
|
cancel-in-progress: true
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
|
gh-release:
|
||||||
|
if: startsWith(github.ref, 'refs/tags/v')
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
permissions:
|
||||||
|
contents: write
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
lfs: true
|
||||||
|
- name: Extract version
|
||||||
|
id: extract_version
|
||||||
|
env:
|
||||||
|
GITHUB_REF: ${{ github.ref }}
|
||||||
|
run: |
|
||||||
|
set -e
|
||||||
|
echo "Extracting tag and version from $GITHUB_REF"
|
||||||
|
if [[ $GITHUB_REF =~ refs/tags/v(.*) ]]; then
|
||||||
|
VERSION=${BASH_REMATCH[1]}
|
||||||
|
TAG=v$VERSION
|
||||||
|
echo "tag=$TAG" >> $GITHUB_OUTPUT
|
||||||
|
echo "version=$VERSION" >> $GITHUB_OUTPUT
|
||||||
|
else
|
||||||
|
echo "Failed to extract version from $GITHUB_REF"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
echo "Extracted version $VERSION from $GITHUB_REF"
|
||||||
|
if [[ $VERSION =~ beta ]]; then
|
||||||
|
echo "This is a beta release"
|
||||||
|
|
||||||
|
# Get last release (that is not this one)
|
||||||
|
FROM_TAG=$(git tag --sort='version:refname' \
|
||||||
|
| grep ^v \
|
||||||
|
| grep -vF "$TAG" \
|
||||||
|
| python ci/semver_sort.py v \
|
||||||
|
| tail -n 1)
|
||||||
|
else
|
||||||
|
echo "This is a stable release"
|
||||||
|
# Get last stable tag (ignore betas)
|
||||||
|
FROM_TAG=$(git tag --sort='version:refname' \
|
||||||
|
| grep ^v \
|
||||||
|
| grep -vF "$TAG" \
|
||||||
|
| grep -v beta \
|
||||||
|
| python ci/semver_sort.py v \
|
||||||
|
| tail -n 1)
|
||||||
|
fi
|
||||||
|
echo "Found from tag $FROM_TAG"
|
||||||
|
echo "from_tag=$FROM_TAG" >> $GITHUB_OUTPUT
|
||||||
|
- name: Create Release Notes
|
||||||
|
id: release_notes
|
||||||
|
uses: mikepenz/release-changelog-builder-action@v4
|
||||||
|
with:
|
||||||
|
configuration: .github/release_notes.json
|
||||||
|
toTag: ${{ steps.extract_version.outputs.tag }}
|
||||||
|
fromTag: ${{ steps.extract_version.outputs.from_tag }}
|
||||||
|
env:
|
||||||
|
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
- name: Create GH release
|
||||||
|
uses: softprops/action-gh-release@v2
|
||||||
|
with:
|
||||||
|
prerelease: ${{ contains('beta', github.ref) }}
|
||||||
|
tag_name: ${{ steps.extract_version.outputs.tag }}
|
||||||
|
token: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
generate_release_notes: false
|
||||||
|
name: Node/Rust LanceDB v${{ steps.extract_version.outputs.version }}
|
||||||
|
body: ${{ steps.release_notes.outputs.changelog }}
|
||||||
|
|
||||||
build-lancedb:
|
build-lancedb:
|
||||||
strategy:
|
strategy:
|
||||||
fail-fast: false
|
fail-fast: false
|
||||||
@@ -40,18 +101,9 @@ jobs:
|
|||||||
- target: aarch64-apple-darwin
|
- target: aarch64-apple-darwin
|
||||||
host: macos-latest
|
host: macos-latest
|
||||||
features: fp16kernels
|
features: fp16kernels
|
||||||
pre_build: |-
|
pre_build: brew install protobuf
|
||||||
brew install protobuf
|
|
||||||
# Fat LTO (the workspace default in .cargo/config.toml) is
|
|
||||||
# single-threaded and is the peak-memory step of the build. On
|
|
||||||
# this runner it accounted for ~111 of the job's ~113 minutes,
|
|
||||||
# making it the critical path of the entire publish pipeline.
|
|
||||||
# ThinLTO parallelizes it across the runner's cores, for a few
|
|
||||||
# percent of runtime performance.
|
|
||||||
export CARGO_PROFILE_RELEASE_LTO=thin
|
|
||||||
export CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16
|
|
||||||
- target: x86_64-pc-windows-msvc
|
- target: x86_64-pc-windows-msvc
|
||||||
host: windows-2025
|
host: windows-latest
|
||||||
features: ","
|
features: ","
|
||||||
pre_build: |-
|
pre_build: |-
|
||||||
choco install --no-progress protoc ninja nasm
|
choco install --no-progress protoc ninja nasm
|
||||||
@@ -59,21 +111,12 @@ jobs:
|
|||||||
# There is an issue where choco doesn't add nasm to the path
|
# There is an issue where choco doesn't add nasm to the path
|
||||||
export PATH="$PATH:/c/Program Files/NASM"
|
export PATH="$PATH:/c/Program Files/NASM"
|
||||||
nasm -v
|
nasm -v
|
||||||
# See the ThinLTO note on aarch64-apple-darwin above. Keeping
|
|
||||||
# peak memory down is also what lets this run on the standard
|
|
||||||
# 4-core runner: the 8-core larger runner was only needed to
|
|
||||||
# stop fat LTO from OOMing rustc-LLVM.
|
|
||||||
export CARGO_PROFILE_RELEASE_LTO=thin
|
|
||||||
export CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16
|
|
||||||
- target: aarch64-pc-windows-msvc
|
- target: aarch64-pc-windows-msvc
|
||||||
host: windows-2025
|
host: windows-latest
|
||||||
features: ","
|
features: ","
|
||||||
pre_build: |-
|
pre_build: |-
|
||||||
choco install --no-progress protoc
|
choco install --no-progress protoc
|
||||||
rustup target add aarch64-pc-windows-msvc
|
rustup target add aarch64-pc-windows-msvc
|
||||||
# See the ThinLTO note on aarch64-apple-darwin above.
|
|
||||||
export CARGO_PROFILE_RELEASE_LTO=thin
|
|
||||||
export CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16
|
|
||||||
- target: x86_64-unknown-linux-gnu
|
- target: x86_64-unknown-linux-gnu
|
||||||
host: ubuntu-latest
|
host: ubuntu-latest
|
||||||
features: fp16kernels
|
features: fp16kernels
|
||||||
@@ -127,13 +170,13 @@ jobs:
|
|||||||
run:
|
run:
|
||||||
working-directory: nodejs
|
working-directory: nodejs
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
- name: Setup pnpm
|
- name: Setup pnpm
|
||||||
uses: pnpm/action-setup@v6
|
uses: pnpm/action-setup@v4
|
||||||
with:
|
with:
|
||||||
version: 11.1.1
|
version: 11.1.1
|
||||||
- name: Setup node
|
- name: Setup node
|
||||||
uses: actions/setup-node@v6
|
uses: actions/setup-node@v4
|
||||||
with:
|
with:
|
||||||
# pnpm 11 requires Node >= 22.13; use 24 since 22 hits EOL
|
# pnpm 11 requires Node >= 22.13; use 24 since 22 hits EOL
|
||||||
# in October.
|
# in October.
|
||||||
@@ -146,49 +189,16 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
toolchain: stable
|
toolchain: stable
|
||||||
targets: ${{ matrix.settings.target }}
|
targets: ${{ matrix.settings.target }}
|
||||||
# These builds were entirely uncached: the old key was static, so
|
- name: Cache cargo
|
||||||
# `actions/cache` (which only writes on a miss) could never refresh it,
|
uses: actions/cache@v4
|
||||||
# and the multi-GB whole-`target/` copy it tried to store never fit the
|
|
||||||
# repo's cache budget, so no entry was ever saved. rust-cache prunes
|
|
||||||
# `target/` to dependency artifacts and keys on Cargo.lock plus the rustc
|
|
||||||
# version, which both fixes the key and keeps entries a sane size.
|
|
||||||
#
|
|
||||||
# This caches dependency *compilation* only. The LTO link of the cdylib
|
|
||||||
# re-runs regardless, since the local crate changes every time, so the
|
|
||||||
# win is larger on the non-LTO jobs than here.
|
|
||||||
- name: Cache cargo (native builds)
|
|
||||||
uses: Swatinem/rust-cache@v2
|
|
||||||
if: ${{ !matrix.settings.docker }}
|
|
||||||
with:
|
with:
|
||||||
# The release profile and per-target dirs differ from what the test
|
path: |
|
||||||
# workflows cache, so these need to be separate entries.
|
~/.cargo/registry/index/
|
||||||
key: release-${{ matrix.settings.target }}
|
~/.cargo/registry/cache/
|
||||||
# Only the nightly run on main writes, so tag and PR runs restore a
|
~/.cargo/git/db/
|
||||||
# warm entry without every dependabot PR writing its own (which would
|
.cargo-cache
|
||||||
# be unreadable elsewhere anyway, since GitHub scopes caches to the
|
target/
|
||||||
# creating ref). The nightly cadence also keeps entries inside
|
key: nodejs-${{ matrix.settings.target }}-cargo-${{ matrix.settings.host }}
|
||||||
# GitHub's 7-day eviction window, which a tag-only trigger would not.
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
# Docker builds can use rust-cache too. `target/` already lives on the
|
|
||||||
# host because the whole workspace is bind-mounted into the container, and
|
|
||||||
# rust-cache's prune and save run host-side, so they can manage it -- which
|
|
||||||
# is what keeps the entry to dependency artifacts rather than a multi-GB
|
|
||||||
# copy of everything.
|
|
||||||
#
|
|
||||||
# Two differences from the native builds. The container's CARGO_HOME is
|
|
||||||
# bind-mounted from `.cargo-cache` rather than the host's ~/.cargo, so that
|
|
||||||
# has to be cached explicitly. And the key is derived from the *host* rustc
|
|
||||||
# version, which is not the compiler that produced these artifacts; that is
|
|
||||||
# safe because cargo fingerprints the real compiler and rebuilds on a
|
|
||||||
# mismatch, it just means a base-image toolchain bump costs one cold build
|
|
||||||
# instead of invalidating the key.
|
|
||||||
- name: Cache cargo (docker builds)
|
|
||||||
uses: Swatinem/rust-cache@v2
|
|
||||||
if: ${{ matrix.settings.docker }}
|
|
||||||
with:
|
|
||||||
key: docker-${{ matrix.settings.target }}
|
|
||||||
cache-directories: .cargo-cache
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: pnpm install --frozen-lockfile
|
run: pnpm install --frozen-lockfile
|
||||||
- name: Install Zig
|
- name: Install Zig
|
||||||
@@ -206,13 +216,9 @@ jobs:
|
|||||||
if: ${{ matrix.settings.docker }}
|
if: ${{ matrix.settings.docker }}
|
||||||
with:
|
with:
|
||||||
image: ${{ matrix.settings.docker }}
|
image: ${{ matrix.settings.docker }}
|
||||||
# All three mounts must live under `.cargo-cache`, which is what the
|
|
||||||
# cache step above saves. Previously the registry mounts pointed at
|
|
||||||
# `.cargo/...`, a path nothing cached, so the container re-downloaded
|
|
||||||
# the whole crate registry on every run.
|
|
||||||
options: "--user 0:0 -v ${{ github.workspace }}/.cargo-cache/git/db:/usr/local/cargo/git/db \
|
options: "--user 0:0 -v ${{ github.workspace }}/.cargo-cache/git/db:/usr/local/cargo/git/db \
|
||||||
-v ${{ github.workspace }}/.cargo-cache/registry/cache:/usr/local/cargo/registry/cache \
|
-v ${{ github.workspace }}/.cargo/registry/cache:/usr/local/cargo/registry/cache \
|
||||||
-v ${{ github.workspace }}/.cargo-cache/registry/index:/usr/local/cargo/registry/index \
|
-v ${{ github.workspace }}/.cargo/registry/index:/usr/local/cargo/registry/index \
|
||||||
-v ${{ github.workspace }}:/build -w /build/nodejs"
|
-v ${{ github.workspace }}:/build -w /build/nodejs"
|
||||||
run: |
|
run: |
|
||||||
set -e
|
set -e
|
||||||
@@ -224,16 +230,6 @@ jobs:
|
|||||||
--js ../lancedb/native.js \
|
--js ../lancedb/native.js \
|
||||||
--strip \
|
--strip \
|
||||||
--output-dir dist/
|
--output-dir dist/
|
||||||
# The container runs as root (`--user 0:0`), so everything it wrote to the
|
|
||||||
# mounted cache dirs is root-owned. rust-cache's post step runs as the
|
|
||||||
# runner user and has to both read these and delete from them while
|
|
||||||
# pruning, so hand them back before it runs.
|
|
||||||
- name: Take ownership of docker build output
|
|
||||||
if: ${{ matrix.settings.docker }}
|
|
||||||
run: |
|
|
||||||
sudo chown -R "$(id -u):$(id -g)" \
|
|
||||||
"${{ github.workspace }}/.cargo-cache" \
|
|
||||||
"${{ github.workspace }}/target"
|
|
||||||
- name: Build
|
- name: Build
|
||||||
run: |
|
run: |
|
||||||
${{ matrix.settings.pre_build }}
|
${{ matrix.settings.pre_build }}
|
||||||
@@ -247,17 +243,8 @@ jobs:
|
|||||||
--output-dir dist/
|
--output-dir dist/
|
||||||
if: ${{ !matrix.settings.docker }}
|
if: ${{ !matrix.settings.docker }}
|
||||||
shell: bash
|
shell: bash
|
||||||
# The standard Windows runners have ~14 GB free, and a release `target/`
|
|
||||||
# for this workspace is a large fraction of that. Report the remaining
|
|
||||||
# headroom so a build that only just fits is visible before a dependency
|
|
||||||
# bump turns it into a failed release. `always()` so the numbers are
|
|
||||||
# still there when the build is what ran out of space.
|
|
||||||
- name: Report disk headroom
|
|
||||||
if: always()
|
|
||||||
run: df -h
|
|
||||||
shell: bash
|
|
||||||
- name: Upload artifact
|
- name: Upload artifact
|
||||||
uses: actions/upload-artifact@v7
|
uses: actions/upload-artifact@v4
|
||||||
with:
|
with:
|
||||||
name: lancedb-${{ matrix.settings.target }}
|
name: lancedb-${{ matrix.settings.target }}
|
||||||
path: nodejs/dist/*.node
|
path: nodejs/dist/*.node
|
||||||
@@ -269,7 +256,7 @@ jobs:
|
|||||||
run: pnpm tsc
|
run: pnpm tsc
|
||||||
- name: Upload Generic Artifacts
|
- name: Upload Generic Artifacts
|
||||||
if: ${{ matrix.settings.target == 'aarch64-apple-darwin' }}
|
if: ${{ matrix.settings.target == 'aarch64-apple-darwin' }}
|
||||||
uses: actions/upload-artifact@v7
|
uses: actions/upload-artifact@v4
|
||||||
with:
|
with:
|
||||||
name: nodejs-dist
|
name: nodejs-dist
|
||||||
path: |
|
path: |
|
||||||
@@ -300,13 +287,13 @@ jobs:
|
|||||||
shell: bash
|
shell: bash
|
||||||
working-directory: nodejs
|
working-directory: nodejs
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
- name: Setup pnpm
|
- name: Setup pnpm
|
||||||
uses: pnpm/action-setup@v6
|
uses: pnpm/action-setup@v4
|
||||||
with:
|
with:
|
||||||
version: 11.1.1
|
version: 11.1.1
|
||||||
- name: Setup Node.js 24 for install
|
- name: Setup Node.js 24 for install
|
||||||
uses: actions/setup-node@v6
|
uses: actions/setup-node@v4
|
||||||
with:
|
with:
|
||||||
# pnpm 11 requires Node >= 22.13; use 24 since 22 hits EOL
|
# pnpm 11 requires Node >= 22.13; use 24 since 22 hits EOL
|
||||||
# in October.
|
# in October.
|
||||||
@@ -316,18 +303,18 @@ jobs:
|
|||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: pnpm install --frozen-lockfile
|
run: pnpm install --frozen-lockfile
|
||||||
- name: Setup Node.js ${{ matrix.node }} for test
|
- name: Setup Node.js ${{ matrix.node }} for test
|
||||||
uses: actions/setup-node@v6
|
uses: actions/setup-node@v4
|
||||||
with:
|
with:
|
||||||
node-version: ${{ matrix.node }}
|
node-version: ${{ matrix.node }}
|
||||||
- name: Download artifacts
|
- name: Download artifacts
|
||||||
uses: actions/download-artifact@v8
|
uses: actions/download-artifact@v4
|
||||||
with:
|
with:
|
||||||
name: lancedb-${{ matrix.settings.target }}
|
name: lancedb-${{ matrix.settings.target }}
|
||||||
path: nodejs/dist/
|
path: nodejs/dist/
|
||||||
# For testing purposes:
|
# For testing purposes:
|
||||||
# run-id: 13982782871
|
# run-id: 13982782871
|
||||||
# github-token: ${{ secrets.GITHUB_TOKEN }} # token with actions:read permissions on target repo
|
# github-token: ${{ secrets.GITHUB_TOKEN }} # token with actions:read permissions on target repo
|
||||||
- uses: actions/download-artifact@v8
|
- uses: actions/download-artifact@v4
|
||||||
with:
|
with:
|
||||||
name: nodejs-dist
|
name: nodejs-dist
|
||||||
path: nodejs/dist
|
path: nodejs/dist
|
||||||
@@ -352,13 +339,13 @@ jobs:
|
|||||||
needs:
|
needs:
|
||||||
- test-lancedb
|
- test-lancedb
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
- name: Setup pnpm
|
- name: Setup pnpm
|
||||||
uses: pnpm/action-setup@v6
|
uses: pnpm/action-setup@v4
|
||||||
with:
|
with:
|
||||||
version: 11.1.1
|
version: 11.1.1
|
||||||
- name: Setup node
|
- name: Setup node
|
||||||
uses: actions/setup-node@v6
|
uses: actions/setup-node@v4
|
||||||
with:
|
with:
|
||||||
node-version: 24
|
node-version: 24
|
||||||
cache: pnpm
|
cache: pnpm
|
||||||
@@ -366,14 +353,14 @@ jobs:
|
|||||||
registry-url: "https://registry.npmjs.org"
|
registry-url: "https://registry.npmjs.org"
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: pnpm install --frozen-lockfile
|
run: pnpm install --frozen-lockfile
|
||||||
- uses: actions/download-artifact@v8
|
- uses: actions/download-artifact@v4
|
||||||
with:
|
with:
|
||||||
name: nodejs-dist
|
name: nodejs-dist
|
||||||
path: nodejs/dist
|
path: nodejs/dist
|
||||||
# For testing purposes:
|
# For testing purposes:
|
||||||
# run-id: 13982782871
|
# run-id: 13982782871
|
||||||
# github-token: ${{ secrets.GITHUB_TOKEN }} # token with actions:read permissions on target repo
|
# github-token: ${{ secrets.GITHUB_TOKEN }} # token with actions:read permissions on target repo
|
||||||
- uses: actions/download-artifact@v8
|
- uses: actions/download-artifact@v4
|
||||||
name: Download arch-specific binaries
|
name: Download arch-specific binaries
|
||||||
with:
|
with:
|
||||||
pattern: lancedb-*
|
pattern: lancedb-*
|
||||||
@@ -406,14 +393,12 @@ jobs:
|
|||||||
name: Report Workflow Failure
|
name: Report Workflow Failure
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: [build-lancedb, test-lancedb, publish]
|
needs: [build-lancedb, test-lancedb, publish]
|
||||||
# Nightly runs are the only thing watching the cross-compiled targets now,
|
if: always() && failure() && startsWith(github.ref, 'refs/tags/v')
|
||||||
# so they have to report failures too or the signal is silently lost.
|
|
||||||
if: always() && failure() && (startsWith(github.ref, 'refs/tags/v') || github.event_name == 'schedule')
|
|
||||||
permissions:
|
permissions:
|
||||||
contents: read
|
contents: read
|
||||||
issues: write
|
issues: write
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
- uses: ./.github/actions/create-failure-issue
|
- uses: ./.github/actions/create-failure-issue
|
||||||
with:
|
with:
|
||||||
job-results: ${{ toJSON(needs) }}
|
job-results: ${{ toJSON(needs) }}
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ name: PyPI Publish
|
|||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
tags:
|
tags:
|
||||||
- 'v*'
|
- 'python-v*'
|
||||||
pull_request:
|
pull_request:
|
||||||
# This should trigger a dry run (we skip the final publish step)
|
# This should trigger a dry run (we skip the final publish step)
|
||||||
paths:
|
paths:
|
||||||
@@ -20,45 +20,30 @@ env:
|
|||||||
permissions:
|
permissions:
|
||||||
contents: read
|
contents: read
|
||||||
|
|
||||||
# Without this, a force-push to a PR leaves the previous run going -- including
|
|
||||||
# a ~74 minute Windows job and a billed arm64 wheel build.
|
|
||||||
concurrency:
|
|
||||||
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
|
|
||||||
cancel-in-progress: true
|
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
linux:
|
linux:
|
||||||
name: Python ${{ matrix.config.package_name }} ${{ matrix.config.platform }} manylinux${{ matrix.config.manylinux }}
|
name: Python ${{ matrix.config.platform }} manylinux${{ matrix.config.manylinux }}
|
||||||
timeout-minutes: 60
|
timeout-minutes: 60
|
||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
config:
|
config:
|
||||||
|
- platform: x86_64
|
||||||
|
manylinux: "2_17"
|
||||||
|
extra_args: ""
|
||||||
|
runner: ubuntu-22.04
|
||||||
- platform: x86_64
|
- platform: x86_64
|
||||||
manylinux: "2_28"
|
manylinux: "2_28"
|
||||||
extra_args: "--features fp16kernels"
|
extra_args: "--features fp16kernels"
|
||||||
runner: ubuntu-22.04
|
runner: ubuntu-22.04
|
||||||
package_name: "lancedb"
|
- platform: aarch64
|
||||||
rustflags: ""
|
manylinux: "2_17"
|
||||||
# For successful fat LTO builds, we need a large runner to avoid OOM errors.
|
extra_args: ""
|
||||||
|
# For successful fat LTO builds, we need a large runner to avoid OOM errors.
|
||||||
|
runner: ubuntu-2404-8x-arm64
|
||||||
- platform: aarch64
|
- platform: aarch64
|
||||||
manylinux: "2_28"
|
manylinux: "2_28"
|
||||||
extra_args: "--features fp16kernels"
|
extra_args: "--features fp16kernels"
|
||||||
runner: ubuntu-2404-8x-arm64
|
runner: ubuntu-2404-8x-arm64
|
||||||
package_name: "lancedb"
|
|
||||||
rustflags: ""
|
|
||||||
# `lancedb-compat`: pre-Haswell-friendly variant for x86_64 hosts
|
|
||||||
# without AVX2 (Sandy Bridge / Ivy Bridge / Westmere on Intel,
|
|
||||||
# Bulldozer / Piledriver / Steamroller on AMD). Compiled at the
|
|
||||||
# `x86-64-v2` baseline; runtime SIMD dispatch in lance-linalg
|
|
||||||
# picks the appropriate tier (scalar / AVX / AVX+FMA / AVX2+FMA
|
|
||||||
# / AVX-512) at load time. Same import as `lancedb` -- conflicts
|
|
||||||
# at install time, so users pick one.
|
|
||||||
- platform: x86_64
|
|
||||||
manylinux: "2_28"
|
|
||||||
extra_args: ""
|
|
||||||
runner: ubuntu-22.04
|
|
||||||
package_name: "lancedb-compat"
|
|
||||||
rustflags: "-Ctarget-cpu=x86-64-v2"
|
|
||||||
runs-on: ${{ matrix.config.runner }}
|
runs-on: ${{ matrix.config.runner }}
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v6
|
||||||
@@ -75,13 +60,11 @@ jobs:
|
|||||||
args: "--release --strip ${{ matrix.config.extra_args }}"
|
args: "--release --strip ${{ matrix.config.extra_args }}"
|
||||||
arm-build: ${{ matrix.config.platform == 'aarch64' }}
|
arm-build: ${{ matrix.config.platform == 'aarch64' }}
|
||||||
manylinux: ${{ matrix.config.manylinux }}
|
manylinux: ${{ matrix.config.manylinux }}
|
||||||
package-name: ${{ matrix.config.package_name }}
|
|
||||||
rustflags: ${{ matrix.config.rustflags }}
|
|
||||||
- uses: actions/upload-artifact@v7
|
- uses: actions/upload-artifact@v7
|
||||||
if: startsWith(github.ref, 'refs/tags/v')
|
if: startsWith(github.ref, 'refs/tags/python-v')
|
||||||
with:
|
with:
|
||||||
name: wheels-linux-${{ matrix.config.package_name }}-${{ matrix.config.platform }}-${{ matrix.config.manylinux }}
|
name: wheels-linux-${{ matrix.config.platform }}-${{ matrix.config.manylinux }}
|
||||||
path: target/wheels/*.whl
|
path: target/wheels/lancedb-*.whl
|
||||||
if-no-files-found: error
|
if-no-files-found: error
|
||||||
mac:
|
mac:
|
||||||
timeout-minutes: 90
|
timeout-minutes: 90
|
||||||
@@ -107,7 +90,7 @@ jobs:
|
|||||||
python-minor-version: 10
|
python-minor-version: 10
|
||||||
args: "--release --strip --target ${{ matrix.config.target }} --features fp16kernels"
|
args: "--release --strip --target ${{ matrix.config.target }} --features fp16kernels"
|
||||||
- uses: actions/upload-artifact@v7
|
- uses: actions/upload-artifact@v7
|
||||||
if: startsWith(github.ref, 'refs/tags/v')
|
if: startsWith(github.ref, 'refs/tags/python-v')
|
||||||
with:
|
with:
|
||||||
name: wheels-mac-${{ matrix.config.target }}
|
name: wheels-mac-${{ matrix.config.target }}
|
||||||
path: target/wheels/lancedb-*.whl
|
path: target/wheels/lancedb-*.whl
|
||||||
@@ -128,26 +111,19 @@ jobs:
|
|||||||
uses: actions/setup-python@v6
|
uses: actions/setup-python@v6
|
||||||
with:
|
with:
|
||||||
python-version: "3.13"
|
python-version: "3.13"
|
||||||
# NOTE: caching cargo here would be a no-op. This workflow only runs on
|
|
||||||
# tags and PRs, and GitHub only lets a run restore caches from its own ref
|
|
||||||
# or the default branch -- so with no run on main there is nothing that
|
|
||||||
# can populate an entry the release build would be allowed to read. Fixing
|
|
||||||
# this needs a main/nightly trigger (which would also catch wheel-build
|
|
||||||
# breakage before a release); the ~74 minutes here is otherwise dominated
|
|
||||||
# by the fat-LTO link, which no cache avoids.
|
|
||||||
- uses: ./.github/workflows/build_windows_wheel
|
- uses: ./.github/workflows/build_windows_wheel
|
||||||
with:
|
with:
|
||||||
python-minor-version: 10
|
python-minor-version: 10
|
||||||
args: "--release --strip"
|
args: "--release --strip"
|
||||||
- uses: actions/upload-artifact@v7
|
- uses: actions/upload-artifact@v7
|
||||||
if: startsWith(github.ref, 'refs/tags/v')
|
if: startsWith(github.ref, 'refs/tags/python-v')
|
||||||
with:
|
with:
|
||||||
name: wheels-windows
|
name: wheels-windows
|
||||||
path: target/wheels/lancedb-*.whl
|
path: target/wheels/lancedb-*.whl
|
||||||
if-no-files-found: error
|
if-no-files-found: error
|
||||||
publish:
|
publish:
|
||||||
name: Publish wheels
|
name: Publish wheels
|
||||||
if: startsWith(github.ref, 'refs/tags/v')
|
if: startsWith(github.ref, 'refs/tags/python-v')
|
||||||
needs: [linux, mac, windows]
|
needs: [linux, mac, windows]
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
permissions:
|
permissions:
|
||||||
@@ -177,7 +153,7 @@ jobs:
|
|||||||
FURY_TOKEN: ${{ secrets.FURY_TOKEN }}
|
FURY_TOKEN: ${{ secrets.FURY_TOKEN }}
|
||||||
run: |
|
run: |
|
||||||
shopt -s nullglob
|
shopt -s nullglob
|
||||||
WHEELS=(target/wheels/*.whl)
|
WHEELS=(target/wheels/lancedb-*.whl)
|
||||||
if [[ ${#WHEELS[@]} -eq 0 ]]; then
|
if [[ ${#WHEELS[@]} -eq 0 ]]; then
|
||||||
echo "No wheels found in target/wheels/" >&2
|
echo "No wheels found in target/wheels/" >&2
|
||||||
exit 1
|
exit 1
|
||||||
@@ -196,6 +172,72 @@ jobs:
|
|||||||
uses: pypa/gh-action-pypi-publish@release/v1
|
uses: pypa/gh-action-pypi-publish@release/v1
|
||||||
with:
|
with:
|
||||||
packages-dir: target/wheels/
|
packages-dir: target/wheels/
|
||||||
|
gh-release:
|
||||||
|
if: startsWith(github.ref, 'refs/tags/python-v')
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
permissions:
|
||||||
|
contents: write
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v6
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
lfs: true
|
||||||
|
- name: Extract version
|
||||||
|
id: extract_version
|
||||||
|
env:
|
||||||
|
GITHUB_REF: ${{ github.ref }}
|
||||||
|
run: |
|
||||||
|
set -e
|
||||||
|
echo "Extracting tag and version from $GITHUB_REF"
|
||||||
|
if [[ $GITHUB_REF =~ refs/tags/python-v(.*) ]]; then
|
||||||
|
VERSION=${BASH_REMATCH[1]}
|
||||||
|
TAG=python-v$VERSION
|
||||||
|
echo "tag=$TAG" >> $GITHUB_OUTPUT
|
||||||
|
echo "version=$VERSION" >> $GITHUB_OUTPUT
|
||||||
|
else
|
||||||
|
echo "Failed to extract version from $GITHUB_REF"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
echo "Extracted version $VERSION from $GITHUB_REF"
|
||||||
|
if [[ $VERSION =~ beta ]]; then
|
||||||
|
echo "This is a beta release"
|
||||||
|
|
||||||
|
# Get last release (that is not this one)
|
||||||
|
FROM_TAG=$(git tag --sort='version:refname' \
|
||||||
|
| grep ^python-v \
|
||||||
|
| grep -vF "$TAG" \
|
||||||
|
| python ci/semver_sort.py python-v \
|
||||||
|
| tail -n 1)
|
||||||
|
else
|
||||||
|
echo "This is a stable release"
|
||||||
|
# Get last stable tag (ignore betas)
|
||||||
|
FROM_TAG=$(git tag --sort='version:refname' \
|
||||||
|
| grep ^python-v \
|
||||||
|
| grep -vF "$TAG" \
|
||||||
|
| grep -v beta \
|
||||||
|
| python ci/semver_sort.py python-v \
|
||||||
|
| tail -n 1)
|
||||||
|
fi
|
||||||
|
echo "Found from tag $FROM_TAG"
|
||||||
|
echo "from_tag=$FROM_TAG" >> $GITHUB_OUTPUT
|
||||||
|
- name: Create Python Release Notes
|
||||||
|
id: python_release_notes
|
||||||
|
uses: mikepenz/release-changelog-builder-action@v4
|
||||||
|
with:
|
||||||
|
configuration: .github/release_notes.json
|
||||||
|
toTag: ${{ steps.extract_version.outputs.tag }}
|
||||||
|
fromTag: ${{ steps.extract_version.outputs.from_tag }}
|
||||||
|
env:
|
||||||
|
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
- name: Create Python GH release
|
||||||
|
uses: softprops/action-gh-release@v2
|
||||||
|
with:
|
||||||
|
prerelease: ${{ contains('beta', github.ref) }}
|
||||||
|
tag_name: ${{ steps.extract_version.outputs.tag }}
|
||||||
|
token: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
generate_release_notes: false
|
||||||
|
name: Python LanceDB v${{ steps.extract_version.outputs.version }}
|
||||||
|
body: ${{ steps.python_release_notes.outputs.changelog }}
|
||||||
report-failure:
|
report-failure:
|
||||||
name: Report Workflow Failure
|
name: Report Workflow Failure
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
@@ -203,7 +245,7 @@ jobs:
|
|||||||
permissions:
|
permissions:
|
||||||
contents: read
|
contents: read
|
||||||
issues: write
|
issues: write
|
||||||
if: always() && failure() && startsWith(github.ref, 'refs/tags/v')
|
if: always() && failure() && startsWith(github.ref, 'refs/tags/python-v')
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v6
|
||||||
- uses: ./.github/actions/create-failure-issue
|
- uses: ./.github/actions/create-failure-issue
|
||||||
|
|||||||
@@ -41,7 +41,7 @@ jobs:
|
|||||||
shell: bash
|
shell: bash
|
||||||
working-directory: python
|
working-directory: python
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
@@ -66,7 +66,7 @@ jobs:
|
|||||||
shell: bash
|
shell: bash
|
||||||
working-directory: python
|
working-directory: python
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
@@ -95,7 +95,7 @@ jobs:
|
|||||||
shell: bash
|
shell: bash
|
||||||
working-directory: python
|
working-directory: python
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
@@ -108,15 +108,6 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
sudo apt update
|
sudo apt update
|
||||||
sudo apt install -y protobuf-compiler
|
sudo apt install -y protobuf-compiler
|
||||||
# `pip install -e .` builds the extension with maturin, which is most of
|
|
||||||
# this job's ~33 minutes. It had no Rust cache, so every dependency was
|
|
||||||
# recompiled from scratch on every run.
|
|
||||||
- uses: Swatinem/rust-cache@v2
|
|
||||||
with:
|
|
||||||
# Restore everywhere, but only save from main. Per-PR saves are
|
|
||||||
# unreadable outside their own branch anyway, since GitHub scopes
|
|
||||||
# caches to the creating ref.
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- name: Install
|
- name: Install
|
||||||
run: |
|
run: |
|
||||||
pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests,dev,embeddings]
|
pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests,dev,embeddings]
|
||||||
@@ -135,7 +126,7 @@ jobs:
|
|||||||
shell: bash
|
shell: bash
|
||||||
working-directory: python
|
working-directory: python
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
@@ -169,7 +160,7 @@ jobs:
|
|||||||
shell: bash
|
shell: bash
|
||||||
working-directory: python
|
working-directory: python
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
@@ -177,14 +168,6 @@ jobs:
|
|||||||
uses: actions/setup-python@v6
|
uses: actions/setup-python@v6
|
||||||
with:
|
with:
|
||||||
python-version: "3.13"
|
python-version: "3.13"
|
||||||
# maturin runs cargo natively on macOS (docker is Linux-only), so the host
|
|
||||||
# target dir is cacheable. This job had no Rust cache.
|
|
||||||
- uses: Swatinem/rust-cache@v2
|
|
||||||
with:
|
|
||||||
# Restore everywhere, but only save from main. Per-PR saves are
|
|
||||||
# unreadable outside their own branch anyway, since GitHub scopes
|
|
||||||
# caches to the creating ref.
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- uses: ./.github/workflows/build_mac_wheel
|
- uses: ./.github/workflows/build_mac_wheel
|
||||||
with:
|
with:
|
||||||
args: --profile ci
|
args: --profile ci
|
||||||
@@ -206,7 +189,7 @@ jobs:
|
|||||||
shell: bash
|
shell: bash
|
||||||
working-directory: python
|
working-directory: python
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
@@ -214,14 +197,6 @@ jobs:
|
|||||||
uses: actions/setup-python@v6
|
uses: actions/setup-python@v6
|
||||||
with:
|
with:
|
||||||
python-version: "3.13"
|
python-version: "3.13"
|
||||||
# maturin runs cargo natively on Windows (docker is Linux-only), so the
|
|
||||||
# host target dir is cacheable. This job had no Rust cache at all and so
|
|
||||||
# rebuilt every dependency from scratch on every run.
|
|
||||||
- uses: Swatinem/rust-cache@v2
|
|
||||||
with:
|
|
||||||
# Restore everywhere, but only save from main. The repo sits at
|
|
||||||
# GitHub's cache cap, so per-PR saves just evict main's entries.
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- uses: ./.github/workflows/build_windows_wheel
|
- uses: ./.github/workflows/build_windows_wheel
|
||||||
with:
|
with:
|
||||||
args: --profile ci
|
args: --profile ci
|
||||||
@@ -237,7 +212,7 @@ jobs:
|
|||||||
shell: bash
|
shell: bash
|
||||||
working-directory: python
|
working-directory: python
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
@@ -249,14 +224,6 @@ jobs:
|
|||||||
uses: actions/setup-python@v6
|
uses: actions/setup-python@v6
|
||||||
with:
|
with:
|
||||||
python-version: "3.10"
|
python-version: "3.10"
|
||||||
# As with Doctest, `pip install -e .` compiles the extension and this job
|
|
||||||
# had no Rust cache, which is most of its ~37 minutes.
|
|
||||||
- uses: Swatinem/rust-cache@v2
|
|
||||||
with:
|
|
||||||
# Restore everywhere, but only save from main. Per-PR saves are
|
|
||||||
# unreadable outside their own branch anyway, since GitHub scopes
|
|
||||||
# caches to the creating ref.
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- name: Install lancedb
|
- name: Install lancedb
|
||||||
run: |
|
run: |
|
||||||
pip install "pydantic<2"
|
pip install "pydantic<2"
|
||||||
|
|||||||
+20
-74
@@ -40,7 +40,7 @@ jobs:
|
|||||||
CC: clang-18
|
CC: clang-18
|
||||||
CXX: clang++-18
|
CXX: clang++-18
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
@@ -48,11 +48,6 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
components: rustfmt, clippy
|
components: rustfmt, clippy
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
|
||||||
# Restore everywhere, but only save from main. Per-PR saves are
|
|
||||||
# unreadable outside their own branch anyway, since GitHub scopes
|
|
||||||
# caches to the creating ref.
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: |
|
run: |
|
||||||
sudo apt update
|
sudo apt update
|
||||||
@@ -70,7 +65,7 @@ jobs:
|
|||||||
timeout-minutes: 10
|
timeout-minutes: 10
|
||||||
runs-on: ubuntu-24.04
|
runs-on: ubuntu-24.04
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
- uses: EmbarkStudios/cargo-deny-action@v2
|
- uses: EmbarkStudios/cargo-deny-action@v2
|
||||||
with:
|
with:
|
||||||
command: check advisories bans licenses sources
|
command: check advisories bans licenses sources
|
||||||
@@ -83,7 +78,7 @@ jobs:
|
|||||||
CC: clang
|
CC: clang
|
||||||
CXX: clang++
|
CXX: clang++
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
# Building without a lock file often requires the latest Rust version since downstream
|
# Building without a lock file often requires the latest Rust version since downstream
|
||||||
# dependencies may have updated their minimum Rust version.
|
# dependencies may have updated their minimum Rust version.
|
||||||
- uses: actions-rust-lang/setup-rust-toolchain@v1
|
- uses: actions-rust-lang/setup-rust-toolchain@v1
|
||||||
@@ -94,11 +89,6 @@ jobs:
|
|||||||
run: rm -f Cargo.lock
|
run: rm -f Cargo.lock
|
||||||
- uses: rui314/setup-mold@v1
|
- uses: rui314/setup-mold@v1
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
|
||||||
# Restore everywhere, but only save from main. Per-PR saves are
|
|
||||||
# unreadable outside their own branch anyway, since GitHub scopes
|
|
||||||
# caches to the creating ref.
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: |
|
run: |
|
||||||
sudo apt update
|
sudo apt update
|
||||||
@@ -108,7 +98,7 @@ jobs:
|
|||||||
cargo build --profile ci --benches --all-features --tests
|
cargo build --profile ci --benches --all-features --tests
|
||||||
|
|
||||||
linux:
|
linux:
|
||||||
timeout-minutes: 60
|
timeout-minutes: 30
|
||||||
# To build all features, we need more disk space than is available
|
# To build all features, we need more disk space than is available
|
||||||
# on the free OSS github runner. This is mostly due to the the
|
# on the free OSS github runner. This is mostly due to the the
|
||||||
# sentence-transformers feature.
|
# sentence-transformers feature.
|
||||||
@@ -123,16 +113,11 @@ jobs:
|
|||||||
CXX: clang++-18
|
CXX: clang++-18
|
||||||
GH_TOKEN: ${{ secrets.SOPHON_READ_TOKEN }}
|
GH_TOKEN: ${{ secrets.SOPHON_READ_TOKEN }}
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
|
||||||
# Restore everywhere, but only save from main. Per-PR saves are
|
|
||||||
# unreadable outside their own branch anyway, since GitHub scopes
|
|
||||||
# caches to the creating ref.
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: |
|
run: |
|
||||||
sudo apt update
|
sudo apt update
|
||||||
@@ -140,26 +125,10 @@ jobs:
|
|||||||
- uses: rui314/setup-mold@v1
|
- uses: rui314/setup-mold@v1
|
||||||
- name: Make Swap
|
- name: Make Swap
|
||||||
run: |
|
run: |
|
||||||
swapfile=/swapfile
|
sudo fallocate -l 16G /swapfile
|
||||||
min_swap_bytes=$((15 * 1024 * 1024 * 1024))
|
sudo chmod 600 /swapfile
|
||||||
active_swap_bytes="$(sudo swapon --show=NAME,SIZE --bytes --noheadings | awk '$1 == "/swapfile" { print $2 }')"
|
sudo mkswap /swapfile
|
||||||
if [ -n "$active_swap_bytes" ]; then
|
sudo swapon /swapfile
|
||||||
if [ "$active_swap_bytes" -ge "$min_swap_bytes" ]; then
|
|
||||||
echo "/swapfile is already active with enough space; skipping swap creation"
|
|
||||||
exit 0
|
|
||||||
fi
|
|
||||||
echo "/swapfile is already active but smaller than 16G; using /mnt/lancedb-swapfile"
|
|
||||||
swapfile=/mnt/lancedb-swapfile
|
|
||||||
fi
|
|
||||||
if sudo swapon --show=NAME --noheadings | grep -Fxq "$swapfile"; then
|
|
||||||
echo "$swapfile is already active; skipping swap creation"
|
|
||||||
exit 0
|
|
||||||
fi
|
|
||||||
sudo rm -f "$swapfile"
|
|
||||||
sudo fallocate -l 16G "$swapfile"
|
|
||||||
sudo chmod 600 "$swapfile"
|
|
||||||
sudo mkswap "$swapfile"
|
|
||||||
sudo swapon "$swapfile"
|
|
||||||
- name: Build
|
- name: Build
|
||||||
run: cargo build --profile ci --all-features --tests --locked --examples
|
run: cargo build --profile ci --all-features --tests --locked --examples
|
||||||
- name: Run feature tests
|
- name: Run feature tests
|
||||||
@@ -173,7 +142,7 @@ jobs:
|
|||||||
run: CARGO_ARGS="--profile ci" make -C ./lancedb remote-tests
|
run: CARGO_ARGS="--profile ci" make -C ./lancedb remote-tests
|
||||||
|
|
||||||
macos:
|
macos:
|
||||||
timeout-minutes: 60
|
timeout-minutes: 30
|
||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
mac-runner: ["macos-14", "macos-15"]
|
mac-runner: ["macos-14", "macos-15"]
|
||||||
@@ -183,18 +152,13 @@ jobs:
|
|||||||
shell: bash
|
shell: bash
|
||||||
working-directory: rust
|
working-directory: rust
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
- name: CPU features
|
- name: CPU features
|
||||||
run: sysctl -a | grep cpu
|
run: sysctl -a | grep cpu
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
|
||||||
# Restore everywhere, but only save from main. Per-PR saves are
|
|
||||||
# unreadable outside their own branch anyway, since GitHub scopes
|
|
||||||
# caches to the creating ref.
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: brew install protobuf
|
run: brew install protobuf
|
||||||
- name: Run tests
|
- name: Run tests
|
||||||
@@ -207,32 +171,20 @@ jobs:
|
|||||||
cargo test --profile ci --features $ALL_FEATURES --locked
|
cargo test --profile ci --features $ALL_FEATURES --locked
|
||||||
|
|
||||||
windows:
|
windows:
|
||||||
|
runs-on: windows-2022
|
||||||
strategy:
|
strategy:
|
||||||
fail-fast: false
|
|
||||||
matrix:
|
matrix:
|
||||||
include:
|
target:
|
||||||
- target: x86_64-pc-windows-msvc
|
- x86_64-pc-windows-msvc
|
||||||
runner: windows-2022
|
- aarch64-pc-windows-msvc
|
||||||
# windows-11-arm is a standard runner, so it is free on public repos.
|
|
||||||
# Running natively lets the aarch64 tests actually execute -- this
|
|
||||||
# job used to cross-compile them and then skip the test step, paying
|
|
||||||
# full codegen and link cost for a compile check.
|
|
||||||
- target: aarch64-pc-windows-msvc
|
|
||||||
runner: windows-11-arm
|
|
||||||
runs-on: ${{ matrix.runner }}
|
|
||||||
defaults:
|
defaults:
|
||||||
run:
|
run:
|
||||||
working-directory: rust/lancedb
|
working-directory: rust/lancedb
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
- name: Set target
|
- name: Set target
|
||||||
run: rustup target add ${{ matrix.target }}
|
run: rustup target add ${{ matrix.target }}
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
|
||||||
# Restore everywhere, but only save from main. Per-PR saves are
|
|
||||||
# unreadable outside their own branch anyway, since GitHub scopes
|
|
||||||
# caches to the creating ref.
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- name: Install Protoc v21.12
|
- name: Install Protoc v21.12
|
||||||
run: choco install --no-progress protoc
|
run: choco install --no-progress protoc
|
||||||
- name: Build
|
- name: Build
|
||||||
@@ -240,12 +192,11 @@ jobs:
|
|||||||
$env:VCPKG_ROOT = $env:VCPKG_INSTALLATION_ROOT
|
$env:VCPKG_ROOT = $env:VCPKG_INSTALLATION_ROOT
|
||||||
cargo build --profile ci --features aws,remote --tests --locked --target ${{ matrix.target }}
|
cargo build --profile ci --features aws,remote --tests --locked --target ${{ matrix.target }}
|
||||||
- name: Run tests
|
- name: Run tests
|
||||||
|
# Can only run tests when target matches host
|
||||||
|
if: ${{ matrix.target == 'x86_64-pc-windows-msvc' }}
|
||||||
run: |
|
run: |
|
||||||
$env:VCPKG_ROOT = $env:VCPKG_INSTALLATION_ROOT
|
$env:VCPKG_ROOT = $env:VCPKG_INSTALLATION_ROOT
|
||||||
# `--target` has to match the build step above. Without it cargo uses
|
cargo test --profile ci --features aws,remote --locked
|
||||||
# target/ci/ rather than target/<triple>/ci/ and rebuilds the entire
|
|
||||||
# dependency graph a second time.
|
|
||||||
cargo test --profile ci --features aws,remote --locked --target ${{ matrix.target }}
|
|
||||||
|
|
||||||
msrv:
|
msrv:
|
||||||
# Check the minimum supported Rust version
|
# Check the minimum supported Rust version
|
||||||
@@ -259,7 +210,7 @@ jobs:
|
|||||||
CC: clang-18
|
CC: clang-18
|
||||||
CXX: clang++-18
|
CXX: clang++-18
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
submodules: true
|
submodules: true
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
@@ -271,11 +222,6 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
toolchain: ${{ matrix.msrv }}
|
toolchain: ${{ matrix.msrv }}
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
|
||||||
# Restore everywhere, but only save from main. Per-PR saves are
|
|
||||||
# unreadable outside their own branch anyway, since GitHub scopes
|
|
||||||
# caches to the creating ref.
|
|
||||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
- name: Downgrade dependencies
|
- name: Downgrade dependencies
|
||||||
# These packages have newer requirements for MSRV
|
# These packages have newer requirements for MSRV
|
||||||
run: |
|
run: |
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ jobs:
|
|||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout
|
- name: Checkout
|
||||||
uses: actions/checkout@v6
|
uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
ref: main
|
ref: main
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ jobs:
|
|||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout
|
- name: Checkout
|
||||||
uses: actions/checkout@v6
|
uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
ref: main
|
ref: main
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
|
|||||||
@@ -27,7 +27,6 @@ python/dist
|
|||||||
*.so
|
*.so
|
||||||
*.dylib
|
*.dylib
|
||||||
*.dll
|
*.dll
|
||||||
*.pdb
|
|
||||||
|
|
||||||
## Javascript
|
## Javascript
|
||||||
*.node
|
*.node
|
||||||
|
|||||||
@@ -92,8 +92,6 @@ Python bindings changes:
|
|||||||
* Should use `LOOP.run()` to call the corresponding `AsyncTable` method.
|
* Should use `LOOP.run()` to call the corresponding `AsyncTable` method.
|
||||||
6. Add concrete sync method to `RemoteTable` class in `python/python/lancedb/remote/table.py`.
|
6. Add concrete sync method to `RemoteTable` class in `python/python/lancedb/remote/table.py`.
|
||||||
7. Add unit test in `python/tests/test_table.py`.
|
7. Add unit test in `python/tests/test_table.py`.
|
||||||
8. If you added a new public class or module-level function (not just a method on an
|
|
||||||
existing class), expose it in the API reference. See "Python API reference" below.
|
|
||||||
|
|
||||||
TypeScript bindings changes:
|
TypeScript bindings changes:
|
||||||
|
|
||||||
@@ -105,33 +103,6 @@ TypeScript bindings changes:
|
|||||||
5. Add test in `nodejs/__test__/table.test.ts`.
|
5. Add test in `nodejs/__test__/table.test.ts`.
|
||||||
6. Run `npm run docs` to generate TypeScript documentation.
|
6. Run `npm run docs` to generate TypeScript documentation.
|
||||||
|
|
||||||
## Python API reference
|
|
||||||
|
|
||||||
`docs/src/python/python.md` is the entire Python API reference. It is maintained by
|
|
||||||
hand, and anything not listed there is not rendered at all, so new public classes and
|
|
||||||
module-level functions have to be added explicitly. How depends on the module:
|
|
||||||
|
|
||||||
* `lancedb.index`, `lancedb.embeddings`, `lancedb.remote`, and `lancedb.rerankers` are
|
|
||||||
rendered by a single directive each, driven by the module's `__all__`. Add the new
|
|
||||||
name to `__all__` and it appears; forget, and it is silently omitted.
|
|
||||||
* Everything else (`lancedb`, `lancedb.table`, `lancedb.query`, `lancedb.db`, ...) is
|
|
||||||
listed symbol by symbol. Add a `::: lancedb.<module>.<Name>` line to the matching
|
|
||||||
section, and remember that the page separates synchronous and asynchronous APIs.
|
|
||||||
|
|
||||||
Deliberately undocumented: concrete implementations reached through an abstract base
|
|
||||||
(`LanceTable`, `LanceDBConnection`, `RemoteDBConnection`), query base classes already
|
|
||||||
covered by `inherited_members`, and internal helpers.
|
|
||||||
|
|
||||||
Cross-references in docstrings use mkdocstrings syntax, `[text][lancedb.table.Table]`.
|
|
||||||
Plain relative links such as `[Table](Table)` do not resolve. To check your work:
|
|
||||||
|
|
||||||
```shell
|
|
||||||
pip install -r docs/requirements.txt
|
|
||||||
cd docs && PYTHONPATH=. mkdocs build
|
|
||||||
```
|
|
||||||
|
|
||||||
The docs site only builds on pushes to `main`, so this is not covered by PR CI.
|
|
||||||
|
|
||||||
## Review Guidelines
|
## Review Guidelines
|
||||||
|
|
||||||
Please consider the following when reviewing code contributions.
|
Please consider the following when reviewing code contributions.
|
||||||
|
|||||||
Generated
+488
-1223
File diff suppressed because it is too large
Load Diff
+24
-26
@@ -13,25 +13,24 @@ categories = ["database-implementations"]
|
|||||||
rust-version = "1.91.0"
|
rust-version = "1.91.0"
|
||||||
|
|
||||||
[workspace.dependencies]
|
[workspace.dependencies]
|
||||||
lance = { "version" = "=10.0.0-beta.5", default-features = false, "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance = { "version" = "=7.0.0", default-features = false }
|
||||||
lance-core = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-core = "=7.0.0"
|
||||||
lance-datagen = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-datagen = "=7.0.0"
|
||||||
lance-file = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-file = "=7.0.0"
|
||||||
lance-io = { "version" = "=10.0.0-beta.5", default-features = false, "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-io = { "version" = "=7.0.0", default-features = false }
|
||||||
lance-index = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-index = "=7.0.0"
|
||||||
lance-linalg = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-linalg = "=7.0.0"
|
||||||
lance-namespace = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-namespace = "=7.0.0"
|
||||||
lance-namespace-impls = { "version" = "=10.0.0-beta.5", default-features = false, "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-namespace-impls = { "version" = "=7.0.0", default-features = false }
|
||||||
lance-table = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-table = "=7.0.0"
|
||||||
lance-testing = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-testing = "=7.0.0"
|
||||||
lance-datafusion = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-datafusion = "=7.0.0"
|
||||||
lance-encoding = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-encoding = "=7.0.0"
|
||||||
lance-arrow = { "version" = "=10.0.0-beta.5", "tag" = "v10.0.0-beta.5", "git" = "https://github.com/lance-format/lance.git" }
|
lance-arrow = "=7.0.0"
|
||||||
ahash = "0.8"
|
ahash = "0.8"
|
||||||
# Note that this one does not include pyarrow
|
# Note that this one does not include pyarrow
|
||||||
arrow = { version = "58.0.0", optional = false }
|
arrow = { version = "58.0.0", optional = false }
|
||||||
arrow-array = "58.0.0"
|
arrow-array = "58.0.0"
|
||||||
arrow-buffer = "58.0.0"
|
|
||||||
arrow-data = "58.0.0"
|
arrow-data = "58.0.0"
|
||||||
arrow-ipc = "58.0.0"
|
arrow-ipc = "58.0.0"
|
||||||
arrow-ord = "58.0.0"
|
arrow-ord = "58.0.0"
|
||||||
@@ -39,23 +38,21 @@ arrow-schema = "58.0.0"
|
|||||||
arrow-select = "58.0.0"
|
arrow-select = "58.0.0"
|
||||||
arrow-cast = "58.0.0"
|
arrow-cast = "58.0.0"
|
||||||
async-trait = "0"
|
async-trait = "0"
|
||||||
datafusion = { version = "54.0.0", default-features = false }
|
datafusion = { version = "53.0.0", default-features = false }
|
||||||
datafusion-catalog = "54.0.0"
|
datafusion-catalog = "53.0.0"
|
||||||
datafusion-common = { version = "54.0.0", default-features = false }
|
datafusion-common = { version = "53.0.0", default-features = false }
|
||||||
datafusion-execution = "54.0.0"
|
datafusion-execution = "53.0.0"
|
||||||
datafusion-expr = "54.0.0"
|
datafusion-expr = "53.0.0"
|
||||||
datafusion-functions = "54.0.0"
|
datafusion-functions = "53.0.0"
|
||||||
datafusion-physical-plan = "54.0.0"
|
datafusion-physical-plan = "53.0.0"
|
||||||
datafusion-physical-expr = "54.0.0"
|
datafusion-physical-expr = "53.0.0"
|
||||||
datafusion-sql = "54.0.0"
|
datafusion-sql = "53.0.0"
|
||||||
env_logger = "0.11"
|
env_logger = "0.11"
|
||||||
half = { "version" = "2.7.1", default-features = false, features = [
|
half = { "version" = "2.7.1", default-features = false, features = [
|
||||||
"num-traits",
|
"num-traits",
|
||||||
] }
|
] }
|
||||||
futures = "0"
|
futures = "0"
|
||||||
log = "0.4"
|
log = "0.4"
|
||||||
metrics = "0.24"
|
|
||||||
metrics-util = "0.19"
|
|
||||||
moka = { version = "0.12", features = ["future"] }
|
moka = { version = "0.12", features = ["future"] }
|
||||||
object_store = "0.13.2"
|
object_store = "0.13.2"
|
||||||
pin-project = "1.0.7"
|
pin-project = "1.0.7"
|
||||||
@@ -64,6 +61,7 @@ snafu = "0.8"
|
|||||||
url = "2"
|
url = "2"
|
||||||
num-traits = "0.2"
|
num-traits = "0.2"
|
||||||
regex = "1.10"
|
regex = "1.10"
|
||||||
|
lazy_static = "1"
|
||||||
semver = "1.0.25"
|
semver = "1.0.25"
|
||||||
chrono = "0.4"
|
chrono = "0.4"
|
||||||
|
|
||||||
|
|||||||
@@ -1,26 +0,0 @@
|
|||||||
# Code review guidelines
|
|
||||||
|
|
||||||
Repo-specific guidance for automated PR reviews.
|
|
||||||
|
|
||||||
## Cross-SDK parity
|
|
||||||
|
|
||||||
LanceDB exposes the same core (`rust/lancedb`) through Python, TypeScript (`nodejs`),
|
|
||||||
and Java bindings. Behavioral drift between SDKs is a recurring problem, so watch for
|
|
||||||
parity gaps when reviewing — but only flag real ones:
|
|
||||||
|
|
||||||
* If the change adds or modifies user-facing API or behavior in the shared core
|
|
||||||
(`rust/lancedb`), check whether each binding that should expose it (`python`,
|
|
||||||
`nodejs`) does. A core change with no corresponding binding update is worth a note.
|
|
||||||
* If the change adds or modifies a public API in one SDK but not the other, open the
|
|
||||||
sibling SDK's corresponding module and state whether an equivalent exists. If not,
|
|
||||||
note it as a possible parity gap and suggest a follow-up issue.
|
|
||||||
* For bug fixes, first read the sibling SDK's analogous code path to check whether the
|
|
||||||
same bug exists there. Only raise parity if it actually does. Do not ask to "port" a
|
|
||||||
fix for a bug that only ever existed in one binding.
|
|
||||||
* Stay silent on internal-only refactors, tests, docs, and changes with no cross-SDK
|
|
||||||
surface.
|
|
||||||
* Parity expectations apply to the Python and TypeScript (`nodejs`) SDKs. Java currently
|
|
||||||
implements only the remote table, not the local/embedded backend, so it is expected to
|
|
||||||
be partial — do not flag Java for missing local-only functionality.
|
|
||||||
* Keep parity feedback to a short, clearly-labeled note (e.g. "Possible SDK parity
|
|
||||||
gap: …"). It is advisory, not a merge blocker.
|
|
||||||
+3
-3
@@ -2,9 +2,9 @@ set -e
|
|||||||
|
|
||||||
RELEASE_TYPE=${1:-"stable"}
|
RELEASE_TYPE=${1:-"stable"}
|
||||||
BUMP_MINOR=${2:-false}
|
BUMP_MINOR=${2:-false}
|
||||||
HEAD_SHA=$(git rev-parse HEAD)
|
TAG_PREFIX=${3:-"v"} # Such as "python-v"
|
||||||
|
HEAD_SHA=${4:-$(git rev-parse HEAD)}
|
||||||
|
|
||||||
readonly TAG_PREFIX="v"
|
|
||||||
readonly SELF_DIR=$(cd "$( dirname "${BASH_SOURCE[0]}" )" && pwd )
|
readonly SELF_DIR=$(cd "$( dirname "${BASH_SOURCE[0]}" )" && pwd )
|
||||||
|
|
||||||
PREV_TAG=$(git tag --sort='version:refname' | grep ^$TAG_PREFIX | python $SELF_DIR/semver_sort.py $TAG_PREFIX | tail -n 1)
|
PREV_TAG=$(git tag --sort='version:refname' | grep ^$TAG_PREFIX | python $SELF_DIR/semver_sort.py $TAG_PREFIX | tail -n 1)
|
||||||
@@ -12,7 +12,7 @@ echo "Found previous tag $PREV_TAG"
|
|||||||
|
|
||||||
# Initially, we don't want to tag if we are doing stable, because we will bump
|
# Initially, we don't want to tag if we are doing stable, because we will bump
|
||||||
# again later. See comment at end for why.
|
# again later. See comment at end for why.
|
||||||
if [[ "$RELEASE_TYPE" == 'stable' ]]; then
|
if [[ "$RELEASE_TYPE" == 'stable' ]]; then
|
||||||
BUMP_ARGS="--no-tag"
|
BUMP_ARGS="--no-tag"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
|||||||
@@ -1,126 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
"""Prepare a Lance dependency update for LanceDB."""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import argparse
|
|
||||||
import json
|
|
||||||
import re
|
|
||||||
import subprocess
|
|
||||||
import sys
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import Sequence
|
|
||||||
|
|
||||||
try:
|
|
||||||
from check_lance_release import parse_semver
|
|
||||||
except ModuleNotFoundError:
|
|
||||||
# Supports importing as ci.update_lance_dependency from tests or ad hoc checks.
|
|
||||||
from ci.check_lance_release import parse_semver # type: ignore
|
|
||||||
|
|
||||||
|
|
||||||
def normalize_version(raw: str) -> str:
|
|
||||||
value = raw.strip()
|
|
||||||
value = value.removeprefix("refs/tags/")
|
|
||||||
value = value.removeprefix("v")
|
|
||||||
try:
|
|
||||||
parse_semver(value)
|
|
||||||
except ValueError:
|
|
||||||
raise ValueError(f"Unsupported Lance version or tag: {raw}")
|
|
||||||
return value
|
|
||||||
|
|
||||||
|
|
||||||
def normalized_tag(version: str) -> str:
|
|
||||||
return f"v{version}"
|
|
||||||
|
|
||||||
|
|
||||||
def branch_name(version: str) -> str:
|
|
||||||
suffix = re.sub(r"[^a-zA-Z0-9]+", "-", version).strip("-")
|
|
||||||
suffix = re.sub(r"-+", "-", suffix)
|
|
||||||
return f"codex/update-lance-{suffix}"
|
|
||||||
|
|
||||||
|
|
||||||
def commit_type(version: str) -> str:
|
|
||||||
prerelease = version.split("-", maxsplit=1)[1] if "-" in version else ""
|
|
||||||
return "chore" if "beta" in prerelease or "rc" in prerelease else "feat"
|
|
||||||
|
|
||||||
|
|
||||||
def metadata_for(version: str) -> dict[str, str]:
|
|
||||||
kind = commit_type(version)
|
|
||||||
message = f"{kind}: update lance dependency to v{version}"
|
|
||||||
return {
|
|
||||||
"version": version,
|
|
||||||
"tag": normalized_tag(version),
|
|
||||||
"branch_name": branch_name(version),
|
|
||||||
"commit_type": kind,
|
|
||||||
"commit_message": message,
|
|
||||||
"pr_title": message,
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def run_command(cmd: Sequence[str], *, cwd: Path) -> None:
|
|
||||||
subprocess.run(cmd, cwd=cwd, check=True)
|
|
||||||
|
|
||||||
|
|
||||||
def update_java_lance_core_version(repo_root: Path, version: str) -> None:
|
|
||||||
pom_path = repo_root / "java" / "pom.xml"
|
|
||||||
contents = pom_path.read_text(encoding="utf-8")
|
|
||||||
updated, count = re.subn(
|
|
||||||
r"(<lance-core\.version>)[^<]+(</lance-core\.version>)",
|
|
||||||
rf"\g<1>{version}\g<2>",
|
|
||||||
contents,
|
|
||||||
count=1,
|
|
||||||
)
|
|
||||||
if count != 1:
|
|
||||||
raise RuntimeError(
|
|
||||||
"Expected exactly one <lance-core.version> entry in java/pom.xml"
|
|
||||||
)
|
|
||||||
pom_path.write_text(updated, encoding="utf-8")
|
|
||||||
|
|
||||||
|
|
||||||
def write_github_outputs(path: str | None, payload: dict[str, str]) -> None:
|
|
||||||
if not path:
|
|
||||||
return
|
|
||||||
with open(path, "a", encoding="utf-8") as output:
|
|
||||||
for key, value in payload.items():
|
|
||||||
output.write(f"{key}={value}\n")
|
|
||||||
|
|
||||||
|
|
||||||
def main(argv: Sequence[str] | None = None) -> int:
|
|
||||||
parser = argparse.ArgumentParser(description=__doc__)
|
|
||||||
parser.add_argument(
|
|
||||||
"tag_or_version",
|
|
||||||
help="Lance tag or version, for example refs/tags/v7.2.0-beta.1 or 7.2.0",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
"--repo-root",
|
|
||||||
type=Path,
|
|
||||||
default=Path(__file__).resolve().parents[1],
|
|
||||||
help="Path to the lancedb repository root",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
"--github-output",
|
|
||||||
default=None,
|
|
||||||
help="Optional GitHub Actions output file to receive metadata fields",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
"--metadata-only",
|
|
||||||
action="store_true",
|
|
||||||
help="Only print derived metadata; do not modify dependency files",
|
|
||||||
)
|
|
||||||
args = parser.parse_args(argv)
|
|
||||||
|
|
||||||
repo_root = args.repo_root.resolve()
|
|
||||||
version = normalize_version(args.tag_or_version)
|
|
||||||
payload = metadata_for(version)
|
|
||||||
|
|
||||||
if not args.metadata_only:
|
|
||||||
run_command([sys.executable, "ci/set_lance_version.py", version], cwd=repo_root)
|
|
||||||
update_java_lance_core_version(repo_root, version)
|
|
||||||
|
|
||||||
write_github_outputs(args.github_output, payload)
|
|
||||||
print(json.dumps(payload, sort_keys=True))
|
|
||||||
return 0
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
sys.exit(main())
|
|
||||||
@@ -51,6 +51,18 @@ ignore = [
|
|||||||
# https://rustsec.org/advisories/RUSTSEC-2024-0436
|
# https://rustsec.org/advisories/RUSTSEC-2024-0436
|
||||||
{ id = "RUSTSEC-2024-0436", reason = "transitive via datafusion; awaiting ecosystem migration" },
|
{ id = "RUSTSEC-2024-0436", reason = "transitive via datafusion; awaiting ecosystem migration" },
|
||||||
|
|
||||||
|
# encoding: unmaintained. Reached through lindera-dictionary, which is
|
||||||
|
# required by the native Lindera tokenizer path. Lindera has not migrated
|
||||||
|
# off this crate yet.
|
||||||
|
# https://rustsec.org/advisories/RUSTSEC-2021-0153
|
||||||
|
{ id = "RUSTSEC-2021-0153", reason = "transitive via lindera-dictionary for native Lindera tokenizer" },
|
||||||
|
|
||||||
|
# fast-float: unsound and unmaintained. Reached only through polars-arrow
|
||||||
|
# from the optional Polars integration; replacement requires a Polars
|
||||||
|
# dependency upgrade.
|
||||||
|
# https://rustsec.org/advisories/RUSTSEC-2024-0379
|
||||||
|
{ id = "RUSTSEC-2024-0379", reason = "transitive via polars-arrow; waiting on Polars migration" },
|
||||||
|
|
||||||
# tantivy: segfault on malformed input due to missing bounds check.
|
# tantivy: segfault on malformed input due to missing bounds check.
|
||||||
# Pulled in via lance for full-text search. We only feed tantivy
|
# Pulled in via lance for full-text search. We only feed tantivy
|
||||||
# documents we construct ourselves, not attacker-controlled bytes.
|
# documents we construct ourselves, not attacker-controlled bytes.
|
||||||
@@ -68,6 +80,18 @@ ignore = [
|
|||||||
# https://rustsec.org/advisories/RUSTSEC-2025-0119
|
# https://rustsec.org/advisories/RUSTSEC-2025-0119
|
||||||
{ id = "RUSTSEC-2025-0119", reason = "transitive via hf-hub/indicatif; cosmetic formatting crate" },
|
{ id = "RUSTSEC-2025-0119", reason = "transitive via hf-hub/indicatif; cosmetic formatting crate" },
|
||||||
|
|
||||||
|
# bincode: unmaintained. Reached through lindera and lindera-dictionary,
|
||||||
|
# which are required by the native Lindera tokenizer path. Lindera has not
|
||||||
|
# migrated to another serialization format yet.
|
||||||
|
# https://rustsec.org/advisories/RUSTSEC-2025-0141
|
||||||
|
{ id = "RUSTSEC-2025-0141", reason = "transitive via lindera/lindera-dictionary for native Lindera tokenizer" },
|
||||||
|
|
||||||
|
# lru: soundness issue in IterMut. Reached only through aws-sdk-s3 in
|
||||||
|
# LanceDB's dev-dependency graph; LanceDB does not use that iterator
|
||||||
|
# directly. Clearing this requires the AWS SDK chain to update lru.
|
||||||
|
# https://rustsec.org/advisories/RUSTSEC-2026-0002
|
||||||
|
{ id = "RUSTSEC-2026-0002", reason = "transitive via aws-sdk-s3 dev-dependency; waiting on AWS SDK lru upgrade" },
|
||||||
|
|
||||||
# rustls-webpki 0.101.7 (old major line): name-constraint checks for
|
# rustls-webpki 0.101.7 (old major line): name-constraint checks for
|
||||||
# URI / wildcard names. Pulled in only via the legacy rustls 0.21 chain
|
# URI / wildcard names. Pulled in only via the legacy rustls 0.21 chain
|
||||||
# from aws-smithy-http-client. The 0.103 line we actively use is patched.
|
# from aws-smithy-http-client. The 0.103 line we actively use is patched.
|
||||||
@@ -84,23 +108,11 @@ ignore = [
|
|||||||
# https://rustsec.org/advisories/RUSTSEC-2026-0104
|
# https://rustsec.org/advisories/RUSTSEC-2026-0104
|
||||||
{ id = "RUSTSEC-2026-0104", reason = "only affects rustls-webpki 0.101 from legacy aws-smithy/rustls 0.21 chain" },
|
{ id = "RUSTSEC-2026-0104", reason = "only affects rustls-webpki 0.101 from legacy aws-smithy/rustls 0.21 chain" },
|
||||||
|
|
||||||
# pyo3 advisories in the Python bindings; tracked pending a patched pyo3 release.
|
# rand 0.8.5: soundness issue only when ThreadRng reseeds inside a custom
|
||||||
# https://rustsec.org/advisories/RUSTSEC-2026-0176
|
# logger. Reached through several transitive chains. LanceDB does not use
|
||||||
# https://rustsec.org/advisories/RUSTSEC-2026-0177
|
# rand from a custom logger; upgrade once all pinned chains accept 0.8.6+.
|
||||||
{ id = "RUSTSEC-2026-0176", reason = "pyo3 in Python bindings; awaiting patched pyo3 release" },
|
# https://rustsec.org/advisories/RUSTSEC-2026-0097
|
||||||
{ id = "RUSTSEC-2026-0177", reason = "pyo3 in Python bindings; awaiting patched pyo3 release" },
|
{ id = "RUSTSEC-2026-0097", reason = "transitive rand 0.8.5; LanceDB does not call ThreadRng from custom logging" },
|
||||||
|
|
||||||
# quick-xml < 0.41.0: quadratic runtime on duplicate attribute names (DoS).
|
|
||||||
# quick-xml < 0.41.0: unbounded namespace-declaration allocation in NsReader (DoS).
|
|
||||||
# Pulled in transitively by inferno (dev-only flame-graph dep), lance-namespace-impls
|
|
||||||
# (git dep from lance), and opendal/reqsign (cloud storage XML parsing). The XML
|
|
||||||
# parsed by opendal/reqsign comes from trusted cloud-storage endpoints (S3, GCS,
|
|
||||||
# Azure), not attacker-controlled input. Clearing requires upstream crates to migrate
|
|
||||||
# to quick-xml >= 0.41.0.
|
|
||||||
# https://rustsec.org/advisories/RUSTSEC-2026-0194
|
|
||||||
# https://rustsec.org/advisories/RUSTSEC-2026-0195
|
|
||||||
{ id = "RUSTSEC-2026-0194", reason = "transitive via inferno/lance/opendal; XML from trusted cloud endpoints, not attacker-controlled" },
|
|
||||||
{ id = "RUSTSEC-2026-0195", reason = "transitive via inferno/lance/opendal; XML from trusted cloud endpoints, not attacker-controlled" },
|
|
||||||
]
|
]
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -135,14 +147,6 @@ allow = [
|
|||||||
"CDLA-Permissive-2.0",
|
"CDLA-Permissive-2.0",
|
||||||
]
|
]
|
||||||
confidence-threshold = 0.8
|
confidence-threshold = 0.8
|
||||||
# Per-crate license exceptions: allow a license for a specific crate only,
|
|
||||||
# rather than globally via the `allow` list above.
|
|
||||||
exceptions = [
|
|
||||||
# CDDL-1.0 (copyleft) is pulled in only as a dev/profiling dependency via
|
|
||||||
# `inferno` -> `pprof` -> `lance-testing`; it is a test dependency that we
|
|
||||||
# do not distribute, so scope the allowance to `inferno` alone.
|
|
||||||
{ allow = ["CDDL-1.0"], crate = "inferno" },
|
|
||||||
]
|
|
||||||
# Crates whose license cannot be determined from Cargo metadata but whose
|
# Crates whose license cannot be determined from Cargo metadata but whose
|
||||||
# license we've manually confirmed from upstream. Keep this list minimal.
|
# license we've manually confirmed from upstream. Keep this list minimal.
|
||||||
[[licenses.clarify]]
|
[[licenses.clarify]]
|
||||||
|
|||||||
@@ -51,11 +51,6 @@ plugins:
|
|||||||
paths: [../python/python]
|
paths: [../python/python]
|
||||||
options:
|
options:
|
||||||
docstring_style: numpy
|
docstring_style: numpy
|
||||||
docstring_options:
|
|
||||||
# Attributes documented in a `Parameters` section, and pydantic
|
|
||||||
# dataclasses whose `__init__` griffe cannot see statically, both
|
|
||||||
# trip this check. It reports nothing actionable here.
|
|
||||||
warn_unknown_params: false
|
|
||||||
heading_level: 3
|
heading_level: 3
|
||||||
show_signature_annotations: true
|
show_signature_annotations: true
|
||||||
show_root_heading: true
|
show_root_heading: true
|
||||||
|
|||||||
+1
-11
@@ -453,16 +453,6 @@ paths:
|
|||||||
The metric type to use for the index. l2, Cosine, Dot are supported.
|
The metric type to use for the index. l2, Cosine, Dot are supported.
|
||||||
index_type:
|
index_type:
|
||||||
type: string
|
type: string
|
||||||
custom_stop_words:
|
|
||||||
type: [array, "null"]
|
|
||||||
items:
|
|
||||||
type: string
|
|
||||||
description: |
|
|
||||||
The custom stop-word list for an FTS index. A non-null
|
|
||||||
array replaces the language's built-in stop-word list and is only
|
|
||||||
applied when remove_stop_words is enabled. Null uses the built-in
|
|
||||||
language list, while an empty array explicitly replaces it with no
|
|
||||||
stop words.
|
|
||||||
responses:
|
responses:
|
||||||
"200":
|
"200":
|
||||||
description: Index successfully created
|
description: Index successfully created
|
||||||
@@ -520,4 +510,4 @@ paths:
|
|||||||
"401":
|
"401":
|
||||||
$ref: "#/components/responses/unauthorized"
|
$ref: "#/components/responses/unauthorized"
|
||||||
"404":
|
"404":
|
||||||
$ref: "#/components/responses/not_found"
|
$ref: "#/components/responses/not_found"
|
||||||
+30
-165
@@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`:
|
|||||||
<dependency>
|
<dependency>
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-core</artifactId>
|
<artifactId>lancedb-core</artifactId>
|
||||||
<version>0.37.1-beta.0</version>
|
<version>0.30.0</version>
|
||||||
</dependency>
|
</dependency>
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -249,57 +249,6 @@ try (BufferAllocator allocator = new RootAllocator();
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
### Creating an Empty Table
|
|
||||||
|
|
||||||
To create an empty table, send an Arrow IPC stream that contains the table schema and no record batches.
|
|
||||||
The schema in the IPC stream becomes the table schema, and rows can be inserted later.
|
|
||||||
|
|
||||||
```java
|
|
||||||
import org.lance.namespace.model.CreateTableRequest;
|
|
||||||
import org.lance.namespace.model.CreateTableResponse;
|
|
||||||
import org.apache.arrow.memory.BufferAllocator;
|
|
||||||
import org.apache.arrow.memory.RootAllocator;
|
|
||||||
import org.apache.arrow.vector.VectorSchemaRoot;
|
|
||||||
import org.apache.arrow.vector.ipc.ArrowStreamWriter;
|
|
||||||
import org.apache.arrow.vector.types.FloatingPointPrecision;
|
|
||||||
import org.apache.arrow.vector.types.pojo.ArrowType;
|
|
||||||
import org.apache.arrow.vector.types.pojo.Field;
|
|
||||||
import org.apache.arrow.vector.types.pojo.FieldType;
|
|
||||||
import org.apache.arrow.vector.types.pojo.Schema;
|
|
||||||
|
|
||||||
import java.io.ByteArrayOutputStream;
|
|
||||||
import java.nio.channels.Channels;
|
|
||||||
import java.util.Arrays;
|
|
||||||
|
|
||||||
Schema schema = new Schema(Arrays.asList(
|
|
||||||
new Field("id", FieldType.nullable(new ArrowType.Int(32, true)), null),
|
|
||||||
new Field("name", FieldType.nullable(new ArrowType.Utf8()), null),
|
|
||||||
new Field("embedding",
|
|
||||||
FieldType.nullable(new ArrowType.FixedSizeList(128)),
|
|
||||||
Arrays.asList(new Field("item",
|
|
||||||
FieldType.nullable(new ArrowType.FloatingPoint(FloatingPointPrecision.SINGLE)),
|
|
||||||
null)))
|
|
||||||
));
|
|
||||||
|
|
||||||
byte[] emptyTableData;
|
|
||||||
try (BufferAllocator allocator = new RootAllocator();
|
|
||||||
VectorSchemaRoot root = VectorSchemaRoot.create(schema, allocator)) {
|
|
||||||
root.setRowCount(0);
|
|
||||||
|
|
||||||
ByteArrayOutputStream out = new ByteArrayOutputStream();
|
|
||||||
try (ArrowStreamWriter writer = new ArrowStreamWriter(root, null, Channels.newChannel(out))) {
|
|
||||||
writer.start();
|
|
||||||
writer.end();
|
|
||||||
}
|
|
||||||
emptyTableData = out.toByteArray();
|
|
||||||
}
|
|
||||||
|
|
||||||
CreateTableRequest request = new CreateTableRequest();
|
|
||||||
request.setId(Arrays.asList("my_namespace", "empty_table"));
|
|
||||||
|
|
||||||
CreateTableResponse response = namespaceClient.createTable(request, emptyTableData);
|
|
||||||
```
|
|
||||||
|
|
||||||
### Insert
|
### Insert
|
||||||
|
|
||||||
```java
|
```java
|
||||||
@@ -482,88 +431,9 @@ query.setVector(vector);
|
|||||||
byte[] result = namespaceClient.queryTable(query);
|
byte[] result = namespaceClient.queryTable(query);
|
||||||
```
|
```
|
||||||
|
|
||||||
## Indexing
|
### Reading Query Results
|
||||||
|
|
||||||
The Java SDK exposes the REST namespace index operations through the same `LanceNamespace` client.
|
Query results are returned in Apache Arrow IPC file format. Here's how to read them:
|
||||||
Index creation runs asynchronously, so use `listTableIndices` or `describeTableIndexStats` to check progress.
|
|
||||||
|
|
||||||
### Creating a Vector Index
|
|
||||||
|
|
||||||
```java
|
|
||||||
import org.lance.namespace.model.CreateTableIndexRequest;
|
|
||||||
import org.lance.namespace.model.CreateTableIndexResponse;
|
|
||||||
|
|
||||||
CreateTableIndexRequest request = new CreateTableIndexRequest();
|
|
||||||
request.setId(Arrays.asList("my_namespace", "my_table"));
|
|
||||||
request.setColumn("embedding");
|
|
||||||
request.setIndexType("IVF_PQ");
|
|
||||||
request.setDistanceType("cosine");
|
|
||||||
request.setName("embedding_idx");
|
|
||||||
|
|
||||||
CreateTableIndexResponse response = namespaceClient.createTableIndex(request);
|
|
||||||
System.out.println("Index transaction: " + response.getTransactionId());
|
|
||||||
```
|
|
||||||
|
|
||||||
### Creating a Scalar Index
|
|
||||||
|
|
||||||
```java
|
|
||||||
import org.lance.namespace.model.CreateTableIndexRequest;
|
|
||||||
import org.lance.namespace.model.CreateTableScalarIndexResponse;
|
|
||||||
|
|
||||||
CreateTableIndexRequest request = new CreateTableIndexRequest();
|
|
||||||
request.setId(Arrays.asList("my_namespace", "my_table"));
|
|
||||||
request.setColumn("category");
|
|
||||||
request.setIndexType("BTREE");
|
|
||||||
request.setName("category_idx");
|
|
||||||
|
|
||||||
CreateTableScalarIndexResponse response = namespaceClient.createTableScalarIndex(request);
|
|
||||||
System.out.println("Index transaction: " + response.getTransactionId());
|
|
||||||
```
|
|
||||||
|
|
||||||
### Creating a Full Text Search Index
|
|
||||||
|
|
||||||
```java
|
|
||||||
import org.lance.namespace.model.CreateTableIndexRequest;
|
|
||||||
import org.lance.namespace.model.CreateTableScalarIndexResponse;
|
|
||||||
|
|
||||||
CreateTableIndexRequest request = new CreateTableIndexRequest();
|
|
||||||
request.setId(Arrays.asList("my_namespace", "my_table"));
|
|
||||||
request.setColumn("text_column");
|
|
||||||
request.setIndexType("FTS");
|
|
||||||
request.setName("text_idx");
|
|
||||||
request.setBaseTokenizer("simple");
|
|
||||||
request.setLowerCase(true);
|
|
||||||
request.setWithPosition(true);
|
|
||||||
|
|
||||||
CreateTableScalarIndexResponse response = namespaceClient.createTableScalarIndex(request);
|
|
||||||
System.out.println("Index transaction: " + response.getTransactionId());
|
|
||||||
```
|
|
||||||
|
|
||||||
### Listing Indexes
|
|
||||||
|
|
||||||
```java
|
|
||||||
import org.lance.namespace.model.IndexContent;
|
|
||||||
import org.lance.namespace.model.ListTableIndicesRequest;
|
|
||||||
import org.lance.namespace.model.ListTableIndicesResponse;
|
|
||||||
|
|
||||||
ListTableIndicesRequest request = new ListTableIndicesRequest();
|
|
||||||
request.setId(Arrays.asList("my_namespace", "my_table"));
|
|
||||||
|
|
||||||
ListTableIndicesResponse response = namespaceClient.listTableIndices(request);
|
|
||||||
for (IndexContent index : response.getIndexes()) {
|
|
||||||
System.out.println(index.getIndexName() + ": " + index.getStatus());
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
!!! note
|
|
||||||
The current Java namespace API exposes index type, index name, distance type, and full text search tokenizer options.
|
|
||||||
IVF training parameters such as `num_partitions` are not exposed by `CreateTableIndexRequest` yet.
|
|
||||||
To make those configurable from Java, the namespace API must add those fields first.
|
|
||||||
|
|
||||||
## Reading Query Results
|
|
||||||
|
|
||||||
Query results are returned as bytes in Apache Arrow IPC file format. Put the byte-channel
|
|
||||||
adapter behind a small helper so query code can work with `ArrowFileReader` directly:
|
|
||||||
|
|
||||||
```java
|
```java
|
||||||
import org.apache.arrow.vector.ipc.ArrowFileReader;
|
import org.apache.arrow.vector.ipc.ArrowFileReader;
|
||||||
@@ -571,50 +441,45 @@ import org.apache.arrow.vector.VectorSchemaRoot;
|
|||||||
import org.apache.arrow.memory.BufferAllocator;
|
import org.apache.arrow.memory.BufferAllocator;
|
||||||
import org.apache.arrow.memory.RootAllocator;
|
import org.apache.arrow.memory.RootAllocator;
|
||||||
|
|
||||||
import java.io.IOException;
|
|
||||||
import java.nio.ByteBuffer;
|
import java.nio.ByteBuffer;
|
||||||
import java.nio.channels.SeekableByteChannel;
|
import java.nio.channels.SeekableByteChannel;
|
||||||
|
|
||||||
final class ArrowIpc {
|
// Helper class to read Arrow data from byte array
|
||||||
static ArrowFileReader openFileReader(byte[] data, BufferAllocator allocator) throws IOException {
|
class ByteArraySeekableByteChannel implements SeekableByteChannel {
|
||||||
return new ArrowFileReader(new ByteArraySeekableByteChannel(data), allocator);
|
private final byte[] data;
|
||||||
|
private long position = 0;
|
||||||
|
private boolean isOpen = true;
|
||||||
|
|
||||||
|
public ByteArraySeekableByteChannel(byte[] data) {
|
||||||
|
this.data = data;
|
||||||
}
|
}
|
||||||
|
|
||||||
private static final class ByteArraySeekableByteChannel implements SeekableByteChannel {
|
@Override
|
||||||
private final byte[] data;
|
public int read(ByteBuffer dst) {
|
||||||
private long position = 0;
|
int remaining = dst.remaining();
|
||||||
private boolean isOpen = true;
|
int available = (int) (data.length - position);
|
||||||
|
if (available <= 0) return -1;
|
||||||
private ByteArraySeekableByteChannel(byte[] data) {
|
int toRead = Math.min(remaining, available);
|
||||||
this.data = data;
|
dst.put(data, (int) position, toRead);
|
||||||
}
|
position += toRead;
|
||||||
|
return toRead;
|
||||||
@Override
|
|
||||||
public int read(ByteBuffer dst) {
|
|
||||||
int remaining = dst.remaining();
|
|
||||||
int available = (int) (data.length - position);
|
|
||||||
if (available <= 0) return -1;
|
|
||||||
int toRead = Math.min(remaining, available);
|
|
||||||
dst.put(data, (int) position, toRead);
|
|
||||||
position += toRead;
|
|
||||||
return toRead;
|
|
||||||
}
|
|
||||||
|
|
||||||
@Override public long position() { return position; }
|
|
||||||
@Override public SeekableByteChannel position(long newPosition) { position = newPosition; return this; }
|
|
||||||
@Override public long size() { return data.length; }
|
|
||||||
@Override public boolean isOpen() { return isOpen; }
|
|
||||||
@Override public void close() { isOpen = false; }
|
|
||||||
@Override public int write(ByteBuffer src) { throw new UnsupportedOperationException(); }
|
|
||||||
@Override public SeekableByteChannel truncate(long size) { throw new UnsupportedOperationException(); }
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Override public long position() { return position; }
|
||||||
|
@Override public SeekableByteChannel position(long newPosition) { position = newPosition; return this; }
|
||||||
|
@Override public long size() { return data.length; }
|
||||||
|
@Override public boolean isOpen() { return isOpen; }
|
||||||
|
@Override public void close() { isOpen = false; }
|
||||||
|
@Override public int write(ByteBuffer src) { throw new UnsupportedOperationException(); }
|
||||||
|
@Override public SeekableByteChannel truncate(long size) { throw new UnsupportedOperationException(); }
|
||||||
}
|
}
|
||||||
|
|
||||||
// Read query results
|
// Read query results
|
||||||
byte[] queryResult = namespaceClient.queryTable(query);
|
byte[] queryResult = namespaceClient.queryTable(query);
|
||||||
|
|
||||||
try (BufferAllocator allocator = new RootAllocator();
|
try (BufferAllocator allocator = new RootAllocator();
|
||||||
ArrowFileReader reader = ArrowIpc.openFileReader(queryResult, allocator)) {
|
ArrowFileReader reader = new ArrowFileReader(
|
||||||
|
new ByteArraySeekableByteChannel(queryResult), allocator)) {
|
||||||
|
|
||||||
for (int i = 0; i < reader.getRecordBlocks().size(); i++) {
|
for (int i = 0; i < reader.getRecordBlocks().size(); i++) {
|
||||||
reader.loadRecordBatch(reader.getRecordBlocks().get(i));
|
reader.loadRecordBatch(reader.getRecordBlocks().get(i));
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# Contributing to LanceDB Typescript
|
# Contributing to LanceDB Typescript
|
||||||
|
|
||||||
This document outlines the process for contributing to LanceDB Typescript.
|
This document outlines the process for contributing to LanceDB Typescript.
|
||||||
For general contribution guidelines, see [CONTRIBUTING.md](https://github.com/lancedb/lancedb/blob/main/CONTRIBUTING.md).
|
For general contribution guidelines, see [CONTRIBUTING.md](../CONTRIBUTING.md).
|
||||||
|
|
||||||
## Project layout
|
## Project layout
|
||||||
|
|
||||||
|
|||||||
@@ -1,43 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / BranchContents
|
|
||||||
|
|
||||||
# Class: BranchContents
|
|
||||||
|
|
||||||
## Constructors
|
|
||||||
|
|
||||||
### new BranchContents()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
new BranchContents(): BranchContents
|
|
||||||
```
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
[`BranchContents`](BranchContents.md)
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### manifestSize
|
|
||||||
|
|
||||||
```ts
|
|
||||||
manifestSize: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### parentBranch?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional parentBranch: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### parentVersion
|
|
||||||
|
|
||||||
```ts
|
|
||||||
parentVersion: number;
|
|
||||||
```
|
|
||||||
@@ -1,139 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / Branches
|
|
||||||
|
|
||||||
# Class: Branches
|
|
||||||
|
|
||||||
Branch manager for a [Table](Table.md).
|
|
||||||
|
|
||||||
Unlike tags, `create` and `checkout` return a new [Table](Table.md) handle scoped
|
|
||||||
to the branch; writes on it do not affect `main`.
|
|
||||||
|
|
||||||
## Methods
|
|
||||||
|
|
||||||
### checkout()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
checkout(name, version?): Promise<Table>
|
|
||||||
```
|
|
||||||
|
|
||||||
Check out an existing branch and return a handle scoped to it.
|
|
||||||
|
|
||||||
With `version` set, the returned handle is pinned to that version of the
|
|
||||||
branch (a read-only, detached view); otherwise it tracks the branch's
|
|
||||||
latest and stays writable.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **name**: `string`
|
|
||||||
|
|
||||||
* **version?**: `number`
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`Table`](Table.md)>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### create()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
create(
|
|
||||||
name,
|
|
||||||
fromRef?,
|
|
||||||
fromVersion?): Promise<Table>
|
|
||||||
```
|
|
||||||
|
|
||||||
Create a branch and return a handle scoped to it.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **name**: `string`
|
|
||||||
Name of the new branch.
|
|
||||||
|
|
||||||
* **fromRef?**: `string`
|
|
||||||
Source branch to fork from. Defaults to `main`.
|
|
||||||
|
|
||||||
* **fromVersion?**: `number`
|
|
||||||
A specific version on `fromRef`. Defaults to latest.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`Table`](Table.md)>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### delete()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
delete(name): Promise<void>
|
|
||||||
```
|
|
||||||
|
|
||||||
Delete a branch.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **name**: `string`
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`void`>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### diff()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
diff(fromBranch): Promise<BranchDiff>
|
|
||||||
```
|
|
||||||
|
|
||||||
Compare a branch against main without modifying either branch.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **fromBranch**: `string`
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`BranchDiff`](../interfaces/BranchDiff.md)>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### list()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
list(): Promise<Record<string, BranchContents>>
|
|
||||||
```
|
|
||||||
|
|
||||||
List all branches, mapping name to branch metadata.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`Record`<`string`, [`BranchContents`](BranchContents.md)>>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### merge()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
merge(fromBranch, dryRun): Promise<MergeBranchResult>
|
|
||||||
```
|
|
||||||
|
|
||||||
Merge a branch into main.
|
|
||||||
|
|
||||||
Set `dryRun` to `true` to preview the merge. A rejected merge resolves
|
|
||||||
with `status: "rejected"` instead of throwing.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **fromBranch**: `string`
|
|
||||||
Branch to merge from.
|
|
||||||
|
|
||||||
* **dryRun**: `boolean` = `false`
|
|
||||||
When true, only preview the merge. Defaults to false.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`MergeBranchResult`](../interfaces/MergeBranchResult.md)>
|
|
||||||
@@ -57,24 +57,6 @@ block size may be added in the future.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### fm()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
static fm(): Index
|
|
||||||
```
|
|
||||||
|
|
||||||
Create an FM-Index.
|
|
||||||
|
|
||||||
An FM-Index is a scalar index on string or binary columns that accelerates
|
|
||||||
substring search, i.e. `contains(col, 'needle')`. Unlike the tokenized
|
|
||||||
full-text-search index, it matches arbitrary substrings of the raw bytes.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
[`Index`](Index.md)
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### fts()
|
### fts()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
|
|||||||
@@ -76,56 +76,6 @@ the query optimizer chooses a suboptimal path.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### useLsm()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
useLsm(enable): MergeInsertBuilder
|
|
||||||
```
|
|
||||||
|
|
||||||
Control MemWAL routing for this merge.
|
|
||||||
|
|
||||||
By default (unset), a `mergeInsert` on a table with an LSM write spec is
|
|
||||||
routed through Lance's MemWAL shard writer, and a table without one uses the
|
|
||||||
standard path.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **enable**: `boolean`
|
|
||||||
`true` forces MemWAL routing and errors if the table has no
|
|
||||||
LSM write spec. `false` forces the standard write path even when a spec is set.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
[`MergeInsertBuilder`](MergeInsertBuilder.md)
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### validateSingleShard()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
validateSingleShard(validateSingleShard): MergeInsertBuilder
|
|
||||||
```
|
|
||||||
|
|
||||||
Controls how an LSM merge checks that its input targets a single shard.
|
|
||||||
|
|
||||||
When a table has an LSM write spec, every row in a `mergeInsert` call must
|
|
||||||
route to the same shard. When `true` (the default), every row is inspected
|
|
||||||
to verify this. When `false`, only the first row is inspected and the
|
|
||||||
shard it routes to is used for the whole input — a faster path for callers
|
|
||||||
that have already pre-sharded their input. Has no effect on tables without
|
|
||||||
an LSM write spec.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **validateSingleShard**: `boolean`
|
|
||||||
Whether to check every row routes to one shard. Defaults to `true`.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
[`MergeInsertBuilder`](MergeInsertBuilder.md)
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### whenMatchedUpdateAll()
|
### whenMatchedUpdateAll()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
|
|||||||
@@ -33,7 +33,7 @@ protected inner: Query | Promise<Query>;
|
|||||||
### analyzePlan()
|
### analyzePlan()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
analyzePlan(distributedMetrics?): Promise<string>
|
analyzePlan(): Promise<string>
|
||||||
```
|
```
|
||||||
|
|
||||||
Executes the query and returns the physical query plan annotated with runtime metrics.
|
Executes the query and returns the physical query plan annotated with runtime metrics.
|
||||||
@@ -41,12 +41,6 @@ Executes the query and returns the physical query plan annotated with runtime me
|
|||||||
This is useful for debugging and performance analysis, as it shows how the query was executed
|
This is useful for debugging and performance analysis, as it shows how the query was executed
|
||||||
and includes metrics such as elapsed time, rows processed, and I/O statistics.
|
and includes metrics such as elapsed time, rows processed, and I/O statistics.
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **distributedMetrics?**: [`AnalyzePlanDistributedMetrics`](../type-aliases/AnalyzePlanDistributedMetrics.md)
|
|
||||||
How distributed worker metrics are displayed for remote query plans.
|
|
||||||
Defaults to `"aggregate"`.
|
|
||||||
|
|
||||||
#### Returns
|
#### Returns
|
||||||
|
|
||||||
`Promise`<`string`>
|
`Promise`<`string`>
|
||||||
@@ -497,42 +491,6 @@ ArrowTable.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### useLsm()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
useLsm(enable): this
|
|
||||||
```
|
|
||||||
|
|
||||||
Control MemWAL read routing for this query.
|
|
||||||
|
|
||||||
By default (unset), when the table carries a MemWAL write spec (see
|
|
||||||
[Table#setLsmWriteSpec](Table.md#setlsmwritespec)), reads are routed through the LSM scanner so
|
|
||||||
they also return data written via the `mergeInsert` LSM path that has not yet
|
|
||||||
been compacted into the base table (the active/frozen in-memory memtables and
|
|
||||||
the flushed generations), deduplicated by primary key; a table without a spec
|
|
||||||
reads the base table.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **enable**: `boolean`
|
|
||||||
`true` forces the LSM scanner and errors if the table has no
|
|
||||||
MemWAL write spec. `false` bypasses the MemWAL and reads the base table only,
|
|
||||||
even when a spec is present.
|
|
||||||
Note: the LSM scanner does not support every query shape (e.g. reranking,
|
|
||||||
hybrid search, `orderBy`). On a MemWAL table those shapes error unless
|
|
||||||
`useLsm(false)` is set, because a base-only read would silently exclude
|
|
||||||
un-compacted MemWAL data.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`this`
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.useLsm`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### where()
|
### where()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -560,9 +518,6 @@ x > 5 OR y = 'test'
|
|||||||
|
|
||||||
Filtering performance can often be improved by creating a scalar index
|
Filtering performance can often be improved by creating a scalar index
|
||||||
on the filter column(s).
|
on the filter column(s).
|
||||||
|
|
||||||
Calling this multiple times combines the filters with a logical AND rather
|
|
||||||
than replacing the previous filter.
|
|
||||||
```
|
```
|
||||||
|
|
||||||
#### Inherited from
|
#### Inherited from
|
||||||
|
|||||||
@@ -38,7 +38,7 @@ protected inner: NativeQueryType | Promise<NativeQueryType>;
|
|||||||
### analyzePlan()
|
### analyzePlan()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
analyzePlan(distributedMetrics?): Promise<string>
|
analyzePlan(): Promise<string>
|
||||||
```
|
```
|
||||||
|
|
||||||
Executes the query and returns the physical query plan annotated with runtime metrics.
|
Executes the query and returns the physical query plan annotated with runtime metrics.
|
||||||
@@ -46,12 +46,6 @@ Executes the query and returns the physical query plan annotated with runtime me
|
|||||||
This is useful for debugging and performance analysis, as it shows how the query was executed
|
This is useful for debugging and performance analysis, as it shows how the query was executed
|
||||||
and includes metrics such as elapsed time, rows processed, and I/O statistics.
|
and includes metrics such as elapsed time, rows processed, and I/O statistics.
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **distributedMetrics?**: [`AnalyzePlanDistributedMetrics`](../type-aliases/AnalyzePlanDistributedMetrics.md)
|
|
||||||
How distributed worker metrics are displayed for remote query plans.
|
|
||||||
Defaults to `"aggregate"`.
|
|
||||||
|
|
||||||
#### Returns
|
#### Returns
|
||||||
|
|
||||||
`Promise`<`string`>
|
`Promise`<`string`>
|
||||||
|
|||||||
@@ -110,23 +110,6 @@ containing the new version number of the table after altering the columns.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### branches()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract branches(): Promise<Branches>
|
|
||||||
```
|
|
||||||
|
|
||||||
Get the branch manager for this table.
|
|
||||||
|
|
||||||
Branches are isolated, writable lines of history forked from another
|
|
||||||
branch (or version). Writes on a branch do not affect `main`.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`Branches`](Branches.md)>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### checkout()
|
### checkout()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -204,25 +187,6 @@ Any attempt to use the table after it is closed will result in an error.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### closeLsmWriters()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract closeLsmWriters(): Promise<void>
|
|
||||||
```
|
|
||||||
|
|
||||||
Drain and close any cached MemWAL shard writers held for this table.
|
|
||||||
|
|
||||||
When an [LsmWriteSpec](../interfaces/LsmWriteSpec.md) is installed, `mergeInsert` opens MemWAL
|
|
||||||
shard writers and caches them for reuse across calls. This closes them,
|
|
||||||
flushing pending data; writers reopen lazily on the next `mergeInsert`.
|
|
||||||
It is a no-op when no writers are cached.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`void`>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### countRows()
|
### countRows()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -295,23 +259,6 @@ await table.createIndex("my_float_col");
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### currentBranch()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract currentBranch(): null | string
|
|
||||||
```
|
|
||||||
|
|
||||||
The branch this table handle is scoped to, or `null` for the main branch.
|
|
||||||
|
|
||||||
A handle returned by [Branches.create](Branches.md#create) or [Branches.checkout](Branches.md#checkout)
|
|
||||||
reports the branch it targets; a handle opened normally reports `null`.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`null` \| `string`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### delete()
|
### delete()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -398,26 +345,6 @@ Drop an index from the table.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### getLsmWriteSpec()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract getLsmWriteSpec(): Promise<undefined | LsmWriteSpec>
|
|
||||||
```
|
|
||||||
|
|
||||||
Read the [LsmWriteSpec](../interfaces/LsmWriteSpec.md) currently installed on this table.
|
|
||||||
|
|
||||||
Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
|
|
||||||
spec has been set, or it was removed with [Table#unsetLsmWriteSpec](Table.md#unsetlsmwritespec)).
|
|
||||||
The returned spec — including its `maintainedIndexes` and
|
|
||||||
`writerConfigDefaults` — mirrors what was passed to
|
|
||||||
[Table#setLsmWriteSpec](Table.md#setlsmwritespec).
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<`undefined` \| [`LsmWriteSpec`](../interfaces/LsmWriteSpec.md)>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### indexStats()
|
### indexStats()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -934,32 +861,6 @@ Return the table as an arrow table
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### tokenize()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract tokenize(query, options): Promise<FtsToken[]>
|
|
||||||
```
|
|
||||||
|
|
||||||
Tokenize a full-text search query using the tokenizer configured on an FTS index.
|
|
||||||
|
|
||||||
Specify exactly one of `column` or `indexName`.
|
|
||||||
|
|
||||||
Model-backed tokenizers such as `jieba/*` and `lindera/*` are rebuilt in
|
|
||||||
the client process from index metadata. For remote tables, this means the
|
|
||||||
same tokenizer model files must also exist locally.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **query**: `string`
|
|
||||||
|
|
||||||
* **options**: [`TokenizeTableOptions`](../type-aliases/TokenizeTableOptions.md)
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`FtsToken`](../interfaces/FtsToken.md)[]>
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### unsetLsmWriteSpec()
|
### unsetLsmWriteSpec()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -1074,29 +975,6 @@ based on the row being updated (e.g. "my_col + 1")
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### updateFieldMetadata()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
abstract updateFieldMetadata(updates): Promise<UpdateFieldMetadataResult>
|
|
||||||
```
|
|
||||||
|
|
||||||
Update per-field (column) metadata.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **updates**: [`FieldMetadataUpdate`](../interfaces/FieldMetadataUpdate.md)[]
|
|
||||||
One or more per-field updates. Each
|
|
||||||
update's metadata is merged into the field's existing metadata by default;
|
|
||||||
a value of `null` deletes that key, and `replace: true` swaps the whole map.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`Promise`<[`UpdateFieldMetadataResult`](../interfaces/UpdateFieldMetadataResult.md)>
|
|
||||||
|
|
||||||
resolves to the new table version.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### vectorSearch()
|
### vectorSearch()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
|
|||||||
@@ -29,7 +29,7 @@ protected inner: TakeQuery | Promise<TakeQuery>;
|
|||||||
### analyzePlan()
|
### analyzePlan()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
analyzePlan(distributedMetrics?): Promise<string>
|
analyzePlan(): Promise<string>
|
||||||
```
|
```
|
||||||
|
|
||||||
Executes the query and returns the physical query plan annotated with runtime metrics.
|
Executes the query and returns the physical query plan annotated with runtime metrics.
|
||||||
@@ -37,12 +37,6 @@ Executes the query and returns the physical query plan annotated with runtime me
|
|||||||
This is useful for debugging and performance analysis, as it shows how the query was executed
|
This is useful for debugging and performance analysis, as it shows how the query was executed
|
||||||
and includes metrics such as elapsed time, rows processed, and I/O statistics.
|
and includes metrics such as elapsed time, rows processed, and I/O statistics.
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **distributedMetrics?**: [`AnalyzePlanDistributedMetrics`](../type-aliases/AnalyzePlanDistributedMetrics.md)
|
|
||||||
How distributed worker metrics are displayed for remote query plans.
|
|
||||||
Defaults to `"aggregate"`.
|
|
||||||
|
|
||||||
#### Returns
|
#### Returns
|
||||||
|
|
||||||
`Promise`<`string`>
|
`Promise`<`string`>
|
||||||
@@ -273,29 +267,6 @@ ArrowTable.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### useLsm()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
useLsm(enable): this
|
|
||||||
```
|
|
||||||
|
|
||||||
Control MemWAL read routing for this take query.
|
|
||||||
|
|
||||||
`false` bypasses the MemWAL and reads the base table only — the escape hatch,
|
|
||||||
since take-by-row-id/offset is not supported on the LSM scanner and, on a
|
|
||||||
MemWAL table, auto-routes to it and errors otherwise.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **enable**: `boolean`
|
|
||||||
`false` reads the base table only.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`this`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### withRowId()
|
### withRowId()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
|
|||||||
@@ -51,7 +51,7 @@ addQueryVector(vector): VectorQuery
|
|||||||
### analyzePlan()
|
### analyzePlan()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
analyzePlan(distributedMetrics?): Promise<string>
|
analyzePlan(): Promise<string>
|
||||||
```
|
```
|
||||||
|
|
||||||
Executes the query and returns the physical query plan annotated with runtime metrics.
|
Executes the query and returns the physical query plan annotated with runtime metrics.
|
||||||
@@ -59,12 +59,6 @@ Executes the query and returns the physical query plan annotated with runtime me
|
|||||||
This is useful for debugging and performance analysis, as it shows how the query was executed
|
This is useful for debugging and performance analysis, as it shows how the query was executed
|
||||||
and includes metrics such as elapsed time, rows processed, and I/O statistics.
|
and includes metrics such as elapsed time, rows processed, and I/O statistics.
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **distributedMetrics?**: [`AnalyzePlanDistributedMetrics`](../type-aliases/AnalyzePlanDistributedMetrics.md)
|
|
||||||
How distributed worker metrics are displayed for remote query plans.
|
|
||||||
Defaults to `"aggregate"`.
|
|
||||||
|
|
||||||
#### Returns
|
#### Returns
|
||||||
|
|
||||||
`Promise`<`string`>
|
`Promise`<`string`>
|
||||||
@@ -746,42 +740,6 @@ ArrowTable.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### useLsm()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
useLsm(enable): this
|
|
||||||
```
|
|
||||||
|
|
||||||
Control MemWAL read routing for this query.
|
|
||||||
|
|
||||||
By default (unset), when the table carries a MemWAL write spec (see
|
|
||||||
[Table#setLsmWriteSpec](Table.md#setlsmwritespec)), reads are routed through the LSM scanner so
|
|
||||||
they also return data written via the `mergeInsert` LSM path that has not yet
|
|
||||||
been compacted into the base table (the active/frozen in-memory memtables and
|
|
||||||
the flushed generations), deduplicated by primary key; a table without a spec
|
|
||||||
reads the base table.
|
|
||||||
|
|
||||||
#### Parameters
|
|
||||||
|
|
||||||
* **enable**: `boolean`
|
|
||||||
`true` forces the LSM scanner and errors if the table has no
|
|
||||||
MemWAL write spec. `false` bypasses the MemWAL and reads the base table only,
|
|
||||||
even when a spec is present.
|
|
||||||
Note: the LSM scanner does not support every query shape (e.g. reranking,
|
|
||||||
hybrid search, `orderBy`). On a MemWAL table those shapes error unless
|
|
||||||
`useLsm(false)` is set, because a base-only read would silently exclude
|
|
||||||
un-compacted MemWAL data.
|
|
||||||
|
|
||||||
#### Returns
|
|
||||||
|
|
||||||
`this`
|
|
||||||
|
|
||||||
#### Inherited from
|
|
||||||
|
|
||||||
`StandardQueryBase.useLsm`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### where()
|
### where()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -809,9 +767,6 @@ x > 5 OR y = 'test'
|
|||||||
|
|
||||||
Filtering performance can often be improved by creating a scalar index
|
Filtering performance can often be improved by creating a scalar index
|
||||||
on the filter column(s).
|
on the filter column(s).
|
||||||
|
|
||||||
Calling this multiple times combines the filters with a logical AND rather
|
|
||||||
than replacing the previous filter.
|
|
||||||
```
|
```
|
||||||
|
|
||||||
#### Inherited from
|
#### Inherited from
|
||||||
|
|||||||
@@ -1,29 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / OAuthFlowType
|
|
||||||
|
|
||||||
# Enumeration: OAuthFlowType
|
|
||||||
|
|
||||||
OAuth authentication flow types.
|
|
||||||
|
|
||||||
## Enumeration Members
|
|
||||||
|
|
||||||
### AzureManagedIdentity
|
|
||||||
|
|
||||||
```ts
|
|
||||||
AzureManagedIdentity: "azure_managed_identity";
|
|
||||||
```
|
|
||||||
|
|
||||||
Azure Managed Identity via IMDS.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### ClientCredentials
|
|
||||||
|
|
||||||
```ts
|
|
||||||
ClientCredentials: "client_credentials";
|
|
||||||
```
|
|
||||||
|
|
||||||
Client Credentials grant (service-to-service / M2M).
|
|
||||||
@@ -1,42 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / instrumentLanceDbMetrics
|
|
||||||
|
|
||||||
# Function: instrumentLanceDbMetrics()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
function instrumentLanceDbMetrics(meterProvider?): boolean
|
|
||||||
```
|
|
||||||
|
|
||||||
Register LanceDB metrics as OpenTelemetry observable instruments.
|
|
||||||
|
|
||||||
Installs a process-global metrics recorder and creates one observable
|
|
||||||
instrument per LanceDB metric (currently object store request counts, bytes,
|
|
||||||
latency, errors, and throttles) on the given (or global) `MeterProvider`. The
|
|
||||||
configured `MetricReader` then collects them on its own schedule.
|
|
||||||
|
|
||||||
Counters and gauges map directly to observable counters/gauges. Because
|
|
||||||
OpenTelemetry has no asynchronous histogram instrument, each histogram is
|
|
||||||
exported Prometheus-style as cumulative `le` bucket counts (`<name>_bucket`,
|
|
||||||
with an `le` attribute) plus `<name>_count` and `<name>_sum`.
|
|
||||||
|
|
||||||
Requires `@opentelemetry/api` (a dependency) and, to actually export, an
|
|
||||||
OpenTelemetry SDK such as `@opentelemetry/sdk-metrics`.
|
|
||||||
|
|
||||||
## Parameters
|
|
||||||
|
|
||||||
* **meterProvider?**: `MeterProvider`
|
|
||||||
The provider to register instruments on. Defaults to the
|
|
||||||
global provider from `@opentelemetry/api`.
|
|
||||||
|
|
||||||
## Returns
|
|
||||||
|
|
||||||
`boolean`
|
|
||||||
|
|
||||||
`true` if the recorder is installed and instruments are registered.
|
|
||||||
`false` if a different `metrics` recorder is already installed in this
|
|
||||||
process (only one global recorder is permitted), in which case a warning is
|
|
||||||
emitted and no instruments are created. Calling this more than once is safe;
|
|
||||||
instruments are created only on the first successful call.
|
|
||||||
@@ -1,26 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / tokenize
|
|
||||||
|
|
||||||
# Function: tokenize()
|
|
||||||
|
|
||||||
```ts
|
|
||||||
function tokenize(query, options?): Promise<FtsToken[]>
|
|
||||||
```
|
|
||||||
|
|
||||||
Tokenize a full-text search query using an explicit tokenizer.
|
|
||||||
|
|
||||||
This does not require a table or FTS index. The tokenizer options match
|
|
||||||
[Index.fts](../classes/Index.md#fts).
|
|
||||||
|
|
||||||
## Parameters
|
|
||||||
|
|
||||||
* **query**: `string`
|
|
||||||
|
|
||||||
* **options?**: `Partial`<[`TokenizeOptions`](../interfaces/TokenizeOptions.md)>
|
|
||||||
|
|
||||||
## Returns
|
|
||||||
|
|
||||||
`Promise`<[`FtsToken`](../interfaces/FtsToken.md)[]>
|
|
||||||
@@ -12,7 +12,6 @@
|
|||||||
## Enumerations
|
## Enumerations
|
||||||
|
|
||||||
- [FullTextQueryType](enumerations/FullTextQueryType.md)
|
- [FullTextQueryType](enumerations/FullTextQueryType.md)
|
||||||
- [OAuthFlowType](enumerations/OAuthFlowType.md)
|
|
||||||
- [Occur](enumerations/Occur.md)
|
- [Occur](enumerations/Occur.md)
|
||||||
- [Operator](enumerations/Operator.md)
|
- [Operator](enumerations/Operator.md)
|
||||||
|
|
||||||
@@ -20,8 +19,6 @@
|
|||||||
|
|
||||||
- [BooleanQuery](classes/BooleanQuery.md)
|
- [BooleanQuery](classes/BooleanQuery.md)
|
||||||
- [BoostQuery](classes/BoostQuery.md)
|
- [BoostQuery](classes/BoostQuery.md)
|
||||||
- [BranchContents](classes/BranchContents.md)
|
|
||||||
- [Branches](classes/Branches.md)
|
|
||||||
- [Connection](classes/Connection.md)
|
- [Connection](classes/Connection.md)
|
||||||
- [HeaderProvider](classes/HeaderProvider.md)
|
- [HeaderProvider](classes/HeaderProvider.md)
|
||||||
- [Index](classes/Index.md)
|
- [Index](classes/Index.md)
|
||||||
@@ -52,11 +49,6 @@
|
|||||||
- [AddDataOptions](interfaces/AddDataOptions.md)
|
- [AddDataOptions](interfaces/AddDataOptions.md)
|
||||||
- [AddResult](interfaces/AddResult.md)
|
- [AddResult](interfaces/AddResult.md)
|
||||||
- [AlterColumnsResult](interfaces/AlterColumnsResult.md)
|
- [AlterColumnsResult](interfaces/AlterColumnsResult.md)
|
||||||
- [BranchColumnChange](interfaces/BranchColumnChange.md)
|
|
||||||
- [BranchColumnSummary](interfaces/BranchColumnSummary.md)
|
|
||||||
- [BranchDiff](interfaces/BranchDiff.md)
|
|
||||||
- [BranchIndexSummary](interfaces/BranchIndexSummary.md)
|
|
||||||
- [BranchRowCountSummary](interfaces/BranchRowCountSummary.md)
|
|
||||||
- [ClientConfig](interfaces/ClientConfig.md)
|
- [ClientConfig](interfaces/ClientConfig.md)
|
||||||
- [ColumnAlteration](interfaces/ColumnAlteration.md)
|
- [ColumnAlteration](interfaces/ColumnAlteration.md)
|
||||||
- [ColumnOrdering](interfaces/ColumnOrdering.md)
|
- [ColumnOrdering](interfaces/ColumnOrdering.md)
|
||||||
@@ -73,11 +65,9 @@
|
|||||||
- [DropNamespaceOptions](interfaces/DropNamespaceOptions.md)
|
- [DropNamespaceOptions](interfaces/DropNamespaceOptions.md)
|
||||||
- [DropNamespaceResponse](interfaces/DropNamespaceResponse.md)
|
- [DropNamespaceResponse](interfaces/DropNamespaceResponse.md)
|
||||||
- [ExecutableQuery](interfaces/ExecutableQuery.md)
|
- [ExecutableQuery](interfaces/ExecutableQuery.md)
|
||||||
- [FieldMetadataUpdate](interfaces/FieldMetadataUpdate.md)
|
|
||||||
- [FragmentStatistics](interfaces/FragmentStatistics.md)
|
- [FragmentStatistics](interfaces/FragmentStatistics.md)
|
||||||
- [FragmentSummaryStats](interfaces/FragmentSummaryStats.md)
|
- [FragmentSummaryStats](interfaces/FragmentSummaryStats.md)
|
||||||
- [FtsOptions](interfaces/FtsOptions.md)
|
- [FtsOptions](interfaces/FtsOptions.md)
|
||||||
- [FtsToken](interfaces/FtsToken.md)
|
|
||||||
- [FullTextQuery](interfaces/FullTextQuery.md)
|
- [FullTextQuery](interfaces/FullTextQuery.md)
|
||||||
- [FullTextSearchOptions](interfaces/FullTextSearchOptions.md)
|
- [FullTextSearchOptions](interfaces/FullTextSearchOptions.md)
|
||||||
- [HnswPqOptions](interfaces/HnswPqOptions.md)
|
- [HnswPqOptions](interfaces/HnswPqOptions.md)
|
||||||
@@ -91,12 +81,7 @@
|
|||||||
- [ListNamespacesOptions](interfaces/ListNamespacesOptions.md)
|
- [ListNamespacesOptions](interfaces/ListNamespacesOptions.md)
|
||||||
- [ListNamespacesResponse](interfaces/ListNamespacesResponse.md)
|
- [ListNamespacesResponse](interfaces/ListNamespacesResponse.md)
|
||||||
- [LsmWriteSpec](interfaces/LsmWriteSpec.md)
|
- [LsmWriteSpec](interfaces/LsmWriteSpec.md)
|
||||||
- [MergeBlocker](interfaces/MergeBlocker.md)
|
|
||||||
- [MergeBranchResult](interfaces/MergeBranchResult.md)
|
|
||||||
- [MergePreview](interfaces/MergePreview.md)
|
|
||||||
- [MergeResult](interfaces/MergeResult.md)
|
- [MergeResult](interfaces/MergeResult.md)
|
||||||
- [NativeOAuthConfig](interfaces/NativeOAuthConfig.md)
|
|
||||||
- [OAuthConfig](interfaces/OAuthConfig.md)
|
|
||||||
- [OpenTableOptions](interfaces/OpenTableOptions.md)
|
- [OpenTableOptions](interfaces/OpenTableOptions.md)
|
||||||
- [OptimizeOptions](interfaces/OptimizeOptions.md)
|
- [OptimizeOptions](interfaces/OptimizeOptions.md)
|
||||||
- [OptimizeStats](interfaces/OptimizeStats.md)
|
- [OptimizeStats](interfaces/OptimizeStats.md)
|
||||||
@@ -116,8 +101,6 @@
|
|||||||
- [TimeoutConfig](interfaces/TimeoutConfig.md)
|
- [TimeoutConfig](interfaces/TimeoutConfig.md)
|
||||||
- [TlsConfig](interfaces/TlsConfig.md)
|
- [TlsConfig](interfaces/TlsConfig.md)
|
||||||
- [TokenResponse](interfaces/TokenResponse.md)
|
- [TokenResponse](interfaces/TokenResponse.md)
|
||||||
- [TokenizeOptions](interfaces/TokenizeOptions.md)
|
|
||||||
- [UpdateFieldMetadataResult](interfaces/UpdateFieldMetadataResult.md)
|
|
||||||
- [UpdateOptions](interfaces/UpdateOptions.md)
|
- [UpdateOptions](interfaces/UpdateOptions.md)
|
||||||
- [UpdateResult](interfaces/UpdateResult.md)
|
- [UpdateResult](interfaces/UpdateResult.md)
|
||||||
- [Version](interfaces/Version.md)
|
- [Version](interfaces/Version.md)
|
||||||
@@ -126,8 +109,6 @@
|
|||||||
|
|
||||||
## Type Aliases
|
## Type Aliases
|
||||||
|
|
||||||
- [AnalyzePlanDistributedMetrics](type-aliases/AnalyzePlanDistributedMetrics.md)
|
|
||||||
- [BaseTokenizer](type-aliases/BaseTokenizer.md)
|
|
||||||
- [Data](type-aliases/Data.md)
|
- [Data](type-aliases/Data.md)
|
||||||
- [DataLike](type-aliases/DataLike.md)
|
- [DataLike](type-aliases/DataLike.md)
|
||||||
- [FieldLike](type-aliases/FieldLike.md)
|
- [FieldLike](type-aliases/FieldLike.md)
|
||||||
@@ -137,15 +118,12 @@
|
|||||||
- [RecordBatchLike](type-aliases/RecordBatchLike.md)
|
- [RecordBatchLike](type-aliases/RecordBatchLike.md)
|
||||||
- [SchemaLike](type-aliases/SchemaLike.md)
|
- [SchemaLike](type-aliases/SchemaLike.md)
|
||||||
- [TableLike](type-aliases/TableLike.md)
|
- [TableLike](type-aliases/TableLike.md)
|
||||||
- [TokenizeTableOptions](type-aliases/TokenizeTableOptions.md)
|
|
||||||
|
|
||||||
## Functions
|
## Functions
|
||||||
|
|
||||||
- [RecordBatchIterator](functions/RecordBatchIterator.md)
|
- [RecordBatchIterator](functions/RecordBatchIterator.md)
|
||||||
- [connect](functions/connect.md)
|
- [connect](functions/connect.md)
|
||||||
- [connectNamespace](functions/connectNamespace.md)
|
- [connectNamespace](functions/connectNamespace.md)
|
||||||
- [instrumentLanceDbMetrics](functions/instrumentLanceDbMetrics.md)
|
|
||||||
- [makeArrowTable](functions/makeArrowTable.md)
|
- [makeArrowTable](functions/makeArrowTable.md)
|
||||||
- [packBits](functions/packBits.md)
|
- [packBits](functions/packBits.md)
|
||||||
- [permutationBuilder](functions/permutationBuilder.md)
|
- [permutationBuilder](functions/permutationBuilder.md)
|
||||||
- [tokenize](functions/tokenize.md)
|
|
||||||
|
|||||||
@@ -1,33 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / BranchColumnChange
|
|
||||||
|
|
||||||
# Interface: BranchColumnChange
|
|
||||||
|
|
||||||
A column whose definition differs between main and the branch.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### branch
|
|
||||||
|
|
||||||
```ts
|
|
||||||
branch: BranchColumnSummary;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### main
|
|
||||||
|
|
||||||
```ts
|
|
||||||
main: BranchColumnSummary;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### name
|
|
||||||
|
|
||||||
```ts
|
|
||||||
name: string;
|
|
||||||
```
|
|
||||||
@@ -1,33 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / BranchColumnSummary
|
|
||||||
|
|
||||||
# Interface: BranchColumnSummary
|
|
||||||
|
|
||||||
Summary of a column in a branch diff.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### dataType
|
|
||||||
|
|
||||||
```ts
|
|
||||||
dataType: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### name
|
|
||||||
|
|
||||||
```ts
|
|
||||||
name: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### nullable
|
|
||||||
|
|
||||||
```ts
|
|
||||||
nullable: boolean;
|
|
||||||
```
|
|
||||||
@@ -1,129 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / BranchDiff
|
|
||||||
|
|
||||||
# Interface: BranchDiff
|
|
||||||
|
|
||||||
Read-only comparison of a branch against main.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### addedColumns
|
|
||||||
|
|
||||||
```ts
|
|
||||||
addedColumns: BranchColumnSummary[];
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### addedIndexes
|
|
||||||
|
|
||||||
```ts
|
|
||||||
addedIndexes: BranchIndexSummary[];
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### baseMoved
|
|
||||||
|
|
||||||
```ts
|
|
||||||
baseMoved: boolean;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### branchVersion
|
|
||||||
|
|
||||||
```ts
|
|
||||||
branchVersion: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### changedColumns
|
|
||||||
|
|
||||||
```ts
|
|
||||||
changedColumns: BranchColumnChange[];
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### fromBranch
|
|
||||||
|
|
||||||
```ts
|
|
||||||
fromBranch: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### mainVersion
|
|
||||||
|
|
||||||
```ts
|
|
||||||
mainVersion: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### mergeBlockers
|
|
||||||
|
|
||||||
```ts
|
|
||||||
mergeBlockers: MergeBlocker[];
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### mergeable
|
|
||||||
|
|
||||||
```ts
|
|
||||||
mergeable: boolean;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### parentVersion
|
|
||||||
|
|
||||||
```ts
|
|
||||||
parentVersion: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### removedColumns
|
|
||||||
|
|
||||||
```ts
|
|
||||||
removedColumns: BranchColumnSummary[];
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### removedIndexes
|
|
||||||
|
|
||||||
```ts
|
|
||||||
removedIndexes: BranchIndexSummary[];
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### rowCountBranch
|
|
||||||
|
|
||||||
```ts
|
|
||||||
rowCountBranch: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### rowCountMain
|
|
||||||
|
|
||||||
```ts
|
|
||||||
rowCountMain: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### rowSummary
|
|
||||||
|
|
||||||
```ts
|
|
||||||
rowSummary: BranchRowCountSummary;
|
|
||||||
```
|
|
||||||
@@ -1,41 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / BranchIndexSummary
|
|
||||||
|
|
||||||
# Interface: BranchIndexSummary
|
|
||||||
|
|
||||||
Summary of an index in a branch diff.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### columns
|
|
||||||
|
|
||||||
```ts
|
|
||||||
columns: string[];
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### indexName
|
|
||||||
|
|
||||||
```ts
|
|
||||||
indexName: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### indexType?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional indexType: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### status
|
|
||||||
|
|
||||||
```ts
|
|
||||||
status: string;
|
|
||||||
```
|
|
||||||
@@ -1,57 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / BranchRowCountSummary
|
|
||||||
|
|
||||||
# Interface: BranchRowCountSummary
|
|
||||||
|
|
||||||
Row-level comparison between main and the branch.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### deltaAvailable
|
|
||||||
|
|
||||||
```ts
|
|
||||||
deltaAvailable: boolean;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### inputsChanged
|
|
||||||
|
|
||||||
```ts
|
|
||||||
inputsChanged: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### newOnBase
|
|
||||||
|
|
||||||
```ts
|
|
||||||
newOnBase: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### newOnBranch
|
|
||||||
|
|
||||||
```ts
|
|
||||||
newOnBranch: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### staleRecompute
|
|
||||||
|
|
||||||
```ts
|
|
||||||
staleRecompute: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### unchanged
|
|
||||||
|
|
||||||
```ts
|
|
||||||
unchanged: number;
|
|
||||||
```
|
|
||||||
@@ -64,19 +64,6 @@ client used by manifest-enabled native connections.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### oauthConfig?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional oauthConfig: NativeOAuthConfig;
|
|
||||||
```
|
|
||||||
|
|
||||||
(For LanceDB cloud only): OAuth configuration for IdP-based
|
|
||||||
authentication (e.g., Azure Entra ID). When set, token acquisition
|
|
||||||
and refresh are handled entirely in Rust. TypeScript users should pass
|
|
||||||
the public `OAuthConfig` type exported from `@lancedb/lancedb`.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### readConsistencyInterval?
|
### readConsistencyInterval?
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
|
|||||||
@@ -1,41 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / FieldMetadataUpdate
|
|
||||||
|
|
||||||
# Interface: FieldMetadataUpdate
|
|
||||||
|
|
||||||
A per-field metadata update, addressed by dot-path.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### metadata
|
|
||||||
|
|
||||||
```ts
|
|
||||||
metadata: Record<string, null | string>;
|
|
||||||
```
|
|
||||||
|
|
||||||
Metadata key/value pairs. Merged into the field's existing metadata by
|
|
||||||
default; a value of `null` deletes that key.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### path
|
|
||||||
|
|
||||||
```ts
|
|
||||||
path: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Dot-separated path to the field. For a top-level column this is just its
|
|
||||||
name; for a nested field it's the path, e.g. "a.b.c".
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### replace?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional replace: boolean;
|
|
||||||
```
|
|
||||||
|
|
||||||
If true, replace the field's entire metadata map instead of merging.
|
|
||||||
@@ -23,7 +23,7 @@ whether to remove punctuation
|
|||||||
### baseTokenizer?
|
### baseTokenizer?
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
optional baseTokenizer: BaseTokenizer;
|
optional baseTokenizer: "raw" | "simple" | "whitespace" | "ngram";
|
||||||
```
|
```
|
||||||
|
|
||||||
The tokenizer to use when building the index.
|
The tokenizer to use when building the index.
|
||||||
@@ -37,38 +37,6 @@ The following tokenizers are available:
|
|||||||
|
|
||||||
"raw" - Raw tokenizer. This tokenizer does not split the text into tokens and indexes the entire text as a single token.
|
"raw" - Raw tokenizer. This tokenizer does not split the text into tokens and indexes the entire text as a single token.
|
||||||
|
|
||||||
"icu" - ICU dictionary-based word segmentation.
|
|
||||||
|
|
||||||
"icu/split" - ICU segmentation with simple-style delimiter splitting.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### blockSize?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional blockSize: 128 | 256;
|
|
||||||
```
|
|
||||||
|
|
||||||
Number of documents per compressed posting block.
|
|
||||||
|
|
||||||
The default is 128. Supported values are 128 and 256. A value of 256 uses
|
|
||||||
the experimental FTS V3 format and may introduce breaking changes.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### customStopWords?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional customStopWords: string[];
|
|
||||||
```
|
|
||||||
|
|
||||||
Custom stop words that replace the built-in list for `language`.
|
|
||||||
|
|
||||||
This option only affects tokenization when `removeStopWords` is true.
|
|
||||||
|
|
||||||
`undefined` keeps the built-in language list. An empty array explicitly
|
|
||||||
replaces it with no stop words.
|
|
||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### language?
|
### language?
|
||||||
|
|||||||
@@ -1,29 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / FtsToken
|
|
||||||
|
|
||||||
# Interface: FtsToken
|
|
||||||
|
|
||||||
Token produced by the tokenizer configured on a full-text search index.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### position
|
|
||||||
|
|
||||||
```ts
|
|
||||||
position: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
Token position used by full-text query matching.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### text
|
|
||||||
|
|
||||||
```ts
|
|
||||||
text: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Token text after tokenizer filters have been applied.
|
|
||||||
@@ -23,31 +23,6 @@ be more columns to represent composite indices.
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### createdAt?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional createdAt: Date;
|
|
||||||
```
|
|
||||||
|
|
||||||
When the index was created.
|
|
||||||
|
|
||||||
`undefined` for remote tables or indices created before timestamps were tracked.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### indexDetails?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional indexDetails: any;
|
|
||||||
```
|
|
||||||
|
|
||||||
Index-type-specific details parsed as a JavaScript object.
|
|
||||||
|
|
||||||
Falls back to a raw string if JSON parsing fails. `undefined` for
|
|
||||||
remote tables or when details are unavailable.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### indexType
|
### indexType
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -58,30 +33,6 @@ The type of the index
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### indexUuid?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional indexUuid: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
The UUID of the first segment of the index.
|
|
||||||
|
|
||||||
`undefined` for remote tables, which do not yet surface this.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### indexVersion?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional indexVersion: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
The on-disk index format version.
|
|
||||||
|
|
||||||
`undefined` for remote tables.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### name
|
### name
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -89,63 +40,3 @@ name: string;
|
|||||||
```
|
```
|
||||||
|
|
||||||
The name of the index
|
The name of the index
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### numIndexedRows?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional numIndexedRows: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
The number of rows indexed, across all segments.
|
|
||||||
|
|
||||||
`undefined` for remote tables.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### numSegments?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional numSegments: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
The number of segments that make up the index.
|
|
||||||
|
|
||||||
`undefined` for remote tables.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### numUnindexedRows?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional numUnindexedRows: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
The number of rows not yet covered by this index.
|
|
||||||
|
|
||||||
`undefined` for remote tables.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### sizeBytes?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional sizeBytes: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
The total size in bytes of all index files across all segments.
|
|
||||||
|
|
||||||
`undefined` for remote tables or indices without size tracking.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### typeUrl?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional typeUrl: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
The protobuf type URL, a precise type identifier for the index.
|
|
||||||
|
|
||||||
`undefined` for remote tables.
|
|
||||||
|
|||||||
@@ -30,6 +30,17 @@ The type of the index
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
### loss?
|
||||||
|
|
||||||
|
```ts
|
||||||
|
optional loss: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
The KMeans loss value of the index,
|
||||||
|
it is only present for vector indices.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### numIndexedRows
|
### numIndexedRows
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
|
|||||||
@@ -11,10 +11,7 @@ Specification selecting Lance's MemWAL LSM-style write path for
|
|||||||
|
|
||||||
`specType` is `"bucket"`, `"identity"`, or `"unsharded"`. For `"bucket"`,
|
`specType` is `"bucket"`, `"identity"`, or `"unsharded"`. For `"bucket"`,
|
||||||
`column` and `numBuckets` are required; for `"identity"`, `column` is
|
`column` and `numBuckets` are required; for `"identity"`, `column` is
|
||||||
required and must be a deterministic function of the unenforced primary
|
required.
|
||||||
key (every row with a given primary key must always produce the same
|
|
||||||
`column` value, or upserts of that key can land in different shards and a
|
|
||||||
stale version can win).
|
|
||||||
|
|
||||||
## Properties
|
## Properties
|
||||||
|
|
||||||
|
|||||||
@@ -1,25 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / MergeBlocker
|
|
||||||
|
|
||||||
# Interface: MergeBlocker
|
|
||||||
|
|
||||||
A reason why a branch cannot currently be merged.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### code
|
|
||||||
|
|
||||||
```ts
|
|
||||||
code: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### message
|
|
||||||
|
|
||||||
```ts
|
|
||||||
message: string;
|
|
||||||
```
|
|
||||||
@@ -1,46 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / MergeBranchResult
|
|
||||||
|
|
||||||
# Interface: MergeBranchResult
|
|
||||||
|
|
||||||
Result of previewing or attempting a branch merge.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### diff
|
|
||||||
|
|
||||||
```ts
|
|
||||||
diff: BranchDiff;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### mainVersionAfter?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional mainVersionAfter: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### preview
|
|
||||||
|
|
||||||
```ts
|
|
||||||
preview: MergePreview;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### status
|
|
||||||
|
|
||||||
```ts
|
|
||||||
status:
|
|
||||||
| "unknown"
|
|
||||||
| "rejected"
|
|
||||||
| "ready"
|
|
||||||
| "notImplemented"
|
|
||||||
| "merged";
|
|
||||||
```
|
|
||||||
@@ -1,17 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / MergePreview
|
|
||||||
|
|
||||||
# Interface: MergePreview
|
|
||||||
|
|
||||||
Changes that would be, or were, promoted by a branch merge.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### promotedColumns
|
|
||||||
|
|
||||||
```ts
|
|
||||||
promotedColumns: string[];
|
|
||||||
```
|
|
||||||
@@ -32,14 +32,6 @@ numInsertedRows: number;
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### numRows
|
|
||||||
|
|
||||||
```ts
|
|
||||||
numRows: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### numUpdatedRows
|
### numUpdatedRows
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
|
|||||||
@@ -1,88 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / NativeOAuthConfig
|
|
||||||
|
|
||||||
# Interface: NativeOAuthConfig
|
|
||||||
|
|
||||||
OAuth configuration for LanceDB authentication.
|
|
||||||
|
|
||||||
This is the generated napi-rs binding shape. TypeScript users should prefer
|
|
||||||
the public `OAuthConfig` type exported from `@lancedb/lancedb`.
|
|
||||||
|
|
||||||
All token acquisition and refresh is handled in the Rust layer.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### clientId
|
|
||||||
|
|
||||||
```ts
|
|
||||||
clientId: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Application / Client ID.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### clientSecret?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional clientSecret: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Client secret (required for client_credentials).
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### flow?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional flow: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Authentication flow: "client_credentials" or "azure_managed_identity"
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### issuerUrl
|
|
||||||
|
|
||||||
```ts
|
|
||||||
issuerUrl: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
OIDC issuer URL or OAuth authority URL.
|
|
||||||
For Azure: `https://login.microsoftonline.com/{tenant_id}/v2.0`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### managedIdentityClientId?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional managedIdentityClientId: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Client ID for user-assigned managed identity (azure_managed_identity).
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### refreshBufferSecs?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional refreshBufferSecs: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
Seconds before expiry to trigger proactive refresh (default: 300).
|
|
||||||
Keep this well below the token TTL; if it is greater than or equal to
|
|
||||||
the TTL, each request refreshes the token.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### scopes
|
|
||||||
|
|
||||||
```ts
|
|
||||||
scopes: string[];
|
|
||||||
```
|
|
||||||
|
|
||||||
OAuth scopes to request. For Azure managed identity, exactly one scope
|
|
||||||
or resource is required. For example: `["api://{app_id}/.default"]`
|
|
||||||
@@ -1,111 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / OAuthConfig
|
|
||||||
|
|
||||||
# Interface: OAuthConfig
|
|
||||||
|
|
||||||
OAuth configuration for LanceDB authentication.
|
|
||||||
|
|
||||||
This is the public TypeScript OAuth configuration type. The generated
|
|
||||||
`NativeOAuthConfig` type has the same runtime shape but is an implementation
|
|
||||||
detail of the napi-rs binding.
|
|
||||||
|
|
||||||
All token acquisition and refresh is handled in the Rust layer.
|
|
||||||
This config is passed through to Rust via napi-rs.
|
|
||||||
|
|
||||||
## Examples
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
const config: OAuthConfig = {
|
|
||||||
issuerUrl: "https://login.microsoftonline.com/{tenant}/v2.0",
|
|
||||||
clientId: "app-id",
|
|
||||||
clientSecret: "secret",
|
|
||||||
scopes: ["api://lancedb-api/.default"],
|
|
||||||
};
|
|
||||||
```
|
|
||||||
|
|
||||||
```typescript
|
|
||||||
const config: OAuthConfig = {
|
|
||||||
issuerUrl: "https://login.microsoftonline.com/{tenant}/v2.0",
|
|
||||||
clientId: "app-id",
|
|
||||||
scopes: ["api://lancedb-api/.default"],
|
|
||||||
flow: OAuthFlowType.AzureManagedIdentity,
|
|
||||||
};
|
|
||||||
```
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### clientId
|
|
||||||
|
|
||||||
```ts
|
|
||||||
clientId: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Application / Client ID.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### clientSecret?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional clientSecret: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Client secret (required for ClientCredentials).
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### flow?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional flow: OAuthFlowType;
|
|
||||||
```
|
|
||||||
|
|
||||||
Authentication flow (default: ClientCredentials).
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### issuerUrl
|
|
||||||
|
|
||||||
```ts
|
|
||||||
issuerUrl: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
OIDC issuer URL or OAuth authority URL.
|
|
||||||
For Azure: `https://login.microsoftonline.com/{tenant_id}/v2.0`
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### managedIdentityClientId?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional managedIdentityClientId: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Client ID for user-assigned managed identity (AzureManagedIdentity).
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### refreshBufferSecs?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional refreshBufferSecs: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
Seconds before expiry to trigger proactive refresh (default: 300).
|
|
||||||
Keep this well below the token TTL; if it is greater than or equal to
|
|
||||||
the TTL, each request refreshes the token.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### scopes
|
|
||||||
|
|
||||||
```ts
|
|
||||||
scopes: string[];
|
|
||||||
```
|
|
||||||
|
|
||||||
OAuth scopes to request.
|
|
||||||
For Azure managed identity, exactly one scope or resource is required.
|
|
||||||
For example: `["api://{app_id}/.default"]`
|
|
||||||
@@ -8,18 +8,6 @@
|
|||||||
|
|
||||||
## Properties
|
## Properties
|
||||||
|
|
||||||
### branch?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional branch: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Open the table scoped to this branch instead of the default branch.
|
|
||||||
|
|
||||||
Reads and writes on the returned table operate in the branch's context.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### ~~indexCacheSize?~~
|
### ~~indexCacheSize?~~
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
@@ -55,17 +43,3 @@ Options already set on the connection will be inherited by the table,
|
|||||||
but can be overridden here.
|
but can be overridden here.
|
||||||
|
|
||||||
The available options are described at https://docs.lancedb.com/storage/
|
The available options are described at https://docs.lancedb.com/storage/
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### version?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional version: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
Open the table pinned to this version, producing a read-only view.
|
|
||||||
|
|
||||||
Composes with [OpenTableOptions.branch](OpenTableOptions.md#branch): when both are set, opens
|
|
||||||
that branch at the version; otherwise opens `main` at the version. Call
|
|
||||||
`checkoutLatest` to return to a writable state.
|
|
||||||
|
|||||||
@@ -8,14 +8,6 @@
|
|||||||
|
|
||||||
## Properties
|
## Properties
|
||||||
|
|
||||||
### clumpSize?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional clumpSize: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### counts?
|
### counts?
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
|
|||||||
@@ -1,124 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / TokenizeOptions
|
|
||||||
|
|
||||||
# Interface: TokenizeOptions
|
|
||||||
|
|
||||||
Options for tokenizing a full-text search query without a table index.
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### asciiFolding?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional asciiFolding: boolean;
|
|
||||||
```
|
|
||||||
|
|
||||||
Whether to fold ASCII characters.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### baseTokenizer?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional baseTokenizer: BaseTokenizer;
|
|
||||||
```
|
|
||||||
|
|
||||||
The tokenizer to use. The default is "simple".
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### customStopWords?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional customStopWords: string[];
|
|
||||||
```
|
|
||||||
|
|
||||||
Custom stop words that replace the built-in list for `language`.
|
|
||||||
|
|
||||||
This option only affects tokenization when `removeStopWords` is true.
|
|
||||||
|
|
||||||
`undefined` keeps the built-in language list. An empty array explicitly
|
|
||||||
replaces it with no stop words.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### language?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional language: string;
|
|
||||||
```
|
|
||||||
|
|
||||||
Language for stemming and stop words.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### lowercase?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional lowercase: boolean;
|
|
||||||
```
|
|
||||||
|
|
||||||
Whether to lowercase tokens.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### maxTokenLength?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional maxTokenLength: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
Maximum token length; tokens longer than this are ignored.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### ngramMaxLength?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional ngramMaxLength: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
N-gram maximum length.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### ngramMinLength?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional ngramMinLength: number;
|
|
||||||
```
|
|
||||||
|
|
||||||
N-gram minimum length.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### prefixOnly?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional prefixOnly: boolean;
|
|
||||||
```
|
|
||||||
|
|
||||||
Whether to only emit token prefixes for the n-gram tokenizer.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### removeStopWords?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional removeStopWords: boolean;
|
|
||||||
```
|
|
||||||
|
|
||||||
Whether to remove stop words.
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
### stem?
|
|
||||||
|
|
||||||
```ts
|
|
||||||
optional stem: boolean;
|
|
||||||
```
|
|
||||||
|
|
||||||
Whether to stem tokens.
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / UpdateFieldMetadataResult
|
|
||||||
|
|
||||||
# Interface: UpdateFieldMetadataResult
|
|
||||||
|
|
||||||
## Properties
|
|
||||||
|
|
||||||
### version
|
|
||||||
|
|
||||||
```ts
|
|
||||||
version: number;
|
|
||||||
```
|
|
||||||
@@ -1,11 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / AnalyzePlanDistributedMetrics
|
|
||||||
|
|
||||||
# Type Alias: AnalyzePlanDistributedMetrics
|
|
||||||
|
|
||||||
```ts
|
|
||||||
type AnalyzePlanDistributedMetrics: "aggregate" | "per_worker" | "full";
|
|
||||||
```
|
|
||||||
@@ -1,19 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / BaseTokenizer
|
|
||||||
|
|
||||||
# Type Alias: BaseTokenizer
|
|
||||||
|
|
||||||
```ts
|
|
||||||
type BaseTokenizer:
|
|
||||||
| "simple"
|
|
||||||
| "whitespace"
|
|
||||||
| "raw"
|
|
||||||
| "ngram"
|
|
||||||
| "icu"
|
|
||||||
| "icu/split"
|
|
||||||
| `jieba/${string}`
|
|
||||||
| `lindera/${string}`;
|
|
||||||
```
|
|
||||||
@@ -1,11 +0,0 @@
|
|||||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
|
||||||
|
|
||||||
***
|
|
||||||
|
|
||||||
[@lancedb/lancedb](../globals.md) / TokenizeTableOptions
|
|
||||||
|
|
||||||
# Type Alias: TokenizeTableOptions
|
|
||||||
|
|
||||||
```ts
|
|
||||||
type TokenizeTableOptions: object | object;
|
|
||||||
```
|
|
||||||
+52
-141
@@ -26,18 +26,6 @@ is also an [asynchronous API client](#connections-asynchronous).
|
|||||||
|
|
||||||
::: lancedb.db.DBConnection
|
::: lancedb.db.DBConnection
|
||||||
|
|
||||||
::: lancedb.Session
|
|
||||||
|
|
||||||
## Namespaces (Synchronous)
|
|
||||||
|
|
||||||
A namespace-backed connection resolves tables through a
|
|
||||||
[Lance namespace](https://lancedb.github.io/lance-namespace/) service instead of
|
|
||||||
listing a storage directory.
|
|
||||||
|
|
||||||
::: lancedb.connect_namespace
|
|
||||||
|
|
||||||
::: lancedb.namespace.LanceNamespaceDBConnection
|
|
||||||
|
|
||||||
## Tables (Synchronous)
|
## Tables (Synchronous)
|
||||||
|
|
||||||
::: lancedb.table.Table
|
::: lancedb.table.Table
|
||||||
@@ -46,12 +34,8 @@ listing a storage directory.
|
|||||||
|
|
||||||
::: lancedb.table.FragmentSummaryStats
|
::: lancedb.table.FragmentSummaryStats
|
||||||
|
|
||||||
::: lancedb.table.TableStatistics
|
|
||||||
|
|
||||||
::: lancedb.table.Tags
|
::: lancedb.table.Tags
|
||||||
|
|
||||||
::: lancedb.table.Branches
|
|
||||||
|
|
||||||
## Expressions
|
## Expressions
|
||||||
|
|
||||||
Type-safe expression builder for filters and projections. Use these instead
|
Type-safe expression builder for filters and projections. Use these instead
|
||||||
@@ -78,46 +62,29 @@ of raw SQL strings with [where][lancedb.query.LanceQueryBuilder.where] and
|
|||||||
|
|
||||||
::: lancedb.query.LanceHybridQueryBuilder
|
::: lancedb.query.LanceHybridQueryBuilder
|
||||||
|
|
||||||
::: lancedb.query.LanceEmptyQueryBuilder
|
|
||||||
|
|
||||||
::: lancedb.query.LanceTakeQueryBuilder
|
|
||||||
|
|
||||||
## Full text queries
|
|
||||||
|
|
||||||
Structured full text queries can be passed to
|
|
||||||
[Table.search][lancedb.table.Table.search] or
|
|
||||||
[AsyncTable.search][lancedb.table.AsyncTable.search] in place of a query string,
|
|
||||||
and combined with [BooleanQuery][lancedb.query.BooleanQuery].
|
|
||||||
|
|
||||||
::: lancedb.query.FullTextQuery
|
|
||||||
|
|
||||||
::: lancedb.query.MatchQuery
|
|
||||||
|
|
||||||
::: lancedb.query.PhraseQuery
|
|
||||||
|
|
||||||
::: lancedb.query.BoostQuery
|
|
||||||
|
|
||||||
::: lancedb.query.MultiMatchQuery
|
|
||||||
|
|
||||||
::: lancedb.query.BooleanQuery
|
|
||||||
|
|
||||||
::: lancedb.query.FullTextOperator
|
|
||||||
|
|
||||||
::: lancedb.query.Occur
|
|
||||||
|
|
||||||
## Embeddings
|
## Embeddings
|
||||||
|
|
||||||
::: lancedb.embeddings
|
::: lancedb.embeddings.registry.EmbeddingFunctionRegistry
|
||||||
options:
|
|
||||||
show_root_heading: false
|
::: lancedb.embeddings.base.EmbeddingFunctionConfig
|
||||||
show_root_toc_entry: false
|
|
||||||
|
::: lancedb.embeddings.base.EmbeddingFunction
|
||||||
|
|
||||||
|
::: lancedb.embeddings.base.TextEmbeddingFunction
|
||||||
|
|
||||||
|
::: lancedb.embeddings.sentence_transformers.SentenceTransformerEmbeddings
|
||||||
|
|
||||||
|
::: lancedb.embeddings.openai.OpenAIEmbeddings
|
||||||
|
|
||||||
|
::: lancedb.embeddings.open_clip.OpenClipEmbeddings
|
||||||
|
|
||||||
## Remote configuration
|
## Remote configuration
|
||||||
|
|
||||||
::: lancedb.remote
|
::: lancedb.remote.ClientConfig
|
||||||
options:
|
|
||||||
show_root_heading: false
|
::: lancedb.remote.TimeoutConfig
|
||||||
show_root_toc_entry: false
|
|
||||||
|
::: lancedb.remote.RetryConfig
|
||||||
|
|
||||||
## Context
|
## Context
|
||||||
|
|
||||||
@@ -127,50 +94,11 @@ and combined with [BooleanQuery][lancedb.query.BooleanQuery].
|
|||||||
|
|
||||||
## Full text search
|
## Full text search
|
||||||
|
|
||||||
Pass `custom_stop_words` to [lancedb.index.FTS][]:
|
Use [lancedb.table.Table.create_fts_index][] for the synchronous API or
|
||||||
|
[lancedb.table.AsyncTable.create_index][] with [lancedb.index.FTS][] for the
|
||||||
|
asynchronous API.
|
||||||
|
|
||||||
```python
|
::: lancedb.index.FTS
|
||||||
from lancedb.index import FTS
|
|
||||||
|
|
||||||
table.create_index(
|
|
||||||
"text",
|
|
||||||
config=FTS(remove_stop_words=True, custom_stop_words=["acme", "internal"]),
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
The list replaces the built-in stop words and is used only when
|
|
||||||
`remove_stop_words=True`:
|
|
||||||
|
|
||||||
- `custom_stop_words=None` uses the built-in list for `language`.
|
|
||||||
- `custom_stop_words=[]` removes no words.
|
|
||||||
- Values are passed through without trimming, lowercasing, or other rewriting.
|
|
||||||
|
|
||||||
The same option is available on `lancedb.tokenize(...)` and the deprecated
|
|
||||||
[lancedb.table.Table.create_fts_index][] compatibility helper:
|
|
||||||
|
|
||||||
```python
|
|
||||||
import lancedb
|
|
||||||
|
|
||||||
tokens = list(lancedb.tokenize("acme makes searchable data",
|
|
||||||
custom_stop_words=["acme"]))
|
|
||||||
```
|
|
||||||
|
|
||||||
::: lancedb.tokenize
|
|
||||||
|
|
||||||
::: lancedb.FtsToken
|
|
||||||
|
|
||||||
## Blobs
|
|
||||||
|
|
||||||
Blob columns store large binary values out of line so they can be read lazily
|
|
||||||
instead of being materialized with the rest of the row.
|
|
||||||
|
|
||||||
::: lancedb.blob
|
|
||||||
|
|
||||||
::: lancedb.BlobType
|
|
||||||
|
|
||||||
::: lancedb._blob.BlobFile
|
|
||||||
options:
|
|
||||||
show_root_full_path: false
|
|
||||||
|
|
||||||
## Utilities
|
## Utilities
|
||||||
|
|
||||||
@@ -178,14 +106,6 @@ instead of being materialized with the rest of the row.
|
|||||||
|
|
||||||
::: lancedb.merge.LanceMergeInsertBuilder
|
::: lancedb.merge.LanceMergeInsertBuilder
|
||||||
|
|
||||||
::: lancedb.otel.instrument_lancedb_metrics
|
|
||||||
|
|
||||||
## Exceptions
|
|
||||||
|
|
||||||
::: lancedb.exceptions.MissingValueError
|
|
||||||
|
|
||||||
::: lancedb.exceptions.MissingColumnError
|
|
||||||
|
|
||||||
## Integrations
|
## Integrations
|
||||||
|
|
||||||
## Pydantic
|
## Pydantic
|
||||||
@@ -194,30 +114,19 @@ instead of being materialized with the rest of the row.
|
|||||||
|
|
||||||
::: lancedb.pydantic.vector
|
::: lancedb.pydantic.vector
|
||||||
|
|
||||||
::: lancedb.pydantic.Vector
|
|
||||||
|
|
||||||
::: lancedb.pydantic.MultiVector
|
|
||||||
|
|
||||||
::: lancedb.pydantic.LanceModel
|
::: lancedb.pydantic.LanceModel
|
||||||
|
|
||||||
## PyTorch
|
|
||||||
|
|
||||||
::: lancedb.streaming.StreamingDataset
|
|
||||||
|
|
||||||
::: lancedb.permutation.permutation_builder
|
|
||||||
|
|
||||||
::: lancedb.permutation.PermutationBuilder
|
|
||||||
|
|
||||||
::: lancedb.permutation.Permutation
|
|
||||||
|
|
||||||
::: lancedb.permutation.Transforms
|
|
||||||
|
|
||||||
## Reranking
|
## Reranking
|
||||||
|
|
||||||
::: lancedb.rerankers
|
::: lancedb.rerankers.linear_combination.LinearCombinationReranker
|
||||||
options:
|
|
||||||
show_root_heading: false
|
::: lancedb.rerankers.cohere.CohereReranker
|
||||||
show_root_toc_entry: false
|
|
||||||
|
::: lancedb.rerankers.colbert.ColbertReranker
|
||||||
|
|
||||||
|
::: lancedb.rerankers.cross_encoder.CrossEncoderReranker
|
||||||
|
|
||||||
|
::: lancedb.rerankers.openai.OpenaiReranker
|
||||||
|
|
||||||
## Connections (Asynchronous)
|
## Connections (Asynchronous)
|
||||||
|
|
||||||
@@ -228,12 +137,6 @@ can be used to create, list, or open tables.
|
|||||||
|
|
||||||
::: lancedb.db.AsyncConnection
|
::: lancedb.db.AsyncConnection
|
||||||
|
|
||||||
## Namespaces (Asynchronous)
|
|
||||||
|
|
||||||
::: lancedb.connect_namespace_async
|
|
||||||
|
|
||||||
::: lancedb.namespace.AsyncLanceNamespaceDBConnection
|
|
||||||
|
|
||||||
## Tables (Asynchronous)
|
## Tables (Asynchronous)
|
||||||
|
|
||||||
Table hold your actual data as a collection of records / rows.
|
Table hold your actual data as a collection of records / rows.
|
||||||
@@ -242,20 +145,32 @@ Table hold your actual data as a collection of records / rows.
|
|||||||
|
|
||||||
::: lancedb.table.AsyncTags
|
::: lancedb.table.AsyncTags
|
||||||
|
|
||||||
::: lancedb.table.AsyncBranches
|
|
||||||
|
|
||||||
## Indices (Asynchronous)
|
## Indices (Asynchronous)
|
||||||
|
|
||||||
Indices can be created on a table to speed up queries. This section
|
Indices can be created on a table to speed up queries. This section
|
||||||
lists the indices that LanceDb supports.
|
lists the indices that LanceDb supports.
|
||||||
|
|
||||||
::: lancedb.index
|
::: lancedb.index.BTree
|
||||||
options:
|
|
||||||
show_root_heading: false
|
::: lancedb.index.Bitmap
|
||||||
show_root_toc_entry: false
|
|
||||||
# `lang_mapping` is defined in the module rather than imported, so it is
|
::: lancedb.index.LabelList
|
||||||
# picked up despite not being in `__all__`. It is an internal lookup table.
|
|
||||||
filters: ["!^_", "!^lang_mapping$"]
|
::: lancedb.index.FTS
|
||||||
|
|
||||||
|
::: lancedb.index.IvfPq
|
||||||
|
|
||||||
|
::: lancedb.index.HnswPq
|
||||||
|
|
||||||
|
::: lancedb.index.HnswSq
|
||||||
|
|
||||||
|
::: lancedb.index.IvfFlat
|
||||||
|
|
||||||
|
::: lancedb.index.IvfSq
|
||||||
|
|
||||||
|
::: lancedb.index.IvfRq
|
||||||
|
|
||||||
|
::: lancedb.index.HnswFlat
|
||||||
|
|
||||||
::: lancedb.table.IndexStatistics
|
::: lancedb.table.IndexStatistics
|
||||||
|
|
||||||
@@ -283,7 +198,3 @@ rows nearest to a query vector and can be created with the
|
|||||||
::: lancedb.query.AsyncHybridQuery
|
::: lancedb.query.AsyncHybridQuery
|
||||||
options:
|
options:
|
||||||
inherited_members: true
|
inherited_members: true
|
||||||
|
|
||||||
::: lancedb.query.AsyncTakeQuery
|
|
||||||
options:
|
|
||||||
inherited_members: true
|
|
||||||
|
|||||||
@@ -8,7 +8,7 @@
|
|||||||
<parent>
|
<parent>
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-parent</artifactId>
|
<artifactId>lancedb-parent</artifactId>
|
||||||
<version>0.37.1-beta.0</version>
|
<version>0.30.0-final.0</version>
|
||||||
<relativePath>../pom.xml</relativePath>
|
<relativePath>../pom.xml</relativePath>
|
||||||
</parent>
|
</parent>
|
||||||
|
|
||||||
|
|||||||
+2
-2
@@ -6,7 +6,7 @@
|
|||||||
|
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-parent</artifactId>
|
<artifactId>lancedb-parent</artifactId>
|
||||||
<version>0.37.1-beta.0</version>
|
<version>0.30.0-final.0</version>
|
||||||
<packaging>pom</packaging>
|
<packaging>pom</packaging>
|
||||||
<name>${project.artifactId}</name>
|
<name>${project.artifactId}</name>
|
||||||
<description>LanceDB Java SDK Parent POM</description>
|
<description>LanceDB Java SDK Parent POM</description>
|
||||||
@@ -28,7 +28,7 @@
|
|||||||
<properties>
|
<properties>
|
||||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||||
<arrow.version>15.0.0</arrow.version>
|
<arrow.version>15.0.0</arrow.version>
|
||||||
<lance-core.version>10.0.0-beta.5</lance-core.version>
|
<lance-core.version>7.0.0</lance-core.version>
|
||||||
<spotless.skip>false</spotless.skip>
|
<spotless.skip>false</spotless.skip>
|
||||||
<spotless.version>2.30.0</spotless.version>
|
<spotless.version>2.30.0</spotless.version>
|
||||||
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
|
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# Contributing to LanceDB Typescript
|
# Contributing to LanceDB Typescript
|
||||||
|
|
||||||
This document outlines the process for contributing to LanceDB Typescript.
|
This document outlines the process for contributing to LanceDB Typescript.
|
||||||
For general contribution guidelines, see [CONTRIBUTING.md](https://github.com/lancedb/lancedb/blob/main/CONTRIBUTING.md).
|
For general contribution guidelines, see [CONTRIBUTING.md](../CONTRIBUTING.md).
|
||||||
|
|
||||||
## Project layout
|
## Project layout
|
||||||
|
|
||||||
|
|||||||
+3
-7
@@ -1,7 +1,7 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "lancedb-nodejs"
|
name = "lancedb-nodejs"
|
||||||
edition.workspace = true
|
edition.workspace = true
|
||||||
version = "0.37.1-beta.0"
|
version = "0.30.0"
|
||||||
publish = false
|
publish = false
|
||||||
license.workspace = true
|
license.workspace = true
|
||||||
description.workspace = true
|
description.workspace = true
|
||||||
@@ -25,12 +25,8 @@ lancedb = { path = "../rust/lancedb", default-features = false }
|
|||||||
lance-namespace.workspace = true
|
lance-namespace.workspace = true
|
||||||
napi = { version = "3.8.3", default-features = false, features = [
|
napi = { version = "3.8.3", default-features = false, features = [
|
||||||
"napi9",
|
"napi9",
|
||||||
"async",
|
"async"
|
||||||
"chrono_date",
|
|
||||||
"serde-json",
|
|
||||||
] }
|
] }
|
||||||
chrono = { version = "0.4", default-features = false, features = ["clock"] }
|
|
||||||
serde_json = "1"
|
|
||||||
napi-derive = "3.5.2"
|
napi-derive = "3.5.2"
|
||||||
# Prevent dynamic linking of lzma, which comes from datafusion
|
# Prevent dynamic linking of lzma, which comes from datafusion
|
||||||
lzma-sys = { version = "0.1", features = ["static"] }
|
lzma-sys = { version = "0.1", features = ["static"] }
|
||||||
@@ -44,6 +40,6 @@ aws-lc-rs = "=1.16.3"
|
|||||||
napi-build = "2.3.1"
|
napi-build = "2.3.1"
|
||||||
|
|
||||||
[features]
|
[features]
|
||||||
default = ["remote", "lancedb/aws", "lancedb/gcs", "lancedb/azure", "lancedb/dynamodb", "lancedb/oss", "lancedb/huggingface", "lancedb/goosefs", "lancedb/metrics-otel"]
|
default = ["remote", "lancedb/aws", "lancedb/gcs", "lancedb/azure", "lancedb/dynamodb", "lancedb/oss", "lancedb/huggingface"]
|
||||||
fp16kernels = ["lancedb/fp16kernels"]
|
fp16kernels = ["lancedb/fp16kernels"]
|
||||||
remote = ["lancedb/remote"]
|
remote = ["lancedb/remote"]
|
||||||
|
|||||||
@@ -52,7 +52,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
Float64,
|
Float64,
|
||||||
Struct,
|
Struct,
|
||||||
List,
|
List,
|
||||||
Map_,
|
|
||||||
Int16,
|
Int16,
|
||||||
Int32,
|
Int32,
|
||||||
Int64,
|
Int64,
|
||||||
@@ -70,30 +69,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
type Schema = ApacheArrow["Schema"];
|
type Schema = ApacheArrow["Schema"];
|
||||||
type Table = ApacheArrow["Table"];
|
type Table = ApacheArrow["Table"];
|
||||||
|
|
||||||
function expectValidMapField(
|
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: Arrow Field types vary across supported versions
|
|
||||||
field: any,
|
|
||||||
): void {
|
|
||||||
expect(DataType.isMap(field.type)).toBe(true);
|
|
||||||
expect(field.type.keysSorted).toBe(true);
|
|
||||||
expect(field.type.children).toHaveLength(1);
|
|
||||||
|
|
||||||
const entries = field.type.children[0];
|
|
||||||
expect(entries.name).toBe("entries");
|
|
||||||
expect(entries.nullable).toBe(false);
|
|
||||||
expect(DataType.isStruct(entries.type)).toBe(true);
|
|
||||||
expect(entries.type.children).toHaveLength(2);
|
|
||||||
|
|
||||||
const [key, value] = entries.type.children;
|
|
||||||
expect([key.name, value.name]).toEqual(["key", "value"]);
|
|
||||||
expect(key.nullable).toBe(false);
|
|
||||||
expect(DataType.isUtf8(key.type)).toBe(true);
|
|
||||||
expect(value.nullable).toBe(true);
|
|
||||||
expect(DataType.isInt(value.type)).toBe(true);
|
|
||||||
expect(value.type.bitWidth).toBe(32);
|
|
||||||
expect(value.type.isSigned).toBe(true);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Helper method to verify various ways to create a table
|
// Helper method to verify various ways to create a table
|
||||||
async function checkTableCreation(
|
async function checkTableCreation(
|
||||||
tableCreationMethod: (
|
tableCreationMethod: (
|
||||||
@@ -963,65 +938,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
false,
|
false,
|
||||||
);
|
);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("will make an empty table with a Map field", async function () {
|
|
||||||
const schema = new Schema([
|
|
||||||
new Field(
|
|
||||||
"attributes",
|
|
||||||
new Map_(
|
|
||||||
new Field(
|
|
||||||
"entries",
|
|
||||||
new Struct([
|
|
||||||
new Field("key", new Utf8(), false),
|
|
||||||
new Field("value", new Int32(), true),
|
|
||||||
]),
|
|
||||||
false,
|
|
||||||
),
|
|
||||||
true,
|
|
||||||
),
|
|
||||||
),
|
|
||||||
]);
|
|
||||||
|
|
||||||
const table = makeEmptyTable(schema);
|
|
||||||
|
|
||||||
expectValidMapField(table.schema.fields[0]);
|
|
||||||
|
|
||||||
const buffer = await fromTableToBuffer(table);
|
|
||||||
const roundTripped = tableFromIPC(buffer);
|
|
||||||
|
|
||||||
expectValidMapField(roundTripped.schema.fields[0]);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("preserves string schema metadata", function () {
|
|
||||||
const metadata = new Map([["source", "fixture"]]);
|
|
||||||
const schema = new Schema(
|
|
||||||
[new Field("value", new Int32(), true)],
|
|
||||||
metadata,
|
|
||||||
);
|
|
||||||
|
|
||||||
expect(makeEmptyTable(schema).schema.metadata.get("source")).toBe(
|
|
||||||
"fixture",
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
it.each([
|
|
||||||
["non-string keys", new Map<unknown, unknown>([[42, "fixture"]])],
|
|
||||||
["non-string values", new Map<unknown, unknown>([["source", 42]])],
|
|
||||||
[
|
|
||||||
"non-string keys and values",
|
|
||||||
new Map<unknown, unknown>([[42, false]]),
|
|
||||||
],
|
|
||||||
])("rejects schema metadata with %s", function (_, metadataLike) {
|
|
||||||
const metadata = metadataLike as unknown as Map<string, string>;
|
|
||||||
const schema = new Schema(
|
|
||||||
[new Field("value", new Int32(), true)],
|
|
||||||
metadata,
|
|
||||||
);
|
|
||||||
|
|
||||||
expect(() => makeEmptyTable(schema)).toThrow(
|
|
||||||
"Expected metadata, if present, to be a Map<string, string> but it had non-string keys or values",
|
|
||||||
);
|
|
||||||
});
|
|
||||||
});
|
});
|
||||||
|
|
||||||
describe("when using two versions of arrow", function () {
|
describe("when using two versions of arrow", function () {
|
||||||
|
|||||||
@@ -1,114 +0,0 @@
|
|||||||
// SPDX-License-Identifier: Apache-2.0
|
|
||||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
|
||||||
|
|
||||||
import {
|
|
||||||
MeterProvider,
|
|
||||||
type MetricData,
|
|
||||||
MetricReader,
|
|
||||||
} from "@opentelemetry/sdk-metrics";
|
|
||||||
import * as tmp from "tmp";
|
|
||||||
import { connect, instrumentLanceDbMetrics } from "../lancedb";
|
|
||||||
// snapshotLancedbMetrics is internal plumbing (not part of the public API), so
|
|
||||||
// it is imported from the native module rather than the package entry point.
|
|
||||||
import { snapshotLancedbMetrics } from "../lancedb/native";
|
|
||||||
|
|
||||||
// The metrics recorder is process-global and installed once, so the whole
|
|
||||||
// bridge is exercised in a single test to avoid cross-test global-state coupling.
|
|
||||||
|
|
||||||
// A minimal pull-based reader whose `collect()` we drive directly, invoking the
|
|
||||||
// observable-instrument callbacks. `@opentelemetry/sdk-metrics` ships no
|
|
||||||
// in-memory reader, so we subclass the abstract base.
|
|
||||||
class TestMetricReader extends MetricReader {
|
|
||||||
protected async onForceFlush(): Promise<void> {
|
|
||||||
// no-op: collection is driven directly via collect()
|
|
||||||
}
|
|
||||||
protected async onShutdown(): Promise<void> {
|
|
||||||
// no-op: nothing to release
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async function metricsByName(
|
|
||||||
reader: TestMetricReader,
|
|
||||||
): Promise<Map<string, MetricData>> {
|
|
||||||
const collected = await reader.collect();
|
|
||||||
const result = new Map<string, MetricData>();
|
|
||||||
for (const scope of collected.resourceMetrics.scopeMetrics) {
|
|
||||||
for (const metric of scope.metrics) {
|
|
||||||
result.set(metric.descriptor.name, metric);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return result;
|
|
||||||
}
|
|
||||||
|
|
||||||
describe("OpenTelemetry metrics bridge", () => {
|
|
||||||
let tmpDir: tmp.DirResult;
|
|
||||||
beforeEach(() => {
|
|
||||||
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
|
||||||
});
|
|
||||||
afterEach(() => tmpDir.removeCallback());
|
|
||||||
|
|
||||||
it("snapshot is safe to call regardless of install state", () => {
|
|
||||||
expect(Array.isArray(snapshotLancedbMetrics())).toBe(true);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("exports object store metrics via observable instruments", async () => {
|
|
||||||
const reader = new TestMetricReader();
|
|
||||||
const provider = new MeterProvider({ readers: [reader] });
|
|
||||||
expect(instrumentLanceDbMetrics(provider)).toBe(true);
|
|
||||||
|
|
||||||
// Generate object store activity on the local filesystem (scheme "file").
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
const data = Array.from({ length: 256 }, (_, i) => ({ id: i }));
|
|
||||||
const table = await db.createTable("t", data);
|
|
||||||
expect(await table.countRows()).toBe(256);
|
|
||||||
|
|
||||||
const metrics = await metricsByName(reader);
|
|
||||||
|
|
||||||
const requests = metrics.get("lance_object_store_requests_total");
|
|
||||||
expect(requests).toBeDefined();
|
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: SDK point shape
|
|
||||||
const requestPoints = (requests!.dataPoints as any[]) ?? [];
|
|
||||||
expect(requestPoints.length).toBeGreaterThan(0);
|
|
||||||
for (const p of requestPoints) {
|
|
||||||
// Labelled by `operation` and `base` (the store scheme by default).
|
|
||||||
expect(p.attributes).toHaveProperty("base");
|
|
||||||
expect(p.attributes).toHaveProperty("operation");
|
|
||||||
}
|
|
||||||
const totalRequests = requestPoints.reduce((acc, p) => acc + p.value, 0);
|
|
||||||
expect(totalRequests).toBeGreaterThan(0);
|
|
||||||
|
|
||||||
// Histograms are decomposed into bucket / count / sum observable counters.
|
|
||||||
const bucket = metrics.get(
|
|
||||||
"lance_object_store_request_duration_seconds_bucket",
|
|
||||||
);
|
|
||||||
expect(bucket).toBeDefined();
|
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: SDK point shape
|
|
||||||
const bucketPoints = (bucket!.dataPoints as any[]) ?? [];
|
|
||||||
expect(bucketPoints.length).toBeGreaterThan(0);
|
|
||||||
expect(bucketPoints.every((p) => "le" in p.attributes)).toBe(true);
|
|
||||||
// The implicit +Inf bucket must be present.
|
|
||||||
expect(bucketPoints.some((p) => p.attributes.le === "+Inf")).toBe(true);
|
|
||||||
|
|
||||||
const count = metrics.get(
|
|
||||||
"lance_object_store_request_duration_seconds_count",
|
|
||||||
);
|
|
||||||
expect(count).toBeDefined();
|
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: SDK point shape
|
|
||||||
const countPoints = (count!.dataPoints as any[]) ?? [];
|
|
||||||
expect(countPoints.reduce((acc, p) => acc + p.value, 0)).toBeGreaterThan(0);
|
|
||||||
|
|
||||||
const sum = metrics.get("lance_object_store_request_duration_seconds_sum");
|
|
||||||
expect(sum).toBeDefined();
|
|
||||||
// biome-ignore lint/suspicious/noExplicitAny: SDK point shape
|
|
||||||
const sumPoints = (sum!.dataPoints as any[]) ?? [];
|
|
||||||
expect(sumPoints.reduce((acc, p) => acc + p.value, 0)).toBeGreaterThan(0);
|
|
||||||
|
|
||||||
// Unit handling: only `_sum` keeps the histogram's unit (seconds); `_bucket`
|
|
||||||
// and `_count` observe cumulative counts and are unitless.
|
|
||||||
expect(sum!.descriptor.unit).toBe("s");
|
|
||||||
expect(bucket!.descriptor.unit).toBe("");
|
|
||||||
expect(count!.descriptor.unit).toBe("");
|
|
||||||
|
|
||||||
await provider.shutdown();
|
|
||||||
});
|
|
||||||
});
|
|
||||||
@@ -215,20 +215,6 @@ describe("Query orderBy", () => {
|
|||||||
expect(results[2].score).toBeCloseTo(4.1, 0.001);
|
expect(results[2].score).toBeCloseTo(4.1, 0.001);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("should combine repeated where clauses with AND", async () => {
|
|
||||||
const results = await table
|
|
||||||
.query()
|
|
||||||
.where("score > 1.0")
|
|
||||||
.where("score < 3.0")
|
|
||||||
.orderBy({ columnName: "score" })
|
|
||||||
.toArray();
|
|
||||||
// Only rows matching both predicates should be returned, rather than the
|
|
||||||
// second where() silently replacing the first.
|
|
||||||
expect(results.length).toBe(2);
|
|
||||||
expect(results[0].score).toBeCloseTo(1.2, 0.001);
|
|
||||||
expect(results[1].score).toBeCloseTo(2.8, 0.001);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("should support method chaining with limit", async () => {
|
it("should support method chaining with limit", async () => {
|
||||||
const results = await table
|
const results = await table
|
||||||
.query()
|
.query()
|
||||||
|
|||||||
@@ -15,7 +15,6 @@ import {
|
|||||||
OAuthHeaderProvider,
|
OAuthHeaderProvider,
|
||||||
StaticHeaderProvider,
|
StaticHeaderProvider,
|
||||||
} from "../lancedb/header";
|
} from "../lancedb/header";
|
||||||
import { Index } from "../lancedb/indices";
|
|
||||||
|
|
||||||
// Test-only header providers
|
// Test-only header providers
|
||||||
class CustomProvider extends HeaderProvider {
|
class CustomProvider extends HeaderProvider {
|
||||||
@@ -192,200 +191,6 @@ describe("remote connection", () => {
|
|||||||
);
|
);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("supports version time-travel and branches on remote", async () => {
|
|
||||||
await withMockDatabase(
|
|
||||||
(req, res) => {
|
|
||||||
const body = req.url?.includes("/branches/list")
|
|
||||||
? JSON.stringify({
|
|
||||||
branches: {
|
|
||||||
exp: { parentVersion: 1, createAt: 1, manifestSize: 1 },
|
|
||||||
},
|
|
||||||
})
|
|
||||||
: JSON.stringify({ name: "t", version: 2, schema: { fields: [] } });
|
|
||||||
res.writeHead(200, { "Content-Type": "application/json" }).end(body);
|
|
||||||
},
|
|
||||||
async (db) => {
|
|
||||||
// version-only (and "main" + version) time-travel the main chain
|
|
||||||
const v2 = await db.openTable("t", undefined, { version: 2 });
|
|
||||||
expect(v2.currentBranch()).toBeNull();
|
|
||||||
const mainV2 = await db.openTable("t", undefined, {
|
|
||||||
branch: "main",
|
|
||||||
version: 2,
|
|
||||||
});
|
|
||||||
expect(mainV2.currentBranch()).toBeNull();
|
|
||||||
|
|
||||||
// a non-main branch opens a handle scoped to that branch
|
|
||||||
const exp = await db.openTable("t", undefined, { branch: "exp" });
|
|
||||||
expect(exp.currentBranch()).toBe("exp");
|
|
||||||
const expV2 = await db.openTable("t", undefined, {
|
|
||||||
branch: "exp",
|
|
||||||
version: 2,
|
|
||||||
});
|
|
||||||
expect(expV2.currentBranch()).toBe("exp");
|
|
||||||
},
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("sends FTS options to remote tables", async () => {
|
|
||||||
let createIndexBody: Record<string, unknown> | undefined;
|
|
||||||
|
|
||||||
await withMockDatabase(
|
|
||||||
(req, res) => {
|
|
||||||
const path = req.url ?? "";
|
|
||||||
if (path.endsWith("/describe/")) {
|
|
||||||
res.writeHead(200, { "Content-Type": "application/json" }).end(
|
|
||||||
JSON.stringify({
|
|
||||||
name: "t",
|
|
||||||
version: 1,
|
|
||||||
schema: {
|
|
||||||
fields: [
|
|
||||||
{ name: "text", type: { type: "string" }, nullable: false },
|
|
||||||
],
|
|
||||||
},
|
|
||||||
}),
|
|
||||||
);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (path.endsWith("/create_index/")) {
|
|
||||||
let raw = "";
|
|
||||||
req.on("data", (chunk) => {
|
|
||||||
raw += chunk;
|
|
||||||
});
|
|
||||||
req.on("end", () => {
|
|
||||||
createIndexBody = JSON.parse(raw);
|
|
||||||
res.writeHead(200).end();
|
|
||||||
});
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
res.writeHead(404).end();
|
|
||||||
},
|
|
||||||
async (db) => {
|
|
||||||
const table = await db.openTable("t");
|
|
||||||
await table.createIndex("text", {
|
|
||||||
config: Index.fts({
|
|
||||||
blockSize: 256,
|
|
||||||
removeStopWords: true,
|
|
||||||
customStopWords: ["the"],
|
|
||||||
}),
|
|
||||||
});
|
|
||||||
},
|
|
||||||
);
|
|
||||||
|
|
||||||
expect(createIndexBody?.["column"]).toBe("text");
|
|
||||||
expect(createIndexBody?.["index_type"]).toBe("FTS");
|
|
||||||
expect(createIndexBody?.["block_size"]).toBe(256);
|
|
||||||
expect(createIndexBody?.["custom_stop_words"]).toEqual(["the"]);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("diffs and merges remote branches", async () => {
|
|
||||||
const sampleDiff = {
|
|
||||||
fromBranch: "exp",
|
|
||||||
parentVersion: 1,
|
|
||||||
mainVersion: 2,
|
|
||||||
branchVersion: 3,
|
|
||||||
baseMoved: false,
|
|
||||||
rowCountMain: 3,
|
|
||||||
rowCountBranch: 3,
|
|
||||||
rowSummary: {
|
|
||||||
unchanged: 3,
|
|
||||||
newOnBase: 0,
|
|
||||||
newOnBranch: 0,
|
|
||||||
staleRecompute: 0,
|
|
||||||
inputsChanged: 0,
|
|
||||||
deltaAvailable: false,
|
|
||||||
},
|
|
||||||
addedColumns: [{ name: "tag", dataType: "utf8", nullable: true }],
|
|
||||||
removedColumns: [],
|
|
||||||
changedColumns: [],
|
|
||||||
addedIndexes: [],
|
|
||||||
removedIndexes: [],
|
|
||||||
mergeable: true,
|
|
||||||
mergeBlockers: [],
|
|
||||||
};
|
|
||||||
const mergeBodies: Record<string, unknown>[] = [];
|
|
||||||
|
|
||||||
await withMockDatabase(
|
|
||||||
(req, res) => {
|
|
||||||
const path = req.url ?? "";
|
|
||||||
if (path.endsWith("/describe/")) {
|
|
||||||
res.writeHead(200, { "Content-Type": "application/json" }).end(
|
|
||||||
JSON.stringify({
|
|
||||||
name: "t",
|
|
||||||
version: 2,
|
|
||||||
schema: { fields: [] },
|
|
||||||
}),
|
|
||||||
);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
let raw = "";
|
|
||||||
req.on("data", (chunk) => {
|
|
||||||
raw += chunk;
|
|
||||||
});
|
|
||||||
req.on("end", () => {
|
|
||||||
const body = raw ? JSON.parse(raw) : {};
|
|
||||||
if (path.endsWith("/branches/diff/")) {
|
|
||||||
// biome-ignore lint/style/useNamingConvention: snake_case mandated by the server wire format
|
|
||||||
expect(body).toEqual({ from_branch: "exp" });
|
|
||||||
res
|
|
||||||
.writeHead(200, { "Content-Type": "application/json" })
|
|
||||||
.end(JSON.stringify(sampleDiff));
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (path.endsWith("/branches/merge/")) {
|
|
||||||
mergeBodies.push(body);
|
|
||||||
const dryRun = body["dry_run"] === true;
|
|
||||||
const response = {
|
|
||||||
status: dryRun ? "ready" : "rejected",
|
|
||||||
diff: dryRun
|
|
||||||
? sampleDiff
|
|
||||||
: {
|
|
||||||
...sampleDiff,
|
|
||||||
mergeable: false,
|
|
||||||
mergeBlockers: [
|
|
||||||
{ code: "baseMoved", message: "main has advanced" },
|
|
||||||
],
|
|
||||||
},
|
|
||||||
preview: { promotedColumns: dryRun ? ["tag"] : [] },
|
|
||||||
};
|
|
||||||
res
|
|
||||||
.writeHead(dryRun ? 200 : 409, {
|
|
||||||
"Content-Type": "application/json",
|
|
||||||
})
|
|
||||||
.end(JSON.stringify(response));
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
res.writeHead(404).end();
|
|
||||||
});
|
|
||||||
},
|
|
||||||
async (db) => {
|
|
||||||
const table = await db.openTable("t");
|
|
||||||
const branches = await table.branches();
|
|
||||||
|
|
||||||
await expect(branches.diff("exp")).resolves.toEqual(sampleDiff);
|
|
||||||
|
|
||||||
const rejected = await branches.merge("exp");
|
|
||||||
expect(rejected.status).toBe("rejected");
|
|
||||||
expect(rejected.diff.mergeBlockers).toEqual([
|
|
||||||
{ code: "baseMoved", message: "main has advanced" },
|
|
||||||
]);
|
|
||||||
|
|
||||||
const preview = await branches.merge("exp", true);
|
|
||||||
expect(preview.status).toBe("ready");
|
|
||||||
expect(preview.preview.promotedColumns).toEqual(["tag"]);
|
|
||||||
},
|
|
||||||
);
|
|
||||||
|
|
||||||
expect(mergeBodies).toEqual([
|
|
||||||
// biome-ignore lint/style/useNamingConvention: snake_case mandated by the server wire format
|
|
||||||
{ from_branch: "exp", dry_run: false },
|
|
||||||
// biome-ignore lint/style/useNamingConvention: snake_case mandated by the server wire format
|
|
||||||
{ from_branch: "exp", dry_run: true },
|
|
||||||
]);
|
|
||||||
});
|
|
||||||
|
|
||||||
describe("TlsConfig", () => {
|
describe("TlsConfig", () => {
|
||||||
it("should create TlsConfig with all fields", () => {
|
it("should create TlsConfig with all fields", () => {
|
||||||
const tlsConfig: TlsConfig = {
|
const tlsConfig: TlsConfig = {
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
import * as arrow from "../lancedb/arrow";
|
import * as arrow from "../lancedb/arrow";
|
||||||
import { sanitizeField, sanitizeMap, sanitizeType } from "../lancedb/sanitize";
|
import { sanitizeField, sanitizeType } from "../lancedb/sanitize";
|
||||||
|
|
||||||
describe("sanitize", function () {
|
describe("sanitize", function () {
|
||||||
describe("sanitizeType function", function () {
|
describe("sanitizeType function", function () {
|
||||||
@@ -181,15 +181,4 @@ describe("sanitize", function () {
|
|||||||
);
|
);
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
describe("sanitizeMap function", function () {
|
|
||||||
it.each([
|
|
||||||
["no children", []],
|
|
||||||
["two children", [{}, {}]],
|
|
||||||
])("should reject a Map type with %s", function (_, children) {
|
|
||||||
expect(() => sanitizeMap({ children, keysSorted: false })).toThrow(
|
|
||||||
"Expected a Map type to have exactly one child",
|
|
||||||
);
|
|
||||||
});
|
|
||||||
});
|
|
||||||
});
|
});
|
||||||
|
|||||||
+25
-690
@@ -16,7 +16,6 @@ import {
|
|||||||
PhraseQuery,
|
PhraseQuery,
|
||||||
Table,
|
Table,
|
||||||
connect,
|
connect,
|
||||||
tokenize,
|
|
||||||
} from "../lancedb";
|
} from "../lancedb";
|
||||||
import {
|
import {
|
||||||
Table as ArrowTable,
|
Table as ArrowTable,
|
||||||
@@ -86,140 +85,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
await expect(table.countRows()).resolves.toBe(3);
|
await expect(table.countRows()).resolves.toBe(3);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("should support branches", async () => {
|
|
||||||
await table.add([{ id: 1 }]);
|
|
||||||
expect(await table.countRows()).toBe(1);
|
|
||||||
|
|
||||||
expect(table.currentBranch()).toBeNull();
|
|
||||||
|
|
||||||
// fork an isolated, writable branch from main
|
|
||||||
const branch = await (await table.branches()).create("exp");
|
|
||||||
expect(branch.currentBranch()).toBe("exp");
|
|
||||||
expect(await branch.countRows()).toBe(1);
|
|
||||||
await branch.add([{ id: 2 }]);
|
|
||||||
expect(await branch.countRows()).toBe(2);
|
|
||||||
// main is untouched by branch writes
|
|
||||||
expect(await table.countRows()).toBe(1);
|
|
||||||
|
|
||||||
// listed, with main (null) as the parent
|
|
||||||
const list = await (await table.branches()).list();
|
|
||||||
expect(Object.keys(list)).toContain("exp");
|
|
||||||
expect(list["exp"].parentBranch).toBeNull();
|
|
||||||
|
|
||||||
// fromRef="main" is equivalent to the default
|
|
||||||
await (await table.branches()).create("exp2", "main");
|
|
||||||
const list2 = await (await table.branches()).list();
|
|
||||||
expect(list2["exp2"].parentBranch).toBeNull();
|
|
||||||
|
|
||||||
// checkout returns a handle scoped to the branch's latest
|
|
||||||
const checkedOut = await (await table.branches()).checkout("exp");
|
|
||||||
expect(checkedOut.currentBranch()).toBe("exp");
|
|
||||||
expect(await checkedOut.countRows()).toBe(2);
|
|
||||||
|
|
||||||
// delete removes it
|
|
||||||
await (await table.branches()).delete("exp");
|
|
||||||
await (await table.branches()).delete("exp2");
|
|
||||||
const after = await (await table.branches()).list();
|
|
||||||
expect(Object.keys(after)).not.toContain("exp");
|
|
||||||
});
|
|
||||||
|
|
||||||
it("should open a branch via open_table", async () => {
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
await table.add([{ id: 1 }]);
|
|
||||||
const branch = await (await table.branches()).create("exp");
|
|
||||||
await branch.add([{ id: 2 }]);
|
|
||||||
|
|
||||||
// open_table(..., { branch }) returns a handle scoped to the branch
|
|
||||||
const opened = await db.openTable("some_table", undefined, {
|
|
||||||
branch: "exp",
|
|
||||||
});
|
|
||||||
expect(await opened.countRows()).toBe(2);
|
|
||||||
// opening without branch still tracks main
|
|
||||||
expect(await (await db.openTable("some_table")).countRows()).toBe(1);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("should open a branch at a version isolated from main and HEAD", async () => {
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
// main: a single fork-point row
|
|
||||||
const t = await db.createTable("bv_table", [{ id: 0 }]);
|
|
||||||
const mainV1 = await t.version();
|
|
||||||
|
|
||||||
// fork "exp", then advance exp AND main independently past the fork so
|
|
||||||
// they diverge while sharing version numbers
|
|
||||||
const exp = await (await t.branches()).create("exp");
|
|
||||||
await exp.add([{ id: 1 }]); // exp: {0, 1}
|
|
||||||
const expV2 = await exp.version();
|
|
||||||
await exp.add([{ id: 2 }]); // exp HEAD: {0, 1, 2}
|
|
||||||
await t.add([{ id: 100 }, { id: 101 }, { id: 102 }]); // main HEAD: {0,100,101,102}
|
|
||||||
expect(await t.version()).toBe(expV2);
|
|
||||||
|
|
||||||
// open exp at the shared version: the data must be exp's, not main's.
|
|
||||||
// count alone cannot prove this (main@v2 also exists), so assert
|
|
||||||
// provenance by content.
|
|
||||||
const pinned = await db.openTable("bv_table", undefined, {
|
|
||||||
branch: "exp",
|
|
||||||
version: expV2,
|
|
||||||
});
|
|
||||||
expect(await pinned.countRows()).toBe(2); // not exp HEAD (3), not main@v2 (4)
|
|
||||||
expect(await pinned.countRows("id = 1")).toBe(1); // exp's post-fork row
|
|
||||||
expect(await pinned.countRows("id = 100")).toBe(0); // main's rows invisible
|
|
||||||
|
|
||||||
// the same coordinate is reachable directly via branches().checkout(name, version)
|
|
||||||
const pinnedDirect = await (await t.branches()).checkout("exp", expV2);
|
|
||||||
expect(await pinnedDirect.countRows()).toBe(2);
|
|
||||||
|
|
||||||
// the HEADs are unaffected
|
|
||||||
expect(
|
|
||||||
await (
|
|
||||||
await db.openTable("bv_table", undefined, { branch: "exp" })
|
|
||||||
).countRows(),
|
|
||||||
).toBe(3);
|
|
||||||
expect(await (await db.openTable("bv_table")).countRows()).toBe(4);
|
|
||||||
|
|
||||||
// version-only (no branch) time-travels main itself: its fork-point
|
|
||||||
// version holds only main's first row, and the shared version number
|
|
||||||
// resolves to main's data, not the branch's ("opens main at the version")
|
|
||||||
const oldMain = await db.openTable("bv_table", undefined, {
|
|
||||||
version: mainV1,
|
|
||||||
});
|
|
||||||
expect(await oldMain.countRows()).toBe(1);
|
|
||||||
const sharedOnMain = await db.openTable("bv_table", undefined, {
|
|
||||||
version: expV2,
|
|
||||||
});
|
|
||||||
expect(await sharedOnMain.countRows()).toBe(4); // main@v2, not exp@v2 (2)
|
|
||||||
|
|
||||||
// detached head: writing to a pinned version is rejected
|
|
||||||
await expect(pinned.add([{ id: 9 }])).rejects.toThrow(
|
|
||||||
/cannot be modified/,
|
|
||||||
);
|
|
||||||
|
|
||||||
// a nonexistent version is rejected -- on main, and on a branch (a
|
|
||||||
// distinct resolution path, on the branch's manifests)
|
|
||||||
await expect(
|
|
||||||
db.openTable("bv_table", undefined, { version: 9999 }),
|
|
||||||
).rejects.toThrow();
|
|
||||||
await expect(
|
|
||||||
db.openTable("bv_table", undefined, { branch: "exp", version: 9999 }),
|
|
||||||
).rejects.toThrow();
|
|
||||||
|
|
||||||
// checkoutLatest re-attaches the pinned handle to the BRANCH's HEAD
|
|
||||||
// (writable again), not main's HEAD (4), and not staying pinned (2)
|
|
||||||
await pinned.checkoutLatest();
|
|
||||||
expect(await pinned.countRows()).toBe(3); // exp HEAD
|
|
||||||
await pinned.add([{ id: 3 }]);
|
|
||||||
expect(await pinned.countRows()).toBe(4); // writable again
|
|
||||||
});
|
|
||||||
|
|
||||||
it("rejects invalid branch inputs", async () => {
|
|
||||||
const branches = await table.branches();
|
|
||||||
await expect(branches.create("")).rejects.toThrow("non-empty");
|
|
||||||
await expect(branches.checkout("")).rejects.toThrow("non-empty");
|
|
||||||
await expect(branches.delete("")).rejects.toThrow("non-empty");
|
|
||||||
await expect(branches.create("bad", "main", -1)).rejects.toThrow(
|
|
||||||
"non-negative",
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("should show table stats", async () => {
|
it("should show table stats", async () => {
|
||||||
await table.add([{ id: 1 }, { id: 2 }]);
|
await table.add([{ id: 1 }, { id: 2 }]);
|
||||||
await table.add([{ id: 1 }]);
|
await table.add([{ id: 1 }]);
|
||||||
@@ -527,14 +392,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
);
|
);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("should expose useLsm on takeRowIds as the base-only escape hatch", async () => {
|
|
||||||
await table.add([{ id: 1 }, { id: 2 }, { id: 3 }]);
|
|
||||||
// useLsm(false) is reachable on TakeQuery (the escape hatch for MemWAL tables,
|
|
||||||
// where take-by-row-id auto-routes to the LSM scanner and is rejected).
|
|
||||||
const res = await table.takeRowIds([0, 2]).useLsm(false).toArray();
|
|
||||||
expect(res.map((r) => r.id)).toEqual([1, 3]);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("should throw for negative number in takeRowIds", () => {
|
it("should throw for negative number in takeRowIds", () => {
|
||||||
expect(() => table.takeRowIds([-1])).toThrow("Row id cannot be negative");
|
expect(() => table.takeRowIds([-1])).toThrow("Row id cannot be negative");
|
||||||
expect(() => table.takeRowIds([0, -5, 2])).toThrow(
|
expect(() => table.takeRowIds([0, -5, 2])).toThrow(
|
||||||
@@ -858,15 +715,13 @@ describe("When creating an index", () => {
|
|||||||
expect(fs.readdirSync(indexDir)).toHaveLength(1);
|
expect(fs.readdirSync(indexDir)).toHaveLength(1);
|
||||||
const indices = await tbl.listIndices();
|
const indices = await tbl.listIndices();
|
||||||
expect(indices.length).toBe(1);
|
expect(indices.length).toBe(1);
|
||||||
expect(indices[0]).toEqual(
|
expect(indices[0]).toEqual({
|
||||||
expect.objectContaining({
|
name: "vec_idx",
|
||||||
name: "vec_idx",
|
indexType: "IvfPq",
|
||||||
indexType: "IvfPq",
|
columns: ["vec"],
|
||||||
columns: ["vec"],
|
});
|
||||||
}),
|
|
||||||
);
|
|
||||||
const stats = await tbl.indexStats("vec_idx");
|
const stats = await tbl.indexStats("vec_idx");
|
||||||
expect(stats).toBeDefined();
|
expect(stats?.loss).toBeDefined();
|
||||||
|
|
||||||
// Search without specifying the column
|
// Search without specifying the column
|
||||||
let rst = await tbl
|
let rst = await tbl
|
||||||
@@ -926,22 +781,10 @@ describe("When creating an index", () => {
|
|||||||
expect(indices2.length).toBe(0);
|
expect(indices2.length).toBe(0);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("should preserve canonical nested field paths across index lifecycle", async () => {
|
it("should create and search a nested vector index", async () => {
|
||||||
const db = await connect(tmpDir.name);
|
const db = await connect(tmpDir.name);
|
||||||
const nestedSchema = new Schema([
|
const nestedSchema = new Schema([
|
||||||
new Field("rowId", new Int32(), true),
|
new Field("id", new Int32(), true),
|
||||||
new Field("row-id", new Int32(), true),
|
|
||||||
new Field("userId", new Int32(), true),
|
|
||||||
new Field(
|
|
||||||
"metadata",
|
|
||||||
new Struct([new Field("user_id", new Int32(), true)]),
|
|
||||||
true,
|
|
||||||
),
|
|
||||||
new Field(
|
|
||||||
"MetaData",
|
|
||||||
new Struct([new Field("userId", new Int32(), true)]),
|
|
||||||
true,
|
|
||||||
),
|
|
||||||
new Field(
|
new Field(
|
||||||
"image",
|
"image",
|
||||||
new Struct([
|
new Struct([
|
||||||
@@ -953,146 +796,27 @@ describe("When creating an index", () => {
|
|||||||
]),
|
]),
|
||||||
true,
|
true,
|
||||||
),
|
),
|
||||||
new Field(
|
|
||||||
"payload",
|
|
||||||
new Struct([new Field("text", new Utf8(), true)]),
|
|
||||||
true,
|
|
||||||
),
|
|
||||||
new Field(
|
|
||||||
"meta-data",
|
|
||||||
new Struct([new Field("user-id", new Int32(), true)]),
|
|
||||||
true,
|
|
||||||
),
|
|
||||||
new Field(
|
|
||||||
"literal",
|
|
||||||
new Struct([new Field("a.b", new Int32(), true)]),
|
|
||||||
true,
|
|
||||||
),
|
|
||||||
]);
|
]);
|
||||||
const nestedTable = await db.createTable(
|
const nestedTable = await db.createTable(
|
||||||
"nested_field_index_lifecycle",
|
"nested_vector",
|
||||||
makeArrowTable(
|
makeArrowTable(
|
||||||
Array.from({ length: 300 }, (_, rowId) => ({
|
Array.from({ length: 300 }, (_, id) => ({
|
||||||
rowId,
|
id,
|
||||||
"row-id": rowId,
|
image: { embedding: [id, id + 1] },
|
||||||
userId: rowId,
|
|
||||||
metadata: { ["user_id"]: rowId },
|
|
||||||
["MetaData"]: { userId: rowId },
|
|
||||||
image: { embedding: [rowId, rowId + 1] },
|
|
||||||
payload: { text: `document ${rowId}` },
|
|
||||||
"meta-data": { "user-id": rowId },
|
|
||||||
literal: { "a.b": rowId },
|
|
||||||
})),
|
})),
|
||||||
{ schema: nestedSchema },
|
{ schema: nestedSchema },
|
||||||
),
|
),
|
||||||
);
|
);
|
||||||
|
|
||||||
await nestedTable.createIndex("rowId", {
|
|
||||||
config: Index.btree(),
|
|
||||||
name: "row_id_idx",
|
|
||||||
});
|
|
||||||
await nestedTable.createIndex("`row-id`", {
|
|
||||||
config: Index.btree(),
|
|
||||||
name: "row_dash_id_idx",
|
|
||||||
});
|
|
||||||
await nestedTable.createIndex("userId", {
|
|
||||||
config: Index.btree(),
|
|
||||||
name: "top_user_id_idx",
|
|
||||||
});
|
|
||||||
await nestedTable.createIndex("metadata.user_id", {
|
|
||||||
config: Index.btree(),
|
|
||||||
name: "nested_user_id_idx",
|
|
||||||
});
|
|
||||||
await nestedTable.createIndex("MetaData.userId", {
|
|
||||||
config: Index.btree(),
|
|
||||||
name: "mixed_case_metadata_user_id_idx",
|
|
||||||
});
|
|
||||||
await nestedTable.createIndex("`meta-data`.`user-id`", {
|
|
||||||
config: Index.btree(),
|
|
||||||
name: "escaped_names_idx",
|
|
||||||
});
|
|
||||||
await nestedTable.createIndex("literal.`a.b`", {
|
|
||||||
config: Index.btree(),
|
|
||||||
name: "literal_dot_idx",
|
|
||||||
});
|
|
||||||
await nestedTable.createIndex("image.embedding", {
|
await nestedTable.createIndex("image.embedding", {
|
||||||
name: "image_embedding_idx",
|
name: "image_embedding_idx",
|
||||||
});
|
});
|
||||||
await nestedTable.createIndex("payload.text", {
|
|
||||||
config: Index.fts({ withPosition: false }),
|
|
||||||
name: "payload_text_idx",
|
|
||||||
});
|
|
||||||
|
|
||||||
const indices = await nestedTable.listIndices();
|
const indices = await nestedTable.listIndices();
|
||||||
expect(indices).toEqual(
|
expect(indices).toContainEqual({
|
||||||
expect.arrayContaining([
|
name: "image_embedding_idx",
|
||||||
expect.objectContaining({
|
indexType: "IvfPq",
|
||||||
name: "row_id_idx",
|
columns: ["image.embedding"],
|
||||||
indexType: "BTree",
|
});
|
||||||
columns: ["rowId"],
|
|
||||||
}),
|
|
||||||
expect.objectContaining({
|
|
||||||
name: "row_dash_id_idx",
|
|
||||||
indexType: "BTree",
|
|
||||||
columns: ["`row-id`"],
|
|
||||||
}),
|
|
||||||
expect.objectContaining({
|
|
||||||
name: "top_user_id_idx",
|
|
||||||
indexType: "BTree",
|
|
||||||
columns: ["userId"],
|
|
||||||
}),
|
|
||||||
expect.objectContaining({
|
|
||||||
name: "nested_user_id_idx",
|
|
||||||
indexType: "BTree",
|
|
||||||
columns: ["metadata.user_id"],
|
|
||||||
}),
|
|
||||||
expect.objectContaining({
|
|
||||||
name: "mixed_case_metadata_user_id_idx",
|
|
||||||
indexType: "BTree",
|
|
||||||
columns: ["MetaData.userId"],
|
|
||||||
}),
|
|
||||||
expect.objectContaining({
|
|
||||||
name: "escaped_names_idx",
|
|
||||||
indexType: "BTree",
|
|
||||||
columns: ["`meta-data`.`user-id`"],
|
|
||||||
}),
|
|
||||||
expect.objectContaining({
|
|
||||||
name: "literal_dot_idx",
|
|
||||||
indexType: "BTree",
|
|
||||||
columns: ["literal.`a.b`"],
|
|
||||||
}),
|
|
||||||
expect.objectContaining({
|
|
||||||
name: "image_embedding_idx",
|
|
||||||
indexType: "IvfPq",
|
|
||||||
columns: ["image.embedding"],
|
|
||||||
}),
|
|
||||||
expect.objectContaining({
|
|
||||||
name: "payload_text_idx",
|
|
||||||
indexType: "FTS",
|
|
||||||
columns: ["payload.text"],
|
|
||||||
}),
|
|
||||||
]),
|
|
||||||
);
|
|
||||||
|
|
||||||
const stats = await nestedTable.indexStats(
|
|
||||||
"mixed_case_metadata_user_id_idx",
|
|
||||||
);
|
|
||||||
expect(stats?.numIndexedRows).toEqual(300);
|
|
||||||
expect(stats?.indexType).toEqual("BTREE");
|
|
||||||
|
|
||||||
const filtered = await nestedTable
|
|
||||||
.query()
|
|
||||||
.where("MetaData.userId = 42")
|
|
||||||
.limit(1)
|
|
||||||
.toArray();
|
|
||||||
expect(filtered[0].MetaData.userId).toEqual(42);
|
|
||||||
|
|
||||||
const escapedFiltered = await nestedTable
|
|
||||||
.query()
|
|
||||||
.where("`row-id` = 43")
|
|
||||||
.limit(1)
|
|
||||||
.toArray();
|
|
||||||
expect(escapedFiltered[0]["row-id"]).toEqual(43);
|
|
||||||
|
|
||||||
const explicit = await nestedTable
|
const explicit = await nestedTable
|
||||||
.query()
|
.query()
|
||||||
@@ -1105,37 +829,7 @@ describe("When creating an index", () => {
|
|||||||
.nearestTo([0.0, 1.0])
|
.nearestTo([0.0, 1.0])
|
||||||
.limit(1)
|
.limit(1)
|
||||||
.toArray();
|
.toArray();
|
||||||
expect(inferred[0].rowId).toEqual(explicit[0].rowId);
|
expect(inferred[0].id).toEqual(explicit[0].id);
|
||||||
|
|
||||||
await nestedTable.add([
|
|
||||||
{
|
|
||||||
rowId: 300,
|
|
||||||
"row-id": 300,
|
|
||||||
userId: 300,
|
|
||||||
metadata: { ["user_id"]: 300 },
|
|
||||||
["MetaData"]: { userId: 300 },
|
|
||||||
image: { embedding: [300.0, 301.0] },
|
|
||||||
payload: { text: "document 300" },
|
|
||||||
"meta-data": { "user-id": 300 },
|
|
||||||
literal: { "a.b": 300 },
|
|
||||||
},
|
|
||||||
]);
|
|
||||||
await nestedTable.optimize();
|
|
||||||
const indicesAfterOptimize = await nestedTable.listIndices();
|
|
||||||
expect(indicesAfterOptimize).toEqual(
|
|
||||||
expect.arrayContaining([
|
|
||||||
expect.objectContaining({
|
|
||||||
name: "mixed_case_metadata_user_id_idx",
|
|
||||||
indexType: "BTree",
|
|
||||||
columns: ["MetaData.userId"],
|
|
||||||
}),
|
|
||||||
expect.objectContaining({
|
|
||||||
name: "image_embedding_idx",
|
|
||||||
indexType: "IvfPq",
|
|
||||||
columns: ["image.embedding"],
|
|
||||||
}),
|
|
||||||
]),
|
|
||||||
);
|
|
||||||
});
|
});
|
||||||
|
|
||||||
it("should report multiple nested vector candidates", async () => {
|
it("should report multiple nested vector candidates", async () => {
|
||||||
@@ -1269,13 +963,11 @@ describe("When creating an index", () => {
|
|||||||
expect(fs.readdirSync(indexDir)).toHaveLength(1);
|
expect(fs.readdirSync(indexDir)).toHaveLength(1);
|
||||||
const indices = await tbl.listIndices();
|
const indices = await tbl.listIndices();
|
||||||
expect(indices.length).toBe(1);
|
expect(indices.length).toBe(1);
|
||||||
expect(indices[0]).toEqual(
|
expect(indices[0]).toEqual({
|
||||||
expect.objectContaining({
|
name: "vec_idx",
|
||||||
name: "vec_idx",
|
indexType: "IvfHnswSq",
|
||||||
indexType: "IvfHnswSq",
|
columns: ["vec"],
|
||||||
columns: ["vec"],
|
});
|
||||||
}),
|
|
||||||
);
|
|
||||||
|
|
||||||
// Search without specifying the column
|
// Search without specifying the column
|
||||||
let rst = await tbl
|
let rst = await tbl
|
||||||
@@ -1448,20 +1140,6 @@ describe("When creating an index", () => {
|
|||||||
expect(fs.readdirSync(indexDir)).toHaveLength(1);
|
expect(fs.readdirSync(indexDir)).toHaveLength(1);
|
||||||
});
|
});
|
||||||
|
|
||||||
test("create an FM index", async () => {
|
|
||||||
// FM-Index accelerates substring search on a string/binary column.
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
const fmTbl = await db.createTable("fm_table", [
|
|
||||||
{ id: 0, text: "hello world" },
|
|
||||||
{ id: 1, text: "foo bar" },
|
|
||||||
]);
|
|
||||||
await fmTbl.createIndex("text", {
|
|
||||||
config: Index.fm(),
|
|
||||||
});
|
|
||||||
const indexDir = path.join(tmpDir.name, "fm_table.lance", "_indices");
|
|
||||||
expect(fs.readdirSync(indexDir)).toHaveLength(1);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("should be able to get index stats", async () => {
|
test("should be able to get index stats", async () => {
|
||||||
await tbl.createIndex("id");
|
await tbl.createIndex("id");
|
||||||
|
|
||||||
@@ -1472,6 +1150,7 @@ describe("When creating an index", () => {
|
|||||||
expect(stats?.distanceType).toBeUndefined();
|
expect(stats?.distanceType).toBeUndefined();
|
||||||
expect(stats?.indexType).toEqual("BTREE");
|
expect(stats?.indexType).toEqual("BTREE");
|
||||||
expect(stats?.numIndices).toEqual(1);
|
expect(stats?.numIndices).toEqual(1);
|
||||||
|
expect(stats?.loss).toBeUndefined();
|
||||||
});
|
});
|
||||||
|
|
||||||
test("when getting stats on non-existent index", async () => {
|
test("when getting stats on non-existent index", async () => {
|
||||||
@@ -1621,35 +1300,6 @@ describe("When creating an index", () => {
|
|||||||
expect(rst64Query.toString()).toEqual(rst64Search.toString());
|
expect(rst64Query.toString()).toEqual(rst64Search.toString());
|
||||||
expect(rst64Query.numRows).toBe(2);
|
expect(rst64Query.numRows).toBe(2);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("should expose rich metadata fields on IndexConfig", async () => {
|
|
||||||
await tbl.createIndex("id", { config: Index.btree() });
|
|
||||||
await tbl.createIndex("vec");
|
|
||||||
|
|
||||||
const indicesByName = Object.fromEntries(
|
|
||||||
(await tbl.listIndices()).map((idx) => [idx.name, idx]),
|
|
||||||
);
|
|
||||||
|
|
||||||
const scalarIdx = indicesByName["id_idx"];
|
|
||||||
expect(scalarIdx).toBeDefined();
|
|
||||||
expect(typeof scalarIdx.indexUuid).toBe("string");
|
|
||||||
expect(scalarIdx.numIndexedRows).toBe(300);
|
|
||||||
expect(scalarIdx.numUnindexedRows).toBe(0);
|
|
||||||
expect(scalarIdx.numSegments).toBeGreaterThanOrEqual(1);
|
|
||||||
expect(scalarIdx.sizeBytes).toBeGreaterThan(0);
|
|
||||||
// Use toString check to avoid cross-realm instanceof failures with native Date objects
|
|
||||||
expect(Object.prototype.toString.call(scalarIdx.createdAt)).toBe(
|
|
||||||
"[object Date]",
|
|
||||||
);
|
|
||||||
expect((scalarIdx.createdAt as Date).getTime()).toBeGreaterThan(0);
|
|
||||||
expect(typeof scalarIdx.indexDetails).toBe("object");
|
|
||||||
|
|
||||||
const vectorIdx = indicesByName["vec_idx"];
|
|
||||||
expect(vectorIdx).toBeDefined();
|
|
||||||
expect(typeof vectorIdx.indexUuid).toBe("string");
|
|
||||||
expect(vectorIdx.numIndexedRows).toBe(300);
|
|
||||||
expect(typeof vectorIdx.indexDetails).toBe("object");
|
|
||||||
});
|
|
||||||
});
|
});
|
||||||
|
|
||||||
describe("When querying a table", () => {
|
describe("When querying a table", () => {
|
||||||
@@ -1921,33 +1571,6 @@ describe("schema evolution", function () {
|
|||||||
expect(await table.schema()).toEqual(expectedSchema3);
|
expect(await table.schema()).toEqual(expectedSchema3);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("can update field metadata", async function () {
|
|
||||||
const con = await connect(tmpDir.name);
|
|
||||||
const table = await con.createTable("fm", [
|
|
||||||
{ id: 1, category: "a" },
|
|
||||||
{ id: 2, category: "b" },
|
|
||||||
]);
|
|
||||||
|
|
||||||
const res = await table.updateFieldMetadata([
|
|
||||||
{ path: "category", metadata: { unit: "label", pii: "false" } },
|
|
||||||
]);
|
|
||||||
expect(res).toHaveProperty("version");
|
|
||||||
expect(res.version).toBe(2);
|
|
||||||
|
|
||||||
let cat = (await table.schema()).fields.find((f) => f.name === "category");
|
|
||||||
expect(cat?.metadata.get("unit")).toBe("label");
|
|
||||||
expect(cat?.metadata.get("pii")).toBe("false");
|
|
||||||
|
|
||||||
// merge: add a key, delete one via null, keep the rest
|
|
||||||
await table.updateFieldMetadata([
|
|
||||||
{ path: "category", metadata: { source: "import", pii: null } },
|
|
||||||
]);
|
|
||||||
cat = (await table.schema()).fields.find((f) => f.name === "category");
|
|
||||||
expect(cat?.metadata.get("unit")).toBe("label"); // preserved
|
|
||||||
expect(cat?.metadata.get("source")).toBe("import"); // added
|
|
||||||
expect(cat?.metadata.has("pii")).toBe(false); // deleted
|
|
||||||
});
|
|
||||||
|
|
||||||
it("can cast to various types", async function () {
|
it("can cast to various types", async function () {
|
||||||
const con = await connect(tmpDir.name);
|
const con = await connect(tmpDir.name);
|
||||||
|
|
||||||
@@ -2316,75 +1939,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
expect(results2[0].text).toBe(data[1].text);
|
expect(results2[0].text).toBe(data[1].text);
|
||||||
});
|
});
|
||||||
|
|
||||||
test("tokenizes FTS queries by column or index name", async () => {
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
const data = [
|
|
||||||
{
|
|
||||||
text: "Running in cafés",
|
|
||||||
japanese: "Hello, こんにちは世界!",
|
|
||||||
vector: [0.1, 0.2, 0.3],
|
|
||||||
},
|
|
||||||
];
|
|
||||||
const table = await db.createTable("test", data);
|
|
||||||
await table.createIndex("text", {
|
|
||||||
config: Index.fts({ baseTokenizer: "simple" }),
|
|
||||||
});
|
|
||||||
await table.createIndex("japanese", {
|
|
||||||
config: Index.fts({
|
|
||||||
baseTokenizer: "icu",
|
|
||||||
stem: false,
|
|
||||||
removeStopWords: false,
|
|
||||||
}),
|
|
||||||
name: "japanese_icu_idx",
|
|
||||||
});
|
|
||||||
|
|
||||||
await expect(table.tokenize("hello", {} as never)).rejects.toThrow(
|
|
||||||
"Specify exactly one",
|
|
||||||
);
|
|
||||||
await expect(
|
|
||||||
table.tokenize("hello", {
|
|
||||||
column: "text",
|
|
||||||
indexName: "text_idx",
|
|
||||||
} as never),
|
|
||||||
).rejects.toThrow("Specify exactly one");
|
|
||||||
|
|
||||||
const simpleTokens = await table.tokenize("Running in cafés", {
|
|
||||||
column: "text",
|
|
||||||
});
|
|
||||||
expect(simpleTokens).toEqual([
|
|
||||||
{ text: "run", position: 0 },
|
|
||||||
{ text: "cafe", position: 2 },
|
|
||||||
]);
|
|
||||||
|
|
||||||
const icuTokens = await table.tokenize("Hello, こんにちは世界!", {
|
|
||||||
indexName: "japanese_icu_idx",
|
|
||||||
});
|
|
||||||
expect(icuTokens).toEqual([
|
|
||||||
{ text: "hello", position: 0 },
|
|
||||||
{ text: "こんにちは", position: 1 },
|
|
||||||
{ text: "世界", position: 2 },
|
|
||||||
]);
|
|
||||||
|
|
||||||
const directSimpleTokens = await tokenize("Running in cafés", {
|
|
||||||
baseTokenizer: "simple",
|
|
||||||
});
|
|
||||||
expect(directSimpleTokens).toEqual([
|
|
||||||
{ text: "run", position: 0 },
|
|
||||||
{ text: "cafe", position: 2 },
|
|
||||||
]);
|
|
||||||
|
|
||||||
const directIcuTokens = await tokenize("Hello, こんにちは世界!", {
|
|
||||||
baseTokenizer: "icu",
|
|
||||||
stem: false,
|
|
||||||
removeStopWords: false,
|
|
||||||
});
|
|
||||||
expect(directIcuTokens).toEqual([
|
|
||||||
{ text: "hello", position: 0 },
|
|
||||||
{ text: "こんにちは", position: 1 },
|
|
||||||
{ text: "世界", position: 2 },
|
|
||||||
]);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("full text search fast search", async () => {
|
test("full text search fast search", async () => {
|
||||||
const db = await connect(tmpDir.name);
|
const db = await connect(tmpDir.name);
|
||||||
const data = [{ text: "hello world", vector: [0.1, 0.2, 0.3], id: 1 }];
|
const data = [{ text: "hello world", vector: [0.1, 0.2, 0.3], id: 1 }];
|
||||||
@@ -2535,35 +2089,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
expect(results3.length).toBe(1);
|
expect(results3.length).toBe(1);
|
||||||
});
|
});
|
||||||
|
|
||||||
test("full text search with custom posting block size", async () => {
|
|
||||||
const db = await connect(tmpDir.name);
|
|
||||||
const data = [
|
|
||||||
{ text: "hello world", vector: [0.1, 0.2, 0.3] },
|
|
||||||
{ text: "goodbye world", vector: [0.4, 0.5, 0.6] },
|
|
||||||
];
|
|
||||||
const table = await db.createTable("test", data);
|
|
||||||
await table.createIndex("text", {
|
|
||||||
config: Index.fts({ blockSize: 256 }),
|
|
||||||
});
|
|
||||||
|
|
||||||
const index = (await table.listIndices()).find(
|
|
||||||
(index) => index.indexType === "FTS",
|
|
||||||
);
|
|
||||||
expect(index?.indexVersion).toBe(3);
|
|
||||||
expect(
|
|
||||||
(index?.indexDetails as Record<string, unknown>)["block_size"],
|
|
||||||
).toBe(256);
|
|
||||||
|
|
||||||
const results = await table.search("hello").toArray();
|
|
||||||
expect(results[0].text).toBe(data[0].text);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("rejects invalid full text posting block size", () => {
|
|
||||||
expect(() => Index.fts({ blockSize: 129 as 128 | 256 })).toThrow(
|
|
||||||
"128 or 256",
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("full text search without lowercase", async () => {
|
test("full text search without lowercase", async () => {
|
||||||
const db = await connect(tmpDir.name);
|
const db = await connect(tmpDir.name);
|
||||||
const data = [
|
const data = [
|
||||||
@@ -2769,15 +2294,6 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
},
|
},
|
||||||
);
|
);
|
||||||
|
|
||||||
test("tokenize supports custom stop words", async () => {
|
|
||||||
const tokens = await tokenize("the lance data", {
|
|
||||||
stem: false,
|
|
||||||
removeStopWords: true,
|
|
||||||
customStopWords: ["lance"],
|
|
||||||
});
|
|
||||||
expect(tokens.map((token) => token.text)).toEqual(["the", "data"]);
|
|
||||||
});
|
|
||||||
|
|
||||||
describe("when calling explainPlan", () => {
|
describe("when calling explainPlan", () => {
|
||||||
let tmpDir: tmp.DirResult;
|
let tmpDir: tmp.DirResult;
|
||||||
let table: Table;
|
let table: Table;
|
||||||
@@ -2821,13 +2337,8 @@ describe("when calling analyzePlan", () => {
|
|||||||
.fill(1)
|
.fill(1)
|
||||||
.map(() => Math.random());
|
.map(() => Math.random());
|
||||||
const plan = await table.query().nearestTo(queryVec).analyzePlan();
|
const plan = await table.query().nearestTo(queryVec).analyzePlan();
|
||||||
|
console.log("Query Plan:\n", plan); // <--- Print the plan
|
||||||
expect(plan).toMatch("AnalyzeExec");
|
expect(plan).toMatch("AnalyzeExec");
|
||||||
|
|
||||||
const fullPlan = await table
|
|
||||||
.query()
|
|
||||||
.nearestTo(queryVec)
|
|
||||||
.analyzePlan("full");
|
|
||||||
expect(fullPlan).toMatch("AnalyzeExec");
|
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
@@ -3113,180 +2624,4 @@ describe("setLsmWriteSpec / unsetLsmWriteSpec", () => {
|
|||||||
}),
|
}),
|
||||||
).rejects.toThrow();
|
).rejects.toThrow();
|
||||||
});
|
});
|
||||||
|
|
||||||
it("reads back the installed spec via getLsmWriteSpec", async () => {
|
|
||||||
const conn = await connect(tmpDir.name);
|
|
||||||
const table = await makeTable(conn);
|
|
||||||
await table.setUnenforcedPrimaryKey("id");
|
|
||||||
|
|
||||||
// Nothing installed yet.
|
|
||||||
expect(await table.getLsmWriteSpec()).toBeUndefined();
|
|
||||||
|
|
||||||
// A real scalar index is needed to name it as a maintained index.
|
|
||||||
await table.add([{ id: 1 }, { id: 2 }, { id: 3 }]);
|
|
||||||
await table.createIndex("id");
|
|
||||||
const indexName = (await table.listIndices())[0].name;
|
|
||||||
|
|
||||||
// Bucket spec round-trips, including maintained indexes and writer config
|
|
||||||
// defaults. Lance writer-config keys are canonically snake_case.
|
|
||||||
// biome-ignore lint/style/useNamingConvention: Lance writer-config keys are snake_case
|
|
||||||
const writerConfigDefaults = { durable_write: "false" };
|
|
||||||
await table.setLsmWriteSpec({
|
|
||||||
specType: "bucket",
|
|
||||||
column: "id",
|
|
||||||
numBuckets: 4,
|
|
||||||
maintainedIndexes: [indexName],
|
|
||||||
writerConfigDefaults,
|
|
||||||
});
|
|
||||||
const spec = await table.getLsmWriteSpec();
|
|
||||||
expect(spec).toBeDefined();
|
|
||||||
expect(spec?.specType).toBe("bucket");
|
|
||||||
expect(spec?.column).toBe("id");
|
|
||||||
expect(spec?.numBuckets).toBe(4);
|
|
||||||
expect(spec?.maintainedIndexes).toEqual([indexName]);
|
|
||||||
expect(spec?.writerConfigDefaults).toEqual(writerConfigDefaults);
|
|
||||||
|
|
||||||
// After unset, undefined again.
|
|
||||||
await table.unsetLsmWriteSpec();
|
|
||||||
expect(await table.getLsmWriteSpec()).toBeUndefined();
|
|
||||||
|
|
||||||
// Identity round-trips (column recovered from the schema).
|
|
||||||
await table.setLsmWriteSpec({ specType: "identity", column: "id" });
|
|
||||||
const identity = await table.getLsmWriteSpec();
|
|
||||||
expect(identity?.specType).toBe("identity");
|
|
||||||
expect(identity?.column).toBe("id");
|
|
||||||
await table.unsetLsmWriteSpec();
|
|
||||||
|
|
||||||
// Unsharded round-trips (no routing column).
|
|
||||||
await table.setLsmWriteSpec({ specType: "unsharded" });
|
|
||||||
const unsharded = await table.getLsmWriteSpec();
|
|
||||||
expect(unsharded?.specType).toBe("unsharded");
|
|
||||||
expect(unsharded?.column).toBeFalsy();
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
describe("LSM merge insert", () => {
|
|
||||||
let tmpDir: tmp.DirResult;
|
|
||||||
|
|
||||||
beforeEach(() => {
|
|
||||||
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
|
||||||
});
|
|
||||||
afterEach(() => tmpDir.removeCallback());
|
|
||||||
|
|
||||||
async function bucketTable(conn: Connection): Promise<Table> {
|
|
||||||
// The primary key column must be non-nullable.
|
|
||||||
const table = await conn.createEmptyTable(
|
|
||||||
"t",
|
|
||||||
new arrow.Schema([
|
|
||||||
new arrow.Field("id", new arrow.Utf8(), false),
|
|
||||||
new arrow.Field("value", new arrow.Float64(), true),
|
|
||||||
]),
|
|
||||||
);
|
|
||||||
await table.add([
|
|
||||||
{ id: "a", value: 1 },
|
|
||||||
{ id: "b", value: 2 },
|
|
||||||
]);
|
|
||||||
await table.setUnenforcedPrimaryKey("id");
|
|
||||||
// numBuckets = 1: every row routes to the single bucket.
|
|
||||||
await table.setLsmWriteSpec({
|
|
||||||
specType: "bucket",
|
|
||||||
column: "id",
|
|
||||||
numBuckets: 1,
|
|
||||||
});
|
|
||||||
return table;
|
|
||||||
}
|
|
||||||
|
|
||||||
it("routes merge_insert through the shard writer", async () => {
|
|
||||||
const conn = await connect(tmpDir.name);
|
|
||||||
const table = await bucketTable(conn);
|
|
||||||
|
|
||||||
const res = await table
|
|
||||||
.mergeInsert("id")
|
|
||||||
.whenMatchedUpdateAll()
|
|
||||||
.whenNotMatchedInsertAll()
|
|
||||||
.execute([
|
|
||||||
{ id: "c", value: 3 },
|
|
||||||
{ id: "d", value: 4 },
|
|
||||||
]);
|
|
||||||
// LSM path: rows go to the MemWAL, so only numRows is populated.
|
|
||||||
expect(res.numRows).toBe(2);
|
|
||||||
expect(res.version).toBe(0);
|
|
||||||
expect(res.numInsertedRows).toBe(0);
|
|
||||||
|
|
||||||
await table.closeLsmWriters();
|
|
||||||
});
|
|
||||||
|
|
||||||
it("falls back to the standard path with useLsm(false)", async () => {
|
|
||||||
const conn = await connect(tmpDir.name);
|
|
||||||
const table = await bucketTable(conn);
|
|
||||||
|
|
||||||
const res = await table
|
|
||||||
.mergeInsert("id")
|
|
||||||
.whenNotMatchedInsertAll()
|
|
||||||
.useLsm(false)
|
|
||||||
.execute([
|
|
||||||
{ id: "b", value: 9 },
|
|
||||||
{ id: "e", value: 5 },
|
|
||||||
]);
|
|
||||||
// Standard path commits: id="e" inserted ("b" already exists).
|
|
||||||
expect(res.numInsertedRows).toBe(1);
|
|
||||||
expect(await table.countRows()).toBe(3);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("supports validateSingleShard(false)", async () => {
|
|
||||||
const conn = await connect(tmpDir.name);
|
|
||||||
const table = await bucketTable(conn);
|
|
||||||
|
|
||||||
const res = await table
|
|
||||||
.mergeInsert("id")
|
|
||||||
.whenMatchedUpdateAll()
|
|
||||||
.whenNotMatchedInsertAll()
|
|
||||||
.validateSingleShard(false)
|
|
||||||
.execute([{ id: "f", value: 6 }]);
|
|
||||||
expect(res.numRows).toBe(1);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("rejects a non-upsert merge under an LSM spec", async () => {
|
|
||||||
const conn = await connect(tmpDir.name);
|
|
||||||
const table = await bucketTable(conn);
|
|
||||||
|
|
||||||
await expect(
|
|
||||||
table
|
|
||||||
.mergeInsert("id")
|
|
||||||
.whenNotMatchedInsertAll()
|
|
||||||
.execute([{ id: "g", value: 7 }]),
|
|
||||||
).rejects.toThrow();
|
|
||||||
});
|
|
||||||
|
|
||||||
it("auto-routes reads through the MemWAL scanner", async () => {
|
|
||||||
const conn = await connect(tmpDir.name);
|
|
||||||
const table = await bucketTable(conn); // base ids "a", "b"
|
|
||||||
|
|
||||||
await table
|
|
||||||
.mergeInsert("id")
|
|
||||||
.whenMatchedUpdateAll()
|
|
||||||
.whenNotMatchedInsertAll()
|
|
||||||
.execute([{ id: "c", value: 3 }]);
|
|
||||||
|
|
||||||
// Default read auto-routes and includes the active memtable row.
|
|
||||||
const lsm = await table.query().toArray();
|
|
||||||
expect(lsm.map((r) => r.id).sort()).toEqual(["a", "b", "c"]);
|
|
||||||
|
|
||||||
// useLsm(false) bypasses the MemWAL and reads the base table only.
|
|
||||||
const baseOnly = await table.query().useLsm(false).toArray();
|
|
||||||
expect(baseOnly.map((r) => r.id).sort()).toEqual(["a", "b"]);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("reads the base table when no LSM spec is installed", async () => {
|
|
||||||
const conn = await connect(tmpDir.name);
|
|
||||||
const table = await conn.createEmptyTable(
|
|
||||||
"plain",
|
|
||||||
new arrow.Schema([new arrow.Field("id", new arrow.Utf8(), false)]),
|
|
||||||
);
|
|
||||||
// No spec: default read and useLsm(false) both succeed against the base table.
|
|
||||||
await expect(table.query().toArray()).resolves.toBeDefined();
|
|
||||||
await expect(table.query().useLsm(false).toArray()).resolves.toBeDefined();
|
|
||||||
// useLsm(true) demands MemWAL routing; without a spec it errors.
|
|
||||||
await expect(table.query().useLsm(true).toArray()).rejects.toThrow();
|
|
||||||
});
|
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -29,14 +29,8 @@ test("full text search", async () => {
|
|||||||
const tbl = await db.createTable("myVectors", data, { mode: "overwrite" });
|
const tbl = await db.createTable("myVectors", data, { mode: "overwrite" });
|
||||||
|
|
||||||
await tbl.createIndex("doc", {
|
await tbl.createIndex("doc", {
|
||||||
config: lancedb.Index.fts({
|
config: lancedb.Index.fts(),
|
||||||
stem: false,
|
|
||||||
removeStopWords: true,
|
|
||||||
customStopWords: ["banana"],
|
|
||||||
}),
|
|
||||||
});
|
});
|
||||||
const tokens = await tbl.tokenize("apple banana", { column: "doc" });
|
|
||||||
expect(tokens.map((token) => token.text)).toEqual(["apple"]);
|
|
||||||
|
|
||||||
// --8<-- [start:full_text_search]
|
// --8<-- [start:full_text_search]
|
||||||
const result = await tbl
|
const result = await tbl
|
||||||
|
|||||||
@@ -84,20 +84,6 @@ export interface CreateTableOptions {
|
|||||||
}
|
}
|
||||||
|
|
||||||
export interface OpenTableOptions {
|
export interface OpenTableOptions {
|
||||||
/**
|
|
||||||
* Open the table scoped to this branch instead of the default branch.
|
|
||||||
*
|
|
||||||
* Reads and writes on the returned table operate in the branch's context.
|
|
||||||
*/
|
|
||||||
branch?: string;
|
|
||||||
/**
|
|
||||||
* Open the table pinned to this version, producing a read-only view.
|
|
||||||
*
|
|
||||||
* Composes with {@link OpenTableOptions.branch}: when both are set, opens
|
|
||||||
* that branch at the version; otherwise opens `main` at the version. Call
|
|
||||||
* `checkoutLatest` to return to a writable state.
|
|
||||||
*/
|
|
||||||
version?: number;
|
|
||||||
/**
|
/**
|
||||||
* Configuration for object storage.
|
* Configuration for object storage.
|
||||||
*
|
*
|
||||||
@@ -497,20 +483,7 @@ export class LocalConnection extends Connection {
|
|||||||
options?.indexCacheSize,
|
options?.indexCacheSize,
|
||||||
);
|
);
|
||||||
|
|
||||||
let table: Table = new LocalTable(innerTable);
|
return new LocalTable(innerTable);
|
||||||
// "main" is the default branch, so treat it as no branch. On a real branch,
|
|
||||||
// scope and pin in one step (yielding "version V of branch B"); otherwise
|
|
||||||
// pin the version, if any, against main.
|
|
||||||
const branch =
|
|
||||||
options?.branch != null && options.branch !== "main"
|
|
||||||
? options.branch
|
|
||||||
: undefined;
|
|
||||||
if (branch != null) {
|
|
||||||
table = await (await table.branches()).checkout(branch, options?.version);
|
|
||||||
} else if (options?.version != null) {
|
|
||||||
await table.checkout(options.version);
|
|
||||||
}
|
|
||||||
return table;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
async cloneTable(
|
async cloneTable(
|
||||||
|
|||||||
@@ -13,21 +13,13 @@ import {
|
|||||||
Connection as LanceDbConnection,
|
Connection as LanceDbConnection,
|
||||||
JsHeaderProvider as NativeJsHeaderProvider,
|
JsHeaderProvider as NativeJsHeaderProvider,
|
||||||
Session,
|
Session,
|
||||||
tokenize as nativeTokenize,
|
|
||||||
} from "./native.js";
|
} from "./native.js";
|
||||||
|
|
||||||
import { HeaderProvider } from "./header";
|
import { HeaderProvider } from "./header";
|
||||||
import type { BaseTokenizer } from "./indices";
|
|
||||||
import type { FtsToken } from "./table";
|
|
||||||
|
|
||||||
// Re-export native header provider for use with connectWithHeaderProvider
|
// Re-export native header provider for use with connectWithHeaderProvider
|
||||||
export { JsHeaderProvider as NativeJsHeaderProvider } from "./native.js";
|
export { JsHeaderProvider as NativeJsHeaderProvider } from "./native.js";
|
||||||
|
|
||||||
// OpenTelemetry metrics bridge. Only the high-level entry point is public; the
|
|
||||||
// underlying recorder/catalog/snapshot functions remain internal plumbing that
|
|
||||||
// `otel.ts` consumes from the native module.
|
|
||||||
export { instrumentLanceDbMetrics } from "./otel";
|
|
||||||
|
|
||||||
export {
|
export {
|
||||||
AddColumnsSql,
|
AddColumnsSql,
|
||||||
ConnectionOptions,
|
ConnectionOptions,
|
||||||
@@ -46,12 +38,10 @@ export {
|
|||||||
FragmentSummaryStats,
|
FragmentSummaryStats,
|
||||||
Tags,
|
Tags,
|
||||||
TagContents,
|
TagContents,
|
||||||
BranchContents,
|
|
||||||
MergeResult,
|
MergeResult,
|
||||||
AddResult,
|
AddResult,
|
||||||
AddColumnsResult,
|
AddColumnsResult,
|
||||||
AlterColumnsResult,
|
AlterColumnsResult,
|
||||||
UpdateFieldMetadataResult,
|
|
||||||
DeleteResult,
|
DeleteResult,
|
||||||
DropColumnsResult,
|
DropColumnsResult,
|
||||||
UpdateResult,
|
UpdateResult,
|
||||||
@@ -60,7 +50,6 @@ export {
|
|||||||
SplitHashOptions,
|
SplitHashOptions,
|
||||||
SplitSequentialOptions,
|
SplitSequentialOptions,
|
||||||
ShuffleOptions,
|
ShuffleOptions,
|
||||||
OAuthConfig as NativeOAuthConfig,
|
|
||||||
} from "./native.js";
|
} from "./native.js";
|
||||||
|
|
||||||
export {
|
export {
|
||||||
@@ -93,7 +82,6 @@ export {
|
|||||||
QueryBase,
|
QueryBase,
|
||||||
VectorQuery,
|
VectorQuery,
|
||||||
TakeQuery,
|
TakeQuery,
|
||||||
AnalyzePlanDistributedMetrics,
|
|
||||||
QueryExecutionOptions,
|
QueryExecutionOptions,
|
||||||
ColumnOrdering,
|
ColumnOrdering,
|
||||||
FullTextSearchOptions,
|
FullTextSearchOptions,
|
||||||
@@ -118,30 +106,17 @@ export {
|
|||||||
HnswPqOptions,
|
HnswPqOptions,
|
||||||
HnswSqOptions,
|
HnswSqOptions,
|
||||||
FtsOptions,
|
FtsOptions,
|
||||||
BaseTokenizer,
|
|
||||||
} from "./indices";
|
} from "./indices";
|
||||||
|
|
||||||
export {
|
export {
|
||||||
Table,
|
Table,
|
||||||
Branches,
|
|
||||||
BranchColumnSummary,
|
|
||||||
BranchColumnChange,
|
|
||||||
BranchIndexSummary,
|
|
||||||
BranchRowCountSummary,
|
|
||||||
MergeBlocker,
|
|
||||||
BranchDiff,
|
|
||||||
MergePreview,
|
|
||||||
MergeBranchResult,
|
|
||||||
AddDataOptions,
|
AddDataOptions,
|
||||||
UpdateOptions,
|
UpdateOptions,
|
||||||
OptimizeOptions,
|
OptimizeOptions,
|
||||||
Version,
|
Version,
|
||||||
WriteProgress,
|
WriteProgress,
|
||||||
FtsToken,
|
|
||||||
TokenizeTableOptions,
|
|
||||||
LsmWriteSpec,
|
LsmWriteSpec,
|
||||||
ColumnAlteration,
|
ColumnAlteration,
|
||||||
FieldMetadataUpdate,
|
|
||||||
} from "./table";
|
} from "./table";
|
||||||
|
|
||||||
export {
|
export {
|
||||||
@@ -151,8 +126,6 @@ export {
|
|||||||
TokenResponse,
|
TokenResponse,
|
||||||
} from "./header";
|
} from "./header";
|
||||||
|
|
||||||
export { OAuthConfig, OAuthFlowType } from "./oauth";
|
|
||||||
|
|
||||||
export { MergeInsertBuilder, WriteExecutionOptions } from "./merge";
|
export { MergeInsertBuilder, WriteExecutionOptions } from "./merge";
|
||||||
|
|
||||||
export * as embedding from "./embedding";
|
export * as embedding from "./embedding";
|
||||||
@@ -170,79 +143,6 @@ export {
|
|||||||
} from "./arrow";
|
} from "./arrow";
|
||||||
export { IntoSql, packBits } from "./util";
|
export { IntoSql, packBits } from "./util";
|
||||||
|
|
||||||
/**
|
|
||||||
* Options for tokenizing a full-text search query without a table index.
|
|
||||||
*/
|
|
||||||
export interface TokenizeOptions {
|
|
||||||
/**
|
|
||||||
* The tokenizer to use. The default is "simple".
|
|
||||||
*/
|
|
||||||
baseTokenizer?: BaseTokenizer;
|
|
||||||
|
|
||||||
/** Language for stemming and stop words. */
|
|
||||||
language?: string;
|
|
||||||
|
|
||||||
/** Maximum token length; tokens longer than this are ignored. */
|
|
||||||
maxTokenLength?: number;
|
|
||||||
|
|
||||||
/** Whether to lowercase tokens. */
|
|
||||||
lowercase?: boolean;
|
|
||||||
|
|
||||||
/** Whether to stem tokens. */
|
|
||||||
stem?: boolean;
|
|
||||||
|
|
||||||
/** Whether to remove stop words. */
|
|
||||||
removeStopWords?: boolean;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Custom stop words that replace the built-in list for `language`.
|
|
||||||
*
|
|
||||||
* This option only affects tokenization when `removeStopWords` is true.
|
|
||||||
*
|
|
||||||
* `undefined` keeps the built-in language list. An empty array explicitly
|
|
||||||
* replaces it with no stop words.
|
|
||||||
*/
|
|
||||||
customStopWords?: string[];
|
|
||||||
|
|
||||||
/** Whether to fold ASCII characters. */
|
|
||||||
asciiFolding?: boolean;
|
|
||||||
|
|
||||||
/** N-gram minimum length. */
|
|
||||||
ngramMinLength?: number;
|
|
||||||
|
|
||||||
/** N-gram maximum length. */
|
|
||||||
ngramMaxLength?: number;
|
|
||||||
|
|
||||||
/** Whether to only emit token prefixes for the n-gram tokenizer. */
|
|
||||||
prefixOnly?: boolean;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Tokenize a full-text search query using an explicit tokenizer.
|
|
||||||
*
|
|
||||||
* This does not require a table or FTS index. The tokenizer options match
|
|
||||||
* {@link Index.fts}.
|
|
||||||
*/
|
|
||||||
export async function tokenize(
|
|
||||||
query: string,
|
|
||||||
options?: Partial<TokenizeOptions>,
|
|
||||||
): Promise<FtsToken[]> {
|
|
||||||
return await nativeTokenize(
|
|
||||||
query,
|
|
||||||
options?.baseTokenizer,
|
|
||||||
options?.language,
|
|
||||||
options?.maxTokenLength,
|
|
||||||
options?.lowercase,
|
|
||||||
options?.stem,
|
|
||||||
options?.removeStopWords,
|
|
||||||
options?.customStopWords,
|
|
||||||
options?.asciiFolding,
|
|
||||||
options?.ngramMinLength,
|
|
||||||
options?.ngramMaxLength,
|
|
||||||
options?.prefixOnly,
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Connect to a LanceDB instance at the given URI.
|
* Connect to a LanceDB instance at the given URI.
|
||||||
*
|
*
|
||||||
|
|||||||
@@ -486,16 +486,6 @@ export interface IvfFlatOptions {
|
|||||||
sampleRate?: number;
|
sampleRate?: number;
|
||||||
}
|
}
|
||||||
|
|
||||||
export type BaseTokenizer =
|
|
||||||
| "simple"
|
|
||||||
| "whitespace"
|
|
||||||
| "raw"
|
|
||||||
| "ngram"
|
|
||||||
| "icu"
|
|
||||||
| "icu/split"
|
|
||||||
| `jieba/${string}`
|
|
||||||
| `lindera/${string}`;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Options to create a full text search index
|
* Options to create a full text search index
|
||||||
*/
|
*/
|
||||||
@@ -519,12 +509,8 @@ export interface FtsOptions {
|
|||||||
* "whitespace" - Whitespace tokenizer. This tokenizer splits the text into tokens using whitespace as a delimiter.
|
* "whitespace" - Whitespace tokenizer. This tokenizer splits the text into tokens using whitespace as a delimiter.
|
||||||
*
|
*
|
||||||
* "raw" - Raw tokenizer. This tokenizer does not split the text into tokens and indexes the entire text as a single token.
|
* "raw" - Raw tokenizer. This tokenizer does not split the text into tokens and indexes the entire text as a single token.
|
||||||
*
|
|
||||||
* "icu" - ICU dictionary-based word segmentation.
|
|
||||||
*
|
|
||||||
* "icu/split" - ICU segmentation with simple-style delimiter splitting.
|
|
||||||
*/
|
*/
|
||||||
baseTokenizer?: BaseTokenizer;
|
baseTokenizer?: "simple" | "whitespace" | "raw" | "ngram";
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* language for stemming and stop words
|
* language for stemming and stop words
|
||||||
@@ -553,16 +539,6 @@ export interface FtsOptions {
|
|||||||
*/
|
*/
|
||||||
removeStopWords?: boolean;
|
removeStopWords?: boolean;
|
||||||
|
|
||||||
/**
|
|
||||||
* Custom stop words that replace the built-in list for `language`.
|
|
||||||
*
|
|
||||||
* This option only affects tokenization when `removeStopWords` is true.
|
|
||||||
*
|
|
||||||
* `undefined` keeps the built-in language list. An empty array explicitly
|
|
||||||
* replaces it with no stop words.
|
|
||||||
*/
|
|
||||||
customStopWords?: string[];
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* whether to remove punctuation
|
* whether to remove punctuation
|
||||||
*/
|
*/
|
||||||
@@ -582,14 +558,6 @@ export interface FtsOptions {
|
|||||||
* whether to only index the prefix of the token for ngram tokenizer
|
* whether to only index the prefix of the token for ngram tokenizer
|
||||||
*/
|
*/
|
||||||
prefixOnly?: boolean;
|
prefixOnly?: boolean;
|
||||||
|
|
||||||
/**
|
|
||||||
* Number of documents per compressed posting block.
|
|
||||||
*
|
|
||||||
* The default is 128. Supported values are 128 and 256. A value of 256 uses
|
|
||||||
* the experimental FTS V3 format and may introduce breaking changes.
|
|
||||||
*/
|
|
||||||
blockSize?: 128 | 256;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
export class Index {
|
export class Index {
|
||||||
@@ -734,17 +702,6 @@ export class Index {
|
|||||||
return new Index(LanceDbIndex.labelList());
|
return new Index(LanceDbIndex.labelList());
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Create an FM-Index.
|
|
||||||
*
|
|
||||||
* An FM-Index is a scalar index on string or binary columns that accelerates
|
|
||||||
* substring search, i.e. `contains(col, 'needle')`. Unlike the tokenized
|
|
||||||
* full-text-search index, it matches arbitrary substrings of the raw bytes.
|
|
||||||
*/
|
|
||||||
static fm() {
|
|
||||||
return new Index(LanceDbIndex.fm());
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Create a full text search index
|
* Create a full text search index
|
||||||
*
|
*
|
||||||
@@ -765,12 +722,10 @@ export class Index {
|
|||||||
options?.lowercase,
|
options?.lowercase,
|
||||||
options?.stem,
|
options?.stem,
|
||||||
options?.removeStopWords,
|
options?.removeStopWords,
|
||||||
options?.customStopWords,
|
|
||||||
options?.asciiFolding,
|
options?.asciiFolding,
|
||||||
options?.ngramMinLength,
|
options?.ngramMinLength,
|
||||||
options?.ngramMaxLength,
|
options?.ngramMaxLength,
|
||||||
options?.prefixOnly,
|
options?.prefixOnly,
|
||||||
options?.blockSize,
|
|
||||||
),
|
),
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -87,37 +87,6 @@ export class MergeInsertBuilder {
|
|||||||
this.#schema,
|
this.#schema,
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
/**
|
|
||||||
* Control MemWAL routing for this merge.
|
|
||||||
*
|
|
||||||
* By default (unset), a `mergeInsert` on a table with an LSM write spec is
|
|
||||||
* routed through Lance's MemWAL shard writer, and a table without one uses the
|
|
||||||
* standard path.
|
|
||||||
*
|
|
||||||
* @param enable - `true` forces MemWAL routing and errors if the table has no
|
|
||||||
* LSM write spec. `false` forces the standard write path even when a spec is set.
|
|
||||||
*/
|
|
||||||
useLsm(enable: boolean): MergeInsertBuilder {
|
|
||||||
return new MergeInsertBuilder(this.#native.useLsm(enable), this.#schema);
|
|
||||||
}
|
|
||||||
/**
|
|
||||||
* Controls how an LSM merge checks that its input targets a single shard.
|
|
||||||
*
|
|
||||||
* When a table has an LSM write spec, every row in a `mergeInsert` call must
|
|
||||||
* route to the same shard. When `true` (the default), every row is inspected
|
|
||||||
* to verify this. When `false`, only the first row is inspected and the
|
|
||||||
* shard it routes to is used for the whole input — a faster path for callers
|
|
||||||
* that have already pre-sharded their input. Has no effect on tables without
|
|
||||||
* an LSM write spec.
|
|
||||||
*
|
|
||||||
* @param validateSingleShard - Whether to check every row routes to one shard. Defaults to `true`.
|
|
||||||
*/
|
|
||||||
validateSingleShard(validateSingleShard: boolean): MergeInsertBuilder {
|
|
||||||
return new MergeInsertBuilder(
|
|
||||||
this.#native.validateSingleShard(validateSingleShard),
|
|
||||||
this.#schema,
|
|
||||||
);
|
|
||||||
}
|
|
||||||
/**
|
/**
|
||||||
* Executes the merge insert operation
|
* Executes the merge insert operation
|
||||||
*
|
*
|
||||||
|
|||||||
@@ -1,76 +0,0 @@
|
|||||||
// SPDX-License-Identifier: Apache-2.0
|
|
||||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
|
||||||
|
|
||||||
/**
|
|
||||||
* OAuth authentication flow types.
|
|
||||||
*/
|
|
||||||
export enum OAuthFlowType {
|
|
||||||
/** Client Credentials grant (service-to-service / M2M). */
|
|
||||||
ClientCredentials = "client_credentials",
|
|
||||||
/** Azure Managed Identity via IMDS. */
|
|
||||||
AzureManagedIdentity = "azure_managed_identity",
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* OAuth configuration for LanceDB authentication.
|
|
||||||
*
|
|
||||||
* This is the public TypeScript OAuth configuration type. The generated
|
|
||||||
* `NativeOAuthConfig` type has the same runtime shape but is an implementation
|
|
||||||
* detail of the napi-rs binding.
|
|
||||||
*
|
|
||||||
* All token acquisition and refresh is handled in the Rust layer.
|
|
||||||
* This config is passed through to Rust via napi-rs.
|
|
||||||
*
|
|
||||||
* @example Client Credentials (service-to-service):
|
|
||||||
* ```typescript
|
|
||||||
* const config: OAuthConfig = {
|
|
||||||
* issuerUrl: "https://login.microsoftonline.com/{tenant}/v2.0",
|
|
||||||
* clientId: "app-id",
|
|
||||||
* clientSecret: "secret",
|
|
||||||
* scopes: ["api://lancedb-api/.default"],
|
|
||||||
* };
|
|
||||||
* ```
|
|
||||||
*
|
|
||||||
* @example Azure Managed Identity:
|
|
||||||
* ```typescript
|
|
||||||
* const config: OAuthConfig = {
|
|
||||||
* issuerUrl: "https://login.microsoftonline.com/{tenant}/v2.0",
|
|
||||||
* clientId: "app-id",
|
|
||||||
* scopes: ["api://lancedb-api/.default"],
|
|
||||||
* flow: OAuthFlowType.AzureManagedIdentity,
|
|
||||||
* };
|
|
||||||
* ```
|
|
||||||
*/
|
|
||||||
export interface OAuthConfig {
|
|
||||||
/**
|
|
||||||
* OIDC issuer URL or OAuth authority URL.
|
|
||||||
* For Azure: `https://login.microsoftonline.com/{tenant_id}/v2.0`
|
|
||||||
*/
|
|
||||||
issuerUrl: string;
|
|
||||||
|
|
||||||
/** Application / Client ID. */
|
|
||||||
clientId: string;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* OAuth scopes to request.
|
|
||||||
* For Azure managed identity, exactly one scope or resource is required.
|
|
||||||
* For example: `["api://{app_id}/.default"]`
|
|
||||||
*/
|
|
||||||
scopes: string[];
|
|
||||||
|
|
||||||
/** Authentication flow (default: ClientCredentials). */
|
|
||||||
flow?: OAuthFlowType;
|
|
||||||
|
|
||||||
/** Client secret (required for ClientCredentials). */
|
|
||||||
clientSecret?: string;
|
|
||||||
|
|
||||||
/** Client ID for user-assigned managed identity (AzureManagedIdentity). */
|
|
||||||
managedIdentityClientId?: string;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Seconds before expiry to trigger proactive refresh (default: 300).
|
|
||||||
* Keep this well below the token TTL; if it is greater than or equal to
|
|
||||||
* the TTL, each request refreshes the token.
|
|
||||||
*/
|
|
||||||
refreshBufferSecs?: number;
|
|
||||||
}
|
|
||||||
@@ -1,137 +0,0 @@
|
|||||||
// SPDX-License-Identifier: Apache-2.0
|
|
||||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
|
||||||
|
|
||||||
import {
|
|
||||||
type Attributes,
|
|
||||||
type MeterProvider,
|
|
||||||
type ObservableResult,
|
|
||||||
metrics,
|
|
||||||
} from "@opentelemetry/api";
|
|
||||||
|
|
||||||
import {
|
|
||||||
lancedbMetricsCatalog,
|
|
||||||
registerLancedbMetricsRecorder,
|
|
||||||
snapshotLancedbMetrics,
|
|
||||||
} from "./native";
|
|
||||||
|
|
||||||
let instrumented = false;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Register LanceDB metrics as OpenTelemetry observable instruments.
|
|
||||||
*
|
|
||||||
* Installs a process-global metrics recorder and creates one observable
|
|
||||||
* instrument per LanceDB metric (currently object store request counts, bytes,
|
|
||||||
* latency, errors, and throttles) on the given (or global) `MeterProvider`. The
|
|
||||||
* configured `MetricReader` then collects them on its own schedule.
|
|
||||||
*
|
|
||||||
* Counters and gauges map directly to observable counters/gauges. Because
|
|
||||||
* OpenTelemetry has no asynchronous histogram instrument, each histogram is
|
|
||||||
* exported Prometheus-style as cumulative `le` bucket counts (`<name>_bucket`,
|
|
||||||
* with an `le` attribute) plus `<name>_count` and `<name>_sum`.
|
|
||||||
*
|
|
||||||
* Requires `@opentelemetry/api` (a dependency) and, to actually export, an
|
|
||||||
* OpenTelemetry SDK such as `@opentelemetry/sdk-metrics`.
|
|
||||||
*
|
|
||||||
* @param meterProvider The provider to register instruments on. Defaults to the
|
|
||||||
* global provider from `@opentelemetry/api`.
|
|
||||||
* @returns `true` if the recorder is installed and instruments are registered.
|
|
||||||
* `false` if a different `metrics` recorder is already installed in this
|
|
||||||
* process (only one global recorder is permitted), in which case a warning is
|
|
||||||
* emitted and no instruments are created. Calling this more than once is safe;
|
|
||||||
* instruments are created only on the first successful call.
|
|
||||||
*/
|
|
||||||
export function instrumentLanceDbMetrics(
|
|
||||||
meterProvider?: MeterProvider,
|
|
||||||
): boolean {
|
|
||||||
if (!registerLancedbMetricsRecorder()) {
|
|
||||||
console.warn(
|
|
||||||
"Could not install the LanceDB metrics recorder: another `metrics` " +
|
|
||||||
"recorder is already installed in this process. LanceDB metrics will " +
|
|
||||||
"not be exported via OpenTelemetry.",
|
|
||||||
);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (instrumented) {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
const provider = meterProvider ?? metrics.getMeterProvider();
|
|
||||||
const meter = provider.getMeter("lancedb");
|
|
||||||
|
|
||||||
const scalarCallback = (metricName: string) => (result: ObservableResult) => {
|
|
||||||
for (const point of snapshotLancedbMetrics()) {
|
|
||||||
if (point.name === metricName && point.value != null) {
|
|
||||||
result.observe(point.value, point.attributes);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
const bucketCallback = (metricName: string) => (result: ObservableResult) => {
|
|
||||||
for (const point of snapshotLancedbMetrics()) {
|
|
||||||
if (point.name !== metricName || point.buckets == null) {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
for (const bucket of point.buckets) {
|
|
||||||
const attributes: Attributes = {
|
|
||||||
...point.attributes,
|
|
||||||
le: bucket.le,
|
|
||||||
};
|
|
||||||
result.observe(bucket.cumulativeCount, attributes);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
const fieldCallback =
|
|
||||||
(metricName: string, field: "count" | "sum") =>
|
|
||||||
(result: ObservableResult) => {
|
|
||||||
for (const point of snapshotLancedbMetrics()) {
|
|
||||||
if (point.name !== metricName) {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
const value = point[field];
|
|
||||||
if (value != null) {
|
|
||||||
result.observe(value, point.attributes);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
for (const desc of lancedbMetricsCatalog()) {
|
|
||||||
const unit = desc.unit ?? "";
|
|
||||||
if (desc.kind === "counter") {
|
|
||||||
const counter = meter.createObservableCounter(desc.name, {
|
|
||||||
unit,
|
|
||||||
description: desc.description,
|
|
||||||
});
|
|
||||||
counter.addCallback(scalarCallback(desc.name));
|
|
||||||
} else if (desc.kind === "gauge") {
|
|
||||||
const gauge = meter.createObservableGauge(desc.name, {
|
|
||||||
unit,
|
|
||||||
description: desc.description,
|
|
||||||
});
|
|
||||||
gauge.addCallback(scalarCallback(desc.name));
|
|
||||||
} else if (desc.kind === "histogram") {
|
|
||||||
// `_bucket` and `_count` observe cumulative sample counts, not the
|
|
||||||
// histogram's measured quantity, so they are unitless; only `_sum`
|
|
||||||
// carries the histogram's unit.
|
|
||||||
const bucket = meter.createObservableCounter(`${desc.name}_bucket`, {
|
|
||||||
description: `${desc.description} (cumulative buckets)`,
|
|
||||||
});
|
|
||||||
bucket.addCallback(bucketCallback(desc.name));
|
|
||||||
|
|
||||||
const count = meter.createObservableCounter(`${desc.name}_count`, {
|
|
||||||
description: `${desc.description} (count)`,
|
|
||||||
});
|
|
||||||
count.addCallback(fieldCallback(desc.name, "count"));
|
|
||||||
|
|
||||||
const sum = meter.createObservableCounter(`${desc.name}_sum`, {
|
|
||||||
unit,
|
|
||||||
description: `${desc.description} (sum)`,
|
|
||||||
});
|
|
||||||
sum.addCallback(fieldCallback(desc.name, "sum"));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
instrumented = true;
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
+3
-53
@@ -79,8 +79,6 @@ export interface QueryExecutionOptions {
|
|||||||
timeoutMs?: number;
|
timeoutMs?: number;
|
||||||
}
|
}
|
||||||
|
|
||||||
export type AnalyzePlanDistributedMetrics = "aggregate" | "per_worker" | "full";
|
|
||||||
|
|
||||||
export interface ColumnOrdering {
|
export interface ColumnOrdering {
|
||||||
columnName: string;
|
columnName: string;
|
||||||
ascending?: boolean;
|
ascending?: boolean;
|
||||||
@@ -313,20 +311,13 @@ export class QueryBase<
|
|||||||
* KNNVectorDistance: metric=l2, metrics=[output_rows=1, elapsed_compute=114.333µs, output_batches=1]
|
* KNNVectorDistance: metric=l2, metrics=[output_rows=1, elapsed_compute=114.333µs, output_batches=1]
|
||||||
* LanceScan: uri=/path/to/data, projection=[vector], row_id=true, row_addr=false, ordered=false, metrics=[output_rows=1, elapsed_compute=103.626µs, bytes_read=549, iops=2, requests=2]
|
* LanceScan: uri=/path/to/data, projection=[vector], row_id=true, row_addr=false, ordered=false, metrics=[output_rows=1, elapsed_compute=103.626µs, bytes_read=549, iops=2, requests=2]
|
||||||
*
|
*
|
||||||
* @param distributedMetrics - How distributed worker metrics are displayed for remote query plans.
|
|
||||||
* Defaults to `"aggregate"`.
|
|
||||||
* @returns A query execution plan with runtime metrics for each step.
|
* @returns A query execution plan with runtime metrics for each step.
|
||||||
*/
|
*/
|
||||||
async analyzePlan(
|
async analyzePlan(): Promise<string> {
|
||||||
distributedMetrics?: AnalyzePlanDistributedMetrics,
|
|
||||||
): Promise<string> {
|
|
||||||
const distributedMetricsMode = distributedMetrics ?? "aggregate";
|
|
||||||
if (this.inner instanceof Promise) {
|
if (this.inner instanceof Promise) {
|
||||||
return this.inner.then((inner) =>
|
return this.inner.then((inner) => inner.analyzePlan());
|
||||||
inner.analyzePlan(distributedMetricsMode),
|
|
||||||
);
|
|
||||||
} else {
|
} else {
|
||||||
return this.inner.analyzePlan(distributedMetricsMode);
|
return this.inner.analyzePlan();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -371,9 +362,6 @@ export class StandardQueryBase<
|
|||||||
*
|
*
|
||||||
* Filtering performance can often be improved by creating a scalar index
|
* Filtering performance can often be improved by creating a scalar index
|
||||||
* on the filter column(s).
|
* on the filter column(s).
|
||||||
*
|
|
||||||
* Calling this multiple times combines the filters with a logical AND rather
|
|
||||||
* than replacing the previous filter.
|
|
||||||
*/
|
*/
|
||||||
where(predicate: string): this {
|
where(predicate: string): this {
|
||||||
this.doCall((inner: NativeQueryType) => inner.onlyIf(predicate));
|
this.doCall((inner: NativeQueryType) => inner.onlyIf(predicate));
|
||||||
@@ -460,30 +448,6 @@ export class StandardQueryBase<
|
|||||||
this.doCall((inner: NativeQueryType) => inner.fastSearch());
|
this.doCall((inner: NativeQueryType) => inner.fastSearch());
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Control MemWAL read routing for this query.
|
|
||||||
*
|
|
||||||
* By default (unset), when the table carries a MemWAL write spec (see
|
|
||||||
* {@link Table#setLsmWriteSpec}), reads are routed through the LSM scanner so
|
|
||||||
* they also return data written via the `mergeInsert` LSM path that has not yet
|
|
||||||
* been compacted into the base table (the active/frozen in-memory memtables and
|
|
||||||
* the flushed generations), deduplicated by primary key; a table without a spec
|
|
||||||
* reads the base table.
|
|
||||||
*
|
|
||||||
* @param enable - `true` forces the LSM scanner and errors if the table has no
|
|
||||||
* MemWAL write spec. `false` bypasses the MemWAL and reads the base table only,
|
|
||||||
* even when a spec is present.
|
|
||||||
*
|
|
||||||
* Note: the LSM scanner does not support every query shape (e.g. reranking,
|
|
||||||
* hybrid search, `orderBy`). On a MemWAL table those shapes error unless
|
|
||||||
* `useLsm(false)` is set, because a base-only read would silently exclude
|
|
||||||
* un-compacted MemWAL data.
|
|
||||||
*/
|
|
||||||
useLsm(enable: boolean): this {
|
|
||||||
this.doCall((inner: NativeQueryType) => inner.useLsm(enable));
|
|
||||||
return this;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -772,20 +736,6 @@ export class TakeQuery extends QueryBase<NativeTakeQuery> {
|
|||||||
constructor(inner: NativeTakeQuery) {
|
constructor(inner: NativeTakeQuery) {
|
||||||
super(inner);
|
super(inner);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Control MemWAL read routing for this take query.
|
|
||||||
*
|
|
||||||
* `false` bypasses the MemWAL and reads the base table only — the escape hatch,
|
|
||||||
* since take-by-row-id/offset is not supported on the LSM scanner and, on a
|
|
||||||
* MemWAL table, auto-routes to it and errors otherwise.
|
|
||||||
*
|
|
||||||
* @param enable - `false` reads the base table only.
|
|
||||||
*/
|
|
||||||
useLsm(enable: boolean): this {
|
|
||||||
this.doCall((inner: NativeTakeQuery) => inner.useLsm(enable));
|
|
||||||
return this;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/** A builder for LanceDB queries.
|
/** A builder for LanceDB queries.
|
||||||
|
|||||||
@@ -84,7 +84,7 @@ export function sanitizeMetadata(
|
|||||||
throw Error("Expected metadata, if present, to be a Map<string, string>");
|
throw Error("Expected metadata, if present, to be a Map<string, string>");
|
||||||
}
|
}
|
||||||
for (const item of metadataLike) {
|
for (const item of metadataLike) {
|
||||||
if (typeof item[0] !== "string" || typeof item[1] !== "string") {
|
if (!(typeof item[0] === "string" || !(typeof item[1] === "string"))) {
|
||||||
throw Error(
|
throw Error(
|
||||||
"Expected metadata, if present, to be a Map<string, string> but it had non-string keys or values",
|
"Expected metadata, if present, to be a Map<string, string> but it had non-string keys or values",
|
||||||
);
|
);
|
||||||
@@ -288,11 +288,12 @@ export function sanitizeMap(typeLike: object) {
|
|||||||
if (!("keysSorted" in typeLike) || typeof typeLike.keysSorted !== "boolean") {
|
if (!("keysSorted" in typeLike) || typeof typeLike.keysSorted !== "boolean") {
|
||||||
throw Error("Expected a Map type to have a `keysSorted` property");
|
throw Error("Expected a Map type to have a `keysSorted` property");
|
||||||
}
|
}
|
||||||
if (typeLike.children.length !== 1) {
|
|
||||||
throw Error("Expected a Map type to have exactly one child");
|
|
||||||
}
|
|
||||||
|
|
||||||
return new Map_(sanitizeField(typeLike.children[0]), typeLike.keysSorted);
|
return new Map_(
|
||||||
|
// biome-ignore lint/suspicious/noExplicitAny: skip
|
||||||
|
typeLike.children.map((field) => sanitizeField(field)) as any,
|
||||||
|
typeLike.keysSorted,
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
export function sanitizeDuration(typeLike: object) {
|
export function sanitizeDuration(typeLike: object) {
|
||||||
|
|||||||
+1
-291
@@ -25,16 +25,13 @@ import {
|
|||||||
AddColumnsSql,
|
AddColumnsSql,
|
||||||
AddResult,
|
AddResult,
|
||||||
AlterColumnsResult,
|
AlterColumnsResult,
|
||||||
BranchContents,
|
|
||||||
DeleteResult,
|
DeleteResult,
|
||||||
DropColumnsResult,
|
DropColumnsResult,
|
||||||
IndexConfig,
|
IndexConfig,
|
||||||
IndexStatistics,
|
IndexStatistics,
|
||||||
Branches as NativeBranches,
|
|
||||||
OptimizeStats,
|
OptimizeStats,
|
||||||
TableStatistics,
|
TableStatistics,
|
||||||
Tags,
|
Tags,
|
||||||
UpdateFieldMetadataResult,
|
|
||||||
UpdateResult,
|
UpdateResult,
|
||||||
Table as _NativeTable,
|
Table as _NativeTable,
|
||||||
} from "./native";
|
} from "./native";
|
||||||
@@ -158,36 +155,13 @@ export interface Version {
|
|||||||
metadata: Record<string, string>;
|
metadata: Record<string, string>;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Token produced by the tokenizer configured on a full-text search index. */
|
|
||||||
export interface FtsToken {
|
|
||||||
/** Token text after tokenizer filters have been applied. */
|
|
||||||
text: string;
|
|
||||||
/** Token position used by full-text query matching. */
|
|
||||||
position: number;
|
|
||||||
}
|
|
||||||
|
|
||||||
export type TokenizeTableOptions =
|
|
||||||
| {
|
|
||||||
/** FTS-indexed column whose tokenizer should be used. */
|
|
||||||
column: string;
|
|
||||||
indexName?: never;
|
|
||||||
}
|
|
||||||
| {
|
|
||||||
/** Name of the FTS index whose tokenizer should be used. */
|
|
||||||
indexName: string;
|
|
||||||
column?: never;
|
|
||||||
};
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Specification selecting Lance's MemWAL LSM-style write path for
|
* Specification selecting Lance's MemWAL LSM-style write path for
|
||||||
* `mergeInsert`.
|
* `mergeInsert`.
|
||||||
*
|
*
|
||||||
* `specType` is `"bucket"`, `"identity"`, or `"unsharded"`. For `"bucket"`,
|
* `specType` is `"bucket"`, `"identity"`, or `"unsharded"`. For `"bucket"`,
|
||||||
* `column` and `numBuckets` are required; for `"identity"`, `column` is
|
* `column` and `numBuckets` are required; for `"identity"`, `column` is
|
||||||
* required and must be a deterministic function of the unenforced primary
|
* required.
|
||||||
* key (every row with a given primary key must always produce the same
|
|
||||||
* `column` value, or upserts of that key can land in different shards and a
|
|
||||||
* stale version can win).
|
|
||||||
*/
|
*/
|
||||||
export interface LsmWriteSpec {
|
export interface LsmWriteSpec {
|
||||||
/** One of `"bucket"`, `"identity"`, or `"unsharded"`. */
|
/** One of `"bucket"`, `"identity"`, or `"unsharded"`. */
|
||||||
@@ -531,18 +505,6 @@ export abstract class Table {
|
|||||||
abstract alterColumns(
|
abstract alterColumns(
|
||||||
columnAlterations: ColumnAlteration[],
|
columnAlterations: ColumnAlteration[],
|
||||||
): Promise<AlterColumnsResult>;
|
): Promise<AlterColumnsResult>;
|
||||||
|
|
||||||
/**
|
|
||||||
* Update per-field (column) metadata.
|
|
||||||
* @param {FieldMetadataUpdate[]} updates One or more per-field updates. Each
|
|
||||||
* update's metadata is merged into the field's existing metadata by default;
|
|
||||||
* a value of `null` deletes that key, and `replace: true` swaps the whole map.
|
|
||||||
* @returns {Promise<UpdateFieldMetadataResult>} resolves to the new table version.
|
|
||||||
*/
|
|
||||||
abstract updateFieldMetadata(
|
|
||||||
updates: FieldMetadataUpdate[],
|
|
||||||
): Promise<UpdateFieldMetadataResult>;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Drop one or more columns from the dataset
|
* Drop one or more columns from the dataset
|
||||||
*
|
*
|
||||||
@@ -605,27 +567,6 @@ export abstract class Table {
|
|||||||
* @returns {Promise<void>}
|
* @returns {Promise<void>}
|
||||||
*/
|
*/
|
||||||
abstract unsetLsmWriteSpec(): Promise<void>;
|
abstract unsetLsmWriteSpec(): Promise<void>;
|
||||||
/**
|
|
||||||
* Read the {@link LsmWriteSpec} currently installed on this table.
|
|
||||||
*
|
|
||||||
* Resolves to `undefined` when the MemWAL LSM write path is not enabled (no
|
|
||||||
* spec has been set, or it was removed with {@link Table#unsetLsmWriteSpec}).
|
|
||||||
* The returned spec — including its `maintainedIndexes` and
|
|
||||||
* `writerConfigDefaults` — mirrors what was passed to
|
|
||||||
* {@link Table#setLsmWriteSpec}.
|
|
||||||
* @returns {Promise<LsmWriteSpec | undefined>}
|
|
||||||
*/
|
|
||||||
abstract getLsmWriteSpec(): Promise<LsmWriteSpec | undefined>;
|
|
||||||
/**
|
|
||||||
* Drain and close any cached MemWAL shard writers held for this table.
|
|
||||||
*
|
|
||||||
* When an {@link LsmWriteSpec} is installed, `mergeInsert` opens MemWAL
|
|
||||||
* shard writers and caches them for reuse across calls. This closes them,
|
|
||||||
* flushing pending data; writers reopen lazily on the next `mergeInsert`.
|
|
||||||
* It is a no-op when no writers are cached.
|
|
||||||
* @returns {Promise<void>}
|
|
||||||
*/
|
|
||||||
abstract closeLsmWriters(): Promise<void>;
|
|
||||||
/** Retrieve the version of the table */
|
/** Retrieve the version of the table */
|
||||||
|
|
||||||
abstract version(): Promise<number>;
|
abstract version(): Promise<number>;
|
||||||
@@ -686,22 +627,6 @@ export abstract class Table {
|
|||||||
*/
|
*/
|
||||||
abstract tags(): Promise<Tags>;
|
abstract tags(): Promise<Tags>;
|
||||||
|
|
||||||
/**
|
|
||||||
* Get the branch manager for this table.
|
|
||||||
*
|
|
||||||
* Branches are isolated, writable lines of history forked from another
|
|
||||||
* branch (or version). Writes on a branch do not affect `main`.
|
|
||||||
*/
|
|
||||||
abstract branches(): Promise<Branches>;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* The branch this table handle is scoped to, or `null` for the main branch.
|
|
||||||
*
|
|
||||||
* A handle returned by {@link Branches.create} or {@link Branches.checkout}
|
|
||||||
* reports the branch it targets; a handle opened normally reports `null`.
|
|
||||||
*/
|
|
||||||
abstract currentBranch(): string | null;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Restore the table to the currently checked out version
|
* Restore the table to the currently checked out version
|
||||||
*
|
*
|
||||||
@@ -736,19 +661,6 @@ export abstract class Table {
|
|||||||
abstract optimize(options?: Partial<OptimizeOptions>): Promise<OptimizeStats>;
|
abstract optimize(options?: Partial<OptimizeOptions>): Promise<OptimizeStats>;
|
||||||
/** List all indices that have been created with {@link Table.createIndex} */
|
/** List all indices that have been created with {@link Table.createIndex} */
|
||||||
abstract listIndices(): Promise<IndexConfig[]>;
|
abstract listIndices(): Promise<IndexConfig[]>;
|
||||||
/**
|
|
||||||
* Tokenize a full-text search query using the tokenizer configured on an FTS index.
|
|
||||||
*
|
|
||||||
* Specify exactly one of `column` or `indexName`.
|
|
||||||
*
|
|
||||||
* Model-backed tokenizers such as `jieba/*` and `lindera/*` are rebuilt in
|
|
||||||
* the client process from index metadata. For remote tables, this means the
|
|
||||||
* same tokenizer model files must also exist locally.
|
|
||||||
*/
|
|
||||||
abstract tokenize(
|
|
||||||
query: string,
|
|
||||||
options: TokenizeTableOptions,
|
|
||||||
): Promise<FtsToken[]>;
|
|
||||||
/** Return the table as an arrow table */
|
/** Return the table as an arrow table */
|
||||||
abstract toArrow(): Promise<ArrowTable>;
|
abstract toArrow(): Promise<ArrowTable>;
|
||||||
|
|
||||||
@@ -1112,12 +1024,6 @@ export class LocalTable extends Table {
|
|||||||
return await this.inner.alterColumns(processedAlterations);
|
return await this.inner.alterColumns(processedAlterations);
|
||||||
}
|
}
|
||||||
|
|
||||||
async updateFieldMetadata(
|
|
||||||
updates: FieldMetadataUpdate[],
|
|
||||||
): Promise<UpdateFieldMetadataResult> {
|
|
||||||
return await this.inner.updateFieldMetadata(updates);
|
|
||||||
}
|
|
||||||
|
|
||||||
async dropColumns(columnNames: string[]): Promise<DropColumnsResult> {
|
async dropColumns(columnNames: string[]): Promise<DropColumnsResult> {
|
||||||
return await this.inner.dropColumns(columnNames);
|
return await this.inner.dropColumns(columnNames);
|
||||||
}
|
}
|
||||||
@@ -1135,19 +1041,6 @@ export class LocalTable extends Table {
|
|||||||
return await this.inner.unsetLsmWriteSpec();
|
return await this.inner.unsetLsmWriteSpec();
|
||||||
}
|
}
|
||||||
|
|
||||||
async getLsmWriteSpec(): Promise<LsmWriteSpec | undefined> {
|
|
||||||
// The native binding types `specType` as a plain `string`; narrow it back
|
|
||||||
// to the public union. The Rust `From` impl only ever emits one of the
|
|
||||||
// three valid values, so the cast is safe.
|
|
||||||
return ((await this.inner.getLsmWriteSpec()) ?? undefined) as
|
|
||||||
| LsmWriteSpec
|
|
||||||
| undefined;
|
|
||||||
}
|
|
||||||
|
|
||||||
async closeLsmWriters(): Promise<void> {
|
|
||||||
return await this.inner.closeLsmWriters();
|
|
||||||
}
|
|
||||||
|
|
||||||
async version(): Promise<number> {
|
async version(): Promise<number> {
|
||||||
return await this.inner.version();
|
return await this.inner.version();
|
||||||
}
|
}
|
||||||
@@ -1179,14 +1072,6 @@ export class LocalTable extends Table {
|
|||||||
return await this.inner.tags();
|
return await this.inner.tags();
|
||||||
}
|
}
|
||||||
|
|
||||||
async branches(): Promise<Branches> {
|
|
||||||
return new Branches(await this.inner.branches());
|
|
||||||
}
|
|
||||||
|
|
||||||
currentBranch(): string | null {
|
|
||||||
return this.inner.currentBranch() ?? null;
|
|
||||||
}
|
|
||||||
|
|
||||||
async optimize(options?: Partial<OptimizeOptions>): Promise<OptimizeStats> {
|
async optimize(options?: Partial<OptimizeOptions>): Promise<OptimizeStats> {
|
||||||
let cleanupOlderThanMs;
|
let cleanupOlderThanMs;
|
||||||
if (
|
if (
|
||||||
@@ -1206,17 +1091,6 @@ export class LocalTable extends Table {
|
|||||||
return await this.inner.listIndices();
|
return await this.inner.listIndices();
|
||||||
}
|
}
|
||||||
|
|
||||||
async tokenize(
|
|
||||||
query: string,
|
|
||||||
options: TokenizeTableOptions,
|
|
||||||
): Promise<FtsToken[]> {
|
|
||||||
return await this.inner.tokenize(
|
|
||||||
query,
|
|
||||||
options?.column,
|
|
||||||
options?.indexName,
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
async toArrow(): Promise<ArrowTable> {
|
async toArrow(): Promise<ArrowTable> {
|
||||||
return await this.query().toArrow();
|
return await this.query().toArrow();
|
||||||
}
|
}
|
||||||
@@ -1312,167 +1186,3 @@ export interface ColumnAlteration {
|
|||||||
/** Set the new nullability. Note that a nullable column cannot be made non-nullable. */
|
/** Set the new nullability. Note that a nullable column cannot be made non-nullable. */
|
||||||
nullable?: boolean;
|
nullable?: boolean;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** A per-field metadata update, addressed by dot-path. */
|
|
||||||
export interface FieldMetadataUpdate {
|
|
||||||
/**
|
|
||||||
* Dot-separated path to the field. For a top-level column this is just its
|
|
||||||
* name; for a nested field it's the path, e.g. "a.b.c".
|
|
||||||
*/
|
|
||||||
path: string;
|
|
||||||
/**
|
|
||||||
* Metadata key/value pairs. Merged into the field's existing metadata by
|
|
||||||
* default; a value of `null` deletes that key.
|
|
||||||
*/
|
|
||||||
metadata: Record<string, string | null>;
|
|
||||||
/** If true, replace the field's entire metadata map instead of merging. */
|
|
||||||
replace?: boolean;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Summary of a column in a branch diff. */
|
|
||||||
export interface BranchColumnSummary {
|
|
||||||
name: string;
|
|
||||||
dataType: string;
|
|
||||||
nullable: boolean;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** A column whose definition differs between main and the branch. */
|
|
||||||
export interface BranchColumnChange {
|
|
||||||
name: string;
|
|
||||||
main: BranchColumnSummary;
|
|
||||||
branch: BranchColumnSummary;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Summary of an index in a branch diff. */
|
|
||||||
export interface BranchIndexSummary {
|
|
||||||
indexName: string;
|
|
||||||
columns: string[];
|
|
||||||
indexType?: string;
|
|
||||||
status: string;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Row-level comparison between main and the branch. */
|
|
||||||
export interface BranchRowCountSummary {
|
|
||||||
unchanged: number;
|
|
||||||
newOnBase: number;
|
|
||||||
newOnBranch: number;
|
|
||||||
staleRecompute: number;
|
|
||||||
inputsChanged: number;
|
|
||||||
deltaAvailable: boolean;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** A reason why a branch cannot currently be merged. */
|
|
||||||
export interface MergeBlocker {
|
|
||||||
code: string;
|
|
||||||
message: string;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Read-only comparison of a branch against main. */
|
|
||||||
export interface BranchDiff {
|
|
||||||
fromBranch: string;
|
|
||||||
parentVersion: number;
|
|
||||||
mainVersion: number;
|
|
||||||
branchVersion: number;
|
|
||||||
baseMoved: boolean;
|
|
||||||
rowCountMain: number;
|
|
||||||
rowCountBranch: number;
|
|
||||||
rowSummary: BranchRowCountSummary;
|
|
||||||
addedColumns: BranchColumnSummary[];
|
|
||||||
removedColumns: BranchColumnSummary[];
|
|
||||||
changedColumns: BranchColumnChange[];
|
|
||||||
addedIndexes: BranchIndexSummary[];
|
|
||||||
removedIndexes: BranchIndexSummary[];
|
|
||||||
mergeable: boolean;
|
|
||||||
mergeBlockers: MergeBlocker[];
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Changes that would be, or were, promoted by a branch merge. */
|
|
||||||
export interface MergePreview {
|
|
||||||
promotedColumns: string[];
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Result of previewing or attempting a branch merge. */
|
|
||||||
export interface MergeBranchResult {
|
|
||||||
status: "ready" | "rejected" | "notImplemented" | "merged" | "unknown";
|
|
||||||
diff: BranchDiff;
|
|
||||||
preview: MergePreview;
|
|
||||||
mainVersionAfter?: number;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Branch manager for a {@link Table}.
|
|
||||||
*
|
|
||||||
* Unlike tags, `create` and `checkout` return a new {@link Table} handle scoped
|
|
||||||
* to the branch; writes on it do not affect `main`.
|
|
||||||
*/
|
|
||||||
export class Branches {
|
|
||||||
#inner: NativeBranches;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Construct a Branches manager. Internal use only.
|
|
||||||
* @hidden
|
|
||||||
*/
|
|
||||||
constructor(inner: NativeBranches) {
|
|
||||||
this.#inner = inner;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** List all branches, mapping name to branch metadata. */
|
|
||||||
async list(): Promise<Record<string, BranchContents>> {
|
|
||||||
return await this.#inner.list();
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Create a branch and return a handle scoped to it.
|
|
||||||
*
|
|
||||||
* @param name Name of the new branch.
|
|
||||||
* @param fromRef Source branch to fork from. Defaults to `main`.
|
|
||||||
* @param fromVersion A specific version on `fromRef`. Defaults to latest.
|
|
||||||
*/
|
|
||||||
async create(
|
|
||||||
name: string,
|
|
||||||
fromRef?: string,
|
|
||||||
fromVersion?: number,
|
|
||||||
): Promise<Table> {
|
|
||||||
return new LocalTable(await this.#inner.create(name, fromRef, fromVersion));
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Check out an existing branch and return a handle scoped to it.
|
|
||||||
*
|
|
||||||
* With `version` set, the returned handle is pinned to that version of the
|
|
||||||
* branch (a read-only, detached view); otherwise it tracks the branch's
|
|
||||||
* latest and stays writable.
|
|
||||||
*/
|
|
||||||
async checkout(name: string, version?: number): Promise<Table> {
|
|
||||||
return new LocalTable(await this.#inner.checkout(name, version));
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Delete a branch. */
|
|
||||||
async delete(name: string): Promise<void> {
|
|
||||||
return await this.#inner.delete(name);
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Compare a branch against main without modifying either branch. */
|
|
||||||
async diff(fromBranch: string): Promise<BranchDiff> {
|
|
||||||
return (await this.#inner.diff(fromBranch)) as unknown as BranchDiff;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Merge a branch into main.
|
|
||||||
*
|
|
||||||
* Set `dryRun` to `true` to preview the merge. A rejected merge resolves
|
|
||||||
* with `status: "rejected"` instead of throwing.
|
|
||||||
*
|
|
||||||
* @param fromBranch Branch to merge from.
|
|
||||||
* @param dryRun When true, only preview the merge. Defaults to false.
|
|
||||||
*/
|
|
||||||
async merge(
|
|
||||||
fromBranch: string,
|
|
||||||
dryRun: boolean = false,
|
|
||||||
): Promise<MergeBranchResult> {
|
|
||||||
return (await this.#inner.merge(
|
|
||||||
fromBranch,
|
|
||||||
dryRun,
|
|
||||||
)) as unknown as MergeBranchResult;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-darwin-arm64",
|
"name": "@lancedb/lancedb-darwin-arm64",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.30.0",
|
||||||
"os": ["darwin"],
|
"os": ["darwin"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.darwin-arm64.node",
|
"main": "lancedb.darwin-arm64.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
||||||
"version": "0.37.1-beta.0",
|
"version": "0.30.0",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.linux-arm64-gnu.node",
|
"main": "lancedb.linux-arm64-gnu.node",
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user