mirror of
https://github.com/lancedb/lancedb.git
synced 2025-12-27 15:12:53 +00:00
Compare commits
35 Commits
python-v0.
...
python-v0.
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9d129c7e86 | ||
|
|
44878dd9a5 | ||
|
|
4b5bb2d76c | ||
|
|
434f4124fc | ||
|
|
03a1a99270 | ||
|
|
0110e3b6f8 | ||
|
|
f1f85b0a84 | ||
|
|
d6daa08b54 | ||
|
|
17b71de22e | ||
|
|
a250d8e7df | ||
|
|
5a2b33581e | ||
|
|
3d254f61b0 | ||
|
|
d15e380be1 | ||
|
|
0baf807be0 | ||
|
|
76bcc78910 | ||
|
|
135dfdc7ec | ||
|
|
6f39108857 | ||
|
|
bb6b0bea0c | ||
|
|
0084eb238b | ||
|
|
28ab29a3f0 | ||
|
|
7d3f5348a7 | ||
|
|
3531393523 | ||
|
|
93b8ac8e3e | ||
|
|
1b78ccedaf | ||
|
|
ca8d118f78 | ||
|
|
386fc9e466 | ||
|
|
ce1bafec1a | ||
|
|
8f18a7feed | ||
|
|
e2be699544 | ||
|
|
f77b0ef37d | ||
|
|
c41401f20f | ||
|
|
1cf3917a87 | ||
|
|
92dbec1f95 | ||
|
|
bbd44e669d | ||
|
|
e2d7640021 |
@@ -1,5 +1,5 @@
|
|||||||
[tool.bumpversion]
|
[tool.bumpversion]
|
||||||
current_version = "0.22.3"
|
current_version = "0.22.4-beta.2"
|
||||||
parse = """(?x)
|
parse = """(?x)
|
||||||
(?P<major>0|[1-9]\\d*)\\.
|
(?P<major>0|[1-9]\\d*)\\.
|
||||||
(?P<minor>0|[1-9]\\d*)\\.
|
(?P<minor>0|[1-9]\\d*)\\.
|
||||||
|
|||||||
@@ -19,7 +19,7 @@ rustflags = [
|
|||||||
"-Wclippy::string_add_assign",
|
"-Wclippy::string_add_assign",
|
||||||
"-Wclippy::string_add",
|
"-Wclippy::string_add",
|
||||||
"-Wclippy::string_lit_as_bytes",
|
"-Wclippy::string_lit_as_bytes",
|
||||||
"-Wclippy::string_to_string",
|
"-Wclippy::implicit_clone",
|
||||||
"-Wclippy::use_self",
|
"-Wclippy::use_self",
|
||||||
"-Dclippy::cargo",
|
"-Dclippy::cargo",
|
||||||
"-Dclippy::dbg_macro",
|
"-Dclippy::dbg_macro",
|
||||||
|
|||||||
2
.github/ISSUE_TEMPLATE/documentation.yml
vendored
2
.github/ISSUE_TEMPLATE/documentation.yml
vendored
@@ -18,6 +18,6 @@ body:
|
|||||||
label: Link
|
label: Link
|
||||||
description: >
|
description: >
|
||||||
Provide a link to the existing documentation, if applicable.
|
Provide a link to the existing documentation, if applicable.
|
||||||
placeholder: ex. https://lancedb.github.io/lancedb/guides/tables/...
|
placeholder: ex. https://lancedb.com/docs/tables/...
|
||||||
validations:
|
validations:
|
||||||
required: false
|
required: false
|
||||||
|
|||||||
@@ -31,7 +31,7 @@ runs:
|
|||||||
with:
|
with:
|
||||||
command: build
|
command: build
|
||||||
working-directory: python
|
working-directory: python
|
||||||
docker-options: "-e PIP_EXTRA_INDEX_URL=https://pypi.fury.io/lancedb/"
|
docker-options: "-e PIP_EXTRA_INDEX_URL='https://pypi.fury.io/lance-format/ https://pypi.fury.io/lancedb/'"
|
||||||
target: x86_64-unknown-linux-gnu
|
target: x86_64-unknown-linux-gnu
|
||||||
manylinux: ${{ inputs.manylinux }}
|
manylinux: ${{ inputs.manylinux }}
|
||||||
args: ${{ inputs.args }}
|
args: ${{ inputs.args }}
|
||||||
@@ -46,7 +46,7 @@ runs:
|
|||||||
with:
|
with:
|
||||||
command: build
|
command: build
|
||||||
working-directory: python
|
working-directory: python
|
||||||
docker-options: "-e PIP_EXTRA_INDEX_URL=https://pypi.fury.io/lancedb/"
|
docker-options: "-e PIP_EXTRA_INDEX_URL='https://pypi.fury.io/lance-format/ https://pypi.fury.io/lancedb/'"
|
||||||
target: aarch64-unknown-linux-gnu
|
target: aarch64-unknown-linux-gnu
|
||||||
manylinux: ${{ inputs.manylinux }}
|
manylinux: ${{ inputs.manylinux }}
|
||||||
args: ${{ inputs.args }}
|
args: ${{ inputs.args }}
|
||||||
|
|||||||
2
.github/workflows/build_mac_wheel/action.yml
vendored
2
.github/workflows/build_mac_wheel/action.yml
vendored
@@ -22,5 +22,5 @@ runs:
|
|||||||
command: build
|
command: build
|
||||||
# TODO: pass through interpreter
|
# TODO: pass through interpreter
|
||||||
args: ${{ inputs.args }}
|
args: ${{ inputs.args }}
|
||||||
docker-options: "-e PIP_EXTRA_INDEX_URL=https://pypi.fury.io/lancedb/"
|
docker-options: "-e PIP_EXTRA_INDEX_URL='https://pypi.fury.io/lance-format/ https://pypi.fury.io/lancedb/'"
|
||||||
working-directory: python
|
working-directory: python
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ runs:
|
|||||||
with:
|
with:
|
||||||
command: build
|
command: build
|
||||||
args: ${{ inputs.args }}
|
args: ${{ inputs.args }}
|
||||||
docker-options: "-e PIP_EXTRA_INDEX_URL=https://pypi.fury.io/lancedb/"
|
docker-options: "-e PIP_EXTRA_INDEX_URL='https://pypi.fury.io/lance-format/ https://pypi.fury.io/lancedb/'"
|
||||||
working-directory: python
|
working-directory: python
|
||||||
- uses: actions/upload-artifact@v4
|
- uses: actions/upload-artifact@v4
|
||||||
with:
|
with:
|
||||||
|
|||||||
@@ -63,27 +63,18 @@ jobs:
|
|||||||
git config user.name "lancedb automation"
|
git config user.name "lancedb automation"
|
||||||
git config user.email "robot@lancedb.com"
|
git config user.email "robot@lancedb.com"
|
||||||
|
|
||||||
- name: Configure Codex authentication
|
|
||||||
env:
|
|
||||||
CODEX_TOKEN_B64: ${{ secrets.CODEX_TOKEN }}
|
|
||||||
run: |
|
|
||||||
if [ -z "${CODEX_TOKEN_B64}" ]; then
|
|
||||||
echo "Repository secret CODEX_TOKEN is not defined; skipping Codex execution."
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
mkdir -p ~/.codex
|
|
||||||
echo "${CODEX_TOKEN_B64}" | base64 --decode > ~/.codex/auth.json
|
|
||||||
|
|
||||||
- name: Run Codex to update Lance dependency
|
- name: Run Codex to update Lance dependency
|
||||||
env:
|
env:
|
||||||
TAG: ${{ inputs.tag }}
|
TAG: ${{ inputs.tag }}
|
||||||
GITHUB_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
GITHUB_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
||||||
GH_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
GH_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
||||||
|
OPENAI_API_KEY: ${{ secrets.CODEX_TOKEN }}
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
VERSION="${TAG#refs/tags/}"
|
VERSION="${TAG#refs/tags/}"
|
||||||
VERSION="${VERSION#v}"
|
VERSION="${VERSION#v}"
|
||||||
BRANCH_NAME="codex/update-lance-${VERSION//[^a-zA-Z0-9]/-}"
|
BRANCH_NAME="codex/update-lance-${VERSION//[^a-zA-Z0-9]/-}"
|
||||||
|
|
||||||
cat <<EOF >/tmp/codex-prompt.txt
|
cat <<EOF >/tmp/codex-prompt.txt
|
||||||
You are running inside the lancedb repository on a GitHub Actions runner. Update the Lance dependency to version ${VERSION} and prepare a pull request for maintainers to review.
|
You are running inside the lancedb repository on a GitHub Actions runner. Update the Lance dependency to version ${VERSION} and prepare a pull request for maintainers to review.
|
||||||
|
|
||||||
@@ -104,4 +95,33 @@ jobs:
|
|||||||
- Do not merge the PR.
|
- Do not merge the PR.
|
||||||
- If any command fails, diagnose and fix the issue instead of aborting.
|
- If any command fails, diagnose and fix the issue instead of aborting.
|
||||||
EOF
|
EOF
|
||||||
|
|
||||||
|
printenv OPENAI_API_KEY | codex login --with-api-key
|
||||||
codex --config shell_environment_policy.ignore_default_excludes=true exec --dangerously-bypass-approvals-and-sandbox "$(cat /tmp/codex-prompt.txt)"
|
codex --config shell_environment_policy.ignore_default_excludes=true exec --dangerously-bypass-approvals-and-sandbox "$(cat /tmp/codex-prompt.txt)"
|
||||||
|
|
||||||
|
- name: Trigger sophon dependency update
|
||||||
|
env:
|
||||||
|
TAG: ${{ inputs.tag }}
|
||||||
|
GH_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
VERSION="${TAG#refs/tags/}"
|
||||||
|
VERSION="${VERSION#v}"
|
||||||
|
LANCEDB_BRANCH="codex/update-lance-${VERSION//[^a-zA-Z0-9]/-}"
|
||||||
|
|
||||||
|
echo "Triggering sophon workflow with:"
|
||||||
|
echo " lance_ref: ${TAG#refs/tags/}"
|
||||||
|
echo " lancedb_ref: ${LANCEDB_BRANCH}"
|
||||||
|
|
||||||
|
gh workflow run codex-bump-lancedb-lance.yml \
|
||||||
|
--repo lancedb/sophon \
|
||||||
|
-f lance_ref="${TAG#refs/tags/}" \
|
||||||
|
-f lancedb_ref="${LANCEDB_BRANCH}"
|
||||||
|
|
||||||
|
- name: Show latest sophon workflow run
|
||||||
|
env:
|
||||||
|
GH_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
echo "Latest sophon workflow run:"
|
||||||
|
gh run list --repo lancedb/sophon --workflow codex-bump-lancedb-lance.yml --limit 1 --json databaseId,url,displayTitle
|
||||||
|
|||||||
6
.github/workflows/docs.yml
vendored
6
.github/workflows/docs.yml
vendored
@@ -24,7 +24,7 @@ env:
|
|||||||
# according to: https://matklad.github.io/2021/09/04/fast-rust-builds.html
|
# according to: https://matklad.github.io/2021/09/04/fast-rust-builds.html
|
||||||
# CI builds are faster with incremental disabled.
|
# CI builds are faster with incremental disabled.
|
||||||
CARGO_INCREMENTAL: "0"
|
CARGO_INCREMENTAL: "0"
|
||||||
PIP_EXTRA_INDEX_URL: "https://pypi.fury.io/lancedb/"
|
PIP_EXTRA_INDEX_URL: "https://pypi.fury.io/lance-format/ https://pypi.fury.io/lancedb/"
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
# Single deploy job since we're just deploying
|
# Single deploy job since we're just deploying
|
||||||
@@ -50,8 +50,8 @@ jobs:
|
|||||||
- name: Build Python
|
- name: Build Python
|
||||||
working-directory: python
|
working-directory: python
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --extra-index-url https://pypi.fury.io/lancedb/ -e .
|
python -m pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -e .
|
||||||
python -m pip install --extra-index-url https://pypi.fury.io/lancedb/ -r ../docs/requirements.txt
|
python -m pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -r ../docs/requirements.txt
|
||||||
- name: Set up node
|
- name: Set up node
|
||||||
uses: actions/setup-node@v3
|
uses: actions/setup-node@v3
|
||||||
with:
|
with:
|
||||||
|
|||||||
62
.github/workflows/lance-release-timer.yml
vendored
Normal file
62
.github/workflows/lance-release-timer.yml
vendored
Normal file
@@ -0,0 +1,62 @@
|
|||||||
|
name: Lance Release Timer
|
||||||
|
|
||||||
|
on:
|
||||||
|
schedule:
|
||||||
|
- cron: "*/10 * * * *"
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
actions: write
|
||||||
|
|
||||||
|
concurrency:
|
||||||
|
group: lance-release-timer
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
trigger-update:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Checkout repository
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Check for new Lance tag
|
||||||
|
id: check
|
||||||
|
env:
|
||||||
|
GH_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
||||||
|
run: |
|
||||||
|
python3 ci/check_lance_release.py --github-output "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
|
- name: Look for existing PR
|
||||||
|
if: steps.check.outputs.needs_update == 'true'
|
||||||
|
id: pr
|
||||||
|
env:
|
||||||
|
GH_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
TITLE="chore: update lance dependency to v${{ steps.check.outputs.latest_version }}"
|
||||||
|
COUNT=$(gh pr list --search "\"$TITLE\" in:title" --state open --limit 1 --json number --jq 'length')
|
||||||
|
if [ "$COUNT" -gt 0 ]; then
|
||||||
|
echo "Open PR already exists for $TITLE"
|
||||||
|
echo "pr_exists=true" >> "$GITHUB_OUTPUT"
|
||||||
|
else
|
||||||
|
echo "No existing PR for $TITLE"
|
||||||
|
echo "pr_exists=false" >> "$GITHUB_OUTPUT"
|
||||||
|
fi
|
||||||
|
|
||||||
|
- name: Trigger codex update workflow
|
||||||
|
if: steps.check.outputs.needs_update == 'true' && steps.pr.outputs.pr_exists != 'true'
|
||||||
|
env:
|
||||||
|
GH_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
TAG=${{ steps.check.outputs.latest_tag }}
|
||||||
|
gh workflow run codex-update-lance-dependency.yml -f tag=refs/tags/$TAG
|
||||||
|
|
||||||
|
- name: Show latest codex workflow run
|
||||||
|
if: steps.check.outputs.needs_update == 'true' && steps.pr.outputs.pr_exists != 'true'
|
||||||
|
env:
|
||||||
|
GH_TOKEN: ${{ secrets.ROBOT_TOKEN }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
gh run list --workflow codex-update-lance-dependency.yml --limit 1 --json databaseId,url,displayTitle
|
||||||
25
.github/workflows/nodejs.yml
vendored
25
.github/workflows/nodejs.yml
vendored
@@ -16,9 +16,6 @@ concurrency:
|
|||||||
cancel-in-progress: true
|
cancel-in-progress: true
|
||||||
|
|
||||||
env:
|
env:
|
||||||
# Disable full debug symbol generation to speed up CI build and keep memory down
|
|
||||||
# "1" means line tables only, which is useful for panic tracebacks.
|
|
||||||
RUSTFLAGS: "-C debuginfo=1"
|
|
||||||
RUST_BACKTRACE: "1"
|
RUST_BACKTRACE: "1"
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
@@ -43,18 +40,20 @@ jobs:
|
|||||||
node-version: 20
|
node-version: 20
|
||||||
cache: 'npm'
|
cache: 'npm'
|
||||||
cache-dependency-path: nodejs/package-lock.json
|
cache-dependency-path: nodejs/package-lock.json
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: actions-rust-lang/setup-rust-toolchain@v1
|
||||||
|
with:
|
||||||
|
components: rustfmt, clippy
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: |
|
run: |
|
||||||
sudo apt update
|
sudo apt update
|
||||||
sudo apt install -y protobuf-compiler libssl-dev
|
sudo apt install -y protobuf-compiler libssl-dev
|
||||||
- uses: actions-rust-lang/setup-rust-toolchain@v1
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
- name: Format Rust
|
||||||
components: rustfmt, clippy
|
run: cargo fmt --all -- --check
|
||||||
- name: Lint
|
- name: Lint Rust
|
||||||
|
run: cargo clippy --profile ci --all --all-features -- -D warnings
|
||||||
|
- name: Lint Typescript
|
||||||
run: |
|
run: |
|
||||||
cargo fmt --all -- --check
|
|
||||||
cargo clippy --all --all-features -- -D warnings
|
|
||||||
npm ci
|
npm ci
|
||||||
npm run lint-ci
|
npm run lint-ci
|
||||||
- name: Lint examples
|
- name: Lint examples
|
||||||
@@ -90,7 +89,8 @@ jobs:
|
|||||||
- name: Build
|
- name: Build
|
||||||
run: |
|
run: |
|
||||||
npm ci
|
npm ci
|
||||||
npm run build
|
npm run build:debug -- --profile ci
|
||||||
|
npm run tsc
|
||||||
- name: Setup localstack
|
- name: Setup localstack
|
||||||
working-directory: .
|
working-directory: .
|
||||||
run: docker compose up --detach --wait
|
run: docker compose up --detach --wait
|
||||||
@@ -147,7 +147,8 @@ jobs:
|
|||||||
- name: Build
|
- name: Build
|
||||||
run: |
|
run: |
|
||||||
npm ci
|
npm ci
|
||||||
npm run build
|
npm run build:debug -- --profile ci
|
||||||
|
npm run tsc
|
||||||
- name: Test
|
- name: Test
|
||||||
run: |
|
run: |
|
||||||
npm run test
|
npm run test
|
||||||
|
|||||||
4
.github/workflows/pypi-publish.yml
vendored
4
.github/workflows/pypi-publish.yml
vendored
@@ -11,7 +11,7 @@ on:
|
|||||||
- Cargo.toml # Change in dependency frequently breaks builds
|
- Cargo.toml # Change in dependency frequently breaks builds
|
||||||
|
|
||||||
env:
|
env:
|
||||||
PIP_EXTRA_INDEX_URL: "https://pypi.fury.io/lancedb/"
|
PIP_EXTRA_INDEX_URL: "https://pypi.fury.io/lance-format/ https://pypi.fury.io/lancedb/"
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
linux:
|
linux:
|
||||||
@@ -65,7 +65,7 @@ jobs:
|
|||||||
matrix:
|
matrix:
|
||||||
config:
|
config:
|
||||||
- target: x86_64-apple-darwin
|
- target: x86_64-apple-darwin
|
||||||
runner: macos-13
|
runner: macos-15-large
|
||||||
- target: aarch64-apple-darwin
|
- target: aarch64-apple-darwin
|
||||||
runner: warp-macos-14-arm64-6x
|
runner: warp-macos-14-arm64-6x
|
||||||
env:
|
env:
|
||||||
|
|||||||
29
.github/workflows/python.yml
vendored
29
.github/workflows/python.yml
vendored
@@ -18,7 +18,8 @@ env:
|
|||||||
# Color output for pytest is off by default.
|
# Color output for pytest is off by default.
|
||||||
PYTEST_ADDOPTS: "--color=yes"
|
PYTEST_ADDOPTS: "--color=yes"
|
||||||
FORCE_COLOR: "1"
|
FORCE_COLOR: "1"
|
||||||
PIP_EXTRA_INDEX_URL: "https://pypi.fury.io/lancedb/"
|
PIP_EXTRA_INDEX_URL: "https://pypi.fury.io/lance-format/ https://pypi.fury.io/lancedb/"
|
||||||
|
RUST_BACKTRACE: "1"
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
lint:
|
lint:
|
||||||
@@ -78,7 +79,7 @@ jobs:
|
|||||||
doctest:
|
doctest:
|
||||||
name: "Doctest"
|
name: "Doctest"
|
||||||
timeout-minutes: 30
|
timeout-minutes: 30
|
||||||
runs-on: "ubuntu-24.04"
|
runs-on: ubuntu-2404-8x-x64
|
||||||
defaults:
|
defaults:
|
||||||
run:
|
run:
|
||||||
shell: bash
|
shell: bash
|
||||||
@@ -97,12 +98,9 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
sudo apt update
|
sudo apt update
|
||||||
sudo apt install -y protobuf-compiler
|
sudo apt install -y protobuf-compiler
|
||||||
- uses: Swatinem/rust-cache@v2
|
|
||||||
with:
|
|
||||||
workspaces: python
|
|
||||||
- name: Install
|
- name: Install
|
||||||
run: |
|
run: |
|
||||||
pip install --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests,dev,embeddings]
|
pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests,dev,embeddings]
|
||||||
pip install tantivy
|
pip install tantivy
|
||||||
pip install mlx
|
pip install mlx
|
||||||
- name: Doctest
|
- name: Doctest
|
||||||
@@ -131,10 +129,9 @@ jobs:
|
|||||||
uses: actions/setup-python@v5
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: 3.${{ matrix.python-minor-version }}
|
python-version: 3.${{ matrix.python-minor-version }}
|
||||||
- uses: Swatinem/rust-cache@v2
|
|
||||||
with:
|
|
||||||
workspaces: python
|
|
||||||
- uses: ./.github/workflows/build_linux_wheel
|
- uses: ./.github/workflows/build_linux_wheel
|
||||||
|
with:
|
||||||
|
args: --profile ci
|
||||||
- uses: ./.github/workflows/run_tests
|
- uses: ./.github/workflows/run_tests
|
||||||
with:
|
with:
|
||||||
integration: true
|
integration: true
|
||||||
@@ -152,7 +149,7 @@ jobs:
|
|||||||
matrix:
|
matrix:
|
||||||
config:
|
config:
|
||||||
- name: x86
|
- name: x86
|
||||||
runner: macos-13
|
runner: macos-15-large
|
||||||
- name: Arm
|
- name: Arm
|
||||||
runner: macos-14
|
runner: macos-14
|
||||||
runs-on: "${{ matrix.config.runner }}"
|
runs-on: "${{ matrix.config.runner }}"
|
||||||
@@ -169,10 +166,9 @@ jobs:
|
|||||||
uses: actions/setup-python@v5
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: "3.12"
|
python-version: "3.12"
|
||||||
- uses: Swatinem/rust-cache@v2
|
|
||||||
with:
|
|
||||||
workspaces: python
|
|
||||||
- uses: ./.github/workflows/build_mac_wheel
|
- uses: ./.github/workflows/build_mac_wheel
|
||||||
|
with:
|
||||||
|
args: --profile ci
|
||||||
- uses: ./.github/workflows/run_tests
|
- uses: ./.github/workflows/run_tests
|
||||||
# Make sure wheels are not included in the Rust cache
|
# Make sure wheels are not included in the Rust cache
|
||||||
- name: Delete wheels
|
- name: Delete wheels
|
||||||
@@ -199,10 +195,9 @@ jobs:
|
|||||||
uses: actions/setup-python@v5
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: "3.12"
|
python-version: "3.12"
|
||||||
- uses: Swatinem/rust-cache@v2
|
|
||||||
with:
|
|
||||||
workspaces: python
|
|
||||||
- uses: ./.github/workflows/build_windows_wheel
|
- uses: ./.github/workflows/build_windows_wheel
|
||||||
|
with:
|
||||||
|
args: --profile ci
|
||||||
- uses: ./.github/workflows/run_tests
|
- uses: ./.github/workflows/run_tests
|
||||||
# Make sure wheels are not included in the Rust cache
|
# Make sure wheels are not included in the Rust cache
|
||||||
- name: Delete wheels
|
- name: Delete wheels
|
||||||
@@ -231,7 +226,7 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
pip install "pydantic<2"
|
pip install "pydantic<2"
|
||||||
pip install pyarrow==16
|
pip install pyarrow==16
|
||||||
pip install --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests]
|
pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests]
|
||||||
pip install tantivy
|
pip install tantivy
|
||||||
- name: Run tests
|
- name: Run tests
|
||||||
run: pytest -m "not slow and not s3_test" -x -v --durations=30 python/tests
|
run: pytest -m "not slow and not s3_test" -x -v --durations=30 python/tests
|
||||||
|
|||||||
2
.github/workflows/run_tests/action.yml
vendored
2
.github/workflows/run_tests/action.yml
vendored
@@ -15,7 +15,7 @@ runs:
|
|||||||
- name: Install lancedb
|
- name: Install lancedb
|
||||||
shell: bash
|
shell: bash
|
||||||
run: |
|
run: |
|
||||||
pip3 install --extra-index-url https://pypi.fury.io/lancedb/ $(ls target/wheels/lancedb-*.whl)[tests,dev]
|
pip3 install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ $(ls target/wheels/lancedb-*.whl)[tests,dev]
|
||||||
- name: Setup localstack for integration tests
|
- name: Setup localstack for integration tests
|
||||||
if: ${{ inputs.integration == 'true' }}
|
if: ${{ inputs.integration == 'true' }}
|
||||||
shell: bash
|
shell: bash
|
||||||
|
|||||||
44
.github/workflows/rust.yml
vendored
44
.github/workflows/rust.yml
vendored
@@ -18,11 +18,7 @@ env:
|
|||||||
# This env var is used by Swatinem/rust-cache@v2 for the cache
|
# This env var is used by Swatinem/rust-cache@v2 for the cache
|
||||||
# key, so we set it to make sure it is always consistent.
|
# key, so we set it to make sure it is always consistent.
|
||||||
CARGO_TERM_COLOR: always
|
CARGO_TERM_COLOR: always
|
||||||
# Disable full debug symbol generation to speed up CI build and keep memory down
|
|
||||||
# "1" means line tables only, which is useful for panic tracebacks.
|
|
||||||
RUSTFLAGS: "-C debuginfo=1"
|
|
||||||
RUST_BACKTRACE: "1"
|
RUST_BACKTRACE: "1"
|
||||||
CARGO_INCREMENTAL: 0
|
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
lint:
|
lint:
|
||||||
@@ -44,8 +40,6 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
components: rustfmt, clippy
|
components: rustfmt, clippy
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
|
||||||
workspaces: rust
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: |
|
run: |
|
||||||
sudo apt update
|
sudo apt update
|
||||||
@@ -53,7 +47,7 @@ jobs:
|
|||||||
- name: Run format
|
- name: Run format
|
||||||
run: cargo fmt --all -- --check
|
run: cargo fmt --all -- --check
|
||||||
- name: Run clippy
|
- name: Run clippy
|
||||||
run: cargo clippy --workspace --tests --all-features -- -D warnings
|
run: cargo clippy --profile ci --workspace --tests --all-features -- -D warnings
|
||||||
|
|
||||||
build-no-lock:
|
build-no-lock:
|
||||||
runs-on: ubuntu-24.04
|
runs-on: ubuntu-24.04
|
||||||
@@ -80,7 +74,7 @@ jobs:
|
|||||||
sudo apt install -y protobuf-compiler libssl-dev
|
sudo apt install -y protobuf-compiler libssl-dev
|
||||||
- name: Build all
|
- name: Build all
|
||||||
run: |
|
run: |
|
||||||
cargo build --benches --all-features --tests
|
cargo build --profile ci --benches --all-features --tests
|
||||||
|
|
||||||
linux:
|
linux:
|
||||||
timeout-minutes: 30
|
timeout-minutes: 30
|
||||||
@@ -103,14 +97,8 @@ jobs:
|
|||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
lfs: true
|
lfs: true
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
|
||||||
workspaces: rust
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: |
|
run: sudo apt install -y protobuf-compiler libssl-dev
|
||||||
# This shaves 2 minutes off this step in CI. This doesn't seem to be
|
|
||||||
# necessary in standard runners, but it is in the 4x runners.
|
|
||||||
sudo rm /var/lib/man-db/auto-update
|
|
||||||
sudo apt install -y protobuf-compiler libssl-dev
|
|
||||||
- uses: rui314/setup-mold@v1
|
- uses: rui314/setup-mold@v1
|
||||||
- name: Make Swap
|
- name: Make Swap
|
||||||
run: |
|
run: |
|
||||||
@@ -119,22 +107,22 @@ jobs:
|
|||||||
sudo mkswap /swapfile
|
sudo mkswap /swapfile
|
||||||
sudo swapon /swapfile
|
sudo swapon /swapfile
|
||||||
- name: Build
|
- name: Build
|
||||||
run: cargo build --all-features --tests --locked --examples
|
run: cargo build --profile ci --all-features --tests --locked --examples
|
||||||
- name: Run feature tests
|
- name: Run feature tests
|
||||||
run: make -C ./lancedb feature-tests
|
run: CARGO_ARGS="--profile ci" make -C ./lancedb feature-tests
|
||||||
- name: Run examples
|
- name: Run examples
|
||||||
run: cargo run --example simple --locked
|
run: cargo run --profile ci --example simple --locked
|
||||||
- name: Run remote tests
|
- name: Run remote tests
|
||||||
# Running this requires access to secrets, so skip if this is
|
# Running this requires access to secrets, so skip if this is
|
||||||
# a PR from a fork.
|
# a PR from a fork.
|
||||||
if: github.event_name != 'pull_request' || !github.event.pull_request.head.repo.fork
|
if: github.event_name != 'pull_request' || !github.event.pull_request.head.repo.fork
|
||||||
run: make -C ./lancedb remote-tests
|
run: CARGO_ARGS="--profile ci" make -C ./lancedb remote-tests
|
||||||
|
|
||||||
macos:
|
macos:
|
||||||
timeout-minutes: 30
|
timeout-minutes: 30
|
||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
mac-runner: ["macos-13", "macos-14"]
|
mac-runner: ["macos-14", "macos-15"]
|
||||||
runs-on: "${{ matrix.mac-runner }}"
|
runs-on: "${{ matrix.mac-runner }}"
|
||||||
defaults:
|
defaults:
|
||||||
run:
|
run:
|
||||||
@@ -148,8 +136,6 @@ jobs:
|
|||||||
- name: CPU features
|
- name: CPU features
|
||||||
run: sysctl -a | grep cpu
|
run: sysctl -a | grep cpu
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
|
||||||
workspaces: rust
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: brew install protobuf
|
run: brew install protobuf
|
||||||
- name: Run tests
|
- name: Run tests
|
||||||
@@ -159,7 +145,7 @@ jobs:
|
|||||||
ALL_FEATURES=`cargo metadata --format-version=1 --no-deps \
|
ALL_FEATURES=`cargo metadata --format-version=1 --no-deps \
|
||||||
| jq -r '.packages[] | .features | keys | .[]' \
|
| jq -r '.packages[] | .features | keys | .[]' \
|
||||||
| grep -v s3-test | sort | uniq | paste -s -d "," -`
|
| grep -v s3-test | sort | uniq | paste -s -d "," -`
|
||||||
cargo test --features $ALL_FEATURES --locked
|
cargo test --profile ci --features $ALL_FEATURES --locked
|
||||||
|
|
||||||
windows:
|
windows:
|
||||||
runs-on: windows-2022
|
runs-on: windows-2022
|
||||||
@@ -173,22 +159,21 @@ jobs:
|
|||||||
working-directory: rust/lancedb
|
working-directory: rust/lancedb
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v4
|
||||||
|
- name: Set target
|
||||||
|
run: rustup target add ${{ matrix.target }}
|
||||||
- uses: Swatinem/rust-cache@v2
|
- uses: Swatinem/rust-cache@v2
|
||||||
with:
|
|
||||||
workspaces: rust
|
|
||||||
- name: Install Protoc v21.12
|
- name: Install Protoc v21.12
|
||||||
run: choco install --no-progress protoc
|
run: choco install --no-progress protoc
|
||||||
- name: Build
|
- name: Build
|
||||||
run: |
|
run: |
|
||||||
rustup target add ${{ matrix.target }}
|
|
||||||
$env:VCPKG_ROOT = $env:VCPKG_INSTALLATION_ROOT
|
$env:VCPKG_ROOT = $env:VCPKG_INSTALLATION_ROOT
|
||||||
cargo build --features remote --tests --locked --target ${{ matrix.target }}
|
cargo build --profile ci --features remote --tests --locked --target ${{ matrix.target }}
|
||||||
- name: Run tests
|
- name: Run tests
|
||||||
# Can only run tests when target matches host
|
# Can only run tests when target matches host
|
||||||
if: ${{ matrix.target == 'x86_64-pc-windows-msvc' }}
|
if: ${{ matrix.target == 'x86_64-pc-windows-msvc' }}
|
||||||
run: |
|
run: |
|
||||||
$env:VCPKG_ROOT = $env:VCPKG_INSTALLATION_ROOT
|
$env:VCPKG_ROOT = $env:VCPKG_INSTALLATION_ROOT
|
||||||
cargo test --features remote --locked
|
cargo test --profile ci --features remote --locked
|
||||||
|
|
||||||
msrv:
|
msrv:
|
||||||
# Check the minimum supported Rust version
|
# Check the minimum supported Rust version
|
||||||
@@ -213,6 +198,7 @@ jobs:
|
|||||||
uses: dtolnay/rust-toolchain@master
|
uses: dtolnay/rust-toolchain@master
|
||||||
with:
|
with:
|
||||||
toolchain: ${{ matrix.msrv }}
|
toolchain: ${{ matrix.msrv }}
|
||||||
|
- uses: Swatinem/rust-cache@v2
|
||||||
- name: Downgrade dependencies
|
- name: Downgrade dependencies
|
||||||
# These packages have newer requirements for MSRV
|
# These packages have newer requirements for MSRV
|
||||||
run: |
|
run: |
|
||||||
@@ -226,4 +212,4 @@ jobs:
|
|||||||
cargo update -p aws-sdk-sts --precise 1.51.0
|
cargo update -p aws-sdk-sts --precise 1.51.0
|
||||||
cargo update -p home --precise 0.5.9
|
cargo update -p home --precise 0.5.9
|
||||||
- name: cargo +${{ matrix.msrv }} check
|
- name: cargo +${{ matrix.msrv }} check
|
||||||
run: cargo check --workspace --tests --benches --all-features
|
run: cargo check --profile ci --workspace --tests --benches --all-features
|
||||||
|
|||||||
773
Cargo.lock
generated
773
Cargo.lock
generated
File diff suppressed because it is too large
Load Diff
42
Cargo.toml
42
Cargo.toml
@@ -15,20 +15,20 @@ categories = ["database-implementations"]
|
|||||||
rust-version = "1.78.0"
|
rust-version = "1.78.0"
|
||||||
|
|
||||||
[workspace.dependencies]
|
[workspace.dependencies]
|
||||||
lance = { "version" = "=1.0.0-beta.2", default-features = false, "tag" = "v1.0.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance = { "version" = "=1.0.0-beta.16", default-features = false, "tag" = "v1.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-core = { "version" = "=1.0.0-beta.2", "tag" = "v1.0.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance-core = { "version" = "=1.0.0-beta.16", "tag" = "v1.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-datagen = { "version" = "=1.0.0-beta.2", "tag" = "v1.0.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance-datagen = { "version" = "=1.0.0-beta.16", "tag" = "v1.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-file = { "version" = "=1.0.0-beta.2", "tag" = "v1.0.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance-file = { "version" = "=1.0.0-beta.16", "tag" = "v1.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-io = { "version" = "=1.0.0-beta.2", default-features = false, "tag" = "v1.0.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance-io = { "version" = "=1.0.0-beta.16", default-features = false, "tag" = "v1.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-index = { "version" = "=1.0.0-beta.2", "tag" = "v1.0.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance-index = { "version" = "=1.0.0-beta.16", "tag" = "v1.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-linalg = { "version" = "=1.0.0-beta.2", "tag" = "v1.0.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance-linalg = { "version" = "=1.0.0-beta.16", "tag" = "v1.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-namespace = { "version" = "=1.0.0-beta.2", "tag" = "v1.0.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance-namespace = { "version" = "=1.0.0-beta.16", "tag" = "v1.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-namespace-impls = { "version" = "=1.0.0-beta.2", "features" = ["dir-aws", "dir-gcp", "dir-azure", "dir-oss", "rest"], "tag" = "v1.0.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance-namespace-impls = { "version" = "=1.0.0-beta.16", default-features = false, "tag" = "v1.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-table = { "version" = "=1.0.0-beta.2", "tag" = "v1.0.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance-table = { "version" = "=1.0.0-beta.16", "tag" = "v1.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-testing = { "version" = "=1.0.0-beta.2", "tag" = "v1.0.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance-testing = { "version" = "=1.0.0-beta.16", "tag" = "v1.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-datafusion = { "version" = "=1.0.0-beta.2", "tag" = "v1.0.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance-datafusion = { "version" = "=1.0.0-beta.16", "tag" = "v1.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-encoding = { "version" = "=1.0.0-beta.2", "tag" = "v1.0.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance-encoding = { "version" = "=1.0.0-beta.16", "tag" = "v1.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-arrow = { "version" = "=1.0.0-beta.2", "tag" = "v1.0.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance-arrow = { "version" = "=1.0.0-beta.16", "tag" = "v1.0.0-beta.16", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
ahash = "0.8"
|
ahash = "0.8"
|
||||||
# Note that this one does not include pyarrow
|
# Note that this one does not include pyarrow
|
||||||
arrow = { version = "56.2", optional = false }
|
arrow = { version = "56.2", optional = false }
|
||||||
@@ -63,3 +63,17 @@ regex = "1.10"
|
|||||||
lazy_static = "1"
|
lazy_static = "1"
|
||||||
semver = "1.0.25"
|
semver = "1.0.25"
|
||||||
chrono = "0.4"
|
chrono = "0.4"
|
||||||
|
|
||||||
|
[profile.ci]
|
||||||
|
debug = "line-tables-only"
|
||||||
|
inherits = "dev"
|
||||||
|
incremental = false
|
||||||
|
|
||||||
|
# This rule applies to every package except workspace members (dependencies
|
||||||
|
# such as `arrow` and `tokio`). It disables debug info and related features on
|
||||||
|
# dependencies so their binaries stay smaller, improving cache reuse.
|
||||||
|
[profile.ci.package."*"]
|
||||||
|
debug = false
|
||||||
|
debug-assertions = false
|
||||||
|
strip = "debuginfo"
|
||||||
|
incremental = false
|
||||||
|
|||||||
@@ -15,7 +15,7 @@
|
|||||||
|
|
||||||
# **The Multimodal AI Lakehouse**
|
# **The Multimodal AI Lakehouse**
|
||||||
|
|
||||||
[**How to Install** ](#how-to-install) ✦ [**Detailed Documentation**](https://lancedb.github.io/lancedb/) ✦ [**Tutorials and Recipes**](https://github.com/lancedb/vectordb-recipes/tree/main) ✦ [**Contributors**](#contributors)
|
[**How to Install** ](#how-to-install) ✦ [**Detailed Documentation**](https://lancedb.com/docs) ✦ [**Tutorials and Recipes**](https://github.com/lancedb/vectordb-recipes/tree/main) ✦ [**Contributors**](#contributors)
|
||||||
|
|
||||||
**The ultimate multimodal data platform for AI/ML applications.**
|
**The ultimate multimodal data platform for AI/ML applications.**
|
||||||
|
|
||||||
|
|||||||
208
ci/check_lance_release.py
Executable file
208
ci/check_lance_release.py
Executable file
@@ -0,0 +1,208 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Determine whether a newer Lance tag exists and expose results for CI."""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Iterable, List, Sequence, Tuple, Union
|
||||||
|
|
||||||
|
try: # Python >=3.11
|
||||||
|
import tomllib # type: ignore
|
||||||
|
except ModuleNotFoundError: # pragma: no cover - fallback for older Python
|
||||||
|
import tomli as tomllib # type: ignore
|
||||||
|
|
||||||
|
LANCE_REPO = "lance-format/lance"
|
||||||
|
|
||||||
|
SEMVER_RE = re.compile(
|
||||||
|
r"^\s*(?P<major>0|[1-9]\d*)\.(?P<minor>0|[1-9]\d*)\.(?P<patch>0|[1-9]\d*)"
|
||||||
|
r"(?:-(?P<prerelease>[0-9A-Za-z.-]+))?"
|
||||||
|
r"(?:\+[0-9A-Za-z.-]+)?\s*$"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class SemVer:
|
||||||
|
major: int
|
||||||
|
minor: int
|
||||||
|
patch: int
|
||||||
|
prerelease: Tuple[Union[int, str], ...]
|
||||||
|
|
||||||
|
def __lt__(self, other: "SemVer") -> bool: # pragma: no cover - simple comparison
|
||||||
|
if (self.major, self.minor, self.patch) != (other.major, other.minor, other.patch):
|
||||||
|
return (self.major, self.minor, self.patch) < (other.major, other.minor, other.patch)
|
||||||
|
if self.prerelease == other.prerelease:
|
||||||
|
return False
|
||||||
|
if not self.prerelease:
|
||||||
|
return False # release > anything else
|
||||||
|
if not other.prerelease:
|
||||||
|
return True
|
||||||
|
for left, right in zip(self.prerelease, other.prerelease):
|
||||||
|
if left == right:
|
||||||
|
continue
|
||||||
|
if isinstance(left, int) and isinstance(right, int):
|
||||||
|
return left < right
|
||||||
|
if isinstance(left, int):
|
||||||
|
return True
|
||||||
|
if isinstance(right, int):
|
||||||
|
return False
|
||||||
|
return str(left) < str(right)
|
||||||
|
return len(self.prerelease) < len(other.prerelease)
|
||||||
|
|
||||||
|
def __eq__(self, other: object) -> bool: # pragma: no cover - trivial
|
||||||
|
if not isinstance(other, SemVer):
|
||||||
|
return NotImplemented
|
||||||
|
return (
|
||||||
|
self.major == other.major
|
||||||
|
and self.minor == other.minor
|
||||||
|
and self.patch == other.patch
|
||||||
|
and self.prerelease == other.prerelease
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def parse_semver(raw: str) -> SemVer:
|
||||||
|
match = SEMVER_RE.match(raw)
|
||||||
|
if not match:
|
||||||
|
raise ValueError(f"Unsupported version format: {raw}")
|
||||||
|
prerelease = match.group("prerelease")
|
||||||
|
parts: Tuple[Union[int, str], ...] = ()
|
||||||
|
if prerelease:
|
||||||
|
parsed: List[Union[int, str]] = []
|
||||||
|
for piece in prerelease.split("."):
|
||||||
|
if piece.isdigit():
|
||||||
|
parsed.append(int(piece))
|
||||||
|
else:
|
||||||
|
parsed.append(piece)
|
||||||
|
parts = tuple(parsed)
|
||||||
|
return SemVer(
|
||||||
|
major=int(match.group("major")),
|
||||||
|
minor=int(match.group("minor")),
|
||||||
|
patch=int(match.group("patch")),
|
||||||
|
prerelease=parts,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class TagInfo:
|
||||||
|
tag: str # e.g. v1.0.0-beta.2
|
||||||
|
version: str # e.g. 1.0.0-beta.2
|
||||||
|
semver: SemVer
|
||||||
|
|
||||||
|
|
||||||
|
def run_command(cmd: Sequence[str]) -> str:
|
||||||
|
result = subprocess.run(cmd, capture_output=True, text=True, check=False)
|
||||||
|
if result.returncode != 0:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"Command {' '.join(cmd)} failed with {result.returncode}: {result.stderr.strip()}"
|
||||||
|
)
|
||||||
|
return result.stdout.strip()
|
||||||
|
|
||||||
|
|
||||||
|
def fetch_remote_tags() -> List[TagInfo]:
|
||||||
|
output = run_command(
|
||||||
|
[
|
||||||
|
"gh",
|
||||||
|
"api",
|
||||||
|
"-X",
|
||||||
|
"GET",
|
||||||
|
f"repos/{LANCE_REPO}/git/refs/tags",
|
||||||
|
"--paginate",
|
||||||
|
"--jq",
|
||||||
|
".[].ref",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
tags: List[TagInfo] = []
|
||||||
|
for line in output.splitlines():
|
||||||
|
ref = line.strip()
|
||||||
|
if not ref.startswith("refs/tags/v"):
|
||||||
|
continue
|
||||||
|
tag = ref.split("refs/tags/")[-1]
|
||||||
|
version = tag.lstrip("v")
|
||||||
|
try:
|
||||||
|
tags.append(TagInfo(tag=tag, version=version, semver=parse_semver(version)))
|
||||||
|
except ValueError:
|
||||||
|
continue
|
||||||
|
if not tags:
|
||||||
|
raise RuntimeError("No Lance tags could be parsed from GitHub API output")
|
||||||
|
return tags
|
||||||
|
|
||||||
|
|
||||||
|
def read_current_version(repo_root: Path) -> str:
|
||||||
|
cargo_path = repo_root / "Cargo.toml"
|
||||||
|
with cargo_path.open("rb") as fh:
|
||||||
|
data = tomllib.load(fh)
|
||||||
|
try:
|
||||||
|
deps = data["workspace"]["dependencies"]
|
||||||
|
entry = deps["lance"]
|
||||||
|
except KeyError as exc: # pragma: no cover - configuration guard
|
||||||
|
raise RuntimeError("Failed to locate workspace.dependencies.lance in Cargo.toml") from exc
|
||||||
|
|
||||||
|
if isinstance(entry, str):
|
||||||
|
raw_version = entry
|
||||||
|
elif isinstance(entry, dict):
|
||||||
|
raw_version = entry.get("version", "")
|
||||||
|
else: # pragma: no cover - defensive
|
||||||
|
raise RuntimeError("Unexpected lance dependency format")
|
||||||
|
|
||||||
|
raw_version = raw_version.strip()
|
||||||
|
if not raw_version:
|
||||||
|
raise RuntimeError("lance dependency does not declare a version")
|
||||||
|
return raw_version.lstrip("=")
|
||||||
|
|
||||||
|
|
||||||
|
def determine_latest_tag(tags: Iterable[TagInfo]) -> TagInfo:
|
||||||
|
return max(tags, key=lambda tag: tag.semver)
|
||||||
|
|
||||||
|
|
||||||
|
def write_outputs(args: argparse.Namespace, payload: dict) -> None:
|
||||||
|
target = getattr(args, "github_output", None)
|
||||||
|
if not target:
|
||||||
|
return
|
||||||
|
with open(target, "a", encoding="utf-8") as handle:
|
||||||
|
for key, value in payload.items():
|
||||||
|
handle.write(f"{key}={value}\n")
|
||||||
|
|
||||||
|
|
||||||
|
def main(argv: Sequence[str] | None = None) -> int:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument(
|
||||||
|
"--repo-root",
|
||||||
|
default=Path(__file__).resolve().parents[1],
|
||||||
|
type=Path,
|
||||||
|
help="Path to the lancedb repository root",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--github-output",
|
||||||
|
default=os.environ.get("GITHUB_OUTPUT"),
|
||||||
|
help="Optional file path for writing GitHub Action outputs",
|
||||||
|
)
|
||||||
|
args = parser.parse_args(argv)
|
||||||
|
|
||||||
|
repo_root = Path(args.repo_root)
|
||||||
|
current_version = read_current_version(repo_root)
|
||||||
|
current_semver = parse_semver(current_version)
|
||||||
|
|
||||||
|
tags = fetch_remote_tags()
|
||||||
|
latest = determine_latest_tag(tags)
|
||||||
|
needs_update = latest.semver > current_semver
|
||||||
|
|
||||||
|
payload = {
|
||||||
|
"current_version": current_version,
|
||||||
|
"current_tag": f"v{current_version}",
|
||||||
|
"latest_version": latest.version,
|
||||||
|
"latest_tag": latest.tag,
|
||||||
|
"needs_update": "true" if needs_update else "false",
|
||||||
|
}
|
||||||
|
|
||||||
|
print(json.dumps(payload))
|
||||||
|
write_outputs(args, payload)
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
@@ -3,6 +3,8 @@ import re
|
|||||||
import sys
|
import sys
|
||||||
import json
|
import json
|
||||||
|
|
||||||
|
LANCE_GIT_URL = "https://github.com/lance-format/lance.git"
|
||||||
|
|
||||||
|
|
||||||
def run_command(command: str) -> str:
|
def run_command(command: str) -> str:
|
||||||
"""
|
"""
|
||||||
@@ -29,7 +31,7 @@ def get_latest_stable_version() -> str:
|
|||||||
|
|
||||||
def get_latest_preview_version() -> str:
|
def get_latest_preview_version() -> str:
|
||||||
lance_tags = run_command(
|
lance_tags = run_command(
|
||||||
"git ls-remote --tags https://github.com/lancedb/lance.git | grep 'refs/tags/v[0-9beta.-]\\+$'"
|
f"git ls-remote --tags {LANCE_GIT_URL} | grep 'refs/tags/v[0-9beta.-]\\+$'"
|
||||||
).splitlines()
|
).splitlines()
|
||||||
lance_tags = (
|
lance_tags = (
|
||||||
tag.split("refs/tags/")[1]
|
tag.split("refs/tags/")[1]
|
||||||
@@ -176,8 +178,8 @@ def set_stable_version(version: str):
|
|||||||
def set_preview_version(version: str):
|
def set_preview_version(version: str):
|
||||||
"""
|
"""
|
||||||
Sets lines to
|
Sets lines to
|
||||||
lance = { "version" = "=0.29.0", default-features = false, "features" = ["dynamodb"], "tag" = "v0.29.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance = { "version" = "=0.29.0", default-features = false, "features" = ["dynamodb"], "tag" = "v0.29.0-beta.2", "git" = LANCE_GIT_URL }
|
||||||
lance-io = { "version" = "=0.29.0", default-features = false, "tag" = "v0.29.0-beta.2", "git" = "https://github.com/lancedb/lance.git" }
|
lance-io = { "version" = "=0.29.0", default-features = false, "tag" = "v0.29.0-beta.2", "git" = LANCE_GIT_URL }
|
||||||
...
|
...
|
||||||
"""
|
"""
|
||||||
|
|
||||||
@@ -194,7 +196,7 @@ def set_preview_version(version: str):
|
|||||||
config["features"] = features
|
config["features"] = features
|
||||||
|
|
||||||
config["tag"] = f"v{version}"
|
config["tag"] = f"v{version}"
|
||||||
config["git"] = "https://github.com/lancedb/lance.git"
|
config["git"] = LANCE_GIT_URL
|
||||||
|
|
||||||
return dict_to_toml_line(package_name, config)
|
return dict_to_toml_line(package_name, config)
|
||||||
|
|
||||||
|
|||||||
@@ -1,8 +1,8 @@
|
|||||||
# LanceDB Documentation
|
# LanceDB Documentation
|
||||||
|
|
||||||
LanceDB docs are deployed to https://lancedb.github.io/lancedb/.
|
LanceDB docs are available at [lancedb.com/docs](https://lancedb.com/docs).
|
||||||
|
|
||||||
Docs is built and deployed automatically by [Github Actions](../.github/workflows/docs.yml)
|
The SDK docs are built and deployed automatically by [Github Actions](../.github/workflows/docs.yml)
|
||||||
whenever a commit is pushed to the `main` branch. So it is possible for the docs to show
|
whenever a commit is pushed to the `main` branch. So it is possible for the docs to show
|
||||||
unreleased features.
|
unreleased features.
|
||||||
|
|
||||||
|
|||||||
@@ -34,7 +34,7 @@ const results = await table.vectorSearch([0.1, 0.3]).limit(20).toArray();
|
|||||||
console.log(results);
|
console.log(results);
|
||||||
```
|
```
|
||||||
|
|
||||||
The [quickstart](https://lancedb.github.io/lancedb/basic/) contains a more complete example.
|
The [quickstart](https://lancedb.com/docs/quickstart/basic-usage/) contains more complete examples.
|
||||||
|
|
||||||
## Development
|
## Development
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# Contributing to LanceDB Typescript
|
# Contributing to LanceDB Typescript
|
||||||
|
|
||||||
This document outlines the process for contributing to LanceDB Typescript.
|
This document outlines the process for contributing to LanceDB Typescript.
|
||||||
For general contribution guidelines, see [CONTRIBUTING.md](../../../../CONTRIBUTING.md).
|
For general contribution guidelines, see [CONTRIBUTING.md](../CONTRIBUTING.md).
|
||||||
|
|
||||||
## Project layout
|
## Project layout
|
||||||
|
|
||||||
|
|||||||
@@ -147,7 +147,7 @@ A new PermutationBuilder instance
|
|||||||
#### Example
|
#### Example
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
builder.splitCalculated("user_id % 3");
|
builder.splitCalculated({ calculation: "user_id % 3" });
|
||||||
```
|
```
|
||||||
|
|
||||||
***
|
***
|
||||||
|
|||||||
@@ -89,4 +89,4 @@ optional storageOptions: Record<string, string>;
|
|||||||
|
|
||||||
(For LanceDB OSS only): configuration for object storage.
|
(For LanceDB OSS only): configuration for object storage.
|
||||||
|
|
||||||
The available options are described at https://lancedb.github.io/lancedb/guides/storage/
|
The available options are described at https://lancedb.com/docs/storage/
|
||||||
|
|||||||
@@ -97,4 +97,4 @@ Configuration for object storage.
|
|||||||
Options already set on the connection will be inherited by the table,
|
Options already set on the connection will be inherited by the table,
|
||||||
but can be overridden here.
|
but can be overridden here.
|
||||||
|
|
||||||
The available options are described at https://lancedb.github.io/lancedb/guides/storage/
|
The available options are described at https://lancedb.com/docs/storage/
|
||||||
|
|||||||
@@ -8,6 +8,14 @@
|
|||||||
|
|
||||||
## Properties
|
## Properties
|
||||||
|
|
||||||
|
### numAttempts
|
||||||
|
|
||||||
|
```ts
|
||||||
|
numAttempts: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### numDeletedRows
|
### numDeletedRows
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
|
|||||||
@@ -42,4 +42,4 @@ Configuration for object storage.
|
|||||||
Options already set on the connection will be inherited by the table,
|
Options already set on the connection will be inherited by the table,
|
||||||
but can be overridden here.
|
but can be overridden here.
|
||||||
|
|
||||||
The available options are described at https://lancedb.github.io/lancedb/guides/storage/
|
The available options are described at https://lancedb.com/docs/storage/
|
||||||
|
|||||||
@@ -30,6 +30,12 @@ is also an [asynchronous API client](#connections-asynchronous).
|
|||||||
|
|
||||||
::: lancedb.table.Table
|
::: lancedb.table.Table
|
||||||
|
|
||||||
|
::: lancedb.table.FragmentStatistics
|
||||||
|
|
||||||
|
::: lancedb.table.FragmentSummaryStats
|
||||||
|
|
||||||
|
::: lancedb.table.Tags
|
||||||
|
|
||||||
## Querying (Synchronous)
|
## Querying (Synchronous)
|
||||||
|
|
||||||
::: lancedb.query.Query
|
::: lancedb.query.Query
|
||||||
@@ -58,6 +64,14 @@ is also an [asynchronous API client](#connections-asynchronous).
|
|||||||
|
|
||||||
::: lancedb.embeddings.open_clip.OpenClipEmbeddings
|
::: lancedb.embeddings.open_clip.OpenClipEmbeddings
|
||||||
|
|
||||||
|
## Remote configuration
|
||||||
|
|
||||||
|
::: lancedb.remote.ClientConfig
|
||||||
|
|
||||||
|
::: lancedb.remote.TimeoutConfig
|
||||||
|
|
||||||
|
::: lancedb.remote.RetryConfig
|
||||||
|
|
||||||
## Context
|
## Context
|
||||||
|
|
||||||
::: lancedb.context.contextualize
|
::: lancedb.context.contextualize
|
||||||
@@ -115,6 +129,8 @@ Table hold your actual data as a collection of records / rows.
|
|||||||
|
|
||||||
::: lancedb.table.AsyncTable
|
::: lancedb.table.AsyncTable
|
||||||
|
|
||||||
|
::: lancedb.table.AsyncTags
|
||||||
|
|
||||||
## Indices (Asynchronous)
|
## Indices (Asynchronous)
|
||||||
|
|
||||||
Indices can be created on a table to speed up queries. This section
|
Indices can be created on a table to speed up queries. This section
|
||||||
@@ -136,6 +152,8 @@ lists the indices that LanceDb supports.
|
|||||||
|
|
||||||
::: lancedb.index.IvfFlat
|
::: lancedb.index.IvfFlat
|
||||||
|
|
||||||
|
::: lancedb.table.IndexStatistics
|
||||||
|
|
||||||
## Querying (Asynchronous)
|
## Querying (Asynchronous)
|
||||||
|
|
||||||
Queries allow you to return data from your database. Basic queries can be
|
Queries allow you to return data from your database. Basic queries can be
|
||||||
|
|||||||
@@ -8,7 +8,7 @@
|
|||||||
<parent>
|
<parent>
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-parent</artifactId>
|
<artifactId>lancedb-parent</artifactId>
|
||||||
<version>0.22.3-final.0</version>
|
<version>0.22.4-beta.2</version>
|
||||||
<relativePath>../pom.xml</relativePath>
|
<relativePath>../pom.xml</relativePath>
|
||||||
</parent>
|
</parent>
|
||||||
|
|
||||||
|
|||||||
@@ -8,7 +8,7 @@
|
|||||||
<parent>
|
<parent>
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-parent</artifactId>
|
<artifactId>lancedb-parent</artifactId>
|
||||||
<version>0.22.3-final.0</version>
|
<version>0.22.4-beta.2</version>
|
||||||
<relativePath>../pom.xml</relativePath>
|
<relativePath>../pom.xml</relativePath>
|
||||||
</parent>
|
</parent>
|
||||||
|
|
||||||
|
|||||||
@@ -6,7 +6,7 @@
|
|||||||
|
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-parent</artifactId>
|
<artifactId>lancedb-parent</artifactId>
|
||||||
<version>0.22.3-final.0</version>
|
<version>0.22.4-beta.2</version>
|
||||||
<packaging>pom</packaging>
|
<packaging>pom</packaging>
|
||||||
<name>${project.artifactId}</name>
|
<name>${project.artifactId}</name>
|
||||||
<description>LanceDB Java SDK Parent POM</description>
|
<description>LanceDB Java SDK Parent POM</description>
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "lancedb-nodejs"
|
name = "lancedb-nodejs"
|
||||||
edition.workspace = true
|
edition.workspace = true
|
||||||
version = "0.22.3"
|
version = "0.22.4-beta.2"
|
||||||
license.workspace = true
|
license.workspace = true
|
||||||
description.workspace = true
|
description.workspace = true
|
||||||
repository.workspace = true
|
repository.workspace = true
|
||||||
|
|||||||
@@ -30,7 +30,7 @@ const results = await table.vectorSearch([0.1, 0.3]).limit(20).toArray();
|
|||||||
console.log(results);
|
console.log(results);
|
||||||
```
|
```
|
||||||
|
|
||||||
The [quickstart](https://lancedb.github.io/lancedb/basic/) contains a more complete example.
|
The [quickstart](https://lancedb.com/docs/quickstart/basic-usage/) contains more complete examples.
|
||||||
|
|
||||||
## Development
|
## Development
|
||||||
|
|
||||||
|
|||||||
@@ -42,7 +42,7 @@ export interface CreateTableOptions {
|
|||||||
* Options already set on the connection will be inherited by the table,
|
* Options already set on the connection will be inherited by the table,
|
||||||
* but can be overridden here.
|
* but can be overridden here.
|
||||||
*
|
*
|
||||||
* The available options are described at https://lancedb.github.io/lancedb/guides/storage/
|
* The available options are described at https://lancedb.com/docs/storage/
|
||||||
*/
|
*/
|
||||||
storageOptions?: Record<string, string>;
|
storageOptions?: Record<string, string>;
|
||||||
|
|
||||||
@@ -78,7 +78,7 @@ export interface OpenTableOptions {
|
|||||||
* Options already set on the connection will be inherited by the table,
|
* Options already set on the connection will be inherited by the table,
|
||||||
* but can be overridden here.
|
* but can be overridden here.
|
||||||
*
|
*
|
||||||
* The available options are described at https://lancedb.github.io/lancedb/guides/storage/
|
* The available options are described at https://lancedb.com/docs/storage/
|
||||||
*/
|
*/
|
||||||
storageOptions?: Record<string, string>;
|
storageOptions?: Record<string, string>;
|
||||||
/**
|
/**
|
||||||
|
|||||||
@@ -118,7 +118,7 @@ export class PermutationBuilder {
|
|||||||
* @returns A new PermutationBuilder instance
|
* @returns A new PermutationBuilder instance
|
||||||
* @example
|
* @example
|
||||||
* ```ts
|
* ```ts
|
||||||
* builder.splitCalculated("user_id % 3");
|
* builder.splitCalculated({ calculation: "user_id % 3" });
|
||||||
* ```
|
* ```
|
||||||
*/
|
*/
|
||||||
splitCalculated(options: SplitCalculatedOptions): PermutationBuilder {
|
splitCalculated(options: SplitCalculatedOptions): PermutationBuilder {
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-darwin-arm64",
|
"name": "@lancedb/lancedb-darwin-arm64",
|
||||||
"version": "0.22.3",
|
"version": "0.22.4-beta.2",
|
||||||
"os": ["darwin"],
|
"os": ["darwin"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.darwin-arm64.node",
|
"main": "lancedb.darwin-arm64.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-darwin-x64",
|
"name": "@lancedb/lancedb-darwin-x64",
|
||||||
"version": "0.22.3",
|
"version": "0.22.4-beta.2",
|
||||||
"os": ["darwin"],
|
"os": ["darwin"],
|
||||||
"cpu": ["x64"],
|
"cpu": ["x64"],
|
||||||
"main": "lancedb.darwin-x64.node",
|
"main": "lancedb.darwin-x64.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
||||||
"version": "0.22.3",
|
"version": "0.22.4-beta.2",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.linux-arm64-gnu.node",
|
"main": "lancedb.linux-arm64-gnu.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-arm64-musl",
|
"name": "@lancedb/lancedb-linux-arm64-musl",
|
||||||
"version": "0.22.3",
|
"version": "0.22.4-beta.2",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.linux-arm64-musl.node",
|
"main": "lancedb.linux-arm64-musl.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-x64-gnu",
|
"name": "@lancedb/lancedb-linux-x64-gnu",
|
||||||
"version": "0.22.3",
|
"version": "0.22.4-beta.2",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["x64"],
|
"cpu": ["x64"],
|
||||||
"main": "lancedb.linux-x64-gnu.node",
|
"main": "lancedb.linux-x64-gnu.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-x64-musl",
|
"name": "@lancedb/lancedb-linux-x64-musl",
|
||||||
"version": "0.22.3",
|
"version": "0.22.4-beta.2",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["x64"],
|
"cpu": ["x64"],
|
||||||
"main": "lancedb.linux-x64-musl.node",
|
"main": "lancedb.linux-x64-musl.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-win32-arm64-msvc",
|
"name": "@lancedb/lancedb-win32-arm64-msvc",
|
||||||
"version": "0.22.3",
|
"version": "0.22.4-beta.2",
|
||||||
"os": [
|
"os": [
|
||||||
"win32"
|
"win32"
|
||||||
],
|
],
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-win32-x64-msvc",
|
"name": "@lancedb/lancedb-win32-x64-msvc",
|
||||||
"version": "0.22.3",
|
"version": "0.22.4-beta.2",
|
||||||
"os": ["win32"],
|
"os": ["win32"],
|
||||||
"cpu": ["x64"],
|
"cpu": ["x64"],
|
||||||
"main": "lancedb.win32-x64-msvc.node",
|
"main": "lancedb.win32-x64-msvc.node",
|
||||||
|
|||||||
4
nodejs/package-lock.json
generated
4
nodejs/package-lock.json
generated
@@ -1,12 +1,12 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb",
|
"name": "@lancedb/lancedb",
|
||||||
"version": "0.22.3",
|
"version": "0.22.4-beta.2",
|
||||||
"lockfileVersion": 3,
|
"lockfileVersion": 3,
|
||||||
"requires": true,
|
"requires": true,
|
||||||
"packages": {
|
"packages": {
|
||||||
"": {
|
"": {
|
||||||
"name": "@lancedb/lancedb",
|
"name": "@lancedb/lancedb",
|
||||||
"version": "0.22.3",
|
"version": "0.22.4-beta.2",
|
||||||
"cpu": [
|
"cpu": [
|
||||||
"x64",
|
"x64",
|
||||||
"arm64"
|
"arm64"
|
||||||
|
|||||||
@@ -11,7 +11,7 @@
|
|||||||
"ann"
|
"ann"
|
||||||
],
|
],
|
||||||
"private": false,
|
"private": false,
|
||||||
"version": "0.22.3",
|
"version": "0.22.4-beta.2",
|
||||||
"main": "dist/index.js",
|
"main": "dist/index.js",
|
||||||
"exports": {
|
"exports": {
|
||||||
".": "./dist/index.js",
|
".": "./dist/index.js",
|
||||||
@@ -73,8 +73,10 @@
|
|||||||
"scripts": {
|
"scripts": {
|
||||||
"artifacts": "napi artifacts",
|
"artifacts": "napi artifacts",
|
||||||
"build:debug": "napi build --platform --no-const-enum --dts ../lancedb/native.d.ts --js ../lancedb/native.js lancedb",
|
"build:debug": "napi build --platform --no-const-enum --dts ../lancedb/native.d.ts --js ../lancedb/native.js lancedb",
|
||||||
|
"postbuild:debug": "shx mkdir -p dist && shx cp lancedb/*.node dist/",
|
||||||
"build:release": "napi build --platform --no-const-enum --release --dts ../lancedb/native.d.ts --js ../lancedb/native.js dist/",
|
"build:release": "napi build --platform --no-const-enum --release --dts ../lancedb/native.d.ts --js ../lancedb/native.js dist/",
|
||||||
"build": "npm run build:debug && npm run tsc && shx cp lancedb/*.node dist/",
|
"postbuild:release": "shx mkdir -p dist && shx cp lancedb/*.node dist/",
|
||||||
|
"build": "npm run build:debug && npm run tsc",
|
||||||
"build-release": "npm run build:release && npm run tsc",
|
"build-release": "npm run build:release && npm run tsc",
|
||||||
"tsc": "tsc -b",
|
"tsc": "tsc -b",
|
||||||
"posttsc": "shx cp lancedb/native.d.ts dist/native.d.ts",
|
"posttsc": "shx cp lancedb/native.d.ts dist/native.d.ts",
|
||||||
|
|||||||
@@ -35,7 +35,7 @@ pub struct ConnectionOptions {
|
|||||||
pub read_consistency_interval: Option<f64>,
|
pub read_consistency_interval: Option<f64>,
|
||||||
/// (For LanceDB OSS only): configuration for object storage.
|
/// (For LanceDB OSS only): configuration for object storage.
|
||||||
///
|
///
|
||||||
/// The available options are described at https://lancedb.github.io/lancedb/guides/storage/
|
/// The available options are described at https://lancedb.com/docs/storage/
|
||||||
pub storage_options: Option<HashMap<String, String>>,
|
pub storage_options: Option<HashMap<String, String>>,
|
||||||
/// (For LanceDB OSS only): the session to use for this connection. Holds
|
/// (For LanceDB OSS only): the session to use for this connection. Holds
|
||||||
/// shared caches and other session-specific state.
|
/// shared caches and other session-specific state.
|
||||||
|
|||||||
@@ -740,6 +740,7 @@ pub struct MergeResult {
|
|||||||
pub num_inserted_rows: i64,
|
pub num_inserted_rows: i64,
|
||||||
pub num_updated_rows: i64,
|
pub num_updated_rows: i64,
|
||||||
pub num_deleted_rows: i64,
|
pub num_deleted_rows: i64,
|
||||||
|
pub num_attempts: i64,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl From<lancedb::table::MergeResult> for MergeResult {
|
impl From<lancedb::table::MergeResult> for MergeResult {
|
||||||
@@ -749,6 +750,7 @@ impl From<lancedb::table::MergeResult> for MergeResult {
|
|||||||
num_inserted_rows: value.num_inserted_rows as i64,
|
num_inserted_rows: value.num_inserted_rows as i64,
|
||||||
num_updated_rows: value.num_updated_rows as i64,
|
num_updated_rows: value.num_updated_rows as i64,
|
||||||
num_deleted_rows: value.num_deleted_rows as i64,
|
num_deleted_rows: value.num_deleted_rows as i64,
|
||||||
|
num_attempts: value.num_attempts as i64,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
[tool.bumpversion]
|
[tool.bumpversion]
|
||||||
current_version = "0.25.4-beta.0"
|
current_version = "0.25.4-beta.3"
|
||||||
parse = """(?x)
|
parse = """(?x)
|
||||||
(?P<major>0|[1-9]\\d*)\\.
|
(?P<major>0|[1-9]\\d*)\\.
|
||||||
(?P<minor>0|[1-9]\\d*)\\.
|
(?P<minor>0|[1-9]\\d*)\\.
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "lancedb-python"
|
name = "lancedb-python"
|
||||||
version = "0.25.4-beta.0"
|
version = "0.25.4-beta.3"
|
||||||
edition.workspace = true
|
edition.workspace = true
|
||||||
description = "Python bindings for LanceDB"
|
description = "Python bindings for LanceDB"
|
||||||
license.workspace = true
|
license.workspace = true
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
PIP_EXTRA_INDEX_URL ?= https://pypi.fury.io/lancedb/
|
PIP_EXTRA_INDEX_URL ?= https://pypi.fury.io/lance-format/ https://pypi.fury.io/lancedb/
|
||||||
|
|
||||||
help: ## Show this help.
|
help: ## Show this help.
|
||||||
@sed -ne '/@sed/!s/## //p' $(MAKEFILE_LIST)
|
@sed -ne '/@sed/!s/## //p' $(MAKEFILE_LIST)
|
||||||
|
|
||||||
.PHONY: develop
|
.PHONY: develop
|
||||||
develop: ## Install the package in development mode.
|
develop: ## Install the package in development mode.
|
||||||
PIP_EXTRA_INDEX_URL=$(PIP_EXTRA_INDEX_URL) maturin develop --extras tests,dev,embeddings
|
PIP_EXTRA_INDEX_URL="$(PIP_EXTRA_INDEX_URL)" maturin develop --extras tests,dev,embeddings
|
||||||
|
|
||||||
.PHONY: format
|
.PHONY: format
|
||||||
format: ## Format the code.
|
format: ## Format the code.
|
||||||
|
|||||||
@@ -10,7 +10,7 @@ dependencies = [
|
|||||||
"pyarrow>=16",
|
"pyarrow>=16",
|
||||||
"pydantic>=1.10",
|
"pydantic>=1.10",
|
||||||
"tqdm>=4.27.0",
|
"tqdm>=4.27.0",
|
||||||
"lance-namespace>=0.0.21"
|
"lance-namespace>=0.2.1"
|
||||||
]
|
]
|
||||||
description = "lancedb"
|
description = "lancedb"
|
||||||
authors = [{ name = "LanceDB Devs", email = "dev@lancedb.com" }]
|
authors = [{ name = "LanceDB Devs", email = "dev@lancedb.com" }]
|
||||||
@@ -45,7 +45,7 @@ repository = "https://github.com/lancedb/lancedb"
|
|||||||
|
|
||||||
[project.optional-dependencies]
|
[project.optional-dependencies]
|
||||||
pylance = [
|
pylance = [
|
||||||
"pylance>=0.25",
|
"pylance>=1.0.0b14",
|
||||||
]
|
]
|
||||||
tests = [
|
tests = [
|
||||||
"aiohttp",
|
"aiohttp",
|
||||||
@@ -59,7 +59,7 @@ tests = [
|
|||||||
"polars>=0.19, <=1.3.0",
|
"polars>=0.19, <=1.3.0",
|
||||||
"tantivy",
|
"tantivy",
|
||||||
"pyarrow-stubs",
|
"pyarrow-stubs",
|
||||||
"pylance>=1.0.0b2",
|
"pylance>=1.0.0b14",
|
||||||
"requests",
|
"requests",
|
||||||
"datafusion",
|
"datafusion",
|
||||||
]
|
]
|
||||||
|
|||||||
@@ -20,7 +20,12 @@ from .remote.db import RemoteDBConnection
|
|||||||
from .schema import vector
|
from .schema import vector
|
||||||
from .table import AsyncTable, Table
|
from .table import AsyncTable, Table
|
||||||
from ._lancedb import Session
|
from ._lancedb import Session
|
||||||
from .namespace import connect_namespace, LanceNamespaceDBConnection
|
from .namespace import (
|
||||||
|
connect_namespace,
|
||||||
|
connect_namespace_async,
|
||||||
|
LanceNamespaceDBConnection,
|
||||||
|
AsyncLanceNamespaceDBConnection,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def connect(
|
def connect(
|
||||||
@@ -36,7 +41,7 @@ def connect(
|
|||||||
session: Optional[Session] = None,
|
session: Optional[Session] = None,
|
||||||
**kwargs: Any,
|
**kwargs: Any,
|
||||||
) -> DBConnection:
|
) -> DBConnection:
|
||||||
"""Connect to a LanceDB database. YAY!
|
"""Connect to a LanceDB database.
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
@@ -67,7 +72,7 @@ def connect(
|
|||||||
default configuration is used.
|
default configuration is used.
|
||||||
storage_options: dict, optional
|
storage_options: dict, optional
|
||||||
Additional options for the storage backend. See available options at
|
Additional options for the storage backend. See available options at
|
||||||
<https://lancedb.github.io/lancedb/guides/storage/>
|
<https://lancedb.com/docs/storage/>
|
||||||
session: Session, optional
|
session: Session, optional
|
||||||
(For LanceDB OSS only)
|
(For LanceDB OSS only)
|
||||||
A session to use for this connection. Sessions allow you to configure
|
A session to use for this connection. Sessions allow you to configure
|
||||||
@@ -169,7 +174,7 @@ async def connect_async(
|
|||||||
default configuration is used.
|
default configuration is used.
|
||||||
storage_options: dict, optional
|
storage_options: dict, optional
|
||||||
Additional options for the storage backend. See available options at
|
Additional options for the storage backend. See available options at
|
||||||
<https://lancedb.github.io/lancedb/guides/storage/>
|
<https://lancedb.com/docs/storage/>
|
||||||
session: Session, optional
|
session: Session, optional
|
||||||
(For LanceDB OSS only)
|
(For LanceDB OSS only)
|
||||||
A session to use for this connection. Sessions allow you to configure
|
A session to use for this connection. Sessions allow you to configure
|
||||||
@@ -224,7 +229,9 @@ __all__ = [
|
|||||||
"connect",
|
"connect",
|
||||||
"connect_async",
|
"connect_async",
|
||||||
"connect_namespace",
|
"connect_namespace",
|
||||||
|
"connect_namespace_async",
|
||||||
"AsyncConnection",
|
"AsyncConnection",
|
||||||
|
"AsyncLanceNamespaceDBConnection",
|
||||||
"AsyncTable",
|
"AsyncTable",
|
||||||
"URI",
|
"URI",
|
||||||
"sanitize_uri",
|
"sanitize_uri",
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ class Connection(object):
|
|||||||
async def close(self): ...
|
async def close(self): ...
|
||||||
async def list_namespaces(
|
async def list_namespaces(
|
||||||
self,
|
self,
|
||||||
namespace: List[str],
|
namespace: Optional[List[str]],
|
||||||
page_token: Optional[str],
|
page_token: Optional[str],
|
||||||
limit: Optional[int],
|
limit: Optional[int],
|
||||||
) -> List[str]: ...
|
) -> List[str]: ...
|
||||||
@@ -34,7 +34,7 @@ class Connection(object):
|
|||||||
async def drop_namespace(self, namespace: List[str]) -> None: ...
|
async def drop_namespace(self, namespace: List[str]) -> None: ...
|
||||||
async def table_names(
|
async def table_names(
|
||||||
self,
|
self,
|
||||||
namespace: List[str],
|
namespace: Optional[List[str]],
|
||||||
start_after: Optional[str],
|
start_after: Optional[str],
|
||||||
limit: Optional[int],
|
limit: Optional[int],
|
||||||
) -> list[str]: ...
|
) -> list[str]: ...
|
||||||
@@ -43,7 +43,7 @@ class Connection(object):
|
|||||||
name: str,
|
name: str,
|
||||||
mode: str,
|
mode: str,
|
||||||
data: pa.RecordBatchReader,
|
data: pa.RecordBatchReader,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
storage_options_provider: Optional[StorageOptionsProvider] = None,
|
storage_options_provider: Optional[StorageOptionsProvider] = None,
|
||||||
location: Optional[str] = None,
|
location: Optional[str] = None,
|
||||||
@@ -53,7 +53,7 @@ class Connection(object):
|
|||||||
name: str,
|
name: str,
|
||||||
mode: str,
|
mode: str,
|
||||||
schema: pa.Schema,
|
schema: pa.Schema,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
storage_options_provider: Optional[StorageOptionsProvider] = None,
|
storage_options_provider: Optional[StorageOptionsProvider] = None,
|
||||||
location: Optional[str] = None,
|
location: Optional[str] = None,
|
||||||
@@ -61,7 +61,7 @@ class Connection(object):
|
|||||||
async def open_table(
|
async def open_table(
|
||||||
self,
|
self,
|
||||||
name: str,
|
name: str,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
storage_options_provider: Optional[StorageOptionsProvider] = None,
|
storage_options_provider: Optional[StorageOptionsProvider] = None,
|
||||||
index_cache_size: Optional[int] = None,
|
index_cache_size: Optional[int] = None,
|
||||||
@@ -71,7 +71,7 @@ class Connection(object):
|
|||||||
self,
|
self,
|
||||||
target_table_name: str,
|
target_table_name: str,
|
||||||
source_uri: str,
|
source_uri: str,
|
||||||
target_namespace: List[str] = [],
|
target_namespace: Optional[List[str]] = None,
|
||||||
source_version: Optional[int] = None,
|
source_version: Optional[int] = None,
|
||||||
source_tag: Optional[str] = None,
|
source_tag: Optional[str] = None,
|
||||||
is_shallow: bool = True,
|
is_shallow: bool = True,
|
||||||
@@ -80,11 +80,13 @@ class Connection(object):
|
|||||||
self,
|
self,
|
||||||
cur_name: str,
|
cur_name: str,
|
||||||
new_name: str,
|
new_name: str,
|
||||||
cur_namespace: List[str] = [],
|
cur_namespace: Optional[List[str]] = None,
|
||||||
new_namespace: List[str] = [],
|
new_namespace: Optional[List[str]] = None,
|
||||||
) -> None: ...
|
) -> None: ...
|
||||||
async def drop_table(self, name: str, namespace: List[str] = []) -> None: ...
|
async def drop_table(
|
||||||
async def drop_all_tables(self, namespace: List[str] = []) -> None: ...
|
self, name: str, namespace: Optional[List[str]] = None
|
||||||
|
) -> None: ...
|
||||||
|
async def drop_all_tables(self, namespace: Optional[List[str]] = None) -> None: ...
|
||||||
|
|
||||||
class Table:
|
class Table:
|
||||||
def name(self) -> str: ...
|
def name(self) -> str: ...
|
||||||
@@ -306,6 +308,7 @@ class MergeResult:
|
|||||||
num_updated_rows: int
|
num_updated_rows: int
|
||||||
num_inserted_rows: int
|
num_inserted_rows: int
|
||||||
num_deleted_rows: int
|
num_deleted_rows: int
|
||||||
|
num_attempts: int
|
||||||
|
|
||||||
class AddColumnsResult:
|
class AddColumnsResult:
|
||||||
version: int
|
version: int
|
||||||
|
|||||||
@@ -96,7 +96,7 @@ def data_to_reader(
|
|||||||
f"Unknown data type {type(data)}. "
|
f"Unknown data type {type(data)}. "
|
||||||
"Supported types: list of dicts, pandas DataFrame, polars DataFrame, "
|
"Supported types: list of dicts, pandas DataFrame, polars DataFrame, "
|
||||||
"pyarrow Table/RecordBatch, or Pydantic models. "
|
"pyarrow Table/RecordBatch, or Pydantic models. "
|
||||||
"See https://lancedb.github.io/lancedb/guides/tables/ for examples."
|
"See https://lancedb.com/docs/tables/ for examples."
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -54,7 +54,7 @@ class DBConnection(EnforceOverrides):
|
|||||||
|
|
||||||
def list_namespaces(
|
def list_namespaces(
|
||||||
self,
|
self,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
page_token: Optional[str] = None,
|
page_token: Optional[str] = None,
|
||||||
limit: int = 10,
|
limit: int = 10,
|
||||||
) -> Iterable[str]:
|
) -> Iterable[str]:
|
||||||
@@ -75,6 +75,8 @@ class DBConnection(EnforceOverrides):
|
|||||||
Iterable of str
|
Iterable of str
|
||||||
List of immediate child namespace names
|
List of immediate child namespace names
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
return []
|
return []
|
||||||
|
|
||||||
def create_namespace(self, namespace: List[str]) -> None:
|
def create_namespace(self, namespace: List[str]) -> None:
|
||||||
@@ -107,7 +109,7 @@ class DBConnection(EnforceOverrides):
|
|||||||
page_token: Optional[str] = None,
|
page_token: Optional[str] = None,
|
||||||
limit: int = 10,
|
limit: int = 10,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
) -> Iterable[str]:
|
) -> Iterable[str]:
|
||||||
"""List all tables in this database, in sorted order
|
"""List all tables in this database, in sorted order
|
||||||
|
|
||||||
@@ -142,7 +144,7 @@ class DBConnection(EnforceOverrides):
|
|||||||
fill_value: float = 0.0,
|
fill_value: float = 0.0,
|
||||||
embedding_functions: Optional[List[EmbeddingFunctionConfig]] = None,
|
embedding_functions: Optional[List[EmbeddingFunctionConfig]] = None,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
||||||
data_storage_version: Optional[str] = None,
|
data_storage_version: Optional[str] = None,
|
||||||
@@ -191,7 +193,11 @@ class DBConnection(EnforceOverrides):
|
|||||||
Additional options for the storage backend. Options already set on the
|
Additional options for the storage backend. Options already set on the
|
||||||
connection will be inherited by the table, but can be overridden here.
|
connection will be inherited by the table, but can be overridden here.
|
||||||
See available options at
|
See available options at
|
||||||
<https://lancedb.github.io/lancedb/guides/storage/>
|
<https://lancedb.com/docs/storage/>
|
||||||
|
|
||||||
|
To enable stable row IDs (row IDs remain stable after compaction,
|
||||||
|
update, delete, and merges), set `new_table_enable_stable_row_ids`
|
||||||
|
to `"true"` in storage_options when connecting to the database.
|
||||||
data_storage_version: optional, str, default "stable"
|
data_storage_version: optional, str, default "stable"
|
||||||
Deprecated. Set `storage_options` when connecting to the database and set
|
Deprecated. Set `storage_options` when connecting to the database and set
|
||||||
`new_table_data_storage_version` in the options.
|
`new_table_data_storage_version` in the options.
|
||||||
@@ -308,7 +314,7 @@ class DBConnection(EnforceOverrides):
|
|||||||
self,
|
self,
|
||||||
name: str,
|
name: str,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
||||||
index_cache_size: Optional[int] = None,
|
index_cache_size: Optional[int] = None,
|
||||||
@@ -339,7 +345,7 @@ class DBConnection(EnforceOverrides):
|
|||||||
Additional options for the storage backend. Options already set on the
|
Additional options for the storage backend. Options already set on the
|
||||||
connection will be inherited by the table, but can be overridden here.
|
connection will be inherited by the table, but can be overridden here.
|
||||||
See available options at
|
See available options at
|
||||||
<https://lancedb.github.io/lancedb/guides/storage/>
|
<https://lancedb.com/docs/storage/>
|
||||||
|
|
||||||
Returns
|
Returns
|
||||||
-------
|
-------
|
||||||
@@ -347,7 +353,7 @@ class DBConnection(EnforceOverrides):
|
|||||||
"""
|
"""
|
||||||
raise NotImplementedError
|
raise NotImplementedError
|
||||||
|
|
||||||
def drop_table(self, name: str, namespace: List[str] = []):
|
def drop_table(self, name: str, namespace: Optional[List[str]] = None):
|
||||||
"""Drop a table from the database.
|
"""Drop a table from the database.
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
@@ -358,14 +364,16 @@ class DBConnection(EnforceOverrides):
|
|||||||
The namespace to drop the table from.
|
The namespace to drop the table from.
|
||||||
Empty list represents root namespace.
|
Empty list represents root namespace.
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
raise NotImplementedError
|
raise NotImplementedError
|
||||||
|
|
||||||
def rename_table(
|
def rename_table(
|
||||||
self,
|
self,
|
||||||
cur_name: str,
|
cur_name: str,
|
||||||
new_name: str,
|
new_name: str,
|
||||||
cur_namespace: List[str] = [],
|
cur_namespace: Optional[List[str]] = None,
|
||||||
new_namespace: List[str] = [],
|
new_namespace: Optional[List[str]] = None,
|
||||||
):
|
):
|
||||||
"""Rename a table in the database.
|
"""Rename a table in the database.
|
||||||
|
|
||||||
@@ -382,6 +390,10 @@ class DBConnection(EnforceOverrides):
|
|||||||
The namespace to move the table to.
|
The namespace to move the table to.
|
||||||
If not specified, defaults to the same as cur_namespace.
|
If not specified, defaults to the same as cur_namespace.
|
||||||
"""
|
"""
|
||||||
|
if cur_namespace is None:
|
||||||
|
cur_namespace = []
|
||||||
|
if new_namespace is None:
|
||||||
|
new_namespace = []
|
||||||
raise NotImplementedError
|
raise NotImplementedError
|
||||||
|
|
||||||
def drop_database(self):
|
def drop_database(self):
|
||||||
@@ -391,7 +403,7 @@ class DBConnection(EnforceOverrides):
|
|||||||
"""
|
"""
|
||||||
raise NotImplementedError
|
raise NotImplementedError
|
||||||
|
|
||||||
def drop_all_tables(self, namespace: List[str] = []):
|
def drop_all_tables(self, namespace: Optional[List[str]] = None):
|
||||||
"""
|
"""
|
||||||
Drop all tables from the database
|
Drop all tables from the database
|
||||||
|
|
||||||
@@ -401,6 +413,8 @@ class DBConnection(EnforceOverrides):
|
|||||||
The namespace to drop all tables from.
|
The namespace to drop all tables from.
|
||||||
None or empty list represents root namespace.
|
None or empty list represents root namespace.
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
raise NotImplementedError
|
raise NotImplementedError
|
||||||
|
|
||||||
@property
|
@property
|
||||||
@@ -472,6 +486,12 @@ class LanceDBConnection(DBConnection):
|
|||||||
uri = uri[7:] # Remove "file://"
|
uri = uri[7:] # Remove "file://"
|
||||||
elif uri.startswith("file:/"):
|
elif uri.startswith("file:/"):
|
||||||
uri = uri[5:] # Remove "file:"
|
uri = uri[5:] # Remove "file:"
|
||||||
|
|
||||||
|
if sys.platform == "win32":
|
||||||
|
# On Windows, a path like /C:/path should become C:/path
|
||||||
|
if len(uri) >= 3 and uri[0] == "/" and uri[2] == ":":
|
||||||
|
uri = uri[1:]
|
||||||
|
|
||||||
uri = Path(uri)
|
uri = Path(uri)
|
||||||
uri = uri.expanduser().absolute()
|
uri = uri.expanduser().absolute()
|
||||||
Path(uri).mkdir(parents=True, exist_ok=True)
|
Path(uri).mkdir(parents=True, exist_ok=True)
|
||||||
@@ -535,7 +555,7 @@ class LanceDBConnection(DBConnection):
|
|||||||
@override
|
@override
|
||||||
def list_namespaces(
|
def list_namespaces(
|
||||||
self,
|
self,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
page_token: Optional[str] = None,
|
page_token: Optional[str] = None,
|
||||||
limit: int = 10,
|
limit: int = 10,
|
||||||
) -> Iterable[str]:
|
) -> Iterable[str]:
|
||||||
@@ -556,6 +576,8 @@ class LanceDBConnection(DBConnection):
|
|||||||
Iterable of str
|
Iterable of str
|
||||||
List of immediate child namespace names
|
List of immediate child namespace names
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
return LOOP.run(
|
return LOOP.run(
|
||||||
self._conn.list_namespaces(
|
self._conn.list_namespaces(
|
||||||
namespace=namespace, page_token=page_token, limit=limit
|
namespace=namespace, page_token=page_token, limit=limit
|
||||||
@@ -590,7 +612,7 @@ class LanceDBConnection(DBConnection):
|
|||||||
page_token: Optional[str] = None,
|
page_token: Optional[str] = None,
|
||||||
limit: int = 10,
|
limit: int = 10,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
) -> Iterable[str]:
|
) -> Iterable[str]:
|
||||||
"""Get the names of all tables in the database. The names are sorted.
|
"""Get the names of all tables in the database. The names are sorted.
|
||||||
|
|
||||||
@@ -608,6 +630,8 @@ class LanceDBConnection(DBConnection):
|
|||||||
Iterator of str.
|
Iterator of str.
|
||||||
A list of table names.
|
A list of table names.
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
return LOOP.run(
|
return LOOP.run(
|
||||||
self._conn.table_names(
|
self._conn.table_names(
|
||||||
namespace=namespace, start_after=page_token, limit=limit
|
namespace=namespace, start_after=page_token, limit=limit
|
||||||
@@ -632,7 +656,7 @@ class LanceDBConnection(DBConnection):
|
|||||||
fill_value: float = 0.0,
|
fill_value: float = 0.0,
|
||||||
embedding_functions: Optional[List[EmbeddingFunctionConfig]] = None,
|
embedding_functions: Optional[List[EmbeddingFunctionConfig]] = None,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
||||||
data_storage_version: Optional[str] = None,
|
data_storage_version: Optional[str] = None,
|
||||||
@@ -649,6 +673,8 @@ class LanceDBConnection(DBConnection):
|
|||||||
---
|
---
|
||||||
DBConnection.create_table
|
DBConnection.create_table
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
if mode.lower() not in ["create", "overwrite"]:
|
if mode.lower() not in ["create", "overwrite"]:
|
||||||
raise ValueError("mode must be either 'create' or 'overwrite'")
|
raise ValueError("mode must be either 'create' or 'overwrite'")
|
||||||
validate_table_name(name)
|
validate_table_name(name)
|
||||||
@@ -674,7 +700,7 @@ class LanceDBConnection(DBConnection):
|
|||||||
self,
|
self,
|
||||||
name: str,
|
name: str,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
||||||
index_cache_size: Optional[int] = None,
|
index_cache_size: Optional[int] = None,
|
||||||
@@ -692,6 +718,8 @@ class LanceDBConnection(DBConnection):
|
|||||||
-------
|
-------
|
||||||
A LanceTable object representing the table.
|
A LanceTable object representing the table.
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
if index_cache_size is not None:
|
if index_cache_size is not None:
|
||||||
import warnings
|
import warnings
|
||||||
|
|
||||||
@@ -717,7 +745,7 @@ class LanceDBConnection(DBConnection):
|
|||||||
target_table_name: str,
|
target_table_name: str,
|
||||||
source_uri: str,
|
source_uri: str,
|
||||||
*,
|
*,
|
||||||
target_namespace: List[str] = [],
|
target_namespace: Optional[List[str]] = None,
|
||||||
source_version: Optional[int] = None,
|
source_version: Optional[int] = None,
|
||||||
source_tag: Optional[str] = None,
|
source_tag: Optional[str] = None,
|
||||||
is_shallow: bool = True,
|
is_shallow: bool = True,
|
||||||
@@ -750,6 +778,8 @@ class LanceDBConnection(DBConnection):
|
|||||||
-------
|
-------
|
||||||
A LanceTable object representing the cloned table.
|
A LanceTable object representing the cloned table.
|
||||||
"""
|
"""
|
||||||
|
if target_namespace is None:
|
||||||
|
target_namespace = []
|
||||||
LOOP.run(
|
LOOP.run(
|
||||||
self._conn.clone_table(
|
self._conn.clone_table(
|
||||||
target_table_name,
|
target_table_name,
|
||||||
@@ -770,7 +800,7 @@ class LanceDBConnection(DBConnection):
|
|||||||
def drop_table(
|
def drop_table(
|
||||||
self,
|
self,
|
||||||
name: str,
|
name: str,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
ignore_missing: bool = False,
|
ignore_missing: bool = False,
|
||||||
):
|
):
|
||||||
"""Drop a table from the database.
|
"""Drop a table from the database.
|
||||||
@@ -784,6 +814,8 @@ class LanceDBConnection(DBConnection):
|
|||||||
ignore_missing: bool, default False
|
ignore_missing: bool, default False
|
||||||
If True, ignore if the table does not exist.
|
If True, ignore if the table does not exist.
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
LOOP.run(
|
LOOP.run(
|
||||||
self._conn.drop_table(
|
self._conn.drop_table(
|
||||||
name, namespace=namespace, ignore_missing=ignore_missing
|
name, namespace=namespace, ignore_missing=ignore_missing
|
||||||
@@ -791,7 +823,9 @@ class LanceDBConnection(DBConnection):
|
|||||||
)
|
)
|
||||||
|
|
||||||
@override
|
@override
|
||||||
def drop_all_tables(self, namespace: List[str] = []):
|
def drop_all_tables(self, namespace: Optional[List[str]] = None):
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
LOOP.run(self._conn.drop_all_tables(namespace=namespace))
|
LOOP.run(self._conn.drop_all_tables(namespace=namespace))
|
||||||
|
|
||||||
@override
|
@override
|
||||||
@@ -799,8 +833,8 @@ class LanceDBConnection(DBConnection):
|
|||||||
self,
|
self,
|
||||||
cur_name: str,
|
cur_name: str,
|
||||||
new_name: str,
|
new_name: str,
|
||||||
cur_namespace: List[str] = [],
|
cur_namespace: Optional[List[str]] = None,
|
||||||
new_namespace: List[str] = [],
|
new_namespace: Optional[List[str]] = None,
|
||||||
):
|
):
|
||||||
"""Rename a table in the database.
|
"""Rename a table in the database.
|
||||||
|
|
||||||
@@ -815,6 +849,10 @@ class LanceDBConnection(DBConnection):
|
|||||||
new_namespace: List[str], optional
|
new_namespace: List[str], optional
|
||||||
The namespace to move the table to.
|
The namespace to move the table to.
|
||||||
"""
|
"""
|
||||||
|
if cur_namespace is None:
|
||||||
|
cur_namespace = []
|
||||||
|
if new_namespace is None:
|
||||||
|
new_namespace = []
|
||||||
LOOP.run(
|
LOOP.run(
|
||||||
self._conn.rename_table(
|
self._conn.rename_table(
|
||||||
cur_name,
|
cur_name,
|
||||||
@@ -904,7 +942,7 @@ class AsyncConnection(object):
|
|||||||
|
|
||||||
async def list_namespaces(
|
async def list_namespaces(
|
||||||
self,
|
self,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
page_token: Optional[str] = None,
|
page_token: Optional[str] = None,
|
||||||
limit: int = 10,
|
limit: int = 10,
|
||||||
) -> Iterable[str]:
|
) -> Iterable[str]:
|
||||||
@@ -925,6 +963,8 @@ class AsyncConnection(object):
|
|||||||
Iterable of str
|
Iterable of str
|
||||||
List of immediate child namespace names (not full paths)
|
List of immediate child namespace names (not full paths)
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
return await self._inner.list_namespaces(
|
return await self._inner.list_namespaces(
|
||||||
namespace=namespace, page_token=page_token, limit=limit
|
namespace=namespace, page_token=page_token, limit=limit
|
||||||
)
|
)
|
||||||
@@ -952,7 +992,7 @@ class AsyncConnection(object):
|
|||||||
async def table_names(
|
async def table_names(
|
||||||
self,
|
self,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
start_after: Optional[str] = None,
|
start_after: Optional[str] = None,
|
||||||
limit: Optional[int] = None,
|
limit: Optional[int] = None,
|
||||||
) -> Iterable[str]:
|
) -> Iterable[str]:
|
||||||
@@ -976,6 +1016,8 @@ class AsyncConnection(object):
|
|||||||
-------
|
-------
|
||||||
Iterable of str
|
Iterable of str
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
return await self._inner.table_names(
|
return await self._inner.table_names(
|
||||||
namespace=namespace, start_after=start_after, limit=limit
|
namespace=namespace, start_after=start_after, limit=limit
|
||||||
)
|
)
|
||||||
@@ -992,7 +1034,7 @@ class AsyncConnection(object):
|
|||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
embedding_functions: Optional[List[EmbeddingFunctionConfig]] = None,
|
embedding_functions: Optional[List[EmbeddingFunctionConfig]] = None,
|
||||||
location: Optional[str] = None,
|
location: Optional[str] = None,
|
||||||
) -> AsyncTable:
|
) -> AsyncTable:
|
||||||
@@ -1039,7 +1081,11 @@ class AsyncConnection(object):
|
|||||||
Additional options for the storage backend. Options already set on the
|
Additional options for the storage backend. Options already set on the
|
||||||
connection will be inherited by the table, but can be overridden here.
|
connection will be inherited by the table, but can be overridden here.
|
||||||
See available options at
|
See available options at
|
||||||
<https://lancedb.github.io/lancedb/guides/storage/>
|
<https://lancedb.com/docs/storage/>
|
||||||
|
|
||||||
|
To enable stable row IDs (row IDs remain stable after compaction,
|
||||||
|
update, delete, and merges), set `new_table_enable_stable_row_ids`
|
||||||
|
to `"true"` in storage_options when connecting to the database.
|
||||||
|
|
||||||
Returns
|
Returns
|
||||||
-------
|
-------
|
||||||
@@ -1149,6 +1195,8 @@ class AsyncConnection(object):
|
|||||||
... await db.create_table("table4", make_batches(), schema=schema)
|
... await db.create_table("table4", make_batches(), schema=schema)
|
||||||
>>> asyncio.run(iterable_example())
|
>>> asyncio.run(iterable_example())
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
metadata = None
|
metadata = None
|
||||||
|
|
||||||
if embedding_functions is not None:
|
if embedding_functions is not None:
|
||||||
@@ -1206,7 +1254,7 @@ class AsyncConnection(object):
|
|||||||
self,
|
self,
|
||||||
name: str,
|
name: str,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
||||||
index_cache_size: Optional[int] = None,
|
index_cache_size: Optional[int] = None,
|
||||||
@@ -1225,7 +1273,7 @@ class AsyncConnection(object):
|
|||||||
Additional options for the storage backend. Options already set on the
|
Additional options for the storage backend. Options already set on the
|
||||||
connection will be inherited by the table, but can be overridden here.
|
connection will be inherited by the table, but can be overridden here.
|
||||||
See available options at
|
See available options at
|
||||||
<https://lancedb.github.io/lancedb/guides/storage/>
|
<https://lancedb.com/docs/storage/>
|
||||||
index_cache_size: int, default 256
|
index_cache_size: int, default 256
|
||||||
**Deprecated**: Use session-level cache configuration instead.
|
**Deprecated**: Use session-level cache configuration instead.
|
||||||
Create a Session with custom cache sizes and pass it to lancedb.connect().
|
Create a Session with custom cache sizes and pass it to lancedb.connect().
|
||||||
@@ -1248,6 +1296,8 @@ class AsyncConnection(object):
|
|||||||
-------
|
-------
|
||||||
A LanceTable object representing the table.
|
A LanceTable object representing the table.
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
table = await self._inner.open_table(
|
table = await self._inner.open_table(
|
||||||
name,
|
name,
|
||||||
namespace=namespace,
|
namespace=namespace,
|
||||||
@@ -1263,7 +1313,7 @@ class AsyncConnection(object):
|
|||||||
target_table_name: str,
|
target_table_name: str,
|
||||||
source_uri: str,
|
source_uri: str,
|
||||||
*,
|
*,
|
||||||
target_namespace: List[str] = [],
|
target_namespace: Optional[List[str]] = None,
|
||||||
source_version: Optional[int] = None,
|
source_version: Optional[int] = None,
|
||||||
source_tag: Optional[str] = None,
|
source_tag: Optional[str] = None,
|
||||||
is_shallow: bool = True,
|
is_shallow: bool = True,
|
||||||
@@ -1296,6 +1346,8 @@ class AsyncConnection(object):
|
|||||||
-------
|
-------
|
||||||
An AsyncTable object representing the cloned table.
|
An AsyncTable object representing the cloned table.
|
||||||
"""
|
"""
|
||||||
|
if target_namespace is None:
|
||||||
|
target_namespace = []
|
||||||
table = await self._inner.clone_table(
|
table = await self._inner.clone_table(
|
||||||
target_table_name,
|
target_table_name,
|
||||||
source_uri,
|
source_uri,
|
||||||
@@ -1310,8 +1362,8 @@ class AsyncConnection(object):
|
|||||||
self,
|
self,
|
||||||
cur_name: str,
|
cur_name: str,
|
||||||
new_name: str,
|
new_name: str,
|
||||||
cur_namespace: List[str] = [],
|
cur_namespace: Optional[List[str]] = None,
|
||||||
new_namespace: List[str] = [],
|
new_namespace: Optional[List[str]] = None,
|
||||||
):
|
):
|
||||||
"""Rename a table in the database.
|
"""Rename a table in the database.
|
||||||
|
|
||||||
@@ -1328,6 +1380,10 @@ class AsyncConnection(object):
|
|||||||
The namespace to move the table to.
|
The namespace to move the table to.
|
||||||
If not specified, defaults to the same as cur_namespace.
|
If not specified, defaults to the same as cur_namespace.
|
||||||
"""
|
"""
|
||||||
|
if cur_namespace is None:
|
||||||
|
cur_namespace = []
|
||||||
|
if new_namespace is None:
|
||||||
|
new_namespace = []
|
||||||
await self._inner.rename_table(
|
await self._inner.rename_table(
|
||||||
cur_name, new_name, cur_namespace=cur_namespace, new_namespace=new_namespace
|
cur_name, new_name, cur_namespace=cur_namespace, new_namespace=new_namespace
|
||||||
)
|
)
|
||||||
@@ -1336,7 +1392,7 @@ class AsyncConnection(object):
|
|||||||
self,
|
self,
|
||||||
name: str,
|
name: str,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
ignore_missing: bool = False,
|
ignore_missing: bool = False,
|
||||||
):
|
):
|
||||||
"""Drop a table from the database.
|
"""Drop a table from the database.
|
||||||
@@ -1351,6 +1407,8 @@ class AsyncConnection(object):
|
|||||||
ignore_missing: bool, default False
|
ignore_missing: bool, default False
|
||||||
If True, ignore if the table does not exist.
|
If True, ignore if the table does not exist.
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
try:
|
try:
|
||||||
await self._inner.drop_table(name, namespace=namespace)
|
await self._inner.drop_table(name, namespace=namespace)
|
||||||
except ValueError as e:
|
except ValueError as e:
|
||||||
@@ -1359,7 +1417,7 @@ class AsyncConnection(object):
|
|||||||
if f"Table '{name}' was not found" not in str(e):
|
if f"Table '{name}' was not found" not in str(e):
|
||||||
raise e
|
raise e
|
||||||
|
|
||||||
async def drop_all_tables(self, namespace: List[str] = []):
|
async def drop_all_tables(self, namespace: Optional[List[str]] = None):
|
||||||
"""Drop all tables from the database.
|
"""Drop all tables from the database.
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
@@ -1368,6 +1426,8 @@ class AsyncConnection(object):
|
|||||||
The namespace to drop all tables from.
|
The namespace to drop all tables from.
|
||||||
None or empty list represents root namespace.
|
None or empty list represents root namespace.
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
await self._inner.drop_all_tables(namespace=namespace)
|
await self._inner.drop_all_tables(namespace=namespace)
|
||||||
|
|
||||||
@deprecation.deprecated(
|
@deprecation.deprecated(
|
||||||
|
|||||||
@@ -609,9 +609,19 @@ class IvfPq:
|
|||||||
class IvfRq:
|
class IvfRq:
|
||||||
"""Describes an IVF RQ Index
|
"""Describes an IVF RQ Index
|
||||||
|
|
||||||
IVF-RQ (Residual Quantization) stores a compressed copy of each vector using
|
IVF-RQ (RabitQ Quantization) compresses vectors using RabitQ quantization
|
||||||
residual quantization and organizes them into IVF partitions. Parameters
|
and organizes them into IVF partitions.
|
||||||
largely mirror IVF-PQ for consistency.
|
|
||||||
|
The compression scheme is called RabitQ quantization. Each dimension is
|
||||||
|
quantized into a small number of bits. The parameters `num_bits` and
|
||||||
|
`num_partitions` control this process, providing a tradeoff between
|
||||||
|
index size (and thus search speed) and index accuracy.
|
||||||
|
|
||||||
|
The partitioning process is called IVF and the `num_partitions` parameter
|
||||||
|
controls how many groups to create.
|
||||||
|
|
||||||
|
Note that training an IVF RQ index on a large dataset is a slow operation
|
||||||
|
and currently is also a memory intensive operation.
|
||||||
|
|
||||||
Attributes
|
Attributes
|
||||||
----------
|
----------
|
||||||
@@ -628,7 +638,7 @@ class IvfRq:
|
|||||||
Number of IVF partitions to create.
|
Number of IVF partitions to create.
|
||||||
|
|
||||||
num_bits: int, default 1
|
num_bits: int, default 1
|
||||||
Number of bits to encode each dimension.
|
Number of bits to encode each dimension in the RabitQ codebook.
|
||||||
|
|
||||||
max_iterations: int, default 50
|
max_iterations: int, default 50
|
||||||
Max iterations to train kmeans when computing IVF partitions.
|
Max iterations to train kmeans when computing IVF partitions.
|
||||||
|
|||||||
@@ -10,6 +10,7 @@ through a namespace abstraction.
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
import sys
|
import sys
|
||||||
from typing import Dict, Iterable, List, Optional, Union
|
from typing import Dict, Iterable, List, Optional, Union
|
||||||
|
|
||||||
@@ -23,7 +24,7 @@ import pyarrow as pa
|
|||||||
|
|
||||||
from lancedb.db import DBConnection, LanceDBConnection
|
from lancedb.db import DBConnection, LanceDBConnection
|
||||||
from lancedb.io import StorageOptionsProvider
|
from lancedb.io import StorageOptionsProvider
|
||||||
from lancedb.table import LanceTable, Table
|
from lancedb.table import AsyncTable, LanceTable, Table
|
||||||
from lancedb.util import validate_table_name
|
from lancedb.util import validate_table_name
|
||||||
from lancedb.common import DATA
|
from lancedb.common import DATA
|
||||||
from lancedb.pydantic import LanceModel
|
from lancedb.pydantic import LanceModel
|
||||||
@@ -99,7 +100,14 @@ def _convert_pyarrow_schema_to_json(schema: pa.Schema) -> JsonArrowSchema:
|
|||||||
)
|
)
|
||||||
fields.append(json_field)
|
fields.append(json_field)
|
||||||
|
|
||||||
return JsonArrowSchema(fields=fields, metadata=schema.metadata)
|
# decode binary metadata to strings for JSON
|
||||||
|
meta = None
|
||||||
|
if schema.metadata:
|
||||||
|
meta = {
|
||||||
|
k.decode("utf-8"): v.decode("utf-8") for k, v in schema.metadata.items()
|
||||||
|
}
|
||||||
|
|
||||||
|
return JsonArrowSchema(fields=fields, metadata=meta)
|
||||||
|
|
||||||
|
|
||||||
class LanceNamespaceStorageOptionsProvider(StorageOptionsProvider):
|
class LanceNamespaceStorageOptionsProvider(StorageOptionsProvider):
|
||||||
@@ -119,13 +127,17 @@ class LanceNamespaceStorageOptionsProvider(StorageOptionsProvider):
|
|||||||
|
|
||||||
Examples
|
Examples
|
||||||
--------
|
--------
|
||||||
>>> from lance_namespace import connect as namespace_connect
|
Create a provider and fetch storage options::
|
||||||
>>> namespace = namespace_connect("rest", {"url": "https://..."})
|
|
||||||
>>> provider = LanceNamespaceStorageOptionsProvider(
|
from lance_namespace import connect as namespace_connect
|
||||||
... namespace=namespace,
|
|
||||||
... table_id=["my_namespace", "my_table"]
|
# Connect to namespace (requires a running namespace server)
|
||||||
... )
|
namespace = namespace_connect("rest", {"uri": "https://..."})
|
||||||
>>> options = provider.fetch_storage_options()
|
provider = LanceNamespaceStorageOptionsProvider(
|
||||||
|
namespace=namespace,
|
||||||
|
table_id=["my_namespace", "my_table"]
|
||||||
|
)
|
||||||
|
options = provider.fetch_storage_options()
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(self, namespace: LanceNamespace, table_id: List[str]):
|
def __init__(self, namespace: LanceNamespace, table_id: List[str]):
|
||||||
@@ -227,8 +239,10 @@ class LanceNamespaceDBConnection(DBConnection):
|
|||||||
page_token: Optional[str] = None,
|
page_token: Optional[str] = None,
|
||||||
limit: int = 10,
|
limit: int = 10,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
) -> Iterable[str]:
|
) -> Iterable[str]:
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
request = ListTablesRequest(id=namespace, page_token=page_token, limit=limit)
|
request = ListTablesRequest(id=namespace, page_token=page_token, limit=limit)
|
||||||
response = self._ns.list_tables(request)
|
response = self._ns.list_tables(request)
|
||||||
return response.tables if response.tables else []
|
return response.tables if response.tables else []
|
||||||
@@ -245,12 +259,14 @@ class LanceNamespaceDBConnection(DBConnection):
|
|||||||
fill_value: float = 0.0,
|
fill_value: float = 0.0,
|
||||||
embedding_functions: Optional[List[EmbeddingFunctionConfig]] = None,
|
embedding_functions: Optional[List[EmbeddingFunctionConfig]] = None,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
storage_options_provider: Optional[StorageOptionsProvider] = None,
|
storage_options_provider: Optional[StorageOptionsProvider] = None,
|
||||||
data_storage_version: Optional[str] = None,
|
data_storage_version: Optional[str] = None,
|
||||||
enable_v2_manifest_paths: Optional[bool] = None,
|
enable_v2_manifest_paths: Optional[bool] = None,
|
||||||
) -> Table:
|
) -> Table:
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
if mode.lower() not in ["create", "overwrite"]:
|
if mode.lower() not in ["create", "overwrite"]:
|
||||||
raise ValueError("mode must be either 'create' or 'overwrite'")
|
raise ValueError("mode must be either 'create' or 'overwrite'")
|
||||||
validate_table_name(name)
|
validate_table_name(name)
|
||||||
@@ -339,11 +355,13 @@ class LanceNamespaceDBConnection(DBConnection):
|
|||||||
self,
|
self,
|
||||||
name: str,
|
name: str,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
storage_options_provider: Optional[StorageOptionsProvider] = None,
|
storage_options_provider: Optional[StorageOptionsProvider] = None,
|
||||||
index_cache_size: Optional[int] = None,
|
index_cache_size: Optional[int] = None,
|
||||||
) -> Table:
|
) -> Table:
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
table_id = namespace + [name]
|
table_id = namespace + [name]
|
||||||
request = DescribeTableRequest(id=table_id)
|
request = DescribeTableRequest(id=table_id)
|
||||||
response = self._ns.describe_table(request)
|
response = self._ns.describe_table(request)
|
||||||
@@ -373,8 +391,10 @@ class LanceNamespaceDBConnection(DBConnection):
|
|||||||
)
|
)
|
||||||
|
|
||||||
@override
|
@override
|
||||||
def drop_table(self, name: str, namespace: List[str] = []):
|
def drop_table(self, name: str, namespace: Optional[List[str]] = None):
|
||||||
# Use namespace drop_table directly
|
# Use namespace drop_table directly
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
table_id = namespace + [name]
|
table_id = namespace + [name]
|
||||||
request = DropTableRequest(id=table_id)
|
request = DropTableRequest(id=table_id)
|
||||||
self._ns.drop_table(request)
|
self._ns.drop_table(request)
|
||||||
@@ -384,9 +404,13 @@ class LanceNamespaceDBConnection(DBConnection):
|
|||||||
self,
|
self,
|
||||||
cur_name: str,
|
cur_name: str,
|
||||||
new_name: str,
|
new_name: str,
|
||||||
cur_namespace: List[str] = [],
|
cur_namespace: Optional[List[str]] = None,
|
||||||
new_namespace: List[str] = [],
|
new_namespace: Optional[List[str]] = None,
|
||||||
):
|
):
|
||||||
|
if cur_namespace is None:
|
||||||
|
cur_namespace = []
|
||||||
|
if new_namespace is None:
|
||||||
|
new_namespace = []
|
||||||
raise NotImplementedError(
|
raise NotImplementedError(
|
||||||
"rename_table is not supported for namespace connections"
|
"rename_table is not supported for namespace connections"
|
||||||
)
|
)
|
||||||
@@ -398,14 +422,16 @@ class LanceNamespaceDBConnection(DBConnection):
|
|||||||
)
|
)
|
||||||
|
|
||||||
@override
|
@override
|
||||||
def drop_all_tables(self, namespace: List[str] = []):
|
def drop_all_tables(self, namespace: Optional[List[str]] = None):
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
for table_name in self.table_names(namespace=namespace):
|
for table_name in self.table_names(namespace=namespace):
|
||||||
self.drop_table(table_name, namespace=namespace)
|
self.drop_table(table_name, namespace=namespace)
|
||||||
|
|
||||||
@override
|
@override
|
||||||
def list_namespaces(
|
def list_namespaces(
|
||||||
self,
|
self,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
page_token: Optional[str] = None,
|
page_token: Optional[str] = None,
|
||||||
limit: int = 10,
|
limit: int = 10,
|
||||||
) -> Iterable[str]:
|
) -> Iterable[str]:
|
||||||
@@ -427,6 +453,8 @@ class LanceNamespaceDBConnection(DBConnection):
|
|||||||
Iterable[str]
|
Iterable[str]
|
||||||
Names of child namespaces.
|
Names of child namespaces.
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
request = ListNamespacesRequest(
|
request = ListNamespacesRequest(
|
||||||
id=namespace, page_token=page_token, limit=limit
|
id=namespace, page_token=page_token, limit=limit
|
||||||
)
|
)
|
||||||
@@ -464,13 +492,15 @@ class LanceNamespaceDBConnection(DBConnection):
|
|||||||
name: str,
|
name: str,
|
||||||
table_uri: str,
|
table_uri: str,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
storage_options_provider: Optional[StorageOptionsProvider] = None,
|
storage_options_provider: Optional[StorageOptionsProvider] = None,
|
||||||
index_cache_size: Optional[int] = None,
|
index_cache_size: Optional[int] = None,
|
||||||
) -> LanceTable:
|
) -> LanceTable:
|
||||||
# Open a table directly from a URI using the location parameter
|
# Open a table directly from a URI using the location parameter
|
||||||
# Note: storage_options should already be merged by the caller
|
# Note: storage_options should already be merged by the caller
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
temp_conn = LanceDBConnection(
|
temp_conn = LanceDBConnection(
|
||||||
table_uri, # Use the table location as the connection URI
|
table_uri, # Use the table location as the connection URI
|
||||||
read_consistency_interval=self.read_consistency_interval,
|
read_consistency_interval=self.read_consistency_interval,
|
||||||
@@ -490,6 +520,310 @@ class LanceNamespaceDBConnection(DBConnection):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class AsyncLanceNamespaceDBConnection:
|
||||||
|
"""
|
||||||
|
An async LanceDB connection that uses a namespace for table management.
|
||||||
|
|
||||||
|
This connection delegates table URI resolution to a lance_namespace instance,
|
||||||
|
while providing async methods for all operations.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
namespace: LanceNamespace,
|
||||||
|
*,
|
||||||
|
read_consistency_interval: Optional[timedelta] = None,
|
||||||
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
|
session: Optional[Session] = None,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Initialize an async namespace-based LanceDB connection.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
namespace : LanceNamespace
|
||||||
|
The namespace instance to use for table management
|
||||||
|
read_consistency_interval : Optional[timedelta]
|
||||||
|
The interval at which to check for updates to the table from other
|
||||||
|
processes. If None, then consistency is not checked.
|
||||||
|
storage_options : Optional[Dict[str, str]]
|
||||||
|
Additional options for the storage backend
|
||||||
|
session : Optional[Session]
|
||||||
|
A session to use for this connection
|
||||||
|
"""
|
||||||
|
self._ns = namespace
|
||||||
|
self.read_consistency_interval = read_consistency_interval
|
||||||
|
self.storage_options = storage_options or {}
|
||||||
|
self.session = session
|
||||||
|
|
||||||
|
async def table_names(
|
||||||
|
self,
|
||||||
|
page_token: Optional[str] = None,
|
||||||
|
limit: int = 10,
|
||||||
|
*,
|
||||||
|
namespace: Optional[List[str]] = None,
|
||||||
|
) -> Iterable[str]:
|
||||||
|
"""List table names in the namespace."""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
|
request = ListTablesRequest(id=namespace, page_token=page_token, limit=limit)
|
||||||
|
response = self._ns.list_tables(request)
|
||||||
|
return response.tables if response.tables else []
|
||||||
|
|
||||||
|
async def create_table(
|
||||||
|
self,
|
||||||
|
name: str,
|
||||||
|
data: Optional[DATA] = None,
|
||||||
|
schema: Optional[Union[pa.Schema, LanceModel]] = None,
|
||||||
|
mode: str = "create",
|
||||||
|
exist_ok: bool = False,
|
||||||
|
on_bad_vectors: str = "error",
|
||||||
|
fill_value: float = 0.0,
|
||||||
|
embedding_functions: Optional[List[EmbeddingFunctionConfig]] = None,
|
||||||
|
*,
|
||||||
|
namespace: Optional[List[str]] = None,
|
||||||
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
|
storage_options_provider: Optional[StorageOptionsProvider] = None,
|
||||||
|
data_storage_version: Optional[str] = None,
|
||||||
|
enable_v2_manifest_paths: Optional[bool] = None,
|
||||||
|
) -> AsyncTable:
|
||||||
|
"""Create a new table in the namespace."""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
|
if mode.lower() not in ["create", "overwrite"]:
|
||||||
|
raise ValueError("mode must be either 'create' or 'overwrite'")
|
||||||
|
validate_table_name(name)
|
||||||
|
|
||||||
|
# Get location from namespace
|
||||||
|
table_id = namespace + [name]
|
||||||
|
|
||||||
|
# Step 1: Get the table location and storage options from namespace
|
||||||
|
location = None
|
||||||
|
namespace_storage_options = None
|
||||||
|
if mode.lower() == "overwrite":
|
||||||
|
# Try to describe the table first to see if it exists
|
||||||
|
try:
|
||||||
|
describe_request = DescribeTableRequest(id=table_id)
|
||||||
|
describe_response = self._ns.describe_table(describe_request)
|
||||||
|
location = describe_response.location
|
||||||
|
namespace_storage_options = describe_response.storage_options
|
||||||
|
except Exception:
|
||||||
|
# Table doesn't exist, will create a new one below
|
||||||
|
pass
|
||||||
|
|
||||||
|
if location is None:
|
||||||
|
# Table doesn't exist or mode is "create", reserve a new location
|
||||||
|
create_empty_request = CreateEmptyTableRequest(
|
||||||
|
id=table_id,
|
||||||
|
location=None,
|
||||||
|
properties=self.storage_options if self.storage_options else None,
|
||||||
|
)
|
||||||
|
create_empty_response = self._ns.create_empty_table(create_empty_request)
|
||||||
|
|
||||||
|
if not create_empty_response.location:
|
||||||
|
raise ValueError(
|
||||||
|
"Table location is missing from create_empty_table response"
|
||||||
|
)
|
||||||
|
|
||||||
|
location = create_empty_response.location
|
||||||
|
namespace_storage_options = create_empty_response.storage_options
|
||||||
|
|
||||||
|
# Merge storage options: self.storage_options < user options < namespace options
|
||||||
|
merged_storage_options = dict(self.storage_options)
|
||||||
|
if storage_options:
|
||||||
|
merged_storage_options.update(storage_options)
|
||||||
|
if namespace_storage_options:
|
||||||
|
merged_storage_options.update(namespace_storage_options)
|
||||||
|
|
||||||
|
# Step 2: Create table using LanceTable.create with the location
|
||||||
|
# Run the sync operation in a thread
|
||||||
|
def _create_table():
|
||||||
|
temp_conn = LanceDBConnection(
|
||||||
|
location,
|
||||||
|
read_consistency_interval=self.read_consistency_interval,
|
||||||
|
storage_options=merged_storage_options,
|
||||||
|
session=self.session,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create a storage options provider if not provided by user
|
||||||
|
if (
|
||||||
|
storage_options_provider is None
|
||||||
|
and namespace_storage_options is not None
|
||||||
|
):
|
||||||
|
provider = LanceNamespaceStorageOptionsProvider(
|
||||||
|
namespace=self._ns,
|
||||||
|
table_id=table_id,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
provider = storage_options_provider
|
||||||
|
|
||||||
|
return LanceTable.create(
|
||||||
|
temp_conn,
|
||||||
|
name,
|
||||||
|
data,
|
||||||
|
schema,
|
||||||
|
mode=mode,
|
||||||
|
exist_ok=exist_ok,
|
||||||
|
on_bad_vectors=on_bad_vectors,
|
||||||
|
fill_value=fill_value,
|
||||||
|
embedding_functions=embedding_functions,
|
||||||
|
namespace=namespace,
|
||||||
|
storage_options=merged_storage_options,
|
||||||
|
storage_options_provider=provider,
|
||||||
|
location=location,
|
||||||
|
)
|
||||||
|
|
||||||
|
lance_table = await asyncio.to_thread(_create_table)
|
||||||
|
# Get the underlying async table from LanceTable
|
||||||
|
return lance_table._table
|
||||||
|
|
||||||
|
async def open_table(
|
||||||
|
self,
|
||||||
|
name: str,
|
||||||
|
*,
|
||||||
|
namespace: Optional[List[str]] = None,
|
||||||
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
|
storage_options_provider: Optional[StorageOptionsProvider] = None,
|
||||||
|
index_cache_size: Optional[int] = None,
|
||||||
|
) -> AsyncTable:
|
||||||
|
"""Open an existing table from the namespace."""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
|
table_id = namespace + [name]
|
||||||
|
request = DescribeTableRequest(id=table_id)
|
||||||
|
response = self._ns.describe_table(request)
|
||||||
|
|
||||||
|
# Merge storage options: self.storage_options < user options < namespace options
|
||||||
|
merged_storage_options = dict(self.storage_options)
|
||||||
|
if storage_options:
|
||||||
|
merged_storage_options.update(storage_options)
|
||||||
|
if response.storage_options:
|
||||||
|
merged_storage_options.update(response.storage_options)
|
||||||
|
|
||||||
|
# Create a storage options provider if not provided by user
|
||||||
|
if storage_options_provider is None and response.storage_options is not None:
|
||||||
|
storage_options_provider = LanceNamespaceStorageOptionsProvider(
|
||||||
|
namespace=self._ns,
|
||||||
|
table_id=table_id,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Open table in a thread
|
||||||
|
def _open_table():
|
||||||
|
temp_conn = LanceDBConnection(
|
||||||
|
response.location,
|
||||||
|
read_consistency_interval=self.read_consistency_interval,
|
||||||
|
storage_options=merged_storage_options,
|
||||||
|
session=self.session,
|
||||||
|
)
|
||||||
|
|
||||||
|
return LanceTable.open(
|
||||||
|
temp_conn,
|
||||||
|
name,
|
||||||
|
namespace=namespace,
|
||||||
|
storage_options=merged_storage_options,
|
||||||
|
storage_options_provider=storage_options_provider,
|
||||||
|
index_cache_size=index_cache_size,
|
||||||
|
location=response.location,
|
||||||
|
)
|
||||||
|
|
||||||
|
lance_table = await asyncio.to_thread(_open_table)
|
||||||
|
return lance_table._table
|
||||||
|
|
||||||
|
async def drop_table(self, name: str, namespace: Optional[List[str]] = None):
|
||||||
|
"""Drop a table from the namespace."""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
|
table_id = namespace + [name]
|
||||||
|
request = DropTableRequest(id=table_id)
|
||||||
|
self._ns.drop_table(request)
|
||||||
|
|
||||||
|
async def rename_table(
|
||||||
|
self,
|
||||||
|
cur_name: str,
|
||||||
|
new_name: str,
|
||||||
|
cur_namespace: Optional[List[str]] = None,
|
||||||
|
new_namespace: Optional[List[str]] = None,
|
||||||
|
):
|
||||||
|
"""Rename is not supported for namespace connections."""
|
||||||
|
if cur_namespace is None:
|
||||||
|
cur_namespace = []
|
||||||
|
if new_namespace is None:
|
||||||
|
new_namespace = []
|
||||||
|
raise NotImplementedError(
|
||||||
|
"rename_table is not supported for namespace connections"
|
||||||
|
)
|
||||||
|
|
||||||
|
async def drop_database(self):
|
||||||
|
"""Deprecated method."""
|
||||||
|
raise NotImplementedError(
|
||||||
|
"drop_database is deprecated, use drop_all_tables instead"
|
||||||
|
)
|
||||||
|
|
||||||
|
async def drop_all_tables(self, namespace: Optional[List[str]] = None):
|
||||||
|
"""Drop all tables in the namespace."""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
|
table_names = await self.table_names(namespace=namespace)
|
||||||
|
for table_name in table_names:
|
||||||
|
await self.drop_table(table_name, namespace=namespace)
|
||||||
|
|
||||||
|
async def list_namespaces(
|
||||||
|
self,
|
||||||
|
namespace: Optional[List[str]] = None,
|
||||||
|
page_token: Optional[str] = None,
|
||||||
|
limit: int = 10,
|
||||||
|
) -> Iterable[str]:
|
||||||
|
"""
|
||||||
|
List child namespaces under the given namespace.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
namespace : Optional[List[str]]
|
||||||
|
The parent namespace to list children from.
|
||||||
|
If None, lists root-level namespaces.
|
||||||
|
page_token : Optional[str]
|
||||||
|
Pagination token for listing results.
|
||||||
|
limit : int
|
||||||
|
Maximum number of namespaces to return.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
Iterable[str]
|
||||||
|
Names of child namespaces.
|
||||||
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
|
request = ListNamespacesRequest(
|
||||||
|
id=namespace, page_token=page_token, limit=limit
|
||||||
|
)
|
||||||
|
response = self._ns.list_namespaces(request)
|
||||||
|
return response.namespaces if response.namespaces else []
|
||||||
|
|
||||||
|
async def create_namespace(self, namespace: List[str]) -> None:
|
||||||
|
"""
|
||||||
|
Create a new namespace.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
namespace : List[str]
|
||||||
|
The namespace path to create.
|
||||||
|
"""
|
||||||
|
request = CreateNamespaceRequest(id=namespace)
|
||||||
|
self._ns.create_namespace(request)
|
||||||
|
|
||||||
|
async def drop_namespace(self, namespace: List[str]) -> None:
|
||||||
|
"""
|
||||||
|
Drop a namespace.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
namespace : List[str]
|
||||||
|
The namespace path to drop.
|
||||||
|
"""
|
||||||
|
request = DropNamespaceRequest(id=namespace)
|
||||||
|
self._ns.drop_namespace(request)
|
||||||
|
|
||||||
|
|
||||||
def connect_namespace(
|
def connect_namespace(
|
||||||
impl: str,
|
impl: str,
|
||||||
properties: Dict[str, str],
|
properties: Dict[str, str],
|
||||||
@@ -534,3 +868,62 @@ def connect_namespace(
|
|||||||
storage_options=storage_options,
|
storage_options=storage_options,
|
||||||
session=session,
|
session=session,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def connect_namespace_async(
|
||||||
|
impl: str,
|
||||||
|
properties: Dict[str, str],
|
||||||
|
*,
|
||||||
|
read_consistency_interval: Optional[timedelta] = None,
|
||||||
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
|
session: Optional[Session] = None,
|
||||||
|
) -> AsyncLanceNamespaceDBConnection:
|
||||||
|
"""
|
||||||
|
Connect to a LanceDB database through a namespace (returns async connection).
|
||||||
|
|
||||||
|
This function is synchronous but returns an AsyncLanceNamespaceDBConnection
|
||||||
|
that provides async methods for all database operations.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
impl : str
|
||||||
|
The namespace implementation to use. For examples:
|
||||||
|
- "dir" for DirectoryNamespace
|
||||||
|
- "rest" for REST-based namespace
|
||||||
|
- Full module path for custom implementations
|
||||||
|
properties : Dict[str, str]
|
||||||
|
Configuration properties for the namespace implementation.
|
||||||
|
Different namespace implemenation has different config properties.
|
||||||
|
For example, use DirectoryNamespace with {"root": "/path/to/directory"}
|
||||||
|
read_consistency_interval : Optional[timedelta]
|
||||||
|
The interval at which to check for updates to the table from other
|
||||||
|
processes. If None, then consistency is not checked.
|
||||||
|
storage_options : Optional[Dict[str, str]]
|
||||||
|
Additional options for the storage backend
|
||||||
|
session : Optional[Session]
|
||||||
|
A session to use for this connection
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
AsyncLanceNamespaceDBConnection
|
||||||
|
An async namespace-based connection to LanceDB
|
||||||
|
|
||||||
|
Examples
|
||||||
|
--------
|
||||||
|
>>> import lancedb
|
||||||
|
>>> # This function is sync, but returns an async connection
|
||||||
|
>>> db = lancedb.connect_namespace_async("dir", {"root": "/path/to/db"})
|
||||||
|
>>> # Use async methods on the connection
|
||||||
|
>>> async def use_db():
|
||||||
|
... tables = await db.table_names()
|
||||||
|
... table = await db.create_table("my_table", schema=schema)
|
||||||
|
"""
|
||||||
|
namespace = namespace_connect(impl, properties)
|
||||||
|
|
||||||
|
# Return the async namespace-based connection
|
||||||
|
return AsyncLanceNamespaceDBConnection(
|
||||||
|
namespace,
|
||||||
|
read_consistency_interval=read_consistency_interval,
|
||||||
|
storage_options=storage_options,
|
||||||
|
session=session,
|
||||||
|
)
|
||||||
|
|||||||
@@ -14,6 +14,7 @@ from typing import (
|
|||||||
Literal,
|
Literal,
|
||||||
Optional,
|
Optional,
|
||||||
Tuple,
|
Tuple,
|
||||||
|
Type,
|
||||||
TypeVar,
|
TypeVar,
|
||||||
Union,
|
Union,
|
||||||
Any,
|
Any,
|
||||||
@@ -786,10 +787,7 @@ class LanceQueryBuilder(ABC):
|
|||||||
-------
|
-------
|
||||||
List[LanceModel]
|
List[LanceModel]
|
||||||
"""
|
"""
|
||||||
return [
|
return [model(**row) for row in self.to_arrow(timeout=timeout).to_pylist()]
|
||||||
model(**{k: v for k, v in row.items() if k in model.field_names()})
|
|
||||||
for row in self.to_arrow(timeout=timeout).to_pylist()
|
|
||||||
]
|
|
||||||
|
|
||||||
def to_polars(self, *, timeout: Optional[timedelta] = None) -> "pl.DataFrame":
|
def to_polars(self, *, timeout: Optional[timedelta] = None) -> "pl.DataFrame":
|
||||||
"""
|
"""
|
||||||
@@ -885,7 +883,7 @@ class LanceQueryBuilder(ABC):
|
|||||||
----------
|
----------
|
||||||
where: str
|
where: str
|
||||||
The where clause which is a valid SQL where clause. See
|
The where clause which is a valid SQL where clause. See
|
||||||
`Lance filter pushdown <https://lancedb.github.io/lance/read_and_write.html#filter-push-down>`_
|
`Lance filter pushdown <https://lance.org/guide/read_and_write#filter-push-down>`_
|
||||||
for valid SQL expressions.
|
for valid SQL expressions.
|
||||||
prefilter: bool, default True
|
prefilter: bool, default True
|
||||||
If True, apply the filter before vector search, otherwise the
|
If True, apply the filter before vector search, otherwise the
|
||||||
@@ -1358,7 +1356,7 @@ class LanceVectorQueryBuilder(LanceQueryBuilder):
|
|||||||
----------
|
----------
|
||||||
where: str
|
where: str
|
||||||
The where clause which is a valid SQL where clause. See
|
The where clause which is a valid SQL where clause. See
|
||||||
`Lance filter pushdown <https://lancedb.github.io/lance/read_and_write.html#filter-push-down>`_
|
`Lance filter pushdown <https://lance.org/guide/read_and_write#filter-push-down>`_
|
||||||
for valid SQL expressions.
|
for valid SQL expressions.
|
||||||
prefilter: bool, default True
|
prefilter: bool, default True
|
||||||
If True, apply the filter before vector search, otherwise the
|
If True, apply the filter before vector search, otherwise the
|
||||||
@@ -1497,7 +1495,7 @@ class LanceFtsQueryBuilder(LanceQueryBuilder):
|
|||||||
if self._phrase_query:
|
if self._phrase_query:
|
||||||
if isinstance(query, str):
|
if isinstance(query, str):
|
||||||
if not query.startswith('"') or not query.endswith('"'):
|
if not query.startswith('"') or not query.endswith('"'):
|
||||||
query = f'"{query}"'
|
self._query = f'"{query}"'
|
||||||
elif isinstance(query, FullTextQuery) and not isinstance(
|
elif isinstance(query, FullTextQuery) and not isinstance(
|
||||||
query, PhraseQuery
|
query, PhraseQuery
|
||||||
):
|
):
|
||||||
@@ -2400,6 +2398,28 @@ class AsyncQueryBase(object):
|
|||||||
|
|
||||||
return pl.from_arrow(await self.to_arrow(timeout=timeout))
|
return pl.from_arrow(await self.to_arrow(timeout=timeout))
|
||||||
|
|
||||||
|
async def to_pydantic(
|
||||||
|
self, model: Type[LanceModel], *, timeout: Optional[timedelta] = None
|
||||||
|
) -> List[LanceModel]:
|
||||||
|
"""
|
||||||
|
Convert results to a list of pydantic models.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
model : Type[LanceModel]
|
||||||
|
The pydantic model to use.
|
||||||
|
timeout : timedelta, optional
|
||||||
|
The maximum time to wait for the query to complete.
|
||||||
|
If None, wait indefinitely.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
list[LanceModel]
|
||||||
|
"""
|
||||||
|
return [
|
||||||
|
model(**row) for row in (await self.to_arrow(timeout=timeout)).to_pylist()
|
||||||
|
]
|
||||||
|
|
||||||
async def explain_plan(self, verbose: Optional[bool] = False):
|
async def explain_plan(self, verbose: Optional[bool] = False):
|
||||||
"""Return the execution plan for this query.
|
"""Return the execution plan for this query.
|
||||||
|
|
||||||
@@ -2409,9 +2429,8 @@ class AsyncQueryBase(object):
|
|||||||
>>> from lancedb import connect_async
|
>>> from lancedb import connect_async
|
||||||
>>> async def doctest_example():
|
>>> async def doctest_example():
|
||||||
... conn = await connect_async("./.lancedb")
|
... conn = await connect_async("./.lancedb")
|
||||||
... table = await conn.create_table("my_table", [{"vector": [99, 99]}])
|
... table = await conn.create_table("my_table", [{"vector": [99.0, 99.0]}])
|
||||||
... query = [100, 100]
|
... plan = await table.query().nearest_to([1.0, 2.0]).explain_plan(True)
|
||||||
... plan = await table.query().nearest_to([1, 2]).explain_plan(True)
|
|
||||||
... print(plan)
|
... print(plan)
|
||||||
>>> asyncio.run(doctest_example()) # doctest: +ELLIPSIS, +NORMALIZE_WHITESPACE
|
>>> asyncio.run(doctest_example()) # doctest: +ELLIPSIS, +NORMALIZE_WHITESPACE
|
||||||
ProjectionExec: expr=[vector@0 as vector, _distance@2 as _distance]
|
ProjectionExec: expr=[vector@0 as vector, _distance@2 as _distance]
|
||||||
@@ -2420,6 +2439,7 @@ class AsyncQueryBase(object):
|
|||||||
SortExec: TopK(fetch=10), expr=[_distance@2 ASC NULLS LAST, _rowid@1 ASC NULLS LAST], preserve_partitioning=[false]
|
SortExec: TopK(fetch=10), expr=[_distance@2 ASC NULLS LAST, _rowid@1 ASC NULLS LAST], preserve_partitioning=[false]
|
||||||
KNNVectorDistance: metric=l2
|
KNNVectorDistance: metric=l2
|
||||||
LanceRead: uri=..., projection=[vector], ...
|
LanceRead: uri=..., projection=[vector], ...
|
||||||
|
<BLANKLINE>
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
@@ -3121,10 +3141,9 @@ class AsyncHybridQuery(AsyncStandardQuery, AsyncVectorQueryBase):
|
|||||||
>>> from lancedb.index import FTS
|
>>> from lancedb.index import FTS
|
||||||
>>> async def doctest_example():
|
>>> async def doctest_example():
|
||||||
... conn = await connect_async("./.lancedb")
|
... conn = await connect_async("./.lancedb")
|
||||||
... table = await conn.create_table("my_table", [{"vector": [99, 99], "text": "hello world"}])
|
... table = await conn.create_table("my_table", [{"vector": [99.0, 99.0], "text": "hello world"}])
|
||||||
... await table.create_index("text", config=FTS(with_position=False))
|
... await table.create_index("text", config=FTS(with_position=False))
|
||||||
... query = [100, 100]
|
... plan = await table.query().nearest_to([1.0, 2.0]).nearest_to_text("hello").explain_plan(True)
|
||||||
... plan = await table.query().nearest_to([1, 2]).nearest_to_text("hello").explain_plan(True)
|
|
||||||
... print(plan)
|
... print(plan)
|
||||||
>>> asyncio.run(doctest_example()) # doctest: +ELLIPSIS, +NORMALIZE_WHITESPACE
|
>>> asyncio.run(doctest_example()) # doctest: +ELLIPSIS, +NORMALIZE_WHITESPACE
|
||||||
Vector Search Plan:
|
Vector Search Plan:
|
||||||
@@ -3398,9 +3417,8 @@ class BaseQueryBuilder(object):
|
|||||||
>>> from lancedb import connect_async
|
>>> from lancedb import connect_async
|
||||||
>>> async def doctest_example():
|
>>> async def doctest_example():
|
||||||
... conn = await connect_async("./.lancedb")
|
... conn = await connect_async("./.lancedb")
|
||||||
... table = await conn.create_table("my_table", [{"vector": [99, 99]}])
|
... table = await conn.create_table("my_table", [{"vector": [99.0, 99.0]}])
|
||||||
... query = [100, 100]
|
... plan = await table.query().nearest_to([1.0, 2.0]).explain_plan(True)
|
||||||
... plan = await table.query().nearest_to([1, 2]).explain_plan(True)
|
|
||||||
... print(plan)
|
... print(plan)
|
||||||
>>> asyncio.run(doctest_example()) # doctest: +ELLIPSIS, +NORMALIZE_WHITESPACE
|
>>> asyncio.run(doctest_example()) # doctest: +ELLIPSIS, +NORMALIZE_WHITESPACE
|
||||||
ProjectionExec: expr=[vector@0 as vector, _distance@2 as _distance]
|
ProjectionExec: expr=[vector@0 as vector, _distance@2 as _distance]
|
||||||
@@ -3409,6 +3427,7 @@ class BaseQueryBuilder(object):
|
|||||||
SortExec: TopK(fetch=10), expr=[_distance@2 ASC NULLS LAST, _rowid@1 ASC NULLS LAST], preserve_partitioning=[false]
|
SortExec: TopK(fetch=10), expr=[_distance@2 ASC NULLS LAST, _rowid@1 ASC NULLS LAST], preserve_partitioning=[false]
|
||||||
KNNVectorDistance: metric=l2
|
KNNVectorDistance: metric=l2
|
||||||
LanceRead: uri=..., projection=[vector], ...
|
LanceRead: uri=..., projection=[vector], ...
|
||||||
|
<BLANKLINE>
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
|
|||||||
@@ -104,7 +104,7 @@ class RemoteDBConnection(DBConnection):
|
|||||||
@override
|
@override
|
||||||
def list_namespaces(
|
def list_namespaces(
|
||||||
self,
|
self,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
page_token: Optional[str] = None,
|
page_token: Optional[str] = None,
|
||||||
limit: int = 10,
|
limit: int = 10,
|
||||||
) -> Iterable[str]:
|
) -> Iterable[str]:
|
||||||
@@ -125,6 +125,8 @@ class RemoteDBConnection(DBConnection):
|
|||||||
Iterable of str
|
Iterable of str
|
||||||
List of immediate child namespace names
|
List of immediate child namespace names
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
return LOOP.run(
|
return LOOP.run(
|
||||||
self._conn.list_namespaces(
|
self._conn.list_namespaces(
|
||||||
namespace=namespace, page_token=page_token, limit=limit
|
namespace=namespace, page_token=page_token, limit=limit
|
||||||
@@ -159,7 +161,7 @@ class RemoteDBConnection(DBConnection):
|
|||||||
page_token: Optional[str] = None,
|
page_token: Optional[str] = None,
|
||||||
limit: int = 10,
|
limit: int = 10,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
) -> Iterable[str]:
|
) -> Iterable[str]:
|
||||||
"""List the names of all tables in the database.
|
"""List the names of all tables in the database.
|
||||||
|
|
||||||
@@ -177,6 +179,8 @@ class RemoteDBConnection(DBConnection):
|
|||||||
-------
|
-------
|
||||||
An iterator of table names.
|
An iterator of table names.
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
return LOOP.run(
|
return LOOP.run(
|
||||||
self._conn.table_names(
|
self._conn.table_names(
|
||||||
namespace=namespace, start_after=page_token, limit=limit
|
namespace=namespace, start_after=page_token, limit=limit
|
||||||
@@ -188,7 +192,7 @@ class RemoteDBConnection(DBConnection):
|
|||||||
self,
|
self,
|
||||||
name: str,
|
name: str,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
index_cache_size: Optional[int] = None,
|
index_cache_size: Optional[int] = None,
|
||||||
) -> Table:
|
) -> Table:
|
||||||
@@ -208,6 +212,8 @@ class RemoteDBConnection(DBConnection):
|
|||||||
"""
|
"""
|
||||||
from .table import RemoteTable
|
from .table import RemoteTable
|
||||||
|
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
if index_cache_size is not None:
|
if index_cache_size is not None:
|
||||||
logging.info(
|
logging.info(
|
||||||
"index_cache_size is ignored in LanceDb Cloud"
|
"index_cache_size is ignored in LanceDb Cloud"
|
||||||
@@ -222,7 +228,7 @@ class RemoteDBConnection(DBConnection):
|
|||||||
target_table_name: str,
|
target_table_name: str,
|
||||||
source_uri: str,
|
source_uri: str,
|
||||||
*,
|
*,
|
||||||
target_namespace: List[str] = [],
|
target_namespace: Optional[List[str]] = None,
|
||||||
source_version: Optional[int] = None,
|
source_version: Optional[int] = None,
|
||||||
source_tag: Optional[str] = None,
|
source_tag: Optional[str] = None,
|
||||||
is_shallow: bool = True,
|
is_shallow: bool = True,
|
||||||
@@ -252,6 +258,8 @@ class RemoteDBConnection(DBConnection):
|
|||||||
"""
|
"""
|
||||||
from .table import RemoteTable
|
from .table import RemoteTable
|
||||||
|
|
||||||
|
if target_namespace is None:
|
||||||
|
target_namespace = []
|
||||||
table = LOOP.run(
|
table = LOOP.run(
|
||||||
self._conn.clone_table(
|
self._conn.clone_table(
|
||||||
target_table_name,
|
target_table_name,
|
||||||
@@ -275,7 +283,7 @@ class RemoteDBConnection(DBConnection):
|
|||||||
mode: Optional[str] = None,
|
mode: Optional[str] = None,
|
||||||
embedding_functions: Optional[List[EmbeddingFunctionConfig]] = None,
|
embedding_functions: Optional[List[EmbeddingFunctionConfig]] = None,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
) -> Table:
|
) -> Table:
|
||||||
"""Create a [Table][lancedb.table.Table] in the database.
|
"""Create a [Table][lancedb.table.Table] in the database.
|
||||||
|
|
||||||
@@ -372,6 +380,8 @@ class RemoteDBConnection(DBConnection):
|
|||||||
LanceTable(table4)
|
LanceTable(table4)
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
validate_table_name(name)
|
validate_table_name(name)
|
||||||
if embedding_functions is not None:
|
if embedding_functions is not None:
|
||||||
logging.warning(
|
logging.warning(
|
||||||
@@ -396,7 +406,7 @@ class RemoteDBConnection(DBConnection):
|
|||||||
return RemoteTable(table, self.db_name)
|
return RemoteTable(table, self.db_name)
|
||||||
|
|
||||||
@override
|
@override
|
||||||
def drop_table(self, name: str, namespace: List[str] = []):
|
def drop_table(self, name: str, namespace: Optional[List[str]] = None):
|
||||||
"""Drop a table from the database.
|
"""Drop a table from the database.
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
@@ -407,6 +417,8 @@ class RemoteDBConnection(DBConnection):
|
|||||||
The namespace to drop the table from.
|
The namespace to drop the table from.
|
||||||
None or empty list represents root namespace.
|
None or empty list represents root namespace.
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
LOOP.run(self._conn.drop_table(name, namespace=namespace))
|
LOOP.run(self._conn.drop_table(name, namespace=namespace))
|
||||||
|
|
||||||
@override
|
@override
|
||||||
@@ -414,8 +426,8 @@ class RemoteDBConnection(DBConnection):
|
|||||||
self,
|
self,
|
||||||
cur_name: str,
|
cur_name: str,
|
||||||
new_name: str,
|
new_name: str,
|
||||||
cur_namespace: List[str] = [],
|
cur_namespace: Optional[List[str]] = None,
|
||||||
new_namespace: List[str] = [],
|
new_namespace: Optional[List[str]] = None,
|
||||||
):
|
):
|
||||||
"""Rename a table in the database.
|
"""Rename a table in the database.
|
||||||
|
|
||||||
@@ -426,6 +438,10 @@ class RemoteDBConnection(DBConnection):
|
|||||||
new_name: str
|
new_name: str
|
||||||
The new name of the table.
|
The new name of the table.
|
||||||
"""
|
"""
|
||||||
|
if cur_namespace is None:
|
||||||
|
cur_namespace = []
|
||||||
|
if new_namespace is None:
|
||||||
|
new_namespace = []
|
||||||
LOOP.run(
|
LOOP.run(
|
||||||
self._conn.rename_table(
|
self._conn.rename_table(
|
||||||
cur_name,
|
cur_name,
|
||||||
|
|||||||
@@ -652,6 +652,17 @@ class RemoteTable(Table):
|
|||||||
"migrate_v2_manifest_paths() is not supported on the LanceDB Cloud"
|
"migrate_v2_manifest_paths() is not supported on the LanceDB Cloud"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def head(self, n=5) -> pa.Table:
|
||||||
|
"""
|
||||||
|
Return the first `n` rows of the table.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
n: int, default 5
|
||||||
|
The number of rows to return.
|
||||||
|
"""
|
||||||
|
return LOOP.run(self._table.query().limit(n).to_arrow())
|
||||||
|
|
||||||
|
|
||||||
def add_index(tbl: pa.Table, i: int) -> pa.Table:
|
def add_index(tbl: pa.Table, i: int) -> pa.Table:
|
||||||
return tbl.add_column(
|
return tbl.add_column(
|
||||||
|
|||||||
@@ -178,7 +178,7 @@ def _into_pyarrow_reader(
|
|||||||
f"Unknown data type {type(data)}. "
|
f"Unknown data type {type(data)}. "
|
||||||
"Supported types: list of dicts, pandas DataFrame, polars DataFrame, "
|
"Supported types: list of dicts, pandas DataFrame, polars DataFrame, "
|
||||||
"pyarrow Table/RecordBatch, or Pydantic models. "
|
"pyarrow Table/RecordBatch, or Pydantic models. "
|
||||||
"See https://lancedb.github.io/lancedb/guides/tables/ for examples."
|
"See https://lancedb.com/docs/tables/ for examples."
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -1018,7 +1018,7 @@ class Table(ABC):
|
|||||||
... .when_not_matched_insert_all() \\
|
... .when_not_matched_insert_all() \\
|
||||||
... .execute(new_data)
|
... .execute(new_data)
|
||||||
>>> res
|
>>> res
|
||||||
MergeResult(version=2, num_updated_rows=2, num_inserted_rows=1, num_deleted_rows=0)
|
MergeResult(version=2, num_updated_rows=2, num_inserted_rows=1, num_deleted_rows=0, num_attempts=1)
|
||||||
>>> # The order of new rows is non-deterministic since we use
|
>>> # The order of new rows is non-deterministic since we use
|
||||||
>>> # a hash-join as part of this operation and so we sort here
|
>>> # a hash-join as part of this operation and so we sort here
|
||||||
>>> table.to_arrow().sort_by("a").to_pandas()
|
>>> table.to_arrow().sort_by("a").to_pandas()
|
||||||
@@ -1708,15 +1708,18 @@ class LanceTable(Table):
|
|||||||
connection: "LanceDBConnection",
|
connection: "LanceDBConnection",
|
||||||
name: str,
|
name: str,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
||||||
index_cache_size: Optional[int] = None,
|
index_cache_size: Optional[int] = None,
|
||||||
location: Optional[str] = None,
|
location: Optional[str] = None,
|
||||||
_async: AsyncTable = None,
|
_async: AsyncTable = None,
|
||||||
):
|
):
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
self._conn = connection
|
self._conn = connection
|
||||||
self._namespace = namespace
|
self._namespace = namespace
|
||||||
|
self._location = location # Store location for use in _dataset_path
|
||||||
if _async is not None:
|
if _async is not None:
|
||||||
self._table = _async
|
self._table = _async
|
||||||
else:
|
else:
|
||||||
@@ -1765,12 +1768,14 @@ class LanceTable(Table):
|
|||||||
db,
|
db,
|
||||||
name,
|
name,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str]] = None,
|
storage_options: Optional[Dict[str, str]] = None,
|
||||||
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
||||||
index_cache_size: Optional[int] = None,
|
index_cache_size: Optional[int] = None,
|
||||||
location: Optional[str] = None,
|
location: Optional[str] = None,
|
||||||
):
|
):
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
tbl = cls(
|
tbl = cls(
|
||||||
db,
|
db,
|
||||||
name,
|
name,
|
||||||
@@ -1794,6 +1799,10 @@ class LanceTable(Table):
|
|||||||
@cached_property
|
@cached_property
|
||||||
def _dataset_path(self) -> str:
|
def _dataset_path(self) -> str:
|
||||||
# Cacheable since it's deterministic
|
# Cacheable since it's deterministic
|
||||||
|
# If table was opened with explicit location (e.g., from namespace),
|
||||||
|
# use that location directly instead of constructing from base URI
|
||||||
|
if self._location is not None:
|
||||||
|
return self._location
|
||||||
return _table_path(self._conn.uri, self.name)
|
return _table_path(self._conn.uri, self.name)
|
||||||
|
|
||||||
def to_lance(self, **kwargs) -> lance.LanceDataset:
|
def to_lance(self, **kwargs) -> lance.LanceDataset:
|
||||||
@@ -2618,7 +2627,7 @@ class LanceTable(Table):
|
|||||||
fill_value: float = 0.0,
|
fill_value: float = 0.0,
|
||||||
embedding_functions: Optional[List[EmbeddingFunctionConfig]] = None,
|
embedding_functions: Optional[List[EmbeddingFunctionConfig]] = None,
|
||||||
*,
|
*,
|
||||||
namespace: List[str] = [],
|
namespace: Optional[List[str]] = None,
|
||||||
storage_options: Optional[Dict[str, str | bool]] = None,
|
storage_options: Optional[Dict[str, str | bool]] = None,
|
||||||
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
storage_options_provider: Optional["StorageOptionsProvider"] = None,
|
||||||
data_storage_version: Optional[str] = None,
|
data_storage_version: Optional[str] = None,
|
||||||
@@ -2678,9 +2687,12 @@ class LanceTable(Table):
|
|||||||
Deprecated. Set `storage_options` when connecting to the database and set
|
Deprecated. Set `storage_options` when connecting to the database and set
|
||||||
`new_table_enable_v2_manifest_paths` in the options.
|
`new_table_enable_v2_manifest_paths` in the options.
|
||||||
"""
|
"""
|
||||||
|
if namespace is None:
|
||||||
|
namespace = []
|
||||||
self = cls.__new__(cls)
|
self = cls.__new__(cls)
|
||||||
self._conn = db
|
self._conn = db
|
||||||
self._namespace = namespace
|
self._namespace = namespace
|
||||||
|
self._location = location
|
||||||
|
|
||||||
if data_storage_version is not None:
|
if data_storage_version is not None:
|
||||||
warnings.warn(
|
warnings.warn(
|
||||||
@@ -3622,7 +3634,7 @@ class AsyncTable:
|
|||||||
... .when_not_matched_insert_all() \\
|
... .when_not_matched_insert_all() \\
|
||||||
... .execute(new_data)
|
... .execute(new_data)
|
||||||
>>> res
|
>>> res
|
||||||
MergeResult(version=2, num_updated_rows=2, num_inserted_rows=1, num_deleted_rows=0)
|
MergeResult(version=2, num_updated_rows=2, num_inserted_rows=1, num_deleted_rows=0, num_attempts=1)
|
||||||
>>> # The order of new rows is non-deterministic since we use
|
>>> # The order of new rows is non-deterministic since we use
|
||||||
>>> # a hash-join as part of this operation and so we sort here
|
>>> # a hash-join as part of this operation and so we sort here
|
||||||
>>> table.to_arrow().sort_by("a").to_pandas()
|
>>> table.to_arrow().sort_by("a").to_pandas()
|
||||||
|
|||||||
@@ -441,6 +441,150 @@ async def test_create_table_v2_manifest_paths_async(tmp_path):
|
|||||||
assert re.match(r"\d{20}\.manifest", manifest)
|
assert re.match(r"\d{20}\.manifest", manifest)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_create_table_stable_row_ids_via_storage_options(tmp_path):
|
||||||
|
"""Test stable_row_ids via storage_options at connect time."""
|
||||||
|
import lance
|
||||||
|
|
||||||
|
# Connect with stable row IDs enabled as default for new tables
|
||||||
|
db_with = await lancedb.connect_async(
|
||||||
|
tmp_path, storage_options={"new_table_enable_stable_row_ids": "true"}
|
||||||
|
)
|
||||||
|
# Connect without stable row IDs (default)
|
||||||
|
db_without = await lancedb.connect_async(
|
||||||
|
tmp_path, storage_options={"new_table_enable_stable_row_ids": "false"}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create table using connection with stable row IDs enabled
|
||||||
|
await db_with.create_table(
|
||||||
|
"with_stable_via_opts",
|
||||||
|
data=[{"id": i} for i in range(10)],
|
||||||
|
)
|
||||||
|
lance_ds_with = lance.dataset(tmp_path / "with_stable_via_opts.lance")
|
||||||
|
fragments_with = lance_ds_with.get_fragments()
|
||||||
|
assert len(fragments_with) > 0
|
||||||
|
assert fragments_with[0].metadata.row_id_meta is not None
|
||||||
|
|
||||||
|
# Create table using connection without stable row IDs
|
||||||
|
await db_without.create_table(
|
||||||
|
"without_stable_via_opts",
|
||||||
|
data=[{"id": i} for i in range(10)],
|
||||||
|
)
|
||||||
|
lance_ds_without = lance.dataset(tmp_path / "without_stable_via_opts.lance")
|
||||||
|
fragments_without = lance_ds_without.get_fragments()
|
||||||
|
assert len(fragments_without) > 0
|
||||||
|
assert fragments_without[0].metadata.row_id_meta is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_create_table_stable_row_ids_via_storage_options_sync(tmp_path):
|
||||||
|
"""Test that enable_stable_row_ids can be set via storage_options (sync API)."""
|
||||||
|
# Connect with stable row IDs enabled as default for new tables
|
||||||
|
db_with = lancedb.connect(
|
||||||
|
tmp_path, storage_options={"new_table_enable_stable_row_ids": "true"}
|
||||||
|
)
|
||||||
|
# Connect without stable row IDs (default)
|
||||||
|
db_without = lancedb.connect(
|
||||||
|
tmp_path, storage_options={"new_table_enable_stable_row_ids": "false"}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create table using connection with stable row IDs enabled
|
||||||
|
tbl_with = db_with.create_table(
|
||||||
|
"with_stable_sync",
|
||||||
|
data=[{"id": i} for i in range(10)],
|
||||||
|
)
|
||||||
|
lance_ds_with = tbl_with.to_lance()
|
||||||
|
fragments_with = lance_ds_with.get_fragments()
|
||||||
|
assert len(fragments_with) > 0
|
||||||
|
assert fragments_with[0].metadata.row_id_meta is not None
|
||||||
|
|
||||||
|
# Create table using connection without stable row IDs
|
||||||
|
tbl_without = db_without.create_table(
|
||||||
|
"without_stable_sync",
|
||||||
|
data=[{"id": i} for i in range(10)],
|
||||||
|
)
|
||||||
|
lance_ds_without = tbl_without.to_lance()
|
||||||
|
fragments_without = lance_ds_without.get_fragments()
|
||||||
|
assert len(fragments_without) > 0
|
||||||
|
assert fragments_without[0].metadata.row_id_meta is None
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_create_table_stable_row_ids_table_level_override(tmp_path):
|
||||||
|
"""Test that stable_row_ids can be enabled/disabled at create_table level."""
|
||||||
|
import lance
|
||||||
|
|
||||||
|
# Connect without any stable row ID setting
|
||||||
|
db_default = await lancedb.connect_async(tmp_path)
|
||||||
|
|
||||||
|
# Connect with stable row IDs enabled at connection level
|
||||||
|
db_with_stable = await lancedb.connect_async(
|
||||||
|
tmp_path, storage_options={"new_table_enable_stable_row_ids": "true"}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Case 1: No connection setting, enable at table level
|
||||||
|
await db_default.create_table(
|
||||||
|
"table_level_enabled",
|
||||||
|
data=[{"id": i} for i in range(10)],
|
||||||
|
storage_options={"new_table_enable_stable_row_ids": "true"},
|
||||||
|
)
|
||||||
|
lance_ds = lance.dataset(tmp_path / "table_level_enabled.lance")
|
||||||
|
fragments = lance_ds.get_fragments()
|
||||||
|
assert len(fragments) > 0
|
||||||
|
assert fragments[0].metadata.row_id_meta is not None, (
|
||||||
|
"Table should have stable row IDs when enabled at table level"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Case 2: Connection has stable row IDs, override with false at table level
|
||||||
|
await db_with_stable.create_table(
|
||||||
|
"table_level_disabled",
|
||||||
|
data=[{"id": i} for i in range(10)],
|
||||||
|
storage_options={"new_table_enable_stable_row_ids": "false"},
|
||||||
|
)
|
||||||
|
lance_ds = lance.dataset(tmp_path / "table_level_disabled.lance")
|
||||||
|
fragments = lance_ds.get_fragments()
|
||||||
|
assert len(fragments) > 0
|
||||||
|
assert fragments[0].metadata.row_id_meta is None, (
|
||||||
|
"Table should NOT have stable row IDs when disabled at table level"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_create_table_stable_row_ids_table_level_override_sync(tmp_path):
|
||||||
|
"""Test that stable_row_ids can be enabled/disabled at create_table level (sync)."""
|
||||||
|
# Connect without any stable row ID setting
|
||||||
|
db_default = lancedb.connect(tmp_path)
|
||||||
|
|
||||||
|
# Connect with stable row IDs enabled at connection level
|
||||||
|
db_with_stable = lancedb.connect(
|
||||||
|
tmp_path, storage_options={"new_table_enable_stable_row_ids": "true"}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Case 1: No connection setting, enable at table level
|
||||||
|
tbl = db_default.create_table(
|
||||||
|
"table_level_enabled_sync",
|
||||||
|
data=[{"id": i} for i in range(10)],
|
||||||
|
storage_options={"new_table_enable_stable_row_ids": "true"},
|
||||||
|
)
|
||||||
|
lance_ds = tbl.to_lance()
|
||||||
|
fragments = lance_ds.get_fragments()
|
||||||
|
assert len(fragments) > 0
|
||||||
|
assert fragments[0].metadata.row_id_meta is not None, (
|
||||||
|
"Table should have stable row IDs when enabled at table level"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Case 2: Connection has stable row IDs, override with false at table level
|
||||||
|
tbl = db_with_stable.create_table(
|
||||||
|
"table_level_disabled_sync",
|
||||||
|
data=[{"id": i} for i in range(10)],
|
||||||
|
storage_options={"new_table_enable_stable_row_ids": "false"},
|
||||||
|
)
|
||||||
|
lance_ds = tbl.to_lance()
|
||||||
|
fragments = lance_ds.get_fragments()
|
||||||
|
assert len(fragments) > 0
|
||||||
|
assert fragments[0].metadata.row_id_meta is None, (
|
||||||
|
"Table should NOT have stable row IDs when disabled at table level"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def test_open_table_sync(tmp_db: lancedb.DBConnection):
|
def test_open_table_sync(tmp_db: lancedb.DBConnection):
|
||||||
tmp_db.create_table("test", data=[{"id": 0}])
|
tmp_db.create_table("test", data=[{"id": 0}])
|
||||||
assert tmp_db.open_table("test").count_rows() == 1
|
assert tmp_db.open_table("test").count_rows() == 1
|
||||||
|
|||||||
@@ -32,6 +32,7 @@ import numpy as np
|
|||||||
import pyarrow as pa
|
import pyarrow as pa
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
import pytest
|
import pytest
|
||||||
|
import pytest_asyncio
|
||||||
from utils import exception_output
|
from utils import exception_output
|
||||||
|
|
||||||
pytest.importorskip("lancedb.fts")
|
pytest.importorskip("lancedb.fts")
|
||||||
@@ -90,7 +91,7 @@ def table(tmp_path) -> ldb.table.LanceTable:
|
|||||||
return table
|
return table
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture
|
@pytest_asyncio.fixture
|
||||||
async def async_table(tmp_path) -> ldb.table.AsyncTable:
|
async def async_table(tmp_path) -> ldb.table.AsyncTable:
|
||||||
# Use local random state to avoid affecting other tests
|
# Use local random state to avoid affecting other tests
|
||||||
rng = np.random.RandomState(42)
|
rng = np.random.RandomState(42)
|
||||||
@@ -253,7 +254,7 @@ def test_search_fts(table, use_tantivy):
|
|||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_fts_select_async(async_table):
|
async def test_fts_select_async(async_table):
|
||||||
tbl = await async_table
|
tbl = async_table
|
||||||
await tbl.create_index("text", config=FTS())
|
await tbl.create_index("text", config=FTS())
|
||||||
await tbl.create_index("text2", config=FTS())
|
await tbl.create_index("text2", config=FTS())
|
||||||
results = (
|
results = (
|
||||||
@@ -324,11 +325,18 @@ def test_search_fts_phrase_query(table):
|
|||||||
pass
|
pass
|
||||||
table.create_fts_index("text", use_tantivy=False, with_position=True, replace=True)
|
table.create_fts_index("text", use_tantivy=False, with_position=True, replace=True)
|
||||||
results = table.search("puppy").limit(100).to_list()
|
results = table.search("puppy").limit(100).to_list()
|
||||||
|
|
||||||
|
# Test with quotation marks
|
||||||
phrase_results = table.search('"puppy runs"').limit(100).to_list()
|
phrase_results = table.search('"puppy runs"').limit(100).to_list()
|
||||||
assert len(results) > len(phrase_results)
|
assert len(results) > len(phrase_results)
|
||||||
assert len(phrase_results) > 0
|
assert len(phrase_results) > 0
|
||||||
|
|
||||||
# Test with a query
|
# Test with .phrase_query()
|
||||||
|
phrase_results = table.search("puppy runs").phrase_query().limit(100).to_list()
|
||||||
|
assert len(results) > len(phrase_results)
|
||||||
|
assert len(phrase_results) > 0
|
||||||
|
|
||||||
|
# Test with PhraseQuery()
|
||||||
phrase_results = (
|
phrase_results = (
|
||||||
table.search(PhraseQuery("puppy runs", "text")).limit(100).to_list()
|
table.search(PhraseQuery("puppy runs", "text")).limit(100).to_list()
|
||||||
)
|
)
|
||||||
@@ -338,7 +346,6 @@ def test_search_fts_phrase_query(table):
|
|||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_search_fts_phrase_query_async(async_table):
|
async def test_search_fts_phrase_query_async(async_table):
|
||||||
async_table = await async_table
|
|
||||||
await async_table.create_index("text", config=FTS(with_position=False))
|
await async_table.create_index("text", config=FTS(with_position=False))
|
||||||
try:
|
try:
|
||||||
phrase_results = (
|
phrase_results = (
|
||||||
@@ -393,7 +400,6 @@ def test_search_fts_specify_column(table):
|
|||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_search_fts_async(async_table):
|
async def test_search_fts_async(async_table):
|
||||||
async_table = await async_table
|
|
||||||
await async_table.create_index("text", config=FTS())
|
await async_table.create_index("text", config=FTS())
|
||||||
results = await async_table.query().nearest_to_text("puppy").limit(5).to_list()
|
results = await async_table.query().nearest_to_text("puppy").limit(5).to_list()
|
||||||
assert len(results) == 5
|
assert len(results) == 5
|
||||||
@@ -424,7 +430,6 @@ async def test_search_fts_async(async_table):
|
|||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_search_fts_specify_column_async(async_table):
|
async def test_search_fts_specify_column_async(async_table):
|
||||||
async_table = await async_table
|
|
||||||
await async_table.create_index("text", config=FTS())
|
await async_table.create_index("text", config=FTS())
|
||||||
await async_table.create_index("text2", config=FTS())
|
await async_table.create_index("text2", config=FTS())
|
||||||
|
|
||||||
|
|||||||
@@ -423,3 +423,218 @@ class TestNamespaceConnection:
|
|||||||
db.drop_table("same_name_table", namespace=["namespace_b"])
|
db.drop_table("same_name_table", namespace=["namespace_b"])
|
||||||
db.drop_namespace(["namespace_a"])
|
db.drop_namespace(["namespace_a"])
|
||||||
db.drop_namespace(["namespace_b"])
|
db.drop_namespace(["namespace_b"])
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
class TestAsyncNamespaceConnection:
|
||||||
|
"""Test async namespace-based LanceDB connection using DirectoryNamespace."""
|
||||||
|
|
||||||
|
def setup_method(self):
|
||||||
|
"""Set up test fixtures."""
|
||||||
|
self.temp_dir = tempfile.mkdtemp()
|
||||||
|
|
||||||
|
def teardown_method(self):
|
||||||
|
"""Clean up test fixtures."""
|
||||||
|
shutil.rmtree(self.temp_dir, ignore_errors=True)
|
||||||
|
|
||||||
|
async def test_connect_namespace_async(self):
|
||||||
|
"""Test connecting to LanceDB through DirectoryNamespace asynchronously."""
|
||||||
|
db = lancedb.connect_namespace_async("dir", {"root": self.temp_dir})
|
||||||
|
|
||||||
|
# Should be an AsyncLanceNamespaceDBConnection
|
||||||
|
assert isinstance(db, lancedb.AsyncLanceNamespaceDBConnection)
|
||||||
|
|
||||||
|
# Initially no tables in root
|
||||||
|
table_names = await db.table_names()
|
||||||
|
assert len(list(table_names)) == 0
|
||||||
|
|
||||||
|
async def test_create_table_async(self):
|
||||||
|
"""Test creating a table asynchronously through namespace."""
|
||||||
|
db = lancedb.connect_namespace_async("dir", {"root": self.temp_dir})
|
||||||
|
|
||||||
|
# Create a child namespace first
|
||||||
|
await db.create_namespace(["test_ns"])
|
||||||
|
|
||||||
|
# Define schema for empty table
|
||||||
|
schema = pa.schema(
|
||||||
|
[
|
||||||
|
pa.field("id", pa.int64()),
|
||||||
|
pa.field("vector", pa.list_(pa.float32(), 2)),
|
||||||
|
pa.field("text", pa.string()),
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create empty table in child namespace
|
||||||
|
table = await db.create_table(
|
||||||
|
"test_table", schema=schema, namespace=["test_ns"]
|
||||||
|
)
|
||||||
|
assert table is not None
|
||||||
|
assert isinstance(table, lancedb.AsyncTable)
|
||||||
|
|
||||||
|
# Table should appear in child namespace
|
||||||
|
table_names = await db.table_names(namespace=["test_ns"])
|
||||||
|
assert "test_table" in list(table_names)
|
||||||
|
|
||||||
|
async def test_open_table_async(self):
|
||||||
|
"""Test opening an existing table asynchronously through namespace."""
|
||||||
|
db = lancedb.connect_namespace_async("dir", {"root": self.temp_dir})
|
||||||
|
|
||||||
|
# Create a child namespace first
|
||||||
|
await db.create_namespace(["test_ns"])
|
||||||
|
|
||||||
|
# Create a table with schema in child namespace
|
||||||
|
schema = pa.schema(
|
||||||
|
[
|
||||||
|
pa.field("id", pa.int64()),
|
||||||
|
pa.field("vector", pa.list_(pa.float32(), 2)),
|
||||||
|
]
|
||||||
|
)
|
||||||
|
await db.create_table("test_table", schema=schema, namespace=["test_ns"])
|
||||||
|
|
||||||
|
# Open the table
|
||||||
|
table = await db.open_table("test_table", namespace=["test_ns"])
|
||||||
|
assert table is not None
|
||||||
|
assert isinstance(table, lancedb.AsyncTable)
|
||||||
|
|
||||||
|
# Test write operation - add data to the table
|
||||||
|
test_data = [
|
||||||
|
{"id": 1, "vector": [1.0, 2.0]},
|
||||||
|
{"id": 2, "vector": [3.0, 4.0]},
|
||||||
|
{"id": 3, "vector": [5.0, 6.0]},
|
||||||
|
]
|
||||||
|
await table.add(test_data)
|
||||||
|
|
||||||
|
# Test read operation - query the table
|
||||||
|
result = await table.to_arrow()
|
||||||
|
assert len(result) == 3
|
||||||
|
assert result.schema.field("id").type == pa.int64()
|
||||||
|
assert result.schema.field("vector").type == pa.list_(pa.float32(), 2)
|
||||||
|
|
||||||
|
# Verify data content
|
||||||
|
result_df = result.to_pandas()
|
||||||
|
assert result_df["id"].tolist() == [1, 2, 3]
|
||||||
|
assert [v.tolist() for v in result_df["vector"]] == [
|
||||||
|
[1.0, 2.0],
|
||||||
|
[3.0, 4.0],
|
||||||
|
[5.0, 6.0],
|
||||||
|
]
|
||||||
|
|
||||||
|
# Test update operation
|
||||||
|
await table.update({"id": 20}, where="id = 2")
|
||||||
|
result = await table.to_arrow()
|
||||||
|
result_df = result.to_pandas().sort_values("id").reset_index(drop=True)
|
||||||
|
assert result_df["id"].tolist() == [1, 3, 20]
|
||||||
|
|
||||||
|
# Test delete operation
|
||||||
|
await table.delete("id = 1")
|
||||||
|
result = await table.to_arrow()
|
||||||
|
assert len(result) == 2
|
||||||
|
result_df = result.to_pandas().sort_values("id").reset_index(drop=True)
|
||||||
|
assert result_df["id"].tolist() == [3, 20]
|
||||||
|
|
||||||
|
async def test_drop_table_async(self):
|
||||||
|
"""Test dropping a table asynchronously through namespace."""
|
||||||
|
db = lancedb.connect_namespace_async("dir", {"root": self.temp_dir})
|
||||||
|
|
||||||
|
# Create a child namespace first
|
||||||
|
await db.create_namespace(["test_ns"])
|
||||||
|
|
||||||
|
# Create tables in child namespace
|
||||||
|
schema = pa.schema(
|
||||||
|
[
|
||||||
|
pa.field("id", pa.int64()),
|
||||||
|
pa.field("vector", pa.list_(pa.float32(), 2)),
|
||||||
|
]
|
||||||
|
)
|
||||||
|
await db.create_table("table1", schema=schema, namespace=["test_ns"])
|
||||||
|
await db.create_table("table2", schema=schema, namespace=["test_ns"])
|
||||||
|
|
||||||
|
# Verify both tables exist in child namespace
|
||||||
|
table_names = list(await db.table_names(namespace=["test_ns"]))
|
||||||
|
assert "table1" in table_names
|
||||||
|
assert "table2" in table_names
|
||||||
|
assert len(table_names) == 2
|
||||||
|
|
||||||
|
# Drop one table
|
||||||
|
await db.drop_table("table1", namespace=["test_ns"])
|
||||||
|
|
||||||
|
# Verify only table2 remains
|
||||||
|
table_names = list(await db.table_names(namespace=["test_ns"]))
|
||||||
|
assert "table1" not in table_names
|
||||||
|
assert "table2" in table_names
|
||||||
|
assert len(table_names) == 1
|
||||||
|
|
||||||
|
async def test_namespace_operations_async(self):
|
||||||
|
"""Test namespace management operations asynchronously."""
|
||||||
|
db = lancedb.connect_namespace_async("dir", {"root": self.temp_dir})
|
||||||
|
|
||||||
|
# Initially no namespaces
|
||||||
|
namespaces = await db.list_namespaces()
|
||||||
|
assert len(list(namespaces)) == 0
|
||||||
|
|
||||||
|
# Create a namespace
|
||||||
|
await db.create_namespace(["test_namespace"])
|
||||||
|
|
||||||
|
# Verify namespace exists
|
||||||
|
namespaces = list(await db.list_namespaces())
|
||||||
|
assert "test_namespace" in namespaces
|
||||||
|
assert len(namespaces) == 1
|
||||||
|
|
||||||
|
# Create table in namespace
|
||||||
|
schema = pa.schema(
|
||||||
|
[
|
||||||
|
pa.field("id", pa.int64()),
|
||||||
|
pa.field("vector", pa.list_(pa.float32(), 2)),
|
||||||
|
]
|
||||||
|
)
|
||||||
|
table = await db.create_table(
|
||||||
|
"test_table", schema=schema, namespace=["test_namespace"]
|
||||||
|
)
|
||||||
|
assert table is not None
|
||||||
|
|
||||||
|
# Verify table exists in namespace
|
||||||
|
tables_in_namespace = list(await db.table_names(namespace=["test_namespace"]))
|
||||||
|
assert "test_table" in tables_in_namespace
|
||||||
|
assert len(tables_in_namespace) == 1
|
||||||
|
|
||||||
|
# Drop table from namespace
|
||||||
|
await db.drop_table("test_table", namespace=["test_namespace"])
|
||||||
|
|
||||||
|
# Verify table no longer exists in namespace
|
||||||
|
tables_in_namespace = list(await db.table_names(namespace=["test_namespace"]))
|
||||||
|
assert len(tables_in_namespace) == 0
|
||||||
|
|
||||||
|
# Drop namespace
|
||||||
|
await db.drop_namespace(["test_namespace"])
|
||||||
|
|
||||||
|
# Verify namespace no longer exists
|
||||||
|
namespaces = list(await db.list_namespaces())
|
||||||
|
assert len(namespaces) == 0
|
||||||
|
|
||||||
|
async def test_drop_all_tables_async(self):
|
||||||
|
"""Test dropping all tables asynchronously through namespace."""
|
||||||
|
db = lancedb.connect_namespace_async("dir", {"root": self.temp_dir})
|
||||||
|
|
||||||
|
# Create a child namespace first
|
||||||
|
await db.create_namespace(["test_ns"])
|
||||||
|
|
||||||
|
# Create multiple tables in child namespace
|
||||||
|
schema = pa.schema(
|
||||||
|
[
|
||||||
|
pa.field("id", pa.int64()),
|
||||||
|
pa.field("vector", pa.list_(pa.float32(), 2)),
|
||||||
|
]
|
||||||
|
)
|
||||||
|
for i in range(3):
|
||||||
|
await db.create_table(f"table{i}", schema=schema, namespace=["test_ns"])
|
||||||
|
|
||||||
|
# Verify tables exist in child namespace
|
||||||
|
table_names = await db.table_names(namespace=["test_ns"])
|
||||||
|
assert len(list(table_names)) == 3
|
||||||
|
|
||||||
|
# Drop all tables in child namespace
|
||||||
|
await db.drop_all_tables(namespace=["test_ns"])
|
||||||
|
|
||||||
|
# Verify all tables are gone from child namespace
|
||||||
|
table_names = await db.table_names(namespace=["test_ns"])
|
||||||
|
assert len(list(table_names)) == 0
|
||||||
|
|||||||
@@ -412,3 +412,50 @@ def test_multi_vector_in_lance_model():
|
|||||||
|
|
||||||
t = TestModel(id=1)
|
t = TestModel(id=1)
|
||||||
assert t.vectors == [[0.0] * 16]
|
assert t.vectors == [[0.0] * 16]
|
||||||
|
|
||||||
|
|
||||||
|
def test_aliases_in_lance_model(mem_db):
|
||||||
|
data = [
|
||||||
|
{"vector": [3.1, 4.1], "item": "foo", "price": 10.0},
|
||||||
|
{"vector": [5.9, 6.5], "item": "bar", "price": 20.0},
|
||||||
|
]
|
||||||
|
tbl = mem_db.create_table("items", data=data)
|
||||||
|
|
||||||
|
class TestModel(LanceModel):
|
||||||
|
name: str = Field(alias="item")
|
||||||
|
price: float
|
||||||
|
distance: float = Field(alias="_distance")
|
||||||
|
|
||||||
|
model = (
|
||||||
|
tbl.search([5.9, 6.5])
|
||||||
|
.distance_type("cosine")
|
||||||
|
.limit(1)
|
||||||
|
.to_pydantic(TestModel)[0]
|
||||||
|
)
|
||||||
|
assert hasattr(model, "name")
|
||||||
|
assert hasattr(model, "distance")
|
||||||
|
assert model.distance < 0.01
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_aliases_in_lance_model_async(mem_db_async):
|
||||||
|
data = [
|
||||||
|
{"vector": [8.3, 2.5], "item": "foo", "price": 12.0},
|
||||||
|
{"vector": [7.7, 3.9], "item": "bar", "price": 11.2},
|
||||||
|
]
|
||||||
|
tbl = await mem_db_async.create_table("items", data=data)
|
||||||
|
|
||||||
|
class TestModel(LanceModel):
|
||||||
|
name: str = Field(alias="item")
|
||||||
|
price: float
|
||||||
|
distance: float = Field(alias="_distance")
|
||||||
|
|
||||||
|
model = (
|
||||||
|
await tbl.vector_search([7.7, 3.9])
|
||||||
|
.distance_type("cosine")
|
||||||
|
.limit(1)
|
||||||
|
.to_pydantic(TestModel)
|
||||||
|
)[0]
|
||||||
|
assert hasattr(model, "name")
|
||||||
|
assert hasattr(model, "distance")
|
||||||
|
assert model.distance < 0.01
|
||||||
|
|||||||
@@ -546,6 +546,22 @@ def query_test_table(query_handler, *, server_version=Version("0.1.0")):
|
|||||||
yield table
|
yield table
|
||||||
|
|
||||||
|
|
||||||
|
def test_head():
|
||||||
|
def handler(body):
|
||||||
|
assert body == {
|
||||||
|
"k": 5,
|
||||||
|
"prefilter": True,
|
||||||
|
"vector": [],
|
||||||
|
"version": None,
|
||||||
|
}
|
||||||
|
|
||||||
|
return pa.table({"id": [1, 2, 3]})
|
||||||
|
|
||||||
|
with query_test_table(handler) as table:
|
||||||
|
data = table.head(5)
|
||||||
|
assert data == pa.table({"id": [1, 2, 3]})
|
||||||
|
|
||||||
|
|
||||||
def test_query_sync_minimal():
|
def test_query_sync_minimal():
|
||||||
def handler(body):
|
def handler(body):
|
||||||
assert body == {
|
assert body == {
|
||||||
|
|||||||
@@ -1487,7 +1487,7 @@ def setup_hybrid_search_table(db: DBConnection, embedding_func):
|
|||||||
table.add([{"text": p} for p in phrases])
|
table.add([{"text": p} for p in phrases])
|
||||||
|
|
||||||
# Create a fts index
|
# Create a fts index
|
||||||
table.create_fts_index("text")
|
table.create_fts_index("text", with_position=True)
|
||||||
|
|
||||||
return table, MyTable, emb
|
return table, MyTable, emb
|
||||||
|
|
||||||
|
|||||||
@@ -690,7 +690,7 @@ impl FTSQuery {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub fn get_query(&self) -> String {
|
pub fn get_query(&self) -> String {
|
||||||
self.fts_query.query.query().to_owned()
|
self.fts_query.query.query().clone()
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn to_query_request(&self) -> PyQueryRequest {
|
pub fn to_query_request(&self) -> PyQueryRequest {
|
||||||
|
|||||||
@@ -134,17 +134,19 @@ pub struct MergeResult {
|
|||||||
pub num_updated_rows: u64,
|
pub num_updated_rows: u64,
|
||||||
pub num_inserted_rows: u64,
|
pub num_inserted_rows: u64,
|
||||||
pub num_deleted_rows: u64,
|
pub num_deleted_rows: u64,
|
||||||
|
pub num_attempts: u32,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[pymethods]
|
#[pymethods]
|
||||||
impl MergeResult {
|
impl MergeResult {
|
||||||
pub fn __repr__(&self) -> String {
|
pub fn __repr__(&self) -> String {
|
||||||
format!(
|
format!(
|
||||||
"MergeResult(version={}, num_updated_rows={}, num_inserted_rows={}, num_deleted_rows={})",
|
"MergeResult(version={}, num_updated_rows={}, num_inserted_rows={}, num_deleted_rows={}, num_attempts={})",
|
||||||
self.version,
|
self.version,
|
||||||
self.num_updated_rows,
|
self.num_updated_rows,
|
||||||
self.num_inserted_rows,
|
self.num_inserted_rows,
|
||||||
self.num_deleted_rows
|
self.num_deleted_rows,
|
||||||
|
self.num_attempts
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -156,6 +158,7 @@ impl From<lancedb::table::MergeResult> for MergeResult {
|
|||||||
num_updated_rows: result.num_updated_rows,
|
num_updated_rows: result.num_updated_rows,
|
||||||
num_inserted_rows: result.num_inserted_rows,
|
num_inserted_rows: result.num_inserted_rows,
|
||||||
num_deleted_rows: result.num_deleted_rows,
|
num_deleted_rows: result.num_deleted_rows,
|
||||||
|
num_attempts: result.num_attempts,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "lancedb"
|
name = "lancedb"
|
||||||
version = "0.22.3"
|
version = "0.22.4-beta.2"
|
||||||
edition.workspace = true
|
edition.workspace = true
|
||||||
description = "LanceDB: A serverless, low-latency vector database for AI applications"
|
description = "LanceDB: A serverless, low-latency vector database for AI applications"
|
||||||
license.workspace = true
|
license.workspace = true
|
||||||
@@ -105,12 +105,12 @@ test-log = "0.2"
|
|||||||
|
|
||||||
[features]
|
[features]
|
||||||
default = ["aws", "gcs", "azure", "dynamodb", "oss"]
|
default = ["aws", "gcs", "azure", "dynamodb", "oss"]
|
||||||
aws = ["lance/aws", "lance-io/aws"]
|
aws = ["lance/aws", "lance-io/aws", "lance-namespace-impls/dir-aws"]
|
||||||
oss = ["lance/oss", "lance-io/oss"]
|
oss = ["lance/oss", "lance-io/oss", "lance-namespace-impls/dir-oss"]
|
||||||
gcs = ["lance/gcp", "lance-io/gcp"]
|
gcs = ["lance/gcp", "lance-io/gcp", "lance-namespace-impls/dir-gcp"]
|
||||||
azure = ["lance/azure", "lance-io/azure"]
|
azure = ["lance/azure", "lance-io/azure", "lance-namespace-impls/dir-azure"]
|
||||||
dynamodb = ["lance/dynamodb", "aws"]
|
dynamodb = ["lance/dynamodb", "aws"]
|
||||||
remote = ["dep:reqwest", "dep:http"]
|
remote = ["dep:reqwest", "dep:http", "lance-namespace-impls/rest"]
|
||||||
fp16kernels = ["lance-linalg/fp16kernels"]
|
fp16kernels = ["lance-linalg/fp16kernels"]
|
||||||
s3-test = []
|
s3-test = []
|
||||||
bedrock = ["dep:aws-sdk-bedrockruntime"]
|
bedrock = ["dep:aws-sdk-bedrockruntime"]
|
||||||
|
|||||||
@@ -9,11 +9,11 @@ all-tests: feature-tests remote-tests
|
|||||||
# the environment.
|
# the environment.
|
||||||
feature-tests:
|
feature-tests:
|
||||||
../../ci/run_with_docker_compose.sh \
|
../../ci/run_with_docker_compose.sh \
|
||||||
cargo test --all-features --tests --locked --examples
|
cargo test --all-features --tests --locked --examples $(CARGO_ARGS)
|
||||||
.PHONY: feature-tests
|
.PHONY: feature-tests
|
||||||
|
|
||||||
# Run tests against remote endpoints.
|
# Run tests against remote endpoints.
|
||||||
remote-tests:
|
remote-tests:
|
||||||
../../ci/run_with_test_connection.sh \
|
../../ci/run_with_test_connection.sh \
|
||||||
cargo test --features remote --locked
|
cargo test --features remote --locked $(CARGO_ARGS)
|
||||||
.PHONY: remote-tests
|
.PHONY: remote-tests
|
||||||
|
|||||||
@@ -239,7 +239,7 @@ impl<const HAS_DATA: bool> CreateTableBuilder<HAS_DATA> {
|
|||||||
/// Options already set on the connection will be inherited by the table,
|
/// Options already set on the connection will be inherited by the table,
|
||||||
/// but can be overridden here.
|
/// but can be overridden here.
|
||||||
///
|
///
|
||||||
/// See available options at <https://lancedb.github.io/lancedb/guides/storage/>
|
/// See available options at <https://lancedb.com/docs/storage/>
|
||||||
pub fn storage_option(mut self, key: impl Into<String>, value: impl Into<String>) -> Self {
|
pub fn storage_option(mut self, key: impl Into<String>, value: impl Into<String>) -> Self {
|
||||||
let store_options = self
|
let store_options = self
|
||||||
.request
|
.request
|
||||||
@@ -259,7 +259,7 @@ impl<const HAS_DATA: bool> CreateTableBuilder<HAS_DATA> {
|
|||||||
/// Options already set on the connection will be inherited by the table,
|
/// Options already set on the connection will be inherited by the table,
|
||||||
/// but can be overridden here.
|
/// but can be overridden here.
|
||||||
///
|
///
|
||||||
/// See available options at <https://lancedb.github.io/lancedb/guides/storage/>
|
/// See available options at <https://lancedb.com/docs/storage/>
|
||||||
pub fn storage_options(
|
pub fn storage_options(
|
||||||
mut self,
|
mut self,
|
||||||
pairs: impl IntoIterator<Item = (impl Into<String>, impl Into<String>)>,
|
pairs: impl IntoIterator<Item = (impl Into<String>, impl Into<String>)>,
|
||||||
@@ -442,7 +442,7 @@ impl OpenTableBuilder {
|
|||||||
/// Options already set on the connection will be inherited by the table,
|
/// Options already set on the connection will be inherited by the table,
|
||||||
/// but can be overridden here.
|
/// but can be overridden here.
|
||||||
///
|
///
|
||||||
/// See available options at <https://lancedb.github.io/lancedb/guides/storage/>
|
/// See available options at <https://lancedb.com/docs/storage/>
|
||||||
pub fn storage_option(mut self, key: impl Into<String>, value: impl Into<String>) -> Self {
|
pub fn storage_option(mut self, key: impl Into<String>, value: impl Into<String>) -> Self {
|
||||||
let storage_options = self
|
let storage_options = self
|
||||||
.request
|
.request
|
||||||
@@ -461,7 +461,7 @@ impl OpenTableBuilder {
|
|||||||
/// Options already set on the connection will be inherited by the table,
|
/// Options already set on the connection will be inherited by the table,
|
||||||
/// but can be overridden here.
|
/// but can be overridden here.
|
||||||
///
|
///
|
||||||
/// See available options at <https://lancedb.github.io/lancedb/guides/storage/>
|
/// See available options at <https://lancedb.com/docs/storage/>
|
||||||
pub fn storage_options(
|
pub fn storage_options(
|
||||||
mut self,
|
mut self,
|
||||||
pairs: impl IntoIterator<Item = (impl Into<String>, impl Into<String>)>,
|
pairs: impl IntoIterator<Item = (impl Into<String>, impl Into<String>)>,
|
||||||
@@ -959,7 +959,7 @@ impl ConnectBuilder {
|
|||||||
|
|
||||||
/// Set an option for the storage layer.
|
/// Set an option for the storage layer.
|
||||||
///
|
///
|
||||||
/// See available options at <https://lancedb.github.io/lancedb/guides/storage/>
|
/// See available options at <https://lancedb.com/docs/storage/>
|
||||||
pub fn storage_option(mut self, key: impl Into<String>, value: impl Into<String>) -> Self {
|
pub fn storage_option(mut self, key: impl Into<String>, value: impl Into<String>) -> Self {
|
||||||
self.request.options.insert(key.into(), value.into());
|
self.request.options.insert(key.into(), value.into());
|
||||||
self
|
self
|
||||||
@@ -967,7 +967,7 @@ impl ConnectBuilder {
|
|||||||
|
|
||||||
/// Set multiple options for the storage layer.
|
/// Set multiple options for the storage layer.
|
||||||
///
|
///
|
||||||
/// See available options at <https://lancedb.github.io/lancedb/guides/storage/>
|
/// See available options at <https://lancedb.com/docs/storage/>
|
||||||
pub fn storage_options(
|
pub fn storage_options(
|
||||||
mut self,
|
mut self,
|
||||||
pairs: impl IntoIterator<Item = (impl Into<String>, impl Into<String>)>,
|
pairs: impl IntoIterator<Item = (impl Into<String>, impl Into<String>)>,
|
||||||
@@ -1102,7 +1102,7 @@ impl ConnectNamespaceBuilder {
|
|||||||
|
|
||||||
/// Set an option for the storage layer.
|
/// Set an option for the storage layer.
|
||||||
///
|
///
|
||||||
/// See available options at <https://lancedb.github.io/lancedb/guides/storage/>
|
/// See available options at <https://lancedb.com/docs/storage/>
|
||||||
pub fn storage_option(mut self, key: impl Into<String>, value: impl Into<String>) -> Self {
|
pub fn storage_option(mut self, key: impl Into<String>, value: impl Into<String>) -> Self {
|
||||||
self.storage_options.insert(key.into(), value.into());
|
self.storage_options.insert(key.into(), value.into());
|
||||||
self
|
self
|
||||||
@@ -1110,7 +1110,7 @@ impl ConnectNamespaceBuilder {
|
|||||||
|
|
||||||
/// Set multiple options for the storage layer.
|
/// Set multiple options for the storage layer.
|
||||||
///
|
///
|
||||||
/// See available options at <https://lancedb.github.io/lancedb/guides/storage/>
|
/// See available options at <https://lancedb.com/docs/storage/>
|
||||||
pub fn storage_options(
|
pub fn storage_options(
|
||||||
mut self,
|
mut self,
|
||||||
pairs: impl IntoIterator<Item = (impl Into<String>, impl Into<String>)>,
|
pairs: impl IntoIterator<Item = (impl Into<String>, impl Into<String>)>,
|
||||||
|
|||||||
@@ -35,6 +35,7 @@ pub const LANCE_FILE_EXTENSION: &str = "lance";
|
|||||||
|
|
||||||
pub const OPT_NEW_TABLE_STORAGE_VERSION: &str = "new_table_data_storage_version";
|
pub const OPT_NEW_TABLE_STORAGE_VERSION: &str = "new_table_data_storage_version";
|
||||||
pub const OPT_NEW_TABLE_V2_MANIFEST_PATHS: &str = "new_table_enable_v2_manifest_paths";
|
pub const OPT_NEW_TABLE_V2_MANIFEST_PATHS: &str = "new_table_enable_v2_manifest_paths";
|
||||||
|
pub const OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS: &str = "new_table_enable_stable_row_ids";
|
||||||
|
|
||||||
/// Controls how new tables should be created
|
/// Controls how new tables should be created
|
||||||
#[derive(Clone, Debug, Default)]
|
#[derive(Clone, Debug, Default)]
|
||||||
@@ -48,6 +49,12 @@ pub struct NewTableConfig {
|
|||||||
/// V2 manifest paths are more efficient than V2 manifest paths but are not
|
/// V2 manifest paths are more efficient than V2 manifest paths but are not
|
||||||
/// supported by old clients.
|
/// supported by old clients.
|
||||||
pub enable_v2_manifest_paths: Option<bool>,
|
pub enable_v2_manifest_paths: Option<bool>,
|
||||||
|
/// Whether to enable stable row IDs for new tables
|
||||||
|
///
|
||||||
|
/// When enabled, row IDs remain stable after compaction, update, delete,
|
||||||
|
/// and merges. This is useful for materialized views and other use cases
|
||||||
|
/// that need to track source rows across these operations.
|
||||||
|
pub enable_stable_row_ids: Option<bool>,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Options specific to the listing database
|
/// Options specific to the listing database
|
||||||
@@ -60,7 +67,7 @@ pub struct ListingDatabaseOptions {
|
|||||||
/// These are used to create/list tables and they are inherited by all tables
|
/// These are used to create/list tables and they are inherited by all tables
|
||||||
/// opened by this database.
|
/// opened by this database.
|
||||||
///
|
///
|
||||||
/// See available options at <https://lancedb.github.io/lancedb/guides/storage/>
|
/// See available options at <https://lancedb.com/docs/storage/>
|
||||||
pub storage_options: HashMap<String, String>,
|
pub storage_options: HashMap<String, String>,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -87,6 +94,14 @@ impl ListingDatabaseOptions {
|
|||||||
})
|
})
|
||||||
})
|
})
|
||||||
.transpose()?,
|
.transpose()?,
|
||||||
|
enable_stable_row_ids: map
|
||||||
|
.get(OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS)
|
||||||
|
.map(|s| {
|
||||||
|
s.parse::<bool>().map_err(|_| Error::InvalidInput {
|
||||||
|
message: format!("enable_stable_row_ids must be a boolean, received {}", s),
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.transpose()?,
|
||||||
};
|
};
|
||||||
// We just assume that any options that are not new table config options are storage options
|
// We just assume that any options that are not new table config options are storage options
|
||||||
let storage_options = map
|
let storage_options = map
|
||||||
@@ -94,6 +109,7 @@ impl ListingDatabaseOptions {
|
|||||||
.filter(|(key, _)| {
|
.filter(|(key, _)| {
|
||||||
key.as_str() != OPT_NEW_TABLE_STORAGE_VERSION
|
key.as_str() != OPT_NEW_TABLE_STORAGE_VERSION
|
||||||
&& key.as_str() != OPT_NEW_TABLE_V2_MANIFEST_PATHS
|
&& key.as_str() != OPT_NEW_TABLE_V2_MANIFEST_PATHS
|
||||||
|
&& key.as_str() != OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS
|
||||||
})
|
})
|
||||||
.map(|(key, value)| (key.clone(), value.clone()))
|
.map(|(key, value)| (key.clone(), value.clone()))
|
||||||
.collect();
|
.collect();
|
||||||
@@ -118,6 +134,12 @@ impl DatabaseOptions for ListingDatabaseOptions {
|
|||||||
enable_v2_manifest_paths.to_string(),
|
enable_v2_manifest_paths.to_string(),
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
if let Some(enable_stable_row_ids) = self.new_table_config.enable_stable_row_ids {
|
||||||
|
map.insert(
|
||||||
|
OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS.to_string(),
|
||||||
|
enable_stable_row_ids.to_string(),
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -157,7 +179,7 @@ impl ListingDatabaseOptionsBuilder {
|
|||||||
|
|
||||||
/// Set an option for the storage layer.
|
/// Set an option for the storage layer.
|
||||||
///
|
///
|
||||||
/// See available options at <https://lancedb.github.io/lancedb/guides/storage/>
|
/// See available options at <https://lancedb.com/docs/storage/>
|
||||||
pub fn storage_option(mut self, key: impl Into<String>, value: impl Into<String>) -> Self {
|
pub fn storage_option(mut self, key: impl Into<String>, value: impl Into<String>) -> Self {
|
||||||
self.options
|
self.options
|
||||||
.storage_options
|
.storage_options
|
||||||
@@ -167,7 +189,7 @@ impl ListingDatabaseOptionsBuilder {
|
|||||||
|
|
||||||
/// Set multiple options for the storage layer.
|
/// Set multiple options for the storage layer.
|
||||||
///
|
///
|
||||||
/// See available options at <https://lancedb.github.io/lancedb/guides/storage/>
|
/// See available options at <https://lancedb.com/docs/storage/>
|
||||||
pub fn storage_options(
|
pub fn storage_options(
|
||||||
mut self,
|
mut self,
|
||||||
pairs: impl IntoIterator<Item = (impl Into<String>, impl Into<String>)>,
|
pairs: impl IntoIterator<Item = (impl Into<String>, impl Into<String>)>,
|
||||||
@@ -475,7 +497,7 @@ impl ListingDatabase {
|
|||||||
// this error is not lance::Error::DatasetNotFound, as the method
|
// this error is not lance::Error::DatasetNotFound, as the method
|
||||||
// `remove_dir_all` may be used to remove something not be a dataset
|
// `remove_dir_all` may be used to remove something not be a dataset
|
||||||
lance::Error::NotFound { .. } => Error::TableNotFound {
|
lance::Error::NotFound { .. } => Error::TableNotFound {
|
||||||
name: name.to_owned(),
|
name: name.clone(),
|
||||||
source: Box::new(err),
|
source: Box::new(err),
|
||||||
},
|
},
|
||||||
_ => Error::from(err),
|
_ => Error::from(err),
|
||||||
@@ -497,7 +519,7 @@ impl ListingDatabase {
|
|||||||
fn extract_storage_overrides(
|
fn extract_storage_overrides(
|
||||||
&self,
|
&self,
|
||||||
request: &CreateTableRequest,
|
request: &CreateTableRequest,
|
||||||
) -> Result<(Option<LanceFileVersion>, Option<bool>)> {
|
) -> Result<(Option<LanceFileVersion>, Option<bool>, Option<bool>)> {
|
||||||
let storage_options = request
|
let storage_options = request
|
||||||
.write_options
|
.write_options
|
||||||
.lance_write_params
|
.lance_write_params
|
||||||
@@ -518,7 +540,19 @@ impl ListingDatabase {
|
|||||||
message: "enable_v2_manifest_paths must be a boolean".to_string(),
|
message: "enable_v2_manifest_paths must be a boolean".to_string(),
|
||||||
})?;
|
})?;
|
||||||
|
|
||||||
Ok((storage_version_override, v2_manifest_override))
|
let stable_row_ids_override = storage_options
|
||||||
|
.and_then(|opts| opts.get(OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS))
|
||||||
|
.map(|s| s.parse::<bool>())
|
||||||
|
.transpose()
|
||||||
|
.map_err(|_| Error::InvalidInput {
|
||||||
|
message: "enable_stable_row_ids must be a boolean".to_string(),
|
||||||
|
})?;
|
||||||
|
|
||||||
|
Ok((
|
||||||
|
storage_version_override,
|
||||||
|
v2_manifest_override,
|
||||||
|
stable_row_ids_override,
|
||||||
|
))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Prepare write parameters for table creation
|
/// Prepare write parameters for table creation
|
||||||
@@ -527,6 +561,7 @@ impl ListingDatabase {
|
|||||||
request: &CreateTableRequest,
|
request: &CreateTableRequest,
|
||||||
storage_version_override: Option<LanceFileVersion>,
|
storage_version_override: Option<LanceFileVersion>,
|
||||||
v2_manifest_override: Option<bool>,
|
v2_manifest_override: Option<bool>,
|
||||||
|
stable_row_ids_override: Option<bool>,
|
||||||
) -> lance::dataset::WriteParams {
|
) -> lance::dataset::WriteParams {
|
||||||
let mut write_params = request
|
let mut write_params = request
|
||||||
.write_options
|
.write_options
|
||||||
@@ -571,6 +606,13 @@ impl ListingDatabase {
|
|||||||
write_params.enable_v2_manifest_paths = enable_v2_manifest_paths;
|
write_params.enable_v2_manifest_paths = enable_v2_manifest_paths;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Apply enable_stable_row_ids: table-level override takes precedence over connection config
|
||||||
|
if let Some(enable_stable_row_ids) =
|
||||||
|
stable_row_ids_override.or(self.new_table_config.enable_stable_row_ids)
|
||||||
|
{
|
||||||
|
write_params.enable_stable_row_ids = enable_stable_row_ids;
|
||||||
|
}
|
||||||
|
|
||||||
if matches!(&request.mode, CreateTableMode::Overwrite) {
|
if matches!(&request.mode, CreateTableMode::Overwrite) {
|
||||||
write_params.mode = WriteMode::Overwrite;
|
write_params.mode = WriteMode::Overwrite;
|
||||||
}
|
}
|
||||||
@@ -706,11 +748,15 @@ impl Database for ListingDatabase {
|
|||||||
.clone()
|
.clone()
|
||||||
.unwrap_or_else(|| self.table_uri(&request.name).unwrap());
|
.unwrap_or_else(|| self.table_uri(&request.name).unwrap());
|
||||||
|
|
||||||
let (storage_version_override, v2_manifest_override) =
|
let (storage_version_override, v2_manifest_override, stable_row_ids_override) =
|
||||||
self.extract_storage_overrides(&request)?;
|
self.extract_storage_overrides(&request)?;
|
||||||
|
|
||||||
let write_params =
|
let write_params = self.prepare_write_params(
|
||||||
self.prepare_write_params(&request, storage_version_override, v2_manifest_override);
|
&request,
|
||||||
|
storage_version_override,
|
||||||
|
v2_manifest_override,
|
||||||
|
stable_row_ids_override,
|
||||||
|
);
|
||||||
|
|
||||||
let data_schema = request.data.arrow_schema();
|
let data_schema = request.data.arrow_schema();
|
||||||
|
|
||||||
@@ -921,7 +967,7 @@ impl Database for ListingDatabase {
|
|||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
use crate::connection::ConnectRequest;
|
use crate::connection::ConnectRequest;
|
||||||
use crate::database::{CreateTableData, CreateTableMode, CreateTableRequest};
|
use crate::database::{CreateTableData, CreateTableMode, CreateTableRequest, WriteOptions};
|
||||||
use crate::table::{Table, TableDefinition};
|
use crate::table::{Table, TableDefinition};
|
||||||
use arrow_array::{Int32Array, RecordBatch, StringArray};
|
use arrow_array::{Int32Array, RecordBatch, StringArray};
|
||||||
use arrow_schema::{DataType, Field, Schema};
|
use arrow_schema::{DataType, Field, Schema};
|
||||||
@@ -1621,4 +1667,267 @@ mod tests {
|
|||||||
// Cloned table should have all 8 rows from the latest version
|
// Cloned table should have all 8 rows from the latest version
|
||||||
assert_eq!(cloned_table.count_rows(None).await.unwrap(), 8);
|
assert_eq!(cloned_table.count_rows(None).await.unwrap(), 8);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_create_table_with_stable_row_ids_connection_level() {
|
||||||
|
let tempdir = tempdir().unwrap();
|
||||||
|
let uri = tempdir.path().to_str().unwrap();
|
||||||
|
|
||||||
|
// Create database with stable row IDs enabled at connection level
|
||||||
|
let mut options = HashMap::new();
|
||||||
|
options.insert(
|
||||||
|
OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS.to_string(),
|
||||||
|
"true".to_string(),
|
||||||
|
);
|
||||||
|
|
||||||
|
let request = ConnectRequest {
|
||||||
|
uri: uri.to_string(),
|
||||||
|
#[cfg(feature = "remote")]
|
||||||
|
client_config: Default::default(),
|
||||||
|
options,
|
||||||
|
read_consistency_interval: None,
|
||||||
|
session: None,
|
||||||
|
};
|
||||||
|
|
||||||
|
let db = ListingDatabase::connect_with_options(&request)
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
// Verify the config was parsed correctly
|
||||||
|
assert_eq!(db.new_table_config.enable_stable_row_ids, Some(true));
|
||||||
|
|
||||||
|
// Create a table - it should inherit the stable row IDs setting
|
||||||
|
let schema = Arc::new(Schema::new(vec![Field::new("id", DataType::Int32, false)]));
|
||||||
|
|
||||||
|
let batch = RecordBatch::try_new(
|
||||||
|
schema.clone(),
|
||||||
|
vec![Arc::new(Int32Array::from(vec![1, 2, 3]))],
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let reader = Box::new(arrow_array::RecordBatchIterator::new(
|
||||||
|
vec![Ok(batch)],
|
||||||
|
schema.clone(),
|
||||||
|
));
|
||||||
|
|
||||||
|
let table = db
|
||||||
|
.create_table(CreateTableRequest {
|
||||||
|
name: "test_stable".to_string(),
|
||||||
|
namespace: vec![],
|
||||||
|
data: CreateTableData::Data(reader),
|
||||||
|
mode: CreateTableMode::Create,
|
||||||
|
write_options: Default::default(),
|
||||||
|
location: None,
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
// Verify table was created successfully
|
||||||
|
assert_eq!(table.count_rows(None).await.unwrap(), 3);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_create_table_with_stable_row_ids_table_level() {
|
||||||
|
let (_tempdir, db) = setup_database().await;
|
||||||
|
|
||||||
|
// Verify connection has no stable row IDs config
|
||||||
|
assert_eq!(db.new_table_config.enable_stable_row_ids, None);
|
||||||
|
|
||||||
|
// Create a table with stable row IDs enabled at table level via storage_options
|
||||||
|
let schema = Arc::new(Schema::new(vec![Field::new("id", DataType::Int32, false)]));
|
||||||
|
|
||||||
|
let batch = RecordBatch::try_new(
|
||||||
|
schema.clone(),
|
||||||
|
vec![Arc::new(Int32Array::from(vec![1, 2, 3]))],
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let reader = Box::new(arrow_array::RecordBatchIterator::new(
|
||||||
|
vec![Ok(batch)],
|
||||||
|
schema.clone(),
|
||||||
|
));
|
||||||
|
|
||||||
|
let mut storage_options = HashMap::new();
|
||||||
|
storage_options.insert(
|
||||||
|
OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS.to_string(),
|
||||||
|
"true".to_string(),
|
||||||
|
);
|
||||||
|
|
||||||
|
let write_options = WriteOptions {
|
||||||
|
lance_write_params: Some(lance::dataset::WriteParams {
|
||||||
|
store_params: Some(lance::io::ObjectStoreParams {
|
||||||
|
storage_options: Some(storage_options),
|
||||||
|
..Default::default()
|
||||||
|
}),
|
||||||
|
..Default::default()
|
||||||
|
}),
|
||||||
|
};
|
||||||
|
|
||||||
|
let table = db
|
||||||
|
.create_table(CreateTableRequest {
|
||||||
|
name: "test_stable_table_level".to_string(),
|
||||||
|
namespace: vec![],
|
||||||
|
data: CreateTableData::Data(reader),
|
||||||
|
mode: CreateTableMode::Create,
|
||||||
|
write_options,
|
||||||
|
location: None,
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
// Verify table was created successfully
|
||||||
|
assert_eq!(table.count_rows(None).await.unwrap(), 3);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_create_table_stable_row_ids_table_overrides_connection() {
|
||||||
|
let tempdir = tempdir().unwrap();
|
||||||
|
let uri = tempdir.path().to_str().unwrap();
|
||||||
|
|
||||||
|
// Create database with stable row IDs enabled at connection level
|
||||||
|
let mut options = HashMap::new();
|
||||||
|
options.insert(
|
||||||
|
OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS.to_string(),
|
||||||
|
"true".to_string(),
|
||||||
|
);
|
||||||
|
|
||||||
|
let request = ConnectRequest {
|
||||||
|
uri: uri.to_string(),
|
||||||
|
#[cfg(feature = "remote")]
|
||||||
|
client_config: Default::default(),
|
||||||
|
options,
|
||||||
|
read_consistency_interval: None,
|
||||||
|
session: None,
|
||||||
|
};
|
||||||
|
|
||||||
|
let db = ListingDatabase::connect_with_options(&request)
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
assert_eq!(db.new_table_config.enable_stable_row_ids, Some(true));
|
||||||
|
|
||||||
|
// Create table with stable row IDs disabled at table level (overrides connection)
|
||||||
|
let schema = Arc::new(Schema::new(vec![Field::new("id", DataType::Int32, false)]));
|
||||||
|
|
||||||
|
let batch = RecordBatch::try_new(
|
||||||
|
schema.clone(),
|
||||||
|
vec![Arc::new(Int32Array::from(vec![1, 2, 3]))],
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let reader = Box::new(arrow_array::RecordBatchIterator::new(
|
||||||
|
vec![Ok(batch)],
|
||||||
|
schema.clone(),
|
||||||
|
));
|
||||||
|
|
||||||
|
let mut storage_options = HashMap::new();
|
||||||
|
storage_options.insert(
|
||||||
|
OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS.to_string(),
|
||||||
|
"false".to_string(),
|
||||||
|
);
|
||||||
|
|
||||||
|
let write_options = WriteOptions {
|
||||||
|
lance_write_params: Some(lance::dataset::WriteParams {
|
||||||
|
store_params: Some(lance::io::ObjectStoreParams {
|
||||||
|
storage_options: Some(storage_options),
|
||||||
|
..Default::default()
|
||||||
|
}),
|
||||||
|
..Default::default()
|
||||||
|
}),
|
||||||
|
};
|
||||||
|
|
||||||
|
let table = db
|
||||||
|
.create_table(CreateTableRequest {
|
||||||
|
name: "test_override".to_string(),
|
||||||
|
namespace: vec![],
|
||||||
|
data: CreateTableData::Data(reader),
|
||||||
|
mode: CreateTableMode::Create,
|
||||||
|
write_options,
|
||||||
|
location: None,
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
// Verify table was created successfully
|
||||||
|
assert_eq!(table.count_rows(None).await.unwrap(), 3);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_stable_row_ids_invalid_value() {
|
||||||
|
let tempdir = tempdir().unwrap();
|
||||||
|
let uri = tempdir.path().to_str().unwrap();
|
||||||
|
|
||||||
|
// Try to create database with invalid stable row IDs value
|
||||||
|
let mut options = HashMap::new();
|
||||||
|
options.insert(
|
||||||
|
OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS.to_string(),
|
||||||
|
"not_a_boolean".to_string(),
|
||||||
|
);
|
||||||
|
|
||||||
|
let request = ConnectRequest {
|
||||||
|
uri: uri.to_string(),
|
||||||
|
#[cfg(feature = "remote")]
|
||||||
|
client_config: Default::default(),
|
||||||
|
options,
|
||||||
|
read_consistency_interval: None,
|
||||||
|
session: None,
|
||||||
|
};
|
||||||
|
|
||||||
|
let result = ListingDatabase::connect_with_options(&request).await;
|
||||||
|
|
||||||
|
assert!(result.is_err());
|
||||||
|
assert!(matches!(
|
||||||
|
result.unwrap_err(),
|
||||||
|
Error::InvalidInput { message } if message.contains("enable_stable_row_ids must be a boolean")
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_stable_row_ids_config_serialization() {
|
||||||
|
// Test that ListingDatabaseOptions correctly serializes stable_row_ids
|
||||||
|
let mut options = HashMap::new();
|
||||||
|
options.insert(
|
||||||
|
OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS.to_string(),
|
||||||
|
"true".to_string(),
|
||||||
|
);
|
||||||
|
|
||||||
|
// Parse the options
|
||||||
|
let db_options = ListingDatabaseOptions::parse_from_map(&options).unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
db_options.new_table_config.enable_stable_row_ids,
|
||||||
|
Some(true)
|
||||||
|
);
|
||||||
|
|
||||||
|
// Serialize back to map
|
||||||
|
let mut serialized = HashMap::new();
|
||||||
|
db_options.serialize_into_map(&mut serialized);
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
serialized.get(OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS),
|
||||||
|
Some(&"true".to_string())
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_stable_row_ids_config_parse_false() {
|
||||||
|
let mut options = HashMap::new();
|
||||||
|
options.insert(
|
||||||
|
OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS.to_string(),
|
||||||
|
"false".to_string(),
|
||||||
|
);
|
||||||
|
|
||||||
|
let db_options = ListingDatabaseOptions::parse_from_map(&options).unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
db_options.new_table_config.enable_stable_row_ids,
|
||||||
|
Some(false)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_stable_row_ids_config_not_set() {
|
||||||
|
let options = HashMap::new();
|
||||||
|
|
||||||
|
let db_options = ListingDatabaseOptions::parse_from_map(&options).unwrap();
|
||||||
|
assert_eq!(db_options.new_table_config.enable_stable_row_ids, None);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -71,7 +71,7 @@
|
|||||||
//! It treats [`FixedSizeList<Float16/Float32>`](https://docs.rs/arrow/latest/arrow/array/struct.FixedSizeListArray.html)
|
//! It treats [`FixedSizeList<Float16/Float32>`](https://docs.rs/arrow/latest/arrow/array/struct.FixedSizeListArray.html)
|
||||||
//! columns as vector columns.
|
//! columns as vector columns.
|
||||||
//!
|
//!
|
||||||
//! For more details, please refer to [LanceDB documentation](https://lancedb.github.io/lancedb/).
|
//! For more details, please refer to the [LanceDB documentation](https://lancedb.com/docs).
|
||||||
//!
|
//!
|
||||||
//! #### Create a table
|
//! #### Create a table
|
||||||
//!
|
//!
|
||||||
|
|||||||
@@ -90,7 +90,7 @@ pub struct RemoteDatabaseOptions {
|
|||||||
pub host_override: Option<String>,
|
pub host_override: Option<String>,
|
||||||
/// Storage options configure the storage layer (e.g. S3, GCS, Azure, etc.)
|
/// Storage options configure the storage layer (e.g. S3, GCS, Azure, etc.)
|
||||||
///
|
///
|
||||||
/// See available options at <https://lancedb.github.io/lancedb/guides/storage/>
|
/// See available options at <https://lancedb.com/docs/storage/>
|
||||||
///
|
///
|
||||||
/// These options are only used for LanceDB Enterprise and only a subset of options
|
/// These options are only used for LanceDB Enterprise and only a subset of options
|
||||||
/// are supported.
|
/// are supported.
|
||||||
|
|||||||
@@ -1183,6 +1183,7 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
|
|||||||
num_deleted_rows: 0,
|
num_deleted_rows: 0,
|
||||||
num_inserted_rows: 0,
|
num_inserted_rows: 0,
|
||||||
num_updated_rows: 0,
|
num_updated_rows: 0,
|
||||||
|
num_attempts: 0,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -467,6 +467,11 @@ pub struct MergeResult {
|
|||||||
/// However those rows are not shared with the user.
|
/// However those rows are not shared with the user.
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
pub num_deleted_rows: u64,
|
pub num_deleted_rows: u64,
|
||||||
|
/// Number of attempts performed during the merge operation.
|
||||||
|
/// This includes the initial attempt plus any retries due to transaction conflicts.
|
||||||
|
/// A value of 1 means the operation succeeded on the first try.
|
||||||
|
#[serde(default)]
|
||||||
|
pub num_attempts: u32,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
|
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
|
||||||
@@ -1810,8 +1815,17 @@ impl NativeTable {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Helper to get num_sub_vectors with default calculation
|
// Helper to get num_sub_vectors with default calculation
|
||||||
fn get_num_sub_vectors(provided: Option<u32>, dim: u32) -> u32 {
|
fn get_num_sub_vectors(provided: Option<u32>, dim: u32, num_bits: Option<u32>) -> u32 {
|
||||||
provided.unwrap_or_else(|| suggested_num_sub_vectors(dim))
|
if let Some(provided) = provided {
|
||||||
|
return provided;
|
||||||
|
}
|
||||||
|
let suggested = suggested_num_sub_vectors(dim);
|
||||||
|
if num_bits.is_some_and(|num_bits| num_bits == 4) && suggested % 2 != 0 {
|
||||||
|
// num_sub_vectors must be even when 4 bits are used
|
||||||
|
suggested + 1
|
||||||
|
} else {
|
||||||
|
suggested
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Helper to extract vector dimension from field
|
// Helper to extract vector dimension from field
|
||||||
@@ -1834,7 +1848,7 @@ impl NativeTable {
|
|||||||
// Use IvfPq as the default for auto vector indices
|
// Use IvfPq as the default for auto vector indices
|
||||||
let dim = Self::get_vector_dimension(field)?;
|
let dim = Self::get_vector_dimension(field)?;
|
||||||
let ivf_params = lance_index::vector::ivf::IvfBuildParams::default();
|
let ivf_params = lance_index::vector::ivf::IvfBuildParams::default();
|
||||||
let num_sub_vectors = Self::get_num_sub_vectors(None, dim);
|
let num_sub_vectors = Self::get_num_sub_vectors(None, dim, None);
|
||||||
let pq_params =
|
let pq_params =
|
||||||
lance_index::vector::pq::PQBuildParams::new(num_sub_vectors as usize, 8);
|
lance_index::vector::pq::PQBuildParams::new(num_sub_vectors as usize, 8);
|
||||||
let lance_idx_params =
|
let lance_idx_params =
|
||||||
@@ -1901,7 +1915,8 @@ impl NativeTable {
|
|||||||
index.sample_rate,
|
index.sample_rate,
|
||||||
index.max_iterations,
|
index.max_iterations,
|
||||||
);
|
);
|
||||||
let num_sub_vectors = Self::get_num_sub_vectors(index.num_sub_vectors, dim);
|
let num_sub_vectors =
|
||||||
|
Self::get_num_sub_vectors(index.num_sub_vectors, dim, index.num_bits);
|
||||||
let num_bits = index.num_bits.unwrap_or(8) as usize;
|
let num_bits = index.num_bits.unwrap_or(8) as usize;
|
||||||
let mut pq_params = PQBuildParams::new(num_sub_vectors as usize, num_bits);
|
let mut pq_params = PQBuildParams::new(num_sub_vectors as usize, num_bits);
|
||||||
pq_params.max_iters = index.max_iterations as usize;
|
pq_params.max_iters = index.max_iterations as usize;
|
||||||
@@ -1937,7 +1952,8 @@ impl NativeTable {
|
|||||||
index.sample_rate,
|
index.sample_rate,
|
||||||
index.max_iterations,
|
index.max_iterations,
|
||||||
);
|
);
|
||||||
let num_sub_vectors = Self::get_num_sub_vectors(index.num_sub_vectors, dim);
|
let num_sub_vectors =
|
||||||
|
Self::get_num_sub_vectors(index.num_sub_vectors, dim, index.num_bits);
|
||||||
let hnsw_params = HnswBuildParams::default()
|
let hnsw_params = HnswBuildParams::default()
|
||||||
.num_edges(index.m as usize)
|
.num_edges(index.m as usize)
|
||||||
.ef_construction(index.ef_construction as usize);
|
.ef_construction(index.ef_construction as usize);
|
||||||
@@ -2520,6 +2536,7 @@ impl BaseTable for NativeTable {
|
|||||||
num_updated_rows: stats.num_updated_rows,
|
num_updated_rows: stats.num_updated_rows,
|
||||||
num_inserted_rows: stats.num_inserted_rows,
|
num_inserted_rows: stats.num_inserted_rows,
|
||||||
num_deleted_rows: stats.num_deleted_rows,
|
num_deleted_rows: stats.num_deleted_rows,
|
||||||
|
num_attempts: stats.num_attempts,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -2979,9 +2996,13 @@ mod tests {
|
|||||||
// Perform a "insert if not exists"
|
// Perform a "insert if not exists"
|
||||||
let mut merge_insert_builder = table.merge_insert(&["i"]);
|
let mut merge_insert_builder = table.merge_insert(&["i"]);
|
||||||
merge_insert_builder.when_not_matched_insert_all();
|
merge_insert_builder.when_not_matched_insert_all();
|
||||||
merge_insert_builder.execute(new_batches).await.unwrap();
|
let result = merge_insert_builder.execute(new_batches).await.unwrap();
|
||||||
// Only 5 rows should actually be inserted
|
// Only 5 rows should actually be inserted
|
||||||
assert_eq!(table.count_rows(None).await.unwrap(), 15);
|
assert_eq!(table.count_rows(None).await.unwrap(), 15);
|
||||||
|
assert_eq!(result.num_inserted_rows, 5);
|
||||||
|
assert_eq!(result.num_updated_rows, 0);
|
||||||
|
assert_eq!(result.num_deleted_rows, 0);
|
||||||
|
assert_eq!(result.num_attempts, 1);
|
||||||
|
|
||||||
// Create new data with i=15..25 (no id matches)
|
// Create new data with i=15..25 (no id matches)
|
||||||
let new_batches = Box::new(merge_insert_test_batches(15, 2));
|
let new_batches = Box::new(merge_insert_test_batches(15, 2));
|
||||||
@@ -4122,6 +4143,8 @@ mod tests {
|
|||||||
table.prewarm_index("text_idx").await.unwrap();
|
table.prewarm_index("text_idx").await.unwrap();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Windows does not support precise sleep durations due to timer resolution limitations.
|
||||||
|
#[cfg(not(target_os = "windows"))]
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn test_read_consistency_interval() {
|
async fn test_read_consistency_interval() {
|
||||||
let intervals = vec![
|
let intervals = vec![
|
||||||
|
|||||||
Reference in New Issue
Block a user