Bump version: 0.27.2-beta.2 → 0.27.2

Bump version: 0.27.2-beta.1 → 0.27.2-beta.2
Bump version: 0.30.2-beta.2 → 0.30.2
2026-04-01 13:30:40 +00:00 · 2026-03-31 21:26:04 +00:00 · 2026-03-31 21:25:36 +00:00 · 2026-03-31 21:25:04 +00:00 · 2026-03-31 21:25:02 +00:00 · 2026-03-31 13:29:17 -07:00
96 changed files with 7657 additions and 5238 deletions
--- a/.bumpversion.toml
+++ b/.bumpversion.toml
@@ -1,5 +1,5 @@
 [tool.bumpversion]
-current_version = "0.27.0-beta.3"
+current_version = "0.27.2"
 parse = """(?x)
    (?P<major>0|[1-9]\\d*)\\.
    (?P<minor>0|[1-9]\\d*)\\.
--- a/.github/workflows/build_linux_wheel/action.yml
+++ b/.github/workflows/build_linux_wheel/action.yml
@@ -23,8 +23,10 @@ runs:
  steps:
    - name: CONFIRM ARM BUILD
      shell: bash
      env:
        ARM_BUILD: ${{ inputs.arm-build }}
      run: |
-        echo "ARM BUILD: ${{ inputs.arm-build }}"
+        echo "ARM BUILD: $ARM_BUILD"
    - name: Build x86_64 Manylinux wheel
      if: ${{ inputs.arm-build == 'false' }}
      uses: PyO3/maturin-action@v1
--- a/.github/workflows/dev.yml
+++ b/.github/workflows/dev.yml
@@ -15,7 +15,7 @@ jobs:
    name: Label PR
    runs-on: ubuntu-latest
    steps:
-      - uses: srvaroa/labeler@master
+      - uses: srvaroa/labeler@v1
        env:
          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
  commitlint:
@@ -24,7 +24,7 @@ jobs:
    name: Verify PR title / description conforms to semantic-release
    runs-on: ubuntu-latest
    steps:
-      - uses: actions/setup-node@v3
+      - uses: actions/setup-node@v4
        with:
          node-version: "18"
      # These rules are disabled because Github will always ensure there
@@ -47,7 +47,7 @@ jobs:
            ${{ github.event.pull_request.body }}
      - if: failure()
-        uses: actions/github-script@v6
+        uses: actions/github-script@v7
        with:
          script: |
            const message = `**ACTION NEEDED**
--- a/.github/workflows/docs.yml
+++ b/.github/workflows/docs.yml
@@ -53,7 +53,7 @@ jobs:
          python -m pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -e .
          python -m pip install --extra-index-url https://pypi.fury.io/lance-format/ --extra-index-url https://pypi.fury.io/lancedb/ -r ../docs/requirements.txt
      - name: Set up node
-        uses: actions/setup-node@v3
+        uses: actions/setup-node@v4
        with:
          node-version: 20
          cache: 'npm'
@@ -68,7 +68,7 @@ jobs:
        run: |
          PYTHONPATH=. mkdocs build
      - name: Setup Pages
-        uses: actions/configure-pages@v2
+        uses: actions/configure-pages@v5
      - name: Upload artifact
        uses: actions/upload-pages-artifact@v3
        with:
--- a/.github/workflows/nodejs.yml
+++ b/.github/workflows/nodejs.yml
@@ -7,6 +7,7 @@ on:
  pull_request:
    paths:
      - Cargo.toml
      - Cargo.lock
      - nodejs/**
      - rust/**
      - docs/src/js/**
@@ -37,7 +38,7 @@ jobs:
      with:
        fetch-depth: 0
        lfs: true
-    - uses: actions/setup-node@v3
+    - uses: actions/setup-node@v4
      with:
        node-version: 20
        cache: 'npm'
@@ -77,7 +78,7 @@ jobs:
      with:
        fetch-depth: 0
        lfs: true
-    - uses: actions/setup-node@v3
+    - uses: actions/setup-node@v4
      name: Setup Node.js 20 for build
      with:
        # @napi-rs/cli v3 requires Node >= 20.12 (via @inquirer/prompts@8).
@@ -94,7 +95,7 @@ jobs:
      run: |
        npm ci --include=optional
        npm run build:debug -- --profile ci
-    - uses: actions/setup-node@v3
+    - uses: actions/setup-node@v4
      name: Setup Node.js ${{ matrix.node-version }} for test
      with:
        node-version: ${{ matrix.node-version }}
@@ -143,7 +144,7 @@ jobs:
      with:
        fetch-depth: 0
        lfs: true
-    - uses: actions/setup-node@v3
+    - uses: actions/setup-node@v4
      with:
        node-version: 20
        cache: 'npm'
--- a/.github/workflows/npm-publish.yml
+++ b/.github/workflows/npm-publish.yml
@@ -19,6 +19,7 @@ on:
    paths:
      - .github/workflows/npm-publish.yml
      - Cargo.toml # Change in dependency frequently breaks builds
      - Cargo.lock
 concurrency:
  group: ${{ github.workflow }}-${{ github.ref }}
@@ -124,7 +125,12 @@ jobs:
            pre_build: |-
              set -e &&
              apt-get update &&
-              apt-get install -y protobuf-compiler pkg-config
+              apt-get install -y protobuf-compiler pkg-config &&
              # The base image (manylinux2014-cross) sets TARGET_CC to the old
              # GCC 4.8 cross-compiler. aws-lc-sys checks TARGET_CC before CC,
              # so it picks up GCC even though the napi-rs image sets CC=clang.
              # Override to use the image's clang-18 which supports -fuse-ld=lld.
              export TARGET_CC=clang TARGET_CXX=clang++
          - target: x86_64-unknown-linux-musl
            # This one seems to need some extra memory
            host: ubuntu-2404-8x-x64
@@ -144,9 +150,10 @@ jobs:
              set -e &&
              apt-get update &&
              apt-get install -y protobuf-compiler pkg-config &&
-              # https://github.com/aws/aws-lc-rs/issues/737#issuecomment-2725918627
+              export TARGET_CC=clang TARGET_CXX=clang++ &&
-              ln -s /usr/aarch64-unknown-linux-gnu/lib/gcc/aarch64-unknown-linux-gnu/4.8.5/crtbeginS.o /usr/aarch64-unknown-linux-gnu/aarch64-unknown-linux-gnu/sysroot/usr/lib/crtbeginS.o &&
+              # The manylinux2014 sysroot has glibc 2.17 headers which lack
-              ln -s /usr/aarch64-unknown-linux-gnu/lib/gcc /usr/aarch64-unknown-linux-gnu/aarch64-unknown-linux-gnu/sysroot/usr/lib/gcc &&
+              # AT_HWCAP2 (added in Linux 3.17). Define it for aws-lc-sys.
              export CFLAGS="$CFLAGS -DAT_HWCAP2=26" &&
              rustup target add aarch64-unknown-linux-gnu
          - target: aarch64-unknown-linux-musl
            host: ubuntu-2404-8x-x64
@@ -266,7 +273,7 @@ jobs:
          - target: x86_64-unknown-linux-gnu
            host: ubuntu-latest
          - target: aarch64-unknown-linux-gnu
-            host: buildjet-16vcpu-ubuntu-2204-arm
+            host: ubuntu-2404-8x-arm64
        node:
          - '20'
    runs-on: ${{ matrix.settings.host }}
--- a/.github/workflows/pypi-publish.yml
+++ b/.github/workflows/pypi-publish.yml
@@ -9,6 +9,7 @@ on:
    paths:
      - .github/workflows/pypi-publish.yml
      - Cargo.toml # Change in dependency frequently breaks builds
      - Cargo.lock
 env:
  PIP_EXTRA_INDEX_URL: "https://pypi.fury.io/lance-format/ https://pypi.fury.io/lancedb/"
--- a/.github/workflows/python.yml
+++ b/.github/workflows/python.yml
@@ -7,6 +7,7 @@ on:
  pull_request:
    paths:
      - Cargo.toml
      - Cargo.lock
      - python/**
      - rust/**
      - .github/workflows/python.yml
--- a/.github/workflows/rust.yml
+++ b/.github/workflows/rust.yml
@@ -7,6 +7,7 @@ on:
  pull_request:
    paths:
      - Cargo.toml
      - Cargo.lock
      - rust/**
      - .github/workflows/rust.yml
@@ -206,14 +207,14 @@ jobs:
      - name: Downgrade  dependencies
        # These packages have newer requirements for MSRV
        run: |
-          cargo update -p aws-sdk-bedrockruntime --precise 1.64.0
+          cargo update -p aws-sdk-bedrockruntime --precise 1.77.0
-          cargo update -p aws-sdk-dynamodb --precise 1.55.0
+          cargo update -p aws-sdk-dynamodb --precise 1.68.0
-          cargo update -p aws-config --precise 1.5.10
+          cargo update -p aws-config --precise 1.6.0
-          cargo update -p aws-sdk-kms --precise 1.51.0
+          cargo update -p aws-sdk-kms --precise 1.63.0
-          cargo update -p aws-sdk-s3 --precise 1.65.0
+          cargo update -p aws-sdk-s3 --precise 1.79.0
-          cargo update -p aws-sdk-sso --precise 1.50.0
+          cargo update -p aws-sdk-sso --precise 1.62.0
-          cargo update -p aws-sdk-ssooidc --precise 1.51.0
+          cargo update -p aws-sdk-ssooidc --precise 1.63.0
-          cargo update -p aws-sdk-sts --precise 1.51.0
+          cargo update -p aws-sdk-sts --precise 1.63.0
          cargo update -p home --precise 0.5.9
      - name: cargo +${{ matrix.msrv }} check
        env:
--- a/Cargo.lock
+++ b/Cargo.lock
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -15,20 +15,20 @@ categories = ["database-implementations"]
 rust-version = "1.91.0"
 [workspace.dependencies]
-lance = { "version" = "=3.0.0-rc.2", default-features = false, "tag" = "v3.0.0-rc.2", "git" = "https://github.com/lance-format/lance.git" }
+lance = { version = "=4.0.0", default-features = false }
-lance-core = { "version" = "=3.0.0-rc.2", "tag" = "v3.0.0-rc.2", "git" = "https://github.com/lance-format/lance.git" }
+lance-core = { version = "=4.0.0" }
-lance-datagen = { "version" = "=3.0.0-rc.2", "tag" = "v3.0.0-rc.2", "git" = "https://github.com/lance-format/lance.git" }
+lance-datagen = { version = "=4.0.0" }
-lance-file = { "version" = "=3.0.0-rc.2", "tag" = "v3.0.0-rc.2", "git" = "https://github.com/lance-format/lance.git" }
+lance-file = { version = "=4.0.0" }
-lance-io = { "version" = "=3.0.0-rc.2", default-features = false, "tag" = "v3.0.0-rc.2", "git" = "https://github.com/lance-format/lance.git" }
+lance-io = { version = "=4.0.0", default-features = false }
-lance-index = { "version" = "=3.0.0-rc.2", "tag" = "v3.0.0-rc.2", "git" = "https://github.com/lance-format/lance.git" }
+lance-index = { version = "=4.0.0" }
-lance-linalg = { "version" = "=3.0.0-rc.2", "tag" = "v3.0.0-rc.2", "git" = "https://github.com/lance-format/lance.git" }
+lance-linalg = { version = "=4.0.0" }
-lance-namespace = { "version" = "=3.0.0-rc.2", "tag" = "v3.0.0-rc.2", "git" = "https://github.com/lance-format/lance.git" }
+lance-namespace = { version = "=4.0.0" }
-lance-namespace-impls = { "version" = "=3.0.0-rc.2", default-features = false, "tag" = "v3.0.0-rc.2", "git" = "https://github.com/lance-format/lance.git" }
+lance-namespace-impls = { version = "=4.0.0", default-features = false }
-lance-table = { "version" = "=3.0.0-rc.2", "tag" = "v3.0.0-rc.2", "git" = "https://github.com/lance-format/lance.git" }
+lance-table = { version = "=4.0.0" }
-lance-testing = { "version" = "=3.0.0-rc.2", "tag" = "v3.0.0-rc.2", "git" = "https://github.com/lance-format/lance.git" }
+lance-testing = { version = "=4.0.0" }
-lance-datafusion = { "version" = "=3.0.0-rc.2", "tag" = "v3.0.0-rc.2", "git" = "https://github.com/lance-format/lance.git" }
+lance-datafusion = { version = "=4.0.0" }
-lance-encoding = { "version" = "=3.0.0-rc.2", "tag" = "v3.0.0-rc.2", "git" = "https://github.com/lance-format/lance.git" }
+lance-encoding = { version = "=4.0.0" }
-lance-arrow = { "version" = "=3.0.0-rc.2", "tag" = "v3.0.0-rc.2", "git" = "https://github.com/lance-format/lance.git" }
+lance-arrow = { version = "=4.0.0" }
 ahash = "0.8"
 # Note that this one does not include pyarrow
 arrow = { version = "57.2", optional = false }
--- a/ci/check_lance_release.py
+++ b/ci/check_lance_release.py
@@ -3,6 +3,7 @@
 from __future__ import annotations
 import argparse
 import functools
 import json
 import os
 import re
@@ -26,6 +27,7 @@ SEMVER_RE = re.compile(
 )
@functools.total_ordering
@dataclass(frozen=True)
 class SemVer:
    major: int
@@ -156,7 +158,9 @@ def read_current_version(repo_root: Path) -> str:
 def determine_latest_tag(tags: Iterable[TagInfo]) -> TagInfo:
-    return max(tags, key=lambda tag: tag.semver)
+    # Stable releases (no prerelease) are always preferred over pre-releases.
    # Within each group, standard semver ordering applies.
    return max(tags, key=lambda tag: (not tag.semver.prerelease, tag.semver))
 def write_outputs(args: argparse.Namespace, payload: dict) -> None:
--- a/docker-compose.yml
+++ b/docker-compose.yml
@@ -1,7 +1,7 @@
 version: "3.9"
 services:
  localstack:
-    image: localstack/localstack:3.3
+    image: localstack/localstack:4.0
    ports:
      - 4566:4566
    environment:
--- a/dockerfiles/Dockerfile
+++ b/dockerfiles/Dockerfile
@@ -1,27 +1,27 @@
-#Simple base dockerfile that supports basic dependencies required to run lance with FTS and Hybrid Search
+# Simple base dockerfile that supports basic dependencies required to run lance with FTS and Hybrid Search
-#Usage docker build -t lancedb:latest -f Dockerfile .
+# Usage: docker build -t lancedb:latest -f Dockerfile .
-FROM python:3.10-slim-buster
+FROM python:3.12-slim-bookworm
-# Install Rust
+# Install build dependencies in a single layer
-RUN apt-get update && apt-get install -y curl build-essential && \
+RUN apt-get update && \
-  curl https://sh.rustup.rs -sSf | sh -s -- -y
+  apt-get install -y --no-install-recommends \
-
+    curl \
-# Set the environment variable for Rust
+    build-essential \
-ENV PATH="/root/.cargo/bin:${PATH}"
+    protobuf-compiler \
-
+    git \
-# Install protobuf compiler
+    ca-certificates && \
 RUN apt-get install -y protobuf-compiler && \
  apt-get clean && \
  rm -rf /var/lib/apt/lists/*
-RUN apt-get -y update &&\
+# Install Rust (pinned installer, non-interactive)
-  apt-get -y upgrade && \
+RUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y --default-toolchain stable --profile minimal
  apt-get -y install git
 # Set the environment variable for Rust
 ENV PATH="/root/.cargo/bin:${PATH}"
 # Verify installations
 RUN python --version && \
  rustc --version && \
  protoc --version
-RUN pip install tantivy lancedb
+RUN pip install --no-cache-dir tantivy lancedb
--- a/docs/requirements.txt
+++ b/docs/requirements.txt
@@ -1,9 +1,9 @@
-mkdocs==1.5.3
+mkdocs==1.6.1
 mkdocs-jupyter==0.24.1
-mkdocs-material==9.5.3
+mkdocs-material==9.6.23
-mkdocs-autorefs<=1.0
+mkdocs-autorefs>=0.5,<=1.0
-mkdocstrings[python]==0.25.2
+mkdocstrings[python]>=0.24,<1.0
-griffe
+griffe>=0.40,<1.0
-mkdocs-render-swagger-plugin
+mkdocs-render-swagger-plugin>=0.1.0
-pydantic
+pydantic>=2.0,<3.0
-mkdocs-redirects
+mkdocs-redirects>=1.2.0
--- a/docs/src/java/java.md
+++ b/docs/src/java/java.md
@@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`:
 <dependency>
    <groupId>com.lancedb</groupId>
    <artifactId>lancedb-core</artifactId>
-    <version>0.27.0-beta.3</version>
+    <version>0.27.2</version>
 </dependency>
 ```
--- a/docs/src/js/classes/Table.md
+++ b/docs/src/js/classes/Table.md
@@ -71,11 +71,12 @@ Add new columns with defined values.
 #### Parameters
-* **newColumnTransforms**: [`AddColumnsSql`](../interfaces/AddColumnsSql.md)[]
+* **newColumnTransforms**: `Field`&lt;`any`&gt; \| `Field`&lt;`any`&gt;[] \| `Schema`&lt;`any`&gt; \| [`AddColumnsSql`](../interfaces/AddColumnsSql.md)[]
-    pairs of column names and
+    Either:
-    the SQL expression to use to calculate the value of the new column. These
+    - An array of objects with column names and SQL expressions to calculate values
-    expressions will be evaluated for each row in the table, and can
+    - A single Arrow Field defining one column with its data type (column will be initialized with null values)
-    reference existing columns in the table.
+    - An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
    - An Arrow Schema defining columns with their data types (columns will be initialized with null values)
 #### Returns
@@ -484,19 +485,7 @@ Modeled after ``VACUUM`` in PostgreSQL.
 - Prune: Removes old versions of the dataset
 - Index: Optimizes the indices, adding new data to existing indices
- Experimental API
+ The frequency an application should call optimize is based on the frequency of
 ----------------
 The optimization process is undergoing active development and may change.
 Our goal with these changes is to improve the performance of optimization and
 reduce the complexity.
 That being said, it is essential today to run optimize if you want the best
 performance.  It should be stable and safe to use in production, but it our
 hope that the API may be simplified (or not even need to be called) in the
 future.
 The frequency an application shoudl call optimize is based on the frequency of
 data modifications.  If data is frequently added, deleted, or updated then
 optimize should be run frequently.  A good rule of thumb is to run optimize if
 you have added or modified 100,000 or more records or run more than 20 data
--- a/docs/src/js/interfaces/OptimizeOptions.md
+++ b/docs/src/js/interfaces/OptimizeOptions.md
@@ -37,3 +37,12 @@ tbl.optimize({cleanupOlderThan: new Date()});
 ```ts
 deleteUnverified: boolean;
 ```
 Because they may be part of an in-progress transaction, files newer than
 7 days old are not deleted by default. If you are sure that there are no
 in-progress transactions, then you can set this to true to delete all
 files older than `cleanupOlderThan`.
 **WARNING**: This should only be set to true if you can guarantee that
 no other process is currently working on this dataset. Otherwise the
 dataset could be put into a corrupted state.
--- a/docs/src/js/namespaces/embedding/classes/EmbeddingFunction.md
+++ b/docs/src/js/namespaces/embedding/classes/EmbeddingFunction.md
@@ -52,7 +52,7 @@ new EmbeddingFunction<T, M>(): EmbeddingFunction<T, M>
 ### computeQueryEmbeddings()
 ```ts
-computeQueryEmbeddings(data): Promise<number[] | Float32Array | Float64Array>
+computeQueryEmbeddings(data): Promise<number[] | Uint8Array | Float32Array | Float64Array>
 ```
 Compute the embeddings for a single query
@@ -63,7 +63,7 @@ Compute the embeddings for a single query
 #### Returns
-`Promise`&lt;`number`[] \| `Float32Array` \| `Float64Array`&gt;
+`Promise`&lt;`number`[] \| `Uint8Array` \| `Float32Array` \| `Float64Array`&gt;
 ***
--- a/docs/src/js/namespaces/embedding/classes/TextEmbeddingFunction.md
+++ b/docs/src/js/namespaces/embedding/classes/TextEmbeddingFunction.md
@@ -37,7 +37,7 @@ new TextEmbeddingFunction<M>(): TextEmbeddingFunction<M>
 ### computeQueryEmbeddings()
 ```ts
-computeQueryEmbeddings(data): Promise<number[] | Float32Array | Float64Array>
+computeQueryEmbeddings(data): Promise<number[] | Uint8Array | Float32Array | Float64Array>
 ```
 Compute the embeddings for a single query
@@ -48,7 +48,7 @@ Compute the embeddings for a single query
 #### Returns
-`Promise`&lt;`number`[] \| `Float32Array` \| `Float64Array`&gt;
+`Promise`&lt;`number`[] \| `Uint8Array` \| `Float32Array` \| `Float64Array`&gt;
 #### Overrides
--- a/docs/src/js/type-aliases/IntoVector.md
+++ b/docs/src/js/type-aliases/IntoVector.md
@@ -7,5 +7,10 @@
 # Type Alias: IntoVector
 ```ts
-type IntoVector: Float32Array | Float64Array | number[] | Promise<Float32Array | Float64Array | number[]>;
+type IntoVector:
  | Float32Array
  | Float64Array
  | Uint8Array
  | number[]
  | Promise<Float32Array | Float64Array | Uint8Array | number[]>;
 ```
--- a/docs/src/python/python.md
+++ b/docs/src/python/python.md
@@ -36,6 +36,20 @@ is also an [asynchronous API client](#connections-asynchronous).
 ::: lancedb.table.Tags
 ## Expressions
 Type-safe expression builder for filters and projections. Use these instead
 of raw SQL strings with [where][lancedb.query.LanceQueryBuilder.where] and
 [select][lancedb.query.LanceQueryBuilder.select].
 ::: lancedb.expr.Expr
 ::: lancedb.expr.col
 ::: lancedb.expr.lit
 ::: lancedb.expr.func
 ## Querying (Synchronous)
 ::: lancedb.query.Query
--- a/java/README.md
+++ b/java/README.md
@@ -1,4 +1,4 @@
-# LanceDB Java SDK
+# LanceDB Java Enterprise Client
 ## Configuration and Initialization
--- a/java/lancedb-core/pom.xml
+++ b/java/lancedb-core/pom.xml
@@ -8,7 +8,7 @@
    <parent>
      <groupId>com.lancedb</groupId>
      <artifactId>lancedb-parent</artifactId>
-      <version>0.27.0-beta.3</version>
+      <version>0.27.2-final.0</version>
      <relativePath>../pom.xml</relativePath>
    </parent>
@@ -56,21 +56,21 @@
        <dependency>
            <groupId>org.apache.logging.log4j</groupId>
            <artifactId>log4j-slf4j2-impl</artifactId>
-            <version>2.24.3</version>
+            <version>2.25.3</version>
            <scope>test</scope>
        </dependency>
        <dependency>
            <groupId>org.apache.logging.log4j</groupId>
            <artifactId>log4j-core</artifactId>
-            <version>2.24.3</version>
+            <version>2.25.3</version>
            <scope>test</scope>
        </dependency>
        <dependency>
            <groupId>org.apache.logging.log4j</groupId>
            <artifactId>log4j-api</artifactId>
-            <version>2.24.3</version>
+            <version>2.25.3</version>
            <scope>test</scope>
        </dependency>
    </dependencies>
--- a/java/pom.xml
+++ b/java/pom.xml
@@ -6,7 +6,7 @@
    <groupId>com.lancedb</groupId>
    <artifactId>lancedb-parent</artifactId>
-    <version>0.27.0-beta.3</version>
+    <version>0.27.2-final.0</version>
    <packaging>pom</packaging>
    <name>${project.artifactId}</name>
    <description>LanceDB Java SDK Parent POM</description>
@@ -28,7 +28,7 @@
    <properties>
        <project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
        <arrow.version>15.0.0</arrow.version>
-        <lance-core.version>3.1.0-beta.2</lance-core.version>
+        <lance-core.version>3.0.1</lance-core.version>
        <spotless.skip>false</spotless.skip>
        <spotless.version>2.30.0</spotless.version>
        <spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
@@ -111,7 +111,7 @@
            <plugin>
                <groupId>org.apache.maven.plugins</groupId>
                <artifactId>maven-source-plugin</artifactId>
-                <version>2.2.1</version>
+                <version>3.3.1</version>
                <executions>
                    <execution>
                        <id>attach-sources</id>
@@ -124,7 +124,7 @@
            <plugin>
                <groupId>org.apache.maven.plugins</groupId>
                <artifactId>maven-javadoc-plugin</artifactId>
-                <version>2.9.1</version>
+                <version>3.11.2</version>
                <executions>
                    <execution>
                        <id>attach-javadocs</id>
@@ -178,15 +178,15 @@
            <plugins>
                <plugin>
                    <artifactId>maven-clean-plugin</artifactId>
-                    <version>3.1.0</version>
+                    <version>3.4.1</version>
                </plugin>
                <plugin>
                    <artifactId>maven-resources-plugin</artifactId>
-                    <version>3.0.2</version>
+                    <version>3.3.1</version>
                </plugin>
                <plugin>
                    <artifactId>maven-compiler-plugin</artifactId>
-                    <version>3.8.1</version>
+                    <version>3.14.0</version>
                    <configuration>
                        <compilerArgs>
                            <arg>-h</arg>
@@ -205,11 +205,11 @@
                </plugin>
                <plugin>
                    <artifactId>maven-jar-plugin</artifactId>
-                    <version>3.0.2</version>
+                    <version>3.4.2</version>
                </plugin>
                <plugin>
                    <artifactId>maven-install-plugin</artifactId>
-                    <version>2.5.2</version>
+                    <version>3.1.3</version>
                </plugin>
                <plugin>
                    <groupId>com.diffplug.spotless</groupId>
@@ -327,7 +327,7 @@
                    <plugin>
                        <groupId>org.apache.maven.plugins</groupId>
                        <artifactId>maven-gpg-plugin</artifactId>
-                        <version>1.5</version>
+                        <version>3.2.7</version>
                        <executions>
                            <execution>
                                <id>sign-artifacts</id>
--- a/nodejs/Cargo.toml
+++ b/nodejs/Cargo.toml
@@ -1,7 +1,7 @@
 [package]
 name = "lancedb-nodejs"
 edition.workspace = true
-version = "0.27.0-beta.3"
+version = "0.27.2"
 license.workspace = true
 description.workspace = true
 repository.workspace = true
@@ -15,6 +15,8 @@ crate-type = ["cdylib"]
 async-trait.workspace = true
 arrow-ipc.workspace = true
 arrow-array.workspace = true
 arrow-buffer = "57.2"
 half.workspace = true
 arrow-schema.workspace = true
 env_logger.workspace = true
 futures.workspace = true
@@ -25,12 +27,12 @@ napi = { version = "3.8.3", default-features = false, features = [
 ] }
 napi-derive = "3.5.2"
 # Prevent dynamic linking of lzma, which comes from datafusion
-lzma-sys = { version = "*", features = ["static"] }
+lzma-sys = { version = "0.1", features = ["static"] }
 log.workspace = true
-# Workaround for build failure until we can fix it.
+# Pin to resolve build failures; update periodically for security patches.
-aws-lc-sys = "=0.28.0"
+aws-lc-sys = "=0.38.0"
-aws-lc-rs = "=1.13.0"
+aws-lc-rs = "=1.16.1"
 [build-dependencies]
 napi-build = "2.3.1"
--- a/nodejs/test/arrow.test.ts
+++ b/nodejs/test/arrow.test.ts
@@ -63,6 +63,7 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
      tableFromIPC,
      DataType,
      Dictionary,
      Uint8: ArrowUint8,
      // biome-ignore lint/suspicious/noExplicitAny: <explanation>
    } = <any>arrow;
    type Schema = ApacheArrow["Schema"];
@@ -362,6 +363,38 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
        ).toEqual(new Float64().toString());
      });
      it("will infer FixedSizeList<Float32> from Float32Array values", async function () {
        const table = makeArrowTable([
          { id: "a", vector: new Float32Array([0.1, 0.2, 0.3]) },
          { id: "b", vector: new Float32Array([0.4, 0.5, 0.6]) },
        ]);
        expect(DataType.isFixedSizeList(table.getChild("vector")?.type)).toBe(
          true,
        );
        const vectorType = table.getChild("vector")?.type;
        expect(vectorType.listSize).toBe(3);
        expect(vectorType.children[0].type.toString()).toEqual(
          new Float32().toString(),
        );
      });
      it("will infer FixedSizeList<Uint8> from Uint8Array values", async function () {
        const table = makeArrowTable([
          { id: "a", vector: new Uint8Array([1, 2, 3]) },
          { id: "b", vector: new Uint8Array([4, 5, 6]) },
        ]);
        expect(DataType.isFixedSizeList(table.getChild("vector")?.type)).toBe(
          true,
        );
        const vectorType = table.getChild("vector")?.type;
        expect(vectorType.listSize).toBe(3);
        expect(vectorType.children[0].type.toString()).toEqual(
          new ArrowUint8().toString(),
        );
      });
      it("will use dictionary encoded strings if asked", async function () {
        const table = makeArrowTable([{ str: "hello" }]);
        expect(DataType.isUtf8(table.getChild("str")?.type)).toBe(true);
--- a/nodejs/test/table.test.ts
+++ b/nodejs/test/table.test.ts
@@ -1259,6 +1259,98 @@ describe("schema evolution", function () {
    expect(await table.schema()).toEqual(expectedSchema);
  });
  it("can add columns with schema for explicit data types", async function () {
    const con = await connect(tmpDir.name);
    const table = await con.createTable("vectors", [
      { id: 1n, vector: [0.1, 0.2] },
    ]);
    // Define schema for new columns with explicit data types
    // Note: All columns must be nullable when using addColumns with Schema
    // because they are initially populated with null values
    const newColumnsSchema = new Schema([
      new Field("price", new Float64(), true),
      new Field("category", new Utf8(), true),
      new Field("rating", new Int32(), true),
    ]);
    const result = await table.addColumns(newColumnsSchema);
    expect(result).toHaveProperty("version");
    expect(result.version).toBe(2);
    const expectedSchema = new Schema([
      new Field("id", new Int64(), true),
      new Field(
        "vector",
        new FixedSizeList(2, new Field("item", new Float32(), true)),
        true,
      ),
      new Field("price", new Float64(), true),
      new Field("category", new Utf8(), true),
      new Field("rating", new Int32(), true),
    ]);
    expect(await table.schema()).toEqual(expectedSchema);
    // Verify that new columns are populated with null values
    const results = await table.query().toArray();
    expect(results).toHaveLength(1);
    expect(results[0].price).toBeNull();
    expect(results[0].category).toBeNull();
    expect(results[0].rating).toBeNull();
  });
  it("can add a single column using Field", async function () {
    const con = await connect(tmpDir.name);
    const table = await con.createTable("vectors", [
      { id: 1n, vector: [0.1, 0.2] },
    ]);
    // Add a single field
    const priceField = new Field("price", new Float64(), true);
    const result = await table.addColumns(priceField);
    expect(result).toHaveProperty("version");
    expect(result.version).toBe(2);
    const expectedSchema = new Schema([
      new Field("id", new Int64(), true),
      new Field(
        "vector",
        new FixedSizeList(2, new Field("item", new Float32(), true)),
        true,
      ),
      new Field("price", new Float64(), true),
    ]);
    expect(await table.schema()).toEqual(expectedSchema);
  });
  it("can add multiple columns using array of Fields", async function () {
    const con = await connect(tmpDir.name);
    const table = await con.createTable("vectors", [
      { id: 1n, vector: [0.1, 0.2] },
    ]);
    // Add multiple fields as array
    const fields = [
      new Field("price", new Float64(), true),
      new Field("category", new Utf8(), true),
    ];
    const result = await table.addColumns(fields);
    expect(result).toHaveProperty("version");
    expect(result.version).toBe(2);
    const expectedSchema = new Schema([
      new Field("id", new Int64(), true),
      new Field(
        "vector",
        new FixedSizeList(2, new Field("item", new Float32(), true)),
        true,
      ),
      new Field("price", new Float64(), true),
      new Field("category", new Utf8(), true),
    ]);
    expect(await table.schema()).toEqual(expectedSchema);
  });
  it("can alter the columns in the schema", async function () {
    const con = await connect(tmpDir.name);
    const schema = new Schema([
@@ -2204,3 +2296,36 @@ describe("when creating an empty table", () => {
    expect((actualSchema.fields[1].type as Float64).precision).toBe(2);
  });
 });
 // Ensure we can create float32 arrays without using Arrow
 // by utilizing native JS TypedArray support
 //
 // https://github.com/lancedb/lancedb/issues/3115
 describe("when creating a table with Float32Array vectors", () => {
  let tmpDir: tmp.DirResult;
  beforeEach(() => {
    tmpDir = tmp.dirSync({ unsafeCleanup: true });
  });
  afterEach(() => {
    tmpDir.removeCallback();
  });
  it("should persist Float32Array as FixedSizeList<Float32> in the LanceDB schema", async () => {
    const db = await connect(tmpDir.name);
    const table = await db.createTable("test", [
      { id: "a", vector: new Float32Array([0.1, 0.2, 0.3]) },
      { id: "b", vector: new Float32Array([0.4, 0.5, 0.6]) },
    ]);
    const schema = await table.schema();
    const vectorField = schema.fields.find((f) => f.name === "vector");
    expect(vectorField).toBeDefined();
    expect(vectorField!.type).toBeInstanceOf(FixedSizeList);
    const fsl = vectorField!.type as FixedSizeList;
    expect(fsl.listSize).toBe(3);
    expect(fsl.children[0].type.typeId).toBe(Type.Float);
    // precision: HALF=0, SINGLE=1, DOUBLE=2
    expect((fsl.children[0].type as Float32).precision).toBe(1);
  });
 });
--- a/nodejs/test/vector_types.test.ts
+++ b/nodejs/test/vector_types.test.ts
@@ -0,0 +1,110 @@
 // SPDX-License-Identifier: Apache-2.0
 // SPDX-FileCopyrightText: Copyright The LanceDB Authors
 import * as tmp from "tmp";
 import { type Table, connect } from "../lancedb";
 import {
  Field,
  FixedSizeList,
  Float32,
  Int64,
  Schema,
  makeArrowTable,
 } from "../lancedb/arrow";
 describe("Vector query with different typed arrays", () => {
  let tmpDir: tmp.DirResult;
  afterEach(() => {
    tmpDir?.removeCallback();
  });
  async function createFloat32Table(): Promise<Table> {
    tmpDir = tmp.dirSync({ unsafeCleanup: true });
    const db = await connect(tmpDir.name);
    const schema = new Schema([
      new Field("id", new Int64(), true),
      new Field(
        "vec",
        new FixedSizeList(2, new Field("item", new Float32())),
        true,
      ),
    ]);
    const data = makeArrowTable(
      [
        { id: 1n, vec: [1.0, 0.0] },
        { id: 2n, vec: [0.0, 1.0] },
        { id: 3n, vec: [1.0, 1.0] },
      ],
      { schema },
    );
    return db.createTable("test_f32", data);
  }
  it("should search with Float32Array (baseline)", async () => {
    const table = await createFloat32Table();
    const results = await table
      .query()
      .nearestTo(new Float32Array([1.0, 0.0]))
      .limit(1)
      .toArray();
    expect(results.length).toBe(1);
    expect(Number(results[0].id)).toBe(1);
  });
  it("should search with number[] (backward compat)", async () => {
    const table = await createFloat32Table();
    const results = await table
      .query()
      .nearestTo([1.0, 0.0])
      .limit(1)
      .toArray();
    expect(results.length).toBe(1);
    expect(Number(results[0].id)).toBe(1);
  });
  it("should search with Float64Array via raw path", async () => {
    const table = await createFloat32Table();
    const results = await table
      .query()
      .nearestTo(new Float64Array([1.0, 0.0]))
      .limit(1)
      .toArray();
    expect(results.length).toBe(1);
    expect(Number(results[0].id)).toBe(1);
  });
  it("should add multiple query vectors with Float64Array", async () => {
    const table = await createFloat32Table();
    const results = await table
      .query()
      .nearestTo(new Float64Array([1.0, 0.0]))
      .addQueryVector(new Float64Array([0.0, 1.0]))
      .limit(2)
      .toArray();
    expect(results.length).toBeGreaterThanOrEqual(2);
  });
  // Float16Array is only available in Node 22+; not in TypeScript's standard lib yet
  const float16ArrayCtor = (globalThis as unknown as Record<string, unknown>)
    .Float16Array as (new (values: number[]) => unknown) | undefined;
  const hasFloat16 = float16ArrayCtor !== undefined;
  const f16it = hasFloat16 ? it : it.skip;
  f16it("should search with Float16Array via raw path", async () => {
    const table = await createFloat32Table();
    const results = await table
      .query()
      .nearestTo(new float16ArrayCtor!([1.0, 0.0]) as Float32Array)
      .limit(1)
      .toArray();
    expect(results.length).toBe(1);
    expect(Number(results[0].id)).toBe(1);
  });
 });
--- a/nodejs/examples/package-lock.json
+++ b/nodejs/examples/package-lock.json
@@ -30,12 +30,15 @@
        "x64",
        "arm64"
      ],
      "dev": true,
      "license": "Apache-2.0",
      "optional": true,
      "os": [
        "darwin",
        "linux",
        "win32"
      ],
      "peer": true,
      "dependencies": {
        "reflect-metadata": "^0.2.2"
      },
@@ -91,14 +94,15 @@
      }
    },
    "node_modules/@babel/code-frame": {
-      "version": "7.26.2",
+      "version": "7.29.0",
-      "resolved": "https://registry.npmjs.org/@babel/code-frame/-/code-frame-7.26.2.tgz",
+      "resolved": "https://registry.npmjs.org/@babel/code-frame/-/code-frame-7.29.0.tgz",
-      "integrity": "sha512-RJlIHRueQgwWitWgF8OdFYGZX328Ax5BCemNGlqHfplnRT9ESi8JkFlvaVYbS+UubVY6dpv87Fs2u5M29iNFVQ==",
+      "integrity": "sha512-9NhCeYjq9+3uxgdtp20LSiJXJvN0FeCtNGpJxuMFZ1Kv3cWUNb6DOhJwUvcVCzKGR66cw4njwM6hrJLqgOwbcw==",
      "dev": true,
      "license": "MIT",
      "dependencies": {
-        "@babel/helper-validator-identifier": "^7.25.9",
+        "@babel/helper-validator-identifier": "^7.28.5",
        "js-tokens": "^4.0.0",
-        "picocolors": "^1.0.0"
+        "picocolors": "^1.1.1"
      },
      "engines": {
        "node": ">=6.9.0"
@@ -233,19 +237,21 @@
      }
    },
    "node_modules/@babel/helper-string-parser": {
-      "version": "7.25.9",
+      "version": "7.27.1",
-      "resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-7.25.9.tgz",
+      "resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-7.27.1.tgz",
-      "integrity": "sha512-4A/SCr/2KLd5jrtOMFzaKjVtAei3+2r/NChoBNoZ3EyP/+GlhoaEGoWOZUmFmoITP7zOJyHIMm+DYRd8o3PvHA==",
+      "integrity": "sha512-qMlSxKbpRlAridDExk92nSobyDdpPijUq2DW6oDnUqd0iOGxmQjyqhMIihI9+zv4LPyZdRje2cavWPbCbWm3eA==",
      "dev": true,
      "license": "MIT",
      "engines": {
        "node": ">=6.9.0"
      }
    },
    "node_modules/@babel/helper-validator-identifier": {
-      "version": "7.25.9",
+      "version": "7.28.5",
-      "resolved": "https://registry.npmjs.org/@babel/helper-validator-identifier/-/helper-validator-identifier-7.25.9.tgz",
+      "resolved": "https://registry.npmjs.org/@babel/helper-validator-identifier/-/helper-validator-identifier-7.28.5.tgz",
-      "integrity": "sha512-Ed61U6XJc3CVRfkERJWDz4dJwKe7iLmmJsbOGu9wSloNSFttHV0I8g6UAgb7qnK5ly5bGLPd4oXZlxCdANBOWQ==",
+      "integrity": "sha512-qSs4ifwzKJSV39ucNjsvc6WVHs6b7S03sOh2OcHF9UHfVPqWWALUsNUVzhSBiItjRZoLHx7nIarVjqKVusUZ1Q==",
      "dev": true,
      "license": "MIT",
      "engines": {
        "node": ">=6.9.0"
      }
@@ -260,25 +266,27 @@
      }
    },
    "node_modules/@babel/helpers": {
-      "version": "7.26.0",
+      "version": "7.28.6",
-      "resolved": "https://registry.npmjs.org/@babel/helpers/-/helpers-7.26.0.tgz",
+      "resolved": "https://registry.npmjs.org/@babel/helpers/-/helpers-7.28.6.tgz",
-      "integrity": "sha512-tbhNuIxNcVb21pInl3ZSjksLCvgdZy9KwJ8brv993QtIVKJBBkYXz4q4ZbAv31GdnC+R90np23L5FbEBlthAEw==",
+      "integrity": "sha512-xOBvwq86HHdB7WUDTfKfT/Vuxh7gElQ+Sfti2Cy6yIWNW05P8iUslOVcZ4/sKbE+/jQaukQAdz/gf3724kYdqw==",
      "dev": true,
      "license": "MIT",
      "dependencies": {
-        "@babel/template": "^7.25.9",
+        "@babel/template": "^7.28.6",
-        "@babel/types": "^7.26.0"
+        "@babel/types": "^7.28.6"
      },
      "engines": {
        "node": ">=6.9.0"
      }
    },
    "node_modules/@babel/parser": {
-      "version": "7.26.2",
+      "version": "7.29.0",
-      "resolved": "https://registry.npmjs.org/@babel/parser/-/parser-7.26.2.tgz",
+      "resolved": "https://registry.npmjs.org/@babel/parser/-/parser-7.29.0.tgz",
-      "integrity": "sha512-DWMCZH9WA4Maitz2q21SRKHo9QXZxkDsbNZoVD62gusNtNBBqDg9i7uOhASfTfIGNzW+O+r7+jAlM8dwphcJKQ==",
+      "integrity": "sha512-IyDgFV5GeDUVX4YdF/3CPULtVGSXXMLh1xVIgdCgxApktqnQV0r7/8Nqthg+8YLGaAtdyIlo2qIdZrbCv4+7ww==",
      "dev": true,
      "license": "MIT",
      "dependencies": {
-        "@babel/types": "^7.26.0"
+        "@babel/types": "^7.29.0"
      },
      "bin": {
        "parser": "bin/babel-parser.js"
@@ -510,14 +518,15 @@
      }
    },
    "node_modules/@babel/template": {
-      "version": "7.25.9",
+      "version": "7.28.6",
-      "resolved": "https://registry.npmjs.org/@babel/template/-/template-7.25.9.tgz",
+      "resolved": "https://registry.npmjs.org/@babel/template/-/template-7.28.6.tgz",
-      "integrity": "sha512-9DGttpmPvIxBb/2uwpVo3dqJ+O6RooAFOS+lB+xDqoE2PVCE8nfoHMdZLpfCQRLwvohzXISPZcgxt80xLfsuwg==",
+      "integrity": "sha512-YA6Ma2KsCdGb+WC6UpBVFJGXL58MDA6oyONbjyF/+5sBgxY/dwkhLogbMT2GXXyU84/IhRw/2D1Os1B/giz+BQ==",
      "dev": true,
      "license": "MIT",
      "dependencies": {
-        "@babel/code-frame": "^7.25.9",
+        "@babel/code-frame": "^7.28.6",
-        "@babel/parser": "^7.25.9",
+        "@babel/parser": "^7.28.6",
-        "@babel/types": "^7.25.9"
+        "@babel/types": "^7.28.6"
      },
      "engines": {
        "node": ">=6.9.0"
@@ -542,13 +551,14 @@
      }
    },
    "node_modules/@babel/types": {
-      "version": "7.26.0",
+      "version": "7.29.0",
-      "resolved": "https://registry.npmjs.org/@babel/types/-/types-7.26.0.tgz",
+      "resolved": "https://registry.npmjs.org/@babel/types/-/types-7.29.0.tgz",
-      "integrity": "sha512-Z/yiTPj+lDVnF7lWeKCIJzaIkI0vYO87dMpZ4bg4TDrFe4XXLFWL1TbXU27gBP3QccxV9mZICCrnjnYlJjXHOA==",
+      "integrity": "sha512-LwdZHpScM4Qz8Xw2iKSzS+cfglZzJGvofQICy7W7v4caru4EaAmyUuO6BGrbyQ2mYV11W0U8j5mBhd14dd3B0A==",
      "dev": true,
      "license": "MIT",
      "dependencies": {
-        "@babel/helper-string-parser": "^7.25.9",
+        "@babel/helper-string-parser": "^7.27.1",
-        "@babel/helper-validator-identifier": "^7.25.9"
+        "@babel/helper-validator-identifier": "^7.28.5"
      },
      "engines": {
        "node": ">=6.9.0"
@@ -1151,95 +1161,6 @@
        "url": "https://opencollective.com/libvips"
      }
    },
    "node_modules/@isaacs/cliui": {
      "version": "8.0.2",
      "resolved": "https://registry.npmjs.org/@isaacs/cliui/-/cliui-8.0.2.tgz",
      "integrity": "sha512-O8jcjabXaleOG9DQ0+ARXWZBTfnP4WNAqzuiJK7ll44AmxGKv/J2M4TPjxjY3znBCfvBXFzucm1twdyFybFqEA==",
      "dependencies": {
        "string-width": "^5.1.2",
        "string-width-cjs": "npm:string-width@^4.2.0",
        "strip-ansi": "^7.0.1",
        "strip-ansi-cjs": "npm:strip-ansi@^6.0.1",
        "wrap-ansi": "^8.1.0",
        "wrap-ansi-cjs": "npm:wrap-ansi@^7.0.0"
      },
      "engines": {
        "node": ">=12"
      }
    },
    "node_modules/@isaacs/cliui/node_modules/ansi-regex": {
      "version": "6.1.0",
      "resolved": "https://registry.npmjs.org/ansi-regex/-/ansi-regex-6.1.0.tgz",
      "integrity": "sha512-7HSX4QQb4CspciLpVFwyRe79O3xsIZDDLER21kERQ71oaPodF8jL725AgJMFAYbooIqolJoRLuM81SpeUkpkvA==",
      "engines": {
        "node": ">=12"
      },
      "funding": {
        "url": "https://github.com/chalk/ansi-regex?sponsor=1"
      }
    },
    "node_modules/@isaacs/cliui/node_modules/ansi-styles": {
      "version": "6.2.1",
      "resolved": "https://registry.npmjs.org/ansi-styles/-/ansi-styles-6.2.1.tgz",
      "integrity": "sha512-bN798gFfQX+viw3R7yrGWRqnrN2oRkEkUjjl4JNn4E8GxxbjtG3FbrEIIY3l8/hrwUwIeCZvi4QuOTP4MErVug==",
      "engines": {
        "node": ">=12"
      },
      "funding": {
        "url": "https://github.com/chalk/ansi-styles?sponsor=1"
      }
    },
    "node_modules/@isaacs/cliui/node_modules/emoji-regex": {
      "version": "9.2.2",
      "resolved": "https://registry.npmjs.org/emoji-regex/-/emoji-regex-9.2.2.tgz",
      "integrity": "sha512-L18DaJsXSUk2+42pv8mLs5jJT2hqFkFE4j21wOmgbUqsZ2hL72NsUU785g9RXgo3s0ZNgVl42TiHp3ZtOv/Vyg=="
    },
    "node_modules/@isaacs/cliui/node_modules/string-width": {
      "version": "5.1.2",
      "resolved": "https://registry.npmjs.org/string-width/-/string-width-5.1.2.tgz",
      "integrity": "sha512-HnLOCR3vjcY8beoNLtcjZ5/nxn2afmME6lhrDrebokqMap+XbeW8n9TXpPDOqdGK5qcI3oT0GKTW6wC7EMiVqA==",
      "dependencies": {
        "eastasianwidth": "^0.2.0",
        "emoji-regex": "^9.2.2",
        "strip-ansi": "^7.0.1"
      },
      "engines": {
        "node": ">=12"
      },
      "funding": {
        "url": "https://github.com/sponsors/sindresorhus"
      }
    },
    "node_modules/@isaacs/cliui/node_modules/strip-ansi": {
      "version": "7.1.0",
      "resolved": "https://registry.npmjs.org/strip-ansi/-/strip-ansi-7.1.0.tgz",
      "integrity": "sha512-iq6eVVI64nQQTRYq2KtEg2d2uU7LElhTJwsH4YzIHZshxlgZms/wIc4VoDQTlG/IvVIrBKG06CrZnp0qv7hkcQ==",
      "dependencies": {
        "ansi-regex": "^6.0.1"
      },
      "engines": {
        "node": ">=12"
      },
      "funding": {
        "url": "https://github.com/chalk/strip-ansi?sponsor=1"
      }
    },
    "node_modules/@isaacs/cliui/node_modules/wrap-ansi": {
      "version": "8.1.0",
      "resolved": "https://registry.npmjs.org/wrap-ansi/-/wrap-ansi-8.1.0.tgz",
      "integrity": "sha512-si7QWI6zUMq56bESFvagtmzMdGOtoxfR+Sez11Mobfc7tm+VkUckk9bW2UeffTGVUbOksxmSw0AA2gs8g71NCQ==",
      "dependencies": {
        "ansi-styles": "^6.1.0",
        "string-width": "^5.0.1",
        "strip-ansi": "^7.0.1"
      },
      "engines": {
        "node": ">=12"
      },
      "funding": {
        "url": "https://github.com/chalk/wrap-ansi?sponsor=1"
      }
    },
    "node_modules/@isaacs/fs-minipass": {
      "version": "4.0.1",
      "resolved": "https://registry.npmjs.org/@isaacs/fs-minipass/-/fs-minipass-4.0.1.tgz",
@@ -1606,15 +1527,6 @@
      "resolved": "../dist",
      "link": true
    },
    "node_modules/@pkgjs/parseargs": {
      "version": "0.11.0",
      "resolved": "https://registry.npmjs.org/@pkgjs/parseargs/-/parseargs-0.11.0.tgz",
      "integrity": "sha512-+1VkjdD0QBLPodGrJUeqarH8VAIvQODIbwh9XpP5Syisf7YoQgsJKPNFoqqLQlu+VQ/tVSshMR6loPMn8U+dPg==",
      "optional": true,
      "engines": {
        "node": ">=14"
      }
    },
    "node_modules/@protobufjs/aspromise": {
      "version": "1.1.2",
      "resolved": "https://registry.npmjs.org/@protobufjs/aspromise/-/aspromise-1.1.2.tgz",
@@ -1846,6 +1758,7 @@
      "version": "5.0.1",
      "resolved": "https://registry.npmjs.org/ansi-regex/-/ansi-regex-5.0.1.tgz",
      "integrity": "sha512-quJQXlTSUGL2LH9SUXo8VwsY4soanhgo6LNSm84E1LBcE8s3O0wpdiRzyR9z/ZZJMlMWv37qOOb9pdJlMUEKFQ==",
      "dev": true,
      "engines": {
        "node": ">=8"
      }
@@ -1854,6 +1767,7 @@
      "version": "4.3.0",
      "resolved": "https://registry.npmjs.org/ansi-styles/-/ansi-styles-4.3.0.tgz",
      "integrity": "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg==",
      "dev": true,
      "dependencies": {
        "color-convert": "^2.0.1"
      },
@@ -2019,13 +1933,15 @@
    "node_modules/balanced-match": {
      "version": "1.0.2",
      "resolved": "https://registry.npmjs.org/balanced-match/-/balanced-match-1.0.2.tgz",
-      "integrity": "sha512-3oSeUO0TMV67hN1AmbXsK4yaqU7tjiHlbxRDZOpH0KW9+CeX4bRAaX0Anxt0tx2MrpRpWwQaPwIlISEJhYU5Pw=="
+      "integrity": "sha512-3oSeUO0TMV67hN1AmbXsK4yaqU7tjiHlbxRDZOpH0KW9+CeX4bRAaX0Anxt0tx2MrpRpWwQaPwIlISEJhYU5Pw==",
      "dev": true
    },
    "node_modules/brace-expansion": {
-      "version": "1.1.11",
+      "version": "1.1.12",
-      "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.11.tgz",
+      "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.12.tgz",
-      "integrity": "sha512-iCuPHDFgrHX7H2vEI/5xpz07zSHB00TpugqhmYtVmMO6518mCuRMoOYFldEBl0g187ufozdaHgWKcYFb61qGiA==",
+      "integrity": "sha512-9T9UjW3r0UW5c1Q7GTwllptXwhvYmEzFhzMfZ9H7FQWt+uZePjZPjBP/W1ZEyZ1twGWom5/56TF4lPcqjnDHcg==",
      "dev": true,
      "license": "MIT",
      "dependencies": {
        "balanced-match": "^1.0.0",
        "concat-map": "0.0.1"
@@ -2102,6 +2018,19 @@
      "integrity": "sha512-E+XQCRwSbaaiChtv6k6Dwgc+bx+Bs6vuKJHHl5kox/BaKbhiXzqQOwK4cO22yElGp2OCmjwVhT3HmxgyPGnJfQ==",
      "dev": true
    },
    "node_modules/call-bind-apply-helpers": {
      "version": "1.0.2",
      "resolved": "https://registry.npmjs.org/call-bind-apply-helpers/-/call-bind-apply-helpers-1.0.2.tgz",
      "integrity": "sha512-Sp1ablJ0ivDkSzjcaJdxEunN5/XvksFJ2sMBFfq6x0ryhQV/2b/KwFe21cMpmHtPOSij8K99/wSfoEuTObmuMQ==",
      "license": "MIT",
      "dependencies": {
        "es-errors": "^1.3.0",
        "function-bind": "^1.1.2"
      },
      "engines": {
        "node": ">= 0.4"
      }
    },
    "node_modules/callsites": {
      "version": "3.1.0",
      "resolved": "https://registry.npmjs.org/callsites/-/callsites-3.1.0.tgz",
@@ -2298,9 +2227,11 @@
      }
    },
    "node_modules/cross-spawn": {
-      "version": "7.0.3",
+      "version": "7.0.6",
-      "resolved": "https://registry.npmjs.org/cross-spawn/-/cross-spawn-7.0.3.tgz",
+      "resolved": "https://registry.npmjs.org/cross-spawn/-/cross-spawn-7.0.6.tgz",
-      "integrity": "sha512-iRDPJKUPVEND7dHPO8rkbOnPpyDygcDFtWjpeWNCgy8WP2rXcxXL8TskReQl6OrB2G7+UJrags1q15Fudc7G6w==",
+      "integrity": "sha512-uV2QOWP2nWzsy2aMp8aRibhi9dlzF5Hgh5SHaB9OiTGEyDTiJJyx0uy51QXdyWbtAHNua4XJzUKca3OzKUd3vA==",
      "dev": true,
      "license": "MIT",
      "dependencies": {
        "path-key": "^3.1.0",
        "shebang-command": "^2.0.0",
@@ -2384,10 +2315,19 @@
        "node": "^14.15.0 || ^16.10.0 || >=18.0.0"
      }
    },
-    "node_modules/eastasianwidth": {
+    "node_modules/dunder-proto": {
-      "version": "0.2.0",
+      "version": "1.0.1",
-      "resolved": "https://registry.npmjs.org/eastasianwidth/-/eastasianwidth-0.2.0.tgz",
+      "resolved": "https://registry.npmjs.org/dunder-proto/-/dunder-proto-1.0.1.tgz",
-      "integrity": "sha512-I88TYZWc9XiYHRQ4/3c5rjjfgkjhLyW2luGIheGERbNQ6OY7yTybanSpDXZa8y7VUP9YmDcYa+eyq4ca7iLqWA=="
+      "integrity": "sha512-KIN/nDJBQRcXw0MLVhZE9iQHmG68qAVIBg9CqmUYjmQIhgij9U5MFvrqkUL5FbtyyzZuOeOt0zdeRe4UY7ct+A==",
      "license": "MIT",
      "dependencies": {
        "call-bind-apply-helpers": "^1.0.1",
        "es-errors": "^1.3.0",
        "gopd": "^1.2.0"
      },
      "engines": {
        "node": ">= 0.4"
      }
    },
    "node_modules/ejs": {
      "version": "3.1.10",
@@ -2425,7 +2365,8 @@
    "node_modules/emoji-regex": {
      "version": "8.0.0",
      "resolved": "https://registry.npmjs.org/emoji-regex/-/emoji-regex-8.0.0.tgz",
-      "integrity": "sha512-MSjYzcWNOA0ewAHpz0MxpYFvwg6yjy1NG3xteoqz644VCo/RPgnr1/GGt+ic3iJTzQ8Eu3TdM14SawnVUmGE6A=="
+      "integrity": "sha512-MSjYzcWNOA0ewAHpz0MxpYFvwg6yjy1NG3xteoqz644VCo/RPgnr1/GGt+ic3iJTzQ8Eu3TdM14SawnVUmGE6A==",
      "dev": true
    },
    "node_modules/error-ex": {
      "version": "1.3.2",
@@ -2442,6 +2383,51 @@
      "integrity": "sha512-zz06S8t0ozoDXMG+ube26zeCTNXcKIPJZJi8hBrF4idCLms4CG9QtK7qBl1boi5ODzFpjswb5JPmHCbMpjaYzg==",
      "dev": true
    },
    "node_modules/es-define-property": {
      "version": "1.0.1",
      "resolved": "https://registry.npmjs.org/es-define-property/-/es-define-property-1.0.1.tgz",
      "integrity": "sha512-e3nRfgfUZ4rNGL232gUgX06QNyyez04KdjFrF+LTRoOXmrOgFKDg4BCdsjW8EnT69eqdYGmRpJwiPVYNrCaW3g==",
      "license": "MIT",
      "engines": {
        "node": ">= 0.4"
      }
    },
    "node_modules/es-errors": {
      "version": "1.3.0",
      "resolved": "https://registry.npmjs.org/es-errors/-/es-errors-1.3.0.tgz",
      "integrity": "sha512-Zf5H2Kxt2xjTvbJvP2ZWLEICxA6j+hAmMzIlypy4xcBg1vKVnx89Wy0GbS+kf5cwCVFFzdCFh2XSCFNULS6csw==",
      "license": "MIT",
      "engines": {
        "node": ">= 0.4"
      }
    },
    "node_modules/es-object-atoms": {
      "version": "1.1.1",
      "resolved": "https://registry.npmjs.org/es-object-atoms/-/es-object-atoms-1.1.1.tgz",
      "integrity": "sha512-FGgH2h8zKNim9ljj7dankFPcICIK9Cp5bm+c2gQSYePhpaG5+esrLODihIorn+Pe6FGJzWhXQotPv73jTaldXA==",
      "license": "MIT",
      "dependencies": {
        "es-errors": "^1.3.0"
      },
      "engines": {
        "node": ">= 0.4"
      }
    },
    "node_modules/es-set-tostringtag": {
      "version": "2.1.0",
      "resolved": "https://registry.npmjs.org/es-set-tostringtag/-/es-set-tostringtag-2.1.0.tgz",
      "integrity": "sha512-j6vWzfrGVfyXxge+O0x5sh6cvxAog0a/4Rdd2K36zCMV5eJ+/+tOAngRO8cODMNWbVRdVlmGZQL2YS3yR8bIUA==",
      "license": "MIT",
      "dependencies": {
        "es-errors": "^1.3.0",
        "get-intrinsic": "^1.2.6",
        "has-tostringtag": "^1.0.2",
        "hasown": "^2.0.2"
      },
      "engines": {
        "node": ">= 0.4"
      }
    },
    "node_modules/escalade": {
      "version": "3.2.0",
      "resolved": "https://registry.npmjs.org/escalade/-/escalade-3.2.0.tgz",
@@ -2554,19 +2540,21 @@
      }
    },
    "node_modules/filelist/node_modules/brace-expansion": {
-      "version": "2.0.1",
+      "version": "2.0.2",
-      "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-2.0.1.tgz",
+      "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-2.0.2.tgz",
-      "integrity": "sha512-XnAIvQ8eM+kC6aULx6wuQiwVsnzsi9d3WxzV3FpWTGA19F621kwdbsAcFKXgKUHZWsy+mY6iL1sHTxWEFCytDA==",
+      "integrity": "sha512-Jt0vHyM+jmUBqojB7E1NIYadt0vI0Qxjxd2TErW94wDz+E2LAm5vKMXXwg6ZZBTHPuUlDgQHKXvjGBdfcF1ZDQ==",
      "dev": true,
      "license": "MIT",
      "dependencies": {
        "balanced-match": "^1.0.0"
      }
    },
    "node_modules/filelist/node_modules/minimatch": {
-      "version": "5.1.6",
+      "version": "5.1.9",
-      "resolved": "https://registry.npmjs.org/minimatch/-/minimatch-5.1.6.tgz",
+      "resolved": "https://registry.npmjs.org/minimatch/-/minimatch-5.1.9.tgz",
-      "integrity": "sha512-lKwV/1brpG6mBUFHtb7NUmtABCb2WZZmm2wNiOA5hAb8VdCS4B3dtMWyvcoViccwAW/COERjXLt0zP1zXUN26g==",
+      "integrity": "sha512-7o1wEA2RyMP7Iu7GNba9vc0RWWGACJOCZBJX2GJWip0ikV+wcOsgVuY9uE8CPiyQhkGFSlhuSkZPavN7u1c2Fw==",
      "dev": true,
      "license": "ISC",
      "dependencies": {
        "brace-expansion": "^2.0.1"
      },
@@ -2604,39 +2592,16 @@
      "resolved": "https://registry.npmjs.org/flatbuffers/-/flatbuffers-1.12.0.tgz",
      "integrity": "sha512-c7CZADjRcl6j0PlvFy0ZqXQ67qSEZfrVPynmnL+2zPc+NtMvrF8Y0QceMo7QqnSPc7+uWjUIAbvCQ5WIKlMVdQ=="
    },
    "node_modules/foreground-child": {
      "version": "3.3.0",
      "resolved": "https://registry.npmjs.org/foreground-child/-/foreground-child-3.3.0.tgz",
      "integrity": "sha512-Ld2g8rrAyMYFXBhEqMz8ZAHBi4J4uS1i/CxGMDnjyFWddMXLVcDp051DZfu+t7+ab7Wv6SMqpWmyFIj5UbfFvg==",
      "dependencies": {
        "cross-spawn": "^7.0.0",
        "signal-exit": "^4.0.1"
      },
      "engines": {
        "node": ">=14"
      },
      "funding": {
        "url": "https://github.com/sponsors/isaacs"
      }
    },
    "node_modules/foreground-child/node_modules/signal-exit": {
      "version": "4.1.0",
      "resolved": "https://registry.npmjs.org/signal-exit/-/signal-exit-4.1.0.tgz",
      "integrity": "sha512-bzyZ1e88w9O1iNJbKnOlvYTrWPDl46O1bG0D3XInv+9tkPrxrN8jUUTiFlDkkmKWgn1M6CfIA13SuGqOa9Korw==",
      "engines": {
        "node": ">=14"
      },
      "funding": {
        "url": "https://github.com/sponsors/isaacs"
      }
    },
    "node_modules/form-data": {
-      "version": "4.0.1",
+      "version": "4.0.5",
-      "resolved": "https://registry.npmjs.org/form-data/-/form-data-4.0.1.tgz",
+      "resolved": "https://registry.npmjs.org/form-data/-/form-data-4.0.5.tgz",
-      "integrity": "sha512-tzN8e4TX8+kkxGPK8D5u0FNmjPUjw3lwC9lSLxxoB/+GtsJG91CO8bSWy73APlgAZzZbXEYZJuxjkHH2w+Ezhw==",
+      "integrity": "sha512-8RipRLol37bNs2bhoV67fiTEvdTrbMUYcFTiy3+wuuOnUog2QBHCZWXDRijWQfAkhBj2Uf5UnVaiWwA5vdd82w==",
      "license": "MIT",
      "dependencies": {
        "asynckit": "^0.4.0",
        "combined-stream": "^1.0.8",
        "es-set-tostringtag": "^2.1.0",
        "hasown": "^2.0.2",
        "mime-types": "^2.1.12"
      },
      "engines": {
@@ -2684,7 +2649,6 @@
      "version": "1.1.2",
      "resolved": "https://registry.npmjs.org/function-bind/-/function-bind-1.1.2.tgz",
      "integrity": "sha512-7XHNxH7qX9xG5mIwxkhumTox/MIRNcOgDrxWsMt2pAr23WHp6MrRlN7FBSFpCpr+oVO0F744iUgR82nJMfG2SA==",
      "dev": true,
      "funding": {
        "url": "https://github.com/sponsors/ljharb"
      }
@@ -2707,6 +2671,30 @@
        "node": "6.* || 8.* || >= 10.*"
      }
    },
    "node_modules/get-intrinsic": {
      "version": "1.3.0",
      "resolved": "https://registry.npmjs.org/get-intrinsic/-/get-intrinsic-1.3.0.tgz",
      "integrity": "sha512-9fSjSaos/fRIVIp+xSJlE6lfwhES7LNtKaCBIamHsjr2na1BiABJPo0mOjjz8GJDURarmCPGqaiVg5mfjb98CQ==",
      "license": "MIT",
      "dependencies": {
        "call-bind-apply-helpers": "^1.0.2",
        "es-define-property": "^1.0.1",
        "es-errors": "^1.3.0",
        "es-object-atoms": "^1.1.1",
        "function-bind": "^1.1.2",
        "get-proto": "^1.0.1",
        "gopd": "^1.2.0",
        "has-symbols": "^1.1.0",
        "hasown": "^2.0.2",
        "math-intrinsics": "^1.1.0"
      },
      "engines": {
        "node": ">= 0.4"
      },
      "funding": {
        "url": "https://github.com/sponsors/ljharb"
      }
    },
    "node_modules/get-package-type": {
      "version": "0.1.0",
      "resolved": "https://registry.npmjs.org/get-package-type/-/get-package-type-0.1.0.tgz",
@@ -2716,6 +2704,19 @@
        "node": ">=8.0.0"
      }
    },
    "node_modules/get-proto": {
      "version": "1.0.1",
      "resolved": "https://registry.npmjs.org/get-proto/-/get-proto-1.0.1.tgz",
      "integrity": "sha512-sTSfBjoXBp89JvIKIefqw7U2CCebsc74kiY6awiGogKtoSGbgjYE/G/+l9sF3MWFPNc9IcoOC4ODfKHfxFmp0g==",
      "license": "MIT",
      "dependencies": {
        "dunder-proto": "^1.0.1",
        "es-object-atoms": "^1.0.0"
      },
      "engines": {
        "node": ">= 0.4"
      }
    },
    "node_modules/get-stream": {
      "version": "6.0.1",
      "resolved": "https://registry.npmjs.org/get-stream/-/get-stream-6.0.1.tgz",
@@ -2758,6 +2759,18 @@
        "node": ">=4"
      }
    },
    "node_modules/gopd": {
      "version": "1.2.0",
      "resolved": "https://registry.npmjs.org/gopd/-/gopd-1.2.0.tgz",
      "integrity": "sha512-ZUKRh6/kUFoAiTAtTYPZJ3hw9wNxx+BIBOijnlG9PnrJsCcSjs1wyyD6vJpaYtgnzDrKYRSqf3OO6Rfa93xsRg==",
      "license": "MIT",
      "engines": {
        "node": ">= 0.4"
      },
      "funding": {
        "url": "https://github.com/sponsors/ljharb"
      }
    },
    "node_modules/graceful-fs": {
      "version": "4.2.11",
      "resolved": "https://registry.npmjs.org/graceful-fs/-/graceful-fs-4.2.11.tgz",
@@ -2778,11 +2791,37 @@
        "node": ">=8"
      }
    },
    "node_modules/has-symbols": {
      "version": "1.1.0",
      "resolved": "https://registry.npmjs.org/has-symbols/-/has-symbols-1.1.0.tgz",
      "integrity": "sha512-1cDNdwJ2Jaohmb3sg4OmKaMBwuC48sYni5HUw2DvsC8LjGTLK9h+eb1X6RyuOHe4hT0ULCW68iomhjUoKUqlPQ==",
      "license": "MIT",
      "engines": {
        "node": ">= 0.4"
      },
      "funding": {
        "url": "https://github.com/sponsors/ljharb"
      }
    },
    "node_modules/has-tostringtag": {
      "version": "1.0.2",
      "resolved": "https://registry.npmjs.org/has-tostringtag/-/has-tostringtag-1.0.2.tgz",
      "integrity": "sha512-NqADB8VjPFLM2V0VvHUewwwsw0ZWBaIdgo+ieHtK3hasLz4qeCRjYcqfB6AQrBggRKppKF8L52/VqdVsO47Dlw==",
      "license": "MIT",
      "dependencies": {
        "has-symbols": "^1.0.3"
      },
      "engines": {
        "node": ">= 0.4"
      },
      "funding": {
        "url": "https://github.com/sponsors/ljharb"
      }
    },
    "node_modules/hasown": {
      "version": "2.0.2",
      "resolved": "https://registry.npmjs.org/hasown/-/hasown-2.0.2.tgz",
      "integrity": "sha512-0hJU9SCPvmMzIBdZFqNPXWa6dqh7WdH0cII9y+CyS8rG3nL48Bclra9HmKhVVUHyPWNH5Y7xDwAB7bfgSjkUMQ==",
      "dev": true,
      "dependencies": {
        "function-bind": "^1.1.2"
      },
@@ -2882,6 +2921,7 @@
      "version": "3.0.0",
      "resolved": "https://registry.npmjs.org/is-fullwidth-code-point/-/is-fullwidth-code-point-3.0.0.tgz",
      "integrity": "sha512-zymm5+u+sCsSWyD9qNaejV3DFvhCKclKdizYaJUuHA83RLjb7nSuGnddCHGv0hk+KY7BMAlsWeK4Ueg6EV6XQg==",
      "dev": true,
      "engines": {
        "node": ">=8"
      }
@@ -2919,7 +2959,8 @@
    "node_modules/isexe": {
      "version": "2.0.0",
      "resolved": "https://registry.npmjs.org/isexe/-/isexe-2.0.0.tgz",
-      "integrity": "sha512-RHxMLp9lnKHGHRng9QFhRCMbYAcVpn69smSGcq3f36xjgVVWThj4qqLbTLlq7Ssj8B+fIQ1EuCEGI2lKsyQeIw=="
+      "integrity": "sha512-RHxMLp9lnKHGHRng9QFhRCMbYAcVpn69smSGcq3f36xjgVVWThj4qqLbTLlq7Ssj8B+fIQ1EuCEGI2lKsyQeIw==",
      "dev": true
    },
    "node_modules/istanbul-lib-coverage": {
      "version": "3.2.2",
@@ -2987,20 +3028,6 @@
        "node": ">=8"
      }
    },
    "node_modules/jackspeak": {
      "version": "3.4.3",
      "resolved": "https://registry.npmjs.org/jackspeak/-/jackspeak-3.4.3.tgz",
      "integrity": "sha512-OGlZQpz2yfahA/Rd1Y8Cd9SIEsqvXkLVoSw/cgwhnhFMDbsQFeZYoJJ7bIZBS9BcamUW96asq/npPWugM+RQBw==",
      "dependencies": {
        "@isaacs/cliui": "^8.0.2"
      },
      "funding": {
        "url": "https://github.com/sponsors/isaacs"
      },
      "optionalDependencies": {
        "@pkgjs/parseargs": "^0.11.0"
      }
    },
    "node_modules/jake": {
      "version": "10.9.2",
      "resolved": "https://registry.npmjs.org/jake/-/jake-10.9.2.tgz",
@@ -3605,10 +3632,11 @@
      "dev": true
    },
    "node_modules/js-yaml": {
-      "version": "3.14.1",
+      "version": "3.14.2",
-      "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-3.14.1.tgz",
+      "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-3.14.2.tgz",
-      "integrity": "sha512-okMH7OXXJ7YrN9Ok3/SXrnu4iX9yOk+25nqX4imS2npuvTYDmo/QEZoqwZkYaIDk3jVvBOTOIEgEhaLOynBS9g==",
+      "integrity": "sha512-PMSmkqxr106Xa156c2M265Z+FTrPl+oxd/rgOQy2tijQeK5TxQ43psO1ZCwhVOSdnn+RzkzlRz/eY4BgJBYVpg==",
      "dev": true,
      "license": "MIT",
      "dependencies": {
        "argparse": "^1.0.7",
        "esprima": "^4.0.0"
@@ -3728,6 +3756,15 @@
        "tmpl": "1.0.5"
      }
    },
    "node_modules/math-intrinsics": {
      "version": "1.1.0",
      "resolved": "https://registry.npmjs.org/math-intrinsics/-/math-intrinsics-1.1.0.tgz",
      "integrity": "sha512-/IXtbwEk5HTPyEwyKX6hGkYXxM9nbj64B+ilVJnC/R6B0pH5G4V3b0pVbL7DBj4tkhBAppbQUlf6F6Xl9LHu1g==",
      "license": "MIT",
      "engines": {
        "node": ">= 0.4"
      }
    },
    "node_modules/merge-stream": {
      "version": "2.0.0",
      "resolved": "https://registry.npmjs.org/merge-stream/-/merge-stream-2.0.0.tgz",
@@ -3776,10 +3813,11 @@
      }
    },
    "node_modules/minimatch": {
-      "version": "3.1.2",
+      "version": "3.1.5",
-      "resolved": "https://registry.npmjs.org/minimatch/-/minimatch-3.1.2.tgz",
+      "resolved": "https://registry.npmjs.org/minimatch/-/minimatch-3.1.5.tgz",
-      "integrity": "sha512-J7p63hRiAjw1NDEww1W7i37+ByIrOWO5XQQAzZ3VOcL0PNybwpfmV/N05zFAzwQ9USyEcX6t3UO+K5aqBQOIHw==",
+      "integrity": "sha512-VgjWUsnnT6n+NUk6eZq77zeFdpW2LWDzP6zFGrCbHXiYNul5Dzqk2HHQ5uFH2DNW5Xbp8+jVzaeNt94ssEEl4w==",
      "dev": true,
      "license": "ISC",
      "dependencies": {
        "brace-expansion": "^1.1.7"
      },
@@ -3796,31 +3834,17 @@
      }
    },
    "node_modules/minizlib": {
-      "version": "3.0.1",
+      "version": "3.1.0",
-      "resolved": "https://registry.npmjs.org/minizlib/-/minizlib-3.0.1.tgz",
+      "resolved": "https://registry.npmjs.org/minizlib/-/minizlib-3.1.0.tgz",
-      "integrity": "sha512-umcy022ILvb5/3Djuu8LWeqUa8D68JaBzlttKeMWen48SjabqS3iY5w/vzeMzMUNhLDifyhbOwKDSznB1vvrwg==",
+      "integrity": "sha512-KZxYo1BUkWD2TVFLr0MQoM8vUUigWD3LlD83a/75BqC+4qE0Hb1Vo5v1FgcfaNXvfXzr+5EhQ6ing/CaBijTlw==",
      "license": "MIT",
      "dependencies": {
-        "minipass": "^7.0.4",
+        "minipass": "^7.1.2"
        "rimraf": "^5.0.5"
      },
      "engines": {
        "node": ">= 18"
      }
    },
    "node_modules/mkdirp": {
      "version": "3.0.1",
      "resolved": "https://registry.npmjs.org/mkdirp/-/mkdirp-3.0.1.tgz",
      "integrity": "sha512-+NsyUUAZDmo6YVHzL/stxSu3t9YS1iljliy3BSDrXJ/dkn1KYdmtZODGGjLcc9XLgVVpH4KshHB8XmZgMhaBXg==",
      "bin": {
        "mkdirp": "dist/cjs/src/bin.js"
      },
      "engines": {
        "node": ">=10"
      },
      "funding": {
        "url": "https://github.com/sponsors/isaacs"
      }
    },
    "node_modules/ms": {
      "version": "2.1.3",
      "resolved": "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz",
@@ -4010,11 +4034,6 @@
        "node": ">=6"
      }
    },
    "node_modules/package-json-from-dist": {
      "version": "1.0.1",
      "resolved": "https://registry.npmjs.org/package-json-from-dist/-/package-json-from-dist-1.0.1.tgz",
      "integrity": "sha512-UEZIS3/by4OC8vL3P2dTXRETpebLI2NiI5vIrjaD/5UtrkFX/tNbwjTSRAGC/+7CAo2pIcBaRgWmcBBHcsaCIw=="
    },
    "node_modules/parse-json": {
      "version": "5.2.0",
      "resolved": "https://registry.npmjs.org/parse-json/-/parse-json-5.2.0.tgz",
@@ -4055,6 +4074,7 @@
      "version": "3.1.1",
      "resolved": "https://registry.npmjs.org/path-key/-/path-key-3.1.1.tgz",
      "integrity": "sha512-ojmeN0qd+y0jszEtoY48r0Peq5dwMEkIlCOu6Q5f41lfkswXuKtYrhgoTpLnyIcHm24Uhqx+5Tqm2InSwLhE6Q==",
      "dev": true,
      "engines": {
        "node": ">=8"
      }
@@ -4065,26 +4085,6 @@
      "integrity": "sha512-LDJzPVEEEPR+y48z93A0Ed0yXb8pAByGWo/k5YYdYgpY2/2EsOsksJrq7lOHxryrVOn1ejG6oAp8ahvOIQD8sw==",
      "dev": true
    },
    "node_modules/path-scurry": {
      "version": "1.11.1",
      "resolved": "https://registry.npmjs.org/path-scurry/-/path-scurry-1.11.1.tgz",
      "integrity": "sha512-Xa4Nw17FS9ApQFJ9umLiJS4orGjm7ZzwUrwamcGQuHSzDyth9boKDaycYdDcZDuqYATXw4HFXgaqWTctW/v1HA==",
      "dependencies": {
        "lru-cache": "^10.2.0",
        "minipass": "^5.0.0 || ^6.0.2 || ^7.0.0"
      },
      "engines": {
        "node": ">=16 || 14 >=14.18"
      },
      "funding": {
        "url": "https://github.com/sponsors/isaacs"
      }
    },
    "node_modules/path-scurry/node_modules/lru-cache": {
      "version": "10.4.3",
      "resolved": "https://registry.npmjs.org/lru-cache/-/lru-cache-10.4.3.tgz",
      "integrity": "sha512-JNAzZcXrCt42VGLuYz0zfAzDfAvJWW6AfYlDBQyDV5DClI2m5sAmK+OIO7s59XfsRsWHp02jAJrRadPRGTt6SQ=="
    },
    "node_modules/picocolors": {
      "version": "1.1.1",
      "resolved": "https://registry.npmjs.org/picocolors/-/picocolors-1.1.1.tgz",
@@ -4246,61 +4246,6 @@
        "node": ">=10"
      }
    },
    "node_modules/rimraf": {
      "version": "5.0.10",
      "resolved": "https://registry.npmjs.org/rimraf/-/rimraf-5.0.10.tgz",
      "integrity": "sha512-l0OE8wL34P4nJH/H2ffoaniAokM2qSmrtXHmlpvYr5AVVX8msAyW0l8NVJFDxlSK4u3Uh/f41cQheDVdnYijwQ==",
      "dependencies": {
        "glob": "^10.3.7"
      },
      "bin": {
        "rimraf": "dist/esm/bin.mjs"
      },
      "funding": {
        "url": "https://github.com/sponsors/isaacs"
      }
    },
    "node_modules/rimraf/node_modules/brace-expansion": {
      "version": "2.0.1",
      "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-2.0.1.tgz",
      "integrity": "sha512-XnAIvQ8eM+kC6aULx6wuQiwVsnzsi9d3WxzV3FpWTGA19F621kwdbsAcFKXgKUHZWsy+mY6iL1sHTxWEFCytDA==",
      "dependencies": {
        "balanced-match": "^1.0.0"
      }
    },
    "node_modules/rimraf/node_modules/glob": {
      "version": "10.4.5",
      "resolved": "https://registry.npmjs.org/glob/-/glob-10.4.5.tgz",
      "integrity": "sha512-7Bv8RF0k6xjo7d4A/PxYLbUCfb6c+Vpd2/mB2yRDlew7Jb5hEXiCD9ibfO7wpk8i4sevK6DFny9h7EYbM3/sHg==",
      "dependencies": {
        "foreground-child": "^3.1.0",
        "jackspeak": "^3.1.2",
        "minimatch": "^9.0.4",
        "minipass": "^7.1.2",
        "package-json-from-dist": "^1.0.0",
        "path-scurry": "^1.11.1"
      },
      "bin": {
        "glob": "dist/esm/bin.mjs"
      },
      "funding": {
        "url": "https://github.com/sponsors/isaacs"
      }
    },
    "node_modules/rimraf/node_modules/minimatch": {
      "version": "9.0.5",
      "resolved": "https://registry.npmjs.org/minimatch/-/minimatch-9.0.5.tgz",
      "integrity": "sha512-G6T0ZX48xgozx7587koeX9Ys2NYy6Gmv//P89sEte9V9whIapMNF4idKxnW2QtCcLiTWlb/wfCabAtAFWhhBow==",
      "dependencies": {
        "brace-expansion": "^2.0.1"
      },
      "engines": {
        "node": ">=16 || 14 >=14.17"
      },
      "funding": {
        "url": "https://github.com/sponsors/isaacs"
      }
    },
    "node_modules/semver": {
      "version": "7.6.3",
      "resolved": "https://registry.npmjs.org/semver/-/semver-7.6.3.tgz",
@@ -4354,6 +4299,7 @@
      "version": "2.0.0",
      "resolved": "https://registry.npmjs.org/shebang-command/-/shebang-command-2.0.0.tgz",
      "integrity": "sha512-kHxr2zZpYtdmrN1qDjrrX/Z1rR1kG8Dx+gkpK1G4eXmvXswmcE1hTWBWYUzlraYw1/yZp6YuDY77YtvbN0dmDA==",
      "dev": true,
      "dependencies": {
        "shebang-regex": "^3.0.0"
      },
@@ -4365,6 +4311,7 @@
      "version": "3.0.0",
      "resolved": "https://registry.npmjs.org/shebang-regex/-/shebang-regex-3.0.0.tgz",
      "integrity": "sha512-7++dFhtcx3353uBaq8DDR4NuxBetBzC7ZQOhmTQInHEd6bSrXdiEyzCvG07Z44UYdLShWUyXt5M/yhz8ekcb1A==",
      "dev": true,
      "engines": {
        "node": ">=8"
      }
@@ -4452,20 +4399,7 @@
      "version": "4.2.3",
      "resolved": "https://registry.npmjs.org/string-width/-/string-width-4.2.3.tgz",
      "integrity": "sha512-wKyQRQpjJ0sIp62ErSZdGsjMJWsap5oRNihHhu6G7JVO/9jIB6UyevL+tXuOqrng8j/cxKTWyWUwvSTriiZz/g==",
-      "dependencies": {
+      "dev": true,
        "emoji-regex": "^8.0.0",
        "is-fullwidth-code-point": "^3.0.0",
        "strip-ansi": "^6.0.1"
      },
      "engines": {
        "node": ">=8"
      }
    },
    "node_modules/string-width-cjs": {
      "name": "string-width",
      "version": "4.2.3",
      "resolved": "https://registry.npmjs.org/string-width/-/string-width-4.2.3.tgz",
      "integrity": "sha512-wKyQRQpjJ0sIp62ErSZdGsjMJWsap5oRNihHhu6G7JVO/9jIB6UyevL+tXuOqrng8j/cxKTWyWUwvSTriiZz/g==",
      "dependencies": {
        "emoji-regex": "^8.0.0",
        "is-fullwidth-code-point": "^3.0.0",
@@ -4479,18 +4413,7 @@
      "version": "6.0.1",
      "resolved": "https://registry.npmjs.org/strip-ansi/-/strip-ansi-6.0.1.tgz",
      "integrity": "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A==",
-      "dependencies": {
+      "dev": true,
        "ansi-regex": "^5.0.1"
      },
      "engines": {
        "node": ">=8"
      }
    },
    "node_modules/strip-ansi-cjs": {
      "name": "strip-ansi",
      "version": "6.0.1",
      "resolved": "https://registry.npmjs.org/strip-ansi/-/strip-ansi-6.0.1.tgz",
      "integrity": "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A==",
      "dependencies": {
        "ansi-regex": "^5.0.1"
      },
@@ -4541,15 +4464,15 @@
      }
    },
    "node_modules/tar": {
-      "version": "7.4.3",
+      "version": "7.5.10",
-      "resolved": "https://registry.npmjs.org/tar/-/tar-7.4.3.tgz",
+      "resolved": "https://registry.npmjs.org/tar/-/tar-7.5.10.tgz",
-      "integrity": "sha512-5S7Va8hKfV7W5U6g3aYxXmlPoZVAwUMy9AOKyF2fVuZa2UD3qZjg578OrLRt8PcNN1PleVaL/5/yYATNL0ICUw==",
+      "integrity": "sha512-8mOPs1//5q/rlkNSPcCegA6hiHJYDmSLEI8aMH/CdSQJNWztHC9WHNam5zdQlfpTwB9Xp7IBEsHfV5LKMJGVAw==",
      "license": "BlueOak-1.0.0",
      "dependencies": {
        "@isaacs/fs-minipass": "^4.0.0",
        "chownr": "^3.0.0",
        "minipass": "^7.1.2",
-        "minizlib": "^3.0.1",
+        "minizlib": "^3.1.0",
        "mkdirp": "^3.0.1",
        "yallist": "^5.0.0"
      },
      "engines": {
@@ -4782,6 +4705,7 @@
      "version": "2.0.2",
      "resolved": "https://registry.npmjs.org/which/-/which-2.0.2.tgz",
      "integrity": "sha512-BLI3Tl1TW3Pvl70l3yq3Y64i+awpwXqsGBYWkkqMtnbXgrMD+yj7rhW0kuEDxzJaYXGjEW5ogapKNMEKNMjibA==",
      "dev": true,
      "dependencies": {
        "isexe": "^2.0.0"
      },
@@ -4809,23 +4733,6 @@
        "url": "https://github.com/chalk/wrap-ansi?sponsor=1"
      }
    },
    "node_modules/wrap-ansi-cjs": {
      "name": "wrap-ansi",
      "version": "7.0.0",
      "resolved": "https://registry.npmjs.org/wrap-ansi/-/wrap-ansi-7.0.0.tgz",
      "integrity": "sha512-YVGIj2kamLSTxw6NsZjoBxfSwsn0ycdesmc4p+Q21c5zPuZ1pl+NfxVdxPtdHvmNVOQ6XSYG4AUtyt/Fi7D16Q==",
      "dependencies": {
        "ansi-styles": "^4.0.0",
        "string-width": "^4.1.0",
        "strip-ansi": "^6.0.0"
      },
      "engines": {
        "node": ">=10"
      },
      "funding": {
        "url": "https://github.com/chalk/wrap-ansi?sponsor=1"
      }
    },
    "node_modules/wrappy": {
      "version": "1.0.2",
      "resolved": "https://registry.npmjs.org/wrappy/-/wrappy-1.0.2.tgz",
--- a/nodejs/lancedb/arrow.ts
+++ b/nodejs/lancedb/arrow.ts
@@ -20,6 +20,8 @@ import {
  Float32,
  Float64,
  Int,
  Int8,
  Int16,
  Int32,
  Int64,
  LargeBinary,
@@ -35,6 +37,8 @@ import {
  Timestamp,
  Type,
  Uint8,
  Uint16,
  Uint32,
  Utf8,
  Vector,
  makeVector as arrowMakeVector,
@@ -113,8 +117,9 @@ export type TableLike =
 export type IntoVector =
  | Float32Array
  | Float64Array
  | Uint8Array
  | number[]
-  | Promise<Float32Array | Float64Array | number[]>;
+  | Promise<Float32Array | Float64Array | Uint8Array | number[]>;
 export type MultiVector = IntoVector[];
@@ -122,14 +127,48 @@ export function isMultiVector(value: unknown): value is MultiVector {
  return Array.isArray(value) && isIntoVector(value[0]);
 }
 // Float16Array is not in TypeScript's standard lib yet; access dynamically
 type Float16ArrayCtor = new (
  ...args: unknown[]
 ) => { buffer: ArrayBuffer; byteOffset: number; byteLength: number };
 const float16ArrayCtor = (globalThis as unknown as Record<string, unknown>)
  .Float16Array as Float16ArrayCtor | undefined;
 export function isIntoVector(value: unknown): value is IntoVector {
  return (
    value instanceof Float32Array ||
    value instanceof Float64Array ||
    value instanceof Uint8Array ||
    (float16ArrayCtor !== undefined && value instanceof float16ArrayCtor) ||
    (Array.isArray(value) && !Array.isArray(value[0]))
  );
 }
 /**
 * Extract the underlying byte buffer and data type from a typed array
 * for passing to the Rust NAPI layer without precision loss.
 */
 export function extractVectorBuffer(
  vector: Float32Array | Float64Array | Uint8Array,
 ): { data: Uint8Array; dtype: string } | null {
  if (float16ArrayCtor !== undefined && vector instanceof float16ArrayCtor) {
    return {
      data: new Uint8Array(vector.buffer, vector.byteOffset, vector.byteLength),
      dtype: "float16",
    };
  }
  if (vector instanceof Float64Array) {
    return {
      data: new Uint8Array(vector.buffer, vector.byteOffset, vector.byteLength),
      dtype: "float64",
    };
  }
  if (vector instanceof Uint8Array && !(vector instanceof Float32Array)) {
    return { data: vector, dtype: "uint8" };
  }
  return null;
 }
 export function isArrowTable(value: object): value is TableLike {
  if (value instanceof ArrowTable) return true;
  return "schema" in value && "batches" in value;
@@ -529,7 +568,8 @@ function isObject(value: unknown): value is Record<string, unknown> {
    !(value instanceof Date) &&
    !(value instanceof Set) &&
    !(value instanceof Map) &&
-    !(value instanceof Buffer)
+    !(value instanceof Buffer) &&
    !ArrayBuffer.isView(value)
  );
 }
@@ -588,6 +628,13 @@ function inferType(
    return new Bool();
  } else if (value instanceof Buffer) {
    return new Binary();
  } else if (ArrayBuffer.isView(value) && !(value instanceof DataView)) {
    const info = typedArrayToArrowType(value);
    if (info !== undefined) {
      const child = new Field("item", info.elementType, true);
      return new FixedSizeList(info.length, child);
    }
    return undefined;
  } else if (Array.isArray(value)) {
    if (value.length === 0) {
      return undefined; // Without any values we can't infer the type
@@ -746,6 +793,32 @@ function makeListVector(lists: unknown[][]): Vector<unknown> {
  return listBuilder.finish().toVector();
 }
 /**
 * Map a JS TypedArray instance to the corresponding Arrow element DataType
 * and its length. Returns undefined if the value is not a recognized TypedArray.
 */
 function typedArrayToArrowType(
  value: ArrayBufferView,
 ): { elementType: DataType; length: number } | undefined {
  if (value instanceof Float32Array)
    return { elementType: new Float32(), length: value.length };
  if (value instanceof Float64Array)
    return { elementType: new Float64(), length: value.length };
  if (value instanceof Uint8Array)
    return { elementType: new Uint8(), length: value.length };
  if (value instanceof Uint16Array)
    return { elementType: new Uint16(), length: value.length };
  if (value instanceof Uint32Array)
    return { elementType: new Uint32(), length: value.length };
  if (value instanceof Int8Array)
    return { elementType: new Int8(), length: value.length };
  if (value instanceof Int16Array)
    return { elementType: new Int16(), length: value.length };
  if (value instanceof Int32Array)
    return { elementType: new Int32(), length: value.length };
  return undefined;
 }
 /** Helper function to convert an Array of JS values to an Arrow Vector */
 function makeVector(
  values: unknown[],
@@ -814,6 +887,16 @@ function makeVector(
      "makeVector cannot infer the type if all values are null or undefined",
    );
  }
  if (ArrayBuffer.isView(sampleValue) && !(sampleValue instanceof DataView)) {
    const info = typedArrayToArrowType(sampleValue);
    if (info !== undefined) {
      const fslType = new FixedSizeList(
        info.length,
        new Field("item", info.elementType, true),
      );
      return vectorFromArray(values, fslType);
    }
  }
  if (Array.isArray(sampleValue)) {
    // Default Arrow inference doesn't handle list types
    return makeListVector(values as unknown[][]);
--- a/nodejs/lancedb/query.ts
+++ b/nodejs/lancedb/query.ts
@@ -5,6 +5,7 @@ import {
  Table as ArrowTable,
  type IntoVector,
  RecordBatch,
  extractVectorBuffer,
  fromBufferToRecordBatch,
  fromRecordBatchToBuffer,
  tableFromIPC,
@@ -661,10 +662,8 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
      const res = (async () => {
        try {
          const v = await vector;
          const arr = Float32Array.from(v);
          //
          // biome-ignore lint/suspicious/noExplicitAny: we need to get the `inner`, but js has no package scoping
-          const value: any = this.addQueryVector(arr);
+          const value: any = this.addQueryVector(v);
          const inner = value.inner as
            | NativeVectorQuery
            | Promise<NativeVectorQuery>;
@@ -676,7 +675,12 @@ export class VectorQuery extends StandardQueryBase<NativeVectorQuery> {
      return new VectorQuery(res);
    } else {
      super.doCall((inner) => {
-        inner.addQueryVector(Float32Array.from(vector));
+        const raw = Array.isArray(vector) ? null : extractVectorBuffer(vector);
        if (raw) {
          inner.addQueryVectorRaw(raw.data, raw.dtype);
        } else {
          inner.addQueryVector(Float32Array.from(vector as number[]));
        }
      });
      return this;
    }
@@ -765,14 +769,23 @@ export class Query extends StandardQueryBase<NativeQuery> {
   * a default `limit` of 10 will be used.  @see {@link Query#limit}
   */
  nearestTo(vector: IntoVector): VectorQuery {
    const callNearestTo = (
      inner: NativeQuery,
      resolved: Float32Array | Float64Array | Uint8Array | number[],
    ): NativeVectorQuery => {
      const raw = Array.isArray(resolved)
        ? null
        : extractVectorBuffer(resolved);
      if (raw) {
        return inner.nearestToRaw(raw.data, raw.dtype);
      }
      return inner.nearestTo(Float32Array.from(resolved as number[]));
    };
    if (this.inner instanceof Promise) {
      const nativeQuery = this.inner.then(async (inner) => {
-        if (vector instanceof Promise) {
+        const resolved = vector instanceof Promise ? await vector : vector;
-          const arr = await vector.then((v) => Float32Array.from(v));
+        return callNearestTo(inner, resolved);
          return inner.nearestTo(arr);
        } else {
          return inner.nearestTo(Float32Array.from(vector));
        }
      });
      return new VectorQuery(nativeQuery);
    }
@@ -780,10 +793,8 @@ export class Query extends StandardQueryBase<NativeQuery> {
      const res = (async () => {
        try {
          const v = await vector;
          const arr = Float32Array.from(v);
          //
          // biome-ignore lint/suspicious/noExplicitAny: we need to get the `inner`, but js has no package scoping
-          const value: any = this.nearestTo(arr);
+          const value: any = this.nearestTo(v);
          const inner = value.inner as
            | NativeVectorQuery
            | Promise<NativeVectorQuery>;
@@ -794,7 +805,7 @@ export class Query extends StandardQueryBase<NativeQuery> {
      })();
      return new VectorQuery(res);
    } else {
-      const vectorQuery = this.inner.nearestTo(Float32Array.from(vector));
+      const vectorQuery = callNearestTo(this.inner, vector);
      return new VectorQuery(vectorQuery);
    }
  }
--- a/nodejs/lancedb/table.ts
+++ b/nodejs/lancedb/table.ts
@@ -5,12 +5,15 @@ import {
  Table as ArrowTable,
  Data,
  DataType,
  Field,
  IntoVector,
  MultiVector,
  Schema,
  dataTypeToJson,
  fromDataToBuffer,
  fromTableToBuffer,
  isMultiVector,
  makeEmptyTable,
  tableFromIPC,
 } from "./arrow";
@@ -84,6 +87,16 @@ export interface OptimizeOptions {
   * tbl.optimize({cleanupOlderThan: new Date()});
   */
  cleanupOlderThan: Date;
  /**
   * Because they may be part of an in-progress transaction, files newer than
   * 7 days old are not deleted by default. If you are sure that there are no
   * in-progress transactions, then you can set this to true to delete all
   * files older than `cleanupOlderThan`.
   *
   * **WARNING**: This should only be set to true if you can guarantee that
   * no other process is currently working on this dataset. Otherwise the
   * dataset could be put into a corrupted state.
   */
  deleteUnverified: boolean;
 }
@@ -381,15 +394,16 @@ export abstract class Table {
  abstract vectorSearch(vector: IntoVector | MultiVector): VectorQuery;
  /**
   * Add new columns with defined values.
-   * @param {AddColumnsSql[]} newColumnTransforms pairs of column names and
+   * @param {AddColumnsSql[] | Field | Field[] | Schema} newColumnTransforms Either:
-   * the SQL expression to use to calculate the value of the new column. These
+   *   - An array of objects with column names and SQL expressions to calculate values
-   * expressions will be evaluated for each row in the table, and can
+   *   - A single Arrow Field defining one column with its data type (column will be initialized with null values)
-   * reference existing columns in the table.
+   *   - An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
   *   - An Arrow Schema defining columns with their data types (columns will be initialized with null values)
   * @returns {Promise<AddColumnsResult>} A promise that resolves to an object
   * containing the new version number of the table after adding the columns.
   */
  abstract addColumns(
-    newColumnTransforms: AddColumnsSql[],
+    newColumnTransforms: AddColumnsSql[] | Field | Field[] | Schema,
  ): Promise<AddColumnsResult>;
  /**
@@ -501,19 +515,7 @@ export abstract class Table {
   *  - Index: Optimizes the indices, adding new data to existing indices
   *
   *
-   *  Experimental API
+   *  The frequency an application should call optimize is based on the frequency of
   *  ----------------
   *
   *  The optimization process is undergoing active development and may change.
   *  Our goal with these changes is to improve the performance of optimization and
   *  reduce the complexity.
   *
   *  That being said, it is essential today to run optimize if you want the best
   *  performance.  It should be stable and safe to use in production, but it our
   *  hope that the API may be simplified (or not even need to be called) in the
   *  future.
   *
   *  The frequency an application shoudl call optimize is based on the frequency of
   *  data modifications.  If data is frequently added, deleted, or updated then
   *  optimize should be run frequently.  A good rule of thumb is to run optimize if
   *  you have added or modified 100,000 or more records or run more than 20 data
@@ -806,9 +808,40 @@ export class LocalTable extends Table {
  // TODO: Support BatchUDF
  async addColumns(
-    newColumnTransforms: AddColumnsSql[],
+    newColumnTransforms: AddColumnsSql[] | Field | Field[] | Schema,
  ): Promise<AddColumnsResult> {
-    return await this.inner.addColumns(newColumnTransforms);
+    // Handle single Field -> convert to array of Fields
    if (newColumnTransforms instanceof Field) {
      newColumnTransforms = [newColumnTransforms];
    }
    // Handle array of Fields -> convert to Schema
    if (
      Array.isArray(newColumnTransforms) &&
      newColumnTransforms.length > 0 &&
      newColumnTransforms[0] instanceof Field
    ) {
      const fields = newColumnTransforms as Field[];
      newColumnTransforms = new Schema(fields);
    }
    // Handle Schema -> use schema-based approach
    if (newColumnTransforms instanceof Schema) {
      const schema = newColumnTransforms;
      // Convert schema to buffer using Arrow IPC format
      const emptyTable = makeEmptyTable(schema);
      const schemaBuf = await fromTableToBuffer(emptyTable);
      return await this.inner.addColumnsWithSchema(schemaBuf);
    }
    // Handle SQL expressions (existing functionality)
    if (Array.isArray(newColumnTransforms)) {
      return await this.inner.addColumns(
        newColumnTransforms as AddColumnsSql[],
      );
    }
    throw new Error("Invalid input type for addColumns");
  }
  async alterColumns(
--- a/nodejs/npm/darwin-arm64/package.json
+++ b/nodejs/npm/darwin-arm64/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-darwin-arm64",
-	"version": "0.27.0-beta.3",
+	"version": "0.27.2",
 	"os": ["darwin"],
 	"cpu": ["arm64"],
 	"main": "lancedb.darwin-arm64.node",
--- a/nodejs/npm/linux-arm64-gnu/package.json
+++ b/nodejs/npm/linux-arm64-gnu/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-linux-arm64-gnu",
-	"version": "0.27.0-beta.3",
+	"version": "0.27.2",
 	"os": ["linux"],
 	"cpu": ["arm64"],
 	"main": "lancedb.linux-arm64-gnu.node",
--- a/nodejs/npm/linux-arm64-musl/package.json
+++ b/nodejs/npm/linux-arm64-musl/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-linux-arm64-musl",
-	"version": "0.27.0-beta.3",
+	"version": "0.27.2",
 	"os": ["linux"],
 	"cpu": ["arm64"],
 	"main": "lancedb.linux-arm64-musl.node",
--- a/nodejs/npm/linux-x64-gnu/package.json
+++ b/nodejs/npm/linux-x64-gnu/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-linux-x64-gnu",
-	"version": "0.27.0-beta.3",
+	"version": "0.27.2",
 	"os": ["linux"],
 	"cpu": ["x64"],
 	"main": "lancedb.linux-x64-gnu.node",
--- a/nodejs/npm/linux-x64-musl/package.json
+++ b/nodejs/npm/linux-x64-musl/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-linux-x64-musl",
-	"version": "0.27.0-beta.3",
+	"version": "0.27.2",
 	"os": ["linux"],
 	"cpu": ["x64"],
 	"main": "lancedb.linux-x64-musl.node",
--- a/nodejs/npm/win32-arm64-msvc/package.json
+++ b/nodejs/npm/win32-arm64-msvc/package.json
@@ -1,6 +1,6 @@
 {
  "name": "@lancedb/lancedb-win32-arm64-msvc",
-  "version": "0.27.0-beta.3",
+  "version": "0.27.2",
  "os": [
    "win32"
  ],
--- a/nodejs/npm/win32-x64-msvc/package.json
+++ b/nodejs/npm/win32-x64-msvc/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-win32-x64-msvc",
-	"version": "0.27.0-beta.3",
+	"version": "0.27.2",
 	"os": ["win32"],
 	"cpu": ["x64"],
 	"main": "lancedb.win32-x64-msvc.node",
--- a/nodejs/package-lock.json
+++ b/nodejs/package-lock.json
--- a/nodejs/package.json
+++ b/nodejs/package.json
@@ -11,7 +11,7 @@
    "ann"
  ],
  "private": false,
-  "version": "0.27.0-beta.3",
+  "version": "0.27.2",
  "main": "dist/index.js",
  "exports": {
    ".": "./dist/index.js",
--- a/nodejs/src/query.rs
+++ b/nodejs/src/query.rs
@@ -3,6 +3,12 @@
 use std::sync::Arc;
 use arrow_array::{
    Array, Float16Array as ArrowFloat16Array, Float32Array as ArrowFloat32Array,
    Float64Array as ArrowFloat64Array, UInt8Array as ArrowUInt8Array,
 };
 use arrow_buffer::ScalarBuffer;
 use half::f16;
 use lancedb::index::scalar::{
    BooleanQuery, BoostQuery, FtsQuery, FullTextSearchQuery, MatchQuery, MultiMatchQuery, Occur,
    Operator, PhraseQuery,
@@ -24,6 +30,33 @@ use crate::rerankers::RerankHybridCallbackArgs;
 use crate::rerankers::Reranker;
 use crate::util::{parse_distance_type, schema_to_buffer};
 fn bytes_to_arrow_array(data: Uint8Array, dtype: String) -> napi::Result<Arc<dyn Array>> {
    let buf = arrow_buffer::Buffer::from(data.to_vec());
    let num_bytes = buf.len();
    match dtype.as_str() {
        "float16" => {
            let scalar_buf = ScalarBuffer::<f16>::new(buf, 0, num_bytes / 2);
            Ok(Arc::new(ArrowFloat16Array::new(scalar_buf, None)))
        }
        "float32" => {
            let scalar_buf = ScalarBuffer::<f32>::new(buf, 0, num_bytes / 4);
            Ok(Arc::new(ArrowFloat32Array::new(scalar_buf, None)))
        }
        "float64" => {
            let scalar_buf = ScalarBuffer::<f64>::new(buf, 0, num_bytes / 8);
            Ok(Arc::new(ArrowFloat64Array::new(scalar_buf, None)))
        }
        "uint8" => {
            let scalar_buf = ScalarBuffer::<u8>::new(buf, 0, num_bytes);
            Ok(Arc::new(ArrowUInt8Array::new(scalar_buf, None)))
        }
        _ => Err(napi::Error::from_reason(format!(
            "Unsupported vector dtype: {}. Expected one of: float16, float32, float64, uint8",
            dtype
        ))),
    }
 }
 #[napi]
 pub struct Query {
    inner: LanceDbQuery,
@@ -78,6 +111,13 @@ impl Query {
        Ok(VectorQuery { inner })
    }
    #[napi]
    pub fn nearest_to_raw(&mut self, data: Uint8Array, dtype: String) -> Result<VectorQuery> {
        let array = bytes_to_arrow_array(data, dtype)?;
        let inner = self.inner.clone().nearest_to(array).default_error()?;
        Ok(VectorQuery { inner })
    }
    #[napi]
    pub fn fast_search(&mut self) {
        self.inner = self.inner.clone().fast_search();
@@ -163,6 +203,13 @@ impl VectorQuery {
        Ok(())
    }
    #[napi]
    pub fn add_query_vector_raw(&mut self, data: Uint8Array, dtype: String) -> Result<()> {
        let array = bytes_to_arrow_array(data, dtype)?;
        self.inner = self.inner.clone().add_query_vector(array).default_error()?;
        Ok(())
    }
    #[napi]
    pub fn distance_type(&mut self, distance_type: String) -> napi::Result<()> {
        let distance_type = parse_distance_type(distance_type)?;
--- a/nodejs/src/table.rs
+++ b/nodejs/src/table.rs
@@ -3,7 +3,7 @@
 use std::collections::HashMap;
-use lancedb::ipc::ipc_file_to_batches;
+use lancedb::ipc::{ipc_file_to_batches, ipc_file_to_schema};
 use lancedb::table::{
    AddDataMode, ColumnAlteration as LanceColumnAlteration, Duration, NewColumnTransform,
    OptimizeAction, OptimizeOptions, Table as LanceDbTable,
@@ -279,6 +279,23 @@ impl Table {
        Ok(res.into())
    }
    #[napi(catch_unwind)]
    pub async fn add_columns_with_schema(
        &self,
        schema_buf: Buffer,
    ) -> napi::Result<AddColumnsResult> {
        let schema = ipc_file_to_schema(schema_buf.to_vec())
            .map_err(|e| napi::Error::from_reason(format!("Failed to read IPC schema: {}", e)))?;
        let transforms = NewColumnTransform::AllNulls(schema);
        let res = self
            .inner_ref()?
            .add_columns(transforms, None)
            .await
            .default_error()?;
        Ok(res.into())
    }
    #[napi(catch_unwind)]
    pub async fn alter_columns(
        &self,
--- a/python/.bumpversion.toml
+++ b/python/.bumpversion.toml
@@ -1,5 +1,5 @@
 [tool.bumpversion]
-current_version = "0.30.0-beta.4"
+current_version = "0.30.2"
 parse = """(?x)
    (?P<major>0|[1-9]\\d*)\\.
    (?P<minor>0|[1-9]\\d*)\\.
--- a/python/.gitignore
+++ b/python/.gitignore
@@ -1,3 +1,5 @@
 # Test data created by some example tests
 data/
 _lancedb.pyd
 # macOS debug symbols bundle generated during build
 *.dSYM/
--- a/python/Cargo.toml
+++ b/python/Cargo.toml
@@ -1,6 +1,6 @@
 [package]
 name = "lancedb-python"
-version = "0.30.0-beta.4"
+version = "0.30.2"
 edition.workspace = true
 description = "Python bindings for LanceDB"
 license.workspace = true
@@ -23,6 +23,7 @@ lance-namespace.workspace = true
 lance-namespace-impls.workspace = true
 lance-io.workspace = true
 env_logger.workspace = true
 log.workspace = true
 pyo3 = { version = "0.26", features = ["extension-module", "abi3-py39"] }
 pyo3-async-runtimes = { version = "0.26", features = [
    "attributes",
--- a/python/README.md
+++ b/python/README.md
@@ -1,4 +1,4 @@
-# LanceDB
+# LanceDB Python SDK
 A Python library for [LanceDB](https://github.com/lancedb/lancedb).
--- a/python/pyproject.toml
+++ b/python/pyproject.toml
@@ -3,10 +3,10 @@ name = "lancedb"
 # version in Cargo.toml
 dynamic = ["version"]
 dependencies = [
-    "deprecation",
+    "deprecation>=2.1.0",
-    "numpy",
+    "numpy>=1.24.0",
    "overrides>=0.7; python_version<'3.12'",
-    "packaging",
+    "packaging>=23.0",
    "pyarrow>=16",
    "pydantic>=1.10",
    "tqdm>=4.27.0",
@@ -48,48 +48,48 @@ pylance = [
    "pylance>=4.0.0b7",
 ]
 tests = [
-    "aiohttp",
+    "aiohttp>=3.9.0",
-    "boto3",
+    "boto3>=1.28.57",
    "pandas>=1.4",
-    "pytest",
+    "pytest>=7.0",
-    "pytest-mock",
+    "pytest-mock>=3.10",
-    "pytest-asyncio",
+    "pytest-asyncio>=0.21",
-    "duckdb",
+    "duckdb>=0.9.0",
-    "pytz",
+    "pytz>=2023.3",
    "polars>=0.19, <=1.3.0",
-    "tantivy",
+    "tantivy>=0.20.0",
-    "pyarrow-stubs",
+    "pyarrow-stubs>=16.0",
    "pylance>=4.0.0b7",
-    "requests",
+    "requests>=2.31.0",
    "datafusion>=52,<53",
 ]
 dev = [
-    "ruff",
+    "ruff>=0.3.0",
-    "pre-commit",
+    "pre-commit>=3.5.0",
-    "pyright",
+    "pyright>=1.1.350",
    'typing-extensions>=4.0.0; python_version < "3.11"',
 ]
 docs = ["mkdocs", "mkdocs-jupyter", "mkdocs-material", "mkdocstrings-python"]
-clip = ["torch", "pillow", "open-clip-torch"]
+clip = ["torch", "pillow>=12.1.1", "open-clip-torch"]
-siglip = ["torch", "pillow", "transformers>=4.41.0","sentencepiece"]
+siglip = ["torch", "pillow>=12.1.1", "transformers>=4.41.0","sentencepiece"]
 embeddings = [
    "requests>=2.31.0",
    "openai>=1.6.1",
-    "sentence-transformers",
+    "sentence-transformers>=2.2.0",
-    "torch",
+    "torch>=2.0.0",
-    "pillow",
+    "pillow>=12.1.1",
-    "open-clip-torch",
+    "open-clip-torch>=2.20.0",
-    "cohere",
+    "cohere>=4.0",
    "colpali-engine>=0.3.10",
-    "huggingface_hub",
+    "huggingface_hub>=0.19.0",
-    "InstructorEmbedding",
+    "InstructorEmbedding>=1.0.1",
-    "google.generativeai",
+    "google.generativeai>=0.3.0",
    "boto3>=1.28.57",
-    "awscli>=1.29.57",
+    "awscli>=1.44.38",
    "botocore>=1.31.57",
    'ibm-watsonx-ai>=1.1.2; python_version >= "3.10"',
    "ollama>=0.3.0",
-    "sentencepiece"
+    "sentencepiece>=0.1.99"
 ]
 azure = ["adlfs>=2024.2.0"]
--- a/python/python/lancedb/init.py
+++ b/python/python/lancedb/init.py
@@ -18,6 +18,7 @@ from .db import AsyncConnection, DBConnection, LanceDBConnection
 from .io import StorageOptionsProvider
 from .remote import ClientConfig
 from .remote.db import RemoteDBConnection
 from .expr import Expr, col, lit, func
 from .schema import vector
 from .table import AsyncTable, Table
 from ._lancedb import Session
@@ -271,6 +272,10 @@ __all__ = [
    "AsyncConnection",
    "AsyncLanceNamespaceDBConnection",
    "AsyncTable",
    "col",
    "Expr",
    "func",
    "lit",
    "URI",
    "sanitize_uri",
    "vector",
--- a/python/python/lancedb/_lancedb.pyi
+++ b/python/python/lancedb/_lancedb.pyi
@@ -27,6 +27,32 @@ from .remote import ClientConfig
 IvfHnswPq: type[HnswPq] = HnswPq
 IvfHnswSq: type[HnswSq] = HnswSq
 class PyExpr:
    """A type-safe DataFusion expression node (Rust-side handle)."""
    def eq(self, other: "PyExpr") -> "PyExpr": ...
    def ne(self, other: "PyExpr") -> "PyExpr": ...
    def lt(self, other: "PyExpr") -> "PyExpr": ...
    def lte(self, other: "PyExpr") -> "PyExpr": ...
    def gt(self, other: "PyExpr") -> "PyExpr": ...
    def gte(self, other: "PyExpr") -> "PyExpr": ...
    def and_(self, other: "PyExpr") -> "PyExpr": ...
    def or_(self, other: "PyExpr") -> "PyExpr": ...
    def not_(self) -> "PyExpr": ...
    def add(self, other: "PyExpr") -> "PyExpr": ...
    def sub(self, other: "PyExpr") -> "PyExpr": ...
    def mul(self, other: "PyExpr") -> "PyExpr": ...
    def div(self, other: "PyExpr") -> "PyExpr": ...
    def lower(self) -> "PyExpr": ...
    def upper(self) -> "PyExpr": ...
    def contains(self, substr: "PyExpr") -> "PyExpr": ...
    def cast(self, data_type: pa.DataType) -> "PyExpr": ...
    def to_sql(self) -> str: ...
 def expr_col(name: str) -> PyExpr: ...
 def expr_lit(value: Union[bool, int, float, str]) -> PyExpr: ...
 def expr_func(name: str, args: List[PyExpr]) -> PyExpr: ...
 class Session:
    def __init__(
        self,
@@ -135,7 +161,10 @@ class Table:
    def close(self) -> None: ...
    async def schema(self) -> pa.Schema: ...
    async def add(
-        self, data: pa.RecordBatchReader, mode: Literal["append", "overwrite"]
+        self,
        data: pa.RecordBatchReader,
        mode: Literal["append", "overwrite"],
        progress: Optional[Any] = None,
    ) -> AddResult: ...
    async def update(
        self, updates: Dict[str, str], where: Optional[str]
@@ -166,6 +195,8 @@ class Table:
    async def checkout(self, version: Union[int, str]): ...
    async def checkout_latest(self): ...
    async def restore(self, version: Optional[Union[int, str]] = None): ...
    async def prewarm_index(self, index_name: str) -> None: ...
    async def prewarm_data(self, columns: Optional[List[str]] = None) -> None: ...
    async def list_indices(self) -> list[IndexConfig]: ...
    async def delete(self, filter: str) -> DeleteResult: ...
    async def add_columns(self, columns: list[tuple[str, str]]) -> AddColumnsResult: ...
@@ -220,7 +251,9 @@ class RecordBatchStream:
 class Query:
    def where(self, filter: str): ...
-    def select(self, columns: Tuple[str, str]): ...
+    def where_expr(self, expr: PyExpr): ...
    def select(self, columns: List[Tuple[str, str]]): ...
    def select_expr(self, columns: List[Tuple[str, PyExpr]]): ...
    def select_columns(self, columns: List[str]): ...
    def limit(self, limit: int): ...
    def offset(self, offset: int): ...
@@ -246,7 +279,9 @@ class TakeQuery:
 class FTSQuery:
    def where(self, filter: str): ...
-    def select(self, columns: List[str]): ...
+    def where_expr(self, expr: PyExpr): ...
    def select(self, columns: List[Tuple[str, str]]): ...
    def select_expr(self, columns: List[Tuple[str, PyExpr]]): ...
    def limit(self, limit: int): ...
    def offset(self, offset: int): ...
    def fast_search(self): ...
@@ -265,7 +300,9 @@ class VectorQuery:
    async def output_schema(self) -> pa.Schema: ...
    async def execute(self) -> RecordBatchStream: ...
    def where(self, filter: str): ...
-    def select(self, columns: List[str]): ...
+    def where_expr(self, expr: PyExpr): ...
    def select(self, columns: List[Tuple[str, str]]): ...
    def select_expr(self, columns: List[Tuple[str, PyExpr]]): ...
    def select_with_projection(self, columns: Tuple[str, str]): ...
    def limit(self, limit: int): ...
    def offset(self, offset: int): ...
@@ -282,7 +319,9 @@ class VectorQuery:
 class HybridQuery:
    def where(self, filter: str): ...
-    def select(self, columns: List[str]): ...
+    def where_expr(self, expr: PyExpr): ...
    def select(self, columns: List[Tuple[str, str]]): ...
    def select_expr(self, columns: List[Tuple[str, PyExpr]]): ...
    def limit(self, limit: int): ...
    def offset(self, offset: int): ...
    def fast_search(self): ...
--- a/python/python/lancedb/embeddings/utils.py
+++ b/python/python/lancedb/embeddings/utils.py
@@ -10,6 +10,7 @@ import sys
 import threading
 import time
 import urllib.error
 import urllib.request
 import weakref
 import logging
 from functools import wraps
--- a/python/python/lancedb/expr.py
+++ b/python/python/lancedb/expr.py
@@ -0,0 +1,298 @@
 # SPDX-License-Identifier: Apache-2.0
 # SPDX-FileCopyrightText: Copyright The LanceDB Authors
 """Type-safe expression builder for filters and projections.
 Instead of writing raw SQL strings you can build expressions with Python
 operators::
    from lancedb.expr import col, lit
    # filter: age > 18 AND status = 'active'
    filt = (col("age") > lit(18)) & (col("status") == lit("active"))
    # projection: compute a derived column
    proj = {"score": col("raw_score") * lit(1.5)}
    table.search().where(filt).select(proj).to_list()
 """
 from __future__ import annotations
 from typing import Union
 import pyarrow as pa
 from lancedb._lancedb import PyExpr, expr_col, expr_lit, expr_func
 __all__ = ["Expr", "col", "lit", "func"]
 _STR_TO_PA_TYPE: dict = {
    "bool": pa.bool_(),
    "boolean": pa.bool_(),
    "int8": pa.int8(),
    "int16": pa.int16(),
    "int32": pa.int32(),
    "int64": pa.int64(),
    "uint8": pa.uint8(),
    "uint16": pa.uint16(),
    "uint32": pa.uint32(),
    "uint64": pa.uint64(),
    "float16": pa.float16(),
    "float32": pa.float32(),
    "float": pa.float32(),
    "float64": pa.float64(),
    "double": pa.float64(),
    "string": pa.string(),
    "utf8": pa.string(),
    "str": pa.string(),
    "large_string": pa.large_utf8(),
    "large_utf8": pa.large_utf8(),
    "date32": pa.date32(),
    "date": pa.date32(),
    "date64": pa.date64(),
 }
 def _coerce(value: "ExprLike") -> "Expr":
    """Return *value* as an :class:`Expr`, wrapping plain Python values via
    :func:`lit` if needed."""
    if isinstance(value, Expr):
        return value
    return lit(value)
 # Type alias used in annotations.
 ExprLike = Union["Expr", bool, int, float, str]
 class Expr:
    """A type-safe expression node.
    Construct instances with :func:`col` and :func:`lit`, then combine them
    using Python operators or the named methods below.
    Examples
    --------
    >>> from lancedb.expr import col, lit
    >>> filt = (col("age") > lit(18)) & (col("name").lower() == lit("alice"))
    >>> proj = {"double": col("x") * lit(2)}
    """
    # Make Expr unhashable so that == returns an Expr rather than being used
    # for dict keys / set membership.
    __hash__ = None  # type: ignore[assignment]
    def __init__(self, inner: PyExpr) -> None:
        self._inner = inner
    # ── comparisons ──────────────────────────────────────────────────────────
    def __eq__(self, other: ExprLike) -> "Expr":  # type: ignore[override]
        """Equal to (``col("x") == 1``)."""
        return Expr(self._inner.eq(_coerce(other)._inner))
    def __ne__(self, other: ExprLike) -> "Expr":  # type: ignore[override]
        """Not equal to (``col("x") != 1``)."""
        return Expr(self._inner.ne(_coerce(other)._inner))
    def __lt__(self, other: ExprLike) -> "Expr":
        """Less than (``col("x") < 1``)."""
        return Expr(self._inner.lt(_coerce(other)._inner))
    def __le__(self, other: ExprLike) -> "Expr":
        """Less than or equal to (``col("x") <= 1``)."""
        return Expr(self._inner.lte(_coerce(other)._inner))
    def __gt__(self, other: ExprLike) -> "Expr":
        """Greater than (``col("x") > 1``)."""
        return Expr(self._inner.gt(_coerce(other)._inner))
    def __ge__(self, other: ExprLike) -> "Expr":
        """Greater than or equal to (``col("x") >= 1``)."""
        return Expr(self._inner.gte(_coerce(other)._inner))
    # ── logical ──────────────────────────────────────────────────────────────
    def __and__(self, other: "Expr") -> "Expr":
        """Logical AND (``expr_a & expr_b``)."""
        return Expr(self._inner.and_(_coerce(other)._inner))
    def __or__(self, other: "Expr") -> "Expr":
        """Logical OR (``expr_a | expr_b``)."""
        return Expr(self._inner.or_(_coerce(other)._inner))
    def __invert__(self) -> "Expr":
        """Logical NOT (``~expr``)."""
        return Expr(self._inner.not_())
    # ── arithmetic ───────────────────────────────────────────────────────────
    def __add__(self, other: ExprLike) -> "Expr":
        """Add (``col("x") + 1``)."""
        return Expr(self._inner.add(_coerce(other)._inner))
    def __radd__(self, other: ExprLike) -> "Expr":
        """Right-hand add (``1 + col("x")``)."""
        return Expr(_coerce(other)._inner.add(self._inner))
    def __sub__(self, other: ExprLike) -> "Expr":
        """Subtract (``col("x") - 1``)."""
        return Expr(self._inner.sub(_coerce(other)._inner))
    def __rsub__(self, other: ExprLike) -> "Expr":
        """Right-hand subtract (``1 - col("x")``)."""
        return Expr(_coerce(other)._inner.sub(self._inner))
    def __mul__(self, other: ExprLike) -> "Expr":
        """Multiply (``col("x") * 2``)."""
        return Expr(self._inner.mul(_coerce(other)._inner))
    def __rmul__(self, other: ExprLike) -> "Expr":
        """Right-hand multiply (``2 * col("x")``)."""
        return Expr(_coerce(other)._inner.mul(self._inner))
    def __truediv__(self, other: ExprLike) -> "Expr":
        """Divide (``col("x") / 2``)."""
        return Expr(self._inner.div(_coerce(other)._inner))
    def __rtruediv__(self, other: ExprLike) -> "Expr":
        """Right-hand divide (``1 / col("x")``)."""
        return Expr(_coerce(other)._inner.div(self._inner))
    # ── string methods ───────────────────────────────────────────────────────
    def lower(self) -> "Expr":
        """Convert string column values to lowercase."""
        return Expr(self._inner.lower())
    def upper(self) -> "Expr":
        """Convert string column values to uppercase."""
        return Expr(self._inner.upper())
    def contains(self, substr: "ExprLike") -> "Expr":
        """Return True where the string contains *substr*."""
        return Expr(self._inner.contains(_coerce(substr)._inner))
    # ── type cast ────────────────────────────────────────────────────────────
    def cast(self, data_type: Union[str, "pa.DataType"]) -> "Expr":
        """Cast values to *data_type*.
        Parameters
        ----------
        data_type:
            A PyArrow ``DataType`` (e.g. ``pa.int32()``) or one of the type
            name strings: ``"bool"``, ``"int8"``, ``"int16"``, ``"int32"``,
            ``"int64"``, ``"uint8"``–``"uint64"``, ``"float32"``,
            ``"float64"``, ``"string"``, ``"date32"``, ``"date64"``.
        """
        if isinstance(data_type, str):
            try:
                data_type = _STR_TO_PA_TYPE[data_type]
            except KeyError:
                raise ValueError(
                    f"unsupported data type: '{data_type}'. Supported: "
                    f"{', '.join(_STR_TO_PA_TYPE)}"
                )
        return Expr(self._inner.cast(data_type))
    # ── named comparison helpers (alternative to operators) ──────────────────
    def eq(self, other: ExprLike) -> "Expr":
        """Equal to."""
        return self.__eq__(other)
    def ne(self, other: ExprLike) -> "Expr":
        """Not equal to."""
        return self.__ne__(other)
    def lt(self, other: ExprLike) -> "Expr":
        """Less than."""
        return self.__lt__(other)
    def lte(self, other: ExprLike) -> "Expr":
        """Less than or equal to."""
        return self.__le__(other)
    def gt(self, other: ExprLike) -> "Expr":
        """Greater than."""
        return self.__gt__(other)
    def gte(self, other: ExprLike) -> "Expr":
        """Greater than or equal to."""
        return self.__ge__(other)
    def and_(self, other: "Expr") -> "Expr":
        """Logical AND."""
        return self.__and__(other)
    def or_(self, other: "Expr") -> "Expr":
        """Logical OR."""
        return self.__or__(other)
    # ── utilities ────────────────────────────────────────────────────────────
    def to_sql(self) -> str:
        """Render the expression as a SQL string (useful for debugging)."""
        return self._inner.to_sql()
    def __repr__(self) -> str:
        return f"Expr({self._inner.to_sql()})"
 # ── free functions ────────────────────────────────────────────────────────────
 def col(name: str) -> Expr:
    """Reference a table column by name.
    Parameters
    ----------
    name:
        The column name.
    Examples
    --------
    >>> from lancedb.expr import col, lit
    >>> col("age") > lit(18)
    Expr((age > 18))
    """
    return Expr(expr_col(name))
 def lit(value: Union[bool, int, float, str]) -> Expr:
    """Create a literal (constant) value expression.
    Parameters
    ----------
    value:
        A Python ``bool``, ``int``, ``float``, or ``str``.
    Examples
    --------
    >>> from lancedb.expr import col, lit
    >>> col("price") * lit(1.1)
    Expr((price * 1.1))
    """
    return Expr(expr_lit(value))
 def func(name: str, *args: ExprLike) -> Expr:
    """Call an arbitrary SQL function by name.
    Parameters
    ----------
    name:
        The SQL function name (e.g. ``"lower"``, ``"upper"``).
    *args:
        The function arguments as :class:`Expr` or plain Python literals.
    Examples
    --------
    >>> from lancedb.expr import col, func
    >>> func("lower", col("name"))
    Expr(lower(name))
    """
    inner_args = [_coerce(a)._inner for a in args]
    return Expr(expr_func(name, inner_args))
--- a/python/python/lancedb/query.py
+++ b/python/python/lancedb/query.py
@@ -38,6 +38,7 @@ from .rerankers.base import Reranker
 from .rerankers.rrf import RRFReranker
 from .rerankers.util import check_reranker_result
 from .util import flatten_columns
 from .expr import Expr
 from lancedb._lancedb import fts_query_to_json
 from typing_extensions import Annotated
@@ -70,7 +71,7 @@ def ensure_vector_query(
 ) -> Union[List[float], List[List[float]], pa.Array, List[pa.Array]]:
    if isinstance(val, list):
        if len(val) == 0:
-            return ValueError("Vector query must be a non-empty list")
+            raise ValueError("Vector query must be a non-empty list")
        sample = val[0]
    else:
        if isinstance(val, float):
@@ -83,7 +84,7 @@ def ensure_vector_query(
        return val
    if isinstance(sample, list):
        if len(sample) == 0:
-            return ValueError("Vector query must be a non-empty list")
+            raise ValueError("Vector query must be a non-empty list")
        if isinstance(sample[0], float):
            # val is list of list of floats
            return val
@@ -449,8 +450,8 @@ class Query(pydantic.BaseModel):
        ensure_vector_query,
    ] = None
-    # sql filter to refine the query with
+    # sql filter or type-safe Expr to refine the query with
-    filter: Optional[str] = None
+    filter: Optional[Union[str, Expr]] = None
    # if True then apply the filter after vector search
    postfilter: Optional[bool] = None
@@ -464,8 +465,8 @@ class Query(pydantic.BaseModel):
    # distance type to use for vector search
    distance_type: Optional[str] = None
-    # which columns to return in the results
+    # which columns to return in the results (dict values may be str or Expr)
-    columns: Optional[Union[List[str], Dict[str, str]]] = None
+    columns: Optional[Union[List[str], Dict[str, Union[str, Expr]]]] = None
    # minimum number of IVF partitions to search
    #
@@ -856,14 +857,15 @@ class LanceQueryBuilder(ABC):
            self._offset = offset
        return self
-    def select(self, columns: Union[list[str], dict[str, str]]) -> Self:
+    def select(self, columns: Union[list[str], dict[str, Union[str, Expr]]]) -> Self:
        """Set the columns to return.
        Parameters
        ----------
-        columns: list of str, or dict of str to str default None
+        columns: list of str, or dict of str to str or Expr
            List of column names to be fetched.
-            Or a dictionary of column names to SQL expressions.
+            Or a dictionary of column names to SQL expressions or
            :class:`~lancedb.expr.Expr` objects.
            All columns are fetched if None or unspecified.
        Returns
@@ -877,15 +879,15 @@ class LanceQueryBuilder(ABC):
            raise ValueError("columns must be a list or a dictionary")
        return self
-    def where(self, where: str, prefilter: bool = True) -> Self:
+    def where(self, where: Union[str, Expr], prefilter: bool = True) -> Self:
        """Set the where clause.
        Parameters
        ----------
-        where: str
+        where: str or :class:`~lancedb.expr.Expr`
-            The where clause which is a valid SQL where clause. See
+            The filter condition.  Can be a SQL string or a type-safe
-            `Lance filter pushdown <https://lance.org/guide/read_and_write#filter-push-down>`_
+            :class:`~lancedb.expr.Expr` built with :func:`~lancedb.expr.col`
-            for valid SQL expressions.
+            and :func:`~lancedb.expr.lit`.
        prefilter: bool, default True
            If True, apply the filter before vector search, otherwise the
            filter is applied on the result of vector search.
@@ -1355,15 +1357,17 @@ class LanceVectorQueryBuilder(LanceQueryBuilder):
        return result_set
-    def where(self, where: str, prefilter: bool = None) -> LanceVectorQueryBuilder:
+    def where(
        self, where: Union[str, Expr], prefilter: bool = None
    ) -> LanceVectorQueryBuilder:
        """Set the where clause.
        Parameters
        ----------
-        where: str
+        where: str or :class:`~lancedb.expr.Expr`
-            The where clause which is a valid SQL where clause. See
+            The filter condition.  Can be a SQL string or a type-safe
-            `Lance filter pushdown <https://lance.org/guide/read_and_write#filter-push-down>`_
+            :class:`~lancedb.expr.Expr` built with :func:`~lancedb.expr.col`
-            for valid SQL expressions.
+            and :func:`~lancedb.expr.lit`.
        prefilter: bool, default True
            If True, apply the filter before vector search, otherwise the
            filter is applied on the result of vector search.
@@ -2205,8 +2209,8 @@ class LanceHybridQueryBuilder(LanceQueryBuilder):
            self._vector_query.select(self._columns)
            self._fts_query.select(self._columns)
        if self._where:
-            self._vector_query.where(self._where, self._postfilter)
+            self._vector_query.where(self._where, not self._postfilter)
-            self._fts_query.where(self._where, self._postfilter)
+            self._fts_query.where(self._where, not self._postfilter)
        if self._with_row_id:
            self._vector_query.with_row_id(True)
            self._fts_query.with_row_id(True)
@@ -2286,10 +2290,20 @@ class AsyncQueryBase(object):
        """
        if isinstance(columns, list) and all(isinstance(c, str) for c in columns):
            self._inner.select_columns(columns)
-        elif isinstance(columns, dict) and all(
+        elif isinstance(columns, dict) and all(isinstance(k, str) for k in columns):
-            isinstance(k, str) and isinstance(v, str) for k, v in columns.items()
+            if any(isinstance(v, Expr) for v in columns.values()):
-        ):
+                # At least one value is an Expr — use the type-safe path.
-            self._inner.select(list(columns.items()))
+                from .expr import _coerce
                pairs = [(k, _coerce(v)._inner) for k, v in columns.items()]
                self._inner.select_expr(pairs)
            elif all(isinstance(v, str) for v in columns.values()):
                self._inner.select(list(columns.items()))
            else:
                raise TypeError(
                    "dict values must be str or Expr, got "
                    + str({k: type(v) for k, v in columns.items()})
                )
        else:
            raise TypeError("columns must be a list of column names or a dict")
        return self
@@ -2529,11 +2543,13 @@ class AsyncStandardQuery(AsyncQueryBase):
        """
        super().__init__(inner)
-    def where(self, predicate: str) -> Self:
+    def where(self, predicate: Union[str, Expr]) -> Self:
        """
        Only return rows matching the given predicate
-        The predicate should be supplied as an SQL query string.
+        The predicate can be a SQL string or a type-safe
        :class:`~lancedb.expr.Expr` built with :func:`~lancedb.expr.col`
        and :func:`~lancedb.expr.lit`.
        Examples
        --------
@@ -2545,7 +2561,10 @@ class AsyncStandardQuery(AsyncQueryBase):
        Filtering performance can often be improved by creating a scalar index
        on the filter column(s).
        """
-        self._inner.where(predicate)
+        if isinstance(predicate, Expr):
            self._inner.where_expr(predicate._inner)
        else:
            self._inner.where(predicate)
        return self
    def limit(self, limit: int) -> Self:
--- a/python/python/lancedb/remote/db.py
+++ b/python/python/lancedb/remote/db.py
@@ -568,4 +568,4 @@ class RemoteDBConnection(DBConnection):
    async def close(self):
        """Close the connection to the database."""
-        self._client.close()
+        self._conn.close()
--- a/python/python/lancedb/remote/table.py
+++ b/python/python/lancedb/remote/table.py
@@ -4,7 +4,7 @@
 from datetime import timedelta
 import logging
 from functools import cached_property
-from typing import Dict, Iterable, List, Optional, Union, Literal
+from typing import Any, Callable, Dict, Iterable, List, Optional, Union, Literal
 import warnings
 from lancedb._lancedb import (
@@ -35,6 +35,7 @@ import pyarrow as pa
 from lancedb.common import DATA, VEC, VECTOR_COLUMN_NAME
 from lancedb.merge import LanceMergeInsertBuilder
 from lancedb.embeddings import EmbeddingFunctionRegistry
 from lancedb.table import _normalize_progress
 from ..query import LanceVectorQueryBuilder, LanceQueryBuilder, LanceTakeQueryBuilder
 from ..table import AsyncTable, IndexStatistics, Query, Table, Tags
@@ -308,6 +309,7 @@ class RemoteTable(Table):
        mode: str = "append",
        on_bad_vectors: str = "error",
        fill_value: float = 0.0,
        progress: Optional[Union[bool, Callable, Any]] = None,
    ) -> AddResult:
        """Add more data to the [Table](Table). It has the same API signature as
        the OSS version.
@@ -330,17 +332,29 @@ class RemoteTable(Table):
            One of "error", "drop", "fill".
        fill_value: float, default 0.
            The value to use when filling vectors. Only used if on_bad_vectors="fill".
        progress: bool, callable, or tqdm-like, optional
            A callback or tqdm-compatible progress bar. See
            :meth:`Table.add` for details.
        Returns
        -------
        AddResult
            An object containing the new version number of the table after adding data.
        """
-        return LOOP.run(
+        progress, owns = _normalize_progress(progress)
-            self._table.add(
+        try:
-                data, mode=mode, on_bad_vectors=on_bad_vectors, fill_value=fill_value
+            return LOOP.run(
                self._table.add(
                    data,
                    mode=mode,
                    on_bad_vectors=on_bad_vectors,
                    fill_value=fill_value,
                    progress=progress,
                )
            )
-        )
+        finally:
            if owns:
                progress.close()
    def search(
        self,
@@ -640,6 +654,45 @@ class RemoteTable(Table):
    def drop_index(self, index_name: str):
        return LOOP.run(self._table.drop_index(index_name))
    def prewarm_index(self, name: str) -> None:
        """Prewarm an index in the table.
        This is a hint to the database that the index will be accessed in the
        future and should be loaded into memory if possible.  This can reduce
        cold-start latency for subsequent queries.
        This call initiates prewarming and returns once the request is accepted.
        It is idempotent and safe to call from multiple clients concurrently.
        Parameters
        ----------
        name: str
            The name of the index to prewarm
        """
        return LOOP.run(self._table.prewarm_index(name))
    def prewarm_data(self, columns: Optional[List[str]] = None) -> None:
        """Prewarm data for the table.
        This is a hint to the database that the given columns will be accessed
        in the future and the database should prefetch the data if possible.
        Currently only supported on remote tables.
        This call initiates prewarming and returns once the request is accepted.
        It is idempotent and safe to call from multiple clients concurrently.
        This operation has a large upfront cost but can speed up future queries
        that need to fetch the given columns.  Large columns such as embeddings
        or binary data may not be practical to prewarm.  This feature is intended
        for workloads that issue many queries against the same columns.
        Parameters
        ----------
        columns: list of str, optional
            The columns to prewarm. If None, all columns are prewarmed.
        """
        return LOOP.run(self._table.prewarm_data(columns))
    def wait_for_index(
        self, index_names: Iterable[str], timeout: timedelta = timedelta(seconds=300)
    ):
--- a/python/python/lancedb/table.py
+++ b/python/python/lancedb/table.py
@@ -14,6 +14,7 @@ from functools import cached_property
 from typing import (
    TYPE_CHECKING,
    Any,
    Callable,
    Dict,
    Iterable,
    List,
@@ -277,7 +278,7 @@ def _sanitize_data(
    if metadata:
        new_metadata = target_schema.metadata or {}
-        new_metadata = new_metadata.update(metadata)
+        new_metadata.update(metadata)
        target_schema = target_schema.with_metadata(new_metadata)
    _validate_schema(target_schema)
@@ -556,6 +557,21 @@ def _table_uri(base: str, table_name: str) -> str:
    return join_uri(base, f"{table_name}.lance")
 def _normalize_progress(progress):
    """Normalize a ``progress`` parameter for :meth:`Table.add`.
    Returns ``(progress_obj, owns)`` where *owns* is True when we created a
    tqdm bar that the caller must close.
    """
    if progress is True:
        from tqdm.auto import tqdm
        return tqdm(unit=" rows"), True
    if progress is False or progress is None:
        return None, False
    return progress, False
 class Table(ABC):
    """
    A Table is a collection of Records in a LanceDB Database.
@@ -974,6 +990,7 @@ class Table(ABC):
        mode: AddMode = "append",
        on_bad_vectors: OnBadVectorsType = "error",
        fill_value: float = 0.0,
        progress: Optional[Union[bool, Callable, Any]] = None,
    ) -> AddResult:
        """Add more data to the [Table](Table).
@@ -995,6 +1012,29 @@ class Table(ABC):
            One of "error", "drop", "fill".
        fill_value: float, default 0.
            The value to use when filling vectors. Only used if on_bad_vectors="fill".
        progress: bool, callable, or tqdm-like, optional
            Progress reporting during the add operation. Can be:
            - ``True`` to automatically create and display a tqdm progress
              bar (requires ``tqdm`` to be installed)::
                table.add(data, progress=True)
            - A **callable** that receives a dict with keys ``output_rows``,
              ``output_bytes``, ``total_rows``, ``elapsed_seconds``,
              ``active_tasks``, ``total_tasks``, and ``done``::
                def on_progress(p):
                    print(f"{p['output_rows']}/{p['total_rows']} rows, "
                          f"{p['active_tasks']}/{p['total_tasks']} workers")
                table.add(data, progress=on_progress)
            - A **tqdm-compatible** progress bar whose ``total`` and
              ``update()`` will be called automatically. The postfix shows
              write throughput (MB/s) and active worker count::
                with tqdm() as pbar:
                    table.add(data, progress=pbar)
        Returns
        -------
@@ -1506,22 +1546,17 @@ class Table(ABC):
            in-progress operation (e.g. appending new data) and these files will not
            be deleted unless they are at least 7 days old. If delete_unverified is True
            then these files will be deleted regardless of their age.
            .. warning::
                This should only be set to True if you can guarantee that no other
                process is currently working on this dataset. Otherwise the dataset
                could be put into a corrupted state.
        retrain: bool, default False
            This parameter is no longer used and is deprecated.
-        Experimental API
+        The frequency an application should call optimize is based on the frequency of
        ----------------
        The optimization process is undergoing active development and may change.
        Our goal with these changes is to improve the performance of optimization and
        reduce the complexity.
        That being said, it is essential today to run optimize if you want the best
        performance.  It should be stable and safe to use in production, but it our
        hope that the API may be simplified (or not even need to be called) in the
        future.
        The frequency an application shoudl call optimize is based on the frequency of
        data modifications.  If data is frequently added, deleted, or updated then
        optimize should be run frequently.  A good rule of thumb is to run optimize if
        you have added or modified 100,000 or more records or run more than 20 data
@@ -2219,12 +2254,18 @@ class LanceTable(Table):
    def prewarm_index(self, name: str) -> None:
        """
-        Prewarms an index in the table
+        Prewarm an index in the table.
-        This loads the entire index into memory
+        This is a hint to the database that the index will be accessed in the
        future and should be loaded into memory if possible.  This can reduce
        cold-start latency for subsequent queries.
-        If the index does not fit into the available cache this call
+        This call initiates prewarming and returns once the request is accepted.
-        may be wasteful
+        It is idempotent and safe to call from multiple clients concurrently.
        It is generally wasteful to call this if the index does not fit into the
        available cache.  Not all index types support prewarming; unsupported
        indices will silently ignore the request.
        Parameters
        ----------
@@ -2233,6 +2274,29 @@ class LanceTable(Table):
        """
        return LOOP.run(self._table.prewarm_index(name))
    def prewarm_data(self, columns: Optional[List[str]] = None) -> None:
        """
        Prewarm data for the table.
        This is a hint to the database that the given columns will be accessed
        in the future and the database should prefetch the data if possible.
        Currently only supported on remote tables.
        This call initiates prewarming and returns once the request is accepted.
        It is idempotent and safe to call from multiple clients concurrently.
        This operation has a large upfront cost but can speed up future queries
        that need to fetch the given columns.  Large columns such as embeddings
        or binary data may not be practical to prewarm.  This feature is intended
        for workloads that issue many queries against the same columns.
        Parameters
        ----------
        columns: list of str, optional
            The columns to prewarm. If None, all columns are prewarmed.
        """
        return LOOP.run(self._table.prewarm_data(columns))
    def wait_for_index(
        self, index_names: Iterable[str], timeout: timedelta = timedelta(seconds=300)
    ) -> None:
@@ -2468,6 +2532,7 @@ class LanceTable(Table):
        mode: AddMode = "append",
        on_bad_vectors: OnBadVectorsType = "error",
        fill_value: float = 0.0,
        progress: Optional[Union[bool, Callable, Any]] = None,
    ) -> AddResult:
        """Add data to the table.
        If vector columns are missing and the table
@@ -2486,17 +2551,29 @@ class LanceTable(Table):
            One of "error", "drop", "fill", "null".
        fill_value: float, default 0.
            The value to use when filling vectors. Only used if on_bad_vectors="fill".
        progress: bool, callable, or tqdm-like, optional
            A callback or tqdm-compatible progress bar. See
            :meth:`Table.add` for details.
        Returns
        -------
        int
            The number of vectors in the table.
        """
-        return LOOP.run(
+        progress, owns = _normalize_progress(progress)
-            self._table.add(
+        try:
-                data, mode=mode, on_bad_vectors=on_bad_vectors, fill_value=fill_value
+            return LOOP.run(
                self._table.add(
                    data,
                    mode=mode,
                    on_bad_vectors=on_bad_vectors,
                    fill_value=fill_value,
                    progress=progress,
                )
            )
-        )
+        finally:
            if owns:
                progress.close()
    def merge(
        self,
@@ -3018,22 +3095,17 @@ class LanceTable(Table):
            in-progress operation (e.g. appending new data) and these files will not
            be deleted unless they are at least 7 days old. If delete_unverified is True
            then these files will be deleted regardless of their age.
            .. warning::
                This should only be set to True if you can guarantee that no other
                process is currently working on this dataset. Otherwise the dataset
                could be put into a corrupted state.
        retrain: bool, default False
            This parameter is no longer used and is deprecated.
-        Experimental API
+        The frequency an application should call optimize is based on the frequency of
        ----------------
        The optimization process is undergoing active development and may change.
        Our goal with these changes is to improve the performance of optimization and
        reduce the complexity.
        That being said, it is essential today to run optimize if you want the best
        performance.  It should be stable and safe to use in production, but it our
        hope that the API may be simplified (or not even need to be called) in the
        future.
        The frequency an application shoudl call optimize is based on the frequency of
        data modifications.  If data is frequently added, deleted, or updated then
        optimize should be run frequently.  A good rule of thumb is to run optimize if
        you have added or modified 100,000 or more records or run more than 20 data
@@ -3634,19 +3706,47 @@ class AsyncTable:
        """
        Prewarm an index in the table.
        This is a hint to the database that the index will be accessed in the
        future and should be loaded into memory if possible.  This can reduce
        cold-start latency for subsequent queries.
        This call initiates prewarming and returns once the request is accepted.
        It is idempotent and safe to call from multiple clients concurrently.
        It is generally wasteful to call this if the index does not fit into the
        available cache.  Not all index types support prewarming; unsupported
        indices will silently ignore the request.
        Parameters
        ----------
        name: str
            The name of the index to prewarm
        Notes
        -----
        This will load the index into memory.  This may reduce the cold-start time for
        future queries.  If the index does not fit in the cache then this call may be
        wasteful.
        """
        await self._inner.prewarm_index(name)
    async def prewarm_data(self, columns: Optional[List[str]] = None) -> None:
        """
        Prewarm data for the table.
        This is a hint to the database that the given columns will be accessed
        in the future and the database should prefetch the data if possible.
        Currently only supported on remote tables.
        This call initiates prewarming and returns once the request is accepted.
        It is idempotent and safe to call from multiple clients concurrently.
        This operation has a large upfront cost but can speed up future queries
        that need to fetch the given columns.  Large columns such as embeddings
        or binary data may not be practical to prewarm.  This feature is intended
        for workloads that issue many queries against the same columns.
        Parameters
        ----------
        columns: list of str, optional
            The columns to prewarm. If None, all columns are prewarmed.
        """
        await self._inner.prewarm_data(columns)
    async def wait_for_index(
        self, index_names: Iterable[str], timeout: timedelta = timedelta(seconds=300)
    ) -> None:
@@ -3722,6 +3822,7 @@ class AsyncTable:
        mode: Optional[Literal["append", "overwrite"]] = "append",
        on_bad_vectors: Optional[OnBadVectorsType] = None,
        fill_value: Optional[float] = None,
        progress: Optional[Union[bool, Callable, Any]] = None,
    ) -> AddResult:
        """Add more data to the [Table](Table).
@@ -3743,6 +3844,9 @@ class AsyncTable:
            One of "error", "drop", "fill", "null".
        fill_value: float, default 0.
            The value to use when filling vectors. Only used if on_bad_vectors="fill".
        progress: callable or tqdm-like, optional
            A callback or tqdm-compatible progress bar. See
            :meth:`Table.add` for details.
        """
        schema = await self.schema()
@@ -3753,7 +3857,13 @@ class AsyncTable:
        # _santitize_data is an old code path, but we will use it until the
        # new code path is ready.
-        if on_bad_vectors != "error" or (
+        if mode == "overwrite":
            # For overwrite, apply the same preprocessing as create_table
            # so vector columns are inferred as FixedSizeList.
            data, _ = sanitize_create_table(
                data, None, on_bad_vectors=on_bad_vectors, fill_value=fill_value
            )
        elif on_bad_vectors != "error" or (
            schema.metadata is not None and b"embedding_functions" in schema.metadata
        ):
            data = _sanitize_data(
@@ -3766,8 +3876,9 @@ class AsyncTable:
            )
        _register_optional_converters()
        data = to_scannable(data)
        progress, owns = _normalize_progress(progress)
        try:
-            return await self._inner.add(data, mode or "append")
+            return await self._inner.add(data, mode or "append", progress=progress)
        except RuntimeError as e:
            if "Cast error" in str(e):
                raise ValueError(e)
@@ -3775,6 +3886,9 @@ class AsyncTable:
                raise ValueError(e)
            else:
                raise
        finally:
            if owns:
                progress.close()
    def merge_insert(self, on: Union[str, Iterable[str]]) -> LanceMergeInsertBuilder:
        """
@@ -4097,7 +4211,7 @@ class AsyncTable:
            async_query = async_query.offset(query.offset)
        if query.columns:
            async_query = async_query.select(query.columns)
-        if query.filter:
+        if query.filter is not None:
            async_query = async_query.where(query.filter)
        if query.fast_search:
            async_query = async_query.fast_search()
@@ -4573,22 +4687,17 @@ class AsyncTable:
            in-progress operation (e.g. appending new data) and these files will not
            be deleted unless they are at least 7 days old. If delete_unverified is True
            then these files will be deleted regardless of their age.
            .. warning::
                This should only be set to True if you can guarantee that no other
                process is currently working on this dataset. Otherwise the dataset
                could be put into a corrupted state.
        retrain: bool, default False
            This parameter is no longer used and is deprecated.
-        Experimental API
+        The frequency an application should call optimize is based on the frequency of
        ----------------
        The optimization process is undergoing active development and may change.
        Our goal with these changes is to improve the performance of optimization and
        reduce the complexity.
        That being said, it is essential today to run optimize if you want the best
        performance.  It should be stable and safe to use in production, but it our
        hope that the API may be simplified (or not even need to be called) in the
        future.
        The frequency an application shoudl call optimize is based on the frequency of
        data modifications.  If data is frequently added, deleted, or updated then
        optimize should be run frequently.  A good rule of thumb is to run optimize if
        you have added or modified 100,000 or more records or run more than 20 data
@@ -4709,7 +4818,16 @@ class IndexStatistics:
    num_indexed_rows: int
    num_unindexed_rows: int
    index_type: Literal[
-        "IVF_PQ", "IVF_HNSW_PQ", "IVF_HNSW_SQ", "FTS", "BTREE", "BITMAP", "LABEL_LIST"
+        "IVF_FLAT",
        "IVF_SQ",
        "IVF_PQ",
        "IVF_RQ",
        "IVF_HNSW_SQ",
        "IVF_HNSW_PQ",
        "FTS",
        "BTREE",
        "BITMAP",
        "LABEL_LIST",
    ]
    distance_type: Optional[Literal["l2", "cosine", "dot"]] = None
    num_indices: Optional[int] = None
--- a/python/python/tests/test_embeddings.py
+++ b/python/python/tests/test_embeddings.py
@@ -546,3 +546,24 @@ def test_openai_no_retry_on_401(mock_sleep):
    assert mock_func.call_count == 1
    # Verify that sleep was never called (no retries)
    assert mock_sleep.call_count == 0
 def test_url_retrieve_downloads_image():
    """
    Embedding functions like open-clip, siglip, and jinaai use url_retrieve()
    to download images from HTTP URLs. For example, open_clip._to_pil() calls:
        PIL_Image.open(io.BytesIO(url_retrieve(image)))
    Verify that url_retrieve() can download an image and open it as PIL Image,
    matching the real usage pattern in embedding functions.
    """
    import io
    Image = pytest.importorskip("PIL.Image")
    from lancedb.embeddings.utils import url_retrieve
    image_url = "http://farm1.staticflickr.com/53/167798175_7c7845bbbd_z.jpg"
    image_bytes = url_retrieve(image_url)
    img = Image.open(io.BytesIO(image_bytes))
    assert img.size[0] > 0 and img.size[1] > 0
--- a/python/python/tests/test_hybrid_query.py
+++ b/python/python/tests/test_hybrid_query.py
@@ -177,6 +177,60 @@ async def test_analyze_plan(table: AsyncTable):
    assert "metrics=" in res
@pytest.fixture
 def table_with_id(tmpdir_factory) -> Table:
    tmp_path = str(tmpdir_factory.mktemp("data"))
    db = lancedb.connect(tmp_path)
    data = pa.table(
        {
            "id": pa.array([1, 2, 3, 4], type=pa.int64()),
            "text": pa.array(["a", "b", "cat", "dog"]),
            "vector": pa.array(
                [[0.1, 0.1], [2, 2], [-0.1, -0.1], [0.5, -0.5]],
                type=pa.list_(pa.float32(), list_size=2),
            ),
        }
    )
    table = db.create_table("test_with_id", data)
    table.create_fts_index("text", with_position=False, use_tantivy=False)
    return table
 def test_hybrid_prefilter_explain_plan(table_with_id: Table):
    """
    Verify that the prefilter logic is not inverted in LanceHybridQueryBuilder.
    """
    plan_prefilter = (
        table_with_id.search(query_type="hybrid")
        .vector([0.0, 0.0])
        .text("dog")
        .where("id = 1", prefilter=True)
        .limit(2)
        .explain_plan(verbose=True)
    )
    plan_postfilter = (
        table_with_id.search(query_type="hybrid")
        .vector([0.0, 0.0])
        .text("dog")
        .where("id = 1", prefilter=False)
        .limit(2)
        .explain_plan(verbose=True)
    )
    # prefilter=True: filter is pushed into the LanceRead scan.
    # The FTS sub-plan exposes this as "full_filter=id = Int64(1)" inside LanceRead.
    assert "full_filter=id = Int64(1)" in plan_prefilter, (
        f"Should push the filter into the scan.\nPlan:\n{plan_prefilter}"
    )
    # prefilter=False: filter is applied as a separate FilterExec after the search.
    # The filter must NOT be embedded in the scan.
    assert "full_filter=id = Int64(1)" not in plan_postfilter, (
        f"Should NOT push the filter into the scan.\nPlan:\n{plan_postfilter}"
    )
 def test_normalize_scores():
    cases = [
        (pa.array([0.1, 0.4]), pa.array([0.0, 1.0])),
--- a/python/python/tests/test_index.py
+++ b/python/python/tests/test_index.py
@@ -3,6 +3,7 @@
 from datetime import timedelta
 import random
 from typing import get_args, get_type_hints
 import pyarrow as pa
 import pytest
@@ -22,6 +23,7 @@ from lancedb.index import (
    HnswSq,
    FTS,
 )
 from lancedb.table import IndexStatistics
@pytest_asyncio.fixture
@@ -283,3 +285,23 @@ async def test_create_index_with_binary_vectors(binary_table: AsyncTable):
    for v in range(256):
        res = await binary_table.query().nearest_to([v] * 128).to_arrow()
        assert res["id"][0].as_py() == v
 def test_index_statistics_index_type_lists_all_supported_values():
    expected_index_types = {
        "IVF_FLAT",
        "IVF_SQ",
        "IVF_PQ",
        "IVF_RQ",
        "IVF_HNSW_SQ",
        "IVF_HNSW_PQ",
        "FTS",
        "BTREE",
        "BITMAP",
        "LABEL_LIST",
    }
    assert (
        set(get_args(get_type_hints(IndexStatistics)["index_type"]))
        == expected_index_types
    )
--- a/python/python/tests/test_namespace.py
+++ b/python/python/tests/test_namespace.py
@@ -8,6 +8,7 @@ import shutil
 import pytest
 import pyarrow as pa
 import lancedb
 from lance_namespace.errors import NamespaceNotEmptyError, TableNotFoundError
 class TestNamespaceConnection:
@@ -130,7 +131,7 @@ class TestNamespaceConnection:
        assert len(list(db.table_names(namespace=["test_ns"]))) == 0
        # Should not be able to open dropped table
-        with pytest.raises(RuntimeError):
+        with pytest.raises(TableNotFoundError):
            db.open_table("table1", namespace=["test_ns"])
    def test_create_table_with_schema(self):
@@ -340,7 +341,7 @@ class TestNamespaceConnection:
        db.create_table("test_table", schema=schema, namespace=["test_namespace"])
        # Try to drop namespace with tables - should fail
-        with pytest.raises(RuntimeError, match="is not empty"):
+        with pytest.raises(NamespaceNotEmptyError):
            db.drop_namespace(["test_namespace"])
        # Drop table first
--- a/python/python/tests/test_namespace_integration.py
+++ b/python/python/tests/test_namespace_integration.py
@@ -147,7 +147,12 @@ class TrackingNamespace(LanceNamespace):
        This simulates a credential rotation system where each call returns
        new credentials that expire after credential_expires_in_seconds.
        """
-        modified = copy.deepcopy(storage_options) if storage_options else {}
+        # Start from base storage options (endpoint, region, allow_http, etc.)
        # because DirectoryNamespace returns None for storage_options from
        # describe_table/declare_table when no credential vendor is configured.
        modified = copy.deepcopy(self.base_storage_options)
        if storage_options:
            modified.update(storage_options)
        # Increment credentials to simulate rotation
        modified["aws_access_key_id"] = f"AKID_{count}"
--- a/python/python/tests/test_query.py
+++ b/python/python/tests/test_query.py
@@ -30,6 +30,7 @@ from lancedb.query import (
    PhraseQuery,
    Query,
    FullTextSearchQuery,
    ensure_vector_query,
 )
 from lancedb.rerankers.cross_encoder import CrossEncoderReranker
 from lancedb.table import AsyncTable, LanceTable
@@ -1501,6 +1502,18 @@ def test_search_empty_table(mem_db):
    assert results == []
 def test_ensure_vector_query_empty_list():
    """Regression: ensure_vector_query used to return instead of raise ValueError."""
    with pytest.raises(ValueError, match="non-empty"):
        ensure_vector_query([])
 def test_ensure_vector_query_nested_empty_list():
    """Regression: ensure_vector_query used to return instead of raise ValueError."""
    with pytest.raises(ValueError, match="non-empty"):
        ensure_vector_query([[]])
 def test_fast_search(tmp_path):
    db = lancedb.connect(tmp_path)
--- a/python/python/tests/test_remote_db.py
+++ b/python/python/tests/test_remote_db.py
@@ -1201,6 +1201,18 @@ async def test_header_provider_overrides_static_headers():
        await db.table_names()
 def test_close():
    """Test that close() works without AttributeError."""
    import asyncio
    def handler(req):
        req.send_response(200)
        req.end_headers()
    with mock_lancedb_connection(handler) as db:
        asyncio.run(db.close())
@pytest.mark.parametrize("exception", [KeyboardInterrupt, SystemExit, GeneratorExit])
 def test_background_loop_cancellation(exception):
    """Test that BackgroundEventLoop.run() cancels the future on interrupt."""
--- a/python/python/tests/test_table.py
+++ b/python/python/tests/test_table.py
@@ -527,6 +527,132 @@ async def test_add_async(mem_db_async: AsyncConnection):
    assert await table.count_rows() == 3
 def test_add_overwrite_infers_vector_schema(mem_db: DBConnection):
    """Overwrite should infer vector columns the same way create_table does.
    Regression test for https://github.com/lancedb/lancedb/issues/3183
    """
    table = mem_db.create_table(
        "test_overwrite_vec",
        data=[
            {"vector": [1.0, 2.0, 3.0, 4.0], "item": "foo"},
            {"vector": [5.0, 6.0, 7.0, 8.0], "item": "bar"},
        ],
    )
    # create_table infers vector as fixed_size_list<float32, 4>
    original_type = table.schema.field("vector").type
    assert pa.types.is_fixed_size_list(original_type)
    # overwrite with plain Python lists (PyArrow infers list<double>)
    table.add(
        [
            {"vector": [10.0, 20.0, 30.0, 40.0], "item": "baz"},
        ],
        mode="overwrite",
    )
    # overwrite should infer vector column the same way as create_table
    new_type = table.schema.field("vector").type
    assert pa.types.is_fixed_size_list(new_type), (
        f"Expected fixed_size_list after overwrite, got {new_type}"
    )
 def test_add_progress_callback(mem_db: DBConnection):
    table = mem_db.create_table(
        "test",
        data=[{"id": 1}, {"id": 2}],
    )
    updates = []
    table.add([{"id": 3}, {"id": 4}], progress=lambda p: updates.append(dict(p)))
    assert len(table) == 4
    # The done callback always fires, so we should always get at least one.
    assert len(updates) >= 1, "expected at least one progress callback"
    for p in updates:
        assert "output_rows" in p
        assert "output_bytes" in p
        assert "total_rows" in p
        assert "elapsed_seconds" in p
        assert "active_tasks" in p
        assert "total_tasks" in p
        assert "done" in p
    # The last callback should have done=True.
    assert updates[-1]["done"] is True
 def test_add_progress_tqdm_like(mem_db: DBConnection):
    """Test that a tqdm-like object gets total set and update() called."""
    class FakeBar:
        def __init__(self):
            self.total = None
            self.n = 0
            self.postfix = None
        def update(self, n):
            self.n += n
        def set_postfix_str(self, s):
            self.postfix = s
        def refresh(self):
            pass
    table = mem_db.create_table(
        "test",
        data=[{"id": 1}, {"id": 2}],
    )
    bar = FakeBar()
    table.add([{"id": 3}, {"id": 4}], progress=bar)
    assert len(table) == 4
    # Postfix should contain throughput and worker count
    if bar.postfix is not None:
        assert "MB/s" in bar.postfix
        assert "workers" in bar.postfix
 def test_add_progress_bool(mem_db: DBConnection):
    """Test that progress=True creates and closes a tqdm bar automatically."""
    table = mem_db.create_table(
        "test",
        data=[{"id": 1}, {"id": 2}],
    )
    table.add([{"id": 3}, {"id": 4}], progress=True)
    assert len(table) == 4
    # progress=False should be the same as None
    table.add([{"id": 5}], progress=False)
    assert len(table) == 5
@pytest.mark.asyncio
 async def test_add_progress_callback_async(mem_db_async: AsyncConnection):
    """Progress callbacks work through the async path too."""
    table = await mem_db_async.create_table("test", data=[{"id": 1}, {"id": 2}])
    updates = []
    await table.add([{"id": 3}, {"id": 4}], progress=lambda p: updates.append(dict(p)))
    assert await table.count_rows() == 4
    assert len(updates) >= 1
    assert updates[-1]["done"] is True
 def test_add_progress_callback_error(mem_db: DBConnection):
    """A failing callback must not prevent the write from succeeding."""
    table = mem_db.create_table("test", data=[{"id": 1}, {"id": 2}])
    def bad_callback(p):
        raise RuntimeError("boom")
    table.add([{"id": 3}, {"id": 4}], progress=bad_callback)
    assert len(table) == 4
 def test_polars(mem_db: DBConnection):
    data = {
        "vector": [[3.1, 4.1], [5.9, 26.5]],
@@ -2047,3 +2173,33 @@ def test_table_uri(tmp_path):
    db = lancedb.connect(tmp_path)
    table = db.create_table("my_table", data=[{"x": 0}])
    assert table.uri == str(tmp_path / "my_table.lance")
 def test_sanitize_data_metadata_not_stripped():
    """Regression test: dict.update() returns None, so assigning its result
    would silently replace metadata with None, causing with_metadata(None)
    to strip all schema metadata from the target schema."""
    from lancedb.table import _sanitize_data
    schema = pa.schema(
        [pa.field("x", pa.int64())],
        metadata={b"existing_key": b"existing_value"},
    )
    batch = pa.record_batch([pa.array([1, 2, 3])], schema=schema)
    # Use a different field type so the reader and target schemas differ,
    # forcing _cast_to_target_schema to rebuild the schema with the
    # target's metadata (instead of taking the fast-path).
    target_schema = pa.schema(
        [pa.field("x", pa.int32())],
        metadata={b"existing_key": b"existing_value"},
    )
    reader = pa.RecordBatchReader.from_batches(schema, [batch])
    metadata = {b"new_key": b"new_value"}
    result = _sanitize_data(reader, target_schema=target_schema, metadata=metadata)
    result_schema = result.schema
    assert result_schema.metadata is not None
    assert result_schema.metadata[b"existing_key"] == b"existing_value"
    assert result_schema.metadata[b"new_key"] == b"new_value"
--- a/python/src/expr.rs
+++ b/python/src/expr.rs
@@ -0,0 +1,175 @@
 // SPDX-License-Identifier: Apache-2.0
 // SPDX-FileCopyrightText: Copyright The LanceDB Authors
 //! PyO3 bindings for the LanceDB expression builder API.
 //!
 //! This module exposes [`PyExpr`] and helper free functions so Python can
 //! build type-safe filter / projection expressions that map directly to
 //! DataFusion [`Expr`] nodes, bypassing SQL string parsing.
 use arrow::{datatypes::DataType, pyarrow::PyArrowType};
 use lancedb::expr::{DfExpr, col as ldb_col, contains, expr_cast, lit as df_lit, lower, upper};
 use pyo3::{Bound, PyAny, PyResult, exceptions::PyValueError, prelude::*, pyfunction};
 /// A type-safe DataFusion expression.
 ///
 /// Instances are constructed via the free functions [`expr_col`] and
 /// [`expr_lit`] and combined with the methods on this struct.  On the Python
 /// side a thin wrapper class (`lancedb.expr.Expr`) delegates to these methods
 /// and adds Python operator overloads.
 #[pyclass(name = "PyExpr")]
 #[derive(Clone)]
 pub struct PyExpr(pub DfExpr);
 #[pymethods]
 impl PyExpr {
    // ── comparisons ──────────────────────────────────────────────────────────
    fn eq(&self, other: &Self) -> Self {
        Self(self.0.clone().eq(other.0.clone()))
    }
    fn ne(&self, other: &Self) -> Self {
        Self(self.0.clone().not_eq(other.0.clone()))
    }
    fn lt(&self, other: &Self) -> Self {
        Self(self.0.clone().lt(other.0.clone()))
    }
    fn lte(&self, other: &Self) -> Self {
        Self(self.0.clone().lt_eq(other.0.clone()))
    }
    fn gt(&self, other: &Self) -> Self {
        Self(self.0.clone().gt(other.0.clone()))
    }
    fn gte(&self, other: &Self) -> Self {
        Self(self.0.clone().gt_eq(other.0.clone()))
    }
    // ── logical ──────────────────────────────────────────────────────────────
    fn and_(&self, other: &Self) -> Self {
        Self(self.0.clone().and(other.0.clone()))
    }
    fn or_(&self, other: &Self) -> Self {
        Self(self.0.clone().or(other.0.clone()))
    }
    fn not_(&self) -> Self {
        use std::ops::Not;
        Self(self.0.clone().not())
    }
    // ── arithmetic ───────────────────────────────────────────────────────────
    fn add(&self, other: &Self) -> Self {
        use std::ops::Add;
        Self(self.0.clone().add(other.0.clone()))
    }
    fn sub(&self, other: &Self) -> Self {
        use std::ops::Sub;
        Self(self.0.clone().sub(other.0.clone()))
    }
    fn mul(&self, other: &Self) -> Self {
        use std::ops::Mul;
        Self(self.0.clone().mul(other.0.clone()))
    }
    fn div(&self, other: &Self) -> Self {
        use std::ops::Div;
        Self(self.0.clone().div(other.0.clone()))
    }
    // ── string functions ─────────────────────────────────────────────────────
    /// Convert string column to lowercase.
    fn lower(&self) -> Self {
        Self(lower(self.0.clone()))
    }
    /// Convert string column to uppercase.
    fn upper(&self) -> Self {
        Self(upper(self.0.clone()))
    }
    /// Test whether the string contains `substr`.
    fn contains(&self, substr: &Self) -> Self {
        Self(contains(self.0.clone(), substr.0.clone()))
    }
    // ── type cast ────────────────────────────────────────────────────────────
    /// Cast the expression to `data_type`.
    ///
    /// `data_type` must be a PyArrow `DataType` (e.g. `pa.int32()`).
    /// On the Python side, `lancedb.expr.Expr.cast` also accepts type name
    /// strings via `pa.lib.ensure_type` before forwarding here.
    fn cast(&self, data_type: PyArrowType<DataType>) -> Self {
        Self(expr_cast(self.0.clone(), data_type.0))
    }
    // ── utilities ────────────────────────────────────────────────────────────
    /// Render the expression as a SQL string (useful for debugging).
    fn to_sql(&self) -> PyResult<String> {
        lancedb::expr::expr_to_sql_string(&self.0).map_err(|e| PyValueError::new_err(e.to_string()))
    }
    fn __repr__(&self) -> PyResult<String> {
        let sql =
            lancedb::expr::expr_to_sql_string(&self.0).unwrap_or_else(|_| "<expr>".to_string());
        Ok(format!("PyExpr({})", sql))
    }
 }
 // ── free functions ────────────────────────────────────────────────────────────
 /// Create a column reference expression.
 ///
 /// The column name is preserved exactly as given (case-sensitive), so
 /// `col("firstName")` correctly references a field named `firstName`.
 #[pyfunction]
 pub fn expr_col(name: &str) -> PyExpr {
    PyExpr(ldb_col(name))
 }
 /// Create a literal value expression.
 ///
 /// Supported Python types: `bool`, `int`, `float`, `str`.
 #[pyfunction]
 pub fn expr_lit(value: Bound<'_, PyAny>) -> PyResult<PyExpr> {
    // bool must be checked before int because bool is a subclass of int in Python
    if let Ok(b) = value.extract::<bool>() {
        return Ok(PyExpr(df_lit(b)));
    }
    if let Ok(i) = value.extract::<i64>() {
        return Ok(PyExpr(df_lit(i)));
    }
    if let Ok(f) = value.extract::<f64>() {
        return Ok(PyExpr(df_lit(f)));
    }
    if let Ok(s) = value.extract::<String>() {
        return Ok(PyExpr(df_lit(s)));
    }
    Err(PyValueError::new_err(format!(
        "unsupported literal type: {}. Supported: bool, int, float, str",
        value.get_type().name()?
    )))
 }
 /// Call an arbitrary registered SQL function by name.
 ///
 /// See `lancedb::expr::func` for the list of supported function names.
 #[pyfunction]
 pub fn expr_func(name: &str, args: Vec<PyExpr>) -> PyResult<PyExpr> {
    let df_args: Vec<DfExpr> = args.into_iter().map(|e| e.0).collect();
    lancedb::expr::func(name, df_args)
        .map(PyExpr)
        .map_err(|e| PyValueError::new_err(e.to_string()))
 }
--- a/python/src/lib.rs
+++ b/python/src/lib.rs
@@ -4,6 +4,7 @@
 use arrow::RecordBatchStream;
 use connection::{Connection, connect};
 use env_logger::Env;
 use expr::{PyExpr, expr_col, expr_func, expr_lit};
 use index::IndexConfig;
 use permutation::{PyAsyncPermutationBuilder, PyPermutationReader};
 use pyo3::{
@@ -21,6 +22,7 @@ use table::{
 pub mod arrow;
 pub mod connection;
 pub mod error;
 pub mod expr;
 pub mod header;
 pub mod index;
 pub mod namespace;
@@ -55,10 +57,14 @@ pub fn _lancedb(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> {
    m.add_class::<UpdateResult>()?;
    m.add_class::<PyAsyncPermutationBuilder>()?;
    m.add_class::<PyPermutationReader>()?;
    m.add_class::<PyExpr>()?;
    m.add_function(wrap_pyfunction!(connect, m)?)?;
    m.add_function(wrap_pyfunction!(permutation::async_permutation_builder, m)?)?;
    m.add_function(wrap_pyfunction!(util::validate_table_name, m)?)?;
    m.add_function(wrap_pyfunction!(query::fts_query_to_json, m)?)?;
    m.add_function(wrap_pyfunction!(expr_col, m)?)?;
    m.add_function(wrap_pyfunction!(expr_lit, m)?)?;
    m.add_function(wrap_pyfunction!(expr_func, m)?)?;
    m.add("__version__", env!("CARGO_PKG_VERSION"))?;
    Ok(())
 }
--- a/python/src/namespace.rs
+++ b/python/src/namespace.rs
@@ -96,10 +96,10 @@ where
    Resp: serde::de::DeserializeOwned + Send + 'static,
 {
    let request_json = serde_json::to_string(&request).map_err(|e| {
-        lance_core::Error::io(
+        lance_core::Error::io(format!(
-            format!("Failed to serialize request for {}: {}", method_name, e),
+            "Failed to serialize request for {}: {}",
-            Default::default(),
+            method_name, e
-        )
+        ))
    })?;
    let response_json = tokio::task::spawn_blocking(move || {
@@ -128,24 +128,14 @@ where
        })
    })
    .await
-    .map_err(|e| {
+    .map_err(|e| lance_core::Error::io(format!("Task join error for {}: {}", method_name, e)))?
-        lance_core::Error::io(
+    .map_err(|e: PyErr| lance_core::Error::io(format!("Python error in {}: {}", method_name, e)))?;
            format!("Task join error for {}: {}", method_name, e),
            Default::default(),
        )
    })?
    .map_err(|e: PyErr| {
        lance_core::Error::io(
            format!("Python error in {}: {}", method_name, e),
            Default::default(),
        )
    })?;
    serde_json::from_str(&response_json).map_err(|e| {
-        lance_core::Error::io(
+        lance_core::Error::io(format!(
-            format!("Failed to deserialize response from {}: {}", method_name, e),
+            "Failed to deserialize response from {}: {}",
-            Default::default(),
+            method_name, e
-        )
+        ))
    })
 }
@@ -159,10 +149,10 @@ where
    Req: serde::Serialize + Send + 'static,
 {
    let request_json = serde_json::to_string(&request).map_err(|e| {
-        lance_core::Error::io(
+        lance_core::Error::io(format!(
-            format!("Failed to serialize request for {}: {}", method_name, e),
+            "Failed to serialize request for {}: {}",
-            Default::default(),
+            method_name, e
-        )
+        ))
    })?;
    tokio::task::spawn_blocking(move || {
@@ -180,18 +170,8 @@ where
        })
    })
    .await
-    .map_err(|e| {
+    .map_err(|e| lance_core::Error::io(format!("Task join error for {}: {}", method_name, e)))?
-        lance_core::Error::io(
+    .map_err(|e: PyErr| lance_core::Error::io(format!("Python error in {}: {}", method_name, e)))
            format!("Task join error for {}: {}", method_name, e),
            Default::default(),
        )
    })?
    .map_err(|e: PyErr| {
        lance_core::Error::io(
            format!("Python error in {}: {}", method_name, e),
            Default::default(),
        )
    })
 }
 /// Helper for methods that return a primitive type
@@ -205,10 +185,10 @@ where
    Resp: for<'py> pyo3::FromPyObject<'py> + Send + 'static,
 {
    let request_json = serde_json::to_string(&request).map_err(|e| {
-        lance_core::Error::io(
+        lance_core::Error::io(format!(
-            format!("Failed to serialize request for {}: {}", method_name, e),
+            "Failed to serialize request for {}: {}",
-            Default::default(),
+            method_name, e
-        )
+        ))
    })?;
    tokio::task::spawn_blocking(move || {
@@ -227,18 +207,8 @@ where
        })
    })
    .await
-    .map_err(|e| {
+    .map_err(|e| lance_core::Error::io(format!("Task join error for {}: {}", method_name, e)))?
-        lance_core::Error::io(
+    .map_err(|e: PyErr| lance_core::Error::io(format!("Python error in {}: {}", method_name, e)))
            format!("Task join error for {}: {}", method_name, e),
            Default::default(),
        )
    })?
    .map_err(|e: PyErr| {
        lance_core::Error::io(
            format!("Python error in {}: {}", method_name, e),
            Default::default(),
        )
    })
 }
 /// Helper for methods that return Bytes
@@ -251,10 +221,10 @@ where
    Req: serde::Serialize + Send + 'static,
 {
    let request_json = serde_json::to_string(&request).map_err(|e| {
-        lance_core::Error::io(
+        lance_core::Error::io(format!(
-            format!("Failed to serialize request for {}: {}", method_name, e),
+            "Failed to serialize request for {}: {}",
-            Default::default(),
+            method_name, e
-        )
+        ))
    })?;
    tokio::task::spawn_blocking(move || {
@@ -273,18 +243,8 @@ where
        })
    })
    .await
-    .map_err(|e| {
+    .map_err(|e| lance_core::Error::io(format!("Task join error for {}: {}", method_name, e)))?
-        lance_core::Error::io(
+    .map_err(|e: PyErr| lance_core::Error::io(format!("Python error in {}: {}", method_name, e)))
            format!("Task join error for {}: {}", method_name, e),
            Default::default(),
        )
    })?
    .map_err(|e: PyErr| {
        lance_core::Error::io(
            format!("Python error in {}: {}", method_name, e),
            Default::default(),
        )
    })
 }
 /// Helper for methods that take request + data and return a response
@@ -299,10 +259,10 @@ where
    Resp: serde::de::DeserializeOwned + Send + 'static,
 {
    let request_json = serde_json::to_string(&request).map_err(|e| {
-        lance_core::Error::io(
+        lance_core::Error::io(format!(
-            format!("Failed to serialize request for {}: {}", method_name, e),
+            "Failed to serialize request for {}: {}",
-            Default::default(),
+            method_name, e
-        )
+        ))
    })?;
    let response_json = tokio::task::spawn_blocking(move || {
@@ -324,24 +284,14 @@ where
        })
    })
    .await
-    .map_err(|e| {
+    .map_err(|e| lance_core::Error::io(format!("Task join error for {}: {}", method_name, e)))?
-        lance_core::Error::io(
+    .map_err(|e: PyErr| lance_core::Error::io(format!("Python error in {}: {}", method_name, e)))?;
            format!("Task join error for {}: {}", method_name, e),
            Default::default(),
        )
    })?
    .map_err(|e: PyErr| {
        lance_core::Error::io(
            format!("Python error in {}: {}", method_name, e),
            Default::default(),
        )
    })?;
    serde_json::from_str(&response_json).map_err(|e| {
-        lance_core::Error::io(
+        lance_core::Error::io(format!(
-            format!("Failed to deserialize response from {}: {}", method_name, e),
+            "Failed to deserialize response from {}: {}",
-            Default::default(),
+            method_name, e
-        )
+        ))
    })
 }
--- a/python/src/query.rs
+++ b/python/src/query.rs
@@ -35,12 +35,10 @@ use pyo3::types::PyList;
 use pyo3::types::{PyDict, PyString};
 use pyo3::{FromPyObject, exceptions::PyRuntimeError};
 use pyo3::{PyErr, pyclass};
-use pyo3::{
+use pyo3::{exceptions::PyValueError, intern};
    exceptions::{PyNotImplementedError, PyValueError},
    intern,
 };
 use pyo3_async_runtimes::tokio::future_into_py;
 use crate::expr::PyExpr;
 use crate::util::parse_distance_type;
 use crate::{arrow::RecordBatchStream, util::PyLanceDB};
 use crate::{error::PythonErrorExt, index::class_name};
@@ -316,6 +314,19 @@ impl<'py> IntoPyObject<'py> for PySelect {
            Select::All => Ok(py.None().into_bound(py).into_any()),
            Select::Columns(columns) => Ok(columns.into_pyobject(py)?.into_any()),
            Select::Dynamic(columns) => Ok(columns.into_pyobject(py)?.into_any()),
            Select::Expr(pairs) => {
                // Serialize DataFusion Expr -> SQL string so Python sees the same
                // format as Select::Dynamic: a list of (name, sql_string) tuples.
                let sql_pairs: PyResult<Vec<(String, String)>> = pairs
                    .into_iter()
                    .map(|(name, expr)| {
                        lancedb::expr::expr_to_sql_string(&expr)
                            .map(|sql| (name, sql))
                            .map_err(|e| PyRuntimeError::new_err(e.to_string()))
                    })
                    .collect();
                Ok(sql_pairs?.into_pyobject(py)?.into_any())
            }
        }
    }
 }
@@ -331,9 +342,13 @@ impl<'py> IntoPyObject<'py> for PyQueryFilter {
    fn into_pyobject(self, py: pyo3::Python<'py>) -> PyResult<Self::Output> {
        match self.0 {
-            QueryFilter::Datafusion(_) => Err(PyNotImplementedError::new_err(
+            QueryFilter::Datafusion(expr) => {
-                "Datafusion filter has no conversion to Python",
+                // Serialize the DataFusion expression to a SQL string so that
-            )),
+                // callers (e.g. remote tables) see the same format as Sql.
                let sql = lancedb::expr::expr_to_sql_string(&expr)
                    .map_err(|e| PyRuntimeError::new_err(e.to_string()))?;
                Ok(sql.into_pyobject(py)?.into_any())
            }
            QueryFilter::Sql(sql) => Ok(sql.into_pyobject(py)?.into_any()),
            QueryFilter::Substrait(substrait) => Ok(substrait.into_pyobject(py)?.into_any()),
        }
@@ -357,10 +372,20 @@ impl Query {
        self.inner = self.inner.clone().only_if(predicate);
    }
    pub fn where_expr(&mut self, expr: PyExpr) {
        self.inner = self.inner.clone().only_if_expr(expr.0);
    }
    pub fn select(&mut self, columns: Vec<(String, String)>) {
        self.inner = self.inner.clone().select(Select::dynamic(&columns));
    }
    pub fn select_expr(&mut self, columns: Vec<(String, PyExpr)>) {
        let pairs: Vec<(String, lancedb::expr::DfExpr)> =
            columns.into_iter().map(|(name, e)| (name, e.0)).collect();
        self.inner = self.inner.clone().select(Select::Expr(pairs));
    }
    pub fn select_columns(&mut self, columns: Vec<String>) {
        self.inner = self.inner.clone().select(Select::columns(&columns));
    }
@@ -594,10 +619,20 @@ impl FTSQuery {
        self.inner = self.inner.clone().only_if(predicate);
    }
    pub fn where_expr(&mut self, expr: PyExpr) {
        self.inner = self.inner.clone().only_if_expr(expr.0);
    }
    pub fn select(&mut self, columns: Vec<(String, String)>) {
        self.inner = self.inner.clone().select(Select::dynamic(&columns));
    }
    pub fn select_expr(&mut self, columns: Vec<(String, PyExpr)>) {
        let pairs: Vec<(String, lancedb::expr::DfExpr)> =
            columns.into_iter().map(|(name, e)| (name, e.0)).collect();
        self.inner = self.inner.clone().select(Select::Expr(pairs));
    }
    pub fn select_columns(&mut self, columns: Vec<String>) {
        self.inner = self.inner.clone().select(Select::columns(&columns));
    }
@@ -712,6 +747,10 @@ impl VectorQuery {
        self.inner = self.inner.clone().only_if(predicate);
    }
    pub fn where_expr(&mut self, expr: PyExpr) {
        self.inner = self.inner.clone().only_if_expr(expr.0);
    }
    pub fn add_query_vector(&mut self, vector: Bound<'_, PyAny>) -> PyResult<()> {
        let data: ArrayData = ArrayData::from_pyarrow_bound(&vector)?;
        let array = make_array(data);
@@ -723,6 +762,12 @@ impl VectorQuery {
        self.inner = self.inner.clone().select(Select::dynamic(&columns));
    }
    pub fn select_expr(&mut self, columns: Vec<(String, PyExpr)>) {
        let pairs: Vec<(String, lancedb::expr::DfExpr)> =
            columns.into_iter().map(|(name, e)| (name, e.0)).collect();
        self.inner = self.inner.clone().select(Select::Expr(pairs));
    }
    pub fn select_columns(&mut self, columns: Vec<String>) {
        self.inner = self.inner.clone().select(Select::columns(&columns));
    }
@@ -877,11 +922,21 @@ impl HybridQuery {
        self.inner_fts.r#where(predicate);
    }
    pub fn where_expr(&mut self, expr: PyExpr) {
        self.inner_vec.where_expr(expr.clone());
        self.inner_fts.where_expr(expr);
    }
    pub fn select(&mut self, columns: Vec<(String, String)>) {
        self.inner_vec.select(columns.clone());
        self.inner_fts.select(columns);
    }
    pub fn select_expr(&mut self, columns: Vec<(String, PyExpr)>) {
        self.inner_vec.select_expr(columns.clone());
        self.inner_fts.select_expr(columns);
    }
    pub fn select_columns(&mut self, columns: Vec<String>) {
        self.inner_vec.select_columns(columns.clone());
        self.inner_fts.select_columns(columns);
--- a/python/src/storage_options.rs
+++ b/python/src/storage_options.rs
@@ -66,13 +66,10 @@ impl StorageOptionsProvider for PyStorageOptionsProviderWrapper {
                    .inner
                    .bind(py)
                    .call_method0("fetch_storage_options")
-                    .map_err(|e| lance_core::Error::IO {
+                    .map_err(|e| lance_core::Error::io_source(Box::new(std::io::Error::other(format!(
-                        source: Box::new(std::io::Error::other(format!(
+                        "Failed to call fetch_storage_options: {}",
-                            "Failed to call fetch_storage_options: {}",
+                        e
-                            e
+                    )))))?;
                        ))),
                        location: snafu::location!(),
                    })?;
                // If result is None, return None
                if result.is_none() {
@@ -81,26 +78,19 @@ impl StorageOptionsProvider for PyStorageOptionsProviderWrapper {
                // Extract the result dict - should be a flat Map<String, String>
                let result_dict = result.downcast::<PyDict>().map_err(|_| {
-                    lance_core::Error::InvalidInput {
+                    lance_core::Error::invalid_input(
-                        source: "fetch_storage_options() must return None or a dict of string key-value pairs".into(),
+                        "fetch_storage_options() must return a dict of string key-value pairs or None",
-                        location: snafu::location!(),
+                    )
                    }
                })?;
                // Convert all entries to HashMap<String, String>
                let mut storage_options = HashMap::new();
                for (key, value) in result_dict.iter() {
                    let key_str: String = key.extract().map_err(|e| {
-                        lance_core::Error::InvalidInput {
+                        lance_core::Error::invalid_input(format!("Storage option key must be a string: {}", e))
                            source: format!("Storage option key must be a string: {}", e).into(),
                            location: snafu::location!(),
                        }
                    })?;
                    let value_str: String = value.extract().map_err(|e| {
-                        lance_core::Error::InvalidInput {
+                        lance_core::Error::invalid_input(format!("Storage option value must be a string: {}", e))
                            source: format!("Storage option value must be a string: {}", e).into(),
                            location: snafu::location!(),
                        }
                    })?;
                    storage_options.insert(key_str, value_str);
                }
@@ -109,13 +99,10 @@ impl StorageOptionsProvider for PyStorageOptionsProviderWrapper {
            })
        })
        .await
-        .map_err(|e| lance_core::Error::IO {
+        .map_err(|e| lance_core::Error::io_source(Box::new(std::io::Error::other(format!(
-            source: Box::new(std::io::Error::other(format!(
+            "Task join error: {}",
-                "Task join error: {}",
+            e
-                e
+        )))))?
            ))),
            location: snafu::location!(),
        })?
    }
    fn provider_id(&self) -> String {
--- a/python/src/table.rs
+++ b/python/src/table.rs
@@ -19,7 +19,7 @@ use lancedb::table::{
    Table as LanceDbTable,
 };
 use pyo3::{
-    Bound, FromPyObject, PyAny, PyRef, PyResult, Python,
+    Bound, FromPyObject, Py, PyAny, PyRef, PyResult, Python,
    exceptions::{PyKeyError, PyRuntimeError, PyValueError},
    pyclass, pymethods,
    types::{IntoPyDict, PyAnyMethods, PyDict, PyDictMethods},
@@ -299,10 +299,12 @@ impl Table {
        })
    }
    #[pyo3(signature = (data, mode, progress=None))]
    pub fn add<'a>(
        self_: PyRef<'a, Self>,
        data: PyScannable,
        mode: String,
        progress: Option<Py<PyAny>>,
    ) -> PyResult<Bound<'a, PyAny>> {
        let mut op = self_.inner_ref()?.add(data);
        if mode == "append" {
@@ -312,6 +314,81 @@ impl Table {
        } else {
            return Err(PyValueError::new_err(format!("Invalid mode: {}", mode)));
        }
        if let Some(progress_obj) = progress {
            let is_callable = Python::attach(|py| progress_obj.bind(py).is_callable());
            if is_callable {
                // Callback: call with a dict of progress info.
                op = op.progress(move |p| {
                    Python::attach(|py| {
                        let dict = PyDict::new(py);
                        if let Err(e) = dict
                            .set_item("output_rows", p.output_rows())
                            .and_then(|_| dict.set_item("output_bytes", p.output_bytes()))
                            .and_then(|_| dict.set_item("total_rows", p.total_rows()))
                            .and_then(|_| {
                                dict.set_item("elapsed_seconds", p.elapsed().as_secs_f64())
                            })
                            .and_then(|_| dict.set_item("active_tasks", p.active_tasks()))
                            .and_then(|_| dict.set_item("total_tasks", p.total_tasks()))
                            .and_then(|_| dict.set_item("done", p.done()))
                        {
                            log::warn!("progress dict error: {e}");
                            return;
                        }
                        if let Err(e) = progress_obj.call1(py, (dict,)) {
                            log::warn!("progress callback error: {e}");
                        }
                    });
                });
            } else {
                // tqdm-like: has update() method.
                let mut last_rows: usize = 0;
                let mut total_set = false;
                op = op.progress(move |p| {
                    let current = p.output_rows();
                    let prev = last_rows;
                    last_rows = current;
                    Python::attach(|py| {
                        if let Some(total) = p.total_rows()
                            && !total_set
                        {
                            if let Err(e) = progress_obj.setattr(py, "total", total) {
                                log::warn!("progress setattr error: {e}");
                            }
                            total_set = true;
                        }
                        let delta = current.saturating_sub(prev);
                        if delta > 0 {
                            if let Err(e) = progress_obj.call_method1(py, "update", (delta,)) {
                                log::warn!("progress update error: {e}");
                            }
                            // Show throughput and active workers in tqdm postfix.
                            let elapsed = p.elapsed().as_secs_f64();
                            if elapsed > 0.0 {
                                let mb_per_sec = p.output_bytes() as f64 / elapsed / 1_000_000.0;
                                let postfix = format!(
                                    "{:.1} MB/s | {}/{} workers",
                                    mb_per_sec,
                                    p.active_tasks(),
                                    p.total_tasks()
                                );
                                if let Err(e) =
                                    progress_obj.call_method1(py, "set_postfix_str", (postfix,))
                                {
                                    log::warn!("progress set_postfix_str error: {e}");
                                }
                            }
                        }
                        if p.done() {
                            // Force a final refresh so the bar shows completion.
                            if let Err(e) = progress_obj.call_method0(py, "refresh") {
                                log::warn!("progress refresh error: {e}");
                            }
                        }
                    });
                });
            }
        }
        future_into_py(self_.py(), async move {
            let result = op.execute().await.infer_error()?;
@@ -426,6 +503,17 @@ impl Table {
        })
    }
    pub fn prewarm_data(
        self_: PyRef<'_, Self>,
        columns: Option<Vec<String>>,
    ) -> PyResult<Bound<'_, PyAny>> {
        let inner = self_.inner_ref()?.clone();
        future_into_py(self_.py(), async move {
            inner.prewarm_data(columns).await.infer_error()?;
            Ok(())
        })
    }
    pub fn list_indices(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
        let inner = self_.inner_ref()?.clone();
        future_into_py(self_.py(), async move {
--- a/python/tests/test_expr.py
+++ b/python/tests/test_expr.py
@@ -0,0 +1,387 @@
 # SPDX-License-Identifier: Apache-2.0
 # SPDX-FileCopyrightText: Copyright The LanceDB Authors
 """Tests for the type-safe expression builder API."""
 import pytest
 import pyarrow as pa
 import lancedb
 from lancedb.expr import Expr, col, lit, func
 # ── unit tests for Expr construction ─────────────────────────────────────────
 class TestExprConstruction:
    def test_col_returns_expr(self):
        e = col("age")
        assert isinstance(e, Expr)
    def test_lit_int(self):
        e = lit(42)
        assert isinstance(e, Expr)
    def test_lit_float(self):
        e = lit(3.14)
        assert isinstance(e, Expr)
    def test_lit_str(self):
        e = lit("hello")
        assert isinstance(e, Expr)
    def test_lit_bool(self):
        e = lit(True)
        assert isinstance(e, Expr)
    def test_lit_unsupported_type_raises(self):
        with pytest.raises(Exception):
            lit([1, 2, 3])
    def test_func(self):
        e = func("lower", col("name"))
        assert isinstance(e, Expr)
        assert e.to_sql() == "lower(name)"
    def test_func_unknown_raises(self):
        with pytest.raises(Exception):
            func("not_a_real_function", col("x"))
 class TestExprOperators:
    def test_eq_operator(self):
        e = col("x") == lit(1)
        assert isinstance(e, Expr)
        assert e.to_sql() == "(x = 1)"
    def test_ne_operator(self):
        e = col("x") != lit(1)
        assert isinstance(e, Expr)
        assert e.to_sql() == "(x <> 1)"
    def test_lt_operator(self):
        e = col("age") < lit(18)
        assert isinstance(e, Expr)
        assert e.to_sql() == "(age < 18)"
    def test_le_operator(self):
        e = col("age") <= lit(18)
        assert isinstance(e, Expr)
        assert e.to_sql() == "(age <= 18)"
    def test_gt_operator(self):
        e = col("age") > lit(18)
        assert isinstance(e, Expr)
        assert e.to_sql() == "(age > 18)"
    def test_ge_operator(self):
        e = col("age") >= lit(18)
        assert isinstance(e, Expr)
        assert e.to_sql() == "(age >= 18)"
    def test_and_operator(self):
        e = (col("age") > lit(18)) & (col("status") == lit("active"))
        assert isinstance(e, Expr)
        assert e.to_sql() == "((age > 18) AND (status = 'active'))"
    def test_or_operator(self):
        e = (col("a") == lit(1)) | (col("b") == lit(2))
        assert isinstance(e, Expr)
        assert e.to_sql() == "((a = 1) OR (b = 2))"
    def test_invert_operator(self):
        e = ~(col("active") == lit(True))
        assert isinstance(e, Expr)
        assert e.to_sql() == "NOT (active = true)"
    def test_add_operator(self):
        e = col("x") + lit(1)
        assert isinstance(e, Expr)
        assert e.to_sql() == "(x + 1)"
    def test_sub_operator(self):
        e = col("x") - lit(1)
        assert isinstance(e, Expr)
        assert e.to_sql() == "(x - 1)"
    def test_mul_operator(self):
        e = col("price") * lit(1.1)
        assert isinstance(e, Expr)
        assert e.to_sql() == "(price * 1.1)"
    def test_div_operator(self):
        e = col("total") / lit(2)
        assert isinstance(e, Expr)
        assert e.to_sql() == "(total / 2)"
    def test_radd(self):
        e = lit(1) + col("x")
        assert isinstance(e, Expr)
        assert e.to_sql() == "(1 + x)"
    def test_rmul(self):
        e = lit(2) * col("x")
        assert isinstance(e, Expr)
        assert e.to_sql() == "(2 * x)"
    def test_coerce_plain_int(self):
        # Operators should auto-wrap plain Python values via lit()
        e = col("age") > 18
        assert isinstance(e, Expr)
        assert e.to_sql() == "(age > 18)"
    def test_coerce_plain_str(self):
        e = col("name") == "alice"
        assert isinstance(e, Expr)
        assert e.to_sql() == "(name = 'alice')"
 class TestExprStringMethods:
    def test_lower(self):
        e = col("name").lower()
        assert isinstance(e, Expr)
        assert e.to_sql() == "lower(name)"
    def test_upper(self):
        e = col("name").upper()
        assert isinstance(e, Expr)
        assert e.to_sql() == "upper(name)"
    def test_contains(self):
        e = col("text").contains(lit("hello"))
        assert isinstance(e, Expr)
        assert e.to_sql() == "contains(text, 'hello')"
    def test_contains_with_str_coerce(self):
        e = col("text").contains("hello")
        assert isinstance(e, Expr)
        assert e.to_sql() == "contains(text, 'hello')"
    def test_chained_lower_eq(self):
        e = col("name").lower() == lit("alice")
        assert isinstance(e, Expr)
        assert e.to_sql() == "(lower(name) = 'alice')"
 class TestExprCast:
    def test_cast_string(self):
        e = col("id").cast("string")
        assert isinstance(e, Expr)
        assert e.to_sql() == "CAST(id AS VARCHAR)"
    def test_cast_int32(self):
        e = col("score").cast("int32")
        assert isinstance(e, Expr)
        assert e.to_sql() == "CAST(score AS INTEGER)"
    def test_cast_float64(self):
        e = col("val").cast("float64")
        assert isinstance(e, Expr)
        assert e.to_sql() == "CAST(val AS DOUBLE)"
    def test_cast_pyarrow_type(self):
        e = col("score").cast(pa.int32())
        assert isinstance(e, Expr)
        assert e.to_sql() == "CAST(score AS INTEGER)"
    def test_cast_pyarrow_float64(self):
        e = col("val").cast(pa.float64())
        assert isinstance(e, Expr)
        assert e.to_sql() == "CAST(val AS DOUBLE)"
    def test_cast_pyarrow_string(self):
        e = col("id").cast(pa.string())
        assert isinstance(e, Expr)
        assert e.to_sql() == "CAST(id AS VARCHAR)"
    def test_cast_pyarrow_and_string_equivalent(self):
        # pa.int32() and "int32" should produce equivalent SQL
        sql_str = col("x").cast("int32").to_sql()
        sql_pa = col("x").cast(pa.int32()).to_sql()
        assert sql_str == sql_pa
 class TestExprNamedMethods:
    def test_eq_method(self):
        e = col("x").eq(lit(1))
        assert isinstance(e, Expr)
        assert e.to_sql() == "(x = 1)"
    def test_gt_method(self):
        e = col("x").gt(lit(0))
        assert isinstance(e, Expr)
        assert e.to_sql() == "(x > 0)"
    def test_and_method(self):
        e = col("x").gt(lit(0)).and_(col("y").lt(lit(10)))
        assert isinstance(e, Expr)
        assert e.to_sql() == "((x > 0) AND (y < 10))"
    def test_or_method(self):
        e = col("x").eq(lit(1)).or_(col("x").eq(lit(2)))
        assert isinstance(e, Expr)
        assert e.to_sql() == "((x = 1) OR (x = 2))"
 class TestExprRepr:
    def test_repr(self):
        e = col("age") > lit(18)
        assert repr(e) == "Expr((age > 18))"
    def test_to_sql(self):
        e = col("age") > 18
        assert e.to_sql() == "(age > 18)"
    def test_unhashable(self):
        e = col("x")
        with pytest.raises(TypeError):
            {e: 1}
 # ── integration tests: end-to-end query against a real table ─────────────────
@pytest.fixture
 def simple_table(tmp_path):
    db = lancedb.connect(str(tmp_path))
    data = pa.table(
        {
            "id": [1, 2, 3, 4, 5],
            "name": ["Alice", "Bob", "Charlie", "alice", "BOB"],
            "age": [25, 17, 30, 22, 15],
            "score": [1.5, 2.0, 3.5, 4.0, 0.5],
        }
    )
    return db.create_table("test", data)
 class TestExprFilter:
    def test_simple_gt_filter(self, simple_table):
        result = simple_table.search().where(col("age") > lit(20)).to_arrow()
        assert result.num_rows == 3  # ages 25, 30, 22
    def test_compound_and_filter(self, simple_table):
        result = (
            simple_table.search()
            .where((col("age") > lit(18)) & (col("score") > lit(2.0)))
            .to_arrow()
        )
        assert result.num_rows == 2  # (30, 3.5) and (22, 4.0)
    def test_string_equality_filter(self, simple_table):
        result = simple_table.search().where(col("name") == lit("Bob")).to_arrow()
        assert result.num_rows == 1
    def test_or_filter(self, simple_table):
        result = (
            simple_table.search()
            .where((col("age") < lit(18)) | (col("age") > lit(28)))
            .to_arrow()
        )
        assert result.num_rows == 3  # ages 17, 30, 15
    def test_coercion_no_lit(self, simple_table):
        # Python values should be auto-coerced
        result = simple_table.search().where(col("age") > 20).to_arrow()
        assert result.num_rows == 3
    def test_string_sql_still_works(self, simple_table):
        # Backwards compatibility: plain strings still accepted
        result = simple_table.search().where("age > 20").to_arrow()
        assert result.num_rows == 3
 class TestExprProjection:
    def test_select_with_expr(self, simple_table):
        result = (
            simple_table.search()
            .select({"double_score": col("score") * lit(2)})
            .to_arrow()
        )
        assert "double_score" in result.schema.names
    def test_select_mixed_str_and_expr(self, simple_table):
        result = (
            simple_table.search()
            .select({"id": "id", "double_score": col("score") * lit(2)})
            .to_arrow()
        )
        assert "id" in result.schema.names
        assert "double_score" in result.schema.names
    def test_select_list_of_columns(self, simple_table):
        # Plain list of str still works
        result = simple_table.search().select(["id", "name"]).to_arrow()
        assert result.schema.names == ["id", "name"]
 # ── column name edge cases ────────────────────────────────────────────────────
 class TestColNaming:
    """Unit tests verifying that col() preserves identifiers exactly.
    Identifiers that need quoting (camelCase, spaces, leading digits, unicode)
    are wrapped in backticks to match the lance SQL parser's dialect.
    """
    def test_camel_case_preserved_in_sql(self):
        # camelCase is quoted with backticks so the case round-trips correctly.
        assert col("firstName").to_sql() == "`firstName`"
    def test_camel_case_in_expression(self):
        assert (col("firstName") > lit(18)).to_sql() == "(`firstName` > 18)"
    def test_space_in_name_quoted(self):
        assert col("first name").to_sql() == "`first name`"
    def test_space_in_expression(self):
        assert (col("first name") == lit("A")).to_sql() == "(`first name` = 'A')"
    def test_leading_digit_quoted(self):
        assert col("2fast").to_sql() == "`2fast`"
    def test_unicode_quoted(self):
        assert col("名前").to_sql() == "`名前`"
    def test_snake_case_unquoted(self):
        # Plain snake_case needs no quoting.
        assert col("first_name").to_sql() == "first_name"
@pytest.fixture
 def special_col_table(tmp_path):
    db = lancedb.connect(str(tmp_path))
    data = pa.table(
        {
            "firstName": ["Alice", "Bob", "Charlie"],
            "first name": ["A", "B", "C"],
            "score": [10, 20, 30],
        }
    )
    return db.create_table("special", data)
 class TestColNamingIntegration:
    def test_camel_case_filter(self, special_col_table):
        result = (
            special_col_table.search()
            .where(col("firstName") == lit("Alice"))
            .to_arrow()
        )
        assert result.num_rows == 1
        assert result["firstName"][0].as_py() == "Alice"
    def test_space_in_col_filter(self, special_col_table):
        result = (
            special_col_table.search().where(col("first name") == lit("B")).to_arrow()
        )
        assert result.num_rows == 1
    def test_camel_case_projection(self, special_col_table):
        result = (
            special_col_table.search()
            .select({"upper_name": col("firstName").upper()})
            .to_arrow()
        )
        assert "upper_name" in result.schema.names
        assert sorted(result["upper_name"].to_pylist()) == ["ALICE", "BOB", "CHARLIE"]
--- a/rust/lancedb/Cargo.toml
+++ b/rust/lancedb/Cargo.toml
@@ -1,6 +1,6 @@
 [package]
 name = "lancedb"
-version = "0.27.0-beta.3"
+version = "0.27.2"
 edition.workspace = true
 description = "LanceDB: A serverless, low-latency vector database for AI applications"
 license.workspace = true
--- a/rust/lancedb/README.md
+++ b/rust/lancedb/README.md
@@ -1,4 +1,4 @@
-# LanceDB Rust
+# LanceDB Rust SDK
 <a href="https://crates.io/crates/vectordb">![img](https://img.shields.io/crates/v/vectordb)</a>
 <a href="https://docs.rs/vectordb/latest/vectordb/">![Docs.rs](https://img.shields.io/docsrs/vectordb)</a>
--- a/rust/lancedb/src/connection.rs
+++ b/rust/lancedb/src/connection.rs
@@ -596,11 +596,8 @@ pub struct ConnectBuilder {
 }
 #[cfg(feature = "remote")]
-const ENV_VARS_TO_STORAGE_OPTS: [(&str, &str); 3] = [
+const ENV_VARS_TO_STORAGE_OPTS: [(&str, &str); 1] =
-    ("AZURE_STORAGE_ACCOUNT_NAME", "azure_storage_account_name"),
+    [("AZURE_STORAGE_ACCOUNT_NAME", "azure_storage_account_name")];
    ("AZURE_CLIENT_ID", "azure_client_id"),
    ("AZURE_TENANT_ID", "azure_tenant_id"),
 ];
 impl ConnectBuilder {
    /// Create a new [`ConnectOptions`] with the given database URI.
--- a/rust/lancedb/src/database/namespace.rs
+++ b/rust/lancedb/src/database/namespace.rs
@@ -213,25 +213,18 @@ impl Database for LanceNamespaceDatabase {
            ..Default::default()
        };
-        let response = self
+        let (location, initial_storage_options, managed_versioning) = {
-            .namespace
+            let response = self.namespace.declare_table(declare_request).await?;
-            .declare_table(declare_request)
+            let loc = response.location.ok_or_else(|| Error::Runtime {
-            .await
+                message: "Table location is missing from declare_table response".to_string(),
            .map_err(|e| Error::Runtime {
                message: format!("Failed to declare table: {}", e),
            })?;
-
+            // Use storage options from response, fall back to self.storage_options
-        let location = response.location.ok_or_else(|| Error::Runtime {
+            let opts = response
-            message: "Table location is missing from declare_table response".to_string(),
+                .storage_options
-        })?;
+                .or_else(|| Some(self.storage_options.clone()))
-
+                .filter(|o| !o.is_empty());
-        // Use storage options from response, fall back to self.storage_options
+            (loc, opts, response.managed_versioning)
-        let initial_storage_options = response
+        };
            .storage_options
            .or_else(|| Some(self.storage_options.clone()))
            .filter(|o| !o.is_empty());
        let managed_versioning = response.managed_versioning;
        // Build write params with storage options and commit handler
        let mut params = request.write_options.lance_write_params.unwrap_or_default();
--- a/rust/lancedb/src/dataloader/permutation/reader.rs
+++ b/rust/lancedb/src/dataloader/permutation/reader.rs
@@ -339,6 +339,12 @@ impl PermutationReader {
                }
                Ok(false)
            }
            Select::Expr(columns) => {
                // For Expr projections, we check if any alias is _rowid.
                // We can't validate the expression itself (it may differ from _rowid)
                // but we allow it through; the column will be included.
                Ok(columns.iter().any(|(alias, _)| alias == ROW_ID))
            }
        }
    }
--- a/rust/lancedb/src/dataloader/permutation/shuffle.rs
+++ b/rust/lancedb/src/dataloader/permutation/shuffle.rs
@@ -240,7 +240,7 @@ impl Shuffler {
                    .await?;
                    // Need to read the entire file in a single batch for in-memory shuffling
                    let batch = reader.read_record_batch(0, reader.num_rows()).await?;
-                    let mut rng = rng.lock().unwrap();
+                    let mut rng = rng.lock().unwrap_or_else(|e| e.into_inner());
                    Self::shuffle_batch(&batch, &mut rng, clump_size)
                }
            })
--- a/rust/lancedb/src/expr.rs
+++ b/rust/lancedb/src/expr.rs
@@ -27,7 +27,17 @@ use arrow_schema::DataType;
 use datafusion_expr::{Expr, ScalarUDF, expr_fn::cast};
 use datafusion_functions::string::expr_fn as string_expr_fn;
-pub use datafusion_expr::{col, lit};
+pub use datafusion_expr::lit;
 /// Create a column reference expression, preserving the name exactly as given.
 ///
 /// Unlike DataFusion's built-in [`col`][datafusion_expr::col], this function
 /// does **not** normalise the identifier to lower-case, so
 /// `col("firstName")` correctly references a field named `firstName`.
 pub fn col(name: impl Into<String>) -> DfExpr {
    use datafusion_common::Column;
    DfExpr::Column(Column::new_unqualified(name))
 }
 pub use datafusion_expr::Expr as DfExpr;
--- a/rust/lancedb/src/expr/sql.rs
+++ b/rust/lancedb/src/expr/sql.rs
@@ -2,11 +2,37 @@
 // SPDX-FileCopyrightText: Copyright The LanceDB Authors
 use datafusion_expr::Expr;
-use datafusion_sql::unparser;
+use datafusion_sql::unparser::{self, dialect::Dialect};
 /// Unparser dialect that matches the quoting style expected by the Lance SQL
 /// parser.  Lance uses backtick (`` ` ``) as the only delimited-identifier
 /// quote character, so we must produce `` `firstName` `` rather than
 /// `"firstName"` for identifiers that require quoting.
 ///
 /// We quote an identifier when it:
 /// * is a SQL reserved word, OR
 /// * contains characters outside `[a-zA-Z0-9_]`, OR
 /// * starts with a digit, OR
 /// * contains upper-case letters (unquoted identifiers are normalised to
 ///   lower-case by the SQL parser, which would break case-sensitive schemas).
 struct LanceSqlDialect;
 impl Dialect for LanceSqlDialect {
    fn identifier_quote_style(&self, identifier: &str) -> Option<char> {
        let needs_quote = identifier.chars().any(|c| c.is_ascii_uppercase())
            || !identifier
                .chars()
                .enumerate()
                .all(|(i, c)| c == '_' || c.is_ascii_alphabetic() || (i > 0 && c.is_ascii_digit()));
        if needs_quote { Some('`') } else { None }
    }
 }
 pub fn expr_to_sql_string(expr: &Expr) -> crate::Result<String> {
-    let ast = unparser::expr_to_sql(expr).map_err(|e| crate::Error::InvalidInput {
+    let ast = unparser::Unparser::new(&LanceSqlDialect)
-        message: format!("failed to serialize expression to SQL: {}", e),
+        .expr_to_sql(expr)
-    })?;
+        .map_err(|e| crate::Error::InvalidInput {
            message: format!("failed to serialize expression to SQL: {}", e),
        })?;
    Ok(ast.to_string())
 }
--- a/rust/lancedb/src/io/object_store/io_tracking.rs
+++ b/rust/lancedb/src/io/object_store/io_tracking.rs
@@ -66,13 +66,13 @@ impl IoTrackingStore {
    }
    fn record_read(&self, num_bytes: u64) {
-        let mut stats = self.stats.lock().unwrap();
+        let mut stats = self.stats.lock().unwrap_or_else(|e| e.into_inner());
        stats.read_iops += 1;
        stats.read_bytes += num_bytes;
    }
    fn record_write(&self, num_bytes: u64) {
-        let mut stats = self.stats.lock().unwrap();
+        let mut stats = self.stats.lock().unwrap_or_else(|e| e.into_inner());
        stats.write_iops += 1;
        stats.write_bytes += num_bytes;
    }
@@ -229,10 +229,63 @@ impl MultipartUpload for IoTrackingMultipartUpload {
    fn put_part(&mut self, payload: PutPayload) -> UploadPart {
        {
-            let mut stats = self.stats.lock().unwrap();
+            let mut stats = self.stats.lock().unwrap_or_else(|e| e.into_inner());
            stats.write_iops += 1;
            stats.write_bytes += payload.content_length() as u64;
        }
        self.target.put_part(payload)
    }
 }
 #[cfg(test)]
 mod tests {
    use super::*;
    /// Helper: poison a Mutex<IoStats> by panicking while holding the lock.
    fn poison_stats(stats: &Arc<Mutex<IoStats>>) {
        let stats_clone = stats.clone();
        let handle = std::thread::spawn(move || {
            let _guard = stats_clone.lock().unwrap();
            panic!("intentional panic to poison stats mutex");
        });
        let _ = handle.join();
        assert!(stats.lock().is_err(), "mutex should be poisoned");
    }
    #[test]
    fn test_record_read_recovers_from_poisoned_lock() {
        let stats = Arc::new(Mutex::new(IoStats::default()));
        let store = IoTrackingStore {
            target: Arc::new(object_store::memory::InMemory::new()),
            stats: stats.clone(),
        };
        poison_stats(&stats);
        // record_read should not panic
        store.record_read(1024);
        // Verify the stats were updated despite poisoning
        let s = stats.lock().unwrap_or_else(|e| e.into_inner());
        assert_eq!(s.read_iops, 1);
        assert_eq!(s.read_bytes, 1024);
    }
    #[test]
    fn test_record_write_recovers_from_poisoned_lock() {
        let stats = Arc::new(Mutex::new(IoStats::default()));
        let store = IoTrackingStore {
            target: Arc::new(object_store::memory::InMemory::new()),
            stats: stats.clone(),
        };
        poison_stats(&stats);
        // record_write should not panic
        store.record_write(2048);
        let s = stats.lock().unwrap_or_else(|e| e.into_inner());
        assert_eq!(s.write_iops, 1);
        assert_eq!(s.write_bytes, 2048);
    }
 }
--- a/rust/lancedb/src/query.rs
+++ b/rust/lancedb/src/query.rs
@@ -5,7 +5,7 @@ use std::sync::Arc;
 use std::{future::Future, time::Duration};
 use arrow::compute::concat_batches;
-use arrow_array::{Array, Float16Array, Float32Array, Float64Array, make_array};
+use arrow_array::{Array, Float16Array, Float32Array, Float64Array, RecordBatch, make_array};
 use arrow_schema::{DataType, SchemaRef};
 use datafusion_expr::Expr;
 use datafusion_physical_plan::ExecutionPlan;
@@ -17,15 +17,17 @@ use lance_datafusion::exec::execute_plan;
 use lance_index::scalar::FullTextSearchQuery;
 use lance_index::scalar::inverted::SCORE_COL;
 use lance_index::vector::DIST_COL;
 use lance_io::stream::RecordBatchStreamAdapter;
 use crate::DistanceType;
 use crate::error::{Error, Result};
 use crate::rerankers::rrf::RRFReranker;
 use crate::rerankers::{NormalizeMethod, Reranker, check_reranker_result};
 use crate::table::BaseTable;
-use crate::utils::TimeoutStream;
+use crate::utils::{MaxBatchLengthStream, TimeoutStream};
-use crate::{arrow::SendableRecordBatchStream, table::AnyQuery};
+use crate::{
    arrow::{SendableRecordBatchStream, SimpleRecordBatchStream},
    table::AnyQuery,
 };
 mod hybrid;
@@ -47,6 +49,25 @@ pub enum Select {
    ///
    /// See [`Query::select`] for more details and examples
    Dynamic(Vec<(String, String)>),
    /// Advanced selection using type-safe DataFusion expressions
    ///
    /// Similar to [`Select::Dynamic`] but uses [`datafusion_expr::Expr`] instead of
    /// raw SQL strings. Use [`crate::expr`] helpers to build expressions:
    ///
    /// ```
    /// use lancedb::expr::{col, lit};
    /// use lancedb::query::Select;
    ///
    /// // SELECT id, id * 2 AS id2 FROM ...
    /// let selection = Select::expr_projection(&[
    ///     ("id", col("id")),
    ///     ("id2", col("id") * lit(2)),
    /// ]);
    /// ```
    ///
    /// Note: For remote/server-side queries the expressions are serialized to SQL strings
    /// automatically (same as [`Select::Dynamic`]).
    Expr(Vec<(String, datafusion_expr::Expr)>),
 }
 impl Select {
@@ -69,6 +90,29 @@ impl Select {
                .collect(),
        )
    }
    /// Create a typed-expression projection.
    ///
    /// This is a convenience method for creating a [`Select::Expr`] variant from
    /// a slice of `(name, expr)` pairs where each `expr` is a [`datafusion_expr::Expr`].
    ///
    /// # Example
    /// ```
    /// use lancedb::expr::{col, lit};
    /// use lancedb::query::Select;
    ///
    /// let selection = Select::expr_projection(&[
    ///     ("id", col("id")),
    ///     ("id2", col("id") * lit(2)),
    /// ]);
    /// ```
    pub fn expr_projection(columns: &[(impl AsRef<str>, datafusion_expr::Expr)]) -> Self {
        Self::Expr(
            columns
                .iter()
                .map(|(name, expr)| (name.as_ref().to_string(), expr.clone()))
                .collect(),
        )
    }
 }
 /// A trait for converting a type to a query vector
@@ -562,6 +606,14 @@ impl Default for QueryExecutionOptions {
    }
 }
 impl QueryExecutionOptions {
    fn without_output_batch_length_limit(&self) -> Self {
        let mut options = self.clone();
        options.max_batch_length = 0;
        options
    }
 }
 /// A trait for a query object that can be executed to get results
 ///
 /// There are various kinds of queries but they all return results
@@ -1138,6 +1190,8 @@ impl VectorQuery {
        &self,
        options: QueryExecutionOptions,
    ) -> Result<SendableRecordBatchStream> {
        let max_batch_length = options.max_batch_length as usize;
        let internal_options = options.without_output_batch_length_limit();
        // clone query and specify we want to include row IDs, which can be needed for reranking
        let mut fts_query = Query::new(self.parent.clone());
        fts_query.request = self.request.base.clone();
@@ -1147,8 +1201,8 @@ impl VectorQuery {
        vector_query.request.base.full_text_search = None;
        let (fts_results, vec_results) = try_join!(
-            fts_query.execute_with_options(options.clone()),
+            fts_query.execute_with_options(internal_options.clone()),
-            vector_query.inner_execute_with_options(options)
+            vector_query.inner_execute_with_options(internal_options)
        )?;
        let (fts_results, vec_results) = try_join!(
@@ -1203,9 +1257,7 @@ impl VectorQuery {
            results = results.drop_column(ROW_ID)?;
        }
-        Ok(SendableRecordBatchStream::from(
+        Ok(single_batch_stream(results, max_batch_length))
            RecordBatchStreamAdapter::new(results.schema(), stream::iter([Ok(results)])),
        ))
    }
    async fn inner_execute_with_options(
@@ -1214,6 +1266,7 @@ impl VectorQuery {
    ) -> Result<SendableRecordBatchStream> {
        let plan = self.create_plan(options.clone()).await?;
        let inner = execute_plan(plan, Default::default())?;
        let inner = MaxBatchLengthStream::new_boxed(inner, options.max_batch_length as usize);
        let inner = if let Some(timeout) = options.timeout {
            TimeoutStream::new_boxed(inner, timeout)
        } else {
@@ -1223,6 +1276,25 @@ impl VectorQuery {
    }
 }
 fn single_batch_stream(batch: RecordBatch, max_batch_length: usize) -> SendableRecordBatchStream {
    let schema = batch.schema();
    if max_batch_length == 0 || batch.num_rows() <= max_batch_length {
        return Box::pin(SimpleRecordBatchStream::new(
            stream::iter([Ok(batch)]),
            schema,
        ));
    }
    let mut batches = Vec::with_capacity(batch.num_rows().div_ceil(max_batch_length));
    let mut offset = 0;
    while offset < batch.num_rows() {
        let length = (batch.num_rows() - offset).min(max_batch_length);
        batches.push(Ok(batch.slice(offset, length)));
        offset += length;
    }
    Box::pin(SimpleRecordBatchStream::new(stream::iter(batches), schema))
 }
 impl ExecutableQuery for VectorQuery {
    async fn create_plan(&self, options: QueryExecutionOptions) -> Result<Arc<dyn ExecutionPlan>> {
        let query = AnyQuery::VectorQuery(self.request.clone());
@@ -1591,6 +1663,58 @@ mod tests {
        });
    }
    #[tokio::test]
    async fn test_select_with_expr_projection() {
        // Mirrors test_select_with_transform but uses Select::Expr instead of Select::Dynamic
        let tmp_dir = tempdir().unwrap();
        let dataset_path = tmp_dir.path().join("test_expr.lance");
        let uri = dataset_path.to_str().unwrap();
        let batches = make_non_empty_batches();
        let conn = connect(uri).execute().await.unwrap();
        let table = conn
            .create_table("my_table", batches)
            .execute()
            .await
            .unwrap();
        use crate::expr::{col, lit};
        let query = table.query().limit(10).select(Select::expr_projection(&[
            ("id2", col("id") * lit(2i32)),
            ("id", col("id")),
        ]));
        let schema = query.output_schema().await.unwrap();
        assert_eq!(
            schema,
            Arc::new(ArrowSchema::new(vec![
                ArrowField::new("id2", DataType::Int32, true),
                ArrowField::new("id", DataType::Int32, true),
            ]))
        );
        let result = query.execute().await;
        let mut batches = result
            .expect("should have result")
            .try_collect::<Vec<_>>()
            .await
            .unwrap();
        assert_eq!(batches.len(), 1);
        let batch = batches.pop().unwrap();
        // id and id2
        assert_eq!(batch.num_columns(), 2);
        let id: &Int32Array = batch.column_by_name("id").unwrap().as_primitive();
        let id2: &Int32Array = batch.column_by_name("id2").unwrap().as_primitive();
        id.iter().zip(id2.iter()).for_each(|(id, id2)| {
            let id = id.unwrap();
            let id2 = id2.unwrap();
            assert_eq!(id * 2, id2);
        });
    }
    #[tokio::test]
    async fn test_execute_no_vector() {
        // TODO: Switch back to memory://foo after https://github.com/lancedb/lancedb/issues/1051
@@ -1659,6 +1783,50 @@ mod tests {
            .unwrap()
    }
    async fn make_large_vector_table(tmp_dir: &tempfile::TempDir, rows: usize) -> Table {
        let dataset_path = tmp_dir.path().join("large_test.lance");
        let uri = dataset_path.to_str().unwrap();
        let schema = Arc::new(ArrowSchema::new(vec![
            ArrowField::new("id", DataType::Utf8, false),
            ArrowField::new(
                "vector",
                DataType::FixedSizeList(
                    Arc::new(ArrowField::new("item", DataType::Float32, true)),
                    4,
                ),
                false,
            ),
        ]));
        let ids = StringArray::from_iter_values((0..rows).map(|i| format!("row-{i}")));
        let vectors = FixedSizeListArray::from_iter_primitive::<Float32Type, _, _>(
            (0..rows).map(|i| Some(vec![Some(i as f32), Some(1.0), Some(2.0), Some(3.0)])),
            4,
        );
        let batch =
            RecordBatch::try_new(schema.clone(), vec![Arc::new(ids), Arc::new(vectors)]).unwrap();
        let conn = connect(uri).execute().await.unwrap();
        conn.create_table("my_table", vec![batch])
            .execute()
            .await
            .unwrap()
    }
    async fn assert_stream_batches_at_most(
        mut results: SendableRecordBatchStream,
        max_batch_length: usize,
    ) {
        let mut saw_batch = false;
        while let Some(batch) = results.next().await {
            let batch = batch.unwrap();
            saw_batch = true;
            assert!(batch.num_rows() <= max_batch_length);
        }
        assert!(saw_batch);
    }
    #[tokio::test]
    async fn test_execute_with_options() {
        let tmp_dir = tempdir().unwrap();
@@ -1678,6 +1846,83 @@ mod tests {
        }
    }
    #[tokio::test]
    async fn test_vector_query_execute_with_options_respects_max_batch_length() {
        let tmp_dir = tempdir().unwrap();
        let table = make_large_vector_table(&tmp_dir, 10_000).await;
        let results = table
            .query()
            .nearest_to(vec![0.0, 1.0, 2.0, 3.0])
            .unwrap()
            .limit(10_000)
            .execute_with_options(QueryExecutionOptions {
                max_batch_length: 100,
                ..Default::default()
            })
            .await
            .unwrap();
        assert_stream_batches_at_most(results, 100).await;
    }
    #[tokio::test]
    async fn test_hybrid_query_execute_with_options_respects_max_batch_length() {
        let tmp_dir = tempdir().unwrap();
        let dataset_path = tmp_dir.path();
        let conn = connect(dataset_path.to_str().unwrap())
            .execute()
            .await
            .unwrap();
        let dims = 2;
        let rows = 512;
        let schema = Arc::new(ArrowSchema::new(vec![
            ArrowField::new("text", DataType::Utf8, false),
            ArrowField::new(
                "vector",
                DataType::FixedSizeList(
                    Arc::new(ArrowField::new("item", DataType::Float32, true)),
                    dims,
                ),
                false,
            ),
        ]));
        let text = StringArray::from_iter_values((0..rows).map(|_| "match"));
        let vectors = FixedSizeListArray::from_iter_primitive::<Float32Type, _, _>(
            (0..rows).map(|i| Some(vec![Some(i as f32), Some(0.0)])),
            dims,
        );
        let record_batch =
            RecordBatch::try_new(schema.clone(), vec![Arc::new(text), Arc::new(vectors)]).unwrap();
        let table = conn
            .create_table("my_table", record_batch)
            .execute()
            .await
            .unwrap();
        table
            .create_index(&["text"], crate::index::Index::FTS(Default::default()))
            .replace(true)
            .execute()
            .await
            .unwrap();
        let results = table
            .query()
            .full_text_search(FullTextSearchQuery::new("match".to_string()))
            .limit(rows)
            .nearest_to(&[0.0, 0.0])
            .unwrap()
            .execute_with_options(QueryExecutionOptions {
                max_batch_length: 100,
                ..Default::default()
            })
            .await
            .unwrap();
        assert_stream_batches_at_most(results, 100).await;
    }
    #[tokio::test]
    async fn test_analyze_plan() {
        let tmp_dir = tempdir().unwrap();
--- a/rust/lancedb/src/remote/client.rs
+++ b/rust/lancedb/src/remote/client.rs
@@ -426,14 +426,11 @@ impl<S: HttpSend> RestfulLanceDbClient<S> {
                })?,
            );
        }
-        if db_prefix.is_some() {
+        if let Some(prefix) = db_prefix {
            headers.insert(
                HeaderName::from_static("x-lancedb-database-prefix"),
-                HeaderValue::from_str(db_prefix.unwrap()).map_err(|_| Error::InvalidInput {
+                HeaderValue::from_str(prefix).map_err(|_| Error::InvalidInput {
-                    message: format!(
+                    message: format!("non-ascii database prefix '{}' provided", prefix),
                        "non-ascii database prefix '{}' provided",
                        db_prefix.unwrap()
                    ),
                })?,
            );
        }
@@ -446,23 +443,13 @@ impl<S: HttpSend> RestfulLanceDbClient<S> {
                })?,
            );
        }
-        // Map azure storage options to x-azure-* headers.
+        if let Some(v) = options.0.get("azure_storage_account_name") {
-        // The option key uses underscores (e.g. "azure_client_id") while the
+            headers.insert(
-        // header uses hyphens (e.g. "x-azure-client-id").
+                HeaderName::from_static("x-azure-storage-account-name"),
-        let azure_opts: [(&str, &str); 3] = [
+                HeaderValue::from_str(v).map_err(|_| Error::InvalidInput {
-            ("azure_storage_account_name", "x-azure-storage-account-name"),
+                    message: format!("non-ascii storage account name '{}' provided", db_name),
-            ("azure_client_id", "x-azure-client-id"),
+                })?,
-            ("azure_tenant_id", "x-azure-tenant-id"),
+            );
        ];
        for (opt_key, header_name) in azure_opts {
            if let Some(v) = options.0.get(opt_key) {
                headers.insert(
                    HeaderName::from_static(header_name),
                    HeaderValue::from_str(v).map_err(|_| Error::InvalidInput {
                        message: format!("non-ascii value '{}' for option '{}'", v, opt_key),
                    })?,
                );
            }
        }
        for (key, value) in &config.extra_headers {
@@ -1085,34 +1072,4 @@ mod tests {
            _ => panic!("Expected Runtime error"),
        }
    }
    #[test]
    fn test_default_headers_azure_opts() {
        let mut opts = HashMap::new();
        opts.insert(
            "azure_storage_account_name".to_string(),
            "myaccount".to_string(),
        );
        opts.insert("azure_client_id".to_string(), "my-client-id".to_string());
        opts.insert("azure_tenant_id".to_string(), "my-tenant-id".to_string());
        let remote_opts = RemoteOptions::new(opts);
        let headers = RestfulLanceDbClient::<Sender>::default_headers(
            "test-key",
            "us-east-1",
            "testdb",
            false,
            &remote_opts,
            None,
            &ClientConfig::default(),
        )
        .unwrap();
        assert_eq!(
            headers.get("x-azure-storage-account-name").unwrap(),
            "myaccount"
        );
        assert_eq!(headers.get("x-azure-client-id").unwrap(), "my-client-id");
        assert_eq!(headers.get("x-azure-tenant-id").unwrap(), "my-tenant-id");
    }
 }
--- a/rust/lancedb/src/remote/db.rs
+++ b/rust/lancedb/src/remote/db.rs
@@ -72,6 +72,10 @@ impl ServerVersion {
    pub fn support_structural_fts(&self) -> bool {
        self.0 >= semver::Version::new(0, 3, 0)
    }
    pub fn support_multipart_write(&self) -> bool {
        self.0 >= semver::Version::new(0, 4, 0)
    }
 }
 pub const OPT_REMOTE_PREFIX: &str = "remote_database_";
@@ -778,12 +782,7 @@ impl RemoteOptions {
 impl From<StorageOptions> for RemoteOptions {
    fn from(options: StorageOptions) -> Self {
-        let supported_opts = vec![
+        let supported_opts = vec!["account_name", "azure_storage_account_name"];
            "account_name",
            "azure_storage_account_name",
            "azure_client_id",
            "azure_tenant_id",
        ];
        let mut filtered = HashMap::new();
        for opt in supported_opts {
            if let Some(v) = options.0.get(opt) {
--- a/rust/lancedb/src/remote/table.rs
+++ b/rust/lancedb/src/remote/table.rs
--- a/rust/lancedb/src/remote/table/insert.rs
+++ b/rust/lancedb/src/remote/table/insert.rs
@@ -11,10 +11,14 @@ use arrow_ipc::CompressionType;
 use datafusion_common::{DataFusionError, Result as DataFusionResult};
 use datafusion_execution::{SendableRecordBatchStream, TaskContext};
 use datafusion_physical_expr::EquivalenceProperties;
 use datafusion_physical_plan::metrics::{ExecutionPlanMetricsSet, MetricsSet};
 use datafusion_physical_plan::stream::RecordBatchStreamAdapter;
-use datafusion_physical_plan::{DisplayAs, DisplayFormatType, ExecutionPlan, PlanProperties};
+use datafusion_physical_plan::{
    DisplayAs, DisplayFormatType, ExecutionPlan, ExecutionPlanProperties, PlanProperties,
 };
 use futures::StreamExt;
 use http::header::CONTENT_TYPE;
 use lance::io::exec::utils::InstrumentedRecordBatchStreamAdapter;
 use crate::Error;
 use crate::remote::ARROW_STREAM_CONTENT_TYPE;
@@ -22,13 +26,16 @@ use crate::remote::client::{HttpSend, RestfulLanceDbClient, Sender};
 use crate::remote::table::RemoteTable;
 use crate::table::AddResult;
 use crate::table::datafusion::insert::COUNT_SCHEMA;
 use crate::table::write_progress::WriteProgressTracker;
 /// ExecutionPlan for inserting data into a remote LanceDB table.
 ///
-/// This plan:
+/// Streams data as Arrow IPC to `/v1/table/{id}/insert/` endpoint.
-/// 1. Requires single partition (no parallel remote inserts yet)
+///
-/// 2. Streams data as Arrow IPC to `/v1/table/{id}/insert/` endpoint
+/// When `upload_id` is set, inserts are staged as part of a multipart write
-/// 3. Stores AddResult for retrieval after execution
+/// session and the plan supports multiple partitions for parallel uploads.
 /// Without `upload_id`, the plan requires a single partition and commits
 /// immediately.
 #[derive(Debug)]
 pub struct RemoteInsertExec<S: HttpSend = Sender> {
    table_name: String,
@@ -38,21 +45,69 @@ pub struct RemoteInsertExec<S: HttpSend = Sender> {
    overwrite: bool,
    properties: PlanProperties,
    add_result: Arc<Mutex<Option<AddResult>>>,
    metrics: ExecutionPlanMetricsSet,
    upload_id: Option<String>,
    tracker: Option<Arc<WriteProgressTracker>>,
 }
 impl<S: HttpSend + 'static> RemoteInsertExec<S> {
-    /// Create a new RemoteInsertExec.
+    /// Create a new single-partition RemoteInsertExec.
    pub fn new(
        table_name: String,
        identifier: String,
        client: RestfulLanceDbClient<S>,
        input: Arc<dyn ExecutionPlan>,
        overwrite: bool,
        tracker: Option<Arc<WriteProgressTracker>>,
    ) -> Self {
        Self::new_inner(
            table_name, identifier, client, input, overwrite, None, tracker,
        )
    }
    /// Create a multi-partition RemoteInsertExec for use with multipart writes.
    ///
    /// Each partition's insert is staged under the given `upload_id` without
    /// committing. The caller is responsible for calling the complete (or abort)
    /// endpoint after all partitions finish.
    pub fn new_multipart(
        table_name: String,
        identifier: String,
        client: RestfulLanceDbClient<S>,
        input: Arc<dyn ExecutionPlan>,
        overwrite: bool,
        upload_id: String,
        tracker: Option<Arc<WriteProgressTracker>>,
    ) -> Self {
        Self::new_inner(
            table_name,
            identifier,
            client,
            input,
            overwrite,
            Some(upload_id),
            tracker,
        )
    }
    fn new_inner(
        table_name: String,
        identifier: String,
        client: RestfulLanceDbClient<S>,
        input: Arc<dyn ExecutionPlan>,
        overwrite: bool,
        upload_id: Option<String>,
        tracker: Option<Arc<WriteProgressTracker>>,
    ) -> Self {
        let num_partitions = if upload_id.is_some() {
            input.output_partitioning().partition_count()
        } else {
            1
        };
        let schema = COUNT_SCHEMA.clone();
        let properties = PlanProperties::new(
            EquivalenceProperties::new(schema),
-            datafusion_physical_plan::Partitioning::UnknownPartitioning(1),
+            datafusion_physical_plan::Partitioning::UnknownPartitioning(num_partitions),
            datafusion_physical_plan::execution_plan::EmissionType::Final,
            datafusion_physical_plan::execution_plan::Boundedness::Bounded,
        );
@@ -65,6 +120,9 @@ impl<S: HttpSend + 'static> RemoteInsertExec<S> {
            overwrite,
            properties,
            add_result: Arc::new(Mutex::new(None)),
            metrics: ExecutionPlanMetricsSet::new(),
            upload_id,
            tracker,
        }
    }
@@ -72,7 +130,10 @@ impl<S: HttpSend + 'static> RemoteInsertExec<S> {
    // TODO: this will be used when we wire this up to Table::add().
    #[allow(dead_code)]
    pub fn add_result(&self) -> Option<AddResult> {
-        self.add_result.lock().unwrap().clone()
+        self.add_result
            .lock()
            .unwrap_or_else(|e| e.into_inner())
            .clone()
    }
    /// Stream the input into an HTTP body as an Arrow IPC stream, capturing any
@@ -83,6 +144,7 @@ impl<S: HttpSend + 'static> RemoteInsertExec<S> {
    fn stream_as_http_body(
        data: SendableRecordBatchStream,
        error_tx: tokio::sync::oneshot::Sender<DataFusionError>,
        tracker: Option<Arc<WriteProgressTracker>>,
    ) -> DataFusionResult<reqwest::Body> {
        let options = arrow_ipc::writer::IpcWriteOptions::default()
            .try_with_compression(Some(CompressionType::LZ4_FRAME))?;
@@ -94,37 +156,46 @@ impl<S: HttpSend + 'static> RemoteInsertExec<S> {
        let stream = futures::stream::try_unfold(
            (data, writer, Some(error_tx), false),
-            move |(mut data, mut writer, error_tx, finished)| async move {
+            move |(mut data, mut writer, error_tx, finished)| {
-                if finished {
+                let tracker = tracker.clone();
-                    return Ok(None);
+                async move {
-                }
+                    if finished {
-                match data.next().await {
+                        return Ok(None);
                    Some(Ok(batch)) => {
                        writer
                            .write(&batch)
                            .map_err(|e| std::io::Error::other(e.to_string()))?;
                        let buffer = std::mem::take(writer.get_mut());
                        Ok(Some((buffer, (data, writer, error_tx, false))))
                    }
-                    Some(Err(e)) => {
+                    match data.next().await {
-                        // Send the original error through the channel before
+                        Some(Ok(batch)) => {
-                        // returning a generic error to reqwest.
+                            writer
-                        if let Some(tx) = error_tx {
+                                .write(&batch)
-                            let _ = tx.send(e);
+                                .map_err(|e| std::io::Error::other(e.to_string()))?;
                            let buffer = std::mem::take(writer.get_mut());
                            if let Some(ref t) = tracker {
                                t.record_bytes(buffer.len());
                            }
                            Ok(Some((buffer, (data, writer, error_tx, false))))
                        }
-                        Err(std::io::Error::other(
+                        Some(Err(e)) => {
-                            "input stream error (see error channel)",
+                            // Send the original error through the channel before
-                        ))
+                            // returning a generic error to reqwest.
-                    }
+                            if let Some(tx) = error_tx {
-                    None => {
+                                let _ = tx.send(e);
-                        writer
+                            }
-                            .finish()
+                            Err(std::io::Error::other(
-                            .map_err(|e| std::io::Error::other(e.to_string()))?;
+                                "input stream error (see error channel)",
-                        let buffer = std::mem::take(writer.get_mut());
+                            ))
-                        if buffer.is_empty() {
+                        }
-                            Ok(None)
+                        None => {
-                        } else {
+                            writer
-                            Ok(Some((buffer, (data, writer, None, true))))
+                                .finish()
                                .map_err(|e| std::io::Error::other(e.to_string()))?;
                            let buffer = std::mem::take(writer.get_mut());
                            if buffer.is_empty() {
                                Ok(None)
                            } else {
                                if let Some(ref t) = tracker {
                                    t.record_bytes(buffer.len());
                                }
                                Ok(Some((buffer, (data, writer, None, true))))
                            }
                        }
                    }
                }
@@ -174,8 +245,11 @@ impl<S: HttpSend + 'static> ExecutionPlan for RemoteInsertExec<S> {
    }
    fn required_input_distribution(&self) -> Vec<datafusion_physical_plan::Distribution> {
-        // Until we have a separate commit endpoint, we need to do all inserts in a single partition
+        if self.upload_id.is_some() {
-        vec![datafusion_physical_plan::Distribution::SinglePartition]
+            vec![datafusion_physical_plan::Distribution::UnspecifiedDistribution]
        } else {
            vec![datafusion_physical_plan::Distribution::SinglePartition]
        }
    }
    fn benefits_from_input_partitioning(&self) -> Vec<bool> {
@@ -191,12 +265,14 @@ impl<S: HttpSend + 'static> ExecutionPlan for RemoteInsertExec<S> {
                "RemoteInsertExec requires exactly one child".to_string(),
            ));
        }
-        Ok(Arc::new(Self::new(
+        Ok(Arc::new(Self::new_inner(
            self.table_name.clone(),
            self.identifier.clone(),
            self.client.clone(),
            children[0].clone(),
            self.overwrite,
            self.upload_id.clone(),
            self.tracker.clone(),
        )))
    }
@@ -205,18 +281,29 @@ impl<S: HttpSend + 'static> ExecutionPlan for RemoteInsertExec<S> {
        partition: usize,
        context: Arc<TaskContext>,
    ) -> DataFusionResult<SendableRecordBatchStream> {
-        if partition != 0 {
+        if self.upload_id.is_none() && partition != 0 {
            return Err(DataFusionError::Internal(
-                "RemoteInsertExec only supports single partition execution".to_string(),
+                "RemoteInsertExec only supports single partition execution without upload_id"
                    .to_string(),
            ));
        }
-        let input_stream = self.input.execute(0, context)?;
+        let input_stream = self.input.execute(partition, context)?;
        let input_schema = input_stream.schema();
        let input_stream: SendableRecordBatchStream =
            Box::pin(InstrumentedRecordBatchStreamAdapter::new(
                input_schema,
                input_stream,
                partition,
                &self.metrics,
            ));
        let client = self.client.clone();
        let identifier = self.identifier.clone();
        let overwrite = self.overwrite;
        let add_result = self.add_result.clone();
        let table_name = self.table_name.clone();
        let upload_id = self.upload_id.clone();
        let tracker = self.tracker.clone();
        let stream = futures::stream::once(async move {
            let mut request = client
@@ -226,9 +313,12 @@ impl<S: HttpSend + 'static> ExecutionPlan for RemoteInsertExec<S> {
            if overwrite {
                request = request.query(&[("mode", "overwrite")]);
            }
            if let Some(ref uid) = upload_id {
                request = request.query(&[("upload_id", uid.as_str())]);
            }
            let (error_tx, mut error_rx) = tokio::sync::oneshot::channel();
-            let body = Self::stream_as_http_body(input_stream, error_tx)?;
+            let body = Self::stream_as_http_body(input_stream, error_tx, tracker)?;
            let request = request.body(body);
            let result: DataFusionResult<(String, _)> = async {
@@ -262,32 +352,43 @@ impl<S: HttpSend + 'static> ExecutionPlan for RemoteInsertExec<S> {
            let (request_id, response) = result?;
-            let body_text = response.text().await.map_err(|e| {
+            // For multipart writes, the staging response is not the final
-                DataFusionError::External(Box::new(Error::Http {
+            // version. Only parse AddResult for non-multipart inserts.
-                    source: Box::new(e),
+            if upload_id.is_none() {
-                    request_id: request_id.clone(),
+                let body_text = response.text().await.map_err(|e| {
                    status_code: None,
                }))
            })?;
            let parsed_result = if body_text.trim().is_empty() {
                // Backward compatible with old servers
                AddResult { version: 0 }
            } else {
                serde_json::from_str(&body_text).map_err(|e| {
                    DataFusionError::External(Box::new(Error::Http {
-                        source: format!("Failed to parse add response: {}", e).into(),
+                        source: Box::new(e),
                        request_id: request_id.clone(),
                        status_code: None,
                    }))
-                })?
+                })?;
-            };
+
                let parsed_result = if body_text.trim().is_empty() {
                    // Backward compatible with old servers
                    AddResult { version: 0 }
                } else {
                    serde_json::from_str(&body_text).map_err(|e| {
                        DataFusionError::External(Box::new(Error::Http {
                            source: format!("Failed to parse add response: {}", e).into(),
                            request_id: request_id.clone(),
                            status_code: None,
                        }))
                    })?
                };
            {
                let mut res_lock = add_result.lock().map_err(|_| {
                    DataFusionError::Execution("Failed to acquire lock for add_result".to_string())
                })?;
                *res_lock = Some(parsed_result);
            } else {
                // We don't use the body in this case, but we should still consume it.
                let _ = response.bytes().await.map_err(|e| {
                    DataFusionError::External(Box::new(Error::Http {
                        source: Box::new(e),
                        request_id: request_id.clone(),
                        status_code: None,
                    }))
                })?;
            }
            // Return a single batch with count 0 (actual count is tracked in add_result)
@@ -301,6 +402,10 @@ impl<S: HttpSend + 'static> ExecutionPlan for RemoteInsertExec<S> {
            stream,
        )))
    }
    fn metrics(&self) -> Option<MetricsSet> {
        Some(self.metrics.clone_inner())
    }
 }
 #[cfg(test)]
--- a/rust/lancedb/src/table.rs
+++ b/rust/lancedb/src/table.rs
@@ -74,7 +74,10 @@ pub mod optimize;
 pub mod query;
 pub mod schema_evolution;
 pub mod update;
 pub mod write_progress;
 use crate::index::waiter::wait_for_index;
 #[cfg(feature = "remote")]
 pub(crate) use add_data::PreprocessingOutput;
 pub use add_data::{AddDataBuilder, AddDataMode, AddResult, NaNVectorBehavior};
 pub use chrono::Duration;
 pub use delete::DeleteResult;
@@ -277,8 +280,13 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
    async fn list_indices(&self) -> Result<Vec<IndexConfig>>;
    /// Drop an index from the table.
    async fn drop_index(&self, name: &str) -> Result<()>;
-    /// Prewarm an index in the table
+    /// Prewarm an index in the table.
    async fn prewarm_index(&self, name: &str) -> Result<()>;
    /// Prewarm data for the table.
    ///
    /// Currently only supported on remote tables.
    /// If `columns` is `None`, all columns are prewarmed.
    async fn prewarm_data(&self, columns: Option<Vec<String>>) -> Result<()>;
    /// Get statistics about the index.
    async fn index_stats(&self, index_name: &str) -> Result<Option<IndexStatistics>>;
    /// Merge insert new records into the table.
@@ -435,6 +443,34 @@ mod test_utils {
                embedding_registry: Arc::new(MemoryRegistry::new()),
            }
        }
        pub fn new_with_handler_version_and_config<T>(
            name: impl Into<String>,
            version: semver::Version,
            handler: impl Fn(reqwest::Request) -> http::Response<T> + Clone + Send + Sync + 'static,
            config: crate::remote::ClientConfig,
        ) -> Self
        where
            T: Into<reqwest::Body>,
        {
            let inner = Arc::new(
                crate::remote::table::RemoteTable::new_mock_with_version_and_config(
                    name.into(),
                    handler.clone(),
                    Some(version),
                    config.clone(),
                ),
            );
            let database = Arc::new(crate::remote::db::RemoteDatabase::new_mock_with_config(
                handler, config,
            ));
            Self {
                inner,
                database: Some(database),
                // Registry is unused.
                embedding_registry: Arc::new(MemoryRegistry::new()),
            }
        }
    }
 }
@@ -946,17 +982,7 @@ impl Table {
    ///  * Prune: Removes old versions of the dataset
    ///  * Index: Optimizes the indices, adding new data to existing indices
    ///
-    /// <section class="warning">Experimental API</section>
+    /// The frequency an application should call optimize is based on the frequency of
    ///
    /// The optimization process is undergoing active development and may change.
    /// Our goal with these changes is to improve the performance of optimization and
    /// reduce the complexity.
    ///
    /// That being said, it is essential today to run optimize if you want the best
    /// performance.  It should be stable and safe to use in production, but it our
    /// hope that the API may be simplified (or not even need to be called) in the future.
    ///
    /// The frequency an application shoudl call optimize is based on the frequency of
    /// data modifications.  If data is frequently added, deleted, or updated then
    /// optimize should be run frequently.  A good rule of thumb is to run optimize if
    /// you have added or modified 100,000 or more records or run more than 20 data
@@ -1123,22 +1149,45 @@ impl Table {
        self.inner.drop_index(name).await
    }
-    /// Prewarm an index in the table
+    /// Prewarm an index in the table.
    ///
-    /// This is a hint to fully load the index into memory.  It can be used to
+    /// This is a hint to the database that the index will be accessed in the
-    /// avoid cold starts
+    /// future and should be loaded into memory if possible.  This can reduce
    /// cold-start latency for subsequent queries.
    ///
    /// This call initiates prewarming and returns once the request is accepted.
    /// It is idempotent and safe to call from multiple clients concurrently.
    ///
    /// It is generally wasteful to call this if the index does not fit into the
-    /// available cache.
+    /// available cache.  Not all index types support prewarming; unsupported
-    ///
+    /// indices will silently ignore the request.
    /// Note: This function is not yet supported on all indices, in which case it
    /// may do nothing.
    ///
    /// Use [`Self::list_indices()`] to find the names of the indices.
    pub async fn prewarm_index(&self, name: &str) -> Result<()> {
        self.inner.prewarm_index(name).await
    }
    /// Prewarm data for the table.
    ///
    /// This is a hint to the database that the given columns will be accessed in
    /// the future and the database should prefetch the data if possible.  This
    /// can reduce cold-start latency for subsequent queries.  Currently only
    /// supported on remote tables.
    ///
    /// This call initiates prewarming and returns once the request is accepted.
    /// It is idempotent and safe to call from multiple clients concurrently —
    /// calling it on already-prewarmed columns is a no-op on the server.
    ///
    /// This operation has a large upfront cost but can speed up future queries
    /// that need to fetch the given columns.  Large columns such as embeddings
    /// or binary data may not be practical to prewarm.  This feature is intended
    /// for workloads that issue many queries against the same columns.
    ///
    /// If `columns` is `None`, all columns are prewarmed.
    pub async fn prewarm_data(&self, columns: Option<Vec<String>>) -> Result<()> {
        self.inner.prewarm_data(columns).await
    }
    /// Poll until the columns are fully indexed. Will return Error::Timeout if the columns
    /// are not fully indexed within the timeout.
    pub async fn wait_for_index(
@@ -2180,21 +2229,26 @@ impl BaseTable for NativeTable {
        let table_schema = Schema::from(&ds.schema().clone());
-        // Peek at the first batch to estimate a good partition count for
+        let num_partitions = if let Some(parallelism) = add.write_parallelism {
-        // write parallelism.
+            parallelism
        let mut peeked = PeekedScannable::new(add.data);
        let num_partitions = if let Some(first_batch) = peeked.peek().await {
            let max_partitions = lance_core::utils::tokio::get_num_compute_intensive_cpus();
            estimate_write_partitions(
                first_batch.get_array_memory_size(),
                first_batch.num_rows(),
                peeked.num_rows(),
                max_partitions,
            )
        } else {
-            1
+            // Peek at the first batch to estimate a good partition count for
            // write parallelism.
            let mut peeked = PeekedScannable::new(add.data);
            let n = if let Some(first_batch) = peeked.peek().await {
                let max_partitions = lance_core::utils::tokio::get_num_compute_intensive_cpus();
                estimate_write_partitions(
                    first_batch.get_array_memory_size(),
                    first_batch.num_rows(),
                    peeked.num_rows(),
                    max_partitions,
                )
            } else {
                1
            };
            add.data = Box::new(peeked);
            n
        };
        add.data = Box::new(peeked);
        let output = add.into_plan(&table_schema, &table_def)?;
@@ -2223,13 +2277,21 @@ impl BaseTable for NativeTable {
        let insert_exec = Arc::new(InsertExec::new(ds_wrapper.clone(), ds, plan, lance_params));
        let tracker_for_tasks = output.tracker.clone();
        if let Some(ref t) = tracker_for_tasks {
            t.set_total_tasks(num_partitions);
        }
        let _finish = write_progress::FinishOnDrop(output.tracker);
        // Execute all partitions in parallel.
        let task_ctx = Arc::new(TaskContext::default());
        let handles = FuturesUnordered::new();
        for partition in 0..num_partitions {
            let exec = insert_exec.clone();
            let ctx = task_ctx.clone();
            let tracker = tracker_for_tasks.clone();
            handles.push(tokio::spawn(async move {
                let _guard = tracker.as_ref().map(|t| t.track_task());
                let mut stream = exec
                    .execute(partition, ctx)
                    .map_err(|e| -> Error { e.into() })?;
@@ -2290,6 +2352,12 @@ impl BaseTable for NativeTable {
        Ok(dataset.prewarm_index(index_name).await?)
    }
    async fn prewarm_data(&self, _columns: Option<Vec<String>>) -> Result<()> {
        Err(Error::NotSupported {
            message: "prewarm_data is currently only supported on remote tables.".into(),
        })
    }
    async fn update(&self, update: UpdateBuilder) -> Result<UpdateResult> {
        // Delegate to the submodule implementation
        update::execute_update(self, update).await
--- a/rust/lancedb/src/table/add_data.rs
+++ b/rust/lancedb/src/table/add_data.rs
@@ -13,6 +13,9 @@ use crate::embeddings::EmbeddingRegistry;
 use crate::table::datafusion::cast::cast_to_table_schema;
 use crate::table::datafusion::reject_nan::reject_nan_vectors;
 use crate::table::datafusion::scannable_exec::ScannableExec;
 use crate::table::write_progress::ProgressCallback;
 use crate::table::write_progress::WriteProgress;
 use crate::table::write_progress::WriteProgressTracker;
 use crate::{Error, Result};
 use super::{BaseTable, TableDefinition, WriteOptions};
@@ -52,6 +55,8 @@ pub struct AddDataBuilder {
    pub(crate) write_options: WriteOptions,
    pub(crate) on_nan_vectors: NaNVectorBehavior,
    pub(crate) embedding_registry: Option<Arc<dyn EmbeddingRegistry>>,
    pub(crate) progress_callback: Option<ProgressCallback>,
    pub(crate) write_parallelism: Option<usize>,
 }
 impl std::fmt::Debug for AddDataBuilder {
@@ -77,6 +82,8 @@ impl AddDataBuilder {
            write_options: WriteOptions::default(),
            on_nan_vectors: NaNVectorBehavior::default(),
            embedding_registry,
            progress_callback: None,
            write_parallelism: None,
        }
    }
@@ -101,7 +108,43 @@ impl AddDataBuilder {
        self
    }
    /// Set a callback to receive progress updates during the add operation.
    ///
    /// The callback is invoked once per batch written, and once more with
    /// [`WriteProgress::done`] set to `true` when the write completes.
    ///
    /// ```
    /// # use lancedb::Table;
    /// # async fn example(table: &Table) -> Result<(), Box<dyn std::error::Error>> {
    /// let batch = arrow_array::record_batch!(("id", Int32, [1, 2, 3])).unwrap();
    /// table.add(batch)
    ///     .progress(|p| println!("{}/{:?} rows", p.output_rows(), p.total_rows()))
    ///     .execute()
    ///     .await?;
    /// # Ok(())
    /// # }
    /// ```
    pub fn progress(mut self, callback: impl FnMut(&WriteProgress) + Send + 'static) -> Self {
        self.progress_callback = Some(Arc::new(std::sync::Mutex::new(callback)));
        self
    }
    /// Set the number of parallel write streams.
    ///
    /// By default, the number of streams is estimated from the data size.
    /// Setting this to `1` disables parallel writes.
    pub fn write_parallelism(mut self, parallelism: usize) -> Self {
        self.write_parallelism = Some(parallelism);
        self
    }
    pub async fn execute(self) -> Result<AddResult> {
        if self.write_parallelism.map(|p| p == 0).unwrap_or(false) {
            return Err(Error::InvalidInput {
                message: "write_parallelism must be greater than 0".to_string(),
            });
        }
        self.parent.clone().add(self).await
    }
@@ -130,8 +173,11 @@ impl AddDataBuilder {
            scannable_with_embeddings(self.data, table_def, self.embedding_registry.as_ref())?;
        let rescannable = self.data.rescannable();
        let tracker = self
            .progress_callback
            .map(|cb| Arc::new(WriteProgressTracker::new(cb, self.data.num_rows())));
        let plan: Arc<dyn datafusion_physical_plan::ExecutionPlan> =
-            Arc::new(ScannableExec::new(self.data));
+            Arc::new(ScannableExec::new(self.data, tracker.clone()));
        // Skip casting when overwriting — the input schema replaces the table schema.
        let plan = if overwrite {
            plan
@@ -149,6 +195,7 @@ impl AddDataBuilder {
            rescannable,
            write_options: self.write_options,
            mode: self.mode,
            tracker,
        })
    }
 }
@@ -161,6 +208,7 @@ pub struct PreprocessingOutput {
    pub rescannable: bool,
    pub write_options: WriteOptions,
    pub mode: AddDataMode,
    pub tracker: Option<Arc<WriteProgressTracker>>,
 }
 /// Check that the input schema is valid for insert.
--- a/rust/lancedb/src/table/datafusion/insert.rs
+++ b/rust/lancedb/src/table/datafusion/insert.rs
@@ -12,13 +12,16 @@ use datafusion_common::{DataFusionError, Result as DataFusionResult};
 use datafusion_execution::{SendableRecordBatchStream, TaskContext};
 use datafusion_physical_expr::{EquivalenceProperties, Partitioning};
 use datafusion_physical_plan::execution_plan::{Boundedness, EmissionType};
 use datafusion_physical_plan::metrics::{ExecutionPlanMetricsSet, MetricBuilder, MetricsSet};
 use datafusion_physical_plan::stream::RecordBatchStreamAdapter;
 use datafusion_physical_plan::{
    DisplayAs, DisplayFormatType, ExecutionPlan, ExecutionPlanProperties, PlanProperties,
 };
 use futures::TryStreamExt;
 use lance::Dataset;
 use lance::dataset::transaction::{Operation, Transaction};
 use lance::dataset::{CommitBuilder, InsertBuilder, WriteParams};
 use lance::io::exec::utils::InstrumentedRecordBatchStreamAdapter;
 use lance_table::format::Fragment;
 use crate::table::dataset::DatasetConsistencyWrapper;
@@ -80,6 +83,7 @@ pub struct InsertExec {
    write_params: WriteParams,
    properties: PlanProperties,
    partial_transactions: Arc<Mutex<Vec<Transaction>>>,
    metrics: ExecutionPlanMetricsSet,
 }
 impl InsertExec {
@@ -105,6 +109,7 @@ impl InsertExec {
            write_params,
            properties,
            partial_transactions: Arc::new(Mutex::new(Vec::with_capacity(num_partitions))),
            metrics: ExecutionPlanMetricsSet::new(),
        }
    }
 }
@@ -176,6 +181,19 @@ impl ExecutionPlan for InsertExec {
        let total_partitions = self.input.output_partitioning().partition_count();
        let ds_wrapper = self.ds_wrapper.clone();
        let output_bytes = MetricBuilder::new(&self.metrics).output_bytes(partition);
        let input_schema = input_stream.schema();
        let input_stream: SendableRecordBatchStream =
            Box::pin(InstrumentedRecordBatchStreamAdapter::new(
                input_schema,
                input_stream.map_ok(move |batch| {
                    output_bytes.add(batch.get_array_memory_size());
                    batch
                }),
                partition,
                &self.metrics,
            ));
        let stream = futures::stream::once(async move {
            let transaction = InsertBuilder::new(dataset.clone())
                .with_params(&write_params)
@@ -186,7 +204,9 @@ impl ExecutionPlan for InsertExec {
            let to_commit = {
                // Don't hold the lock over an await point.
-                let mut txns = partial_transactions.lock().unwrap();
+                let mut txns = partial_transactions
                    .lock()
                    .unwrap_or_else(|e| e.into_inner());
                txns.push(transaction);
                if txns.len() == total_partitions {
                    Some(std::mem::take(&mut *txns))
@@ -215,6 +235,10 @@ impl ExecutionPlan for InsertExec {
            stream,
        )))
    }
    fn metrics(&self) -> Option<MetricsSet> {
        Some(self.metrics.clone_inner())
    }
 }
 #[cfg(test)]
--- a/rust/lancedb/src/table/datafusion/scannable_exec.rs
+++ b/rust/lancedb/src/table/datafusion/scannable_exec.rs
@@ -7,17 +7,21 @@ use std::sync::{Arc, Mutex};
 use datafusion_common::{DataFusionError, Result as DFResult, Statistics, stats::Precision};
 use datafusion_execution::{SendableRecordBatchStream, TaskContext};
 use datafusion_physical_expr::{EquivalenceProperties, Partitioning};
 use datafusion_physical_plan::stream::RecordBatchStreamAdapter;
 use datafusion_physical_plan::{
    DisplayAs, DisplayFormatType, ExecutionPlan, PlanProperties, execution_plan::EmissionType,
 };
 use futures::TryStreamExt;
 use crate::table::write_progress::WriteProgressTracker;
 use crate::{arrow::SendableRecordBatchStreamExt, data::scannable::Scannable};
-pub struct ScannableExec {
+pub(crate) struct ScannableExec {
-    // We don't require Scannable to by Sync, so we wrap it in a Mutex to allow safe concurrent access.
+    // We don't require Scannable to be Sync, so we wrap it in a Mutex to allow safe concurrent access.
    source: Mutex<Box<dyn Scannable>>,
    num_rows: Option<usize>,
    properties: PlanProperties,
    tracker: Option<Arc<WriteProgressTracker>>,
 }
 impl std::fmt::Debug for ScannableExec {
@@ -30,7 +34,7 @@ impl std::fmt::Debug for ScannableExec {
 }
 impl ScannableExec {
-    pub fn new(source: Box<dyn Scannable>) -> Self {
+    pub fn new(source: Box<dyn Scannable>, tracker: Option<Arc<WriteProgressTracker>>) -> Self {
        let schema = source.schema();
        let eq_properties = EquivalenceProperties::new(schema);
        let properties = PlanProperties::new(
@@ -46,6 +50,7 @@ impl ScannableExec {
            source,
            num_rows,
            properties,
            tracker,
        }
    }
 }
@@ -102,7 +107,18 @@ impl ExecutionPlan for ScannableExec {
            Err(poison) => poison.into_inner().scan_as_stream(),
        };
-        Ok(stream.into_df_stream())
+        let tracker = self.tracker.clone();
        let stream = stream.into_df_stream().map_ok(move |batch| {
            if let Some(ref t) = tracker {
                t.record_batch(batch.num_rows(), batch.get_array_memory_size());
            }
            batch
        });
        Ok(Box::pin(RecordBatchStreamAdapter::new(
            self.schema(),
            stream,
        )))
    }
    fn partition_statistics(&self, _partition: Option<usize>) -> DFResult<Statistics> {
--- a/rust/lancedb/src/table/dataset.rs
+++ b/rust/lancedb/src/table/dataset.rs
@@ -82,7 +82,7 @@ impl DatasetConsistencyWrapper {
    /// pinned dataset regardless of consistency mode.
    pub async fn get(&self) -> Result<Arc<Dataset>> {
        {
-            let state = self.state.lock().unwrap();
+            let state = self.state.lock()?;
            if state.pinned_version.is_some() {
                return Ok(state.dataset.clone());
            }
@@ -101,7 +101,7 @@ impl DatasetConsistencyWrapper {
            }
            ConsistencyMode::Strong => refresh_latest(self.state.clone()).await,
            ConsistencyMode::Lazy => {
-                let state = self.state.lock().unwrap();
+                let state = self.state.lock()?;
                Ok(state.dataset.clone())
            }
        }
@@ -116,7 +116,7 @@ impl DatasetConsistencyWrapper {
    /// concurrent [`as_time_travel`](Self::as_time_travel) call), the update
    /// is silently ignored — the write already committed to storage.
    pub fn update(&self, dataset: Dataset) {
-        let mut state = self.state.lock().unwrap();
+        let mut state = self.state.lock().unwrap_or_else(|e| e.into_inner());
        if state.pinned_version.is_some() {
            // A concurrent as_time_travel() beat us here. The write succeeded
            // in storage, but since we're now pinned we don't advance the
@@ -139,7 +139,7 @@ impl DatasetConsistencyWrapper {
    /// Check that the dataset is in a mutable mode (Latest).
    pub fn ensure_mutable(&self) -> Result<()> {
-        let state = self.state.lock().unwrap();
+        let state = self.state.lock()?;
        if state.pinned_version.is_some() {
            Err(crate::Error::InvalidInput {
                message: "table cannot be modified when a specific version is checked out"
@@ -152,13 +152,16 @@ impl DatasetConsistencyWrapper {
    /// Returns the version, if in time travel mode, or None otherwise.
    pub fn time_travel_version(&self) -> Option<u64> {
-        self.state.lock().unwrap().pinned_version
+        self.state
            .lock()
            .unwrap_or_else(|e| e.into_inner())
            .pinned_version
    }
    /// Convert into a wrapper in latest version mode.
    pub async fn as_latest(&self) -> Result<()> {
        let dataset = {
-            let state = self.state.lock().unwrap();
+            let state = self.state.lock()?;
            if state.pinned_version.is_none() {
                return Ok(());
            }
@@ -168,7 +171,7 @@ impl DatasetConsistencyWrapper {
        let latest_version = dataset.latest_version_id().await?;
        let new_dataset = dataset.checkout_version(latest_version).await?;
-        let mut state = self.state.lock().unwrap();
+        let mut state = self.state.lock()?;
        if state.pinned_version.is_some() {
            state.dataset = Arc::new(new_dataset);
            state.pinned_version = None;
@@ -184,7 +187,7 @@ impl DatasetConsistencyWrapper {
        let target_ref = target_version.into();
        let (should_checkout, dataset) = {
-            let state = self.state.lock().unwrap();
+            let state = self.state.lock()?;
            let should = match state.pinned_version {
                None => true,
                Some(version) => match &target_ref {
@@ -204,7 +207,7 @@ impl DatasetConsistencyWrapper {
        let new_dataset = dataset.checkout_version(target_ref).await?;
        let version_value = new_dataset.version().version;
-        let mut state = self.state.lock().unwrap();
+        let mut state = self.state.lock()?;
        state.dataset = Arc::new(new_dataset);
        state.pinned_version = Some(version_value);
        Ok(())
@@ -212,7 +215,7 @@ impl DatasetConsistencyWrapper {
    pub async fn reload(&self) -> Result<()> {
        let (dataset, pinned_version) = {
-            let state = self.state.lock().unwrap();
+            let state = self.state.lock()?;
            (state.dataset.clone(), state.pinned_version)
        };
@@ -230,7 +233,7 @@ impl DatasetConsistencyWrapper {
                let new_dataset = dataset.checkout_version(version).await?;
-                let mut state = self.state.lock().unwrap();
+                let mut state = self.state.lock()?;
                if state.pinned_version == Some(version) {
                    state.dataset = Arc::new(new_dataset);
                }
@@ -242,14 +245,14 @@ impl DatasetConsistencyWrapper {
 }
 async fn refresh_latest(state: Arc<Mutex<DatasetState>>) -> Result<Arc<Dataset>> {
-    let dataset = { state.lock().unwrap().dataset.clone() };
+    let dataset = { state.lock()?.dataset.clone() };
    let mut ds = (*dataset).clone();
    ds.checkout_latest().await?;
    let new_arc = Arc::new(ds);
    {
-        let mut state = state.lock().unwrap();
+        let mut state = state.lock()?;
        if state.pinned_version.is_none()
            && new_arc.manifest().version >= state.dataset.manifest().version
        {
@@ -612,4 +615,108 @@ mod tests {
        let s = io_stats.incremental_stats();
        assert_eq!(s.read_iops, 0, "step 5, elapsed={:?}", start.elapsed());
    }
    /// Helper: poison the mutex inside a DatasetConsistencyWrapper.
    fn poison_state(wrapper: &DatasetConsistencyWrapper) {
        let state = wrapper.state.clone();
        let handle = std::thread::spawn(move || {
            let _guard = state.lock().unwrap();
            panic!("intentional panic to poison mutex");
        });
        let _ = handle.join(); // join collects the panic
        assert!(wrapper.state.lock().is_err(), "mutex should be poisoned");
    }
    #[tokio::test]
    async fn test_get_returns_error_on_poisoned_lock() {
        let dir = tempfile::tempdir().unwrap();
        let uri = dir.path().to_str().unwrap();
        let ds = create_test_dataset(uri).await;
        let wrapper = DatasetConsistencyWrapper::new_latest(ds, None);
        poison_state(&wrapper);
        // get() should return Err, not panic
        let result = wrapper.get().await;
        assert!(result.is_err());
    }
    #[tokio::test]
    async fn test_ensure_mutable_returns_error_on_poisoned_lock() {
        let dir = tempfile::tempdir().unwrap();
        let uri = dir.path().to_str().unwrap();
        let ds = create_test_dataset(uri).await;
        let wrapper = DatasetConsistencyWrapper::new_latest(ds, None);
        poison_state(&wrapper);
        let result = wrapper.ensure_mutable();
        assert!(result.is_err());
    }
    #[tokio::test]
    async fn test_update_recovers_from_poisoned_lock() {
        let dir = tempfile::tempdir().unwrap();
        let uri = dir.path().to_str().unwrap();
        let ds = create_test_dataset(uri).await;
        let ds_v2 = append_to_dataset(uri).await;
        let wrapper = DatasetConsistencyWrapper::new_latest(ds, None);
        poison_state(&wrapper);
        // update() returns (), should not panic
        wrapper.update(ds_v2);
    }
    #[tokio::test]
    async fn test_time_travel_version_recovers_from_poisoned_lock() {
        let dir = tempfile::tempdir().unwrap();
        let uri = dir.path().to_str().unwrap();
        let ds = create_test_dataset(uri).await;
        let wrapper = DatasetConsistencyWrapper::new_latest(ds, None);
        poison_state(&wrapper);
        // Should not panic, returns whatever was in the mutex
        let _version = wrapper.time_travel_version();
    }
    #[tokio::test]
    async fn test_as_latest_returns_error_on_poisoned_lock() {
        let dir = tempfile::tempdir().unwrap();
        let uri = dir.path().to_str().unwrap();
        let ds = create_test_dataset(uri).await;
        let wrapper = DatasetConsistencyWrapper::new_latest(ds, None);
        poison_state(&wrapper);
        let result = wrapper.as_latest().await;
        assert!(result.is_err());
    }
    #[tokio::test]
    async fn test_as_time_travel_returns_error_on_poisoned_lock() {
        let dir = tempfile::tempdir().unwrap();
        let uri = dir.path().to_str().unwrap();
        let ds = create_test_dataset(uri).await;
        let wrapper = DatasetConsistencyWrapper::new_latest(ds, None);
        poison_state(&wrapper);
        let result = wrapper.as_time_travel(1u64).await;
        assert!(result.is_err());
    }
    #[tokio::test]
    async fn test_reload_returns_error_on_poisoned_lock() {
        let dir = tempfile::tempdir().unwrap();
        let uri = dir.path().to_str().unwrap();
        let ds = create_test_dataset(uri).await;
        let wrapper = DatasetConsistencyWrapper::new_latest(ds, None);
        poison_state(&wrapper);
        let result = wrapper.reload().await;
        assert!(result.is_err());
    }
 }
--- a/rust/lancedb/src/table/optimize.rs
+++ b/rust/lancedb/src/table/optimize.rs
@@ -64,6 +64,9 @@ pub enum OptimizeAction {
        older_than: Option<Duration>,
        /// Because they may be part of an in-progress transaction, files newer than 7 days old are not deleted by default.
        /// If you are sure that there are no in-progress transactions, then you can set this to True to delete all files older than `older_than`.
        ///
        /// **WARNING**: This should only be set to true if you can guarantee that no other process is
        /// currently working on this dataset.  Otherwise the dataset could be put into a corrupted state.
        delete_unverified: Option<bool>,
        /// If true, an error will be returned if there are any old versions that are still tagged.
        error_if_tagged_old_versions: Option<bool>,
@@ -117,6 +120,10 @@ pub(crate) async fn optimize_indices(table: &NativeTable, options: &OptimizeOpti
 ///   If you are sure that there are no in-progress transactions, then you
 ///   can set this to True to delete all files older than `older_than`.
 ///
 ///   **WARNING**: This should only be set to true if you can guarantee that
 ///   no other process is currently working on this dataset.  Otherwise the
 ///   dataset could be put into a corrupted state.
 ///
 /// This calls into [lance::dataset::Dataset::cleanup_old_versions] and
 /// returns the result.
 pub(crate) async fn cleanup_old_versions(
--- a/rust/lancedb/src/table/query.rs
+++ b/rust/lancedb/src/table/query.rs
@@ -9,7 +9,7 @@ use crate::expr::expr_to_sql_string;
 use crate::query::{
    DEFAULT_TOP_K, QueryExecutionOptions, QueryFilter, QueryRequest, Select, VectorQueryRequest,
 };
-use crate::utils::{TimeoutStream, default_vector_column};
+use crate::utils::{MaxBatchLengthStream, TimeoutStream, default_vector_column};
 use arrow::array::{AsArray, FixedSizeListBuilder, Float32Builder};
 use arrow::datatypes::{Float32Type, UInt8Type};
 use arrow_array::Array;
@@ -66,6 +66,7 @@ async fn execute_generic_query(
 ) -> Result<DatasetRecordBatchStream> {
    let plan = create_plan(table, query, options.clone()).await?;
    let inner = execute_plan(plan, Default::default())?;
    let inner = MaxBatchLengthStream::new_boxed(inner, options.max_batch_length as usize);
    let inner = if let Some(timeout) = options.timeout {
        TimeoutStream::new_boxed(inner, timeout)
    } else {
@@ -186,6 +187,13 @@ pub async fn create_plan(
        Select::Dynamic(ref select_with_transform) => {
            scanner.project_with_transform(select_with_transform.as_slice())?;
        }
        Select::Expr(ref expr_pairs) => {
            let sql_pairs: crate::Result<Vec<(String, String)>> = expr_pairs
                .iter()
                .map(|(name, expr)| expr_to_sql_string(expr).map(|sql| (name.clone(), sql)))
                .collect();
            scanner.project_with_transform(sql_pairs?.as_slice())?;
        }
        Select::All => {}
    }
@@ -193,7 +201,9 @@ pub async fn create_plan(
        scanner.with_row_id();
    }
-    scanner.batch_size(options.max_batch_length as usize);
+    if options.max_batch_length > 0 {
        scanner.batch_size(options.max_batch_length as usize);
    }
    if query.base.fast_search {
        scanner.fast_search();
@@ -340,6 +350,17 @@ fn convert_to_namespace_query(query: &AnyQuery) -> Result<NsQueryTableRequest> {
                                .to_string(),
                    });
                }
                Select::Expr(pairs) => {
                    let sql_pairs: crate::Result<Vec<(String, String)>> = pairs
                        .iter()
                        .map(|(name, expr)| expr_to_sql_string(expr).map(|sql| (name.clone(), sql)))
                        .collect();
                    let sql_pairs = sql_pairs?;
                    Some(Box::new(QueryTableRequestColumns {
                        column_names: None,
                        column_aliases: Some(sql_pairs.into_iter().collect()),
                    }))
                }
            };
            // Check for unsupported features
@@ -411,6 +432,17 @@ fn convert_to_namespace_query(query: &AnyQuery) -> Result<NsQueryTableRequest> {
                            .to_string(),
                    });
                }
                Select::Expr(pairs) => {
                    let sql_pairs: crate::Result<Vec<(String, String)>> = pairs
                        .iter()
                        .map(|(name, expr)| expr_to_sql_string(expr).map(|sql| (name.clone(), sql)))
                        .collect();
                    let sql_pairs = sql_pairs?;
                    Some(Box::new(QueryTableRequestColumns {
                        column_names: None,
                        column_aliases: Some(sql_pairs.into_iter().collect()),
                    }))
                }
            };
            // Handle full text search if present
--- a/rust/lancedb/src/table/write_progress.rs
+++ b/rust/lancedb/src/table/write_progress.rs
@@ -0,0 +1,431 @@
 // SPDX-License-Identifier: Apache-2.0
 // SPDX-FileCopyrightText: Copyright The LanceDB Authors
 //! Progress monitoring for write operations.
 //!
 //! You can add a callback to process progress in [`crate::table::AddDataBuilder::progress`].
 //! [`WriteProgress`] is the struct passed to the callback.
 use std::sync::atomic::{AtomicUsize, Ordering};
 use std::sync::{Arc, Mutex};
 use std::time::{Duration, Instant};
 /// Progress snapshot for a write operation.
 #[derive(Debug, Clone)]
 pub struct WriteProgress {
    // These are private and only accessible via getters, to make it easy to add
    // new fields without breaking existing callbacks.
    elapsed: Duration,
    output_rows: usize,
    output_bytes: usize,
    total_rows: Option<usize>,
    active_tasks: usize,
    total_tasks: usize,
    done: bool,
 }
 impl WriteProgress {
    /// Wall-clock time since monitoring started.
    pub fn elapsed(&self) -> Duration {
        self.elapsed
    }
    /// Number of rows written so far.
    pub fn output_rows(&self) -> usize {
        self.output_rows
    }
    /// Number of bytes written so far.
    pub fn output_bytes(&self) -> usize {
        self.output_bytes
    }
    /// Total rows expected.
    ///
    /// Populated when the input source reports a row count (e.g. a
    /// [`arrow_array::RecordBatch`]).  Always `Some` when [`WriteProgress::done`]
    /// is `true` — falling back to the actual number of rows written.
    pub fn total_rows(&self) -> Option<usize> {
        self.total_rows
    }
    /// Number of parallel write tasks currently in flight.
    pub fn active_tasks(&self) -> usize {
        self.active_tasks
    }
    /// Total number of parallel write tasks (i.e. the write parallelism).
    pub fn total_tasks(&self) -> usize {
        self.total_tasks
    }
    /// Whether the write operation has completed.
    ///
    /// The final callback always has `done = true`.  Callers can use this to
    /// finalize progress bars or perform cleanup.
    pub fn done(&self) -> bool {
        self.done
    }
 }
 /// Callback type for progress updates.
 ///
 /// Callbacks are serialized by the tracker and are never invoked reentrantly,
 /// so `FnMut` is safe to use here.
 pub type ProgressCallback = Arc<Mutex<dyn FnMut(&WriteProgress) + Send>>;
 /// Tracks progress of a write operation and invokes a [`ProgressCallback`].
 ///
 /// Call [`WriteProgressTracker::record_batch`] for each batch written.
 /// Call [`WriteProgressTracker::finish`] once after all data is written.
 ///
 /// The callback is never invoked reentrantly: all state updates and callback
 /// invocations are serialized behind a single lock.
 impl std::fmt::Debug for WriteProgressTracker {
    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
        f.debug_struct("WriteProgressTracker")
            .field("total_rows", &self.total_rows)
            .finish()
    }
 }
 pub(crate) struct WriteProgressTracker {
    rows_and_bytes: std::sync::Mutex<(usize, usize)>,
    /// Wire bytes tracked separately by the insert layer. When set (> 0),
    /// this takes precedence over the in-memory bytes from `rows_and_bytes`.
    wire_bytes: AtomicUsize,
    active_tasks: Arc<AtomicUsize>,
    total_tasks: AtomicUsize,
    start: Instant,
    /// Known total rows from the input source, if available.
    total_rows: Option<usize>,
    callback: ProgressCallback,
 }
 impl WriteProgressTracker {
    pub fn new(callback: ProgressCallback, total_rows: Option<usize>) -> Self {
        Self {
            rows_and_bytes: std::sync::Mutex::new((0, 0)),
            wire_bytes: AtomicUsize::new(0),
            active_tasks: Arc::new(AtomicUsize::new(0)),
            total_tasks: AtomicUsize::new(1),
            start: Instant::now(),
            total_rows,
            callback,
        }
    }
    /// Set the total number of parallel write tasks (the write parallelism).
    pub fn set_total_tasks(&self, n: usize) {
        self.total_tasks.store(n, Ordering::Relaxed);
    }
    /// Increment the active task count. Returns a guard that decrements on drop.
    pub fn track_task(&self) -> ActiveTaskGuard {
        self.active_tasks.fetch_add(1, Ordering::Relaxed);
        ActiveTaskGuard(self.active_tasks.clone())
    }
    /// Record a batch of rows passing through the scan node.
    pub fn record_batch(&self, rows: usize, bytes: usize) {
        // Lock order: callback first, then rows_and_bytes. This is the only
        // order used anywhere, so deadlocks cannot occur.
        let mut cb = self.callback.lock().unwrap_or_else(|e| e.into_inner());
        let mut guard = self
            .rows_and_bytes
            .lock()
            .unwrap_or_else(|e| e.into_inner());
        guard.0 += rows;
        guard.1 += bytes;
        let progress = self.snapshot(guard.0, guard.1, false);
        drop(guard);
        cb(&progress);
    }
    /// Record wire bytes from the insert layer (e.g. IPC-encoded bytes for
    /// remote writes). When wire bytes are recorded, they take precedence over
    /// the in-memory Arrow bytes tracked by [`record_batch`].
    pub fn record_bytes(&self, bytes: usize) {
        self.wire_bytes.fetch_add(bytes, Ordering::Relaxed);
    }
    /// Emit the final progress callback indicating the write is complete.
    ///
    /// `total_rows` is always `Some` on the final callback: it uses the known
    /// total if available, or falls back to the number of rows actually written.
    pub fn finish(&self) {
        let mut cb = self.callback.lock().unwrap_or_else(|e| e.into_inner());
        let guard = self
            .rows_and_bytes
            .lock()
            .unwrap_or_else(|e| e.into_inner());
        let mut snap = self.snapshot(guard.0, guard.1, true);
        snap.total_rows = Some(self.total_rows.unwrap_or(guard.0));
        drop(guard);
        cb(&snap);
    }
    fn snapshot(&self, rows: usize, in_memory_bytes: usize, done: bool) -> WriteProgress {
        let wire = self.wire_bytes.load(Ordering::Relaxed);
        // Prefer wire bytes (actual I/O size) when the insert layer is
        // tracking them; fall back to in-memory Arrow size otherwise.
        // TODO: for local writes, track actual bytes written by Lance
        // instead of using in-memory Arrow size as a proxy.
        let output_bytes = if wire > 0 { wire } else { in_memory_bytes };
        WriteProgress {
            elapsed: self.start.elapsed(),
            output_rows: rows,
            output_bytes,
            total_rows: self.total_rows,
            active_tasks: self.active_tasks.load(Ordering::Relaxed),
            total_tasks: self.total_tasks.load(Ordering::Relaxed),
            done,
        }
    }
 }
 /// RAII guard that decrements the active task count when dropped.
 pub(crate) struct ActiveTaskGuard(Arc<AtomicUsize>);
 impl Drop for ActiveTaskGuard {
    fn drop(&mut self) {
        self.0.fetch_sub(1, Ordering::Relaxed);
    }
 }
 /// RAII guard that calls [`WriteProgressTracker::finish`] on drop.
 ///
 /// This ensures the final `done=true` callback fires even if the write
 /// errors or the future is cancelled.
 pub(crate) struct FinishOnDrop(pub Option<Arc<WriteProgressTracker>>);
 impl Drop for FinishOnDrop {
    fn drop(&mut self) {
        if let Some(t) = self.0.take() {
            t.finish();
        }
    }
 }
 #[cfg(test)]
 mod tests {
    use std::sync::Arc;
    use std::sync::atomic::{AtomicUsize, Ordering};
    use arrow_array::record_batch;
    use crate::connect;
    #[tokio::test]
    async fn test_progress_monitor_fires_callback() {
        let db = connect("memory://").execute().await.unwrap();
        let batch = record_batch!(("id", Int32, [1, 2, 3])).unwrap();
        let table = db
            .create_table("progress_test", batch)
            .execute()
            .await
            .unwrap();
        let callback_count = Arc::new(AtomicUsize::new(0));
        let last_rows = Arc::new(AtomicUsize::new(0));
        let max_active = Arc::new(AtomicUsize::new(0));
        let last_total_tasks = Arc::new(AtomicUsize::new(0));
        let cb_count = callback_count.clone();
        let cb_rows = last_rows.clone();
        let cb_active = max_active.clone();
        let cb_total_tasks = last_total_tasks.clone();
        let new_data = record_batch!(("id", Int32, [4, 5, 6])).unwrap();
        table
            .add(new_data)
            .progress(move |p| {
                cb_count.fetch_add(1, Ordering::SeqCst);
                cb_rows.store(p.output_rows(), Ordering::SeqCst);
                cb_active.fetch_max(p.active_tasks(), Ordering::SeqCst);
                cb_total_tasks.store(p.total_tasks(), Ordering::SeqCst);
            })
            .execute()
            .await
            .unwrap();
        assert_eq!(table.count_rows(None).await.unwrap(), 6);
        assert!(callback_count.load(Ordering::SeqCst) >= 1);
        // Progress tracks the newly inserted rows, not the total table size.
        assert_eq!(last_rows.load(Ordering::SeqCst), 3);
        // At least one callback should have seen an active task.
        assert!(max_active.load(Ordering::SeqCst) >= 1);
        // total_tasks should reflect the write parallelism.
        assert!(last_total_tasks.load(Ordering::SeqCst) >= 1);
    }
    #[tokio::test]
    async fn test_progress_done_fires_at_end() {
        let db = connect("memory://").execute().await.unwrap();
        let batch = record_batch!(("id", Int32, [1, 2, 3])).unwrap();
        let table = db
            .create_table("progress_done", batch)
            .execute()
            .await
            .unwrap();
        let seen_done = Arc::new(std::sync::Mutex::new(Vec::<bool>::new()));
        let seen = seen_done.clone();
        let new_data = record_batch!(("id", Int32, [4, 5, 6])).unwrap();
        table
            .add(new_data)
            .progress(move |p| {
                seen.lock().unwrap().push(p.done());
            })
            .execute()
            .await
            .unwrap();
        let done_flags = seen_done.lock().unwrap();
        assert!(!done_flags.is_empty(), "at least one callback must fire");
        // Only the last callback should have done=true.
        let last = *done_flags.last().unwrap();
        assert!(last, "last callback must have done=true");
        // All earlier callbacks should have done=false.
        for &d in done_flags.iter().rev().skip(1) {
            assert!(!d, "non-final callbacks must have done=false");
        }
    }
    #[tokio::test]
    async fn test_progress_total_rows_known() {
        let db = connect("memory://").execute().await.unwrap();
        let batch = record_batch!(("id", Int32, [1, 2, 3])).unwrap();
        let table = db
            .create_table("total_known", batch)
            .execute()
            .await
            .unwrap();
        let seen_total = Arc::new(std::sync::Mutex::new(Vec::new()));
        let seen = seen_total.clone();
        // RecordBatch implements Scannable with num_rows() -> Some(3)
        let new_data = record_batch!(("id", Int32, [4, 5, 6])).unwrap();
        table
            .add(new_data)
            .progress(move |p| {
                seen.lock().unwrap().push(p.total_rows());
            })
            .execute()
            .await
            .unwrap();
        let totals = seen_total.lock().unwrap();
        // All callbacks (including done) should have total_rows = Some(3)
        assert!(
            totals.contains(&Some(3)),
            "expected total_rows=Some(3) in at least one callback, got: {:?}",
            *totals
        );
    }
    #[tokio::test]
    async fn test_progress_total_rows_unknown() {
        use arrow_array::RecordBatchIterator;
        let db = connect("memory://").execute().await.unwrap();
        let batch = record_batch!(("id", Int32, [1, 2, 3])).unwrap();
        let table = db
            .create_table("total_unknown", batch)
            .execute()
            .await
            .unwrap();
        let seen_total = Arc::new(std::sync::Mutex::new(Vec::new()));
        let seen = seen_total.clone();
        // RecordBatchReader does not provide num_rows, so total_rows should be
        // None in intermediate callbacks but always Some on the done callback.
        let schema = arrow_schema::Schema::new(vec![arrow_schema::Field::new(
            "id",
            arrow_schema::DataType::Int32,
            false,
        )]);
        let new_data: Box<dyn arrow_array::RecordBatchReader + Send> =
            Box::new(RecordBatchIterator::new(
                vec![Ok(record_batch!(("id", Int32, [4, 5, 6])).unwrap())],
                Arc::new(schema),
            ));
        table
            .add(new_data)
            .progress(move |p| {
                seen.lock().unwrap().push((p.total_rows(), p.done()));
            })
            .execute()
            .await
            .unwrap();
        let entries = seen_total.lock().unwrap();
        assert!(!entries.is_empty(), "at least one callback must fire");
        for (total, done) in entries.iter() {
            if *done {
                assert!(
                    total.is_some(),
                    "done callback must have total_rows set, got: {:?}",
                    total
                );
            } else {
                assert_eq!(
                    *total, None,
                    "intermediate callback must have total_rows=None, got: {:?}",
                    total
                );
            }
        }
    }
    #[test]
    fn test_record_batch_recovers_from_poisoned_callback_lock() {
        use super::{ProgressCallback, WriteProgressTracker};
        use std::sync::Mutex;
        let callback: ProgressCallback = Arc::new(Mutex::new(|_: &super::WriteProgress| {}));
        // Poison the callback mutex
        let cb_clone = callback.clone();
        let handle = std::thread::spawn(move || {
            let _guard = cb_clone.lock().unwrap();
            panic!("intentional panic to poison callback mutex");
        });
        let _ = handle.join();
        assert!(
            callback.lock().is_err(),
            "callback mutex should be poisoned"
        );
        let tracker = WriteProgressTracker::new(callback, Some(100));
        // record_batch should not panic
        tracker.record_batch(10, 1024);
    }
    #[test]
    fn test_finish_recovers_from_poisoned_callback_lock() {
        use super::{ProgressCallback, WriteProgressTracker};
        use std::sync::Mutex;
        let callback: ProgressCallback = Arc::new(Mutex::new(|_: &super::WriteProgress| {}));
        // Poison the callback mutex
        let cb_clone = callback.clone();
        let handle = std::thread::spawn(move || {
            let _guard = cb_clone.lock().unwrap();
            panic!("intentional panic to poison callback mutex");
        });
        let _ = handle.join();
        let tracker = WriteProgressTracker::new(callback, Some(100));
        // finish should not panic
        tracker.finish();
    }
 }
--- a/rust/lancedb/src/utils/background_cache.rs
+++ b/rust/lancedb/src/utils/background_cache.rs
@@ -122,7 +122,7 @@ where
    /// This is a cheap synchronous check useful as a fast path before
    /// constructing a fetch closure for [`get()`](Self::get).
    pub fn try_get(&self) -> Option<V> {
-        let cache = self.inner.lock().unwrap();
+        let cache = self.inner.lock().unwrap_or_else(|e| e.into_inner());
        cache.state.fresh_value(self.ttl, self.refresh_window)
    }
@@ -138,7 +138,7 @@ where
    {
        // Fast path: check if cache is fresh
        {
-            let cache = self.inner.lock().unwrap();
+            let cache = self.inner.lock().unwrap_or_else(|e| e.into_inner());
            if let Some(value) = cache.state.fresh_value(self.ttl, self.refresh_window) {
                return Ok(value);
            }
@@ -147,7 +147,7 @@ where
        // Slow path
        let mut fetch = Some(fetch);
        let action = {
-            let mut cache = self.inner.lock().unwrap();
+            let mut cache = self.inner.lock().unwrap_or_else(|e| e.into_inner());
            self.determine_action(&mut cache, &mut fetch)
        };
@@ -161,7 +161,7 @@ where
    ///
    /// This avoids a blocking fetch on the first [`get()`](Self::get) call.
    pub fn seed(&self, value: V) {
-        let mut cache = self.inner.lock().unwrap();
+        let mut cache = self.inner.lock().unwrap_or_else(|e| e.into_inner());
        cache.state = State::Current(value, clock::now());
    }
@@ -170,7 +170,7 @@ where
    /// Any in-flight background fetch from before this call will not update the
    /// cache (the generation counter prevents stale writes).
    pub fn invalidate(&self) {
-        let mut cache = self.inner.lock().unwrap();
+        let mut cache = self.inner.lock().unwrap_or_else(|e| e.into_inner());
        cache.state = State::Empty;
        cache.generation += 1;
    }
@@ -267,7 +267,7 @@ where
        let fut_for_spawn = shared.clone();
        tokio::spawn(async move {
            let result = fut_for_spawn.await;
-            let mut cache = inner.lock().unwrap();
+            let mut cache = inner.lock().unwrap_or_else(|e| e.into_inner());
            // Only update if no invalidation has happened since we started
            if cache.generation != generation {
                return;
@@ -590,4 +590,67 @@ mod tests {
        let v = cache.get(ok_fetcher(count.clone(), "fresh")).await.unwrap();
        assert_eq!(v, "fresh");
    }
    /// Helper: poison the inner mutex of a BackgroundCache.
    fn poison_cache(cache: &BackgroundCache<String, TestError>) {
        let inner = cache.inner.clone();
        let handle = std::thread::spawn(move || {
            let _guard = inner.lock().unwrap();
            panic!("intentional panic to poison mutex");
        });
        let _ = handle.join();
        assert!(cache.inner.lock().is_err(), "mutex should be poisoned");
    }
    #[tokio::test]
    async fn test_try_get_recovers_from_poisoned_lock() {
        let cache = new_cache();
        let count = Arc::new(AtomicUsize::new(0));
        // Seed a value first
        cache.get(ok_fetcher(count.clone(), "hello")).await.unwrap();
        cache.get(ok_fetcher(count.clone(), "hello")).await.unwrap(); // peek
        poison_cache(&cache);
        // try_get() should not panic — it recovers via unwrap_or_else
        let result = cache.try_get();
        // The value may or may not be fresh depending on timing, but it must not panic
        let _ = result;
    }
    #[tokio::test]
    async fn test_get_recovers_from_poisoned_lock() {
        let cache = new_cache();
        let count = Arc::new(AtomicUsize::new(0));
        poison_cache(&cache);
        // get() should not panic — it recovers and can still fetch
        let result = cache.get(ok_fetcher(count.clone(), "recovered")).await;
        assert!(result.is_ok());
        assert_eq!(result.unwrap(), "recovered");
    }
    #[tokio::test]
    async fn test_seed_recovers_from_poisoned_lock() {
        let cache = new_cache();
        poison_cache(&cache);
        // seed() should not panic
        cache.seed("seeded".to_string());
    }
    #[tokio::test]
    async fn test_invalidate_recovers_from_poisoned_lock() {
        let cache = new_cache();
        let count = Arc::new(AtomicUsize::new(0));
        cache.get(ok_fetcher(count.clone(), "hello")).await.unwrap();
        poison_cache(&cache);
        // invalidate() should not panic
        cache.invalidate();
    }
 }
--- a/rust/lancedb/src/utils/mod.rs
+++ b/rust/lancedb/src/utils/mod.rs
@@ -335,6 +335,85 @@ impl Stream for TimeoutStream {
    }
 }
 /// A `Stream` wrapper that slices oversized batches to enforce a maximum batch length.
 pub struct MaxBatchLengthStream {
    inner: SendableRecordBatchStream,
    max_batch_length: Option<usize>,
    buffered_batch: Option<RecordBatch>,
    buffered_offset: usize,
 }
 impl MaxBatchLengthStream {
    pub fn new(inner: SendableRecordBatchStream, max_batch_length: usize) -> Self {
        Self {
            inner,
            max_batch_length: (max_batch_length > 0).then_some(max_batch_length),
            buffered_batch: None,
            buffered_offset: 0,
        }
    }
    pub fn new_boxed(
        inner: SendableRecordBatchStream,
        max_batch_length: usize,
    ) -> SendableRecordBatchStream {
        if max_batch_length == 0 {
            inner
        } else {
            Box::pin(Self::new(inner, max_batch_length))
        }
    }
 }
 impl RecordBatchStream for MaxBatchLengthStream {
    fn schema(&self) -> SchemaRef {
        self.inner.schema()
    }
 }
 impl Stream for MaxBatchLengthStream {
    type Item = DataFusionResult<RecordBatch>;
    fn poll_next(
        mut self: Pin<&mut Self>,
        cx: &mut std::task::Context<'_>,
    ) -> std::task::Poll<Option<Self::Item>> {
        loop {
            let Some(max_batch_length) = self.max_batch_length else {
                return Pin::new(&mut self.inner).poll_next(cx);
            };
            if let Some(batch) = self.buffered_batch.clone() {
                if self.buffered_offset < batch.num_rows() {
                    let remaining = batch.num_rows() - self.buffered_offset;
                    let length = remaining.min(max_batch_length);
                    let sliced = batch.slice(self.buffered_offset, length);
                    self.buffered_offset += length;
                    if self.buffered_offset >= batch.num_rows() {
                        self.buffered_batch = None;
                        self.buffered_offset = 0;
                    }
                    return std::task::Poll::Ready(Some(Ok(sliced)));
                }
                self.buffered_batch = None;
                self.buffered_offset = 0;
            }
            match Pin::new(&mut self.inner).poll_next(cx) {
                std::task::Poll::Ready(Some(Ok(batch))) => {
                    if batch.num_rows() <= max_batch_length {
                        return std::task::Poll::Ready(Some(Ok(batch)));
                    }
                    self.buffered_batch = Some(batch);
                    self.buffered_offset = 0;
                }
                other => return other,
            }
        }
    }
 }
 #[cfg(test)]
 mod tests {
    use arrow_array::Int32Array;
@@ -470,7 +549,7 @@ mod tests {
        assert_eq!(string_to_datatype(string), Some(expected));
    }
-    fn sample_batch() -> RecordBatch {
+    fn sample_batch(num_rows: i32) -> RecordBatch {
        let schema = Arc::new(Schema::new(vec![Field::new(
            "col1",
            DataType::Int32,
@@ -478,14 +557,14 @@ mod tests {
        )]));
        RecordBatch::try_new(
            schema.clone(),
-            vec![Arc::new(Int32Array::from(vec![1, 2, 3]))],
+            vec![Arc::new(Int32Array::from_iter_values(0..num_rows))],
        )
        .unwrap()
    }
    #[tokio::test]
    async fn test_timeout_stream() {
-        let batch = sample_batch();
+        let batch = sample_batch(3);
        let schema = batch.schema();
        let mock_stream = stream::iter(vec![Ok(batch.clone()), Ok(batch.clone())]);
@@ -515,7 +594,7 @@ mod tests {
    #[tokio::test]
    async fn test_timeout_stream_zero_duration() {
-        let batch = sample_batch();
+        let batch = sample_batch(3);
        let schema = batch.schema();
        let mock_stream = stream::iter(vec![Ok(batch.clone()), Ok(batch.clone())]);
@@ -534,7 +613,7 @@ mod tests {
    #[tokio::test]
    async fn test_timeout_stream_completes_normally() {
-        let batch = sample_batch();
+        let batch = sample_batch(3);
        let schema = batch.schema();
        let mock_stream = stream::iter(vec![Ok(batch.clone()), Ok(batch.clone())]);
@@ -552,4 +631,35 @@ mod tests {
        // Stream should be empty now
        assert!(timeout_stream.next().await.is_none());
    }
    async fn collect_batch_sizes(
        stream: SendableRecordBatchStream,
        max_batch_length: usize,
    ) -> Vec<usize> {
        let mut sliced_stream = MaxBatchLengthStream::new(stream, max_batch_length);
        sliced_stream
            .by_ref()
            .map(|batch| batch.unwrap().num_rows())
            .collect::<Vec<_>>()
            .await
    }
    #[tokio::test]
    async fn test_max_batch_length_stream_behaviors() {
        let schema = sample_batch(7).schema();
        let mock_stream = stream::iter(vec![Ok(sample_batch(2)), Ok(sample_batch(7))]);
        let sendable_stream: SendableRecordBatchStream =
            Box::pin(RecordBatchStreamAdapter::new(schema.clone(), mock_stream));
        assert_eq!(
            collect_batch_sizes(sendable_stream, 3).await,
            vec![2, 3, 3, 1]
        );
        let sendable_stream: SendableRecordBatchStream = Box::pin(RecordBatchStreamAdapter::new(
            schema,
            stream::iter(vec![Ok(sample_batch(2)), Ok(sample_batch(7))]),
        ));
        assert_eq!(collect_batch_sizes(sendable_stream, 0).await, vec![2, 7]);
    }
 }
`@@ -1,4 +1,4 @@`
	`# LanceDB Java SDK`	`# LanceDB Java Enterprise Client`

	`## Configuration and Initialization`	`## Configuration and Initialization`
`@@ -1,4 +1,4 @@`
	`# LanceDB`	`# LanceDB Python SDK`

	`A Python library for [LanceDB](https://github.com/lancedb/lancedb).`	`A Python library for [LanceDB](https://github.com/lancedb/lancedb).`