Bump version: 0.23.0-beta.3 → 0.23.0

Bump version: 0.23.0-beta.2 → 0.23.0-beta.3
feat: upgrade lance to stable version (#2420 )
2025-12-23 21:39:57 +00:00 · 2025-06-04 21:07:39 +00:00 · 2025-06-04 21:07:38 +00:00 · 2025-06-04 13:34:30 -07:00 · 2025-06-04 08:40:38 -07:00 · 2025-06-04 07:15:07 +00:00
94 changed files with 5727 additions and 1627 deletions
--- a/.bumpversion.toml
+++ b/.bumpversion.toml
@@ -1,5 +1,5 @@
 [tool.bumpversion]
-current_version = "0.19.0-beta.11"
+current_version = "0.20.0-beta.2"
 parse = """(?x)
    (?P<major>0|[1-9]\\d*)\\.
    (?P<minor>0|[1-9]\\d*)\\.
--- a/.github/workflows/java.yml
+++ b/.github/workflows/java.yml
@@ -35,6 +35,9 @@ jobs:
      - uses: Swatinem/rust-cache@v2
        with:
          workspaces: java/core/lancedb-jni
      - uses: actions-rust-lang/setup-rust-toolchain@v1
        with:
          components: rustfmt
      - name: Run cargo fmt
        run: cargo fmt --check
        working-directory: ./java/core/lancedb-jni
@@ -68,6 +71,9 @@ jobs:
      - uses: Swatinem/rust-cache@v2
        with:
          workspaces: java/core/lancedb-jni
      - uses: actions-rust-lang/setup-rust-toolchain@v1
        with:
          components: rustfmt
      - name: Run cargo fmt
        run: cargo fmt --check
        working-directory: ./java/core/lancedb-jni
@@ -110,4 +116,3 @@ jobs:
          -Djdk.reflect.useDirectMethodHandle=false \
          -Dio.netty.tryReflectionSetAccessible=true"
          JAVA_HOME=$JAVA_17 mvn clean test
--- a/.github/workflows/make-release-commit.yml
+++ b/.github/workflows/make-release-commit.yml
@@ -84,6 +84,7 @@ jobs:
        run: |
          pip install bump-my-version PyGithub packaging
          bash ci/bump_version.sh ${{ inputs.type }} ${{ inputs.bump-minor }} v $COMMIT_BEFORE_BUMP
          bash ci/update_lockfiles.sh --amend
      - name: Push new version tag
        if: ${{ !inputs.dry_run }}
        uses: ad-m/github-push-action@master
@@ -92,11 +93,3 @@ jobs:
          github_token: ${{ secrets.LANCEDB_RELEASE_TOKEN }}
          branch: ${{ github.ref }}
          tags: true
      - uses: ./.github/workflows/update_package_lock
        if: ${{ !inputs.dry_run && inputs.other }}
        with:
          github_token: ${{ secrets.GITHUB_TOKEN }}
      - uses: ./.github/workflows/update_package_lock_nodejs
        if: ${{ !inputs.dry_run && inputs.other }}
        with:
          github_token: ${{ secrets.GITHUB_TOKEN }}
--- a/.github/workflows/nodejs.yml
+++ b/.github/workflows/nodejs.yml
@@ -47,6 +47,9 @@ jobs:
      run: |
        sudo apt update
        sudo apt install -y protobuf-compiler libssl-dev
    - uses: actions-rust-lang/setup-rust-toolchain@v1
      with:
        components: rustfmt, clippy
    - name: Lint
      run: |
        cargo fmt --all -- --check
@@ -113,7 +116,7 @@ jobs:
        set -e
        npm ci
        npm run docs
-        if ! git diff --exit-code; then
+        if ! git diff --exit-code -- . ':(exclude)Cargo.lock'; then
          echo "Docs need to be updated"
          echo "Run 'npm run docs', fix any warnings, and commit the changes."
          exit 1
--- a/.github/workflows/npm-publish.yml
+++ b/.github/workflows/npm-publish.yml
@@ -505,6 +505,8 @@ jobs:
    name: vectordb NPM Publish
    needs: [node, node-macos, node-linux-gnu, node-windows]
    runs-on: ubuntu-latest
    permissions:
      contents: write
    # Only runs on tags that matches the make-release action
    if: startsWith(github.ref, 'refs/tags/v')
    steps:
@@ -537,6 +539,10 @@ jobs:
        # We need to deprecate the old package to avoid confusion.
        # Each time we publish a new version, it gets undeprecated.
        run: npm deprecate vectordb "Use @lancedb/lancedb instead."
      - name: Update package-lock.json
        run: bash ci/update_lockfiles.sh
      - name: Push new commit
        uses: ad-m/github-push-action@master
      - name: Notify Slack Action
        uses: ravsamhq/notify-slack-action@2.3.0
        if: ${{ always() }}
@@ -546,21 +552,3 @@ jobs:
          notification_title: "{workflow} is failing"
        env:
          SLACK_WEBHOOK_URL: ${{ secrets.ACTION_MONITORING_SLACK }}
  update-package-lock:
    if: startsWith(github.ref, 'refs/tags/v')
    needs: [release]
    runs-on: ubuntu-latest
    permissions:
      contents: write
    steps:
      - name: Checkout
        uses: actions/checkout@v4
        with:
          ref: main
          token: ${{ secrets.LANCEDB_RELEASE_TOKEN }}
          fetch-depth: 0
          lfs: true
      - uses: ./.github/workflows/update_package_lock
        with:
          github_token: ${{ secrets.GITHUB_TOKEN }}
--- a/.github/workflows/python.yml
+++ b/.github/workflows/python.yml
@@ -228,6 +228,7 @@ jobs:
      - name: Install lancedb
        run: |
          pip install "pydantic<2"
          pip install pyarrow==16
          pip install --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests]
          pip install tantivy
      - name: Run tests
--- a/.github/workflows/run_tests/action.yml
+++ b/.github/workflows/run_tests/action.yml
@@ -24,8 +24,8 @@ runs:
    - name: pytest (with integration)
      shell: bash
      if: ${{ inputs.integration == 'true' }}
-      run: pytest -m "not slow" -x -v --durations=30 python/python/tests
+      run: pytest -m "not slow" -vv --durations=30 python/python/tests
    - name: pytest (no integration tests)
      shell: bash
      if: ${{ inputs.integration != 'true' }}
-      run: pytest -m "not slow and not s3_test" -x -v --durations=30 python/python/tests
+      run: pytest -m "not slow and not s3_test" -vv --durations=30 python/python/tests
--- a/.github/workflows/rust.yml
+++ b/.github/workflows/rust.yml
@@ -40,6 +40,9 @@ jobs:
        with:
          fetch-depth: 0
          lfs: true
      - uses: actions-rust-lang/setup-rust-toolchain@v1
        with:
          components: rustfmt, clippy
      - uses: Swatinem/rust-cache@v2
        with:
          workspaces: rust
@@ -160,8 +163,8 @@ jobs:
    strategy:
      matrix:
        target:
-        - x86_64-pc-windows-msvc
+          - x86_64-pc-windows-msvc
-        - aarch64-pc-windows-msvc
+          - aarch64-pc-windows-msvc
    defaults:
      run:
        working-directory: rust/lancedb
--- a/.github/workflows/update_package_lock/action.yml
+++ b/.github/workflows/update_package_lock/action.yml
@@ -1,33 +0,0 @@
 name: update_package_lock
 description: "Update node's package.lock"
 inputs:
  github_token:
    required: true
    description: "github token for the repo"
 runs:
  using: "composite"
  steps:
    - uses: actions/setup-node@v3
      with:
        node-version: 20
    - name: Set git configs
      shell: bash
      run: |
        git config user.name 'Lance Release'
        git config user.email 'lance-dev@lancedb.com'
    - name: Update package-lock.json file
      working-directory: ./node
      run: |
        npm install
        git add package-lock.json
        git commit -m "Updating package-lock.json"
      shell: bash
    - name: Push changes
      if: ${{ inputs.dry_run }} == "false"
      uses: ad-m/github-push-action@master
      with:
        github_token: ${{ inputs.github_token }}
        branch: main
        tags: true
--- a/.github/workflows/update_package_lock_nodejs/action.yml
+++ b/.github/workflows/update_package_lock_nodejs/action.yml
@@ -1,33 +0,0 @@
 name: update_package_lock_nodejs
 description: "Update nodejs's package.lock"
 inputs:
  github_token:
    required: true
    description: "github token for the repo"
 runs:
  using: "composite"
  steps:
    - uses: actions/setup-node@v3
      with:
        node-version: 20
    - name: Set git configs
      shell: bash
      run: |
        git config user.name 'Lance Release'
        git config user.email 'lance-dev@lancedb.com'
    - name: Update package-lock.json file
      working-directory: ./nodejs
      run: |
        npm install
        git add package-lock.json
        git commit -m "Updating package-lock.json"
      shell: bash
    - name: Push changes
      if: ${{ inputs.dry_run }} == "false"
      uses: ad-m/github-push-action@master
      with:
        github_token: ${{ inputs.github_token }}
        branch: main
        tags: true
--- a/Cargo.lock
+++ b/Cargo.lock
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -21,32 +21,32 @@ categories = ["database-implementations"]
 rust-version = "1.78.0"
 [workspace.dependencies]
-lance = { "version" = "=0.26.0", "features" = ["dynamodb"] }
+lance = { "version" = "=0.29.0", "features" = ["dynamodb"] }
-lance-io = "=0.26.0"
+lance-io = "=0.29.0"
-lance-index = "=0.26.0"
+lance-index = "=0.29.0"
-lance-linalg = "=0.26.0"
+lance-linalg = "=0.29.0"
-lance-table = "=0.26.0"
+lance-table = "=0.29.0"
-lance-testing = "=0.26.0"
+lance-testing = "=0.29.0"
-lance-datafusion = "=0.26.0"
+lance-datafusion = "=0.29.0"
-lance-encoding = "=0.26.0"
+lance-encoding = "=0.29.0"
 # Note that this one does not include pyarrow
-arrow = { version = "54.1", optional = false }
+arrow = { version = "55.1", optional = false }
-arrow-array = "54.1"
+arrow-array = "55.1"
-arrow-data = "54.1"
+arrow-data = "55.1"
-arrow-ipc = "54.1"
+arrow-ipc = "55.1"
-arrow-ord = "54.1"
+arrow-ord = "55.1"
-arrow-schema = "54.1"
+arrow-schema = "55.1"
-arrow-arith = "54.1"
+arrow-arith = "55.1"
-arrow-cast = "54.1"
+arrow-cast = "55.1"
 async-trait = "0"
-datafusion = { version = "46.0", default-features = false }
+datafusion = { version = "47.0", default-features = false }
-datafusion-catalog = "46.0"
+datafusion-catalog = "47.0"
-datafusion-common = { version = "46.0", default-features = false }
+datafusion-common = { version = "47.0", default-features = false }
-datafusion-execution = "46.0"
+datafusion-execution = "47.0"
-datafusion-expr = "46.0"
+datafusion-expr = "47.0"
-datafusion-physical-plan = "46.0"
+datafusion-physical-plan = "47.0"
 env_logger = "0.11"
-half = { "version" = "=2.4.1", default-features = false, features = [
+half = { "version" = "=2.5.0", default-features = false, features = [
    "num-traits",
 ] }
 futures = "0"
@@ -57,19 +57,16 @@ pin-project = "1.0.7"
 snafu = "0.8"
 url = "2"
 num-traits = "0.2"
-rand = "0.8"
+rand = "0.9"
 regex = "1.10"
 lazy_static = "1"
 semver = "1.0.25"
 # Temporary pins to work around downstream issues
 # https://github.com/apache/arrow-rs/commit/2fddf85afcd20110ce783ed5b4cdeb82293da30b
-chrono = "=0.4.39"
+chrono = "=0.4.41"
 # https://github.com/RustCrypto/formats/issues/1684
 base64ct = "=1.6.0"
 # Workaround for: https://github.com/eira-fransham/crunchy/issues/13
 crunchy = "=0.2.2"
 # Workaround for: https://github.com/Lokathor/bytemuck/issues/306
 bytemuck_derive = ">=1.8.1, <1.9.0"
--- a/README.md
+++ b/README.md
@@ -1,94 +1,97 @@
 <a href="https://cloud.lancedb.com" target="_blank">
  <img src="https://github.com/user-attachments/assets/92dad0a2-2a37-4ce1-b783-0d1b4f30a00c" alt="LanceDB Cloud Public Beta" width="100%" style="max-width: 100%;">
 </a>
 <div align="center">
 <p align="center">
-<picture>
+[![LanceDB](docs/src/assets/hero-header.png)](https://lancedb.com)
-  <source media="(prefers-color-scheme: dark)" srcset="https://github.com/user-attachments/assets/ac270358-333e-4bea-a132-acefaa94040e">
+[![Website](https://img.shields.io/badge/-Website-100000?style=for-the-badge&labelColor=645cfb&color=645cfb)](https://lancedb.com/)
-  <source media="(prefers-color-scheme: light)" srcset="https://github.com/user-attachments/assets/b864d814-0d29-4784-8fd9-807297c758c0">
+[![Blog](https://img.shields.io/badge/Blog-100000?style=for-the-badge&labelColor=645cfb&color=645cfb)](https://blog.lancedb.com/)
-  <img alt="LanceDB Logo" src="https://github.com/user-attachments/assets/b864d814-0d29-4784-8fd9-807297c758c0" width=300>
+[![Discord](https://img.shields.io/badge/-Discord-100000?style=for-the-badge&logo=discord&logoColor=white&labelColor=645cfb&color=645cfb)](https://discord.gg/zMM32dvNtd)
-</picture>
+[![Twitter](https://img.shields.io/badge/-Twitter-100000?style=for-the-badge&logo=x&logoColor=white&labelColor=645cfb&color=645cfb)](https://twitter.com/lancedb)
 [![LinkedIn](https://img.shields.io/badge/-LinkedIn-100000?style=for-the-badge&logo=linkedin&logoColor=white&labelColor=645cfb&color=645cfb)](https://www.linkedin.com/company/lancedb/)
 **Search More, Manage Less**
-<a href='https://github.com/lancedb/vectordb-recipes/tree/main' target="_blank"><img alt='LanceDB' src='https://img.shields.io/badge/VectorDB_Recipes-100000?style=for-the-badge&logo=LanceDB&logoColor=white&labelColor=645cfb&color=645cfb'/></a>
+<img src="docs/src/assets/lancedb.png" alt="LanceDB" width="50%">
 <a href='https://lancedb.github.io/lancedb/' target="_blank"><img alt='lancdb' src='https://img.shields.io/badge/DOCS-100000?style=for-the-badge&logo=lancdb&logoColor=white&labelColor=645cfb&color=645cfb'/></a>
 [![Blog](https://img.shields.io/badge/Blog-12100E?style=for-the-badge&logoColor=white)](https://blog.lancedb.com/)
 [![Discord](https://img.shields.io/badge/Discord-%235865F2.svg?style=for-the-badge&logo=discord&logoColor=white)](https://discord.gg/zMM32dvNtd)
 [![Twitter](https://img.shields.io/badge/Twitter-%231DA1F2.svg?style=for-the-badge&logo=Twitter&logoColor=white)](https://twitter.com/lancedb)
 [![Gurubase](https://img.shields.io/badge/Gurubase-Ask%20LanceDB%20Guru-006BFF?style=for-the-badge)](https://gurubase.io/g/lancedb)
-</p>
+# **The Multimodal AI Lakehouse**
-<img max-width="750px" alt="LanceDB Multimodal Search" src="https://github.com/lancedb/lancedb/assets/917119/09c5afc5-7816-4687-bae4-f2ca194426ec">
+[**How to Install** ](#how-to-install) ✦ [**Detailed Documentation**](https://lancedb.github.io/lancedb/) ✦ [**Tutorials and Recipes**](https://github.com/lancedb/vectordb-recipes/tree/main) ✦  [**Contributors**](#contributors) 
 **The ultimate multimodal data platform for AI/ML applications.** 
 LanceDB is designed for fast, scalable, and production-ready vector search. It is built on top of the Lance columnar format. You can store, index, and search over petabytes of multimodal data and vectors with ease. 
 LanceDB is a central location where developers can build, train and analyze their AI workloads.
 </p>
 </div>
-<hr />
+<br>
-LanceDB is an open-source database for vector-search built with persistent storage, which greatly simplifies retrieval, filtering and management of embeddings.
+## **Demo: Multimodal Search by Keyword, Vector or with SQL**
 <img max-width="750px" alt="LanceDB Multimodal Search" src="https://github.com/lancedb/lancedb/assets/917119/09c5afc5-7816-4687-bae4-f2ca194426ec">
-The key features of LanceDB include:
+## **Star LanceDB to get updates!**
-* Production-scale vector search with no servers to manage.
+<details>
 <summary>⭐ Click here ⭐  to see how fast we're growing!</summary>
 <picture>
  <source media="(prefers-color-scheme: dark)" srcset="https://api.star-history.com/svg?repos=lancedb/lancedb&theme=dark&type=Date">
  <img width="100%" src="https://api.star-history.com/svg?repos=lancedb/lancedb&theme=dark&type=Date">
 </picture>
 </details>
-* Store, query and filter vectors, metadata and multi-modal data (text, images, videos, point clouds, and more).
+## **Key Features**:
-* Support for vector similarity search, full-text search and SQL.
+- **Fast Vector Search**: Search billions of vectors in milliseconds with state-of-the-art indexing.
 - **Comprehensive Search**: Support for vector similarity search, full-text search and SQL.
 - **Multimodal Support**: Store, query and filter vectors, metadata and multimodal data (text, images, videos, point clouds, and more).
 - **Advanced Features**: Zero-copy, automatic versioning, manage versions of your data without needing extra infrastructure. GPU support in building vector index.
-* Native Python and Javascript/Typescript support.
+### **Products**:
 - **Open Source & Local**: 100% open source, runs locally or in your cloud. No vendor lock-in.
 - **Cloud and Enterprise**: Production-scale vector search with no servers to manage. Complete data sovereignty and security.
-* Zero-copy, automatic versioning, manage versions of your data without needing extra infrastructure.
+### **Ecosystem**:
 - **Columnar Storage**: Built on the Lance columnar format for efficient storage and analytics.
 - **Seamless Integration**: Python, Node.js, Rust, and REST APIs for easy integration. Native Python and Javascript/Typescript support.
 - **Rich Ecosystem**: Integrations with [**LangChain** 🦜️🔗](https://python.langchain.com/docs/integrations/vectorstores/lancedb/), [**LlamaIndex** 🦙](https://gpt-index.readthedocs.io/en/latest/examples/vector_stores/LanceDBIndexDemo.html), Apache-Arrow, Pandas, Polars, DuckDB and more on the way.
-* GPU support in building vector index(*).
+## **How to Install**:
-* Ecosystem integrations with [LangChain 🦜️🔗](https://python.langchain.com/docs/integrations/vectorstores/lancedb/), [LlamaIndex 🦙](https://gpt-index.readthedocs.io/en/latest/examples/vector_stores/LanceDBIndexDemo.html), Apache-Arrow, Pandas, Polars, DuckDB and more on the way.
+Follow the [Quickstart](https://lancedb.github.io/lancedb/basic/) doc to set up LanceDB locally. 
-LanceDB's core is written in Rust 🦀 and is built using <a href="https://github.com/lancedb/lance">Lance</a>, an open-source columnar format designed for performant ML workloads.
+**API & SDK:** We also support Python, Typescript and Rust SDKs
-## Quick Start
+| Interface | Documentation |
 |-----------|---------------|
 | Python SDK | https://lancedb.github.io/lancedb/python/python/ |
 | Typescript SDK | https://lancedb.github.io/lancedb/js/globals/ |
 | Rust SDK | https://docs.rs/lancedb/latest/lancedb/index.html |
 | REST API | https://docs.lancedb.com/api-reference/introduction |
-**Javascript**
+## **Join Us and Contribute**
 ```shell
 npm install @lancedb/lancedb
 ```
-```javascript
+We welcome contributions from everyone! Whether you're a developer, researcher, or just someone who wants to help out. 
 import * as lancedb from "@lancedb/lancedb";
-const db = await lancedb.connect("data/sample-lancedb");
+If you have any suggestions or feature requests, please feel free to open an issue on GitHub or discuss it on our [**Discord**](https://discord.gg/G5DcmnZWKB) server.
-const table = await db.createTable("vectors", [
+
-	{ id: 1, vector: [0.1, 0.2], item: "foo", price: 10 },
+[**Check out the GitHub Issues**](https://github.com/lancedb/lancedb/issues) if you would like to work on the features that are planned for the future. If you have any suggestions or feature requests, please feel free to open an issue on GitHub. 
-	{ id: 2, vector: [1.1, 1.2], item: "bar", price: 50 },
+
-], {mode: 'overwrite'});
+## **Contributors**
 <a href="https://github.com/lancedb/lancedb/graphs/contributors">
  <img src="https://contrib.rocks/image?repo=lancedb/lancedb" />
 </a>
-const query = table.vectorSearch([0.1, 0.3]).limit(2);
+## **Stay in Touch With Us**
-const results = await query.toArray();
+<div align="center">
-// You can also search for rows by specific criteria without involving a vector search.
+</br>
 const rowsByCriteria = await table.query().where("price >= 10").toArray();
 ```
-**Python**
+[![Website](https://img.shields.io/badge/-Website-100000?style=for-the-badge&labelColor=645cfb&color=645cfb)](https://lancedb.com/)
-```shell
+[![Blog](https://img.shields.io/badge/Blog-100000?style=for-the-badge&labelColor=645cfb&color=645cfb)](https://blog.lancedb.com/)
-pip install lancedb
+[![Discord](https://img.shields.io/badge/-Discord-100000?style=for-the-badge&logo=discord&logoColor=white&labelColor=645cfb&color=645cfb)](https://discord.gg/zMM32dvNtd)
-```
+[![Twitter](https://img.shields.io/badge/-Twitter-100000?style=for-the-badge&logo=x&logoColor=white&labelColor=645cfb&color=645cfb)](https://twitter.com/lancedb)
 [![LinkedIn](https://img.shields.io/badge/-LinkedIn-100000?style=for-the-badge&logo=linkedin&logoColor=white&labelColor=645cfb&color=645cfb)](https://www.linkedin.com/company/lancedb/)
-```python
+</div>
 import lancedb
 uri = "data/sample-lancedb"
 db = lancedb.connect(uri)
 table = db.create_table("my_table",
                         data=[{"vector": [3.1, 4.1], "item": "foo", "price": 10.0},
                               {"vector": [5.9, 26.5], "item": "bar", "price": 20.0}])
 result = table.search([100, 100]).limit(2).to_pandas()
 ```
 ## Blogs, Tutorials & Videos
 * 📈 <a href="https://blog.lancedb.com/benchmarking-random-access-in-lance/">2000x better performance with Lance over Parquet</a>
 * 🤖 <a href="https://github.com/lancedb/vectordb-recipes/tree/main/examples/Youtube-Search-QA-Bot">Build a question and answer bot with LanceDB</a>
--- a/ci/set_lance_version.py
+++ b/ci/set_lance_version.py
@@ -0,0 +1,174 @@
 import argparse
 import sys
 import json
 def run_command(command: str) -> str:
    """
    Run a shell command and return stdout as a string.
    If exit code is not 0, raise an exception with the stderr output.
    """
    import subprocess
    result = subprocess.run(command, shell=True, capture_output=True, text=True)
    if result.returncode != 0:
        raise Exception(f"Command failed with error: {result.stderr.strip()}")
    return result.stdout.strip()
 def get_latest_stable_version() -> str:
    version_line = run_command("cargo info lance | grep '^version:'")
    version = version_line.split(" ")[1].strip()
    return version
 def get_latest_preview_version() -> str:
    lance_tags = run_command(
        "git ls-remote --tags https://github.com/lancedb/lance.git | grep 'refs/tags/v[0-9beta.-]\\+$'"
    ).splitlines()
    lance_tags = (
        tag.split("refs/tags/")[1]
        for tag in lance_tags
        if "refs/tags/" in tag and "beta" in tag
    )
    from packaging.version import Version
    latest = max(
        (tag[1:] for tag in lance_tags if tag.startswith("v")), key=lambda t: Version(t)
    )
    return str(latest)
 def extract_features(line: str) -> list:
    """
    Extracts the features from a line in Cargo.toml.
    Example: 'lance = { "version" = "=0.29.0", "features" = ["dynamodb"] }'
    Returns: ['dynamodb']
    """
    import re
    match = re.search(r'"features"\s*=\s*\[(.*?)\]', line)
    if match:
        features_str = match.group(1)
        return [f.strip('"') for f in features_str.split(",")]
    return []
 def update_cargo_toml(line_updater):
    """
    Updates the Cargo.toml file by applying the line_updater function to each line.
    The line_updater function should take a line as input and return the updated line.
    """
    with open("Cargo.toml", "r") as f:
        lines = f.readlines()
    new_lines = []
    for line in lines:
        if line.startswith("lance"):
            # Update the line using the provided function
            new_lines.append(line_updater(line))
        else:
            # Keep the line unchanged
            new_lines.append(line)
    with open("Cargo.toml", "w") as f:
        f.writelines(new_lines)
 def set_stable_version(version: str):
    """
    Sets lines to
    lance = { "version" = "=0.29.0", "features" = ["dynamodb"] }
    lance-io = "=0.29.0"
    ...
    """
    def line_updater(line: str) -> str:
        package_name = line.split("=", maxsplit=1)[0].strip()
        features = extract_features(line)
        if features:
            return f'{package_name} = {{ "version" = "={version}", "features" = {json.dumps(features)} }}\n'
        else:
            return f'{package_name} = "={version}"\n'
    update_cargo_toml(line_updater)
 def set_preview_version(version: str):
    """
    Sets lines to
    lance = { "version" = "=0.29.0", "features" = ["dynamodb"], tag = "v0.29.0-beta.2", git="https://github.com/lancedb/lance.git" }
    lance-io = { version = "=0.29.0", tag = "v0.29.0-beta.2", git="https://github.com/lancedb/lance.git" }
    ...
    """
    def line_updater(line: str) -> str:
        package_name = line.split("=", maxsplit=1)[0].strip()
        features = extract_features(line)
        base_version = version.split("-")[0]  # Get the base version without beta suffix
        if features:
            return f'{package_name} = {{ "version" = "={base_version}", "features" = {json.dumps(features)}, "tag" = "v{version}", "git" = "https://github.com/lancedb/lance.git" }}\n'
        else:
            return f'{package_name} = {{ "version" = "={base_version}", "tag" = "v{version}", "git" = "https://github.com/lancedb/lance.git" }}\n'
    update_cargo_toml(line_updater)
 def set_local_version():
    """
    Sets lines to
    lance = { path = "../lance/rust/lance", features = ["dynamodb"] }
    lance-io = { path = "../lance/rust/lance-io" }
    ...
    """
    def line_updater(line: str) -> str:
        package_name = line.split("=", maxsplit=1)[0].strip()
        features = extract_features(line)
        if features:
            return f'{package_name} = {{ "path" = "../lance/rust/{package_name}", "features" = {json.dumps(features)} }}\n'
        else:
            return f'{package_name} = {{ "path" = "../lance/rust/{package_name}" }}\n'
    update_cargo_toml(line_updater)
 parser = argparse.ArgumentParser(description="Set the version of the Lance package.")
 parser.add_argument(
    "version",
    type=str,
    help="The version to set for the Lance package. Use 'stable' for the latest stable version, 'preview' for latest preview version, or a specific version number (e.g., '0.1.0'). You can also specify 'local' to use a local path.",
 )
 args = parser.parse_args()
 if args.version == "stable":
    latest_stable_version = get_latest_stable_version()
    print(
        f"Found latest stable version: \033[1mv{latest_stable_version}\033[0m",
        file=sys.stderr,
    )
    set_stable_version(latest_stable_version)
 elif args.version == "preview":
    latest_preview_version = get_latest_preview_version()
    print(
        f"Found latest preview version: \033[1mv{latest_preview_version}\033[0m",
        file=sys.stderr,
    )
    set_preview_version(latest_preview_version)
 elif args.version == "local":
    set_local_version()
 else:
    # Parse the version number.
    version = args.version
    # Ignore initial v if present.
    if version.startswith("v"):
        version = version[1:]
    if "beta" in version:
        set_preview_version(version)
    else:
        set_stable_version(version)
 print("Updating lockfiles...", file=sys.stderr, end="")
 run_command("cargo metadata > /dev/null")
 print(" done.", file=sys.stderr)
--- a/ci/update_lockfiles.sh
+++ b/ci/update_lockfiles.sh
@@ -0,0 +1,30 @@
 #!/usr/bin/env bash
 set -euo pipefail
 AMEND=false
 for arg in "$@"; do
  if [[ "$arg" == "--amend" ]]; then
    AMEND=true
  fi
 done
 # This updates the lockfile without building
 cargo metadata --quiet > /dev/null
 pushd nodejs || exit 1
 npm install --package-lock-only --silent
 popd
 pushd node || exit 1
 npm install --package-lock-only --silent
 popd
 if git diff --quiet --exit-code; then
  echo "No lockfile changes to commit; skipping amend."
 elif $AMEND; then
  git add Cargo.lock nodejs/package-lock.json node/package-lock.json
  git commit --amend --no-edit
 else
  git add Cargo.lock nodejs/package-lock.json node/package-lock.json
  git commit -m "Update lockfiles"
 fi
--- a/docs/mkdocs.yml
+++ b/docs/mkdocs.yml
@@ -193,6 +193,7 @@ nav:
          - Pandas and PyArrow: python/pandas_and_pyarrow.md
          - Polars: python/polars_arrow.md
          - DuckDB: python/duckdb.md
          - Datafusion: python/datafusion.md
          - LangChain:
              - LangChain 🔗: integrations/langchain.md
              - LangChain demo: notebooks/langchain_demo.ipynb
@@ -205,6 +206,7 @@ nav:
          - PromptTools: integrations/prompttools.md
          - dlt: integrations/dlt.md
          - phidata: integrations/phidata.md
          - Genkit: integrations/genkit.md
      - 🎯 Examples:
          - Overview: examples/index.md
          - 🐍 Python:
@@ -247,6 +249,7 @@ nav:
      - Data management: concepts/data_management.md
  - Guides:
      - Working with tables: guides/tables.md
      - Working with SQL: guides/sql_querying.md
      - Building an ANN index: ann_indexes.md
      - Vector Search: search.md
      - Full-text search (native): fts.md
@@ -323,6 +326,7 @@ nav:
      - Pandas and PyArrow: python/pandas_and_pyarrow.md
      - Polars: python/polars_arrow.md
      - DuckDB: python/duckdb.md
      - Datafusion: python/datafusion.md
      - LangChain 🦜️🔗↗: integrations/langchain.md
      - LangChain.js 🦜️🔗↗: https://js.langchain.com/docs/integrations/vectorstores/lancedb
      - LlamaIndex 🦙↗: integrations/llamaIndex.md
@@ -331,6 +335,7 @@ nav:
      - PromptTools: integrations/prompttools.md
      - dlt: integrations/dlt.md
      - phidata: integrations/phidata.md
      - Genkit: integrations/genkit.md
  - Examples:
      - examples/index.md
      - 🐍 Python:
--- a/docs/overrides/partials/main.html
+++ b/docs/overrides/partials/main.html
@@ -0,0 +1,5 @@
 {% extends "base.html" %}
 {% block announce %}
  📚 Starting June 1st, 2025, please use <a href="https://lancedb.github.io/documentation" target="_blank" rel="noopener noreferrer">lancedb.github.io/documentation</a> for the latest docs.
 {% endblock %}
--- a/docs/src/ann_indexes.md
+++ b/docs/src/ann_indexes.md
@@ -291,7 +291,7 @@ Product quantization can lead to approximately `16 * sizeof(float32) / 1 = 64` t
 `num_partitions` is used to decide how many partitions the first level `IVF` index uses.
 Higher number of partitions could lead to more efficient I/O during queries and better accuracy, but it takes much more time to train.
-On `SIFT-1M` dataset, our benchmark shows that keeping each partition 1K-4K rows lead to a good latency / recall.
+On `SIFT-1M` dataset, our benchmark shows that keeping each partition 4K-8K rows lead to a good latency / recall.
 `num_sub_vectors` specifies how many Product Quantization (PQ) short codes to generate on each vector. The number should be a factor of the vector dimension. Because
 PQ is a lossy compression of the original vector, a higher `num_sub_vectors` usually results in
--- a/docs/src/assets/hero-header.png
+++ b/docs/src/assets/hero-header.png
--- a/docs/src/assets/lancedb.png
+++ b/docs/src/assets/lancedb.png
--- a/docs/src/guides/sql_querying.md
+++ b/docs/src/guides/sql_querying.md
@@ -0,0 +1,66 @@
 You can use DuckDB and Apache Datafusion to query your LanceDB tables using SQL.
 This guide will show how to query Lance tables them using both.
 We will re-use the dataset [created previously](./pandas_and_pyarrow.md):
 ```python
 import lancedb
 db = lancedb.connect("data/sample-lancedb")
 data = [
    {"vector": [3.1, 4.1], "item": "foo", "price": 10.0},
    {"vector": [5.9, 26.5], "item": "bar", "price": 20.0}
 ]
 table = db.create_table("pd_table", data=data)
 ```
 ## Querying a LanceDB Table with DuckDb
 The `to_lance` method converts the LanceDB table to a `LanceDataset`, which is accessible to DuckDB through the Arrow compatibility layer.
 To query the resulting Lance dataset in DuckDB, all you need to do is reference the dataset by the same name in your SQL query.
 ```python
 import duckdb
 arrow_table = table.to_lance()
 duckdb.query("SELECT * FROM arrow_table")
 ```
 ```
 ┌─────────────┬─────────┬────────┐
 │   vector    │  item   │ price  │
 │   float[]   │ varchar │ double │
 ├─────────────┼─────────┼────────┤
 │ [3.1, 4.1]  │ foo     │   10.0 │
 │ [5.9, 26.5] │ bar     │   20.0 │
 └─────────────┴─────────┴────────┘
 ```
 ## Querying a LanceDB Table with Apache Datafusion
 Have the required imports before doing any querying.
 === "Python"
    ```python
    --8<-- "python/python/tests/docs/test_guide_tables.py:import-lancedb"
    --8<-- "python/python/tests/docs/test_guide_tables.py:import-session-context"
    --8<-- "python/python/tests/docs/test_guide_tables.py:import-ffi-dataset"
    ```
 Register the table created with the Datafusion session context.
 === "Python"
    ```python
    --8<-- "python/python/tests/docs/test_guide_tables.py:lance_sql_basic"
    ```
 ```
 ┌─────────────┬─────────┬────────┐
 │   vector    │  item   │ price  │
 │   float[]   │ varchar │ double │
 ├─────────────┼─────────┼────────┤
 │ [3.1, 4.1]  │ foo     │   10.0 │
 │ [5.9, 26.5] │ bar     │   20.0 │
 └─────────────┴─────────┴────────┘
 ```
--- a/docs/src/guides/tables.md
+++ b/docs/src/guides/tables.md
@@ -765,7 +765,7 @@ This can be used to update zero to all rows depending on how many rows match the
        ];
        const tbl = await db.createTable("my_table", data)
-        await tbl.update({ 
+        await tbl.update({
            values: { vector: [10, 10] },
            where: "x = 2"
        });
@@ -787,9 +787,9 @@ This can be used to update zero to all rows depending on how many rows match the
        ];
        const tbl = await db.createTable("my_table", data)
-        await tbl.update({ 
+        await tbl.update({
-            where: "x = 2", 
+            where: "x = 2",
-            values: { vector: [10, 10] } 
+            values: { vector: [10, 10] }
        });
        ```
--- a/docs/src/integrations/genkit.md
+++ b/docs/src/integrations/genkit.md
@@ -0,0 +1,183 @@
 ### genkitx-lancedb
 This is a lancedb plugin for genkit framework. It allows you to use LanceDB for ingesting and rereiving data using genkit framework.
 ![integration-banner-genkit](https://github.com/user-attachments/assets/a6cc28af-98e9-4425-b87c-7ab139bd7893)
 ### Installation
 ```bash
 pnpm install genkitx-lancedb
 ```
 ### Usage
 Adding LanceDB plugin to your genkit instance.
 ```ts
 import { lancedbIndexerRef, lancedb, lancedbRetrieverRef, WriteMode } from 'genkitx-lancedb';
 import { textEmbedding004, vertexAI } from '@genkit-ai/vertexai';
 import { gemini } from '@genkit-ai/vertexai';
 import { z, genkit } from 'genkit';
 import { Document } from 'genkit/retriever';
 import { chunk } from 'llm-chunk';
 import { readFile } from 'fs/promises';
 import path from 'path';
 import pdf from 'pdf-parse/lib/pdf-parse';
 const ai = genkit({
  plugins: [
    // vertexAI provides the textEmbedding004 embedder
    vertexAI(),
    // the local vector store requires an embedder to translate from text to vector
    lancedb([
      {
        dbUri: '.db', // optional lancedb uri, default to .db
        tableName: 'table', // optional table name, default to table
        embedder: textEmbedding004,
      },
    ]),
  ],
 });
 ```
 You can run this app with the following command:
 ```bash
 genkit start -- tsx --watch src/index.ts
 ```
 This'll add LanceDB as a retriever and indexer to the genkit instance. You can see it in the GUI view
 <img width="1710" alt="Screenshot 2025-05-11 at 7 21 05 PM" src="https://github.com/user-attachments/assets/e752f7f4-785b-4797-a11e-72ab06a531b7" />
 **Testing retrieval on a sample table**
 Let's see the raw retrieval results
 <img width="1710" alt="Screenshot 2025-05-11 at 7 21 05 PM" src="https://github.com/user-attachments/assets/b8d356ed-8421-4790-8fc0-d6af563b9657" />
 On running this query, you'll 5 results fetched from the lancedb table, where each result looks something like this:
 <img width="1417" alt="Screenshot 2025-05-11 at 7 21 18 PM" src="https://github.com/user-attachments/assets/77429525-36e2-4da6-a694-e58c1cf9eb83" />
 ## Creating a custom RAG flow
 Now that we've seen how you can use LanceDB for in a genkit pipeline, let's refine the flow and create a RAG. A RAG flow will consist of an index and a retreiver with its outputs postprocessed an fed into an LLM for final response
 ### Creating custom indexer flows
 You can also create custom indexer flows, utilizing more options and features provided by LanceDB.
 ```ts
 export const menuPdfIndexer = lancedbIndexerRef({
   // Using all defaults, for dbUri, tableName, and embedder, etc
 });
 const chunkingConfig = {
  minLength: 1000,
  maxLength: 2000,
  splitter: 'sentence',
  overlap: 100,
  delimiters: '',
 } as any;
 async function extractTextFromPdf(filePath: string) {
  const pdfFile = path.resolve(filePath);
  const dataBuffer = await readFile(pdfFile);
  const data = await pdf(dataBuffer);
  return data.text;
 }
 export const indexMenu = ai.defineFlow(
  {
    name: 'indexMenu',
    inputSchema: z.string().describe('PDF file path'),
    outputSchema: z.void(),
  },
  async (filePath: string) => {
    filePath = path.resolve(filePath);
    // Read the pdf.
    const pdfTxt = await ai.run('extract-text', () =>
      extractTextFromPdf(filePath)
    );
    // Divide the pdf text into segments.
    const chunks = await ai.run('chunk-it', async () =>
      chunk(pdfTxt, chunkingConfig)
    );
    // Convert chunks of text into documents to store in the index.
    const documents = chunks.map((text) => {
      return Document.fromText(text, { filePath });
    });
    // Add documents to the index.
    await ai.index({
      indexer: menuPdfIndexer,
      documents,
      options: {
        writeMode: WriteMode.Overwrite,
      } as any
    });
  }
 );
 ```
 <img width="1316" alt="Screenshot 2025-05-11 at 8 35 56 PM" src="https://github.com/user-attachments/assets/e2a20ce4-d1d0-4fa2-9a84-f2cc26e3a29f" />
 In your console, you can see the logs
 <img width="511" alt="Screenshot 2025-05-11 at 7 19 14 PM" src="https://github.com/user-attachments/assets/243f26c5-ed38-40b6-b661-002f40f0423a" />
 ### Creating custom retriever flows
 You can also create custom retriever flows, utilizing more options and features provided by LanceDB.
 ```ts
 export const menuRetriever = lancedbRetrieverRef({
  tableName: "table", // Use the same table name as the indexer.
  displayName: "Menu", // Use a custom display name.
 export const menuQAFlow = ai.defineFlow(
  { name: "Menu", inputSchema: z.string(), outputSchema: z.string() },
  async (input: string) => {
    // retrieve relevant documents
    const docs = await ai.retrieve({
      retriever: menuRetriever,
      query: input,
      options: { 
        k: 3,
      },
    });
    const extractedContent = docs.map(doc => {
      if (doc.content && Array.isArray(doc.content) && doc.content.length > 0) {
        if (doc.content[0].media && doc.content[0].media.url) {
          return doc.content[0].media.url;
        }
      }
      return "No content found";
    });
    console.log("Extracted content:", extractedContent);
    const { text } = await ai.generate({
      model: gemini('gemini-2.0-flash'),
      prompt: `
 You are acting as a helpful AI assistant that can answer 
 questions about the food available on the menu at Genkit Grub Pub.
 Use only the context provided to answer the question.
 If you don't know, do not make up an answer.
 Do not add or change items on the menu.
 Context:
 ${extractedContent.join('\n\n')}
 Question: ${input}`,
      docs,
    });
    return text;
  }
 );
 ```
 Now using our retrieval flow, we can ask question about the ingsted PDF
 <img width="1306" alt="Screenshot 2025-05-11 at 7 18 45 PM" src="https://github.com/user-attachments/assets/86c66b13-7c12-4d5f-9d81-ae36bfb1c346" />
--- a/docs/src/js/classes/MergeInsertBuilder.md
+++ b/docs/src/js/classes/MergeInsertBuilder.md
@@ -33,20 +33,22 @@ Construct a MergeInsertBuilder. __Internal use only.__
 ### execute()
 ```ts
-execute(data): Promise<void>
+execute(data, execOptions?): Promise<MergeResult>
 ```
 Executes the merge insert operation
 Nothing is returned but the `Table` is updated
 #### Parameters
 * **data**: [`Data`](../type-aliases/Data.md)
 * **execOptions?**: `Partial`&lt;[`WriteExecutionOptions`](../interfaces/WriteExecutionOptions.md)&gt;
 #### Returns
-`Promise`&lt;`void`&gt;
+`Promise`&lt;[`MergeResult`](../interfaces/MergeResult.md)&gt;
 the merge result
 ***
--- a/docs/src/js/classes/Table.md
+++ b/docs/src/js/classes/Table.md
@@ -40,7 +40,7 @@ Returns the name of the table
 ### add()
 ```ts
-abstract add(data, options?): Promise<void>
+abstract add(data, options?): Promise<AddResult>
 ```
 Insert records into this Table.
@@ -54,14 +54,17 @@ Insert records into this Table.
 #### Returns
-`Promise`&lt;`void`&gt;
+`Promise`&lt;[`AddResult`](../interfaces/AddResult.md)&gt;
 A promise that resolves to an object
 containing the new version number of the table
 ***
 ### addColumns()
 ```ts
-abstract addColumns(newColumnTransforms): Promise<void>
+abstract addColumns(newColumnTransforms): Promise<AddColumnsResult>
 ```
 Add new columns with defined values.
@@ -76,14 +79,17 @@ Add new columns with defined values.
 #### Returns
-`Promise`&lt;`void`&gt;
+`Promise`&lt;[`AddColumnsResult`](../interfaces/AddColumnsResult.md)&gt;
 A promise that resolves to an object
 containing the new version number of the table after adding the columns.
 ***
 ### alterColumns()
 ```ts
-abstract alterColumns(columnAlterations): Promise<void>
+abstract alterColumns(columnAlterations): Promise<AlterColumnsResult>
 ```
 Alter the name or nullability of columns.
@@ -96,7 +102,10 @@ Alter the name or nullability of columns.
 #### Returns
-`Promise`&lt;`void`&gt;
+`Promise`&lt;[`AlterColumnsResult`](../interfaces/AlterColumnsResult.md)&gt;
 A promise that resolves to an object
 containing the new version number of the table after altering the columns.
 ***
@@ -117,8 +126,8 @@ wish to return to standard mode, call `checkoutLatest`.
 #### Parameters
-* **version**: `number`
+* **version**: `string` \| `number`
-    The version to checkout
+    The version to checkout, could be version number or tag
 #### Returns
@@ -252,7 +261,7 @@ await table.createIndex("my_float_col");
 ### delete()
 ```ts
-abstract delete(predicate): Promise<void>
+abstract delete(predicate): Promise<DeleteResult>
 ```
 Delete the rows that satisfy the predicate.
@@ -263,7 +272,10 @@ Delete the rows that satisfy the predicate.
 #### Returns
-`Promise`&lt;`void`&gt;
+`Promise`&lt;[`DeleteResult`](../interfaces/DeleteResult.md)&gt;
 A promise that resolves to an object
 containing the new version number of the table
 ***
@@ -284,7 +296,7 @@ Return a brief description of the table
 ### dropColumns()
 ```ts
-abstract dropColumns(columnNames): Promise<void>
+abstract dropColumns(columnNames): Promise<DropColumnsResult>
 ```
 Drop one or more columns from the dataset
@@ -303,7 +315,10 @@ then call ``cleanup_files`` to remove the old files.
 #### Returns
-`Promise`&lt;`void`&gt;
+`Promise`&lt;[`DropColumnsResult`](../interfaces/DropColumnsResult.md)&gt;
 A promise that resolves to an object
 containing the new version number of the table after dropping the columns.
 ***
@@ -615,6 +630,50 @@ of the given query
 ***
 ### stats()
 ```ts
 abstract stats(): Promise<TableStatistics>
 ```
 Returns table and fragment statistics
 #### Returns
 `Promise`&lt;[`TableStatistics`](../interfaces/TableStatistics.md)&gt;
 The table and fragment statistics
 ***
 ### tags()
 ```ts
 abstract tags(): Promise<Tags>
 ```
 Get a tags manager for this table.
 Tags allow you to label specific versions of a table with a human-readable name.
 The returned tags manager can be used to list, create, update, or delete tags.
 #### Returns
 `Promise`&lt;[`Tags`](Tags.md)&gt;
 A tags manager for this table
 #### Example
 ```typescript
 const tagsManager = await table.tags();
 await tagsManager.create("v1", 1);
 const tags = await tagsManager.list();
 console.log(tags); // { "v1": { version: 1, manifestSize: ... } }
 ```
 ***
 ### toArrow()
 ```ts
@@ -634,7 +693,7 @@ Return the table as an arrow table
 #### update(opts)
 ```ts
-abstract update(opts): Promise<void>
+abstract update(opts): Promise<UpdateResult>
 ```
 Update existing records in the Table
@@ -645,7 +704,10 @@ Update existing records in the Table
 ##### Returns
-`Promise`&lt;`void`&gt;
+`Promise`&lt;[`UpdateResult`](../interfaces/UpdateResult.md)&gt;
 A promise that resolves to an object containing
 the number of rows updated and the new version number
 ##### Example
@@ -656,7 +718,7 @@ table.update({where:"x = 2", values:{"vector": [10, 10]}})
 #### update(opts)
 ```ts
-abstract update(opts): Promise<void>
+abstract update(opts): Promise<UpdateResult>
 ```
 Update existing records in the Table
@@ -667,7 +729,10 @@ Update existing records in the Table
 ##### Returns
-`Promise`&lt;`void`&gt;
+`Promise`&lt;[`UpdateResult`](../interfaces/UpdateResult.md)&gt;
 A promise that resolves to an object containing
 the number of rows updated and the new version number
 ##### Example
@@ -678,7 +743,7 @@ table.update({where:"x = 2", valuesSql:{"x": "x + 1"}})
 #### update(updates, options)
 ```ts
-abstract update(updates, options?): Promise<void>
+abstract update(updates, options?): Promise<UpdateResult>
 ```
 Update existing records in the Table
@@ -701,10 +766,6 @@ repeatedly calilng this method.
 * **updates**: `Record`&lt;`string`, `string`&gt; \| `Map`&lt;`string`, `string`&gt;
    the
    columns to update
    Keys in the map should specify the name of the column to update.
    Values in the map provide the new value of the column.  These can
    be SQL literal strings (e.g. "7" or "'foo'") or they can be expressions
    based on the row being updated (e.g. "my_col + 1")
 * **options?**: `Partial`&lt;[`UpdateOptions`](../interfaces/UpdateOptions.md)&gt;
    additional options to control
@@ -712,7 +773,15 @@ repeatedly calilng this method.
 ##### Returns
-`Promise`&lt;`void`&gt;
+`Promise`&lt;[`UpdateResult`](../interfaces/UpdateResult.md)&gt;
 A promise that resolves to an object
 containing the number of rows updated and the new version number
 Keys in the map should specify the name of the column to update.
 Values in the map provide the new value of the column.  These can
 be SQL literal strings (e.g. "7" or "'foo'") or they can be expressions
 based on the row being updated (e.g. "my_col + 1")
 ***
--- a/docs/src/js/classes/TagContents.md
+++ b/docs/src/js/classes/TagContents.md
@@ -0,0 +1,35 @@
 [**@lancedb/lancedb**](../README.md) • **Docs**
 ***
 [@lancedb/lancedb](../globals.md) / TagContents
 # Class: TagContents
 ## Constructors
 ### new TagContents()
 ```ts
 new TagContents(): TagContents
 ```
 #### Returns
 [`TagContents`](TagContents.md)
 ## Properties
 ### manifestSize
 ```ts
 manifestSize: number;
 ```
 ***
 ### version
 ```ts
 version: number;
 ```
--- a/docs/src/js/classes/Tags.md
+++ b/docs/src/js/classes/Tags.md
@@ -0,0 +1,99 @@
 [**@lancedb/lancedb**](../README.md) • **Docs**
 ***
 [@lancedb/lancedb](../globals.md) / Tags
 # Class: Tags
 ## Constructors
 ### new Tags()
 ```ts
 new Tags(): Tags
 ```
 #### Returns
 [`Tags`](Tags.md)
 ## Methods
 ### create()
 ```ts
 create(tag, version): Promise<void>
 ```
 #### Parameters
 * **tag**: `string`
 * **version**: `number`
 #### Returns
 `Promise`&lt;`void`&gt;
 ***
 ### delete()
 ```ts
 delete(tag): Promise<void>
 ```
 #### Parameters
 * **tag**: `string`
 #### Returns
 `Promise`&lt;`void`&gt;
 ***
 ### getVersion()
 ```ts
 getVersion(tag): Promise<number>
 ```
 #### Parameters
 * **tag**: `string`
 #### Returns
 `Promise`&lt;`number`&gt;
 ***
 ### list()
 ```ts
 list(): Promise<Record<string, TagContents>>
 ```
 #### Returns
 `Promise`&lt;`Record`&lt;`string`, [`TagContents`](TagContents.md)&gt;&gt;
 ***
 ### update()
 ```ts
 update(tag, version): Promise<void>
 ```
 #### Parameters
 * **tag**: `string`
 * **version**: `number`
 #### Returns
 `Promise`&lt;`void`&gt;
--- a/docs/src/js/globals.md
+++ b/docs/src/js/globals.md
@@ -27,19 +27,28 @@
 - [QueryBase](classes/QueryBase.md)
 - [RecordBatchIterator](classes/RecordBatchIterator.md)
 - [Table](classes/Table.md)
 - [TagContents](classes/TagContents.md)
 - [Tags](classes/Tags.md)
 - [VectorColumnOptions](classes/VectorColumnOptions.md)
 - [VectorQuery](classes/VectorQuery.md)
 ## Interfaces
 - [AddColumnsResult](interfaces/AddColumnsResult.md)
 - [AddColumnsSql](interfaces/AddColumnsSql.md)
 - [AddDataOptions](interfaces/AddDataOptions.md)
 - [AddResult](interfaces/AddResult.md)
 - [AlterColumnsResult](interfaces/AlterColumnsResult.md)
 - [ClientConfig](interfaces/ClientConfig.md)
 - [ColumnAlteration](interfaces/ColumnAlteration.md)
 - [CompactionStats](interfaces/CompactionStats.md)
 - [ConnectionOptions](interfaces/ConnectionOptions.md)
 - [CreateTableOptions](interfaces/CreateTableOptions.md)
 - [DeleteResult](interfaces/DeleteResult.md)
 - [DropColumnsResult](interfaces/DropColumnsResult.md)
 - [ExecutableQuery](interfaces/ExecutableQuery.md)
 - [FragmentStatistics](interfaces/FragmentStatistics.md)
 - [FragmentSummaryStats](interfaces/FragmentSummaryStats.md)
 - [FtsOptions](interfaces/FtsOptions.md)
 - [FullTextQuery](interfaces/FullTextQuery.md)
 - [FullTextSearchOptions](interfaces/FullTextSearchOptions.md)
@@ -50,6 +59,7 @@
 - [IndexStatistics](interfaces/IndexStatistics.md)
 - [IvfFlatOptions](interfaces/IvfFlatOptions.md)
 - [IvfPqOptions](interfaces/IvfPqOptions.md)
 - [MergeResult](interfaces/MergeResult.md)
 - [OpenTableOptions](interfaces/OpenTableOptions.md)
 - [OptimizeOptions](interfaces/OptimizeOptions.md)
 - [OptimizeStats](interfaces/OptimizeStats.md)
@@ -57,9 +67,12 @@
 - [RemovalStats](interfaces/RemovalStats.md)
 - [RetryConfig](interfaces/RetryConfig.md)
 - [TableNamesOptions](interfaces/TableNamesOptions.md)
 - [TableStatistics](interfaces/TableStatistics.md)
 - [TimeoutConfig](interfaces/TimeoutConfig.md)
 - [UpdateOptions](interfaces/UpdateOptions.md)
 - [UpdateResult](interfaces/UpdateResult.md)
 - [Version](interfaces/Version.md)
 - [WriteExecutionOptions](interfaces/WriteExecutionOptions.md)
 ## Type Aliases
--- a/docs/src/js/interfaces/AddColumnsResult.md
+++ b/docs/src/js/interfaces/AddColumnsResult.md
@@ -0,0 +1,15 @@
 [**@lancedb/lancedb**](../README.md) • **Docs**
 ***
 [@lancedb/lancedb](../globals.md) / AddColumnsResult
 # Interface: AddColumnsResult
 ## Properties
 ### version
 ```ts
 version: number;
 ```
--- a/docs/src/js/interfaces/AddResult.md
+++ b/docs/src/js/interfaces/AddResult.md
@@ -0,0 +1,15 @@
 [**@lancedb/lancedb**](../README.md) • **Docs**
 ***
 [@lancedb/lancedb](../globals.md) / AddResult
 # Interface: AddResult
 ## Properties
 ### version
 ```ts
 version: number;
 ```
--- a/docs/src/js/interfaces/AlterColumnsResult.md
+++ b/docs/src/js/interfaces/AlterColumnsResult.md
@@ -0,0 +1,15 @@
 [**@lancedb/lancedb**](../README.md) • **Docs**
 ***
 [@lancedb/lancedb](../globals.md) / AlterColumnsResult
 # Interface: AlterColumnsResult
 ## Properties
 ### version
 ```ts
 version: number;
 ```
--- a/docs/src/js/interfaces/DeleteResult.md
+++ b/docs/src/js/interfaces/DeleteResult.md
@@ -0,0 +1,15 @@
 [**@lancedb/lancedb**](../README.md) • **Docs**
 ***
 [@lancedb/lancedb](../globals.md) / DeleteResult
 # Interface: DeleteResult
 ## Properties
 ### version
 ```ts
 version: number;
 ```
--- a/docs/src/js/interfaces/DropColumnsResult.md
+++ b/docs/src/js/interfaces/DropColumnsResult.md
@@ -0,0 +1,15 @@
 [**@lancedb/lancedb**](../README.md) • **Docs**
 ***
 [@lancedb/lancedb](../globals.md) / DropColumnsResult
 # Interface: DropColumnsResult
 ## Properties
 ### version
 ```ts
 version: number;
 ```
--- a/docs/src/js/interfaces/FragmentStatistics.md
+++ b/docs/src/js/interfaces/FragmentStatistics.md
@@ -0,0 +1,37 @@
 [**@lancedb/lancedb**](../README.md) • **Docs**
 ***
 [@lancedb/lancedb](../globals.md) / FragmentStatistics
 # Interface: FragmentStatistics
 ## Properties
 ### lengths
 ```ts
 lengths: FragmentSummaryStats;
 ```
 Statistics on the number of rows in the table fragments
 ***
 ### numFragments
 ```ts
 numFragments: number;
 ```
 The number of fragments in the table
 ***
 ### numSmallFragments
 ```ts
 numSmallFragments: number;
 ```
 The number of uncompacted fragments in the table
--- a/docs/src/js/interfaces/FragmentSummaryStats.md
+++ b/docs/src/js/interfaces/FragmentSummaryStats.md
@@ -0,0 +1,77 @@
 [**@lancedb/lancedb**](../README.md) • **Docs**
 ***
 [@lancedb/lancedb](../globals.md) / FragmentSummaryStats
 # Interface: FragmentSummaryStats
 ## Properties
 ### max
 ```ts
 max: number;
 ```
 The number of rows in the fragment with the most rows
 ***
 ### mean
 ```ts
 mean: number;
 ```
 The mean number of rows in the fragments
 ***
 ### min
 ```ts
 min: number;
 ```
 The number of rows in the fragment with the fewest rows
 ***
 ### p25
 ```ts
 p25: number;
 ```
 The 25th percentile of number of rows in the fragments
 ***
 ### p50
 ```ts
 p50: number;
 ```
 The 50th percentile of number of rows in the fragments
 ***
 ### p75
 ```ts
 p75: number;
 ```
 The 75th percentile of number of rows in the fragments
 ***
 ### p99
 ```ts
 p99: number;
 ```
 The 99th percentile of number of rows in the fragments
--- a/docs/src/js/interfaces/MergeResult.md
+++ b/docs/src/js/interfaces/MergeResult.md
@@ -0,0 +1,39 @@
 [**@lancedb/lancedb**](../README.md) • **Docs**
 ***
 [@lancedb/lancedb](../globals.md) / MergeResult
 # Interface: MergeResult
 ## Properties
 ### numDeletedRows
 ```ts
 numDeletedRows: number;
 ```
 ***
 ### numInsertedRows
 ```ts
 numInsertedRows: number;
 ```
 ***
 ### numUpdatedRows
 ```ts
 numUpdatedRows: number;
 ```
 ***
 ### version
 ```ts
 version: number;
 ```
--- a/docs/src/js/interfaces/TableStatistics.md
+++ b/docs/src/js/interfaces/TableStatistics.md
@@ -0,0 +1,47 @@
 [**@lancedb/lancedb**](../README.md) • **Docs**
 ***
 [@lancedb/lancedb](../globals.md) / TableStatistics
 # Interface: TableStatistics
 ## Properties
 ### fragmentStats
 ```ts
 fragmentStats: FragmentStatistics;
 ```
 Statistics on table fragments
 ***
 ### numIndices
 ```ts
 numIndices: number;
 ```
 The number of indices in the table
 ***
 ### numRows
 ```ts
 numRows: number;
 ```
 The number of rows in the table
 ***
 ### totalBytes
 ```ts
 totalBytes: number;
 ```
 The total number of bytes in the table
--- a/docs/src/js/interfaces/UpdateResult.md
+++ b/docs/src/js/interfaces/UpdateResult.md
@@ -0,0 +1,23 @@
 [**@lancedb/lancedb**](../README.md) • **Docs**
 ***
 [@lancedb/lancedb](../globals.md) / UpdateResult
 # Interface: UpdateResult
 ## Properties
 ### rowsUpdated
 ```ts
 rowsUpdated: number;
 ```
 ***
 ### version
 ```ts
 version: number;
 ```
--- a/docs/src/js/interfaces/WriteExecutionOptions.md
+++ b/docs/src/js/interfaces/WriteExecutionOptions.md
@@ -0,0 +1,26 @@
 [**@lancedb/lancedb**](../README.md) • **Docs**
 ***
 [@lancedb/lancedb](../globals.md) / WriteExecutionOptions
 # Interface: WriteExecutionOptions
 ## Properties
 ### timeoutMs?
 ```ts
 optional timeoutMs: number;
 ```
 Maximum time to run the operation before cancelling it.
 By default, there is a 30-second timeout that is only enforced after the
 first attempt. This is to prevent spending too long retrying to resolve
 conflicts. For example, if a write attempt takes 20 seconds and fails,
 the second attempt will be cancelled after 10 seconds, hitting the
 30-second timeout. However, a write that takes one hour and succeeds on the
 first attempt will not be cancelled.
 When this is set, the timeout is enforced on all attempts, including the first.
--- a/docs/src/python/datafusion.md
+++ b/docs/src/python/datafusion.md
@@ -0,0 +1,53 @@
 # Apache Datafusion
 In Python, LanceDB tables can also be queried with [Apache Datafusion](https://datafusion.apache.org/), an extensible query engine written in Rust that uses Apache Arrow as its in-memory format. This means you can write complex SQL queries to analyze your data in LanceDB.
 This integration is done via [Datafusion FFI](https://docs.rs/datafusion-ffi/latest/datafusion_ffi/), which provides a native integration between LanceDB and Datafusion.
 The Datafusion FFI allows to pass down column selections and basic filters to LanceDB, reducing the amount of scanned data when executing your query. Additionally, the integration allows streaming data from LanceDB tables which allows to do aggregation larger-than-memory.
 We can demonstrate this by first installing `datafusion` and `lancedb`.
 ```shell
 pip install datafusion lancedb
 ```
 We will re-use the dataset [created previously](./pandas_and_pyarrow.md):
 ```python
 import lancedb
 from datafusion import SessionContext
 from lance import FFILanceTableProvider
 db = lancedb.connect("data/sample-lancedb")
 data = [
    {"vector": [3.1, 4.1], "item": "foo", "price": 10.0},
    {"vector": [5.9, 26.5], "item": "bar", "price": 20.0}
 ]
 lance_table = db.create_table("lance_table", data)
 ctx = SessionContext()
 ffi_lance_table = FFILanceTableProvider(
    lance_table.to_lance(), with_row_id=True, with_row_addr=True
 )
 ctx.register_table_provider("ffi_lance_table", ffi_lance_table)
 ```
 The `to_lance` method converts the LanceDB table to a `LanceDataset`, which is accessible to Datafusion through the Datafusion FFI integration layer.
 To query the resulting Lance dataset in Datafusion, you first need to register the dataset with Datafusion and then just reference it by the same name in your SQL query.
 ```python
 ctx.table("ffi_lance_table")
 ctx.sql("SELECT * FROM ffi_lance_table")
 ```
 ```
 ┌─────────────┬─────────┬────────┬─────────────────┬─────────────────┐
 │   vector    │  item   │ price  │ _rowid          │ _rowaddr        │
 │   float[]   │ varchar │ double │ bigint unsigned │ bigint unsigned │
 ├─────────────┼─────────┼────────┼─────────────────┼─────────────────┤
 │ [3.1, 4.1]  │ foo     │   10.0 │               0 │               0 │
 │ [5.9, 26.5] │ bar     │   20.0 │               1 │               1 │
 └─────────────┴─────────┴────────┴─────────────────┴─────────────────┘
 ```
--- a/java/core/pom.xml
+++ b/java/core/pom.xml
@@ -8,7 +8,7 @@
    <parent>
        <groupId>com.lancedb</groupId>
        <artifactId>lancedb-parent</artifactId>
-        <version>0.19.0-beta.11</version>
+        <version>0.20.0-beta.2</version>
        <relativePath>../pom.xml</relativePath>
    </parent>
--- a/java/pom.xml
+++ b/java/pom.xml
@@ -6,7 +6,7 @@
    <groupId>com.lancedb</groupId>
    <artifactId>lancedb-parent</artifactId>
-    <version>0.19.0-beta.11</version>
+    <version>0.20.0-beta.2</version>
    <packaging>pom</packaging>
    <name>LanceDB Parent</name>
--- a/node/package-lock.json
+++ b/node/package-lock.json
@@ -1,12 +1,12 @@
 {
  "name": "vectordb",
-  "version": "0.19.0-beta.11",
+  "version": "0.20.0-beta.2",
  "lockfileVersion": 3,
  "requires": true,
  "packages": {
    "": {
      "name": "vectordb",
-      "version": "0.19.0-beta.11",
+      "version": "0.20.0-beta.2",
      "cpu": [
        "x64",
        "arm64"
@@ -52,11 +52,11 @@
        "uuid": "^9.0.0"
      },
      "optionalDependencies": {
-        "@lancedb/vectordb-darwin-arm64": "0.19.0-beta.11",
+        "@lancedb/vectordb-darwin-arm64": "0.20.0-beta.2",
-        "@lancedb/vectordb-darwin-x64": "0.19.0-beta.11",
+        "@lancedb/vectordb-darwin-x64": "0.20.0-beta.2",
-        "@lancedb/vectordb-linux-arm64-gnu": "0.19.0-beta.11",
+        "@lancedb/vectordb-linux-arm64-gnu": "0.20.0-beta.2",
-        "@lancedb/vectordb-linux-x64-gnu": "0.19.0-beta.11",
+        "@lancedb/vectordb-linux-x64-gnu": "0.20.0-beta.2",
-        "@lancedb/vectordb-win32-x64-msvc": "0.19.0-beta.11"
+        "@lancedb/vectordb-win32-x64-msvc": "0.20.0-beta.2"
      },
      "peerDependencies": {
        "@apache-arrow/ts": "^14.0.2",
@@ -327,65 +327,60 @@
      }
    },
    "node_modules/@lancedb/vectordb-darwin-arm64": {
-      "version": "0.19.0-beta.11",
+      "version": "0.20.0-beta.2",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-darwin-arm64/-/vectordb-darwin-arm64-0.19.0-beta.11.tgz",
+      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-darwin-arm64/-/vectordb-darwin-arm64-0.20.0-beta.2.tgz",
-      "integrity": "sha512-fLefGJYdlIRIjrJj8MU1r5Zix5LpKktpCYilA7tZrfvBWkubGceJ+U6RPsWz7VGBfWcETo3g5CBooUPhbtSMlQ==",
+      "integrity": "sha512-H9PmJ/5KSvstVzR8Q7T22+eHRjJZ2ef3aA3gdFxXvoMi3xQ0MGIxz23HuKHGTRT4tfl1nNnpOPb2W7Na8etK9w==",
      "cpu": [
        "arm64"
      ],
      "license": "Apache-2.0",
      "optional": true,
      "os": [
        "darwin"
      ]
    },
    "node_modules/@lancedb/vectordb-darwin-x64": {
-      "version": "0.19.0-beta.11",
+      "version": "0.20.0-beta.2",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-darwin-x64/-/vectordb-darwin-x64-0.19.0-beta.11.tgz",
+      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-darwin-x64/-/vectordb-darwin-x64-0.20.0-beta.2.tgz",
-      "integrity": "sha512-FkCa1TbPLDXAGhlRI4tafcltzApCsyvgi+I+kX07u5DKPNQVALpQ3R6X6GLlbiFsAFBdyv9t2fqQ9DlgjJIZpA==",
+      "integrity": "sha512-9AQkv4tIys+vg0cplZtSE48o61jd7EnmuMkUht+vLORL5/HAma84eAoU9lXHT7zAtPAQmL+98Bfvcsx7fJ6mVw==",
      "cpu": [
        "x64"
      ],
      "license": "Apache-2.0",
      "optional": true,
      "os": [
        "darwin"
      ]
    },
    "node_modules/@lancedb/vectordb-linux-arm64-gnu": {
-      "version": "0.19.0-beta.11",
+      "version": "0.20.0-beta.2",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-linux-arm64-gnu/-/vectordb-linux-arm64-gnu-0.19.0-beta.11.tgz",
+      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-linux-arm64-gnu/-/vectordb-linux-arm64-gnu-0.20.0-beta.2.tgz",
-      "integrity": "sha512-iZkL/01HNUNQ8pGK0+hoNyrM7P1YtShsyIQVzJMfo41SAofCBf9qvi9YaYyd49sDb+dQXeRn1+cfaJ9siz1OHw==",
+      "integrity": "sha512-eQWoJz2ePml7NyEInTBeakWx56+5c6r2p3F+iHC5tsLuznn6eFX90koXJunRxH1WXHDN48ECUlEmKypgfEmn4w==",
      "cpu": [
        "arm64"
      ],
      "license": "Apache-2.0",
      "optional": true,
      "os": [
        "linux"
      ]
    },
    "node_modules/@lancedb/vectordb-linux-x64-gnu": {
-      "version": "0.19.0-beta.11",
+      "version": "0.20.0-beta.2",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-linux-x64-gnu/-/vectordb-linux-x64-gnu-0.19.0-beta.11.tgz",
+      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-linux-x64-gnu/-/vectordb-linux-x64-gnu-0.20.0-beta.2.tgz",
-      "integrity": "sha512-MdKRHxe2tRQqmExNLv3f6Wvx1mEi98eFtD0ysm4tNrQdaS1MJbTp+DUehrRKkfDWsooalHkIi9d02BVw5qseUQ==",
+      "integrity": "sha512-/+84U+Dt07m8Jk0b8h+SvOzlrynITPP3SDBOlB+OonwmGSxirXhc8gkfNZctgXOJYKMyRIRSsMHP/QNjOp2ajA==",
      "cpu": [
        "x64"
      ],
      "license": "Apache-2.0",
      "optional": true,
      "os": [
        "linux"
      ]
    },
    "node_modules/@lancedb/vectordb-win32-x64-msvc": {
-      "version": "0.19.0-beta.11",
+      "version": "0.20.0-beta.2",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-win32-x64-msvc/-/vectordb-win32-x64-msvc-0.19.0-beta.11.tgz",
+      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-win32-x64-msvc/-/vectordb-win32-x64-msvc-0.20.0-beta.2.tgz",
-      "integrity": "sha512-KWy+t9jr0feJAW9NkmM/w9kfdpp78+7mkeh9lb0g3xI3OdYU1yizNqFjbIQqJf7/L4sou4wmOjAC+FcP8qCtzg==",
+      "integrity": "sha512-bgdunAPnknBh/5oO+vr6RXMr6wb3hHugNPXcIidxYMQvgFa8uhaAKtgYkAKuoyUReOYo8DGtVkZxNUUpZbF7/A==",
      "cpu": [
        "x64"
      ],
      "license": "Apache-2.0",
      "optional": true,
      "os": [
        "win32"
--- a/node/package.json
+++ b/node/package.json
@@ -1,6 +1,6 @@
 {
  "name": "vectordb",
-  "version": "0.19.0-beta.11",
+  "version": "0.20.0-beta.2",
  "description": " Serverless, low-latency vector database for AI applications",
  "private": false,
  "main": "dist/index.js",
@@ -89,10 +89,10 @@
    }
  },
  "optionalDependencies": {
-    "@lancedb/vectordb-darwin-x64": "0.19.0-beta.11",
+    "@lancedb/vectordb-darwin-x64": "0.20.0-beta.2",
-    "@lancedb/vectordb-darwin-arm64": "0.19.0-beta.11",
+    "@lancedb/vectordb-darwin-arm64": "0.20.0-beta.2",
-    "@lancedb/vectordb-linux-x64-gnu": "0.19.0-beta.11",
+    "@lancedb/vectordb-linux-x64-gnu": "0.20.0-beta.2",
-    "@lancedb/vectordb-linux-arm64-gnu": "0.19.0-beta.11",
+    "@lancedb/vectordb-linux-arm64-gnu": "0.20.0-beta.2",
-    "@lancedb/vectordb-win32-x64-msvc": "0.19.0-beta.11"
+    "@lancedb/vectordb-win32-x64-msvc": "0.20.0-beta.2"
  }
 }
--- a/nodejs/Cargo.toml
+++ b/nodejs/Cargo.toml
@@ -1,7 +1,7 @@
 [package]
 name = "lancedb-nodejs"
 edition.workspace = true
-version = "0.19.0-beta.11"
+version = "0.20.0-beta.2"
 license.workspace = true
 description.workspace = true
 repository.workspace = true
@@ -30,6 +30,7 @@ log.workspace = true
 # Workaround for build failure until we can fix it.
 aws-lc-sys = "=0.28.0"
 aws-lc-rs = "=1.13.0"
 [build-dependencies]
 napi-build = "2.1"
--- a/nodejs/test/arrow.test.ts
+++ b/nodejs/test/arrow.test.ts
@@ -374,6 +374,71 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
        expect(table2.numRows).toBe(4);
        expect(table2.schema).toEqual(schema);
      });
      it("should correctly retain values in nested struct fields", async function () {
        // Define test data with nested struct
        const testData = [
          {
            id: "doc1",
            vector: [1, 2, 3],
            metadata: {
              filePath: "/path/to/file1.ts",
              startLine: 10,
              endLine: 20,
              text: "function test() { return true; }",
            },
          },
          {
            id: "doc2",
            vector: [4, 5, 6],
            metadata: {
              filePath: "/path/to/file2.ts",
              startLine: 30,
              endLine: 40,
              text: "function test2() { return false; }",
            },
          },
        ];
        // Create Arrow table from the data
        const table = makeArrowTable(testData);
        // Verify schema has the nested struct fields
        const metadataField = table.schema.fields.find(
          (f) => f.name === "metadata",
        );
        expect(metadataField).toBeDefined();
        // biome-ignore lint/suspicious/noExplicitAny: accessing fields in different Arrow versions
        const childNames = metadataField?.type.children.map((c: any) => c.name);
        expect(childNames).toEqual([
          "filePath",
          "startLine",
          "endLine",
          "text",
        ]);
        // Convert to buffer and back (simulating storage and retrieval)
        const buf = await fromTableToBuffer(table);
        const retrievedTable = tableFromIPC(buf);
        // Verify the retrieved table has the same structure
        const rows = [];
        for (let i = 0; i < retrievedTable.numRows; i++) {
          rows.push(retrievedTable.get(i));
        }
        // Check values in the first row
        const firstRow = rows[0];
        expect(firstRow.id).toBe("doc1");
        expect(firstRow.vector.toJSON()).toEqual([1, 2, 3]);
        // Verify metadata values are preserved (this is where the bug is)
        expect(firstRow.metadata).toBeDefined();
        expect(firstRow.metadata.filePath).toBe("/path/to/file1.ts");
        expect(firstRow.metadata.startLine).toBe(10);
        expect(firstRow.metadata.endLine).toBe(20);
        expect(firstRow.metadata.text).toBe("function test() { return true; }");
      });
    });
    class DummyEmbedding extends EmbeddingFunction<string> {
--- a/nodejs/test/table.test.ts
+++ b/nodejs/test/table.test.ts
@@ -34,6 +34,7 @@ import {
 } from "../lancedb/embedding";
 import { Index } from "../lancedb/indices";
 import { instanceOfFullTextQuery } from "../lancedb/query";
 import exp = require("constants");
 describe.each([arrow15, arrow16, arrow17, arrow18])(
  "Given a table",
@@ -71,8 +72,33 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
      await expect(table.countRows()).resolves.toBe(3);
    });
-    it("should overwrite data if asked", async () => {
+    it("should show table stats", async () => {
      await table.add([{ id: 1 }, { id: 2 }]);
      await table.add([{ id: 1 }]);
      await expect(table.stats()).resolves.toEqual({
        fragmentStats: {
          lengths: {
            max: 2,
            mean: 1,
            min: 1,
            p25: 1,
            p50: 2,
            p75: 2,
            p99: 2,
          },
          numFragments: 2,
          numSmallFragments: 2,
        },
        numIndices: 0,
        numRows: 3,
        totalBytes: 24,
      });
    });
    it("should overwrite data if asked", async () => {
      const addRes = await table.add([{ id: 1 }, { id: 2 }]);
      expect(addRes).toHaveProperty("version");
      expect(addRes.version).toBe(2);
      await table.add([{ id: 1 }], { mode: "overwrite" });
      await expect(table.countRows()).resolves.toBe(1);
    });
@@ -88,7 +114,11 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
      await table.add([{ id: 1 }]);
      expect(await table.countRows("id == 1")).toBe(1);
      expect(await table.countRows("id == 7")).toBe(0);
-      await table.update({ id: "7" });
+      const updateRes = await table.update({ id: "7" });
      expect(updateRes).toHaveProperty("version");
      expect(updateRes.version).toBe(3);
      expect(updateRes).toHaveProperty("rowsUpdated");
      expect(updateRes.rowsUpdated).toBe(1);
      expect(await table.countRows("id == 1")).toBe(0);
      expect(await table.countRows("id == 7")).toBe(1);
      await table.add([{ id: 2 }]);
@@ -315,11 +345,17 @@ describe("merge insert", () => {
      { a: 3, b: "y" },
      { a: 4, b: "z" },
    ];
-    await table
+    const mergeInsertRes = await table
      .mergeInsert("a")
      .whenMatchedUpdateAll()
      .whenNotMatchedInsertAll()
-      .execute(newData);
+      .execute(newData, { timeoutMs: 10_000 });
    expect(mergeInsertRes).toHaveProperty("version");
    expect(mergeInsertRes.version).toBe(2);
    expect(mergeInsertRes.numInsertedRows).toBe(1);
    expect(mergeInsertRes.numUpdatedRows).toBe(2);
    expect(mergeInsertRes.numDeletedRows).toBe(0);
    const expected = [
      { a: 1, b: "a" },
      { a: 2, b: "x" },
@@ -337,10 +373,12 @@ describe("merge insert", () => {
      { a: 3, b: "y" },
      { a: 4, b: "z" },
    ];
-    await table
+    const mergeInsertRes = await table
      .mergeInsert("a")
      .whenMatchedUpdateAll({ where: "target.b = 'b'" })
      .execute(newData);
    expect(mergeInsertRes).toHaveProperty("version");
    expect(mergeInsertRes.version).toBe(2);
    const expected = [
      { a: 1, b: "a" },
@@ -425,6 +463,20 @@ describe("merge insert", () => {
    res = res.sort((a, b) => a.a - b.a);
    expect(res).toEqual(expected);
  });
  test("timeout", async () => {
    const newData = [
      { a: 2, b: "x" },
      { a: 4, b: "z" },
    ];
    await expect(
      table
        .mergeInsert("a")
        .whenMatchedUpdateAll()
        .whenNotMatchedInsertAll()
        .execute(newData, { timeoutMs: 0 }),
    ).rejects.toThrow("merge insert timed out");
  });
 });
 describe("When creating an index", () => {
@@ -1000,15 +1052,19 @@ describe("schema evolution", function () {
      { id: 1n, vector: [0.1, 0.2] },
    ]);
    // Can create a non-nullable column only through addColumns at the moment.
-    await table.addColumns([
+    const addColumnsRes = await table.addColumns([
      { name: "price", valueSql: "cast(10.0 as double)" },
    ]);
    expect(addColumnsRes).toHaveProperty("version");
    expect(addColumnsRes.version).toBe(2);
    expect(await table.schema()).toEqual(schema);
-    await table.alterColumns([
+    const alterColumnsRes = await table.alterColumns([
      { path: "id", rename: "new_id" },
      { path: "price", nullable: true },
    ]);
    expect(alterColumnsRes).toHaveProperty("version");
    expect(alterColumnsRes.version).toBe(3);
    const expectedSchema = new Schema([
      new Field("new_id", new Int64(), true),
@@ -1126,7 +1182,9 @@ describe("schema evolution", function () {
    const table = await con.createTable("vectors", [
      { id: 1n, vector: [0.1, 0.2] },
    ]);
-    await table.dropColumns(["vector"]);
+    const dropColumnsRes = await table.dropColumns(["vector"]);
    expect(dropColumnsRes).toHaveProperty("version");
    expect(dropColumnsRes.version).toBe(2);
    const expectedSchema = new Schema([new Field("id", new Int64(), true)]);
    expect(await table.schema()).toEqual(expectedSchema);
@@ -1178,6 +1236,99 @@ describe("when dealing with versioning", () => {
  });
 });
 describe("when dealing with tags", () => {
  let tmpDir: tmp.DirResult;
  beforeEach(() => {
    tmpDir = tmp.dirSync({ unsafeCleanup: true });
  });
  afterEach(() => {
    tmpDir.removeCallback();
  });
  it("can manage tags", async () => {
    const conn = await connect(tmpDir.name, {
      readConsistencyInterval: 0,
    });
    const table = await conn.createTable("my_table", [
      { id: 1n, vector: [0.1, 0.2] },
    ]);
    expect(await table.version()).toBe(1);
    await table.add([{ id: 2n, vector: [0.3, 0.4] }]);
    expect(await table.version()).toBe(2);
    const tagsManager = await table.tags();
    const initialTags = await tagsManager.list();
    expect(Object.keys(initialTags).length).toBe(0);
    const tag1 = "tag1";
    await tagsManager.create(tag1, 1);
    expect(await tagsManager.getVersion(tag1)).toBe(1);
    const tagsAfterFirst = await tagsManager.list();
    expect(Object.keys(tagsAfterFirst).length).toBe(1);
    expect(tagsAfterFirst).toHaveProperty(tag1);
    expect(tagsAfterFirst[tag1].version).toBe(1);
    await tagsManager.create("tag2", 2);
    expect(await tagsManager.getVersion("tag2")).toBe(2);
    const tagsAfterSecond = await tagsManager.list();
    expect(Object.keys(tagsAfterSecond).length).toBe(2);
    expect(tagsAfterSecond).toHaveProperty(tag1);
    expect(tagsAfterSecond[tag1].version).toBe(1);
    expect(tagsAfterSecond).toHaveProperty("tag2");
    expect(tagsAfterSecond["tag2"].version).toBe(2);
    await table.add([{ id: 3n, vector: [0.5, 0.6] }]);
    await tagsManager.update(tag1, 3);
    expect(await tagsManager.getVersion(tag1)).toBe(3);
    await tagsManager.delete("tag2");
    const tagsAfterDelete = await tagsManager.list();
    expect(Object.keys(tagsAfterDelete).length).toBe(1);
    expect(tagsAfterDelete).toHaveProperty(tag1);
    expect(tagsAfterDelete[tag1].version).toBe(3);
    await table.add([{ id: 4n, vector: [0.7, 0.8] }]);
    expect(await table.version()).toBe(4);
    await table.checkout(tag1);
    expect(await table.version()).toBe(3);
    await table.checkoutLatest();
    expect(await table.version()).toBe(4);
  });
  it("can checkout and restore tags", async () => {
    const conn = await connect(tmpDir.name, {
      readConsistencyInterval: 0,
    });
    const table = await conn.createTable("my_table", [
      { id: 1n, vector: [0.1, 0.2] },
    ]);
    expect(await table.version()).toBe(1);
    expect(await table.countRows()).toBe(1);
    const tagsManager = await table.tags();
    const tag1 = "tag1";
    await tagsManager.create(tag1, 1);
    await table.add([{ id: 2n, vector: [0.3, 0.4] }]);
    const tag2 = "tag2";
    await tagsManager.create(tag2, 2);
    expect(await table.version()).toBe(2);
    await table.checkout(tag1);
    expect(await table.version()).toBe(1);
    await table.restore();
    expect(await table.version()).toBe(3);
    expect(await table.countRows()).toBe(1);
    await table.add([{ id: 3n, vector: [0.5, 0.6] }]);
    expect(await table.countRows()).toBe(2);
  });
 });
 describe("when optimizing a dataset", () => {
  let tmpDir: tmp.DirResult;
  let table: Table;
@@ -1355,7 +1506,9 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
      ];
      const table = await db.createTable("test", data);
      await table.createIndex("text", {
-        config: Index.fts(),
+        config: Index.fts({
          withPosition: true,
        }),
      });
      const results = await table.search("lance").toArray();
@@ -1408,7 +1561,9 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
      ];
      const table = await db.createTable("test", data);
      await table.createIndex("text", {
-        config: Index.fts(),
+        config: Index.fts({
          withPosition: true,
        }),
      });
      const results = await table.search("world").toArray();
--- a/nodejs/lancedb/arrow.ts
+++ b/nodejs/lancedb/arrow.ts
@@ -639,8 +639,9 @@ function transposeData(
 ): Vector {
  if (field.type instanceof Struct) {
    const childFields = field.type.children;
    const fullPath = [...path, field.name];
    const childVectors = childFields.map((child) => {
-      return transposeData(data, child, [...path, child.name]);
+      return transposeData(data, child, fullPath);
    });
    const structData = makeData({
      type: field.type,
@@ -652,7 +653,14 @@ function transposeData(
    const values = data.map((datum) => {
      let current: unknown = datum;
      for (const key of valuesPath) {
-        if (isObject(current) && Object.hasOwn(current, key)) {
+        if (current == null) {
          return null;
        }
        if (
          isObject(current) &&
          (Object.hasOwn(current, key) || key in current)
        ) {
          current = current[key];
        } else {
          return null;
--- a/nodejs/lancedb/index.ts
+++ b/nodejs/lancedb/index.ts
@@ -23,6 +23,18 @@ export {
  OptimizeStats,
  CompactionStats,
  RemovalStats,
  TableStatistics,
  FragmentStatistics,
  FragmentSummaryStats,
  Tags,
  TagContents,
  MergeResult,
  AddResult,
  AddColumnsResult,
  AlterColumnsResult,
  DeleteResult,
  DropColumnsResult,
  UpdateResult,
 } from "./native.js";
 export {
@@ -74,7 +86,7 @@ export {
  ColumnAlteration,
 } from "./table";
-export { MergeInsertBuilder } from "./merge";
+export { MergeInsertBuilder, WriteExecutionOptions } from "./merge";
 export * as embedding from "./embedding";
 export * as rerankers from "./rerankers";
--- a/nodejs/lancedb/merge.ts
+++ b/nodejs/lancedb/merge.ts
@@ -1,7 +1,7 @@
 // SPDX-License-Identifier: Apache-2.0
 // SPDX-FileCopyrightText: Copyright The LanceDB Authors
 import { Data, Schema, fromDataToBuffer } from "./arrow";
-import { NativeMergeInsertBuilder } from "./native";
+import { MergeResult, NativeMergeInsertBuilder } from "./native";
 /** A builder used to create and run a merge insert operation */
 export class MergeInsertBuilder {
@@ -73,9 +73,12 @@ export class MergeInsertBuilder {
  /**
   * Executes the merge insert operation
   *
-   * Nothing is returned but the `Table` is updated
+   * @returns {Promise<MergeResult>} the merge result
   */
-  async execute(data: Data): Promise<void> {
+  async execute(
    data: Data,
    execOptions?: Partial<WriteExecutionOptions>,
  ): Promise<MergeResult> {
    let schema: Schema;
    if (this.#schema instanceof Promise) {
      schema = await this.#schema;
@@ -83,7 +86,28 @@ export class MergeInsertBuilder {
    } else {
      schema = this.#schema;
    }
    if (execOptions?.timeoutMs !== undefined) {
      this.#native.setTimeout(execOptions.timeoutMs);
    }
    const buffer = await fromDataToBuffer(data, undefined, schema);
-    await this.#native.execute(buffer);
+    return await this.#native.execute(buffer);
  }
 }
 export interface WriteExecutionOptions {
  /**
   * Maximum time to run the operation before cancelling it.
   *
   * By default, there is a 30-second timeout that is only enforced after the
   * first attempt. This is to prevent spending too long retrying to resolve
   * conflicts. For example, if a write attempt takes 20 seconds and fails,
   * the second attempt will be cancelled after 10 seconds, hitting the
   * 30-second timeout. However, a write that takes one hour and succeeds on the
   * first attempt will not be cancelled.
   *
   * When this is set, the timeout is enforced on all attempts, including the first.
   */
  timeoutMs?: number;
 }
--- a/nodejs/lancedb/table.ts
+++ b/nodejs/lancedb/table.ts
@@ -16,10 +16,18 @@ import { EmbeddingFunctionConfig, getRegistry } from "./embedding/registry";
 import { IndexOptions } from "./indices";
 import { MergeInsertBuilder } from "./merge";
 import {
  AddColumnsResult,
  AddColumnsSql,
  AddResult,
  AlterColumnsResult,
  DeleteResult,
  DropColumnsResult,
  IndexConfig,
  IndexStatistics,
  OptimizeStats,
  TableStatistics,
  Tags,
  UpdateResult,
  Table as _NativeTable,
 } from "./native";
 import {
@@ -124,12 +132,19 @@ export abstract class Table {
  /**
   * Insert records into this Table.
   * @param {Data} data Records to be inserted into the Table
   * @returns {Promise<AddResult>} A promise that resolves to an object
   * containing the new version number of the table
   */
-  abstract add(data: Data, options?: Partial<AddDataOptions>): Promise<void>;
+  abstract add(
    data: Data,
    options?: Partial<AddDataOptions>,
  ): Promise<AddResult>;
  /**
   * Update existing records in the Table
   * @param opts.values The values to update. The keys are the column names and the values
   * are the values to set.
   * @returns {Promise<UpdateResult>} A promise that resolves to an object containing
   * the number of rows updated and the new version number
   * @example
   * ```ts
   * table.update({where:"x = 2", values:{"vector": [10, 10]}})
@@ -139,11 +154,13 @@ export abstract class Table {
    opts: {
      values: Map<string, IntoSql> | Record<string, IntoSql>;
    } & Partial<UpdateOptions>,
-  ): Promise<void>;
+  ): Promise<UpdateResult>;
  /**
   * Update existing records in the Table
   * @param opts.valuesSql The values to update. The keys are the column names and the values
   * are the values to set. The values are SQL expressions.
   * @returns {Promise<UpdateResult>} A promise that resolves to an object containing
   * the number of rows updated and the new version number
   * @example
   * ```ts
   * table.update({where:"x = 2", valuesSql:{"x": "x + 1"}})
@@ -153,7 +170,7 @@ export abstract class Table {
    opts: {
      valuesSql: Map<string, string> | Record<string, string>;
    } & Partial<UpdateOptions>,
-  ): Promise<void>;
+  ): Promise<UpdateResult>;
  /**
   * Update existing records in the Table
   *
@@ -171,6 +188,8 @@ export abstract class Table {
   * repeatedly calilng this method.
   * @param {Map<string, string> | Record<string, string>} updates - the
   * columns to update
   * @returns {Promise<UpdateResult>} A promise that resolves to an object
   * containing the number of rows updated and the new version number
   *
   * Keys in the map should specify the name of the column to update.
   * Values in the map provide the new value of the column.  These can
@@ -182,12 +201,16 @@ export abstract class Table {
  abstract update(
    updates: Map<string, string> | Record<string, string>,
    options?: Partial<UpdateOptions>,
-  ): Promise<void>;
+  ): Promise<UpdateResult>;
  /** Count the total number of rows in the dataset. */
  abstract countRows(filter?: string): Promise<number>;
-  /** Delete the rows that satisfy the predicate. */
+  /**
-  abstract delete(predicate: string): Promise<void>;
+   * Delete the rows that satisfy the predicate.
   * @returns {Promise<DeleteResult>} A promise that resolves to an object
   * containing the new version number of the table
   */
  abstract delete(predicate: string): Promise<DeleteResult>;
  /**
   * Create an index to speed up queries.
   *
@@ -341,15 +364,23 @@ export abstract class Table {
   * the SQL expression to use to calculate the value of the new column. These
   * expressions will be evaluated for each row in the table, and can
   * reference existing columns in the table.
   * @returns {Promise<AddColumnsResult>} A promise that resolves to an object
   * containing the new version number of the table after adding the columns.
   */
-  abstract addColumns(newColumnTransforms: AddColumnsSql[]): Promise<void>;
+  abstract addColumns(
    newColumnTransforms: AddColumnsSql[],
  ): Promise<AddColumnsResult>;
  /**
   * Alter the name or nullability of columns.
   * @param {ColumnAlteration[]} columnAlterations One or more alterations to
   * apply to columns.
   * @returns {Promise<AlterColumnsResult>} A promise that resolves to an object
   * containing the new version number of the table after altering the columns.
   */
-  abstract alterColumns(columnAlterations: ColumnAlteration[]): Promise<void>;
+  abstract alterColumns(
    columnAlterations: ColumnAlteration[],
  ): Promise<AlterColumnsResult>;
  /**
   * Drop one or more columns from the dataset
   *
@@ -360,8 +391,10 @@ export abstract class Table {
   * @param {string[]} columnNames The names of the columns to drop. These can
   * be nested column references (e.g. "a.b.c") or top-level column names
   * (e.g. "a").
   * @returns {Promise<DropColumnsResult>} A promise that resolves to an object
   * containing the new version number of the table after dropping the columns.
   */
-  abstract dropColumns(columnNames: string[]): Promise<void>;
+  abstract dropColumns(columnNames: string[]): Promise<DropColumnsResult>;
  /** Retrieve the version of the table */
  abstract version(): Promise<number>;
@@ -374,7 +407,7 @@ export abstract class Table {
   *
   * Calling this method will set the table into time-travel mode. If you
   * wish to return to standard mode, call `checkoutLatest`.
-   * @param {number} version The version to checkout
+   * @param {number | string} version The version to checkout, could be version number or tag
   * @example
   * ```typescript
   * import * as lancedb from "@lancedb/lancedb"
@@ -390,7 +423,8 @@ export abstract class Table {
   * console.log(await table.version()); // 2
   * ```
   */
-  abstract checkout(version: number): Promise<void>;
+  abstract checkout(version: number | string): Promise<void>;
  /**
   * Checkout the latest version of the table. _This is an in-place operation._
   *
@@ -404,6 +438,23 @@ export abstract class Table {
   */
  abstract listVersions(): Promise<Version[]>;
  /**
   * Get a tags manager for this table.
   *
   * Tags allow you to label specific versions of a table with a human-readable name.
   * The returned tags manager can be used to list, create, update, or delete tags.
   *
   * @returns {Tags} A tags manager for this table
   * @example
   * ```typescript
   * const tagsManager = await table.tags();
   * await tagsManager.create("v1", 1);
   * const tags = await tagsManager.list();
   * console.log(tags); // { "v1": { version: 1, manifestSize: ... } }
   * ```
   */
  abstract tags(): Promise<Tags>;
  /**
   * Restore the table to the currently checked out version
   *
@@ -463,6 +514,13 @@ export abstract class Table {
   * Use {@link Table.listIndices} to find the names of the indices.
   */
  abstract indexStats(name: string): Promise<IndexStatistics | undefined>;
  /** Returns table and fragment statistics
   *
   * @returns {TableStatistics} The table and fragment statistics
   *
   */
  abstract stats(): Promise<TableStatistics>;
 }
 export class LocalTable extends Table {
@@ -502,12 +560,12 @@ export class LocalTable extends Table {
    return tbl.schema;
  }
-  async add(data: Data, options?: Partial<AddDataOptions>): Promise<void> {
+  async add(data: Data, options?: Partial<AddDataOptions>): Promise<AddResult> {
    const mode = options?.mode ?? "append";
    const schema = await this.schema();
    const buffer = await fromDataToBuffer(data, undefined, schema);
-    await this.inner.add(buffer, mode);
+    return await this.inner.add(buffer, mode);
  }
  async update(
@@ -520,7 +578,7 @@ export class LocalTable extends Table {
          valuesSql: Map<string, string> | Record<string, string>;
        } & Partial<UpdateOptions>),
    options?: Partial<UpdateOptions>,
-  ) {
+  ): Promise<UpdateResult> {
    const isValues =
      "values" in optsOrUpdates && typeof optsOrUpdates.values !== "string";
    const isValuesSql =
@@ -567,15 +625,15 @@ export class LocalTable extends Table {
        columns = Object.entries(optsOrUpdates as Record<string, string>);
        predicate = options?.where;
    }
-    await this.inner.update(predicate, columns);
+    return await this.inner.update(predicate, columns);
  }
  async countRows(filter?: string): Promise<number> {
    return await this.inner.countRows(filter);
  }
-  async delete(predicate: string): Promise<void> {
+  async delete(predicate: string): Promise<DeleteResult> {
-    await this.inner.delete(predicate);
+    return await this.inner.delete(predicate);
  }
  async createIndex(column: string, options?: Partial<IndexOptions>) {
@@ -663,11 +721,15 @@ export class LocalTable extends Table {
  // TODO: Support BatchUDF
-  async addColumns(newColumnTransforms: AddColumnsSql[]): Promise<void> {
+  async addColumns(
-    await this.inner.addColumns(newColumnTransforms);
+    newColumnTransforms: AddColumnsSql[],
  ): Promise<AddColumnsResult> {
    return await this.inner.addColumns(newColumnTransforms);
  }
-  async alterColumns(columnAlterations: ColumnAlteration[]): Promise<void> {
+  async alterColumns(
    columnAlterations: ColumnAlteration[],
  ): Promise<AlterColumnsResult> {
    const processedAlterations = columnAlterations.map((alteration) => {
      if (typeof alteration.dataType === "string") {
        return {
@@ -688,19 +750,22 @@ export class LocalTable extends Table {
      }
    });
-    await this.inner.alterColumns(processedAlterations);
+    return await this.inner.alterColumns(processedAlterations);
  }
-  async dropColumns(columnNames: string[]): Promise<void> {
+  async dropColumns(columnNames: string[]): Promise<DropColumnsResult> {
-    await this.inner.dropColumns(columnNames);
+    return await this.inner.dropColumns(columnNames);
  }
  async version(): Promise<number> {
    return await this.inner.version();
  }
-  async checkout(version: number): Promise<void> {
+  async checkout(version: number | string): Promise<void> {
-    await this.inner.checkout(version);
+    if (typeof version === "string") {
      return this.inner.checkoutTag(version);
    }
    return this.inner.checkout(version);
  }
  async checkoutLatest(): Promise<void> {
@@ -719,6 +784,10 @@ export class LocalTable extends Table {
    await this.inner.restore();
  }
  async tags(): Promise<Tags> {
    return await this.inner.tags();
  }
  async optimize(options?: Partial<OptimizeOptions>): Promise<OptimizeStats> {
    let cleanupOlderThanMs;
    if (
@@ -749,6 +818,11 @@ export class LocalTable extends Table {
    }
    return stats;
  }
  async stats(): Promise<TableStatistics> {
    return await this.inner.stats();
  }
  mergeInsert(on: string | string[]): MergeInsertBuilder {
    on = Array.isArray(on) ? on : [on];
    return new MergeInsertBuilder(this.inner.mergeInsert(on), this.schema());
--- a/nodejs/npm/darwin-arm64/package.json
+++ b/nodejs/npm/darwin-arm64/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-darwin-arm64",
-	"version": "0.19.0-beta.11",
+	"version": "0.20.0-beta.2",
 	"os": ["darwin"],
 	"cpu": ["arm64"],
 	"main": "lancedb.darwin-arm64.node",
--- a/nodejs/npm/darwin-x64/package.json
+++ b/nodejs/npm/darwin-x64/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-darwin-x64",
-	"version": "0.19.0-beta.11",
+	"version": "0.20.0-beta.2",
 	"os": ["darwin"],
 	"cpu": ["x64"],
 	"main": "lancedb.darwin-x64.node",
--- a/nodejs/npm/linux-arm64-gnu/package.json
+++ b/nodejs/npm/linux-arm64-gnu/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-linux-arm64-gnu",
-	"version": "0.19.0-beta.11",
+	"version": "0.20.0-beta.2",
 	"os": ["linux"],
 	"cpu": ["arm64"],
 	"main": "lancedb.linux-arm64-gnu.node",
--- a/nodejs/npm/linux-arm64-musl/package.json
+++ b/nodejs/npm/linux-arm64-musl/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-linux-arm64-musl",
-	"version": "0.19.0-beta.11",
+	"version": "0.20.0-beta.2",
 	"os": ["linux"],
 	"cpu": ["arm64"],
 	"main": "lancedb.linux-arm64-musl.node",
--- a/nodejs/npm/linux-x64-gnu/package.json
+++ b/nodejs/npm/linux-x64-gnu/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-linux-x64-gnu",
-	"version": "0.19.0-beta.11",
+	"version": "0.20.0-beta.2",
 	"os": ["linux"],
 	"cpu": ["x64"],
 	"main": "lancedb.linux-x64-gnu.node",
--- a/nodejs/npm/linux-x64-musl/package.json
+++ b/nodejs/npm/linux-x64-musl/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-linux-x64-musl",
-	"version": "0.19.0-beta.11",
+	"version": "0.20.0-beta.2",
 	"os": ["linux"],
 	"cpu": ["x64"],
 	"main": "lancedb.linux-x64-musl.node",
--- a/nodejs/npm/win32-arm64-msvc/package.json
+++ b/nodejs/npm/win32-arm64-msvc/package.json
@@ -1,6 +1,6 @@
 {
  "name": "@lancedb/lancedb-win32-arm64-msvc",
-  "version": "0.19.0-beta.11",
+  "version": "0.20.0-beta.2",
  "os": [
    "win32"
  ],
--- a/nodejs/npm/win32-x64-msvc/package.json
+++ b/nodejs/npm/win32-x64-msvc/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-win32-x64-msvc",
-	"version": "0.19.0-beta.11",
+	"version": "0.20.0-beta.2",
 	"os": ["win32"],
 	"cpu": ["x64"],
 	"main": "lancedb.win32-x64-msvc.node",
--- a/nodejs/package-lock.json
+++ b/nodejs/package-lock.json
@@ -1,12 +1,12 @@
 {
  "name": "@lancedb/lancedb",
-  "version": "0.19.0-beta.11",
+  "version": "0.20.0-beta.2",
  "lockfileVersion": 3,
  "requires": true,
  "packages": {
    "": {
      "name": "@lancedb/lancedb",
-      "version": "0.19.0-beta.11",
+      "version": "0.20.0-beta.2",
      "cpu": [
        "x64",
        "arm64"
--- a/nodejs/package.json
+++ b/nodejs/package.json
@@ -11,7 +11,7 @@
    "ann"
  ],
  "private": false,
-  "version": "0.19.0-beta.11",
+  "version": "0.20.0-beta.2",
  "main": "dist/index.js",
  "exports": {
    ".": "./dist/index.js",
--- a/nodejs/src/index.rs
+++ b/nodejs/src/index.rs
@@ -125,32 +125,30 @@ impl Index {
        ascii_folding: Option<bool>,
    ) -> Self {
        let mut opts = FtsIndexBuilder::default();
        let mut tokenizer_configs = opts.tokenizer_configs.clone();
        if let Some(with_position) = with_position {
            opts = opts.with_position(with_position);
        }
        if let Some(base_tokenizer) = base_tokenizer {
-            tokenizer_configs = tokenizer_configs.base_tokenizer(base_tokenizer);
+            opts = opts.base_tokenizer(base_tokenizer);
        }
        if let Some(language) = language {
-            tokenizer_configs = tokenizer_configs.language(&language).unwrap();
+            opts = opts.language(&language).unwrap();
        }
        if let Some(max_token_length) = max_token_length {
-            tokenizer_configs = tokenizer_configs.max_token_length(Some(max_token_length as usize));
+            opts = opts.max_token_length(Some(max_token_length as usize));
        }
        if let Some(lower_case) = lower_case {
-            tokenizer_configs = tokenizer_configs.lower_case(lower_case);
+            opts = opts.lower_case(lower_case);
        }
        if let Some(stem) = stem {
-            tokenizer_configs = tokenizer_configs.stem(stem);
+            opts = opts.stem(stem);
        }
        if let Some(remove_stop_words) = remove_stop_words {
-            tokenizer_configs = tokenizer_configs.remove_stop_words(remove_stop_words);
+            opts = opts.remove_stop_words(remove_stop_words);
        }
        if let Some(ascii_folding) = ascii_folding {
-            tokenizer_configs = tokenizer_configs.ascii_folding(ascii_folding);
+            opts = opts.ascii_folding(ascii_folding);
        }
        opts.tokenizer_configs = tokenizer_configs;
        Self {
            inner: Mutex::new(Some(LanceDbIndex::FTS(opts))),
--- a/nodejs/src/merge.rs
+++ b/nodejs/src/merge.rs
@@ -1,11 +1,13 @@
 // SPDX-License-Identifier: Apache-2.0
 // SPDX-FileCopyrightText: Copyright The LanceDB Authors
 use std::time::Duration;
 use lancedb::{arrow::IntoArrow, ipc::ipc_file_to_batches, table::merge::MergeInsertBuilder};
 use napi::bindgen_prelude::*;
 use napi_derive::napi;
-use crate::error::convert_error;
+use crate::{error::convert_error, table::MergeResult};
 #[napi]
 #[derive(Clone)]
@@ -36,8 +38,13 @@ impl NativeMergeInsertBuilder {
        this
    }
    #[napi]
    pub fn set_timeout(&mut self, timeout: u32) {
        self.inner.timeout(Duration::from_millis(timeout as u64));
    }
    #[napi(catch_unwind)]
-    pub async fn execute(&self, buf: Buffer) -> napi::Result<()> {
+    pub async fn execute(&self, buf: Buffer) -> napi::Result<MergeResult> {
        let data = ipc_file_to_batches(buf.to_vec())
            .and_then(IntoArrow::into_arrow)
            .map_err(|e| {
@@ -46,12 +53,13 @@ impl NativeMergeInsertBuilder {
        let this = self.clone();
-        this.inner.execute(data).await.map_err(|e| {
+        let res = this.inner.execute(data).await.map_err(|e| {
            napi::Error::from_reason(format!(
                "Failed to execute merge insert: {}",
                convert_error(&e)
            ))
-        })
+        })?;
        Ok(res.into())
    }
 }
--- a/nodejs/src/table.rs
+++ b/nodejs/src/table.rs
@@ -75,7 +75,7 @@ impl Table {
    }
    #[napi(catch_unwind)]
-    pub async fn add(&self, buf: Buffer, mode: String) -> napi::Result<()> {
+    pub async fn add(&self, buf: Buffer, mode: String) -> napi::Result<AddResult> {
        let batches = ipc_file_to_batches(buf.to_vec())
            .map_err(|e| napi::Error::from_reason(format!("Failed to read IPC file: {}", e)))?;
        let mut op = self.inner_ref()?.add(batches);
@@ -88,7 +88,8 @@ impl Table {
            return Err(napi::Error::from_reason(format!("Invalid mode: {}", mode)));
        };
-        op.execute().await.default_error()
+        let res = op.execute().await.default_error()?;
        Ok(res.into())
    }
    #[napi(catch_unwind)]
@@ -101,8 +102,9 @@ impl Table {
    }
    #[napi(catch_unwind)]
-    pub async fn delete(&self, predicate: String) -> napi::Result<()> {
+    pub async fn delete(&self, predicate: String) -> napi::Result<DeleteResult> {
-        self.inner_ref()?.delete(&predicate).await.default_error()
+        let res = self.inner_ref()?.delete(&predicate).await.default_error()?;
        Ok(res.into())
    }
    #[napi(catch_unwind)]
@@ -157,12 +159,18 @@ impl Table {
            .default_error()
    }
    #[napi(catch_unwind)]
    pub async fn stats(&self) -> Result<TableStatistics> {
        let stats = self.inner_ref()?.stats().await.default_error()?;
        Ok(stats.into())
    }
    #[napi(catch_unwind)]
    pub async fn update(
        &self,
        only_if: Option<String>,
        columns: Vec<(String, String)>,
-    ) -> napi::Result<u64> {
+    ) -> napi::Result<UpdateResult> {
        let mut op = self.inner_ref()?.update();
        if let Some(only_if) = only_if {
            op = op.only_if(only_if);
@@ -170,7 +178,8 @@ impl Table {
        for (column_name, value) in columns {
            op = op.column(column_name, value);
        }
-        op.execute().await.default_error()
+        let res = op.execute().await.default_error()?;
        Ok(res.into())
    }
    #[napi(catch_unwind)]
@@ -184,21 +193,28 @@ impl Table {
    }
    #[napi(catch_unwind)]
-    pub async fn add_columns(&self, transforms: Vec<AddColumnsSql>) -> napi::Result<()> {
+    pub async fn add_columns(
        &self,
        transforms: Vec<AddColumnsSql>,
    ) -> napi::Result<AddColumnsResult> {
        let transforms = transforms
            .into_iter()
            .map(|sql| (sql.name, sql.value_sql))
            .collect::<Vec<_>>();
        let transforms = NewColumnTransform::SqlExpressions(transforms);
-        self.inner_ref()?
+        let res = self
            .inner_ref()?
            .add_columns(transforms, None)
            .await
            .default_error()?;
-        Ok(())
+        Ok(res.into())
    }
    #[napi(catch_unwind)]
-    pub async fn alter_columns(&self, alterations: Vec<ColumnAlteration>) -> napi::Result<()> {
+    pub async fn alter_columns(
        &self,
        alterations: Vec<ColumnAlteration>,
    ) -> napi::Result<AlterColumnsResult> {
        for alteration in &alterations {
            if alteration.rename.is_none()
                && alteration.nullable.is_none()
@@ -215,21 +231,23 @@ impl Table {
            .collect::<std::result::Result<Vec<_>, String>>()
            .map_err(napi::Error::from_reason)?;
-        self.inner_ref()?
+        let res = self
            .inner_ref()?
            .alter_columns(&alterations)
            .await
            .default_error()?;
-        Ok(())
+        Ok(res.into())
    }
    #[napi(catch_unwind)]
-    pub async fn drop_columns(&self, columns: Vec<String>) -> napi::Result<()> {
+    pub async fn drop_columns(&self, columns: Vec<String>) -> napi::Result<DropColumnsResult> {
        let col_refs = columns.iter().map(String::as_str).collect::<Vec<_>>();
-        self.inner_ref()?
+        let res = self
            .inner_ref()?
            .drop_columns(&col_refs)
            .await
            .default_error()?;
-        Ok(())
+        Ok(res.into())
    }
    #[napi(catch_unwind)]
@@ -249,6 +267,14 @@ impl Table {
            .default_error()
    }
    #[napi(catch_unwind)]
    pub async fn checkout_tag(&self, tag: String) -> napi::Result<()> {
        self.inner_ref()?
            .checkout_tag(tag.as_str())
            .await
            .default_error()
    }
    #[napi(catch_unwind)]
    pub async fn checkout_latest(&self) -> napi::Result<()> {
        self.inner_ref()?.checkout_latest().await.default_error()
@@ -281,6 +307,13 @@ impl Table {
        self.inner_ref()?.restore().await.default_error()
    }
    #[napi(catch_unwind)]
    pub async fn tags(&self) -> napi::Result<Tags> {
        Ok(Tags {
            inner: self.inner_ref()?.clone(),
        })
    }
    #[napi(catch_unwind)]
    pub async fn optimize(
        &self,
@@ -540,9 +573,257 @@ impl From<lancedb::index::IndexStatistics> for IndexStatistics {
    }
 }
 #[napi(object)]
 pub struct TableStatistics {
    /// The total number of bytes in the table
    pub total_bytes: i64,
    /// The number of rows in the table
    pub num_rows: i64,
    /// The number of indices in the table
    pub num_indices: i64,
    /// Statistics on table fragments
    pub fragment_stats: FragmentStatistics,
 }
 #[napi(object)]
 pub struct FragmentStatistics {
    /// The number of fragments in the table
    pub num_fragments: i64,
    /// The number of uncompacted fragments in the table
    pub num_small_fragments: i64,
    /// Statistics on the number of rows in the table fragments
    pub lengths: FragmentSummaryStats,
 }
 #[napi(object)]
 pub struct FragmentSummaryStats {
    /// The number of rows in the fragment with the fewest rows
    pub min: i64,
    /// The number of rows in the fragment with the most rows
    pub max: i64,
    /// The mean number of rows in the fragments
    pub mean: i64,
    /// The 25th percentile of number of rows in the fragments
    pub p25: i64,
    /// The 50th percentile of number of rows in the fragments
    pub p50: i64,
    /// The 75th percentile of number of rows in the fragments
    pub p75: i64,
    /// The 99th percentile of number of rows in the fragments
    pub p99: i64,
 }
 impl From<lancedb::table::TableStatistics> for TableStatistics {
    fn from(v: lancedb::table::TableStatistics) -> Self {
        Self {
            total_bytes: v.total_bytes as i64,
            num_rows: v.num_rows as i64,
            num_indices: v.num_indices as i64,
            fragment_stats: FragmentStatistics {
                num_fragments: v.fragment_stats.num_fragments as i64,
                num_small_fragments: v.fragment_stats.num_small_fragments as i64,
                lengths: FragmentSummaryStats {
                    min: v.fragment_stats.lengths.min as i64,
                    max: v.fragment_stats.lengths.max as i64,
                    mean: v.fragment_stats.lengths.mean as i64,
                    p25: v.fragment_stats.lengths.p25 as i64,
                    p50: v.fragment_stats.lengths.p50 as i64,
                    p75: v.fragment_stats.lengths.p75 as i64,
                    p99: v.fragment_stats.lengths.p99 as i64,
                },
            },
        }
    }
 }
 #[napi(object)]
 pub struct Version {
    pub version: i64,
    pub timestamp: i64,
    pub metadata: HashMap<String, String>,
 }
 #[napi(object)]
 pub struct UpdateResult {
    pub rows_updated: i64,
    pub version: i64,
 }
 impl From<lancedb::table::UpdateResult> for UpdateResult {
    fn from(value: lancedb::table::UpdateResult) -> Self {
        Self {
            rows_updated: value.rows_updated as i64,
            version: value.version as i64,
        }
    }
 }
 #[napi(object)]
 pub struct AddResult {
    pub version: i64,
 }
 impl From<lancedb::table::AddResult> for AddResult {
    fn from(value: lancedb::table::AddResult) -> Self {
        Self {
            version: value.version as i64,
        }
    }
 }
 #[napi(object)]
 pub struct DeleteResult {
    pub version: i64,
 }
 impl From<lancedb::table::DeleteResult> for DeleteResult {
    fn from(value: lancedb::table::DeleteResult) -> Self {
        Self {
            version: value.version as i64,
        }
    }
 }
 #[napi(object)]
 pub struct MergeResult {
    pub version: i64,
    pub num_inserted_rows: i64,
    pub num_updated_rows: i64,
    pub num_deleted_rows: i64,
 }
 impl From<lancedb::table::MergeResult> for MergeResult {
    fn from(value: lancedb::table::MergeResult) -> Self {
        Self {
            version: value.version as i64,
            num_inserted_rows: value.num_inserted_rows as i64,
            num_updated_rows: value.num_updated_rows as i64,
            num_deleted_rows: value.num_deleted_rows as i64,
        }
    }
 }
 #[napi(object)]
 pub struct AddColumnsResult {
    pub version: i64,
 }
 impl From<lancedb::table::AddColumnsResult> for AddColumnsResult {
    fn from(value: lancedb::table::AddColumnsResult) -> Self {
        Self {
            version: value.version as i64,
        }
    }
 }
 #[napi(object)]
 pub struct AlterColumnsResult {
    pub version: i64,
 }
 impl From<lancedb::table::AlterColumnsResult> for AlterColumnsResult {
    fn from(value: lancedb::table::AlterColumnsResult) -> Self {
        Self {
            version: value.version as i64,
        }
    }
 }
 #[napi(object)]
 pub struct DropColumnsResult {
    pub version: i64,
 }
 impl From<lancedb::table::DropColumnsResult> for DropColumnsResult {
    fn from(value: lancedb::table::DropColumnsResult) -> Self {
        Self {
            version: value.version as i64,
        }
    }
 }
 #[napi]
 pub struct TagContents {
    pub version: i64,
    pub manifest_size: i64,
 }
 #[napi]
 pub struct Tags {
    inner: LanceDbTable,
 }
 #[napi]
 impl Tags {
    #[napi]
    pub async fn list(&self) -> napi::Result<HashMap<String, TagContents>> {
        let rust_tags = self.inner.tags().await.default_error()?;
        let tag_list = rust_tags.as_ref().list().await.default_error()?;
        let tag_contents = tag_list
            .into_iter()
            .map(|(k, v)| {
                (
                    k,
                    TagContents {
                        version: v.version as i64,
                        manifest_size: v.manifest_size as i64,
                    },
                )
            })
            .collect();
        Ok(tag_contents)
    }
    #[napi]
    pub async fn get_version(&self, tag: String) -> napi::Result<i64> {
        let rust_tags = self.inner.tags().await.default_error()?;
        rust_tags
            .as_ref()
            .get_version(tag.as_str())
            .await
            .map(|v| v as i64)
            .default_error()
    }
    #[napi]
    pub async unsafe fn create(&mut self, tag: String, version: i64) -> napi::Result<()> {
        let mut rust_tags = self.inner.tags().await.default_error()?;
        rust_tags
            .as_mut()
            .create(tag.as_str(), version as u64)
            .await
            .default_error()
    }
    #[napi]
    pub async unsafe fn delete(&mut self, tag: String) -> napi::Result<()> {
        let mut rust_tags = self.inner.tags().await.default_error()?;
        rust_tags
            .as_mut()
            .delete(tag.as_str())
            .await
            .default_error()
    }
    #[napi]
    pub async unsafe fn update(&mut self, tag: String, version: i64) -> napi::Result<()> {
        let mut rust_tags = self.inner.tags().await.default_error()?;
        rust_tags
            .as_mut()
            .update(tag.as_str(), version as u64)
            .await
            .default_error()
    }
 }
--- a/python/.bumpversion.toml
+++ b/python/.bumpversion.toml
@@ -1,5 +1,5 @@
 [tool.bumpversion]
-current_version = "0.22.0"
+current_version = "0.23.0"
 parse = """(?x)
    (?P<major>0|[1-9]\\d*)\\.
    (?P<minor>0|[1-9]\\d*)\\.
--- a/python/Cargo.toml
+++ b/python/Cargo.toml
@@ -1,6 +1,6 @@
 [package]
 name = "lancedb-python"
-version = "0.22.0"
+version = "0.23.0"
 edition.workspace = true
 description = "Python bindings for LanceDB"
 license.workspace = true
@@ -14,11 +14,11 @@ name = "_lancedb"
 crate-type = ["cdylib"]
 [dependencies]
-arrow = { version = "54.1", features = ["pyarrow"] }
+arrow = { version = "55.1", features = ["pyarrow"] }
 lancedb = { path = "../rust/lancedb", default-features = false }
 env_logger.workspace = true
-pyo3 = { version = "0.23", features = ["extension-module", "abi3-py39"] }
+pyo3 = { version = "0.24", features = ["extension-module", "abi3-py39"] }
-pyo3-async-runtimes = { version = "0.23", features = [
+pyo3-async-runtimes = { version = "0.24", features = [
    "attributes",
    "tokio-runtime",
 ] }
@@ -27,7 +27,7 @@ futures.workspace = true
 tokio = { version = "1.40", features = ["sync"] }
 [build-dependencies]
-pyo3-build-config = { version = "0.23", features = [
+pyo3-build-config = { version = "0.24", features = [
    "extension-module",
    "abi3-py39",
 ] }
--- a/python/pyproject.toml
+++ b/python/pyproject.toml
@@ -7,7 +7,7 @@ dependencies = [
    "numpy",
    "overrides>=0.7",
    "packaging",
-    "pyarrow>=14",
+    "pyarrow>=16",
    "pydantic>=1.10",
    "tqdm>=4.27.0",
 ]
@@ -60,6 +60,7 @@ tests = [
    "pyarrow-stubs",
    "pylance>=0.25",
    "requests",
    "datafusion",
 ]
 dev = [
    "ruff",
--- a/python/python/lancedb/_lancedb.pyi
+++ b/python/python/lancedb/_lancedb.pyi
@@ -1,5 +1,5 @@
 from datetime import timedelta
-from typing import Dict, List, Optional, Tuple, Any, Union, Literal
+from typing import Dict, List, Optional, Tuple, Any, TypedDict, Union, Literal
 import pyarrow as pa
@@ -36,8 +36,10 @@ class Table:
    async def schema(self) -> pa.Schema: ...
    async def add(
        self, data: pa.RecordBatchReader, mode: Literal["append", "overwrite"]
-    ) -> None: ...
+    ) -> AddResult: ...
-    async def update(self, updates: Dict[str, str], where: Optional[str]) -> None: ...
+    async def update(
        self, updates: Dict[str, str], where: Optional[str]
    ) -> UpdateResult: ...
    async def count_rows(self, filter: Optional[str]) -> int: ...
    async def create_index(
        self,
@@ -47,23 +49,34 @@ class Table:
    ): ...
    async def list_versions(self) -> List[Dict[str, Any]]: ...
    async def version(self) -> int: ...
-    async def checkout(self, version: int): ...
+    async def checkout(self, version: Union[int, str]): ...
    async def checkout_latest(self): ...
-    async def restore(self, version: Optional[int] = None): ...
+    async def restore(self, version: Optional[Union[int, str]] = None): ...
    async def list_indices(self) -> list[IndexConfig]: ...
-    async def delete(self, filter: str): ...
+    async def delete(self, filter: str) -> DeleteResult: ...
-    async def add_columns(self, columns: list[tuple[str, str]]) -> None: ...
+    async def add_columns(self, columns: list[tuple[str, str]]) -> AddColumnsResult: ...
-    async def add_columns_with_schema(self, schema: pa.Schema) -> None: ...
+    async def add_columns_with_schema(self, schema: pa.Schema) -> AddColumnsResult: ...
-    async def alter_columns(self, columns: list[dict[str, Any]]) -> None: ...
+    async def alter_columns(
        self, columns: list[dict[str, Any]]
    ) -> AlterColumnsResult: ...
    async def optimize(
        self,
        *,
        cleanup_since_ms: Optional[int] = None,
        delete_unverified: Optional[bool] = None,
    ) -> OptimizeStats: ...
    @property
    def tags(self) -> Tags: ...
    def query(self) -> Query: ...
    def vector_search(self) -> VectorQuery: ...
 class Tags:
    async def list(self) -> Dict[str, Tag]: ...
    async def get_version(self, tag: str) -> int: ...
    async def create(self, tag: str, version: int): ...
    async def delete(self, tag: str): ...
    async def update(self, tag: str, version: int): ...
 class IndexConfig:
    index_type: str
    columns: List[str]
@@ -195,3 +208,32 @@ class RemovalStats:
 class OptimizeStats:
    compaction: CompactionStats
    prune: RemovalStats
 class Tag(TypedDict):
    version: int
    manifest_size: int
 class AddResult:
    version: int
 class DeleteResult:
    version: int
 class UpdateResult:
    rows_updated: int
    version: int
 class MergeResult:
    version: int
    num_updated_rows: int
    num_inserted_rows: int
    num_deleted_rows: int
 class AddColumnsResult:
    version: int
 class AlterColumnsResult:
    version: int
 class DropColumnsResult:
    version: int
--- a/python/python/lancedb/index.py
+++ b/python/python/lancedb/index.py
@@ -102,7 +102,7 @@ class FTS:
    Attributes
    ----------
-    with_position : bool, default True
+    with_position : bool, default False
        Whether to store the position of the token in the document. Setting this
        to False can reduce the size of the index and improve indexing speed,
        but it will disable support for phrase queries.
@@ -118,25 +118,25 @@ class FTS:
        ignored.
    lower_case : bool, default True
        Whether to convert the token to lower case. This makes queries case-insensitive.
-    stem : bool, default False
+    stem : bool, default True
        Whether to stem the token. Stemming reduces words to their root form.
        For example, in English "running" and "runs" would both be reduced to "run".
-    remove_stop_words : bool, default False
+    remove_stop_words : bool, default True
        Whether to remove stop words. Stop words are common words that are often
        removed from text before indexing. For example, in English "the" and "and".
-    ascii_folding : bool, default False
+    ascii_folding : bool, default True
        Whether to fold ASCII characters. This converts accented characters to
        their ASCII equivalent. For example, "café" would be converted to "cafe".
    """
-    with_position: bool = True
+    with_position: bool = False
    base_tokenizer: Literal["simple", "raw", "whitespace"] = "simple"
    language: str = "English"
    max_token_length: Optional[int] = 40
    lower_case: bool = True
-    stem: bool = False
+    stem: bool = True
-    remove_stop_words: bool = False
+    remove_stop_words: bool = True
-    ascii_folding: bool = False
+    ascii_folding: bool = True
@dataclass
--- a/python/python/lancedb/merge.py
+++ b/python/python/lancedb/merge.py
@@ -4,10 +4,14 @@
 from __future__ import annotations
 from datetime import timedelta
 from typing import TYPE_CHECKING, List, Optional
 if TYPE_CHECKING:
    from .common import DATA
    from ._lancedb import (
        MergeInsertResult,
    )
 class LanceMergeInsertBuilder(object):
@@ -28,6 +32,7 @@ class LanceMergeInsertBuilder(object):
        self._when_not_matched_insert_all = False
        self._when_not_matched_by_source_delete = False
        self._when_not_matched_by_source_condition = None
        self._timeout = None
    def when_matched_update_all(
        self, *, where: Optional[str] = None
@@ -78,7 +83,8 @@ class LanceMergeInsertBuilder(object):
        new_data: DATA,
        on_bad_vectors: str = "error",
        fill_value: float = 0.0,
-    ):
+        timeout: Optional[timedelta] = None,
    ) -> MergeInsertResult:
        """
        Executes the merge insert operation
@@ -95,5 +101,24 @@ class LanceMergeInsertBuilder(object):
            One of "error", "drop", "fill".
        fill_value: float, default 0.
            The value to use when filling vectors. Only used if on_bad_vectors="fill".
        timeout: Optional[timedelta], default None
            Maximum time to run the operation before cancelling it.
            By default, there is a 30-second timeout that is only enforced after the
            first attempt. This is to prevent spending too long retrying to resolve
            conflicts. For example, if a write attempt takes 20 seconds and fails,
            the second attempt will be cancelled after 10 seconds, hitting the
            30-second timeout. However, a write that takes one hour and succeeds on the
            first attempt will not be cancelled.
            When this is set, the timeout is enforced on all attempts, including
            the first.
        Returns
        -------
        MergeInsertResult
            version: the new version number of the table after doing merge insert.
        """
        if timeout is not None:
            self._timeout = timeout
        return self._table._do_merge(self, new_data, on_bad_vectors, fill_value)
--- a/python/python/lancedb/pydantic.py
+++ b/python/python/lancedb/pydantic.py
@@ -415,6 +415,7 @@ class LanceModel(pydantic.BaseModel):
    >>> table.add([
    ...     TestModel(name="test", vector=[1.0, 2.0])
    ... ])
    AddResult(version=2)
    >>> table.search([0., 0.]).limit(1).to_pydantic(TestModel)
    [TestModel(name='test', vector=FixedSizeList(dim=2))]
    """
--- a/python/python/lancedb/query.py
+++ b/python/python/lancedb/query.py
@@ -1636,51 +1636,7 @@ class LanceHybridQueryBuilder(LanceQueryBuilder):
        raise NotImplementedError("to_query_object not yet supported on a hybrid query")
    def to_arrow(self, *, timeout: Optional[timedelta] = None) -> pa.Table:
-        vector_query, fts_query = self._validate_query(
+        self._create_query_builders()
            self._query, self._vector, self._text
        )
        self._fts_query = LanceFtsQueryBuilder(
            self._table, fts_query, fts_columns=self._fts_columns
        )
        vector_query = self._query_to_vector(
            self._table, vector_query, self._vector_column
        )
        self._vector_query = LanceVectorQueryBuilder(
            self._table, vector_query, self._vector_column
        )
        if self._limit:
            self._vector_query.limit(self._limit)
            self._fts_query.limit(self._limit)
        if self._columns:
            self._vector_query.select(self._columns)
            self._fts_query.select(self._columns)
        if self._where:
            self._vector_query.where(self._where, self._postfilter)
            self._fts_query.where(self._where, self._postfilter)
        if self._with_row_id:
            self._vector_query.with_row_id(True)
            self._fts_query.with_row_id(True)
        if self._phrase_query:
            self._fts_query.phrase_query(True)
        if self._distance_type:
            self._vector_query.metric(self._distance_type)
        if self._nprobes:
            self._vector_query.nprobes(self._nprobes)
        if self._refine_factor:
            self._vector_query.refine_factor(self._refine_factor)
        if self._ef:
            self._vector_query.ef(self._ef)
        if self._bypass_vector_index:
            self._vector_query.bypass_vector_index()
        if self._lower_bound or self._upper_bound:
            self._vector_query.distance_range(
                lower_bound=self._lower_bound, upper_bound=self._upper_bound
            )
        if self._reranker is None:
            self._reranker = RRFReranker()
        with ThreadPoolExecutor() as executor:
            fts_future = executor.submit(
                self._fts_query.with_row_id(True).to_arrow, timeout=timeout
@@ -2003,6 +1959,112 @@ class LanceHybridQueryBuilder(LanceQueryBuilder):
        self._bypass_vector_index = True
        return self
    def explain_plan(self, verbose: Optional[bool] = False) -> str:
        """Return the execution plan for this query.
        Examples
        --------
        >>> import lancedb
        >>> db = lancedb.connect("./.lancedb")
        >>> table = db.create_table("my_table", [{"vector": [99.0, 99]}])
        >>> query = [100, 100]
        >>> plan = table.search(query).explain_plan(True)
        >>> print(plan) # doctest: +ELLIPSIS, +NORMALIZE_WHITESPACE
        ProjectionExec: expr=[vector@0 as vector, _distance@2 as _distance]
        GlobalLimitExec: skip=0, fetch=10
          FilterExec: _distance@2 IS NOT NULL
            SortExec: TopK(fetch=10), expr=[_distance@2 ASC NULLS LAST], preserve_partitioning=[false]
              KNNVectorDistance: metric=l2
                LanceScan: uri=..., projection=[vector], row_id=true, row_addr=false, ordered=false
        Parameters
        ----------
        verbose : bool, default False
            Use a verbose output format.
        Returns
        -------
        plan : str
        """  # noqa: E501
        self._create_query_builders()
        results = ["Vector Search Plan:"]
        results.append(
            self._table._explain_plan(
                self._vector_query.to_query_object(), verbose=verbose
            )
        )
        results.append("FTS Search Plan:")
        results.append(
            self._table._explain_plan(
                self._fts_query.to_query_object(), verbose=verbose
            )
        )
        return "\n".join(results)
    def analyze_plan(self):
        """Execute the query and display with runtime metrics.
        Returns
        -------
        plan : str
        """
        self._create_query_builders()
        results = ["Vector Search Plan:"]
        results.append(self._table._analyze_plan(self._vector_query.to_query_object()))
        results.append("FTS Search Plan:")
        results.append(self._table._analyze_plan(self._fts_query.to_query_object()))
        return "\n".join(results)
    def _create_query_builders(self):
        """Set up and configure the vector and FTS query builders."""
        vector_query, fts_query = self._validate_query(
            self._query, self._vector, self._text
        )
        self._fts_query = LanceFtsQueryBuilder(
            self._table, fts_query, fts_columns=self._fts_columns
        )
        vector_query = self._query_to_vector(
            self._table, vector_query, self._vector_column
        )
        self._vector_query = LanceVectorQueryBuilder(
            self._table, vector_query, self._vector_column
        )
        # Apply common configurations
        if self._limit:
            self._vector_query.limit(self._limit)
            self._fts_query.limit(self._limit)
        if self._columns:
            self._vector_query.select(self._columns)
            self._fts_query.select(self._columns)
        if self._where:
            self._vector_query.where(self._where, self._postfilter)
            self._fts_query.where(self._where, self._postfilter)
        if self._with_row_id:
            self._vector_query.with_row_id(True)
            self._fts_query.with_row_id(True)
        if self._phrase_query:
            self._fts_query.phrase_query(True)
        if self._distance_type:
            self._vector_query.metric(self._distance_type)
        if self._nprobes:
            self._vector_query.nprobes(self._nprobes)
        if self._refine_factor:
            self._vector_query.refine_factor(self._refine_factor)
        if self._ef:
            self._vector_query.ef(self._ef)
        if self._bypass_vector_index:
            self._vector_query.bypass_vector_index()
        if self._lower_bound or self._upper_bound:
            self._vector_query.distance_range(
                lower_bound=self._lower_bound, upper_bound=self._upper_bound
            )
        if self._reranker is None:
            self._reranker = RRFReranker()
 class AsyncQueryBase(object):
    def __init__(self, inner: Union[LanceQuery, LanceVectorQuery]):
--- a/python/python/lancedb/remote/table.py
+++ b/python/python/lancedb/remote/table.py
@@ -7,7 +7,16 @@ from functools import cached_property
 from typing import Dict, Iterable, List, Optional, Union, Literal
 import warnings
-from lancedb._lancedb import IndexConfig
+from lancedb._lancedb import (
    AddColumnsResult,
    AddResult,
    AlterColumnsResult,
    DeleteResult,
    DropColumnsResult,
    IndexConfig,
    MergeResult,
    UpdateResult,
 )
 from lancedb.embeddings.base import EmbeddingFunctionConfig
 from lancedb.index import FTS, BTree, Bitmap, HnswPq, HnswSq, IvfFlat, IvfPq, LabelList
 from lancedb.remote.db import LOOP
@@ -18,7 +27,7 @@ from lancedb.merge import LanceMergeInsertBuilder
 from lancedb.embeddings import EmbeddingFunctionRegistry
 from ..query import LanceVectorQueryBuilder, LanceQueryBuilder
-from ..table import AsyncTable, IndexStatistics, Query, Table
+from ..table import AsyncTable, IndexStatistics, Query, Table, Tags
 class RemoteTable(Table):
@@ -38,9 +47,6 @@ class RemoteTable(Table):
    def __repr__(self) -> str:
        return f"RemoteTable({self.db_name}.{self.name})"
    def __len__(self) -> int:
        self.count_rows(None)
    @property
    def schema(self) -> pa.Schema:
        """The [Arrow Schema](https://arrow.apache.org/docs/python/api/datatypes.html#)
@@ -54,6 +60,10 @@ class RemoteTable(Table):
        """Get the current version of the table"""
        return LOOP.run(self._table.version())
    @property
    def tags(self) -> Tags:
        return Tags(self._table)
    @cached_property
    def embedding_functions(self) -> Dict[str, EmbeddingFunctionConfig]:
        """
@@ -81,13 +91,13 @@ class RemoteTable(Table):
        """to_pandas() is not yet supported on LanceDB cloud."""
        return NotImplementedError("to_pandas() is not yet supported on LanceDB cloud.")
-    def checkout(self, version: int):
+    def checkout(self, version: Union[int, str]):
        return LOOP.run(self._table.checkout(version))
    def checkout_latest(self):
        return LOOP.run(self._table.checkout_latest())
-    def restore(self, version: Optional[int] = None):
+    def restore(self, version: Optional[Union[int, str]] = None):
        return LOOP.run(self._table.restore(version))
    def list_indices(self) -> Iterable[IndexConfig]:
@@ -139,15 +149,15 @@ class RemoteTable(Table):
        *,
        replace: bool = False,
        wait_timeout: timedelta = None,
-        with_position: bool = True,
+        with_position: bool = False,
        # tokenizer configs:
        base_tokenizer: str = "simple",
        language: str = "English",
        max_token_length: Optional[int] = 40,
        lower_case: bool = True,
-        stem: bool = False,
+        stem: bool = True,
-        remove_stop_words: bool = False,
+        remove_stop_words: bool = True,
-        ascii_folding: bool = False,
+        ascii_folding: bool = True,
    ):
        config = FTS(
            with_position=with_position,
@@ -259,7 +269,7 @@ class RemoteTable(Table):
        mode: str = "append",
        on_bad_vectors: str = "error",
        fill_value: float = 0.0,
-    ) -> int:
+    ) -> AddResult:
        """Add more data to the [Table](Table). It has the same API signature as
        the OSS version.
@@ -282,8 +292,12 @@ class RemoteTable(Table):
        fill_value: float, default 0.
            The value to use when filling vectors. Only used if on_bad_vectors="fill".
        Returns
        -------
        AddResult
            An object containing the new version number of the table after adding data.
        """
-        LOOP.run(
+        return LOOP.run(
            self._table.add(
                data, mode=mode, on_bad_vectors=on_bad_vectors, fill_value=fill_value
            )
@@ -409,10 +423,12 @@ class RemoteTable(Table):
        new_data: DATA,
        on_bad_vectors: str,
        fill_value: float,
-    ):
+    ) -> MergeResult:
-        LOOP.run(self._table._do_merge(merge, new_data, on_bad_vectors, fill_value))
+        return LOOP.run(
            self._table._do_merge(merge, new_data, on_bad_vectors, fill_value)
        )
-    def delete(self, predicate: str):
+    def delete(self, predicate: str) -> DeleteResult:
        """Delete rows from the table.
        This can be used to delete a single row, many rows, all rows, or
@@ -427,6 +443,11 @@ class RemoteTable(Table):
            The filter must not be empty, or it will error.
        Returns
        -------
        DeleteResult
            An object containing the new version number of the table after deletion.
        Examples
        --------
        >>> import lancedb
@@ -459,7 +480,7 @@ class RemoteTable(Table):
           x      vector  _distance # doctest: +SKIP
        0  2  [3.0, 4.0]       85.0 # doctest: +SKIP
        """
-        LOOP.run(self._table.delete(predicate))
+        return LOOP.run(self._table.delete(predicate))
    def update(
        self,
@@ -467,7 +488,7 @@ class RemoteTable(Table):
        values: Optional[dict] = None,
        *,
        values_sql: Optional[Dict[str, str]] = None,
-    ):
+    ) -> UpdateResult:
        """
        This can be used to update zero to all rows depending on how many
        rows match the where clause.
@@ -485,6 +506,12 @@ class RemoteTable(Table):
            reference existing columns. For example, {"x": "x + 1"} will increment
            the x column by 1.
        Returns
        -------
        UpdateResult
            - rows_updated: The number of rows that were updated
            - version: The new version number of the table after the update
        Examples
        --------
        >>> import lancedb
@@ -509,7 +536,7 @@ class RemoteTable(Table):
        2  2  [10.0, 10.0] # doctest: +SKIP
        """
-        LOOP.run(
+        return LOOP.run(
            self._table.update(where=where, updates=values, updates_sql=values_sql)
        )
@@ -557,13 +584,15 @@ class RemoteTable(Table):
    def count_rows(self, filter: Optional[str] = None) -> int:
        return LOOP.run(self._table.count_rows(filter))
-    def add_columns(self, transforms: Dict[str, str]):
+    def add_columns(self, transforms: Dict[str, str]) -> AddColumnsResult:
        return LOOP.run(self._table.add_columns(transforms))
-    def alter_columns(self, *alterations: Iterable[Dict[str, str]]):
+    def alter_columns(
        self, *alterations: Iterable[Dict[str, str]]
    ) -> AlterColumnsResult:
        return LOOP.run(self._table.alter_columns(*alterations))
-    def drop_columns(self, columns: Iterable[str]):
+    def drop_columns(self, columns: Iterable[str]) -> DropColumnsResult:
        return LOOP.run(self._table.drop_columns(columns))
    def drop_index(self, index_name: str):
@@ -574,6 +603,9 @@ class RemoteTable(Table):
    ):
        return LOOP.run(self._table.wait_for_index(index_names, timeout))
    def stats(self):
        return LOOP.run(self._table.stats())
    def uses_v2_manifest_paths(self) -> bool:
        raise NotImplementedError(
            "uses_v2_manifest_paths() is not supported on the LanceDB Cloud"
--- a/python/python/lancedb/table.py
+++ b/python/python/lancedb/table.py
@@ -77,6 +77,14 @@ if TYPE_CHECKING:
        OptimizeStats,
        CleanupStats,
        CompactionStats,
        Tag,
        AddColumnsResult,
        AddResult,
        AlterColumnsResult,
        DeleteResult,
        DropColumnsResult,
        MergeResult,
        UpdateResult,
    )
    from .db import LanceDBConnection
    from .index import IndexConfig
@@ -549,6 +557,7 @@ class Table(ABC):
    Can append new data with [Table.add()][lancedb.table.Table.add].
    >>> table.add([{"vector": [0.5, 1.3], "b": 4}])
    AddResult(version=2)
    Can query the table with [Table.search][lancedb.table.Table.search].
@@ -582,6 +591,39 @@ class Table(ABC):
        """
        raise NotImplementedError
    @property
    @abstractmethod
    def tags(self) -> Tags:
        """Tag management for the table.
        Similar to Git, tags are a way to add metadata to a specific version of the
        table.
        .. warning::
            Tagged versions are exempted from the :py:meth:`cleanup_old_versions()`
            process.
            To remove a version that has been tagged, you must first
            :py:meth:`~Tags.delete` the associated tag.
        Examples
        --------
        .. code-block:: python
            table = db.open_table("my_table")
            table.tags.create("v2-prod-20250203", 10)
            tags = table.tags.list()
        """
        raise NotImplementedError
    def __len__(self) -> int:
        """The number of rows in this Table"""
        return self.count_rows(None)
    @property
    @abstractmethod
    def embedding_functions(self) -> Dict[str, EmbeddingFunctionConfig]:
@@ -709,6 +751,13 @@ class Table(ABC):
        """
        raise NotImplementedError
    @abstractmethod
    def stats(self) -> TableStatistics:
        """
        Retrieve table and fragment statistics.
        """
        raise NotImplementedError
    @abstractmethod
    def create_scalar_index(
        self,
@@ -780,15 +829,15 @@ class Table(ABC):
        writer_heap_size: Optional[int] = 1024 * 1024 * 1024,
        use_tantivy: bool = True,
        tokenizer_name: Optional[str] = None,
-        with_position: bool = True,
+        with_position: bool = False,
        # tokenizer configs:
        base_tokenizer: BaseTokenizerType = "simple",
        language: str = "English",
        max_token_length: Optional[int] = 40,
        lower_case: bool = True,
-        stem: bool = False,
+        stem: bool = True,
-        remove_stop_words: bool = False,
+        remove_stop_words: bool = True,
-        ascii_folding: bool = False,
+        ascii_folding: bool = True,
        wait_timeout: Optional[timedelta] = None,
    ):
        """Create a full-text search index on the table.
@@ -818,7 +867,7 @@ class Table(ABC):
        use_tantivy: bool, default True
            If True, use the legacy full-text search implementation based on tantivy.
            If False, use the new full-text search implementation based on lance-index.
-        with_position: bool, default True
+        with_position: bool, default False
            Only available with use_tantivy=False
            If False, do not store the positions of the terms in the text.
            This can reduce the size of the index and improve indexing speed.
@@ -836,13 +885,13 @@ class Table(ABC):
        lower_case : bool, default True
            Whether to convert the token to lower case. This makes queries
            case-insensitive.
-        stem : bool, default False
+        stem : bool, default True
            Whether to stem the token. Stemming reduces words to their root form.
            For example, in English "running" and "runs" would both be reduced to "run".
-        remove_stop_words : bool, default False
+        remove_stop_words : bool, default True
            Whether to remove stop words. Stop words are common words that are often
            removed from text before indexing. For example, in English "the" and "and".
-        ascii_folding : bool, default False
+        ascii_folding : bool, default True
            Whether to fold ASCII characters. This converts accented characters to
            their ASCII equivalent. For example, "café" would be converted to "cafe".
        wait_timeout: timedelta, optional
@@ -857,7 +906,7 @@ class Table(ABC):
        mode: AddMode = "append",
        on_bad_vectors: OnBadVectorsType = "error",
        fill_value: float = 0.0,
-    ):
+    ) -> AddResult:
        """Add more data to the [Table](Table).
        Parameters
@@ -879,6 +928,10 @@ class Table(ABC):
        fill_value: float, default 0.
            The value to use when filling vectors. Only used if on_bad_vectors="fill".
        Returns
        -------
        AddResult
            An object containing the new version number of the table after adding data.
        """
        raise NotImplementedError
@@ -925,10 +978,12 @@ class Table(ABC):
        >>> table = db.create_table("my_table", data)
        >>> new_data = pa.table({"a": [2, 3, 4], "b": ["x", "y", "z"]})
        >>> # Perform a "upsert" operation
-        >>> table.merge_insert("a")             \\
+        >>> res = table.merge_insert("a")     \\
        ...      .when_matched_update_all()     \\
        ...      .when_not_matched_insert_all() \\
        ...      .execute(new_data)
        >>> res
        MergeResult(version=2, num_updated_rows=2, num_inserted_rows=1, num_deleted_rows=0)
        >>> # The order of new rows is non-deterministic since we use
        >>> # a hash-join as part of this operation and so we sort here
        >>> table.to_arrow().sort_by("a").to_pandas()
@@ -937,7 +992,7 @@ class Table(ABC):
        1  2  x
        2  3  y
        3  4  z
-        """
+        """  # noqa: E501
        on = [on] if isinstance(on, str) else list(iter(on))
        return LanceMergeInsertBuilder(self, on)
@@ -1052,10 +1107,10 @@ class Table(ABC):
        new_data: DATA,
        on_bad_vectors: OnBadVectorsType,
        fill_value: float,
-    ): ...
+    ) -> MergeResult: ...
    @abstractmethod
-    def delete(self, where: str):
+    def delete(self, where: str) -> DeleteResult:
        """Delete rows from the table.
        This can be used to delete a single row, many rows, all rows, or
@@ -1070,6 +1125,11 @@ class Table(ABC):
            The filter must not be empty, or it will error.
        Returns
        -------
        DeleteResult
            An object containing the new version number of the table after deletion.
        Examples
        --------
        >>> import lancedb
@@ -1086,6 +1146,7 @@ class Table(ABC):
        1  2  [3.0, 4.0]
        2  3  [5.0, 6.0]
        >>> table.delete("x = 2")
        DeleteResult(version=2)
        >>> table.to_pandas()
           x      vector
        0  1  [1.0, 2.0]
@@ -1099,6 +1160,7 @@ class Table(ABC):
        >>> to_remove
        '1, 5'
        >>> table.delete(f"x IN ({to_remove})")
        DeleteResult(version=3)
        >>> table.to_pandas()
           x      vector
        0  3  [5.0, 6.0]
@@ -1112,7 +1174,7 @@ class Table(ABC):
        values: Optional[dict] = None,
        *,
        values_sql: Optional[Dict[str, str]] = None,
-    ):
+    ) -> UpdateResult:
        """
        This can be used to update zero to all rows depending on how many
        rows match the where clause. If no where clause is provided, then
@@ -1134,6 +1196,12 @@ class Table(ABC):
            reference existing columns. For example, {"x": "x + 1"} will increment
            the x column by 1.
        Returns
        -------
        UpdateResult
            - rows_updated: The number of rows that were updated
            - version: The new version number of the table after the update
        Examples
        --------
        >>> import lancedb
@@ -1147,12 +1215,14 @@ class Table(ABC):
        1  2  [3.0, 4.0]
        2  3  [5.0, 6.0]
        >>> table.update(where="x = 2", values={"vector": [10.0, 10]})
        UpdateResult(rows_updated=1, version=2)
        >>> table.to_pandas()
           x        vector
        0  1    [1.0, 2.0]
        1  3    [5.0, 6.0]
        2  2  [10.0, 10.0]
        >>> table.update(values_sql={"x": "x + 1"})
        UpdateResult(rows_updated=3, version=3)
        >>> table.to_pandas()
           x        vector
        0  2    [1.0, 2.0]
@@ -1315,6 +1385,11 @@ class Table(ABC):
            Alternatively, a pyarrow Field or Schema can be provided to add
            new columns with the specified data types. The new columns will
            be initialized with null values.
        Returns
        -------
        AddColumnsResult
            version: the new version number of the table after adding columns.
        """
    @abstractmethod
@@ -1340,10 +1415,15 @@ class Table(ABC):
                nullability is not changed. Only non-nullable columns can be changed
                to nullable. Currently, you cannot change a nullable column to
                non-nullable.
        Returns
        -------
        AlterColumnsResult
            version: the new version number of the table after the alteration.
        """
    @abstractmethod
-    def drop_columns(self, columns: Iterable[str]):
+    def drop_columns(self, columns: Iterable[str]) -> DropColumnsResult:
        """
        Drop columns from the table.
@@ -1351,10 +1431,15 @@ class Table(ABC):
        ----------
        columns : Iterable[str]
            The names of the columns to drop.
        Returns
        -------
        DropColumnsResult
            version: the new version number of the table dropping the columns.
        """
    @abstractmethod
-    def checkout(self, version: int):
+    def checkout(self, version: Union[int, str]):
        """
        Checks out a specific version of the Table
@@ -1369,6 +1454,12 @@ class Table(ABC):
        Any operation that modifies the table will fail while the table is in a checked
        out state.
        Parameters
        ----------
        version: int | str,
            The version to check out. A version number (`int`) or a tag
            (`str`) can be provided.
        To return the table to a normal state use `[Self::checkout_latest]`
        """
@@ -1383,7 +1474,7 @@ class Table(ABC):
        """
    @abstractmethod
-    def restore(self, version: Optional[int] = None):
+    def restore(self, version: Optional[Union[int, str]] = None):
        """Restore a version of the table. This is an in-place operation.
        This creates a new version where the data is equivalent to the
@@ -1391,9 +1482,10 @@ class Table(ABC):
        Parameters
        ----------
-        version : int, default None
+        version : int or str, default None
-            The version to restore. If unspecified then restores the currently
+            The version number or version tag to restore.
-            checked out version. If the currently checked out version is the
+            If unspecified then restores the currently checked out version.
            If the currently checked out version is the
            latest version then this is a no-op.
        """
@@ -1538,7 +1630,46 @@ class LanceTable(Table):
        """Get the current version of the table"""
        return LOOP.run(self._table.version())
-    def checkout(self, version: int):
+    @property
    def tags(self) -> Tags:
        """Tag management for the table.
        Similar to Git, tags are a way to add metadata to a specific version of the
        table.
        .. warning::
            Tagged versions are exempted from the :py:meth:`cleanup_old_versions()`
            process.
            To remove a version that has been tagged, you must first
            :py:meth:`~Tags.delete` the associated tag.
        Returns
        -------
        Tags
            The tag manager for managing tags for the table.
        Examples
        --------
        >>> import lancedb
        >>> db = lancedb.connect("./.lancedb")
        >>> table = db.create_table("my_table",
        ...    [{"vector": [1.1, 0.9], "type": "vector"}])
        >>> table.tags.create("v1", table.version)
        >>> table.add([{"vector": [0.5, 0.2], "type": "vector"}])
        AddResult(version=2)
        >>> tags = table.tags.list()
        >>> print(tags["v1"]["version"])
        1
        >>> table.checkout("v1")
        >>> table.to_pandas()
               vector    type
        0  [1.1, 0.9]  vector
        """
        return Tags(self._table)
    def checkout(self, version: Union[int, str]):
        """Checkout a version of the table. This is an in-place operation.
        This allows viewing previous versions of the table. If you wish to
@@ -1550,8 +1681,9 @@ class LanceTable(Table):
        Parameters
        ----------
-        version : int
+        version: int | str,
-            The version to checkout.
+            The version to check out. A version number (`int`) or a tag
            (`str`) can be provided.
        Examples
        --------
@@ -1565,6 +1697,7 @@ class LanceTable(Table):
               vector    type
        0  [1.1, 0.9]  vector
        >>> table.add([{"vector": [0.5, 0.2], "type": "vector"}])
        AddResult(version=2)
        >>> table.version
        2
        >>> table.checkout(1)
@@ -1582,7 +1715,7 @@ class LanceTable(Table):
        """
        LOOP.run(self._table.checkout_latest())
-    def restore(self, version: Optional[int] = None):
+    def restore(self, version: Optional[Union[int, str]] = None):
        """Restore a version of the table. This is an in-place operation.
        This creates a new version where the data is equivalent to the
@@ -1590,9 +1723,10 @@ class LanceTable(Table):
        Parameters
        ----------
-        version : int, default None
+        version : int or str, default None
-            The version to restore. If unspecified then restores the currently
+            The version number or version tag to restore.
-            checked out version. If the currently checked out version is the
+            If unspecified then restores the currently checked out version.
            If the currently checked out version is the
            latest version then this is a no-op.
        Examples
@@ -1607,14 +1741,23 @@ class LanceTable(Table):
               vector    type
        0  [1.1, 0.9]  vector
        >>> table.add([{"vector": [0.5, 0.2], "type": "vector"}])
        AddResult(version=2)
        >>> table.version
        2
        >>> table.tags.create("v2", 2)
        >>> table.restore(1)
        >>> table.to_pandas()
               vector    type
        0  [1.1, 0.9]  vector
        >>> len(table.list_versions())
        3
        >>> table.restore("v2")
        >>> table.to_pandas()
               vector    type
        0  [1.1, 0.9]  vector
        1  [0.5, 0.2]  vector
        >>> len(table.list_versions())
        4
        """
        if version is not None:
            LOOP.run(self._table.checkout(version))
@@ -1623,9 +1766,6 @@ class LanceTable(Table):
    def count_rows(self, filter: Optional[str] = None) -> int:
        return LOOP.run(self._table.count_rows(filter))
    def __len__(self) -> int:
        return self.count_rows()
    def __repr__(self) -> str:
        val = f"{self.__class__.__name__}(name={self.name!r}, version={self.version}"
        if self._conn.read_consistency_interval is not None:
@@ -1801,6 +1941,9 @@ class LanceTable(Table):
    ) -> None:
        return LOOP.run(self._table.wait_for_index(index_names, timeout))
    def stats(self) -> TableStatistics:
        return LOOP.run(self._table.stats())
    def create_scalar_index(
        self,
        column: str,
@@ -1829,15 +1972,15 @@ class LanceTable(Table):
        writer_heap_size: Optional[int] = 1024 * 1024 * 1024,
        use_tantivy: bool = True,
        tokenizer_name: Optional[str] = None,
-        with_position: bool = True,
+        with_position: bool = False,
        # tokenizer configs:
        base_tokenizer: BaseTokenizerType = "simple",
        language: str = "English",
        max_token_length: Optional[int] = 40,
        lower_case: bool = True,
-        stem: bool = False,
+        stem: bool = True,
-        remove_stop_words: bool = False,
+        remove_stop_words: bool = True,
-        ascii_folding: bool = False,
+        ascii_folding: bool = True,
    ):
        if not use_tantivy:
            if not isinstance(field_names, str):
@@ -1847,6 +1990,7 @@ class LanceTable(Table):
                tokenizer_configs = {
                    "base_tokenizer": base_tokenizer,
                    "language": language,
                    "with_position": with_position,
                    "max_token_length": max_token_length,
                    "lower_case": lower_case,
                    "stem": stem,
@@ -1857,7 +2001,6 @@ class LanceTable(Table):
                tokenizer_configs = self.infer_tokenizer_configs(tokenizer_name)
            config = FTS(
                with_position=with_position,
                **tokenizer_configs,
            )
@@ -1968,7 +2111,7 @@ class LanceTable(Table):
        mode: AddMode = "append",
        on_bad_vectors: OnBadVectorsType = "error",
        fill_value: float = 0.0,
-    ):
+    ) -> AddResult:
        """Add data to the table.
        If vector columns are missing and the table
        has embedding functions, then the vector columns
@@ -1992,7 +2135,7 @@ class LanceTable(Table):
        int
            The number of vectors in the table.
        """
-        LOOP.run(
+        return LOOP.run(
            self._table.add(
                data, mode=mode, on_bad_vectors=on_bad_vectors, fill_value=fill_value
            )
@@ -2322,8 +2465,8 @@ class LanceTable(Table):
        )
        return self
-    def delete(self, where: str):
+    def delete(self, where: str) -> DeleteResult:
-        LOOP.run(self._table.delete(where))
+        return LOOP.run(self._table.delete(where))
    def update(
        self,
@@ -2331,7 +2474,7 @@ class LanceTable(Table):
        values: Optional[dict] = None,
        *,
        values_sql: Optional[Dict[str, str]] = None,
-    ):
+    ) -> UpdateResult:
        """
        This can be used to update zero to all rows depending on how many
        rows match the where clause.
@@ -2349,6 +2492,12 @@ class LanceTable(Table):
            reference existing columns. For example, {"x": "x + 1"} will increment
            the x column by 1.
        Returns
        -------
        UpdateResult
            - rows_updated: The number of rows that were updated
            - version: The new version number of the table after the update
        Examples
        --------
        >>> import lancedb
@@ -2362,6 +2511,7 @@ class LanceTable(Table):
        1  2  [3.0, 4.0]
        2  3  [5.0, 6.0]
        >>> table.update(where="x = 2", values={"vector": [10.0, 10]})
        UpdateResult(rows_updated=1, version=2)
        >>> table.to_pandas()
           x        vector
        0  1    [1.0, 2.0]
@@ -2369,7 +2519,7 @@ class LanceTable(Table):
        2  2  [10.0, 10.0]
        """
-        LOOP.run(self._table.update(values, where=where, updates_sql=values_sql))
+        return LOOP.run(self._table.update(values, where=where, updates_sql=values_sql))
    def _execute_query(
        self,
@@ -2403,8 +2553,10 @@ class LanceTable(Table):
        new_data: DATA,
        on_bad_vectors: OnBadVectorsType,
        fill_value: float,
-    ):
+    ) -> MergeResult:
-        LOOP.run(self._table._do_merge(merge, new_data, on_bad_vectors, fill_value))
+        return LOOP.run(
            self._table._do_merge(merge, new_data, on_bad_vectors, fill_value)
        )
    @deprecation.deprecated(
        deprecated_in="0.21.0",
@@ -2546,14 +2698,16 @@ class LanceTable(Table):
    def add_columns(
        self, transforms: Dict[str, str] | pa.field | List[pa.field] | pa.Schema
-    ):
+    ) -> AddColumnsResult:
-        LOOP.run(self._table.add_columns(transforms))
+        return LOOP.run(self._table.add_columns(transforms))
-    def alter_columns(self, *alterations: Iterable[Dict[str, str]]):
+    def alter_columns(
-        LOOP.run(self._table.alter_columns(*alterations))
+        self, *alterations: Iterable[Dict[str, str]]
    ) -> AlterColumnsResult:
        return LOOP.run(self._table.alter_columns(*alterations))
-    def drop_columns(self, columns: Iterable[str]):
+    def drop_columns(self, columns: Iterable[str]) -> DropColumnsResult:
-        LOOP.run(self._table.drop_columns(columns))
+        return LOOP.run(self._table.drop_columns(columns))
    def uses_v2_manifest_paths(self) -> bool:
        """
@@ -3095,6 +3249,12 @@ class AsyncTable:
        """
        await self._inner.wait_for_index(index_names, timeout)
    async def stats(self) -> TableStatistics:
        """
        Retrieve table and fragment statistics.
        """
        return await self._inner.stats()
    async def add(
        self,
        data: DATA,
@@ -3102,7 +3262,7 @@ class AsyncTable:
        mode: Optional[Literal["append", "overwrite"]] = "append",
        on_bad_vectors: Optional[OnBadVectorsType] = None,
        fill_value: Optional[float] = None,
-    ):
+    ) -> AddResult:
        """Add more data to the [Table](Table).
        Parameters
@@ -3141,7 +3301,7 @@ class AsyncTable:
        if isinstance(data, pa.Table):
            data = data.to_reader()
-        await self._inner.add(data, mode or "append")
+        return await self._inner.add(data, mode or "append")
    def merge_insert(self, on: Union[str, Iterable[str]]) -> LanceMergeInsertBuilder:
        """
@@ -3186,10 +3346,12 @@ class AsyncTable:
        >>> table = db.create_table("my_table", data)
        >>> new_data = pa.table({"a": [2, 3, 4], "b": ["x", "y", "z"]})
        >>> # Perform a "upsert" operation
-        >>> table.merge_insert("a")             \\
+        >>> res = table.merge_insert("a")     \\
        ...      .when_matched_update_all()     \\
        ...      .when_not_matched_insert_all() \\
        ...      .execute(new_data)
        >>> res
        MergeResult(version=2, num_updated_rows=2, num_inserted_rows=1, num_deleted_rows=0)
        >>> # The order of new rows is non-deterministic since we use
        >>> # a hash-join as part of this operation and so we sort here
        >>> table.to_arrow().sort_by("a").to_pandas()
@@ -3198,7 +3360,7 @@ class AsyncTable:
        1  2  x
        2  3  y
        3  4  z
-        """
+        """  # noqa: E501
        on = [on] if isinstance(on, str) else list(iter(on))
        return LanceMergeInsertBuilder(self, on)
@@ -3529,7 +3691,7 @@ class AsyncTable:
        new_data: DATA,
        on_bad_vectors: OnBadVectorsType,
        fill_value: float,
-    ):
+    ) -> MergeResult:
        schema = await self.schema()
        if on_bad_vectors is None:
            on_bad_vectors = "error"
@@ -3545,7 +3707,7 @@ class AsyncTable:
        )
        if isinstance(data, pa.Table):
            data = pa.RecordBatchReader.from_batches(data.schema, data.to_batches())
-        await self._inner.execute_merge_insert(
+        return await self._inner.execute_merge_insert(
            data,
            dict(
                on=merge._on,
@@ -3554,10 +3716,11 @@ class AsyncTable:
                when_not_matched_insert_all=merge._when_not_matched_insert_all,
                when_not_matched_by_source_delete=merge._when_not_matched_by_source_delete,
                when_not_matched_by_source_condition=merge._when_not_matched_by_source_condition,
                timeout=merge._timeout,
            ),
        )
-    async def delete(self, where: str):
+    async def delete(self, where: str) -> DeleteResult:
        """Delete rows from the table.
        This can be used to delete a single row, many rows, all rows, or
@@ -3588,6 +3751,7 @@ class AsyncTable:
        1  2  [3.0, 4.0]
        2  3  [5.0, 6.0]
        >>> table.delete("x = 2")
        DeleteResult(version=2)
        >>> table.to_pandas()
           x      vector
        0  1  [1.0, 2.0]
@@ -3601,6 +3765,7 @@ class AsyncTable:
        >>> to_remove
        '1, 5'
        >>> table.delete(f"x IN ({to_remove})")
        DeleteResult(version=3)
        >>> table.to_pandas()
           x      vector
        0  3  [5.0, 6.0]
@@ -3613,7 +3778,7 @@ class AsyncTable:
        *,
        where: Optional[str] = None,
        updates_sql: Optional[Dict[str, str]] = None,
-    ):
+    ) -> UpdateResult:
        """
        This can be used to update zero to all rows in the table.
@@ -3635,6 +3800,13 @@ class AsyncTable:
            literals (e.g. "7" or "'foo'") or they can be expressions based on the
            previous value of the row (e.g. "x + 1" to increment the x column by 1)
        Returns
        -------
        UpdateResult
            An object containing:
            - rows_updated: The number of rows that were updated
            - version: The new version number of the table after the update
        Examples
        --------
        >>> import asyncio
@@ -3663,7 +3835,7 @@ class AsyncTable:
    async def add_columns(
        self, transforms: dict[str, str] | pa.field | List[pa.field] | pa.Schema
-    ):
+    ) -> AddColumnsResult:
        """
        Add new columns with defined values.
@@ -3675,6 +3847,12 @@ class AsyncTable:
            each row in the table, and can reference existing columns.
            Alternatively, you can pass a pyarrow field or schema to add
            new columns with NULLs.
        Returns
        -------
        AddColumnsResult
            version: the new version number of the table after adding columns.
        """
        if isinstance(transforms, pa.Field):
            transforms = [transforms]
@@ -3683,11 +3861,13 @@ class AsyncTable:
        ):
            transforms = pa.schema(transforms)
        if isinstance(transforms, pa.Schema):
-            await self._inner.add_columns_with_schema(transforms)
+            return await self._inner.add_columns_with_schema(transforms)
        else:
-            await self._inner.add_columns(list(transforms.items()))
+            return await self._inner.add_columns(list(transforms.items()))
-    async def alter_columns(self, *alterations: Iterable[dict[str, Any]]):
+    async def alter_columns(
        self, *alterations: Iterable[dict[str, Any]]
    ) -> AlterColumnsResult:
        """
        Alter column names and nullability.
@@ -3707,8 +3887,13 @@ class AsyncTable:
                nullability is not changed. Only non-nullable columns can be changed
                to nullable. Currently, you cannot change a nullable column to
                non-nullable.
        Returns
        -------
        AlterColumnsResult
            version: the new version number of the table after the alteration.
        """
-        await self._inner.alter_columns(alterations)
+        return await self._inner.alter_columns(alterations)
    async def drop_columns(self, columns: Iterable[str]):
        """
@@ -3719,7 +3904,7 @@ class AsyncTable:
        columns : Iterable[str]
            The names of the columns to drop.
        """
-        await self._inner.drop_columns(columns)
+        return await self._inner.drop_columns(columns)
    async def version(self) -> int:
        """
@@ -3746,7 +3931,7 @@ class AsyncTable:
        return versions
-    async def checkout(self, version: int):
+    async def checkout(self, version: int | str):
        """
        Checks out a specific version of the Table
@@ -3761,6 +3946,12 @@ class AsyncTable:
        Any operation that modifies the table will fail while the table is in a checked
        out state.
        Parameters
        ----------
        version: int | str,
            The version to check out. A version number (`int`) or a tag
            (`str`) can be provided.
        To return the table to a normal state use `[Self::checkout_latest]`
        """
        try:
@@ -3783,7 +3974,7 @@ class AsyncTable:
        """
        await self._inner.checkout_latest()
-    async def restore(self, version: Optional[int] = None):
+    async def restore(self, version: Optional[int | str] = None):
        """
        Restore the table to the currently checked out version
@@ -3798,6 +3989,24 @@ class AsyncTable:
        """
        await self._inner.restore(version)
    @property
    def tags(self) -> AsyncTags:
        """Tag management for the dataset.
        Similar to Git, tags are a way to add metadata to a specific version of the
        dataset.
        .. warning::
            Tagged versions are exempted from the
            :py:meth:`optimize(cleanup_older_than)` process.
            To remove a version that has been tagged, you must first
            :py:meth:`~Tags.delete` the associated tag.
        """
        return AsyncTags(self._inner)
    async def optimize(
        self,
        *,
@@ -3967,3 +4176,217 @@ class IndexStatistics:
    # a dictionary instead of a class.
    def __getitem__(self, key):
        return getattr(self, key)
@dataclass
 class TableStatistics:
    """
    Statistics about a table and fragments.
    Attributes
    ----------
    total_bytes: int
        The total number of bytes in the table.
    num_rows: int
        The total number of rows in the table.
    num_indices: int
        The total number of indices in the table.
    fragment_stats: FragmentStatistics
        Statistics about fragments in the table.
    """
    total_bytes: int
    num_rows: int
    num_indices: int
    fragment_stats: FragmentStatistics
@dataclass
 class FragmentStatistics:
    """
    Statistics about fragments.
    Attributes
    ----------
    num_fragments: int
        The total number of fragments in the table.
    num_small_fragments: int
        The total number of small fragments in the table.
        Small fragments have low row counts and may need to be compacted.
    lengths: FragmentSummaryStats
        Statistics about the number of rows in the table fragments.
    """
    num_fragments: int
    num_small_fragments: int
    lengths: FragmentSummaryStats
@dataclass
 class FragmentSummaryStats:
    """
    Statistics about fragments sizes
    Attributes
    ----------
    min: int
        The number of rows in the fragment with the fewest rows.
    max: int
        The number of rows in the fragment with the most rows.
    mean: int
        The mean number of rows in the fragments.
    p25: int
        The 25th percentile of number of rows in the fragments.
    p50: int
        The 50th percentile of number of rows in the fragments.
    p75: int
        The 75th percentile of number of rows in the fragments.
    p99: int
        The 99th percentile of number of rows in the fragments.
    """
    min: int
    max: int
    mean: int
    p25: int
    p50: int
    p75: int
    p99: int
 class Tags:
    """
    Table tag manager.
    """
    def __init__(self, table):
        self._table = table
    def list(self) -> Dict[str, Tag]:
        """
        List all table tags.
        Returns
        -------
        dict[str, Tag]
            A dictionary mapping tag names to version numbers.
        """
        return LOOP.run(self._table.tags.list())
    def get_version(self, tag: str) -> int:
        """
        Get the version of a tag.
        Parameters
        ----------
        tag: str,
            The name of the tag to get the version for.
        """
        return LOOP.run(self._table.tags.get_version(tag))
    def create(self, tag: str, version: int) -> None:
        """
        Create a tag for a given table version.
        Parameters
        ----------
        tag: str,
            The name of the tag to create. This name must be unique among all tag
            names for the table.
        version: int,
            The table version to tag.
        """
        LOOP.run(self._table.tags.create(tag, version))
    def delete(self, tag: str) -> None:
        """
        Delete tag from the table.
        Parameters
        ----------
        tag: str,
            The name of the tag to delete.
        """
        LOOP.run(self._table.tags.delete(tag))
    def update(self, tag: str, version: int) -> None:
        """
        Update tag to a new version.
        Parameters
        ----------
        tag: str,
            The name of the tag to update.
        version: int,
            The new table version to tag.
        """
        LOOP.run(self._table.tags.update(tag, version))
 class AsyncTags:
    """
    Async table tag manager.
    """
    def __init__(self, table):
        self._table = table
    async def list(self) -> Dict[str, Tag]:
        """
        List all table tags.
        Returns
        -------
        dict[str, Tag]
            A dictionary mapping tag names to version numbers.
        """
        return await self._table.tags.list()
    async def get_version(self, tag: str) -> int:
        """
        Get the version of a tag.
        Parameters
        ----------
        tag: str,
            The name of the tag to get the version for.
        """
        return await self._table.tags.get_version(tag)
    async def create(self, tag: str, version: int) -> None:
        """
        Create a tag for a given table version.
        Parameters
        ----------
        tag: str,
            The name of the tag to create. This name must be unique among all tag
            names for the table.
        version: int,
            The table version to tag.
        """
        await self._table.tags.create(tag, version)
    async def delete(self, tag: str) -> None:
        """
        Delete tag from the table.
        Parameters
        ----------
        tag: str,
            The name of the tag to delete.
        """
        await self._table.tags.delete(tag)
    async def update(self, tag: str, version: int) -> None:
        """
        Update tag to a new version.
        Parameters
        ----------
        tag: str,
            The name of the tag to update.
        version: int,
            The new table version to tag.
        """
        await self._table.tags.update(tag, version)
--- a/python/python/tests/docs/test_guide_tables.py
+++ b/python/python/tests/docs/test_guide_tables.py
@@ -25,6 +25,10 @@ import numpy as np
 from lancedb.pydantic import Vector, LanceModel
 # --8<-- [end:import-lancedb-pydantic]
 # --8<-- [start:import-session-context]
 from datafusion import SessionContext
 # --8<-- [end:import-session-context]
 # --8<-- [start:import-datetime]
 from datetime import timedelta
@@ -33,6 +37,10 @@ from datetime import timedelta
 from lancedb.embeddings import get_registry
 # --8<-- [end:import-embeddings]
 # --8<-- [start:import-ffi-dataset]
 from lance import FFILanceTableProvider
 # --8<-- [end:import-ffi-dataset]
 # --8<-- [start:import-pydantic-basemodel]
 from pydantic import BaseModel
@@ -341,6 +349,27 @@ def test_table_with_embedding():
    # --8<-- [end:create_table_with_embedding]
 def test_sql_query():
    db = lancedb.connect("data/sample-lancedb")
    data = [
        {"vector": [1.1, 1.2], "lat": 45.5, "long": -122.7},
        {"vector": [0.2, 1.8], "lat": 40.1, "long": -74.1},
    ]
    table = db.create_table("lance_table", data)
    # --8<-- [start:lance_sql_basic]
    ctx = SessionContext()
    ffi_lance_table = FFILanceTableProvider(
        table.to_lance(), with_row_id=False, with_row_addr=False
    )
    ctx.register_table_provider("ffi_lance_table", ffi_lance_table)
    ctx.table("ffi_lance_table")
    ctx.sql("SELECT vector FROM ffi_lance_table")
    # --8<-- [end:lance_sql_basic]
@pytest.mark.skip
 async def test_table_with_embedding_async():
    async_db = await lancedb.connect_async("data/sample-lancedb")
--- a/python/python/tests/docs/test_merge_insert.py
+++ b/python/python/tests/docs/test_merge_insert.py
@@ -18,15 +18,19 @@ def test_upsert(mem_db):
        {"id": 1, "name": "Bobby"},
        {"id": 2, "name": "Charlie"},
    ]
-    (
+    res = (
        table.merge_insert("id")
        .when_matched_update_all()
        .when_not_matched_insert_all()
        .execute(new_users)
    )
    table.count_rows()  # 3
    res  # {'num_inserted_rows': 1, 'num_updated_rows': 1, 'num_deleted_rows': 0}
    # --8<-- [end:upsert_basic]
    assert table.count_rows() == 3
    assert res.num_inserted_rows == 1
    assert res.num_deleted_rows == 0
    assert res.num_updated_rows == 1
@pytest.mark.asyncio
@@ -44,15 +48,22 @@ async def test_upsert_async(mem_db_async):
        {"id": 1, "name": "Bobby"},
        {"id": 2, "name": "Charlie"},
    ]
-    await (
+    res = await (
        table.merge_insert("id")
        .when_matched_update_all()
        .when_not_matched_insert_all()
        .execute(new_users)
    )
    await table.count_rows()  # 3
    res
    # MergeResult(version=2, num_updated_rows=1,
    # num_inserted_rows=1, num_deleted_rows=0)
    # --8<-- [end:upsert_basic_async]
    assert await table.count_rows() == 3
    assert res.version == 2
    assert res.num_inserted_rows == 1
    assert res.num_deleted_rows == 0
    assert res.num_updated_rows == 1
 def test_insert_if_not_exists(mem_db):
@@ -69,10 +80,19 @@ def test_insert_if_not_exists(mem_db):
        {"domain": "google.com", "name": "Google"},
        {"domain": "facebook.com", "name": "Facebook"},
    ]
-    (table.merge_insert("domain").when_not_matched_insert_all().execute(new_domains))
+    res = (
        table.merge_insert("domain").when_not_matched_insert_all().execute(new_domains)
    )
    table.count_rows()  # 3
    res
    # MergeResult(version=2, num_updated_rows=0,
    # num_inserted_rows=1, num_deleted_rows=0)
    # --8<-- [end:insert_if_not_exists]
    assert table.count_rows() == 3
    assert res.version == 2
    assert res.num_inserted_rows == 1
    assert res.num_deleted_rows == 0
    assert res.num_updated_rows == 0
@pytest.mark.asyncio
@@ -90,12 +110,19 @@ async def test_insert_if_not_exists_async(mem_db_async):
        {"domain": "google.com", "name": "Google"},
        {"domain": "facebook.com", "name": "Facebook"},
    ]
-    await (
+    res = await (
        table.merge_insert("domain").when_not_matched_insert_all().execute(new_domains)
    )
    await table.count_rows()  # 3
-    # --8<-- [end:insert_if_not_exists_async]
+    res
    # MergeResult(version=2, num_updated_rows=0,
    # num_inserted_rows=1, num_deleted_rows=0)
    # --8<-- [end:insert_if_not_exists]
    assert await table.count_rows() == 3
    assert res.version == 2
    assert res.num_inserted_rows == 1
    assert res.num_deleted_rows == 0
    assert res.num_updated_rows == 0
 def test_replace_range(mem_db):
@@ -113,7 +140,7 @@ def test_replace_range(mem_db):
    new_chunks = [
        {"doc_id": 1, "chunk_id": 0, "text": "Baz"},
    ]
-    (
+    res = (
        table.merge_insert(["doc_id", "chunk_id"])
        .when_matched_update_all()
        .when_not_matched_insert_all()
@@ -121,8 +148,15 @@ def test_replace_range(mem_db):
        .execute(new_chunks)
    )
    table.count_rows("doc_id = 1")  # 1
-    # --8<-- [end:replace_range]
+    res
    # MergeResult(version=2, num_updated_rows=1,
    # num_inserted_rows=0, num_deleted_rows=1)
    # --8<-- [end:insert_if_not_exists]
    assert table.count_rows("doc_id = 1") == 1
    assert res.version == 2
    assert res.num_inserted_rows == 0
    assert res.num_deleted_rows == 1
    assert res.num_updated_rows == 1
@pytest.mark.asyncio
@@ -141,7 +175,7 @@ async def test_replace_range_async(mem_db_async):
    new_chunks = [
        {"doc_id": 1, "chunk_id": 0, "text": "Baz"},
    ]
-    await (
+    res = await (
        table.merge_insert(["doc_id", "chunk_id"])
        .when_matched_update_all()
        .when_not_matched_insert_all()
@@ -149,5 +183,12 @@ async def test_replace_range_async(mem_db_async):
        .execute(new_chunks)
    )
    await table.count_rows("doc_id = 1")  # 1
-    # --8<-- [end:replace_range_async]
+    res
    # MergeResult(version=2, num_updated_rows=1,
    # num_inserted_rows=0, num_deleted_rows=1)
    # --8<-- [end:insert_if_not_exists]
    assert await table.count_rows("doc_id = 1") == 1
    assert res.version == 2
    assert res.num_inserted_rows == 0
    assert res.num_deleted_rows == 1
    assert res.num_updated_rows == 1
--- a/python/python/tests/docs/test_search.py
+++ b/python/python/tests/docs/test_search.py
@@ -156,6 +156,9 @@ async def test_vector_search_async():
    # --8<-- [end:search_result_async_as_list]
@pytest.mark.skipif(
    os.name == "nt", reason="Need to fix https://github.com/lancedb/lance/issues/3905"
 )
 def test_fts_fuzzy_query():
    uri = "data/fuzzy-example"
    db = lancedb.connect(uri)
@@ -189,6 +192,9 @@ def test_fts_fuzzy_query():
    }
@pytest.mark.skipif(
    os.name == "nt", reason="Need to fix https://github.com/lancedb/lance/issues/3905"
 )
 def test_fts_boost_query():
    uri = "data/boost-example"
    db = lancedb.connect(uri)
@@ -234,6 +240,9 @@ def test_fts_boost_query():
    )
@pytest.mark.skipif(
    os.name == "nt", reason="Need to fix https://github.com/lancedb/lance/issues/3905"
 )
 def test_fts_native():
    # --8<-- [start:basic_fts]
    uri = "data/sample-lancedb"
@@ -282,6 +291,9 @@ def test_fts_native():
    # --8<-- [end:fts_incremental_index]
@pytest.mark.skipif(
    os.name == "nt", reason="Need to fix https://github.com/lancedb/lance/issues/3905"
 )
@pytest.mark.asyncio
 async def test_fts_native_async():
    # --8<-- [start:basic_fts_async]
--- a/python/python/tests/test_fts.py
+++ b/python/python/tests/test_fts.py
@@ -287,7 +287,7 @@ def test_search_fts_phrase_query(table):
        assert False
    except Exception:
        pass
-    table.create_fts_index("text", use_tantivy=False, replace=True)
+    table.create_fts_index("text", use_tantivy=False, with_position=True, replace=True)
    results = table.search("puppy").limit(100).to_list()
    phrase_results = table.search('"puppy runs"').limit(100).to_list()
    assert len(results) > len(phrase_results)
@@ -312,7 +312,7 @@ async def test_search_fts_phrase_query_async(async_table):
        assert False
    except Exception:
        pass
-    await async_table.create_index("text", config=FTS())
+    await async_table.create_index("text", config=FTS(with_position=True))
    results = await async_table.query().nearest_to_text("puppy").limit(100).to_list()
    phrase_results = (
        await async_table.query().nearest_to_text('"puppy runs"').limit(100).to_list()
@@ -649,7 +649,7 @@ def test_fts_on_list(mem_db: DBConnection):
        }
    )
    table = mem_db.create_table("test", data=data)
-    table.create_fts_index("text", use_tantivy=False)
+    table.create_fts_index("text", use_tantivy=False, with_position=True)
    res = table.search("lance").limit(5).to_list()
    assert len(res) == 3
--- a/python/python/tests/test_remote_db.py
+++ b/python/python/tests/test_remote_db.py
@@ -149,6 +149,24 @@ async def test_async_checkout():
        assert await table.count_rows() == 300
 def test_table_len_sync():
    def handler(request):
        if request.path == "/v1/table/test/create/?mode=create":
            request.send_response(200)
            request.send_header("Content-Type", "application/json")
            request.end_headers()
            request.wfile.write(b"{}")
        request.send_response(200)
        request.send_header("Content-Type", "application/json")
        request.end_headers()
        request.wfile.write(json.dumps(1).encode())
    with mock_lancedb_connection(handler) as db:
        table = db.create_table("test", [{"id": 1}])
        assert len(table) == 1
@pytest.mark.asyncio
 async def test_http_error():
    request_id_holder = {"request_id": None}
@@ -389,6 +407,50 @@ def test_table_wait_for_index_timeout():
            table.wait_for_index(["id_idx"], timedelta(seconds=1))
 def test_stats():
    stats = {
        "total_bytes": 38,
        "num_rows": 2,
        "num_indices": 0,
        "fragment_stats": {
            "num_fragments": 1,
            "num_small_fragments": 1,
            "lengths": {
                "min": 2,
                "max": 2,
                "mean": 2,
                "p25": 2,
                "p50": 2,
                "p75": 2,
                "p99": 2,
            },
        },
    }
    def handler(request):
        if request.path == "/v1/table/test/create/?mode=create":
            request.send_response(200)
            request.send_header("Content-Type", "application/json")
            request.end_headers()
            request.wfile.write(b"{}")
        elif request.path == "/v1/table/test/stats/":
            request.send_response(200)
            request.send_header("Content-Type", "application/json")
            request.end_headers()
            payload = json.dumps(stats)
            request.wfile.write(payload.encode())
        else:
            print(request.path)
            request.send_response(404)
            request.end_headers()
    with mock_lancedb_connection(handler) as db:
        table = db.create_table("test", [{"id": 1}])
        res = table.stats()
        print(f"{res=}")
        assert res == stats
@contextlib.contextmanager
 def query_test_table(query_handler, *, server_version=Version("0.1.0")):
    def handler(request):
--- a/python/python/tests/test_table.py
+++ b/python/python/tests/test_table.py
@@ -106,15 +106,22 @@ async def test_update_async(mem_db_async: AsyncConnection):
    table = await mem_db_async.create_table("some_table", data=[{"id": 0}])
    assert await table.count_rows("id == 0") == 1
    assert await table.count_rows("id == 7") == 0
-    await table.update({"id": 7})
+    update_res = await table.update({"id": 7})
    assert update_res.rows_updated == 1
    assert update_res.version == 2
    assert await table.count_rows("id == 7") == 1
    assert await table.count_rows("id == 0") == 0
-    await table.add([{"id": 2}])
+    add_res = await table.add([{"id": 2}])
-    await table.update(where="id % 2 == 0", updates_sql={"id": "5"})
+    assert add_res.version == 3
    update_res = await table.update(where="id % 2 == 0", updates_sql={"id": "5"})
    assert update_res.rows_updated == 1
    assert update_res.version == 4
    assert await table.count_rows("id == 7") == 1
    assert await table.count_rows("id == 2") == 0
    assert await table.count_rows("id == 5") == 1
-    await table.update({"id": 10}, where="id == 5")
+    update_res = await table.update({"id": 10}, where="id == 5")
    assert update_res.rows_updated == 1
    assert update_res.version == 5
    assert await table.count_rows("id == 10") == 1
@@ -437,7 +444,8 @@ def test_add_pydantic_model(mem_db: DBConnection):
            content="foo", meta=Metadata(source="bar", timestamp=datetime.now())
        ),
    )
-    tbl.add([expected])
+    add_res = tbl.add([expected])
    assert add_res.version == 2
    result = tbl.search([0.0, 0.0]).limit(1).to_pydantic(LanceSchema)[0]
    assert result == expected
@@ -459,11 +467,12 @@ async def test_add_async(mem_db_async: AsyncConnection):
        ],
    )
    assert await table.count_rows() == 2
-    await table.add(
+    add_res = await table.add(
        data=[
            {"vector": [10.0, 11.0], "item": "baz", "price": 30.0},
        ],
    )
    assert add_res.version == 2
    assert await table.count_rows() == 3
@@ -529,6 +538,113 @@ def test_versioning(mem_db: DBConnection):
    assert len(table) == 2
 def test_tags(mem_db: DBConnection):
    table = mem_db.create_table(
        "test",
        data=[
            {"vector": [3.1, 4.1], "item": "foo", "price": 10.0},
            {"vector": [5.9, 26.5], "item": "bar", "price": 20.0},
        ],
    )
    table.tags.create("tag1", 1)
    tags = table.tags.list()
    assert "tag1" in tags
    assert tags["tag1"]["version"] == 1
    table.add(
        data=[
            {"vector": [10.0, 11.0], "item": "baz", "price": 30.0},
        ],
    )
    table.tags.create("tag2", 2)
    tags = table.tags.list()
    assert "tag1" in tags
    assert "tag2" in tags
    assert tags["tag1"]["version"] == 1
    assert tags["tag2"]["version"] == 2
    table.tags.delete("tag2")
    table.tags.update("tag1", 2)
    tags = table.tags.list()
    assert "tag1" in tags
    assert tags["tag1"]["version"] == 2
    table.tags.update("tag1", 1)
    tags = table.tags.list()
    assert "tag1" in tags
    assert tags["tag1"]["version"] == 1
    table.checkout("tag1")
    assert table.version == 1
    assert table.count_rows() == 2
    table.tags.create("tag2", 2)
    table.checkout("tag2")
    assert table.version == 2
    assert table.count_rows() == 3
    table.checkout_latest()
    table.add(
        data=[
            {"vector": [12.0, 13.0], "item": "baz", "price": 40.0},
        ],
    )
@pytest.mark.asyncio
 async def test_async_tags(mem_db_async: AsyncConnection):
    table = await mem_db_async.create_table(
        "test",
        data=[
            {"vector": [3.1, 4.1], "item": "foo", "price": 10.0},
            {"vector": [5.9, 26.5], "item": "bar", "price": 20.0},
        ],
    )
    await table.tags.create("tag1", 1)
    tags = await table.tags.list()
    assert "tag1" in tags
    assert tags["tag1"]["version"] == 1
    await table.add(
        data=[
            {"vector": [10.0, 11.0], "item": "baz", "price": 30.0},
        ],
    )
    await table.tags.create("tag2", 2)
    tags = await table.tags.list()
    assert "tag1" in tags
    assert "tag2" in tags
    assert tags["tag1"]["version"] == 1
    assert tags["tag2"]["version"] == 2
    await table.tags.delete("tag2")
    await table.tags.update("tag1", 2)
    tags = await table.tags.list()
    assert "tag1" in tags
    assert tags["tag1"]["version"] == 2
    await table.tags.update("tag1", 1)
    tags = await table.tags.list()
    assert "tag1" in tags
    assert tags["tag1"]["version"] == 1
    await table.checkout("tag1")
    assert await table.version() == 1
    assert await table.count_rows() == 2
    await table.tags.create("tag2", 2)
    await table.checkout("tag2")
    assert await table.version() == 2
    assert await table.count_rows() == 3
    await table.checkout_latest()
    await table.add(
        data=[
            {"vector": [12.0, 13.0], "item": "baz", "price": 40.0},
        ],
    )
@patch("lancedb.table.AsyncTable.create_index")
 def test_create_index_method(mock_create_index, mem_db: DBConnection):
    table = mem_db.create_table(
@@ -653,6 +769,29 @@ def test_restore(mem_db: DBConnection):
        table.restore(0)
 def test_restore_with_tags(mem_db: DBConnection):
    table = mem_db.create_table(
        "my_table",
        data=[{"vector": [1.1, 0.9], "type": "vector"}],
    )
    tag = "tag1"
    table.tags.create(tag, 1)
    table.add([{"vector": [0.5, 0.2], "type": "vector"}])
    table.restore(tag)
    assert len(table.list_versions()) == 3
    assert len(table) == 1
    expected = table.to_arrow()
    table.add([{"vector": [0.3, 0.3], "type": "vector"}])
    table.checkout("tag1")
    table.restore()
    assert len(table.list_versions()) == 5
    assert table.to_arrow() == expected
    with pytest.raises(ValueError):
        table.restore("tag_unknown")
 def test_merge(tmp_db: DBConnection, tmp_path):
    pytest.importorskip("lance")
    import lance
@@ -688,7 +827,8 @@ def test_delete(mem_db: DBConnection):
    )
    assert len(table) == 2
    assert len(table.list_versions()) == 1
-    table.delete("id=0")
+    delete_res = table.delete("id=0")
    assert delete_res.version == 2
    assert len(table.list_versions()) == 2
    assert table.version == 2
    assert len(table) == 1
@@ -702,7 +842,9 @@ def test_update(mem_db: DBConnection):
    )
    assert len(table) == 2
    assert len(table.list_versions()) == 1
-    table.update(where="id=0", values={"vector": [1.1, 1.1]})
+    update_res = table.update(where="id=0", values={"vector": [1.1, 1.1]})
    assert update_res.version == 2
    assert update_res.rows_updated == 1
    assert len(table.list_versions()) == 2
    assert table.version == 2
    assert len(table) == 2
@@ -791,9 +933,16 @@ def test_merge_insert(mem_db: DBConnection):
    new_data = pa.table({"a": [2, 3, 4], "b": ["x", "y", "z"]})
    # upsert
-    table.merge_insert(
+    merge_insert_res = (
-        "a"
+        table.merge_insert("a")
-    ).when_matched_update_all().when_not_matched_insert_all().execute(new_data)
+        .when_matched_update_all()
        .when_not_matched_insert_all()
        .execute(new_data, timeout=timedelta(seconds=10))
    )
    assert merge_insert_res.version == 2
    assert merge_insert_res.num_inserted_rows == 1
    assert merge_insert_res.num_updated_rows == 2
    assert merge_insert_res.num_deleted_rows == 0
    expected = pa.table({"a": [1, 2, 3, 4], "b": ["a", "x", "y", "z"]})
    assert table.to_arrow().sort_by("a") == expected
@@ -801,17 +950,28 @@ def test_merge_insert(mem_db: DBConnection):
    table.restore(version)
    # conditional update
-    table.merge_insert("a").when_matched_update_all(where="target.b = 'b'").execute(
+    merge_insert_res = (
-        new_data
+        table.merge_insert("a")
        .when_matched_update_all(where="target.b = 'b'")
        .execute(new_data)
    )
    assert merge_insert_res.version == 4
    assert merge_insert_res.num_inserted_rows == 0
    assert merge_insert_res.num_updated_rows == 1
    assert merge_insert_res.num_deleted_rows == 0
    expected = pa.table({"a": [1, 2, 3], "b": ["a", "x", "c"]})
    assert table.to_arrow().sort_by("a") == expected
    table.restore(version)
    # insert-if-not-exists
-    table.merge_insert("a").when_not_matched_insert_all().execute(new_data)
+    merge_insert_res = (
-
+        table.merge_insert("a").when_not_matched_insert_all().execute(new_data)
    )
    assert merge_insert_res.version == 6
    assert merge_insert_res.num_inserted_rows == 1
    assert merge_insert_res.num_updated_rows == 0
    assert merge_insert_res.num_deleted_rows == 0
    expected = pa.table({"a": [1, 2, 3, 4], "b": ["a", "b", "c", "z"]})
    assert table.to_arrow().sort_by("a") == expected
@@ -820,13 +980,17 @@ def test_merge_insert(mem_db: DBConnection):
    new_data = pa.table({"a": [2, 4], "b": ["x", "z"]})
    # replace-range
-    (
+    merge_insert_res = (
        table.merge_insert("a")
        .when_matched_update_all()
        .when_not_matched_insert_all()
        .when_not_matched_by_source_delete("a > 2")
        .execute(new_data)
    )
    assert merge_insert_res.version == 8
    assert merge_insert_res.num_inserted_rows == 1
    assert merge_insert_res.num_updated_rows == 1
    assert merge_insert_res.num_deleted_rows == 1
    expected = pa.table({"a": [1, 2, 4], "b": ["a", "x", "z"]})
    assert table.to_arrow().sort_by("a") == expected
@@ -834,15 +998,27 @@ def test_merge_insert(mem_db: DBConnection):
    table.restore(version)
    # replace-range no condition
-    table.merge_insert(
+    merge_insert_res = (
-        "a"
+        table.merge_insert("a")
-    ).when_matched_update_all().when_not_matched_insert_all().when_not_matched_by_source_delete().execute(
+        .when_matched_update_all()
-        new_data
+        .when_not_matched_insert_all()
        .when_not_matched_by_source_delete()
        .execute(new_data)
    )
    assert merge_insert_res.version == 10
    assert merge_insert_res.num_inserted_rows == 1
    assert merge_insert_res.num_updated_rows == 1
    assert merge_insert_res.num_deleted_rows == 2
    expected = pa.table({"a": [2, 4], "b": ["x", "z"]})
    assert table.to_arrow().sort_by("a") == expected
    # timeout
    with pytest.raises(Exception, match="merge insert timed out"):
        table.merge_insert("a").when_matched_update_all().execute(
            new_data, timeout=timedelta(0)
        )
 # We vary the data format because there are slight differences in how
 # subschemas are handled in different formats
@@ -1371,11 +1547,13 @@ def test_restore_consistency(tmp_path):
 def test_add_columns(mem_db: DBConnection):
    data = pa.table({"id": [0, 1]})
    table = LanceTable.create(mem_db, "my_table", data=data)
-    table.add_columns({"new_col": "id + 2"})
+    add_columns_res = table.add_columns({"new_col": "id + 2"})
    assert add_columns_res.version == 2
    assert table.to_arrow().column_names == ["id", "new_col"]
    assert table.to_arrow()["new_col"].to_pylist() == [2, 3]
-    table.add_columns({"null_int": "cast(null as bigint)"})
+    add_columns_res = table.add_columns({"null_int": "cast(null as bigint)"})
    assert add_columns_res.version == 3
    assert table.schema.field("null_int").type == pa.int64()
@@ -1383,7 +1561,8 @@ def test_add_columns(mem_db: DBConnection):
 async def test_add_columns_async(mem_db_async: AsyncConnection):
    data = pa.table({"id": [0, 1]})
    table = await mem_db_async.create_table("my_table", data=data)
-    await table.add_columns({"new_col": "id + 2"})
+    add_columns_res = await table.add_columns({"new_col": "id + 2"})
    assert add_columns_res.version == 2
    data = await table.to_arrow()
    assert data.column_names == ["id", "new_col"]
    assert data["new_col"].to_pylist() == [2, 3]
@@ -1393,9 +1572,10 @@ async def test_add_columns_async(mem_db_async: AsyncConnection):
 async def test_add_columns_with_schema(mem_db_async: AsyncConnection):
    data = pa.table({"id": [0, 1]})
    table = await mem_db_async.create_table("my_table", data=data)
-    await table.add_columns(
+    add_columns_res = await table.add_columns(
        [pa.field("x", pa.int64()), pa.field("vector", pa.list_(pa.float32(), 8))]
    )
    assert add_columns_res.version == 2
    assert await table.schema() == pa.schema(
        [
@@ -1406,11 +1586,12 @@ async def test_add_columns_with_schema(mem_db_async: AsyncConnection):
    )
    table = await mem_db_async.create_table("table2", data=data)
-    await table.add_columns(
+    add_columns_res = await table.add_columns(
        pa.schema(
            [pa.field("y", pa.int64()), pa.field("emb", pa.list_(pa.float32(), 8))]
        )
    )
    assert add_columns_res.version == 2
    assert await table.schema() == pa.schema(
        [
            pa.field("id", pa.int64()),
@@ -1423,7 +1604,8 @@ async def test_add_columns_with_schema(mem_db_async: AsyncConnection):
 def test_alter_columns(mem_db: DBConnection):
    data = pa.table({"id": [0, 1]})
    table = mem_db.create_table("my_table", data=data)
-    table.alter_columns({"path": "id", "rename": "new_id"})
+    alter_columns_res = table.alter_columns({"path": "id", "rename": "new_id"})
    assert alter_columns_res.version == 2
    assert table.to_arrow().column_names == ["new_id"]
@@ -1431,9 +1613,13 @@ def test_alter_columns(mem_db: DBConnection):
 async def test_alter_columns_async(mem_db_async: AsyncConnection):
    data = pa.table({"id": [0, 1]})
    table = await mem_db_async.create_table("my_table", data=data)
-    await table.alter_columns({"path": "id", "rename": "new_id"})
+    alter_columns_res = await table.alter_columns({"path": "id", "rename": "new_id"})
    assert alter_columns_res.version == 2
    assert (await table.to_arrow()).column_names == ["new_id"]
-    await table.alter_columns(dict(path="new_id", data_type=pa.int16(), nullable=True))
+    alter_columns_res = await table.alter_columns(
        dict(path="new_id", data_type=pa.int16(), nullable=True)
    )
    assert alter_columns_res.version == 3
    data = await table.to_arrow()
    assert data.column(0).type == pa.int16()
    assert data.schema.field(0).nullable
@@ -1442,7 +1628,8 @@ async def test_alter_columns_async(mem_db_async: AsyncConnection):
 def test_drop_columns(mem_db: DBConnection):
    data = pa.table({"id": [0, 1], "category": ["a", "b"]})
    table = mem_db.create_table("my_table", data=data)
-    table.drop_columns(["category"])
+    drop_columns_res = table.drop_columns(["category"])
    assert drop_columns_res.version == 2
    assert table.to_arrow().column_names == ["id"]
@@ -1450,7 +1637,8 @@ def test_drop_columns(mem_db: DBConnection):
 async def test_drop_columns_async(mem_db_async: AsyncConnection):
    data = pa.table({"id": [0, 1], "category": ["a", "b"]})
    table = await mem_db_async.create_table("my_table", data=data)
-    await table.drop_columns(["category"])
+    drop_columns_res = await table.drop_columns(["category"])
    assert drop_columns_res.version == 2
    assert (await table.to_arrow()).column_names == ["id"]
@@ -1588,3 +1776,31 @@ def test_replace_field_metadata(tmp_path):
    schema = table.schema
    field = schema[0].metadata
    assert field == {b"foo": b"bar"}
 def test_stats(mem_db: DBConnection):
    table = mem_db.create_table(
        "my_table",
        data=[{"text": "foo", "id": 0}, {"text": "bar", "id": 1}],
    )
    assert len(table) == 2
    stats = table.stats()
    print(f"{stats=}")
    assert stats == {
        "total_bytes": 38,
        "num_rows": 2,
        "num_indices": 0,
        "fragment_stats": {
            "num_fragments": 1,
            "num_small_fragments": 1,
            "lengths": {
                "min": 2,
                "max": 2,
                "mean": 2,
                "p25": 2,
                "p50": 2,
                "p75": 2,
                "p99": 2,
            },
        },
    }
--- a/python/src/index.rs
+++ b/python/src/index.rs
@@ -3,7 +3,7 @@
 use lancedb::index::vector::IvfFlatIndexBuilder;
 use lancedb::index::{
-    scalar::{BTreeIndexBuilder, FtsIndexBuilder, TokenizerConfig},
+    scalar::{BTreeIndexBuilder, FtsIndexBuilder},
    vector::{IvfHnswPqIndexBuilder, IvfHnswSqIndexBuilder, IvfPqIndexBuilder},
    Index as LanceDbIndex,
 };
@@ -38,19 +38,17 @@ pub fn extract_index_params(source: &Option<Bound<'_, PyAny>>) -> PyResult<Lance
            "LabelList" => Ok(LanceDbIndex::LabelList(Default::default())),
            "FTS" => {
                let params = source.extract::<FtsParams>()?;
-                let inner_opts = TokenizerConfig::default()
+                let inner_opts = FtsIndexBuilder::default()
                    .base_tokenizer(params.base_tokenizer)
                    .language(&params.language)
                    .map_err(|_| PyValueError::new_err(format!("LanceDB does not support the requested language: '{}'", params.language)))?
                    .with_position(params.with_position)
                    .lower_case(params.lower_case)
                    .max_token_length(params.max_token_length)
                    .remove_stop_words(params.remove_stop_words)
                    .stem(params.stem)
                    .ascii_folding(params.ascii_folding);
-                let mut opts = FtsIndexBuilder::default()
+                Ok(LanceDbIndex::FTS(inner_opts))
                    .with_position(params.with_position);
                opts.tokenizer_configs = inner_opts;
                Ok(LanceDbIndex::FTS(opts))
            },
            "IvfFlat" => {
                let params = source.extract::<IvfFlatParams>()?;
--- a/python/src/lib.rs
+++ b/python/src/lib.rs
@@ -11,7 +11,10 @@ use pyo3::{
    wrap_pyfunction, Bound, PyResult, Python,
 };
 use query::{FTSQuery, HybridQuery, Query, VectorQuery};
-use table::Table;
+use table::{
    AddColumnsResult, AddResult, AlterColumnsResult, DeleteResult, DropColumnsResult, MergeResult,
    Table, UpdateResult,
 };
 pub mod arrow;
 pub mod connection;
@@ -35,6 +38,13 @@ pub fn _lancedb(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> {
    m.add_class::<HybridQuery>()?;
    m.add_class::<VectorQuery>()?;
    m.add_class::<RecordBatchStream>()?;
    m.add_class::<AddColumnsResult>()?;
    m.add_class::<AlterColumnsResult>()?;
    m.add_class::<AddResult>()?;
    m.add_class::<MergeResult>()?;
    m.add_class::<DeleteResult>()?;
    m.add_class::<DropColumnsResult>()?;
    m.add_class::<UpdateResult>()?;
    m.add_function(wrap_pyfunction!(connect, m)?)?;
    m.add_function(wrap_pyfunction!(util::validate_table_name, m)?)?;
    m.add("__version__", env!("CARGO_PKG_VERSION"))?;
--- a/python/src/table.rs
+++ b/python/src/table.rs
@@ -2,6 +2,11 @@
 // SPDX-FileCopyrightText: Copyright The LanceDB Authors
 use std::{collections::HashMap, sync::Arc};
 use crate::{
    error::PythonErrorExt,
    index::{extract_index_params, IndexConfig},
    query::Query,
 };
 use arrow::{
    datatypes::{DataType, Schema},
    ffi_stream::ArrowArrayStreamReader,
@@ -19,12 +24,6 @@ use pyo3::{
 };
 use pyo3_async_runtimes::tokio::future_into_py;
 use crate::{
    error::PythonErrorExt,
    index::{extract_index_params, IndexConfig},
    query::Query,
 };
 /// Statistics about a compaction operation.
 #[pyclass(get_all)]
 #[derive(Clone, Debug)]
@@ -59,6 +58,170 @@ pub struct OptimizeStats {
    pub prune: RemovalStats,
 }
 #[pyclass(get_all)]
 #[derive(Clone, Debug)]
 pub struct UpdateResult {
    pub rows_updated: u64,
    pub version: u64,
 }
 #[pymethods]
 impl UpdateResult {
    pub fn __repr__(&self) -> String {
        format!(
            "UpdateResult(rows_updated={}, version={})",
            self.rows_updated, self.version
        )
    }
 }
 impl From<lancedb::table::UpdateResult> for UpdateResult {
    fn from(result: lancedb::table::UpdateResult) -> Self {
        Self {
            rows_updated: result.rows_updated,
            version: result.version,
        }
    }
 }
 #[pyclass(get_all)]
 #[derive(Clone, Debug)]
 pub struct AddResult {
    pub version: u64,
 }
 #[pymethods]
 impl AddResult {
    pub fn __repr__(&self) -> String {
        format!("AddResult(version={})", self.version)
    }
 }
 impl From<lancedb::table::AddResult> for AddResult {
    fn from(result: lancedb::table::AddResult) -> Self {
        Self {
            version: result.version,
        }
    }
 }
 #[pyclass(get_all)]
 #[derive(Clone, Debug)]
 pub struct DeleteResult {
    pub version: u64,
 }
 #[pymethods]
 impl DeleteResult {
    pub fn __repr__(&self) -> String {
        format!("DeleteResult(version={})", self.version)
    }
 }
 impl From<lancedb::table::DeleteResult> for DeleteResult {
    fn from(result: lancedb::table::DeleteResult) -> Self {
        Self {
            version: result.version,
        }
    }
 }
 #[pyclass(get_all)]
 #[derive(Clone, Debug)]
 pub struct MergeResult {
    pub version: u64,
    pub num_updated_rows: u64,
    pub num_inserted_rows: u64,
    pub num_deleted_rows: u64,
 }
 #[pymethods]
 impl MergeResult {
    pub fn __repr__(&self) -> String {
        format!(
            "MergeResult(version={}, num_updated_rows={}, num_inserted_rows={}, num_deleted_rows={})",
            self.version,
            self.num_updated_rows,
            self.num_inserted_rows,
            self.num_deleted_rows
        )
    }
 }
 impl From<lancedb::table::MergeResult> for MergeResult {
    fn from(result: lancedb::table::MergeResult) -> Self {
        Self {
            version: result.version,
            num_updated_rows: result.num_updated_rows,
            num_inserted_rows: result.num_inserted_rows,
            num_deleted_rows: result.num_deleted_rows,
        }
    }
 }
 #[pyclass(get_all)]
 #[derive(Clone, Debug)]
 pub struct AddColumnsResult {
    pub version: u64,
 }
 #[pymethods]
 impl AddColumnsResult {
    pub fn __repr__(&self) -> String {
        format!("AddColumnsResult(version={})", self.version)
    }
 }
 impl From<lancedb::table::AddColumnsResult> for AddColumnsResult {
    fn from(result: lancedb::table::AddColumnsResult) -> Self {
        Self {
            version: result.version,
        }
    }
 }
 #[pyclass(get_all)]
 #[derive(Clone, Debug)]
 pub struct AlterColumnsResult {
    pub version: u64,
 }
 #[pymethods]
 impl AlterColumnsResult {
    pub fn __repr__(&self) -> String {
        format!("AlterColumnsResult(version={})", self.version)
    }
 }
 impl From<lancedb::table::AlterColumnsResult> for AlterColumnsResult {
    fn from(result: lancedb::table::AlterColumnsResult) -> Self {
        Self {
            version: result.version,
        }
    }
 }
 #[pyclass(get_all)]
 #[derive(Clone, Debug)]
 pub struct DropColumnsResult {
    pub version: u64,
 }
 #[pymethods]
 impl DropColumnsResult {
    pub fn __repr__(&self) -> String {
        format!("DropColumnsResult(version={})", self.version)
    }
 }
 impl From<lancedb::table::DropColumnsResult> for DropColumnsResult {
    fn from(result: lancedb::table::DropColumnsResult) -> Self {
        Self {
            version: result.version,
        }
    }
 }
 #[pyclass]
 pub struct Table {
    // We keep a copy of the name to use if the inner table is dropped
@@ -133,15 +296,16 @@ impl Table {
        }
        future_into_py(self_.py(), async move {
-            op.execute().await.infer_error()?;
+            let result = op.execute().await.infer_error()?;
-            Ok(())
+            Ok(AddResult::from(result))
        })
    }
    pub fn delete(self_: PyRef<'_, Self>, condition: String) -> PyResult<Bound<'_, PyAny>> {
        let inner = self_.inner_ref()?.clone();
        future_into_py(self_.py(), async move {
-            inner.delete(&condition).await.infer_error()
+            let result = inner.delete(&condition).await.infer_error()?;
            Ok(DeleteResult::from(result))
        })
    }
@@ -161,8 +325,8 @@ impl Table {
            op = op.column(column_name, value);
        }
        future_into_py(self_.py(), async move {
-            op.execute().await.infer_error()?;
+            let result = op.execute().await.infer_error()?;
-            Ok(())
+            Ok(UpdateResult::from(result))
        })
    }
@@ -280,6 +444,40 @@ impl Table {
        })
    }
    pub fn stats(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
        let inner = self_.inner_ref()?.clone();
        future_into_py(self_.py(), async move {
            let stats = inner.stats().await.infer_error()?;
            Python::with_gil(|py| {
                let dict = PyDict::new(py);
                dict.set_item("total_bytes", stats.total_bytes)?;
                dict.set_item("num_rows", stats.num_rows)?;
                dict.set_item("num_indices", stats.num_indices)?;
                let fragment_stats = PyDict::new(py);
                fragment_stats.set_item("num_fragments", stats.fragment_stats.num_fragments)?;
                fragment_stats.set_item(
                    "num_small_fragments",
                    stats.fragment_stats.num_small_fragments,
                )?;
                let fragment_lengths = PyDict::new(py);
                fragment_lengths.set_item("min", stats.fragment_stats.lengths.min)?;
                fragment_lengths.set_item("max", stats.fragment_stats.lengths.max)?;
                fragment_lengths.set_item("mean", stats.fragment_stats.lengths.mean)?;
                fragment_lengths.set_item("p25", stats.fragment_stats.lengths.p25)?;
                fragment_lengths.set_item("p50", stats.fragment_stats.lengths.p50)?;
                fragment_lengths.set_item("p75", stats.fragment_stats.lengths.p75)?;
                fragment_lengths.set_item("p99", stats.fragment_stats.lengths.p99)?;
                fragment_stats.set_item("lengths", fragment_lengths)?;
                dict.set_item("fragment_stats", fragment_stats)?;
                Ok(Some(dict.unbind()))
            })
        })
    }
    pub fn __repr__(&self) -> String {
        match &self.inner {
            None => format!("ClosedTable({})", self.name),
@@ -322,10 +520,16 @@ impl Table {
        })
    }
-    pub fn checkout(self_: PyRef<'_, Self>, version: u64) -> PyResult<Bound<'_, PyAny>> {
+    pub fn checkout(self_: PyRef<'_, Self>, version: LanceVersion) -> PyResult<Bound<'_, PyAny>> {
        let inner = self_.inner_ref()?.clone();
-        future_into_py(self_.py(), async move {
+        let py = self_.py();
-            inner.checkout(version).await.infer_error()
+        future_into_py(py, async move {
            match version {
                LanceVersion::Version(version_num) => {
                    inner.checkout(version_num).await.infer_error()
                }
                LanceVersion::Tag(tag) => inner.checkout_tag(&tag).await.infer_error(),
            }
        })
    }
@@ -337,12 +541,19 @@ impl Table {
    }
    #[pyo3(signature = (version=None))]
-    pub fn restore(self_: PyRef<'_, Self>, version: Option<u64>) -> PyResult<Bound<'_, PyAny>> {
+    pub fn restore(
        self_: PyRef<'_, Self>,
        version: Option<LanceVersion>,
    ) -> PyResult<Bound<'_, PyAny>> {
        let inner = self_.inner_ref()?.clone();
        let py = self_.py();
-        future_into_py(self_.py(), async move {
+        future_into_py(py, async move {
            if let Some(version) = version {
-                inner.checkout(version).await.infer_error()?;
+                match version {
                    LanceVersion::Version(num) => inner.checkout(num).await.infer_error()?,
                    LanceVersion::Tag(tag) => inner.checkout_tag(&tag).await.infer_error()?,
                }
            }
            inner.restore().await.infer_error()
        })
@@ -352,6 +563,11 @@ impl Table {
        Query::new(self.inner_ref().unwrap().query())
    }
    #[getter]
    pub fn tags(&self) -> PyResult<Tags> {
        Ok(Tags::new(self.inner_ref()?.clone()))
    }
    /// Optimize the on-disk data by compacting and pruning old data, for better performance.
    #[pyo3(signature = (cleanup_since_ms=None, delete_unverified=None, retrain=None))]
    pub fn optimize(
@@ -433,10 +649,13 @@ impl Table {
            builder
                .when_not_matched_by_source_delete(parameters.when_not_matched_by_source_condition);
        }
        if let Some(timeout) = parameters.timeout {
            builder.timeout(timeout);
        }
        future_into_py(self_.py(), async move {
-            builder.execute(Box::new(batches)).await.infer_error()?;
+            let res = builder.execute(Box::new(batches)).await.infer_error()?;
-            Ok(())
+            Ok(MergeResult::from(res))
        })
    }
@@ -472,8 +691,8 @@ impl Table {
        let inner = self_.inner_ref()?.clone();
        future_into_py(self_.py(), async move {
-            inner.add_columns(definitions, None).await.infer_error()?;
+            let result = inner.add_columns(definitions, None).await.infer_error()?;
-            Ok(())
+            Ok(AddColumnsResult::from(result))
        })
    }
@@ -486,8 +705,8 @@ impl Table {
        let inner = self_.inner_ref()?.clone();
        future_into_py(self_.py(), async move {
-            inner.add_columns(transform, None).await.infer_error()?;
+            let result = inner.add_columns(transform, None).await.infer_error()?;
-            Ok(())
+            Ok(AddColumnsResult::from(result))
        })
    }
@@ -530,8 +749,8 @@ impl Table {
        let inner = self_.inner_ref()?.clone();
        future_into_py(self_.py(), async move {
-            inner.alter_columns(&alterations).await.infer_error()?;
+            let result = inner.alter_columns(&alterations).await.infer_error()?;
-            Ok(())
+            Ok(AlterColumnsResult::from(result))
        })
    }
@@ -539,8 +758,8 @@ impl Table {
        let inner = self_.inner_ref()?.clone();
        future_into_py(self_.py(), async move {
            let column_refs = columns.iter().map(String::as_str).collect::<Vec<&str>>();
-            inner.drop_columns(&column_refs).await.infer_error()?;
+            let result = inner.drop_columns(&column_refs).await.infer_error()?;
-            Ok(())
+            Ok(DropColumnsResult::from(result))
        })
    }
@@ -576,6 +795,12 @@ impl Table {
    }
 }
 #[derive(FromPyObject)]
 pub enum LanceVersion {
    Version(u64),
    Tag(String),
 }
 #[derive(FromPyObject)]
 #[pyo3(from_item_all)]
 pub struct MergeInsertParams {
@@ -585,4 +810,74 @@ pub struct MergeInsertParams {
    when_not_matched_insert_all: bool,
    when_not_matched_by_source_delete: bool,
    when_not_matched_by_source_condition: Option<String>,
    timeout: Option<std::time::Duration>,
 }
 #[pyclass]
 pub struct Tags {
    inner: LanceDbTable,
 }
 impl Tags {
    pub fn new(table: LanceDbTable) -> Self {
        Self { inner: table }
    }
 }
 #[pymethods]
 impl Tags {
    pub fn list(self_: PyRef<'_, Self>) -> PyResult<Bound<'_, PyAny>> {
        let inner = self_.inner.clone();
        future_into_py(self_.py(), async move {
            let tags = inner.tags().await.infer_error()?;
            let res = tags.list().await.infer_error()?;
            Python::with_gil(|py| {
                let py_dict = PyDict::new(py);
                for (key, contents) in res {
                    let value_dict = PyDict::new(py);
                    value_dict.set_item("version", contents.version)?;
                    value_dict.set_item("manifest_size", contents.manifest_size)?;
                    py_dict.set_item(key, value_dict)?;
                }
                Ok(py_dict.unbind())
            })
        })
    }
    pub fn get_version(self_: PyRef<'_, Self>, tag: String) -> PyResult<Bound<'_, PyAny>> {
        let inner = self_.inner.clone();
        future_into_py(self_.py(), async move {
            let tags = inner.tags().await.infer_error()?;
            let res = tags.get_version(tag.as_str()).await.infer_error()?;
            Ok(res)
        })
    }
    pub fn create(self_: PyRef<Self>, tag: String, version: u64) -> PyResult<Bound<PyAny>> {
        let inner = self_.inner.clone();
        future_into_py(self_.py(), async move {
            let mut tags = inner.tags().await.infer_error()?;
            tags.create(tag.as_str(), version).await.infer_error()?;
            Ok(())
        })
    }
    pub fn delete(self_: PyRef<Self>, tag: String) -> PyResult<Bound<PyAny>> {
        let inner = self_.inner.clone();
        future_into_py(self_.py(), async move {
            let mut tags = inner.tags().await.infer_error()?;
            tags.delete(tag.as_str()).await.infer_error()?;
            Ok(())
        })
    }
    pub fn update(self_: PyRef<Self>, tag: String, version: u64) -> PyResult<Bound<PyAny>> {
        let inner = self_.inner.clone();
        future_into_py(self_.py(), async move {
            let mut tags = inner.tags().await.infer_error()?;
            tags.update(tag.as_str(), version).await.infer_error()?;
            Ok(())
        })
    }
 }
--- a/rust-toolchain.toml
+++ b/rust-toolchain.toml
@@ -1,2 +1,2 @@
 [toolchain]
-channel = "1.83.0"
+channel = "1.86.0"
--- a/rust/ffi/node/Cargo.toml
+++ b/rust/ffi/node/Cargo.toml
@@ -1,6 +1,6 @@
 [package]
 name = "lancedb-node"
-version = "0.19.0-beta.11"
+version = "0.20.0-beta.2"
 description = "Serverless, low-latency vector database for AI applications"
 license.workspace = true
 edition.workspace = true
--- a/rust/lancedb/Cargo.toml
+++ b/rust/lancedb/Cargo.toml
@@ -1,6 +1,6 @@
 [package]
 name = "lancedb"
-version = "0.19.0-beta.11"
+version = "0.20.0-beta.2"
 edition.workspace = true
 description = "LanceDB: A serverless, low-latency vector database for AI applications"
 license.workspace = true
@@ -60,15 +60,15 @@ reqwest = { version = "0.12.0", default-features = false, features = [
    "macos-system-configuration",
    "stream",
 ], optional = true }
-rand = { version = "0.8.3", features = ["small_rng"], optional = true }
+rand = { version = "0.9", features = ["small_rng"], optional = true }
 http = { version = "1", optional = true } # Matching what is in reqwest
 uuid = { version = "1.7.0", features = ["v4"], optional = true }
 polars-arrow = { version = ">=0.37,<0.40.0", optional = true }
 polars = { version = ">=0.37,<0.40.0", optional = true }
 hf-hub = { version = "0.4.1", optional = true, default-features = false, features = ["rustls-tls", "tokio", "ureq"]}
-candle-core = { version = "0.6.0", optional = true }
+candle-core = { version = "0.9.1", optional = true }
-candle-transformers = { version = "0.6.0", optional = true }
+candle-transformers = { version = "0.9.1", optional = true }
-candle-nn = { version = "0.6.0", optional = true }
+candle-nn = { version = "0.9.1", optional = true }
 tokenizers = { version = "0.19.1", optional = true }
 semver = { workspace = true }
@@ -78,7 +78,7 @@ bytemuck_derive.workspace = true
 [dev-dependencies]
 tempfile = "3.5.0"
-rand = { version = "0.8.3", features = ["small_rng"] }
+rand = { version = "0.9", features = ["small_rng"] }
 random_word = { version = "0.4.3", features = ["en"] }
 uuid = { version = "1.7.0", features = ["v4"] }
 walkdir = "2"
--- a/rust/lancedb/examples/full_text_search.rs
+++ b/rust/lancedb/examples/full_text_search.rs
@@ -51,7 +51,7 @@ fn create_some_records() -> Result<Box<dyn RecordBatchReader + Send>> {
                Arc::new(Int32Array::from_iter_values(0..TOTAL as i32)),
                Arc::new(StringArray::from_iter_values((0..TOTAL).map(|_| {
                    (0..n_terms)
-                        .map(|_| words[random::<usize>() % words.len()])
+                        .map(|_| words[random::<u32>() as usize % words.len()])
                        .collect::<Vec<_>>()
                        .join(" ")
                }))),
--- a/rust/lancedb/src/embeddings/sentence_transformers.rs
+++ b/rust/lancedb/src/embeddings/sentence_transformers.rs
@@ -214,7 +214,7 @@ impl SentenceTransformersEmbeddings {
        let embeddings = self
            .model
-            .forward(&input_ids, &token_type_ids)
+            .forward(&input_ids, &token_type_ids, None)
            // TODO: it'd be nice to support other devices
            .and_then(|output| output.to_device(&Device::Cpu))?;
@@ -310,7 +310,7 @@ impl SentenceTransformersEmbeddings {
        let embeddings = Tensor::stack(&tokens, 0)
            .and_then(|tokens| {
                let token_type_ids = tokens.zeros_like()?;
-                self.model.forward(&tokens, &token_type_ids)
+                self.model.forward(&tokens, &token_type_ids, None)
            })
            // TODO: it'd be nice to support other devices
            .and_then(|tokens| tokens.to_device(&Device::Cpu))
--- a/rust/lancedb/src/index/scalar.rs
+++ b/rust/lancedb/src/index/scalar.rs
@@ -51,35 +51,7 @@ pub struct BitmapIndexBuilder {}
 #[derive(Debug, Clone, Default)]
 pub struct LabelListIndexBuilder {}
 /// Builder for a full text search index
 ///
 /// A full text search index is an index on a string column that allows for full text search
 #[derive(Debug, Clone)]
 pub struct FtsIndexBuilder {
    /// Whether to store the position of the tokens
    /// This is used for phrase queries
    pub with_position: bool,
    pub tokenizer_configs: TokenizerConfig,
 }
 impl Default for FtsIndexBuilder {
    fn default() -> Self {
        Self {
            with_position: true,
            tokenizer_configs: TokenizerConfig::default(),
        }
    }
 }
 impl FtsIndexBuilder {
    /// Set the with_position flag
    pub fn with_position(mut self, with_position: bool) -> Self {
        self.with_position = with_position;
        self
    }
 }
 pub use lance_index::scalar::inverted::query::*;
 pub use lance_index::scalar::inverted::TokenizerConfig;
 pub use lance_index::scalar::FullTextSearchQuery;
 pub use lance_index::scalar::InvertedIndexParams as FtsIndexBuilder;
 pub use lance_index::scalar::InvertedIndexParams;
--- a/rust/lancedb/src/io/object_store.rs
+++ b/rust/lancedb/src/io/object_store.rs
@@ -197,16 +197,8 @@ mod test {
    #[tokio::test]
    async fn test_e2e() {
-        let dir1 = tempfile::tempdir()
+        let dir1 = tempfile::tempdir().unwrap().keep().canonicalize().unwrap();
-            .unwrap()
+        let dir2 = tempfile::tempdir().unwrap().keep().canonicalize().unwrap();
            .into_path()
            .canonicalize()
            .unwrap();
        let dir2 = tempfile::tempdir()
            .unwrap()
            .into_path()
            .canonicalize()
            .unwrap();
        let secondary_store = LocalFileSystem::new_with_prefix(dir2.to_str().unwrap()).unwrap();
        let object_store_wrapper = Arc::new(MirroringObjectStoreWrapper {
--- a/rust/lancedb/src/remote/table.rs
+++ b/rust/lancedb/src/remote/table.rs
--- a/rust/lancedb/src/table.rs
+++ b/rust/lancedb/src/table.rs
@@ -14,7 +14,7 @@ use datafusion_physical_plan::projection::ProjectionExec;
 use datafusion_physical_plan::repartition::RepartitionExec;
 use datafusion_physical_plan::union::UnionExec;
 use datafusion_physical_plan::ExecutionPlan;
-use futures::{StreamExt, TryStreamExt};
+use futures::{FutureExt, StreamExt, TryFutureExt};
 use lance::dataset::builder::DatasetBuilder;
 use lance::dataset::cleanup::RemovalStats;
 use lance::dataset::optimize::{compact_files, CompactionMetrics, IndexRemapperOptions};
@@ -80,9 +80,14 @@ pub mod merge;
 use crate::index::waiter::wait_for_index;
 pub use chrono::Duration;
 use futures::future::{join_all, Either};
 pub use lance::dataset::optimize::CompactionOptions;
 pub use lance::dataset::refs::{TagContents, Tags as LanceTags};
 pub use lance::dataset::scanner::DatasetRecordBatchStream;
 use lance::dataset::statistics::DatasetStatisticsExt;
 use lance_index::frag_reuse::FRAG_REUSE_INDEX_NAME;
 pub use lance_index::optimize::OptimizeOptions;
 use serde_with::skip_serializing_none;
 /// Defines the type of column
 #[derive(Debug, Clone, Serialize, Deserialize)]
@@ -307,7 +312,7 @@ impl<T: IntoArrow> AddDataBuilder<T> {
        self
    }
-    pub async fn execute(self) -> Result<()> {
+    pub async fn execute(self) -> Result<AddResult> {
        let parent = self.parent.clone();
        let data = self.data.into_arrow()?;
        let without_data = AddDataBuilder::<NoData> {
@@ -375,8 +380,8 @@ impl UpdateBuilder {
    }
    /// Executes the update operation.
-    /// Returns the number of rows that were updated.
+    /// Returns the update result
-    pub async fn execute(self) -> Result<u64> {
+    pub async fn execute(self) -> Result<UpdateResult> {
        if self.columns.is_empty() {
            Err(Error::InvalidInput {
                message: "at least one column must be specified in an update operation".to_string(),
@@ -401,6 +406,100 @@ pub enum AnyQuery {
    VectorQuery(VectorQueryRequest),
 }
 #[async_trait]
 pub trait Tags: Send + Sync {
    /// List the tags of the table.
    async fn list(&self) -> Result<HashMap<String, TagContents>>;
    /// Get the version of the table referenced by a tag.
    async fn get_version(&self, tag: &str) -> Result<u64>;
    /// Create a new tag for the given version of the table.
    async fn create(&mut self, tag: &str, version: u64) -> Result<()>;
    /// Delete a tag from the table.
    async fn delete(&mut self, tag: &str) -> Result<()>;
    /// Update an existing tag to point to a new version of the table.
    async fn update(&mut self, tag: &str, version: u64) -> Result<()>;
 }
 #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
 pub struct UpdateResult {
    #[serde(default)]
    pub rows_updated: u64,
    // The commit version associated with the operation.
    // A version of `0` indicates compatibility with legacy servers that do not return
    /// a commit version.
    #[serde(default)]
    pub version: u64,
 }
 #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
 pub struct AddResult {
    // The commit version associated with the operation.
    // A version of `0` indicates compatibility with legacy servers that do not return
    /// a commit version.
    #[serde(default)]
    pub version: u64,
 }
 #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
 pub struct DeleteResult {
    // The commit version associated with the operation.
    // A version of `0` indicates compatibility with legacy servers that do not return
    /// a commit version.
    #[serde(default)]
    pub version: u64,
 }
 #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
 pub struct MergeResult {
    // The commit version associated with the operation.
    // A version of `0` indicates compatibility with legacy servers that do not return
    /// a commit version.
    #[serde(default)]
    pub version: u64,
    /// Number of inserted rows (for user statistics)
    #[serde(default)]
    pub num_inserted_rows: u64,
    /// Number of updated rows (for user statistics)
    #[serde(default)]
    pub num_updated_rows: u64,
    /// Number of deleted rows (for user statistics)
    /// Note: This is different from internal references to 'deleted_rows', since we technically "delete" updated rows during processing.
    /// However those rows are not shared with the user.
    #[serde(default)]
    pub num_deleted_rows: u64,
 }
 #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
 pub struct AddColumnsResult {
    // The commit version associated with the operation.
    // A version of `0` indicates compatibility with legacy servers that do not return
    /// a commit version.
    #[serde(default)]
    pub version: u64,
 }
 #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
 pub struct AlterColumnsResult {
    // The commit version associated with the operation.
    // A version of `0` indicates compatibility with legacy servers that do not return
    /// a commit version.
    #[serde(default)]
    pub version: u64,
 }
 #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
 pub struct DropColumnsResult {
    // The commit version associated with the operation.
    // A version of `0` indicates compatibility with legacy servers that do not return
    /// a commit version.
    #[serde(default)]
    pub version: u64,
 }
 /// A trait for anything "table-like".  This is used for both native tables (which target
 /// Lance datasets) and remote tables (which target LanceDB cloud)
 ///
@@ -445,11 +544,11 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
        &self,
        add: AddDataBuilder<NoData>,
        data: Box<dyn arrow_array::RecordBatchReader + Send>,
-    ) -> Result<()>;
+    ) -> Result<AddResult>;
    /// Delete rows from the table.
-    async fn delete(&self, predicate: &str) -> Result<()>;
+    async fn delete(&self, predicate: &str) -> Result<DeleteResult>;
    /// Update rows in the table.
-    async fn update(&self, update: UpdateBuilder) -> Result<u64>;
+    async fn update(&self, update: UpdateBuilder) -> Result<UpdateResult>;
    /// Create an index on the provided column(s).
    async fn create_index(&self, index: IndexBuilder) -> Result<()>;
    /// List the indices on the table.
@@ -465,7 +564,9 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
        &self,
        params: MergeInsertBuilder,
        new_data: Box<dyn RecordBatchReader + Send>,
-    ) -> Result<()>;
+    ) -> Result<MergeResult>;
    /// Gets the table tag manager.
    async fn tags(&self) -> Result<Box<dyn Tags + '_>>;
    /// Optimize the dataset.
    async fn optimize(&self, action: OptimizeAction) -> Result<OptimizeStats>;
    /// Add columns to the table.
@@ -473,15 +574,18 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
        &self,
        transforms: NewColumnTransform,
        read_columns: Option<Vec<String>>,
-    ) -> Result<()>;
+    ) -> Result<AddColumnsResult>;
    /// Alter columns in the table.
-    async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<()>;
+    async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult>;
    /// Drop columns from the table.
-    async fn drop_columns(&self, columns: &[&str]) -> Result<()>;
+    async fn drop_columns(&self, columns: &[&str]) -> Result<DropColumnsResult>;
    /// Get the version of the table.
    async fn version(&self) -> Result<u64>;
    /// Checkout a specific version of the table.
    async fn checkout(&self, version: u64) -> Result<()>;
    /// Checkout a table version referenced by a tag.
    /// Tags provide a human-readable way to reference specific versions of the table.
    async fn checkout_tag(&self, tag: &str) -> Result<()>;
    /// Checkout the latest version of the table.
    async fn checkout_latest(&self) -> Result<()>;
    /// Restore the table to the currently checked out version.
@@ -499,6 +603,8 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
        index_names: &[&str],
        timeout: std::time::Duration,
    ) -> Result<()>;
    /// Get statistics on the table
    async fn stats(&self) -> Result<TableStatistics>;
 }
 /// A Table is a collection of strong typed Rows.
@@ -701,7 +807,7 @@ impl Table {
    /// tbl.delete("id > 5").await.unwrap();
    /// # });
    /// ```
-    pub async fn delete(&self, predicate: &str) -> Result<()> {
+    pub async fn delete(&self, predicate: &str) -> Result<DeleteResult> {
        self.inner.delete(predicate).await
    }
@@ -1016,17 +1122,20 @@ impl Table {
        &self,
        transforms: NewColumnTransform,
        read_columns: Option<Vec<String>>,
-    ) -> Result<()> {
+    ) -> Result<AddColumnsResult> {
        self.inner.add_columns(transforms, read_columns).await
    }
    /// Change a column's name or nullability.
-    pub async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<()> {
+    pub async fn alter_columns(
        &self,
        alterations: &[ColumnAlteration],
    ) -> Result<AlterColumnsResult> {
        self.inner.alter_columns(alterations).await
    }
    /// Remove columns from the table.
-    pub async fn drop_columns(&self, columns: &[&str]) -> Result<()> {
+    pub async fn drop_columns(&self, columns: &[&str]) -> Result<DropColumnsResult> {
        self.inner.drop_columns(columns).await
    }
@@ -1058,6 +1167,24 @@ impl Table {
        self.inner.checkout(version).await
    }
    /// Checks out a specific version of the Table by tag
    ///
    /// Any read operation on the table will now access the data at the version referenced by the tag.
    /// As a consequence, calling this method will disable any read consistency interval
    /// that was previously set.
    ///
    /// This is a read-only operation that turns the table into a sort of "view"
    /// or "detached head".  Other table instances will not be affected.  To make the change
    /// permanent you can use the `[Self::restore]` method.
    ///
    /// Any operation that modifies the table will fail while the table is in a checked
    /// out state.
    ///
    /// To return the table to a normal state use `[Self::checkout_latest]`
    pub async fn checkout_tag(&self, tag: &str) -> Result<()> {
        self.inner.checkout_tag(tag).await
    }
    /// Ensures the table is pointing at the latest version
    ///
    /// This can be used to manually update a table when the read_consistency_interval is None
@@ -1144,6 +1271,11 @@ impl Table {
        self.inner.wait_for_index(index_names, timeout).await
    }
    /// Get the tags manager.
    pub async fn tags(&self) -> Result<Box<dyn Tags + '_>> {
        self.inner.tags().await
    }
    // Take many execution plans and map them into a single plan that adds
    // a query_index column and unions them.
    pub(crate) fn multi_vector_plan(
@@ -1194,6 +1326,40 @@ impl Table {
        .unwrap();
        Ok(Arc::new(repartitioned))
    }
    /// Retrieve statistics on the table
    pub async fn stats(&self) -> Result<TableStatistics> {
        self.inner.stats().await
    }
 }
 pub struct NativeTags {
    inner: LanceTags,
 }
 #[async_trait]
 impl Tags for NativeTags {
    async fn list(&self) -> Result<HashMap<String, TagContents>> {
        Ok(self.inner.list().await?)
    }
    async fn get_version(&self, tag: &str) -> Result<u64> {
        Ok(self.inner.get_version(tag).await?)
    }
    async fn create(&mut self, tag: &str, version: u64) -> Result<()> {
        self.inner.create(tag, version).await?;
        Ok(())
    }
    async fn delete(&mut self, tag: &str) -> Result<()> {
        self.inner.delete(tag).await?;
        Ok(())
    }
    async fn update(&mut self, tag: &str, version: u64) -> Result<()> {
        self.inner.update(tag, version).await?;
        Ok(())
    }
 }
 impl From<NativeTable> for Table {
@@ -1812,16 +1978,12 @@ impl NativeTable {
        }
        let mut dataset = self.dataset.get_mut().await?;
        let fts_params = lance_index::scalar::InvertedIndexParams {
            with_position: fts_opts.with_position,
            tokenizer_config: fts_opts.tokenizer_configs,
        };
        dataset
            .create_index(
                &[field.name()],
                IndexType::Inverted,
                None,
-                &fts_params,
+                &fts_opts,
                replace,
            )
            .await?;
@@ -1849,7 +2011,7 @@ impl NativeTable {
    /// more information.
    pub async fn uses_v2_manifest_paths(&self) -> Result<bool> {
        let dataset = self.dataset.get().await?;
-        Ok(dataset.manifest_naming_scheme == ManifestNamingScheme::V2)
+        Ok(dataset.manifest_location().naming_scheme == ManifestNamingScheme::V2)
    }
    /// Migrate the table to use the new manifest path scheme.
@@ -1940,6 +2102,10 @@ impl BaseTable for NativeTable {
        self.dataset.as_time_travel(version).await
    }
    async fn checkout_tag(&self, tag: &str) -> Result<()> {
        self.dataset.as_time_travel(tag).await
    }
    async fn checkout_latest(&self) -> Result<()> {
        self.dataset
            .as_latest(self.read_consistency_interval)
@@ -1998,7 +2164,7 @@ impl BaseTable for NativeTable {
        &self,
        add: AddDataBuilder<NoData>,
        data: Box<dyn RecordBatchReader + Send>,
-    ) -> Result<()> {
+    ) -> Result<AddResult> {
        let data = Box::new(MaybeEmbedded::try_new(
            data,
            self.table_definition().await?,
@@ -2021,9 +2187,9 @@ impl BaseTable for NativeTable {
                .execute_stream(data)
                .await?
        };
-
+        let version = dataset.manifest().version;
        self.dataset.set_latest(dataset).await;
-        Ok(())
+        Ok(AddResult { version })
    }
    async fn create_index(&self, opts: IndexBuilder) -> Result<()> {
@@ -2069,7 +2235,7 @@ impl BaseTable for NativeTable {
        Ok(dataset.prewarm_index(index_name).await?)
    }
-    async fn update(&self, update: UpdateBuilder) -> Result<u64> {
+    async fn update(&self, update: UpdateBuilder) -> Result<UpdateResult> {
        let dataset = self.dataset.get().await?.clone();
        let mut builder = LanceUpdateBuilder::new(Arc::new(dataset));
        if let Some(predicate) = update.filter {
@@ -2085,7 +2251,10 @@ impl BaseTable for NativeTable {
        self.dataset
            .set_latest(res.new_dataset.as_ref().clone())
            .await;
-        Ok(res.rows_updated)
+        Ok(UpdateResult {
            rows_updated: res.rows_updated,
            version: res.new_dataset.version().version,
        })
    }
    async fn create_plan(
@@ -2277,7 +2446,7 @@ impl BaseTable for NativeTable {
        &self,
        params: MergeInsertBuilder,
        new_data: Box<dyn RecordBatchReader + Send>,
-    ) -> Result<()> {
+    ) -> Result<MergeResult> {
        let dataset = Arc::new(self.dataset.get().await?.clone());
        let mut builder = LanceMergeInsertBuilder::try_new(dataset.clone(), params.on)?;
        match (
@@ -2303,16 +2472,51 @@ impl BaseTable for NativeTable {
        } else {
            builder.when_not_matched_by_source(WhenNotMatchedBySource::Keep);
        }
-        let job = builder.try_build()?;
+
-        let (new_dataset, _stats) = job.execute_reader(new_data).await?;
+        let future = if let Some(timeout) = params.timeout {
            // The default retry timeout is 30s, so we pass the full timeout down
            // as well in case it is longer than that.
            let future = builder
                .retry_timeout(timeout)
                .try_build()?
                .execute_reader(new_data);
            Either::Left(tokio::time::timeout(timeout, future).map(|res| match res {
                Ok(Ok((new_dataset, stats))) => Ok((new_dataset, stats)),
                Ok(Err(e)) => Err(e.into()),
                Err(_) => Err(Error::Runtime {
                    message: "merge insert timed out".to_string(),
                }),
            }))
        } else {
            let job = builder.try_build()?;
            Either::Right(job.execute_reader(new_data).map_err(|e| e.into()))
        };
        let (new_dataset, stats) = future.await?;
        let version = new_dataset.manifest().version;
        self.dataset.set_latest(new_dataset.as_ref().clone()).await;
-        Ok(())
+        Ok(MergeResult {
            version,
            num_updated_rows: stats.num_updated_rows,
            num_inserted_rows: stats.num_inserted_rows,
            num_deleted_rows: stats.num_deleted_rows,
        })
    }
    /// Delete rows from the table
-    async fn delete(&self, predicate: &str) -> Result<()> {
+    async fn delete(&self, predicate: &str) -> Result<DeleteResult> {
-        self.dataset.get_mut().await?.delete(predicate).await?;
+        let mut dataset = self.dataset.get_mut().await?;
-        Ok(())
+        dataset.delete(predicate).await?;
        Ok(DeleteResult {
            version: dataset.version().version,
        })
    }
    async fn tags(&self) -> Result<Box<dyn Tags + '_>> {
        let dataset = self.dataset.get().await?;
        Ok(Box::new(NativeTags {
            inner: dataset.tags.clone(),
        }))
    }
    async fn optimize(&self, action: OptimizeAction) -> Result<OptimizeStats> {
@@ -2371,54 +2575,83 @@ impl BaseTable for NativeTable {
        &self,
        transforms: NewColumnTransform,
        read_columns: Option<Vec<String>>,
-    ) -> Result<()> {
+    ) -> Result<AddColumnsResult> {
-        self.dataset
+        let mut dataset = self.dataset.get_mut().await?;
-            .get_mut()
+        dataset.add_columns(transforms, read_columns, None).await?;
-            .await?
+        Ok(AddColumnsResult {
-            .add_columns(transforms, read_columns, None)
+            version: dataset.version().version,
-            .await?;
+        })
        Ok(())
    }
-    async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<()> {
+    async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult> {
-        self.dataset
+        let mut dataset = self.dataset.get_mut().await?;
-            .get_mut()
+        dataset.alter_columns(alterations).await?;
-            .await?
+        Ok(AlterColumnsResult {
-            .alter_columns(alterations)
+            version: dataset.version().version,
-            .await?;
+        })
        Ok(())
    }
-    async fn drop_columns(&self, columns: &[&str]) -> Result<()> {
+    async fn drop_columns(&self, columns: &[&str]) -> Result<DropColumnsResult> {
-        self.dataset.get_mut().await?.drop_columns(columns).await?;
+        let mut dataset = self.dataset.get_mut().await?;
-        Ok(())
+        dataset.drop_columns(columns).await?;
        Ok(DropColumnsResult {
            version: dataset.version().version,
        })
    }
    async fn list_indices(&self) -> Result<Vec<IndexConfig>> {
        let dataset = self.dataset.get().await?;
        let indices = dataset.load_indices().await?;
-        futures::stream::iter(indices.as_slice()).then(|idx| async {
+        let results = futures::stream::iter(indices.as_slice()).then(|idx| async {
-            let stats = dataset.index_statistics(idx.name.as_str()).await?;
+
-            let stats: serde_json::Value = serde_json::from_str(&stats).map_err(|e| Error::Runtime {
+            // skip Lance internal indexes
-                message: format!("error deserializing index statistics: {}", e),
+            if idx.name == FRAG_REUSE_INDEX_NAME {
-            })?;
+                return None;
-            let index_type = stats.get("index_type").and_then(|v| v.as_str())
+            }
-            .ok_or_else(|| Error::Runtime {
+
-                message: "index statistics was missing index type".to_string(),
+            let stats = match dataset.index_statistics(idx.name.as_str()).await {
-            })?;
+                Ok(stats) => stats,
-            let index_type: crate::index::IndexType = index_type.parse().map_err(|e| Error::Runtime {
+                Err(e) => {
-                message: format!("error parsing index type: {}", e),
+                    log::warn!("Failed to get statistics for index {} ({}): {}", idx.name, idx.uuid, e);
-            })?;
+                    return None;
                }
            };
            let stats: serde_json::Value = match serde_json::from_str(&stats) {
                Ok(stats) => stats,
                Err(e) => {
                    log::warn!("Failed to deserialize index statistics for index {} ({}): {}", idx.name, idx.uuid, e);
                    return None;
                }
            };
            let Some(index_type) = stats.get("index_type").and_then(|v| v.as_str()) else {
                log::warn!("Index statistics was missing 'index_type' field for index {} ({})", idx.name, idx.uuid);
                return None;
            };
            let index_type: crate::index::IndexType = match index_type.parse() {
                Ok(index_type) => index_type,
                Err(e) => {
                    log::warn!("Failed to parse index type for index {} ({}): {}", idx.name, idx.uuid, e);
                    return None;
                }
            };
            let mut columns = Vec::with_capacity(idx.fields.len());
            for field_id in &idx.fields {
-                let field = dataset.schema().field_by_id(*field_id).ok_or_else(|| Error::Runtime { message: format!("The index with name {} and uuid {} referenced a field with id {} which does not exist in the schema", idx.name, idx.uuid, field_id) })?;
+                let Some(field) = dataset.schema().field_by_id(*field_id) else {
                    log::warn!("The index {} ({}) referenced a field with id {} which does not exist in the schema", idx.name, idx.uuid, field_id);
                    return None;
                };
                columns.push(field.name.clone());
            }
            let name = idx.name.clone();
-            Ok(IndexConfig { index_type, columns, name })
+            Some(IndexConfig { index_type, columns, name })
-        }).try_collect::<Vec<_>>().await
+        }).collect::<Vec<_>>().await;
        Ok(results.into_iter().flatten().collect())
    }
    fn dataset_uri(&self) -> &str {
@@ -2480,6 +2713,108 @@ impl BaseTable for NativeTable {
    ) -> Result<()> {
        wait_for_index(self, index_names, timeout).await
    }
    async fn stats(&self) -> Result<TableStatistics> {
        let num_rows = self.count_rows(None).await?;
        let num_indices = self.list_indices().await?.len();
        let ds = self.dataset.get().await?;
        let ds_clone = (*ds).clone();
        let ds_stats = Arc::new(ds_clone).calculate_data_stats().await?;
        let total_bytes = ds_stats.fields.iter().map(|f| f.bytes_on_disk).sum::<u64>() as usize;
        let frags = ds.get_fragments();
        let mut sorted_sizes = join_all(
            frags
                .iter()
                .map(|frag| async move { frag.physical_rows().await.unwrap_or(0) }),
        )
        .await;
        sorted_sizes.sort();
        let small_frag_threshold = 100000;
        let num_fragments = sorted_sizes.len();
        let num_small_fragments = sorted_sizes
            .iter()
            .filter(|&&size| size < small_frag_threshold)
            .count();
        let p25 = *sorted_sizes.get(num_fragments / 4).unwrap_or(&0);
        let p50 = *sorted_sizes.get(num_fragments / 2).unwrap_or(&0);
        let p75 = *sorted_sizes.get(num_fragments * 3 / 4).unwrap_or(&0);
        let p99 = *sorted_sizes.get(num_fragments * 99 / 100).unwrap_or(&0);
        let min = sorted_sizes.first().copied().unwrap_or(0);
        let max = sorted_sizes.last().copied().unwrap_or(0);
        let mean = if num_fragments == 0 {
            0
        } else {
            sorted_sizes.iter().copied().sum::<usize>() / num_fragments
        };
        let frag_stats = FragmentStatistics {
            num_fragments,
            num_small_fragments,
            lengths: FragmentSummaryStats {
                min,
                max,
                mean,
                p25,
                p50,
                p75,
                p99,
            },
        };
        let stats = TableStatistics {
            total_bytes,
            num_rows,
            num_indices,
            fragment_stats: frag_stats,
        };
        Ok(stats)
    }
 }
 #[skip_serializing_none]
 #[derive(Debug, Deserialize, PartialEq)]
 pub struct TableStatistics {
    /// The total number of bytes in the table
    pub total_bytes: usize,
    /// The number of rows in the table
    pub num_rows: usize,
    /// The number of indices in the table
    pub num_indices: usize,
    /// Statistics on table fragments
    pub fragment_stats: FragmentStatistics,
 }
 #[skip_serializing_none]
 #[derive(Debug, Deserialize, PartialEq)]
 pub struct FragmentStatistics {
    /// The number of fragments in the table
    pub num_fragments: usize,
    /// The number of uncompacted fragments in the table
    pub num_small_fragments: usize,
    /// Statistics on the number of rows in the table fragments
    pub lengths: FragmentSummaryStats,
    // todo: add size statistics
    // /// Statistics on the number of bytes in the table fragments
    // sizes: FragmentStats,
 }
 #[skip_serializing_none]
 #[derive(Debug, Deserialize, PartialEq)]
 pub struct FragmentSummaryStats {
    pub min: usize,
    pub max: usize,
    pub mean: usize,
    pub p25: usize,
    pub p50: usize,
    pub p75: usize,
    pub p99: usize,
 }
 #[cfg(test)]
@@ -2509,7 +2844,7 @@ mod tests {
    use super::*;
    use crate::connect;
    use crate::connection::ConnectBuilder;
-    use crate::index::scalar::BTreeIndexBuilder;
+    use crate::index::scalar::{BTreeIndexBuilder, BitmapIndexBuilder};
    use crate::query::{ExecutableQuery, QueryBase};
    #[tokio::test]
@@ -3081,6 +3416,60 @@ mod tests {
        )
    }
    #[tokio::test]
    async fn test_tags() {
        let tmp_dir = tempdir().unwrap();
        let uri = tmp_dir.path().to_str().unwrap();
        let conn = ConnectBuilder::new(uri)
            .read_consistency_interval(Duration::from_secs(0))
            .execute()
            .await
            .unwrap();
        let table = conn
            .create_table("my_table", some_sample_data())
            .execute()
            .await
            .unwrap();
        assert_eq!(table.version().await.unwrap(), 1);
        table.add(some_sample_data()).execute().await.unwrap();
        assert_eq!(table.version().await.unwrap(), 2);
        let mut tags_manager = table.tags().await.unwrap();
        let tags = tags_manager.list().await.unwrap();
        assert!(tags.is_empty(), "Tags should be empty initially");
        let tag1 = "tag1";
        tags_manager.create(tag1, 1).await.unwrap();
        assert_eq!(tags_manager.get_version(tag1).await.unwrap(), 1);
        let tags = tags_manager.list().await.unwrap();
        assert_eq!(tags.len(), 1);
        assert!(tags.contains_key(tag1));
        assert_eq!(tags.get(tag1).unwrap().version, 1);
        tags_manager.create("tag2", 2).await.unwrap();
        assert_eq!(tags_manager.get_version("tag2").await.unwrap(), 2);
        let tags = tags_manager.list().await.unwrap();
        assert_eq!(tags.len(), 2);
        assert!(tags.contains_key(tag1));
        assert_eq!(tags.get(tag1).unwrap().version, 1);
        assert!(tags.contains_key("tag2"));
        assert_eq!(tags.get("tag2").unwrap().version, 2);
        // Test update and delete
        table.add(some_sample_data()).execute().await.unwrap();
        tags_manager.update(tag1, 3).await.unwrap();
        assert_eq!(tags_manager.get_version(tag1).await.unwrap(), 3);
        tags_manager.delete("tag2").await.unwrap();
        let tags = tags_manager.list().await.unwrap();
        assert_eq!(tags.len(), 1);
        assert!(tags.contains_key(tag1));
        assert_eq!(tags.get(tag1).unwrap().version, 3);
        // Test checkout tag
        table.add(some_sample_data()).execute().await.unwrap();
        assert_eq!(table.version().await.unwrap(), 4);
        table.checkout_tag(tag1).await.unwrap();
        assert_eq!(table.version().await.unwrap(), 3);
        table.checkout_latest().await.unwrap();
        assert_eq!(table.version().await.unwrap(), 4);
    }
    #[tokio::test]
    async fn test_create_index() {
        use arrow_array::RecordBatch;
@@ -3803,4 +4192,169 @@ mod tests {
            Some(&"test_field_val1".to_string())
        );
    }
    #[tokio::test]
    pub async fn test_stats() {
        let tmp_dir = tempdir().unwrap();
        let uri = tmp_dir.path().to_str().unwrap();
        let conn = ConnectBuilder::new(uri).execute().await.unwrap();
        let schema = Arc::new(Schema::new(vec![
            Field::new("id", DataType::Int32, false),
            Field::new("foo", DataType::Int32, true),
        ]));
        let batch = RecordBatch::try_new(
            schema.clone(),
            vec![
                Arc::new(Int32Array::from_iter_values(0..100)),
                Arc::new(Int32Array::from_iter_values(0..100)),
            ],
        )
        .unwrap();
        let table = conn
            .create_table(
                "test_stats",
                RecordBatchIterator::new(vec![Ok(batch.clone())], batch.schema()),
            )
            .execute()
            .await
            .unwrap();
        for _ in 0..10 {
            let batch = RecordBatch::try_new(
                schema.clone(),
                vec![
                    Arc::new(Int32Array::from_iter_values(0..15)),
                    Arc::new(Int32Array::from_iter_values(0..15)),
                ],
            )
            .unwrap();
            table
                .add(RecordBatchIterator::new(
                    vec![Ok(batch.clone())],
                    batch.schema(),
                ))
                .execute()
                .await
                .unwrap();
        }
        let empty_table = conn
            .create_table(
                "test_stats_empty",
                RecordBatchIterator::new(vec![], batch.schema()),
            )
            .execute()
            .await
            .unwrap();
        let res = table.stats().await.unwrap();
        println!("{:#?}", res);
        assert_eq!(
            res,
            TableStatistics {
                num_rows: 250,
                num_indices: 0,
                total_bytes: 2000,
                fragment_stats: FragmentStatistics {
                    num_fragments: 11,
                    num_small_fragments: 11,
                    lengths: FragmentSummaryStats {
                        min: 15,
                        max: 100,
                        mean: 22,
                        p25: 15,
                        p50: 15,
                        p75: 15,
                        p99: 100,
                    },
                },
            }
        );
        let res = empty_table.stats().await.unwrap();
        println!("{:#?}", res);
        assert_eq!(
            res,
            TableStatistics {
                num_rows: 0,
                num_indices: 0,
                total_bytes: 0,
                fragment_stats: FragmentStatistics {
                    num_fragments: 0,
                    num_small_fragments: 0,
                    lengths: FragmentSummaryStats {
                        min: 0,
                        max: 0,
                        mean: 0,
                        p25: 0,
                        p50: 0,
                        p75: 0,
                        p99: 0,
                    },
                },
            }
        )
    }
    #[tokio::test]
    pub async fn test_list_indices_skip_frag_reuse() {
        let tmp_dir = tempdir().unwrap();
        let uri = tmp_dir.path().to_str().unwrap();
        let conn = ConnectBuilder::new(uri).execute().await.unwrap();
        let schema = Arc::new(Schema::new(vec![
            Field::new("id", DataType::Int32, false),
            Field::new("foo", DataType::Int32, true),
        ]));
        let batch = RecordBatch::try_new(
            schema.clone(),
            vec![
                Arc::new(Int32Array::from_iter_values(0..100)),
                Arc::new(Int32Array::from_iter_values(0..100)),
            ],
        )
        .unwrap();
        let table = conn
            .create_table(
                "test_list_indices_skip_frag_reuse",
                RecordBatchIterator::new(vec![Ok(batch.clone())], batch.schema()),
            )
            .execute()
            .await
            .unwrap();
        table
            .add(RecordBatchIterator::new(
                vec![Ok(batch.clone())],
                batch.schema(),
            ))
            .execute()
            .await
            .unwrap();
        table
            .create_index(&["id"], Index::Bitmap(BitmapIndexBuilder {}))
            .execute()
            .await
            .unwrap();
        table
            .optimize(OptimizeAction::Compact {
                options: CompactionOptions {
                    target_rows_per_fragment: 2_000,
                    defer_index_remap: true,
                    ..Default::default()
                },
                remap_options: None,
            })
            .await
            .unwrap();
        let result = table.list_indices().await.unwrap();
        assert_eq!(result.len(), 1);
        assert_eq!(result[0].index_type, crate::index::IndexType::Bitmap);
    }
 }
--- a/rust/lancedb/src/table/dataset.rs
+++ b/rust/lancedb/src/table/dataset.rs
@@ -7,7 +7,7 @@ use std::{
    time::{self, Duration, Instant},
 };
-use lance::Dataset;
+use lance::{dataset::refs, Dataset};
 use tokio::sync::{RwLock, RwLockReadGuard, RwLockWriteGuard};
 use crate::error::Result;
@@ -83,19 +83,32 @@ impl DatasetRef {
        }
    }
-    async fn as_time_travel(&mut self, target_version: u64) -> Result<()> {
+    async fn as_time_travel(&mut self, target_version: impl Into<refs::Ref>) -> Result<()> {
        let target_ref = target_version.into();
        match self {
            Self::Latest { dataset, .. } => {
                let new_dataset = dataset.checkout_version(target_ref.clone()).await?;
                let version_value = new_dataset.version().version;
                *self = Self::TimeTravel {
-                    dataset: dataset.checkout_version(target_version).await?,
+                    dataset: new_dataset,
-                    version: target_version,
+                    version: version_value,
                };
            }
            Self::TimeTravel { dataset, version } => {
-                if *version != target_version {
+                let should_checkout = match &target_ref {
                    refs::Ref::Version(target_ver) => version != target_ver,
                    refs::Ref::Tag(_) => true, // Always checkout for tags
                };
                if should_checkout {
                    let new_dataset = dataset.checkout_version(target_ref).await?;
                    let version_value = new_dataset.version().version;
                    *self = Self::TimeTravel {
-                        dataset: dataset.checkout_version(target_version).await?,
+                        dataset: new_dataset,
-                        version: target_version,
+                        version: version_value,
                    };
                }
            }
@@ -175,7 +188,7 @@ impl DatasetConsistencyWrapper {
        write_guard.as_latest(read_consistency_interval).await
    }
-    pub async fn as_time_travel(&self, target_version: u64) -> Result<()> {
+    pub async fn as_time_travel(&self, target_version: impl Into<refs::Ref>) -> Result<()> {
        self.0.write().await.as_time_travel(target_version).await
    }
--- a/rust/lancedb/src/table/merge.rs
+++ b/rust/lancedb/src/table/merge.rs
@@ -1,13 +1,13 @@
 // SPDX-License-Identifier: Apache-2.0
 // SPDX-FileCopyrightText: Copyright The LanceDB Authors
-use std::sync::Arc;
+use std::{sync::Arc, time::Duration};
 use arrow_array::RecordBatchReader;
 use crate::Result;
-use super::BaseTable;
+use super::{BaseTable, MergeResult};
 /// A builder used to create and run a merge insert operation
 ///
@@ -21,6 +21,7 @@ pub struct MergeInsertBuilder {
    pub(crate) when_not_matched_insert_all: bool,
    pub(crate) when_not_matched_by_source_delete: bool,
    pub(crate) when_not_matched_by_source_delete_filt: Option<String>,
    pub(crate) timeout: Option<Duration>,
 }
 impl MergeInsertBuilder {
@@ -33,6 +34,7 @@ impl MergeInsertBuilder {
            when_not_matched_insert_all: false,
            when_not_matched_by_source_delete: false,
            when_not_matched_by_source_delete_filt: None,
            timeout: None,
        }
    }
@@ -84,10 +86,26 @@ impl MergeInsertBuilder {
        self
    }
    /// Maximum time to run the operation before cancelling it.
    ///
    /// By default, there is a 30-second timeout that is only enforced after the
    /// first attempt. This is to prevent spending too long retrying to resolve
    /// conflicts. For example, if a write attempt takes 20 seconds and fails,
    /// the second attempt will be cancelled after 10 seconds, hitting the
    /// 30-second timeout. However, a write that takes one hour and succeeds on the
    /// first attempt will not be cancelled.
    ///
    /// When this is set, the timeout is enforced on all attempts, including the first.
    pub fn timeout(&mut self, timeout: Duration) -> &mut Self {
        self.timeout = Some(timeout);
        self
    }
    /// Executes the merge insert operation
    ///
-    /// Nothing is returned but the [`super::Table`] is updated
+    /// Returns version and statistics about the merge operation including the number of rows
-    pub async fn execute(self, new_data: Box<dyn RecordBatchReader + Send>) -> Result<()> {
+    /// inserted, updated, and deleted.
    pub async fn execute(self, new_data: Box<dyn RecordBatchReader + Send>) -> Result<MergeResult> {
        self.table.clone().merge_insert(self, new_data).await
    }
 }
`@@ -1,2 +1,2 @@`
	`[toolchain]`	`[toolchain]`
	`channel = "1.83.0"`	`channel = "1.86.0"`