Bump version: 0.23.0-beta.3 → 0.23.0

Bump version: 0.23.0-beta.2 → 0.23.0-beta.3
feat: upgrade lance to stable version (#2420 )
2025-12-23 05:19:58 +00:00 · 2025-06-04 21:07:39 +00:00 · 2025-06-04 21:07:38 +00:00 · 2025-06-04 13:34:30 -07:00 · 2025-06-04 08:40:38 -07:00 · 2025-06-04 07:15:07 +00:00
71 changed files with 2150 additions and 1030 deletions
--- a/.bumpversion.toml
+++ b/.bumpversion.toml
@@ -1,5 +1,5 @@
 [tool.bumpversion]
-current_version = "0.19.1-beta.1"
+current_version = "0.20.0-beta.2"
 parse = """(?x)
    (?P<major>0|[1-9]\\d*)\\.
    (?P<minor>0|[1-9]\\d*)\\.
--- a/.github/workflows/java.yml
+++ b/.github/workflows/java.yml
@@ -35,6 +35,9 @@ jobs:
      - uses: Swatinem/rust-cache@v2
        with:
          workspaces: java/core/lancedb-jni
+      - uses: actions-rust-lang/setup-rust-toolchain@v1
+        with:
+          components: rustfmt
      - name: Run cargo fmt
        run: cargo fmt --check
        working-directory: ./java/core/lancedb-jni
@@ -68,6 +71,9 @@ jobs:
      - uses: Swatinem/rust-cache@v2
        with:
          workspaces: java/core/lancedb-jni
+      - uses: actions-rust-lang/setup-rust-toolchain@v1
+        with:
+          components: rustfmt
      - name: Run cargo fmt
        run: cargo fmt --check
        working-directory: ./java/core/lancedb-jni
@@ -110,4 +116,3 @@ jobs:
          -Djdk.reflect.useDirectMethodHandle=false \
          -Dio.netty.tryReflectionSetAccessible=true"
          JAVA_HOME=$JAVA_17 mvn clean test
-  
--- a/.github/workflows/make-release-commit.yml
+++ b/.github/workflows/make-release-commit.yml
@@ -84,6 +84,7 @@ jobs:
        run: |
          pip install bump-my-version PyGithub packaging
          bash ci/bump_version.sh ${{ inputs.type }} ${{ inputs.bump-minor }} v $COMMIT_BEFORE_BUMP
+          bash ci/update_lockfiles.sh --amend
      - name: Push new version tag
        if: ${{ !inputs.dry_run }}
        uses: ad-m/github-push-action@master
@@ -92,11 +93,3 @@ jobs:
          github_token: ${{ secrets.LANCEDB_RELEASE_TOKEN }}
          branch: ${{ github.ref }}
          tags: true
-      - uses: ./.github/workflows/update_package_lock
-        if: ${{ !inputs.dry_run && inputs.other }}
-        with:
-          github_token: ${{ secrets.GITHUB_TOKEN }}
-      - uses: ./.github/workflows/update_package_lock_nodejs
-        if: ${{ !inputs.dry_run && inputs.other }}
-        with:
-          github_token: ${{ secrets.GITHUB_TOKEN }}
--- a/.github/workflows/nodejs.yml
+++ b/.github/workflows/nodejs.yml
@@ -47,6 +47,9 @@ jobs:
      run: |
        sudo apt update
        sudo apt install -y protobuf-compiler libssl-dev
+    - uses: actions-rust-lang/setup-rust-toolchain@v1
+      with:
+        components: rustfmt, clippy
    - name: Lint
      run: |
        cargo fmt --all -- --check
@@ -113,7 +116,7 @@ jobs:
        set -e
        npm ci
        npm run docs
-        if ! git diff --exit-code; then
+        if ! git diff --exit-code -- . ':(exclude)Cargo.lock'; then
          echo "Docs need to be updated"
          echo "Run 'npm run docs', fix any warnings, and commit the changes."
          exit 1
--- a/.github/workflows/npm-publish.yml
+++ b/.github/workflows/npm-publish.yml
@@ -505,6 +505,8 @@ jobs:
    name: vectordb NPM Publish
    needs: [node, node-macos, node-linux-gnu, node-windows]
    runs-on: ubuntu-latest
+    permissions:
+      contents: write
    # Only runs on tags that matches the make-release action
    if: startsWith(github.ref, 'refs/tags/v')
    steps:
@@ -537,6 +539,10 @@ jobs:
        # We need to deprecate the old package to avoid confusion.
        # Each time we publish a new version, it gets undeprecated.
        run: npm deprecate vectordb "Use @lancedb/lancedb instead."
+      - name: Update package-lock.json
+        run: bash ci/update_lockfiles.sh
+      - name: Push new commit
+        uses: ad-m/github-push-action@master
      - name: Notify Slack Action
        uses: ravsamhq/notify-slack-action@2.3.0
        if: ${{ always() }}
@@ -546,21 +552,3 @@ jobs:
          notification_title: "{workflow} is failing"
        env:
          SLACK_WEBHOOK_URL: ${{ secrets.ACTION_MONITORING_SLACK }}
-
-  update-package-lock:
-    if: startsWith(github.ref, 'refs/tags/v')
-    needs: [release]
-    runs-on: ubuntu-latest
-    permissions:
-      contents: write
-    steps:
-      - name: Checkout
-        uses: actions/checkout@v4
-        with:
-          ref: main
-          token: ${{ secrets.LANCEDB_RELEASE_TOKEN }}
-          fetch-depth: 0
-          lfs: true
-      - uses: ./.github/workflows/update_package_lock
-        with:
-          github_token: ${{ secrets.GITHUB_TOKEN }}
--- a/.github/workflows/run_tests/action.yml
+++ b/.github/workflows/run_tests/action.yml
@@ -24,8 +24,8 @@ runs:
    - name: pytest (with integration)
      shell: bash
      if: ${{ inputs.integration == 'true' }}
-      run: pytest -m "not slow" -x -v --durations=30 python/python/tests
+      run: pytest -m "not slow" -vv --durations=30 python/python/tests
    - name: pytest (no integration tests)
      shell: bash
      if: ${{ inputs.integration != 'true' }}
-      run: pytest -m "not slow and not s3_test" -x -v --durations=30 python/python/tests
+      run: pytest -m "not slow and not s3_test" -vv --durations=30 python/python/tests
--- a/.github/workflows/rust.yml
+++ b/.github/workflows/rust.yml
@@ -40,6 +40,9 @@ jobs:
        with:
          fetch-depth: 0
          lfs: true
+      - uses: actions-rust-lang/setup-rust-toolchain@v1
+        with:
+          components: rustfmt, clippy
      - uses: Swatinem/rust-cache@v2
        with:
          workspaces: rust
@@ -160,8 +163,8 @@ jobs:
    strategy:
      matrix:
        target:
-        - x86_64-pc-windows-msvc
-        - aarch64-pc-windows-msvc
+          - x86_64-pc-windows-msvc
+          - aarch64-pc-windows-msvc
    defaults:
      run:
        working-directory: rust/lancedb
--- a/.github/workflows/update_package_lock/action.yml
+++ b/.github/workflows/update_package_lock/action.yml
@@ -1,33 +0,0 @@
-name: update_package_lock
-description: "Update node's package.lock"
-
-inputs:
-  github_token:
-    required: true
-    description: "github token for the repo"
-
-runs:
-  using: "composite"
-  steps:
-    - uses: actions/setup-node@v3
-      with:
-        node-version: 20
-    - name: Set git configs
-      shell: bash
-      run: |
-        git config user.name 'Lance Release'
-        git config user.email 'lance-dev@lancedb.com'
-    - name: Update package-lock.json file
-      working-directory: ./node
-      run: |
-        npm install
-        git add package-lock.json
-        git commit -m "Updating package-lock.json"
-      shell: bash
-    - name: Push changes
-      if: ${{ inputs.dry_run }} == "false"
-      uses: ad-m/github-push-action@master
-      with:
-        github_token: ${{ inputs.github_token }}
-        branch: main
-        tags: true
--- a/.github/workflows/update_package_lock_nodejs/action.yml
+++ b/.github/workflows/update_package_lock_nodejs/action.yml
@@ -1,33 +0,0 @@
-name: update_package_lock_nodejs
-description: "Update nodejs's package.lock"
-
-inputs:
-  github_token:
-    required: true
-    description: "github token for the repo"
-
-runs:
-  using: "composite"
-  steps:
-    - uses: actions/setup-node@v3
-      with:
-        node-version: 20
-    - name: Set git configs
-      shell: bash
-      run: |
-        git config user.name 'Lance Release'
-        git config user.email 'lance-dev@lancedb.com'
-    - name: Update package-lock.json file
-      working-directory: ./nodejs
-      run: |
-        npm install
-        git add package-lock.json
-        git commit -m "Updating package-lock.json"
-      shell: bash
-    - name: Push changes
-      if: ${{ inputs.dry_run }} == "false"
-      uses: ad-m/github-push-action@master
-      with:
-        github_token: ${{ inputs.github_token }}
-        branch: main
-        tags: true
--- a/Cargo.lock
+++ b/Cargo.lock
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -21,32 +21,32 @@ categories = ["database-implementations"]
 rust-version = "1.78.0"

 [workspace.dependencies]
-lance = { "version" = "=0.27.0", "features" = ["dynamodb"], tag = "v0.27.0-beta.2", git="https://github.com/lancedb/lance.git" }
-lance-io = { version = "=0.27.0", tag = "v0.27.0-beta.2", git="https://github.com/lancedb/lance.git" }
-lance-index = { version = "=0.27.0", tag = "v0.27.0-beta.2", git="https://github.com/lancedb/lance.git" }
-lance-linalg = { version = "=0.27.0", tag = "v0.27.0-beta.2", git="https://github.com/lancedb/lance.git" }
-lance-table = { version = "=0.27.0", tag = "v0.27.0-beta.2", git="https://github.com/lancedb/lance.git" }
-lance-testing = { version = "=0.27.0", tag = "v0.27.0-beta.2", git="https://github.com/lancedb/lance.git" }
-lance-datafusion = { version = "=0.27.0", tag = "v0.27.0-beta.2", git="https://github.com/lancedb/lance.git" }
-lance-encoding = { version = "=0.27.0", tag = "v0.27.0-beta.2", git="https://github.com/lancedb/lance.git" }
+lance = { "version" = "=0.29.0", "features" = ["dynamodb"] }
+lance-io = "=0.29.0"
+lance-index = "=0.29.0"
+lance-linalg = "=0.29.0"
+lance-table = "=0.29.0"
+lance-testing = "=0.29.0"
+lance-datafusion = "=0.29.0"
+lance-encoding = "=0.29.0"
 # Note that this one does not include pyarrow
-arrow = { version = "54.1", optional = false }
-arrow-array = "54.1"
-arrow-data = "54.1"
-arrow-ipc = "54.1"
-arrow-ord = "54.1"
-arrow-schema = "54.1"
-arrow-arith = "54.1"
-arrow-cast = "54.1"
+arrow = { version = "55.1", optional = false }
+arrow-array = "55.1"
+arrow-data = "55.1"
+arrow-ipc = "55.1"
+arrow-ord = "55.1"
+arrow-schema = "55.1"
+arrow-arith = "55.1"
+arrow-cast = "55.1"
 async-trait = "0"
-datafusion = { version = "46.0", default-features = false }
-datafusion-catalog = "46.0"
-datafusion-common = { version = "46.0", default-features = false }
-datafusion-execution = "46.0"
-datafusion-expr = "46.0"
-datafusion-physical-plan = "46.0"
+datafusion = { version = "47.0", default-features = false }
+datafusion-catalog = "47.0"
+datafusion-common = { version = "47.0", default-features = false }
+datafusion-execution = "47.0"
+datafusion-expr = "47.0"
+datafusion-physical-plan = "47.0"
 env_logger = "0.11"
-half = { "version" = "=2.4.1", default-features = false, features = [
+half = { "version" = "=2.5.0", default-features = false, features = [
    "num-traits",
 ] }
 futures = "0"
@@ -57,19 +57,16 @@ pin-project = "1.0.7"
 snafu = "0.8"
 url = "2"
 num-traits = "0.2"
-rand = "0.8"
+rand = "0.9"
 regex = "1.10"
 lazy_static = "1"
 semver = "1.0.25"
-
 # Temporary pins to work around downstream issues
 # https://github.com/apache/arrow-rs/commit/2fddf85afcd20110ce783ed5b4cdeb82293da30b
-chrono = "=0.4.39"
+chrono = "=0.4.41"
 # https://github.com/RustCrypto/formats/issues/1684
 base64ct = "=1.6.0"
-
 # Workaround for: https://github.com/eira-fransham/crunchy/issues/13
 crunchy = "=0.2.2"
-
 # Workaround for: https://github.com/Lokathor/bytemuck/issues/306
 bytemuck_derive = ">=1.8.1, <1.9.0"
--- a/README.md
+++ b/README.md
@@ -1,94 +1,97 @@
 <a href="https://cloud.lancedb.com" target="_blank">
  <img src="https://github.com/user-attachments/assets/92dad0a2-2a37-4ce1-b783-0d1b4f30a00c" alt="LanceDB Cloud Public Beta" width="100%" style="max-width: 100%;">
 </a>
-
 <div align="center">
-<p align="center">

-<picture>
-  <source media="(prefers-color-scheme: dark)" srcset="https://github.com/user-attachments/assets/ac270358-333e-4bea-a132-acefaa94040e">
-  <source media="(prefers-color-scheme: light)" srcset="https://github.com/user-attachments/assets/b864d814-0d29-4784-8fd9-807297c758c0">
-  <img alt="LanceDB Logo" src="https://github.com/user-attachments/assets/b864d814-0d29-4784-8fd9-807297c758c0" width=300>
-</picture>
+[![LanceDB](docs/src/assets/hero-header.png)](https://lancedb.com)
+[![Website](https://img.shields.io/badge/-Website-100000?style=for-the-badge&labelColor=645cfb&color=645cfb)](https://lancedb.com/)
+[![Blog](https://img.shields.io/badge/Blog-100000?style=for-the-badge&labelColor=645cfb&color=645cfb)](https://blog.lancedb.com/)
+[![Discord](https://img.shields.io/badge/-Discord-100000?style=for-the-badge&logo=discord&logoColor=white&labelColor=645cfb&color=645cfb)](https://discord.gg/zMM32dvNtd)
+[![Twitter](https://img.shields.io/badge/-Twitter-100000?style=for-the-badge&logo=x&logoColor=white&labelColor=645cfb&color=645cfb)](https://twitter.com/lancedb)
+[![LinkedIn](https://img.shields.io/badge/-LinkedIn-100000?style=for-the-badge&logo=linkedin&logoColor=white&labelColor=645cfb&color=645cfb)](https://www.linkedin.com/company/lancedb/)

-**Search More, Manage Less**

-<a href='https://github.com/lancedb/vectordb-recipes/tree/main' target="_blank"><img alt='LanceDB' src='https://img.shields.io/badge/VectorDB_Recipes-100000?style=for-the-badge&logo=LanceDB&logoColor=white&labelColor=645cfb&color=645cfb'/></a>
-<a href='https://lancedb.github.io/lancedb/' target="_blank"><img alt='lancdb' src='https://img.shields.io/badge/DOCS-100000?style=for-the-badge&logo=lancdb&logoColor=white&labelColor=645cfb&color=645cfb'/></a>
-[![Blog](https://img.shields.io/badge/Blog-12100E?style=for-the-badge&logoColor=white)](https://blog.lancedb.com/)
-[![Discord](https://img.shields.io/badge/Discord-%235865F2.svg?style=for-the-badge&logo=discord&logoColor=white)](https://discord.gg/zMM32dvNtd)
-[![Twitter](https://img.shields.io/badge/Twitter-%231DA1F2.svg?style=for-the-badge&logo=Twitter&logoColor=white)](https://twitter.com/lancedb)
-[![Gurubase](https://img.shields.io/badge/Gurubase-Ask%20LanceDB%20Guru-006BFF?style=for-the-badge)](https://gurubase.io/g/lancedb)
+<img src="docs/src/assets/lancedb.png" alt="LanceDB" width="50%">

-</p>
+# **The Multimodal AI Lakehouse**

-<img max-width="750px" alt="LanceDB Multimodal Search" src="https://github.com/lancedb/lancedb/assets/917119/09c5afc5-7816-4687-bae4-f2ca194426ec">
+[**How to Install** ](#how-to-install) ✦ [**Detailed Documentation**](https://lancedb.github.io/lancedb/) ✦ [**Tutorials and Recipes**](https://github.com/lancedb/vectordb-recipes/tree/main) ✦  [**Contributors**](#contributors) 
+
+**The ultimate multimodal data platform for AI/ML applications.** 
+
+LanceDB is designed for fast, scalable, and production-ready vector search. It is built on top of the Lance columnar format. You can store, index, and search over petabytes of multimodal data and vectors with ease. 
+LanceDB is a central location where developers can build, train and analyze their AI workloads.

-</p>
 </div>

-<hr />
+<br>

-LanceDB is an open-source database for vector-search built with persistent storage, which greatly simplifies retrieval, filtering and management of embeddings.
+## **Demo: Multimodal Search by Keyword, Vector or with SQL**
+<img max-width="750px" alt="LanceDB Multimodal Search" src="https://github.com/lancedb/lancedb/assets/917119/09c5afc5-7816-4687-bae4-f2ca194426ec">

-The key features of LanceDB include:
+## **Star LanceDB to get updates!**

-* Production-scale vector search with no servers to manage.
+<details>
+<summary>⭐ Click here ⭐  to see how fast we're growing!</summary>
+<picture>
+  <source media="(prefers-color-scheme: dark)" srcset="https://api.star-history.com/svg?repos=lancedb/lancedb&theme=dark&type=Date">
+  <img width="100%" src="https://api.star-history.com/svg?repos=lancedb/lancedb&theme=dark&type=Date">
+</picture>
+</details>

-* Store, query and filter vectors, metadata and multi-modal data (text, images, videos, point clouds, and more).
+## **Key Features**:

-* Support for vector similarity search, full-text search and SQL.
+- **Fast Vector Search**: Search billions of vectors in milliseconds with state-of-the-art indexing.
+- **Comprehensive Search**: Support for vector similarity search, full-text search and SQL.
+- **Multimodal Support**: Store, query and filter vectors, metadata and multimodal data (text, images, videos, point clouds, and more).
+- **Advanced Features**: Zero-copy, automatic versioning, manage versions of your data without needing extra infrastructure. GPU support in building vector index.

-* Native Python and Javascript/Typescript support.
+### **Products**:
+- **Open Source & Local**: 100% open source, runs locally or in your cloud. No vendor lock-in.
+- **Cloud and Enterprise**: Production-scale vector search with no servers to manage. Complete data sovereignty and security.

-* Zero-copy, automatic versioning, manage versions of your data without needing extra infrastructure.
+### **Ecosystem**:
+- **Columnar Storage**: Built on the Lance columnar format for efficient storage and analytics.
+- **Seamless Integration**: Python, Node.js, Rust, and REST APIs for easy integration. Native Python and Javascript/Typescript support.
+- **Rich Ecosystem**: Integrations with [**LangChain** 🦜️🔗](https://python.langchain.com/docs/integrations/vectorstores/lancedb/), [**LlamaIndex** 🦙](https://gpt-index.readthedocs.io/en/latest/examples/vector_stores/LanceDBIndexDemo.html), Apache-Arrow, Pandas, Polars, DuckDB and more on the way.

-* GPU support in building vector index(*).
+## **How to Install**:

-* Ecosystem integrations with [LangChain 🦜️🔗](https://python.langchain.com/docs/integrations/vectorstores/lancedb/), [LlamaIndex 🦙](https://gpt-index.readthedocs.io/en/latest/examples/vector_stores/LanceDBIndexDemo.html), Apache-Arrow, Pandas, Polars, DuckDB and more on the way.
+Follow the [Quickstart](https://lancedb.github.io/lancedb/basic/) doc to set up LanceDB locally. 

-LanceDB's core is written in Rust 🦀 and is built using <a href="https://github.com/lancedb/lance">Lance</a>, an open-source columnar format designed for performant ML workloads.
+**API & SDK:** We also support Python, Typescript and Rust SDKs

-## Quick Start
+| Interface | Documentation |
+|-----------|---------------|
+| Python SDK | https://lancedb.github.io/lancedb/python/python/ |
+| Typescript SDK | https://lancedb.github.io/lancedb/js/globals/ |
+| Rust SDK | https://docs.rs/lancedb/latest/lancedb/index.html |
+| REST API | https://docs.lancedb.com/api-reference/introduction |

-**Javascript**
-```shell
-npm install @lancedb/lancedb
-```
+## **Join Us and Contribute**

-```javascript
-import * as lancedb from "@lancedb/lancedb";
+We welcome contributions from everyone! Whether you're a developer, researcher, or just someone who wants to help out. 

-const db = await lancedb.connect("data/sample-lancedb");
-const table = await db.createTable("vectors", [
-	{ id: 1, vector: [0.1, 0.2], item: "foo", price: 10 },
-	{ id: 2, vector: [1.1, 1.2], item: "bar", price: 50 },
-], {mode: 'overwrite'});
+If you have any suggestions or feature requests, please feel free to open an issue on GitHub or discuss it on our [**Discord**](https://discord.gg/G5DcmnZWKB) server.
+
+[**Check out the GitHub Issues**](https://github.com/lancedb/lancedb/issues) if you would like to work on the features that are planned for the future. If you have any suggestions or feature requests, please feel free to open an issue on GitHub. 
+
+## **Contributors**
+
+<a href="https://github.com/lancedb/lancedb/graphs/contributors">
+  <img src="https://contrib.rocks/image?repo=lancedb/lancedb" />
+</a>


-const query = table.vectorSearch([0.1, 0.3]).limit(2);
-const results = await query.toArray();
+## **Stay in Touch With Us**
+<div align="center">

-// You can also search for rows by specific criteria without involving a vector search.
-const rowsByCriteria = await table.query().where("price >= 10").toArray();
-```
+</br>

-**Python**
-```shell
-pip install lancedb
-```
+[![Website](https://img.shields.io/badge/-Website-100000?style=for-the-badge&labelColor=645cfb&color=645cfb)](https://lancedb.com/)
+[![Blog](https://img.shields.io/badge/Blog-100000?style=for-the-badge&labelColor=645cfb&color=645cfb)](https://blog.lancedb.com/)
+[![Discord](https://img.shields.io/badge/-Discord-100000?style=for-the-badge&logo=discord&logoColor=white&labelColor=645cfb&color=645cfb)](https://discord.gg/zMM32dvNtd)
+[![Twitter](https://img.shields.io/badge/-Twitter-100000?style=for-the-badge&logo=x&logoColor=white&labelColor=645cfb&color=645cfb)](https://twitter.com/lancedb)
+[![LinkedIn](https://img.shields.io/badge/-LinkedIn-100000?style=for-the-badge&logo=linkedin&logoColor=white&labelColor=645cfb&color=645cfb)](https://www.linkedin.com/company/lancedb/)

-```python
-import lancedb
-
-uri = "data/sample-lancedb"
-db = lancedb.connect(uri)
-table = db.create_table("my_table",
-                         data=[{"vector": [3.1, 4.1], "item": "foo", "price": 10.0},
-                               {"vector": [5.9, 26.5], "item": "bar", "price": 20.0}])
-result = table.search([100, 100]).limit(2).to_pandas()
-```
-
-## Blogs, Tutorials & Videos
-* 📈 <a href="https://blog.lancedb.com/benchmarking-random-access-in-lance/">2000x better performance with Lance over Parquet</a>
-* 🤖 <a href="https://github.com/lancedb/vectordb-recipes/tree/main/examples/Youtube-Search-QA-Bot">Build a question and answer bot with LanceDB</a>
+</div>
--- a/ci/set_lance_version.py
+++ b/ci/set_lance_version.py
@@ -0,0 +1,174 @@
+import argparse
+import sys
+import json
+
+
+def run_command(command: str) -> str:
+    """
+    Run a shell command and return stdout as a string.
+    If exit code is not 0, raise an exception with the stderr output.
+    """
+    import subprocess
+
+    result = subprocess.run(command, shell=True, capture_output=True, text=True)
+    if result.returncode != 0:
+        raise Exception(f"Command failed with error: {result.stderr.strip()}")
+    return result.stdout.strip()
+
+
+def get_latest_stable_version() -> str:
+    version_line = run_command("cargo info lance | grep '^version:'")
+    version = version_line.split(" ")[1].strip()
+    return version
+
+
+def get_latest_preview_version() -> str:
+    lance_tags = run_command(
+        "git ls-remote --tags https://github.com/lancedb/lance.git | grep 'refs/tags/v[0-9beta.-]\\+$'"
+    ).splitlines()
+    lance_tags = (
+        tag.split("refs/tags/")[1]
+        for tag in lance_tags
+        if "refs/tags/" in tag and "beta" in tag
+    )
+    from packaging.version import Version
+
+    latest = max(
+        (tag[1:] for tag in lance_tags if tag.startswith("v")), key=lambda t: Version(t)
+    )
+    return str(latest)
+
+
+def extract_features(line: str) -> list:
+    """
+    Extracts the features from a line in Cargo.toml.
+    Example: 'lance = { "version" = "=0.29.0", "features" = ["dynamodb"] }'
+    Returns: ['dynamodb']
+    """
+    import re
+
+    match = re.search(r'"features"\s*=\s*\[(.*?)\]', line)
+    if match:
+        features_str = match.group(1)
+        return [f.strip('"') for f in features_str.split(",")]
+    return []
+
+
+def update_cargo_toml(line_updater):
+    """
+    Updates the Cargo.toml file by applying the line_updater function to each line.
+    The line_updater function should take a line as input and return the updated line.
+    """
+    with open("Cargo.toml", "r") as f:
+        lines = f.readlines()
+
+    new_lines = []
+    for line in lines:
+        if line.startswith("lance"):
+            # Update the line using the provided function
+            new_lines.append(line_updater(line))
+        else:
+            # Keep the line unchanged
+            new_lines.append(line)
+
+    with open("Cargo.toml", "w") as f:
+        f.writelines(new_lines)
+
+
+def set_stable_version(version: str):
+    """
+    Sets lines to
+    lance = { "version" = "=0.29.0", "features" = ["dynamodb"] }
+    lance-io = "=0.29.0"
+    ...
+    """
+
+    def line_updater(line: str) -> str:
+        package_name = line.split("=", maxsplit=1)[0].strip()
+        features = extract_features(line)
+        if features:
+            return f'{package_name} = {{ "version" = "={version}", "features" = {json.dumps(features)} }}\n'
+        else:
+            return f'{package_name} = "={version}"\n'
+
+    update_cargo_toml(line_updater)
+
+
+def set_preview_version(version: str):
+    """
+    Sets lines to
+    lance = { "version" = "=0.29.0", "features" = ["dynamodb"], tag = "v0.29.0-beta.2", git="https://github.com/lancedb/lance.git" }
+    lance-io = { version = "=0.29.0", tag = "v0.29.0-beta.2", git="https://github.com/lancedb/lance.git" }
+    ...
+    """
+
+    def line_updater(line: str) -> str:
+        package_name = line.split("=", maxsplit=1)[0].strip()
+        features = extract_features(line)
+        base_version = version.split("-")[0]  # Get the base version without beta suffix
+        if features:
+            return f'{package_name} = {{ "version" = "={base_version}", "features" = {json.dumps(features)}, "tag" = "v{version}", "git" = "https://github.com/lancedb/lance.git" }}\n'
+        else:
+            return f'{package_name} = {{ "version" = "={base_version}", "tag" = "v{version}", "git" = "https://github.com/lancedb/lance.git" }}\n'
+
+    update_cargo_toml(line_updater)
+
+
+def set_local_version():
+    """
+    Sets lines to
+    lance = { path = "../lance/rust/lance", features = ["dynamodb"] }
+    lance-io = { path = "../lance/rust/lance-io" }
+    ...
+    """
+
+    def line_updater(line: str) -> str:
+        package_name = line.split("=", maxsplit=1)[0].strip()
+        features = extract_features(line)
+        if features:
+            return f'{package_name} = {{ "path" = "../lance/rust/{package_name}", "features" = {json.dumps(features)} }}\n'
+        else:
+            return f'{package_name} = {{ "path" = "../lance/rust/{package_name}" }}\n'
+
+    update_cargo_toml(line_updater)
+
+
+parser = argparse.ArgumentParser(description="Set the version of the Lance package.")
+parser.add_argument(
+    "version",
+    type=str,
+    help="The version to set for the Lance package. Use 'stable' for the latest stable version, 'preview' for latest preview version, or a specific version number (e.g., '0.1.0'). You can also specify 'local' to use a local path.",
+)
+args = parser.parse_args()
+
+if args.version == "stable":
+    latest_stable_version = get_latest_stable_version()
+    print(
+        f"Found latest stable version: \033[1mv{latest_stable_version}\033[0m",
+        file=sys.stderr,
+    )
+    set_stable_version(latest_stable_version)
+elif args.version == "preview":
+    latest_preview_version = get_latest_preview_version()
+    print(
+        f"Found latest preview version: \033[1mv{latest_preview_version}\033[0m",
+        file=sys.stderr,
+    )
+    set_preview_version(latest_preview_version)
+elif args.version == "local":
+    set_local_version()
+else:
+    # Parse the version number.
+    version = args.version
+    # Ignore initial v if present.
+    if version.startswith("v"):
+        version = version[1:]
+
+    if "beta" in version:
+        set_preview_version(version)
+    else:
+        set_stable_version(version)
+
+print("Updating lockfiles...", file=sys.stderr, end="")
+run_command("cargo metadata > /dev/null")
+print(" done.", file=sys.stderr)
--- a/ci/update_lockfiles.sh
+++ b/ci/update_lockfiles.sh
@@ -0,0 +1,30 @@
+#!/usr/bin/env bash
+set -euo pipefail
+
+AMEND=false
+
+for arg in "$@"; do
+  if [[ "$arg" == "--amend" ]]; then
+    AMEND=true
+  fi
+done
+
+# This updates the lockfile without building
+cargo metadata --quiet > /dev/null
+
+pushd nodejs || exit 1
+npm install --package-lock-only --silent
+popd
+pushd node || exit 1
+npm install --package-lock-only --silent
+popd
+
+if git diff --quiet --exit-code; then
+  echo "No lockfile changes to commit; skipping amend."
+elif $AMEND; then
+  git add Cargo.lock nodejs/package-lock.json node/package-lock.json
+  git commit --amend --no-edit
+else
+  git add Cargo.lock nodejs/package-lock.json node/package-lock.json
+  git commit -m "Update lockfiles"
+fi
--- a/docs/mkdocs.yml
+++ b/docs/mkdocs.yml
@@ -193,6 +193,7 @@ nav:
          - Pandas and PyArrow: python/pandas_and_pyarrow.md
          - Polars: python/polars_arrow.md
          - DuckDB: python/duckdb.md
+          - Datafusion: python/datafusion.md
          - LangChain:
              - LangChain 🔗: integrations/langchain.md
              - LangChain demo: notebooks/langchain_demo.ipynb
@@ -205,6 +206,7 @@ nav:
          - PromptTools: integrations/prompttools.md
          - dlt: integrations/dlt.md
          - phidata: integrations/phidata.md
+          - Genkit: integrations/genkit.md
      - 🎯 Examples:
          - Overview: examples/index.md
          - 🐍 Python:
@@ -247,6 +249,7 @@ nav:
      - Data management: concepts/data_management.md
  - Guides:
      - Working with tables: guides/tables.md
+      - Working with SQL: guides/sql_querying.md
      - Building an ANN index: ann_indexes.md
      - Vector Search: search.md
      - Full-text search (native): fts.md
@@ -323,6 +326,7 @@ nav:
      - Pandas and PyArrow: python/pandas_and_pyarrow.md
      - Polars: python/polars_arrow.md
      - DuckDB: python/duckdb.md
+      - Datafusion: python/datafusion.md
      - LangChain 🦜️🔗↗: integrations/langchain.md
      - LangChain.js 🦜️🔗↗: https://js.langchain.com/docs/integrations/vectorstores/lancedb
      - LlamaIndex 🦙↗: integrations/llamaIndex.md
@@ -331,6 +335,7 @@ nav:
      - PromptTools: integrations/prompttools.md
      - dlt: integrations/dlt.md
      - phidata: integrations/phidata.md
+      - Genkit: integrations/genkit.md
  - Examples:
      - examples/index.md
      - 🐍 Python:
--- a/docs/overrides/partials/main.html
+++ b/docs/overrides/partials/main.html
@@ -0,0 +1,5 @@
+{% extends "base.html" %}
+
+{% block announce %}
+  📚 Starting June 1st, 2025, please use <a href="https://lancedb.github.io/documentation" target="_blank" rel="noopener noreferrer">lancedb.github.io/documentation</a> for the latest docs.
+{% endblock %}
--- a/docs/src/ann_indexes.md
+++ b/docs/src/ann_indexes.md
@@ -291,7 +291,7 @@ Product quantization can lead to approximately `16 * sizeof(float32) / 1 = 64` t

 `num_partitions` is used to decide how many partitions the first level `IVF` index uses.
 Higher number of partitions could lead to more efficient I/O during queries and better accuracy, but it takes much more time to train.
-On `SIFT-1M` dataset, our benchmark shows that keeping each partition 1K-4K rows lead to a good latency / recall.
+On `SIFT-1M` dataset, our benchmark shows that keeping each partition 4K-8K rows lead to a good latency / recall.

 `num_sub_vectors` specifies how many Product Quantization (PQ) short codes to generate on each vector. The number should be a factor of the vector dimension. Because
 PQ is a lossy compression of the original vector, a higher `num_sub_vectors` usually results in
--- a/docs/src/assets/hero-header.png
+++ b/docs/src/assets/hero-header.png
--- a/docs/src/assets/lancedb.png
+++ b/docs/src/assets/lancedb.png
--- a/docs/src/guides/sql_querying.md
+++ b/docs/src/guides/sql_querying.md
@@ -0,0 +1,66 @@
+You can use DuckDB and Apache Datafusion to query your LanceDB tables using SQL.
+This guide will show how to query Lance tables them using both.
+
+We will re-use the dataset [created previously](./pandas_and_pyarrow.md):
+
+```python
+import lancedb
+
+db = lancedb.connect("data/sample-lancedb")
+data = [
+    {"vector": [3.1, 4.1], "item": "foo", "price": 10.0},
+    {"vector": [5.9, 26.5], "item": "bar", "price": 20.0}
+]
+table = db.create_table("pd_table", data=data)
+```
+
+## Querying a LanceDB Table with DuckDb
+
+The `to_lance` method converts the LanceDB table to a `LanceDataset`, which is accessible to DuckDB through the Arrow compatibility layer.
+To query the resulting Lance dataset in DuckDB, all you need to do is reference the dataset by the same name in your SQL query.
+
+```python
+import duckdb
+
+arrow_table = table.to_lance()
+
+duckdb.query("SELECT * FROM arrow_table")
+```
+
+```
+┌─────────────┬─────────┬────────┐
+│   vector    │  item   │ price  │
+│   float[]   │ varchar │ double │
+├─────────────┼─────────┼────────┤
+│ [3.1, 4.1]  │ foo     │   10.0 │
+│ [5.9, 26.5] │ bar     │   20.0 │
+└─────────────┴─────────┴────────┘
+```
+
+## Querying a LanceDB Table with Apache Datafusion
+
+Have the required imports before doing any querying.
+
+=== "Python"
+    ```python
+    --8<-- "python/python/tests/docs/test_guide_tables.py:import-lancedb"
+    --8<-- "python/python/tests/docs/test_guide_tables.py:import-session-context"
+    --8<-- "python/python/tests/docs/test_guide_tables.py:import-ffi-dataset"
+    ```
+
+Register the table created with the Datafusion session context.
+
+=== "Python"
+    ```python
+    --8<-- "python/python/tests/docs/test_guide_tables.py:lance_sql_basic"
+    ```
+
+```
+┌─────────────┬─────────┬────────┐
+│   vector    │  item   │ price  │
+│   float[]   │ varchar │ double │
+├─────────────┼─────────┼────────┤
+│ [3.1, 4.1]  │ foo     │   10.0 │
+│ [5.9, 26.5] │ bar     │   20.0 │
+└─────────────┴─────────┴────────┘
+```
--- a/docs/src/guides/tables.md
+++ b/docs/src/guides/tables.md
@@ -765,7 +765,7 @@ This can be used to update zero to all rows depending on how many rows match the
        ];
        const tbl = await db.createTable("my_table", data)

-        await tbl.update({ 
+        await tbl.update({
            values: { vector: [10, 10] },
            where: "x = 2"
        });
@@ -787,9 +787,9 @@ This can be used to update zero to all rows depending on how many rows match the
        ];
        const tbl = await db.createTable("my_table", data)

-        await tbl.update({ 
-            where: "x = 2", 
-            values: { vector: [10, 10] } 
+        await tbl.update({
+            where: "x = 2",
+            values: { vector: [10, 10] }
        });
        ```

--- a/docs/src/integrations/genkit.md
+++ b/docs/src/integrations/genkit.md
@@ -0,0 +1,183 @@
+### genkitx-lancedb
+This is a lancedb plugin for genkit framework. It allows you to use LanceDB for ingesting and rereiving data using genkit framework.
+
+![integration-banner-genkit](https://github.com/user-attachments/assets/a6cc28af-98e9-4425-b87c-7ab139bd7893)
+
+### Installation
+```bash
+pnpm install genkitx-lancedb
+```
+
+### Usage
+
+Adding LanceDB plugin to your genkit instance.
+
+```ts
+import { lancedbIndexerRef, lancedb, lancedbRetrieverRef, WriteMode } from 'genkitx-lancedb';
+import { textEmbedding004, vertexAI } from '@genkit-ai/vertexai';
+import { gemini } from '@genkit-ai/vertexai';
+import { z, genkit } from 'genkit';
+import { Document } from 'genkit/retriever';
+import { chunk } from 'llm-chunk';
+import { readFile } from 'fs/promises';
+import path from 'path';
+import pdf from 'pdf-parse/lib/pdf-parse';
+
+const ai = genkit({
+  plugins: [
+    // vertexAI provides the textEmbedding004 embedder
+    vertexAI(),
+
+    // the local vector store requires an embedder to translate from text to vector
+    lancedb([
+      {
+        dbUri: '.db', // optional lancedb uri, default to .db
+        tableName: 'table', // optional table name, default to table
+        embedder: textEmbedding004,
+      },
+    ]),
+  ],
+});
+```
+
+You can run this app with the following command:
+```bash
+genkit start -- tsx --watch src/index.ts
+```
+
+This'll add LanceDB as a retriever and indexer to the genkit instance. You can see it in the GUI view
+<img width="1710" alt="Screenshot 2025-05-11 at 7 21 05 PM" src="https://github.com/user-attachments/assets/e752f7f4-785b-4797-a11e-72ab06a531b7" />
+
+**Testing retrieval on a sample table**
+Let's see the raw retrieval results
+
+<img width="1710" alt="Screenshot 2025-05-11 at 7 21 05 PM" src="https://github.com/user-attachments/assets/b8d356ed-8421-4790-8fc0-d6af563b9657" />
+On running this query, you'll 5 results fetched from the lancedb table, where each result looks something like this:
+<img width="1417" alt="Screenshot 2025-05-11 at 7 21 18 PM" src="https://github.com/user-attachments/assets/77429525-36e2-4da6-a694-e58c1cf9eb83" />
+
+
+
+## Creating a custom RAG flow
+
+Now that we've seen how you can use LanceDB for in a genkit pipeline, let's refine the flow and create a RAG. A RAG flow will consist of an index and a retreiver with its outputs postprocessed an fed into an LLM for final response
+
+### Creating custom indexer flows
+You can also create custom indexer flows, utilizing more options and features provided by LanceDB.
+
+```ts
+export const menuPdfIndexer = lancedbIndexerRef({
+   // Using all defaults, for dbUri, tableName, and embedder, etc
+});
+
+const chunkingConfig = {
+  minLength: 1000,
+  maxLength: 2000,
+  splitter: 'sentence',
+  overlap: 100,
+  delimiters: '',
+} as any;
+
+
+async function extractTextFromPdf(filePath: string) {
+  const pdfFile = path.resolve(filePath);
+  const dataBuffer = await readFile(pdfFile);
+  const data = await pdf(dataBuffer);
+  return data.text;
+}
+
+export const indexMenu = ai.defineFlow(
+  {
+    name: 'indexMenu',
+    inputSchema: z.string().describe('PDF file path'),
+    outputSchema: z.void(),
+  },
+  async (filePath: string) => {
+    filePath = path.resolve(filePath);
+
+    // Read the pdf.
+    const pdfTxt = await ai.run('extract-text', () =>
+      extractTextFromPdf(filePath)
+    );
+
+    // Divide the pdf text into segments.
+    const chunks = await ai.run('chunk-it', async () =>
+      chunk(pdfTxt, chunkingConfig)
+    );
+
+    // Convert chunks of text into documents to store in the index.
+    const documents = chunks.map((text) => {
+      return Document.fromText(text, { filePath });
+    });
+
+    // Add documents to the index.
+    await ai.index({
+      indexer: menuPdfIndexer,
+      documents,
+      options: {
+        writeMode: WriteMode.Overwrite,
+      } as any
+    });
+  }
+);
+```
+
+<img width="1316" alt="Screenshot 2025-05-11 at 8 35 56 PM" src="https://github.com/user-attachments/assets/e2a20ce4-d1d0-4fa2-9a84-f2cc26e3a29f" />
+
+In your console, you can see the logs
+
+<img width="511" alt="Screenshot 2025-05-11 at 7 19 14 PM" src="https://github.com/user-attachments/assets/243f26c5-ed38-40b6-b661-002f40f0423a" />
+
+### Creating custom retriever flows
+You can also create custom retriever flows, utilizing more options and features provided by LanceDB.
+```ts
+export const menuRetriever = lancedbRetrieverRef({
+  tableName: "table", // Use the same table name as the indexer.
+  displayName: "Menu", // Use a custom display name.
+
+export const menuQAFlow = ai.defineFlow(
+  { name: "Menu", inputSchema: z.string(), outputSchema: z.string() },
+  async (input: string) => {
+    // retrieve relevant documents
+    const docs = await ai.retrieve({
+      retriever: menuRetriever,
+      query: input,
+      options: { 
+        k: 3,
+      },
+    });
+
+    const extractedContent = docs.map(doc => {
+      if (doc.content && Array.isArray(doc.content) && doc.content.length > 0) {
+        if (doc.content[0].media && doc.content[0].media.url) {
+          return doc.content[0].media.url;
+        }
+      }
+      return "No content found";
+    });
+
+    console.log("Extracted content:", extractedContent);
+
+    const { text } = await ai.generate({
+      model: gemini('gemini-2.0-flash'),
+      prompt: `
+You are acting as a helpful AI assistant that can answer 
+questions about the food available on the menu at Genkit Grub Pub.
+
+Use only the context provided to answer the question.
+If you don't know, do not make up an answer.
+Do not add or change items on the menu.
+
+Context:
+${extractedContent.join('\n\n')}
+
+Question: ${input}`,
+      docs,
+    });
+    
+    return text;
+  }
+);
+```
+Now using our retrieval flow, we can ask question about the ingsted PDF
+<img width="1306" alt="Screenshot 2025-05-11 at 7 18 45 PM" src="https://github.com/user-attachments/assets/86c66b13-7c12-4d5f-9d81-ae36bfb1c346" />
+
--- a/docs/src/js/classes/MergeInsertBuilder.md
+++ b/docs/src/js/classes/MergeInsertBuilder.md
@@ -33,7 +33,7 @@ Construct a MergeInsertBuilder. __Internal use only.__
 ### execute()

 ```ts
-execute(data): Promise<MergeResult>
+execute(data, execOptions?): Promise<MergeResult>
 ```

 Executes the merge insert operation
@@ -42,6 +42,8 @@ Executes the merge insert operation

 * **data**: [`Data`](../type-aliases/Data.md)

+* **execOptions?**: `Partial`&lt;[`WriteExecutionOptions`](../interfaces/WriteExecutionOptions.md)&gt;
+
 #### Returns

 `Promise`&lt;[`MergeResult`](../interfaces/MergeResult.md)&gt;
--- a/docs/src/js/globals.md
+++ b/docs/src/js/globals.md
@@ -72,6 +72,7 @@
 - [UpdateOptions](interfaces/UpdateOptions.md)
 - [UpdateResult](interfaces/UpdateResult.md)
 - [Version](interfaces/Version.md)
+- [WriteExecutionOptions](interfaces/WriteExecutionOptions.md)

 ## Type Aliases

--- a/docs/src/js/interfaces/WriteExecutionOptions.md
+++ b/docs/src/js/interfaces/WriteExecutionOptions.md
@@ -0,0 +1,26 @@
+[**@lancedb/lancedb**](../README.md) • **Docs**
+
+***
+
+[@lancedb/lancedb](../globals.md) / WriteExecutionOptions
+
+# Interface: WriteExecutionOptions
+
+## Properties
+
+### timeoutMs?
+
+```ts
+optional timeoutMs: number;
+```
+
+Maximum time to run the operation before cancelling it.
+
+By default, there is a 30-second timeout that is only enforced after the
+first attempt. This is to prevent spending too long retrying to resolve
+conflicts. For example, if a write attempt takes 20 seconds and fails,
+the second attempt will be cancelled after 10 seconds, hitting the
+30-second timeout. However, a write that takes one hour and succeeds on the
+first attempt will not be cancelled.
+
+When this is set, the timeout is enforced on all attempts, including the first.
--- a/docs/src/python/datafusion.md
+++ b/docs/src/python/datafusion.md
@@ -0,0 +1,53 @@
+# Apache Datafusion
+
+In Python, LanceDB tables can also be queried with [Apache Datafusion](https://datafusion.apache.org/), an extensible query engine written in Rust that uses Apache Arrow as its in-memory format. This means you can write complex SQL queries to analyze your data in LanceDB.
+
+This integration is done via [Datafusion FFI](https://docs.rs/datafusion-ffi/latest/datafusion_ffi/), which provides a native integration between LanceDB and Datafusion.
+The Datafusion FFI allows to pass down column selections and basic filters to LanceDB, reducing the amount of scanned data when executing your query. Additionally, the integration allows streaming data from LanceDB tables which allows to do aggregation larger-than-memory.
+
+We can demonstrate this by first installing `datafusion` and `lancedb`.
+
+```shell
+pip install datafusion lancedb
+```
+
+We will re-use the dataset [created previously](./pandas_and_pyarrow.md):
+
+```python
+import lancedb
+
+from datafusion import SessionContext
+from lance import FFILanceTableProvider
+
+db = lancedb.connect("data/sample-lancedb")
+data = [
+    {"vector": [3.1, 4.1], "item": "foo", "price": 10.0},
+    {"vector": [5.9, 26.5], "item": "bar", "price": 20.0}
+]
+lance_table = db.create_table("lance_table", data)
+
+ctx = SessionContext()
+
+ffi_lance_table = FFILanceTableProvider(
+    lance_table.to_lance(), with_row_id=True, with_row_addr=True
+)
+ctx.register_table_provider("ffi_lance_table", ffi_lance_table)
+```
+
+The `to_lance` method converts the LanceDB table to a `LanceDataset`, which is accessible to Datafusion through the Datafusion FFI integration layer.
+To query the resulting Lance dataset in Datafusion, you first need to register the dataset with Datafusion and then just reference it by the same name in your SQL query.
+
+```python
+ctx.table("ffi_lance_table")
+ctx.sql("SELECT * FROM ffi_lance_table")
+```
+
+```
+┌─────────────┬─────────┬────────┬─────────────────┬─────────────────┐
+│   vector    │  item   │ price  │ _rowid          │ _rowaddr        │
+│   float[]   │ varchar │ double │ bigint unsigned │ bigint unsigned │
+├─────────────┼─────────┼────────┼─────────────────┼─────────────────┤
+│ [3.1, 4.1]  │ foo     │   10.0 │               0 │               0 │
+│ [5.9, 26.5] │ bar     │   20.0 │               1 │               1 │
+└─────────────┴─────────┴────────┴─────────────────┴─────────────────┘
+```
--- a/java/core/pom.xml
+++ b/java/core/pom.xml
@@ -8,7 +8,7 @@
    <parent>
        <groupId>com.lancedb</groupId>
        <artifactId>lancedb-parent</artifactId>
-        <version>0.19.1-beta.1</version>
+        <version>0.20.0-beta.2</version>
        <relativePath>../pom.xml</relativePath>
    </parent>

--- a/java/pom.xml
+++ b/java/pom.xml
@@ -6,7 +6,7 @@

    <groupId>com.lancedb</groupId>
    <artifactId>lancedb-parent</artifactId>
-    <version>0.19.1-beta.1</version>
+    <version>0.20.0-beta.2</version>
    <packaging>pom</packaging>

    <name>LanceDB Parent</name>
--- a/node/package-lock.json
+++ b/node/package-lock.json
@@ -1,12 +1,12 @@
 {
  "name": "vectordb",
-  "version": "0.19.1-beta.1",
+  "version": "0.20.0-beta.2",
  "lockfileVersion": 3,
  "requires": true,
  "packages": {
    "": {
      "name": "vectordb",
-      "version": "0.19.1-beta.1",
+      "version": "0.20.0-beta.2",
      "cpu": [
        "x64",
        "arm64"
@@ -52,11 +52,11 @@
        "uuid": "^9.0.0"
      },
      "optionalDependencies": {
-        "@lancedb/vectordb-darwin-arm64": "0.19.1-beta.1",
-        "@lancedb/vectordb-darwin-x64": "0.19.1-beta.1",
-        "@lancedb/vectordb-linux-arm64-gnu": "0.19.1-beta.1",
-        "@lancedb/vectordb-linux-x64-gnu": "0.19.1-beta.1",
-        "@lancedb/vectordb-win32-x64-msvc": "0.19.1-beta.1"
+        "@lancedb/vectordb-darwin-arm64": "0.20.0-beta.2",
+        "@lancedb/vectordb-darwin-x64": "0.20.0-beta.2",
+        "@lancedb/vectordb-linux-arm64-gnu": "0.20.0-beta.2",
+        "@lancedb/vectordb-linux-x64-gnu": "0.20.0-beta.2",
+        "@lancedb/vectordb-win32-x64-msvc": "0.20.0-beta.2"
      },
      "peerDependencies": {
        "@apache-arrow/ts": "^14.0.2",
@@ -327,65 +327,60 @@
      }
    },
    "node_modules/@lancedb/vectordb-darwin-arm64": {
-      "version": "0.19.1-beta.1",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-darwin-arm64/-/vectordb-darwin-arm64-0.19.1-beta.1.tgz",
-      "integrity": "sha512-Epvel0pF5TM6MtIWQ2KhqezqSSHTL3Wr7a2rGAwz6X/XY23i6DbMPpPs0HyeIDzDrhxNfE3cz3S+SiCA6xpR0g==",
+      "version": "0.20.0-beta.2",
+      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-darwin-arm64/-/vectordb-darwin-arm64-0.20.0-beta.2.tgz",
+      "integrity": "sha512-H9PmJ/5KSvstVzR8Q7T22+eHRjJZ2ef3aA3gdFxXvoMi3xQ0MGIxz23HuKHGTRT4tfl1nNnpOPb2W7Na8etK9w==",
      "cpu": [
        "arm64"
      ],
-      "license": "Apache-2.0",
      "optional": true,
      "os": [
        "darwin"
      ]
    },
    "node_modules/@lancedb/vectordb-darwin-x64": {
-      "version": "0.19.1-beta.1",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-darwin-x64/-/vectordb-darwin-x64-0.19.1-beta.1.tgz",
-      "integrity": "sha512-hOiUSlIoISbiXytp46hToi/r6sF5pImAsfbzCsIq8ExDV4TPa8fjbhcIT80vxxOwc2mpSSK4HsVJYod95RSbEQ==",
+      "version": "0.20.0-beta.2",
+      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-darwin-x64/-/vectordb-darwin-x64-0.20.0-beta.2.tgz",
+      "integrity": "sha512-9AQkv4tIys+vg0cplZtSE48o61jd7EnmuMkUht+vLORL5/HAma84eAoU9lXHT7zAtPAQmL+98Bfvcsx7fJ6mVw==",
      "cpu": [
        "x64"
      ],
-      "license": "Apache-2.0",
      "optional": true,
      "os": [
        "darwin"
      ]
    },
    "node_modules/@lancedb/vectordb-linux-arm64-gnu": {
-      "version": "0.19.1-beta.1",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-linux-arm64-gnu/-/vectordb-linux-arm64-gnu-0.19.1-beta.1.tgz",
-      "integrity": "sha512-/1JhGVDEngwrlM8o2TNW8G6nJ9U/VgHKAORmj/cTA7O30helJIoo9jfvUAUy+vZ4VoEwRXQbMI+gaYTg0l3MTg==",
+      "version": "0.20.0-beta.2",
+      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-linux-arm64-gnu/-/vectordb-linux-arm64-gnu-0.20.0-beta.2.tgz",
+      "integrity": "sha512-eQWoJz2ePml7NyEInTBeakWx56+5c6r2p3F+iHC5tsLuznn6eFX90koXJunRxH1WXHDN48ECUlEmKypgfEmn4w==",
      "cpu": [
        "arm64"
      ],
-      "license": "Apache-2.0",
      "optional": true,
      "os": [
        "linux"
      ]
    },
    "node_modules/@lancedb/vectordb-linux-x64-gnu": {
-      "version": "0.19.1-beta.1",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-linux-x64-gnu/-/vectordb-linux-x64-gnu-0.19.1-beta.1.tgz",
-      "integrity": "sha512-zNRGSSUt8nTJMmll4NdxhQjwxR8Rezq3T4dsRoiDts5ienMam5HFjYiZ3FkDZQo16rgq2BcbFuH1G8u1chywlg==",
+      "version": "0.20.0-beta.2",
+      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-linux-x64-gnu/-/vectordb-linux-x64-gnu-0.20.0-beta.2.tgz",
+      "integrity": "sha512-/+84U+Dt07m8Jk0b8h+SvOzlrynITPP3SDBOlB+OonwmGSxirXhc8gkfNZctgXOJYKMyRIRSsMHP/QNjOp2ajA==",
      "cpu": [
        "x64"
      ],
-      "license": "Apache-2.0",
      "optional": true,
      "os": [
        "linux"
      ]
    },
    "node_modules/@lancedb/vectordb-win32-x64-msvc": {
-      "version": "0.19.1-beta.1",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-win32-x64-msvc/-/vectordb-win32-x64-msvc-0.19.1-beta.1.tgz",
-      "integrity": "sha512-yV550AJGlsIFdm1KoHQPJ1TZx121ZXCIdebBtBZj3wOObIhyB/i0kZAtGvwjkmr7EYyfzt1EHZzbjSGVdehIAA==",
+      "version": "0.20.0-beta.2",
+      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-win32-x64-msvc/-/vectordb-win32-x64-msvc-0.20.0-beta.2.tgz",
+      "integrity": "sha512-bgdunAPnknBh/5oO+vr6RXMr6wb3hHugNPXcIidxYMQvgFa8uhaAKtgYkAKuoyUReOYo8DGtVkZxNUUpZbF7/A==",
      "cpu": [
        "x64"
      ],
-      "license": "Apache-2.0",
      "optional": true,
      "os": [
        "win32"
--- a/node/package.json
+++ b/node/package.json
@@ -1,6 +1,6 @@
 {
  "name": "vectordb",
-  "version": "0.19.1-beta.1",
+  "version": "0.20.0-beta.2",
  "description": " Serverless, low-latency vector database for AI applications",
  "private": false,
  "main": "dist/index.js",
@@ -89,10 +89,10 @@
    }
  },
  "optionalDependencies": {
-    "@lancedb/vectordb-darwin-x64": "0.19.1-beta.1",
-    "@lancedb/vectordb-darwin-arm64": "0.19.1-beta.1",
-    "@lancedb/vectordb-linux-x64-gnu": "0.19.1-beta.1",
-    "@lancedb/vectordb-linux-arm64-gnu": "0.19.1-beta.1",
-    "@lancedb/vectordb-win32-x64-msvc": "0.19.1-beta.1"
+    "@lancedb/vectordb-darwin-x64": "0.20.0-beta.2",
+    "@lancedb/vectordb-darwin-arm64": "0.20.0-beta.2",
+    "@lancedb/vectordb-linux-x64-gnu": "0.20.0-beta.2",
+    "@lancedb/vectordb-linux-arm64-gnu": "0.20.0-beta.2",
+    "@lancedb/vectordb-win32-x64-msvc": "0.20.0-beta.2"
  }
 }
--- a/nodejs/Cargo.toml
+++ b/nodejs/Cargo.toml
@@ -1,7 +1,7 @@
 [package]
 name = "lancedb-nodejs"
 edition.workspace = true
-version = "0.19.1-beta.1"
+version = "0.20.0-beta.2"
 license.workspace = true
 description.workspace = true
 repository.workspace = true
@@ -30,6 +30,7 @@ log.workspace = true

 # Workaround for build failure until we can fix it.
 aws-lc-sys = "=0.28.0"
+aws-lc-rs = "=1.13.0"

 [build-dependencies]
 napi-build = "2.1"
--- a/nodejs/test/table.test.ts
+++ b/nodejs/test/table.test.ts
@@ -349,7 +349,7 @@ describe("merge insert", () => {
      .mergeInsert("a")
      .whenMatchedUpdateAll()
      .whenNotMatchedInsertAll()
-      .execute(newData);
+      .execute(newData, { timeoutMs: 10_000 });
    expect(mergeInsertRes).toHaveProperty("version");
    expect(mergeInsertRes.version).toBe(2);
    expect(mergeInsertRes.numInsertedRows).toBe(1);
@@ -463,6 +463,20 @@ describe("merge insert", () => {
    res = res.sort((a, b) => a.a - b.a);
    expect(res).toEqual(expected);
  });
+
+  test("timeout", async () => {
+    const newData = [
+      { a: 2, b: "x" },
+      { a: 4, b: "z" },
+    ];
+    await expect(
+      table
+        .mergeInsert("a")
+        .whenMatchedUpdateAll()
+        .whenNotMatchedInsertAll()
+        .execute(newData, { timeoutMs: 0 }),
+    ).rejects.toThrow("merge insert timed out");
+  });
 });

 describe("When creating an index", () => {
@@ -1287,6 +1301,32 @@ describe("when dealing with tags", () => {
    await table.checkoutLatest();
    expect(await table.version()).toBe(4);
  });
+
+  it("can checkout and restore tags", async () => {
+    const conn = await connect(tmpDir.name, {
+      readConsistencyInterval: 0,
+    });
+
+    const table = await conn.createTable("my_table", [
+      { id: 1n, vector: [0.1, 0.2] },
+    ]);
+    expect(await table.version()).toBe(1);
+    expect(await table.countRows()).toBe(1);
+    const tagsManager = await table.tags();
+    const tag1 = "tag1";
+    await tagsManager.create(tag1, 1);
+    await table.add([{ id: 2n, vector: [0.3, 0.4] }]);
+    const tag2 = "tag2";
+    await tagsManager.create(tag2, 2);
+    expect(await table.version()).toBe(2);
+    await table.checkout(tag1);
+    expect(await table.version()).toBe(1);
+    await table.restore();
+    expect(await table.version()).toBe(3);
+    expect(await table.countRows()).toBe(1);
+    await table.add([{ id: 3n, vector: [0.5, 0.6] }]);
+    expect(await table.countRows()).toBe(2);
+  });
 });

 describe("when optimizing a dataset", () => {
@@ -1466,7 +1506,9 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
      ];
      const table = await db.createTable("test", data);
      await table.createIndex("text", {
-        config: Index.fts(),
+        config: Index.fts({
+          withPosition: true,
+        }),
      });

      const results = await table.search("lance").toArray();
@@ -1519,7 +1561,9 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
      ];
      const table = await db.createTable("test", data);
      await table.createIndex("text", {
-        config: Index.fts(),
+        config: Index.fts({
+          withPosition: true,
+        }),
      });

      const results = await table.search("world").toArray();
--- a/nodejs/lancedb/index.ts
+++ b/nodejs/lancedb/index.ts
@@ -86,7 +86,7 @@ export {
  ColumnAlteration,
 } from "./table";

-export { MergeInsertBuilder } from "./merge";
+export { MergeInsertBuilder, WriteExecutionOptions } from "./merge";

 export * as embedding from "./embedding";
 export * as rerankers from "./rerankers";
--- a/nodejs/lancedb/merge.ts
+++ b/nodejs/lancedb/merge.ts
@@ -75,7 +75,10 @@ export class MergeInsertBuilder {
   *
   * @returns {Promise<MergeResult>} the merge result
   */
-  async execute(data: Data): Promise<MergeResult> {
+  async execute(
+    data: Data,
+    execOptions?: Partial<WriteExecutionOptions>,
+  ): Promise<MergeResult> {
    let schema: Schema;
    if (this.#schema instanceof Promise) {
      schema = await this.#schema;
@@ -83,7 +86,28 @@ export class MergeInsertBuilder {
    } else {
      schema = this.#schema;
    }
+
+    if (execOptions?.timeoutMs !== undefined) {
+      this.#native.setTimeout(execOptions.timeoutMs);
+    }
+
    const buffer = await fromDataToBuffer(data, undefined, schema);
    return await this.#native.execute(buffer);
  }
 }
+
+export interface WriteExecutionOptions {
+  /**
+   * Maximum time to run the operation before cancelling it.
+   *
+   * By default, there is a 30-second timeout that is only enforced after the
+   * first attempt. This is to prevent spending too long retrying to resolve
+   * conflicts. For example, if a write attempt takes 20 seconds and fails,
+   * the second attempt will be cancelled after 10 seconds, hitting the
+   * 30-second timeout. However, a write that takes one hour and succeeds on the
+   * first attempt will not be cancelled.
+   *
+   * When this is set, the timeout is enforced on all attempts, including the first.
+   */
+  timeoutMs?: number;
+}
--- a/nodejs/npm/darwin-arm64/package.json
+++ b/nodejs/npm/darwin-arm64/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-darwin-arm64",
-	"version": "0.19.1-beta.1",
+	"version": "0.20.0-beta.2",
 	"os": ["darwin"],
 	"cpu": ["arm64"],
 	"main": "lancedb.darwin-arm64.node",
--- a/nodejs/npm/darwin-x64/package.json
+++ b/nodejs/npm/darwin-x64/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-darwin-x64",
-	"version": "0.19.1-beta.1",
+	"version": "0.20.0-beta.2",
 	"os": ["darwin"],
 	"cpu": ["x64"],
 	"main": "lancedb.darwin-x64.node",
--- a/nodejs/npm/linux-arm64-gnu/package.json
+++ b/nodejs/npm/linux-arm64-gnu/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-linux-arm64-gnu",
-	"version": "0.19.1-beta.1",
+	"version": "0.20.0-beta.2",
 	"os": ["linux"],
 	"cpu": ["arm64"],
 	"main": "lancedb.linux-arm64-gnu.node",
--- a/nodejs/npm/linux-arm64-musl/package.json
+++ b/nodejs/npm/linux-arm64-musl/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-linux-arm64-musl",
-	"version": "0.19.1-beta.1",
+	"version": "0.20.0-beta.2",
 	"os": ["linux"],
 	"cpu": ["arm64"],
 	"main": "lancedb.linux-arm64-musl.node",
--- a/nodejs/npm/linux-x64-gnu/package.json
+++ b/nodejs/npm/linux-x64-gnu/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-linux-x64-gnu",
-	"version": "0.19.1-beta.1",
+	"version": "0.20.0-beta.2",
 	"os": ["linux"],
 	"cpu": ["x64"],
 	"main": "lancedb.linux-x64-gnu.node",
--- a/nodejs/npm/linux-x64-musl/package.json
+++ b/nodejs/npm/linux-x64-musl/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-linux-x64-musl",
-	"version": "0.19.1-beta.1",
+	"version": "0.20.0-beta.2",
 	"os": ["linux"],
 	"cpu": ["x64"],
 	"main": "lancedb.linux-x64-musl.node",
--- a/nodejs/npm/win32-arm64-msvc/package.json
+++ b/nodejs/npm/win32-arm64-msvc/package.json
@@ -1,6 +1,6 @@
 {
  "name": "@lancedb/lancedb-win32-arm64-msvc",
-  "version": "0.19.1-beta.1",
+  "version": "0.20.0-beta.2",
  "os": [
    "win32"
  ],
--- a/nodejs/npm/win32-x64-msvc/package.json
+++ b/nodejs/npm/win32-x64-msvc/package.json
@@ -1,6 +1,6 @@
 {
 	"name": "@lancedb/lancedb-win32-x64-msvc",
-	"version": "0.19.1-beta.1",
+	"version": "0.20.0-beta.2",
 	"os": ["win32"],
 	"cpu": ["x64"],
 	"main": "lancedb.win32-x64-msvc.node",
--- a/nodejs/package-lock.json
+++ b/nodejs/package-lock.json
@@ -1,12 +1,12 @@
 {
  "name": "@lancedb/lancedb",
-  "version": "0.19.1-beta.1",
+  "version": "0.20.0-beta.2",
  "lockfileVersion": 3,
  "requires": true,
  "packages": {
    "": {
      "name": "@lancedb/lancedb",
-      "version": "0.19.1-beta.1",
+      "version": "0.20.0-beta.2",
      "cpu": [
        "x64",
        "arm64"
--- a/nodejs/package.json
+++ b/nodejs/package.json
@@ -11,7 +11,7 @@
    "ann"
  ],
  "private": false,
-  "version": "0.19.1-beta.1",
+  "version": "0.20.0-beta.2",
  "main": "dist/index.js",
  "exports": {
    ".": "./dist/index.js",
--- a/nodejs/src/index.rs
+++ b/nodejs/src/index.rs
@@ -125,32 +125,30 @@ impl Index {
        ascii_folding: Option<bool>,
    ) -> Self {
        let mut opts = FtsIndexBuilder::default();
-        let mut tokenizer_configs = opts.tokenizer_configs.clone();
        if let Some(with_position) = with_position {
            opts = opts.with_position(with_position);
        }
        if let Some(base_tokenizer) = base_tokenizer {
-            tokenizer_configs = tokenizer_configs.base_tokenizer(base_tokenizer);
+            opts = opts.base_tokenizer(base_tokenizer);
        }
        if let Some(language) = language {
-            tokenizer_configs = tokenizer_configs.language(&language).unwrap();
+            opts = opts.language(&language).unwrap();
        }
        if let Some(max_token_length) = max_token_length {
-            tokenizer_configs = tokenizer_configs.max_token_length(Some(max_token_length as usize));
+            opts = opts.max_token_length(Some(max_token_length as usize));
        }
        if let Some(lower_case) = lower_case {
-            tokenizer_configs = tokenizer_configs.lower_case(lower_case);
+            opts = opts.lower_case(lower_case);
        }
        if let Some(stem) = stem {
-            tokenizer_configs = tokenizer_configs.stem(stem);
+            opts = opts.stem(stem);
        }
        if let Some(remove_stop_words) = remove_stop_words {
-            tokenizer_configs = tokenizer_configs.remove_stop_words(remove_stop_words);
+            opts = opts.remove_stop_words(remove_stop_words);
        }
        if let Some(ascii_folding) = ascii_folding {
-            tokenizer_configs = tokenizer_configs.ascii_folding(ascii_folding);
+            opts = opts.ascii_folding(ascii_folding);
        }
-        opts.tokenizer_configs = tokenizer_configs;

        Self {
            inner: Mutex::new(Some(LanceDbIndex::FTS(opts))),
--- a/nodejs/src/merge.rs
+++ b/nodejs/src/merge.rs
@@ -1,6 +1,8 @@
 // SPDX-License-Identifier: Apache-2.0
 // SPDX-FileCopyrightText: Copyright The LanceDB Authors

+use std::time::Duration;
+
 use lancedb::{arrow::IntoArrow, ipc::ipc_file_to_batches, table::merge::MergeInsertBuilder};
 use napi::bindgen_prelude::*;
 use napi_derive::napi;
@@ -36,6 +38,11 @@ impl NativeMergeInsertBuilder {
        this
    }

+    #[napi]
+    pub fn set_timeout(&mut self, timeout: u32) {
+        self.inner.timeout(Duration::from_millis(timeout as u64));
+    }
+
    #[napi(catch_unwind)]
    pub async fn execute(&self, buf: Buffer) -> napi::Result<MergeResult> {
        let data = ipc_file_to_batches(buf.to_vec())
--- a/python/.bumpversion.toml
+++ b/python/.bumpversion.toml
@@ -1,5 +1,5 @@
 [tool.bumpversion]
-current_version = "0.22.1-beta.2"
+current_version = "0.23.0"
 parse = """(?x)
    (?P<major>0|[1-9]\\d*)\\.
    (?P<minor>0|[1-9]\\d*)\\.
--- a/python/Cargo.toml
+++ b/python/Cargo.toml
@@ -1,6 +1,6 @@
 [package]
 name = "lancedb-python"
-version = "0.22.1-beta.2"
+version = "0.23.0"
 edition.workspace = true
 description = "Python bindings for LanceDB"
 license.workspace = true
@@ -14,11 +14,11 @@ name = "_lancedb"
 crate-type = ["cdylib"]

 [dependencies]
-arrow = { version = "54.1", features = ["pyarrow"] }
+arrow = { version = "55.1", features = ["pyarrow"] }
 lancedb = { path = "../rust/lancedb", default-features = false }
 env_logger.workspace = true
-pyo3 = { version = "0.23", features = ["extension-module", "abi3-py39"] }
-pyo3-async-runtimes = { version = "0.23", features = [
+pyo3 = { version = "0.24", features = ["extension-module", "abi3-py39"] }
+pyo3-async-runtimes = { version = "0.24", features = [
    "attributes",
    "tokio-runtime",
 ] }
@@ -27,7 +27,7 @@ futures.workspace = true
 tokio = { version = "1.40", features = ["sync"] }

 [build-dependencies]
-pyo3-build-config = { version = "0.23", features = [
+pyo3-build-config = { version = "0.24", features = [
    "extension-module",
    "abi3-py39",
 ] }
--- a/python/pyproject.toml
+++ b/python/pyproject.toml
@@ -60,6 +60,7 @@ tests = [
    "pyarrow-stubs",
    "pylance>=0.25",
    "requests",
+    "datafusion",
 ]
 dev = [
    "ruff",
--- a/python/python/lancedb/_lancedb.pyi
+++ b/python/python/lancedb/_lancedb.pyi
@@ -51,7 +51,7 @@ class Table:
    async def version(self) -> int: ...
    async def checkout(self, version: Union[int, str]): ...
    async def checkout_latest(self): ...
-    async def restore(self, version: Optional[int] = None): ...
+    async def restore(self, version: Optional[Union[int, str]] = None): ...
    async def list_indices(self) -> list[IndexConfig]: ...
    async def delete(self, filter: str) -> DeleteResult: ...
    async def add_columns(self, columns: list[tuple[str, str]]) -> AddColumnsResult: ...
--- a/python/python/lancedb/index.py
+++ b/python/python/lancedb/index.py
@@ -102,7 +102,7 @@ class FTS:

    Attributes
    ----------
-    with_position : bool, default True
+    with_position : bool, default False
        Whether to store the position of the token in the document. Setting this
        to False can reduce the size of the index and improve indexing speed,
        but it will disable support for phrase queries.
@@ -118,25 +118,25 @@ class FTS:
        ignored.
    lower_case : bool, default True
        Whether to convert the token to lower case. This makes queries case-insensitive.
-    stem : bool, default False
+    stem : bool, default True
        Whether to stem the token. Stemming reduces words to their root form.
        For example, in English "running" and "runs" would both be reduced to "run".
-    remove_stop_words : bool, default False
+    remove_stop_words : bool, default True
        Whether to remove stop words. Stop words are common words that are often
        removed from text before indexing. For example, in English "the" and "and".
-    ascii_folding : bool, default False
+    ascii_folding : bool, default True
        Whether to fold ASCII characters. This converts accented characters to
        their ASCII equivalent. For example, "café" would be converted to "cafe".
    """

-    with_position: bool = True
+    with_position: bool = False
    base_tokenizer: Literal["simple", "raw", "whitespace"] = "simple"
    language: str = "English"
    max_token_length: Optional[int] = 40
    lower_case: bool = True
-    stem: bool = False
-    remove_stop_words: bool = False
-    ascii_folding: bool = False
+    stem: bool = True
+    remove_stop_words: bool = True
+    ascii_folding: bool = True


@dataclass
--- a/python/python/lancedb/merge.py
+++ b/python/python/lancedb/merge.py
@@ -4,6 +4,7 @@

 from __future__ import annotations

+from datetime import timedelta
 from typing import TYPE_CHECKING, List, Optional

 if TYPE_CHECKING:
@@ -31,6 +32,7 @@ class LanceMergeInsertBuilder(object):
        self._when_not_matched_insert_all = False
        self._when_not_matched_by_source_delete = False
        self._when_not_matched_by_source_condition = None
+        self._timeout = None

    def when_matched_update_all(
        self, *, where: Optional[str] = None
@@ -81,6 +83,7 @@ class LanceMergeInsertBuilder(object):
        new_data: DATA,
        on_bad_vectors: str = "error",
        fill_value: float = 0.0,
+        timeout: Optional[timedelta] = None,
    ) -> MergeInsertResult:
        """
        Executes the merge insert operation
@@ -98,10 +101,24 @@ class LanceMergeInsertBuilder(object):
            One of "error", "drop", "fill".
        fill_value: float, default 0.
            The value to use when filling vectors. Only used if on_bad_vectors="fill".
+        timeout: Optional[timedelta], default None
+            Maximum time to run the operation before cancelling it.
+
+            By default, there is a 30-second timeout that is only enforced after the
+            first attempt. This is to prevent spending too long retrying to resolve
+            conflicts. For example, if a write attempt takes 20 seconds and fails,
+            the second attempt will be cancelled after 10 seconds, hitting the
+            30-second timeout. However, a write that takes one hour and succeeds on the
+            first attempt will not be cancelled.
+
+            When this is set, the timeout is enforced on all attempts, including
+            the first.

        Returns
        -------
        MergeInsertResult
            version: the new version number of the table after doing merge insert.
        """
+        if timeout is not None:
+            self._timeout = timeout
        return self._table._do_merge(self, new_data, on_bad_vectors, fill_value)
--- a/python/python/lancedb/remote/table.py
+++ b/python/python/lancedb/remote/table.py
@@ -47,9 +47,6 @@ class RemoteTable(Table):
    def __repr__(self) -> str:
        return f"RemoteTable({self.db_name}.{self.name})"

-    def __len__(self) -> int:
-        self.count_rows(None)
-
    @property
    def schema(self) -> pa.Schema:
        """The [Arrow Schema](https://arrow.apache.org/docs/python/api/datatypes.html#)
@@ -100,7 +97,7 @@ class RemoteTable(Table):
    def checkout_latest(self):
        return LOOP.run(self._table.checkout_latest())

-    def restore(self, version: Optional[int] = None):
+    def restore(self, version: Optional[Union[int, str]] = None):
        return LOOP.run(self._table.restore(version))

    def list_indices(self) -> Iterable[IndexConfig]:
@@ -152,15 +149,15 @@ class RemoteTable(Table):
        *,
        replace: bool = False,
        wait_timeout: timedelta = None,
-        with_position: bool = True,
+        with_position: bool = False,
        # tokenizer configs:
        base_tokenizer: str = "simple",
        language: str = "English",
        max_token_length: Optional[int] = 40,
        lower_case: bool = True,
-        stem: bool = False,
-        remove_stop_words: bool = False,
-        ascii_folding: bool = False,
+        stem: bool = True,
+        remove_stop_words: bool = True,
+        ascii_folding: bool = True,
    ):
        config = FTS(
            with_position=with_position,
--- a/python/python/lancedb/table.py
+++ b/python/python/lancedb/table.py
@@ -620,6 +620,10 @@ class Table(ABC):
        """
        raise NotImplementedError

+    def __len__(self) -> int:
+        """The number of rows in this Table"""
+        return self.count_rows(None)
+
    @property
    @abstractmethod
    def embedding_functions(self) -> Dict[str, EmbeddingFunctionConfig]:
@@ -825,15 +829,15 @@ class Table(ABC):
        writer_heap_size: Optional[int] = 1024 * 1024 * 1024,
        use_tantivy: bool = True,
        tokenizer_name: Optional[str] = None,
-        with_position: bool = True,
+        with_position: bool = False,
        # tokenizer configs:
        base_tokenizer: BaseTokenizerType = "simple",
        language: str = "English",
        max_token_length: Optional[int] = 40,
        lower_case: bool = True,
-        stem: bool = False,
-        remove_stop_words: bool = False,
-        ascii_folding: bool = False,
+        stem: bool = True,
+        remove_stop_words: bool = True,
+        ascii_folding: bool = True,
        wait_timeout: Optional[timedelta] = None,
    ):
        """Create a full-text search index on the table.
@@ -863,7 +867,7 @@ class Table(ABC):
        use_tantivy: bool, default True
            If True, use the legacy full-text search implementation based on tantivy.
            If False, use the new full-text search implementation based on lance-index.
-        with_position: bool, default True
+        with_position: bool, default False
            Only available with use_tantivy=False
            If False, do not store the positions of the terms in the text.
            This can reduce the size of the index and improve indexing speed.
@@ -881,13 +885,13 @@ class Table(ABC):
        lower_case : bool, default True
            Whether to convert the token to lower case. This makes queries
            case-insensitive.
-        stem : bool, default False
+        stem : bool, default True
            Whether to stem the token. Stemming reduces words to their root form.
            For example, in English "running" and "runs" would both be reduced to "run".
-        remove_stop_words : bool, default False
+        remove_stop_words : bool, default True
            Whether to remove stop words. Stop words are common words that are often
            removed from text before indexing. For example, in English "the" and "and".
-        ascii_folding : bool, default False
+        ascii_folding : bool, default True
            Whether to fold ASCII characters. This converts accented characters to
            their ASCII equivalent. For example, "café" would be converted to "cafe".
        wait_timeout: timedelta, optional
@@ -1470,7 +1474,7 @@ class Table(ABC):
        """

    @abstractmethod
-    def restore(self, version: Optional[int] = None):
+    def restore(self, version: Optional[Union[int, str]] = None):
        """Restore a version of the table. This is an in-place operation.

        This creates a new version where the data is equivalent to the
@@ -1478,9 +1482,10 @@ class Table(ABC):

        Parameters
        ----------
-        version : int, default None
-            The version to restore. If unspecified then restores the currently
-            checked out version. If the currently checked out version is the
+        version : int or str, default None
+            The version number or version tag to restore.
+            If unspecified then restores the currently checked out version.
+            If the currently checked out version is the
            latest version then this is a no-op.
        """

@@ -1710,7 +1715,7 @@ class LanceTable(Table):
        """
        LOOP.run(self._table.checkout_latest())

-    def restore(self, version: Optional[int] = None):
+    def restore(self, version: Optional[Union[int, str]] = None):
        """Restore a version of the table. This is an in-place operation.

        This creates a new version where the data is equivalent to the
@@ -1718,9 +1723,10 @@ class LanceTable(Table):

        Parameters
        ----------
-        version : int, default None
-            The version to restore. If unspecified then restores the currently
-            checked out version. If the currently checked out version is the
+        version : int or str, default None
+            The version number or version tag to restore.
+            If unspecified then restores the currently checked out version.
+            If the currently checked out version is the
            latest version then this is a no-op.

        Examples
@@ -1738,12 +1744,20 @@ class LanceTable(Table):
        AddResult(version=2)
        >>> table.version
        2
+        >>> table.tags.create("v2", 2)
        >>> table.restore(1)
        >>> table.to_pandas()
               vector    type
        0  [1.1, 0.9]  vector
        >>> len(table.list_versions())
        3
+        >>> table.restore("v2")
+        >>> table.to_pandas()
+               vector    type
+        0  [1.1, 0.9]  vector
+        1  [0.5, 0.2]  vector
+        >>> len(table.list_versions())
+        4
        """
        if version is not None:
            LOOP.run(self._table.checkout(version))
@@ -1752,9 +1766,6 @@ class LanceTable(Table):
    def count_rows(self, filter: Optional[str] = None) -> int:
        return LOOP.run(self._table.count_rows(filter))

-    def __len__(self) -> int:
-        return self.count_rows()
-
    def __repr__(self) -> str:
        val = f"{self.__class__.__name__}(name={self.name!r}, version={self.version}"
        if self._conn.read_consistency_interval is not None:
@@ -1961,15 +1972,15 @@ class LanceTable(Table):
        writer_heap_size: Optional[int] = 1024 * 1024 * 1024,
        use_tantivy: bool = True,
        tokenizer_name: Optional[str] = None,
-        with_position: bool = True,
+        with_position: bool = False,
        # tokenizer configs:
        base_tokenizer: BaseTokenizerType = "simple",
        language: str = "English",
        max_token_length: Optional[int] = 40,
        lower_case: bool = True,
-        stem: bool = False,
-        remove_stop_words: bool = False,
-        ascii_folding: bool = False,
+        stem: bool = True,
+        remove_stop_words: bool = True,
+        ascii_folding: bool = True,
    ):
        if not use_tantivy:
            if not isinstance(field_names, str):
@@ -1979,6 +1990,7 @@ class LanceTable(Table):
                tokenizer_configs = {
                    "base_tokenizer": base_tokenizer,
                    "language": language,
+                    "with_position": with_position,
                    "max_token_length": max_token_length,
                    "lower_case": lower_case,
                    "stem": stem,
@@ -1989,7 +2001,6 @@ class LanceTable(Table):
                tokenizer_configs = self.infer_tokenizer_configs(tokenizer_name)

            config = FTS(
-                with_position=with_position,
                **tokenizer_configs,
            )

@@ -3705,6 +3716,7 @@ class AsyncTable:
                when_not_matched_insert_all=merge._when_not_matched_insert_all,
                when_not_matched_by_source_delete=merge._when_not_matched_by_source_delete,
                when_not_matched_by_source_condition=merge._when_not_matched_by_source_condition,
+                timeout=merge._timeout,
            ),
        )

@@ -3962,7 +3974,7 @@ class AsyncTable:
        """
        await self._inner.checkout_latest()

-    async def restore(self, version: Optional[int] = None):
+    async def restore(self, version: Optional[int | str] = None):
        """
        Restore the table to the currently checked out version

--- a/python/python/tests/docs/test_guide_tables.py
+++ b/python/python/tests/docs/test_guide_tables.py
@@ -25,6 +25,10 @@ import numpy as np
 from lancedb.pydantic import Vector, LanceModel

 # --8<-- [end:import-lancedb-pydantic]
+# --8<-- [start:import-session-context]
+from datafusion import SessionContext
+
+# --8<-- [end:import-session-context]
 # --8<-- [start:import-datetime]
 from datetime import timedelta

@@ -33,6 +37,10 @@ from datetime import timedelta
 from lancedb.embeddings import get_registry

 # --8<-- [end:import-embeddings]
+# --8<-- [start:import-ffi-dataset]
+from lance import FFILanceTableProvider
+
+# --8<-- [end:import-ffi-dataset]
 # --8<-- [start:import-pydantic-basemodel]
 from pydantic import BaseModel

@@ -341,6 +349,27 @@ def test_table_with_embedding():
    # --8<-- [end:create_table_with_embedding]


+def test_sql_query():
+    db = lancedb.connect("data/sample-lancedb")
+    data = [
+        {"vector": [1.1, 1.2], "lat": 45.5, "long": -122.7},
+        {"vector": [0.2, 1.8], "lat": 40.1, "long": -74.1},
+    ]
+    table = db.create_table("lance_table", data)
+
+    # --8<-- [start:lance_sql_basic]
+    ctx = SessionContext()
+    ffi_lance_table = FFILanceTableProvider(
+        table.to_lance(), with_row_id=False, with_row_addr=False
+    )
+
+    ctx.register_table_provider("ffi_lance_table", ffi_lance_table)
+    ctx.table("ffi_lance_table")
+
+    ctx.sql("SELECT vector FROM ffi_lance_table")
+    # --8<-- [end:lance_sql_basic]
+
+
@pytest.mark.skip
 async def test_table_with_embedding_async():
    async_db = await lancedb.connect_async("data/sample-lancedb")
--- a/python/python/tests/docs/test_search.py
+++ b/python/python/tests/docs/test_search.py
@@ -156,6 +156,9 @@ async def test_vector_search_async():
    # --8<-- [end:search_result_async_as_list]


+@pytest.mark.skipif(
+    os.name == "nt", reason="Need to fix https://github.com/lancedb/lance/issues/3905"
+)
 def test_fts_fuzzy_query():
    uri = "data/fuzzy-example"
    db = lancedb.connect(uri)
@@ -189,6 +192,9 @@ def test_fts_fuzzy_query():
    }


+@pytest.mark.skipif(
+    os.name == "nt", reason="Need to fix https://github.com/lancedb/lance/issues/3905"
+)
 def test_fts_boost_query():
    uri = "data/boost-example"
    db = lancedb.connect(uri)
@@ -234,6 +240,9 @@ def test_fts_boost_query():
    )


+@pytest.mark.skipif(
+    os.name == "nt", reason="Need to fix https://github.com/lancedb/lance/issues/3905"
+)
 def test_fts_native():
    # --8<-- [start:basic_fts]
    uri = "data/sample-lancedb"
@@ -282,6 +291,9 @@ def test_fts_native():
    # --8<-- [end:fts_incremental_index]


+@pytest.mark.skipif(
+    os.name == "nt", reason="Need to fix https://github.com/lancedb/lance/issues/3905"
+)
@pytest.mark.asyncio
 async def test_fts_native_async():
    # --8<-- [start:basic_fts_async]
--- a/python/python/tests/test_fts.py
+++ b/python/python/tests/test_fts.py
@@ -287,7 +287,7 @@ def test_search_fts_phrase_query(table):
        assert False
    except Exception:
        pass
-    table.create_fts_index("text", use_tantivy=False, replace=True)
+    table.create_fts_index("text", use_tantivy=False, with_position=True, replace=True)
    results = table.search("puppy").limit(100).to_list()
    phrase_results = table.search('"puppy runs"').limit(100).to_list()
    assert len(results) > len(phrase_results)
@@ -312,7 +312,7 @@ async def test_search_fts_phrase_query_async(async_table):
        assert False
    except Exception:
        pass
-    await async_table.create_index("text", config=FTS())
+    await async_table.create_index("text", config=FTS(with_position=True))
    results = await async_table.query().nearest_to_text("puppy").limit(100).to_list()
    phrase_results = (
        await async_table.query().nearest_to_text('"puppy runs"').limit(100).to_list()
@@ -649,7 +649,7 @@ def test_fts_on_list(mem_db: DBConnection):
        }
    )
    table = mem_db.create_table("test", data=data)
-    table.create_fts_index("text", use_tantivy=False)
+    table.create_fts_index("text", use_tantivy=False, with_position=True)

    res = table.search("lance").limit(5).to_list()
    assert len(res) == 3
--- a/python/python/tests/test_remote_db.py
+++ b/python/python/tests/test_remote_db.py
@@ -149,6 +149,24 @@ async def test_async_checkout():
        assert await table.count_rows() == 300


+def test_table_len_sync():
+    def handler(request):
+        if request.path == "/v1/table/test/create/?mode=create":
+            request.send_response(200)
+            request.send_header("Content-Type", "application/json")
+            request.end_headers()
+            request.wfile.write(b"{}")
+
+        request.send_response(200)
+        request.send_header("Content-Type", "application/json")
+        request.end_headers()
+        request.wfile.write(json.dumps(1).encode())
+
+    with mock_lancedb_connection(handler) as db:
+        table = db.create_table("test", [{"id": 1}])
+        assert len(table) == 1
+
+
@pytest.mark.asyncio
 async def test_http_error():
    request_id_holder = {"request_id": None}
--- a/python/python/tests/test_table.py
+++ b/python/python/tests/test_table.py
@@ -769,6 +769,29 @@ def test_restore(mem_db: DBConnection):
        table.restore(0)


+def test_restore_with_tags(mem_db: DBConnection):
+    table = mem_db.create_table(
+        "my_table",
+        data=[{"vector": [1.1, 0.9], "type": "vector"}],
+    )
+    tag = "tag1"
+    table.tags.create(tag, 1)
+    table.add([{"vector": [0.5, 0.2], "type": "vector"}])
+    table.restore(tag)
+    assert len(table.list_versions()) == 3
+    assert len(table) == 1
+    expected = table.to_arrow()
+
+    table.add([{"vector": [0.3, 0.3], "type": "vector"}])
+    table.checkout("tag1")
+    table.restore()
+    assert len(table.list_versions()) == 5
+    assert table.to_arrow() == expected
+
+    with pytest.raises(ValueError):
+        table.restore("tag_unknown")
+
+
 def test_merge(tmp_db: DBConnection, tmp_path):
    pytest.importorskip("lance")
    import lance
@@ -914,7 +937,7 @@ def test_merge_insert(mem_db: DBConnection):
        table.merge_insert("a")
        .when_matched_update_all()
        .when_not_matched_insert_all()
-        .execute(new_data)
+        .execute(new_data, timeout=timedelta(seconds=10))
    )
    assert merge_insert_res.version == 2
    assert merge_insert_res.num_inserted_rows == 1
@@ -990,6 +1013,12 @@ def test_merge_insert(mem_db: DBConnection):
    expected = pa.table({"a": [2, 4], "b": ["x", "z"]})
    assert table.to_arrow().sort_by("a") == expected

+    # timeout
+    with pytest.raises(Exception, match="merge insert timed out"):
+        table.merge_insert("a").when_matched_update_all().execute(
+            new_data, timeout=timedelta(0)
+        )
+

 # We vary the data format because there are slight differences in how
 # subschemas are handled in different formats
--- a/python/src/index.rs
+++ b/python/src/index.rs
@@ -3,7 +3,7 @@

 use lancedb::index::vector::IvfFlatIndexBuilder;
 use lancedb::index::{
-    scalar::{BTreeIndexBuilder, FtsIndexBuilder, TokenizerConfig},
+    scalar::{BTreeIndexBuilder, FtsIndexBuilder},
    vector::{IvfHnswPqIndexBuilder, IvfHnswSqIndexBuilder, IvfPqIndexBuilder},
    Index as LanceDbIndex,
 };
@@ -38,19 +38,17 @@ pub fn extract_index_params(source: &Option<Bound<'_, PyAny>>) -> PyResult<Lance
            "LabelList" => Ok(LanceDbIndex::LabelList(Default::default())),
            "FTS" => {
                let params = source.extract::<FtsParams>()?;
-                let inner_opts = TokenizerConfig::default()
+                let inner_opts = FtsIndexBuilder::default()
                    .base_tokenizer(params.base_tokenizer)
                    .language(&params.language)
                    .map_err(|_| PyValueError::new_err(format!("LanceDB does not support the requested language: '{}'", params.language)))?
+                    .with_position(params.with_position)
                    .lower_case(params.lower_case)
                    .max_token_length(params.max_token_length)
                    .remove_stop_words(params.remove_stop_words)
                    .stem(params.stem)
                    .ascii_folding(params.ascii_folding);
-                let mut opts = FtsIndexBuilder::default()
-                    .with_position(params.with_position);
-                opts.tokenizer_configs = inner_opts;
-                Ok(LanceDbIndex::FTS(opts))
+                Ok(LanceDbIndex::FTS(inner_opts))
            },
            "IvfFlat" => {
                let params = source.extract::<IvfFlatParams>()?;
--- a/python/src/table.rs
+++ b/python/src/table.rs
@@ -17,10 +17,10 @@ use lancedb::table::{
    Table as LanceDbTable,
 };
 use pyo3::{
-    exceptions::{PyIOError, PyKeyError, PyRuntimeError, PyValueError},
+    exceptions::{PyKeyError, PyRuntimeError, PyValueError},
    pyclass, pymethods,
-    types::{IntoPyDict, PyAnyMethods, PyDict, PyDictMethods, PyInt, PyString},
-    Bound, FromPyObject, PyAny, PyObject, PyRef, PyResult, Python,
+    types::{IntoPyDict, PyAnyMethods, PyDict, PyDictMethods},
+    Bound, FromPyObject, PyAny, PyRef, PyResult, Python,
 };
 use pyo3_async_runtimes::tokio::future_into_py;

@@ -520,25 +520,15 @@ impl Table {
        })
    }

-    pub fn checkout(self_: PyRef<'_, Self>, version: PyObject) -> PyResult<Bound<'_, PyAny>> {
+    pub fn checkout(self_: PyRef<'_, Self>, version: LanceVersion) -> PyResult<Bound<'_, PyAny>> {
        let inner = self_.inner_ref()?.clone();
        let py = self_.py();
-        let (is_int, int_value, string_value) = if let Ok(i) = version.downcast_bound::<PyInt>(py) {
-            let num: u64 = i.extract()?;
-            (true, num, String::new())
-        } else if let Ok(s) = version.downcast_bound::<PyString>(py) {
-            let str_value = s.to_string();
-            (false, 0, str_value)
-        } else {
-            return Err(PyIOError::new_err(
-                "version must be an integer or a string.",
-            ));
-        };
        future_into_py(py, async move {
-            if is_int {
-                inner.checkout(int_value).await.infer_error()
-            } else {
-                inner.checkout_tag(&string_value).await.infer_error()
+            match version {
+                LanceVersion::Version(version_num) => {
+                    inner.checkout(version_num).await.infer_error()
+                }
+                LanceVersion::Tag(tag) => inner.checkout_tag(&tag).await.infer_error(),
            }
        })
    }
@@ -551,12 +541,19 @@ impl Table {
    }

    #[pyo3(signature = (version=None))]
-    pub fn restore(self_: PyRef<'_, Self>, version: Option<u64>) -> PyResult<Bound<'_, PyAny>> {
+    pub fn restore(
+        self_: PyRef<'_, Self>,
+        version: Option<LanceVersion>,
+    ) -> PyResult<Bound<'_, PyAny>> {
        let inner = self_.inner_ref()?.clone();
+        let py = self_.py();

-        future_into_py(self_.py(), async move {
+        future_into_py(py, async move {
            if let Some(version) = version {
-                inner.checkout(version).await.infer_error()?;
+                match version {
+                    LanceVersion::Version(num) => inner.checkout(num).await.infer_error()?,
+                    LanceVersion::Tag(tag) => inner.checkout_tag(&tag).await.infer_error()?,
+                }
            }
            inner.restore().await.infer_error()
        })
@@ -652,6 +649,9 @@ impl Table {
            builder
                .when_not_matched_by_source_delete(parameters.when_not_matched_by_source_condition);
        }
+        if let Some(timeout) = parameters.timeout {
+            builder.timeout(timeout);
+        }

        future_into_py(self_.py(), async move {
            let res = builder.execute(Box::new(batches)).await.infer_error()?;
@@ -795,6 +795,12 @@ impl Table {
    }
 }

+#[derive(FromPyObject)]
+pub enum LanceVersion {
+    Version(u64),
+    Tag(String),
+}
+
 #[derive(FromPyObject)]
 #[pyo3(from_item_all)]
 pub struct MergeInsertParams {
@@ -804,6 +810,7 @@ pub struct MergeInsertParams {
    when_not_matched_insert_all: bool,
    when_not_matched_by_source_delete: bool,
    when_not_matched_by_source_condition: Option<String>,
+    timeout: Option<std::time::Duration>,
 }

 #[pyclass]
--- a/rust-toolchain.toml
+++ b/rust-toolchain.toml
@@ -1,2 +1,2 @@
 [toolchain]
-channel = "1.83.0"
+channel = "1.86.0"
--- a/rust/ffi/node/Cargo.toml
+++ b/rust/ffi/node/Cargo.toml
@@ -1,6 +1,6 @@
 [package]
 name = "lancedb-node"
-version = "0.19.1-beta.1"
+version = "0.20.0-beta.2"
 description = "Serverless, low-latency vector database for AI applications"
 license.workspace = true
 edition.workspace = true
--- a/rust/lancedb/Cargo.toml
+++ b/rust/lancedb/Cargo.toml
@@ -1,6 +1,6 @@
 [package]
 name = "lancedb"
-version = "0.19.1-beta.1"
+version = "0.20.0-beta.2"
 edition.workspace = true
 description = "LanceDB: A serverless, low-latency vector database for AI applications"
 license.workspace = true
@@ -60,15 +60,15 @@ reqwest = { version = "0.12.0", default-features = false, features = [
    "macos-system-configuration",
    "stream",
 ], optional = true }
-rand = { version = "0.8.3", features = ["small_rng"], optional = true }
+rand = { version = "0.9", features = ["small_rng"], optional = true }
 http = { version = "1", optional = true } # Matching what is in reqwest
 uuid = { version = "1.7.0", features = ["v4"], optional = true }
 polars-arrow = { version = ">=0.37,<0.40.0", optional = true }
 polars = { version = ">=0.37,<0.40.0", optional = true }
 hf-hub = { version = "0.4.1", optional = true, default-features = false, features = ["rustls-tls", "tokio", "ureq"]}
-candle-core = { version = "0.6.0", optional = true }
-candle-transformers = { version = "0.6.0", optional = true }
-candle-nn = { version = "0.6.0", optional = true }
+candle-core = { version = "0.9.1", optional = true }
+candle-transformers = { version = "0.9.1", optional = true }
+candle-nn = { version = "0.9.1", optional = true }
 tokenizers = { version = "0.19.1", optional = true }
 semver = { workspace = true }

@@ -78,7 +78,7 @@ bytemuck_derive.workspace = true

 [dev-dependencies]
 tempfile = "3.5.0"
-rand = { version = "0.8.3", features = ["small_rng"] }
+rand = { version = "0.9", features = ["small_rng"] }
 random_word = { version = "0.4.3", features = ["en"] }
 uuid = { version = "1.7.0", features = ["v4"] }
 walkdir = "2"
--- a/rust/lancedb/examples/full_text_search.rs
+++ b/rust/lancedb/examples/full_text_search.rs
@@ -51,7 +51,7 @@ fn create_some_records() -> Result<Box<dyn RecordBatchReader + Send>> {
                Arc::new(Int32Array::from_iter_values(0..TOTAL as i32)),
                Arc::new(StringArray::from_iter_values((0..TOTAL).map(|_| {
                    (0..n_terms)
-                        .map(|_| words[random::<usize>() % words.len()])
+                        .map(|_| words[random::<u32>() as usize % words.len()])
                        .collect::<Vec<_>>()
                        .join(" ")
                }))),
--- a/rust/lancedb/src/embeddings/sentence_transformers.rs
+++ b/rust/lancedb/src/embeddings/sentence_transformers.rs
@@ -214,7 +214,7 @@ impl SentenceTransformersEmbeddings {

        let embeddings = self
            .model
-            .forward(&input_ids, &token_type_ids)
+            .forward(&input_ids, &token_type_ids, None)
            // TODO: it'd be nice to support other devices
            .and_then(|output| output.to_device(&Device::Cpu))?;

@@ -310,7 +310,7 @@ impl SentenceTransformersEmbeddings {
        let embeddings = Tensor::stack(&tokens, 0)
            .and_then(|tokens| {
                let token_type_ids = tokens.zeros_like()?;
-                self.model.forward(&tokens, &token_type_ids)
+                self.model.forward(&tokens, &token_type_ids, None)
            })
            // TODO: it'd be nice to support other devices
            .and_then(|tokens| tokens.to_device(&Device::Cpu))
--- a/rust/lancedb/src/index/scalar.rs
+++ b/rust/lancedb/src/index/scalar.rs
@@ -51,35 +51,7 @@ pub struct BitmapIndexBuilder {}
 #[derive(Debug, Clone, Default)]
 pub struct LabelListIndexBuilder {}

-/// Builder for a full text search index
-///
-/// A full text search index is an index on a string column that allows for full text search
-#[derive(Debug, Clone)]
-pub struct FtsIndexBuilder {
-    /// Whether to store the position of the tokens
-    /// This is used for phrase queries
-    pub with_position: bool,
-
-    pub tokenizer_configs: TokenizerConfig,
-}
-
-impl Default for FtsIndexBuilder {
-    fn default() -> Self {
-        Self {
-            with_position: true,
-            tokenizer_configs: TokenizerConfig::default(),
-        }
-    }
-}
-
-impl FtsIndexBuilder {
-    /// Set the with_position flag
-    pub fn with_position(mut self, with_position: bool) -> Self {
-        self.with_position = with_position;
-        self
-    }
-}
-
 pub use lance_index::scalar::inverted::query::*;
-pub use lance_index::scalar::inverted::TokenizerConfig;
 pub use lance_index::scalar::FullTextSearchQuery;
+pub use lance_index::scalar::InvertedIndexParams as FtsIndexBuilder;
+pub use lance_index::scalar::InvertedIndexParams;
--- a/rust/lancedb/src/io/object_store.rs
+++ b/rust/lancedb/src/io/object_store.rs
@@ -197,16 +197,8 @@ mod test {

    #[tokio::test]
    async fn test_e2e() {
-        let dir1 = tempfile::tempdir()
-            .unwrap()
-            .into_path()
-            .canonicalize()
-            .unwrap();
-        let dir2 = tempfile::tempdir()
-            .unwrap()
-            .into_path()
-            .canonicalize()
-            .unwrap();
+        let dir1 = tempfile::tempdir().unwrap().keep().canonicalize().unwrap();
+        let dir2 = tempfile::tempdir().unwrap().keep().canonicalize().unwrap();

        let secondary_store = LocalFileSystem::new_with_prefix(dir2.to_str().unwrap()).unwrap();
        let object_store_wrapper = Arc::new(MirroringObjectStoreWrapper {
--- a/rust/lancedb/src/remote/table.rs
+++ b/rust/lancedb/src/remote/table.rs
@@ -758,8 +758,7 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
        let (request_id, response) = self.send_streaming(request, data, true).await?;
        let response = self.check_table_response(&request_id, response).await?;
        let body = response.text().await.err_to_http(request_id.clone())?;
-
-        if body.trim().is_empty() || body == "{}" {
+        if body.trim().is_empty() {
            // Backward compatible with old servers
            return Ok(AddResult { version: 0 });
        }
@@ -922,7 +921,7 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
        let response = self.check_table_response(&request_id, response).await?;
        let body = response.text().await.err_to_http(request_id.clone())?;

-        if body.trim().is_empty() || body == "{}" {
+        if body.trim().is_empty() {
            // Backward compatible with old servers
            return Ok(UpdateResult {
                rows_updated: 0,
@@ -950,12 +949,10 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
        let (request_id, response) = self.send(request, true).await?;
        let response = self.check_table_response(&request_id, response).await?;
        let body = response.text().await.err_to_http(request_id.clone())?;
-
-        if body == "{}" {
+        if body.trim().is_empty() {
            // Backward compatible with old servers
            return Ok(DeleteResult { version: 0 });
        }
-
        let delete_response: DeleteResult =
            serde_json::from_str(&body).map_err(|e| Error::Http {
                source: format!("Failed to parse delete response: {}", e).into(),
@@ -998,16 +995,12 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
            Index::Bitmap(_) => ("BITMAP", None),
            Index::LabelList(_) => ("LABEL_LIST", None),
            Index::FTS(fts) => {
-                let with_position = fts.with_position;
-                let configs = serde_json::to_value(fts.tokenizer_configs).map_err(|e| {
-                    Error::InvalidInput {
-                        message: format!("failed to serialize FTS index params {:?}", e),
-                    }
+                let params = serde_json::to_value(&fts).map_err(|e| Error::InvalidInput {
+                    message: format!("failed to serialize FTS index params {:?}", e),
                })?;
-                for (key, value) in configs.as_object().unwrap() {
+                for (key, value) in params.as_object().unwrap() {
                    body[key] = value.clone();
                }
-                body["with_position"] = serde_json::Value::Bool(with_position);
                ("FTS", None)
            }
            Index::Auto => {
@@ -1071,19 +1064,28 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
    ) -> Result<MergeResult> {
        self.check_mutable().await?;

+        let timeout = params.timeout;
+
        let query = MergeInsertRequest::try_from(params)?;
-        let request = self
+        let mut request = self
            .client
            .post(&format!("/v1/table/{}/merge_insert/", self.name))
            .query(&query)
            .header(CONTENT_TYPE, ARROW_STREAM_CONTENT_TYPE);

+        if let Some(timeout) = timeout {
+            // (If it doesn't fit into u64, it's not worth sending anyways.)
+            if let Ok(timeout_ms) = u64::try_from(timeout.as_millis()) {
+                request = request.header(REQUEST_TIMEOUT_HEADER, timeout_ms);
+            }
+        }
+
        let (request_id, response) = self.send_streaming(request, new_data, true).await?;

        let response = self.check_table_response(&request_id, response).await?;
        let body = response.text().await.err_to_http(request_id.clone())?;

-        if body.trim().is_empty() || body == "{}" {
+        if body.trim().is_empty() {
            // Backward compatible with old servers
            return Ok(MergeResult {
                version: 0,
@@ -1145,7 +1147,7 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
                let response = self.check_table_response(&request_id, response).await?;
                let body = response.text().await.err_to_http(request_id.clone())?;

-                if body.trim().is_empty() || body == "{}" {
+                if body.trim().is_empty() {
                    // Backward compatible with old servers
                    return Ok(AddColumnsResult { version: 0 });
                }
@@ -1198,7 +1200,7 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
        let response = self.check_table_response(&request_id, response).await?;
        let body = response.text().await.err_to_http(request_id.clone())?;

-        if body.trim().is_empty() || body == "{}" {
+        if body.trim().is_empty() {
            // Backward compatible with old servers
            return Ok(AlterColumnsResult { version: 0 });
        }
@@ -1223,7 +1225,7 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
        let response = self.check_table_response(&request_id, response).await?;
        let body = response.text().await.err_to_http(request_id.clone())?;

-        if body.trim().is_empty() || body == "{}" {
+        if body.trim().is_empty() {
            // Backward compatible with old servers
            return Ok(DropColumnsResult { version: 0 });
        }
@@ -1328,7 +1330,12 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
            self.name, index_name
        ));
        let (request_id, response) = self.send(request, true).await?;
-        self.check_table_response(&request_id, response).await?;
+        if response.status() == StatusCode::NOT_FOUND {
+            return Err(Error::IndexNotFound {
+                name: index_name.to_string(),
+            });
+        };
+        self.client.check_response(&request_id, response).await?;
        Ok(())
    }

@@ -1603,16 +1610,21 @@ mod tests {
    }

    #[rstest]
-    #[case(true)]
-    #[case(false)]
+    #[case("", 0)]
+    #[case("{}", 0)]
+    #[case(r#"{"request_id": "test-request-id"}"#, 0)]
+    #[case(r#"{"version": 43}"#, 43)]
    #[tokio::test]
-    async fn test_add_append(#[case] old_server: bool) {
+    async fn test_add_append(#[case] response_body: &str, #[case] expected_version: u64) {
        let data = RecordBatch::try_new(
            Arc::new(Schema::new(vec![Field::new("a", DataType::Int32, false)])),
            vec![Arc::new(Int32Array::from(vec![1, 2, 3]))],
        )
        .unwrap();

+        // Clone response_body to give it 'static lifetime for the closure
+        let response_body = response_body.to_string();
+
        let (sender, receiver) = std::sync::mpsc::channel();
        let table = Table::new_with_handler("my_table", move |mut request| {
            if request.url().path() == "/v1/table/my_table/insert/" {
@@ -1622,36 +1634,29 @@ mod tests {
                    .query_pairs()
                    .filter(|(k, _)| k == "mode")
                    .all(|(_, v)| v == "append"));
-
                assert_eq!(
                    request.headers().get("Content-Type").unwrap(),
                    ARROW_STREAM_CONTENT_TYPE
                );
-
                let mut body_out = reqwest::Body::from(Vec::new());
                std::mem::swap(request.body_mut().as_mut().unwrap(), &mut body_out);
                sender.send(body_out).unwrap();
-
-                if old_server {
-                    http::Response::builder().status(200).body("").unwrap()
-                } else {
-                    http::Response::builder()
-                        .status(200)
-                        .body(r#"{"version": 43}"#)
-                        .unwrap()
-                }
+                http::Response::builder()
+                    .status(200)
+                    .body(response_body.clone())
+                    .unwrap()
            } else {
                panic!("Unexpected request path: {}", request.url().path());
            }
        });
-
        let result = table
            .add(RecordBatchIterator::new([Ok(data.clone())], data.schema()))
            .execute()
            .await
            .unwrap();

-        assert_eq!(result.version, if old_server { 0 } else { 43 });
+        // Check version matches expected value
+        assert_eq!(result.version, expected_version);

        let body = receiver.recv().unwrap();
        let body = collect_body(body).await;
@@ -2451,14 +2456,10 @@ mod tests {
                    expected_body["metric_type"] = distance_type.to_lowercase().into();
                }
                if let Index::FTS(fts) = &params {
-                    expected_body["with_position"] = fts.with_position.into();
-                    expected_body["base_tokenizer"] = "simple".into();
-                    expected_body["language"] = "English".into();
-                    expected_body["max_token_length"] = 40.into();
-                    expected_body["lower_case"] = true.into();
-                    expected_body["stem"] = false.into();
-                    expected_body["remove_stop_words"] = false.into();
-                    expected_body["ascii_folding"] = false.into();
+                    let params = serde_json::to_value(fts).unwrap();
+                    for (key, value) in params.as_object().unwrap() {
+                        expected_body[key] = value.clone();
+                    }
                }

                assert_eq!(body, expected_body);
@@ -2884,6 +2885,22 @@ mod tests {
        table.drop_index("my_index").await.unwrap();
    }

+    #[tokio::test]
+    async fn test_drop_index_not_exists() {
+        let table = Table::new_with_handler("my_table", |request| {
+            assert_eq!(request.method(), "POST");
+            assert_eq!(
+                request.url().path(),
+                "/v1/table/my_table/index/my_index/drop/"
+            );
+            http::Response::builder().status(404).body("{}").unwrap()
+        });
+
+        // Assert that the error is IndexNotFound
+        let e = table.drop_index("my_index").await.unwrap_err();
+        assert!(matches!(e, Error::IndexNotFound { .. }));
+    }
+
    #[tokio::test]
    async fn test_wait_for_index() {
        let table = _make_table_with_indices(0);
--- a/rust/lancedb/src/table.rs
+++ b/rust/lancedb/src/table.rs
@@ -14,7 +14,7 @@ use datafusion_physical_plan::projection::ProjectionExec;
 use datafusion_physical_plan::repartition::RepartitionExec;
 use datafusion_physical_plan::union::UnionExec;
 use datafusion_physical_plan::ExecutionPlan;
-use futures::{StreamExt, TryStreamExt};
+use futures::{FutureExt, StreamExt, TryFutureExt};
 use lance::dataset::builder::DatasetBuilder;
 use lance::dataset::cleanup::RemovalStats;
 use lance::dataset::optimize::{compact_files, CompactionMetrics, IndexRemapperOptions};
@@ -80,11 +80,12 @@ pub mod merge;

 use crate::index::waiter::wait_for_index;
 pub use chrono::Duration;
-use futures::future::join_all;
+use futures::future::{join_all, Either};
 pub use lance::dataset::optimize::CompactionOptions;
 pub use lance::dataset::refs::{TagContents, Tags as LanceTags};
 pub use lance::dataset::scanner::DatasetRecordBatchStream;
 use lance::dataset::statistics::DatasetStatisticsExt;
+use lance_index::frag_reuse::FRAG_REUSE_INDEX_NAME;
 pub use lance_index::optimize::OptimizeOptions;
 use serde_with::skip_serializing_none;

@@ -423,68 +424,79 @@ pub trait Tags: Send + Sync {
    async fn update(&mut self, tag: &str, version: u64) -> Result<()>;
 }

-#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
 pub struct UpdateResult {
+    #[serde(default)]
    pub rows_updated: u64,
    // The commit version associated with the operation.
    // A version of `0` indicates compatibility with legacy servers that do not return
    /// a commit version.
+    #[serde(default)]
    pub version: u64,
 }

-#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
 pub struct AddResult {
    // The commit version associated with the operation.
    // A version of `0` indicates compatibility with legacy servers that do not return
    /// a commit version.
+    #[serde(default)]
    pub version: u64,
 }

-#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
 pub struct DeleteResult {
    // The commit version associated with the operation.
    // A version of `0` indicates compatibility with legacy servers that do not return
    /// a commit version.
+    #[serde(default)]
    pub version: u64,
 }

-#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
 pub struct MergeResult {
    // The commit version associated with the operation.
    // A version of `0` indicates compatibility with legacy servers that do not return
    /// a commit version.
+    #[serde(default)]
    pub version: u64,
    /// Number of inserted rows (for user statistics)
+    #[serde(default)]
    pub num_inserted_rows: u64,
    /// Number of updated rows (for user statistics)
+    #[serde(default)]
    pub num_updated_rows: u64,
    /// Number of deleted rows (for user statistics)
    /// Note: This is different from internal references to 'deleted_rows', since we technically "delete" updated rows during processing.
    /// However those rows are not shared with the user.
+    #[serde(default)]
    pub num_deleted_rows: u64,
 }

-#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
 pub struct AddColumnsResult {
    // The commit version associated with the operation.
    // A version of `0` indicates compatibility with legacy servers that do not return
    /// a commit version.
+    #[serde(default)]
    pub version: u64,
 }

-#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
 pub struct AlterColumnsResult {
    // The commit version associated with the operation.
    // A version of `0` indicates compatibility with legacy servers that do not return
    /// a commit version.
+    #[serde(default)]
    pub version: u64,
 }

-#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
 pub struct DropColumnsResult {
    // The commit version associated with the operation.
    // A version of `0` indicates compatibility with legacy servers that do not return
    /// a commit version.
+    #[serde(default)]
    pub version: u64,
 }

@@ -1966,16 +1978,12 @@ impl NativeTable {
        }

        let mut dataset = self.dataset.get_mut().await?;
-        let fts_params = lance_index::scalar::InvertedIndexParams {
-            with_position: fts_opts.with_position,
-            tokenizer_config: fts_opts.tokenizer_configs,
-        };
        dataset
            .create_index(
                &[field.name()],
                IndexType::Inverted,
                None,
-                &fts_params,
+                &fts_opts,
                replace,
            )
            .await?;
@@ -2003,7 +2011,7 @@ impl NativeTable {
    /// more information.
    pub async fn uses_v2_manifest_paths(&self) -> Result<bool> {
        let dataset = self.dataset.get().await?;
-        Ok(dataset.manifest_naming_scheme == ManifestNamingScheme::V2)
+        Ok(dataset.manifest_location().naming_scheme == ManifestNamingScheme::V2)
    }

    /// Migrate the table to use the new manifest path scheme.
@@ -2464,8 +2472,26 @@ impl BaseTable for NativeTable {
        } else {
            builder.when_not_matched_by_source(WhenNotMatchedBySource::Keep);
        }
-        let job = builder.try_build()?;
-        let (new_dataset, stats) = job.execute_reader(new_data).await?;
+
+        let future = if let Some(timeout) = params.timeout {
+            // The default retry timeout is 30s, so we pass the full timeout down
+            // as well in case it is longer than that.
+            let future = builder
+                .retry_timeout(timeout)
+                .try_build()?
+                .execute_reader(new_data);
+            Either::Left(tokio::time::timeout(timeout, future).map(|res| match res {
+                Ok(Ok((new_dataset, stats))) => Ok((new_dataset, stats)),
+                Ok(Err(e)) => Err(e.into()),
+                Err(_) => Err(Error::Runtime {
+                    message: "merge insert timed out".to_string(),
+                }),
+            }))
+        } else {
+            let job = builder.try_build()?;
+            Either::Right(job.execute_reader(new_data).map_err(|e| e.into()))
+        };
+        let (new_dataset, stats) = future.await?;
        let version = new_dataset.manifest().version;
        self.dataset.set_latest(new_dataset.as_ref().clone()).await;
        Ok(MergeResult {
@@ -2576,28 +2602,56 @@ impl BaseTable for NativeTable {
    async fn list_indices(&self) -> Result<Vec<IndexConfig>> {
        let dataset = self.dataset.get().await?;
        let indices = dataset.load_indices().await?;
-        futures::stream::iter(indices.as_slice()).then(|idx| async {
-            let stats = dataset.index_statistics(idx.name.as_str()).await?;
-            let stats: serde_json::Value = serde_json::from_str(&stats).map_err(|e| Error::Runtime {
-                message: format!("error deserializing index statistics: {}", e),
-            })?;
-            let index_type = stats.get("index_type").and_then(|v| v.as_str())
-            .ok_or_else(|| Error::Runtime {
-                message: "index statistics was missing index type".to_string(),
-            })?;
-            let index_type: crate::index::IndexType = index_type.parse().map_err(|e| Error::Runtime {
-                message: format!("error parsing index type: {}", e),
-            })?;
+        let results = futures::stream::iter(indices.as_slice()).then(|idx| async {
+
+            // skip Lance internal indexes
+            if idx.name == FRAG_REUSE_INDEX_NAME {
+                return None;
+            }
+
+            let stats = match dataset.index_statistics(idx.name.as_str()).await {
+                Ok(stats) => stats,
+                Err(e) => {
+                    log::warn!("Failed to get statistics for index {} ({}): {}", idx.name, idx.uuid, e);
+                    return None;
+                }
+            };
+
+            let stats: serde_json::Value = match serde_json::from_str(&stats) {
+                Ok(stats) => stats,
+                Err(e) => {
+                    log::warn!("Failed to deserialize index statistics for index {} ({}): {}", idx.name, idx.uuid, e);
+                    return None;
+                }
+            };
+
+            let Some(index_type) = stats.get("index_type").and_then(|v| v.as_str()) else {
+                log::warn!("Index statistics was missing 'index_type' field for index {} ({})", idx.name, idx.uuid);
+                return None;
+            };
+
+            let index_type: crate::index::IndexType = match index_type.parse() {
+                Ok(index_type) => index_type,
+                Err(e) => {
+                    log::warn!("Failed to parse index type for index {} ({}): {}", idx.name, idx.uuid, e);
+                    return None;
+                }
+            };

            let mut columns = Vec::with_capacity(idx.fields.len());
            for field_id in &idx.fields {
-                let field = dataset.schema().field_by_id(*field_id).ok_or_else(|| Error::Runtime { message: format!("The index with name {} and uuid {} referenced a field with id {} which does not exist in the schema", idx.name, idx.uuid, field_id) })?;
+                let Some(field) = dataset.schema().field_by_id(*field_id) else {
+                    log::warn!("The index {} ({}) referenced a field with id {} which does not exist in the schema", idx.name, idx.uuid, field_id);
+                    return None;
+                };
                columns.push(field.name.clone());
            }

            let name = idx.name.clone();
-            Ok(IndexConfig { index_type, columns, name })
-        }).try_collect::<Vec<_>>().await
+            Some(IndexConfig { index_type, columns, name })
+        }).collect::<Vec<_>>().await;
+
+        Ok(results.into_iter().flatten().collect())
    }

    fn dataset_uri(&self) -> &str {
@@ -2790,7 +2844,7 @@ mod tests {
    use super::*;
    use crate::connect;
    use crate::connection::ConnectBuilder;
-    use crate::index::scalar::BTreeIndexBuilder;
+    use crate::index::scalar::{BTreeIndexBuilder, BitmapIndexBuilder};
    use crate::query::{ExecutableQuery, QueryBase};

    #[tokio::test]
@@ -4242,4 +4296,65 @@ mod tests {
            }
        )
    }
+
+    #[tokio::test]
+    pub async fn test_list_indices_skip_frag_reuse() {
+        let tmp_dir = tempdir().unwrap();
+        let uri = tmp_dir.path().to_str().unwrap();
+
+        let conn = ConnectBuilder::new(uri).execute().await.unwrap();
+
+        let schema = Arc::new(Schema::new(vec![
+            Field::new("id", DataType::Int32, false),
+            Field::new("foo", DataType::Int32, true),
+        ]));
+        let batch = RecordBatch::try_new(
+            schema.clone(),
+            vec![
+                Arc::new(Int32Array::from_iter_values(0..100)),
+                Arc::new(Int32Array::from_iter_values(0..100)),
+            ],
+        )
+        .unwrap();
+
+        let table = conn
+            .create_table(
+                "test_list_indices_skip_frag_reuse",
+                RecordBatchIterator::new(vec![Ok(batch.clone())], batch.schema()),
+            )
+            .execute()
+            .await
+            .unwrap();
+
+        table
+            .add(RecordBatchIterator::new(
+                vec![Ok(batch.clone())],
+                batch.schema(),
+            ))
+            .execute()
+            .await
+            .unwrap();
+
+        table
+            .create_index(&["id"], Index::Bitmap(BitmapIndexBuilder {}))
+            .execute()
+            .await
+            .unwrap();
+
+        table
+            .optimize(OptimizeAction::Compact {
+                options: CompactionOptions {
+                    target_rows_per_fragment: 2_000,
+                    defer_index_remap: true,
+                    ..Default::default()
+                },
+                remap_options: None,
+            })
+            .await
+            .unwrap();
+
+        let result = table.list_indices().await.unwrap();
+        assert_eq!(result.len(), 1);
+        assert_eq!(result[0].index_type, crate::index::IndexType::Bitmap);
+    }
 }
--- a/rust/lancedb/src/table/merge.rs
+++ b/rust/lancedb/src/table/merge.rs
@@ -1,7 +1,7 @@
 // SPDX-License-Identifier: Apache-2.0
 // SPDX-FileCopyrightText: Copyright The LanceDB Authors

-use std::sync::Arc;
+use std::{sync::Arc, time::Duration};

 use arrow_array::RecordBatchReader;

@@ -21,6 +21,7 @@ pub struct MergeInsertBuilder {
    pub(crate) when_not_matched_insert_all: bool,
    pub(crate) when_not_matched_by_source_delete: bool,
    pub(crate) when_not_matched_by_source_delete_filt: Option<String>,
+    pub(crate) timeout: Option<Duration>,
 }

 impl MergeInsertBuilder {
@@ -33,6 +34,7 @@ impl MergeInsertBuilder {
            when_not_matched_insert_all: false,
            when_not_matched_by_source_delete: false,
            when_not_matched_by_source_delete_filt: None,
+            timeout: None,
        }
    }

@@ -84,6 +86,21 @@ impl MergeInsertBuilder {
        self
    }

+    /// Maximum time to run the operation before cancelling it.
+    ///
+    /// By default, there is a 30-second timeout that is only enforced after the
+    /// first attempt. This is to prevent spending too long retrying to resolve
+    /// conflicts. For example, if a write attempt takes 20 seconds and fails,
+    /// the second attempt will be cancelled after 10 seconds, hitting the
+    /// 30-second timeout. However, a write that takes one hour and succeeds on the
+    /// first attempt will not be cancelled.
+    ///
+    /// When this is set, the timeout is enforced on all attempts, including the first.
+    pub fn timeout(&mut self, timeout: Duration) -> &mut Self {
+        self.timeout = Some(timeout);
+        self
+    }
+
    /// Executes the merge insert operation
    ///
    /// Returns version and statistics about the merge operation including the number of rows
Author	SHA1	Message	Date
Lance Release	8f42b5874e	Bump version: 0.23.0-beta.3 → 0.23.0	2025-06-04 21:07:39 +00:00
Lance Release	274f19f560	Bump version: 0.23.0-beta.2 → 0.23.0-beta.3	2025-06-04 21:07:38 +00:00
Will Jones	fbcbc75b5b	feat: upgrade lance to stable version (#2420 ) Adds a script to change the lance dependency easily. To make this change, I just had to run: ```bash python ci/set_lance_version.py stable ``` <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - New Features - Added a script to automate updating the Lance package version in project dependencies. - Chores - Updated workflows to improve lockfile management and automate updates during releases and publishing. - Switched Lance dependencies from git-based references to fixed version numbers for improved stability. - Enhanced lockfile update script with an option to amend commits and quieter output. <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Co-authored-by: coderabbitai[bot] <136622811+coderabbitai[bot]@users.noreply.github.com>	2025-06-04 13:34:30 -07:00
Will Jones	008f389bd0	ci: commit updated Cargo.lock (#2418 ) Follow up to #2416 Forgot to do `git add`. Also need to delete old actions updating package lock. <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Chores - Removed legacy workflows related to updating package lock files. - Improved the update lockfiles script to ensure updated lockfiles are always included in amended commits. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-06-04 08:40:38 -07:00
Lance Release	91af6518d9	Updating package-lock.json	2025-06-04 07:15:07 +00:00
Lance Release	af6819762c	Updating package-lock.json	2025-06-04 07:14:50 +00:00
Lance Release	7acece493d	Bump version: 0.20.0-beta.1 → 0.20.0-beta.2	2025-06-04 07:14:39 +00:00
Lance Release	20e017fedc	Bump version: 0.23.0-beta.1 → 0.23.0-beta.2	2025-06-04 07:13:44 +00:00
Jack Ye	74e578b3c8	feat: upgrade lance to v0.29.0-beta.2 (#2419 ) <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Chores - Updated various internal dependencies to newer versions for improved stability and compatibility. - Increased the version number for the Python package. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-06-03 15:16:26 -07:00
Lance Release	d92d9eb3d2	Updating package-lock.json	2025-06-03 16:28:18 +00:00
Lance Release	b6cdce7bc9	Updating package-lock.json	2025-06-03 16:28:02 +00:00
Lance Release	316b406265	Bump version: 0.20.0-beta.0 → 0.20.0-beta.1	2025-06-03 16:27:53 +00:00
Lance Release	8825c7c1dd	Bump version: 0.23.0-beta.0 → 0.23.0-beta.1	2025-06-03 16:26:58 +00:00
David Myriel	81c85ff702	docs: announcement for Documentation (#2410 ) Just letting people know where to look starting June 1st. Both docsites should be pointing to lancedb.github.io/documentation. <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Documentation - Added a notification banner to the documentation site informing users about a new URL for accessing the latest documentation starting June 1st, 2025. The message includes a clickable link that opens in a new tab. <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Co-authored-by: Will Jones <willjones127@gmail.com> Co-authored-by: coderabbitai[bot] <136622811+coderabbitai[bot]@users.noreply.github.com>	2025-06-03 08:55:02 -07:00
Will Jones	570f2154d5	ci: automatically update Cargo.lock (#2416 ) <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Chores - Updated workflow to ignore changes in the `Cargo.lock` file during documentation checks, reducing unnecessary workflow failures. - Enhanced release process by adding automated lockfile updates for Node.js and Rust components. - Removed an obsolete package-lock update job from the publishing workflow to streamline releases. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-06-03 07:49:21 -07:00
Lance Release	0525c055fc	Updating package-lock.json	2025-05-31 04:29:20 +00:00
Lance Release	38d11291da	Updating package-lock.json	2025-05-31 03:48:11 +00:00
Lance Release	258e682574	Updating package-lock.json	2025-05-31 03:47:55 +00:00
Lance Release	d7afa600b8	Bump version: 0.19.2-beta.0 → 0.20.0-beta.0	2025-05-31 03:47:37 +00:00
Lance Release	5c7303ab2e	Bump version: 0.22.2-beta.0 → 0.23.0-beta.0	2025-05-31 03:47:13 +00:00
Will Jones	5895ef4039	ci: revert unnecessary version bump (#2415 ) <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Chores - Downgraded version numbers for the Node.js, Python, and Rust packages. No other user-facing changes were made. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-30 16:51:14 -07:00
Jack Ye	0528cd858a	fix: avoid failing list_indices for any unknown index (#2413 ) Closes #2412 <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Bug Fixes - Improved the reliability of listing indices by logging warnings for errors and skipping problematic entries, ensuring successful results are returned. - Internal indices used for optimization are now excluded from the visible list of indices. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-30 14:43:12 -07:00
Jack Ye	6582f43422	feat: upgrade lance to v0.29.0-beta.1 (#2414 ) <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Chores - Updated internal dependencies for improved stability and compatibility. No user-facing changes. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-30 13:47:41 -07:00
BubbleCal	5c7f63388d	feat!: upgrade lance to v0.28.0 (#2404 ) this introduces some breaking changes in terms of rust API of creating FTS index, and the default index params changed Signed-off-by: BubbleCal <bubble-cal@outlook.com> <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - New Features - Updated default settings for full-text search (FTS) index creation: stemming, stop word removal, and ASCII folding are now enabled by default, while token position storage is disabled by default. - Refactor - Simplified and streamlined the configuration and handling of FTS index parameters for improved maintainability and consistency across interfaces. - Enhanced serialization and request construction for FTS index parameters to reduce manual handling and improve code clarity. - Improved test coverage by explicitly enabling positional indexing in FTS tests to support phrase queries. - Chores - Upgraded all internal dependencies related to FTS indexing to the latest version for enhanced compatibility and performance. - Updated package versions for Node.js, Python, and Rust components to the latest beta releases. - Improved CI workflows by adding Rust toolchain setup with formatting and linting tools. <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Signed-off-by: BubbleCal <bubble-cal@outlook.com> Co-authored-by: Will Jones <willjones127@gmail.com>	2025-05-29 15:19:24 -07:00
Renato Marroquin	d0bc671cac	docs: add example for querying a lance table with SQL (#2389 ) Adds example for querying a dataset with SQL <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Documentation - Added new guides on querying LanceDB tables using SQL with DuckDB and Apache Datafusion. - Included detailed instructions for integrating LanceDB with Datafusion in Python. - Updated navigation to include Datafusion and SQL querying documentation. - Improved formatting in TypeScript and vectordb update examples for consistency. - Tests - Added a new test demonstrating SQL querying on Lance tables via DataFusion integration. <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Co-authored-by: Weston Pace <weston.pace@gmail.com>	2025-05-29 06:14:38 -07:00
David Myriel	d37e17593d	[doc] Add New Readme Page (#2405 ) Added a new readme for better navigation, updated language and more detail <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Documentation - Updated the README with a modernized header, improved structure, and clearer descriptions of features and architecture. - Expanded and reorganized key features and product offerings for better clarity. - Simplified installation instructions and added a table of SDK interfaces with documentation links. - Enhanced community and contributor sections with new visuals and links to social and support channels. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-27 17:45:17 +02:00
Lance Release	cb726d370e	Updating package-lock.json	2025-05-23 22:36:54 +00:00
Lance Release	23ee132546	Updating package-lock.json	2025-05-23 21:58:58 +00:00
Lance Release	7fa090d330	Updating package-lock.json	2025-05-23 21:58:43 +00:00
Lance Release	07bc1c5397	Bump version: 0.19.1 → 0.19.2-beta.0	2025-05-23 21:58:31 +00:00
Lance Release	d7a9dbb9fc	Bump version: 0.22.1 → 0.22.2-beta.0	2025-05-23 21:58:17 +00:00
Jack Ye	00487afc7d	feat: upgrade lance to v0.27.3-beta.2 (#2403 ) <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Chores - Updated internal dependencies for improved compatibility and stability. No changes to user-facing features. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-23 14:53:13 -07:00
BubbleCal	1902d65aad	docs: update the `num_partitions` recommendation (#2401 )	2025-05-23 23:45:37 +08:00
Lance Release	c4fbb65b8e	Updating package-lock.json	2025-05-22 07:06:03 +00:00
Lance Release	875ed7ae6f	Updating package-lock.json	2025-05-22 05:58:59 +00:00
Lance Release	95a46a57ba	Updating package-lock.json	2025-05-22 05:58:43 +00:00
Lance Release	51561e31a0	Bump version: 0.19.1-beta.6 → 0.19.1	2025-05-22 05:58:05 +00:00
Lance Release	7b19120578	Bump version: 0.19.1-beta.5 → 0.19.1-beta.6	2025-05-22 05:58:00 +00:00
Lance Release	745c34a6a9	Bump version: 0.22.1-beta.6 → 0.22.1	2025-05-22 05:57:20 +00:00
Lance Release	db8fa2454d	Bump version: 0.22.1-beta.5 → 0.22.1-beta.6	2025-05-22 05:57:20 +00:00
Lei Xu	a67a7b4b42	chore: use stable lance (#2398 ) <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Chores - Updated workspace dependencies to use a stable release version for improved consistency and reliability. No changes to application features or functionality. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-21 22:34:29 -07:00
Lei Xu	496846e532	chore: bump lance version (#2397 ) - Bump lance version and prepare a new release. - Bump rust toolchain to 1.86, because GHA ubuntu does not have 1.83 `cargo-fmt` anymore	2025-05-21 14:15:55 -07:00
Ayush Chaurasia	dadcfebf8e	docs: add logos in genkit docs page (#2391 ) <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Documentation - Added an integration banner image to the beginning of the Genkitx-LanceDB documentation. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-20 01:40:12 +05:30
Lance Release	67033dbd7f	Updating package-lock.json	2025-05-16 00:25:41 +00:00
Lance Release	05a85cfc2a	Updating package-lock.json	2025-05-15 23:44:27 +00:00
Lance Release	40c5d3d72b	Updating package-lock.json	2025-05-15 23:44:10 +00:00
Lance Release	198f0f80c6	Bump version: 0.19.1-beta.4 → 0.19.1-beta.5	2025-05-15 23:43:32 +00:00
Lance Release	e3f2fd3892	Bump version: 0.22.1-beta.4 → 0.22.1-beta.5	2025-05-15 23:42:46 +00:00
Wyatt Alt	f401ccc599	chore: update lance to 0.27.1-beta.1 (#2388 ) This is for fe14671f1 <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Chores - Updated internal dependencies to newer versions for improved stability and performance. No changes to features or functionality. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-15 16:09:01 -07:00
Ayush Chaurasia	81b59139f8	docs: add genkit integration docs (#2383 ) <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Documentation - Added a comprehensive guide for integrating LanceDB with Genkit, including installation instructions, setup, indexing, retrieval, and building a custom RAG pipeline with example code and screenshots. - Updated the documentation navigation to include the new Genkit integration, making it accessible from the site menu. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-12 18:18:07 +05:30
ayush chaurasia	1026781ab6	Revert "update" This reverts commit `9c699b8cd9`.	2025-05-11 21:04:59 +05:30
ayush chaurasia	9c699b8cd9	update	2025-05-11 21:01:53 +05:30
Lance Release	34bec59bc3	Updating package-lock.json	2025-05-08 21:34:37 +00:00
Lance Release	a5fbbf0d66	Updating package-lock.json	2025-05-08 20:20:18 +00:00
Lance Release	b42721167b	Updating package-lock.json	2025-05-08 20:20:00 +00:00
Lance Release	543dec9ff0	Bump version: 0.19.1-beta.3 → 0.19.1-beta.4	2025-05-08 20:19:17 +00:00
Lance Release	04f962f6b0	Bump version: 0.22.1-beta.3 → 0.22.1-beta.4	2025-05-08 20:18:40 +00:00
LuQQiu	19e896ff69	chore: add default for result structs (#2377 ) add default for result structs, when values are not provided, will go with the default values <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Chores - Improved internal handling of table operation results to support default values. No changes to user-facing features or functionality. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-08 13:09:11 -07:00
Will Jones	272e4103b2	feat: provide timeout parameter for merge_insert (#2378 ) Provides the ability to set a timeout for merge insert. The default underlying timeout is however long the first attempt takes, or if there are multiple attempts, 30 seconds. This has two use cases: 1. Make the timeout shorter, when you want to fail if it takes too long. 2. Allow taking more time to do retries. <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - New Features - Added support for specifying a timeout when performing merge insert operations in Python, Node.js, and Rust APIs. - Introduced a new option to control the maximum allowed execution time for merge inserts, including retry timeout handling. - Documentation - Updated and added documentation to describe the new timeout option and its usage in APIs. - Tests - Added and updated tests to verify correct timeout behavior during merge insert operations. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-08 13:07:05 -07:00
Wyatt Alt	75c257ebb6	fix: return IndexNotExist on remote drop index 404 (#2380 ) Prior to this commit, attempting to drop an index that did not exist would return a TableNotFound error stating that the target table does not exist -- even when it did exist. Instead, we now return an IndexNotFound error. <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Bug Fixes - Improved error handling when attempting to drop a non-existent index, providing a more accurate error message. - Tests - Added a test to verify correct error reporting when dropping an index that does not exist. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-07 17:24:05 -07:00
Wyatt Alt	9ee152eb42	fix: support __len__ on remote table (#2379 ) This moves the __len__ method from LanceTable and RemoteTable to Table so that child classes don't need to implement their own. In the process, it fixes the implementation of RemoteTable's length method, which was previously missing a return statement. <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Refactor - Centralized the table length functionality in the base table class, simplifying subclass behavior. - Removed redundant or non-functional length methods from specific table classes. - Tests - Added a new test to verify correct table length reporting for remote tables. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-07 17:23:39 -07:00
LuQQiu	c9ae1b1737	fix: add restore with tag in python and nodejs API (#2374 ) add restore with tag API in python and nodejs API and add tests to guard them <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - New Features - The restore functionality now supports using version tags in addition to numeric version identifiers, allowing you to revert tables to a state marked by a tag. - Bug Fixes - Restoring with an unknown tag now properly raises an error. - Documentation - Updated documentation and examples to clarify that restore accepts both version numbers and tags. - Tests - Added new tests to verify restore behavior with version tags and error handling for unknown tags. - Added tests for checkout and restore operations involving tags. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-06 16:12:58 -07:00
Lance Release	89dc80c42a	Updating package-lock.json	2025-05-06 03:53:49 +00:00
Wyatt Alt	7b020ac799	chore: run cargo update (#2376 )	2025-05-05 20:26:42 -07:00
Lance Release	529e774bbb	Updating package-lock.json	2025-05-06 02:45:45 +00:00
Lance Release	7c12239305	Updating package-lock.json	2025-05-06 02:45:29 +00:00
Lance Release	d83424d6b4	Bump version: 0.19.1-beta.2 → 0.19.1-beta.3	2025-05-06 02:45:06 +00:00
Lance Release	8bf89f887c	Bump version: 0.22.1-beta.2 → 0.22.1-beta.3	2025-05-06 02:44:39 +00:00
LuQQiu	b2160b2304	fix: fix backward compatibility with the add API (#2375 ) add API originally returns a struct with request_id, add backward compatibility for that <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - Bug Fixes - Improved handling of empty server responses for various data operations to ensure consistent behavior across server versions. - Added default values to version and numeric fields to prevent errors when response data is incomplete. - Tests - Expanded tests to cover multiple server response scenarios, validating correct version handling in data operations. <!-- end of auto-generated comment: release notes by coderabbit.ai -->	2025-05-05 19:26:27 -07:00
Lance Release	1bb82597be	Updating package-lock.json	2025-05-06 01:21:13 +00:00
Lance Release	e4eee38b3c	Updating package-lock.json	2025-05-06 00:09:39 +00:00
Lance Release	64fc2be503	Updating package-lock.json	2025-05-06 00:09:19 +00:00
Lance Release	dc8054e90d	Bump version: 0.19.1-beta.1 → 0.19.1-beta.2	2025-05-06 00:08:55 +00:00