ruff

update benchmark script
2025-12-24 22:09:58 +00:00 · 2024-04-16 12:11:57 +05:30 · 2024-04-16 12:09:18 +05:30 · 2024-04-16 11:13:52 +05:30 · 2024-04-16 10:00:47 +05:30 · 2024-04-16 09:51:08 +05:30
209 changed files with 3938 additions and 17270 deletions
--- a/.bumpversion.cfg
+++ b/.bumpversion.cfg
@@ -0,0 +1,22 @@
+[bumpversion]
+current_version = 0.4.17
+commit = True
+message = Bump version: {current_version} → {new_version}
+tag = True
+tag_name = v{new_version}
+
+[bumpversion:file:node/package.json]
+
+[bumpversion:file:nodejs/package.json]
+
+[bumpversion:file:nodejs/npm/darwin-x64/package.json]
+
+[bumpversion:file:nodejs/npm/darwin-arm64/package.json]
+
+[bumpversion:file:nodejs/npm/linux-x64-gnu/package.json]
+
+[bumpversion:file:nodejs/npm/linux-arm64-gnu/package.json]
+
+[bumpversion:file:rust/ffi/node/Cargo.toml]
+
+[bumpversion:file:rust/lancedb/Cargo.toml]
--- a/.bumpversion.toml
+++ b/.bumpversion.toml
@@ -1,57 +0,0 @@
-[tool.bumpversion]
-current_version = "0.6.0"
-parse = """(?x)
-    (?P<major>0|[1-9]\\d*)\\.
-    (?P<minor>0|[1-9]\\d*)\\.
-    (?P<patch>0|[1-9]\\d*)
-    (?:-(?P<pre_l>[a-zA-Z-]+)\\.(?P<pre_n>0|[1-9]\\d*))?
-"""
-serialize = [
-    "{major}.{minor}.{patch}-{pre_l}.{pre_n}",
-    "{major}.{minor}.{patch}",
-]
-search = "{current_version}"
-replace = "{new_version}"
-regex = false
-ignore_missing_version = false
-ignore_missing_files = false
-tag = true
-sign_tags = false
-tag_name = "v{new_version}"
-tag_message = "Bump version: {current_version} → {new_version}"
-allow_dirty = true
-commit = true
-message = "Bump version: {current_version} → {new_version}"
-commit_args = ""
-
-[tool.bumpversion.parts.pre_l]
-values = ["beta", "final"]
-optional_value = "final"
-
-[[tool.bumpversion.files]]
-filename = "node/package.json"
-search = "\"version\": \"{current_version}\","
-replace = "\"version\": \"{new_version}\","
-
-[[tool.bumpversion.files]]
-filename = "nodejs/package.json"
-search = "\"version\": \"{current_version}\","
-replace = "\"version\": \"{new_version}\","
-
-# nodejs binary packages
-[[tool.bumpversion.files]]
-glob = "nodejs/npm/*/package.json"
-search = "\"version\": \"{current_version}\","
-replace = "\"version\": \"{new_version}\","
-
-# Cargo files
-# ------------
-[[tool.bumpversion.files]]
-filename = "rust/ffi/node/Cargo.toml"
-search = "\nversion = \"{current_version}\""
-replace = "\nversion = \"{new_version}\""
-
-[[tool.bumpversion.files]]
-filename = "rust/lancedb/Cargo.toml"
-search = "\nversion = \"{current_version}\""
-replace = "\nversion = \"{new_version}\""
--- a/.github/labeler.yml
+++ b/.github/labeler.yml
@@ -1,33 +0,0 @@
-version: 1
-appendOnly: true
-# Labels are applied based on conventional commits standard
-# https://www.conventionalcommits.org/en/v1.0.0/
-# These labels are later used in release notes. See .github/release.yml
-labels:
-# If the PR title has an ! before the : it will be considered a breaking change
-# For example, `feat!: add new feature` will be considered a breaking change
- label: breaking-change
-  title: "^[^:]+!:.*"
- label: breaking-change
-  body: "BREAKING CHANGE"
- label: enhancement
-  title: "^feat(\\(.+\\))?!?:.*"
- label: bug
-  title: "^fix(\\(.+\\))?!?:.*"
- label: documentation
-  title: "^docs(\\(.+\\))?!?:.*"
- label: performance
-  title: "^perf(\\(.+\\))?!?:.*"
- label: ci
-  title: "^ci(\\(.+\\))?!?:.*"
- label: chore
-  title: "^(chore|test|build|style)(\\(.+\\))?!?:.*"
- label: Python
-  files:
-    - "^python\\/.*"
- label: Rust
-  files:
-    - "^rust\\/.*"
- label: typescript
-  files:
-    - "^node\\/.*"
--- a/.github/release_notes.json
+++ b/.github/release_notes.json
@@ -1,41 +0,0 @@
-{
-    "ignore_labels": ["chore"],
-    "pr_template": "- ${{TITLE}} by @${{AUTHOR}} in ${{URL}}",
-    "categories": [
-        {
-            "title": "## 🏆 Highlights",
-            "labels": ["highlight"]
-        },
-        {
-            "title": "## 🛠 Breaking Changes",
-            "labels": ["breaking-change"]
-        },
-        {
-            "title": "## ⚠️ Deprecations ",
-            "labels": ["deprecation"]
-        },
-        {
-            "title": "## 🎉 New Features",
-            "labels": ["enhancement"]
-        },
-        {
-            "title": "## 🐛 Bug Fixes",
-            "labels": ["bug"]
-        },
-        {
-            "title": "## 📚 Documentation",
-            "labels": ["documentation"]
-        },
-        {
-            "title": "## 🚀 Performance Improvements",
-            "labels": ["performance"]
-        },
-        {
-            "title": "## Other Changes"
-        },
-        {
-            "title": "## 🔧 Build and CI",
-            "labels": ["ci"]
-        }
-    ]
-}
--- a/.github/workflows/build_linux_wheel/action.yml
+++ b/.github/workflows/build_linux_wheel/action.yml
@@ -46,7 +46,6 @@ runs:
      with:
        command: build
        working-directory: python
-        docker-options: "-e PIP_EXTRA_INDEX_URL=https://pypi.fury.io/lancedb/"
        target: aarch64-unknown-linux-gnu
        manylinux: ${{ inputs.manylinux }}
        args: ${{ inputs.args }}
--- a/.github/workflows/build_mac_wheel/action.yml
+++ b/.github/workflows/build_mac_wheel/action.yml
@@ -21,6 +21,5 @@ runs:
      with:
        command: build
        args: ${{ inputs.args }}
-        docker-options: "-e PIP_EXTRA_INDEX_URL=https://pypi.fury.io/lancedb/"
        working-directory: python
        interpreter: 3.${{ inputs.python-minor-version }}
--- a/.github/workflows/build_windows_wheel/action.yml
+++ b/.github/workflows/build_windows_wheel/action.yml
@@ -26,7 +26,6 @@ runs:
      with:
        command: build
        args: ${{ inputs.args }}
-        docker-options: "-e PIP_EXTRA_INDEX_URL=https://pypi.fury.io/lancedb/"
        working-directory: python
    - uses: actions/upload-artifact@v3
      with:
--- a/.github/workflows/cargo-publish.yml
+++ b/.github/workflows/cargo-publish.yml
@@ -1,12 +1,8 @@
 name: Cargo Publish

 on:
-  push:
-    tags-ignore:
-      # We don't publish pre-releases for Rust. Crates.io is just a source
-      # distribution, so we don't need to publish pre-releases.
-      - 'v*-beta*'
-      - '*-v*' # for example, python-vX.Y.Z
+  release:
+    types: [ published ]

 env:
  # This env var is used by Swatinem/rust-cache@v2 for the cache
--- a/.github/workflows/dev.yml
+++ b/.github/workflows/dev.yml
@@ -1,81 +0,0 @@
-name: PR Checks
-
-on:
-  pull_request_target:
-    types: [opened, edited, synchronize, reopened]
-
-concurrency:
-  group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
-  cancel-in-progress: true
-
-jobs:
-  labeler:
-    permissions:
-      pull-requests: write
-    name: Label PR
-    runs-on: ubuntu-latest
-    steps:
-      - uses: srvaroa/labeler@master
-        env:
-          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
-  commitlint:
-    permissions:
-      pull-requests: write
-    name: Verify PR title / description conforms to semantic-release
-    runs-on: ubuntu-latest
-    steps:
-      - uses: actions/setup-node@v3
-        with:
-          node-version: "18"
-      # These rules are disabled because Github will always ensure there
-      # is a blank line between the title and the body and Github will
-      # word wrap the description field to ensure a reasonable max line
-      # length.
-      - run: npm install @commitlint/config-conventional
-      - run: >
-          echo 'module.exports = {
-            "rules": {
-              "body-max-line-length": [0, "always", Infinity],
-              "footer-max-line-length": [0, "always", Infinity],
-              "body-leading-blank": [0, "always"]
-            }
-          }' > .commitlintrc.js
-      - run: npx commitlint --extends @commitlint/config-conventional --verbose <<< $COMMIT_MSG
-        env:
-          COMMIT_MSG: >
-            ${{ github.event.pull_request.title }}
-
-            ${{ github.event.pull_request.body }}
-      - if: failure()
-        uses: actions/github-script@v6
-        with:
-          script: |
-            const message = `**ACTION NEEDED**
-              
-              Lance follows the [Conventional Commits specification](https://www.conventionalcommits.org/en/v1.0.0/) for release automation.
-
-              The PR title and description are used as the merge commit message.\
-              Please update your PR title and description to match the specification.
-
-              For details on the error please inspect the "PR Title Check" action.
-              `
-            // Get list of current comments
-            const comments = await github.paginate(github.rest.issues.listComments, {
-              owner: context.repo.owner,
-              repo: context.repo.repo,
-              issue_number: context.issue.number
-            });
-            // Check if this job already commented
-            for (const comment of comments) {
-              if (comment.body === message) {
-                return // Already commented
-              }
-            }
-            // Post the comment about Conventional Commits
-            github.rest.issues.createComment({
-              owner: context.repo.owner,
-              repo: context.repo.repo,
-              issue_number: context.issue.number,
-              body: message
-            })
-            core.setFailed(message)
--- a/.github/workflows/docs_test.yml
+++ b/.github/workflows/docs_test.yml
@@ -24,7 +24,7 @@ env:
 jobs:
  test-python:
    name: Test doc python code
-    runs-on: "warp-ubuntu-latest-x64-4x"
+    runs-on: "buildjet-8vcpu-ubuntu-2204"
    steps:
    - name: Checkout
      uses: actions/checkout@v4
@@ -56,7 +56,7 @@ jobs:
        for d in *; do cd "$d"; echo "$d".py; python "$d".py; cd ..; done
  test-node:
    name: Test doc nodejs code
-    runs-on: "warp-ubuntu-latest-x64-4x"
+    runs-on: "buildjet-8vcpu-ubuntu-2204"
    timeout-minutes: 60
    strategy:
      fail-fast: false
--- a/.github/workflows/java.yml
+++ b/.github/workflows/java.yml
@@ -1,85 +0,0 @@
-name: Build and Run Java JNI Tests
-on:
-  push:
-    branches:
-      - main
-  pull_request:
-    paths:
-      - java/**
-      - rust/**
-      - .github/workflows/java.yml
-env:
-  # This env var is used by Swatinem/rust-cache@v2 for the cache
-  # key, so we set it to make sure it is always consistent.
-  CARGO_TERM_COLOR: always
-  # Disable full debug symbol generation to speed up CI build and keep memory down
-  # "1" means line tables only, which is useful for panic tracebacks.
-  RUSTFLAGS: "-C debuginfo=1"
-  RUST_BACKTRACE: "1"
-  # according to: https://matklad.github.io/2021/09/04/fast-rust-builds.html
-  # CI builds are faster with incremental disabled.
-  CARGO_INCREMENTAL: "0"
-  CARGO_BUILD_JOBS: "1"
-jobs:
-  linux-build:
-    runs-on: ubuntu-22.04
-    name: ubuntu-22.04 + Java 11 & 17
-    defaults:
-      run:
-        working-directory: ./java
-    steps:
-      - name: Checkout repository
-        uses: actions/checkout@v4
-      - uses: Swatinem/rust-cache@v2
-        with:
-          workspaces: java/core/lancedb-jni
-      - name: Run cargo fmt
-        run: cargo fmt --check
-        working-directory: ./java/core/lancedb-jni
-      - name: Install dependencies
-        run: |
-          sudo apt update
-          sudo apt install -y protobuf-compiler libssl-dev
-      - name: Install Java 17
-        uses: actions/setup-java@v4
-        with:
-          distribution: temurin
-          java-version: 17
-          cache: "maven"
-      - run: echo "JAVA_17=$JAVA_HOME" >> $GITHUB_ENV
-      - name: Install Java 11
-        uses: actions/setup-java@v4
-        with:
-          distribution: temurin
-          java-version: 11
-          cache: "maven"
-      - name: Java Style Check
-        run: mvn checkstyle:check
-      # Disable because of issues in lancedb rust core code
-      # - name: Rust Clippy
-      #   working-directory: java/core/lancedb-jni
-      #   run: cargo clippy --all-targets -- -D warnings
-      - name: Running tests with Java 11
-        run: mvn clean test
-      - name: Running tests with Java 17
-        run: |
-          export JAVA_TOOL_OPTIONS="$JAVA_TOOL_OPTIONS \
-          -XX:+IgnoreUnrecognizedVMOptions \
-          --add-opens=java.base/java.lang=ALL-UNNAMED \
-          --add-opens=java.base/java.lang.invoke=ALL-UNNAMED \
-          --add-opens=java.base/java.lang.reflect=ALL-UNNAMED \
-          --add-opens=java.base/java.io=ALL-UNNAMED \
-          --add-opens=java.base/java.net=ALL-UNNAMED \
-          --add-opens=java.base/java.nio=ALL-UNNAMED \
-          --add-opens=java.base/java.util=ALL-UNNAMED \
-          --add-opens=java.base/java.util.concurrent=ALL-UNNAMED \
-          --add-opens=java.base/java.util.concurrent.atomic=ALL-UNNAMED \
-          --add-opens=java.base/jdk.internal.ref=ALL-UNNAMED \
-          --add-opens=java.base/sun.nio.ch=ALL-UNNAMED \
-          --add-opens=java.base/sun.nio.cs=ALL-UNNAMED \
-          --add-opens=java.base/sun.security.action=ALL-UNNAMED \
-          --add-opens=java.base/sun.util.calendar=ALL-UNNAMED \
-          --add-opens=java.security.jgss/sun.security.krb5=ALL-UNNAMED \
-          -Djdk.reflect.useDirectMethodHandle=false \
-          -Dio.netty.tryReflectionSetAccessible=true"
-          JAVA_HOME=$JAVA_17 mvn clean test
--- a/.github/workflows/make-release-commit.yml
+++ b/.github/workflows/make-release-commit.yml
@@ -1,62 +1,37 @@
 name: Create release commit

-# This workflow increments versions, tags the version, and pushes it.
-# When a tag is pushed, another workflow is triggered that creates a GH release
-# and uploads the binaries. This workflow is only for creating the tag.
-
-# This script will enforce that a minor version is incremented if there are any
-# breaking changes since the last minor increment. However, it isn't able to
-# differentiate between breaking changes in Node versus Python. If you wish to
-# bypass this check, you can manually increment the version and push the tag.
 on:
  workflow_dispatch:
    inputs:
      dry_run:
        description: 'Dry run (create the local commit/tags but do not push it)'
        required: true
-        default: false
-        type: boolean
-      type:
-        description: 'What kind of release is this?'
-        required: true
-        default: 'preview'
+        default: "false"
        type: choice
        options:
-          - preview
-          - stable
-      python:
-        description: 'Make a Python release'
+          - "true"
+          - "false"
+      part:
+        description: 'What kind of release is this?'
        required: true
-        default: true
-        type: boolean
-      other:
-        description: 'Make a Node/Rust release'
-        required: true
-        default: true
-        type: boolean
-      bump-minor:
-        description: 'Bump minor version'
-        required: true
-        default: false
-        type: boolean
+        default: 'patch'
+        type: choice
+        options:
+          - patch
+          - minor
+          - major

 jobs:
-  make-release:
-    # Creates tag and GH release. The GH release will trigger the build and release jobs.
+  bump-version:
    runs-on: ubuntu-latest
-    permissions:
-      contents: write
    steps:
-      - name: Output Inputs
-        run: echo "${{ toJSON(github.event.inputs) }}"
-      - uses: actions/checkout@v4
+      - name: Check out main
+        uses: actions/checkout@v4
        with:
+          ref: main
+          persist-credentials: false
          fetch-depth: 0
          lfs: true
-          # It's important we use our token here, as the default token will NOT
-          # trigger any workflows watching for new tags. See:
-          # https://docs.github.com/en/actions/using-workflows/triggering-a-workflow#triggering-a-workflow-from-a-workflow
-          token: ${{ secrets.LANCEDB_RELEASE_TOKEN }}
      - name: Set git configs for bumpversion
        shell: bash
        run: |
@@ -66,34 +41,19 @@ jobs:
        uses: actions/setup-python@v5
        with:
          python-version: "3.11"
-      - name: Bump Python version
-        if: ${{ inputs.python }}
-        working-directory: python
-        env:
-          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+      - name: Bump version, create tag and commit
        run: |
-          # Need to get the commit before bumping the version, so we can
-          # determine if there are breaking changes in the next step as well.
-          echo "COMMIT_BEFORE_BUMP=$(git rev-parse HEAD)" >> $GITHUB_ENV
-
-          pip install bump-my-version PyGithub packaging
-          bash ../ci/bump_version.sh ${{ inputs.type }} ${{ inputs.bump-minor }} python-v
-      - name: Bump Node/Rust version
-        if: ${{ inputs.other }}
-        env:
-          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
-        run: |
-          pip install bump-my-version PyGithub packaging
-          bash ci/bump_version.sh ${{ inputs.type }} ${{ inputs.bump-minor }} v $COMMIT_BEFORE_BUMP
-      - name: Push new version tag
-        if: ${{ !inputs.dry_run }}
+          pip install bump2version
+          bumpversion --verbose ${{ inputs.part }}
+      - name: Push new version and tag
+        if: ${{ inputs.dry_run }} == "false"
        uses: ad-m/github-push-action@master
        with:
-          # Need to use PAT here too to trigger next workflow. See comment above.
          github_token: ${{ secrets.LANCEDB_RELEASE_TOKEN }}
-          branch: ${{ github.ref }}
+          branch: main
          tags: true
      - uses: ./.github/workflows/update_package_lock
-        if: ${{ !inputs.dry_run && inputs.other }}
+        if: ${{ inputs.dry_run }} == "false"
        with:
-          github_token: ${{ secrets.GITHUB_TOKEN }}
+          github_token: ${{ secrets.LANCEDB_RELEASE_TOKEN }}
+
--- a/.github/workflows/nodejs.yml
+++ b/.github/workflows/nodejs.yml
@@ -52,7 +52,8 @@ jobs:
        cargo fmt --all -- --check
        cargo clippy --all --all-features -- -D warnings
        npm ci
-        npm run lint-ci
+        npm run lint
+        npm run chkformat
  linux:
    name: Linux (NodeJS ${{ matrix.node-version }})
    timeout-minutes: 30
--- a/.github/workflows/npm-publish.yml
+++ b/.github/workflows/npm-publish.yml
@@ -1,9 +1,8 @@
 name: NPM Publish

 on:
-  push:
-    tags:
-      - "v*"
+  release:
+    types: [published]

 jobs:
  node:
@@ -111,11 +110,12 @@ jobs:
            runner: ubuntu-latest
          - arch: aarch64
            # For successful fat LTO builds, we need a large runner to avoid OOM errors.
-            runner: warp-ubuntu-latest-arm64-4x
+            runner: buildjet-16vcpu-ubuntu-2204-arm
    steps:
      - name: Checkout
        uses: actions/checkout@v4
-      # To avoid OOM errors on ARM, we create a swap file.
+      # Buildjet aarch64 runners have only 1.5 GB RAM per core, vs 3.5 GB per core for
+      # x86_64 runners. To avoid OOM errors on ARM, we create a swap file.
      - name: Configure aarch64 build
        if: ${{ matrix.config.arch == 'aarch64' }}
        run: |
@@ -274,15 +274,9 @@ jobs:
        env:
          NODE_AUTH_TOKEN: ${{ secrets.LANCEDB_NPM_REGISTRY_TOKEN }}
        run: |
-          # Tag beta as "preview" instead of default "latest". See lancedb 
-          # npm publish step for more info.
-          if [[ $GITHUB_REF =~ refs/tags/v(.*)-beta.* ]]; then
-            PUBLISH_ARGS="--tag preview"
-          fi
-
          mv */*.tgz .
          for filename in *.tgz; do
-            npm publish $PUBLISH_ARGS $filename
+            npm publish $filename
          done

  release-nodejs:
@@ -322,23 +316,11 @@ jobs:
      - name: Publish to NPM
        env:
          NODE_AUTH_TOKEN: ${{ secrets.LANCEDB_NPM_REGISTRY_TOKEN }}
-        # By default, things are published to the latest tag. This is what is
-        # installed by default if the user does not specify a version. This is
-        # good for stable releases, but for pre-releases, we want to publish to
-        # the "preview" tag so they can install with `npm install lancedb@preview`.
-        # See: https://medium.com/@mbostock/prereleases-and-npm-e778fc5e2420
-        run: |
-          if [[ $GITHUB_REF =~ refs/tags/v(.*)-beta.* ]]; then
-            npm publish --access public --tag preview
-          else
-            npm publish --access public
-          fi
+        run: npm publish --access public

  update-package-lock:
    needs: [release]
    runs-on: ubuntu-latest
-    permissions:
-      contents: write
    steps:
      - name: Checkout
        uses: actions/checkout@v4
@@ -349,13 +331,11 @@ jobs:
          lfs: true
      - uses: ./.github/workflows/update_package_lock
        with:
-          github_token: ${{ secrets.GITHUB_TOKEN }}
+          github_token: ${{ secrets.LANCEDB_RELEASE_TOKEN }}

  update-package-lock-nodejs:
    needs: [release-nodejs]
    runs-on: ubuntu-latest
-    permissions:
-      contents: write
    steps:
      - name: Checkout
        uses: actions/checkout@v4
@@ -366,70 +346,4 @@ jobs:
          lfs: true
      - uses: ./.github/workflows/update_package_lock_nodejs
        with:
-          github_token: ${{ secrets.GITHUB_TOKEN }}
-
-  gh-release:
-    runs-on: ubuntu-latest
-    permissions:
-      contents: write
-    steps:
-      - uses: actions/checkout@v4
-        with:
-          fetch-depth: 0
-          lfs: true
-      - name: Extract version
-        id: extract_version
-        env:
-          GITHUB_REF: ${{ github.ref }}
-        run: |
-          set -e
-          echo "Extracting tag and version from $GITHUB_REF"
-          if [[ $GITHUB_REF =~ refs/tags/v(.*) ]]; then
-            VERSION=${BASH_REMATCH[1]}
-            TAG=v$VERSION
-            echo "tag=$TAG" >> $GITHUB_OUTPUT
-            echo "version=$VERSION" >> $GITHUB_OUTPUT
-          else
-            echo "Failed to extract version from $GITHUB_REF"
-            exit 1
-          fi
-          echo "Extracted version $VERSION from $GITHUB_REF"
-          if [[ $VERSION =~ beta ]]; then
-            echo "This is a beta release"
-
-            # Get last release (that is not this one)
-            FROM_TAG=$(git tag --sort='version:refname' \
-              | grep ^v \
-              | grep -vF "$TAG" \
-              | python ci/semver_sort.py v \
-              | tail -n 1)
-          else
-            echo "This is a stable release"
-            # Get last stable tag (ignore betas)
-            FROM_TAG=$(git tag --sort='version:refname' \
-              | grep ^v \
-              | grep -vF "$TAG" \
-              | grep -v beta \
-              | python ci/semver_sort.py v \
-              | tail -n 1)
-          fi
-          echo "Found from tag $FROM_TAG"
-          echo "from_tag=$FROM_TAG" >> $GITHUB_OUTPUT
-      - name: Create Release Notes
-        id: release_notes
-        uses: mikepenz/release-changelog-builder-action@v4
-        with:
-          configuration: .github/release_notes.json
-          toTag: ${{ steps.extract_version.outputs.tag }}
-          fromTag: ${{ steps.extract_version.outputs.from_tag }}
-        env:
-          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
-      - name: Create GH release
-        uses: softprops/action-gh-release@v2
-        with:
-          prerelease: ${{ contains('beta', github.ref) }}
-          tag_name: ${{ steps.extract_version.outputs.tag }}
-          token: ${{ secrets.GITHUB_TOKEN }}
-          generate_release_notes: false
-          name: Node/Rust LanceDB v${{ steps.extract_version.outputs.version }}
-          body: ${{ steps.release_notes.outputs.changelog }}
+          github_token: ${{ secrets.LANCEDB_RELEASE_TOKEN }}
--- a/.github/workflows/pypi-publish.yml
+++ b/.github/workflows/pypi-publish.yml
@@ -1,16 +1,18 @@
 name: PyPI Publish

 on:
-  push:
-    tags:
-      - 'python-v*'
+  release:
+    types: [published]

 jobs:
  linux:
+    # Only runs on tags that matches the python-make-release action
+    if: startsWith(github.ref, 'refs/tags/python-v')
    name: Python ${{ matrix.config.platform }} manylinux${{ matrix.config.manylinux }}
    timeout-minutes: 60
    strategy:
      matrix:
+        python-minor-version: ["8"]
        config:
          - platform: x86_64
            manylinux: "2_17"
@@ -32,22 +34,25 @@ jobs:
      - name: Set up Python
        uses: actions/setup-python@v4
        with:
-          python-version: 3.8
+          python-version: 3.${{ matrix.python-minor-version }}
      - uses: ./.github/workflows/build_linux_wheel
        with:
-          python-minor-version: 8
+          python-minor-version: ${{ matrix.python-minor-version }}
          args: "--release --strip ${{ matrix.config.extra_args }}"
          arm-build: ${{ matrix.config.platform == 'aarch64' }}
          manylinux: ${{ matrix.config.manylinux }}
      - uses: ./.github/workflows/upload_wheel
        with:
-          pypi_token: ${{ secrets.LANCEDB_PYPI_API_TOKEN }}
-          fury_token: ${{ secrets.FURY_TOKEN }}
+          token: ${{ secrets.LANCEDB_PYPI_API_TOKEN }}
+          repo: "pypi"
  mac:
+    # Only runs on tags that matches the python-make-release action
+    if: startsWith(github.ref, 'refs/tags/python-v')
    timeout-minutes: 60
    runs-on: ${{ matrix.config.runner }}
    strategy:
      matrix:
+        python-minor-version: ["8"]
        config:
          - target: x86_64-apple-darwin
            runner: macos-13
@@ -58,6 +63,7 @@ jobs:
    steps:
      - uses: actions/checkout@v4
        with:
+          ref: ${{ inputs.ref }}
          fetch-depth: 0
          lfs: true
      - name: Set up Python
@@ -66,95 +72,38 @@ jobs:
          python-version: 3.12
      - uses: ./.github/workflows/build_mac_wheel
        with:
-          python-minor-version: 8
+          python-minor-version: ${{ matrix.python-minor-version }}
          args: "--release --strip --target ${{ matrix.config.target }} --features fp16kernels"
      - uses: ./.github/workflows/upload_wheel
        with:
-          pypi_token: ${{ secrets.LANCEDB_PYPI_API_TOKEN }}
-          fury_token: ${{ secrets.FURY_TOKEN }}
+          python-minor-version: ${{ matrix.python-minor-version }}
+          token: ${{ secrets.LANCEDB_PYPI_API_TOKEN }}
+          repo: "pypi"
  windows:
+    # Only runs on tags that matches the python-make-release action
+    if: startsWith(github.ref, 'refs/tags/python-v')
    timeout-minutes: 60
    runs-on: windows-latest
+    strategy:
+      matrix:
+        python-minor-version: ["8"]
    steps:
      - uses: actions/checkout@v4
        with:
+          ref: ${{ inputs.ref }}
          fetch-depth: 0
          lfs: true
      - name: Set up Python
        uses: actions/setup-python@v4
        with:
-          python-version: 3.8
+          python-version: 3.${{ matrix.python-minor-version }}
      - uses: ./.github/workflows/build_windows_wheel
        with:
-          python-minor-version: 8
+          python-minor-version: ${{ matrix.python-minor-version }}
          args: "--release --strip"
          vcpkg_token: ${{ secrets.VCPKG_GITHUB_PACKAGES }}
      - uses: ./.github/workflows/upload_wheel
        with:
-          pypi_token: ${{ secrets.LANCEDB_PYPI_API_TOKEN }}
-          fury_token: ${{ secrets.FURY_TOKEN }}
-  gh-release:
-    runs-on: ubuntu-latest
-    permissions:
-      contents: write
-    steps:
-      - uses: actions/checkout@v4
-        with:
-          fetch-depth: 0
-          lfs: true
-      - name: Extract version
-        id: extract_version
-        env:
-          GITHUB_REF: ${{ github.ref }}
-        run: |
-          set -e
-          echo "Extracting tag and version from $GITHUB_REF"
-          if [[ $GITHUB_REF =~ refs/tags/python-v(.*) ]]; then
-            VERSION=${BASH_REMATCH[1]}
-            TAG=python-v$VERSION
-            echo "tag=$TAG" >> $GITHUB_OUTPUT
-            echo "version=$VERSION" >> $GITHUB_OUTPUT
-          else
-            echo "Failed to extract version from $GITHUB_REF"
-            exit 1
-          fi
-          echo "Extracted version $VERSION from $GITHUB_REF"
-          if [[ $VERSION =~ beta ]]; then
-            echo "This is a beta release"
-
-            # Get last release (that is not this one)
-            FROM_TAG=$(git tag --sort='version:refname' \
-              | grep ^python-v \
-              | grep -vF "$TAG" \
-              | python ci/semver_sort.py python-v \
-              | tail -n 1)
-          else
-            echo "This is a stable release"
-            # Get last stable tag (ignore betas)
-            FROM_TAG=$(git tag --sort='version:refname' \
-              | grep ^python-v \
-              | grep -vF "$TAG" \
-              | grep -v beta \
-              | python ci/semver_sort.py python-v \
-              | tail -n 1)
-          fi
-          echo "Found from tag $FROM_TAG"
-          echo "from_tag=$FROM_TAG" >> $GITHUB_OUTPUT
-      - name: Create Python Release Notes
-        id: python_release_notes
-        uses: mikepenz/release-changelog-builder-action@v4
-        with:
-          configuration: .github/release_notes.json
-          toTag: ${{ steps.extract_version.outputs.tag }}
-          fromTag: ${{ steps.extract_version.outputs.from_tag }}
-        env:
-          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
-      - name: Create Python GH release
-        uses: softprops/action-gh-release@v2
-        with:
-          prerelease: ${{ contains('beta', github.ref) }}
-          tag_name: ${{ steps.extract_version.outputs.tag }}
-          token: ${{ secrets.GITHUB_TOKEN }}
-          generate_release_notes: false
-          name: Python LanceDB v${{ steps.extract_version.outputs.version }}
-          body: ${{ steps.python_release_notes.outputs.changelog }}
+          python-minor-version: ${{ matrix.python-minor-version }}
+          token: ${{ secrets.LANCEDB_PYPI_API_TOKEN }}
+          repo: "pypi"
--- a/.github/workflows/python-make-release-commit.yml
+++ b/.github/workflows/python-make-release-commit.yml
@@ -0,0 +1,56 @@
+name: Python - Create release commit
+
+on:
+  workflow_dispatch:
+    inputs:
+      dry_run:
+        description: 'Dry run (create the local commit/tags but do not push it)'
+        required: true
+        default: "false"
+        type: choice
+        options:
+          - "true"
+          - "false"
+      part:
+        description: 'What kind of release is this?'
+        required: true
+        default: 'patch'
+        type: choice
+        options:
+          - patch
+          - minor
+          - major
+
+jobs:
+  bump-version:
+    runs-on: ubuntu-latest
+    steps:
+    - name: Check out main
+      uses: actions/checkout@v4
+      with:
+        ref: main
+        persist-credentials: false
+        fetch-depth: 0
+        lfs: true
+    - name: Set git configs for bumpversion
+      shell: bash
+      run: |
+        git config user.name 'Lance Release'
+        git config user.email 'lance-dev@lancedb.com'
+    - name: Set up Python
+      uses: actions/setup-python@v5
+      with:
+        python-version: "3.11"
+    - name: Bump version, create tag and commit
+      working-directory: python
+      run: |
+        pip install bump2version
+        bumpversion --verbose ${{ inputs.part }}
+    - name: Push new version and tag
+      if: ${{ inputs.dry_run }} == "false"
+      uses: ad-m/github-push-action@master
+      with:
+        github_token: ${{ secrets.LANCEDB_RELEASE_TOKEN }}
+        branch: main
+        tags: true
+
--- a/.github/workflows/python.yml
+++ b/.github/workflows/python.yml
@@ -65,7 +65,7 @@ jobs:
          workspaces: python
      - name: Install
        run: |
-          pip install --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests,dev,embeddings]
+          pip install -e .[tests,dev,embeddings]
          pip install tantivy
          pip install mlx
      - name: Doctest
@@ -75,7 +75,7 @@ jobs:
    timeout-minutes: 30
    strategy:
      matrix:
-        python-minor-version: ["9", "11"]
+        python-minor-version: ["8", "11"]
    runs-on: "ubuntu-22.04"
    defaults:
      run:
@@ -189,7 +189,7 @@ jobs:
      - name: Install lancedb
        run: |
          pip install "pydantic<2"
-          pip install --extra-index-url https://pypi.fury.io/lancedb/ -e .[tests]
+          pip install -e .[tests]
          pip install tantivy
      - name: Run tests
        run: pytest -m "not slow and not s3_test" -x -v --durations=30 python/tests
--- a/.github/workflows/run_tests/action.yml
+++ b/.github/workflows/run_tests/action.yml
@@ -15,7 +15,7 @@ runs:
    - name: Install lancedb
      shell: bash
      run: |
-        pip3 install --extra-index-url https://pypi.fury.io/lancedb/ $(ls target/wheels/lancedb-*.whl)[tests,dev]
+        pip3 install $(ls target/wheels/lancedb-*.whl)[tests,dev]
    - name: Setup localstack for integration tests
      if: ${{ inputs.integration == 'true' }}
      shell: bash
--- a/.github/workflows/rust.yml
+++ b/.github/workflows/rust.yml
@@ -74,11 +74,11 @@ jobs:
      run: |
          sudo apt update
          sudo apt install -y protobuf-compiler libssl-dev
+    - name: Build
+      run: cargo build --all-features
    - name: Start S3 integration test environment
      working-directory: .
      run: docker compose up --detach --wait
-    - name: Build
-      run: cargo build --all-features
    - name: Run tests
      run: cargo test --all-features
    - name: Run examples
--- a/.github/workflows/upload_wheel/action.yml
+++ b/.github/workflows/upload_wheel/action.yml
@@ -2,43 +2,28 @@ name: upload-wheel

 description: "Upload wheels to Pypi"
 inputs:
-  pypi_token:
+  os:
+    required: true
+    description: "ubuntu-22.04 or macos-13"
+  repo:
+    required: false
+    description: "pypi or testpypi"
+    default: "pypi"
+  token:
    required: true
    description: "release token for the repo"
-  fury_token:
-    required: true
-    description: "release token for the fury repo"

 runs:
  using: "composite"
  steps:
-  - name: Install dependencies
-    shell: bash
-    run: |
-      python -m pip install --upgrade pip
-      pip install twine
-  - name: Choose repo
-    shell: bash
-    id: choose_repo
-    run: |
-      if [ ${{ github.ref }} == "*beta*" ]; then
-        echo "repo=fury" >> $GITHUB_OUTPUT
-      else
-        echo "repo=pypi" >> $GITHUB_OUTPUT
-      fi
-  - name: Publish to PyPI
-    shell: bash
-    env:
-      FURY_TOKEN: ${{ inputs.fury_token }}
-      PYPI_TOKEN: ${{ inputs.pypi_token }}
-    run: |
-      if [ ${{ steps.choose_repo.outputs.repo }} == "fury" ]; then
-        WHEEL=$(ls target/wheels/lancedb-*.whl 2> /dev/null | head -n 1)
-        echo "Uploading $WHEEL to Fury"
-        curl -f -F package=@$WHEEL https://$FURY_TOKEN@push.fury.io/lancedb/
-      else
-        twine upload --repository ${{ steps.choose_repo.outputs.repo }} \
-          --username __token__ \
-          --password $PYPI_TOKEN \
-          target/wheels/lancedb-*.whl
-      fi
+    - name: Install dependencies
+      shell: bash
+      run: |
+        python -m pip install --upgrade pip
+        pip install twine
+    - name: Publish wheel
+      env:
+        TWINE_USERNAME: __token__
+        TWINE_PASSWORD: ${{ inputs.token }}
+      shell: bash
+      run: twine upload --repository ${{ inputs.repo }} target/wheels/lancedb-*.whl
--- a/.gitignore
+++ b/.gitignore
@@ -6,7 +6,7 @@
 venv

 .vscode
-.zed
+
 rust/target
 rust/Cargo.lock

--- a/.pre-commit-config.yaml
+++ b/.pre-commit-config.yaml
@@ -10,12 +10,9 @@ repos:
    rev: v0.2.2
    hooks:
    - id: ruff
- repo: local
+- repo: https://github.com/pre-commit/mirrors-prettier
+  rev: v3.1.0
  hooks:
-    - id: local-biome-check
-      name: biome check
-      entry: npx @biomejs/biome@1.8.3 check --config-path nodejs/biome.json nodejs/
-      language: system
-      types: [text]
+    - id: prettier
      files: "nodejs/.*"
      exclude: nodejs/lancedb/native.d.ts|nodejs/dist/.*
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -1,11 +1,5 @@
 [workspace]
-members = [
-    "rust/ffi/node",
-    "rust/lancedb",
-    "nodejs",
-    "python",
-    "java/core/lancedb-jni",
-]
+members = ["rust/ffi/node", "rust/lancedb", "nodejs", "python"]
 # Python package needs to be built by maturin.
 exclude = ["python"]
 resolver = "2"
@@ -20,31 +14,22 @@ keywords = ["lancedb", "lance", "database", "vector", "search"]
 categories = ["database-implementations"]

 [workspace.dependencies]
-# lance = { "version" = "=0.14.0", "features" = ["dynamodb"] }
-# lance-index = { "version" = "=0.14.0" }
-# lance-linalg = { "version" = "=0.14.0" }
-# lance-testing = { "version" = "=0.14.0" }
-# lance-datafusion = { "version" = "=0.14.0" }
-
-lance = { path = "../lance/rust/lance", "features" = ["dynamodb"] }
-lance-index = { path = "../lance/rust/lance-index" }
-lance-linalg = { path = "../lance/rust/lance-linalg" }
-lance-testing = { path = "../lance/rust/lance-testing" }
-lance-datafusion = { path = "../lance/rust/lance-datafusion" }
-
+lance = { "version" = "=0.10.12", "features" = ["dynamodb"] }
+lance-index = { "version" = "=0.10.12" }
+lance-linalg = { "version" = "=0.10.12" }
+lance-testing = { "version" = "=0.10.12" }
 # Note that this one does not include pyarrow
-arrow = { version = "51.0", optional = false }
-arrow-array = "51.0"
-arrow-data = "51.0"
-arrow-ipc = "51.0"
-arrow-ord = "51.0"
-arrow-schema = "51.0"
-arrow-arith = "51.0"
-arrow-cast = "51.0"
+arrow = { version = "50.0", optional = false }
+arrow-array = "50.0"
+arrow-data = "50.0"
+arrow-ipc = "50.0"
+arrow-ord = "50.0"
+arrow-schema = "50.0"
+arrow-arith = "50.0"
+arrow-cast = "50.0"
 async-trait = "0"
 chrono = "0.4.35"
-datafusion-physical-plan = "37.1"
-half = { "version" = "=2.4.1", default-features = false, features = [
+half = { "version" = "=2.3.1", default-features = false, features = [
    "num-traits",
 ] }
 futures = "0"
--- a/README.md
+++ b/README.md
@@ -20,7 +20,7 @@

 <hr />

-LanceDB is an open-source database for vector-search built with persistent storage, which greatly simplifies retrieval, filtering and management of embeddings.
+LanceDB is an open-source database for vector-search built with persistent storage, which greatly simplifies retrevial, filtering and management of embeddings.

 The key features of LanceDB include:

@@ -36,7 +36,7 @@ The key features of LanceDB include:

 * GPU support in building vector index(*).

-* Ecosystem integrations with [LangChain 🦜️🔗](https://python.langchain.com/docs/integrations/vectorstores/lancedb/), [LlamaIndex 🦙](https://gpt-index.readthedocs.io/en/latest/examples/vector_stores/LanceDBIndexDemo.html), Apache-Arrow, Pandas, Polars, DuckDB and more on the way.
+* Ecosystem integrations with [LangChain 🦜️🔗](https://python.langchain.com/en/latest/modules/indexes/vectorstores/examples/lanecdb.html), [LlamaIndex 🦙](https://gpt-index.readthedocs.io/en/latest/examples/vector_stores/LanceDBIndexDemo.html), Apache-Arrow, Pandas, Polars, DuckDB and more on the way.

 LanceDB's core is written in Rust 🦀 and is built using <a href="https://github.com/lancedb/lance">Lance</a>, an open-source columnar format designed for performant ML workloads.

@@ -83,5 +83,5 @@ result = table.search([100, 100]).limit(2).to_pandas()
 ```

 ## Blogs, Tutorials & Videos
-* 📈 <a href="https://blog.lancedb.com/benchmarking-random-access-in-lance/">2000x better performance with Lance over Parquet</a>
+* 📈 <a href="https://blog.eto.ai/benchmarking-random-access-in-lance-ed690757a826">2000x better performance with Lance over Parquet</a>
 * 🤖 <a href="https://github.com/lancedb/lancedb/blob/main/docs/src/notebooks/youtube_transcript_search.ipynb">Build a question and answer bot with LanceDB</a>
--- a/ci/bump_version.sh
+++ b/ci/bump_version.sh
@@ -1,51 +0,0 @@
-set -e
-
-RELEASE_TYPE=${1:-"stable"}
-BUMP_MINOR=${2:-false}
-TAG_PREFIX=${3:-"v"} # Such as "python-v"
-HEAD_SHA=${4:-$(git rev-parse HEAD)}
-
-readonly SELF_DIR=$(cd "$( dirname "${BASH_SOURCE[0]}" )" && pwd )
-
-PREV_TAG=$(git tag --sort='version:refname' | grep ^$TAG_PREFIX | python $SELF_DIR/semver_sort.py $TAG_PREFIX | tail -n 1)
-echo "Found previous tag $PREV_TAG"
-
-# Initially, we don't want to tag if we are doing stable, because we will bump
-# again later. See comment at end for why.
-if [[ "$RELEASE_TYPE" == 'stable' ]]; then 
-  BUMP_ARGS="--no-tag"
-fi
-
-# If last is stable and not bumping minor
-if [[ $PREV_TAG != *beta* ]]; then
-    if [[ "$BUMP_MINOR" != "false" ]]; then
-      # X.Y.Z -> X.(Y+1).0-beta.0
-      bump-my-version bump -vv $BUMP_ARGS minor
-    else
-      # X.Y.Z -> X.Y.(Z+1)-beta.0
-      bump-my-version bump -vv $BUMP_ARGS patch
-    fi
-else
-  if [[ "$BUMP_MINOR" != "false" ]]; then
-    # X.Y.Z-beta.N -> X.(Y+1).0-beta.0
-    bump-my-version bump -vv $BUMP_ARGS minor
-  else
-    # X.Y.Z-beta.N -> X.Y.Z-beta.(N+1)
-    bump-my-version bump -vv $BUMP_ARGS pre_n
-  fi
-fi
-
-# The above bump will always bump to a pre-release version. If we are releasing
-# a stable version, bump the pre-release level ("pre_l") to make it stable.
-if [[ $RELEASE_TYPE == 'stable' ]]; then
-  # X.Y.Z-beta.N -> X.Y.Z
-  bump-my-version bump -vv pre_l
-fi
-
-# Validate that we have incremented version appropriately for breaking changes
-NEW_TAG=$(git describe --tags --exact-match HEAD)
-NEW_VERSION=$(echo $NEW_TAG | sed "s/^$TAG_PREFIX//")
-LAST_STABLE_RELEASE=$(git tag --sort='version:refname' | grep ^$TAG_PREFIX | grep -v beta | grep -vF "$NEW_TAG" | python $SELF_DIR/semver_sort.py $TAG_PREFIX | tail -n 1)
-LAST_STABLE_VERSION=$(echo $LAST_STABLE_RELEASE | sed "s/^$TAG_PREFIX//")
-
-python $SELF_DIR/check_breaking_changes.py $LAST_STABLE_RELEASE $HEAD_SHA $LAST_STABLE_VERSION $NEW_VERSION
--- a/ci/check_breaking_changes.py
+++ b/ci/check_breaking_changes.py
@@ -1,35 +0,0 @@
-"""
-Check whether there are any breaking changes in the PRs between the base and head commits.
-If there are, assert that we have incremented the minor version.
-"""
-import argparse
-import os
-from packaging.version import parse
-
-from github import Github
-
-if __name__ == "__main__":
-    parser = argparse.ArgumentParser()
-    parser.add_argument("base")
-    parser.add_argument("head")
-    parser.add_argument("last_stable_version")
-    parser.add_argument("current_version")
-    args = parser.parse_args()
-
-    repo = Github(os.environ["GITHUB_TOKEN"]).get_repo(os.environ["GITHUB_REPOSITORY"])
-    commits = repo.compare(args.base, args.head).commits
-    prs = (pr for commit in commits for pr in commit.get_pulls())
-
-    for pr in prs:
-        if any(label.name == "breaking-change" for label in pr.labels):
-            print(f"Breaking change in PR: {pr.html_url}")
-            break
-    else:
-        print("No breaking changes found.")
-        exit(0)
-    
-    last_stable_version = parse(args.last_stable_version)
-    current_version = parse(args.current_version)
-    if current_version.minor <= last_stable_version.minor:
-        print("Minor version is not greater than the last stable version.")
-        exit(1)
--- a/ci/semver_sort.py
+++ b/ci/semver_sort.py
@@ -1,35 +0,0 @@
-"""
-Takes a list of semver strings and sorts them in ascending order.
-"""
-
-import sys
-from packaging.version import parse, InvalidVersion
-
-if __name__ == "__main__":
-    import argparse
-    parser = argparse.ArgumentParser()
-    parser.add_argument("prefix", default="v")
-    args = parser.parse_args()
-
-    # Read the input from stdin
-    lines = sys.stdin.readlines()
-
-    # Parse the versions
-    versions = []
-    for line in lines:
-        line = line.strip()
-        try:
-            version_str = line.removeprefix(args.prefix)
-            version = parse(version_str)
-        except InvalidVersion:
-            # There are old tags that don't follow the semver format
-            print(f"Invalid version: {line}", file=sys.stderr)
-            continue
-        versions.append((line, version))
-
-    # Sort the versions
-    versions.sort(key=lambda x: x[1])
-
-    # Print the sorted versions as original strings
-    for line, _ in versions:
-        print(line)
--- a/docs/benchmarks/llama-index-datasets.py
+++ b/docs/benchmarks/llama-index-datasets.py
@@ -0,0 +1,280 @@
+import argparse
+import os
+from llama_index.core import SimpleDirectoryReader
+from llama_index.core.llama_dataset import LabelledRagDataset
+from llama_index.core.node_parser import SentenceSplitter
+from lancedb.embeddings.fine_tuner.dataset import QADataset, TextChunk
+from lancedb.embeddings.fine_tuner.llm import Openai
+from lancedb.embeddings import get_registry
+from llama_index.vector_stores.lancedb import LanceDBVectorStore
+from llama_index.core import ServiceContext, VectorStoreIndex, StorageContext
+from llama_index.embeddings.huggingface import HuggingFaceEmbedding
+from llama_index.core.llama_pack import download_llama_pack
+
+import time
+import wandb
+
+
+import pandas as pd
+
+
+def get_paths_from_dataset(dataset: str, split=True):
+    """
+    Returns paths of:
+    - downloaded dataset, lance train dataset, lance test dataset, finetuned model
+    """
+    if split:
+        return (
+            f"./data/{dataset}",
+            f"./data/{dataset}_lance_train",
+            f"./data/{dataset}_lance_test",
+            f"./data/tuned_{dataset}",
+        )
+    return f"./data/{dataset}", f"./data/{dataset}_lance", f"./data/tuned_{dataset}"
+
+
+def get_llama_dataset(dataset: str):
+    """
+    returns:
+    - nodes, documents, rag_dataset
+    """
+    if not os.path.exists(f"./data/{dataset}"):
+        os.system(
+            f"llamaindex-cli download-llamadataset {dataset} --download-dir ./data/{dataset}"  # noqa
+        )
+    rag_dataset = LabelledRagDataset.from_json(f"./data/{dataset}/rag_dataset.json")
+    docs = SimpleDirectoryReader(input_dir=f"./data/{dataset}/source_files").load_data()
+
+    parser = SentenceSplitter()
+    nodes = parser.get_nodes_from_documents(docs)
+
+    return nodes, docs, rag_dataset
+
+
+def lance_dataset_from_llama_nodes(nodes: list, name: str, split=True):
+    llm = Openai()
+    chunks = [TextChunk.from_llama_index_node(node) for node in nodes]
+    # train test split 75-35
+    if not split:
+        if os.path.exists(f"./data/{name}_lance"):
+            ds = QADataset.load(f"./data/{name}_lance")
+            return ds
+        ds = QADataset.from_llm(chunks, llm)
+        ds.save(f"./data/{name}_lance")
+        return ds
+
+    if os.path.exists(f"./data/{name}_lance_train") and os.path.exists(
+        f"./data/{name}_lance_test"
+    ):
+        train_ds = QADataset.load(f"./data/{name}_lance_train")
+        test_ds = QADataset.load(f"./data/{name}_lance_test")
+        return train_ds, test_ds
+    # split chunks random
+    train_size = int(len(chunks) * 0.65)
+    train_chunks = chunks[:train_size]
+    test_chunks = chunks[train_size:]
+    train_ds = QADataset.from_llm(train_chunks, llm)
+    test_ds = QADataset.from_llm(test_chunks, llm)
+    train_ds.save(f"./data/{name}_lance_train")
+    test_ds.save(f"./data/{name}_lance_test")
+    return train_ds, test_ds
+
+
+def finetune(
+    trainset: str, model: str, epochs: int, path: str, valset: str = None, top_k=5
+):
+    print(f"Finetuning {model} for {epochs} epochs")
+    print(f"trainset query instances: {len(trainset.queries)}")
+    print(f"valset query instances: {len(valset.queries)}")
+
+    valset = valset if valset is not None else trainset
+    model = get_registry().get("sentence-transformers").create(name=model)
+    base_result = model.evaluate(valset, path="./data/eval/", top_k=top_k)
+    base_hit_rate = pd.DataFrame(base_result)["is_hit"].mean()
+
+    model.finetune(trainset=trainset, valset=valset, path=path, epochs=epochs)
+    tuned = get_registry().get("sentence-transformers").create(name=path)
+    tuned_result = tuned.evaluate(
+        valset, path=f"./data/eval/{str(time.time())}", top_k=top_k
+    )
+    tuned_hit_rate = pd.DataFrame(tuned_result)["is_hit"].mean()
+
+    return base_hit_rate, tuned_hit_rate
+
+
+def do_eval_rag(dataset: str, model: str):
+    # Requires - pip install llama-index-vector-stores-lancedb
+    # Requires - pip install llama-index-embeddings-huggingface
+    nodes, docs, rag_dataset = get_llama_dataset(dataset)
+
+    embed_model = HuggingFaceEmbedding(model)
+    vector_store = LanceDBVectorStore(uri="/tmp/lancedb")
+    storage_context = StorageContext.from_defaults(vector_store=vector_store)
+    service_context = ServiceContext.from_defaults(embed_model=embed_model)
+    index = VectorStoreIndex(
+        nodes,
+        service_context=service_context,
+        show_progress=True,
+        storage_context=storage_context,
+    )
+
+    # build basic RAG system
+    index = VectorStoreIndex.from_documents(documents=docs)
+    query_engine = index.as_query_engine()
+
+    # evaluate using the RagEvaluatorPack
+    RagEvaluatorPack = download_llama_pack("RagEvaluatorPack", "./rag_evaluator_pack")
+    rag_evaluator_pack = RagEvaluatorPack(
+        rag_dataset=rag_dataset, query_engine=query_engine
+    )
+
+    metrics = rag_evaluator_pack.run(
+        batch_size=20,  # batches the number of openai api calls to make
+        sleep_time_in_seconds=1,  # seconds to sleep before making an api call
+    )
+
+    return metrics
+
+
+def main(
+    dataset,
+    model,
+    epochs,
+    top_k=5,
+    eval_rag=False,
+    split=True,
+    project: str = "lancedb_finetune",
+):
+    nodes, _, _ = get_llama_dataset(dataset)
+    trainset = None
+    valset = None
+    if split:
+        trainset, valset = lance_dataset_from_llama_nodes(nodes, dataset, split)
+        data_path, lance_train_path, lance_test_path, tuned_path = (
+            get_paths_from_dataset(dataset, split=split)
+        )
+    else:
+        trainset = lance_dataset_from_llama_nodes(nodes, dataset, split)
+        valset = trainset
+        data_path, lance_path, tuned_path = get_paths_from_dataset(dataset, split=split)
+
+    base_hit_rate, tuned_hit_rate = finetune(
+        trainset, model, epochs, tuned_path, valset, top_k=top_k
+    )
+
+    # Base model model metrics
+    metrics = do_eval_rag(dataset, model) if eval_rag else {}
+
+    # Tuned model metrics
+    metrics_tuned = do_eval_rag(dataset, tuned_path) if eval_rag else {}
+
+    wandb.init(project="lancedb_finetune", name=f"{dataset}_{model}_{epochs}")
+    wandb.log(
+        {
+            "hit_rate": tuned_hit_rate,
+        }
+    )
+    wandb.log(metrics_tuned)
+    wandb.finish()
+
+    wandb.init(project="lancedb_finetune", name=f"{dataset}_{model}_base")
+    wandb.log(
+        {
+            "hit_rate": base_hit_rate,
+        }
+    )
+    wandb.log(metrics)
+    wandb.finish()
+
+
+def banchmark_all():
+    datasets = [
+        "Uber10KDataset2021",
+        "MiniTruthfulQADataset",
+        "MiniSquadV2Dataset",
+        "MiniEsgBenchDataset",
+        "MiniCovidQaDataset",
+        "Llama2PaperDataset",
+        "HistoryOfAlexnetDataset",
+        "PatronusAIFinanceBenchDataset",
+    ]
+    models = ["BAAI/bge-small-en-v1.5"]
+    top_ks = [5]
+    for top_k in top_ks:
+        for model in models:
+            for dataset in datasets:
+                main(dataset, model, 5, top_k=top_k)
+
+
+if __name__ == "__main__":
+    """
+    Benchmark the fine-tuning process for a given dataset and model.
+
+    Usage:
+    - For a single dataset
+    python lancedb/docs/benchmarks/llama-index-datasets.py --dataset Uber10KDataset2021 --model BAAI/bge-small-en-v1.5 --epochs 4 --top_k 5 --split 1 --project lancedb_finetune
+    
+    - For all datasets and models across all top_ks
+    python lancedb/docs/benchmarks/llama-index-datasets.py --benchmark-all
+    """
+
+    parser = argparse.ArgumentParser()
+    parser.add_argument(
+        "--dataset",
+        type=str,
+        default="BraintrustCodaHelpDeskDataset",
+        help="The dataset to use for fine-tuning",
+    )
+    parser.add_argument(
+        "--model",
+        type=str,
+        default="BAAI/bge-small-en-v1.5",
+        help="The model to use for fine-tuning",
+    )
+    parser.add_argument(
+        "--epochs",
+        type=int,
+        default=4,
+        help="The number of epochs to fine-tune the model",
+    )
+    parser.add_argument(
+        "--project",
+        type=str,
+        default="lancedb_finetune",
+        help="The wandb project to log the results",
+    )
+    parser.add_argument(
+        "--top_k", type=int, default=5, help="The number of top results to evaluate"
+    )
+    parser.add_argument(
+        "--split",
+        type=int,
+        default=1,
+        help="Whether to split the dataset into train and test(65-35 split), default is 1",
+    )
+    parser.add_argument(
+        "--eval-rag",
+        action="store_true",
+        default=False,
+        help="Whether to evaluate the model using RAG",
+    )
+    parser.add_argument(
+        "--benchmark-all",
+        action="store_true",
+        default=False,
+        help="Benchmark all datasets across all models and top_ks",
+    )
+    args = parser.parse_args()
+
+    if args.benchmark_all:
+        banchmark_all()
+    else:
+        main(
+            args.dataset,
+            args.model,
+            args.epochs,
+            args.top_k,
+            args.eval_rag,
+            args.split,
+            args.project,
+        )
--- a/docs/mkdocs.yml
+++ b/docs/mkdocs.yml
@@ -57,8 +57,6 @@ plugins:
            - https://arrow.apache.org/docs/objects.inv
            - https://pandas.pydata.org/docs/objects.inv
  - mkdocs-jupyter
-  - render_swagger:
-      allow_arbitrary_locations : true

 markdown_extensions:
  - admonition
@@ -108,9 +106,6 @@ nav:
          - Versioning & Reproducibility: notebooks/reproducibility.ipynb
          - Configuring Storage: guides/storage.md
          - Sync -> Async Migration Guide: migration.md
-          - Tuning retrieval performance:
-              - Choosing right query type: guides/tuning_retrievers/1_query_types.md
-              - Reranking: guides/tuning_retrievers/2_reranking.md
      - 🧬 Managing embeddings:
          - Overview: embeddings/index.md
          - Embedding functions: embeddings/embedding_functions.md
@@ -124,12 +119,9 @@ nav:
          - Polars: python/polars_arrow.md
          - DuckDB: python/duckdb.md
          - LangChain:
-            - LangChain 🔗: integrations/langchain.md
-            - LangChain demo: notebooks/langchain_demo.ipynb
+            - LangChain 🔗: https://python.langchain.com/docs/integrations/vectorstores/lancedb/
            - LangChain JS/TS 🔗: https://js.langchain.com/docs/integrations/vectorstores/lancedb
-          - LlamaIndex 🦙:
-            - LlamaIndex docs: integrations/llamaIndex.md
-            - LlamaIndex demo: notebooks/llamaIndex_demo.ipynb
+          - LlamaIndex 🦙: https://docs.llamaindex.ai/en/stable/examples/vector_stores/LanceDBIndexDemo/
          - Pydantic: python/pydantic.md
          - Voxel51: integrations/voxel51.md
          - PromptTools: integrations/prompttools.md
@@ -160,8 +152,7 @@ nav:
          - Overview: cloud/index.md
          - API reference:
              - 🐍 Python: python/saas-python.md
-              - 👾 JavaScript: javascript/modules.md
-              - REST API: cloud/rest.md
+              - 👾 JavaScript: javascript/saas-modules.md

  - Quick start: basic.md
  - Concepts:
@@ -190,9 +181,6 @@ nav:
      - Versioning & Reproducibility: notebooks/reproducibility.ipynb
      - Configuring Storage: guides/storage.md
      - Sync -> Async Migration Guide: migration.md
-      - Tuning retrieval performance:
-          - Choosing right query type: guides/tuning_retrievers/1_query_types.md
-          - Reranking: guides/tuning_retrievers/2_reranking.md
  - Managing Embeddings:
      - Overview: embeddings/index.md
      - Embedding functions: embeddings/embedding_functions.md
@@ -205,9 +193,9 @@ nav:
      - Pandas and PyArrow: python/pandas_and_pyarrow.md
      - Polars: python/polars_arrow.md
      - DuckDB: python/duckdb.md
-      - LangChain 🦜️🔗↗: integrations/langchain.md
+      - LangChain 🦜️🔗↗: https://python.langchain.com/docs/integrations/vectorstores/lancedb
      - LangChain.js 🦜️🔗↗: https://js.langchain.com/docs/integrations/vectorstores/lancedb
-      - LlamaIndex 🦙↗: integrations/llamaIndex.md
+      - LlamaIndex 🦙↗: https://gpt-index.readthedocs.io/en/latest/examples/vector_stores/LanceDBIndexDemo.html
      - Pydantic: python/pydantic.md
      - Voxel51: integrations/voxel51.md
      - PromptTools: integrations/prompttools.md
@@ -231,8 +219,7 @@ nav:
      - Overview: cloud/index.md
      - API reference:
          - 🐍 Python: python/saas-python.md
-          - 👾 JavaScript: javascript/modules.md
-          - REST API: cloud/rest.md
+          - 👾 JavaScript: javascript/saas-modules.md

 extra_css:
  - styles/global.css
--- a/docs/openapi.yml
+++ b/docs/openapi.yml
@@ -1,479 +0,0 @@
-openapi: 3.1.0
-info:
-  version: 1.0.0
-  title: LanceDB Cloud API
-  description: |
-    LanceDB Cloud API is a RESTful API that allows users to access and modify data stored in LanceDB Cloud.
-    Table actions are considered temporary resource creations and all use POST method.
-  contact:
-    name: LanceDB support
-    url: https://lancedb.com
-    email: contact@lancedb.com
-
-servers:
-  - url: https://{db}.{region}.api.lancedb.com
-    description: LanceDB Cloud REST endpoint.
-    variables:
-      db:
-        default: ""
-        description: the name of DB
-      region:
-        default: "us-east-1"
-        description: the service region of the DB
-
-security:
-  - key_auth: []
-
-components:
-  securitySchemes:
-    key_auth:
-      name: x-api-key
-      type: apiKey
-      in: header
-  parameters:
-    table_name:
-      name: name
-      in: path
-      description: name of the table
-      required: true
-      schema:
-        type: string
-  responses:
-    invalid_request:
-      description: Invalid request
-      content:
-        text/plain:
-          schema:
-            type: string
-    not_found:
-      description: Not found
-      content:
-        text/plain:
-          schema:
-            type: string
-    unauthorized:
-      description: Unauthorized
-      content:
-        text/plain:
-          schema:
-            type: string
-  requestBodies:
-    arrow_stream_buffer:
-      description: Arrow IPC stream buffer
-      required: true
-      content:
-        application/vnd.apache.arrow.stream:
-          schema:
-            type: string
-            format: binary
-
-paths:
-  /v1/table/:
-    get:
-      description: List tables, optionally, with pagination.
-      tags:
-        - Tables
-      summary: List Tables
-      operationId: listTables
-      parameters:
-        - name: limit
-          in: query
-          description: Limits the number of items to return.
-          schema:
-            type: integer
-        - name: page_token
-          in: query
-          description: Specifies the starting position of the next query
-          schema:
-            type: string
-      responses:
-        "200":
-          description: Successfully returned a list of tables in the DB
-          content:
-            application/json:
-              schema:
-                type: object
-                properties:
-                  tables:
-                    type: array
-                    items:
-                      type: string
-                  page_token:
-                    type: string
-
-        "400":
-          $ref: "#/components/responses/invalid_request"
-        "401":
-          $ref: "#/components/responses/unauthorized"
-        "404":
-          $ref: "#/components/responses/not_found"
-
-  /v1/table/{name}/create/:
-    post:
-      description: Create a new table
-      summary: Create a new table
-      operationId: createTable
-      tags:
-        - Tables
-      parameters:
-        - $ref: "#/components/parameters/table_name"
-      requestBody:
-        $ref: "#/components/requestBodies/arrow_stream_buffer"
-      responses:
-        "200":
-          description: Table successfully created
-        "400":
-          $ref: "#/components/responses/invalid_request"
-        "401":
-          $ref: "#/components/responses/unauthorized"
-        "404":
-          $ref: "#/components/responses/not_found"
-
-  /v1/table/{name}/query/:
-    post:
-      description: Vector Query
-      url: https://{db-uri}.{aws-region}.api.lancedb.com/v1/table/{name}/query/
-      tags:
-        - Data
-      summary: Vector Query
-      parameters:
-        - $ref: "#/components/parameters/table_name"
-      requestBody:
-        required: true
-        content:
-          application/json:
-            schema:
-              type: object
-              properties:
-                vector:
-                  type: FixedSizeList
-                  description: |
-                    The targetted vector to search for. Required.
-                vector_column:
-                  type: string
-                  description: |
-                    The column to query, it can be inferred from the schema if there is only one vector column.
-                prefilter:
-                  type: boolean
-                  description: |
-                    Whether to prefilter the data. Optional.
-                k:
-                  type: integer
-                  description: |
-                    The number of search results to return. Default is 10.
-                distance_type:
-                  type: string
-                  description: |
-                    The distance metric to use for search. L2, Cosine, Dot and Hamming are supported. Default is L2.
-                bypass_vector_index:
-                  type: boolean
-                  description: |
-                    Whether to bypass vector index. Optional.
-                filter:
-                  type: string
-                  description: |
-                    A filter expression that specifies the rows to query. Optional.
-                columns:
-                  type: array
-                  items:
-                    type: string
-                  description: |
-                    The columns to return. Optional.
-                nprobe:
-                  type: integer
-                  description: |
-                    The number of probes to use for search. Optional.
-                refine_factor:
-                  type: integer
-                  description: |
-                    The refine factor to use for search. Optional.
-
-      responses:
-        "200":
-          description: top k results if query is successfully executed
-          content:
-            application/json:
-              schema:
-                type: object
-                properties:
-                  results:
-                    type: array
-                    items:
-                      type: object
-                      properties:
-                        id:
-                          type: integer
-                        selected_col_1_to_return:
-                          type: col_1_type
-                        selected_col_n_to_return:
-                          type: col_n_type
-                        _distance:
-                          type: float
-
-        "400":
-          $ref: "#/components/responses/invalid_request"
-        "401":
-          $ref: "#/components/responses/unauthorized"
-        "404":
-          $ref: "#/components/responses/not_found"
-
-  /v1/table/{name}/insert/:
-    post:
-      description: Insert new data to the Table.
-      tags:
-        - Data
-      operationId: insertData
-      summary: Insert new data.
-      parameters:
-        - $ref: "#/components/parameters/table_name"
-      requestBody:
-        $ref: "#/components/requestBodies/arrow_stream_buffer"
-      responses:
-        "200":
-          description: Insert successful
-        "400":
-          $ref: "#/components/responses/invalid_request"
-        "401":
-          $ref: "#/components/responses/unauthorized"
-        "404":
-          $ref: "#/components/responses/not_found"
-  /v1/table/{name}/merge_insert/:
-    post:
-      description: Create a "merge insert" operation
-        This operation can add rows, update rows, and remove rows all in a single
-        transaction. See python method `lancedb.table.Table.merge_insert` for examples.
-      tags:
-        - Data
-      summary: Merge Insert
-      operationId: mergeInsert
-      parameters:
-        - $ref: "#/components/parameters/table_name"
-        - name: on
-          in: query
-          description: |
-            The column to use as the primary key for the merge operation.
-          required: true
-          schema:
-            type: string
-        - name: when_matched_update_all
-          in: query
-          description: |
-            Rows that exist in both the source table (new data) and
-            the target table (old data) will be updated, replacing
-            the old row with the corresponding matching row.
-          required: false
-          schema:
-            type: boolean
-        - name: when_matched_update_all_filt
-          in: query
-          description: |
-            If present then only rows that satisfy the filter expression will
-            be updated
-          required: false
-          schema:
-            type: string
-        - name: when_not_matched_insert_all
-          in: query
-          description: |
-            Rows that exist only in the source table (new data) will be
-            inserted into the target table (old data).
-          required: false
-          schema:
-            type: boolean
-        - name: when_not_matched_by_source_delete
-          in: query
-          description: |
-            Rows that exist only in the target table (old data) will be
-            deleted. An optional condition (`when_not_matched_by_source_delete_filt`)
-            can be provided to limit what data is deleted.
-          required: false
-          schema:
-            type: boolean
-        - name: when_not_matched_by_source_delete_filt
-          in: query
-          description: |
-            The filter expression that specifies the rows to delete.
-          required: false
-          schema:
-            type: string
-      requestBody:
-        $ref: "#/components/requestBodies/arrow_stream_buffer"
-      responses:
-        "200":
-          description: Merge Insert successful
-        "400":
-          $ref: "#/components/responses/invalid_request"
-        "401":
-          $ref: "#/components/responses/unauthorized"
-        "404":
-          $ref: "#/components/responses/not_found"
-  /v1/table/{name}/delete/:
-    post:
-      description: Delete rows from a table.
-      tags:
-        - Data
-      summary: Delete rows from a table
-      operationId: deleteData
-      parameters:
-        - $ref: "#/components/parameters/table_name"
-      requestBody:
-        required: true
-        content:
-          application/json:
-            schema:
-              type: object
-              properties:
-                predicate:
-                  type: string
-                  description: |
-                    A filter expression that specifies the rows to delete.
-      responses:
-        "200":
-          description: Delete successful
-        "401":
-          $ref: "#/components/responses/unauthorized"
-  /v1/table/{name}/drop/:
-    post:
-      description: Drop a table
-      tags:
-        - Tables
-      summary: Drop a table
-      operationId: dropTable
-      parameters:
-        - $ref: "#/components/parameters/table_name"
-      requestBody:
-        $ref: "#/components/requestBodies/arrow_stream_buffer"
-      responses:
-        "200":
-          description: Drop successful
-        "401":
-          $ref: "#/components/responses/unauthorized"
-
-  /v1/table/{name}/describe/:
-    post:
-      description: Describe a table and return Table Information.
-      tags:
-        - Tables
-      summary: Describe a table
-      operationId: describeTable
-      parameters:
-        - $ref: "#/components/parameters/table_name"
-      responses:
-        "200":
-          description: Table information
-          content:
-            application/json:
-              schema:
-                type: object
-                properties:
-                  table:
-                    type: string
-                  version:
-                    type: integer
-                  schema:
-                    type: string
-                  stats:
-                    type: object
-        "401":
-          $ref: "#/components/responses/unauthorized"
-        "404":
-          $ref: "#/components/responses/not_found"
-
-  /v1/table/{name}/index/list/:
-    post:
-      description: List indexes of a table
-      tags:
-        - Tables
-      summary: List indexes of a table
-      operationId: listIndexes
-      parameters:
-        - $ref: "#/components/parameters/table_name"
-      responses:
-        "200":
-          description: Available list of indexes on the table.
-          content:
-            application/json:
-              schema:
-                type: object
-                properties:
-                  indexes:
-                    type: array
-                    items:
-                      type: object
-                      properties:
-                        columns:
-                          type: array
-                          items:
-                            type: string
-                        index_name:
-                          type: string
-                        index_uuid:
-                          type: string
-        "401":
-          $ref: "#/components/responses/unauthorized"
-        "404":
-          $ref: "#/components/responses/not_found"
-  /v1/table/{name}/create_index/:
-    post:
-      description: Create vector index on a Table
-      tags:
-        - Tables
-      summary: Create vector index on a Table
-      operationId: createIndex
-      parameters:
-        - $ref: "#/components/parameters/table_name"
-      requestBody:
-        required: true
-        content:
-          application/json:
-            schema:
-              type: object
-              properties:
-                column:
-                  type: string
-                metric_type:
-                  type: string
-                  nullable: false
-                  description: |
-                    The metric type to use for the index. L2, Cosine, Dot are supported.
-                index_type:
-                  type: string
-      responses:
-        "200":
-          description: Index successfully created
-        "400":
-          $ref: "#/components/responses/invalid_request"
-        "401":
-          $ref: "#/components/responses/unauthorized"
-        "404":
-          $ref: "#/components/responses/not_found"
-  /v1/table/{name}/create_scalar_index/:
-    post:
-      description: Create a scalar index on a table
-      tags:
-        - Tables
-      summary: Create a scalar index on a table
-      operationId: createScalarIndex
-      parameters:
-        - $ref: "#/components/parameters/table_name"
-      requestBody:
-        required: true
-        content:
-          application/json:
-            schema:
-              type: object
-              properties:
-                column:
-                  type: string
-                index_type:
-                  type: string
-                  required: false
-      responses:
-        "200":
-          description: Scalar Index successfully created
-        "400":
-          $ref: "#/components/responses/invalid_request"
-        "401":
-          $ref: "#/components/responses/unauthorized"
-        "404":
-          $ref: "#/components/responses/not_found"
--- a/docs/requirements.txt
+++ b/docs/requirements.txt
@@ -2,5 +2,4 @@ mkdocs==1.5.3
 mkdocs-jupyter==0.24.1
 mkdocs-material==9.5.3
 mkdocstrings[python]==0.20.0
-mkdocs-render-swagger-plugin
-pydantic
+pydantic
--- a/docs/src/basic.md
+++ b/docs/src/basic.md
@@ -44,36 +44,6 @@

    !!! info "Please also make sure you're using the same version of Arrow as in the [lancedb crate](https://github.com/lancedb/lancedb/blob/main/Cargo.toml)"

-### Preview releases
-
-Stable releases are created about every 2 weeks. For the latest features and bug
-fixes, you can install the preview release. These releases receive the same
-level of testing as stable releases, but are not guaranteed to be available for
-more than 6 months after they are released. Once your application is stable, we
-recommend switching to stable releases.
-
-=== "Python"
-
-      ```shell
-      pip install --pre --extra-index-url https://pypi.fury.io/lancedb/ lancedb
-      ```
-
-=== "Typescript"
-
-      ```shell
-      npm install vectordb@preview
-      ```
-
-=== "Rust"
-    
-    We don't push preview releases to crates.io, but you can referent the tag
-    in GitHub within your Cargo dependencies:
-
-    ```toml
-    [dependencies]
-    lancedb = { git = "https://github.com/lancedb/lancedb.git", tag = "vX.Y.Z-beta.N" }
-    ```
-
 ## Connect to a database

 === "Python"
@@ -180,9 +150,6 @@ table.

 !!! info "Under the hood, LanceDB reads in the Apache Arrow data and persists it to disk using the [Lance format](https://www.github.com/lancedb/lance)."

-!!! info "Automatic embedding generation with Embedding API"
-    When working with embedding models, it is recommended to use the LanceDB embedding API to automatically create vector representation of the data and queries in the background. See the [quickstart example](#using-the-embedding-api) or the embedding API [guide](./embeddings/)
-
 ### Create an empty table

 Sometimes you may not have the data to insert into the table at creation time.
@@ -197,9 +164,6 @@ similar to a `CREATE TABLE` statement in SQL.
      --8<-- "python/python/tests/docs/test_basic.py:create_empty_table_async"
      ```

-    !!! note "You can define schema in Pydantic"
-        LanceDB comes with Pydantic support, which allows you to define the schema of your data using Pydantic models. This makes it easy to work with LanceDB tables and data. Learn more about all supported types in [tables guide](./guides/tables.md).
-
 === "Typescript"

    ```typescript
@@ -430,19 +394,6 @@ Use the `drop_table()` method on the database to remove a table.
    })
    ```

-## Using the Embedding API
-You can use the embedding API when working with embedding models. It automatically vectorizes the data at ingestion and query time and comes with built-in integrations with popular embedding models like Openai, Hugging Face, Sentence Transformers, CLIP and more.
-
-=== "Python"
-
-    ```python
-    --8<-- "python/python/tests/docs/test_embeddings_optional.py:imports"
-    --8<-- "python/python/tests/docs/test_embeddings_optional.py:openai_embeddings"
-    ```
-
-Learn about using the existing integrations and creating custom embedding functions in the [embedding API guide](./embeddings/).
-
-
 ## What's next

 This section covered the very basics of using LanceDB. If you're learning about vector databases for the first time, you may want to read the page on [indexing](concepts/index_ivfpq.md) to get familiar with the concepts.
--- a/docs/src/cloud/rest.md
+++ b/docs/src/cloud/rest.md
@@ -1 +0,0 @@
-!!swagger ../../openapi.yml!!
--- a/docs/src/embeddings/default_embedding_functions.md
+++ b/docs/src/embeddings/default_embedding_functions.md
@@ -159,7 +159,7 @@ Allows you to set parameters when registering a `sentence-transformers` object.
    from lancedb.embeddings import get_registry

    db = lancedb.connect("/tmp/db")
-    model = get_registry().get("sentence-transformers").create(name="BAAI/bge-small-en-v1.5", device="cpu")
+    model = get_registry.get("sentence-transformers").create(name="BAAI/bge-small-en-v1.5", device="cpu")

    class Words(LanceModel):
        text: str = model.SourceField()
@@ -193,57 +193,19 @@ from lancedb.pydantic import LanceModel, Vector

 model = get_registry().get("huggingface").create(name='facebook/bart-base')

-class Words(LanceModel):
+class TextModel(LanceModel):
    text: str = model.SourceField()
    vector: Vector(model.ndims()) = model.VectorField()

 df = pd.DataFrame({"text": ["hi hello sayonara", "goodbye world"]})
 table = db.create_table("greets", schema=Words)
-table.add(df)
+table.add()
 query = "old greeting"
 actual = table.search(query).limit(1).to_pydantic(Words)[0]
 print(actual.text)
 ```


-### Ollama embeddings
-Generate embeddings via the [ollama](https://github.com/ollama/ollama-python) python library. More details:
-
- [Ollama docs on embeddings](https://github.com/ollama/ollama/blob/main/docs/api.md#generate-embeddings)
- [Ollama blog on embeddings](https://ollama.com/blog/embedding-models)
-
-| Parameter              | Type                       | Default Value            | Description                                                                                                                                    |
-|------------------------|----------------------------|--------------------------|------------------------------------------------------------------------------------------------------------------------------------------------|
-| `name`                 | `str`                      | `nomic-embed-text`       | The name of the model.                                                                                                                         |
-| `host`                 | `str`                      | `http://localhost:11434` | The Ollama host to connect to.                                                                                                                 |
-| `options`              | `ollama.Options` or `dict` | `None`                   | Additional model parameters listed in the documentation for the Modelfile such as `temperature`. |
-| `keep_alive`           | `float` or `str`           | `"5m"`                   | Controls how long the model will stay loaded into memory following the request.                                                                |
-| `ollama_client_kwargs` | `dict`                     | `{}`                     | kwargs that can be past to the `ollama.Client`.                                                                                                |
-
-```python
-import lancedb
-from lancedb.pydantic import LanceModel, Vector
-from lancedb.embeddings import get_registry
-
-db = lancedb.connect("/tmp/db")
-func = get_registry().get("ollama").create(name="nomic-embed-text")
-
-class Words(LanceModel):
-    text: str = func.SourceField()
-    vector: Vector(func.ndims()) = func.VectorField()
-
-table = db.create_table("words", schema=Words, mode="overwrite")
-table.add([
-    {"text": "hello world"},
-    {"text": "goodbye world"}
-])
-
-query = "greetings"
-actual = table.search(query).limit(1).to_pydantic(Words)[0]
-print(actual.text)
-```
-
-
 ### OpenAI embeddings
 LanceDB registers the OpenAI embeddings function in the registry by default, as `openai`. Below are the parameters that you can customize when creating the instances:

@@ -365,68 +327,6 @@ tbl.add(df)
 rs = tbl.search("hello").limit(1).to_pandas()
 ```

-### Cohere Embeddings
-Using cohere API requires cohere package, which can be installed using `pip install cohere`. Cohere embeddings are used to generate embeddings for text data. The embeddings can be used for various tasks like semantic search, clustering, and classification.
-You also need to set the `COHERE_API_KEY` environment variable to use the Cohere API.
-
-Supported models are:
-```
-    * embed-english-v3.0
-    * embed-multilingual-v3.0
-    * embed-english-light-v3.0
-    * embed-multilingual-light-v3.0
-    * embed-english-v2.0
-    * embed-english-light-v2.0
-    * embed-multilingual-v2.0
-```
-
-Supported parameters (to be passed in `create` method) are:
-
-| Parameter | Type | Default Value | Description |
-|---|---|---|---|
-| `name` | `str` | `"embed-english-v2.0"` | The model ID of the cohere model to use. Supported base models for Text Embeddings: embed-english-v3.0, embed-multilingual-v3.0, embed-english-light-v3.0, embed-multilingual-light-v3.0, embed-english-v2.0, embed-english-light-v2.0, embed-multilingual-v2.0 |
-| `source_input_type` | `str` | `"search_document"` | The type of input data to be used for the source column. |
-| `query_input_type` | `str` | `"search_query"` | The type of input data to be used for the query. |
-
-Cohere supports following input types:
-| Input Type               | Description                          |
-|-------------------------|---------------------------------------|
-| "`search_document`"     | Used for embeddings stored in a vector|
-|                         | database for search use-cases.        |
-| "`search_query`"        | Used for embeddings of search queries |
-|                         | run against a vector DB               |
-| "`semantic_similarity`" | Specifies the given text will be used |
-|                         | for Semantic Textual Similarity (STS) |
-| "`classification`"      | Used for embeddings passed through a  |
-|                         | text classifier.                      |
-| "`clustering`"          | Used for the embeddings run through a |
-|                         | clustering algorithm                  |
-
-Usage Example:
-    
-    ```python
-        import lancedb
-        from lancedb.pydantic import LanceModel, Vector
-        from lancedb.embeddings import EmbeddingFunctionRegistry
-
-        cohere = EmbeddingFunctionRegistry
-            .get_instance()
-            .get("cohere")
-            .create(name="embed-multilingual-v2.0")
-
-        class TextModel(LanceModel):
-            text: str = cohere.SourceField()
-            vector: Vector(cohere.ndims()) =  cohere.VectorField()
-
-        data = [ { "text": "hello world" },
-                { "text": "goodbye world" }]
-
-        db = lancedb.connect("~/.lancedb")
-        tbl = db.create_table("test", schema=TextModel, mode="overwrite")
-
-        tbl.add(data)
-    ```
-
 ### AWS Bedrock Text Embedding Functions
 AWS Bedrock supports multiple base models for generating text embeddings. You need to setup the AWS credentials to use this embedding function.
 You can do so by using `awscli` and also add your session_token:
--- a/docs/src/embeddings/embedding_functions.md
+++ b/docs/src/embeddings/embedding_functions.md
@@ -2,9 +2,6 @@ Representing multi-modal data as vector embeddings is becoming a standard practi

 For this purpose, LanceDB introduces an **embedding functions API**, that allow you simply set up once, during the configuration stage of your project. After this, the table remembers it, effectively making the embedding functions *disappear in the background* so you don't have to worry about manually passing callables, and instead, simply focus on the rest of your data engineering pipeline.

-!!! Note "LanceDB cloud doesn't support embedding functions yet"
-    LanceDB Cloud does not support embedding functions yet. You need to generate embeddings before ingesting into the table or querying.
-
 !!! warning
    Using the embedding function registry means that you don't have to explicitly generate the embeddings yourself. 
    However, if your embedding function changes, you'll have to re-configure your table with the new embedding function 
@@ -49,7 +46,7 @@ For this purpose, LanceDB introduces an **embedding functions API**, that allow

    ```python
    class Pets(LanceModel):
-        vector: Vector(clip.ndims()) = clip.VectorField()
+        vector: Vector(clip.ndims) = clip.VectorField()
        image_uri: str = clip.SourceField()
    ```

@@ -152,7 +149,7 @@ You can also use the integration for adding utility operations in the schema. Fo

 ```python
 class Pets(LanceModel):
-    vector: Vector(clip.ndims()) = clip.VectorField()
+    vector: Vector(clip.ndims) = clip.VectorField()
    image_uri: str = clip.SourceField()

    @property
@@ -169,4 +166,4 @@ rs[2].image
 ![](../assets/dog_clip_output.png)

 Now that you have the basic idea about LanceDB embedding functions and the embedding function registry,
-let's dive deeper into defining your own [custom functions](./custom_embedding_function.md).
+let's dive deeper into defining your own [custom functions](./custom_embedding_function.md).
--- a/docs/src/embeddings/index.md
+++ b/docs/src/embeddings/index.md
@@ -68,39 +68,6 @@ table.add(
    ]
 )

-query = "greetings"
-actual = table.search(query).limit(1).to_pydantic(Words)[0]
-print(actual.text)
-```
-
-### Jina Embeddings
-LanceDB registers the JinaAI embeddings function in the registry as `jina`. You can pass any supported model name to the `create`. By default it uses `"jina-clip-v1"`.
-`jina-clip-v1` can handle both text and images and other models only support `text`.
-
-You need to pass `JINA_API_KEY` in the environment variable or pass it as `api_key` to `create` method.
-
-```python
-import os
-import lancedb
-from lancedb.pydantic import LanceModel, Vector
-from lancedb.embeddings import get_registry
-os.environ['JINA_API_KEY'] = "jina_*"
-
-db = lancedb.connect("/tmp/db")
-func = get_registry().get("jina").create(name="jina-clip-v1")
-
-class Words(LanceModel):
-    text: str = func.SourceField()
-    vector: Vector(func.ndims()) = func.VectorField()
-
-table = db.create_table("words", schema=Words, mode="overwrite")
-table.add(
-    [
-        {"text": "hello world"},
-        {"text": "goodbye world"}
-    ]
-    )
-
 query = "greetings"
 actual = table.search(query).limit(1).to_pydantic(Words)[0]
 print(actual.text)
--- a/docs/src/fts.md
+++ b/docs/src/fts.md
@@ -2,6 +2,7 @@

 LanceDB provides support for full-text search via [Tantivy](https://github.com/quickwit-oss/tantivy) (currently Python only), allowing you to incorporate keyword-based search (based on BM25) in your retrieval solutions. Our goal is to push the FTS integration down to the Rust level in the future, so that it's available for Rust and JavaScript users as well.  Follow along at [this Github issue](https://github.com/lancedb/lance/issues/1195)

+A hybrid search solution combining vector and full-text search is also on the way.

 ## Installation

@@ -54,16 +55,6 @@ This returns the result as a list of dictionaries as follows.
 !!! note
    LanceDB automatically searches on the existing FTS index if the input to the search is of type `str`. If you provide a vector as input, LanceDB will search the ANN index instead.

-## Tokenization
-By default the text is tokenized by splitting on punctuation and whitespaces and then removing tokens that are longer than 40 chars. For more language specific tokenization then provide the argument tokenizer_name with the 2 letter language code followed by "_stem". So for english it would be "en_stem".
-
-```python
-table.create_fts_index("text", tokenizer_name="en_stem")
-```
-
-The following [languages](https://docs.rs/tantivy/latest/tantivy/tokenizer/enum.Language.html) are currently supported.
-
-
 ## Index multiple columns

 If you have multiple string columns to index, there's no need to combine them manually -- simply pass them all as a list to `create_fts_index`:
@@ -149,7 +140,6 @@ is treated as a phrase query.
 In general, a query that's declared as a phrase query will be wrapped in double quotes during parsing, with nested
 double quotes replaced by single quotes.

-
 ## Configurations

 By default, LanceDB configures a 1GB heap size limit for creating the index. You can
--- a/docs/src/guides/storage.md
+++ b/docs/src/guides/storage.md
@@ -265,108 +265,6 @@ For **read-only access**, LanceDB will need a policy such as:
 }
 ```

-#### DynamoDB Commit Store for concurrent writes
-
-By default, S3 does not support concurrent writes. Having two or more processes
-writing to the same table at the same time can lead to data corruption. This is
-because S3, unlike other object stores, does not have any atomic put or copy
-operation.
-
-To enable concurrent writes, you can configure LanceDB to use a DynamoDB table
-as a commit store. This table will be used to coordinate writes between
-different processes. To enable this feature, you must modify your connection
-URI to use the `s3+ddb` scheme and add a query parameter `ddbTableName` with the
-name of the table to use.
-
-=== "Python"
-
-    ```python
-    import lancedb
-    db = await lancedb.connect_async(
-        "s3+ddb://bucket/path?ddbTableName=my-dynamodb-table",
-    )
-    ```
-
-=== "JavaScript"
-
-    ```javascript
-    const lancedb = require("lancedb");
-
-    const db = await lancedb.connect(
-        "s3+ddb://bucket/path?ddbTableName=my-dynamodb-table",
-    );
-    ```
-
-The DynamoDB table must be created with the following schema:
-
- Hash key: `base_uri` (string)
- Range key: `version` (number)
-
-You can create this programmatically with:
-
-=== "Python"
-
-    <!-- skip-test -->
-    ```python
-    import boto3
-
-    dynamodb = boto3.client("dynamodb")
-    table = dynamodb.create_table(
-        TableName=table_name,
-        KeySchema=[
-            {"AttributeName": "base_uri", "KeyType": "HASH"},
-            {"AttributeName": "version", "KeyType": "RANGE"},
-        ],
-        AttributeDefinitions=[
-            {"AttributeName": "base_uri", "AttributeType": "S"},
-            {"AttributeName": "version", "AttributeType": "N"},
-        ],
-        ProvisionedThroughput={"ReadCapacityUnits": 1, "WriteCapacityUnits": 1},
-    )
-    ```
-
-=== "JavaScript"
-
-    <!-- skip-test -->
-    ```javascript
-    import {
-      CreateTableCommand,
-      DynamoDBClient,
-    } from "@aws-sdk/client-dynamodb";
-
-    const dynamodb = new DynamoDBClient({
-      region: CONFIG.awsRegion,
-      credentials: {
-        accessKeyId: CONFIG.awsAccessKeyId,
-        secretAccessKey: CONFIG.awsSecretAccessKey,
-      },
-      endpoint: CONFIG.awsEndpoint,
-    });
-    const command = new CreateTableCommand({
-      TableName: table_name,
-      AttributeDefinitions: [
-        {
-          AttributeName: "base_uri",
-          AttributeType: "S",
-        },
-        {
-          AttributeName: "version",
-          AttributeType: "N",
-        },
-      ],
-      KeySchema: [
-        { AttributeName: "base_uri", KeyType: "HASH" },
-        { AttributeName: "version", KeyType: "RANGE" },
-      ],
-      ProvisionedThroughput: {
-        ReadCapacityUnits: 1,
-        WriteCapacityUnits: 1,
-      },
-    });
-    await client.send(command);
-    ```
-
-
 #### S3-compatible stores

 LanceDB can also connect to S3-compatible stores, such as MinIO. To do so, you must specify both region and endpoint:
@@ -401,14 +299,6 @@ LanceDB can also connect to S3-compatible stores, such as MinIO. To do so, you m

 This can also be done with the ``AWS_ENDPOINT`` and ``AWS_DEFAULT_REGION`` environment variables.

-!!! tip "Local servers"
-
-    For local development, the server often has a `http` endpoint rather than a
-    secure `https` endpoint. In this case, you must also set the `ALLOW_HTTP`
-    environment variable to `true` to allow non-TLS connections, or pass the
-    storage option `allow_http` as `true`. If you do not do this, you will get
-    an error like `URL scheme is not allowed`.
-
 #### S3 Express

 LanceDB supports [S3 Express One Zone](https://aws.amazon.com/s3/storage-classes/express-one-zone/) endpoints, but requires additional configuration. Also, S3 Express endpoints only support connecting from an EC2 instance within the same region.
--- a/docs/src/guides/tables.md
+++ b/docs/src/guides/tables.md
@@ -116,21 +116,21 @@ This guide will show how to create tables, insert data into them, and update the

 ### From a Polars DataFrame

-LanceDB supports [Polars](https://pola.rs/), a modern, fast DataFrame library
-written in Rust. Just like in Pandas, the Polars integration is enabled by PyArrow
-under the hood. A deeper integration between LanceDB Tables and Polars DataFrames
-is on the way.
+    LanceDB supports [Polars](https://pola.rs/), a modern, fast DataFrame library
+    written in Rust. Just like in Pandas, the Polars integration is enabled by PyArrow
+    under the hood. A deeper integration between LanceDB Tables and Polars DataFrames
+    is on the way.

-```python
-import polars as pl
+    ```python
+    import polars as pl

-data = pl.DataFrame({
-    "vector": [[3.1, 4.1], [5.9, 26.5]],
-    "item": ["foo", "bar"],
-    "price": [10.0, 20.0]
-})
-table = db.create_table("pl_table", data=data)
-```
+    data = pl.DataFrame({
+        "vector": [[3.1, 4.1], [5.9, 26.5]],
+        "item": ["foo", "bar"],
+        "price": [10.0, 20.0]
+    })
+    table = db.create_table("pl_table", data=data)
+    ```

 ### From an Arrow Table
 === "Python"
@@ -452,27 +452,6 @@ After a table has been created, you can always add more data to it using the var
    tbl.add(pydantic_model_items)
    ```

-    ??? "Ingesting Pydantic models with LanceDB embedding API"
-        When using LanceDB's embedding API, you can add Pydantic models directly to the table. LanceDB will automatically convert the `vector` field to a vector before adding it to the table. You need to specify the default value of `vector` feild as None to allow LanceDB to automatically vectorize the data.
-
-        ```python
-        import lancedb
-        from lancedb.pydantic import LanceModel, Vector
-        from lancedb.embeddings import get_registry
-
-        db = lancedb.connect("~/tmp")
-        embed_fcn = get_registry().get("huggingface").create(name="BAAI/bge-small-en-v1.5")
-
-        class Schema(LanceModel):
-            text: str = embed_fcn.SourceField()
-            vector: Vector(embed_fcn.ndims()) = embed_fcn.VectorField(default=None)
-
-        tbl = db.create_table("my_table", schema=Schema, mode="overwrite")
-        models = [Schema(text="hello"), Schema(text="world")]
-        tbl.add(models)
-        ```
-
-

 === "JavaScript"

@@ -657,31 +636,6 @@ The `values` parameter is used to provide the new values for the columns as lite

    When rows are updated, they are moved out of the index. The row will still show up in ANN queries, but the query will not be as fast as it would be if the row was in the index. If you update a large proportion of rows, consider rebuilding the index afterwards.

-## Drop a table
-
-Use the `drop_table()` method on the database to remove a table.
-
-=== "Python"
-
-      ```python
-      --8<-- "python/python/tests/docs/test_basic.py:drop_table"
-      --8<-- "python/python/tests/docs/test_basic.py:drop_table_async"
-      ```
-
-      This permanently removes the table and is not recoverable, unlike deleting rows.
-      By default, if the table does not exist an exception is raised. To suppress this,
-      you can pass in `ignore_missing=True`.
-
-=== "Javascript/Typescript"
-
-      ```typescript
-      --8<-- "docs/src/basic_legacy.ts:drop_table"
-      ```
-
-      This permanently removes the table and is not recoverable, unlike deleting rows.
-      If the table does not exist an exception is raised.
-
-
 ## Consistency

 In LanceDB OSS, users can set the `read_consistency_interval` parameter on connections to achieve different levels of read consistency. This parameter determines how frequently the database synchronizes with the underlying storage system to check for updates made by other processes. If another process updates a table, the database will not see the changes until the next synchronization.
--- a/docs/src/guides/tuning_retrievers/1_query_types.md
+++ b/docs/src/guides/tuning_retrievers/1_query_types.md
@@ -1,128 +0,0 @@
-## Improving retriever performance
-VectorDBs are used as retreivers in recommender or chatbot-based systems for retrieving relevant data based on user queries. For example, retriever is a critical component of Retrieval Augmented Generation (RAG) acrhitectures. In this section, we will discuss how to improve the performance of retrievers.
-
-There are serveral ways to improve the performance of retrievers. Some of the common techniques are:
-
-* Using different query types
-* Using hybrid search
-* Fine-tuning the embedding models
-* Using different embedding models
-
-Using different embedding models is something that's very specific to the use case and the data. So we will not discuss it here. In this section, we will discuss the first three techniques.
-
-
-!!! note "Note"
-    We'll be using a simple metric called "hit-rate" for evaluating the performance of the retriever across this guide. Hit-rate is the percentage of queries for which the retriever returned the correct answer in the top-k results. For example, if the retriever returned the correct answer in the top-3 results for 70% of the queries, then the hit-rate@3 is 0.7.
-
-
-## The dataset
-We'll be using a QA dataset generated using a LLama2 review paper. The dataset contains 221 query, context and answer triplets. The queries and answers are generated using GPT-4 based on a given query. Full script used to generate the dataset can be found on this [repo](https://github.com/lancedb/ragged). It can be downloaded from [here](https://github.com/AyushExel/assets/blob/main/data_qa.csv)
-
-### Using different query types
-Let's setup the embeddings and the dataset first. We'll use the LanceDB's `huggingface` embeddings integration for this guide. 
-
-```python
-import lancedb
-import pandas as pd
-from lancedb.embeddings import get_registry
-from lancedb.pydantic import Vector, LanceModel
-
-db = lancedb.connect("~/lancedb/query_types")
-df = pd.read_csv("data_qa.csv")
-
-embed_fcn = get_registry().get("huggingface").create(name="BAAI/bge-small-en-v1.")
-
-class Schema(LanceModel):
-    context: str = embed_fcn.SourceField()
-    vector: Vector(embed_fcn.ndims()) = embed_fcn.VectorField()
-
-table = db.create_table("qa", schema=Schema)
-table.add(df[["context"]].to_dict(orient="records"))
-
-queries = df["query"].tolist()
-```
-
-Now that we have the dataset and embeddings table set up, here's how you can run different query types on the dataset.
-
-* <b> Vector Search: </b>
-
-    ```python
-    table.search(quries[0], query_type="vector").limit(5).to_pandas()
-    ```
-    By default, LanceDB uses vector search query type for searching and it automatically converts the input query to a vector before searching when using embedding API. So, the following statement is equivalent to the above statement.
-
-    ```python
-    table.search(quries[0]).limit(5).to_pandas()
-    ```
-
-    Vector or semantic search is useful when you want to find documents that are similar to the query in terms of meaning.
-
---
-
-* <b> Full-text Search: </b>
-    
-    FTS requires creating an index on the column you want to search on. `replace=True` will replace the existing index if it exists.
-    Once the index is created, you can search using the `fts` query type.
-    ```python
-    table.create_fts_index("context", replace=True)
-    table.search(quries[0], query_type="fts").limit(5).to_pandas()
-    ```
-
-    Full-text search is useful when you want to find documents that contain the query terms.
-
---
-
-* <b> Hybrid Search: </b>
-
-    Hybrid search is a combination of vector and full-text search. Here's how you can run a hybrid search query on the dataset.
-    ```python
-    table.search(quries[0], query_type="hybrid").limit(5).to_pandas()
-    ```
-    Hybrid search requires a reranker to combine and rank the results from vector and full-text search. We'll cover reranking as a concept in the next section.
-
-    Hybrid search is useful when you want to combine the benefits of both vector and full-text search.
-
-    !!! note "Note"
-        By default, it uses `LinearCombinationReranker` that combines the scores from vector and full-text search using a weighted linear combination. It is the simplest reranker implementation available in LanceDB. You can also use other rerankers like `CrossEncoderReranker` or `CohereReranker` for reranking the results.
-        Learn more about rerankers [here](https://lancedb.github.io/lancedb/reranking/)
-
-    
-
-### Hit rate evaluation results
-
-Now that we have seen how to run different query types on the dataset, let's evaluate the hit-rate of each query type on the dataset.
-For brevity, the entire evaluation script is not shown here. You can find the complete evaluation and benchmarking utility scripts [here](https://github.com/lancedb/ragged).
-
-Here are the hit-rate results for the dataset:
-
-| Query Type | Hit-rate@5 |
-| --- | --- |
-| Vector Search | 0.640 |
-| Full-text Search | 0.595 |
-| Hybrid Search (w/ LinearCombinationReranker) | 0.645 |
-
-**Choosing query type** is very specific to the use case and the data. This synthetic dataset has been generated to be semantically challenging, i.e, the queries don't have a lot of keywords in common with the context. So, vector search performs better than full-text search. However, in real-world scenarios, full-text search might perform better than vector search. Hybrid search is a good choice when you want to combine the benefits of both vector and full-text search.
-
-### Evaluation results on other datasets
-
-The hit-rate results can vary based on the dataset and the query type. Here are the hit-rate results for the other datasets using the same embedding function.
-
-* <b> SQuAD Dataset: </b>
-
-    | Query Type | Hit-rate@5 |
-    | --- | --- |
-    | Vector Search | 0.822 |
-    | Full-text Search | 0.835 |
-    | Hybrid Search (w/ LinearCombinationReranker) | 0.8874 |
-
-* <b> Uber10K sec filing Dataset: </b>
-
-    | Query Type | Hit-rate@5 |
-    | --- | --- |
-    | Vector Search | 0.608 |
-    | Full-text Search | 0.82 |
-    | Hybrid Search (w/ LinearCombinationReranker) | 0.80 |
-
-In these standard datasets, FTS seems to perform much better than vector search because the queries have a lot of keywords in common with the context. So, in general choosing the query type is very specific to the use case and the data.
-
-
--- a/docs/src/guides/tuning_retrievers/2_reranking.md
+++ b/docs/src/guides/tuning_retrievers/2_reranking.md
@@ -1,78 +0,0 @@
-Continuing from the previous example, we can now rerank the results using more complex rerankers.
-
-## Reranking search results
-You can rerank any search results using a reranker. The syntax for reranking is as follows:
-
-```python
-from lancedb.rerankers import LinearCombinationReranker
-
-reranker = LinearCombinationReranker()
-table.search(quries[0], query_type="hybrid").rerank(reranker=reranker).limit(5).to_pandas()
-```
-Based on the `query_type`, the `rerank()` function can accept other arguments as well. For example, hybrid search accepts a `normalize` param to determine the score normalization method.
-
-!!! note "Note"
-    LanceDB provides a `Reranker` base class that can be extended to implement custom rerankers. Each reranker must implement the `rerank_hybrid` method. `rerank_vector` and `rerank_fts` methods are optional. For example, the `LinearCombinationReranker` only implements the `rerank_hybrid` method and so it can only be used for reranking hybrid search results.
-
-## Choosing a Reranker
-There are many rerankers available in LanceDB like `CrossEncoderReranker`, `CohereReranker`, and `ColBERT`. The choice of reranker depends on the dataset and the application. You can even implement you own custom reranker by extending the `Reranker` class. For more details about each available reranker and performance comparison, refer to the [rerankers](https://lancedb.github.io/lancedb/reranking/) documentation.
-
-In this example, we'll use the `CohereReranker` to rerank the search results. It requires  `cohere` to be installed and `COHERE_API_KEY` to be set in the environment. To get your API key, sign up on [Cohere](https://cohere.ai/).
-
-```python
-from lancedb.rerankers import CohereReranker
-
-# use Cohere reranker v3
-reranker = CohereReranker(model_name="rerank-english-v3.0") # default model is "rerank-english-v2.0"
-```
-
-### Reranking search results
-Now we can rerank all query type results using the `CohereReranker`:
-
-```python
-
-# rerank hybrid search results
-table.search(quries[0], query_type="hybrid").rerank(reranker=reranker).limit(5).to_pandas()
-
-# rerank vector search results
-table.search(quries[0], query_type="vector").rerank(reranker=reranker).limit(5).to_pandas()
-
-# rerank fts search results
-table.search(quries[0], query_type="fts").rerank(reranker=reranker).limit(5).to_pandas()
-```
-
-Each reranker can accept additional arguments. For example, `CohereReranker` accepts `top_k` and `batch_size` params to control the number of documents to rerank and the batch size for reranking respectively. Similarly, a custom reranker can accept any number of arguments based on the implementation. For example, a reranker can accept a `filter` that implements some custom logic to filter out documents before reranking.
-
-## Results
-
-Let us take a look at the same datasets from the previous sections, using the same embedding table but with Cohere reranker applied to all query types.
-
-!!! note "Note"
-    When reranking fts or vector search results, the search results are over-fetched by a factor of 2 and then reranked. From the reranked set, `top_k` (5 in this case) results are taken. This is done because reranking will have no effect on the hit-rate if we only fetch the `top_k` results.
-
-### Synthetic LLama2 paper dataset
-
-| Query Type | Hit-rate@5 |
-| --- | --- |
-| Vector |  0.640 |
-| FTS   |  0.595  |
-| Reranked vector | 0.677    |
-| Reranked fts  | 0.672    |
-| Hybrid | 0.759 |
-
-### SQuAD Dataset
-
-
-### Uber10K sec filing Dataset
-
-| Query Type | Hit-rate@5 |
-| --- | --- |
-| Vector |  0.608 |
-| FTS   |  0.824  |
-| Reranked vector | 0.671    |
-| Reranked fts  | 0.843    |
-| Hybrid | 0.849 |
-
-
-
-
--- a/docs/src/hybrid_search/eval.md
+++ b/docs/src/hybrid_search/eval.md
@@ -5,9 +5,7 @@ Hybrid Search is a broad (often misused) term. It can mean anything from combini
 ## The challenge of (re)ranking search results
 Once you have a group of the most relevant search results from multiple search sources, you'd likely standardize the score and rank them accordingly. This process can also be seen as another independent step - reranking.
 There are two approaches for reranking search results from multiple sources.
-
 * <b>Score-based</b>: Calculate final relevance scores based on a weighted linear combination of individual search algorithm scores. Example - Weighted linear combination of semantic search & keyword-based search results.
-
 * <b>Relevance-based</b>: Discards the existing scores and calculates the relevance of each search result - query pair. Example - Cross Encoder models

 Even though there are many strategies for reranking search results, none works for all cases. Moreover, evaluating them itself is a challenge. Also, reranking can be dataset, application specific so it's hard to generalize.
--- a/docs/src/integrations/index.md
+++ b/docs/src/integrations/index.md
@@ -13,7 +13,7 @@ Get started using these examples and quick links.
 | Integrations | |
 |---|---:|
 | <h3> LlamaIndex </h3>LlamaIndex is a simple, flexible data framework for connecting custom data sources to large language models. Llama index integrates with LanceDB as the serverless VectorDB. <h3>[Lean More](https://gpt-index.readthedocs.io/en/latest/examples/vector_stores/LanceDBIndexDemo.html) </h3> |<img src="../assets/llama-index.jpg" alt="image" width="150" height="auto">|
-| <h3>Langchain</h3>Langchain allows building applications with LLMs through composability <h3>[Lean More](https://lancedb.github.io/lancedb/integrations/langchain/) | <img src="../assets/langchain.png" alt="image" width="150" height="auto">|
+| <h3>Langchain</h3>Langchain allows building applications with LLMs through composability <h3>[Lean More](https://python.langchain.com/docs/integrations/vectorstores/lancedb) | <img src="../assets/langchain.png" alt="image" width="150" height="auto">|
 | <h3>Langchain TS</h3> Javascript bindings for Langchain. It integrates with LanceDB's serverless vectordb allowing you to build powerful AI applications through composibility using only serverless functions. <h3>[Learn More]( https://js.langchain.com/docs/modules/data_connection/vectorstores/integrations/lancedb) | <img src="../assets/langchain.png" alt="image" width="150" height="auto">|
 | <h3>Voxel51</h3>  It is an open source toolkit that enables you to build better computer vision workflows by improving the quality of your datasets and delivering insights about your models.<h3>[Learn More](./voxel51.md) | <img src="../assets/voxel.gif" alt="image" width="150" height="auto">|
 | <h3>PromptTools</h3>  Offers a set of free, open-source tools for testing and experimenting with models, prompts, and configurations. The core idea is to enable developers to evaluate prompts using familiar interfaces like code and notebooks. You can use it to experiment with different configurations of LanceDB, and test how LanceDB integrates with the LLM of your choice.<h3>[Learn More](./prompttools.md) | <img src="../assets/prompttools.jpeg" alt="image" width="150" height="auto">|
--- a/docs/src/integrations/langchain.md
+++ b/docs/src/integrations/langchain.md
@@ -1,201 +0,0 @@
-# Langchain
-![Illustration](../assets/langchain.png)
-
-## Quick Start
-You can load your document data using langchain's loaders, for this example we are using `TextLoader` and `OpenAIEmbeddings` as the embedding model. Checkout Complete example here - [LangChain demo](../notebooks/langchain_example.ipynb)
-```python
-import os
-from langchain.document_loaders import TextLoader
-from langchain.vectorstores import LanceDB
-from langchain_openai import OpenAIEmbeddings
-from langchain_text_splitters import CharacterTextSplitter
-
-os.environ["OPENAI_API_KEY"] = "sk-..."
-
-loader = TextLoader("../../modules/state_of_the_union.txt") # Replace with your data path
-documents = loader.load()
-
-documents = CharacterTextSplitter().split_documents(documents)
-embeddings = OpenAIEmbeddings()
-
-docsearch = LanceDB.from_documents(documents, embeddings)
-query = "What did the president say about Ketanji Brown Jackson"
-docs = docsearch.similarity_search(query)
-print(docs[0].page_content)
-```
-
-## Documentation
-In the above example `LanceDB` vector store class object is created using `from_documents()` method  which is a `classmethod` and returns the initialized class object. 
-You can also use `LanceDB.from_texts(texts: List[str],embedding: Embeddings)` class method.  
-
-The exhaustive list of parameters for `LanceDB` vector store are :  
- `connection`: (Optional) `lancedb.db.LanceDBConnection` connection object to use.  If not provided, a new connection will be created.  
- `embedding`: Langchain embedding model.  
- `vector_key`: (Optional) Column name to use for vector's in the table. Defaults to `'vector'`.   
- `id_key`: (Optional) Column name to use for id's in the table. Defaults to `'id'`.  
- `text_key`: (Optional) Column name to use for text in the table. Defaults to `'text'`.  
- `table_name`: (Optional) Name of your table in the database. Defaults to `'vectorstore'`.  
- `api_key`: (Optional) API key to use for LanceDB cloud database. Defaults to `None`.  
- `region`: (Optional) Region to use for LanceDB cloud database. Only for LanceDB Cloud, defaults to `None`.  
- `mode`: (Optional) Mode to use for adding data to the table. Defaults to `'overwrite'`.  
- `reranker`: (Optional) The reranker to use for LanceDB.
- `relevance_score_fn`: (Optional[Callable[[float], float]]) Langchain relevance score function to be used. Defaults to `None`. 
-
-```python
-db_url = "db://lang_test" # url of db you created
-api_key = "xxxxx" # your API key
-region="us-east-1-dev"  # your selected region
-
-vector_store = LanceDB(
-    uri=db_url,
-    api_key=api_key, #(dont include for local API)
-    region=region, #(dont include for local API)
-    embedding=embeddings,
-    table_name='langchain_test' #Optional
-    )
-```
-
-### Methods 
-
-##### add_texts()
- `texts`: `Iterable` of strings to add to the vectorstore.
- `metadatas`: Optional `list[dict()]` of metadatas associated with the texts.
- `ids`: Optional `list` of ids to associate with the texts. 
- `kwargs`: `Any`
-
-This method adds texts and stores respective embeddings automatically.
-
-```python
-vector_store.add_texts(texts = ['test_123'], metadatas =[{'source' :'wiki'}]) 
-
-#Additionaly, to explore the table you can load it into a df or save it in a csv file:
-
-tbl = vector_store.get_table()
-print("tbl:", tbl)
-pd_df = tbl.to_pandas()
-pd_df.to_csv("docsearch.csv", index=False)
-
-# you can also create a new vector store object using an older connection object:
-vector_store = LanceDB(connection=tbl, embedding=embeddings)
-```
-##### create_index() 
- `col_name`: `Optional[str] = None`
- `vector_col`: `Optional[str] = None`
- `num_partitions`: `Optional[int] = 256`
- `num_sub_vectors`: `Optional[int] = 96`
- `index_cache_size`: `Optional[int] = None`
-
-This method creates an index for the vector store. For index creation make sure your table has enough data in it. An ANN index is ususally not needed for datasets ~100K vectors. For large-scale (>1M) or higher dimension vectors, it is beneficial to create an ANN index.
-
-```python
-# for creating vector index
-vector_store.create_index(vector_col='vector', metric = 'cosine')
-
-# for creating scalar index(for non-vector columns)
-vector_store.create_index(col_name='text')
-
-```
-
-##### similarity_search()
- `query`: `str`
- `k`: `Optional[int] = None`
- `filter`: `Optional[Dict[str, str]] = None`
- `fts`: `Optional[bool] = False`
- `name`: `Optional[str] = None`
- `kwargs`: `Any`
-
-Return documents most similar to the query without relevance scores
-
-```python
-docs = docsearch.similarity_search(query)
-print(docs[0].page_content)
-```
-
-##### similarity_search_by_vector()
- `embedding`: `List[float]`
- `k`: `Optional[int] = None`
- `filter`: `Optional[Dict[str, str]] = None`
- `name`: `Optional[str] = None`
- `kwargs`: `Any`
-
-Returns documents most similar to the query vector.
-
-```python
-docs = docsearch.similarity_search_by_vector(query)
-print(docs[0].page_content)
-```
-
-##### similarity_search_with_score()
- `query`: `str`
- `k`: `Optional[int] = None`
- `filter`: `Optional[Dict[str, str]] = None`
- `kwargs`: `Any`
-
-Returns documents most similar to the query string with relevance scores, gets called by base class's `similarity_search_with_relevance_scores` which selects relevance score based on our `_select_relevance_score_fn`.
-
-```python
-docs = docsearch.similarity_search_with_relevance_scores(query)
-print("relevance score - ", docs[0][1])
-print("text- ", docs[0][0].page_content[:1000])
-```
-
-##### similarity_search_by_vector_with_relevance_scores()
- `embedding`: `List[float]`
- `k`: `Optional[int] = None`
- `filter`: `Optional[Dict[str, str]] = None`
- `name`: `Optional[str] = None`
- `kwargs`: `Any`
-
-Return documents most similar to the query vector with relevance scores.
-Relevance score 
-
-```python
-docs = docsearch.similarity_search_by_vector_with_relevance_scores(query_embedding)
-print("relevance score - ", docs[0][1])
-print("text- ", docs[0][0].page_content[:1000])
-```
-
-##### max_marginal_relevance_search()
- `query`: `str`
- `k`: `Optional[int] = None`
- `fetch_k` : Number of Documents to fetch to pass to MMR algorithm, `Optional[int] = None`
- `lambda_mult`: Number between 0 and 1 that determines the degree
-                        of diversity among the results with 0 corresponding
-                        to maximum diversity and 1 to minimum diversity.
-                        Defaults to 0.5. `float = 0.5`
- `filter`: `Optional[Dict[str, str]] = None`
- `kwargs`: `Any`
-
-Returns docs selected using the maximal marginal relevance(MMR).
-Maximal marginal relevance optimizes for similarity to query AND diversity among selected documents.
-
-Similarly, `max_marginal_relevance_search_by_vector()` function returns docs most similar to the embedding passed to the function using MMR. instead of a string query you need to pass the embedding to be searched for. 
-
-```python
-result = docsearch.max_marginal_relevance_search(
-        query="text"
-    )
-result_texts = [doc.page_content for doc in result]
-print(result_texts)
-
-## search by vector :
-result = docsearch.max_marginal_relevance_search_by_vector(
-        embeddings.embed_query("text")
-    )
-result_texts = [doc.page_content for doc in result]
-print(result_texts)
-```
-
-##### add_images()
- `uris` : File path to the image. `List[str]`.
- `metadatas` : Optional list of metadatas. `(Optional[List[dict]], optional)`
- `ids` : Optional list of IDs. `(Optional[List[str]], optional)`
-
-Adds images by automatically creating their embeddings and adds them to the vectorstore.
-
-```python
-vec_store.add_images(uris=image_uris) 
-# here image_uris are local fs paths to the images.
-```
-
-
--- a/docs/src/integrations/llamaIndex.md
+++ b/docs/src/integrations/llamaIndex.md
@@ -1,142 +0,0 @@
-# Llama-Index
-![Illustration](../assets/llama-index.jpg)
-
-## Quick start
-You would need to install the integration via `pip install llama-index-vector-stores-lancedb` in order to use it. 
-You can run the below script to try it out :
-```python
-import logging
-import sys
-
-# Uncomment to see debug logs
-# logging.basicConfig(stream=sys.stdout, level=logging.DEBUG)
-# logging.getLogger().addHandler(logging.StreamHandler(stream=sys.stdout))
-
-from llama_index.core import SimpleDirectoryReader, Document, StorageContext
-from llama_index.core import VectorStoreIndex
-from llama_index.vector_stores.lancedb import LanceDBVectorStore
-import textwrap
-import openai
-
-openai.api_key = "sk-..."
-
-documents = SimpleDirectoryReader("./data/your-data-dir/").load_data()
-print("Document ID:", documents[0].doc_id, "Document Hash:", documents[0].hash)
-
-## For LanceDB cloud :
-# vector_store = LanceDBVectorStore( 
-#     uri="db://db_name", # your remote DB URI
-#     api_key="sk_..", # lancedb cloud api key
-#     region="your-region" # the region you configured
-#     ...
-# )
-
-vector_store = LanceDBVectorStore(
-    uri="./lancedb", mode="overwrite", query_type="vector"
-)
-storage_context = StorageContext.from_defaults(vector_store=vector_store)
-
-index = VectorStoreIndex.from_documents(
-    documents, storage_context=storage_context
-)
-lance_filter = "metadata.file_name = 'paul_graham_essay.txt' "
-retriever = index.as_retriever(vector_store_kwargs={"where": lance_filter})
-response = retriever.retrieve("What did the author do growing up?")
-```
-
-Checkout Complete example here - [LlamaIndex demo](../notebooks/LlamaIndex_example.ipynb)
-
-### Filtering
-For metadata filtering, you can use a Lance SQL-like string filter as demonstrated in the example above. Additionally, you can also filter using the `MetadataFilters` class from LlamaIndex:
-```python
-from llama_index.core.vector_stores import (
-    MetadataFilters,
-    FilterOperator,
-    FilterCondition,
-    MetadataFilter,
-)
-
-query_filters = MetadataFilters(
-    filters=[
-        MetadataFilter(
-            key="creation_date", operator=FilterOperator.EQ, value="2024-05-23"
-        ),
-        MetadataFilter(
-            key="file_size", value=75040, operator=FilterOperator.GT
-        ),
-    ],
-    condition=FilterCondition.AND,
-)
-```
-
-### Hybrid Search
-For complete documentation, refer [here](https://lancedb.github.io/lancedb/hybrid_search/hybrid_search/). This example uses the `colbert` reranker. Make sure to install necessary dependencies for the reranker you choose.
-```python
-from lancedb.rerankers import ColbertReranker
-
-reranker = ColbertReranker()
-vector_store._add_reranker(reranker)
-
-query_engine = index.as_query_engine(
-    filters=query_filters,
-    vector_store_kwargs={
-        "query_type": "hybrid",
-    }
-)
-
-response = query_engine.query("How much did Viaweb charge per month?")
-```
-
-In the above snippet, you can change/specify query_type again when creating the engine/retriever.
-
-## API reference
-The exhaustive list of parameters for `LanceDBVectorStore` vector store are :  
- `connection`: Optional, `lancedb.db.LanceDBConnection` connection object to use. If not provided, a new connection will be created.
- `uri`: Optional[str], the uri of your database. Defaults to `"/tmp/lancedb"`.
- `table_name` : Optional[str], Name of your table in the database. Defaults to `"vectors"`.
- `table`: Optional[Any], `lancedb.db.LanceTable` object to be passed. Defaults to `None`. 
- `vector_column_name`: Optional[Any], Column name to use for vector's in the table. Defaults to `'vector'`.   
- `doc_id_key`: Optional[str], Column name to use for document id's in the table. Defaults to `'doc_id'`.  
- `text_key`: Optional[str], Column name to use for text in the table. Defaults to `'text'`.  
- `api_key`: Optional[str], API key to use for LanceDB cloud database. Defaults to `None`.  
- `region`: Optional[str], Region to use for LanceDB cloud database. Only for LanceDB Cloud, defaults to `None`.  
- `nprobes` : Optional[int], Set the number of probes to use. Only applicable if ANN index is created on the table else its ignored. Defaults to `20`.
- `refine_factor` : Optional[int], Refine the results by reading extra elements and re-ranking them in memory. Defaults to `None`.
- `reranker`: Optional[Any], The reranker to use for LanceDB.
-        Defaults to `None`.
- `overfetch_factor`: Optional[int], The factor by which to fetch more results.
-        Defaults to `1`.
- `mode`: Optional[str], The mode to use for LanceDB.
-            Defaults to `"overwrite"`.
- `query_type`:Optional[str], The type of query to use for LanceDB.
-            Defaults to `"vector"`.
-
-
-### Methods 
-
- __from_table(cls, table: lancedb.db.LanceTable) -> `LanceDBVectorStore`__ : (class method) Creates instance from lancedb table. 
-
- **_add_reranker(self, reranker: lancedb.rerankers.Reranker) -> `None`** : Add a reranker to an existing vector store. 
-    - Usage :
-        ```python
-        from lancedb.rerankers import ColbertReranker
-        reranker = ColbertReranker()
-        vector_store._add_reranker(reranker)
-        ```               
- **_table_exists(self, tbl_name: `Optional[str]` = `None`) -> `bool`** : Returns `True` if `tbl_name` exists in database.
- __create_index(  
-  self, scalar: `Optional[bool]` = False, col_name: `Optional[str]` = None, num_partitions: `Optional[int]` = 256, num_sub_vectors: `Optional[int]` = 96, index_cache_size: `Optional[int]` = None, metric: `Optional[str]` = "L2",  
-) -> `None`__ : Creates a scalar(for non-vector cols) or a vector index on a table.
-        Make sure your vector column has enough data before creating an index on it.
-
- __add(self, nodes: `List[BaseNode]`, **add_kwargs: `Any`, ) -> `List[str]`__ :
-adds Nodes to the table
-
- **delete(self, ref_doc_id: `str`) -> `None`**: Delete nodes using with node_ids.
- **delete_nodes(self, node_ids: `List[str]`) -> `None`** : Delete nodes using with node_ids.
- __query(
-        self,
-        query: `VectorStoreQuery`,
-        **kwargs: `Any`,
-    ) -> `VectorStoreQueryResult`__:
-        Query index(`VectorStoreIndex`) for top k most similar nodes. Accepts llamaIndex `VectorStoreQuery` object.
--- a/docs/src/notebooks/LlamaIndex_example.ipynb
+++ b/docs/src/notebooks/LlamaIndex_example.ipynb
@@ -1,538 +0,0 @@
-{
- "cells": [
-  {
-   "attachments": {},
-   "cell_type": "markdown",
-   "id": "2db56c9b",
-   "metadata": {},
-   "source": [
-    "<a href=\"https://colab.research.google.com/github/run-llama/llama_index/blob/main/docs/docs/examples/vector_stores/LanceDBIndexDemo.ipynb\" target=\"_parent\"><img src=\"https://colab.research.google.com/assets/colab-badge.svg\" alt=\"Open In Colab\"/></a>"
-   ]
-  },
-  {
-   "attachments": {},
-   "cell_type": "markdown",
-   "id": "db0855d0",
-   "metadata": {},
-   "source": [
-    "# LanceDB Vector Store\n",
-    "In this notebook we are going to show how to use [LanceDB](https://www.lancedb.com) to perform vector searches in LlamaIndex"
-   ]
-  },
-  {
-   "attachments": {},
-   "cell_type": "markdown",
-   "id": "f44170b2",
-   "metadata": {},
-   "source": [
-    "If you're opening this Notebook on colab, you will probably need to install LlamaIndex 🦙."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "6c84199c",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "%pip install llama-index llama-index-vector-stores-lancedb"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "1a90ce34",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "%pip install lancedb==0.6.13 #Only required if the above cell installs an older version of lancedb (pypi package may not be released yet)"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "39c62671",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "# Refresh vector store URI if restarting or re-using the same notebook\n",
-    "! rm -rf ./lancedb"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "59b54276",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "import logging\n",
-    "import sys\n",
-    "\n",
-    "# Uncomment to see debug logs\n",
-    "# logging.basicConfig(stream=sys.stdout, level=logging.DEBUG)\n",
-    "# logging.getLogger().addHandler(logging.StreamHandler(stream=sys.stdout))\n",
-    "\n",
-    "\n",
-    "from llama_index.core import SimpleDirectoryReader, Document, StorageContext\n",
-    "from llama_index.core import VectorStoreIndex\n",
-    "from llama_index.vector_stores.lancedb import LanceDBVectorStore\n",
-    "import textwrap"
-   ]
-  },
-  {
-   "attachments": {},
-   "cell_type": "markdown",
-   "id": "26c71b6d",
-   "metadata": {},
-   "source": [
-    "### Setup OpenAI\n",
-    "The first step is to configure the openai key. It will be used to created embeddings for the documents loaded into the index"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "67b86621",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "import openai\n",
-    "\n",
-    "openai.api_key = \"sk-\""
-   ]
-  },
-  {
-   "attachments": {},
-   "cell_type": "markdown",
-   "id": "073f0a68",
-   "metadata": {},
-   "source": [
-    "Download Data"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "eef1b911",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "--2024-06-11 16:42:37--  https://raw.githubusercontent.com/run-llama/llama_index/main/docs/docs/examples/data/paul_graham/paul_graham_essay.txt\n",
-      "Resolving raw.githubusercontent.com (raw.githubusercontent.com)... 185.199.109.133, 185.199.110.133, 185.199.108.133, ...\n",
-      "Connecting to raw.githubusercontent.com (raw.githubusercontent.com)|185.199.109.133|:443... connected.\n",
-      "HTTP request sent, awaiting response... 200 OK\n",
-      "Length: 75042 (73K) [text/plain]\n",
-      "Saving to: ‘data/paul_graham/paul_graham_essay.txt’\n",
-      "\n",
-      "data/paul_graham/pa 100%[===================>]  73.28K  --.-KB/s    in 0.02s   \n",
-      "\n",
-      "2024-06-11 16:42:37 (3.97 MB/s) - ‘data/paul_graham/paul_graham_essay.txt’ saved [75042/75042]\n",
-      "\n"
-     ]
-    }
-   ],
-   "source": [
-    "!mkdir -p 'data/paul_graham/'\n",
-    "!wget 'https://raw.githubusercontent.com/run-llama/llama_index/main/docs/docs/examples/data/paul_graham/paul_graham_essay.txt' -O 'data/paul_graham/paul_graham_essay.txt'"
-   ]
-  },
-  {
-   "attachments": {},
-   "cell_type": "markdown",
-   "id": "f7010b1d-d1bb-4f08-9309-a328bb4ea396",
-   "metadata": {},
-   "source": [
-    "### Loading documents\n",
-    "Load the documents stored in the `data/paul_graham/` using the SimpleDirectoryReader"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "c154dd4b",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "Document ID: cac1ba78-5007-4cf8-89ba-280264790115 Document Hash: fe2d4d3ef3a860780f6c2599808caa587c8be6516fe0ba4ca53cf117044ba953\n"
-     ]
-    }
-   ],
-   "source": [
-    "documents = SimpleDirectoryReader(\"./data/paul_graham/\").load_data()\n",
-    "print(\"Document ID:\", documents[0].doc_id, \"Document Hash:\", documents[0].hash)"
-   ]
-  },
-  {
-   "attachments": {},
-   "cell_type": "markdown",
-   "id": "c0232fd1",
-   "metadata": {},
-   "source": [
-    "### Create the index\n",
-    "Here we create an index backed by LanceDB using the documents loaded previously. LanceDBVectorStore takes a few arguments.\n",
-    "- uri (str, required): Location where LanceDB will store its files.\n",
-    "- table_name (str, optional): The table name where the embeddings will be stored. Defaults to \"vectors\".\n",
-    "- nprobes (int, optional): The number of probes used. A higher number makes search more accurate but also slower. Defaults to 20.\n",
-    "- refine_factor: (int, optional): Refine the results by reading extra elements and re-ranking them in memory. Defaults to None\n",
-    "\n",
-    "- More details can be found at [LanceDB docs](https://lancedb.github.io/lancedb/ann_indexes)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "1f2e20ef",
-   "metadata": {},
-   "source": [
-    "##### For LanceDB cloud :\n",
-    "```python\n",
-    "vector_store = LanceDBVectorStore( \n",
-    "    uri=\"db://db_name\", # your remote DB URI\n",
-    "    api_key=\"sk_..\", # lancedb cloud api key\n",
-    "    region=\"your-region\" # the region you configured\n",
-    "    ...\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "8731da62",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "vector_store = LanceDBVectorStore(\n",
-    "    uri=\"./lancedb\", mode=\"overwrite\", query_type=\"hybrid\"\n",
-    ")\n",
-    "storage_context = StorageContext.from_defaults(vector_store=vector_store)\n",
-    "\n",
-    "index = VectorStoreIndex.from_documents(\n",
-    "    documents, storage_context=storage_context\n",
-    ")"
-   ]
-  },
-  {
-   "attachments": {},
-   "cell_type": "markdown",
-   "id": "8ee4473a-094f-4d0a-a825-e1213db07240",
-   "metadata": {},
-   "source": [
-    "### Query the index\n",
-    "We can now ask questions using our index. We can use filtering via `MetadataFilters` or use native lance `where` clause."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "5eb6419b",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "from llama_index.core.vector_stores import (\n",
-    "    MetadataFilters,\n",
-    "    FilterOperator,\n",
-    "    FilterCondition,\n",
-    "    MetadataFilter,\n",
-    ")\n",
-    "\n",
-    "from datetime import datetime\n",
-    "\n",
-    "\n",
-    "query_filters = MetadataFilters(\n",
-    "    filters=[\n",
-    "        MetadataFilter(\n",
-    "            key=\"creation_date\",\n",
-    "            operator=FilterOperator.EQ,\n",
-    "            value=datetime.now().strftime(\"%Y-%m-%d\"),\n",
-    "        ),\n",
-    "        MetadataFilter(\n",
-    "            key=\"file_size\", value=75040, operator=FilterOperator.GT\n",
-    "        ),\n",
-    "    ],\n",
-    "    condition=FilterCondition.AND,\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "ee201930",
-   "metadata": {},
-   "source": [
-    "### Hybrid Search\n",
-    "\n",
-    "LanceDB offers hybrid search with reranking capabilities. For complete documentation, refer [here](https://lancedb.github.io/lancedb/hybrid_search/hybrid_search/).\n",
-    "\n",
-    "This example uses the `colbert` reranker. The following cell installs the necessary dependencies for `colbert`. If you choose a different reranker, make sure to adjust the dependencies accordingly."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "e12d1454",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "! pip install -U torch transformers tantivy@git+https://github.com/quickwit-oss/tantivy-py#164adc87e1a033117001cf70e38c82a53014d985"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "c742cb07",
-   "metadata": {},
-   "source": [
-    "if you want to add a reranker at vector store initialization, you can pass it in the arguments like below :\n",
-    "```\n",
-    "from lancedb.rerankers import ColbertReranker\n",
-    "reranker = ColbertReranker()\n",
-    "vector_store = LanceDBVectorStore(uri=\"./lancedb\", reranker=reranker, mode=\"overwrite\")\n",
-    "```"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "27ea047b",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "import lancedb"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "8414517f",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "from lancedb.rerankers import ColbertReranker\n",
-    "\n",
-    "reranker = ColbertReranker()\n",
-    "vector_store._add_reranker(reranker)\n",
-    "\n",
-    "query_engine = index.as_query_engine(\n",
-    "    filters=query_filters,\n",
-    "    # vector_store_kwargs={\n",
-    "    #     \"query_type\": \"fts\",\n",
-    "    # },\n",
-    ")\n",
-    "\n",
-    "response = query_engine.query(\"How much did Viaweb charge per month?\")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "dc6ccb7a",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "Viaweb charged $100 a month for a small store and $300 a month for a big one.\n",
-      "metadata - {'65ed5f07-5b8a-4143-a939-e8764884828e': {'file_path': '/Users/raghavdixit/Desktop/open_source/llama_index_lance/docs/docs/examples/vector_stores/data/paul_graham/paul_graham_essay.txt', 'file_name': 'paul_graham_essay.txt', 'file_type': 'text/plain', 'file_size': 75042, 'creation_date': '2024-06-11', 'last_modified_date': '2024-06-11'}, 'be231827-20b8-4988-ac75-94fa79b3c22e': {'file_path': '/Users/raghavdixit/Desktop/open_source/llama_index_lance/docs/docs/examples/vector_stores/data/paul_graham/paul_graham_essay.txt', 'file_name': 'paul_graham_essay.txt', 'file_type': 'text/plain', 'file_size': 75042, 'creation_date': '2024-06-11', 'last_modified_date': '2024-06-11'}}\n"
-     ]
-    }
-   ],
-   "source": [
-    "print(response)\n",
-    "print(\"metadata -\", response.metadata)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "0c1c6c73",
-   "metadata": {},
-   "source": [
-    "##### lance filters(SQL like) directly via the `where` clause :"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "0a2bcc07",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "lance_filter = \"metadata.file_name = 'paul_graham_essay.txt' \"\n",
-    "retriever = index.as_retriever(vector_store_kwargs={\"where\": lance_filter})\n",
-    "response = retriever.retrieve(\"What did the author do growing up?\")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "7ac47cf9",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "What I Worked On\n",
-      "\n",
-      "February 2021\n",
-      "\n",
-      "Before college the two main things I worked on, outside of school, were writing and programming. I didn't write essays. I wrote what beginning writers were supposed to write then, and probably still are: short stories. My stories were awful. They had hardly any plot, just characters with strong feelings, which I imagined made them deep.\n",
-      "\n",
-      "The first programs I tried writing were on the IBM 1401 that our school district used for what was then called \"data processing.\" This was in 9th grade, so I was 13 or 14. The school district's 1401 happened to be in the basement of our junior high school, and my friend Rich Draves and I got permission to use it. It was like a mini Bond villain's lair down there, with all these alien-looking machines — CPU, disk drives, printer, card reader — sitting up on a raised floor under bright fluorescent lights.\n",
-      "\n",
-      "The language we used was an early version of Fortran. You had to type programs on punch cards, then stack them in the card reader and press a button to load the program into memory and run it. The result would ordinarily be to print something on the spectacularly loud printer.\n",
-      "\n",
-      "I was puzzled by the 1401. I couldn't figure out what to do with it. And in retrospect there's not much I could have done with it. The only form of input to programs was data stored on punched cards, and I didn't have any data stored on punched cards. The only other option was to do things that didn't rely on any input, like calculate approximations of pi, but I didn't know enough math to do anything interesting of that type. So I'm not surprised I can't remember any programs I wrote, because they can't have done much. My clearest memory is of the moment I learned it was possible for programs not to terminate, when one of mine didn't. On a machine without time-sharing, this was a social as well as a technical error, as the data center manager's expression made clear.\n",
-      "\n",
-      "With microcomputers, everything changed. Now you could have a computer sitting right in front of you, on a desk, that could respond to your keystrokes as it was running instead of just churning through a stack of punch cards and then stopping. [1]\n",
-      "\n",
-      "The first of my friends to get a microcomputer built it himself. It was sold as a kit by Heathkit. I remember vividly how impressed and envious I felt watching him sitting in front of it, typing programs right into the computer.\n",
-      "\n",
-      "Computers were expensive in those days and it took me years of nagging before I convinced my father to buy one, a TRS-80, in about 1980. The gold standard then was the Apple II, but a TRS-80 was good enough. This was when I really started programming. I wrote simple games, a program to predict how high my model rockets would fly, and a word processor that my father used to write at least one book. There was only room in memory for about 2 pages of text, so he'd write 2 pages at a time and then print them out, but it was a lot better than a typewriter.\n",
-      "\n",
-      "Though I liked programming, I didn't plan to study it in college. In college I was going to study philosophy, which sounded much more powerful. It seemed, to my naive high school self, to be the study of the ultimate truths, compared to which the things studied in other fields would be mere domain knowledge. What I discovered when I got to college was that the other fields took up so much of the space of ideas that there wasn't much left for these supposed ultimate truths. All that seemed left for philosophy were edge cases that people in other fields felt could safely be ignored.\n",
-      "\n",
-      "I couldn't have put this into words when I was 18. All I knew at the time was that I kept taking philosophy courses and they kept being boring. So I decided to switch to AI.\n",
-      "\n",
-      "AI was in the air in the mid 1980s, but there were two things especially that made me want to work on it: a novel by Heinlein called The Moon is a Harsh Mistress, which featured an intelligent computer called Mike, and a PBS documentary that showed Terry Winograd using SHRDLU. I haven't tried rereading The Moon is a Harsh Mistress, so I don't know how well it has aged, but when I read it I was drawn entirely into its world.\n",
-      "metadata - {'file_path': '/Users/raghavdixit/Desktop/open_source/llama_index_lance/docs/docs/examples/vector_stores/data/paul_graham/paul_graham_essay.txt', 'file_name': 'paul_graham_essay.txt', 'file_type': 'text/plain', 'file_size': 75042, 'creation_date': '2024-06-11', 'last_modified_date': '2024-06-11'}\n"
-     ]
-    }
-   ],
-   "source": [
-    "print(response[0].get_content())\n",
-    "print(\"metadata -\", response[0].metadata)"
-   ]
-  },
-  {
-   "attachments": {},
-   "cell_type": "markdown",
-   "id": "6afc84ac",
-   "metadata": {},
-   "source": [
-    "### Appending data\n",
-    "You can also add data to an existing index"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "759a532e",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "nodes = [node.node for node in response]"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "069fc099",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "del index\n",
-    "\n",
-    "index = VectorStoreIndex.from_documents(\n",
-    "    [Document(text=\"The sky is purple in Portland, Maine\")],\n",
-    "    uri=\"/tmp/new_dataset\",\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "a64ed441",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "index.insert_nodes(nodes)"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "b5cffcfe",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "Portland, Maine\n"
-     ]
-    }
-   ],
-   "source": [
-    "query_engine = index.as_query_engine()\n",
-    "response = query_engine.query(\"Where is the sky purple?\")\n",
-    "print(textwrap.fill(str(response), 100))"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "ec548a02",
-   "metadata": {},
-   "source": [
-    "You can also create an index from an existing table"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "dc99404d",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "del index\n",
-    "\n",
-    "vec_store = LanceDBVectorStore.from_table(vector_store._table)\n",
-    "index = VectorStoreIndex.from_vector_store(vec_store)"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "7b2e8cca",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "The author started Viaweb and Aspra.\n"
-     ]
-    }
-   ],
-   "source": [
-    "query_engine = index.as_query_engine()\n",
-    "response = query_engine.query(\"What companies did the author start?\")\n",
-    "print(textwrap.fill(str(response), 100))"
-   ]
-  }
- ],
- "metadata": {
-  "kernelspec": {
-   "display_name": "Python 3 (ipykernel)",
-   "language": "python",
-   "name": "python3"
-  },
-  "language_info": {
-   "codemirror_mode": {
-    "name": "ipython",
-    "version": 3
-   },
-   "file_extension": ".py",
-   "mimetype": "text/x-python",
-   "name": "python",
-   "nbconvert_exporter": "python",
-   "pygments_lexer": "ipython3"
-  }
- },
- "nbformat": 4,
- "nbformat_minor": 5
-}
--- a/docs/src/notebooks/lancedb_reranking.ipynb
+++ b/docs/src/notebooks/lancedb_reranking.ipynb
--- a/docs/src/notebooks/langchain_example.ipynb
+++ b/docs/src/notebooks/langchain_example.ipynb
@@ -1,566 +0,0 @@
-{
- "cells": [
-  {
-   "cell_type": "markdown",
-   "id": "683953b3",
-   "metadata": {},
-   "source": [
-    "# LanceDB\n",
-    "\n",
-    ">[LanceDB](https://lancedb.com/) is an open-source database for vector-search built with persistent storage, which greatly simplifies retrevial, filtering and management of embeddings. Fully open source.\n",
-    "\n",
-    "This notebook shows how to use functionality related to the `LanceDB` vector database based on the Lance data format."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "b1051ba9",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "! pip install tantivy"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "88ac92c0",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "! pip install -U langchain-openai langchain-community"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "5a1c84d6-a10f-428c-95cd-46d3a1702e07",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "! pip install lancedb"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "99134dd1-b91e-486f-8d90-534248e43b9d",
-   "metadata": {},
-   "source": [
-    "We want to use OpenAIEmbeddings so we have to get the OpenAI API Key. "
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 1,
-   "id": "a0361f5c-e6f4-45f4-b829-11680cf03cec",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [],
-   "source": [
-    "import getpass\n",
-    "import os\n",
-    "\n",
-    "os.environ[\"OPENAI_API_KEY\"] = getpass.getpass(\"OpenAI API Key:\")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 2,
-   "id": "d114ed78",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "! rm -rf /tmp/lancedb"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 3,
-   "id": "a3c3999a",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "from langchain_community.document_loaders import TextLoader\n",
-    "from langchain_community.vectorstores import LanceDB\n",
-    "from langchain_openai import OpenAIEmbeddings\n",
-    "from langchain_text_splitters import CharacterTextSplitter\n",
-    "\n",
-    "loader = TextLoader(\"../../how_to/state_of_the_union.txt\")\n",
-    "documents = loader.load()\n",
-    "\n",
-    "documents = CharacterTextSplitter().split_documents(documents)\n",
-    "embeddings = OpenAIEmbeddings()"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "e9517bb0",
-   "metadata": {},
-   "source": [
-    "##### For LanceDB cloud, you can invoke the vector store as follows :\n",
-    "\n",
-    "\n",
-    "```python\n",
-    "db_url = \"db://lang_test\" # url of db you created\n",
-    "api_key = \"xxxxx\" # your API key\n",
-    "region=\"us-east-1-dev\"  # your selected region\n",
-    "\n",
-    "vector_store = LanceDB(\n",
-    "    uri=db_url,\n",
-    "    api_key=api_key,\n",
-    "    region=region,\n",
-    "    embedding=embeddings,\n",
-    "    table_name='langchain_test'\n",
-    "    )\n",
-    "```\n",
-    "\n",
-    "You can also add `region`, `api_key`, `uri` to `from_documents()` classmethod\n"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 4,
-   "id": "6e104aee",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "from lancedb.rerankers import LinearCombinationReranker\n",
-    "\n",
-    "reranker = LinearCombinationReranker(weight=0.3)\n",
-    "\n",
-    "docsearch = LanceDB.from_documents(documents, embeddings, reranker=reranker)\n",
-    "query = \"What did the president say about Ketanji Brown Jackson\""
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 31,
-   "id": "259c7988",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "relevance score -  0.7066475030191711\n",
-      "text-  They were responding to a 9-1-1 call when a man shot and killed them with a stolen gun. \n",
-      "\n",
-      "Officer Mora was 27 years old. \n",
-      "\n",
-      "Officer Rivera was 22. \n",
-      "\n",
-      "Both Dominican Americans who’d grown up on the same streets they later chose to patrol as police officers. \n",
-      "\n",
-      "I spoke with their families and told them that we are forever in debt for their sacrifice, and we will carry on their mission to restore the trust and safety every community deserves. \n",
-      "\n",
-      "I’ve worked on these issues a long time. \n",
-      "\n",
-      "I know what works: Investing in crime prevention and community police officers who’ll walk the beat, who’ll know the neighborhood, and who can restore trust and safety. \n",
-      "\n",
-      "So let’s not abandon our streets. Or choose between safety and equal justice. \n",
-      "\n",
-      "Let’s come together to protect our communities, restore trust, and hold law enforcement accountable. \n",
-      "\n",
-      "That’s why the Justice Department required body cameras, banned chokeholds, and restricted no-knock warrants for its officers. \n",
-      "\n",
-      "That’s why the American Rescue \n"
-     ]
-    }
-   ],
-   "source": [
-    "docs = docsearch.similarity_search_with_relevance_scores(query)\n",
-    "print(\"relevance score - \", docs[0][1])\n",
-    "print(\"text- \", docs[0][0].page_content[:1000])"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 33,
-   "id": "9fa29dae",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "distance -  0.30000001192092896\n",
-      "text-  My administration is providing assistance with job training and housing, and now helping lower-income veterans get VA care debt-free.  \n",
-      "\n",
-      "Our troops in Iraq and Afghanistan faced many dangers. \n",
-      "\n",
-      "One was stationed at bases and breathing in toxic smoke from “burn pits” that incinerated wastes of war—medical and hazard material, jet fuel, and more. \n",
-      "\n",
-      "When they came home, many of the world’s fittest and best trained warriors were never the same. \n",
-      "\n",
-      "Headaches. Numbness. Dizziness. \n",
-      "\n",
-      "A cancer that would put them in a flag-draped coffin. \n",
-      "\n",
-      "I know. \n",
-      "\n",
-      "One of those soldiers was my son Major Beau Biden. \n",
-      "\n",
-      "We don’t know for sure if a burn pit was the cause of his brain cancer, or the diseases of so many of our troops. \n",
-      "\n",
-      "But I’m committed to finding out everything we can. \n",
-      "\n",
-      "Committed to military families like Danielle Robinson from Ohio. \n",
-      "\n",
-      "The widow of Sergeant First Class Heath Robinson.  \n",
-      "\n",
-      "He was born a soldier. Army National Guard. Combat medic in Kosovo and Iraq. \n",
-      "\n",
-      "Stationed near Baghdad, just ya\n"
-     ]
-    }
-   ],
-   "source": [
-    "docs = docsearch.similarity_search_with_score(query=\"Headaches\", query_type=\"hybrid\")\n",
-    "print(\"distance - \", docs[0][1])\n",
-    "print(\"text- \", docs[0][0].page_content[:1000])"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 8,
-   "id": "e70ad201",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "reranker :  <lancedb.rerankers.linear_combination.LinearCombinationReranker object at 0x107ef1130>\n"
-     ]
-    }
-   ],
-   "source": [
-    "print(\"reranker : \", docsearch._reranker)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "f5e1cdfd",
-   "metadata": {},
-   "source": [
-    "Additionaly, to explore the table you can load it into a df or save it in a csv file: \n",
-    "```python\n",
-    "tbl = docsearch.get_table()\n",
-    "print(\"tbl:\", tbl)\n",
-    "pd_df = tbl.to_pandas()\n",
-    "# pd_df.to_csv(\"docsearch.csv\", index=False)\n",
-    "\n",
-    "# you can also create a new vector store object using an older connection object:\n",
-    "vector_store = LanceDB(connection=tbl, embedding=embeddings)\n",
-    "```"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 15,
-   "id": "9c608226",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "metadata : {'source': '../../how_to/state_of_the_union.txt'}\n",
-      "\n",
-      "SQL filtering :\n",
-      "\n",
-      "They were responding to a 9-1-1 call when a man shot and killed them with a stolen gun. \n",
-      "\n",
-      "Officer Mora was 27 years old. \n",
-      "\n",
-      "Officer Rivera was 22. \n",
-      "\n",
-      "Both Dominican Americans who’d grown up on the same streets they later chose to patrol as police officers. \n",
-      "\n",
-      "I spoke with their families and told them that we are forever in debt for their sacrifice, and we will carry on their mission to restore the trust and safety every community deserves. \n",
-      "\n",
-      "I’ve worked on these issues a long time. \n",
-      "\n",
-      "I know what works: Investing in crime prevention and community police officers who’ll walk the beat, who’ll know the neighborhood, and who can restore trust and safety. \n",
-      "\n",
-      "So let’s not abandon our streets. Or choose between safety and equal justice. \n",
-      "\n",
-      "Let’s come together to protect our communities, restore trust, and hold law enforcement accountable. \n",
-      "\n",
-      "That’s why the Justice Department required body cameras, banned chokeholds, and restricted no-knock warrants for its officers. \n",
-      "\n",
-      "That’s why the American Rescue Plan provided $350 Billion that cities, states, and counties can use to hire more police and invest in proven strategies like community violence interruption—trusted messengers breaking the cycle of violence and trauma and giving young people hope.  \n",
-      "\n",
-      "We should all agree: The answer is not to Defund the police. The answer is to FUND the police with the resources and training they need to protect our communities. \n",
-      "\n",
-      "I ask Democrats and Republicans alike: Pass my budget and keep our neighborhoods safe.  \n",
-      "\n",
-      "And I will keep doing everything in my power to crack down on gun trafficking and ghost guns you can buy online and make at home—they have no serial numbers and can’t be traced. \n",
-      "\n",
-      "And I ask Congress to pass proven measures to reduce gun violence. Pass universal background checks. Why should anyone on a terrorist list be able to purchase a weapon? \n",
-      "\n",
-      "Ban assault weapons and high-capacity magazines. \n",
-      "\n",
-      "Repeal the liability shield that makes gun manufacturers the only industry in America that can’t be sued. \n",
-      "\n",
-      "These laws don’t infringe on the Second Amendment. They save lives. \n",
-      "\n",
-      "The most fundamental right in America is the right to vote – and to have it counted. And it’s under assault. \n",
-      "\n",
-      "In state after state, new laws have been passed, not only to suppress the vote, but to subvert entire elections. \n",
-      "\n",
-      "We cannot let this happen. \n",
-      "\n",
-      "Tonight. I call on the Senate to: Pass the Freedom to Vote Act. Pass the John Lewis Voting Rights Act. And while you’re at it, pass the Disclose Act so Americans can know who is funding our elections. \n",
-      "\n",
-      "Tonight, I’d like to honor someone who has dedicated his life to serve this country: Justice Stephen Breyer—an Army veteran, Constitutional scholar, and retiring Justice of the United States Supreme Court. Justice Breyer, thank you for your service. \n",
-      "\n",
-      "One of the most serious constitutional responsibilities a President has is nominating someone to serve on the United States Supreme Court. \n",
-      "\n",
-      "And I did that 4 days ago, when I nominated Circuit Court of Appeals Judge Ketanji Brown Jackson. One of our nation’s top legal minds, who will continue Justice Breyer’s legacy of excellence. \n",
-      "\n",
-      "A former top litigator in private practice. A former federal public defender. And from a family of public school educators and police officers. A consensus builder. Since she’s been nominated, she’s received a broad range of support—from the Fraternal Order of Police to former judges appointed by Democrats and Republicans. \n",
-      "\n",
-      "And if we are to advance liberty and justice, we need to secure the Border and fix the immigration system. \n",
-      "\n",
-      "We can do both. At our border, we’ve installed new technology like cutting-edge scanners to better detect drug smuggling.  \n",
-      "\n",
-      "We’ve set up joint patrols with Mexico and Guatemala to catch more human traffickers.  \n",
-      "\n",
-      "We’re putting in place dedicated immigration judges so families fleeing persecution and violence can have their cases heard faster.\n"
-     ]
-    }
-   ],
-   "source": [
-    "docs = docsearch.similarity_search(\n",
-    "    query=query, filter={\"metadata.source\": \"../../how_to/state_of_the_union.txt\"}\n",
-    ")\n",
-    "\n",
-    "print(\"metadata :\", docs[0].metadata)\n",
-    "\n",
-    "# or you can directly supply SQL string filters :\n",
-    "\n",
-    "print(\"\\nSQL filtering :\\n\")\n",
-    "docs = docsearch.similarity_search(query=query, filter=\"text LIKE '%Officer Rivera%'\")\n",
-    "print(docs[0].page_content)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "9a173c94",
-   "metadata": {},
-   "source": [
-    "## Adding images "
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "05f669d7",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "! pip install -U langchain-experimental"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "3ed69810",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "! pip install open_clip_torch torch"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 16,
-   "id": "2cacb5ee",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "! rm -rf '/tmp/multimmodal_lance'"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 17,
-   "id": "b3456e2c",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "from langchain_experimental.open_clip import OpenCLIPEmbeddings"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 18,
-   "id": "3848eba2",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "import os\n",
-    "\n",
-    "import requests\n",
-    "\n",
-    "# List of image URLs to download\n",
-    "image_urls = [\n",
-    "    \"https://github.com/raghavdixit99/assets/assets/34462078/abf47cc4-d979-4aaa-83be-53a2115bf318\",\n",
-    "    \"https://github.com/raghavdixit99/assets/assets/34462078/93be928e-522b-4e37-889d-d4efd54b2112\",\n",
-    "]\n",
-    "\n",
-    "texts = [\"bird\", \"dragon\"]\n",
-    "\n",
-    "# Directory to save images\n",
-    "dir_name = \"./photos/\"\n",
-    "\n",
-    "# Create directory if it doesn't exist\n",
-    "os.makedirs(dir_name, exist_ok=True)\n",
-    "\n",
-    "image_uris = []\n",
-    "# Download and save each image\n",
-    "for i, url in enumerate(image_urls, start=1):\n",
-    "    response = requests.get(url)\n",
-    "    path = os.path.join(dir_name, f\"image{i}.jpg\")\n",
-    "    image_uris.append(path)\n",
-    "    with open(path, \"wb\") as f:\n",
-    "        f.write(response.content)"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 21,
-   "id": "3d62c2a0",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "from langchain_community.vectorstores import LanceDB\n",
-    "\n",
-    "vec_store = LanceDB(\n",
-    "    table_name=\"multimodal_test\",\n",
-    "    embedding=OpenCLIPEmbeddings(),\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 22,
-   "id": "ebbb4881",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "['b673620b-01f0-42ca-a92e-d033bb92c0a6',\n",
-       " '99c3a5b0-b577-417a-8177-92f4a655dbfb']"
-      ]
-     },
-     "execution_count": 22,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "vec_store.add_images(uris=image_uris)"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 23,
-   "id": "3c29dea3",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "['f7adde5d-a4a3-402b-9e73-088b230722c3',\n",
-       " 'cbed59da-0aec-4bff-8820-9e59d81a2140']"
-      ]
-     },
-     "execution_count": 23,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "vec_store.add_texts(texts)"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 24,
-   "id": "8b2f25ce",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "img_embed = vec_store._embedding.embed_query(\"bird\")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 25,
-   "id": "87a24079",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "Document(page_content='bird', metadata={'id': 'f7adde5d-a4a3-402b-9e73-088b230722c3'})"
-      ]
-     },
-     "execution_count": 25,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "vec_store.similarity_search_by_vector(img_embed)[0]"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 26,
-   "id": "78557867",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "LanceTable(connection=LanceDBConnection(/tmp/lancedb), name=\"multimodal_test\")"
-      ]
-     },
-     "execution_count": 26,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "vec_store._table"
-   ]
-  }
- ],
- "metadata": {
-  "kernelspec": {
-   "display_name": "Python 3 (ipykernel)",
-   "language": "python",
-   "name": "python3"
-  },
-  "language_info": {
-   "codemirror_mode": {
-    "name": "ipython",
-    "version": 3
-   },
-   "file_extension": ".py",
-   "mimetype": "text/x-python",
-   "name": "python",
-   "nbconvert_exporter": "python",
-   "pygments_lexer": "ipython3",
-   "version": "3.12.2"
-  }
- },
- "nbformat": 4,
- "nbformat_minor": 5
-}
--- a/docs/src/notebooks/youtube_transcript_search.ipynb
+++ b/docs/src/notebooks/youtube_transcript_search.ipynb
@@ -36,7 +36,7 @@
    }
   ],
   "source": [
-    "!pip install --quiet openai datasets\n",
+    "!pip install --quiet openai datasets \n",
    "!pip install --quiet -U lancedb"
   ]
  },
@@ -213,7 +213,7 @@
    "if \"OPENAI_API_KEY\" not in os.environ:\n",
    "    # OR set the key here as a variable\n",
    "    os.environ[\"OPENAI_API_KEY\"] = \"sk-...\"\n",
-    "\n",
+    "    \n",
    "client = OpenAI()\n",
    "assert len(client.models.list().data) > 0"
   ]
@@ -234,12 +234,9 @@
   "metadata": {},
   "outputs": [],
   "source": [
-    "def embed_func(c):\n",
+    "def embed_func(c):    \n",
    "    rs = client.embeddings.create(input=c, model=\"text-embedding-ada-002\")\n",
-    "    return [\n",
-    "        data.embedding\n",
-    "        for data in rs.data\n",
-    "    ]"
+    "    return [rs.data[0].embedding]"
   ]
  },
  {
@@ -517,7 +514,7 @@
    "                prompt_start +\n",
    "                \"\\n\\n---\\n\\n\".join(context.text) +\n",
    "                prompt_end\n",
-    "            )\n",
+    "            )    \n",
    "    return prompt"
   ]
  },
--- a/docs/src/reranking/jina.md
+++ b/docs/src/reranking/jina.md
@@ -1,78 +0,0 @@
-# Jina Reranker
-
-This re-ranker uses the [Jina](https://jina.ai/reranker/) API to rerank the search results. You can use this re-ranker by passing `JinaReranker()` to the `rerank()` method. Note that you'll either need to set the `JINA_API_KEY` environment variable or pass the `api_key` argument to use this re-ranker.
-
-
-!!! note
-    Supported Query Types: Hybrid, Vector, FTS
-
-
-```python
-import os
-import lancedb
-from lancedb.embeddings import get_registry
-from lancedb.pydantic import LanceModel, Vector
-from lancedb.rerankers import JinaReranker
-
-os.environ['JINA_API_KEY'] = "jina_*"
-
-
-embedder = get_registry().get("jina").create()
-db = lancedb.connect("~/.lancedb")
-
-class Schema(LanceModel):
-    text: str = embedder.SourceField()
-    vector: Vector(embedder.ndims()) = embedder.VectorField()
-
-data = [
-    {"text": "hello world"},
-    {"text": "goodbye world"}
-    ]
-tbl = db.create_table("test", schema=Schema, mode="overwrite")
-tbl.add(data)
-reranker = JinaReranker(api_key="key")
-
-# Run vector search with a reranker
-result = tbl.search("hello").rerank(reranker=reranker).to_list() 
-
-# Run FTS search with a reranker
-result = tbl.search("hello", query_type="fts").rerank(reranker=reranker).to_list()
-
-# Run hybrid search with a reranker
-tbl.create_fts_index("text", replace=True)
-result = tbl.search("hello", query_type="hybrid").rerank(reranker=reranker).to_list()
-
-```
-
-Accepted Arguments
----------------
-| Argument | Type | Default | Description |
-| --- | --- | --- | --- |
-| `model_name` | `str` | `"jina-reranker-v2-base-multilingual"` | The name of the reranker model to use. You can find the list of available models in https://jina.ai/reranker/|
-| `column` | `str` | `"text"` | The name of the column to use as input to the cross encoder model. |
-| `top_n` | `str` | `None` | The number of results to return. If None, will return all results. |
-| `api_key` | `str` | `None` | The API key for the Jina API. If not provided, the `JINA_API_KEY` environment variable is used. |
-| `return_score` | str | `"relevance"` | Options are "relevance" or "all". The type of score to return. If "relevance", will return only the `_relevance_score. If "all" is supported, will return relevance score along with the vector and/or fts scores depending on query type |
-
-
-
-## Supported Scores for each query type
-You can specify the type of scores you want the reranker to return. The following are the supported scores for each query type:
-
-### Hybrid Search
-|`return_score`| Status | Description |
-| --- | --- | --- |
-| `relevance` | ✅ Supported | Returns only have the `_relevance_score` column |
-| `all` | ❌ Not Supported | Returns have vector(`_distance`) and FTS(`score`) along with Hybrid Search score(`_relevance_score`) |
-
-### Vector Search
-|`return_score`| Status | Description |
-| --- | --- | --- |
-| `relevance` | ✅ Supported | Returns only have the `_relevance_score` column |
-| `all` | ✅ Supported | Returns have vector(`_distance`) along with Hybrid Search score(`_relevance_score`) |
-
-### FTS Search
-|`return_score`| Status | Description |
-| --- | --- | --- |
-| `relevance` | ✅ Supported | Returns only have the `_relevance_score` column |
-| `all` | ✅ Supported | Returns have FTS(`score`) along with Hybrid Search score(`_relevance_score`) |
--- a/docs/test/md_testing.py
+++ b/docs/test/md_testing.py
@@ -7,7 +7,7 @@ excluded_globs = [
    "../src/fts.md",
    "../src/embedding.md",
    "../src/examples/*.md",
-    "../src/integrations/*.md",
+    "../src/integrations/voxel51.md",
    "../src/guides/tables.md",
    "../src/python/duckdb.md",
    "../src/embeddings/*.md",
@@ -16,7 +16,6 @@ excluded_globs = [
    "../src/basic.md",
    "../src/hybrid_search/hybrid_search.md",
    "../src/reranking/*.md",
-    "../src/guides/tuning_retrievers/*.md",
 ]

 python_prefix = "py"
--- a/java/core/lancedb-jni/Cargo.toml
+++ b/java/core/lancedb-jni/Cargo.toml
@@ -1,27 +0,0 @@
-[package]
-name = "lancedb-jni"
-description = "JNI bindings for LanceDB"
-# TODO modify lancedb/Cargo.toml for version and dependencies
-version = "0.4.18"
-edition.workspace = true
-repository.workspace = true
-readme.workspace = true
-license.workspace = true
-keywords.workspace = true
-categories.workspace = true
-publish = false
-
-[lib]
-crate-type = ["cdylib"]
-
-[dependencies]
-lancedb = { path = "../../../rust/lancedb" }
-lance = { workspace = true }
-arrow = { workspace = true, features = ["ffi"] }
-arrow-schema.workspace = true
-tokio = "1.23"
-jni = "0.21.1"
-snafu.workspace = true
-lazy_static.workspace = true
-serde = { version = "^1" }
-serde_json = { version = "1" }
--- a/java/core/lancedb-jni/src/connection.rs
+++ b/java/core/lancedb-jni/src/connection.rs
@@ -1,130 +0,0 @@
-use crate::ffi::JNIEnvExt;
-use crate::traits::IntoJava;
-use crate::{Error, RT};
-use jni::objects::{JObject, JString, JValue};
-use jni::JNIEnv;
-pub const NATIVE_CONNECTION: &str = "nativeConnectionHandle";
-use crate::Result;
-use lancedb::connection::{connect, Connection};
-
-#[derive(Clone)]
-pub struct BlockingConnection {
-    pub(crate) inner: Connection,
-}
-
-impl BlockingConnection {
-    pub fn create(dataset_uri: &str) -> Result<Self> {
-        let inner = RT.block_on(connect(dataset_uri).execute())?;
-        Ok(Self { inner })
-    }
-
-    pub fn table_names(
-        &self,
-        start_after: Option<String>,
-        limit: Option<i32>,
-    ) -> Result<Vec<String>> {
-        let mut op = self.inner.table_names();
-        if let Some(start_after) = start_after {
-            op = op.start_after(start_after);
-        }
-        if let Some(limit) = limit {
-            op = op.limit(limit as u32);
-        }
-        Ok(RT.block_on(op.execute())?)
-    }
-}
-
-impl IntoJava for BlockingConnection {
-    fn into_java<'a>(self, env: &mut JNIEnv<'a>) -> JObject<'a> {
-        attach_native_connection(env, self)
-    }
-}
-
-fn attach_native_connection<'local>(
-    env: &mut JNIEnv<'local>,
-    connection: BlockingConnection,
-) -> JObject<'local> {
-    let j_connection = create_java_connection_object(env);
-    // This block sets a native Rust object (Connection) as a field in the Java object (j_Connection).
-    // Caution: This creates a potential for memory leaks. The Rust object (Connection) is not
-    // automatically garbage-collected by Java, and its memory will not be freed unless
-    // explicitly handled.
-    //
-    // To prevent memory leaks, ensure the following:
-    // 1. The Java object (`j_Connection`) should implement the `java.io.Closeable` interface.
-    // 2. Users of this Java object should be instructed to always use it within a try-with-resources
-    //    statement (or manually call the `close()` method) to ensure that `self.close()` is invoked.
-    match unsafe { env.set_rust_field(&j_connection, NATIVE_CONNECTION, connection) } {
-        Ok(_) => j_connection,
-        Err(err) => {
-            env.throw_new(
-                "java/lang/RuntimeException",
-                format!("Failed to set native handle for Connection: {}", err),
-            )
-            .expect("Error throwing exception");
-            JObject::null()
-        }
-    }
-}
-
-fn create_java_connection_object<'a>(env: &mut JNIEnv<'a>) -> JObject<'a> {
-    env.new_object("com/lancedb/lancedb/Connection", "()V", &[])
-        .expect("Failed to create Java Lance Connection instance")
-}
-
-#[no_mangle]
-pub extern "system" fn Java_com_lancedb_lancedb_Connection_releaseNativeConnection(
-    mut env: JNIEnv,
-    j_connection: JObject,
-) {
-    let _: BlockingConnection = unsafe {
-        env.take_rust_field(j_connection, NATIVE_CONNECTION)
-            .expect("Failed to take native Connection handle")
-    };
-}
-
-#[no_mangle]
-pub extern "system" fn Java_com_lancedb_lancedb_Connection_connect<'local>(
-    mut env: JNIEnv<'local>,
-    _obj: JObject,
-    dataset_uri_object: JString,
-) -> JObject<'local> {
-    let dataset_uri: String = ok_or_throw!(env, env.get_string(&dataset_uri_object)).into();
-    let blocking_connection = ok_or_throw!(env, BlockingConnection::create(&dataset_uri));
-    blocking_connection.into_java(&mut env)
-}
-
-#[no_mangle]
-pub extern "system" fn Java_com_lancedb_lancedb_Connection_tableNames<'local>(
-    mut env: JNIEnv<'local>,
-    j_connection: JObject,
-    start_after_obj: JObject, // Optional<String>
-    limit_obj: JObject,       // Optional<Integer>
-) -> JObject<'local> {
-    ok_or_throw!(
-        env,
-        inner_table_names(&mut env, j_connection, start_after_obj, limit_obj)
-    )
-}
-
-fn inner_table_names<'local>(
-    env: &mut JNIEnv<'local>,
-    j_connection: JObject,
-    start_after_obj: JObject, // Optional<String>
-    limit_obj: JObject,       // Optional<Integer>
-) -> Result<JObject<'local>> {
-    let start_after = env.get_string_opt(&start_after_obj)?;
-    let limit = env.get_int_opt(&limit_obj)?;
-    let conn =
-        unsafe { env.get_rust_field::<_, _, BlockingConnection>(j_connection, NATIVE_CONNECTION) }?;
-    let table_names = conn.table_names(start_after, limit)?;
-    drop(conn);
-    let j_names = env.new_object("java/util/ArrayList", "()V", &[])?;
-    for item in table_names {
-        let jstr_item = env.new_string(item)?;
-        let item_jobj = JObject::from(jstr_item);
-        let item_gen = JValue::Object(&item_jobj);
-        env.call_method(&j_names, "add", "(Ljava/lang/Object;)Z", &[item_gen])?;
-    }
-    Ok(j_names)
-}
--- a/java/core/lancedb-jni/src/error.rs
+++ b/java/core/lancedb-jni/src/error.rs
@@ -1,225 +0,0 @@
-// Copyright 2024 Lance Developers.
-//
-// Licensed under the Apache License, Version 2.0 (the "License");
-// you may not use this file except in compliance with the License.
-// You may obtain a copy of the License at
-//
-//     http://www.apache.org/licenses/LICENSE-2.0
-//
-// Unless required by applicable law or agreed to in writing, software
-// distributed under the License is distributed on an "AS IS" BASIS,
-// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-// See the License for the specific language governing permissions and
-// limitations under the License.
-
-use std::str::Utf8Error;
-
-use arrow_schema::ArrowError;
-use jni::errors::Error as JniError;
-use serde_json::Error as JsonError;
-use snafu::{Location, Snafu};
-
-type BoxedError = Box<dyn std::error::Error + Send + Sync + 'static>;
-
-/// Java Exception types
-pub enum JavaException {
-    IllegalArgumentException,
-    IOException,
-    RuntimeException,
-}
-
-impl JavaException {
-    pub fn as_str(&self) -> &str {
-        match self {
-            Self::IllegalArgumentException => "java/lang/IllegalArgumentException",
-            Self::IOException => "java/io/IOException",
-            Self::RuntimeException => "java/lang/RuntimeException",
-        }
-    }
-}
-/// TODO(lu) change to lancedb-jni
-#[derive(Debug, Snafu)]
-#[snafu(visibility(pub))]
-pub enum Error {
-    #[snafu(display("JNI error: {message}, {location}"))]
-    Jni { message: String, location: Location },
-    #[snafu(display("Invalid argument: {message}, {location}"))]
-    InvalidArgument { message: String, location: Location },
-    #[snafu(display("IO error: {source}, {location}"))]
-    IO {
-        source: BoxedError,
-        location: Location,
-    },
-    #[snafu(display("Arrow error: {message}, {location}"))]
-    Arrow { message: String, location: Location },
-    #[snafu(display("Index error: {message}, {location}"))]
-    Index { message: String, location: Location },
-    #[snafu(display("JSON error: {message}, {location}"))]
-    JSON { message: String, location: Location },
-    #[snafu(display("Dataset at path {path} was not found, {location}"))]
-    DatasetNotFound { path: String, location: Location },
-    #[snafu(display("Dataset already exists: {uri}, {location}"))]
-    DatasetAlreadyExists { uri: String, location: Location },
-    #[snafu(display("Table '{name}' already exists"))]
-    TableAlreadyExists { name: String },
-    #[snafu(display("Table '{name}' was not found"))]
-    TableNotFound { name: String },
-    #[snafu(display("Invalid table name '{name}': {reason}"))]
-    InvalidTableName { name: String, reason: String },
-    #[snafu(display("Embedding function '{name}' was not found: {reason}, {location}"))]
-    EmbeddingFunctionNotFound {
-        name: String,
-        reason: String,
-        location: Location,
-    },
-    #[snafu(display("Other Lance error: {message}, {location}"))]
-    OtherLance { message: String, location: Location },
-    #[snafu(display("Other LanceDB error: {message}, {location}"))]
-    OtherLanceDB { message: String, location: Location },
-}
-
-impl Error {
-    /// Throw as Java Exception
-    pub fn throw(&self, env: &mut jni::JNIEnv) {
-        match self {
-            Self::InvalidArgument { .. }
-            | Self::DatasetNotFound { .. }
-            | Self::DatasetAlreadyExists { .. }
-            | Self::TableAlreadyExists { .. }
-            | Self::TableNotFound { .. }
-            | Self::InvalidTableName { .. }
-            | Self::EmbeddingFunctionNotFound { .. } => {
-                self.throw_as(env, JavaException::IllegalArgumentException)
-            }
-            Self::IO { .. } | Self::Index { .. } => self.throw_as(env, JavaException::IOException),
-            Self::Arrow { .. }
-            | Self::JSON { .. }
-            | Self::OtherLance { .. }
-            | Self::OtherLanceDB { .. }
-            | Self::Jni { .. } => self.throw_as(env, JavaException::RuntimeException),
-        }
-    }
-
-    /// Throw as an concrete Java Exception
-    pub fn throw_as(&self, env: &mut jni::JNIEnv, exception: JavaException) {
-        let message = &format!(
-            "Error when throwing Java exception: {}:{}",
-            exception.as_str(),
-            self
-        );
-        env.throw_new(exception.as_str(), self.to_string())
-            .expect(message);
-    }
-}
-
-pub type Result<T> = std::result::Result<T, Error>;
-
-trait ToSnafuLocation {
-    fn to_snafu_location(&'static self) -> snafu::Location;
-}
-
-impl ToSnafuLocation for std::panic::Location<'static> {
-    fn to_snafu_location(&'static self) -> snafu::Location {
-        snafu::Location::new(self.file(), self.line(), self.column())
-    }
-}
-
-impl From<JniError> for Error {
-    #[track_caller]
-    fn from(source: JniError) -> Self {
-        Self::Jni {
-            message: source.to_string(),
-            location: std::panic::Location::caller().to_snafu_location(),
-        }
-    }
-}
-
-impl From<Utf8Error> for Error {
-    #[track_caller]
-    fn from(source: Utf8Error) -> Self {
-        Self::InvalidArgument {
-            message: source.to_string(),
-            location: std::panic::Location::caller().to_snafu_location(),
-        }
-    }
-}
-
-impl From<ArrowError> for Error {
-    #[track_caller]
-    fn from(source: ArrowError) -> Self {
-        Self::Arrow {
-            message: source.to_string(),
-            location: std::panic::Location::caller().to_snafu_location(),
-        }
-    }
-}
-
-impl From<JsonError> for Error {
-    #[track_caller]
-    fn from(source: JsonError) -> Self {
-        Self::JSON {
-            message: source.to_string(),
-            location: std::panic::Location::caller().to_snafu_location(),
-        }
-    }
-}
-
-impl From<lance::Error> for Error {
-    #[track_caller]
-    fn from(source: lance::Error) -> Self {
-        match source {
-            lance::Error::DatasetNotFound {
-                path,
-                source: _,
-                location,
-            } => Self::DatasetNotFound { path, location },
-            lance::Error::DatasetAlreadyExists { uri, location } => {
-                Self::DatasetAlreadyExists { uri, location }
-            }
-            lance::Error::IO { source, location } => Self::IO { source, location },
-            lance::Error::Arrow { message, location } => Self::Arrow { message, location },
-            lance::Error::Index { message, location } => Self::Index { message, location },
-            lance::Error::InvalidInput { source, location } => Self::InvalidArgument {
-                message: source.to_string(),
-                location,
-            },
-            _ => Self::OtherLance {
-                message: source.to_string(),
-                location: std::panic::Location::caller().to_snafu_location(),
-            },
-        }
-    }
-}
-
-impl From<lancedb::Error> for Error {
-    #[track_caller]
-    fn from(source: lancedb::Error) -> Self {
-        match source {
-            lancedb::Error::InvalidTableName { name, reason } => {
-                Self::InvalidTableName { name, reason }
-            }
-            lancedb::Error::InvalidInput { message } => Self::InvalidArgument {
-                message,
-                location: std::panic::Location::caller().to_snafu_location(),
-            },
-            lancedb::Error::TableNotFound { name } => Self::TableNotFound { name },
-            lancedb::Error::TableAlreadyExists { name } => Self::TableAlreadyExists { name },
-            lancedb::Error::EmbeddingFunctionNotFound { name, reason } => {
-                Self::EmbeddingFunctionNotFound {
-                    name,
-                    reason,
-                    location: std::panic::Location::caller().to_snafu_location(),
-                }
-            }
-            lancedb::Error::Arrow { source } => Self::Arrow {
-                message: source.to_string(),
-                location: std::panic::Location::caller().to_snafu_location(),
-            },
-            lancedb::Error::Lance { source } => Self::from(source),
-            _ => Self::OtherLanceDB {
-                message: source.to_string(),
-                location: std::panic::Location::caller().to_snafu_location(),
-            },
-        }
-    }
-}
--- a/java/core/lancedb-jni/src/ffi.rs
+++ b/java/core/lancedb-jni/src/ffi.rs
@@ -1,204 +0,0 @@
-// Copyright 2024 Lance Developers.
-//
-// Licensed under the Apache License, Version 2.0 (the "License");
-// you may not use this file except in compliance with the License.
-// You may obtain a copy of the License at
-//
-//     http://www.apache.org/licenses/LICENSE-2.0
-//
-// Unless required by applicable law or agreed to in writing, software
-// distributed under the License is distributed on an "AS IS" BASIS,
-// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-// See the License for the specific language governing permissions and
-// limitations under the License.
-
-use core::slice;
-
-use jni::objects::{JByteBuffer, JObjectArray, JString};
-use jni::sys::jobjectArray;
-use jni::{objects::JObject, JNIEnv};
-
-use crate::error::{Error, Result};
-
-/// TODO(lu) import from lance-jni without duplicate
-/// Extend JNIEnv with helper functions.
-pub trait JNIEnvExt {
-    /// Get integers from Java List<Integer> object.
-    fn get_integers(&mut self, obj: &JObject) -> Result<Vec<i32>>;
-
-    /// Get strings from Java List<String> object.
-    fn get_strings(&mut self, obj: &JObject) -> Result<Vec<String>>;
-
-    /// Get strings from Java String[] object.
-    /// Note that get Option<Vec<String>> from Java Optional<String[]> just doesn't work.
-    #[allow(unused)]
-    fn get_strings_array(&mut self, obj: jobjectArray) -> Result<Vec<String>>;
-
-    /// Get Option<String> from Java Optional<String>.
-    fn get_string_opt(&mut self, obj: &JObject) -> Result<Option<String>>;
-
-    /// Get Option<Vec<String>> from Java Optional<List<String>>.
-    #[allow(unused)]
-    fn get_strings_opt(&mut self, obj: &JObject) -> Result<Option<Vec<String>>>;
-
-    /// Get Option<i32> from Java Optional<Integer>.
-    fn get_int_opt(&mut self, obj: &JObject) -> Result<Option<i32>>;
-
-    /// Get Option<Vec<i32>> from Java Optional<List<Integer>>.
-    fn get_ints_opt(&mut self, obj: &JObject) -> Result<Option<Vec<i32>>>;
-
-    /// Get Option<i64> from Java Optional<Long>.
-    #[allow(unused)]
-    fn get_long_opt(&mut self, obj: &JObject) -> Result<Option<i64>>;
-
-    /// Get Option<u64> from Java Optional<Long>.
-    #[allow(unused)]
-    fn get_u64_opt(&mut self, obj: &JObject) -> Result<Option<u64>>;
-
-    /// Get Option<&[u8]> from Java Optional<ByteBuffer>.
-    #[allow(unused)]
-    fn get_bytes_opt(&mut self, obj: &JObject) -> Result<Option<&[u8]>>;
-
-    fn get_optional<T, F>(&mut self, obj: &JObject, f: F) -> Result<Option<T>>
-    where
-        F: FnOnce(&mut JNIEnv, &JObject) -> Result<T>;
-}
-
-impl JNIEnvExt for JNIEnv<'_> {
-    fn get_integers(&mut self, obj: &JObject) -> Result<Vec<i32>> {
-        let list = self.get_list(obj)?;
-        let mut iter = list.iter(self)?;
-        let mut results = Vec::with_capacity(list.size(self)? as usize);
-        while let Some(elem) = iter.next(self)? {
-            let int_obj = self.call_method(elem, "intValue", "()I", &[])?;
-            let int_value = int_obj.i()?;
-            results.push(int_value);
-        }
-        Ok(results)
-    }
-
-    fn get_strings(&mut self, obj: &JObject) -> Result<Vec<String>> {
-        let list = self.get_list(obj)?;
-        let mut iter = list.iter(self)?;
-        let mut results = Vec::with_capacity(list.size(self)? as usize);
-        while let Some(elem) = iter.next(self)? {
-            let jstr = JString::from(elem);
-            let val = self.get_string(&jstr)?;
-            results.push(val.to_str()?.to_string())
-        }
-        Ok(results)
-    }
-
-    fn get_strings_array(&mut self, obj: jobjectArray) -> Result<Vec<String>> {
-        let jobject_array = unsafe { JObjectArray::from_raw(obj) };
-        let array_len = self.get_array_length(&jobject_array)?;
-        let mut res: Vec<String> = Vec::new();
-        for i in 0..array_len {
-            let item: JString = self.get_object_array_element(&jobject_array, i)?.into();
-            res.push(self.get_string(&item)?.into());
-        }
-        Ok(res)
-    }
-
-    fn get_string_opt(&mut self, obj: &JObject) -> Result<Option<String>> {
-        self.get_optional(obj, |env, inner_obj| {
-            let java_obj_gen = env.call_method(inner_obj, "get", "()Ljava/lang/Object;", &[])?;
-            let java_string_obj = java_obj_gen.l()?;
-            let jstr = JString::from(java_string_obj);
-            let val = env.get_string(&jstr)?;
-            Ok(val.to_str()?.to_string())
-        })
-    }
-
-    fn get_strings_opt(&mut self, obj: &JObject) -> Result<Option<Vec<String>>> {
-        self.get_optional(obj, |env, inner_obj| {
-            let java_obj_gen = env.call_method(inner_obj, "get", "()Ljava/lang/Object;", &[])?;
-            let java_list_obj = java_obj_gen.l()?;
-            env.get_strings(&java_list_obj)
-        })
-    }
-
-    fn get_int_opt(&mut self, obj: &JObject) -> Result<Option<i32>> {
-        self.get_optional(obj, |env, inner_obj| {
-            let java_obj_gen = env.call_method(inner_obj, "get", "()Ljava/lang/Object;", &[])?;
-            let java_int_obj = java_obj_gen.l()?;
-            let int_obj = env.call_method(java_int_obj, "intValue", "()I", &[])?;
-            let int_value = int_obj.i()?;
-            Ok(int_value)
-        })
-    }
-
-    fn get_ints_opt(&mut self, obj: &JObject) -> Result<Option<Vec<i32>>> {
-        self.get_optional(obj, |env, inner_obj| {
-            let java_obj_gen = env.call_method(inner_obj, "get", "()Ljava/lang/Object;", &[])?;
-            let java_list_obj = java_obj_gen.l()?;
-            env.get_integers(&java_list_obj)
-        })
-    }
-
-    fn get_long_opt(&mut self, obj: &JObject) -> Result<Option<i64>> {
-        self.get_optional(obj, |env, inner_obj| {
-            let java_obj_gen = env.call_method(inner_obj, "get", "()Ljava/lang/Object;", &[])?;
-            let java_long_obj = java_obj_gen.l()?;
-            let long_obj = env.call_method(java_long_obj, "longValue", "()J", &[])?;
-            let long_value = long_obj.j()?;
-            Ok(long_value)
-        })
-    }
-
-    fn get_u64_opt(&mut self, obj: &JObject) -> Result<Option<u64>> {
-        self.get_optional(obj, |env, inner_obj| {
-            let java_obj_gen = env.call_method(inner_obj, "get", "()Ljava/lang/Object;", &[])?;
-            let java_long_obj = java_obj_gen.l()?;
-            let long_obj = env.call_method(java_long_obj, "longValue", "()J", &[])?;
-            let long_value = long_obj.j()?;
-            Ok(long_value as u64)
-        })
-    }
-
-    fn get_bytes_opt(&mut self, obj: &JObject) -> Result<Option<&[u8]>> {
-        self.get_optional(obj, |env, inner_obj| {
-            let java_obj_gen = env.call_method(inner_obj, "get", "()Ljava/lang/Object;", &[])?;
-            let java_byte_buffer_obj = java_obj_gen.l()?;
-            let j_byte_buffer = JByteBuffer::from(java_byte_buffer_obj);
-            let raw_data = env.get_direct_buffer_address(&j_byte_buffer)?;
-            let capacity = env.get_direct_buffer_capacity(&j_byte_buffer)?;
-            let data = unsafe { slice::from_raw_parts(raw_data, capacity) };
-            Ok(data)
-        })
-    }
-
-    fn get_optional<T, F>(&mut self, obj: &JObject, f: F) -> Result<Option<T>>
-    where
-        F: FnOnce(&mut JNIEnv, &JObject) -> Result<T>,
-    {
-        if obj.is_null() {
-            return Ok(None);
-        }
-        let is_present = self.call_method(obj, "isPresent", "()Z", &[])?;
-        if !is_present.z()? {
-            // TODO(lu): put get java object into here cuz can only get java Object
-            Ok(None)
-        } else {
-            f(self, obj).map(Some)
-        }
-    }
-}
-
-#[no_mangle]
-pub extern "system" fn Java_com_lancedb_lance_test_JniTestHelper_parseInts(
-    mut env: JNIEnv,
-    _obj: JObject,
-    list_obj: JObject, // List<Integer>
-) {
-    ok_or_throw_without_return!(env, env.get_integers(&list_obj));
-}
-
-#[no_mangle]
-pub extern "system" fn Java_com_lancedb_lance_test_JniTestHelper_parseIntsOpt(
-    mut env: JNIEnv,
-    _obj: JObject,
-    list_obj: JObject, // Optional<List<Integer>>
-) {
-    ok_or_throw_without_return!(env, env.get_ints_opt(&list_obj));
-}
--- a/java/core/lancedb-jni/src/lib.rs
+++ b/java/core/lancedb-jni/src/lib.rs
@@ -1,68 +0,0 @@
-// Copyright 2024 Lance Developers.
-//
-// Licensed under the Apache License, Version 2.0 (the "License");
-// you may not use this file except in compliance with the License.
-// You may obtain a copy of the License at
-//
-//     http://www.apache.org/licenses/LICENSE-2.0
-//
-// Unless required by applicable law or agreed to in writing, software
-// distributed under the License is distributed on an "AS IS" BASIS,
-// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-// See the License for the specific language governing permissions and
-// limitations under the License.
-
-use lazy_static::lazy_static;
-
-// TODO import from lance-jni without duplicate
-#[macro_export]
-macro_rules! ok_or_throw {
-    ($env:expr, $result:expr) => {
-        match $result {
-            Ok(value) => value,
-            Err(err) => {
-                Error::from(err).throw(&mut $env);
-                return JObject::null();
-            }
-        }
-    };
-}
-
-macro_rules! ok_or_throw_without_return {
-    ($env:expr, $result:expr) => {
-        match $result {
-            Ok(value) => value,
-            Err(err) => {
-                Error::from(err).throw(&mut $env);
-                return;
-            }
-        }
-    };
-}
-
-#[macro_export]
-macro_rules! ok_or_throw_with_return {
-    ($env:expr, $result:expr, $ret:expr) => {
-        match $result {
-            Ok(value) => value,
-            Err(err) => {
-                Error::from(err).throw(&mut $env);
-                return $ret;
-            }
-        }
-    };
-}
-
-mod connection;
-pub mod error;
-mod ffi;
-mod traits;
-
-pub use error::{Error, Result};
-
-lazy_static! {
-    static ref RT: tokio::runtime::Runtime = tokio::runtime::Builder::new_multi_thread()
-        .enable_all()
-        .build()
-        .expect("Failed to create tokio runtime");
-}
--- a/java/core/lancedb-jni/src/traits.rs
+++ b/java/core/lancedb-jni/src/traits.rs
@@ -1,122 +0,0 @@
-// Copyright 2024 Lance Developers.
-//
-// Licensed under the Apache License, Version 2.0 (the "License");
-// you may not use this file except in compliance with the License.
-// You may obtain a copy of the License at
-//
-//     http://www.apache.org/licenses/LICENSE-2.0
-//
-// Unless required by applicable law or agreed to in writing, software
-// distributed under the License is distributed on an "AS IS" BASIS,
-// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-// See the License for the specific language governing permissions and
-// limitations under the License.
-
-use jni::objects::{JMap, JObject, JString, JValue};
-use jni::JNIEnv;
-
-use crate::Result;
-
-pub trait FromJObject<T> {
-    fn extract(&self) -> Result<T>;
-}
-
-/// Convert a Rust type into a Java Object.
-pub trait IntoJava {
-    fn into_java<'a>(self, env: &mut JNIEnv<'a>) -> JObject<'a>;
-}
-
-impl FromJObject<i32> for JObject<'_> {
-    fn extract(&self) -> Result<i32> {
-        Ok(JValue::from(self).i()?)
-    }
-}
-
-impl FromJObject<i64> for JObject<'_> {
-    fn extract(&self) -> Result<i64> {
-        Ok(JValue::from(self).j()?)
-    }
-}
-
-impl FromJObject<f32> for JObject<'_> {
-    fn extract(&self) -> Result<f32> {
-        Ok(JValue::from(self).f()?)
-    }
-}
-
-impl FromJObject<f64> for JObject<'_> {
-    fn extract(&self) -> Result<f64> {
-        Ok(JValue::from(self).d()?)
-    }
-}
-
-pub trait FromJString {
-    fn extract(&self, env: &mut JNIEnv) -> Result<String>;
-}
-
-impl FromJString for JString<'_> {
-    fn extract(&self, env: &mut JNIEnv) -> Result<String> {
-        Ok(env.get_string(self)?.into())
-    }
-}
-
-pub trait JMapExt {
-    #[allow(dead_code)]
-    fn get_string(&self, env: &mut JNIEnv, key: &str) -> Result<Option<String>>;
-
-    #[allow(dead_code)]
-    fn get_i32(&self, env: &mut JNIEnv, key: &str) -> Result<Option<i32>>;
-
-    #[allow(dead_code)]
-    fn get_i64(&self, env: &mut JNIEnv, key: &str) -> Result<Option<i64>>;
-
-    #[allow(dead_code)]
-    fn get_f32(&self, env: &mut JNIEnv, key: &str) -> Result<Option<f32>>;
-
-    #[allow(dead_code)]
-    fn get_f64(&self, env: &mut JNIEnv, key: &str) -> Result<Option<f64>>;
-}
-
-fn get_map_value<T>(env: &mut JNIEnv, map: &JMap, key: &str) -> Result<Option<T>>
-where
-    for<'a> JObject<'a>: FromJObject<T>,
-{
-    let key_obj: JObject = env.new_string(key)?.into();
-    if let Some(value) = map.get(env, &key_obj)? {
-        if value.is_null() {
-            Ok(None)
-        } else {
-            Ok(Some(value.extract()?))
-        }
-    } else {
-        Ok(None)
-    }
-}
-
-impl JMapExt for JMap<'_, '_, '_> {
-    fn get_string(&self, env: &mut JNIEnv, key: &str) -> Result<Option<String>> {
-        let key_obj: JObject = env.new_string(key)?.into();
-        if let Some(value) = self.get(env, &key_obj)? {
-            let value_str: JString = value.into();
-            Ok(Some(value_str.extract(env)?))
-        } else {
-            Ok(None)
-        }
-    }
-
-    fn get_i32(&self, env: &mut JNIEnv, key: &str) -> Result<Option<i32>> {
-        get_map_value(env, self, key)
-    }
-
-    fn get_i64(&self, env: &mut JNIEnv, key: &str) -> Result<Option<i64>> {
-        get_map_value(env, self, key)
-    }
-
-    fn get_f32(&self, env: &mut JNIEnv, key: &str) -> Result<Option<f32>> {
-        get_map_value(env, self, key)
-    }
-
-    fn get_f64(&self, env: &mut JNIEnv, key: &str) -> Result<Option<f64>> {
-        get_map_value(env, self, key)
-    }
-}
--- a/java/core/pom.xml
+++ b/java/core/pom.xml
@@ -1,94 +0,0 @@
-<?xml version="1.0" encoding="UTF-8"?>
-
-<project xmlns="http://maven.apache.org/POM/4.0.0"
-    xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
-    xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/xsd/maven-4.0.0.xsd">
-    <modelVersion>4.0.0</modelVersion>
-
-    <parent>
-        <groupId>com.lancedb</groupId>
-        <artifactId>lancedb-parent</artifactId>
-        <version>0.1-SNAPSHOT</version>
-        <relativePath>../pom.xml</relativePath>
-    </parent>
-
-    <artifactId>lancedb-core</artifactId>
-    <name>LanceDB Core</name>
-    <packaging>jar</packaging>
-
-    <dependencies>
-        <dependency>
-            <groupId>org.apache.arrow</groupId>
-            <artifactId>arrow-vector</artifactId>
-        </dependency>
-        <dependency>
-            <groupId>org.apache.arrow</groupId>
-            <artifactId>arrow-memory-netty</artifactId>
-        </dependency>
-        <dependency>
-            <groupId>org.apache.arrow</groupId>
-            <artifactId>arrow-c-data</artifactId>
-        </dependency>
-        <dependency>
-            <groupId>org.apache.arrow</groupId>
-            <artifactId>arrow-dataset</artifactId>
-        </dependency>
-        <dependency>
-            <groupId>org.json</groupId>
-            <artifactId>json</artifactId>
-        </dependency>
-        <dependency>
-            <groupId>org.questdb</groupId>
-            <artifactId>jar-jni</artifactId>
-        </dependency>
-        <dependency>
-            <groupId>org.junit.jupiter</groupId>
-            <artifactId>junit-jupiter</artifactId>
-             <scope>test</scope>
-        </dependency>
-    </dependencies>
-
-    <profiles>
-        <profile>
-            <id>build-jni</id>
-            <activation>
-                <activeByDefault>true</activeByDefault>
-            </activation>
-            <build>
-                <plugins>
-                    <plugin>
-                        <groupId>org.questdb</groupId>
-                        <artifactId>rust-maven-plugin</artifactId>
-                        <version>1.1.1</version>
-                        <executions>
-                            <execution>
-                                <id>lancedb-jni</id>
-                                <goals>
-                                    <goal>build</goal>
-                                </goals>
-                                <configuration>
-                                    <path>lancedb-jni</path>
-                                    <!--<release>true</release>-->
-                                    <!-- Copy native libraries to target/classes for runtime access -->
-                                    <copyTo>${project.build.directory}/classes/nativelib</copyTo>
-                                    <copyWithPlatformDir>true</copyWithPlatformDir>
-                                </configuration>
-                            </execution>
-                            <execution>
-                                <id>lancedb-jni-test</id>
-                                <goals>
-                                    <goal>test</goal>
-                                </goals>
-                                <configuration>
-                                    <path>lancedb-jni</path>
-                                    <release>false</release>
-                                    <verbosity>-v</verbosity>
-                                </configuration>
-                            </execution>
-                        </executions>
-                    </plugin>
-                </plugins>
-            </build>
-        </profile>
-    </profiles>
-</project>
--- a/java/core/src/main/java/com/lancedb/lancedb/Connection.java
+++ b/java/core/src/main/java/com/lancedb/lancedb/Connection.java
@@ -1,120 +0,0 @@
-/*
- * Licensed under the Apache License, Version 2.0 (the "License");
- * you may not use this file except in compliance with the License.
- * You may obtain a copy of the License at
- *
- *     http://www.apache.org/licenses/LICENSE-2.0
- *
- * Unless required by applicable law or agreed to in writing, software
- * distributed under the License is distributed on an "AS IS" BASIS,
- * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
- * See the License for the specific language governing permissions and
- * limitations under the License.
- */
-
-package com.lancedb.lancedb;
-
-import io.questdb.jar.jni.JarJniLoader;
-import java.io.Closeable;
-import java.util.List;
-import java.util.Optional;
-
-/**
- * Represents LanceDB database.
- */
-public class Connection implements Closeable {
-  static {
-    JarJniLoader.loadLib(Connection.class, "/nativelib", "lancedb_jni");
-  }
-
-  private long nativeConnectionHandle;
-
-  /**
-   * Connect to a LanceDB instance.
-   */
-  public static native Connection connect(String uri);
-
-  /**
-   * Get the names of all tables in the database. The names are sorted in
-   * ascending order.
-   *
-   * @return the table names
-   */
-  public List<String> tableNames() {
-    return tableNames(Optional.empty(), Optional.empty());
-  }
-
-  /**
-   * Get the names of filtered tables in the database. The names are sorted in
-   * ascending order.
-   *
-   * @param limit The number of results to return.
-   * @return the table names
-   */
-  public List<String> tableNames(int limit) {
-    return tableNames(Optional.empty(), Optional.of(limit));
-  }
-
-  /**
-   * Get the names of filtered tables in the database. The names are sorted in
-   * ascending order.
-   *
-   * @param startAfter If present, only return names that come lexicographically after the supplied
-   *                   value. This can be combined with limit to implement pagination
-   *                   by setting this to the last table name from the previous page.
-   * @return the table names
-   */
-  public List<String> tableNames(String startAfter) {
-    return tableNames(Optional.of(startAfter), Optional.empty());
-  }
-
-  /**
-   * Get the names of filtered tables in the database. The names are sorted in
-   * ascending order.
-   *
-   * @param startAfter If present, only return names that come lexicographically after the supplied
-   *                   value. This can be combined with limit to implement pagination
-   *                   by setting this to the last table name from the previous page.
-   * @param limit The number of results to return.
-   * @return the table names
-   */
-  public List<String> tableNames(String startAfter, int limit) {
-    return tableNames(Optional.of(startAfter), Optional.of(limit));
-  }
-
-  /**
-   * Get the names of filtered tables in the database. The names are sorted in
-   * ascending order.
-   *
-   * @param startAfter If present, only return names that come lexicographically after the supplied
-   *                   value. This can be combined with limit to implement pagination
-   *                   by setting this to the last table name from the previous page.
-   * @param limit The number of results to return.
-   * @return the table names
-   */
-  public native List<String> tableNames(
-      Optional<String> startAfter, Optional<Integer> limit);
-
-  /**
-   * Closes this connection and releases any system resources associated with it. If
-   * the connection is
-   * already closed, then invoking this method has no effect.
-   */
-  @Override
-  public void close() {
-    if (nativeConnectionHandle != 0) {
-      releaseNativeConnection(nativeConnectionHandle);
-      nativeConnectionHandle = 0;
-    }
-  }
-
-  /**
-   * Native method to release the Lance connection resources associated with the
-   * given handle.
-   *
-   * @param handle The native handle to the connection resource.
-   */
-  private native void releaseNativeConnection(long handle);
-
-  private Connection() {}
-}
--- a/java/core/src/test/java/com/lancedb/lancedb/ConnectionTest.java
+++ b/java/core/src/test/java/com/lancedb/lancedb/ConnectionTest.java
@@ -1,135 +0,0 @@
-/*
- * Licensed under the Apache License, Version 2.0 (the "License");
- * you may not use this file except in compliance with the License.
- * You may obtain a copy of the License at
- *
- *     http://www.apache.org/licenses/LICENSE-2.0
- *
- * Unless required by applicable law or agreed to in writing, software
- * distributed under the License is distributed on an "AS IS" BASIS,
- * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
- * See the License for the specific language governing permissions and
- * limitations under the License.
- */
-package com.lancedb.lancedb;
-
-import static org.junit.jupiter.api.Assertions.assertEquals;
-import static org.junit.jupiter.api.Assertions.assertTrue;
-
-import java.nio.file.Path;
-import java.util.List;
-import java.net.URL;
-import org.junit.jupiter.api.BeforeAll;
-import org.junit.jupiter.api.Test;
-import org.junit.jupiter.api.io.TempDir;
-
-public class ConnectionTest {
-  private static final String[] TABLE_NAMES = {
-      "dataset_version",
-      "new_empty_dataset",
-      "test",
-      "write_stream"
-  };
-
-  @TempDir
-  static Path tempDir; // Temporary directory for the tests
-  private static URL lanceDbURL;
-
-  @BeforeAll
-  static void setUp() {
-    ClassLoader classLoader = ConnectionTest.class.getClassLoader();
-    lanceDbURL = classLoader.getResource("example_db");
-  }
-
-  @Test
-  void emptyDB() {
-    String databaseUri = tempDir.resolve("emptyDB").toString();
-    try (Connection conn = Connection.connect(databaseUri)) {
-      List<String> tableNames = conn.tableNames();
-      assertTrue(tableNames.isEmpty());
-    }
-  }
-
-  @Test
-  void tableNames() {
-    try (Connection conn = Connection.connect(lanceDbURL.toString())) {
-      List<String> tableNames = conn.tableNames();
-      assertEquals(4, tableNames.size());
-      for (int i = 0; i < TABLE_NAMES.length; i++) {
-        assertEquals(TABLE_NAMES[i], tableNames.get(i));
-      }
-    }
-  }
-
-  @Test
-  void tableNamesStartAfter() {
-    try (Connection conn = Connection.connect(lanceDbURL.toString())) {
-      assertTableNamesStartAfter(conn, TABLE_NAMES[0], 3, TABLE_NAMES[1], TABLE_NAMES[2], TABLE_NAMES[3]);
-      assertTableNamesStartAfter(conn, TABLE_NAMES[1], 2, TABLE_NAMES[2], TABLE_NAMES[3]);
-      assertTableNamesStartAfter(conn, TABLE_NAMES[2], 1, TABLE_NAMES[3]);
-      assertTableNamesStartAfter(conn, TABLE_NAMES[3], 0);
-      assertTableNamesStartAfter(conn, "a_dataset", 4, TABLE_NAMES[0], TABLE_NAMES[1], TABLE_NAMES[2], TABLE_NAMES[3]);
-      assertTableNamesStartAfter(conn, "o_dataset", 2, TABLE_NAMES[2], TABLE_NAMES[3]);
-      assertTableNamesStartAfter(conn, "v_dataset", 1, TABLE_NAMES[3]);
-      assertTableNamesStartAfter(conn, "z_dataset", 0);
-    }
-  }
-
-  private void assertTableNamesStartAfter(Connection conn, String startAfter, int expectedSize, String... expectedNames) {
-    List<String> tableNames = conn.tableNames(startAfter);
-    assertEquals(expectedSize, tableNames.size());
-    for (int i = 0; i < expectedNames.length; i++) {
-      assertEquals(expectedNames[i], tableNames.get(i));
-    }
-  }
-
-  @Test
-  void tableNamesLimit() {
-      try (Connection conn = Connection.connect(lanceDbURL.toString())) {
-      for (int i = 0; i <= TABLE_NAMES.length; i++) {
-        List<String> tableNames = conn.tableNames(i);
-        assertEquals(i, tableNames.size());
-        for (int j = 0; j < i; j++) {
-          assertEquals(TABLE_NAMES[j], tableNames.get(j));
-        }
-      }
-    }
-  }
-
-  @Test
-  void tableNamesStartAfterLimit() {
-    try (Connection conn = Connection.connect(lanceDbURL.toString())) {
-      List<String> tableNames = conn.tableNames(TABLE_NAMES[0], 2);
-      assertEquals(2, tableNames.size());
-      assertEquals(TABLE_NAMES[1], tableNames.get(0));
-      assertEquals(TABLE_NAMES[2], tableNames.get(1));
-      tableNames = conn.tableNames(TABLE_NAMES[1], 1);
-      assertEquals(1, tableNames.size());
-      assertEquals(TABLE_NAMES[2], tableNames.get(0));
-      tableNames = conn.tableNames(TABLE_NAMES[2], 2);
-      assertEquals(1, tableNames.size());
-      assertEquals(TABLE_NAMES[3], tableNames.get(0));
-      tableNames = conn.tableNames(TABLE_NAMES[3], 2);
-      assertEquals(0, tableNames.size());
-      tableNames = conn.tableNames(TABLE_NAMES[0], 0);
-      assertEquals(0, tableNames.size());
-
-      // Limit larger than the number of remaining tables
-      tableNames = conn.tableNames(TABLE_NAMES[0], 10);
-      assertEquals(3, tableNames.size());
-      assertEquals(TABLE_NAMES[1], tableNames.get(0));
-      assertEquals(TABLE_NAMES[2], tableNames.get(1));
-      assertEquals(TABLE_NAMES[3], tableNames.get(2));
-
-      // Start after a value not in the list
-      tableNames = conn.tableNames("non_existent_table", 2);
-      assertEquals(2, tableNames.size());
-      assertEquals(TABLE_NAMES[2], tableNames.get(0));
-      assertEquals(TABLE_NAMES[3], tableNames.get(1));
-
-      // Start after the last table with a limit
-      tableNames = conn.tableNames(TABLE_NAMES[3], 1);
-      assertEquals(0, tableNames.size());
-    }
-  }
-}
--- a/java/core/src/test/resources/example_db/dataset_version.lance/_latest.manifest
+++ b/java/core/src/test/resources/example_db/dataset_version.lance/_latest.manifest
--- a/java/core/src/test/resources/example_db/dataset_version.lance/_transactions/0-d51afd07-e3cd-4c76-9b9b-787e13fd55b0.txn
+++ b/java/core/src/test/resources/example_db/dataset_version.lance/_transactions/0-d51afd07-e3cd-4c76-9b9b-787e13fd55b0.txn
@@ -1 +0,0 @@
-$d51afd07-e3cd-4c76-9b9b-787e13fd55b0<62>=id <20><><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD>*int3208name <20><><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD>*string08
--- a/java/core/src/test/resources/example_db/dataset_version.lance/_transactions/1-336c3e56-33fd-45d8-bbfb-95ebb563cbe0.txn
+++ b/java/core/src/test/resources/example_db/dataset_version.lance/_transactions/1-336c3e56-33fd-45d8-bbfb-95ebb563cbe0.txn
--- a/java/core/src/test/resources/example_db/dataset_version.lance/_transactions/2-3344b369-7471-4e23-8865-c949b6e19bc2.txn
+++ b/java/core/src/test/resources/example_db/dataset_version.lance/_transactions/2-3344b369-7471-4e23-8865-c949b6e19bc2.txn
--- a/java/core/src/test/resources/example_db/dataset_version.lance/_versions/1.manifest
+++ b/java/core/src/test/resources/example_db/dataset_version.lance/_versions/1.manifest
--- a/java/core/src/test/resources/example_db/dataset_version.lance/_versions/2.manifest
+++ b/java/core/src/test/resources/example_db/dataset_version.lance/_versions/2.manifest
--- a/java/core/src/test/resources/example_db/dataset_version.lance/_versions/3.manifest
+++ b/java/core/src/test/resources/example_db/dataset_version.lance/_versions/3.manifest
--- a/java/core/src/test/resources/example_db/dataset_version.lance/data/60a9b599-f79f-48a8-bffa-b495762b622a.lance
+++ b/java/core/src/test/resources/example_db/dataset_version.lance/data/60a9b599-f79f-48a8-bffa-b495762b622a.lance
--- a/java/core/src/test/resources/example_db/dataset_version.lance/data/a13f68ba-04e6-48b5-bec0-bf54444be5f0.lance
+++ b/java/core/src/test/resources/example_db/dataset_version.lance/data/a13f68ba-04e6-48b5-bec0-bf54444be5f0.lance
--- a/java/core/src/test/resources/example_db/new_empty_dataset.lance/_latest.manifest
+++ b/java/core/src/test/resources/example_db/new_empty_dataset.lance/_latest.manifest
--- a/java/core/src/test/resources/example_db/new_empty_dataset.lance/_transactions/0-15648e72-076f-4ef1-8b90-10d305b95b3b.txn
+++ b/java/core/src/test/resources/example_db/new_empty_dataset.lance/_transactions/0-15648e72-076f-4ef1-8b90-10d305b95b3b.txn
@@ -1 +0,0 @@
-$15648e72-076f-4ef1-8b90-10d305b95b3b<33>=id <20><><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD>*int3208name <20><><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD>*string08
--- a/java/core/src/test/resources/example_db/new_empty_dataset.lance/_versions/1.manifest
+++ b/java/core/src/test/resources/example_db/new_empty_dataset.lance/_versions/1.manifest
--- a/java/core/src/test/resources/example_db/test.lance/_latest.manifest
+++ b/java/core/src/test/resources/example_db/test.lance/_latest.manifest
--- a/java/core/src/test/resources/example_db/test.lance/_transactions/0-a3689caf-4f6b-4afc-a3c7-97af75661843.txn
+++ b/java/core/src/test/resources/example_db/test.lance/_transactions/0-a3689caf-4f6b-4afc-a3c7-97af75661843.txn
@@ -1 +0,0 @@
-$a3689caf-4f6b-4afc-a3c7-97af75661843<34>oitem <20><><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD>*string8price <20><><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD>*double80vector <20><><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD>*fixed_size_list:float:28
--- a/java/core/src/test/resources/example_db/test.lance/_transactions/1-3f0fa7b9-7311-4945-9b0f-57dff4c04ee2.txn
+++ b/java/core/src/test/resources/example_db/test.lance/_transactions/1-3f0fa7b9-7311-4945-9b0f-57dff4c04ee2.txn
--- a/java/core/src/test/resources/example_db/test.lance/_versions/1.manifest
+++ b/java/core/src/test/resources/example_db/test.lance/_versions/1.manifest
--- a/java/core/src/test/resources/example_db/test.lance/_versions/2.manifest
+++ b/java/core/src/test/resources/example_db/test.lance/_versions/2.manifest
--- a/java/core/src/test/resources/example_db/test.lance/data/cd209a1b-00e0-4adf-93b2-2547c866e1ef.lance
+++ b/java/core/src/test/resources/example_db/test.lance/data/cd209a1b-00e0-4adf-93b2-2547c866e1ef.lance
--- a/java/core/src/test/resources/example_db/write_stream.lance/_latest.manifest
+++ b/java/core/src/test/resources/example_db/write_stream.lance/_latest.manifest
--- a/java/core/src/test/resources/example_db/write_stream.lance/_transactions/0-ea2f0479-36d1-4302-908a-dae45b9eb443.txn
+++ b/java/core/src/test/resources/example_db/write_stream.lance/_transactions/0-ea2f0479-36d1-4302-908a-dae45b9eb443.txn
--- a/java/core/src/test/resources/example_db/write_stream.lance/_versions/1.manifest
+++ b/java/core/src/test/resources/example_db/write_stream.lance/_versions/1.manifest
--- a/java/core/src/test/resources/example_db/write_stream.lance/data/665ff491-6dc5-4496-b292-166ed5c2a309.lance
+++ b/java/core/src/test/resources/example_db/write_stream.lance/data/665ff491-6dc5-4496-b292-166ed5c2a309.lance
--- a/java/pom.xml
+++ b/java/pom.xml
@@ -1,129 +0,0 @@
-<?xml version="1.0" encoding="UTF-8"?>
-<project xmlns="http://maven.apache.org/POM/4.0.0"
-    xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
-    xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/xsd/maven-4.0.0.xsd">
-    <modelVersion>4.0.0</modelVersion>
-
-    <groupId>com.lancedb</groupId>
-    <artifactId>lancedb-parent</artifactId>
-    <version>0.1-SNAPSHOT</version>
-    <packaging>pom</packaging>
-
-    <name>Lance Parent</name>
-
-    <properties>
-        <project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
-        <maven.compiler.source>11</maven.compiler.source>
-        <maven.compiler.target>11</maven.compiler.target>
-        <arrow.version>15.0.0</arrow.version>
-    </properties>
-
-    <modules>
-        <module>core</module>
-    </modules>
-
-    <dependencyManagement>
-        <dependencies>
-            <dependency>
-                <groupId>org.apache.arrow</groupId>
-                <artifactId>arrow-vector</artifactId>
-                <version>${arrow.version}</version>
-            </dependency>
-            <dependency>
-                <groupId>org.apache.arrow</groupId>
-                <artifactId>arrow-memory-netty</artifactId>
-                <version>${arrow.version}</version>
-            </dependency>
-            <dependency>
-                <groupId>org.apache.arrow</groupId>
-                <artifactId>arrow-c-data</artifactId>
-                <version>${arrow.version}</version>
-            </dependency>
-            <dependency>
-                <groupId>org.apache.arrow</groupId>
-                <artifactId>arrow-dataset</artifactId>
-                <version>${arrow.version}</version>
-            </dependency>
-            <dependency>
-                <groupId>org.questdb</groupId>
-                <artifactId>jar-jni</artifactId>
-                <version>1.1.1</version>
-            </dependency>
-            <dependency>
-                <groupId>org.junit.jupiter</groupId>
-                <artifactId>junit-jupiter</artifactId>
-                <version>5.10.1</version>
-            </dependency>
-            <dependency>
-                <groupId>org.json</groupId>
-                <artifactId>json</artifactId>
-                <version>20210307</version>
-            </dependency>
-        </dependencies>
-    </dependencyManagement>
-
-     <build>
-        <plugins>
-            <plugin>
-                <groupId>org.apache.maven.plugins</groupId>
-                <artifactId>maven-checkstyle-plugin</artifactId>
-                <version>3.3.1</version>
-                <configuration>
-                    <configLocation>google_checks.xml</configLocation>
-                    <consoleOutput>true</consoleOutput>
-                    <failsOnError>true</failsOnError>
-                    <violationSeverity>warning</violationSeverity>
-                    <linkXRef>false</linkXRef>
-                </configuration>
-                <executions>
-                    <execution>
-                        <id>validate</id>
-                        <phase>validate</phase>
-                        <goals>
-                            <goal>check</goal>
-                        </goals>
-                    </execution>
-                </executions>
-            </plugin>
-        </plugins>
-        <pluginManagement>
-            <plugins>
-                <plugin>
-                    <artifactId>maven-clean-plugin</artifactId>
-                    <version>3.1.0</version>
-                </plugin>
-                <plugin>
-                    <artifactId>maven-resources-plugin</artifactId>
-                    <version>3.0.2</version>
-                </plugin>
-                <plugin>
-                    <artifactId>maven-compiler-plugin</artifactId>
-                    <version>3.8.1</version>
-                    <configuration>
-                        <compilerArgs>
-                            <arg>-h</arg>
-                            <arg>target/headers</arg>
-                        </compilerArgs>
-                    </configuration>
-                </plugin>
-                <plugin>
-                    <artifactId>maven-surefire-plugin</artifactId>
-                    <version>3.2.5</version>
-                    <configuration>
-                        <argLine>--add-opens=java.base/java.nio=ALL-UNNAMED</argLine>
-                        <forkNode implementation="org.apache.maven.plugin.surefire.extensions.SurefireForkNodeFactory"/>
-                        <useSystemClassLoader>false</useSystemClassLoader>
-                    </configuration>
-                </plugin>
-                <plugin>
-                    <artifactId>maven-jar-plugin</artifactId>
-                    <version>3.0.2</version>
-                </plugin>
-                <plugin>
-                    <artifactId>maven-install-plugin</artifactId>
-                    <version>2.5.2</version>
-                </plugin>
-            </plugins>
-        </pluginManagement>
-    </build>
-</project>
--- a/node/package-lock.json
+++ b/node/package-lock.json
@@ -1,12 +1,12 @@
 {
  "name": "vectordb",
-  "version": "0.6.0",
+  "version": "0.4.17",
  "lockfileVersion": 3,
  "requires": true,
  "packages": {
    "": {
      "name": "vectordb",
-      "version": "0.6.0",
+      "version": "0.4.17",
      "cpu": [
        "x64",
        "arm64"
@@ -52,11 +52,11 @@
        "uuid": "^9.0.0"
      },
      "optionalDependencies": {
-        "@lancedb/vectordb-darwin-arm64": "0.4.20",
-        "@lancedb/vectordb-darwin-x64": "0.4.20",
-        "@lancedb/vectordb-linux-arm64-gnu": "0.4.20",
-        "@lancedb/vectordb-linux-x64-gnu": "0.4.20",
-        "@lancedb/vectordb-win32-x64-msvc": "0.4.20"
+        "@lancedb/vectordb-darwin-arm64": "0.4.17",
+        "@lancedb/vectordb-darwin-x64": "0.4.17",
+        "@lancedb/vectordb-linux-arm64-gnu": "0.4.17",
+        "@lancedb/vectordb-linux-x64-gnu": "0.4.17",
+        "@lancedb/vectordb-win32-x64-msvc": "0.4.17"
      },
      "peerDependencies": {
        "@apache-arrow/ts": "^14.0.2",
@@ -333,66 +333,6 @@
        "@jridgewell/sourcemap-codec": "^1.4.10"
      }
    },
-    "node_modules/@lancedb/vectordb-darwin-arm64": {
-      "version": "0.4.20",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-darwin-arm64/-/vectordb-darwin-arm64-0.4.20.tgz",
-      "integrity": "sha512-ffP2K4sA5mQTgePyARw1y8dPN996FmpvyAYoWO+TSItaXlhcXvc+KVa5udNMCZMDYeEnEv2Xpj6k4PwW3oBz+A==",
-      "cpu": [
-        "arm64"
-      ],
-      "optional": true,
-      "os": [
-        "darwin"
-      ]
-    },
-    "node_modules/@lancedb/vectordb-darwin-x64": {
-      "version": "0.4.20",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-darwin-x64/-/vectordb-darwin-x64-0.4.20.tgz",
-      "integrity": "sha512-GSYsXE20RIehDu30FjREhJdEzhnwOTV7ZsrSXagStzLY1gr7pyd7sfqxmmUtdD09di7LnQoiM71AOpPTa01YwQ==",
-      "cpu": [
-        "x64"
-      ],
-      "optional": true,
-      "os": [
-        "darwin"
-      ]
-    },
-    "node_modules/@lancedb/vectordb-linux-arm64-gnu": {
-      "version": "0.4.20",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-linux-arm64-gnu/-/vectordb-linux-arm64-gnu-0.4.20.tgz",
-      "integrity": "sha512-FpNOjOsz3nJVm6EBGyNgbOW2aFhsWZ/igeY45Z8hbZaaK2YBwrg/DASoNlUzgv6IR8cUaGJ2irNVJfsKR2cG6g==",
-      "cpu": [
-        "arm64"
-      ],
-      "optional": true,
-      "os": [
-        "linux"
-      ]
-    },
-    "node_modules/@lancedb/vectordb-linux-x64-gnu": {
-      "version": "0.4.20",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-linux-x64-gnu/-/vectordb-linux-x64-gnu-0.4.20.tgz",
-      "integrity": "sha512-pOqWjrRZQSrLTlQPkjidRii7NZDw8Xu9pN6ouVu2JAK8n81FXaPtFCyAI+Y3v9GpnYDN0rvD4eQ36aHAVPsa2g==",
-      "cpu": [
-        "x64"
-      ],
-      "optional": true,
-      "os": [
-        "linux"
-      ]
-    },
-    "node_modules/@lancedb/vectordb-win32-x64-msvc": {
-      "version": "0.4.20",
-      "resolved": "https://registry.npmjs.org/@lancedb/vectordb-win32-x64-msvc/-/vectordb-win32-x64-msvc-0.4.20.tgz",
-      "integrity": "sha512-5J5SsYSJ7jRCmU/sgwVHdrGz43B/7R2T9OEoFTKyVAtqTZdu75rkytXyn9SyEayXVhlUOaw76N0ASm0hAoDS/A==",
-      "cpu": [
-        "x64"
-      ],
-      "optional": true,
-      "os": [
-        "win32"
-      ]
-    },
    "node_modules/@neon-rs/cli": {
      "version": "0.0.160",
      "resolved": "https://registry.npmjs.org/@neon-rs/cli/-/cli-0.0.160.tgz",
--- a/node/package.json
+++ b/node/package.json
@@ -1,12 +1,12 @@
 {
  "name": "vectordb",
-  "version": "0.6.0",
+  "version": "0.4.17",
  "description": " Serverless, low-latency vector database for AI applications",
  "main": "dist/index.js",
  "types": "dist/index.d.ts",
  "scripts": {
    "tsc": "tsc -b",
-    "build": "npm run tsc && cargo-cp-artifact --artifact cdylib lancedb_node index.node -- cargo build -p lancedb-node --message-format=json",
+    "build": "npm run tsc && cargo-cp-artifact --artifact cdylib lancedb-node index.node -- cargo build --message-format=json",
    "build-release": "npm run build -- --release",
    "test": "npm run tsc && mocha -recursive dist/test",
    "integration-test": "npm run tsc && mocha -recursive dist/integration_test",
@@ -88,10 +88,10 @@
    }
  },
  "optionalDependencies": {
-    "@lancedb/vectordb-darwin-arm64": "0.4.20",
-    "@lancedb/vectordb-darwin-x64": "0.4.20",
-    "@lancedb/vectordb-linux-arm64-gnu": "0.4.20",
-    "@lancedb/vectordb-linux-x64-gnu": "0.4.20",
-    "@lancedb/vectordb-win32-x64-msvc": "0.4.20"
+    "@lancedb/vectordb-darwin-arm64": "0.4.17",
+    "@lancedb/vectordb-darwin-x64": "0.4.17",
+    "@lancedb/vectordb-linux-arm64-gnu": "0.4.17",
+    "@lancedb/vectordb-linux-x64-gnu": "0.4.17",
+    "@lancedb/vectordb-win32-x64-msvc": "0.4.17"
  }
 }
--- a/node/src/arrow.ts
+++ b/node/src/arrow.ts
@@ -27,23 +27,23 @@ import {
  RecordBatch,
  makeData,
  Struct,
-  type Float,
+  Float,
  DataType,
  Binary,
  Float32
-} from "apache-arrow";
-import { type EmbeddingFunction } from "./index";
-import { sanitizeSchema } from "./sanitize";
+} from 'apache-arrow'
+import { type EmbeddingFunction } from './index'
+import { sanitizeSchema } from './sanitize'

 /*
 * Options to control how a column should be converted to a vector array
 */
 export class VectorColumnOptions {
  /** Vector column type. */
-  type: Float = new Float32();
+  type: Float = new Float32()

-  constructor(values?: Partial<VectorColumnOptions>) {
-    Object.assign(this, values);
+  constructor (values?: Partial<VectorColumnOptions>) {
+    Object.assign(this, values)
  }
 }

@@ -60,7 +60,7 @@ export class MakeArrowTableOptions {
   * The schema must be specified if there are no records (e.g. to make
   * an empty table)
   */
-  schema?: Schema;
+  schema?: Schema

  /*
   * Mapping from vector column name to expected type
@@ -80,9 +80,7 @@ export class MakeArrowTableOptions {
   */
  vectorColumns: Record<string, VectorColumnOptions> = {
    vector: new VectorColumnOptions()
-  };
-
-  embeddings?: EmbeddingFunction<any>;
+  }

  /**
   * If true then string columns will be encoded with dictionary encoding
@@ -93,10 +91,10 @@ export class MakeArrowTableOptions {
   *
   * If `schema` is provided then this property is ignored.
   */
-  dictionaryEncodeStrings: boolean = false;
+  dictionaryEncodeStrings: boolean = false

-  constructor(values?: Partial<MakeArrowTableOptions>) {
-    Object.assign(this, values);
+  constructor (values?: Partial<MakeArrowTableOptions>) {
+    Object.assign(this, values)
  }
 }

@@ -195,68 +193,59 @@ export class MakeArrowTableOptions {
 * assert.deepEqual(table.schema, schema)
 * ```
 */
-export function makeArrowTable(
+export function makeArrowTable (
  data: Array<Record<string, any>>,
  options?: Partial<MakeArrowTableOptions>
 ): ArrowTable {
-  if (
-    data.length === 0 &&
-    (options?.schema === undefined || options?.schema === null)
-  ) {
-    throw new Error("At least one record or a schema needs to be provided");
+  if (data.length === 0 && (options?.schema === undefined || options?.schema === null)) {
+    throw new Error('At least one record or a schema needs to be provided')
  }

-  const opt = new MakeArrowTableOptions(options !== undefined ? options : {});
+  const opt = new MakeArrowTableOptions(options !== undefined ? options : {})
  if (opt.schema !== undefined && opt.schema !== null) {
-    opt.schema = sanitizeSchema(opt.schema);
-    opt.schema = validateSchemaEmbeddings(opt.schema, data, opt.embeddings);
+    opt.schema = sanitizeSchema(opt.schema)
  }
-
-  const columns: Record<string, Vector> = {};
+  const columns: Record<string, Vector> = {}
  // TODO: sample dataset to find missing columns
  // Prefer the field ordering of the schema, if present
-  const columnNames =
-    opt.schema != null ? (opt.schema.names as string[]) : Object.keys(data[0]);
+  const columnNames = ((opt.schema) != null) ? (opt.schema.names as string[]) : Object.keys(data[0])
  for (const colName of columnNames) {
-    if (
-      data.length !== 0 &&
-      !Object.prototype.hasOwnProperty.call(data[0], colName)
-    ) {
+    if (data.length !== 0 && !Object.prototype.hasOwnProperty.call(data[0], colName)) {
      // The field is present in the schema, but not in the data, skip it
-      continue;
+      continue
    }
    // Extract a single column from the records (transpose from row-major to col-major)
-    let values = data.map((datum) => datum[colName]);
+    let values = data.map((datum) => datum[colName])

    // By default (type === undefined) arrow will infer the type from the JS type
-    let type;
+    let type
    if (opt.schema !== undefined) {
      // If there is a schema provided, then use that for the type instead
-      type = opt.schema?.fields.filter((f) => f.name === colName)[0]?.type;
+      type = opt.schema?.fields.filter((f) => f.name === colName)[0]?.type
      if (DataType.isInt(type) && type.bitWidth === 64) {
        // wrap in BigInt to avoid bug: https://github.com/apache/arrow/issues/40051
        values = values.map((v) => {
          if (v === null) {
-            return v;
+            return v
          }
-          return BigInt(v);
-        });
+          return BigInt(v)
+        })
      }
    } else {
      // Otherwise, check to see if this column is one of the vector columns
      // defined by opt.vectorColumns and, if so, use the fixed size list type
-      const vectorColumnOptions = opt.vectorColumns[colName];
+      const vectorColumnOptions = opt.vectorColumns[colName]
      if (vectorColumnOptions !== undefined) {
-        type = newVectorType(values[0].length, vectorColumnOptions.type);
+        type = newVectorType(values[0].length, vectorColumnOptions.type)
      }
    }

    try {
      // Convert an Array of JS values to an arrow vector
-      columns[colName] = makeVector(values, type, opt.dictionaryEncodeStrings);
+      columns[colName] = makeVector(values, type, opt.dictionaryEncodeStrings)
    } catch (error: unknown) {
      // eslint-disable-next-line @typescript-eslint/restrict-template-expressions
-      throw Error(`Could not convert column "${colName}" to Arrow: ${error}`);
+      throw Error(`Could not convert column "${colName}" to Arrow: ${error}`)
    }
  }

@@ -271,116 +260,97 @@ export function makeArrowTable(
    // To work around this we first create a table with the wrong schema and
    // then patch the schema of the batches so we can use
    // `new ArrowTable(schema, batches)` which does not do any schema inference
-    const firstTable = new ArrowTable(columns);
-    const batchesFixed = firstTable.batches.map(
-      // eslint-disable-next-line @typescript-eslint/no-non-null-assertion
-      (batch) => new RecordBatch(opt.schema!, batch.data)
-    );
-    return new ArrowTable(opt.schema, batchesFixed);
+    const firstTable = new ArrowTable(columns)
+    // eslint-disable-next-line @typescript-eslint/no-non-null-assertion
+    const batchesFixed = firstTable.batches.map(batch => new RecordBatch(opt.schema!, batch.data))
+    return new ArrowTable(opt.schema, batchesFixed)
  } else {
-    return new ArrowTable(columns);
+    return new ArrowTable(columns)
  }
 }

 /**
 * Create an empty Arrow table with the provided schema
 */
-export function makeEmptyTable(schema: Schema): ArrowTable {
-  return makeArrowTable([], { schema });
+export function makeEmptyTable (schema: Schema): ArrowTable {
+  return makeArrowTable([], { schema })
 }

 // Helper function to convert Array<Array<any>> to a variable sized list array
-function makeListVector(lists: any[][]): Vector<any> {
+function makeListVector (lists: any[][]): Vector<any> {
  if (lists.length === 0 || lists[0].length === 0) {
-    throw Error("Cannot infer list vector from empty array or empty list");
+    throw Error('Cannot infer list vector from empty array or empty list')
  }
-  const sampleList = lists[0];
-  let inferredType;
+  const sampleList = lists[0]
+  let inferredType
  try {
-    const sampleVector = makeVector(sampleList);
-    inferredType = sampleVector.type;
+    const sampleVector = makeVector(sampleList)
+    inferredType = sampleVector.type
  } catch (error: unknown) {
    // eslint-disable-next-line @typescript-eslint/restrict-template-expressions
-    throw Error(`Cannot infer list vector.  Cannot infer inner type: ${error}`);
+    throw Error(`Cannot infer list vector.  Cannot infer inner type: ${error}`)
  }

  const listBuilder = makeBuilder({
-    type: new List(new Field("item", inferredType, true))
-  });
+    type: new List(new Field('item', inferredType, true))
+  })
  for (const list of lists) {
-    listBuilder.append(list);
+    listBuilder.append(list)
  }
-  return listBuilder.finish().toVector();
+  return listBuilder.finish().toVector()
 }

 // Helper function to convert an Array of JS values to an Arrow Vector
-function makeVector(
-  values: any[],
-  type?: DataType,
-  stringAsDictionary?: boolean
-): Vector<any> {
+function makeVector (values: any[], type?: DataType, stringAsDictionary?: boolean): Vector<any> {
  if (type !== undefined) {
    // No need for inference, let Arrow create it
-    return vectorFromArray(values, type);
+    return vectorFromArray(values, type)
  }
  if (values.length === 0) {
-    throw Error(
-      "makeVector requires at least one value or the type must be specfied"
-    );
+    throw Error('makeVector requires at least one value or the type must be specfied')
  }
-  const sampleValue = values.find((val) => val !== null && val !== undefined);
+  const sampleValue = values.find(val => val !== null && val !== undefined)
  if (sampleValue === undefined) {
-    throw Error(
-      "makeVector cannot infer the type if all values are null or undefined"
-    );
+    throw Error('makeVector cannot infer the type if all values are null or undefined')
  }
  if (Array.isArray(sampleValue)) {
    // Default Arrow inference doesn't handle list types
-    return makeListVector(values);
+    return makeListVector(values)
  } else if (Buffer.isBuffer(sampleValue)) {
    // Default Arrow inference doesn't handle Buffer
-    return vectorFromArray(values, new Binary());
-  } else if (
-    !(stringAsDictionary ?? false) &&
-    (typeof sampleValue === "string" || sampleValue instanceof String)
-  ) {
+    return vectorFromArray(values, new Binary())
+  } else if (!(stringAsDictionary ?? false) && (typeof sampleValue === 'string' || sampleValue instanceof String)) {
    // If the type is string then don't use Arrow's default inference unless dictionaries are requested
    // because it will always use dictionary encoding for strings
-    return vectorFromArray(values, new Utf8());
+    return vectorFromArray(values, new Utf8())
  } else {
    // Convert a JS array of values to an arrow vector
-    return vectorFromArray(values);
+    return vectorFromArray(values)
  }
 }

-async function applyEmbeddings<T>(
-  table: ArrowTable,
-  embeddings?: EmbeddingFunction<T>,
-  schema?: Schema
-): Promise<ArrowTable> {
+async function applyEmbeddings<T> (table: ArrowTable, embeddings?: EmbeddingFunction<T>, schema?: Schema): Promise<ArrowTable> {
  if (embeddings == null) {
-    return table;
+    return table
  }
  if (schema !== undefined && schema !== null) {
-    schema = sanitizeSchema(schema);
+    schema = sanitizeSchema(schema)
  }

  // Convert from ArrowTable to Record<String, Vector>
  const colEntries = [...Array(table.numCols).keys()].map((_, idx) => {
-    const name = table.schema.fields[idx].name;
+    const name = table.schema.fields[idx].name
    // eslint-disable-next-line @typescript-eslint/no-non-null-assertion
-    const vec = table.getChildAt(idx)!;
-    return [name, vec];
-  });
-  const newColumns = Object.fromEntries(colEntries);
+    const vec = table.getChildAt(idx)!
+    return [name, vec]
+  })
+  const newColumns = Object.fromEntries(colEntries)

-  const sourceColumn = newColumns[embeddings.sourceColumn];
-  const destColumn = embeddings.destColumn ?? "vector";
-  const innerDestType = embeddings.embeddingDataType ?? new Float32();
+  const sourceColumn = newColumns[embeddings.sourceColumn]
+  const destColumn = embeddings.destColumn ?? 'vector'
+  const innerDestType = embeddings.embeddingDataType ?? new Float32()
  if (sourceColumn === undefined) {
-    throw new Error(
-      `Cannot apply embedding function because the source column '${embeddings.sourceColumn}' was not present in the data`
-    );
+    throw new Error(`Cannot apply embedding function because the source column '${embeddings.sourceColumn}' was not present in the data`)
  }

  if (table.numRows === 0) {
@@ -388,60 +358,45 @@ async function applyEmbeddings<T>(
      // We have an empty table and it already has the embedding column so no work needs to be done
      // Note: we don't return an error like we did below because this is a common occurrence.  For example,
      // if we call convertToTable with 0 records and a schema that includes the embedding
-      return table;
+      return table
    }
    if (embeddings.embeddingDimension !== undefined) {
-      const destType = newVectorType(
-        embeddings.embeddingDimension,
-        innerDestType
-      );
-      newColumns[destColumn] = makeVector([], destType);
+      const destType = newVectorType(embeddings.embeddingDimension, innerDestType)
+      newColumns[destColumn] = makeVector([], destType)
    } else if (schema != null) {
-      const destField = schema.fields.find((f) => f.name === destColumn);
+      const destField = schema.fields.find(f => f.name === destColumn)
      if (destField != null) {
-        newColumns[destColumn] = makeVector([], destField.type);
+        newColumns[destColumn] = makeVector([], destField.type)
      } else {
-        throw new Error(
-          `Attempt to apply embeddings to an empty table failed because schema was missing embedding column '${destColumn}'`
-        );
+        throw new Error(`Attempt to apply embeddings to an empty table failed because schema was missing embedding column '${destColumn}'`)
      }
    } else {
-      throw new Error(
-        "Attempt to apply embeddings to an empty table when the embeddings function does not specify `embeddingDimension`"
-      );
+      throw new Error('Attempt to apply embeddings to an empty table when the embeddings function does not specify `embeddingDimension`')
    }
  } else {
    if (Object.prototype.hasOwnProperty.call(newColumns, destColumn)) {
-      throw new Error(
-        `Attempt to apply embeddings to table failed because column ${destColumn} already existed`
-      );
+      throw new Error(`Attempt to apply embeddings to table failed because column ${destColumn} already existed`)
    }
    if (table.batches.length > 1) {
-      throw new Error(
-        "Internal error: `makeArrowTable` unexpectedly created a table with more than one batch"
-      );
+      throw new Error('Internal error: `makeArrowTable` unexpectedly created a table with more than one batch')
    }
-    const values = sourceColumn.toArray();
-    const vectors = await embeddings.embed(values as T[]);
+    const values = sourceColumn.toArray()
+    const vectors = await embeddings.embed(values as T[])
    if (vectors.length !== values.length) {
-      throw new Error(
-        "Embedding function did not return an embedding for each input element"
-      );
+      throw new Error('Embedding function did not return an embedding for each input element')
    }
-    const destType = newVectorType(vectors[0].length, innerDestType);
-    newColumns[destColumn] = makeVector(vectors, destType);
+    const destType = newVectorType(vectors[0].length, innerDestType)
+    newColumns[destColumn] = makeVector(vectors, destType)
  }

-  const newTable = new ArrowTable(newColumns);
+  const newTable = new ArrowTable(newColumns)
  if (schema != null) {
-    if (schema.fields.find((f) => f.name === destColumn) === undefined) {
-      throw new Error(
-        `When using embedding functions and specifying a schema the schema should include the embedding column but the column ${destColumn} was missing`
-      );
+    if (schema.fields.find(f => f.name === destColumn) === undefined) {
+      throw new Error(`When using embedding functions and specifying a schema the schema should include the embedding column but the column ${destColumn} was missing`)
    }
-    return alignTable(newTable, schema);
+    return alignTable(newTable, schema)
  }
-  return newTable;
+  return newTable
 }

 /*
@@ -462,24 +417,21 @@ async function applyEmbeddings<T>(
 * embedding columns.  If no schema is provded then embedding columns will
 * be placed at the end of the table, after all of the input columns.
 */
-export async function convertToTable<T>(
+export async function convertToTable<T> (
  data: Array<Record<string, unknown>>,
  embeddings?: EmbeddingFunction<T>,
  makeTableOptions?: Partial<MakeArrowTableOptions>
 ): Promise<ArrowTable> {
-  const table = makeArrowTable(data, makeTableOptions);
-  return await applyEmbeddings(table, embeddings, makeTableOptions?.schema);
+  const table = makeArrowTable(data, makeTableOptions)
+  return await applyEmbeddings(table, embeddings, makeTableOptions?.schema)
 }

 // Creates the Arrow Type for a Vector column with dimension `dim`
-function newVectorType<T extends Float>(
-  dim: number,
-  innerType: T
-): FixedSizeList<T> {
+function newVectorType <T extends Float> (dim: number, innerType: T): FixedSizeList<T> {
  // Somewhere we always default to have the elements nullable, so we need to set it to true
  // otherwise we often get schema mismatches because the stored data always has schema with nullable elements
-  const children = new Field<T>("item", innerType, true);
-  return new FixedSizeList(dim, children);
+  const children = new Field<T>('item', innerType, true)
+  return new FixedSizeList(dim, children)
 }

 /**
@@ -489,17 +441,17 @@ function newVectorType<T extends Float>(
 *
 * `schema` is required if data is empty
 */
-export async function fromRecordsToBuffer<T>(
+export async function fromRecordsToBuffer<T> (
  data: Array<Record<string, unknown>>,
  embeddings?: EmbeddingFunction<T>,
  schema?: Schema
 ): Promise<Buffer> {
  if (schema !== undefined && schema !== null) {
-    schema = sanitizeSchema(schema);
+    schema = sanitizeSchema(schema)
  }
-  const table = await convertToTable(data, embeddings, { schema, embeddings });
-  const writer = RecordBatchFileWriter.writeAll(table);
-  return Buffer.from(await writer.toUint8Array());
+  const table = await convertToTable(data, embeddings, { schema })
+  const writer = RecordBatchFileWriter.writeAll(table)
+  return Buffer.from(await writer.toUint8Array())
 }

 /**
@@ -509,17 +461,17 @@ export async function fromRecordsToBuffer<T>(
 *
 * `schema` is required if data is empty
 */
-export async function fromRecordsToStreamBuffer<T>(
+export async function fromRecordsToStreamBuffer<T> (
  data: Array<Record<string, unknown>>,
  embeddings?: EmbeddingFunction<T>,
  schema?: Schema
 ): Promise<Buffer> {
  if (schema !== null && schema !== undefined) {
-    schema = sanitizeSchema(schema);
+    schema = sanitizeSchema(schema)
  }
-  const table = await convertToTable(data, embeddings, { schema });
-  const writer = RecordBatchStreamWriter.writeAll(table);
-  return Buffer.from(await writer.toUint8Array());
+  const table = await convertToTable(data, embeddings, { schema })
+  const writer = RecordBatchStreamWriter.writeAll(table)
+  return Buffer.from(await writer.toUint8Array())
 }

 /**
@@ -530,17 +482,17 @@ export async function fromRecordsToStreamBuffer<T>(
 *
 * `schema` is required if the table is empty
 */
-export async function fromTableToBuffer<T>(
+export async function fromTableToBuffer<T> (
  table: ArrowTable,
  embeddings?: EmbeddingFunction<T>,
  schema?: Schema
 ): Promise<Buffer> {
  if (schema !== null && schema !== undefined) {
-    schema = sanitizeSchema(schema);
+    schema = sanitizeSchema(schema)
  }
-  const tableWithEmbeddings = await applyEmbeddings(table, embeddings, schema);
-  const writer = RecordBatchFileWriter.writeAll(tableWithEmbeddings);
-  return Buffer.from(await writer.toUint8Array());
+  const tableWithEmbeddings = await applyEmbeddings(table, embeddings, schema)
+  const writer = RecordBatchFileWriter.writeAll(tableWithEmbeddings)
+  return Buffer.from(await writer.toUint8Array())
 }

 /**
@@ -551,85 +503,49 @@ export async function fromTableToBuffer<T>(
 *
 * `schema` is required if the table is empty
 */
-export async function fromTableToStreamBuffer<T>(
+export async function fromTableToStreamBuffer<T> (
  table: ArrowTable,
  embeddings?: EmbeddingFunction<T>,
  schema?: Schema
 ): Promise<Buffer> {
  if (schema !== null && schema !== undefined) {
-    schema = sanitizeSchema(schema);
+    schema = sanitizeSchema(schema)
  }
-  const tableWithEmbeddings = await applyEmbeddings(table, embeddings, schema);
-  const writer = RecordBatchStreamWriter.writeAll(tableWithEmbeddings);
-  return Buffer.from(await writer.toUint8Array());
+  const tableWithEmbeddings = await applyEmbeddings(table, embeddings, schema)
+  const writer = RecordBatchStreamWriter.writeAll(tableWithEmbeddings)
+  return Buffer.from(await writer.toUint8Array())
 }

-function alignBatch(batch: RecordBatch, schema: Schema): RecordBatch {
-  const alignedChildren = [];
+function alignBatch (batch: RecordBatch, schema: Schema): RecordBatch {
+  const alignedChildren = []
  for (const field of schema.fields) {
    const indexInBatch = batch.schema.fields?.findIndex(
      (f) => f.name === field.name
-    );
+    )
    if (indexInBatch < 0) {
      throw new Error(
        `The column ${field.name} was not found in the Arrow Table`
-      );
+      )
    }
-    alignedChildren.push(batch.data.children[indexInBatch]);
+    alignedChildren.push(batch.data.children[indexInBatch])
  }
  const newData = makeData({
    type: new Struct(schema.fields),
    length: batch.numRows,
    nullCount: batch.nullCount,
    children: alignedChildren
-  });
-  return new RecordBatch(schema, newData);
+  })
+  return new RecordBatch(schema, newData)
 }

-function alignTable(table: ArrowTable, schema: Schema): ArrowTable {
+function alignTable (table: ArrowTable, schema: Schema): ArrowTable {
  const alignedBatches = table.batches.map((batch) =>
    alignBatch(batch, schema)
-  );
-  return new ArrowTable(schema, alignedBatches);
+  )
+  return new ArrowTable(schema, alignedBatches)
 }

 // Creates an empty Arrow Table
-export function createEmptyTable(schema: Schema): ArrowTable {
-  return new ArrowTable(sanitizeSchema(schema));
-}
-
-function validateSchemaEmbeddings(
-  schema: Schema<any>,
-  data: Array<Record<string, unknown>>,
-  embeddings: EmbeddingFunction<any> | undefined
-) {
-  const fields = [];
-  const missingEmbeddingFields = [];
-
-  // First we check if the field is a `FixedSizeList`
-  // Then we check if the data contains the field
-  // if it does not, we add it to the list of missing embedding fields
-  // Finally, we check if those missing embedding fields are `this._embeddings`
-  // if they are not, we throw an error
-  for (const field of schema.fields) {
-    if (field.type instanceof FixedSizeList) {
-      if (data.length !== 0 && data?.[0]?.[field.name] === undefined) {
-        missingEmbeddingFields.push(field);
-      } else {
-        fields.push(field);
-      }
-    } else {
-      fields.push(field);
-    }
-  }
-
-  if (missingEmbeddingFields.length > 0 && embeddings === undefined) {
-    throw new Error(
-      `Table has embeddings: "${missingEmbeddingFields
-        .map((f) => f.name)
-        .join(",")}", but no embedding function was provided`
-    );
-  }
-
-  return new Schema(fields, schema.metadata);
+export function createEmptyTable (schema: Schema): ArrowTable {
+  return new ArrowTable(sanitizeSchema(schema))
 }
--- a/node/src/index.ts
+++ b/node/src/index.ts
@@ -12,20 +12,19 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-import { type Schema, Table as ArrowTable, tableFromIPC } from "apache-arrow";
+import { type Schema, Table as ArrowTable, tableFromIPC } from 'apache-arrow'
 import {
  createEmptyTable,
  fromRecordsToBuffer,
  fromTableToBuffer,
  makeArrowTable
-} from "./arrow";
-import type { EmbeddingFunction } from "./embedding/embedding_function";
-import { RemoteConnection } from "./remote";
-import { Query } from "./query";
-import { isEmbeddingFunction } from "./embedding/embedding_function";
-import { type Literal, toSQL } from "./util";
-
-import { type HttpMiddleware } from "./middleware";
+} from './arrow'
+import type { EmbeddingFunction } from './embedding/embedding_function'
+import { RemoteConnection } from './remote'
+import { Query } from './query'
+import { isEmbeddingFunction } from './embedding/embedding_function'
+import { type Literal, toSQL } from './util'
+import { type HttpMiddleware } from './middleware'

 const {
  databaseNew,
@@ -49,18 +48,14 @@ const {
  tableAlterColumns,
  tableDropColumns
  // eslint-disable-next-line @typescript-eslint/no-var-requires
-} = require("../native.js");
+} = require('../native.js')

-export { Query };
-export type { EmbeddingFunction };
-export { OpenAIEmbeddingFunction } from "./embedding/openai";
-export {
-  convertToTable,
-  makeArrowTable,
-  type MakeArrowTableOptions
-} from "./arrow";
+export { Query }
+export type { EmbeddingFunction }
+export { OpenAIEmbeddingFunction } from './embedding/openai'
+export { convertToTable, makeArrowTable, type MakeArrowTableOptions } from './arrow'

-const defaultAwsRegion = "us-west-2";
+const defaultAwsRegion = 'us-west-2'

 export interface AwsCredentials {
  accessKeyId: string
@@ -133,19 +128,19 @@ export interface ConnectionOptions {
  readConsistencyInterval?: number
 }

-function getAwsArgs(opts: ConnectionOptions): any[] {
-  const callArgs: any[] = [];
-  const awsCredentials = opts.awsCredentials;
+function getAwsArgs (opts: ConnectionOptions): any[] {
+  const callArgs: any[] = []
+  const awsCredentials = opts.awsCredentials
  if (awsCredentials !== undefined) {
-    callArgs.push(awsCredentials.accessKeyId);
-    callArgs.push(awsCredentials.secretKey);
-    callArgs.push(awsCredentials.sessionToken);
+    callArgs.push(awsCredentials.accessKeyId)
+    callArgs.push(awsCredentials.secretKey)
+    callArgs.push(awsCredentials.sessionToken)
  } else {
-    callArgs.fill(undefined, 0, 3);
+    callArgs.fill(undefined, 0, 3)
  }

-  callArgs.push(opts.awsRegion);
-  return callArgs;
+  callArgs.push(opts.awsRegion)
+  return callArgs
 }

 export interface CreateTableOptions<T> {
@@ -178,56 +173,56 @@ export interface CreateTableOptions<T> {
 *
 * @see {@link ConnectionOptions} for more details on the URI format.
 */
-export async function connect(uri: string): Promise<Connection>;
+export async function connect (uri: string): Promise<Connection>
 /**
 * Connect to a LanceDB instance with connection options.
 *
 * @param opts The {@link ConnectionOptions} to use when connecting to the database.
 */
-export async function connect(
+export async function connect (
  opts: Partial<ConnectionOptions>
-): Promise<Connection>;
-export async function connect(
+): Promise<Connection>
+export async function connect (
  arg: string | Partial<ConnectionOptions>
 ): Promise<Connection> {
-  let opts: ConnectionOptions;
-  if (typeof arg === "string") {
-    opts = { uri: arg };
+  let opts: ConnectionOptions
+  if (typeof arg === 'string') {
+    opts = { uri: arg }
  } else {
-    const keys = Object.keys(arg);
-    if (keys.length === 1 && keys[0] === "uri" && typeof arg.uri === "string") {
-      opts = { uri: arg.uri };
+    const keys = Object.keys(arg)
+    if (keys.length === 1 && keys[0] === 'uri' && typeof arg.uri === 'string') {
+      opts = { uri: arg.uri }
    } else {
      opts = Object.assign(
        {
-          uri: "",
+          uri: '',
          awsCredentials: undefined,
          awsRegion: defaultAwsRegion,
          apiKey: undefined,
          region: defaultAwsRegion
        },
        arg
-      );
+      )
    }
  }

-  if (opts.uri.startsWith("db://")) {
+  if (opts.uri.startsWith('db://')) {
    // Remote connection
-    return new RemoteConnection(opts);
+    return new RemoteConnection(opts)
  }

  const storageOptions = opts.storageOptions ?? {};
  if (opts.awsCredentials?.accessKeyId !== undefined) {
-    storageOptions.aws_access_key_id = opts.awsCredentials.accessKeyId;
+    storageOptions.aws_access_key_id = opts.awsCredentials.accessKeyId
  }
  if (opts.awsCredentials?.secretKey !== undefined) {
-    storageOptions.aws_secret_access_key = opts.awsCredentials.secretKey;
+    storageOptions.aws_secret_access_key = opts.awsCredentials.secretKey
  }
  if (opts.awsCredentials?.sessionToken !== undefined) {
-    storageOptions.aws_session_token = opts.awsCredentials.sessionToken;
+    storageOptions.aws_session_token = opts.awsCredentials.sessionToken
  }
  if (opts.awsRegion !== undefined) {
-    storageOptions.region = opts.awsRegion;
+    storageOptions.region = opts.awsRegion
  }
  // It's a pain to pass a record to Rust, so we convert it to an array of key-value pairs
  const storageOptionsArr = Object.entries(storageOptions);
@@ -236,8 +231,8 @@ export async function connect(
    opts.uri,
    storageOptionsArr,
    opts.readConsistencyInterval
-  );
-  return new LocalConnection(db, opts);
+  )
+  return new LocalConnection(db, opts)
 }

 /**
@@ -538,11 +533,7 @@ export interface Table<T = number[]> {
   * @param data the new data to insert
   * @param args parameters controlling how the operation should behave
   */
-  mergeInsert: (
-    on: string,
-    data: Array<Record<string, unknown>> | ArrowTable,
-    args: MergeInsertArgs
-  ) => Promise<void>
+  mergeInsert: (on: string, data: Array<Record<string, unknown>> | ArrowTable, args: MergeInsertArgs) => Promise<void>

  /**
   * List the indicies on this table.
@@ -567,9 +558,7 @@ export interface Table<T = number[]> {
   *                            expressions will be evaluated for each row in the
   *                            table, and can reference existing columns in the table.
   */
-  addColumns(
-    newColumnTransforms: Array<{ name: string, valueSql: string }>
-  ): Promise<void>
+  addColumns(newColumnTransforms: Array<{ name: string, valueSql: string }>): Promise<void>

  /**
   * Alter the name or nullability of columns.
@@ -695,49 +684,38 @@ export interface MergeInsertArgs {
  whenNotMatchedBySourceDelete?: string | boolean
 }

-export enum IndexStatus {
-  Pending = "pending",
-  Indexing = "indexing",
-  Done = "done",
-  Failed = "failed"
-}
-
 export interface VectorIndex {
  columns: string[]
  name: string
  uuid: string
-  status: IndexStatus
 }

 export interface IndexStats {
  numIndexedRows: number | null
  numUnindexedRows: number | null
-  indexType: string | null
-  distanceType: string | null
-  completedAt: string | null
 }

 /**
 * A connection to a LanceDB database.
 */
 export class LocalConnection implements Connection {
-  private readonly _options: () => ConnectionOptions;
-  private readonly _db: any;
+  private readonly _options: () => ConnectionOptions
+  private readonly _db: any

-  constructor(db: any, options: ConnectionOptions) {
-    this._options = () => options;
-    this._db = db;
+  constructor (db: any, options: ConnectionOptions) {
+    this._options = () => options
+    this._db = db
  }

-  get uri(): string {
-    return this._options().uri;
+  get uri (): string {
+    return this._options().uri
  }

  /**
   * Get the names of all tables in the database.
   */
-  async tableNames(): Promise<string[]> {
-    return databaseTableNames.call(this._db);
+  async tableNames (): Promise<string[]> {
+    return databaseTableNames.call(this._db)
  }

  /**
@@ -745,7 +723,7 @@ export class LocalConnection implements Connection {
   *
   * @param name The name of the table.
   */
-  async openTable(name: string): Promise<Table>;
+  async openTable (name: string): Promise<Table>

  /**
   * Open a table in the database.
@@ -756,20 +734,23 @@ export class LocalConnection implements Connection {
  async openTable<T>(
    name: string,
    embeddings: EmbeddingFunction<T>
-  ): Promise<Table<T>>;
+  ): Promise<Table<T>>
  async openTable<T>(
    name: string,
    embeddings?: EmbeddingFunction<T>
-  ): Promise<Table<T>>;
+  ): Promise<Table<T>>
  async openTable<T>(
    name: string,
    embeddings?: EmbeddingFunction<T>
  ): Promise<Table<T>> {
-    const tbl = await databaseOpenTable.call(this._db, name);
+    const tbl = await databaseOpenTable.call(
+      this._db,
+      name,
+    )
    if (embeddings !== undefined) {
-      return new LocalTable(tbl, name, this._options(), embeddings);
+      return new LocalTable(tbl, name, this._options(), embeddings)
    } else {
-      return new LocalTable(tbl, name, this._options());
+      return new LocalTable(tbl, name, this._options())
    }
  }

@@ -779,32 +760,32 @@ export class LocalConnection implements Connection {
    optsOrEmbedding?: WriteOptions | EmbeddingFunction<T>,
    opt?: WriteOptions
  ): Promise<Table<T>> {
-    if (typeof name === "string") {
-      let writeOptions: WriteOptions = new DefaultWriteOptions();
+    if (typeof name === 'string') {
+      let writeOptions: WriteOptions = new DefaultWriteOptions()
      if (opt !== undefined && isWriteOptions(opt)) {
-        writeOptions = opt;
+        writeOptions = opt
      } else if (
        optsOrEmbedding !== undefined &&
        isWriteOptions(optsOrEmbedding)
      ) {
-        writeOptions = optsOrEmbedding;
+        writeOptions = optsOrEmbedding
      }

-      let embeddings: undefined | EmbeddingFunction<T>;
+      let embeddings: undefined | EmbeddingFunction<T>
      if (
        optsOrEmbedding !== undefined &&
        isEmbeddingFunction(optsOrEmbedding)
      ) {
-        embeddings = optsOrEmbedding;
+        embeddings = optsOrEmbedding
      }
      return await this.createTableImpl({
        name,
        data,
        embeddingFunction: embeddings,
        writeOptions
-      });
+      })
    }
-    return await this.createTableImpl(name);
+    return await this.createTableImpl(name)
  }

  private async createTableImpl<T>({
@@ -820,27 +801,27 @@ export class LocalConnection implements Connection {
    embeddingFunction?: EmbeddingFunction<T> | undefined
    writeOptions?: WriteOptions | undefined
  }): Promise<Table<T>> {
-    let buffer: Buffer;
+    let buffer: Buffer

-    function isEmpty(
+    function isEmpty (
      data: Array<Record<string, unknown>> | ArrowTable<any>
    ): boolean {
      if (data instanceof ArrowTable) {
-        return data.data.length === 0;
+        return data.data.length === 0
      }
-      return data.length === 0;
+      return data.length === 0
    }

    if (data === undefined || isEmpty(data)) {
      if (schema === undefined) {
-        throw new Error("Either data or schema needs to defined");
+        throw new Error('Either data or schema needs to defined')
      }
-      buffer = await fromTableToBuffer(createEmptyTable(schema));
+      buffer = await fromTableToBuffer(createEmptyTable(schema))
    } else if (data instanceof ArrowTable) {
-      buffer = await fromTableToBuffer(data, embeddingFunction, schema);
+      buffer = await fromTableToBuffer(data, embeddingFunction, schema)
    } else {
      // data is Array<Record<...>>
-      buffer = await fromRecordsToBuffer(data, embeddingFunction, schema);
+      buffer = await fromRecordsToBuffer(data, embeddingFunction, schema)
    }

    const tbl = await tableCreate.call(
@@ -849,11 +830,11 @@ export class LocalConnection implements Connection {
      buffer,
      writeOptions?.writeMode?.toString(),
      ...getAwsArgs(this._options())
-    );
+    )
    if (embeddingFunction !== undefined) {
-      return new LocalTable(tbl, name, this._options(), embeddingFunction);
+      return new LocalTable(tbl, name, this._options(), embeddingFunction)
    } else {
-      return new LocalTable(tbl, name, this._options());
+      return new LocalTable(tbl, name, this._options())
    }
  }

@@ -861,69 +842,69 @@ export class LocalConnection implements Connection {
   * Drop an existing table.
   * @param name The name of the table to drop.
   */
-  async dropTable(name: string): Promise<void> {
-    await databaseDropTable.call(this._db, name);
+  async dropTable (name: string): Promise<void> {
+    await databaseDropTable.call(this._db, name)
  }

-  withMiddleware(middleware: HttpMiddleware): Connection {
-    return this;
+  withMiddleware (middleware: HttpMiddleware): Connection {
+    return this
  }
 }

 export class LocalTable<T = number[]> implements Table<T> {
-  private _tbl: any;
-  private readonly _name: string;
-  private readonly _isElectron: boolean;
-  private readonly _embeddings?: EmbeddingFunction<T>;
-  private readonly _options: () => ConnectionOptions;
+  private _tbl: any
+  private readonly _name: string
+  private readonly _isElectron: boolean
+  private readonly _embeddings?: EmbeddingFunction<T>
+  private readonly _options: () => ConnectionOptions

-  constructor(tbl: any, name: string, options: ConnectionOptions);
+  constructor (tbl: any, name: string, options: ConnectionOptions)
  /**
   * @param tbl
   * @param name
   * @param options
   * @param embeddings An embedding function to use when interacting with this table
   */
-  constructor(
+  constructor (
    tbl: any,
    name: string,
    options: ConnectionOptions,
    embeddings: EmbeddingFunction<T>
-  );
-  constructor(
+  )
+  constructor (
    tbl: any,
    name: string,
    options: ConnectionOptions,
    embeddings?: EmbeddingFunction<T>
  ) {
-    this._tbl = tbl;
-    this._name = name;
-    this._embeddings = embeddings;
-    this._options = () => options;
-    this._isElectron = this.checkElectron();
+    this._tbl = tbl
+    this._name = name
+    this._embeddings = embeddings
+    this._options = () => options
+    this._isElectron = this.checkElectron()
  }

-  get name(): string {
-    return this._name;
+  get name (): string {
+    return this._name
  }

  /**
   * Creates a search query to find the nearest neighbors of the given search term
   * @param query The query search term
   */
-  search(query: T): Query<T> {
-    return new Query(query, this._tbl, this._embeddings);
+  search (query: T): Query<T> {
+    return new Query(query, this._tbl, this._embeddings)
  }

  /**
   * Creates a filter query to find all rows matching the specified criteria
   * @param value The filter criteria (like SQL where clause syntax)
   */
-  filter(value: string): Query<T> {
-    return new Query(undefined, this._tbl, this._embeddings).filter(value);
+  filter (value: string): Query<T> {
+    return new Query(undefined, this._tbl, this._embeddings).filter(value)
  }

-  where = this.filter;
+  where = this.filter

  /**
   * Insert records into this Table.
@@ -931,19 +912,16 @@ export class LocalTable<T = number[]> implements Table<T> {
   * @param data Records to be inserted into the Table
   * @return The number of rows added to the table
   */
-  async add(
+  async add (
    data: Array<Record<string, unknown>> | ArrowTable
  ): Promise<number> {
-    const schema = await this.schema;
-
-    let tbl: ArrowTable;
-
+    const schema = await this.schema
+    let tbl: ArrowTable
    if (data instanceof ArrowTable) {
-      tbl = data;
+      tbl = data
    } else {
-      tbl = makeArrowTable(data, { schema, embeddings: this._embeddings });
+      tbl = makeArrowTable(data, { schema })
    }
-
    return tableAdd
      .call(
        this._tbl,
@@ -952,8 +930,8 @@ export class LocalTable<T = number[]> implements Table<T> {
        ...getAwsArgs(this._options())
      )
      .then((newTable: any) => {
-        this._tbl = newTable;
-      });
+        this._tbl = newTable
+      })
  }

  /**
@@ -962,14 +940,14 @@ export class LocalTable<T = number[]> implements Table<T> {
   * @param data Records to be inserted into the Table
   * @return The number of rows added to the table
   */
-  async overwrite(
+  async overwrite (
    data: Array<Record<string, unknown>> | ArrowTable
  ): Promise<number> {
-    let buffer: Buffer;
+    let buffer: Buffer
    if (data instanceof ArrowTable) {
-      buffer = await fromTableToBuffer(data, this._embeddings);
+      buffer = await fromTableToBuffer(data, this._embeddings)
    } else {
-      buffer = await fromRecordsToBuffer(data, this._embeddings);
+      buffer = await fromRecordsToBuffer(data, this._embeddings)
    }
    return tableAdd
      .call(
@@ -979,8 +957,8 @@ export class LocalTable<T = number[]> implements Table<T> {
        ...getAwsArgs(this._options())
      )
      .then((newTable: any) => {
-        this._tbl = newTable;
-      });
+        this._tbl = newTable
+      })
  }

  /**
@@ -988,26 +966,26 @@ export class LocalTable<T = number[]> implements Table<T> {
   *
   * @param indexParams The parameters of this Index, @see VectorIndexParams.
   */
-  async createIndex(indexParams: VectorIndexParams): Promise<any> {
+  async createIndex (indexParams: VectorIndexParams): Promise<any> {
    return tableCreateVectorIndex
      .call(this._tbl, indexParams)
      .then((newTable: any) => {
-        this._tbl = newTable;
-      });
+        this._tbl = newTable
+      })
  }

-  async createScalarIndex(column: string, replace?: boolean): Promise<void> {
+  async createScalarIndex (column: string, replace?: boolean): Promise<void> {
    if (replace === undefined) {
-      replace = true;
+      replace = true
    }
-    return tableCreateScalarIndex.call(this._tbl, column, replace);
+    return tableCreateScalarIndex.call(this._tbl, column, replace)
  }

  /**
   * Returns the number of rows in this table.
   */
-  async countRows(filter?: string): Promise<number> {
-    return tableCountRows.call(this._tbl, filter);
+  async countRows (filter?: string): Promise<number> {
+    return tableCountRows.call(this._tbl, filter)
  }

  /**
@@ -1015,10 +993,10 @@ export class LocalTable<T = number[]> implements Table<T> {
   *
   * @param filter A filter in the same format used by a sql WHERE clause.
   */
-  async delete(filter: string): Promise<void> {
+  async delete (filter: string): Promise<void> {
    return tableDelete.call(this._tbl, filter).then((newTable: any) => {
-      this._tbl = newTable;
-    });
+      this._tbl = newTable
+    })
  }

  /**
@@ -1028,65 +1006,55 @@ export class LocalTable<T = number[]> implements Table<T> {
   *
   * @returns
   */
-  async update(args: UpdateArgs | UpdateSqlArgs): Promise<void> {
-    let filter: string | null;
-    let updates: Record<string, string>;
+  async update (args: UpdateArgs | UpdateSqlArgs): Promise<void> {
+    let filter: string | null
+    let updates: Record<string, string>

-    if ("valuesSql" in args) {
-      filter = args.where ?? null;
-      updates = args.valuesSql;
+    if ('valuesSql' in args) {
+      filter = args.where ?? null
+      updates = args.valuesSql
    } else {
-      filter = args.where ?? null;
-      updates = {};
+      filter = args.where ?? null
+      updates = {}
      for (const [key, value] of Object.entries(args.values)) {
-        updates[key] = toSQL(value);
+        updates[key] = toSQL(value)
      }
    }

    return tableUpdate
      .call(this._tbl, filter, updates)
      .then((newTable: any) => {
-        this._tbl = newTable;
-      });
+        this._tbl = newTable
+      })
  }

-  async mergeInsert(
-    on: string,
-    data: Array<Record<string, unknown>> | ArrowTable,
-    args: MergeInsertArgs
-  ): Promise<void> {
-    let whenMatchedUpdateAll = false;
-    let whenMatchedUpdateAllFilt = null;
-    if (
-      args.whenMatchedUpdateAll !== undefined &&
-      args.whenMatchedUpdateAll !== null
-    ) {
-      whenMatchedUpdateAll = true;
+  async mergeInsert (on: string, data: Array<Record<string, unknown>> | ArrowTable, args: MergeInsertArgs): Promise<void> {
+    let whenMatchedUpdateAll = false
+    let whenMatchedUpdateAllFilt = null
+    if (args.whenMatchedUpdateAll !== undefined && args.whenMatchedUpdateAll !== null) {
+      whenMatchedUpdateAll = true
      if (args.whenMatchedUpdateAll !== true) {
-        whenMatchedUpdateAllFilt = args.whenMatchedUpdateAll;
+        whenMatchedUpdateAllFilt = args.whenMatchedUpdateAll
      }
    }
-    const whenNotMatchedInsertAll = args.whenNotMatchedInsertAll ?? false;
-    let whenNotMatchedBySourceDelete = false;
-    let whenNotMatchedBySourceDeleteFilt = null;
-    if (
-      args.whenNotMatchedBySourceDelete !== undefined &&
-      args.whenNotMatchedBySourceDelete !== null
-    ) {
-      whenNotMatchedBySourceDelete = true;
+    const whenNotMatchedInsertAll = args.whenNotMatchedInsertAll ?? false
+    let whenNotMatchedBySourceDelete = false
+    let whenNotMatchedBySourceDeleteFilt = null
+    if (args.whenNotMatchedBySourceDelete !== undefined && args.whenNotMatchedBySourceDelete !== null) {
+      whenNotMatchedBySourceDelete = true
      if (args.whenNotMatchedBySourceDelete !== true) {
-        whenNotMatchedBySourceDeleteFilt = args.whenNotMatchedBySourceDelete;
+        whenNotMatchedBySourceDeleteFilt = args.whenNotMatchedBySourceDelete
      }
    }

-    const schema = await this.schema;
-    let tbl: ArrowTable;
+    const schema = await this.schema
+    let tbl: ArrowTable
    if (data instanceof ArrowTable) {
-      tbl = data;
+      tbl = data
    } else {
-      tbl = makeArrowTable(data, { schema });
+      tbl = makeArrowTable(data, { schema })
    }
-    const buffer = await fromTableToBuffer(tbl, this._embeddings, schema);
+    const buffer = await fromTableToBuffer(tbl, this._embeddings, schema)

    this._tbl = await tableMergeInsert.call(
      this._tbl,
@@ -1097,7 +1065,7 @@ export class LocalTable<T = number[]> implements Table<T> {
      whenNotMatchedBySourceDelete,
      whenNotMatchedBySourceDeleteFilt,
      buffer
-    );
+    )
  }

  /**
@@ -1115,16 +1083,16 @@ export class LocalTable<T = number[]> implements Table<T> {
   *                 uphold this promise can lead to corrupted tables.
   * @returns
   */
-  async cleanupOldVersions(
+  async cleanupOldVersions (
    olderThan?: number,
    deleteUnverified?: boolean
  ): Promise<CleanupStats> {
    return tableCleanupOldVersions
      .call(this._tbl, olderThan, deleteUnverified)
      .then((res: { newTable: any, metrics: CleanupStats }) => {
-        this._tbl = res.newTable;
-        return res.metrics;
-      });
+        this._tbl = res.newTable
+        return res.metrics
+      })
  }

  /**
@@ -1138,64 +1106,62 @@ export class LocalTable<T = number[]> implements Table<T> {
   *               for most tables.
   * @returns Metrics about the compaction operation.
   */
-  async compactFiles(options?: CompactionOptions): Promise<CompactionMetrics> {
-    const optionsArg = options ?? {};
+  async compactFiles (options?: CompactionOptions): Promise<CompactionMetrics> {
+    const optionsArg = options ?? {}
    return tableCompactFiles
      .call(this._tbl, optionsArg)
      .then((res: { newTable: any, metrics: CompactionMetrics }) => {
-        this._tbl = res.newTable;
-        return res.metrics;
-      });
+        this._tbl = res.newTable
+        return res.metrics
+      })
  }

-  async listIndices(): Promise<VectorIndex[]> {
-    return tableListIndices.call(this._tbl);
+  async listIndices (): Promise<VectorIndex[]> {
+    return tableListIndices.call(this._tbl)
  }

-  async indexStats(indexUuid: string): Promise<IndexStats> {
-    return tableIndexStats.call(this._tbl, indexUuid);
+  async indexStats (indexUuid: string): Promise<IndexStats> {
+    return tableIndexStats.call(this._tbl, indexUuid)
  }

-  get schema(): Promise<Schema> {
+  get schema (): Promise<Schema> {
    // empty table
-    return this.getSchema();
+    return this.getSchema()
  }

-  private async getSchema(): Promise<Schema> {
-    const buffer = await tableSchema.call(this._tbl, this._isElectron);
-    const table = tableFromIPC(buffer);
-    return table.schema;
+  private async getSchema (): Promise<Schema> {
+    const buffer = await tableSchema.call(this._tbl, this._isElectron)
+    const table = tableFromIPC(buffer)
+    return table.schema
  }

  // See https://github.com/electron/electron/issues/2288
-  private checkElectron(): boolean {
+  private checkElectron (): boolean {
    try {
      // eslint-disable-next-line no-prototype-builtins
      return (
-        Object.prototype.hasOwnProperty.call(process?.versions, "electron") ||
-        navigator?.userAgent?.toLowerCase()?.includes(" electron")
-      );
+        Object.prototype.hasOwnProperty.call(process?.versions, 'electron') ||
+        navigator?.userAgent?.toLowerCase()?.includes(' electron')
+      )
    } catch (e) {
-      return false;
+      return false
    }
  }

-  async addColumns(
-    newColumnTransforms: Array<{ name: string, valueSql: string }>
-  ): Promise<void> {
-    return tableAddColumns.call(this._tbl, newColumnTransforms);
+  async addColumns (newColumnTransforms: Array<{ name: string, valueSql: string }>): Promise<void> {
+    return tableAddColumns.call(this._tbl, newColumnTransforms)
  }

-  async alterColumns(columnAlterations: ColumnAlteration[]): Promise<void> {
-    return tableAlterColumns.call(this._tbl, columnAlterations);
+  async alterColumns (columnAlterations: ColumnAlteration[]): Promise<void> {
+    return tableAlterColumns.call(this._tbl, columnAlterations)
  }

-  async dropColumns(columnNames: string[]): Promise<void> {
-    return tableDropColumns.call(this._tbl, columnNames);
+  async dropColumns (columnNames: string[]): Promise<void> {
+    return tableDropColumns.call(this._tbl, columnNames)
  }

-  withMiddleware(middleware: HttpMiddleware): Table<T> {
-    return this;
+  withMiddleware (middleware: HttpMiddleware): Table<T> {
+    return this
  }
 }

@@ -1218,7 +1184,7 @@ export interface CompactionOptions {
   */
  targetRowsPerFragment?: number
  /**
-   * The maximum number of T per group. Defaults to 1024.
+   * The maximum number of rows per group. Defaults to 1024.
   */
  maxRowsPerGroup?: number
  /**
@@ -1318,21 +1284,21 @@ export interface IvfPQIndexConfig {
   */
  index_cache_size?: number

-  type: "ivf_pq"
+  type: 'ivf_pq'
 }

-export type VectorIndexParams = IvfPQIndexConfig;
+export type VectorIndexParams = IvfPQIndexConfig

 /**
 * Write mode for writing a table.
 */
 export enum WriteMode {
  /** Create a new {@link Table}. */
-  Create = "create",
+  Create = 'create',
  /** Overwrite the existing {@link Table} if presented. */
-  Overwrite = "overwrite",
+  Overwrite = 'overwrite',
  /** Append new data to the table. */
-  Append = "append",
+  Append = 'append',
 }

 /**
@@ -1344,14 +1310,14 @@ export interface WriteOptions {
 }

 export class DefaultWriteOptions implements WriteOptions {
-  writeMode = WriteMode.Create;
+  writeMode = WriteMode.Create
 }

-export function isWriteOptions(value: any): value is WriteOptions {
+export function isWriteOptions (value: any): value is WriteOptions {
  return (
    Object.keys(value).length === 1 &&
-    (value.writeMode === undefined || typeof value.writeMode === "string")
-  );
+    (value.writeMode === undefined || typeof value.writeMode === 'string')
+  )
 }

 /**
@@ -1361,15 +1327,15 @@ export enum MetricType {
  /**
   * Euclidean distance
   */
-  L2 = "l2",
+  L2 = 'l2',

  /**
   * Cosine distance
   */
-  Cosine = "cosine",
+  Cosine = 'cosine',

  /**
   * Dot product
   */
-  Dot = "dot",
+  Dot = 'dot',
 }
--- a/node/src/integration_test/test.ts
+++ b/node/src/integration_test/test.ts
@@ -51,7 +51,7 @@ describe('LanceDB Mirrored Store Integration test', function () {

    const dir = tmpdir()
    console.log(dir)
-    const conn = await lancedb.connect({ uri: `s3://lancedb-integtest?mirroredStore=${dir}`, storageOptions: { allowHttp: 'true' } })
+    const conn = await lancedb.connect(`s3://lancedb-integtest?mirroredStore=${dir}`)
    const data = Array(200).fill({ vector: Array(128).fill(1.0), id: 0 })
    data.push(...Array(200).fill({ vector: Array(128).fill(1.0), id: 1 }))
    data.push(...Array(200).fill({ vector: Array(128).fill(1.0), id: 2 }))
--- a/node/src/remote/index.ts
+++ b/node/src/remote/index.ts
@@ -140,9 +140,6 @@ export class RemoteConnection implements Connection {
      schema = nameOrOpts.schema
      embeddings = nameOrOpts.embeddingFunction
      tableName = nameOrOpts.name
-      if (data === undefined) {
-        data = nameOrOpts.data
-      }
    }

    let buffer: Buffer
@@ -509,8 +506,7 @@ export class RemoteTable<T = number[]> implements Table<T> {
    return (await results.body()).indexes?.map((index: any) => ({
      columns: index.columns,
      name: index.index_name,
-      uuid: index.index_uuid,
-      status: index.status
+      uuid: index.index_uuid
    }))
  }

@@ -521,10 +517,7 @@ export class RemoteTable<T = number[]> implements Table<T> {
    const body = await results.body()
    return {
      numIndexedRows: body?.num_indexed_rows,
-      numUnindexedRows: body?.num_unindexed_rows,
-      indexType: body?.index_type,
-      distanceType: body?.distance_type,
-      completedAt: body?.completed_at
+      numUnindexedRows: body?.num_unindexed_rows
    }
  }

--- a/node/src/sanitize.ts
+++ b/node/src/sanitize.ts
@@ -32,7 +32,7 @@ import {
  Bool,
  Date_,
  Decimal,
-  type DataType,
+  DataType,
  Dictionary,
  Binary,
  Float32,
@@ -74,12 +74,12 @@ import {
  DurationNanosecond,
  DurationMicrosecond,
  DurationMillisecond,
-  DurationSecond
+  DurationSecond,
 } from "apache-arrow";
 import type { IntBitWidth, TimeBitWidth } from "apache-arrow/type";

 function sanitizeMetadata(
-  metadataLike?: unknown
+  metadataLike?: unknown,
 ): Map<string, string> | undefined {
  if (metadataLike === undefined || metadataLike === null) {
    return undefined;
@@ -90,7 +90,7 @@ function sanitizeMetadata(
  for (const item of metadataLike) {
    if (!(typeof item[0] === "string" || !(typeof item[1] === "string"))) {
      throw Error(
-        "Expected metadata, if present, to be a Map<string, string> but it had non-string keys or values"
+        "Expected metadata, if present, to be a Map<string, string> but it had non-string keys or values",
      );
    }
  }
@@ -105,7 +105,7 @@ function sanitizeInt(typeLike: object) {
    typeof typeLike.isSigned !== "boolean"
  ) {
    throw Error(
-      "Expected an Int Type to have a `bitWidth` and `isSigned` property"
+      "Expected an Int Type to have a `bitWidth` and `isSigned` property",
    );
  }
  return new Int(typeLike.isSigned, typeLike.bitWidth as IntBitWidth);
@@ -128,7 +128,7 @@ function sanitizeDecimal(typeLike: object) {
    typeof typeLike.bitWidth !== "number"
  ) {
    throw Error(
-      "Expected a Decimal Type to have `scale`, `precision`, and `bitWidth` properties"
+      "Expected a Decimal Type to have `scale`, `precision`, and `bitWidth` properties",
    );
  }
  return new Decimal(typeLike.scale, typeLike.precision, typeLike.bitWidth);
@@ -149,7 +149,7 @@ function sanitizeTime(typeLike: object) {
    typeof typeLike.bitWidth !== "number"
  ) {
    throw Error(
-      "Expected a Time type to have `unit` and `bitWidth` properties"
+      "Expected a Time type to have `unit` and `bitWidth` properties",
    );
  }
  return new Time(typeLike.unit, typeLike.bitWidth as TimeBitWidth);
@@ -172,7 +172,7 @@ function sanitizeTypedTimestamp(
    | typeof TimestampNanosecond
    | typeof TimestampMicrosecond
    | typeof TimestampMillisecond
-    | typeof TimestampSecond
+    | typeof TimestampSecond,
 ) {
  let timezone = null;
  if ("timezone" in typeLike && typeof typeLike.timezone === "string") {
@@ -191,7 +191,7 @@ function sanitizeInterval(typeLike: object) {
 function sanitizeList(typeLike: object) {
  if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
    throw Error(
-      "Expected a List type to have an array-like `children` property"
+      "Expected a List type to have an array-like `children` property",
    );
  }
  if (typeLike.children.length !== 1) {
@@ -203,7 +203,7 @@ function sanitizeList(typeLike: object) {
 function sanitizeStruct(typeLike: object) {
  if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
    throw Error(
-      "Expected a Struct type to have an array-like `children` property"
+      "Expected a Struct type to have an array-like `children` property",
    );
  }
  return new Struct(typeLike.children.map((child) => sanitizeField(child)));
@@ -216,47 +216,47 @@ function sanitizeUnion(typeLike: object) {
    typeof typeLike.mode !== "number"
  ) {
    throw Error(
-      "Expected a Union type to have `typeIds` and `mode` properties"
+      "Expected a Union type to have `typeIds` and `mode` properties",
    );
  }
  if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
    throw Error(
-      "Expected a Union type to have an array-like `children` property"
+      "Expected a Union type to have an array-like `children` property",
    );
  }

  return new Union(
    typeLike.mode,
    typeLike.typeIds as any,
-    typeLike.children.map((child) => sanitizeField(child))
+    typeLike.children.map((child) => sanitizeField(child)),
  );
 }

 function sanitizeTypedUnion(
  typeLike: object,
-  UnionType: typeof DenseUnion | typeof SparseUnion
+  UnionType: typeof DenseUnion | typeof SparseUnion,
 ) {
  if (!("typeIds" in typeLike)) {
    throw Error(
-      "Expected a DenseUnion/SparseUnion type to have a `typeIds` property"
+      "Expected a DenseUnion/SparseUnion type to have a `typeIds` property",
    );
  }
  if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
    throw Error(
-      "Expected a DenseUnion/SparseUnion type to have an array-like `children` property"
+      "Expected a DenseUnion/SparseUnion type to have an array-like `children` property",
    );
  }

  return new UnionType(
    typeLike.typeIds as any,
-    typeLike.children.map((child) => sanitizeField(child))
+    typeLike.children.map((child) => sanitizeField(child)),
  );
 }

 function sanitizeFixedSizeBinary(typeLike: object) {
  if (!("byteWidth" in typeLike) || typeof typeLike.byteWidth !== "number") {
    throw Error(
-      "Expected a FixedSizeBinary type to have a `byteWidth` property"
+      "Expected a FixedSizeBinary type to have a `byteWidth` property",
    );
  }
  return new FixedSizeBinary(typeLike.byteWidth);
@@ -268,7 +268,7 @@ function sanitizeFixedSizeList(typeLike: object) {
  }
  if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
    throw Error(
-      "Expected a FixedSizeList type to have an array-like `children` property"
+      "Expected a FixedSizeList type to have an array-like `children` property",
    );
  }
  if (typeLike.children.length !== 1) {
@@ -276,14 +276,14 @@ function sanitizeFixedSizeList(typeLike: object) {
  }
  return new FixedSizeList(
    typeLike.listSize,
-    sanitizeField(typeLike.children[0])
+    sanitizeField(typeLike.children[0]),
  );
 }

 function sanitizeMap(typeLike: object) {
  if (!("children" in typeLike) || !Array.isArray(typeLike.children)) {
    throw Error(
-      "Expected a Map type to have an array-like `children` property"
+      "Expected a Map type to have an array-like `children` property",
    );
  }
  if (!("keysSorted" in typeLike) || typeof typeLike.keysSorted !== "boolean") {
@@ -291,7 +291,7 @@ function sanitizeMap(typeLike: object) {
  }
  return new Map_(
    typeLike.children.map((field) => sanitizeField(field)) as any,
-    typeLike.keysSorted
+    typeLike.keysSorted,
  );
 }

@@ -319,7 +319,7 @@ function sanitizeDictionary(typeLike: object) {
    sanitizeType(typeLike.dictionary),
    sanitizeType(typeLike.indices) as any,
    typeLike.id,
-    typeLike.isOrdered
+    typeLike.isOrdered,
  );
 }

@@ -454,7 +454,7 @@ function sanitizeField(fieldLike: unknown): Field {
    !("nullable" in fieldLike)
  ) {
    throw Error(
-      "The field passed in is missing a `type`/`name`/`nullable` property"
+      "The field passed in is missing a `type`/`name`/`nullable` property",
    );
  }
  const type = sanitizeType(fieldLike.type);
@@ -489,7 +489,7 @@ export function sanitizeSchema(schemaLike: unknown): Schema {
  }
  if (!("fields" in schemaLike)) {
    throw Error(
-      "The schema passed in does not appear to be a schema (no 'fields' property)"
+      "The schema passed in does not appear to be a schema (no 'fields' property)",
    );
  }
  let metadata;
@@ -498,11 +498,11 @@ export function sanitizeSchema(schemaLike: unknown): Schema {
  }
  if (!Array.isArray(schemaLike.fields)) {
    throw Error(
-      "The schema passed in had a 'fields' property but it was not an array"
+      "The schema passed in had a 'fields' property but it was not an array",
    );
  }
  const sanitizedFields = schemaLike.fields.map((field) =>
-    sanitizeField(field)
+    sanitizeField(field),
  );
  return new Schema(sanitizedFields, metadata);
 }
--- a/node/src/test/test.ts
+++ b/node/src/test/test.ts
--- a/nodejs/.eslintignore
+++ b/nodejs/.eslintignore
@@ -0,0 +1,3 @@
+**/dist/**/*
+**/native.js
+**/native.d.ts
--- a/nodejs/.gitignore
+++ b/nodejs/.gitignore
@@ -1 +0,0 @@
-yarn.lock
--- a/nodejs/.prettierignore
+++ b/nodejs/.prettierignore
@@ -0,0 +1 @@
+.eslintignore
--- a/nodejs/Cargo.toml
+++ b/nodejs/Cargo.toml
@@ -15,11 +15,11 @@ crate-type = ["cdylib"]
 arrow-ipc.workspace = true
 futures.workspace = true
 lancedb = { path = "../rust/lancedb" }
-napi = { version = "2.16.8", default-features = false, features = [
-    "napi9",
+napi = { version = "2.15", default-features = false, features = [
+    "napi7",
    "async",
 ] }
-napi-derive = "2.16.4"
+napi-derive = "2"

 # Prevent dynamic linking of lzma, which comes from datafusion
 lzma-sys = { version = "*", features = ["static"] }
--- a/nodejs/README.md
+++ b/nodejs/README.md
@@ -43,20 +43,29 @@ npm run test

 ### Running lint / format

-LanceDb uses [biome](https://biomejs.dev/) for linting and formatting. if you are using VSCode you will need to install the official [Biome](https://marketplace.visualstudio.com/items?itemName=biomejs.biome) extension.
-To manually lint your code you can run:
+LanceDb uses eslint for linting. VSCode does not need any plugins to use eslint. However, it
+may need some additional configuration. Make sure that eslint.experimental.useFlatConfig is
+set to true. Also, if your vscode root folder is the repo root then you will need to set
+the eslint.workingDirectories to ["nodejs"]. To manually lint your code you can run:

 ```sh
 npm run lint
 ```

-to automatically fix all fixable issues:
+LanceDb uses prettier for formatting. If you are using VSCode you will need to install the
+"Prettier - Code formatter" extension. You should then configure it to be the default formatter
+for typescript and you should enable format on save. To manually check your code's format you
+can run:

 ```sh
-npm run lint-fix
+npm run chkformat
 ```

-If you do not have your workspace root set to the `nodejs` directory, unfortunately the extension will not work. You can still run the linting and formatting commands manually.
+If you need to manually format your code you can run:
+
+```sh
+npx prettier --write .
+```

 ### Generating docs

--- a/nodejs/test/arrow.test.ts
+++ b/nodejs/test/arrow.test.ts
@@ -13,27 +13,32 @@
 // limitations under the License.

 import {
-  Binary,
-  Bool,
-  DataType,
-  Dictionary,
+  convertToTable,
+  fromTableToBuffer,
+  makeArrowTable,
+  makeEmptyTable,
+} from "../dist/arrow";
+import {
  Field,
  FixedSizeList,
-  Float,
  Float16,
  Float32,
-  Float64,
  Int32,
-  Int64,
-  List,
-  MetadataVersion,
-  Precision,
-  Schema,
-  Struct,
-  type Table,
-  Type,
-  Utf8,
  tableFromIPC,
+  Schema,
+  Float64,
+  type Table,
+  Binary,
+  Bool,
+  Utf8,
+  Struct,
+  List,
+  DataType,
+  Dictionary,
+  Int64,
+  Float,
+  Precision,
+  MetadataVersion,
 } from "apache-arrow";
 import {
  Dictionary as OldDictionary,
@@ -41,25 +46,14 @@ import {
  FixedSizeList as OldFixedSizeList,
  Float32 as OldFloat32,
  Int32 as OldInt32,
-  Schema as OldSchema,
  Struct as OldStruct,
+  Schema as OldSchema,
  TimestampNanosecond as OldTimestampNanosecond,
  Utf8 as OldUtf8,
 } from "apache-arrow-old";
-import {
-  convertToTable,
-  fromTableToBuffer,
-  makeArrowTable,
-  makeEmptyTable,
-} from "../lancedb/arrow";
-import {
-  EmbeddingFunction,
-  FieldOptions,
-  FunctionOptions,
-} from "../lancedb/embedding/embedding_function";
-import { EmbeddingFunctionConfig } from "../lancedb/embedding/registry";
+import { type EmbeddingFunction } from "../dist/embedding/embedding_function";

-// biome-ignore lint/suspicious/noExplicitAny: skip
+// eslint-disable-next-line @typescript-eslint/no-explicit-any
 function sampleRecords(): Array<Record<string, any>> {
  return [
    {
@@ -286,46 +280,23 @@ describe("The function makeArrowTable", function () {
  });
 });

-class DummyEmbedding extends EmbeddingFunction<string> {
-  toJSON(): Partial<FunctionOptions> {
-    return {};
-  }
+class DummyEmbedding implements EmbeddingFunction<string> {
+  public readonly sourceColumn = "string";
+  public readonly embeddingDimension = 2;
+  public readonly embeddingDataType = new Float16();

-  async computeSourceEmbeddings(data: string[]): Promise<number[][]> {
-    return data.map(() => [0.0, 0.0]);
-  }
-
-  ndims(): number {
-    return 2;
-  }
-
-  embeddingDataType() {
-    return new Float16();
-  }
-}
-
-class DummyEmbeddingWithNoDimension extends EmbeddingFunction<string> {
-  toJSON(): Partial<FunctionOptions> {
-    return {};
-  }
-
-  embeddingDataType(): Float {
-    return new Float16();
-  }
-
-  async computeSourceEmbeddings(data: string[]): Promise<number[][]> {
+  async embed(data: string[]): Promise<number[][]> {
    return data.map(() => [0.0, 0.0]);
  }
 }
-const dummyEmbeddingConfig: EmbeddingFunctionConfig = {
-  sourceColumn: "string",
-  function: new DummyEmbedding(),
-};

-const dummyEmbeddingConfigWithNoDimension: EmbeddingFunctionConfig = {
-  sourceColumn: "string",
-  function: new DummyEmbeddingWithNoDimension(),
-};
+class DummyEmbeddingWithNoDimension implements EmbeddingFunction<string> {
+  public readonly sourceColumn = "string";
+
+  async embed(data: string[]): Promise<number[][]> {
+    return data.map(() => [0.0, 0.0]);
+  }
+}

 describe("convertToTable", function () {
  it("will infer data types correctly", async function () {
@@ -360,7 +331,7 @@ describe("convertToTable", function () {

  it("will apply embeddings", async function () {
    const records = sampleRecords();
-    const table = await convertToTable(records, dummyEmbeddingConfig);
+    const table = await convertToTable(records, new DummyEmbedding());
    expect(DataType.isFixedSizeList(table.getChild("vector")?.type)).toBe(true);
    expect(table.getChild("vector")?.type.children[0].type.toString()).toEqual(
      new Float16().toString(),
@@ -369,7 +340,7 @@ describe("convertToTable", function () {

  it("will fail if missing the embedding source column", async function () {
    await expect(
-      convertToTable([{ id: 1 }], dummyEmbeddingConfig),
+      convertToTable([{ id: 1 }], new DummyEmbedding()),
    ).rejects.toThrow("'string' was not present");
  });

@@ -380,7 +351,7 @@ describe("convertToTable", function () {
    const table = makeEmptyTable(schema);

    // If the embedding specifies the dimension we are fine
-    await fromTableToBuffer(table, dummyEmbeddingConfig);
+    await fromTableToBuffer(table, new DummyEmbedding());

    // We can also supply a schema and should be ok
    const schemaWithEmbedding = new Schema([
@@ -393,13 +364,13 @@ describe("convertToTable", function () {
    ]);
    await fromTableToBuffer(
      table,
-      dummyEmbeddingConfigWithNoDimension,
+      new DummyEmbeddingWithNoDimension(),
      schemaWithEmbedding,
    );

    // Otherwise we will get an error
    await expect(
-      fromTableToBuffer(table, dummyEmbeddingConfigWithNoDimension),
+      fromTableToBuffer(table, new DummyEmbeddingWithNoDimension()),
    ).rejects.toThrow("does not specify `embeddingDimension`");
  });

@@ -412,7 +383,7 @@ describe("convertToTable", function () {
        false,
      ),
    ]);
-    const table = await convertToTable([], dummyEmbeddingConfig, { schema });
+    const table = await convertToTable([], new DummyEmbedding(), { schema });
    expect(DataType.isFixedSizeList(table.getChild("vector")?.type)).toBe(true);
    expect(table.getChild("vector")?.type.children[0].type.toString()).toEqual(
      new Float16().toString(),
@@ -422,17 +393,16 @@ describe("convertToTable", function () {
  it("will complain if embeddings present but schema missing embedding column", async function () {
    const schema = new Schema([new Field("string", new Utf8(), false)]);
    await expect(
-      convertToTable([], dummyEmbeddingConfig, { schema }),
+      convertToTable([], new DummyEmbedding(), { schema }),
    ).rejects.toThrow("column vector was missing");
  });

  it("will provide a nice error if run twice", async function () {
    const records = sampleRecords();
-    const table = await convertToTable(records, dummyEmbeddingConfig);
-
+    const table = await convertToTable(records, new DummyEmbedding());
    // fromTableToBuffer will try and apply the embeddings again
    await expect(
-      fromTableToBuffer(table, dummyEmbeddingConfig),
+      fromTableToBuffer(table, new DummyEmbedding()),
    ).rejects.toThrow("already existed");
  });
 });
@@ -468,7 +438,7 @@ describe("when using two versions of arrow", function () {
          new OldField("ts_no_tz", new OldTimestampNanosecond(null)),
        ]),
      ),
-      // biome-ignore lint/suspicious/noExplicitAny: skip
+      // eslint-disable-next-line @typescript-eslint/no-explicit-any
    ]) as any;
    schema.metadataVersion = MetadataVersion.V5;
    const table = makeArrowTable([], { schema });
--- a/nodejs/test/connection.test.ts
+++ b/nodejs/test/connection.test.ts
@@ -12,15 +12,13 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-import { Field, Float64, Schema } from "apache-arrow";
 import * as tmp from "tmp";
-import { Connection, Table, connect } from "../lancedb";
+
+import { Connection, connect } from "../dist/index.js";

 describe("when connecting", () => {
  let tmpDir: tmp.DirResult;
-  beforeEach(() => {
-    tmpDir = tmp.dirSync({ unsafeCleanup: true });
-  });
+  beforeEach(() => (tmpDir = tmp.dirSync({ unsafeCleanup: true })));
  afterEach(() => tmpDir.removeCallback());

  it("should connect", async () => {
@@ -57,18 +55,6 @@ describe("given a connection", () => {
    expect(db.isOpen()).toBe(false);
    await expect(db.tableNames()).rejects.toThrow("Connection is closed");
  });
-  it("should be able to create a table from an object arg `createTable(options)`, or args `createTable(name, data, options)`", async () => {
-    let tbl = await db.createTable("test", [{ id: 1 }, { id: 2 }]);
-    await expect(tbl.countRows()).resolves.toBe(2);
-
-    tbl = await db.createTable({
-      name: "test",
-      data: [{ id: 3 }],
-      mode: "overwrite",
-    });
-
-    await expect(tbl.countRows()).resolves.toBe(1);
-  });

  it("should fail if creating table twice, unless overwrite is true", async () => {
    let tbl = await db.createTable("test", [{ id: 1 }, { id: 2 }]);
@@ -99,39 +85,4 @@ describe("given a connection", () => {
    tables = await db.tableNames({ startAfter: "a" });
    expect(tables).toEqual(["b", "c"]);
  });
-
-  it("should create tables in v2 mode", async () => {
-    const db = await connect(tmpDir.name);
-    const data = [...Array(10000).keys()].map((i) => ({ id: i }));
-
-    // Create in v1 mode
-    let table = await db.createTable("test", data);
-
-    const isV2 = async (table: Table) => {
-      const data = await table.query().toArrow({ maxBatchLength: 100000 });
-      console.log(data.batches.length);
-      return data.batches.length < 5;
-    };
-
-    await expect(isV2(table)).resolves.toBe(false);
-
-    // Create in v2 mode
-    table = await db.createTable("test_v2", data, { useLegacyFormat: false });
-
-    await expect(isV2(table)).resolves.toBe(true);
-
-    await table.add(data);
-
-    await expect(isV2(table)).resolves.toBe(true);
-
-    // Create empty in v2 mode
-    const schema = new Schema([new Field("id", new Float64(), true)]);
-
-    table = await db.createEmptyTable("test_v2_empty", schema, {
-      useLegacyFormat: false,
-    });
-
-    await table.add(data);
-    await expect(isV2(table)).resolves.toBe(true);
-  });
 });
--- a/nodejs/test/embedding.test.ts
+++ b/nodejs/test/embedding.test.ts
@@ -1,314 +0,0 @@
-// Copyright 2024 Lance Developers.
-//
-// Licensed under the Apache License, Version 2.0 (the "License");
-// you may not use this file except in compliance with the License.
-// You may obtain a copy of the License at
-//
-//     http://www.apache.org/licenses/LICENSE-2.0
-//
-// Unless required by applicable law or agreed to in writing, software
-// distributed under the License is distributed on an "AS IS" BASIS,
-// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-// See the License for the specific language governing permissions and
-// limitations under the License.
-
-import * as tmp from "tmp";
-
-import { connect } from "../lancedb";
-import {
-  Field,
-  FixedSizeList,
-  Float,
-  Float16,
-  Float32,
-  Float64,
-  Schema,
-  Utf8,
-} from "../lancedb/arrow";
-import { EmbeddingFunction, LanceSchema } from "../lancedb/embedding";
-import { getRegistry, register } from "../lancedb/embedding/registry";
-
-describe("embedding functions", () => {
-  let tmpDir: tmp.DirResult;
-  beforeEach(() => {
-    tmpDir = tmp.dirSync({ unsafeCleanup: true });
-  });
-  afterEach(() => {
-    tmpDir.removeCallback();
-    getRegistry().reset();
-  });
-
-  it("should be able to create a table with an embedding function", async () => {
-    class MockEmbeddingFunction extends EmbeddingFunction<string> {
-      toJSON(): object {
-        return {};
-      }
-      ndims() {
-        return 3;
-      }
-      embeddingDataType(): Float {
-        return new Float32();
-      }
-      async computeQueryEmbeddings(_data: string) {
-        return [1, 2, 3];
-      }
-      async computeSourceEmbeddings(data: string[]) {
-        return Array.from({ length: data.length }).fill([
-          1, 2, 3,
-        ]) as number[][];
-      }
-    }
-    const func = new MockEmbeddingFunction();
-    const db = await connect(tmpDir.name);
-    const table = await db.createTable(
-      "test",
-      [
-        { id: 1, text: "hello" },
-        { id: 2, text: "world" },
-      ],
-      {
-        embeddingFunction: {
-          function: func,
-          sourceColumn: "text",
-        },
-      },
-    );
-    // biome-ignore lint/suspicious/noExplicitAny: test
-    const arr = (await table.query().toArray()) as any;
-    expect(arr[0].vector).toBeDefined();
-
-    // we round trip through JSON to make sure the vector properly gets converted to an array
-    // otherwise it'll be a TypedArray or Vector
-    const vector0 = JSON.parse(JSON.stringify(arr[0].vector));
-    expect(vector0).toEqual([1, 2, 3]);
-  });
-
-  it("should be able to create an empty table with an embedding function", async () => {
-    @register()
-    class MockEmbeddingFunction extends EmbeddingFunction<string> {
-      toJSON(): object {
-        return {};
-      }
-      ndims() {
-        return 3;
-      }
-      embeddingDataType(): Float {
-        return new Float32();
-      }
-      async computeQueryEmbeddings(_data: string) {
-        return [1, 2, 3];
-      }
-      async computeSourceEmbeddings(data: string[]) {
-        return Array.from({ length: data.length }).fill([
-          1, 2, 3,
-        ]) as number[][];
-      }
-    }
-    const schema = new Schema([
-      new Field("text", new Utf8(), true),
-      new Field(
-        "vector",
-        new FixedSizeList(3, new Field("item", new Float32(), true)),
-        true,
-      ),
-    ]);
-
-    const func = new MockEmbeddingFunction();
-    const db = await connect(tmpDir.name);
-    const table = await db.createEmptyTable("test", schema, {
-      embeddingFunction: {
-        function: func,
-        sourceColumn: "text",
-      },
-    });
-    const outSchema = await table.schema();
-    expect(outSchema.metadata.get("embedding_functions")).toBeDefined();
-    await table.add([{ text: "hello world" }]);
-
-    // biome-ignore lint/suspicious/noExplicitAny: test
-    const arr = (await table.query().toArray()) as any;
-    expect(arr[0].vector).toBeDefined();
-
-    // we round trip through JSON to make sure the vector properly gets converted to an array
-    // otherwise it'll be a TypedArray or Vector
-    const vector0 = JSON.parse(JSON.stringify(arr[0].vector));
-    expect(vector0).toEqual([1, 2, 3]);
-  });
-  it("should error when appending to a table with an unregistered embedding function", async () => {
-    @register("mock")
-    class MockEmbeddingFunction extends EmbeddingFunction<string> {
-      toJSON(): object {
-        return {};
-      }
-      ndims() {
-        return 3;
-      }
-      embeddingDataType(): Float {
-        return new Float32();
-      }
-      async computeQueryEmbeddings(_data: string) {
-        return [1, 2, 3];
-      }
-      async computeSourceEmbeddings(data: string[]) {
-        return Array.from({ length: data.length }).fill([
-          1, 2, 3,
-        ]) as number[][];
-      }
-    }
-    const func = getRegistry().get<MockEmbeddingFunction>("mock")!.create();
-
-    const schema = LanceSchema({
-      id: new Float64(),
-      text: func.sourceField(new Utf8()),
-      vector: func.vectorField(),
-    });
-
-    const db = await connect(tmpDir.name);
-    await db.createTable(
-      "test",
-      [
-        { id: 1, text: "hello" },
-        { id: 2, text: "world" },
-      ],
-      {
-        schema,
-      },
-    );
-
-    getRegistry().reset();
-    const db2 = await connect(tmpDir.name);
-
-    const tbl = await db2.openTable("test");
-
-    expect(tbl.add([{ id: 3, text: "hello" }])).rejects.toThrow(
-      `Function "mock" not found in registry`,
-    );
-  });
-  test.each([new Float16(), new Float32(), new Float64()])(
-    "should be able to provide manual embeddings with multiple float datatype",
-    async (floatType) => {
-      class MockEmbeddingFunction extends EmbeddingFunction<string> {
-        toJSON(): object {
-          return {};
-        }
-        ndims() {
-          return 3;
-        }
-        embeddingDataType(): Float {
-          return floatType;
-        }
-        async computeQueryEmbeddings(_data: string) {
-          return [1, 2, 3];
-        }
-        async computeSourceEmbeddings(data: string[]) {
-          return Array.from({ length: data.length }).fill([
-            1, 2, 3,
-          ]) as number[][];
-        }
-      }
-      const data = [{ text: "hello" }, { text: "hello world" }];
-
-      const schema = new Schema([
-        new Field("vector", new FixedSizeList(3, new Field("item", floatType))),
-        new Field("text", new Utf8()),
-      ]);
-      const func = new MockEmbeddingFunction();
-
-      const name = "test";
-      const db = await connect(tmpDir.name);
-
-      const table = await db.createTable(name, data, {
-        schema,
-        embeddingFunction: {
-          sourceColumn: "text",
-          function: func,
-        },
-      });
-      const res = await table.query().toArray();
-
-      expect([...res[0].vector]).toEqual([1, 2, 3]);
-    },
-  );
-
-  test.each([new Float16(), new Float32(), new Float64()])(
-    "should be able to provide auto embeddings with multiple float datatypes",
-    async (floatType) => {
-      @register("test1")
-      class MockEmbeddingFunctionWithoutNDims extends EmbeddingFunction<string> {
-        toJSON(): object {
-          return {};
-        }
-
-        embeddingDataType(): Float {
-          return floatType;
-        }
-        async computeQueryEmbeddings(_data: string) {
-          return [1, 2, 3];
-        }
-        async computeSourceEmbeddings(data: string[]) {
-          return Array.from({ length: data.length }).fill([
-            1, 2, 3,
-          ]) as number[][];
-        }
-      }
-      @register("test")
-      class MockEmbeddingFunction extends EmbeddingFunction<string> {
-        toJSON(): object {
-          return {};
-        }
-        ndims() {
-          return 3;
-        }
-        embeddingDataType(): Float {
-          return floatType;
-        }
-        async computeQueryEmbeddings(_data: string) {
-          return [1, 2, 3];
-        }
-        async computeSourceEmbeddings(data: string[]) {
-          return Array.from({ length: data.length }).fill([
-            1, 2, 3,
-          ]) as number[][];
-        }
-      }
-      const func = getRegistry().get<MockEmbeddingFunction>("test")!.create();
-      const func2 = getRegistry()
-        .get<MockEmbeddingFunctionWithoutNDims>("test1")!
-        .create();
-
-      const schema = LanceSchema({
-        text: func.sourceField(new Utf8()),
-        vector: func.vectorField(floatType),
-      });
-
-      const schema2 = LanceSchema({
-        text: func2.sourceField(new Utf8()),
-        vector: func2.vectorField({ datatype: floatType, dims: 3 }),
-      });
-      const schema3 = LanceSchema({
-        text: func2.sourceField(new Utf8()),
-        vector: func.vectorField({
-          datatype: new FixedSizeList(3, new Field("item", floatType, true)),
-          dims: 3,
-        }),
-      });
-
-      const expectedSchema = new Schema([
-        new Field("text", new Utf8(), true),
-        new Field(
-          "vector",
-          new FixedSizeList(3, new Field("item", floatType, true)),
-          true,
-        ),
-      ]);
-      const stringSchema = JSON.stringify(schema, null, 2);
-      const stringSchema2 = JSON.stringify(schema2, null, 2);
-      const stringSchema3 = JSON.stringify(schema3, null, 2);
-      const stringExpectedSchema = JSON.stringify(expectedSchema, null, 2);
-
-      expect(stringSchema).toEqual(stringExpectedSchema);
-      expect(stringSchema2).toEqual(stringExpectedSchema);
-      expect(stringSchema3).toEqual(stringExpectedSchema);
-    },
-  );
-});
--- a/nodejs/test/registry.test.ts
+++ b/nodejs/test/registry.test.ts
@@ -1,170 +0,0 @@
-// Copyright 2024 Lance Developers.
-//
-// Licensed under the Apache License, Version 2.0 (the "License");
-// you may not use this file except in compliance with the License.
-// You may obtain a copy of the License at
-//
-//     http://www.apache.org/licenses/LICENSE-2.0
-//
-// Unless required by applicable law or agreed to in writing, software
-// distributed under the License is distributed on an "AS IS" BASIS,
-// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-// See the License for the specific language governing permissions and
-// limitations under the License.
-import * as arrow from "apache-arrow";
-import * as arrowOld from "apache-arrow-old";
-
-import * as tmp from "tmp";
-
-import { connect } from "../lancedb";
-import { EmbeddingFunction, LanceSchema } from "../lancedb/embedding";
-import { getRegistry, register } from "../lancedb/embedding/registry";
-
-describe.each([arrow, arrowOld])("LanceSchema", (arrow) => {
-  test("should preserve input order", async () => {
-    const schema = LanceSchema({
-      id: new arrow.Int32(),
-      text: new arrow.Utf8(),
-      vector: new arrow.Float32(),
-    });
-    expect(schema.fields.map((x) => x.name)).toEqual(["id", "text", "vector"]);
-  });
-});
-
-describe("Registry", () => {
-  let tmpDir: tmp.DirResult;
-  beforeEach(() => {
-    tmpDir = tmp.dirSync({ unsafeCleanup: true });
-  });
-
-  afterEach(() => {
-    tmpDir.removeCallback();
-    getRegistry().reset();
-  });
-
-  it("should register a new item to the registry", async () => {
-    @register("mock-embedding")
-    class MockEmbeddingFunction extends EmbeddingFunction<string> {
-      toJSON(): object {
-        return {
-          someText: "hello",
-        };
-      }
-      constructor() {
-        super();
-      }
-      ndims() {
-        return 3;
-      }
-      embeddingDataType(): arrow.Float {
-        return new arrow.Float32();
-      }
-      async computeSourceEmbeddings(data: string[]) {
-        return data.map(() => [1, 2, 3]);
-      }
-    }
-
-    const func = getRegistry()
-      .get<MockEmbeddingFunction>("mock-embedding")!
-      .create();
-
-    const schema = LanceSchema({
-      id: new arrow.Int32(),
-      text: func.sourceField(new arrow.Utf8()),
-      vector: func.vectorField(),
-    });
-
-    const db = await connect(tmpDir.name);
-    const table = await db.createTable(
-      "test",
-      [
-        { id: 1, text: "hello" },
-        { id: 2, text: "world" },
-      ],
-      { schema },
-    );
-    const expected = [
-      [1, 2, 3],
-      [1, 2, 3],
-    ];
-    const actual = await table.query().toArrow();
-    const vectors = actual
-      .getChild("vector")
-      ?.toArray()
-      .map((x: unknown) => {
-        if (x instanceof arrow.Vector) {
-          return [...x];
-        } else {
-          return x;
-        }
-      });
-    expect(vectors).toEqual(expected);
-  });
-  test("should error if registering with the same name", async () => {
-    class MockEmbeddingFunction extends EmbeddingFunction<string> {
-      toJSON(): object {
-        return {
-          someText: "hello",
-        };
-      }
-      constructor() {
-        super();
-      }
-      ndims() {
-        return 3;
-      }
-      embeddingDataType(): arrow.Float {
-        return new arrow.Float32();
-      }
-      async computeSourceEmbeddings(data: string[]) {
-        return data.map(() => [1, 2, 3]);
-      }
-    }
-    register("mock-embedding")(MockEmbeddingFunction);
-    expect(() => register("mock-embedding")(MockEmbeddingFunction)).toThrow(
-      'Embedding function with alias "mock-embedding" already exists',
-    );
-  });
-  test("schema should contain correct metadata", async () => {
-    class MockEmbeddingFunction extends EmbeddingFunction<string> {
-      toJSON(): object {
-        return {
-          someText: "hello",
-        };
-      }
-      constructor() {
-        super();
-      }
-      ndims() {
-        return 3;
-      }
-      embeddingDataType(): arrow.Float {
-        return new arrow.Float32();
-      }
-      async computeSourceEmbeddings(data: string[]) {
-        return data.map(() => [1, 2, 3]);
-      }
-    }
-    const func = new MockEmbeddingFunction();
-
-    const schema = LanceSchema({
-      id: new arrow.Int32(),
-      text: func.sourceField(new arrow.Utf8()),
-      vector: func.vectorField(),
-    });
-    const expectedMetadata = new Map<string, string>([
-      [
-        "embedding_functions",
-        JSON.stringify([
-          {
-            sourceColumn: "text",
-            vectorColumn: "vector",
-            name: "MockEmbeddingFunction",
-            model: { someText: "hello" },
-          },
-        ]),
-      ],
-    ]);
-    expect(schema.metadata).toEqual(expectedMetadata);
-  });
-});
--- a/Show More
+++ b/Show More
Author	SHA1	Message	Date
ayush chaurasia	db67b27a42	ruff	2024-04-16 12:11:57 +05:30
ayush chaurasia	4b0820ef15	ruff	2024-04-16 12:09:18 +05:30
ayush chaurasia	c7bb919561	update benchmark script	2024-04-16 11:13:52 +05:30
ayush chaurasia	df404b726e	ruff	2024-04-16 10:00:47 +05:30
ayush chaurasia	ffbb104648	remove protected namespaces	2024-04-16 09:51:08 +05:30
ayush chaurasia	3ebd561fd9	ruff	2024-04-16 09:26:47 +05:30
ayush chaurasia	6bc488f674	update	2024-04-16 09:24:29 +05:30
ayush chaurasia	ea34c0b4c4	update	2024-04-16 09:07:40 +05:30
ayush chaurasia	1a827925eb	update docs	2024-04-16 08:59:36 +05:30
ayush chaurasia	fe5888d661	update usage	2024-04-16 08:24:40 +05:30
ayush chaurasia	6074e6b7ee	update	2024-04-15 17:19:04 +05:30
ayush chaurasia	fd8de238bb	ruff	2024-04-15 17:11:59 +05:30
ayush chaurasia	d0c1113417	add test	2024-04-15 17:07:29 +05:30
ayush chaurasia	3ca96a852f	remove test file	2024-04-15 17:02:58 +05:30
ayush chaurasia	9428c6b565	update	2024-04-15 16:59:16 +05:30
ayush chaurasia	ff00a3242c	update	2024-04-15 07:52:04 +05:30
ayush chaurasia	878deb73a0	update	2024-04-15 07:51:05 +05:30
ayush chaurasia	c75bb65609	update	2024-04-15 05:59:26 +05:30
				`@@ -1 +0,0 @@`
				`$d51afd07-e3cd-4c76-9b9b-787e13fd55b0<62>=id <20><><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD>int3208name <20><><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD>string08`
				`@@ -1 +0,0 @@`
				`$15648e72-076f-4ef1-8b90-10d305b95b3b<33>=id <20><><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD>int3208name <20><><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD>string08`
				`@@ -1 +0,0 @@`
				`$a3689caf-4f6b-4afc-a3c7-97af75661843<34>oitem <20><><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD>string8price <20><><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD>double80vector <20><><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD><EFBFBD>*fixed_size_list:float:28`