mirror of
https://github.com/lancedb/lancedb.git
synced 2026-09-11 15:52:17 +00:00
Compare commits
10
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b5c3344985 | ||
|
|
13f9dd630b | ||
|
|
bc4497b21a | ||
|
|
1da5876870 | ||
|
|
577fb48376 | ||
|
|
c7b051aff7 | ||
|
|
2e205ac9bb | ||
|
|
3e3878b223 | ||
|
|
19fb665c76 | ||
|
|
1f95398c34 |
+1
-1
@@ -1,5 +1,5 @@
|
|||||||
[tool.bumpversion]
|
[tool.bumpversion]
|
||||||
current_version = "0.39.0-beta.4"
|
current_version = "0.39.0-beta.6"
|
||||||
parse = """(?x)
|
parse = """(?x)
|
||||||
(?P<major>0|[1-9]\\d*)\\.
|
(?P<major>0|[1-9]\\d*)\\.
|
||||||
(?P<minor>0|[1-9]\\d*)\\.
|
(?P<minor>0|[1-9]\\d*)\\.
|
||||||
|
|||||||
@@ -0,0 +1,20 @@
|
|||||||
|
name: Typo checker
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches:
|
||||||
|
- main
|
||||||
|
pull_request:
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
run:
|
||||||
|
name: Spell Check with Typos
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Check out code
|
||||||
|
uses: actions/checkout@v6
|
||||||
|
|
||||||
|
- name: Check spelling of the entire repository
|
||||||
|
uses: crate-ci/typos@6802cc60d4e7f78b9d5454f6cf3935c042d5e1e3 # v1.26.0
|
||||||
@@ -10,6 +10,10 @@ repos:
|
|||||||
rev: v0.9.9
|
rev: v0.9.9
|
||||||
hooks:
|
hooks:
|
||||||
- id: ruff
|
- id: ruff
|
||||||
|
- repo: https://github.com/crate-ci/typos
|
||||||
|
rev: v1.26.0
|
||||||
|
hooks:
|
||||||
|
- id: typos
|
||||||
# - repo: https://github.com/RobertCraigie/pyright-python
|
# - repo: https://github.com/RobertCraigie/pyright-python
|
||||||
# rev: v1.1.395
|
# rev: v1.1.395
|
||||||
# hooks:
|
# hooks:
|
||||||
|
|||||||
+19
@@ -0,0 +1,19 @@
|
|||||||
|
[default]
|
||||||
|
extend-ignore-re = ["(?Rm)^.*(#|//)\\s*spellchecker:disable-line$"]
|
||||||
|
|
||||||
|
[default.extend-words]
|
||||||
|
# Azure Kubernetes Service, mentioned in rust/lancedb/src/remote/oauth.rs.
|
||||||
|
AKS = "AKS"
|
||||||
|
# RabitQ is the name of a vector quantization algorithm, not a typo of "Rabbit".
|
||||||
|
Rabit = "Rabit"
|
||||||
|
# `VarBuilder::from_mmaped_safetensors` is the real (if oddly-spelled) name of
|
||||||
|
# the candle-core API we call in rust/lancedb/src/embeddings/sentence_transformers.rs.
|
||||||
|
mmaped = "mmaped"
|
||||||
|
# `WriteableBuffer` is the real name of a type from Python's `_typeshed` stubs,
|
||||||
|
# used in python/python/lancedb/_blob.py.
|
||||||
|
Writeable = "Writeable"
|
||||||
|
|
||||||
|
[files]
|
||||||
|
extend-exclude = [
|
||||||
|
"*_THIRD_PARTY_LICENSES.*",
|
||||||
|
]
|
||||||
Generated
+257
-246
File diff suppressed because it is too large
Load Diff
+15
-15
@@ -13,20 +13,20 @@ categories = ["database-implementations"]
|
|||||||
rust-version = "1.91.0"
|
rust-version = "1.91.0"
|
||||||
|
|
||||||
[workspace.dependencies]
|
[workspace.dependencies]
|
||||||
lance = { "version" = "=12.0.0-beta.14", default-features = false, "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" }
|
lance = { "version" = "=12.0.0-beta.17", default-features = false, "tag" = "v12.0.0-beta.17", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-core = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" }
|
lance-core = { "version" = "=12.0.0-beta.17", "tag" = "v12.0.0-beta.17", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-datagen = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" }
|
lance-datagen = { "version" = "=12.0.0-beta.17", "tag" = "v12.0.0-beta.17", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-file = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" }
|
lance-file = { "version" = "=12.0.0-beta.17", "tag" = "v12.0.0-beta.17", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-io = { "version" = "=12.0.0-beta.14", default-features = false, "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" }
|
lance-io = { "version" = "=12.0.0-beta.17", default-features = false, "tag" = "v12.0.0-beta.17", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-index = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" }
|
lance-index = { "version" = "=12.0.0-beta.17", "tag" = "v12.0.0-beta.17", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-linalg = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" }
|
lance-linalg = { "version" = "=12.0.0-beta.17", "tag" = "v12.0.0-beta.17", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-namespace = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" }
|
lance-namespace = { "version" = "=12.0.0-beta.17", "tag" = "v12.0.0-beta.17", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-namespace-impls = { "version" = "=12.0.0-beta.14", default-features = false, "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" }
|
lance-namespace-impls = { "version" = "=12.0.0-beta.17", default-features = false, "tag" = "v12.0.0-beta.17", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-table = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" }
|
lance-table = { "version" = "=12.0.0-beta.17", "tag" = "v12.0.0-beta.17", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-testing = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" }
|
lance-testing = { "version" = "=12.0.0-beta.17", "tag" = "v12.0.0-beta.17", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-datafusion = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" }
|
lance-datafusion = { "version" = "=12.0.0-beta.17", "tag" = "v12.0.0-beta.17", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-encoding = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" }
|
lance-encoding = { "version" = "=12.0.0-beta.17", "tag" = "v12.0.0-beta.17", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-arrow = { "version" = "=12.0.0-beta.14", "tag" = "v12.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" }
|
lance-arrow = { "version" = "=12.0.0-beta.17", "tag" = "v12.0.0-beta.17", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lancedb = { path = "rust/lancedb", default-features = false }
|
lancedb = { path = "rust/lancedb", default-features = false }
|
||||||
ahash = "0.8"
|
ahash = "0.8"
|
||||||
# Note that this one does not include pyarrow
|
# Note that this one does not include pyarrow
|
||||||
@@ -60,7 +60,7 @@ log = "0.4"
|
|||||||
metrics = "0.24"
|
metrics = "0.24"
|
||||||
metrics-util = "0.19"
|
metrics-util = "0.19"
|
||||||
moka = { version = "0.12", features = ["future"] }
|
moka = { version = "0.12", features = ["future"] }
|
||||||
object_store = "0.13.2"
|
object_store = "0.14.1"
|
||||||
pin-project = "1.0.7"
|
pin-project = "1.0.7"
|
||||||
rand = "0.9"
|
rand = "0.9"
|
||||||
snafu = "0.8"
|
snafu = "0.8"
|
||||||
|
|||||||
+1
-1
@@ -155,7 +155,7 @@ paths:
|
|||||||
vector:
|
vector:
|
||||||
type: FixedSizeList
|
type: FixedSizeList
|
||||||
description: |
|
description: |
|
||||||
The targetted vector to search for. Required.
|
The targeted vector to search for. Required.
|
||||||
vector_column:
|
vector_column:
|
||||||
type: string
|
type: string
|
||||||
description: |
|
description: |
|
||||||
|
|||||||
@@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`:
|
|||||||
<dependency>
|
<dependency>
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-core</artifactId>
|
<artifactId>lancedb-core</artifactId>
|
||||||
<version>0.39.0-beta.4</version>
|
<version>0.39.0-beta.6</version>
|
||||||
</dependency>
|
</dependency>
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|||||||
@@ -141,7 +141,7 @@ Currently this causes multiple copies of the row to be created
|
|||||||
but that behavior is subject to change.
|
but that behavior is subject to change.
|
||||||
|
|
||||||
An optional condition may be specified. If it is, then only
|
An optional condition may be specified. If it is, then only
|
||||||
matched rows that satisfy the condtion will be updated. Any
|
matched rows that satisfy the condition will be updated. Any
|
||||||
rows that do not satisfy the condition will be left as they
|
rows that do not satisfy the condition will be left as they
|
||||||
are. Failing to satisfy the condition does not cause a
|
are. Failing to satisfy the condition does not cause a
|
||||||
"matched row" to become a "not matched" row.
|
"matched row" to become a "not matched" row.
|
||||||
|
|||||||
@@ -1266,7 +1266,7 @@ value is 0")
|
|||||||
Note: if your condition is something like "some_id_column == 7" and
|
Note: if your condition is something like "some_id_column == 7" and
|
||||||
you are updating many rows (with different ids) then you will get
|
you are updating many rows (with different ids) then you will get
|
||||||
better performance with a single [`merge_insert`] call instead of
|
better performance with a single [`merge_insert`] call instead of
|
||||||
repeatedly calilng this method.
|
repeatedly calling this method.
|
||||||
|
|
||||||
##### Parameters
|
##### Parameters
|
||||||
|
|
||||||
|
|||||||
@@ -118,7 +118,7 @@ Number of sub-vectors of PQ.
|
|||||||
This value controls how much the vector is compressed during the quantization step.
|
This value controls how much the vector is compressed during the quantization step.
|
||||||
The more sub vectors there are the less the vector is compressed. The default is
|
The more sub vectors there are the less the vector is compressed. The default is
|
||||||
the dimension of the vector divided by 16. If the dimension is not evenly divisible
|
the dimension of the vector divided by 16. If the dimension is not evenly divisible
|
||||||
by 16 we use the dimension divded by 8.
|
by 16 we use the dimension divided by 8.
|
||||||
|
|
||||||
The above two cases are highly preferred. Having 8 or 16 values per subvector allows
|
The above two cases are highly preferred. Having 8 or 16 values per subvector allows
|
||||||
us to use efficient SIMD instructions.
|
us to use efficient SIMD instructions.
|
||||||
|
|||||||
@@ -16,7 +16,7 @@ optional config: Index;
|
|||||||
|
|
||||||
Advanced index configuration
|
Advanced index configuration
|
||||||
|
|
||||||
This option allows you to specify a specfic index to create and also
|
This option allows you to specify a specific index to create and also
|
||||||
allows you to pass in configuration for training the index.
|
allows you to pass in configuration for training the index.
|
||||||
|
|
||||||
See the static methods on Index for details on the various index types.
|
See the static methods on Index for details on the various index types.
|
||||||
|
|||||||
@@ -112,7 +112,7 @@ Number of sub-vectors of PQ.
|
|||||||
This value controls how much the vector is compressed during the quantization step.
|
This value controls how much the vector is compressed during the quantization step.
|
||||||
The more sub vectors there are the less the vector is compressed. The default is
|
The more sub vectors there are the less the vector is compressed. The default is
|
||||||
the dimension of the vector divided by 16. If the dimension is not evenly divisible
|
the dimension of the vector divided by 16. If the dimension is not evenly divisible
|
||||||
by 16 we use the dimension divded by 8.
|
by 16 we use the dimension divided by 8.
|
||||||
|
|
||||||
The above two cases are highly preferred. Having 8 or 16 values per subvector allows
|
The above two cases are highly preferred. Having 8 or 16 values per subvector allows
|
||||||
us to use efficient SIMD instructions.
|
us to use efficient SIMD instructions.
|
||||||
|
|||||||
@@ -8,7 +8,7 @@
|
|||||||
<parent>
|
<parent>
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-parent</artifactId>
|
<artifactId>lancedb-parent</artifactId>
|
||||||
<version>0.39.0-beta.4</version>
|
<version>0.39.0-beta.6</version>
|
||||||
<relativePath>../pom.xml</relativePath>
|
<relativePath>../pom.xml</relativePath>
|
||||||
</parent>
|
</parent>
|
||||||
|
|
||||||
|
|||||||
+2
-2
@@ -6,7 +6,7 @@
|
|||||||
|
|
||||||
<groupId>com.lancedb</groupId>
|
<groupId>com.lancedb</groupId>
|
||||||
<artifactId>lancedb-parent</artifactId>
|
<artifactId>lancedb-parent</artifactId>
|
||||||
<version>0.39.0-beta.4</version>
|
<version>0.39.0-beta.6</version>
|
||||||
<packaging>pom</packaging>
|
<packaging>pom</packaging>
|
||||||
<name>${project.artifactId}</name>
|
<name>${project.artifactId}</name>
|
||||||
<description>LanceDB Java SDK Parent POM</description>
|
<description>LanceDB Java SDK Parent POM</description>
|
||||||
@@ -28,7 +28,7 @@
|
|||||||
<properties>
|
<properties>
|
||||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||||
<arrow.version>15.0.0</arrow.version>
|
<arrow.version>15.0.0</arrow.version>
|
||||||
<lance-core.version>12.0.0-beta.14</lance-core.version>
|
<lance-core.version>12.0.0-beta.17</lance-core.version>
|
||||||
<spotless.skip>false</spotless.skip>
|
<spotless.skip>false</spotless.skip>
|
||||||
<spotless.version>2.30.0</spotless.version>
|
<spotless.version>2.30.0</spotless.version>
|
||||||
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
|
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
|
||||||
|
|||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "lancedb-nodejs"
|
name = "lancedb-nodejs"
|
||||||
edition.workspace = true
|
edition.workspace = true
|
||||||
version = "0.39.0-beta.4"
|
version = "0.39.0-beta.6"
|
||||||
publish = false
|
publish = false
|
||||||
license.workspace = true
|
license.workspace = true
|
||||||
description.workspace = true
|
description.workspace = true
|
||||||
|
|||||||
@@ -281,7 +281,7 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
numIndices: 0,
|
numIndices: 0,
|
||||||
numRows: 3,
|
numRows: 3,
|
||||||
// Full on-disk size of the two data files, footers and metadata included.
|
// Full on-disk size of the two data files, footers and metadata included.
|
||||||
totalBytes: 684,
|
totalBytes: 550,
|
||||||
});
|
});
|
||||||
|
|
||||||
// Index files count toward totalBytes too (only deletion files and
|
// Index files count toward totalBytes too (only deletion files and
|
||||||
@@ -289,7 +289,7 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
await table.createIndex("id", { config: Index.btree() });
|
await table.createIndex("id", { config: Index.btree() });
|
||||||
const statsWithIndex = await table.stats();
|
const statsWithIndex = await table.stats();
|
||||||
expect(statsWithIndex.numIndices).toBe(1);
|
expect(statsWithIndex.numIndices).toBe(1);
|
||||||
expect(statsWithIndex.totalBytes).toBeGreaterThan(684);
|
expect(statsWithIndex.totalBytes).toBeGreaterThan(550);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("should overwrite data if asked", async () => {
|
it("should overwrite data if asked", async () => {
|
||||||
@@ -3252,7 +3252,7 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
const db = await connect(tmpDir.name);
|
const db = await connect(tmpDir.name);
|
||||||
const data = [
|
const data = [
|
||||||
{ text: "fa", vector: [0.1, 0.2, 0.3] },
|
{ text: "fa", vector: [0.1, 0.2, 0.3] },
|
||||||
{ text: "fo", vector: [0.4, 0.5, 0.6] },
|
{ text: "fo", vector: [0.4, 0.5, 0.6] }, // spellchecker:disable-line
|
||||||
{ text: "fob", vector: [0.4, 0.5, 0.6] },
|
{ text: "fob", vector: [0.4, 0.5, 0.6] },
|
||||||
{ text: "focus", vector: [0.4, 0.5, 0.6] },
|
{ text: "focus", vector: [0.4, 0.5, 0.6] },
|
||||||
{ text: "foo", vector: [0.4, 0.5, 0.6] },
|
{ text: "foo", vector: [0.4, 0.5, 0.6] },
|
||||||
@@ -3277,7 +3277,7 @@ describe.each([arrow15, arrow16, arrow17, arrow18])(
|
|||||||
const resultSet = new Set(fuzzyResults.map((r) => r.text));
|
const resultSet = new Set(fuzzyResults.map((r) => r.text));
|
||||||
expect(resultSet.has("foo")).toBe(true);
|
expect(resultSet.has("foo")).toBe(true);
|
||||||
expect(resultSet.has("fob")).toBe(true);
|
expect(resultSet.has("fob")).toBe(true);
|
||||||
expect(resultSet.has("fo")).toBe(true);
|
expect(resultSet.has("fo")).toBe(true); // spellchecker:disable-line
|
||||||
expect(resultSet.has("food")).toBe(true);
|
expect(resultSet.has("food")).toBe(true);
|
||||||
|
|
||||||
const prefixResults = await table
|
const prefixResults = await table
|
||||||
|
|||||||
@@ -600,7 +600,7 @@ function makeVector(
|
|||||||
}
|
}
|
||||||
if (values.length === 0) {
|
if (values.length === 0) {
|
||||||
throw Error(
|
throw Error(
|
||||||
"makeVector requires at least one value or the type must be specfied",
|
"makeVector requires at least one value or the type must be specified",
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
const sampleValue = values.find((val) => val !== null && val !== undefined);
|
const sampleValue = values.find((val) => val !== null && val !== undefined);
|
||||||
@@ -858,7 +858,7 @@ async function applyEmbeddings<T>(
|
|||||||
* customized by the `embeddingDataType` property of the embedding function.
|
* customized by the `embeddingDataType` property of the embedding function.
|
||||||
*
|
*
|
||||||
* If a schema is provided in `makeTableOptions` then it should include the
|
* If a schema is provided in `makeTableOptions` then it should include the
|
||||||
* embedding columns. If no schema is provded then embedding columns will
|
* embedding columns. If no schema is provided then embedding columns will
|
||||||
* be placed at the end of the table, after all of the input columns.
|
* be placed at the end of the table, after all of the input columns.
|
||||||
*/
|
*/
|
||||||
export async function convertToTable(
|
export async function convertToTable(
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ export interface IvfPqOptions {
|
|||||||
* This value controls how much the vector is compressed during the quantization step.
|
* This value controls how much the vector is compressed during the quantization step.
|
||||||
* The more sub vectors there are the less the vector is compressed. The default is
|
* The more sub vectors there are the less the vector is compressed. The default is
|
||||||
* the dimension of the vector divided by 16. If the dimension is not evenly divisible
|
* the dimension of the vector divided by 16. If the dimension is not evenly divisible
|
||||||
* by 16 we use the dimension divded by 8.
|
* by 16 we use the dimension divided by 8.
|
||||||
*
|
*
|
||||||
* The above two cases are highly preferred. Having 8 or 16 values per subvector allows
|
* The above two cases are highly preferred. Having 8 or 16 values per subvector allows
|
||||||
* us to use efficient SIMD instructions.
|
* us to use efficient SIMD instructions.
|
||||||
@@ -228,7 +228,7 @@ export interface HnswPqOptions {
|
|||||||
* This value controls how much the vector is compressed during the quantization step.
|
* This value controls how much the vector is compressed during the quantization step.
|
||||||
* The more sub vectors there are the less the vector is compressed. The default is
|
* The more sub vectors there are the less the vector is compressed. The default is
|
||||||
* the dimension of the vector divided by 16. If the dimension is not evenly divisible
|
* the dimension of the vector divided by 16. If the dimension is not evenly divisible
|
||||||
* by 16 we use the dimension divded by 8.
|
* by 16 we use the dimension divided by 8.
|
||||||
*
|
*
|
||||||
* The above two cases are highly preferred. Having 8 or 16 values per subvector allows
|
* The above two cases are highly preferred. Having 8 or 16 values per subvector allows
|
||||||
* us to use efficient SIMD instructions.
|
* us to use efficient SIMD instructions.
|
||||||
@@ -825,7 +825,7 @@ export interface IndexOptions {
|
|||||||
/**
|
/**
|
||||||
* Advanced index configuration
|
* Advanced index configuration
|
||||||
*
|
*
|
||||||
* This option allows you to specify a specfic index to create and also
|
* This option allows you to specify a specific index to create and also
|
||||||
* allows you to pass in configuration for training the index.
|
* allows you to pass in configuration for training the index.
|
||||||
*
|
*
|
||||||
* See the static methods on Index for details on the various index types.
|
* See the static methods on Index for details on the various index types.
|
||||||
|
|||||||
@@ -27,7 +27,7 @@ export class MergeInsertBuilder {
|
|||||||
* but that behavior is subject to change.
|
* but that behavior is subject to change.
|
||||||
*
|
*
|
||||||
* An optional condition may be specified. If it is, then only
|
* An optional condition may be specified. If it is, then only
|
||||||
* matched rows that satisfy the condtion will be updated. Any
|
* matched rows that satisfy the condition will be updated. Any
|
||||||
* rows that do not satisfy the condition will be left as they
|
* rows that do not satisfy the condition will be left as they
|
||||||
* are. Failing to satisfy the condition does not cause a
|
* are. Failing to satisfy the condition does not cause a
|
||||||
* "matched row" to become a "not matched" row.
|
* "matched row" to become a "not matched" row.
|
||||||
|
|||||||
@@ -3,7 +3,7 @@
|
|||||||
|
|
||||||
// The utilities in this file help sanitize data from the user's arrow
|
// The utilities in this file help sanitize data from the user's arrow
|
||||||
// library into the types expected by vectordb's arrow library. Node
|
// library into the types expected by vectordb's arrow library. Node
|
||||||
// generally allows for mulitple versions of the same library (and sometimes
|
// generally allows for multiple versions of the same library (and sometimes
|
||||||
// even multiple copies of the same version) to be installed at the same
|
// even multiple copies of the same version) to be installed at the same
|
||||||
// time. However, arrow-js uses instanceof which expected that the input
|
// time. However, arrow-js uses instanceof which expected that the input
|
||||||
// comes from the exact same library instance. This is not always the case
|
// comes from the exact same library instance. This is not always the case
|
||||||
|
|||||||
@@ -313,7 +313,7 @@ export abstract class Table {
|
|||||||
* Note: if your condition is something like "some_id_column == 7" and
|
* Note: if your condition is something like "some_id_column == 7" and
|
||||||
* you are updating many rows (with different ids) then you will get
|
* you are updating many rows (with different ids) then you will get
|
||||||
* better performance with a single [`merge_insert`] call instead of
|
* better performance with a single [`merge_insert`] call instead of
|
||||||
* repeatedly calilng this method.
|
* repeatedly calling this method.
|
||||||
* @param {Map<string, string> | Record<string, string>} updates - the
|
* @param {Map<string, string> | Record<string, string>} updates - the
|
||||||
* columns to update
|
* columns to update
|
||||||
* @returns {Promise<UpdateResult>} A promise that resolves to an object
|
* @returns {Promise<UpdateResult>} A promise that resolves to an object
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-darwin-arm64",
|
"name": "@lancedb/lancedb-darwin-arm64",
|
||||||
"version": "0.39.0-beta.4",
|
"version": "0.39.0-beta.6",
|
||||||
"os": ["darwin"],
|
"os": ["darwin"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.darwin-arm64.node",
|
"main": "lancedb.darwin-arm64.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
||||||
"version": "0.39.0-beta.4",
|
"version": "0.39.0-beta.6",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.linux-arm64-gnu.node",
|
"main": "lancedb.linux-arm64-gnu.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-arm64-musl",
|
"name": "@lancedb/lancedb-linux-arm64-musl",
|
||||||
"version": "0.39.0-beta.4",
|
"version": "0.39.0-beta.6",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["arm64"],
|
"cpu": ["arm64"],
|
||||||
"main": "lancedb.linux-arm64-musl.node",
|
"main": "lancedb.linux-arm64-musl.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-x64-gnu",
|
"name": "@lancedb/lancedb-linux-x64-gnu",
|
||||||
"version": "0.39.0-beta.4",
|
"version": "0.39.0-beta.6",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["x64"],
|
"cpu": ["x64"],
|
||||||
"main": "lancedb.linux-x64-gnu.node",
|
"main": "lancedb.linux-x64-gnu.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-linux-x64-musl",
|
"name": "@lancedb/lancedb-linux-x64-musl",
|
||||||
"version": "0.39.0-beta.4",
|
"version": "0.39.0-beta.6",
|
||||||
"os": ["linux"],
|
"os": ["linux"],
|
||||||
"cpu": ["x64"],
|
"cpu": ["x64"],
|
||||||
"main": "lancedb.linux-x64-musl.node",
|
"main": "lancedb.linux-x64-musl.node",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-win32-arm64-msvc",
|
"name": "@lancedb/lancedb-win32-arm64-msvc",
|
||||||
"version": "0.39.0-beta.4",
|
"version": "0.39.0-beta.6",
|
||||||
"os": [
|
"os": [
|
||||||
"win32"
|
"win32"
|
||||||
],
|
],
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@lancedb/lancedb-win32-x64-msvc",
|
"name": "@lancedb/lancedb-win32-x64-msvc",
|
||||||
"version": "0.39.0-beta.4",
|
"version": "0.39.0-beta.6",
|
||||||
"os": ["win32"],
|
"os": ["win32"],
|
||||||
"cpu": ["x64"],
|
"cpu": ["x64"],
|
||||||
"main": "lancedb.win32-x64-msvc.node",
|
"main": "lancedb.win32-x64-msvc.node",
|
||||||
|
|||||||
+1
-1
@@ -11,7 +11,7 @@
|
|||||||
"ann"
|
"ann"
|
||||||
],
|
],
|
||||||
"private": false,
|
"private": false,
|
||||||
"version": "0.39.0-beta.4",
|
"version": "0.39.0-beta.6",
|
||||||
"main": "dist/index.js",
|
"main": "dist/index.js",
|
||||||
"exports": {
|
"exports": {
|
||||||
".": "./dist/index.js",
|
".": "./dist/index.js",
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "lancedb-python"
|
name = "lancedb-python"
|
||||||
version = "0.39.0-beta.4"
|
version = "0.39.0-beta.6"
|
||||||
publish = false
|
publish = false
|
||||||
edition.workspace = true
|
edition.workspace = true
|
||||||
description = "Python bindings for LanceDB"
|
description = "Python bindings for LanceDB"
|
||||||
|
|||||||
@@ -21,7 +21,7 @@ class GteEmbeddings(TextEmbeddingFunction):
|
|||||||
An embedding function that uses GTE-LARGE MLX format(for Apple silicon devices only)
|
An embedding function that uses GTE-LARGE MLX format(for Apple silicon devices only)
|
||||||
as well as the standard cpu/gpu version from: https://huggingface.co/thenlper/gte-large.
|
as well as the standard cpu/gpu version from: https://huggingface.co/thenlper/gte-large.
|
||||||
|
|
||||||
For Apple users, you will need the mlx package insalled, which can be done with:
|
For Apple users, you will need the mlx package installed, which can be done with:
|
||||||
pip install mlx
|
pip install mlx
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
|
|||||||
@@ -60,7 +60,7 @@ class InstructorEmbeddingFunction(TextEmbeddingFunction):
|
|||||||
|
|
||||||
import lancedb
|
import lancedb
|
||||||
from lancedb.pydantic import LanceModel, Vector
|
from lancedb.pydantic import LanceModel, Vector
|
||||||
from lancedb.embeddings import get_registry, InstuctorEmbeddingFunction
|
from lancedb.embeddings import get_registry, InstructorEmbeddingFunction
|
||||||
|
|
||||||
instructor = get_registry().get("instructor").create(
|
instructor = get_registry().get("instructor").create(
|
||||||
source_instruction="represent the document for retrieval",
|
source_instruction="represent the document for retrieval",
|
||||||
|
|||||||
@@ -751,7 +751,7 @@ class IvfPq:
|
|||||||
This value controls how much the vector is compressed during the
|
This value controls how much the vector is compressed during the
|
||||||
quantization step. The more sub vectors there are the less the vector is
|
quantization step. The more sub vectors there are the less the vector is
|
||||||
compressed. The default is the dimension of the vector divided by 16. If
|
compressed. The default is the dimension of the vector divided by 16. If
|
||||||
the dimension is not evenly divisible by 16 we use the dimension divded by
|
the dimension is not evenly divisible by 16 we use the dimension divided by
|
||||||
8.
|
8.
|
||||||
|
|
||||||
The above two cases are highly preferred. Having 8 or 16 values per
|
The above two cases are highly preferred. Having 8 or 16 values per
|
||||||
|
|||||||
@@ -78,6 +78,10 @@ if TYPE_CHECKING:
|
|||||||
T = TypeVar("T", bound="LanceModel")
|
T = TypeVar("T", bound="LanceModel")
|
||||||
AnalyzePlanDistributedMetrics = Literal["aggregate", "per_worker", "full"]
|
AnalyzePlanDistributedMetrics = Literal["aggregate", "per_worker", "full"]
|
||||||
|
|
||||||
|
# Number of rows a hybrid query returns when no limit was set on it. This
|
||||||
|
# mirrors the default the Rust query builder applies to its sub-queries.
|
||||||
|
DEFAULT_HYBRID_LIMIT = 10
|
||||||
|
|
||||||
|
|
||||||
@runtime_checkable
|
@runtime_checkable
|
||||||
class _LanceScanner(Protocol):
|
class _LanceScanner(Protocol):
|
||||||
@@ -859,7 +863,7 @@ class Query(pydantic.BaseModel):
|
|||||||
return query
|
return query
|
||||||
|
|
||||||
# This tells pydantic to allow custom types (needed for the `vector` query since
|
# This tells pydantic to allow custom types (needed for the `vector` query since
|
||||||
# pa.Array wouln't be allowed otherwise)
|
# pa.Array wouldn't be allowed otherwise)
|
||||||
model_config = pydantic.ConfigDict(arbitrary_types_allowed=True)
|
model_config = pydantic.ConfigDict(arbitrary_types_allowed=True)
|
||||||
|
|
||||||
|
|
||||||
@@ -3893,14 +3897,54 @@ class AsyncHybridQuery(AsyncStandardQuery, AsyncVectorQueryBase):
|
|||||||
|
|
||||||
return self
|
return self
|
||||||
|
|
||||||
|
def _create_child_queries(
|
||||||
|
self,
|
||||||
|
) -> Tuple["AsyncFTSQuery", "AsyncVectorQuery", int, int]:
|
||||||
|
"""Build the sub-queries that make up this hybrid query.
|
||||||
|
|
||||||
|
Execution, `explain_plan` and `analyze_plan` all go through here so that
|
||||||
|
the plans that are reported are the plans that actually run.
|
||||||
|
|
||||||
|
Returns the two sub-queries along with the effective limit and offset of
|
||||||
|
the hybrid query itself.
|
||||||
|
"""
|
||||||
|
fts_query = AsyncFTSQuery(self._inner.to_fts_query(), self._table)
|
||||||
|
vec_query = AsyncVectorQuery(self._inner.to_vector_query(), self._table)
|
||||||
|
|
||||||
|
fts_req = fts_query._inner.to_query_request()
|
||||||
|
vec_req = vec_query._inner.to_query_request()
|
||||||
|
|
||||||
|
# Only one of the two sub-queries carries the limit when it was never
|
||||||
|
# set explicitly: nearest_to()/nearest_to_text() build the sibling query
|
||||||
|
# from scratch, and that is where the default gets filled in. Which one
|
||||||
|
# that is depends on the order the hybrid query was built in, so look at
|
||||||
|
# both rather than at a single side.
|
||||||
|
limit = fts_req.limit if fts_req.limit is not None else vec_req.limit
|
||||||
|
if limit is None:
|
||||||
|
limit = DEFAULT_HYBRID_LIMIT
|
||||||
|
offset = fts_req.offset or vec_req.offset or 0
|
||||||
|
|
||||||
|
fts_query.with_row_id()
|
||||||
|
vec_query.with_row_id()
|
||||||
|
|
||||||
|
# offset() pushes the offset down into both sub-queries, which would make
|
||||||
|
# each of them skip its own first `offset` rows. The window has to be
|
||||||
|
# taken out of the combined, reranked results instead, so fetch the
|
||||||
|
# skipped prefix here too and slice it off afterwards.
|
||||||
|
fts_query.limit(limit + offset)
|
||||||
|
vec_query.limit(limit + offset)
|
||||||
|
fts_query.offset(0)
|
||||||
|
vec_query.offset(0)
|
||||||
|
|
||||||
|
return fts_query, vec_query, limit, offset
|
||||||
|
|
||||||
async def to_batches(
|
async def to_batches(
|
||||||
self,
|
self,
|
||||||
*,
|
*,
|
||||||
max_batch_length: Optional[int] = None,
|
max_batch_length: Optional[int] = None,
|
||||||
timeout: Optional[timedelta] = None,
|
timeout: Optional[timedelta] = None,
|
||||||
) -> AsyncRecordBatchReader:
|
) -> AsyncRecordBatchReader:
|
||||||
fts_query = AsyncFTSQuery(self._inner.to_fts_query(), self._table)
|
fts_query, vec_query, limit, offset = self._create_child_queries()
|
||||||
vec_query = AsyncVectorQuery(self._inner.to_vector_query(), self._table)
|
|
||||||
|
|
||||||
req = fts_query._inner.to_query_request()
|
req = fts_query._inner.to_query_request()
|
||||||
blob_auto_row_id = False
|
blob_auto_row_id = False
|
||||||
@@ -3920,9 +3964,6 @@ class AsyncHybridQuery(AsyncStandardQuery, AsyncVectorQueryBase):
|
|||||||
self._blob_auto_row_id = blob_auto_row_id
|
self._blob_auto_row_id = blob_auto_row_id
|
||||||
self._blob_paths = blob_paths
|
self._blob_paths = blob_paths
|
||||||
|
|
||||||
fts_query.with_row_id()
|
|
||||||
vec_query.with_row_id()
|
|
||||||
|
|
||||||
fts_results, vector_results = await asyncio.gather(
|
fts_results, vector_results = await asyncio.gather(
|
||||||
fts_query.to_arrow(timeout=timeout),
|
fts_query.to_arrow(timeout=timeout),
|
||||||
vec_query.to_arrow(timeout=timeout),
|
vec_query.to_arrow(timeout=timeout),
|
||||||
@@ -3934,8 +3975,9 @@ class AsyncHybridQuery(AsyncStandardQuery, AsyncVectorQueryBase):
|
|||||||
norm=self._norm,
|
norm=self._norm,
|
||||||
fts_query=fts_query.get_query(),
|
fts_query=fts_query.get_query(),
|
||||||
reranker=self._reranker,
|
reranker=self._reranker,
|
||||||
limit=self._inner.get_limit(),
|
limit=limit,
|
||||||
with_row_ids=True,
|
with_row_ids=True,
|
||||||
|
offset=offset,
|
||||||
)
|
)
|
||||||
if (
|
if (
|
||||||
not self._user_requested_row_id()
|
not self._user_requested_row_id()
|
||||||
@@ -3964,14 +4006,14 @@ class AsyncHybridQuery(AsyncStandardQuery, AsyncVectorQueryBase):
|
|||||||
... print(plan)
|
... print(plan)
|
||||||
>>> asyncio.run(doctest_example()) # doctest: +ELLIPSIS, +NORMALIZE_WHITESPACE
|
>>> asyncio.run(doctest_example()) # doctest: +ELLIPSIS, +NORMALIZE_WHITESPACE
|
||||||
RRFReranker(K=60)
|
RRFReranker(K=60)
|
||||||
ProjectionExec: expr=[vector@0 as vector, text@3 as text, _distance@2 as _distance]
|
ProjectionExec: expr=[vector@0 as vector, text@3 as text, _distance@2 as _distance, _rowid@1 as _rowid]
|
||||||
LanceRead: uri=..., projection=[text], source=stream(_rowid)
|
LanceRead: uri=..., projection=[text], source=stream(_rowid)
|
||||||
GlobalLimitExec: skip=0, fetch=10
|
GlobalLimitExec: skip=0, fetch=10
|
||||||
FilterExec: _distance@2 IS NOT NULL
|
FilterExec: _distance@2 IS NOT NULL
|
||||||
SortExec: TopK(fetch=10), expr=[_distance@2 ASC NULLS LAST, _rowid@1 ASC NULLS LAST], preserve_partitioning=[false]
|
SortExec: TopK(fetch=10), expr=[_distance@2 ASC NULLS LAST, _rowid@1 ASC NULLS LAST], preserve_partitioning=[false]
|
||||||
KNNVectorDistance: metric=l2
|
KNNVectorDistance: metric=l2
|
||||||
LanceRead: uri=..., projection=[vector], ...
|
LanceRead: uri=..., projection=[vector], ...
|
||||||
ProjectionExec: expr=[vector@2 as vector, text@3 as text, _score@1 as _score]
|
ProjectionExec: expr=[vector@2 as vector, text@3 as text, _score@1 as _score, _rowid@0 as _rowid]
|
||||||
LanceRead: uri=..., projection=[vector, text], source=stream(_rowid)
|
LanceRead: uri=..., projection=[vector, text], source=stream(_rowid)
|
||||||
GlobalLimitExec: skip=0, fetch=10
|
GlobalLimitExec: skip=0, fetch=10
|
||||||
MatchQuery: column=text, query=[hello]
|
MatchQuery: column=text, query=[hello]
|
||||||
@@ -3986,8 +4028,9 @@ class AsyncHybridQuery(AsyncStandardQuery, AsyncVectorQueryBase):
|
|||||||
plan : str
|
plan : str
|
||||||
""" # noqa: E501
|
""" # noqa: E501
|
||||||
|
|
||||||
vector_plan = await self._inner.to_vector_query().explain_plan(verbose)
|
fts_query, vec_query, _, _ = self._create_child_queries()
|
||||||
fts_plan = await self._inner.to_fts_query().explain_plan(verbose)
|
vector_plan = await vec_query.explain_plan(verbose)
|
||||||
|
fts_plan = await fts_query.explain_plan(verbose)
|
||||||
# Indent sub-plans under the reranker
|
# Indent sub-plans under the reranker
|
||||||
indented_vector = "\n".join(" " + line for line in vector_plan.splitlines())
|
indented_vector = "\n".join(" " + line for line in vector_plan.splitlines())
|
||||||
indented_fts = "\n".join(" " + line for line in fts_plan.splitlines())
|
indented_fts = "\n".join(" " + line for line in fts_plan.splitlines())
|
||||||
@@ -4014,14 +4057,12 @@ class AsyncHybridQuery(AsyncStandardQuery, AsyncVectorQueryBase):
|
|||||||
-------
|
-------
|
||||||
plan : str
|
plan : str
|
||||||
"""
|
"""
|
||||||
|
fts_query, vec_query, _, _ = self._create_child_queries()
|
||||||
|
|
||||||
results = ["Vector Search Query:"]
|
results = ["Vector Search Query:"]
|
||||||
results.append(
|
results.append(await vec_query.analyze_plan(distributed_metrics))
|
||||||
await self._inner.to_vector_query().analyze_plan(distributed_metrics)
|
|
||||||
)
|
|
||||||
results.append("FTS Search Query:")
|
results.append("FTS Search Query:")
|
||||||
results.append(
|
results.append(await fts_query.analyze_plan(distributed_metrics))
|
||||||
await self._inner.to_fts_query().analyze_plan(distributed_metrics)
|
|
||||||
)
|
|
||||||
|
|
||||||
return "\n".join(results)
|
return "\n".join(results)
|
||||||
|
|
||||||
|
|||||||
@@ -720,7 +720,7 @@ class RemoteTable(Table):
|
|||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
query: list/np.ndarray/str/PIL.Image.Image, default None
|
query: list/np.ndarray/str/PIL.Image.Image, default None
|
||||||
The targetted vector to search for.
|
The targeted vector to search for.
|
||||||
|
|
||||||
- *default None*.
|
- *default None*.
|
||||||
Acceptable types are: list, np.ndarray, PIL.Image.Image
|
Acceptable types are: list, np.ndarray, PIL.Image.Image
|
||||||
|
|||||||
@@ -175,7 +175,7 @@ class Reranker(ABC):
|
|||||||
if the results haven't been executed yet or the results in arrow format.
|
if the results haven't been executed yet or the results in arrow format.
|
||||||
query : str or None,
|
query : str or None,
|
||||||
The input query. Some rerankers might not need the query to rerank.
|
The input query. Some rerankers might not need the query to rerank.
|
||||||
In that case, it can be set to None explicitly. This is inteded to
|
In that case, it can be set to None explicitly. This is intended to
|
||||||
be handled by the reranker implementations.
|
be handled by the reranker implementations.
|
||||||
deduplicate : bool, optional
|
deduplicate : bool, optional
|
||||||
Whether to deduplicate the results based on the `_rowid` column,
|
Whether to deduplicate the results based on the `_rowid` column,
|
||||||
|
|||||||
@@ -1619,7 +1619,7 @@ class Table(ABC):
|
|||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
query: list/np.ndarray/str/PIL.Image.Image, default None
|
query: list/np.ndarray/str/PIL.Image.Image, default None
|
||||||
The targetted vector to search for.
|
The targeted vector to search for.
|
||||||
|
|
||||||
- *default None*.
|
- *default None*.
|
||||||
Acceptable types are: list, np.ndarray, PIL.Image.Image
|
Acceptable types are: list, np.ndarray, PIL.Image.Image
|
||||||
@@ -3841,7 +3841,7 @@ class LanceTable(Table):
|
|||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
query: list/np.ndarray/str/PIL.Image.Image, default None
|
query: list/np.ndarray/str/PIL.Image.Image, default None
|
||||||
The targetted vector to search for.
|
The targeted vector to search for.
|
||||||
|
|
||||||
- *default None*.
|
- *default None*.
|
||||||
Acceptable types are: list, np.ndarray, PIL.Image.Image
|
Acceptable types are: list, np.ndarray, PIL.Image.Image
|
||||||
@@ -5638,7 +5638,7 @@ class AsyncTable:
|
|||||||
if fill_value is None:
|
if fill_value is None:
|
||||||
fill_value = 0.0
|
fill_value = 0.0
|
||||||
|
|
||||||
# _santitize_data is an old code path, but we will use it until the
|
# _sanitize_data is an old code path, but we will use it until the
|
||||||
# new code path is ready.
|
# new code path is ready.
|
||||||
if mode == "overwrite":
|
if mode == "overwrite":
|
||||||
# For overwrite, apply the same preprocessing as create_table
|
# For overwrite, apply the same preprocessing as create_table
|
||||||
@@ -5814,7 +5814,7 @@ class AsyncTable:
|
|||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
query: list/np.ndarray/str/PIL.Image.Image, default None
|
query: list/np.ndarray/str/PIL.Image.Image, default None
|
||||||
The targetted vector to search for.
|
The targeted vector to search for.
|
||||||
|
|
||||||
- *default None*.
|
- *default None*.
|
||||||
Acceptable types are: list, np.ndarray, PIL.Image.Image
|
Acceptable types are: list, np.ndarray, PIL.Image.Image
|
||||||
|
|||||||
@@ -297,7 +297,10 @@ def test_blob_v2_projection_sources_use_typed_column_name():
|
|||||||
|
|
||||||
|
|
||||||
def _legacy_v1_table(name):
|
def _legacy_v1_table(name):
|
||||||
db = lancedb.connect("memory:///")
|
# Legacy v1 blob columns are only writable at file version <= 2.1.
|
||||||
|
db = lancedb.connect(
|
||||||
|
"memory:///", storage_options={"new_table_data_storage_version": "2.1"}
|
||||||
|
)
|
||||||
schema = pa.schema(
|
schema = pa.schema(
|
||||||
[
|
[
|
||||||
pa.field("id", pa.int64()),
|
pa.field("id", pa.int64()),
|
||||||
|
|||||||
@@ -327,8 +327,8 @@ def test_embedding_function_with_pandas(tmp_path):
|
|||||||
) -> List[np.array]:
|
) -> List[np.array]:
|
||||||
return [np.random.randn(self.ndims()).tolist() for _ in range(len(texts))]
|
return [np.random.randn(self.ndims()).tolist() for _ in range(len(texts))]
|
||||||
|
|
||||||
registery = get_registry()
|
registry = get_registry()
|
||||||
func = registery.get("mock-embedding").create()
|
func = registry.get("mock-embedding").create()
|
||||||
|
|
||||||
class TestSchema(LanceModel):
|
class TestSchema(LanceModel):
|
||||||
text: str = func.SourceField()
|
text: str = func.SourceField()
|
||||||
@@ -394,9 +394,9 @@ def test_multiple_embeddings_for_pandas(tmp_path):
|
|||||||
) -> List[np.array]:
|
) -> List[np.array]:
|
||||||
return [np.random.randn(self.ndims()).tolist() for _ in range(len(texts))]
|
return [np.random.randn(self.ndims()).tolist() for _ in range(len(texts))]
|
||||||
|
|
||||||
registery = get_registry()
|
registry = get_registry()
|
||||||
func1 = registery.get("mock-embedding").create()
|
func1 = registry.get("mock-embedding").create()
|
||||||
func2 = registery.get("mock-embedding2").create()
|
func2 = registry.get("mock-embedding2").create()
|
||||||
|
|
||||||
class TestSchema(LanceModel):
|
class TestSchema(LanceModel):
|
||||||
text: str = func1.SourceField()
|
text: str = func1.SourceField()
|
||||||
|
|||||||
@@ -1011,8 +1011,13 @@ def test_fts_ngram(mem_db: DBConnection):
|
|||||||
assert set(r["text"] for r in results) == {"lance database", "lance is cool"}
|
assert set(r["text"] for r in results) == {"lance database", "lance is cool"}
|
||||||
|
|
||||||
results = (
|
results = (
|
||||||
table.search("nce", query_type="fts").limit(10).to_list()
|
table.search(
|
||||||
) # spellchecker:disable-line
|
"nce", # spellchecker:disable-line
|
||||||
|
query_type="fts",
|
||||||
|
)
|
||||||
|
.limit(10)
|
||||||
|
.to_list()
|
||||||
|
)
|
||||||
assert len(results) == 2
|
assert len(results) == 2
|
||||||
assert set(r["text"] for r in results) == {"lance database", "lance is cool"}
|
assert set(r["text"] for r in results) == {"lance database", "lance is cool"}
|
||||||
|
|
||||||
@@ -1034,8 +1039,13 @@ def test_fts_ngram(mem_db: DBConnection):
|
|||||||
assert set(r["text"] for r in results) == {"lance database", "lance is cool"}
|
assert set(r["text"] for r in results) == {"lance database", "lance is cool"}
|
||||||
|
|
||||||
results = (
|
results = (
|
||||||
table.search("nce", query_type="fts").limit(10).to_list()
|
table.search(
|
||||||
) # spellchecker:disable-line
|
"nce", # spellchecker:disable-line
|
||||||
|
query_type="fts",
|
||||||
|
)
|
||||||
|
.limit(10)
|
||||||
|
.to_list()
|
||||||
|
)
|
||||||
assert len(results) == 0
|
assert len(results) == 0
|
||||||
|
|
||||||
results = table.search("la", query_type="fts").limit(10).to_list()
|
results = table.search("la", query_type="fts").limit(10).to_list()
|
||||||
|
|||||||
@@ -203,6 +203,93 @@ async def test_async_hybrid_query_default_limit(table: AsyncTable):
|
|||||||
assert texts.count("a") == 1
|
assert texts.count("a") == 1
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_async_hybrid_query_offset(table: AsyncTable):
|
||||||
|
# The offset window of a hybrid query must be a suffix of the same query
|
||||||
|
# run without an offset. Skipping the first rows of each sub-query instead
|
||||||
|
# of the first rows of the fused result silently changes which rows land in
|
||||||
|
# the window.
|
||||||
|
full = await (
|
||||||
|
table.query()
|
||||||
|
.nearest_to([0.0, 0.4])
|
||||||
|
.nearest_to_text("dog")
|
||||||
|
.limit(4)
|
||||||
|
.with_row_id()
|
||||||
|
.to_arrow()
|
||||||
|
)
|
||||||
|
assert len(full) == 4
|
||||||
|
|
||||||
|
second_page = await (
|
||||||
|
table.query()
|
||||||
|
.nearest_to([0.0, 0.4])
|
||||||
|
.nearest_to_text("dog")
|
||||||
|
.offset(2)
|
||||||
|
.limit(2)
|
||||||
|
.with_row_id()
|
||||||
|
.to_arrow()
|
||||||
|
)
|
||||||
|
assert second_page["_rowid"].to_pylist() == full["_rowid"].to_pylist()[2:]
|
||||||
|
|
||||||
|
first_page = await (
|
||||||
|
table.query()
|
||||||
|
.nearest_to([0.0, 0.4])
|
||||||
|
.nearest_to_text("dog")
|
||||||
|
.limit(2)
|
||||||
|
.with_row_id()
|
||||||
|
.to_arrow()
|
||||||
|
)
|
||||||
|
# Paging through the result must visit every row exactly once: no row
|
||||||
|
# repeated from the previous page and none dropped between the two.
|
||||||
|
paged = first_page["_rowid"].to_pylist() + second_page["_rowid"].to_pylist()
|
||||||
|
assert sorted(paged) == sorted(full["_rowid"].to_pylist())
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_async_hybrid_query_fts_first_default_limit(table: AsyncTable):
|
||||||
|
# nearest_to() and nearest_to_text() build their new sibling sub-query from
|
||||||
|
# scratch, and that is the sub-query the default limit ends up on. So the
|
||||||
|
# side that carries the limit depends on the order the hybrid query was
|
||||||
|
# built in, and looking at only one side loses the limit for half the ways
|
||||||
|
# a hybrid query can be written. Without a limit the combined results are
|
||||||
|
# not truncated at all and the whole union of both candidate lists is
|
||||||
|
# returned.
|
||||||
|
await table.add([{"text": "dog", "vector": [50.0 + i, 50.0]} for i in range(10)])
|
||||||
|
|
||||||
|
result = await (
|
||||||
|
table.query().nearest_to_text("dog").nearest_to([0.1, 0.1]).to_arrow()
|
||||||
|
)
|
||||||
|
assert len(result) == 10
|
||||||
|
|
||||||
|
offset_result = await (
|
||||||
|
table.query().nearest_to_text("dog").nearest_to([0.1, 0.1]).offset(2).to_arrow()
|
||||||
|
)
|
||||||
|
assert len(offset_result) == 10
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_async_hybrid_query_explain_plan_matches_execution(table: AsyncTable):
|
||||||
|
# Paging rewrites the sub-queries: each one fetches limit + offset rows with
|
||||||
|
# no offset of its own, and the window is sliced out after fusion. The plans
|
||||||
|
# have to be built from those rewritten sub-queries, otherwise explain_plan
|
||||||
|
# and analyze_plan describe a query that is never run.
|
||||||
|
query = (
|
||||||
|
table.query().nearest_to([0.0, 0.4]).nearest_to_text("dog").offset(2).limit(2)
|
||||||
|
)
|
||||||
|
await query.to_arrow()
|
||||||
|
|
||||||
|
plan = await query.explain_plan()
|
||||||
|
assert [
|
||||||
|
line.strip() for line in plan.splitlines() if "GlobalLimitExec" in line
|
||||||
|
] == [
|
||||||
|
"GlobalLimitExec: skip=0, fetch=4",
|
||||||
|
"GlobalLimitExec: skip=0, fetch=4",
|
||||||
|
]
|
||||||
|
|
||||||
|
analyzed = await query.analyze_plan()
|
||||||
|
assert analyzed.count("skip=0, fetch=4") == 2
|
||||||
|
assert "skip=2" not in analyzed
|
||||||
|
|
||||||
|
|
||||||
def test_hybrid_query_offset(sync_table: Table):
|
def test_hybrid_query_offset(sync_table: Table):
|
||||||
# The offset window of a hybrid query must be a suffix of the same query
|
# The offset window of a hybrid query must be a suffix of the same query
|
||||||
# run without an offset -- it must not be silently ignored.
|
# run without an offset -- it must not be silently ignored.
|
||||||
|
|||||||
@@ -193,7 +193,13 @@ class TestNamespaceConnection:
|
|||||||
),
|
),
|
||||||
)
|
)
|
||||||
|
|
||||||
table = db.create_table("blob_table", data, namespace_path=["test_ns"])
|
# Legacy v1 blob columns are only writable at file version <= 2.1.
|
||||||
|
table = db.create_table(
|
||||||
|
"blob_table",
|
||||||
|
data,
|
||||||
|
namespace_path=["test_ns"],
|
||||||
|
storage_options={"new_table_data_storage_version": "2.1"},
|
||||||
|
)
|
||||||
df = table.to_pandas(blob_mode="lazy").sort_values("id")
|
df = table.to_pandas(blob_mode="lazy").sort_values("id")
|
||||||
|
|
||||||
blob = df["blob"].iloc[0]
|
blob = df["blob"].iloc[0]
|
||||||
|
|||||||
@@ -40,6 +40,10 @@ from utils import exception_output
|
|||||||
from importlib.util import find_spec
|
from importlib.util import find_spec
|
||||||
|
|
||||||
|
|
||||||
|
# Legacy v1 blob columns are only writable at file version <= 2.1.
|
||||||
|
LEGACY_BLOB_STORAGE_OPTIONS = {"new_table_data_storage_version": "2.1"}
|
||||||
|
|
||||||
|
|
||||||
def _blob_query_data():
|
def _blob_query_data():
|
||||||
return pa.table(
|
return pa.table(
|
||||||
{
|
{
|
||||||
@@ -119,13 +123,17 @@ def _assert_blob_bytes_projection(df):
|
|||||||
|
|
||||||
def _blob_query_table(db, name, blob_schema):
|
def _blob_query_table(db, name, blob_schema):
|
||||||
if blob_schema == "v1":
|
if blob_schema == "v1":
|
||||||
return db.create_table(name, _blob_query_data())
|
return db.create_table(
|
||||||
|
name, _blob_query_data(), storage_options=LEGACY_BLOB_STORAGE_OPTIONS
|
||||||
|
)
|
||||||
return _create_blob_v2_query_table(db, name)
|
return _create_blob_v2_query_table(db, name)
|
||||||
|
|
||||||
|
|
||||||
async def _blob_query_table_async(db, name, blob_schema):
|
async def _blob_query_table_async(db, name, blob_schema):
|
||||||
if blob_schema == "v1":
|
if blob_schema == "v1":
|
||||||
return await db.create_table(name, _blob_query_data())
|
return await db.create_table(
|
||||||
|
name, _blob_query_data(), storage_options=LEGACY_BLOB_STORAGE_OPTIONS
|
||||||
|
)
|
||||||
return await _create_blob_v2_query_table_async(db, name)
|
return await _create_blob_v2_query_table_async(db, name)
|
||||||
|
|
||||||
|
|
||||||
@@ -275,7 +283,9 @@ async def test_query_to_pandas_kwargs(table, table_async):
|
|||||||
def test_plain_scan_query_to_pandas_blob_modes(tmp_db, blob_mode):
|
def test_plain_scan_query_to_pandas_blob_modes(tmp_db, blob_mode):
|
||||||
pytest.importorskip("lance")
|
pytest.importorskip("lance")
|
||||||
table = tmp_db.create_table(
|
table = tmp_db.create_table(
|
||||||
f"test_query_to_pandas_blob_{blob_mode}", _blob_query_data()
|
f"test_query_to_pandas_blob_{blob_mode}",
|
||||||
|
_blob_query_data(),
|
||||||
|
storage_options=LEGACY_BLOB_STORAGE_OPTIONS,
|
||||||
)
|
)
|
||||||
|
|
||||||
df = (
|
df = (
|
||||||
@@ -322,7 +332,9 @@ def test_plain_scan_query_to_pandas_blob_mode_does_not_collect_arrow(
|
|||||||
):
|
):
|
||||||
pytest.importorskip("lance")
|
pytest.importorskip("lance")
|
||||||
table = tmp_db.create_table(
|
table = tmp_db.create_table(
|
||||||
"test_query_to_pandas_blob_no_arrow_collect", _blob_query_data()
|
"test_query_to_pandas_blob_no_arrow_collect",
|
||||||
|
_blob_query_data(),
|
||||||
|
storage_options=LEGACY_BLOB_STORAGE_OPTIONS,
|
||||||
)
|
)
|
||||||
query = table.search().where("id = 1").select(["id", "blob"])
|
query = table.search().where("id = 1").select(["id", "blob"])
|
||||||
|
|
||||||
@@ -347,7 +359,9 @@ def test_plain_scan_query_to_pandas_blob_descriptions_flatten_uses_scanner(
|
|||||||
):
|
):
|
||||||
pytest.importorskip("lance")
|
pytest.importorskip("lance")
|
||||||
table = tmp_db.create_table(
|
table = tmp_db.create_table(
|
||||||
"test_query_to_pandas_blob_desc_flatten", _blob_query_data()
|
"test_query_to_pandas_blob_desc_flatten",
|
||||||
|
_blob_query_data(),
|
||||||
|
storage_options=LEGACY_BLOB_STORAGE_OPTIONS,
|
||||||
)
|
)
|
||||||
query = table.search().where("id = 1").select(["id", "blob"])
|
query = table.search().where("id = 1").select(["id", "blob"])
|
||||||
|
|
||||||
@@ -365,7 +379,11 @@ def test_plain_scan_query_to_pandas_blob_descriptions_flatten_uses_scanner(
|
|||||||
def test_plain_scan_query_to_pandas_scanner_state(tmp_db):
|
def test_plain_scan_query_to_pandas_scanner_state(tmp_db):
|
||||||
pytest.importorskip("lance")
|
pytest.importorskip("lance")
|
||||||
data = _blob_query_data()
|
data = _blob_query_data()
|
||||||
table = tmp_db.create_table("test_query_to_pandas_scanner_state", data.slice(0, 2))
|
table = tmp_db.create_table(
|
||||||
|
"test_query_to_pandas_scanner_state",
|
||||||
|
data.slice(0, 2),
|
||||||
|
storage_options=LEGACY_BLOB_STORAGE_OPTIONS,
|
||||||
|
)
|
||||||
table.add(data.slice(2, 2))
|
table.add(data.slice(2, 2))
|
||||||
|
|
||||||
fragments = table.to_lance().get_fragments()
|
fragments = table.to_lance().get_fragments()
|
||||||
@@ -400,7 +418,9 @@ def test_plain_scan_query_to_pandas_scanner_state(tmp_db):
|
|||||||
async def test_async_plain_scan_query_to_pandas_blob_projection(tmp_db_async):
|
async def test_async_plain_scan_query_to_pandas_blob_projection(tmp_db_async):
|
||||||
pytest.importorskip("lance")
|
pytest.importorskip("lance")
|
||||||
table = await tmp_db_async.create_table(
|
table = await tmp_db_async.create_table(
|
||||||
"test_async_query_to_pandas_blob_projection", _blob_query_data()
|
"test_async_query_to_pandas_blob_projection",
|
||||||
|
_blob_query_data(),
|
||||||
|
storage_options=LEGACY_BLOB_STORAGE_OPTIONS,
|
||||||
)
|
)
|
||||||
|
|
||||||
lazy_df = await (
|
lazy_df = await (
|
||||||
@@ -452,7 +472,9 @@ async def test_async_plain_scan_query_to_pandas_blob_mode_does_not_collect_arrow
|
|||||||
):
|
):
|
||||||
pytest.importorskip("lance")
|
pytest.importorskip("lance")
|
||||||
table = await tmp_db_async.create_table(
|
table = await tmp_db_async.create_table(
|
||||||
"test_async_query_to_pandas_blob_no_arrow_collect", _blob_query_data()
|
"test_async_query_to_pandas_blob_no_arrow_collect",
|
||||||
|
_blob_query_data(),
|
||||||
|
storage_options=LEGACY_BLOB_STORAGE_OPTIONS,
|
||||||
)
|
)
|
||||||
query = table.query().where("id = 1").select(["id", "blob"])
|
query = table.query().where("id = 1").select(["id", "blob"])
|
||||||
|
|
||||||
@@ -474,7 +496,11 @@ async def test_async_plain_scan_query_to_pandas_blob_mode_does_not_collect_arrow
|
|||||||
|
|
||||||
def test_vector_query_to_pandas_blob_mode_requires_native_path(tmp_db):
|
def test_vector_query_to_pandas_blob_mode_requires_native_path(tmp_db):
|
||||||
pytest.importorskip("lance")
|
pytest.importorskip("lance")
|
||||||
table = tmp_db.create_table("test_vector_query_blob_mode", _blob_query_data())
|
table = tmp_db.create_table(
|
||||||
|
"test_vector_query_blob_mode",
|
||||||
|
_blob_query_data(),
|
||||||
|
storage_options=LEGACY_BLOB_STORAGE_OPTIONS,
|
||||||
|
)
|
||||||
|
|
||||||
with pytest.raises(RuntimeError, match="Lance native pandas conversion"):
|
with pytest.raises(RuntimeError, match="Lance native pandas conversion"):
|
||||||
table.search([1.0, 0.0]).select(["blob", "vector"]).limit(1).to_pandas(
|
table.search([1.0, 0.0]).select(["blob", "vector"]).limit(1).to_pandas(
|
||||||
@@ -485,7 +511,9 @@ def test_vector_query_to_pandas_blob_mode_requires_native_path(tmp_db):
|
|||||||
def test_vector_query_to_pandas_blob_descriptions_requires_plain_scan(tmp_db):
|
def test_vector_query_to_pandas_blob_descriptions_requires_plain_scan(tmp_db):
|
||||||
pytest.importorskip("lance")
|
pytest.importorskip("lance")
|
||||||
table = tmp_db.create_table(
|
table = tmp_db.create_table(
|
||||||
"test_vector_query_blob_descriptions", _blob_query_data()
|
"test_vector_query_blob_descriptions",
|
||||||
|
_blob_query_data(),
|
||||||
|
storage_options=LEGACY_BLOB_STORAGE_OPTIONS,
|
||||||
)
|
)
|
||||||
|
|
||||||
with pytest.raises(RuntimeError, match="plain scan query"):
|
with pytest.raises(RuntimeError, match="plain scan query"):
|
||||||
|
|||||||
@@ -81,7 +81,7 @@ def get_test_table(tmp_path):
|
|||||||
"but his son was mortal",
|
"but his son was mortal",
|
||||||
"there hasn't been a good battlefield game since 2142",
|
"there hasn't been a good battlefield game since 2142",
|
||||||
"I wish they would make another one",
|
"I wish they would make another one",
|
||||||
"campains are not as good as they used to be",
|
"campaigns are not as good as they used to be",
|
||||||
"Multiplayer and open world games have destroyed the single player experience",
|
"Multiplayer and open world games have destroyed the single player experience",
|
||||||
"Maybe the future is console games",
|
"Maybe the future is console games",
|
||||||
"I don't know",
|
"I don't know",
|
||||||
|
|||||||
@@ -64,15 +64,23 @@ async def _blob_v2_table_async(db: AsyncConnection, name: str):
|
|||||||
return table
|
return table
|
||||||
|
|
||||||
|
|
||||||
|
# Legacy v1 blob columns are only writable at file version <= 2.1.
|
||||||
|
LEGACY_BLOB_STORAGE_OPTIONS = {"new_table_data_storage_version": "2.1"}
|
||||||
|
|
||||||
|
|
||||||
def _blob_table(db: DBConnection, name: str, blob_schema: str):
|
def _blob_table(db: DBConnection, name: str, blob_schema: str):
|
||||||
if blob_schema == "v1":
|
if blob_schema == "v1":
|
||||||
return db.create_table(name, data=_blob_test_data())
|
return db.create_table(
|
||||||
|
name, data=_blob_test_data(), storage_options=LEGACY_BLOB_STORAGE_OPTIONS
|
||||||
|
)
|
||||||
return _blob_v2_table(db, name)
|
return _blob_v2_table(db, name)
|
||||||
|
|
||||||
|
|
||||||
async def _blob_table_async(db: AsyncConnection, name: str, blob_schema: str):
|
async def _blob_table_async(db: AsyncConnection, name: str, blob_schema: str):
|
||||||
if blob_schema == "v1":
|
if blob_schema == "v1":
|
||||||
return await db.create_table(name, data=_blob_test_data())
|
return await db.create_table(
|
||||||
|
name, data=_blob_test_data(), storage_options=LEGACY_BLOB_STORAGE_OPTIONS
|
||||||
|
)
|
||||||
return await _blob_v2_table_async(db, name)
|
return await _blob_v2_table_async(db, name)
|
||||||
|
|
||||||
|
|
||||||
@@ -147,7 +155,11 @@ def test_table_to_pandas_invalid_blob_mode_non_blob_table(tmp_db: DBConnection):
|
|||||||
@pytest.mark.parametrize("blob_mode", ["lazy", "bytes", "descriptions"])
|
@pytest.mark.parametrize("blob_mode", ["lazy", "bytes", "descriptions"])
|
||||||
def test_table_to_pandas_blob_modes(tmp_db: DBConnection, blob_mode):
|
def test_table_to_pandas_blob_modes(tmp_db: DBConnection, blob_mode):
|
||||||
pytest.importorskip("lance")
|
pytest.importorskip("lance")
|
||||||
table = tmp_db.create_table(f"test_to_pandas_blob_{blob_mode}", _blob_test_data())
|
table = tmp_db.create_table(
|
||||||
|
f"test_to_pandas_blob_{blob_mode}",
|
||||||
|
_blob_test_data(),
|
||||||
|
storage_options=LEGACY_BLOB_STORAGE_OPTIONS,
|
||||||
|
)
|
||||||
|
|
||||||
df = table.to_pandas(blob_mode=blob_mode)
|
df = table.to_pandas(blob_mode=blob_mode)
|
||||||
|
|
||||||
@@ -3342,7 +3354,7 @@ def test_empty_query(mem_db: DBConnection):
|
|||||||
# None is the same as default
|
# None is the same as default
|
||||||
df = table.search().select(["id"]).limit(None).to_arrow()
|
df = table.search().select(["id"]).limit(None).to_arrow()
|
||||||
assert df.num_rows == 100
|
assert df.num_rows == 100
|
||||||
# invalid limist is the same as None, wihch is the same as default
|
# invalid limist is the same as None, which is the same as default
|
||||||
df = table.search().select(["id"]).limit(-1).to_arrow()
|
df = table.search().select(["id"]).limit(-1).to_arrow()
|
||||||
assert df.num_rows == 100
|
assert df.num_rows == 100
|
||||||
# valid limit should work
|
# valid limit should work
|
||||||
@@ -3959,7 +3971,7 @@ def test_stats(mem_db: DBConnection):
|
|||||||
print(f"{stats=}")
|
print(f"{stats=}")
|
||||||
assert stats == {
|
assert stats == {
|
||||||
# Full on-disk size of the data file, footer and metadata included.
|
# Full on-disk size of the data file, footer and metadata included.
|
||||||
"total_bytes": 633,
|
"total_bytes": 637,
|
||||||
"num_rows": 2,
|
"num_rows": 2,
|
||||||
"num_indices": 0,
|
"num_indices": 0,
|
||||||
"fragment_stats": {
|
"fragment_stats": {
|
||||||
|
|||||||
+1
-1
@@ -334,7 +334,7 @@ pub struct PyQueryRequest {
|
|||||||
pub column: Option<String>,
|
pub column: Option<String>,
|
||||||
pub query_vector: Option<PyQueryVectors>,
|
pub query_vector: Option<PyQueryVectors>,
|
||||||
pub minimum_nprobes: Option<usize>,
|
pub minimum_nprobes: Option<usize>,
|
||||||
// None means user did not set it and default shoud be used (currenty 20)
|
// None means user did not set it and default should be used (currently 20)
|
||||||
// Some(0) means user set it to None and there is no limit
|
// Some(0) means user set it to None and there is no limit
|
||||||
pub maximum_nprobes: Option<usize>,
|
pub maximum_nprobes: Option<usize>,
|
||||||
pub lower_bound: Option<f32>,
|
pub lower_bound: Option<f32>,
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "lancedb"
|
name = "lancedb"
|
||||||
version = "0.39.0-beta.4"
|
version = "0.39.0-beta.6"
|
||||||
edition.workspace = true
|
edition.workspace = true
|
||||||
description = "LanceDB: A serverless, low-latency vector database for AI applications"
|
description = "LanceDB: A serverless, low-latency vector database for AI applications"
|
||||||
license.workspace = true
|
license.workspace = true
|
||||||
|
|||||||
@@ -163,7 +163,7 @@ pub struct PolarsDataFrameRecordBatchReader {
|
|||||||
impl PolarsDataFrameRecordBatchReader {
|
impl PolarsDataFrameRecordBatchReader {
|
||||||
/// Creates a new `PolarsDataFrameRecordBatchReader` from a given Polars DataFrame.
|
/// Creates a new `PolarsDataFrameRecordBatchReader` from a given Polars DataFrame.
|
||||||
/// If the input dataframe does not have aligned chunks, this function undergoes
|
/// If the input dataframe does not have aligned chunks, this function undergoes
|
||||||
/// the costly operation of reallocating each series as a single contigous chunk.
|
/// the costly operation of reallocating each series as a single contiguous chunk.
|
||||||
pub fn new(mut df: DataFrame) -> Result<Self> {
|
pub fn new(mut df: DataFrame) -> Result<Self> {
|
||||||
df.align_chunks();
|
df.align_chunks();
|
||||||
let arrow_schema =
|
let arrow_schema =
|
||||||
|
|||||||
@@ -532,10 +532,11 @@ mod tests {
|
|||||||
fn storage_version_bumps_to_v2_2() {
|
fn storage_version_bumps_to_v2_2() {
|
||||||
let mut params = WriteParams::default();
|
let mut params = WriteParams::default();
|
||||||
ensure_blob_storage_version(&blob_schema(), &mut params);
|
ensure_blob_storage_version(&blob_schema(), &mut params);
|
||||||
assert_eq!(
|
let resolved = params
|
||||||
params.data_storage_version.unwrap().resolve(),
|
.data_storage_version
|
||||||
ConcreteFileVersion::V2_2
|
.unwrap_or(LanceFileVersion::Stable)
|
||||||
);
|
.resolve();
|
||||||
|
assert_eq!(resolved, ConcreteFileVersion::V2_2);
|
||||||
assert!(!params.enable_stable_row_ids);
|
assert!(!params.enable_stable_row_ids);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -547,10 +548,11 @@ mod tests {
|
|||||||
};
|
};
|
||||||
ensure_blob_storage_version(&blob_schema(), &mut params);
|
ensure_blob_storage_version(&blob_schema(), &mut params);
|
||||||
assert!(params.enable_stable_row_ids);
|
assert!(params.enable_stable_row_ids);
|
||||||
assert_eq!(
|
let resolved = params
|
||||||
params.data_storage_version.unwrap().resolve(),
|
.data_storage_version
|
||||||
ConcreteFileVersion::V2_2
|
.unwrap_or(LanceFileVersion::Stable)
|
||||||
);
|
.resolve();
|
||||||
|
assert_eq!(resolved, ConcreteFileVersion::V2_2);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|||||||
@@ -827,7 +827,7 @@ impl Connection {
|
|||||||
pub struct ConnectRequest {
|
pub struct ConnectRequest {
|
||||||
/// Database URI
|
/// Database URI
|
||||||
///
|
///
|
||||||
/// ### Accpeted URI formats
|
/// ### Accepted URI formats
|
||||||
///
|
///
|
||||||
/// - `/path/to/database` - local database on file system.
|
/// - `/path/to/database` - local database on file system.
|
||||||
/// - `s3://bucket/path/to/database` or `gs://bucket/path/to/database` - database on cloud object store
|
/// - `s3://bucket/path/to/database` or `gs://bucket/path/to/database` - database on cloud object store
|
||||||
|
|||||||
@@ -512,7 +512,7 @@ impl ListingDatabase {
|
|||||||
// iter thru the query params and extract the commit store param
|
// iter thru the query params and extract the commit store param
|
||||||
let mut engine = None;
|
let mut engine = None;
|
||||||
let mut mirrored_store = None;
|
let mut mirrored_store = None;
|
||||||
let mut filtered_querys = vec![];
|
let mut filtered_queries = vec![];
|
||||||
|
|
||||||
// WARNING: specifying engine is NOT a publicly supported feature in lancedb yet
|
// WARNING: specifying engine is NOT a publicly supported feature in lancedb yet
|
||||||
// THE API WILL CHANGE
|
// THE API WILL CHANGE
|
||||||
@@ -528,13 +528,13 @@ impl ListingDatabase {
|
|||||||
mirrored_store = Some(value.to_string());
|
mirrored_store = Some(value.to_string());
|
||||||
} else {
|
} else {
|
||||||
// to owned so we can modify the url
|
// to owned so we can modify the url
|
||||||
filtered_querys.push((key.to_string(), value.to_string()));
|
filtered_queries.push((key.to_string(), value.to_string()));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Filter out the commit store query param -- it's a lancedb param
|
// Filter out the commit store query param -- it's a lancedb param
|
||||||
url.query_pairs_mut().clear();
|
url.query_pairs_mut().clear();
|
||||||
url.query_pairs_mut().extend_pairs(filtered_querys);
|
url.query_pairs_mut().extend_pairs(filtered_queries);
|
||||||
// Take a copy of the query string so we can propagate it to lance.
|
// Take a copy of the query string so we can propagate it to lance.
|
||||||
// `query_pairs_mut()` leaves the URL with `Some("")` even when no
|
// `query_pairs_mut()` leaves the URL with `Some("")` even when no
|
||||||
// pairs survive (or none existed in the first place), so an empty
|
// pairs survive (or none existed in the first place), so an empty
|
||||||
@@ -896,11 +896,11 @@ impl Database for ListingDatabase {
|
|||||||
}
|
}
|
||||||
|
|
||||||
async fn read_consistency(&self) -> Result<ReadConsistency> {
|
async fn read_consistency(&self) -> Result<ReadConsistency> {
|
||||||
if let Some(read_consistency_inverval) = self.read_consistency_interval {
|
if let Some(interval) = self.read_consistency_interval {
|
||||||
if read_consistency_inverval.is_zero() {
|
if interval.is_zero() {
|
||||||
Ok(ReadConsistency::Strong)
|
Ok(ReadConsistency::Strong)
|
||||||
} else {
|
} else {
|
||||||
Ok(ReadConsistency::Eventual(read_consistency_inverval))
|
Ok(ReadConsistency::Eventual(interval))
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
Ok(ReadConsistency::Manual)
|
Ok(ReadConsistency::Manual)
|
||||||
@@ -3043,15 +3043,15 @@ mod tests {
|
|||||||
/// across platforms — see the `file://` test below).
|
/// across platforms — see the `file://` test below).
|
||||||
fn capture_query_like_connect(input_uri: &str) -> Option<String> {
|
fn capture_query_like_connect(input_uri: &str) -> Option<String> {
|
||||||
let mut url = url::Url::parse(input_uri).unwrap();
|
let mut url = url::Url::parse(input_uri).unwrap();
|
||||||
let mut filtered_querys = Vec::new();
|
let mut filtered_queries = Vec::new();
|
||||||
for (key, value) in url.query_pairs() {
|
for (key, value) in url.query_pairs() {
|
||||||
if key == ENGINE || key == MIRRORED_STORE {
|
if key == ENGINE || key == MIRRORED_STORE {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
filtered_querys.push((key.to_string(), value.to_string()));
|
filtered_queries.push((key.to_string(), value.to_string()));
|
||||||
}
|
}
|
||||||
url.query_pairs_mut().clear();
|
url.query_pairs_mut().clear();
|
||||||
url.query_pairs_mut().extend_pairs(filtered_querys);
|
url.query_pairs_mut().extend_pairs(filtered_queries);
|
||||||
url.query().filter(|q| !q.is_empty()).map(|s| s.to_string())
|
url.query().filter(|q| !q.is_empty()).map(|s| s.to_string())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -3,6 +3,7 @@
|
|||||||
|
|
||||||
//! Namespace-based database implementation that delegates table management to lance-namespace
|
//! Namespace-based database implementation that delegates table management to lance-namespace
|
||||||
|
|
||||||
|
use lance_datafusion::utils::StreamingWriteSource;
|
||||||
use std::collections::{HashMap, HashSet};
|
use std::collections::{HashMap, HashSet};
|
||||||
use std::sync::{Arc, Mutex};
|
use std::sync::{Arc, Mutex};
|
||||||
|
|
||||||
@@ -250,11 +251,11 @@ impl Database for LanceNamespaceDatabase {
|
|||||||
}
|
}
|
||||||
|
|
||||||
async fn read_consistency(&self) -> Result<ReadConsistency> {
|
async fn read_consistency(&self) -> Result<ReadConsistency> {
|
||||||
if let Some(read_consistency_inverval) = self.read_consistency_interval {
|
if let Some(interval) = self.read_consistency_interval {
|
||||||
if read_consistency_inverval.is_zero() {
|
if interval.is_zero() {
|
||||||
Ok(ReadConsistency::Strong)
|
Ok(ReadConsistency::Strong)
|
||||||
} else {
|
} else {
|
||||||
Ok(ReadConsistency::Eventual(read_consistency_inverval))
|
Ok(ReadConsistency::Eventual(interval))
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
Ok(ReadConsistency::Manual)
|
Ok(ReadConsistency::Manual)
|
||||||
@@ -304,6 +305,10 @@ impl Database for LanceNamespaceDatabase {
|
|||||||
}
|
}
|
||||||
|
|
||||||
async fn create_table(&self, request: DbCreateTableRequest) -> Result<Arc<dyn BaseTable>> {
|
async fn create_table(&self, request: DbCreateTableRequest) -> Result<Arc<dyn BaseTable>> {
|
||||||
|
// Refuse a bad declaration before the namespace records a table.
|
||||||
|
crate::table::computed_columns::ensure_declarations_are_planned(
|
||||||
|
&request.data.arrow_schema(),
|
||||||
|
)?;
|
||||||
let mut table_id = request.namespace_path.clone();
|
let mut table_id = request.namespace_path.clone();
|
||||||
table_id.push(request.name.clone());
|
table_id.push(request.name.clone());
|
||||||
let mut existing_table = None;
|
let mut existing_table = None;
|
||||||
|
|||||||
@@ -125,7 +125,7 @@ macro_rules! impl_pq_params_setter {
|
|||||||
/// This value controls how much the vector is compressed during the quantization step.
|
/// This value controls how much the vector is compressed during the quantization step.
|
||||||
/// The more sub vectors there are the less the vector is compressed. The default is
|
/// The more sub vectors there are the less the vector is compressed. The default is
|
||||||
/// the dimension of the vector divided by 16. If the dimension is not evenly divisible
|
/// the dimension of the vector divided by 16. If the dimension is not evenly divisible
|
||||||
/// by 16 we use the dimension divded by 8.
|
/// by 16 we use the dimension divided by 8.
|
||||||
///
|
///
|
||||||
/// The above two cases are highly preferred. Having 8 or 16 values per subvector allows
|
/// The above two cases are highly preferred. Having 8 or 16 values per subvector allows
|
||||||
/// us to use efficient SIMD instructions.
|
/// us to use efficient SIMD instructions.
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -24,8 +24,8 @@ use std::time::{SystemTime, UNIX_EPOCH};
|
|||||||
|
|
||||||
use arrow_array::cast::AsArray;
|
use arrow_array::cast::AsArray;
|
||||||
use arrow_array::types::UInt64Type;
|
use arrow_array::types::UInt64Type;
|
||||||
use arrow_array::{RecordBatch, UInt64Array};
|
use arrow_array::{RecordBatch, UInt64Array, new_null_array};
|
||||||
use arrow_schema::{Schema as ArrowSchema, SchemaRef};
|
use arrow_schema::{FieldRef, Schema as ArrowSchema, SchemaRef};
|
||||||
use datafusion::common::ScalarValue;
|
use datafusion::common::ScalarValue;
|
||||||
use datafusion::error::DataFusionError;
|
use datafusion::error::DataFusionError;
|
||||||
use datafusion::physical_plan::SendableRecordBatchStream;
|
use datafusion::physical_plan::SendableRecordBatchStream;
|
||||||
@@ -34,7 +34,7 @@ use datafusion::prelude::{col, lit};
|
|||||||
use futures::{StreamExt, TryStreamExt};
|
use futures::{StreamExt, TryStreamExt};
|
||||||
use lance::Dataset;
|
use lance::Dataset;
|
||||||
use lance::dataset::mem_wal::DatasetMemWalExt;
|
use lance::dataset::mem_wal::DatasetMemWalExt;
|
||||||
use lance::dataset::transaction::{Operation, Transaction};
|
use lance::dataset::transaction::{Operation, Transaction, UpdateMode};
|
||||||
use lance::dataset::write::delete::DeleteBuilder;
|
use lance::dataset::write::delete::DeleteBuilder;
|
||||||
use lance::dataset::write::merge_insert::inserted_rows::{
|
use lance::dataset::write::merge_insert::inserted_rows::{
|
||||||
KeyExistenceFilter, KeyExistenceFilterBuilder, KeyValue,
|
KeyExistenceFilter, KeyExistenceFilterBuilder, KeyValue,
|
||||||
@@ -51,6 +51,9 @@ use super::{
|
|||||||
definition_to_metadata,
|
definition_to_metadata,
|
||||||
};
|
};
|
||||||
use crate::database::OpenTableRequest;
|
use crate::database::OpenTableRequest;
|
||||||
|
use crate::table::computed_columns::{
|
||||||
|
computed_column_from_field, computed_columns, ensure_declarations_are_planned,
|
||||||
|
};
|
||||||
use crate::table::{NativeTable, NativeTableExt, Table};
|
use crate::table::{NativeTable, NativeTableExt, Table};
|
||||||
use crate::{Error, Result};
|
use crate::{Error, Result};
|
||||||
|
|
||||||
@@ -167,30 +170,52 @@ pub(crate) async fn execute_refresh(
|
|||||||
.map(|p| (p.output.clone(), p.expression.clone()))
|
.map(|p| (p.output.clone(), p.expression.clone()))
|
||||||
.collect();
|
.collect();
|
||||||
validate_inputs(&source_ds, definition)?;
|
validate_inputs(&source_ds, definition)?;
|
||||||
let (replanned, mut planned_fields, _renames) = super::plan(
|
let (replanned, planned_fields, _renames) = super::plan(
|
||||||
source_schema,
|
source_schema,
|
||||||
&definition.source_table,
|
&definition.source_table,
|
||||||
&definition.source_namespace,
|
&definition.source_namespace,
|
||||||
&projections,
|
Some(&projections),
|
||||||
definition.filter.as_deref(),
|
definition.filter.as_deref(),
|
||||||
definition.limit,
|
definition.limit,
|
||||||
)?;
|
)?;
|
||||||
|
let mut planned_fields = planned_fields;
|
||||||
planned_fields.push(arrow_schema::Field::new(
|
planned_fields.push(arrow_schema::Field::new(
|
||||||
SOURCE_ROW_ID_COLUMN,
|
SOURCE_ROW_ID_COLUMN,
|
||||||
arrow_schema::DataType::UInt64,
|
arrow_schema::DataType::UInt64,
|
||||||
false,
|
false,
|
||||||
));
|
));
|
||||||
|
// A computed column is not planned from the source: refresh writes it
|
||||||
|
// NULL and its declaration's owner fills it. Its declaration must still
|
||||||
|
// be complete, and it must be able to hold NULL.
|
||||||
let physical = ArrowSchema::from(view_ds.schema());
|
let physical = ArrowSchema::from(view_ds.schema());
|
||||||
let planned_shape: Vec<_> = planned_fields
|
let mut computed = computed_columns(&physical).into_iter().map(|c| c.name);
|
||||||
.iter()
|
if let Some(name) = computed.by_ref().find(|name| {
|
||||||
.map(|f| (f.name().clone(), f.data_type().clone(), f.is_nullable()))
|
physical
|
||||||
.collect();
|
.field_with_name(name)
|
||||||
let physical_shape: Vec<_> = physical
|
.is_ok_and(|f| !f.is_nullable())
|
||||||
|
}) {
|
||||||
|
return Err(Error::Schema {
|
||||||
|
message: format!(
|
||||||
|
"computed column '{name}' of view '{}' cannot hold NULL; recreate the view",
|
||||||
|
view.name()
|
||||||
|
),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
ensure_declarations_are_planned(&physical)?;
|
||||||
|
let physical_planned: Vec<&FieldRef> = physical
|
||||||
.fields()
|
.fields()
|
||||||
.iter()
|
.iter()
|
||||||
.map(|f| (f.name().clone(), f.data_type().clone(), f.is_nullable()))
|
.filter(|f| computed_column_from_field(f).is_none())
|
||||||
.collect();
|
.collect();
|
||||||
if planned_shape != physical_shape {
|
// A projected column that became nullable at the source still fits the
|
||||||
|
// view's nullable field; the reverse would not.
|
||||||
|
let matches = planned_fields.len() == physical_planned.len()
|
||||||
|
&& planned_fields.iter().zip(&physical_planned).all(|(e, p)| {
|
||||||
|
e.name() == p.name()
|
||||||
|
&& e.data_type() == p.data_type()
|
||||||
|
&& (p.is_nullable() || !e.is_nullable())
|
||||||
|
});
|
||||||
|
if !matches {
|
||||||
return Err(Error::Schema {
|
return Err(Error::Schema {
|
||||||
message: format!(
|
message: format!(
|
||||||
"the stored definition of view '{}' does not produce this \
|
"the stored definition of view '{}' does not produce this \
|
||||||
@@ -229,11 +254,18 @@ pub(crate) async fn execute_refresh(
|
|||||||
.get(SOURCE_VERSION_TS_META_KEY)
|
.get(SOURCE_VERSION_TS_META_KEY)
|
||||||
.and_then(|raw| raw.parse().ok());
|
.and_then(|raw| raw.parse().ok());
|
||||||
// The watermark speaks only for the view state its refresh left behind;
|
// The watermark speaks only for the view state its refresh left behind;
|
||||||
// any other commit on the view since then is drift.
|
// any other commit on the view since then is drift, except a fill of its
|
||||||
let view_intact = metadata
|
// computed columns, which rewrites nothing refresh certifies.
|
||||||
|
let recorded_view_version = metadata
|
||||||
.get(VIEW_VERSION_META_KEY)
|
.get(VIEW_VERSION_META_KEY)
|
||||||
.and_then(|raw| raw.parse::<u64>().ok())
|
.and_then(|raw| raw.parse::<u64>().ok());
|
||||||
== Some(view_ds.version().version);
|
let view_intact = match recorded_view_version {
|
||||||
|
Some(recorded) if recorded == view_ds.version().version => true,
|
||||||
|
Some(recorded) if recorded < view_ds.version().version => {
|
||||||
|
only_computed_rewrites_since(&view_ds, recorded).await?
|
||||||
|
}
|
||||||
|
_ => false,
|
||||||
|
};
|
||||||
|
|
||||||
if !full && watermark == Some(source_version) && view_intact && recorded_ts == Some(source_ts) {
|
if !full && watermark == Some(source_version) && view_intact && recorded_ts == Some(source_ts) {
|
||||||
return Ok(RefreshMaterializedViewResult {
|
return Ok(RefreshMaterializedViewResult {
|
||||||
@@ -1090,6 +1122,69 @@ struct RowScope {
|
|||||||
limit: Option<u64>,
|
limit: Option<u64>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Whether every commit on the view after `recorded` is a fill of its
|
||||||
|
/// computed columns: a column rewrite or data replacement touching only
|
||||||
|
/// those fields and neither adding nor removing rows. A version whose
|
||||||
|
/// transaction cannot be read is not proven, so it counts as drift.
|
||||||
|
async fn only_computed_rewrites_since(view_ds: &Dataset, recorded: u64) -> Result<bool> {
|
||||||
|
// A fill may write any field under a computed column, so the whole
|
||||||
|
// subtree counts, not only the root.
|
||||||
|
let physical = ArrowSchema::from(view_ds.schema());
|
||||||
|
fn subtree(field: &lance_core::datatypes::Field, ids: &mut Vec<u32>) {
|
||||||
|
ids.push(field.id as u32);
|
||||||
|
for child in &field.children {
|
||||||
|
subtree(child, ids);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let mut computed_fields = Vec::new();
|
||||||
|
for column in computed_columns(&physical) {
|
||||||
|
if let Some(field) = view_ds.schema().field(&column.name) {
|
||||||
|
subtree(field, &mut computed_fields);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if computed_fields.is_empty() {
|
||||||
|
return Ok(false);
|
||||||
|
}
|
||||||
|
for version in recorded + 1..=view_ds.version().version {
|
||||||
|
let Some(transaction) = view_ds.read_transaction_by_version(version).await? else {
|
||||||
|
return Ok(false);
|
||||||
|
};
|
||||||
|
let fill = match &transaction.operation {
|
||||||
|
Operation::Update {
|
||||||
|
removed_fragment_ids,
|
||||||
|
new_fragments,
|
||||||
|
fields_modified,
|
||||||
|
update_mode: Some(UpdateMode::RewriteColumns),
|
||||||
|
..
|
||||||
|
} => {
|
||||||
|
removed_fragment_ids.is_empty()
|
||||||
|
&& new_fragments.is_empty()
|
||||||
|
&& !fields_modified.is_empty()
|
||||||
|
&& fields_modified
|
||||||
|
.iter()
|
||||||
|
.all(|field| computed_fields.contains(field))
|
||||||
|
}
|
||||||
|
// What `refresh_column` commits for a SQL declaration.
|
||||||
|
Operation::DataReplacement { replacements } => {
|
||||||
|
!replacements.is_empty()
|
||||||
|
&& replacements.iter().all(|group| {
|
||||||
|
!group.1.fields.is_empty()
|
||||||
|
&& group
|
||||||
|
.1
|
||||||
|
.fields
|
||||||
|
.iter()
|
||||||
|
.all(|field| computed_fields.contains(&(*field as u32)))
|
||||||
|
})
|
||||||
|
}
|
||||||
|
_ => false,
|
||||||
|
};
|
||||||
|
if !fill {
|
||||||
|
return Ok(false);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(true)
|
||||||
|
}
|
||||||
|
|
||||||
async fn compute_stream(
|
async fn compute_stream(
|
||||||
source: &Dataset,
|
source: &Dataset,
|
||||||
definition: &MaterializedViewDefinition,
|
definition: &MaterializedViewDefinition,
|
||||||
@@ -1158,6 +1253,10 @@ async fn compute_stream(
|
|||||||
let batch = batch.map_err(|e| DataFusionError::External(Box::new(e)))?;
|
let batch = batch.map_err(|e| DataFusionError::External(Box::new(e)))?;
|
||||||
let mut columns = Vec::with_capacity(out_schema.fields().len());
|
let mut columns = Vec::with_capacity(out_schema.fields().len());
|
||||||
for field in out_schema.fields() {
|
for field in out_schema.fields() {
|
||||||
|
if computed_column_from_field(field).is_some() {
|
||||||
|
columns.push(new_null_array(field.data_type(), batch.num_rows()));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
let name = if field.name() == SOURCE_ROW_ID_COLUMN {
|
let name = if field.name() == SOURCE_ROW_ID_COLUMN {
|
||||||
ROW_ID
|
ROW_ID
|
||||||
} else {
|
} else {
|
||||||
@@ -2768,7 +2867,7 @@ mod tests {
|
|||||||
let (conn, source) = db_with_source(vec![1]).await;
|
let (conn, source) = db_with_source(vec![1]).await;
|
||||||
let prepared = crate::materialized_view::prepare_declaration(
|
let prepared = crate::materialized_view::prepare_declaration(
|
||||||
&source,
|
&source,
|
||||||
&[("x".into(), "x".into()), ("twice".into(), "x * 2".into())],
|
Some(&[("x".into(), "x".into()), ("twice".into(), "x * 2".into())]),
|
||||||
None,
|
None,
|
||||||
None,
|
None,
|
||||||
)
|
)
|
||||||
@@ -3132,4 +3231,424 @@ mod tests {
|
|||||||
let err = view.refresh().execute().await.unwrap_err();
|
let err = view.refresh().execute().await.unwrap_err();
|
||||||
assert!(err.to_string().contains("source table 'src'"), "{err}");
|
assert!(err.to_string().contains("source table 'src'"), "{err}");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A view with a computed column, declared over `people` and refreshed.
|
||||||
|
async fn refreshed_computed_view(conn: &Connection) -> MaterializedView {
|
||||||
|
use crate::materialized_view::tests::{computed_field, people, test_binding};
|
||||||
|
let source = people(conn).await;
|
||||||
|
let view = crate::materialized_view::prepare_declaration(
|
||||||
|
&source,
|
||||||
|
Some(&[
|
||||||
|
("id".to_string(), "id".to_string()),
|
||||||
|
("name".to_string(), "name".to_string()),
|
||||||
|
]),
|
||||||
|
None,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.with_computed_columns(
|
||||||
|
vec![(2, computed_field("emb", "fb_1", "name"))],
|
||||||
|
&[test_binding("fb_1", "name", "emb")],
|
||||||
|
)
|
||||||
|
.unwrap()
|
||||||
|
.create("v")
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
let result = view.refresh().execute().await.unwrap();
|
||||||
|
assert_eq!(result.mode, RefreshMode::Rebuild);
|
||||||
|
view
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn unfilled(view: &MaterializedView) -> usize {
|
||||||
|
view.table()
|
||||||
|
.count_rows(Some("emb IS NULL".to_string()))
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn append_people(conn: &Connection, ids: Vec<i32>, names: Vec<&str>) {
|
||||||
|
let batch = record_batch!(("id", Int32, ids), ("name", Utf8, names)).unwrap();
|
||||||
|
conn.open_table("people")
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.add(batch)
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Commit the fill job's shape on the view: a column rewrite of
|
||||||
|
/// `fields`, touching no rows. The data is left as it is; what matters
|
||||||
|
/// here is how the next refresh classifies the commit.
|
||||||
|
async fn commit_column_rewrite(view: &MaterializedView, fields: &[&str]) {
|
||||||
|
let native = view.table().as_native().unwrap();
|
||||||
|
native.dataset.reload().await.unwrap();
|
||||||
|
let dataset = native.dataset.get().await.unwrap().as_ref().clone();
|
||||||
|
let fields_modified = fields
|
||||||
|
.iter()
|
||||||
|
.map(|name| dataset.schema().field(name).unwrap().id as u32)
|
||||||
|
.collect();
|
||||||
|
let updated_fragments = dataset
|
||||||
|
.get_fragments()
|
||||||
|
.iter()
|
||||||
|
.map(|fragment| fragment.metadata().clone())
|
||||||
|
.collect();
|
||||||
|
let operation = Operation::Update {
|
||||||
|
removed_fragment_ids: Vec::new(),
|
||||||
|
updated_fragments,
|
||||||
|
new_fragments: Vec::new(),
|
||||||
|
fields_modified,
|
||||||
|
compacted_sstables: Vec::new(),
|
||||||
|
fields_for_preserving_frag_bitmap: Vec::new(),
|
||||||
|
update_mode: Some(UpdateMode::RewriteColumns),
|
||||||
|
inserted_rows_filter: None,
|
||||||
|
updated_fragment_offsets: None,
|
||||||
|
};
|
||||||
|
let read_version = dataset.version().version;
|
||||||
|
CommitBuilder::new(WriteDestination::Dataset(Arc::new(dataset)))
|
||||||
|
.execute(Transaction::new(read_version, operation, None))
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Refresh never computes a computed column: every row it writes, on a
|
||||||
|
/// rebuild, an append and a rewrite, carries NULL there, and the
|
||||||
|
/// declaration survives all three.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_computed_columns_are_written_null_and_kept() {
|
||||||
|
let conn = connect("memory://").execute().await.unwrap();
|
||||||
|
let view = refreshed_computed_view(&conn).await;
|
||||||
|
assert_eq!(unfilled(&view).await, 3);
|
||||||
|
|
||||||
|
append_people(&conn, vec![4, 5], vec!["d", "e"]).await;
|
||||||
|
let result = view.refresh().execute().await.unwrap();
|
||||||
|
assert_eq!(result.mode, RefreshMode::Incremental);
|
||||||
|
assert_eq!(unfilled(&view).await, 5);
|
||||||
|
|
||||||
|
conn.open_table("people")
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.update()
|
||||||
|
.column("name", "'z'")
|
||||||
|
.only_if("id = 1")
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
view.refresh().execute().await.unwrap();
|
||||||
|
assert_eq!(unfilled(&view).await, 5);
|
||||||
|
assert_eq!(read(view.table(), "id").await, vec![1, 2, 3, 4, 5]);
|
||||||
|
|
||||||
|
let schema = view.table().schema().await.unwrap();
|
||||||
|
assert!(
|
||||||
|
crate::table::computed_columns::function_bindings(&schema)
|
||||||
|
.unwrap()
|
||||||
|
.iter()
|
||||||
|
.any(|b| b.binding_id() == "fb_1"),
|
||||||
|
"the binding envelope was lost"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
computed_column_from_field(schema.field_with_name("emb").unwrap()).is_some(),
|
||||||
|
"the declaration was lost"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
view.refresh().execute().await.unwrap().mode,
|
||||||
|
RefreshMode::NoOp
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The fill job's commit rewrites only computed columns. It is the one
|
||||||
|
/// commit on a view that is not drift: the next refresh carries on from
|
||||||
|
/// its watermark instead of rebuilding, which would null what the fill
|
||||||
|
/// just wrote.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_a_computed_column_fill_is_not_drift() {
|
||||||
|
let conn = connect("memory://").execute().await.unwrap();
|
||||||
|
let view = refreshed_computed_view(&conn).await;
|
||||||
|
|
||||||
|
commit_column_rewrite(&view, &["emb"]).await;
|
||||||
|
assert_eq!(
|
||||||
|
view.refresh().execute().await.unwrap().mode,
|
||||||
|
RefreshMode::NoOp
|
||||||
|
);
|
||||||
|
|
||||||
|
commit_column_rewrite(&view, &["emb"]).await;
|
||||||
|
append_people(&conn, vec![4], vec!["d"]).await;
|
||||||
|
let result = view.refresh().execute().await.unwrap();
|
||||||
|
assert_eq!(result.mode, RefreshMode::Incremental);
|
||||||
|
assert_eq!(result.rows_written, 1);
|
||||||
|
assert_eq!(read(view.table(), "id").await, vec![1, 2, 3, 4]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A column rewrite that reaches a projected column is drift like any
|
||||||
|
/// other write: refresh certifies those columns and must recompute them.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_a_rewrite_of_a_projected_column_is_drift() {
|
||||||
|
let conn = connect("memory://").execute().await.unwrap();
|
||||||
|
let view = refreshed_computed_view(&conn).await;
|
||||||
|
|
||||||
|
commit_column_rewrite(&view, &["emb", "name"]).await;
|
||||||
|
assert_eq!(
|
||||||
|
view.refresh().execute().await.unwrap().mode,
|
||||||
|
RefreshMode::Rebuild
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The declaration contract is checked before any refresh mutation: a
|
||||||
|
/// missing binding envelope and a column that lost its declaration both
|
||||||
|
/// fail closed.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_a_broken_declaration_is_refused_before_refresh() {
|
||||||
|
let conn = connect("memory://").execute().await.unwrap();
|
||||||
|
let view = refreshed_computed_view(&conn).await;
|
||||||
|
let native = view.table().as_native().unwrap();
|
||||||
|
let mut dataset = native.dataset.get().await.unwrap().as_ref().clone();
|
||||||
|
dataset
|
||||||
|
.update_schema_metadata(vec![(
|
||||||
|
crate::table::computed_columns::FUNCTION_BINDINGS_META_KEY.to_string(),
|
||||||
|
None,
|
||||||
|
)])
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
let err = view.refresh().execute().await.unwrap_err().to_string();
|
||||||
|
assert!(err.contains("references missing binding 'fb_1'"), "{err}");
|
||||||
|
|
||||||
|
let conn = connect("memory://").execute().await.unwrap();
|
||||||
|
let view = refreshed_computed_view(&conn).await;
|
||||||
|
let native = view.table().as_native().unwrap();
|
||||||
|
let mut dataset = native.dataset.get().await.unwrap().as_ref().clone();
|
||||||
|
dataset
|
||||||
|
.replace_field_metadata(vec![(
|
||||||
|
dataset.schema().field("emb").unwrap().id as u32,
|
||||||
|
HashMap::new(),
|
||||||
|
)])
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
let err = view.refresh().execute().await.unwrap_err().to_string();
|
||||||
|
assert!(err.contains("does not match binding 'fb_1'"), "{err}");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An input the view does not project is materialized on every refresh
|
||||||
|
/// path, before the provenance column, with the source's values.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_internal_inputs_are_materialized_and_refreshed() {
|
||||||
|
use crate::materialized_view::tests::{computed_field, strict_people, test_binding};
|
||||||
|
let conn = connect("memory://").execute().await.unwrap();
|
||||||
|
let source = strict_people(&conn).await;
|
||||||
|
let mut prepared = crate::materialized_view::prepare_declaration(
|
||||||
|
&source,
|
||||||
|
Some(&[("id".to_string(), "id".to_string())]),
|
||||||
|
None,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
let input = prepared.input_column("name").unwrap();
|
||||||
|
let view = prepared
|
||||||
|
.with_computed_columns(
|
||||||
|
vec![(1, computed_field("emb", "fb_1", &input))],
|
||||||
|
&[test_binding("fb_1", &input, "emb")],
|
||||||
|
)
|
||||||
|
.unwrap()
|
||||||
|
.create("v")
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
let names: Vec<String> = view
|
||||||
|
.table()
|
||||||
|
.schema()
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.fields()
|
||||||
|
.iter()
|
||||||
|
.map(|f| f.name().clone())
|
||||||
|
.collect();
|
||||||
|
assert_eq!(names, ["id", "emb", "__input_name", SOURCE_ROW_ID_COLUMN]);
|
||||||
|
|
||||||
|
let unfilled_inputs = || async {
|
||||||
|
view.table()
|
||||||
|
.count_rows(Some("__input_name IS NULL".to_string()))
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
};
|
||||||
|
assert_eq!(
|
||||||
|
view.refresh().execute().await.unwrap().mode,
|
||||||
|
RefreshMode::Rebuild
|
||||||
|
);
|
||||||
|
assert_eq!(view.table().count_rows(None).await.unwrap(), 3);
|
||||||
|
assert_eq!(unfilled_inputs().await, 0);
|
||||||
|
|
||||||
|
let more = arrow_array::RecordBatch::try_new(
|
||||||
|
source.schema().await.unwrap(),
|
||||||
|
vec![
|
||||||
|
Arc::new(Int32Array::from(vec![4])),
|
||||||
|
Arc::new(arrow_array::StringArray::from(vec!["d"])),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
source.add(more).execute().await.unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
view.refresh().execute().await.unwrap().mode,
|
||||||
|
RefreshMode::Incremental
|
||||||
|
);
|
||||||
|
assert_eq!(unfilled_inputs().await, 0);
|
||||||
|
assert_eq!(
|
||||||
|
view.table()
|
||||||
|
.count_rows(Some("__input_name = 'd'".to_string()))
|
||||||
|
.await
|
||||||
|
.unwrap(),
|
||||||
|
1
|
||||||
|
);
|
||||||
|
|
||||||
|
source
|
||||||
|
.update()
|
||||||
|
.column("name", "'z'")
|
||||||
|
.only_if("id = 1")
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
view.refresh().execute().await.unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
view.table()
|
||||||
|
.count_rows(Some("__input_name = 'z'".to_string()))
|
||||||
|
.await
|
||||||
|
.unwrap(),
|
||||||
|
1
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
unfilled(&view).await,
|
||||||
|
4,
|
||||||
|
"rewritten and new rows are unfilled"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A SQL declaration is filled by `refresh_column` on the view, which
|
||||||
|
/// commits a data replacement; the next refresh continues from its
|
||||||
|
/// watermark and keeps what the fill wrote, and only rows the view added
|
||||||
|
/// since come back unfilled.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_a_sql_fill_is_not_drift() {
|
||||||
|
use crate::materialized_view::tests::{people, sql_field};
|
||||||
|
let conn = connect("memory://").execute().await.unwrap();
|
||||||
|
let source = people(&conn).await;
|
||||||
|
let view = crate::materialized_view::prepare_declaration(
|
||||||
|
&source,
|
||||||
|
Some(&[("id".to_string(), "id".to_string())]),
|
||||||
|
None,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.with_computed_columns(
|
||||||
|
vec![(
|
||||||
|
1,
|
||||||
|
sql_field("next", arrow_schema::DataType::Int32, "id + 1", r#"["id"]"#),
|
||||||
|
)],
|
||||||
|
&[],
|
||||||
|
)
|
||||||
|
.unwrap()
|
||||||
|
.create("v")
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
let filled = || async {
|
||||||
|
view.table()
|
||||||
|
.count_rows(Some("next = id + 1".to_string()))
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
};
|
||||||
|
assert_eq!(
|
||||||
|
view.refresh().execute().await.unwrap().mode,
|
||||||
|
RefreshMode::Rebuild
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
view.table()
|
||||||
|
.refresh_column("next")
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.rows_filled,
|
||||||
|
3
|
||||||
|
);
|
||||||
|
assert_eq!(filled().await, 3);
|
||||||
|
assert_eq!(
|
||||||
|
view.refresh().execute().await.unwrap().mode,
|
||||||
|
RefreshMode::NoOp
|
||||||
|
);
|
||||||
|
assert_eq!(filled().await, 3);
|
||||||
|
|
||||||
|
append_people(&conn, vec![4], vec!["d"]).await;
|
||||||
|
assert_eq!(
|
||||||
|
view.refresh().execute().await.unwrap().mode,
|
||||||
|
RefreshMode::Incremental
|
||||||
|
);
|
||||||
|
assert_eq!(filled().await, 3);
|
||||||
|
assert_eq!(
|
||||||
|
view.table()
|
||||||
|
.refresh_column("next")
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.rows_filled,
|
||||||
|
1
|
||||||
|
);
|
||||||
|
assert_eq!(filled().await, 4);
|
||||||
|
assert_eq!(
|
||||||
|
view.refresh().execute().await.unwrap().mode,
|
||||||
|
RefreshMode::NoOp
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A fill of a nested computed column writes its child fields; that is
|
||||||
|
/// still a fill, not drift.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_a_nested_sql_fill_is_not_drift() {
|
||||||
|
use crate::materialized_view::tests::{people, sql_field};
|
||||||
|
let conn = connect("memory://").execute().await.unwrap();
|
||||||
|
let source = people(&conn).await;
|
||||||
|
let payload = sql_field(
|
||||||
|
"payload",
|
||||||
|
arrow_schema::DataType::Struct(
|
||||||
|
vec![arrow_schema::Field::new(
|
||||||
|
"value",
|
||||||
|
arrow_schema::DataType::Utf8,
|
||||||
|
true,
|
||||||
|
)]
|
||||||
|
.into(),
|
||||||
|
),
|
||||||
|
"named_struct('value', name)",
|
||||||
|
r#"["name"]"#,
|
||||||
|
);
|
||||||
|
let view = crate::materialized_view::prepare_declaration(
|
||||||
|
&source,
|
||||||
|
Some(&[("name".to_string(), "name".to_string())]),
|
||||||
|
None,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.with_computed_columns(vec![(1, payload)], &[])
|
||||||
|
.unwrap()
|
||||||
|
.create("v")
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
view.refresh().execute().await.unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
view.table()
|
||||||
|
.refresh_column("payload")
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.rows_filled,
|
||||||
|
3
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
view.refresh().execute().await.unwrap().mode,
|
||||||
|
RefreshMode::NoOp
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
view.table()
|
||||||
|
.count_rows(Some("payload.value = name".to_string()))
|
||||||
|
.await
|
||||||
|
.unwrap(),
|
||||||
|
3
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1299,7 +1299,7 @@ impl VectorQuery {
|
|||||||
/// This can be useful when there is a narrow filter to allow these queries to
|
/// This can be useful when there is a narrow filter to allow these queries to
|
||||||
/// spend more time searching and avoid potential false negatives.
|
/// spend more time searching and avoid potential false negatives.
|
||||||
///
|
///
|
||||||
/// Set to None to search all partitions, if needed, to satsify the limit
|
/// Set to None to search all partitions, if needed, to satisfy the limit
|
||||||
pub fn maximum_nprobes(mut self, maximum_nprobes: Option<usize>) -> Result<Self> {
|
pub fn maximum_nprobes(mut self, maximum_nprobes: Option<usize>) -> Result<Self> {
|
||||||
if let Some(maximum_nprobes) = maximum_nprobes {
|
if let Some(maximum_nprobes) = maximum_nprobes {
|
||||||
if maximum_nprobes == 0 {
|
if maximum_nprobes == 0 {
|
||||||
|
|||||||
@@ -240,7 +240,7 @@ enum BadVectorHandling {
|
|||||||
/// An error is returned
|
/// An error is returned
|
||||||
#[default]
|
#[default]
|
||||||
Error,
|
Error,
|
||||||
/// The offending row is droppped
|
/// The offending row is dropped
|
||||||
Drop,
|
Drop,
|
||||||
/// The invalid/missing items are replaced by fill_value
|
/// The invalid/missing items are replaced by fill_value
|
||||||
Fill(f32),
|
Fill(f32),
|
||||||
@@ -1326,7 +1326,7 @@ impl Table {
|
|||||||
/// Note: if your condition is something like "some_id_column == 7" and
|
/// Note: if your condition is something like "some_id_column == 7" and
|
||||||
/// you are updating many rows (with different ids) then you will get
|
/// you are updating many rows (with different ids) then you will get
|
||||||
/// better performance with a single [`merge_insert`] call instead of
|
/// better performance with a single [`merge_insert`] call instead of
|
||||||
/// repeatedly calilng this method.
|
/// repeatedly calling this method.
|
||||||
pub fn update(&self) -> UpdateBuilder {
|
pub fn update(&self) -> UpdateBuilder {
|
||||||
UpdateBuilder::new(self.inner.clone())
|
UpdateBuilder::new(self.inner.clone())
|
||||||
}
|
}
|
||||||
@@ -2804,7 +2804,7 @@ impl NativeTable {
|
|||||||
namespace_client: Option<Arc<dyn LanceNamespace>>,
|
namespace_client: Option<Arc<dyn LanceNamespace>>,
|
||||||
pushdown_operations: HashSet<NamespaceClientPushdownOperation>,
|
pushdown_operations: HashSet<NamespaceClientPushdownOperation>,
|
||||||
) -> Result<Self> {
|
) -> Result<Self> {
|
||||||
computed_columns::ensure_no_foreign_declarations(batches.arrow_schema().fields())?;
|
let batches = computed_columns::admit_create_source(batches)?;
|
||||||
// Default params uses format v1.
|
// Default params uses format v1.
|
||||||
let params = params.unwrap_or(WriteParams {
|
let params = params.unwrap_or(WriteParams {
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -2904,6 +2904,7 @@ impl NativeTable {
|
|||||||
pushdown_operations: HashSet<NamespaceClientPushdownOperation>,
|
pushdown_operations: HashSet<NamespaceClientPushdownOperation>,
|
||||||
session: Option<Arc<lance::session::Session>>,
|
session: Option<Arc<lance::session::Session>>,
|
||||||
) -> Result<Self> {
|
) -> Result<Self> {
|
||||||
|
let batches = computed_columns::admit_create_source(batches)?;
|
||||||
// Build table_id from namespace + name for the storage options provider
|
// Build table_id from namespace + name for the storage options provider
|
||||||
let mut table_id = namespace.clone();
|
let mut table_id = namespace.clone();
|
||||||
table_id.push(name.to_string());
|
table_id.push(name.to_string());
|
||||||
@@ -5677,7 +5678,7 @@ mod tests {
|
|||||||
TableStatistics {
|
TableStatistics {
|
||||||
num_rows: 250,
|
num_rows: 250,
|
||||||
num_indices: 0,
|
num_indices: 0,
|
||||||
total_bytes: 8925,
|
total_bytes: 8969,
|
||||||
fragment_stats: FragmentStatistics {
|
fragment_stats: FragmentStatistics {
|
||||||
num_fragments: 11,
|
num_fragments: 11,
|
||||||
num_small_fragments: 11,
|
num_small_fragments: 11,
|
||||||
|
|||||||
@@ -21,6 +21,7 @@
|
|||||||
//! [`computed_columns`] and [`computed_column_from_field`] read declarations
|
//! [`computed_columns`] and [`computed_column_from_field`] read declarations
|
||||||
//! back off a schema.
|
//! back off a schema.
|
||||||
|
|
||||||
|
use futures::StreamExt;
|
||||||
use std::collections::{BTreeSet, HashMap, HashSet};
|
use std::collections::{BTreeSet, HashMap, HashSet};
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
|
|
||||||
@@ -1338,6 +1339,106 @@ pub(crate) fn ensure_batch_writes_no_computed_values(
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Validate every computed-column declaration `schema` carries against the
|
||||||
|
/// schema itself: every field with declaration metadata is a complete
|
||||||
|
/// declaration, a SQL declaration re-plans to the field it declares, a
|
||||||
|
/// Function declaration satisfies the binding contract, and no declaration
|
||||||
|
/// reads another computed column. What passes here is what `refresh_column`
|
||||||
|
/// can execute.
|
||||||
|
pub(crate) fn ensure_declarations_are_planned(schema: &ArrowSchema) -> Result<()> {
|
||||||
|
let invalid = |message: String| Error::InvalidInput { message };
|
||||||
|
// A field with any declaration key is a declaration; a partial one is
|
||||||
|
// not "no declaration", it is a broken one.
|
||||||
|
for field in schema.fields() {
|
||||||
|
if field.metadata().keys().any(|k| is_declaration_key(k))
|
||||||
|
&& computed_column_from_field(field).is_none()
|
||||||
|
{
|
||||||
|
return Err(invalid(format!(
|
||||||
|
"field '{}' carries an incomplete computed-column declaration",
|
||||||
|
field.name()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let declared: HashSet<String> = computed_columns(schema)
|
||||||
|
.into_iter()
|
||||||
|
.map(|c| c.name)
|
||||||
|
.collect();
|
||||||
|
for column in computed_columns(schema) {
|
||||||
|
let field = schema.field_with_name(&column.name)?;
|
||||||
|
if !field.is_nullable() {
|
||||||
|
return Err(invalid(format!(
|
||||||
|
"computed column '{}' must be nullable until a refresh fills it",
|
||||||
|
column.name
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
match &column.kind {
|
||||||
|
ComputedColumnKind::Sql { expression } => {
|
||||||
|
let others: Vec<ArrowField> = schema
|
||||||
|
.fields()
|
||||||
|
.iter()
|
||||||
|
.filter(|f| f.name() != &column.name)
|
||||||
|
.map(|f| f.as_ref().clone())
|
||||||
|
.collect();
|
||||||
|
let bound = bind(Arc::new(ArrowSchema::new(others)), &column.name, expression)?;
|
||||||
|
if let Some(input) = bound.roots.iter().find(|r| declared.contains(*r)) {
|
||||||
|
return Err(invalid(format!(
|
||||||
|
"computed column '{}' reads computed column '{input}'",
|
||||||
|
column.name
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
if &bound.data_type != field.data_type() {
|
||||||
|
return Err(invalid(format!(
|
||||||
|
"computed column '{}' is declared as {} but its expression yields {}",
|
||||||
|
column.name,
|
||||||
|
field.data_type(),
|
||||||
|
bound.data_type
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let mut declared_inputs = column.inputs.clone();
|
||||||
|
declared_inputs.sort();
|
||||||
|
if declared_inputs != bound.inputs {
|
||||||
|
return Err(invalid(format!(
|
||||||
|
"computed column '{}' declares inputs {:?} but its expression reads {:?}",
|
||||||
|
column.name, declared_inputs, bound.inputs
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
ComputedColumnKind::Function { binding_id, .. } => {
|
||||||
|
// The binding validator resolves each input's leaf; the
|
||||||
|
// no-computed-input rule is about the root it hangs from.
|
||||||
|
let bindings = function_bindings(schema)?;
|
||||||
|
let Some(binding) = bindings.iter().find(|b| b.binding_id() == binding_id) else {
|
||||||
|
continue; // reported by the binding validator below
|
||||||
|
};
|
||||||
|
// Roots come from the canonical path parser: a quoted
|
||||||
|
// top-level name may itself contain a dot.
|
||||||
|
if let Some(input) = binding
|
||||||
|
.inputs()
|
||||||
|
.iter()
|
||||||
|
.filter_map(|input| resolve_field_path(schema, &input.field_path).ok())
|
||||||
|
.map(|resolved| resolved.root.name().as_str())
|
||||||
|
.find(|r| declared.contains(*r))
|
||||||
|
{
|
||||||
|
return Err(invalid(format!(
|
||||||
|
"computed column '{}' reads computed column '{input}'",
|
||||||
|
column.name
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
ComputedColumnKind::Unrecognized { kind } => {
|
||||||
|
return Err(Error::NotSupported {
|
||||||
|
message: format!(
|
||||||
|
"computed column '{}' is defined by '{kind}', which this version \
|
||||||
|
of lancedb cannot fill",
|
||||||
|
column.name
|
||||||
|
),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
ensure_supported_function_metadata(schema)
|
||||||
|
}
|
||||||
|
|
||||||
/// Reject fields carrying declaration metadata that did not come through
|
/// Reject fields carrying declaration metadata that did not come through
|
||||||
/// [`plan`]. One authority for creation, overwrite and raw transforms.
|
/// [`plan`]. One authority for creation, overwrite and raw transforms.
|
||||||
pub(crate) fn ensure_no_foreign_declarations<'a>(
|
pub(crate) fn ensure_no_foreign_declarations<'a>(
|
||||||
@@ -1796,6 +1897,54 @@ pub(super) async fn add_foreign_kind(table: &crate::Table, name: &str, kind: &st
|
|||||||
.unwrap();
|
.unwrap();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Admit a table's initial data: every declaration it carries is validated,
|
||||||
|
/// and the stream refuses any batch with values in a computed column, whose
|
||||||
|
/// values come from refresh alone. One boundary for every way a table is
|
||||||
|
/// created.
|
||||||
|
pub(crate) fn admit_create_source<S: lance_datafusion::utils::StreamingWriteSource>(
|
||||||
|
batches: S,
|
||||||
|
) -> Result<UnfilledDeclarations<S>> {
|
||||||
|
let schema = batches.arrow_schema();
|
||||||
|
ensure_declarations_are_planned(&schema)?;
|
||||||
|
let declared = computed_columns(&schema)
|
||||||
|
.into_iter()
|
||||||
|
.map(|c| c.name)
|
||||||
|
.collect();
|
||||||
|
Ok(UnfilledDeclarations {
|
||||||
|
inner: batches,
|
||||||
|
declared,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A write source whose computed columns must arrive unfilled.
|
||||||
|
pub(crate) struct UnfilledDeclarations<S> {
|
||||||
|
inner: S,
|
||||||
|
declared: Vec<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<S: lance_datafusion::utils::StreamingWriteSource> lance_datafusion::utils::StreamingWriteSource
|
||||||
|
for UnfilledDeclarations<S>
|
||||||
|
{
|
||||||
|
fn arrow_schema(&self) -> SchemaRef {
|
||||||
|
self.inner.arrow_schema()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn into_stream(self) -> datafusion_physical_plan::SendableRecordBatchStream {
|
||||||
|
if self.declared.is_empty() {
|
||||||
|
return self.inner.into_stream();
|
||||||
|
}
|
||||||
|
let schema = self.inner.arrow_schema();
|
||||||
|
let declared = self.declared;
|
||||||
|
let stream = self.inner.into_stream().map(move |batch| {
|
||||||
|
let batch = batch?;
|
||||||
|
ensure_batch_writes_no_computed_values(&declared, &batch)
|
||||||
|
.map_err(|e| datafusion_common::DataFusionError::External(Box::new(e)))?;
|
||||||
|
Ok(batch)
|
||||||
|
});
|
||||||
|
Box::pin(datafusion_physical_plan::stream::RecordBatchStreamAdapter::new(schema, stream))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
/// The gate's reproducer: the validator applies the same schema-level
|
/// The gate's reproducer: the validator applies the same schema-level
|
||||||
@@ -2646,6 +2795,8 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A create carries a declaration only if it re-plans completely; this
|
||||||
|
/// one lacks its inputs and is refused before its forged value matters.
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn test_create_table_cannot_inject_a_declaration() {
|
async fn test_create_table_cannot_inject_a_declaration() {
|
||||||
let conn = connect("memory://").execute().await.unwrap();
|
let conn = connect("memory://").execute().await.unwrap();
|
||||||
@@ -2673,7 +2824,7 @@ mod tests {
|
|||||||
.await
|
.await
|
||||||
.unwrap_err();
|
.unwrap_err();
|
||||||
assert!(
|
assert!(
|
||||||
matches!(&err, Error::InvalidInput { message } if message.contains("computed()")),
|
matches!(&err, Error::InvalidInput { message } if message.contains("computed column 'doubled'")),
|
||||||
"{err:?}"
|
"{err:?}"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -52,7 +52,7 @@ enum ConsistencyMode {
|
|||||||
/// refresh_window = min(3s, TTL/4)
|
/// refresh_window = min(3s, TTL/4)
|
||||||
///
|
///
|
||||||
/// | t < TTL - refresh_window | t < TTL | t >= TTL |
|
/// | t < TTL - refresh_window | t < TTL | t >= TTL |
|
||||||
/// | Return value | Background refresh & return value | syncronous refresh |
|
/// | Return value | Background refresh & return value | synchronous refresh |
|
||||||
Eventual(BackgroundCache<Arc<Dataset>, Error>),
|
Eventual(BackgroundCache<Arc<Dataset>, Error>),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -103,7 +103,7 @@ impl MergeInsertBuilder {
|
|||||||
/// but that behavior is subject to change.
|
/// but that behavior is subject to change.
|
||||||
///
|
///
|
||||||
/// An optional condition may be specified. If it is, then only
|
/// An optional condition may be specified. If it is, then only
|
||||||
/// matched rows that satisfy the condtion will be updated. Any
|
/// matched rows that satisfy the condition will be updated. Any
|
||||||
/// rows that do not satisfy the condition will be left as they
|
/// rows that do not satisfy the condition will be left as they
|
||||||
/// are. Failing to satisfy the condition does not cause a
|
/// are. Failing to satisfy the condition does not cause a
|
||||||
/// "matched row" to become a "not matched" row.
|
/// "matched row" to become a "not matched" row.
|
||||||
|
|||||||
@@ -904,7 +904,7 @@ fn unsharded_shard_id() -> Uuid {
|
|||||||
|
|
||||||
/// Build a [`ShardWriterConfig`] from the persisted `writer_config_defaults`.
|
/// Build a [`ShardWriterConfig`] from the persisted `writer_config_defaults`.
|
||||||
///
|
///
|
||||||
/// Unknown or unparseable keys are ignored; absent keys keep the
|
/// Unknown or unparsable keys are ignored; absent keys keep the
|
||||||
/// [`ShardWriterConfig`] default. The shard id is set by `mem_wal_writer`.
|
/// [`ShardWriterConfig`] default. The shard id is set by `mem_wal_writer`.
|
||||||
fn shard_writer_config_from_defaults(defaults: &HashMap<String, String>) -> ShardWriterConfig {
|
fn shard_writer_config_from_defaults(defaults: &HashMap<String, String>) -> ShardWriterConfig {
|
||||||
let mut config = ShardWriterConfig::default().with_shard_spec_id(SHARDING_SPEC_ID);
|
let mut config = ShardWriterConfig::default().with_shard_spec_id(SHARDING_SPEC_ID);
|
||||||
|
|||||||
@@ -19,6 +19,7 @@ use lancedb::{
|
|||||||
connect, connect_namespace,
|
connect, connect_namespace,
|
||||||
database::listing::{
|
database::listing::{
|
||||||
ListingDatabaseOptions, NewTableConfig, OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS,
|
ListingDatabaseOptions, NewTableConfig, OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS,
|
||||||
|
OPT_NEW_TABLE_STORAGE_VERSION,
|
||||||
},
|
},
|
||||||
query::{ExecutableQuery, QueryBase},
|
query::{ExecutableQuery, QueryBase},
|
||||||
table::{AddDataMode, CompactionOptions, OptimizeAction, OptimizeStats, WriteOptions},
|
table::{AddDataMode, CompactionOptions, OptimizeAction, OptimizeStats, WriteOptions},
|
||||||
@@ -146,7 +147,10 @@ async fn non_blob_table_keeps_default_format_and_row_id_setting() -> Result<()>
|
|||||||
let schema = Arc::new(Schema::new(vec![Field::new("id", DataType::Int64, false)]));
|
let schema = Arc::new(Schema::new(vec![Field::new("id", DataType::Int64, false)]));
|
||||||
let table = db.create_empty_table("t", schema).execute().await?;
|
let table = db.create_empty_table("t", schema).execute().await?;
|
||||||
|
|
||||||
assert!(!supports_blob_v2(storage_format_version(&table).await));
|
assert_eq!(
|
||||||
|
storage_format_version(&table).await,
|
||||||
|
LanceFileVersion::Stable.resolve()
|
||||||
|
);
|
||||||
assert!(!uses_stable_row_ids(&table).await);
|
assert!(!uses_stable_row_ids(&table).await);
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
@@ -809,7 +813,11 @@ async fn fetch_blobs_rejects_unknown_column() -> Result<()> {
|
|||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn fetch_blobs_rejects_legacy_v1_blob_column() -> Result<()> {
|
async fn fetch_blobs_rejects_legacy_v1_blob_column() -> Result<()> {
|
||||||
let tmp = tempdir().unwrap();
|
let tmp = tempdir().unwrap();
|
||||||
let db = connect(tmp.path().to_str().unwrap()).execute().await?;
|
// Legacy v1 blob columns are only writable at file version <= 2.1.
|
||||||
|
let db = connect(tmp.path().to_str().unwrap())
|
||||||
|
.storage_options([(OPT_NEW_TABLE_STORAGE_VERSION, "2.1")])
|
||||||
|
.execute()
|
||||||
|
.await?;
|
||||||
let legacy = Field::new("image", DataType::LargeBinary, true).with_metadata(
|
let legacy = Field::new("image", DataType::LargeBinary, true).with_metadata(
|
||||||
std::collections::HashMap::from([("lance-encoding:blob".to_string(), "true".to_string())]),
|
std::collections::HashMap::from([("lance-encoding:blob".to_string(), "true".to_string())]),
|
||||||
);
|
);
|
||||||
|
|||||||
Reference in New Issue
Block a user