mirror of
https://github.com/lancedb/lancedb.git
synced 2026-09-21 20:45:57 +00:00
Merge remote-tracking branch 'origin/main' into gatekeeper/fix-2325-1
# Conflicts: # Cargo.lock
This commit is contained in:
+1
-1
@@ -1,5 +1,5 @@
|
||||
[tool.bumpversion]
|
||||
current_version = "0.40.0-beta.1"
|
||||
current_version = "0.40.0-beta.3"
|
||||
parse = """(?x)
|
||||
(?P<major>0|[1-9]\\d*)\\.
|
||||
(?P<minor>0|[1-9]\\d*)\\.
|
||||
|
||||
Generated
+47
-47
@@ -3523,8 +3523,8 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c"
|
||||
|
||||
[[package]]
|
||||
name = "fsst"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"rand 0.9.5",
|
||||
@@ -5073,8 +5073,8 @@ checksum = "e037a2e1d8d5fdbd49b16a4ea09d5d6401c1f29eca5ff29d03d3824dba16256a"
|
||||
|
||||
[[package]]
|
||||
name = "lance"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"arrow",
|
||||
@@ -5146,8 +5146,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-arrow"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -5169,7 +5169,7 @@ dependencies = [
|
||||
[[package]]
|
||||
name = "lance-arrow-scalar"
|
||||
version = "58.0.0"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -5183,7 +5183,7 @@ dependencies = [
|
||||
[[package]]
|
||||
name = "lance-arrow-stats"
|
||||
version = "58.0.0"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -5192,8 +5192,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-bitpacking"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrayref",
|
||||
"crunchy",
|
||||
@@ -5203,8 +5203,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-core"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -5241,8 +5241,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-datafusion"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5273,8 +5273,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-datagen"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5291,8 +5291,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-derive"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -5301,8 +5301,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-encoding"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow-arith",
|
||||
"arrow-array",
|
||||
@@ -5335,8 +5335,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-file"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow-arith",
|
||||
"arrow-array",
|
||||
@@ -5368,8 +5368,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-geo"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"datafusion",
|
||||
"geo-traits",
|
||||
@@ -5383,8 +5383,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-index"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"arrow",
|
||||
@@ -5452,8 +5452,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-index-core"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -5475,8 +5475,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-io"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5516,8 +5516,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-linalg"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -5531,8 +5531,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-namespace"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"async-trait",
|
||||
@@ -5546,8 +5546,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-namespace-impls"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-ipc",
|
||||
@@ -5600,8 +5600,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-select"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -5615,8 +5615,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-table"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5656,8 +5656,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-testing"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -5670,8 +5670,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-tokenizer"
|
||||
version = "13.0.0-beta.4"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.4#fea665e0d96d73acd1be330ce850973ff00b7126"
|
||||
version = "13.0.0-beta.6"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v13.0.0-beta.6#93c055dee58553c8b7c816c29892fe32877be3d0"
|
||||
dependencies = [
|
||||
"frostem",
|
||||
"icu_segmenter",
|
||||
@@ -5684,7 +5684,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lancedb"
|
||||
version = "0.40.0-beta.1"
|
||||
version = "0.40.0-beta.3"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"anyhow",
|
||||
@@ -5781,7 +5781,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lancedb-nodejs"
|
||||
version = "0.40.0-beta.1"
|
||||
version = "0.40.0-beta.3"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -5806,7 +5806,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lancedb-python"
|
||||
version = "0.40.0-beta.1"
|
||||
version = "0.40.0-beta.3"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"arrow",
|
||||
|
||||
+14
-14
@@ -13,20 +13,20 @@ categories = ["database-implementations"]
|
||||
rust-version = "1.91.0"
|
||||
|
||||
[workspace.dependencies]
|
||||
lance = { "version" = "=13.0.0-beta.4", default-features = false, "tag" = "v13.0.0-beta.4", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-core = { "version" = "=13.0.0-beta.4", "tag" = "v13.0.0-beta.4", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datagen = { "version" = "=13.0.0-beta.4", "tag" = "v13.0.0-beta.4", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-file = { "version" = "=13.0.0-beta.4", "tag" = "v13.0.0-beta.4", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-io = { "version" = "=13.0.0-beta.4", default-features = false, "tag" = "v13.0.0-beta.4", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-index = { "version" = "=13.0.0-beta.4", "tag" = "v13.0.0-beta.4", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-linalg = { "version" = "=13.0.0-beta.4", "tag" = "v13.0.0-beta.4", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace = { "version" = "=13.0.0-beta.4", "tag" = "v13.0.0-beta.4", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace-impls = { "version" = "=13.0.0-beta.4", default-features = false, "tag" = "v13.0.0-beta.4", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-table = { "version" = "=13.0.0-beta.4", "tag" = "v13.0.0-beta.4", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-testing = { "version" = "=13.0.0-beta.4", "tag" = "v13.0.0-beta.4", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datafusion = { "version" = "=13.0.0-beta.4", "tag" = "v13.0.0-beta.4", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-encoding = { "version" = "=13.0.0-beta.4", "tag" = "v13.0.0-beta.4", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-arrow = { "version" = "=13.0.0-beta.4", "tag" = "v13.0.0-beta.4", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance = { "version" = "=13.0.0-beta.6", default-features = false, "tag" = "v13.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-core = { "version" = "=13.0.0-beta.6", "tag" = "v13.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datagen = { "version" = "=13.0.0-beta.6", "tag" = "v13.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-file = { "version" = "=13.0.0-beta.6", "tag" = "v13.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-io = { "version" = "=13.0.0-beta.6", default-features = false, "tag" = "v13.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-index = { "version" = "=13.0.0-beta.6", "tag" = "v13.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-linalg = { "version" = "=13.0.0-beta.6", "tag" = "v13.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace = { "version" = "=13.0.0-beta.6", "tag" = "v13.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace-impls = { "version" = "=13.0.0-beta.6", default-features = false, "tag" = "v13.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-table = { "version" = "=13.0.0-beta.6", "tag" = "v13.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-testing = { "version" = "=13.0.0-beta.6", "tag" = "v13.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datafusion = { "version" = "=13.0.0-beta.6", "tag" = "v13.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-encoding = { "version" = "=13.0.0-beta.6", "tag" = "v13.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-arrow = { "version" = "=13.0.0-beta.6", "tag" = "v13.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lancedb = { path = "rust/lancedb", default-features = false }
|
||||
ahash = "0.8"
|
||||
# Note that this one does not include pyarrow
|
||||
|
||||
@@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`:
|
||||
<dependency>
|
||||
<groupId>com.lancedb</groupId>
|
||||
<artifactId>lancedb-core</artifactId>
|
||||
<version>0.40.0-beta.1</version>
|
||||
<version>0.40.0-beta.3</version>
|
||||
</dependency>
|
||||
```
|
||||
|
||||
|
||||
@@ -578,6 +578,41 @@ so this is safe to call repeatedly.
|
||||
|
||||
***
|
||||
|
||||
### functionErrors()
|
||||
|
||||
```ts
|
||||
abstract functionErrors(options?): Promise<FunctionErrors>
|
||||
```
|
||||
|
||||
The per-row errors Function refreshes recorded on this table.
|
||||
|
||||
A refresh running under a skip policy records each row it skipped with
|
||||
the input that failed and the error. This lists those records, newest
|
||||
job first, plus a summary for any fragment whose per-row detail was
|
||||
capped. LanceDB Cloud and Enterprise only; reading errors needs read
|
||||
access to the table, since a message carries the value that failed.
|
||||
|
||||
#### Parameters
|
||||
|
||||
* **options?**: [`FunctionErrorsOptions`](../interfaces/FunctionErrorsOptions.md)
|
||||
Optional filters: `jobId`,
|
||||
`column`, and `limit` (server default 10000, cap 100000).
|
||||
|
||||
#### Returns
|
||||
|
||||
`Promise`<[`FunctionErrors`](../interfaces/FunctionErrors.md)>
|
||||
|
||||
The records, the capped fragments,
|
||||
and whether the listing stopped at its limit.
|
||||
|
||||
#### Example
|
||||
|
||||
```ts
|
||||
const { records, truncated } = await table.functionErrors({ column: "embedding" });
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### getLsmStats()
|
||||
|
||||
```ts
|
||||
|
||||
@@ -92,6 +92,10 @@
|
||||
- [FtsToken](interfaces/FtsToken.md)
|
||||
- [FullTextQuery](interfaces/FullTextQuery.md)
|
||||
- [FullTextSearchOptions](interfaces/FullTextSearchOptions.md)
|
||||
- [FunctionErrorFragment](interfaces/FunctionErrorFragment.md)
|
||||
- [FunctionErrorRecord](interfaces/FunctionErrorRecord.md)
|
||||
- [FunctionErrors](interfaces/FunctionErrors.md)
|
||||
- [FunctionErrorsOptions](interfaces/FunctionErrorsOptions.md)
|
||||
- [GenerationStats](interfaces/GenerationStats.md)
|
||||
- [HnswPqOptions](interfaces/HnswPqOptions.md)
|
||||
- [HnswSqOptions](interfaces/HnswSqOptions.md)
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||
|
||||
***
|
||||
|
||||
[@lancedb/lancedb](../globals.md) / FunctionErrorFragment
|
||||
|
||||
# Interface: FunctionErrorFragment
|
||||
|
||||
A fragment whose per-row error detail was capped: `rowsSkipped` rows
|
||||
failed, of which only `rowsRecorded` have a record of their own.
|
||||
|
||||
## Properties
|
||||
|
||||
### fragmentId
|
||||
|
||||
```ts
|
||||
fragmentId: number;
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### jobId
|
||||
|
||||
```ts
|
||||
jobId: string;
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### rowsRecorded
|
||||
|
||||
```ts
|
||||
rowsRecorded: number;
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### rowsSkipped
|
||||
|
||||
```ts
|
||||
rowsSkipped: number;
|
||||
```
|
||||
@@ -0,0 +1,92 @@
|
||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||
|
||||
***
|
||||
|
||||
[@lancedb/lancedb](../globals.md) / FunctionErrorRecord
|
||||
|
||||
# Interface: FunctionErrorRecord
|
||||
|
||||
One row a Function refresh skipped, as the server recorded it.
|
||||
|
||||
## Properties
|
||||
|
||||
### column
|
||||
|
||||
```ts
|
||||
column: string;
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### createdAtMillis
|
||||
|
||||
```ts
|
||||
createdAtMillis: number;
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### errorMessage
|
||||
|
||||
```ts
|
||||
errorMessage: string;
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### errorType
|
||||
|
||||
```ts
|
||||
errorType: string;
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### fragmentId
|
||||
|
||||
```ts
|
||||
fragmentId: number;
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### function
|
||||
|
||||
```ts
|
||||
function: string;
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### functionVersion
|
||||
|
||||
```ts
|
||||
functionVersion: string;
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### jobId
|
||||
|
||||
```ts
|
||||
jobId: string;
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### rowOffset?
|
||||
|
||||
```ts
|
||||
optional rowOffset: number;
|
||||
```
|
||||
|
||||
The row's offset within the fragment; absent when the fragment's
|
||||
detail was capped.
|
||||
|
||||
***
|
||||
|
||||
### tableVersion
|
||||
|
||||
```ts
|
||||
tableVersion: number;
|
||||
```
|
||||
@@ -0,0 +1,39 @@
|
||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||
|
||||
***
|
||||
|
||||
[@lancedb/lancedb](../globals.md) / FunctionErrors
|
||||
|
||||
# Interface: FunctionErrors
|
||||
|
||||
A table's per-row Function errors.
|
||||
|
||||
## Properties
|
||||
|
||||
### fragments
|
||||
|
||||
```ts
|
||||
fragments: FunctionErrorFragment[];
|
||||
```
|
||||
|
||||
Fragments whose detail was capped.
|
||||
|
||||
***
|
||||
|
||||
### records
|
||||
|
||||
```ts
|
||||
records: FunctionErrorRecord[];
|
||||
```
|
||||
|
||||
The recorded rows, newest job first.
|
||||
|
||||
***
|
||||
|
||||
### truncated
|
||||
|
||||
```ts
|
||||
truncated: boolean;
|
||||
```
|
||||
|
||||
Whether the listing stopped at its limit.
|
||||
@@ -0,0 +1,39 @@
|
||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||
|
||||
***
|
||||
|
||||
[@lancedb/lancedb](../globals.md) / FunctionErrorsOptions
|
||||
|
||||
# Interface: FunctionErrorsOptions
|
||||
|
||||
Which per-row Function errors to list; every filter is optional.
|
||||
|
||||
## Properties
|
||||
|
||||
### column?
|
||||
|
||||
```ts
|
||||
optional column: string;
|
||||
```
|
||||
|
||||
Only errors on this column.
|
||||
|
||||
***
|
||||
|
||||
### jobId?
|
||||
|
||||
```ts
|
||||
optional jobId: string;
|
||||
```
|
||||
|
||||
Only errors recorded by this job.
|
||||
|
||||
***
|
||||
|
||||
### limit?
|
||||
|
||||
```ts
|
||||
optional limit: number;
|
||||
```
|
||||
|
||||
At most this many records (server default 10000, cap 100000).
|
||||
@@ -6,64 +6,17 @@
|
||||
|
||||
# Interface: MaterializedViewDefinition
|
||||
|
||||
The query that defines a materialized view.
|
||||
The query that defines a materialized view, as stored:
|
||||
`SELECT columns FROM [ns.]table [, function(args) AS alias | , UNNEST(column) AS alias]
|
||||
[WHERE predicate] [LIMIT n]`. A Function in `FROM` position yields one row
|
||||
per element it returns.
|
||||
|
||||
## Properties
|
||||
|
||||
### filter?
|
||||
### query
|
||||
|
||||
```ts
|
||||
optional filter: string;
|
||||
query: string;
|
||||
```
|
||||
|
||||
SQL predicate selecting the source rows the view holds.
|
||||
|
||||
***
|
||||
|
||||
### inputs
|
||||
|
||||
```ts
|
||||
inputs: string[];
|
||||
```
|
||||
|
||||
Source columns the projections and filter read.
|
||||
|
||||
***
|
||||
|
||||
### limit?
|
||||
|
||||
```ts
|
||||
optional limit: number;
|
||||
```
|
||||
|
||||
Cap on the number of rows the view holds.
|
||||
|
||||
***
|
||||
|
||||
### projections
|
||||
|
||||
```ts
|
||||
projections: [string, string][];
|
||||
```
|
||||
|
||||
`[output column, SQL expression]` pairs, in view schema order.
|
||||
|
||||
***
|
||||
|
||||
### sourceNamespace
|
||||
|
||||
```ts
|
||||
sourceNamespace: string[];
|
||||
```
|
||||
|
||||
Namespace holding the source table; empty is the root namespace.
|
||||
|
||||
***
|
||||
|
||||
### sourceTable
|
||||
|
||||
```ts
|
||||
sourceTable: string;
|
||||
```
|
||||
|
||||
Name of the source table, in the same database as the view.
|
||||
The defining query, in the canonical spelling the server stores.
|
||||
|
||||
@@ -172,6 +172,12 @@ listing a storage directory.
|
||||
|
||||
::: lancedb.functions.RefreshColumnResult
|
||||
|
||||
::: lancedb.FunctionErrors
|
||||
|
||||
::: lancedb.FunctionErrorRecord
|
||||
|
||||
::: lancedb.FunctionErrorFragment
|
||||
|
||||
::: lancedb.job.Job
|
||||
|
||||
::: lancedb.job.AsyncJob
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
<parent>
|
||||
<groupId>com.lancedb</groupId>
|
||||
<artifactId>lancedb-parent</artifactId>
|
||||
<version>0.40.0-beta.1</version>
|
||||
<version>0.40.0-beta.3</version>
|
||||
<relativePath>../pom.xml</relativePath>
|
||||
</parent>
|
||||
|
||||
|
||||
+2
-2
@@ -6,7 +6,7 @@
|
||||
|
||||
<groupId>com.lancedb</groupId>
|
||||
<artifactId>lancedb-parent</artifactId>
|
||||
<version>0.40.0-beta.1</version>
|
||||
<version>0.40.0-beta.3</version>
|
||||
<packaging>pom</packaging>
|
||||
<name>${project.artifactId}</name>
|
||||
<description>LanceDB Java SDK Parent POM</description>
|
||||
@@ -28,7 +28,7 @@
|
||||
<properties>
|
||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||
<arrow.version>15.0.0</arrow.version>
|
||||
<lance-core.version>13.0.0-beta.4</lance-core.version>
|
||||
<lance-core.version>13.0.0-beta.6</lance-core.version>
|
||||
<spotless.skip>false</spotless.skip>
|
||||
<spotless.version>2.30.0</spotless.version>
|
||||
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
[package]
|
||||
name = "lancedb-nodejs"
|
||||
edition.workspace = true
|
||||
version = "0.40.0-beta.1"
|
||||
version = "0.40.0-beta.3"
|
||||
publish = false
|
||||
license.workspace = true
|
||||
description.workspace = true
|
||||
|
||||
@@ -28,6 +28,39 @@ describe("materialized views", () => {
|
||||
});
|
||||
afterEach(() => tmpDir.removeCallback());
|
||||
|
||||
it("reads stored queries and legacy layouts", () => {
|
||||
const read = (stored: string) =>
|
||||
definitionFromMetadata(new Map([[DEFINITION_META_KEY, stored]]), "v");
|
||||
const query =
|
||||
"SELECT id, c.chunk FROM ns.docs, UNNEST(chunks) AS c WHERE id > 1";
|
||||
expect(read(`{"format":1,"query":${JSON.stringify(query)}}`).query).toBe(
|
||||
query,
|
||||
);
|
||||
|
||||
// The structured layout written before the format number reads as the
|
||||
// query it described, under either of its kind tags.
|
||||
expect(
|
||||
read(
|
||||
'{"kind":"namespaced_select","source_table":"people","source_namespace":["ns"],' +
|
||||
'"projections":[{"output":"name","expression":"`name`"},' +
|
||||
'{"output":"Shout","expression":"upper(name)"}],"filter":"age >= 18","limit":42}',
|
||||
).query,
|
||||
).toBe(
|
||||
"SELECT `name`, upper(name) AS `Shout` FROM ns.people WHERE age >= 18 LIMIT 42",
|
||||
);
|
||||
expect(read('{"kind":"select","source_table":"people"}').query).toBe(
|
||||
"SELECT * FROM people",
|
||||
);
|
||||
|
||||
// A newer writer's layout is reported, never guessed at.
|
||||
for (const newer of [
|
||||
`{"format":2,"query":${JSON.stringify(query)}}`,
|
||||
'{"kind":"select_v3","source_table":"people"}',
|
||||
]) {
|
||||
expect(() => read(newer)).toThrow(/cannot refresh/);
|
||||
}
|
||||
});
|
||||
|
||||
it("rejects a stored limit a number cannot carry", () => {
|
||||
const big = new Map([
|
||||
[
|
||||
@@ -38,36 +71,6 @@ describe("materialized views", () => {
|
||||
expect(() => definitionFromMetadata(big, "v")).toThrow(
|
||||
/too large to represent exactly/,
|
||||
);
|
||||
|
||||
const safe = new Map([
|
||||
[
|
||||
DEFINITION_META_KEY,
|
||||
'{"kind":"select","source_table":"people","limit":42}',
|
||||
],
|
||||
]);
|
||||
expect(definitionFromMetadata(safe, "v").limit).toBe(42);
|
||||
});
|
||||
|
||||
it("reads the namespaced select kind and refuses unknown kinds", () => {
|
||||
// "namespaced_select" is the namespaced form of "select": same shape, a
|
||||
// separate kind so readers that predate it refuse instead of resolving
|
||||
// the source at the root.
|
||||
const namespaced = new Map([
|
||||
[
|
||||
DEFINITION_META_KEY,
|
||||
'{"kind":"namespaced_select","source_table":"people","source_namespace":["ns"]}',
|
||||
],
|
||||
]);
|
||||
const definition = definitionFromMetadata(namespaced, "v");
|
||||
expect(definition.sourceTable).toBe("people");
|
||||
expect(definition.sourceNamespace).toEqual(["ns"]);
|
||||
|
||||
const unknown = new Map([
|
||||
[DEFINITION_META_KEY, '{"kind":"select_v3","source_table":"people"}'],
|
||||
]);
|
||||
expect(() => definitionFromMetadata(unknown, "v")).toThrow(
|
||||
/cannot refresh/,
|
||||
);
|
||||
});
|
||||
|
||||
it("creates, refreshes and queries a view", async () => {
|
||||
@@ -88,13 +91,9 @@ describe("materialized views", () => {
|
||||
});
|
||||
const view = await db.openMaterializedView("adults");
|
||||
const definition = await view.definition();
|
||||
expect(definition.sourceTable).toBe("people");
|
||||
expect(definition.filter).toBe("age >= 18");
|
||||
expect(definition.projections).toEqual([
|
||||
["name", "`name`"],
|
||||
["age", "`age`"],
|
||||
]);
|
||||
expect(definition.inputs).toEqual(["age", "name"]);
|
||||
expect(definition.query).toBe(
|
||||
"SELECT name, age FROM people WHERE age >= 18",
|
||||
);
|
||||
});
|
||||
|
||||
it("refreshes incrementally after an append", async () => {
|
||||
|
||||
@@ -195,6 +195,66 @@ describe("remote connection", () => {
|
||||
);
|
||||
});
|
||||
|
||||
it("lists the rows a Function refresh skipped", async () => {
|
||||
const bodies: unknown[] = [];
|
||||
await withMockDatabase(
|
||||
(req, res) => {
|
||||
const path = req.url ?? "";
|
||||
if (path.endsWith("/describe/")) {
|
||||
res.writeHead(200, { "Content-Type": "application/json" }).end(
|
||||
JSON.stringify({
|
||||
name: "docs",
|
||||
version: 1,
|
||||
schema: { fields: [] },
|
||||
}),
|
||||
);
|
||||
return;
|
||||
}
|
||||
if (path === "/v1/table/docs/errors") {
|
||||
let body = "";
|
||||
req.on("data", (chunk) => {
|
||||
body += chunk;
|
||||
});
|
||||
req.on("end", () => {
|
||||
bodies.push(JSON.parse(body));
|
||||
res.writeHead(200, { "Content-Type": "application/json" }).end(
|
||||
`{"records": [{"job_id": "j-7", "fragment_id": 3, "row_offset": 9,
|
||||
"column": "embedding", "function": "embed", "function_version": "2",
|
||||
"table_version": 11, "error_type": "ValueError",
|
||||
"error_message": "bad input 'x'", "created_at_millis": 1700000000000}],
|
||||
"fragments": [{"job_id": "j-7", "fragment_id": 4, "rows_skipped": 500,
|
||||
"rows_recorded": 100}], "truncated": true}`,
|
||||
);
|
||||
});
|
||||
return;
|
||||
}
|
||||
res.writeHead(404).end();
|
||||
},
|
||||
async (db) => {
|
||||
const table = await db.openTable("docs");
|
||||
const errors = await table.functionErrors({
|
||||
jobId: "j-7",
|
||||
column: "embedding",
|
||||
limit: 2,
|
||||
});
|
||||
expect(errors.truncated).toBe(true);
|
||||
expect(errors.records.map((r) => r.errorMessage)).toEqual([
|
||||
"bad input 'x'",
|
||||
]);
|
||||
expect(errors.records[0].rowOffset).toBe(9);
|
||||
expect(errors.fragments[0].rowsSkipped).toBe(500);
|
||||
await table.functionErrors();
|
||||
await expect(table.functionErrors({ limit: -1 })).rejects.toThrow(
|
||||
"limit must be a non-negative integer",
|
||||
);
|
||||
},
|
||||
);
|
||||
expect(bodies).toEqual([
|
||||
JSON.parse('{"job_id": "j-7", "column": "embedding", "limit": 2}'),
|
||||
{},
|
||||
]);
|
||||
});
|
||||
|
||||
it("surfaces JSON server errors from remote table operations", async () => {
|
||||
await withMockDatabase(
|
||||
(req, res) => {
|
||||
|
||||
@@ -4358,6 +4358,14 @@ describe("computed columns", () => {
|
||||
expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]);
|
||||
});
|
||||
|
||||
it("records Function errors only on remote tables", async () => {
|
||||
const db = await connect(tmpDir.name);
|
||||
const table = await db.createTable("errors_local", [{ x: 1 }]);
|
||||
await expect(table.functionErrors()).rejects.toThrow(
|
||||
"LanceDB Cloud and Enterprise",
|
||||
);
|
||||
});
|
||||
|
||||
it("returns a job handle from refreshColumnAsync", async () => {
|
||||
const db = await connect(tmpDir.name);
|
||||
const table = await db.createTable("computed_job", [{ x: 1 }, { x: 2 }]);
|
||||
|
||||
@@ -56,6 +56,10 @@ export {
|
||||
AddResult,
|
||||
AddColumnsResult,
|
||||
RefreshColumnResult,
|
||||
FunctionErrors,
|
||||
FunctionErrorsOptions,
|
||||
FunctionErrorRecord,
|
||||
FunctionErrorFragment,
|
||||
RefreshMaterializedViewResult,
|
||||
AlterColumnsResult,
|
||||
UpdateFieldMetadataResult,
|
||||
|
||||
@@ -7,20 +7,18 @@ import { Table } from "./table";
|
||||
/** Schema metadata key holding a materialized view's definition. */
|
||||
export const DEFINITION_META_KEY = "mv.definition";
|
||||
|
||||
/** The query that defines a materialized view. */
|
||||
/** The stored layout this version reads: `{"format": 1, "query": "<SQL>"}`. */
|
||||
export const DEFINITION_FORMAT = 1;
|
||||
|
||||
/**
|
||||
* The query that defines a materialized view, as stored:
|
||||
* `SELECT columns FROM [ns.]table [, function(args) AS alias | , UNNEST(column) AS alias]
|
||||
* [WHERE predicate] [LIMIT n]`. A Function in `FROM` position yields one row
|
||||
* per element it returns.
|
||||
*/
|
||||
export interface MaterializedViewDefinition {
|
||||
/** Name of the source table, in the same database as the view. */
|
||||
sourceTable: string;
|
||||
/** `[output column, SQL expression]` pairs, in view schema order. */
|
||||
projections: [string, string][];
|
||||
/** SQL predicate selecting the source rows the view holds. */
|
||||
filter?: string;
|
||||
/** Cap on the number of rows the view holds. */
|
||||
limit?: number;
|
||||
/** Source columns the projections and filter read. */
|
||||
inputs: string[];
|
||||
/** Namespace holding the source table; empty is the root namespace. */
|
||||
sourceNamespace: string[];
|
||||
/** The defining query, in the canonical spelling the server stores. */
|
||||
query: string;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -88,15 +86,21 @@ export function definitionFromJson(
|
||||
): MaterializedViewDefinition {
|
||||
// biome-ignore lint/suspicious/noExplicitAny: raw JSON
|
||||
const value: any = JSON.parse(raw);
|
||||
// "namespaced_select" keeps older readers from resolving the source at root.
|
||||
if (
|
||||
value.kind !== undefined &&
|
||||
value.kind !== "select" &&
|
||||
value.kind !== "namespaced_select"
|
||||
) {
|
||||
if (value.format !== undefined) {
|
||||
// A newer writer's layout is reported, never guessed at.
|
||||
if (!Number.isInteger(value.format) || value.format > DEFINITION_FORMAT) {
|
||||
throw new Error(
|
||||
`materialized view '${name}' is stored in format ${value.format}, ` +
|
||||
"which this version of lancedb cannot refresh",
|
||||
);
|
||||
}
|
||||
return { query: value.query };
|
||||
}
|
||||
// The structured layout written before the format number.
|
||||
if (value.kind !== "select" && value.kind !== "namespaced_select") {
|
||||
throw new Error(
|
||||
`materialized view '${name}' is defined by '${value.kind}', which this ` +
|
||||
"version of lancedb cannot refresh",
|
||||
`materialized view '${name}' is stored in format kind '${value.kind}', ` +
|
||||
"which this version of lancedb cannot refresh",
|
||||
);
|
||||
}
|
||||
const limit = value.limit ?? undefined;
|
||||
@@ -108,18 +112,41 @@ export function definitionFromJson(
|
||||
`materialized view '${name}' has a stored limit too large to represent exactly`,
|
||||
);
|
||||
}
|
||||
return {
|
||||
sourceTable: value.source_table,
|
||||
// biome-ignore lint/suspicious/noExplicitAny: raw JSON
|
||||
projections: (value.projections ?? []).map((p: any) => [
|
||||
p.output,
|
||||
p.expression,
|
||||
]),
|
||||
filter: value.filter ?? undefined,
|
||||
limit,
|
||||
inputs: value.inputs ?? [],
|
||||
sourceNamespace: value.source_namespace ?? [],
|
||||
};
|
||||
return { query: legacyQuery(value, limit) };
|
||||
}
|
||||
|
||||
function legacyIdent(name: string): string {
|
||||
return /^[a-z_][a-z0-9_]*$/.test(name)
|
||||
? name
|
||||
: `\`${name.replace(/`/g, "``")}\``;
|
||||
}
|
||||
|
||||
/** Render the pre-format structured layout as the query it described. */
|
||||
// biome-ignore lint/suspicious/noExplicitAny: raw JSON
|
||||
function legacyQuery(value: any, limit: number | undefined): string {
|
||||
// biome-ignore lint/suspicious/noExplicitAny: raw JSON
|
||||
const projections: any[] = value.projections ?? [];
|
||||
const columns =
|
||||
projections.length === 0
|
||||
? "*"
|
||||
: projections
|
||||
.map((p) =>
|
||||
p.expression === p.output || p.expression === `\`${p.output}\``
|
||||
? p.expression
|
||||
: `${p.expression} AS ${legacyIdent(p.output)}`,
|
||||
)
|
||||
.join(", ");
|
||||
const table = [...(value.source_namespace ?? []), value.source_table]
|
||||
.map(legacyIdent)
|
||||
.join(".");
|
||||
let query = `SELECT ${columns} FROM ${table}`;
|
||||
if (value.filter !== undefined && value.filter !== null) {
|
||||
query += ` WHERE ${value.filter}`;
|
||||
}
|
||||
if (limit !== undefined) {
|
||||
query += ` LIMIT ${limit}`;
|
||||
}
|
||||
return query;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -21,6 +21,7 @@ import { BlobFile } from "./blob";
|
||||
import { EmbeddingFunctionConfig, getRegistry } from "./embedding/registry";
|
||||
import { IndexOptions } from "./indices";
|
||||
import { Job } from "./job";
|
||||
import { validateNonNegativeInteger } from "./materialized_view";
|
||||
import { MergeInsertBuilder } from "./merge";
|
||||
import {
|
||||
AddColumnsResult,
|
||||
@@ -30,6 +31,8 @@ import {
|
||||
BranchContents,
|
||||
DeleteResult,
|
||||
DropColumnsResult,
|
||||
FunctionErrors,
|
||||
FunctionErrorsOptions,
|
||||
IndexConfig,
|
||||
IndexStatistics,
|
||||
LsmStats,
|
||||
@@ -636,6 +639,27 @@ export abstract class Table {
|
||||
*/
|
||||
abstract refreshColumnAsync(column: string): Promise<Job>;
|
||||
|
||||
/**
|
||||
* The per-row errors Function refreshes recorded on this table.
|
||||
*
|
||||
* A refresh running under a skip policy records each row it skipped with
|
||||
* the input that failed and the error. This lists those records, newest
|
||||
* job first, plus a summary for any fragment whose per-row detail was
|
||||
* capped. LanceDB Cloud and Enterprise only; reading errors needs read
|
||||
* access to the table, since a message carries the value that failed.
|
||||
* @param {FunctionErrorsOptions} options Optional filters: `jobId`,
|
||||
* `column`, and `limit` (server default 10000, cap 100000).
|
||||
* @returns {Promise<FunctionErrors>} The records, the capped fragments,
|
||||
* and whether the listing stopped at its limit.
|
||||
* @example
|
||||
* ```ts
|
||||
* const { records, truncated } = await table.functionErrors({ column: "embedding" });
|
||||
* ```
|
||||
*/
|
||||
abstract functionErrors(
|
||||
options?: FunctionErrorsOptions,
|
||||
): Promise<FunctionErrors>;
|
||||
|
||||
/**
|
||||
* Recompute this table's contents from its materialized-view definition.
|
||||
*
|
||||
@@ -1363,6 +1387,13 @@ export class LocalTable extends Table {
|
||||
return new Job(await this.inner.refreshColumnAsync(column));
|
||||
}
|
||||
|
||||
async functionErrors(
|
||||
options?: FunctionErrorsOptions,
|
||||
): Promise<FunctionErrors> {
|
||||
validateNonNegativeInteger(options?.limit, "limit");
|
||||
return await this.inner.functionErrors(options);
|
||||
}
|
||||
|
||||
async refreshMaterializedView(
|
||||
full?: boolean,
|
||||
sourceVersion?: number,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-darwin-arm64",
|
||||
"version": "0.40.0-beta.1",
|
||||
"version": "0.40.0-beta.3",
|
||||
"os": ["darwin"],
|
||||
"cpu": ["arm64"],
|
||||
"main": "lancedb.darwin-arm64.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
||||
"version": "0.40.0-beta.1",
|
||||
"version": "0.40.0-beta.3",
|
||||
"os": ["linux"],
|
||||
"cpu": ["arm64"],
|
||||
"main": "lancedb.linux-arm64-gnu.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-arm64-musl",
|
||||
"version": "0.40.0-beta.1",
|
||||
"version": "0.40.0-beta.3",
|
||||
"os": ["linux"],
|
||||
"cpu": ["arm64"],
|
||||
"main": "lancedb.linux-arm64-musl.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-x64-gnu",
|
||||
"version": "0.40.0-beta.1",
|
||||
"version": "0.40.0-beta.3",
|
||||
"os": ["linux"],
|
||||
"cpu": ["x64"],
|
||||
"main": "lancedb.linux-x64-gnu.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-x64-musl",
|
||||
"version": "0.40.0-beta.1",
|
||||
"version": "0.40.0-beta.3",
|
||||
"os": ["linux"],
|
||||
"cpu": ["x64"],
|
||||
"main": "lancedb.linux-x64-musl.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-win32-arm64-msvc",
|
||||
"version": "0.40.0-beta.1",
|
||||
"version": "0.40.0-beta.3",
|
||||
"os": [
|
||||
"win32"
|
||||
],
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-win32-x64-msvc",
|
||||
"version": "0.40.0-beta.1",
|
||||
"version": "0.40.0-beta.3",
|
||||
"os": ["win32"],
|
||||
"cpu": ["x64"],
|
||||
"main": "lancedb.win32-x64-msvc.node",
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@
|
||||
"ann"
|
||||
],
|
||||
"private": false,
|
||||
"version": "0.40.0-beta.1",
|
||||
"version": "0.40.0-beta.3",
|
||||
"main": "dist/index.js",
|
||||
"exports": {
|
||||
".": "./dist/index.js",
|
||||
|
||||
+118
-1
@@ -427,6 +427,32 @@ impl Table {
|
||||
Ok(crate::job::Job::new(job))
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn function_errors(
|
||||
&self,
|
||||
options: Option<FunctionErrorsOptions>,
|
||||
) -> napi::Result<FunctionErrors> {
|
||||
let options = options.unwrap_or_default();
|
||||
let limit = options
|
||||
.limit
|
||||
.map(|limit| {
|
||||
usize::try_from(limit)
|
||||
.map_err(|_| napi::Error::from_reason("limit must be a non-negative integer"))
|
||||
})
|
||||
.transpose()?;
|
||||
let request = lancedb::function::FunctionErrorsRequest {
|
||||
job_id: options.job_id,
|
||||
column: options.column,
|
||||
limit,
|
||||
};
|
||||
let errors = self
|
||||
.inner_ref()?
|
||||
.function_errors(request)
|
||||
.await
|
||||
.default_error()?;
|
||||
Ok(errors.into())
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn refresh_materialized_view(
|
||||
&self,
|
||||
@@ -453,7 +479,7 @@ impl Table {
|
||||
let view = lancedb::MaterializedView::from_table(inner)
|
||||
.await
|
||||
.default_error()?;
|
||||
serde_json::to_string(view.definition()).map_err(|err| {
|
||||
view.definition().to_json().map_err(|err| {
|
||||
napi::Error::from_reason(format!(
|
||||
"failed to serialize materialized-view definition: {err}"
|
||||
))
|
||||
@@ -1476,6 +1502,97 @@ pub struct RefreshColumnResult {
|
||||
pub version: i64,
|
||||
}
|
||||
|
||||
/// Which per-row Function errors to list; every filter is optional.
|
||||
#[napi(object)]
|
||||
#[derive(Clone, Debug, Default)]
|
||||
pub struct FunctionErrorsOptions {
|
||||
/// Only errors recorded by this job.
|
||||
pub job_id: Option<String>,
|
||||
/// Only errors on this column.
|
||||
pub column: Option<String>,
|
||||
/// At most this many records (server default 10000, cap 100000).
|
||||
pub limit: Option<i64>,
|
||||
}
|
||||
|
||||
/// One row a Function refresh skipped, as the server recorded it.
|
||||
#[napi(object)]
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct FunctionErrorRecord {
|
||||
pub job_id: String,
|
||||
pub fragment_id: i64,
|
||||
/// The row's offset within the fragment; absent when the fragment's
|
||||
/// detail was capped.
|
||||
pub row_offset: Option<i64>,
|
||||
pub column: String,
|
||||
pub function: String,
|
||||
pub function_version: String,
|
||||
pub table_version: i64,
|
||||
pub error_type: String,
|
||||
pub error_message: String,
|
||||
pub created_at_millis: i64,
|
||||
}
|
||||
|
||||
impl From<lancedb::function::FunctionErrorRecord> for FunctionErrorRecord {
|
||||
fn from(record: lancedb::function::FunctionErrorRecord) -> Self {
|
||||
Self {
|
||||
job_id: record.job_id,
|
||||
fragment_id: record.fragment_id as i64,
|
||||
row_offset: record.row_offset.map(i64::from),
|
||||
column: record.column,
|
||||
function: record.function,
|
||||
function_version: record.function_version,
|
||||
table_version: record.table_version as i64,
|
||||
error_type: record.error_type,
|
||||
error_message: record.error_message,
|
||||
created_at_millis: record.created_at_millis,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A fragment whose per-row error detail was capped: `rowsSkipped` rows
|
||||
/// failed, of which only `rowsRecorded` have a record of their own.
|
||||
#[napi(object)]
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct FunctionErrorFragment {
|
||||
pub job_id: String,
|
||||
pub fragment_id: i64,
|
||||
pub rows_skipped: i64,
|
||||
pub rows_recorded: i64,
|
||||
}
|
||||
|
||||
impl From<lancedb::function::FunctionErrorFragment> for FunctionErrorFragment {
|
||||
fn from(fragment: lancedb::function::FunctionErrorFragment) -> Self {
|
||||
Self {
|
||||
job_id: fragment.job_id,
|
||||
fragment_id: fragment.fragment_id as i64,
|
||||
rows_skipped: fragment.rows_skipped as i64,
|
||||
rows_recorded: fragment.rows_recorded as i64,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A table's per-row Function errors.
|
||||
#[napi(object)]
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct FunctionErrors {
|
||||
/// The recorded rows, newest job first.
|
||||
pub records: Vec<FunctionErrorRecord>,
|
||||
/// Fragments whose detail was capped.
|
||||
pub fragments: Vec<FunctionErrorFragment>,
|
||||
/// Whether the listing stopped at its limit.
|
||||
pub truncated: bool,
|
||||
}
|
||||
|
||||
impl From<lancedb::function::FunctionErrors> for FunctionErrors {
|
||||
fn from(errors: lancedb::function::FunctionErrors) -> Self {
|
||||
Self {
|
||||
records: errors.records.into_iter().map(Into::into).collect(),
|
||||
fragments: errors.fragments.into_iter().map(Into::into).collect(),
|
||||
truncated: errors.truncated,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[napi(object)]
|
||||
pub struct RefreshMaterializedViewResult {
|
||||
/// How the view was brought up to date: "rebuild", "incremental" or "no_op".
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "lancedb-python"
|
||||
version = "0.40.0-beta.1"
|
||||
version = "0.40.0-beta.3"
|
||||
publish = false
|
||||
edition.workspace = true
|
||||
description = "Python bindings for LanceDB"
|
||||
|
||||
@@ -126,6 +126,7 @@ include = [
|
||||
"python/lancedb/rerankers/util.py",
|
||||
"python/lancedb/rerankers/__init__.py",
|
||||
"python/lancedb/rerankers/voyageai.py",
|
||||
"python/lancedb/rerankers/typesafe.py",
|
||||
"python/lancedb/rerankers/jinaai.py",
|
||||
"python/lancedb/rerankers/openai.py",
|
||||
"python/lancedb/rerankers/cross_encoder.py",
|
||||
|
||||
@@ -12,6 +12,9 @@ __version__ = importlib.metadata.version("lancedb")
|
||||
|
||||
from ._lancedb import connect as lancedb_connect
|
||||
from ._lancedb import FtsToken
|
||||
from ._lancedb import FunctionErrorFragment as FunctionErrorFragment
|
||||
from ._lancedb import FunctionErrorRecord as FunctionErrorRecord
|
||||
from ._lancedb import FunctionErrors as FunctionErrors
|
||||
from ._lancedb import LsmWriteSpec
|
||||
from ._lancedb import tokenize as _tokenize
|
||||
from .common import URI, sanitize_uri
|
||||
|
||||
@@ -461,6 +461,12 @@ class Table:
|
||||
) -> AddColumnsResult: ...
|
||||
async def refresh_column(self, column: str) -> RefreshColumnResult: ...
|
||||
async def refresh_column_async(self, column: str) -> Job: ...
|
||||
async def function_errors(
|
||||
self,
|
||||
job_id: Optional[str] = None,
|
||||
column: Optional[str] = None,
|
||||
limit: Optional[int] = None,
|
||||
) -> FunctionErrors: ...
|
||||
async def refresh_materialized_view(
|
||||
self, full: bool = False, source_version: Optional[int] = None
|
||||
) -> RefreshMaterializedViewResult: ...
|
||||
@@ -821,6 +827,29 @@ class RefreshColumnResult:
|
||||
rows_filled: int
|
||||
version: int
|
||||
|
||||
class FunctionErrorRecord:
|
||||
job_id: str
|
||||
fragment_id: int
|
||||
row_offset: Optional[int]
|
||||
column: str
|
||||
function: str
|
||||
function_version: str
|
||||
table_version: int
|
||||
error_type: str
|
||||
error_message: str
|
||||
created_at_millis: int
|
||||
|
||||
class FunctionErrorFragment:
|
||||
job_id: str
|
||||
fragment_id: int
|
||||
rows_skipped: int
|
||||
rows_recorded: int
|
||||
|
||||
class FunctionErrors:
|
||||
records: list[FunctionErrorRecord]
|
||||
fragments: list[FunctionErrorFragment]
|
||||
truncated: bool
|
||||
|
||||
class RefreshMaterializedViewResult:
|
||||
@staticmethod
|
||||
def from_json(value: str) -> RefreshMaterializedViewResult: ...
|
||||
|
||||
@@ -7,6 +7,7 @@ import io
|
||||
import os
|
||||
import urllib.parse as urlparse
|
||||
from typing import TYPE_CHECKING, List, Union
|
||||
from urllib.request import url2pathname
|
||||
|
||||
import numpy as np
|
||||
import pyarrow as pa
|
||||
@@ -154,7 +155,7 @@ class OpenClipEmbeddings(EmbeddingFunction):
|
||||
parsed = urlparse.urlparse(image)
|
||||
# TODO handle drive letter on windows.
|
||||
if parsed.scheme == "file":
|
||||
return PIL_Image.open(parsed.path)
|
||||
return PIL_Image.open(url2pathname(parsed.path))
|
||||
elif parsed.scheme == "":
|
||||
return PIL_Image.open(image if os.name == "nt" else parsed.path)
|
||||
elif parsed.scheme.startswith("http"):
|
||||
|
||||
@@ -7,7 +7,7 @@ maintained by refresh. See ``DBConnection.create_materialized_view``."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass, field
|
||||
from dataclasses import dataclass
|
||||
from typing import TYPE_CHECKING, Dict, List, Optional, Sequence, Tuple, Union
|
||||
|
||||
from .background_loop import LOOP
|
||||
@@ -29,22 +29,24 @@ SelectArg = Union[
|
||||
]
|
||||
|
||||
|
||||
DEFINITION_FORMAT = 1
|
||||
"""The stored layout this version reads: ``{"format": 1, "query": "<SQL>"}``.
|
||||
A ``kind`` key beside it is for readers older than the format number."""
|
||||
|
||||
|
||||
@dataclass
|
||||
class MaterializedViewDefinition:
|
||||
"""The query that defines a materialized view."""
|
||||
"""The query that defines a materialized view, as stored::
|
||||
|
||||
source_table: str
|
||||
"""Name of the source table, in the same database as the view."""
|
||||
projections: List[Tuple[str, str]]
|
||||
"""``(output column, SQL expression)`` pairs, in view schema order."""
|
||||
filter: Optional[str] = None
|
||||
"""SQL predicate selecting the source rows the view holds."""
|
||||
limit: Optional[int] = None
|
||||
"""Cap on the number of rows the view holds."""
|
||||
inputs: List[str] = field(default_factory=list)
|
||||
"""Source columns the projections and filter read."""
|
||||
source_namespace: List[str] = field(default_factory=list)
|
||||
"""Namespace holding the source table; empty is the root namespace."""
|
||||
SELECT columns
|
||||
FROM [ns.]table [, function(args) AS alias | , UNNEST(column) AS alias]
|
||||
[WHERE predicate] [LIMIT n]
|
||||
|
||||
A Function in ``FROM`` position yields one row per element it returns.
|
||||
"""
|
||||
|
||||
query: str
|
||||
"""The defining query, in the canonical spelling the server stores."""
|
||||
|
||||
|
||||
def _definition_from_schema(
|
||||
@@ -54,38 +56,64 @@ def _definition_from_schema(
|
||||
raw = metadata.get(DEFINITION_META_KEY)
|
||||
if raw is None:
|
||||
raise ValueError(f"Table '{name}' is not a materialized view")
|
||||
value = json.loads(raw)
|
||||
return _definition_from_value(json.loads(raw), name)
|
||||
|
||||
|
||||
def _definition_from_json(raw: str, name: str = "") -> MaterializedViewDefinition:
|
||||
"""Parse the definition native code hands over, in its stored layout."""
|
||||
return _definition_from_value(json.loads(raw), name)
|
||||
|
||||
|
||||
def _definition_from_value(value: dict, name: str) -> MaterializedViewDefinition:
|
||||
fmt = value.get("format")
|
||||
if fmt is not None:
|
||||
# A newer writer's layout is reported, never guessed at.
|
||||
if not isinstance(fmt, int) or fmt > DEFINITION_FORMAT:
|
||||
raise NotImplementedError(
|
||||
f"materialized view '{name}' is stored in format {fmt}, which "
|
||||
"this version of lancedb cannot refresh"
|
||||
)
|
||||
return MaterializedViewDefinition(query=value["query"])
|
||||
# The structured layout written before the format number.
|
||||
kind = value.get("kind")
|
||||
# "namespaced_select" keeps older readers from resolving the source at root.
|
||||
if kind not in ("select", "namespaced_select"):
|
||||
raise NotImplementedError(
|
||||
f"materialized view '{name}' is defined by '{kind}', which this "
|
||||
"version of lancedb cannot refresh"
|
||||
f"materialized view '{name}' is stored in format kind '{kind}', "
|
||||
"which this version of lancedb cannot refresh"
|
||||
)
|
||||
return MaterializedViewDefinition(
|
||||
source_table=value["source_table"],
|
||||
projections=[
|
||||
(p["output"], p["expression"]) for p in value.get("projections", [])
|
||||
],
|
||||
filter=value.get("filter"),
|
||||
limit=value.get("limit"),
|
||||
inputs=value.get("inputs", []),
|
||||
source_namespace=value.get("source_namespace", []),
|
||||
)
|
||||
return MaterializedViewDefinition(query=_legacy_query(value))
|
||||
|
||||
|
||||
def _definition_from_json(raw: str) -> MaterializedViewDefinition:
|
||||
value = json.loads(raw)
|
||||
return MaterializedViewDefinition(
|
||||
source_table=value["source_table"],
|
||||
projections=[
|
||||
(p["output"], p["expression"]) for p in value.get("projections", [])
|
||||
],
|
||||
filter=value.get("filter"),
|
||||
limit=value.get("limit"),
|
||||
inputs=value.get("inputs", []),
|
||||
source_namespace=value.get("source_namespace", []),
|
||||
def _legacy_ident(name: str) -> str:
|
||||
if name and all(c == "_" or c.islower() or c.isdigit() for c in name):
|
||||
return name
|
||||
return _quote_identifier(name)
|
||||
|
||||
|
||||
def _legacy_query(value: dict) -> str:
|
||||
"""Render the pre-format structured layout as the query it described."""
|
||||
projections = value.get("projections", [])
|
||||
if projections:
|
||||
items = []
|
||||
for p in projections:
|
||||
output, expression = p["output"], p["expression"]
|
||||
if expression in (output, _quote_identifier(output)):
|
||||
items.append(expression)
|
||||
else:
|
||||
items.append(f"{expression} AS {_legacy_ident(output)}")
|
||||
columns = ", ".join(items)
|
||||
else:
|
||||
columns = "*"
|
||||
table = ".".join(
|
||||
_legacy_ident(part)
|
||||
for part in [*value.get("source_namespace", []), value["source_table"]]
|
||||
)
|
||||
query = f"SELECT {columns} FROM {table}"
|
||||
if value.get("filter") is not None:
|
||||
query += f" WHERE {value['filter']}"
|
||||
if value.get("limit") is not None:
|
||||
query += f" LIMIT {value['limit']}"
|
||||
return query
|
||||
|
||||
|
||||
def _quote_identifier(name: str) -> str:
|
||||
|
||||
@@ -996,6 +996,16 @@ class RemoteTable(Table):
|
||||
def refresh_column_async(self, column: str) -> Job[RefreshColumnResult]:
|
||||
return Job(LOOP.run(self._table.refresh_column_async(column)))
|
||||
|
||||
def function_errors(
|
||||
self,
|
||||
job_id: Optional[str] = None,
|
||||
column: Optional[str] = None,
|
||||
limit: Optional[int] = None,
|
||||
):
|
||||
return LOOP.run(
|
||||
self._table.function_errors(job_id=job_id, column=column, limit=limit)
|
||||
)
|
||||
|
||||
def alter_columns(
|
||||
self, *alterations: Iterable[Dict[str, str]]
|
||||
) -> AlterColumnsResult:
|
||||
|
||||
@@ -11,6 +11,7 @@ from .jinaai import JinaReranker
|
||||
from .rrf import RRFReranker
|
||||
from .mrr import MRRReranker
|
||||
from .answerdotai import AnswerdotaiRerankers
|
||||
from .typesafe import TypeSafeReranker
|
||||
from .voyageai import VoyageAIReranker
|
||||
from .watsonx import WatsonxReranker
|
||||
|
||||
@@ -27,6 +28,7 @@ __all__ = [
|
||||
"JinaReranker",
|
||||
"RRFReranker",
|
||||
"AnswerdotaiRerankers",
|
||||
"TypeSafeReranker",
|
||||
"VoyageAIReranker",
|
||||
"MRRReranker",
|
||||
"WatsonxReranker",
|
||||
|
||||
@@ -0,0 +1,159 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||
|
||||
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from functools import cached_property
|
||||
from typing import Any, Dict, Mapping, Optional
|
||||
|
||||
import pyarrow as pa
|
||||
|
||||
from ..util import attempt_import_or_raise
|
||||
from .base import Reranker
|
||||
|
||||
DEFAULT_INSTRUCTIONS = (
|
||||
"Does the document contain information that answers or directly addresses "
|
||||
"the query?"
|
||||
)
|
||||
DEFAULT_CRITERIA = {
|
||||
"true": "The document answers the query or states facts the query asks about.",
|
||||
"false": "The document is off-topic, or only shares keywords or a general "
|
||||
"subject with the query without addressing it.",
|
||||
}
|
||||
# The key TypeSafe returns the answer under. It is not sent to the model.
|
||||
_QUESTION_ID = "relevance"
|
||||
|
||||
|
||||
class TypeSafeReranker(Reranker):
|
||||
"""
|
||||
Reranks the results using the TypeSafe System One API.
|
||||
https://docs.typesafe.ai/cookbooks/rerank_typesafe
|
||||
|
||||
Each result is scored independently: TypeSafe reads the query together with
|
||||
the result's ``column`` value and answers a yes/no ("noul") question about
|
||||
whether the document is relevant. The probability of "yes", between 0 and 1,
|
||||
becomes the ``_relevance_score``. Because every score is an absolute
|
||||
probability rather than a position in the list, scores are comparable
|
||||
across queries and can be thresholded. They are model estimates, though,
|
||||
and can vary slightly between identical calls, so results with close scores
|
||||
may swap places when the same search is repeated.
|
||||
|
||||
One request is sent per result, up to ``max_concurrency`` at a time.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
model_name : str, default "jev-latest"
|
||||
The TypeSafe model to use.
|
||||
column : str, default "text"
|
||||
The name of the column holding the document text to score.
|
||||
instructions : str, optional
|
||||
The yes/no question asked about each query and document pair. The state
|
||||
TypeSafe reads is ``{"query": <query>, "document": <column value>}``.
|
||||
Defaults to a generic relevance question.
|
||||
criteria : Mapping[str, str], optional
|
||||
What a "yes" and a "no" mean, as a mapping with the keys ``"true"`` and
|
||||
``"false"``. Domain-specific criteria usually rank better than the
|
||||
generic default. Pass an empty mapping to send no criteria.
|
||||
return_score : str, default "relevance"
|
||||
Options are "relevance" or "all". If "all", keeps the vector and FTS
|
||||
scores alongside the relevance score.
|
||||
api_key : str, optional
|
||||
The API key to use. If None, the TypeSafe SDK reads the
|
||||
``TYPESAFE_API_KEY`` environment variable.
|
||||
max_concurrency : int, default 8
|
||||
The maximum number of TypeSafe requests in flight for one rerank call.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
model_name: str = "jev-latest",
|
||||
column: str = "text",
|
||||
instructions: Optional[str] = None,
|
||||
criteria: Optional[Mapping[str, str]] = None,
|
||||
return_score: str = "relevance",
|
||||
api_key: Optional[str] = None,
|
||||
max_concurrency: int = 8,
|
||||
):
|
||||
super().__init__(return_score)
|
||||
if max_concurrency < 1:
|
||||
raise ValueError("max_concurrency must be at least 1")
|
||||
criteria = DEFAULT_CRITERIA if criteria is None else dict(criteria)
|
||||
unknown = set(criteria) - {"true", "false"}
|
||||
if unknown:
|
||||
raise ValueError(
|
||||
f"criteria keys must be 'true' or 'false', got {sorted(unknown)}"
|
||||
)
|
||||
self.model_name = model_name
|
||||
self.column = column
|
||||
self.instructions = instructions or DEFAULT_INSTRUCTIONS
|
||||
self.criteria = criteria
|
||||
self.api_key = api_key
|
||||
self.max_concurrency = max_concurrency
|
||||
|
||||
def __str__(self):
|
||||
return f"TypeSafeReranker(model_name={self.model_name})"
|
||||
|
||||
@cached_property
|
||||
def _client(self):
|
||||
typesafe_sdk = attempt_import_or_raise("typesafe_sdk", "typesafe-sdk")
|
||||
return typesafe_sdk.TypeSafeClient(api_key=self.api_key)
|
||||
|
||||
@cached_property
|
||||
def _question(self) -> Dict[str, Any]:
|
||||
question: Dict[str, Any] = {"type": "noul", "instructions": self.instructions}
|
||||
if self.criteria:
|
||||
question["criteria"] = self.criteria
|
||||
return question
|
||||
|
||||
def _score(self, query: str, document: Optional[str]) -> float:
|
||||
if document is None:
|
||||
return 0.0
|
||||
response = self._client.system_one(
|
||||
state={"query": query, "document": document},
|
||||
questions={_QUESTION_ID: self._question},
|
||||
model=self.model_name,
|
||||
)
|
||||
return response.answers[_QUESTION_ID].noul
|
||||
|
||||
def _rerank(self, result_set: pa.Table, query: str) -> pa.Table:
|
||||
result_set = self._handle_empty_results(result_set)
|
||||
if len(result_set) == 0:
|
||||
return result_set
|
||||
docs = result_set[self.column].to_pylist()
|
||||
# Rerankers are also called synchronously from inside the async query
|
||||
# APIs, so the requests run on threads rather than on an event loop.
|
||||
with ThreadPoolExecutor(
|
||||
max_workers=min(self.max_concurrency, len(docs))
|
||||
) as pool:
|
||||
scores = list(pool.map(lambda doc: self._score(query, doc), docs))
|
||||
result_set = result_set.append_column(
|
||||
"_relevance_score", pa.array(scores, type=pa.float32())
|
||||
)
|
||||
return result_set.sort_by([("_relevance_score", "descending")])
|
||||
|
||||
def rerank_hybrid(
|
||||
self,
|
||||
query: str,
|
||||
vector_results: pa.Table,
|
||||
fts_results: pa.Table,
|
||||
):
|
||||
if self.score == "all":
|
||||
combined_results = self._merge_and_keep_scores(vector_results, fts_results)
|
||||
else:
|
||||
combined_results = self.merge_results(vector_results, fts_results)
|
||||
combined_results = self._rerank(combined_results, query)
|
||||
if self.score == "relevance":
|
||||
combined_results = self._keep_relevance_score(combined_results)
|
||||
return combined_results
|
||||
|
||||
def rerank_vector(self, query: str, vector_results: pa.Table):
|
||||
vector_results = self._rerank(vector_results, query)
|
||||
if self.score == "relevance":
|
||||
vector_results = vector_results.drop_columns(["_distance"])
|
||||
return vector_results
|
||||
|
||||
def rerank_fts(self, query: str, fts_results: pa.Table):
|
||||
fts_results = self._rerank(fts_results, query)
|
||||
if self.score == "relevance":
|
||||
fts_results = fts_results.drop_columns(["_score"])
|
||||
return fts_results
|
||||
@@ -187,6 +187,7 @@ if TYPE_CHECKING:
|
||||
CompactionStats,
|
||||
Tag,
|
||||
AddColumnsResult,
|
||||
FunctionErrors,
|
||||
RefreshColumnResult,
|
||||
AddResult,
|
||||
AlterColumnsResult,
|
||||
@@ -2400,6 +2401,47 @@ class Table(ABC):
|
||||
'finished'
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
def function_errors(
|
||||
self,
|
||||
job_id: Optional[str] = None,
|
||||
column: Optional[str] = None,
|
||||
limit: Optional[int] = None,
|
||||
) -> "FunctionErrors":
|
||||
"""
|
||||
The per-row errors Function refreshes recorded on this table.
|
||||
|
||||
A refresh running under a skip policy records each row it skipped
|
||||
with the input that failed and the error. This lists those records,
|
||||
newest job first, plus a summary for any fragment whose per-row
|
||||
detail was capped. LanceDB Cloud and Enterprise only; reading errors
|
||||
needs read access to the table, since a message carries the value
|
||||
that failed.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
job_id: str, optional
|
||||
Only errors recorded by this job.
|
||||
column: str, optional
|
||||
Only errors on this column.
|
||||
limit: int, optional
|
||||
At most this many records (server default 10000, cap 100000).
|
||||
|
||||
Returns
|
||||
-------
|
||||
FunctionErrors
|
||||
``records``, ``fragments`` and ``truncated``, the last saying
|
||||
whether the listing stopped at its limit.
|
||||
|
||||
Examples
|
||||
--------
|
||||
>>> errors = table.function_errors(column="embedding") # doctest: +SKIP
|
||||
>>> for record in errors.records: # doctest: +SKIP
|
||||
... print(record.job_id, record.row_offset, record.error_message)
|
||||
>>> if errors.truncated: # doctest: +SKIP
|
||||
... print("listing stopped at the limit")
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
def alter_columns(self, *alterations: Iterable[Dict[str, str]]):
|
||||
"""
|
||||
@@ -4461,6 +4503,18 @@ class LanceTable(Table):
|
||||
"""
|
||||
return Job(LOOP.run(self._table.refresh_column_async(column)))
|
||||
|
||||
def function_errors(
|
||||
self,
|
||||
job_id: Optional[str] = None,
|
||||
column: Optional[str] = None,
|
||||
limit: Optional[int] = None,
|
||||
) -> "FunctionErrors":
|
||||
"""The per-row errors Function refreshes recorded on this table. See
|
||||
[`Table.function_errors`][lancedb.table.Table.function_errors]."""
|
||||
return LOOP.run(
|
||||
self._table.function_errors(job_id=job_id, column=column, limit=limit)
|
||||
)
|
||||
|
||||
def alter_columns(
|
||||
self, *alterations: Iterable[Dict[str, str]]
|
||||
) -> AlterColumnsResult:
|
||||
@@ -6527,6 +6581,49 @@ class AsyncTable:
|
||||
"""
|
||||
return await self._inner.refresh_column(column)
|
||||
|
||||
async def function_errors(
|
||||
self,
|
||||
job_id: Optional[str] = None,
|
||||
column: Optional[str] = None,
|
||||
limit: Optional[int] = None,
|
||||
) -> "FunctionErrors":
|
||||
"""
|
||||
The per-row errors Function refreshes recorded on this table.
|
||||
|
||||
A refresh running under a skip policy records each row it skipped
|
||||
with the input that failed and the error. This lists those records,
|
||||
newest job first, plus a summary for any fragment whose per-row
|
||||
detail was capped. LanceDB Cloud and Enterprise only; reading errors
|
||||
needs read access to the table, since a message carries the value
|
||||
that failed.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
job_id: str, optional
|
||||
Only errors recorded by this job.
|
||||
column: str, optional
|
||||
Only errors on this column.
|
||||
limit: int, optional
|
||||
At most this many records (server default 10000, cap 100000).
|
||||
|
||||
Returns
|
||||
-------
|
||||
FunctionErrors
|
||||
``records``, ``fragments`` and ``truncated``, the last saying
|
||||
whether the listing stopped at its limit.
|
||||
|
||||
Examples
|
||||
--------
|
||||
>>> errors = await table.function_errors(column="embedding") # doctest: +SKIP
|
||||
>>> for record in errors.records: # doctest: +SKIP
|
||||
... print(record.job_id, record.row_offset, record.error_message)
|
||||
>>> if errors.truncated: # doctest: +SKIP
|
||||
... print("listing stopped at the limit")
|
||||
"""
|
||||
return await self._inner.function_errors(
|
||||
job_id=job_id, column=column, limit=limit
|
||||
)
|
||||
|
||||
async def refresh_column_async(
|
||||
self, column: str
|
||||
) -> AsyncJob[RefreshColumnJobResult]:
|
||||
|
||||
@@ -633,6 +633,18 @@ def test_url_retrieve_downloads_image():
|
||||
assert img.size[0] > 0 and img.size[1] > 0
|
||||
|
||||
|
||||
def test_open_clip_opens_percent_encoded_file_uri(tmp_path):
|
||||
"""OpenCLIP should decode local file URIs before opening them."""
|
||||
Image = pytest.importorskip("PIL.Image")
|
||||
from lancedb.embeddings.open_clip import OpenClipEmbeddings
|
||||
|
||||
image_path = tmp_path / "test image.png"
|
||||
Image.new("RGB", (4, 4), color="red").save(image_path, format="PNG")
|
||||
|
||||
with OpenClipEmbeddings._to_pil(None, image_path.as_uri()) as image:
|
||||
assert image.size == (4, 4)
|
||||
|
||||
|
||||
def test_jina_generate_image_input_dict_local_path(tmp_path):
|
||||
"""
|
||||
JinaEmbeddings._generate_image_input_dict must accept a local image path
|
||||
|
||||
@@ -217,10 +217,7 @@ def test_definition_round_trips(tmp_path):
|
||||
|
||||
view = db.open_materialized_view("adults")
|
||||
assert view.definition == MaterializedViewDefinition(
|
||||
source_table="people",
|
||||
projections=[("name", "`name`"), ("age", "`age`")],
|
||||
filter="age >= 18",
|
||||
inputs=["age", "name"],
|
||||
query="SELECT name, age FROM people WHERE age >= 18"
|
||||
)
|
||||
|
||||
|
||||
@@ -305,7 +302,7 @@ async def test_async_create_refresh_and_open(tmp_path):
|
||||
|
||||
reopened = await db.open_materialized_view("shouts")
|
||||
definition = await reopened.definition()
|
||||
assert definition.projections == [("shout", "upper(name)")]
|
||||
assert definition.query == "SELECT upper(name) AS shout FROM people"
|
||||
assert await db.list_materialized_views() == ["shouts"]
|
||||
|
||||
|
||||
@@ -423,7 +420,7 @@ def test_namespace_connection_materialized_views(tmp_path):
|
||||
assert db.list_materialized_views() == ["adults"]
|
||||
|
||||
reopened = db.open_materialized_view("adults")
|
||||
assert reopened.definition.source_table == "people"
|
||||
assert reopened.definition.query.startswith("SELECT name, age FROM ")
|
||||
with pytest.raises(ValueError, match="not a materialized view"):
|
||||
db.open_materialized_view("people")
|
||||
|
||||
@@ -458,7 +455,7 @@ async def test_async_namespace_connection_materialized_views(tmp_path):
|
||||
assert await db.list_materialized_views() == ["adults"]
|
||||
|
||||
reopened = await db.open_materialized_view("adults")
|
||||
assert (await reopened.definition()).source_table == "people"
|
||||
assert (await reopened.definition()).query.startswith("SELECT name, age FROM ")
|
||||
|
||||
# The view's table came through the namespace, not straight from the
|
||||
# inner connection: a bare inner table carries no namespace context, so
|
||||
@@ -484,36 +481,49 @@ async def test_async_namespace_connection_materialized_views(tmp_path):
|
||||
assert await db.list_materialized_views() == []
|
||||
|
||||
|
||||
def test_namespaced_select_kind_is_read_and_unknown_kinds_are_refused():
|
||||
def test_stored_queries_and_legacy_layouts_are_read():
|
||||
import json
|
||||
|
||||
import pyarrow as pa
|
||||
|
||||
from lancedb.materialized_view import _definition_from_schema
|
||||
|
||||
def schema_with(definition: dict) -> pa.Schema:
|
||||
return pa.schema([pa.field("id", pa.int32())]).with_metadata(
|
||||
def read(definition: dict) -> MaterializedViewDefinition:
|
||||
schema = pa.schema([pa.field("id", pa.int32())]).with_metadata(
|
||||
{b"mv.definition": json.dumps(definition).encode()}
|
||||
)
|
||||
return _definition_from_schema(schema, "v")
|
||||
|
||||
# "namespaced_select" is the namespaced form of "select": same shape,
|
||||
# a separate kind so readers that predate it refuse instead of
|
||||
# resolving the source at the root.
|
||||
definition = _definition_from_schema(
|
||||
schema_with(
|
||||
query = "SELECT id, c.chunk FROM ns.docs, UNNEST(chunks) AS c WHERE id > 1"
|
||||
assert read({"format": 1, "query": query}).query == query
|
||||
|
||||
# The structured layout written before the format number reads as the
|
||||
# query it described, under either of its kind tags.
|
||||
assert (
|
||||
read(
|
||||
{
|
||||
"kind": "namespaced_select",
|
||||
"source_table": "people",
|
||||
"source_namespace": ["ns"],
|
||||
"projections": [{"output": "name", "expression": "name"}],
|
||||
"projections": [
|
||||
{"output": "name", "expression": "`name`"},
|
||||
{"output": "Shout", "expression": "upper(name)"},
|
||||
],
|
||||
"filter": "age >= 18",
|
||||
"limit": 10,
|
||||
}
|
||||
),
|
||||
"v",
|
||||
).query
|
||||
== "SELECT `name`, upper(name) AS `Shout` FROM ns.people "
|
||||
"WHERE age >= 18 LIMIT 10"
|
||||
)
|
||||
assert read({"kind": "select", "source_table": "people"}).query == (
|
||||
"SELECT * FROM people"
|
||||
)
|
||||
assert definition.source_table == "people"
|
||||
assert definition.source_namespace == ["ns"]
|
||||
|
||||
with pytest.raises(NotImplementedError, match="cannot refresh"):
|
||||
_definition_from_schema(
|
||||
schema_with({"kind": "select_v3", "source_table": "people"}), "v"
|
||||
)
|
||||
# A newer writer's layout is reported, never guessed at.
|
||||
for newer in (
|
||||
{"format": 2, "query": query},
|
||||
{"kind": "select_v3", "source_table": "people"},
|
||||
):
|
||||
with pytest.raises(NotImplementedError, match="cannot refresh"):
|
||||
read(newer)
|
||||
|
||||
@@ -982,6 +982,71 @@ def test_remote_refresh_async_returns_typed_terminal_result():
|
||||
assert result.version == 8
|
||||
|
||||
|
||||
def test_remote_function_errors_lists_the_rows_a_refresh_skipped():
|
||||
listing = {
|
||||
"records": [
|
||||
{
|
||||
"job_id": "j-7",
|
||||
"fragment_id": 3,
|
||||
"row_offset": 9,
|
||||
"column": "embedding",
|
||||
"function": "embed",
|
||||
"function_version": "2",
|
||||
"table_version": 11,
|
||||
"error_type": "ValueError",
|
||||
"error_message": "bad input 'x'",
|
||||
"created_at_millis": 1700000000000,
|
||||
}
|
||||
],
|
||||
"fragments": [
|
||||
{
|
||||
"job_id": "j-7",
|
||||
"fragment_id": 4,
|
||||
"rows_skipped": 500,
|
||||
"rows_recorded": 100,
|
||||
}
|
||||
],
|
||||
"truncated": True,
|
||||
}
|
||||
bodies = []
|
||||
|
||||
def handler(request):
|
||||
content_len = int(request.headers.get("Content-Length", 0))
|
||||
body = request.rfile.read(content_len) if content_len > 0 else b""
|
||||
if request.path == "/v1/table/test/errors":
|
||||
bodies.append(json.loads(body))
|
||||
request.send_response(200)
|
||||
request.send_header("Content-Type", "application/json")
|
||||
request.end_headers()
|
||||
request.wfile.write(json.dumps(listing).encode())
|
||||
elif request.path == "/v1/table/test/describe/":
|
||||
request.send_response(200)
|
||||
request.send_header("Content-Type", "application/json")
|
||||
request.end_headers()
|
||||
request.wfile.write(
|
||||
json.dumps({"version": 1, "schema": {"fields": []}}).encode()
|
||||
)
|
||||
else:
|
||||
request.send_response(404)
|
||||
request.end_headers()
|
||||
|
||||
with mock_lancedb_connection(handler) as db:
|
||||
table = db.open_table("test")
|
||||
errors = table.function_errors(job_id="j-7", column="embedding", limit=2)
|
||||
everything = table.function_errors()
|
||||
|
||||
assert bodies == [{"job_id": "j-7", "column": "embedding", "limit": 2}, {}]
|
||||
assert errors.truncated is True
|
||||
assert [r.error_message for r in errors.records] == ["bad input 'x'"]
|
||||
assert errors.records[0].row_offset == 9
|
||||
assert errors.records[0].function_version == "2"
|
||||
assert (errors.fragments[0].rows_skipped, errors.fragments[0].rows_recorded) == (
|
||||
500,
|
||||
100,
|
||||
)
|
||||
assert everything.truncated is True
|
||||
|
||||
|
||||
def test_remote_job_wait_raises_on_failure():
|
||||
from lancedb.exceptions import JobFailedError
|
||||
from lancedb.index import BTree
|
||||
|
||||
@@ -21,6 +21,7 @@ from lancedb.rerankers import (
|
||||
OpenaiReranker,
|
||||
JinaReranker,
|
||||
AnswerdotaiRerankers,
|
||||
TypeSafeReranker,
|
||||
VoyageAIReranker,
|
||||
MRRReranker,
|
||||
WatsonxReranker,
|
||||
@@ -101,7 +102,15 @@ def get_test_table(tmp_path):
|
||||
return table, MyTable
|
||||
|
||||
|
||||
def _run_test_reranker(reranker, table, query, query_vector, schema):
|
||||
def _run_test_reranker(
|
||||
reranker, table, query, query_vector, schema, deterministic=True
|
||||
):
|
||||
"""Exercise a reranker across search types.
|
||||
|
||||
Set ``deterministic=False`` for rerankers whose scores can vary between
|
||||
identical calls, such as remote model APIs; repeated searches are then not
|
||||
expected to return the same order.
|
||||
"""
|
||||
# Hybrid search setting
|
||||
result1 = (
|
||||
table.search(query, query_type="hybrid", vector_column_name="vector")
|
||||
@@ -113,7 +122,8 @@ def _run_test_reranker(reranker, table, query, query_vector, schema):
|
||||
.rerank(reranker=reranker)
|
||||
.to_pydantic(schema)
|
||||
)
|
||||
assert result1 == result2
|
||||
if deterministic:
|
||||
assert result1 == result2
|
||||
|
||||
query_vector = table.to_pandas()["vector"][0]
|
||||
result = (
|
||||
@@ -203,7 +213,9 @@ def _run_test_reranker(reranker, table, query, query_vector, schema):
|
||||
)
|
||||
assert len(result_deduped) <= 20
|
||||
result_arrow = reranker.rerank_multivector([rs1.to_arrow(), rs2.to_arrow()], query)
|
||||
assert len(result) == 20 and result == result_arrow
|
||||
assert len(result) == 20 and len(result_arrow) == 20
|
||||
if deterministic:
|
||||
assert result == result_arrow
|
||||
|
||||
|
||||
def _run_test_hybrid_reranker(reranker, tmp_path):
|
||||
@@ -515,6 +527,95 @@ def test_voyageai_reranker(tmp_path):
|
||||
_run_test_reranker(reranker, table, "single player experience", None, schema)
|
||||
|
||||
|
||||
class _FakeTypeSafeClient:
|
||||
"""Stands in for ``typesafe_sdk.TypeSafeClient``; scores by word overlap."""
|
||||
|
||||
def __init__(self):
|
||||
self.requests = []
|
||||
|
||||
def system_one(self, state, questions, model):
|
||||
self.requests.append((state, questions, model))
|
||||
query_words = set(state["query"].lower().split())
|
||||
doc_words = set(state["document"].lower().split())
|
||||
noul = len(query_words & doc_words) / len(query_words)
|
||||
answers = {key: type("NoulAnswer", (), {"noul": noul})() for key in questions}
|
||||
return type("SystemOneResponse", (), {"answers": answers})()
|
||||
|
||||
|
||||
def test_typesafe_reranker_with_fake_client(tmp_path):
|
||||
reranker = TypeSafeReranker(max_concurrency=4)
|
||||
reranker._client = _FakeTypeSafeClient()
|
||||
table, schema = get_test_table(tmp_path)
|
||||
_run_test_reranker(reranker, table, "single player experience", None, schema)
|
||||
|
||||
state, questions, model = reranker._client.requests[0]
|
||||
assert model == "jev-latest"
|
||||
assert set(state) == {"query", "document"}
|
||||
assert questions == {
|
||||
"relevance": {
|
||||
"type": "noul",
|
||||
"instructions": reranker.instructions,
|
||||
"criteria": reranker.criteria,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
def test_typesafe_reranker_scores_each_row():
|
||||
reranker = TypeSafeReranker(
|
||||
column="body",
|
||||
instructions="Is this about cats?",
|
||||
criteria={},
|
||||
return_score="all",
|
||||
)
|
||||
reranker._client = _FakeTypeSafeClient()
|
||||
results = pa.table(
|
||||
{
|
||||
"body": ["dogs bark", "cats purr and cats nap", None, "cats"],
|
||||
"_distance": [0.1, 0.2, 0.3, 0.4],
|
||||
}
|
||||
)
|
||||
|
||||
reranked = reranker.rerank_vector("cats nap", results)
|
||||
|
||||
assert reranked["body"].to_pylist() == [
|
||||
"cats purr and cats nap",
|
||||
"cats",
|
||||
"dogs bark",
|
||||
None,
|
||||
]
|
||||
assert reranked["_relevance_score"].to_pylist() == [1.0, 0.5, 0.0, 0.0]
|
||||
assert reranked["_distance"].to_pylist() == [0.2, 0.4, 0.1, 0.3]
|
||||
# Null documents are scored 0 without a request.
|
||||
assert len(reranker._client.requests) == 3
|
||||
assert reranker._client.requests[0][1] == {
|
||||
"relevance": {"type": "noul", "instructions": "Is this about cats?"}
|
||||
}
|
||||
|
||||
|
||||
def test_typesafe_reranker_rejects_invalid_arguments():
|
||||
with pytest.raises(ValueError, match="criteria keys"):
|
||||
TypeSafeReranker(criteria={"yes": "relevant"})
|
||||
with pytest.raises(ValueError, match="max_concurrency"):
|
||||
TypeSafeReranker(max_concurrency=0)
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
os.environ.get("TYPESAFE_API_KEY") is None, reason="TYPESAFE_API_KEY not set"
|
||||
)
|
||||
def test_typesafe_reranker(tmp_path):
|
||||
pytest.importorskip("typesafe_sdk")
|
||||
reranker = TypeSafeReranker()
|
||||
table, schema = get_test_table(tmp_path)
|
||||
_run_test_reranker(
|
||||
reranker,
|
||||
table,
|
||||
"single player experience",
|
||||
None,
|
||||
schema,
|
||||
deterministic=False,
|
||||
)
|
||||
|
||||
|
||||
def test_empty_result_reranker():
|
||||
pytest.importorskip("sentence_transformers")
|
||||
db = lancedb.connect("memory://")
|
||||
|
||||
@@ -4449,6 +4449,13 @@ async def test_computed_column_async(tmp_path):
|
||||
assert (await table.to_arrow())["tripled"].to_pylist() == [9]
|
||||
|
||||
|
||||
def test_function_errors_are_remote_only(tmp_path):
|
||||
db = lancedb.connect(tmp_path)
|
||||
table = db.create_table("t", [{"x": 1}])
|
||||
with pytest.raises(NotImplementedError, match="LanceDB Cloud and Enterprise"):
|
||||
table.function_errors()
|
||||
|
||||
|
||||
def test_refresh_column_async_returns_job(tmp_path):
|
||||
db = lancedb.connect(tmp_path)
|
||||
table = db.create_table("computed_job", [{"x": 1}, {"x": 2}])
|
||||
|
||||
+6
-2
@@ -16,8 +16,9 @@ use query::{FTSQuery, HybridQuery, Query, VectorQuery};
|
||||
use session::Session;
|
||||
use table::{
|
||||
AddColumnsResult, AddResult, AlterColumnsResult, DeleteResult, DropColumnsResult, FtsToken,
|
||||
LsmWriteSpec, MergeResult, PyBlobFile, RefreshColumnResult, RefreshMaterializedViewResult,
|
||||
Table, UpdateFieldMetadataResult, UpdateResult,
|
||||
FunctionErrorFragment, FunctionErrorRecord, FunctionErrors, LsmWriteSpec, MergeResult,
|
||||
PyBlobFile, RefreshColumnResult, RefreshMaterializedViewResult, Table,
|
||||
UpdateFieldMetadataResult, UpdateResult,
|
||||
};
|
||||
|
||||
pub mod arrow;
|
||||
@@ -84,6 +85,9 @@ pub fn _lancedb(py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_class::<RecordBatchStream>()?;
|
||||
m.add_class::<AddColumnsResult>()?;
|
||||
m.add_class::<RefreshColumnResult>()?;
|
||||
m.add_class::<FunctionErrors>()?;
|
||||
m.add_class::<FunctionErrorRecord>()?;
|
||||
m.add_class::<FunctionErrorFragment>()?;
|
||||
m.add_class::<RefreshMaterializedViewResult>()?;
|
||||
m.add_class::<AlterColumnsResult>()?;
|
||||
m.add_class::<UpdateFieldMetadataResult>()?;
|
||||
|
||||
+131
-1
@@ -573,6 +573,117 @@ impl From<lancedb::table::RefreshColumnResult> for RefreshColumnResult {
|
||||
}
|
||||
}
|
||||
|
||||
/// One row a Function refresh skipped, as the server recorded it.
|
||||
#[pyclass(module = "lancedb._lancedb", get_all, from_py_object)]
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct FunctionErrorRecord {
|
||||
pub job_id: String,
|
||||
pub fragment_id: u64,
|
||||
pub row_offset: Option<u32>,
|
||||
pub column: String,
|
||||
pub function: String,
|
||||
pub function_version: String,
|
||||
pub table_version: u64,
|
||||
pub error_type: String,
|
||||
pub error_message: String,
|
||||
pub created_at_millis: i64,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl FunctionErrorRecord {
|
||||
pub fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"FunctionErrorRecord(job_id={:?}, fragment_id={}, row_offset={:?}, column={:?}, \
|
||||
error_type={:?}, error_message={:?})",
|
||||
self.job_id,
|
||||
self.fragment_id,
|
||||
self.row_offset,
|
||||
self.column,
|
||||
self.error_type,
|
||||
self.error_message
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
impl From<lancedb::function::FunctionErrorRecord> for FunctionErrorRecord {
|
||||
fn from(record: lancedb::function::FunctionErrorRecord) -> Self {
|
||||
Self {
|
||||
job_id: record.job_id,
|
||||
fragment_id: record.fragment_id,
|
||||
row_offset: record.row_offset,
|
||||
column: record.column,
|
||||
function: record.function,
|
||||
function_version: record.function_version,
|
||||
table_version: record.table_version,
|
||||
error_type: record.error_type,
|
||||
error_message: record.error_message,
|
||||
created_at_millis: record.created_at_millis,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A fragment whose per-row error detail was capped.
|
||||
#[pyclass(module = "lancedb._lancedb", get_all, from_py_object)]
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct FunctionErrorFragment {
|
||||
pub job_id: String,
|
||||
pub fragment_id: u64,
|
||||
pub rows_skipped: u64,
|
||||
pub rows_recorded: u64,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl FunctionErrorFragment {
|
||||
pub fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"FunctionErrorFragment(job_id={:?}, fragment_id={}, rows_skipped={}, rows_recorded={})",
|
||||
self.job_id, self.fragment_id, self.rows_skipped, self.rows_recorded
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
impl From<lancedb::function::FunctionErrorFragment> for FunctionErrorFragment {
|
||||
fn from(fragment: lancedb::function::FunctionErrorFragment) -> Self {
|
||||
Self {
|
||||
job_id: fragment.job_id,
|
||||
fragment_id: fragment.fragment_id,
|
||||
rows_skipped: fragment.rows_skipped,
|
||||
rows_recorded: fragment.rows_recorded,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A table's per-row Function errors.
|
||||
#[pyclass(module = "lancedb._lancedb", get_all, from_py_object)]
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct FunctionErrors {
|
||||
pub records: Vec<FunctionErrorRecord>,
|
||||
pub fragments: Vec<FunctionErrorFragment>,
|
||||
pub truncated: bool,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl FunctionErrors {
|
||||
pub fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"FunctionErrors(records={}, fragments={}, truncated={})",
|
||||
self.records.len(),
|
||||
self.fragments.len(),
|
||||
self.truncated
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
impl From<lancedb::function::FunctionErrors> for FunctionErrors {
|
||||
fn from(errors: lancedb::function::FunctionErrors) -> Self {
|
||||
Self {
|
||||
records: errors.records.into_iter().map(Into::into).collect(),
|
||||
fragments: errors.fragments.into_iter().map(Into::into).collect(),
|
||||
truncated: errors.truncated,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[pyclass(get_all, from_py_object)]
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct RefreshMaterializedViewResult {
|
||||
@@ -1772,6 +1883,25 @@ impl Table {
|
||||
})
|
||||
}
|
||||
|
||||
#[pyo3(signature = (job_id=None, column=None, limit=None))]
|
||||
pub fn function_errors(
|
||||
self_: PyRef<'_, Self>,
|
||||
job_id: Option<String>,
|
||||
column: Option<String>,
|
||||
limit: Option<usize>,
|
||||
) -> PyResult<Bound<'_, PyAny>> {
|
||||
let inner = self_.inner_ref()?.clone();
|
||||
let request = lancedb::function::FunctionErrorsRequest {
|
||||
job_id,
|
||||
column,
|
||||
limit,
|
||||
};
|
||||
future_into_py(self_.py(), async move {
|
||||
let errors = inner.function_errors(request).await.infer_error()?;
|
||||
Ok(FunctionErrors::from(errors))
|
||||
})
|
||||
}
|
||||
|
||||
#[pyo3(signature = (full=false, source_version=None))]
|
||||
pub fn refresh_materialized_view(
|
||||
self_: PyRef<'_, Self>,
|
||||
@@ -1818,7 +1948,7 @@ impl Table {
|
||||
let view = lancedb::MaterializedView::from_table(inner)
|
||||
.await
|
||||
.infer_error()?;
|
||||
serde_json::to_string(view.definition()).map_err(|err| {
|
||||
view.definition().to_json().map_err(|err| {
|
||||
PyRuntimeError::new_err(format!(
|
||||
"failed to serialize materialized-view definition: {err}"
|
||||
))
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "lancedb"
|
||||
version = "0.40.0-beta.1"
|
||||
version = "0.40.0-beta.3"
|
||||
edition.workspace = true
|
||||
description = "LanceDB: A serverless, low-latency vector database for AI applications"
|
||||
license.workspace = true
|
||||
|
||||
@@ -360,7 +360,7 @@ pub trait Database:
|
||||
continue;
|
||||
};
|
||||
let schema = table.schema().await?;
|
||||
if crate::materialized_view::materialized_view_kind(schema.metadata())?.is_some() {
|
||||
if crate::materialized_view::read_definition(schema.metadata())?.is_some() {
|
||||
names.push(name);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -731,6 +731,101 @@ pub struct RefreshColumnResult {
|
||||
pub published_version: Option<u64>,
|
||||
}
|
||||
|
||||
/// Which per-row errors [`crate::Table::function_errors`] lists. Every
|
||||
/// filter is optional; the listing is table-addressed, so with none set it
|
||||
/// covers every refresh of every column.
|
||||
#[derive(Debug, Clone, Default, PartialEq, Eq)]
|
||||
pub struct FunctionErrorsRequest {
|
||||
/// Only errors recorded by this job.
|
||||
pub job_id: Option<String>,
|
||||
/// Only errors on this column.
|
||||
pub column: Option<String>,
|
||||
/// At most this many records; the server default is 10000 and its cap
|
||||
/// 100000. [`FunctionErrors::truncated`] says whether the cap was hit.
|
||||
pub limit: Option<usize>,
|
||||
}
|
||||
|
||||
impl FunctionErrorsRequest {
|
||||
/// A request with no filter.
|
||||
pub fn new() -> Self {
|
||||
Self::default()
|
||||
}
|
||||
|
||||
/// Only errors recorded by `job_id`.
|
||||
pub fn job_id(mut self, job_id: impl Into<String>) -> Self {
|
||||
self.job_id = Some(job_id.into());
|
||||
self
|
||||
}
|
||||
|
||||
/// Only errors on `column`.
|
||||
pub fn column(mut self, column: impl Into<String>) -> Self {
|
||||
self.column = Some(column.into());
|
||||
self
|
||||
}
|
||||
|
||||
/// At most `limit` records.
|
||||
pub fn limit(mut self, limit: usize) -> Self {
|
||||
self.limit = Some(limit);
|
||||
self
|
||||
}
|
||||
}
|
||||
|
||||
/// One row a Function refresh skipped, as the server recorded it. The
|
||||
/// message carries the input that failed, which is why reading errors needs
|
||||
/// read access to the table.
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub struct FunctionErrorRecord {
|
||||
/// The refresh job that recorded the error.
|
||||
pub job_id: String,
|
||||
/// The fragment holding the row.
|
||||
pub fragment_id: u64,
|
||||
/// The row's offset within the fragment; `None` when the fragment's
|
||||
/// detail was capped and only the fragment summary remains.
|
||||
#[serde(default)]
|
||||
pub row_offset: Option<u32>,
|
||||
/// The column being computed.
|
||||
pub column: String,
|
||||
/// The Function that failed.
|
||||
pub function: String,
|
||||
/// The Function's version.
|
||||
pub function_version: String,
|
||||
/// The table version the refresh read.
|
||||
pub table_version: u64,
|
||||
/// The error's class, as the executor reported it.
|
||||
pub error_type: String,
|
||||
/// The error's text.
|
||||
pub error_message: String,
|
||||
/// When the error was recorded, in milliseconds since the epoch.
|
||||
pub created_at_millis: i64,
|
||||
}
|
||||
|
||||
/// A fragment whose per-row detail was capped: `rows_skipped` rows failed,
|
||||
/// of which only `rows_recorded` have a [`FunctionErrorRecord`].
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub struct FunctionErrorFragment {
|
||||
/// The refresh job that recorded the errors.
|
||||
pub job_id: String,
|
||||
/// The fragment.
|
||||
pub fragment_id: u64,
|
||||
/// Rows the refresh skipped in this fragment.
|
||||
pub rows_skipped: u64,
|
||||
/// Rows with a record of their own.
|
||||
pub rows_recorded: u64,
|
||||
}
|
||||
|
||||
/// A table's per-row Function errors; see [`crate::Table::function_errors`].
|
||||
#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub struct FunctionErrors {
|
||||
/// The recorded rows, newest job first.
|
||||
pub records: Vec<FunctionErrorRecord>,
|
||||
/// Fragments whose detail was capped.
|
||||
#[serde(default)]
|
||||
pub fragments: Vec<FunctionErrorFragment>,
|
||||
/// Whether the listing stopped at its limit.
|
||||
#[serde(default)]
|
||||
pub truncated: bool,
|
||||
}
|
||||
|
||||
impl RefreshColumnResult {
|
||||
/// Deprecated compatibility alias for `rows_assigned`.
|
||||
pub fn rows_filled(&self) -> u64 {
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,633 @@
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||
|
||||
//! The SQL a materialized view is defined by. A definition is stored as one
|
||||
//! canonical query, parsed here into the relational shape refresh maintains:
|
||||
//!
|
||||
//! ```sql
|
||||
//! SELECT <column | expr AS name | *>, ...
|
||||
//! FROM [ns.]table [, function(args) AS alias | , UNNEST(column) AS alias]
|
||||
//! [WHERE predicate] [LIMIT n]
|
||||
//! ```
|
||||
//!
|
||||
//! A Function in `FROM` position yields one row per element it returns;
|
||||
//! `UNNEST` does the same for a list column the table already holds.
|
||||
//!
|
||||
//! Anything else is a query this engine cannot maintain yet and is refused
|
||||
//! at parse time, which is also what an older engine does with a newer
|
||||
//! query: fail closed, never materialize it at the wrong cardinality.
|
||||
|
||||
use datafusion_sql::sqlparser::ast::{
|
||||
Expr, FunctionArg, FunctionArgExpr, JoinOperator, LimitClause, ObjectName, ObjectNamePart,
|
||||
Query, SelectItem, SetExpr, Statement, TableFactor, TableFunctionArgs, TableWithJoins, Value,
|
||||
};
|
||||
use datafusion_sql::sqlparser::dialect::GenericDialect;
|
||||
use datafusion_sql::sqlparser::keywords::{
|
||||
ALL_KEYWORDS, ALL_KEYWORDS_INDEX, RESERVED_FOR_COLUMN_ALIAS, RESERVED_FOR_IDENTIFIER,
|
||||
RESERVED_FOR_TABLE_ALIAS,
|
||||
};
|
||||
use datafusion_sql::sqlparser::parser::Parser;
|
||||
use datafusion_sql::sqlparser::tokenizer::{Token, Tokenizer};
|
||||
use lance_datafusion::planner::Planner;
|
||||
|
||||
use super::{LateralSource, MaterializedViewDefinition, ViewLateral, ViewProjection};
|
||||
use crate::{Error, Result};
|
||||
|
||||
fn invalid(message: impl Into<String>) -> Error {
|
||||
Error::InvalidInput {
|
||||
message: message.into(),
|
||||
}
|
||||
}
|
||||
|
||||
const SHAPE: &str = "a materialized view is defined by `SELECT columns FROM table \
|
||||
[, function(args) AS alias | , UNNEST(column) AS alias] [WHERE predicate] [LIMIT n]`";
|
||||
|
||||
/// Whether `name` must be delimited to read back as this identifier: bare,
|
||||
/// the parser would take it as a keyword, or a different spelling.
|
||||
fn needs_quote(name: &str) -> bool {
|
||||
let plain = !name.is_empty()
|
||||
&& name
|
||||
.chars()
|
||||
.enumerate()
|
||||
.all(|(i, c)| c == '_' || c.is_ascii_lowercase() || (i > 0 && c.is_ascii_digit()));
|
||||
if !plain {
|
||||
return true;
|
||||
}
|
||||
let reserved = ALL_KEYWORDS
|
||||
.binary_search(&name.to_ascii_uppercase().as_str())
|
||||
.is_ok_and(|i| {
|
||||
let keyword = &ALL_KEYWORDS_INDEX[i];
|
||||
RESERVED_FOR_TABLE_ALIAS.contains(keyword)
|
||||
|| RESERVED_FOR_COLUMN_ALIAS.contains(keyword)
|
||||
|| RESERVED_FOR_IDENTIFIER.contains(keyword)
|
||||
});
|
||||
if reserved {
|
||||
return true;
|
||||
}
|
||||
// Then the parsers' own judgement: lance's in an expression, where a
|
||||
// column name is read, and sqlparser's in a `FROM`, where a table's is.
|
||||
let schema = std::sync::Arc::new(arrow_schema::Schema::new(vec![arrow_schema::Field::new(
|
||||
name,
|
||||
arrow_schema::DataType::Int32,
|
||||
true,
|
||||
)]));
|
||||
let planner = Planner::new(schema);
|
||||
let column = planner
|
||||
.parse_expr(&format!("{name} IS NOT NULL"))
|
||||
.is_ok_and(|expr| Planner::column_names_in_expr(&expr) == [name]);
|
||||
if !column {
|
||||
return true;
|
||||
}
|
||||
let sql = format!("SELECT 1 FROM {name}");
|
||||
!matches!(
|
||||
Parser::parse_sql(&GenericDialect {}, &sql).as_deref(),
|
||||
Ok([Statement::Query(query)]) if matches!(
|
||||
query.body.as_ref(),
|
||||
SetExpr::Select(select) if matches!(
|
||||
select.from.as_slice(),
|
||||
[TableWithJoins { relation: TableFactor::Table { name: table, .. }, .. }]
|
||||
if table.to_string() == name
|
||||
)
|
||||
)
|
||||
)
|
||||
}
|
||||
|
||||
/// `name` as a lance SQL identifier: bare when the parser reads it back
|
||||
/// unchanged, backtick-delimited otherwise.
|
||||
pub fn ident_sql(name: &str) -> String {
|
||||
if needs_quote(name) {
|
||||
format!("`{}`", name.replace('`', "``"))
|
||||
} else {
|
||||
name.to_string()
|
||||
}
|
||||
}
|
||||
|
||||
/// Rewrite every delimited identifier in `sql` to the form [`ident_sql`]
|
||||
/// produces, so the same name is spelled one way wherever it appears.
|
||||
/// Lance's parser delimits with backticks only; a `"name"` is rewritten
|
||||
/// rather than read as a string.
|
||||
pub fn canonical_tokens(sql: &str) -> Result<String> {
|
||||
let tokens = Tokenizer::new(&GenericDialect {}, sql)
|
||||
.with_unescape(false)
|
||||
.tokenize()
|
||||
.map_err(|err| invalid(format!("invalid SQL: {err}")))?;
|
||||
Ok(tokens
|
||||
.into_iter()
|
||||
.map(|token| match token {
|
||||
Token::Word(word) if word.quote_style == Some('"') => {
|
||||
ident_sql(&word.value.replace("\"\"", "\""))
|
||||
}
|
||||
Token::Word(word) if word.quote_style == Some('`') => {
|
||||
ident_sql(&word.value.replace("``", "`"))
|
||||
}
|
||||
other => other.to_string(),
|
||||
})
|
||||
.collect())
|
||||
}
|
||||
|
||||
/// One expression, in the spelling the stored query uses.
|
||||
pub fn canonical_expr(sql: &str) -> Result<String> {
|
||||
let text = canonical_tokens(sql)?;
|
||||
let expr = Parser::new(&GenericDialect {})
|
||||
.try_with_sql(&text)
|
||||
.and_then(|mut parser| {
|
||||
let expr = parser.parse_expr()?;
|
||||
parser.expect_token(&Token::EOF)?;
|
||||
Ok(expr)
|
||||
})
|
||||
.map_err(|err| invalid(format!("invalid SQL expression '{sql}': {err}")))?;
|
||||
Ok(expr.to_string())
|
||||
}
|
||||
|
||||
/// The column a bare `SELECT` item names: the last part of a plain or
|
||||
/// compound identifier, `None` for any other expression.
|
||||
fn column_ref_name(expr: &Expr) -> Option<String> {
|
||||
match expr {
|
||||
Expr::Identifier(ident) => Some(ident.value.clone()),
|
||||
Expr::CompoundIdentifier(parts) => parts.last().map(|p| p.value.clone()),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse `sql` into a definition. The query is re-rendered and compared
|
||||
/// with what was parsed, so any clause this shape does not carry is
|
||||
/// refused rather than dropped.
|
||||
pub fn parse(sql: &str) -> Result<MaterializedViewDefinition> {
|
||||
let text = canonical_tokens(sql)?;
|
||||
let mut statements = Parser::parse_sql(&GenericDialect {}, &text)
|
||||
.map_err(|err| invalid(format!("invalid SQL: {err}")))?;
|
||||
let query = match (statements.pop(), statements.is_empty()) {
|
||||
(Some(Statement::Query(query)), true) => normalize_from(*query),
|
||||
_ => {
|
||||
return Err(invalid(format!(
|
||||
"expected a single SELECT statement; {SHAPE}"
|
||||
)));
|
||||
}
|
||||
};
|
||||
let definition = extract(&query)?;
|
||||
let rendered = render(&definition);
|
||||
if canonical_tokens(&query.to_string())? != rendered {
|
||||
return Err(invalid(format!(
|
||||
"unsupported clause in the view query; {SHAPE}"
|
||||
)));
|
||||
}
|
||||
Ok(definition)
|
||||
}
|
||||
|
||||
/// One spelling per relation: `FROM t CROSS JOIN UNNEST(..)` is
|
||||
/// `FROM t, UNNEST(..)`, and the alias always takes `AS`.
|
||||
fn normalize_from(mut query: Query) -> Query {
|
||||
if let SetExpr::Select(select) = query.body.as_mut() {
|
||||
if select.from.len() == 1
|
||||
&& select.from[0].joins.len() == 1
|
||||
&& matches!(
|
||||
select.from[0].joins[0].join_operator,
|
||||
JoinOperator::CrossJoin(_)
|
||||
)
|
||||
&& is_lateral_item(&select.from[0].joins[0].relation)
|
||||
{
|
||||
let join = select.from[0].joins.pop().expect("checked above");
|
||||
select.from.push(TableWithJoins {
|
||||
relation: join.relation,
|
||||
joins: Vec::new(),
|
||||
});
|
||||
}
|
||||
// `meta.title AS title` names what `meta.title` already names.
|
||||
for item in &mut select.projection {
|
||||
if let SelectItem::ExprWithAlias { expr, alias } = item
|
||||
&& column_ref_name(expr).as_deref() == Some(alias.value.as_str())
|
||||
{
|
||||
*item = SelectItem::UnnamedExpr(expr.clone());
|
||||
}
|
||||
}
|
||||
if let Some(item) = select.from.get_mut(1) {
|
||||
// `LATERAL f(x) AS c` and `f(x) AS c` are one relation: a function
|
||||
// in FROM position is lateral by nature.
|
||||
if let TableFactor::Function {
|
||||
name, args, alias, ..
|
||||
} = &item.relation
|
||||
{
|
||||
item.relation = TableFactor::Table {
|
||||
name: name.clone(),
|
||||
alias: alias.clone(),
|
||||
args: Some(TableFunctionArgs {
|
||||
args: args.clone(),
|
||||
settings: None,
|
||||
}),
|
||||
with_hints: Vec::new(),
|
||||
version: None,
|
||||
with_ordinality: false,
|
||||
partitions: Vec::new(),
|
||||
json_path: None,
|
||||
sample: None,
|
||||
index_hints: Vec::new(),
|
||||
};
|
||||
}
|
||||
// `UNNEST(c) e` and `f(x) e` take `AS`.
|
||||
match &mut item.relation {
|
||||
TableFactor::UNNEST {
|
||||
alias: Some(alias), ..
|
||||
}
|
||||
| TableFactor::Table {
|
||||
alias: Some(alias), ..
|
||||
} => alias.explicit = true,
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
}
|
||||
query
|
||||
}
|
||||
|
||||
fn is_lateral_item(factor: &TableFactor) -> bool {
|
||||
matches!(
|
||||
factor,
|
||||
TableFactor::UNNEST { .. }
|
||||
| TableFactor::Function { .. }
|
||||
| TableFactor::Table { args: Some(_), .. }
|
||||
)
|
||||
}
|
||||
|
||||
fn single_name(name: &ObjectName, what: &str) -> Result<String> {
|
||||
match name.0.as_slice() {
|
||||
[ObjectNamePart::Identifier(ident)] => Ok(ident.value.clone()),
|
||||
_ => Err(invalid(format!(
|
||||
"{what} must be a single name, not '{name}'"
|
||||
))),
|
||||
}
|
||||
}
|
||||
|
||||
fn extract(query: &Query) -> Result<MaterializedViewDefinition> {
|
||||
let SetExpr::Select(select) = query.body.as_ref() else {
|
||||
return Err(invalid(format!("expected a SELECT; {SHAPE}")));
|
||||
};
|
||||
|
||||
let mut from = select.from.iter();
|
||||
let (source_namespace, source_table) = match from.next().map(|f| &f.relation) {
|
||||
Some(TableFactor::Table { args: Some(_), .. }) => {
|
||||
return Err(invalid(
|
||||
"a view reads a table; a Function in FROM position follows it: \
|
||||
`FROM table, function(args) AS alias`",
|
||||
));
|
||||
}
|
||||
Some(TableFactor::Table {
|
||||
alias: Some(alias), ..
|
||||
}) => {
|
||||
return Err(invalid(format!(
|
||||
"table aliases are not supported (`AS {}`); refer to columns unqualified",
|
||||
alias.name
|
||||
)));
|
||||
}
|
||||
Some(TableFactor::Table { name, .. }) => {
|
||||
let mut parts = Vec::with_capacity(name.0.len());
|
||||
for part in &name.0 {
|
||||
match part {
|
||||
ObjectNamePart::Identifier(ident) => parts.push(ident.value.clone()),
|
||||
other => return Err(invalid(format!("unsupported table name part '{other}'"))),
|
||||
}
|
||||
}
|
||||
let table = parts.pop().ok_or_else(|| invalid("empty table name"))?;
|
||||
(parts, table)
|
||||
}
|
||||
_ => return Err(invalid(format!("the view must read one table; {SHAPE}"))),
|
||||
};
|
||||
let lateral = match from.next().map(|f| &f.relation) {
|
||||
None => None,
|
||||
Some(TableFactor::UNNEST {
|
||||
alias, array_exprs, ..
|
||||
}) => {
|
||||
let column = match array_exprs.as_slice() {
|
||||
[Expr::Identifier(ident)] => ident.value.clone(),
|
||||
_ => {
|
||||
return Err(invalid(
|
||||
"UNNEST takes one top-level list column of the table",
|
||||
));
|
||||
}
|
||||
};
|
||||
let alias = alias
|
||||
.as_ref()
|
||||
.ok_or_else(|| invalid("UNNEST needs an alias: `UNNEST(column) AS alias`"))?;
|
||||
Some(ViewLateral {
|
||||
source: LateralSource::Unnest { column },
|
||||
alias: alias.name.value.clone(),
|
||||
})
|
||||
}
|
||||
Some(TableFactor::Table {
|
||||
name,
|
||||
args: Some(TableFunctionArgs { args, .. }),
|
||||
alias,
|
||||
..
|
||||
}) => {
|
||||
let function = single_name(name, "a Function in FROM position")?;
|
||||
let mut rendered = Vec::with_capacity(args.len());
|
||||
for arg in args {
|
||||
match arg {
|
||||
FunctionArg::Unnamed(FunctionArgExpr::Expr(expr)) => {
|
||||
rendered.push(expr.to_string())
|
||||
}
|
||||
other => {
|
||||
return Err(invalid(format!(
|
||||
"'{function}' takes positional expression arguments, not '{other}'"
|
||||
)));
|
||||
}
|
||||
}
|
||||
}
|
||||
let alias = alias.as_ref().ok_or_else(|| {
|
||||
invalid(format!(
|
||||
"'{function}' in FROM position needs an alias: `{function}(...) AS alias`"
|
||||
))
|
||||
})?;
|
||||
Some(ViewLateral {
|
||||
source: LateralSource::Function {
|
||||
name: function,
|
||||
args: rendered,
|
||||
},
|
||||
alias: alias.name.value.clone(),
|
||||
})
|
||||
}
|
||||
Some(_) => return Err(invalid(format!("the view must read one table; {SHAPE}"))),
|
||||
};
|
||||
if from.next().is_some() {
|
||||
return Err(invalid(format!("the view must read one table; {SHAPE}")));
|
||||
}
|
||||
|
||||
let mut projections = Vec::with_capacity(select.projection.len());
|
||||
for item in &select.projection {
|
||||
match item {
|
||||
SelectItem::Wildcard(_) if select.projection.len() == 1 => {
|
||||
projections.push(ViewProjection::star());
|
||||
}
|
||||
SelectItem::Wildcard(_) => {
|
||||
return Err(invalid("`*` must be the only column selected"));
|
||||
}
|
||||
SelectItem::UnnamedExpr(expr) => {
|
||||
let output = column_ref_name(expr).ok_or_else(|| {
|
||||
invalid(format!(
|
||||
"view column `{expr}` needs a name: `{expr} AS name`"
|
||||
))
|
||||
})?;
|
||||
projections.push(ViewProjection {
|
||||
output,
|
||||
expression: expr.to_string(),
|
||||
});
|
||||
}
|
||||
SelectItem::ExprWithAlias { expr, alias } => projections.push(ViewProjection {
|
||||
output: alias.value.clone(),
|
||||
expression: expr.to_string(),
|
||||
}),
|
||||
other => return Err(invalid(format!("unsupported select item '{other}'"))),
|
||||
}
|
||||
}
|
||||
|
||||
let limit =
|
||||
match &query.limit_clause {
|
||||
None => None,
|
||||
Some(LimitClause::LimitOffset {
|
||||
limit: Some(Expr::Value(value)),
|
||||
offset: None,
|
||||
limit_by,
|
||||
}) if limit_by.is_empty() => match &value.value {
|
||||
Value::Number(n, _) => Some(n.parse::<u64>().map_err(|_| {
|
||||
invalid(format!("view limit {n} is not a non-negative integer"))
|
||||
})?),
|
||||
_ => return Err(invalid("view limit must be an integer literal")),
|
||||
},
|
||||
Some(_) => return Err(invalid("view limit must be a plain `LIMIT n`")),
|
||||
};
|
||||
|
||||
Ok(MaterializedViewDefinition {
|
||||
source_table,
|
||||
source_namespace,
|
||||
lateral,
|
||||
projections,
|
||||
filter: select.selection.as_ref().map(|e| e.to_string()),
|
||||
limit,
|
||||
})
|
||||
}
|
||||
|
||||
/// The canonical query for `definition`; [`parse`] reads it back equal.
|
||||
pub fn render(definition: &MaterializedViewDefinition) -> String {
|
||||
let mut sql = String::from("SELECT ");
|
||||
if definition.selects_star() {
|
||||
sql.push('*');
|
||||
} else {
|
||||
let items: Vec<String> = definition
|
||||
.projections
|
||||
.iter()
|
||||
.map(|p| {
|
||||
let bare = Parser::new(&GenericDialect {})
|
||||
.try_with_sql(&p.expression)
|
||||
.and_then(|mut parser| parser.parse_expr())
|
||||
.ok()
|
||||
.and_then(|expr| column_ref_name(&expr))
|
||||
.is_some_and(|name| name == p.output);
|
||||
if bare {
|
||||
p.expression.clone()
|
||||
} else {
|
||||
format!("{} AS {}", p.expression, ident_sql(&p.output))
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
sql.push_str(&items.join(", "));
|
||||
}
|
||||
sql.push_str(" FROM ");
|
||||
let table: Vec<String> = definition
|
||||
.source_namespace
|
||||
.iter()
|
||||
.chain(std::iter::once(&definition.source_table))
|
||||
.map(|part| ident_sql(part))
|
||||
.collect();
|
||||
sql.push_str(&table.join("."));
|
||||
if let Some(lateral) = &definition.lateral {
|
||||
match &lateral.source {
|
||||
LateralSource::Unnest { column } => {
|
||||
sql.push_str(&format!(", UNNEST({})", ident_sql(column)))
|
||||
}
|
||||
LateralSource::Function { name, args } => {
|
||||
sql.push_str(&format!(", {}({})", ident_sql(name), args.join(", ")))
|
||||
}
|
||||
}
|
||||
sql.push_str(&format!(" AS {}", ident_sql(&lateral.alias)));
|
||||
}
|
||||
if let Some(filter) = &definition.filter {
|
||||
sql.push_str(&format!(" WHERE {filter}"));
|
||||
}
|
||||
if let Some(limit) = definition.limit {
|
||||
sql.push_str(&format!(" LIMIT {limit}"));
|
||||
}
|
||||
sql
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn a_query_round_trips_through_its_canonical_form() {
|
||||
for (sql, canonical) in [
|
||||
(
|
||||
r#"select "Name", x*2 as twice from ns.docs where x > 1 limit 5"#,
|
||||
"SELECT `Name`, x * 2 AS twice FROM ns.docs WHERE x > 1 LIMIT 5",
|
||||
),
|
||||
(
|
||||
"SELECT id, c.chunk, c.ordinal + 1 AS nth FROM docs, UNNEST(chunks) AS c WHERE c.ordinal < 5",
|
||||
"SELECT id, c.chunk, c.ordinal + 1 AS nth FROM docs, UNNEST(chunks) AS c WHERE c.ordinal < 5",
|
||||
),
|
||||
(
|
||||
"SELECT id FROM docs CROSS JOIN UNNEST(chunks) c",
|
||||
"SELECT id FROM docs, UNNEST(chunks) AS c",
|
||||
),
|
||||
(
|
||||
"select d.id, c.text from docs, chunk(body, 512) c where c.ordinal < 3",
|
||||
"SELECT d.id, c.text FROM docs, chunk(body, 512) AS c WHERE c.ordinal < 3",
|
||||
),
|
||||
(
|
||||
"SELECT id, c.text FROM docs CROSS JOIN LATERAL chunk(body) AS c",
|
||||
"SELECT id, c.text FROM docs, chunk(body) AS c",
|
||||
),
|
||||
(
|
||||
"SELECT id, c.text FROM docs, LATERAL chunk(upper(body)) AS c",
|
||||
"SELECT id, c.text FROM docs, chunk(upper(body)) AS c",
|
||||
),
|
||||
("SELECT * FROM `select`.t", "SELECT * FROM `select`.t"),
|
||||
(
|
||||
"SELECT meta.title AS title, id AS id FROM t",
|
||||
"SELECT meta.title, id FROM t",
|
||||
),
|
||||
] {
|
||||
let definition = parse(sql).unwrap();
|
||||
assert_eq!(render(&definition), canonical, "{sql}");
|
||||
assert_eq!(parse(canonical).unwrap(), definition, "{sql}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parsed_parts_are_the_relational_shape() {
|
||||
let definition =
|
||||
parse("SELECT id, c.chunk FROM ns.docs, UNNEST(chunks) AS c WHERE id > 1").unwrap();
|
||||
assert_eq!(definition.source_namespace, ["ns"]);
|
||||
assert_eq!(definition.source_table, "docs");
|
||||
assert_eq!(
|
||||
definition.lateral,
|
||||
Some(ViewLateral {
|
||||
source: LateralSource::Unnest {
|
||||
column: "chunks".into()
|
||||
},
|
||||
alias: "c".into()
|
||||
})
|
||||
);
|
||||
let function = parse("SELECT id, c.text FROM docs, chunk(body, 512) AS c").unwrap();
|
||||
assert_eq!(
|
||||
function.lateral,
|
||||
Some(ViewLateral {
|
||||
source: LateralSource::Function {
|
||||
name: "chunk".into(),
|
||||
args: vec!["body".into(), "512".into()]
|
||||
},
|
||||
alias: "c".into()
|
||||
})
|
||||
);
|
||||
assert_eq!(
|
||||
definition.projections,
|
||||
[
|
||||
ViewProjection {
|
||||
output: "id".into(),
|
||||
expression: "id".into()
|
||||
},
|
||||
ViewProjection {
|
||||
output: "chunk".into(),
|
||||
expression: "c.chunk".into()
|
||||
},
|
||||
]
|
||||
);
|
||||
assert_eq!(definition.filter.as_deref(), Some("id > 1"));
|
||||
assert!(parse("SELECT * FROM t").unwrap().selects_star());
|
||||
}
|
||||
|
||||
/// A clause the engine cannot maintain is refused, never dropped.
|
||||
#[test]
|
||||
fn unsupported_clauses_are_refused() {
|
||||
for sql in [
|
||||
"SELECT id FROM t GROUP BY id",
|
||||
"SELECT id FROM t ORDER BY id",
|
||||
"SELECT DISTINCT id FROM t",
|
||||
"SELECT id FROM t LIMIT 5 OFFSET 2",
|
||||
"SELECT id FROM t JOIN u ON t.id = u.id",
|
||||
"SELECT id FROM t, u",
|
||||
"SELECT id FROM t, UNNEST(c)",
|
||||
"SELECT id FROM t, UNNEST(a.b) AS c",
|
||||
"SELECT id FROM t, chunk(body)",
|
||||
"SELECT id FROM t, ns.chunk(body) AS c",
|
||||
"SELECT id FROM t, chunk(size => 5) AS c",
|
||||
"SELECT id FROM t AS d, chunk(d.body) AS c",
|
||||
"SELECT id FROM chunk(body) AS c",
|
||||
"SELECT id FROM t, chunk(body) AS c, UNNEST(x) AS u",
|
||||
"SELECT count(*) FROM t",
|
||||
"SELECT x * 2 FROM t",
|
||||
"SELECT id FROM t; SELECT id FROM t",
|
||||
"SELECT *, id FROM t",
|
||||
"WITH q AS (SELECT 1) SELECT id FROM t",
|
||||
"SELECT id FROM t HAVING id > 1",
|
||||
] {
|
||||
assert!(parse(sql).is_err(), "{sql}");
|
||||
}
|
||||
}
|
||||
|
||||
/// The canonical spelling is what lance's planner reads back, for the
|
||||
/// expression forms a view is likely to carry.
|
||||
#[test]
|
||||
fn canonical_expressions_plan_in_lance() {
|
||||
use arrow_schema::{DataType, Field, Schema};
|
||||
|
||||
let planner = Planner::new(std::sync::Arc::new(Schema::new(vec![
|
||||
Field::new("x", DataType::Int32, true),
|
||||
Field::new("Name", DataType::Utf8, true),
|
||||
Field::new(
|
||||
"when",
|
||||
DataType::Timestamp(arrow_schema::TimeUnit::Microsecond, None),
|
||||
true,
|
||||
),
|
||||
Field::new(
|
||||
"meta",
|
||||
DataType::Struct(vec![Field::new("title", DataType::Utf8, true)].into()),
|
||||
true,
|
||||
),
|
||||
])));
|
||||
for (raw, canonical) in [
|
||||
("x*2+1", "x * 2 + 1"),
|
||||
(r#"CAST(x as bigint)"#, "CAST(x AS BIGINT)"),
|
||||
(r#"upper("Name") like 'A%'"#, "upper(`Name`) LIKE 'A%'"),
|
||||
(
|
||||
"x is not null and x between 1 and 3",
|
||||
"x IS NOT NULL AND x BETWEEN 1 AND 3",
|
||||
),
|
||||
("meta.title", "meta.title"),
|
||||
(r#"`Name` = 'it''s'"#, "`Name` = 'it''s'"),
|
||||
("x in (1, 2)", "x IN (1, 2)"),
|
||||
(
|
||||
"`when` > timestamp '2024-01-01'",
|
||||
"when > TIMESTAMP '2024-01-01'",
|
||||
),
|
||||
("-x", "-x"),
|
||||
] {
|
||||
let text = canonical_expr(raw).unwrap();
|
||||
assert_eq!(text, canonical, "{raw}");
|
||||
let expr = planner
|
||||
.parse_expr(&text)
|
||||
.unwrap_or_else(|e| panic!("{text}: {e}"));
|
||||
planner
|
||||
.optimize_expr(expr)
|
||||
.unwrap_or_else(|e| panic!("{text}: {e}"));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn identifiers_are_delimited_only_when_the_parser_needs_it() {
|
||||
assert_eq!(ident_sql("name"), "name");
|
||||
assert_eq!(ident_sql("Name"), "`Name`");
|
||||
assert_eq!(ident_sql("select"), "`select`");
|
||||
assert_eq!(ident_sql("1st"), "`1st`");
|
||||
assert_eq!(ident_sql("a`b"), "`a``b`");
|
||||
assert_eq!(canonical_expr(r#""Party" = 'D'"#).unwrap(), "`Party` = 'D'");
|
||||
assert_eq!(canonical_expr("`name`").unwrap(), "name");
|
||||
}
|
||||
}
|
||||
@@ -25,9 +25,10 @@ use std::time::{SystemTime, UNIX_EPOCH};
|
||||
use arrow_array::cast::AsArray;
|
||||
use arrow_array::types::UInt64Type;
|
||||
use arrow_array::{RecordBatch, UInt64Array, new_null_array};
|
||||
use arrow_schema::{FieldRef, Schema as ArrowSchema, SchemaRef};
|
||||
use arrow_schema::{DataType, Field as ArrowField, FieldRef, Schema as ArrowSchema, SchemaRef};
|
||||
use datafusion::common::ScalarValue;
|
||||
use datafusion::error::DataFusionError;
|
||||
use datafusion::physical_expr::PhysicalExpr;
|
||||
use datafusion::physical_plan::SendableRecordBatchStream;
|
||||
use datafusion::physical_plan::stream::RecordBatchStreamAdapter;
|
||||
use datafusion::prelude::{col, lit};
|
||||
@@ -41,6 +42,8 @@ use lance::dataset::write::merge_insert::inserted_rows::{
|
||||
};
|
||||
use lance::dataset::{CommitBuilder, InsertBuilder, WriteDestination, WriteMode, WriteParams};
|
||||
use lance_core::{ROW_CREATED_AT_VERSION, ROW_ID, ROW_LAST_UPDATED_AT_VERSION};
|
||||
|
||||
use lance_datafusion::planner::Planner;
|
||||
use lance_file::version::ConcreteFileVersion;
|
||||
use lance_table::format::Fragment;
|
||||
use serde::{Deserialize, Serialize};
|
||||
@@ -131,12 +134,12 @@ pub(crate) async fn execute_refresh(
|
||||
|
||||
// The definition a handle cached at open may since have been replaced;
|
||||
// what refresh executes and what it stamps must be one generation.
|
||||
let definition = match super::materialized_view_kind(&view_ds.schema().metadata)? {
|
||||
Some(super::MaterializedViewKind::Select(definition)) => definition,
|
||||
Some(super::MaterializedViewKind::Unrecognized { kind }) => {
|
||||
let definition = match super::read_definition(&view_ds.schema().metadata)? {
|
||||
Some(super::StoredDefinition::Query(definition)) => definition,
|
||||
Some(super::StoredDefinition::Newer { format }) => {
|
||||
return Err(Error::NotSupported {
|
||||
message: format!(
|
||||
"materialized view '{}' is defined by '{kind}', which this \
|
||||
"materialized view '{}' is stored in format {format}, which this \
|
||||
version of lancedb cannot refresh",
|
||||
view.name()
|
||||
),
|
||||
@@ -149,9 +152,10 @@ pub(crate) async fn execute_refresh(
|
||||
}
|
||||
};
|
||||
let definition = &definition;
|
||||
let staging = super::read_staging(&view_ds.schema().metadata)?;
|
||||
ensure_no_mem_wal(&view_ds, "materialized view", view.name()).await?;
|
||||
|
||||
let source_ds = open_source(view, definition).await?;
|
||||
let source_ds = open_source(view, definition, staging.as_ref()).await?;
|
||||
let source_ds = match pinned {
|
||||
Some(version) => source_ds.checkout_version(version).await?,
|
||||
None => source_ds,
|
||||
@@ -164,20 +168,30 @@ pub(crate) async fn execute_refresh(
|
||||
// require its planned output to be exactly the view's physical schema: a
|
||||
// definition the stored table cannot represent must not be certified.
|
||||
let source_schema = Arc::new(ArrowSchema::from(source_ds.schema()));
|
||||
let projections: Vec<(String, String)> = definition
|
||||
.projections
|
||||
.iter()
|
||||
.map(|p| (p.output.clone(), p.expression.clone()))
|
||||
.collect();
|
||||
validate_inputs(&source_ds, definition)?;
|
||||
let (replanned, planned_fields, _renames) = super::plan(
|
||||
source_schema,
|
||||
&definition.source_table,
|
||||
&definition.source_namespace,
|
||||
Some(&projections),
|
||||
definition.filter.as_deref(),
|
||||
definition.limit,
|
||||
)?;
|
||||
let super::Planned {
|
||||
definition: replanned,
|
||||
fields: planned_fields,
|
||||
inputs,
|
||||
..
|
||||
} = super::plan(source_schema.clone(), definition, staging.as_ref()).map_err(|e| match e {
|
||||
// The stored query planned when the view was declared; what changed
|
||||
// since is the source.
|
||||
Error::InvalidExpression { column, message } => Error::Schema {
|
||||
message: format!(
|
||||
"view column '{column}' no longer plans against '{}' (a source column \
|
||||
was dropped or renamed): {message}",
|
||||
definition.source_table
|
||||
),
|
||||
},
|
||||
Error::InvalidInput { message } => Error::Schema {
|
||||
message: format!(
|
||||
"the stored query no longer plans against '{}' (a source column was \
|
||||
dropped or renamed): {message}",
|
||||
definition.source_table
|
||||
),
|
||||
},
|
||||
e => e,
|
||||
})?;
|
||||
let mut planned_fields = planned_fields;
|
||||
planned_fields.push(arrow_schema::Field::new(
|
||||
SOURCE_ROW_ID_COLUMN,
|
||||
@@ -224,14 +238,19 @@ pub(crate) async fn execute_refresh(
|
||||
),
|
||||
});
|
||||
}
|
||||
let definition_changed =
|
||||
definition.filter != replanned.filter || definition.inputs != replanned.inputs;
|
||||
// The stored query is rewritten whenever its stored form differs from
|
||||
// the current one: a legacy layout, or a spelling the canonicalizer no
|
||||
// longer produces. Whether the rows change is a separate question: a
|
||||
// legacy raw filter like `"Party" = 'D'` read the double quotes as a
|
||||
// string literal, so its watermark certifies different rows than the
|
||||
// canonical predicate, and only a rebuild can replace them.
|
||||
let current = definition_to_metadata(&replanned)?;
|
||||
let persist = view_ds.schema().metadata.get(DEFINITION_META_KEY) != Some(¤t);
|
||||
let unnest = super::physical_unnest(&replanned, staging.as_ref())?;
|
||||
let definition_changed = !same_meaning(&source_schema, definition, &replanned, unnest.as_ref());
|
||||
let definition = &replanned;
|
||||
let persist = persist.then_some(definition);
|
||||
|
||||
// A watermark written for a legacy raw filter certifies the rows that
|
||||
// filter produced, not the canonical predicate above. Rebuild instead of
|
||||
// accepting or advancing it, and persist the migrated definition in the
|
||||
// same metadata commit that certifies the replacement rows.
|
||||
if definition_changed {
|
||||
return rebuild(
|
||||
view_native,
|
||||
@@ -240,6 +259,8 @@ pub(crate) async fn execute_refresh(
|
||||
source_version,
|
||||
source_ts,
|
||||
definition,
|
||||
&inputs,
|
||||
unnest.as_ref(),
|
||||
true,
|
||||
expected_incarnation,
|
||||
)
|
||||
@@ -284,6 +305,7 @@ pub(crate) async fn execute_refresh(
|
||||
recorded_ts,
|
||||
full,
|
||||
definition,
|
||||
&inputs,
|
||||
)
|
||||
.await
|
||||
{
|
||||
@@ -296,6 +318,9 @@ pub(crate) async fn execute_refresh(
|
||||
source_ts,
|
||||
increment,
|
||||
definition,
|
||||
&inputs,
|
||||
unnest.as_ref(),
|
||||
persist,
|
||||
watermark,
|
||||
expected_incarnation,
|
||||
)
|
||||
@@ -311,7 +336,9 @@ pub(crate) async fn execute_refresh(
|
||||
source_version,
|
||||
source_ts,
|
||||
definition,
|
||||
false,
|
||||
&inputs,
|
||||
unnest.as_ref(),
|
||||
persist.is_some(),
|
||||
expected_incarnation,
|
||||
)
|
||||
.await
|
||||
@@ -326,7 +353,9 @@ pub(crate) async fn execute_refresh(
|
||||
source_version,
|
||||
source_ts,
|
||||
definition,
|
||||
false,
|
||||
&inputs,
|
||||
unnest.as_ref(),
|
||||
persist.is_some(),
|
||||
expected_incarnation,
|
||||
)
|
||||
.await
|
||||
@@ -345,6 +374,7 @@ async fn plan_increment(
|
||||
recorded_ts: Option<u128>,
|
||||
full: bool,
|
||||
definition: &MaterializedViewDefinition,
|
||||
inputs: &[String],
|
||||
) -> Option<Increment> {
|
||||
if full {
|
||||
return None;
|
||||
@@ -414,7 +444,7 @@ async fn plan_increment(
|
||||
});
|
||||
}
|
||||
|
||||
is_pure_append(&old, source_ds, &relevant_field_ids(source_ds, definition)).then(|| Increment {
|
||||
is_pure_append(&old, source_ds, &relevant_field_ids(source_ds, inputs)).then(|| Increment {
|
||||
appended: live
|
||||
.into_iter()
|
||||
.filter(|f| !old_ids.contains(&f.id))
|
||||
@@ -567,7 +597,7 @@ fn fragment_signature(metadata: &Fragment, relevant: &HashSet<i32>) -> (u64, Str
|
||||
}
|
||||
|
||||
/// Field ids (with struct descendants) of the source columns the view reads.
|
||||
fn relevant_field_ids(source: &Dataset, definition: &MaterializedViewDefinition) -> HashSet<i32> {
|
||||
fn relevant_field_ids(source: &Dataset, inputs: &[String]) -> HashSet<i32> {
|
||||
fn collect(field: &lance_core::datatypes::Field, ids: &mut HashSet<i32>) {
|
||||
ids.insert(field.id);
|
||||
for child in &field.children {
|
||||
@@ -575,7 +605,7 @@ fn relevant_field_ids(source: &Dataset, definition: &MaterializedViewDefinition)
|
||||
}
|
||||
}
|
||||
let mut ids = HashSet::new();
|
||||
for input in &definition.inputs {
|
||||
for input in inputs {
|
||||
if let Some(field) = source.schema().field(input) {
|
||||
collect(field, &mut ids);
|
||||
}
|
||||
@@ -583,20 +613,46 @@ fn relevant_field_ids(source: &Dataset, definition: &MaterializedViewDefinition)
|
||||
ids
|
||||
}
|
||||
|
||||
/// Error if a column the view reads no longer exists in the source.
|
||||
fn validate_inputs(source: &Dataset, definition: &MaterializedViewDefinition) -> Result<()> {
|
||||
for input in &definition.inputs {
|
||||
if source.schema().field(input).is_none() {
|
||||
return Err(Error::Schema {
|
||||
message: format!(
|
||||
"source column '{input}' read by the view no longer exists \
|
||||
(dropped or renamed in '{}')",
|
||||
definition.source_table
|
||||
),
|
||||
});
|
||||
}
|
||||
/// Whether two plannings of a view compute the same rows and columns: the
|
||||
/// same source, unnest and limit, and expressions the planner reads as the
|
||||
/// same logical expression, whatever their spelling. A definition that does
|
||||
/// not plan compares as different.
|
||||
fn same_meaning(
|
||||
source_schema: &SchemaRef,
|
||||
stored: &MaterializedViewDefinition,
|
||||
replanned: &MaterializedViewDefinition,
|
||||
unnest: Option<&super::ViewUnnest>,
|
||||
) -> bool {
|
||||
if stored.source_table != replanned.source_table
|
||||
|| stored.source_namespace != replanned.source_namespace
|
||||
|| stored.lateral != replanned.lateral
|
||||
|| stored.limit != replanned.limit
|
||||
|| stored.projections.len() != replanned.projections.len()
|
||||
|| stored.filter.is_some() != replanned.filter.is_some()
|
||||
{
|
||||
return false;
|
||||
}
|
||||
Ok(())
|
||||
let schema = match unnest {
|
||||
None => source_schema.clone(),
|
||||
Some(unnest) => match super::flattened_schema(source_schema, unnest) {
|
||||
Ok(schema) => schema,
|
||||
Err(_) => return false,
|
||||
},
|
||||
};
|
||||
let planner = Planner::new(schema);
|
||||
let same_expr = |a: &str, b: &str| match (planner.parse_expr(a), planner.parse_expr(b)) {
|
||||
(Ok(a), Ok(b)) => a == b,
|
||||
_ => false,
|
||||
};
|
||||
stored
|
||||
.projections
|
||||
.iter()
|
||||
.zip(&replanned.projections)
|
||||
.all(|(a, b)| a.output == b.output && same_expr(&a.expression, &b.expression))
|
||||
&& match (&stored.filter, &replanned.filter) {
|
||||
(Some(a), Some(b)) => same_expr(a, b),
|
||||
_ => true,
|
||||
}
|
||||
}
|
||||
|
||||
/// Reject MemWAL/LSM state on a refresh participant: un-compacted tiers are
|
||||
@@ -616,14 +672,27 @@ pub(crate) async fn ensure_no_mem_wal(dataset: &Dataset, role: &str, name: &str)
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn open_source(view: &Table, definition: &MaterializedViewDefinition) -> Result<Dataset> {
|
||||
/// The table refresh scans: the staging table when the query calls a
|
||||
/// Function in FROM position, otherwise the query's source.
|
||||
async fn open_source(
|
||||
view: &Table,
|
||||
definition: &MaterializedViewDefinition,
|
||||
staging: Option<&super::StagingBinding>,
|
||||
) -> Result<Dataset> {
|
||||
let database = view.database_opt().ok_or_else(|| Error::InvalidInput {
|
||||
message: "the view was not opened through a database connection".into(),
|
||||
})?;
|
||||
let (name, namespace_path) = match staging {
|
||||
Some(staging) => (staging.table.clone(), staging.namespace.clone()),
|
||||
None => (
|
||||
definition.source_table.clone(),
|
||||
definition.source_namespace.clone(),
|
||||
),
|
||||
};
|
||||
let source = database
|
||||
.open_table(OpenTableRequest {
|
||||
name: definition.source_table.clone(),
|
||||
namespace_path: definition.source_namespace.clone(),
|
||||
name,
|
||||
namespace_path,
|
||||
index_cache_size: None,
|
||||
lance_read_params: None,
|
||||
location: None,
|
||||
@@ -656,6 +725,9 @@ async fn incremental(
|
||||
source_ts: u128,
|
||||
increment: Increment,
|
||||
definition: &MaterializedViewDefinition,
|
||||
inputs: &[String],
|
||||
unnest: Option<&super::ViewUnnest>,
|
||||
persist: Option<&MaterializedViewDefinition>,
|
||||
watermark: Option<u64>,
|
||||
expected_incarnation: Option<&str>,
|
||||
) -> Result<Option<RefreshMaterializedViewResult>> {
|
||||
@@ -739,7 +811,7 @@ async fn incremental(
|
||||
view_ds.clone(),
|
||||
source_version,
|
||||
source_ts,
|
||||
None,
|
||||
persist,
|
||||
expected_incarnation,
|
||||
)
|
||||
.await?;
|
||||
@@ -761,7 +833,7 @@ async fn incremental(
|
||||
published,
|
||||
source_version,
|
||||
source_ts,
|
||||
None,
|
||||
persist,
|
||||
expected_incarnation,
|
||||
)
|
||||
.await?;
|
||||
@@ -778,6 +850,8 @@ async fn incremental(
|
||||
let mut stream = compute_stream(
|
||||
source_ds,
|
||||
definition,
|
||||
inputs,
|
||||
unnest,
|
||||
RowScope {
|
||||
fragments: Some(new_fragments),
|
||||
// An update rewrites whole fragments, so a fragment new at head
|
||||
@@ -799,6 +873,8 @@ async fn incremental(
|
||||
let recomputed = compute_stream(
|
||||
source_ds,
|
||||
definition,
|
||||
inputs,
|
||||
unnest,
|
||||
RowScope {
|
||||
updated_between: Some((watermark_version, source_version)),
|
||||
..Default::default()
|
||||
@@ -833,7 +909,7 @@ async fn incremental(
|
||||
published,
|
||||
source_version,
|
||||
source_ts,
|
||||
None,
|
||||
persist,
|
||||
expected_incarnation,
|
||||
)
|
||||
.await?;
|
||||
@@ -883,7 +959,7 @@ async fn incremental(
|
||||
appended,
|
||||
source_version,
|
||||
source_ts,
|
||||
None,
|
||||
persist,
|
||||
expected_incarnation,
|
||||
)
|
||||
.await?;
|
||||
@@ -898,6 +974,8 @@ async fn rebuild(
|
||||
source_version: u64,
|
||||
source_ts: u128,
|
||||
definition: &MaterializedViewDefinition,
|
||||
inputs: &[String],
|
||||
unnest: Option<&super::ViewUnnest>,
|
||||
persist_definition: bool,
|
||||
expected_incarnation: Option<&str>,
|
||||
) -> Result<RefreshMaterializedViewResult> {
|
||||
@@ -906,6 +984,8 @@ async fn rebuild(
|
||||
let stream = compute_stream(
|
||||
source_ds,
|
||||
definition,
|
||||
inputs,
|
||||
unnest,
|
||||
RowScope {
|
||||
limit: definition.limit,
|
||||
..Default::default()
|
||||
@@ -1202,6 +1282,8 @@ async fn only_computed_rewrites_since(view_ds: &Dataset, recorded: u64) -> Resul
|
||||
async fn compute_stream(
|
||||
source: &Dataset,
|
||||
definition: &MaterializedViewDefinition,
|
||||
inputs: &[String],
|
||||
unnest: Option<&super::ViewUnnest>,
|
||||
scope: RowScope,
|
||||
schema: SchemaRef,
|
||||
rows_written: Arc<AtomicU64>,
|
||||
@@ -1233,6 +1315,7 @@ async fn compute_stream(
|
||||
let clauses: Vec<String> = definition
|
||||
.filter
|
||||
.clone()
|
||||
.filter(|_| unnest.is_none())
|
||||
.map(|f| format!("({f})"))
|
||||
.into_iter()
|
||||
.chain(updated_filter)
|
||||
@@ -1241,12 +1324,23 @@ async fn compute_stream(
|
||||
if !clauses.is_empty() {
|
||||
scanner.filter(&clauses.join(" AND "))?;
|
||||
}
|
||||
let transforms: Vec<(&str, &str)> = definition
|
||||
.projections
|
||||
.iter()
|
||||
.map(|p| (p.output.as_str(), p.expression.as_str()))
|
||||
.collect();
|
||||
scanner.project_with_transform(&transforms)?;
|
||||
// An expanded view cannot project or filter in the scan: both read the
|
||||
// unnested element, which exists only after the per-batch expansion.
|
||||
let expanded = match unnest {
|
||||
Some(unnest) => Some(UnnestPlan::new(source, definition, inputs, unnest)?),
|
||||
None => {
|
||||
let transforms: Vec<(&str, &str)> = definition
|
||||
.projections
|
||||
.iter()
|
||||
.map(|p| (p.output.as_str(), p.expression.as_str()))
|
||||
.collect();
|
||||
scanner.project_with_transform(&transforms)?;
|
||||
None
|
||||
}
|
||||
};
|
||||
if let Some(expanded) = &expanded {
|
||||
scanner.project(&expanded.raw_inputs)?;
|
||||
}
|
||||
// A scan reads a limit of zero as no limit at all, so a view capped at
|
||||
// nothing is answered without one.
|
||||
if limit == Some(0) {
|
||||
@@ -1265,6 +1359,10 @@ async fn compute_stream(
|
||||
let out_schema = schema.clone();
|
||||
let mapped = scanner.try_into_stream().await?.map(move |batch| {
|
||||
let batch = batch.map_err(|e| DataFusionError::External(Box::new(e)))?;
|
||||
let batch = match &expanded {
|
||||
None => batch,
|
||||
Some(expanded) => expanded.apply(&batch)?,
|
||||
};
|
||||
let mut columns = Vec::with_capacity(out_schema.fields().len());
|
||||
for field in out_schema.fields() {
|
||||
if computed_column_from_field(field).is_some() {
|
||||
@@ -1290,6 +1388,192 @@ async fn compute_stream(
|
||||
Ok(Box::pin(RecordBatchStreamAdapter::new(schema, mapped)))
|
||||
}
|
||||
|
||||
/// The post-scan half of an unnested view's refresh: the scan reads
|
||||
/// `raw_inputs` plus the row id, and each batch is unnested on the list
|
||||
/// column, filtered, then projected by expressions typed against
|
||||
/// `read_schema`, where the element sits under the alias.
|
||||
struct UnnestPlan {
|
||||
column: String,
|
||||
raw_inputs: Vec<String>,
|
||||
read_schema: SchemaRef,
|
||||
projections: Vec<(String, Arc<dyn PhysicalExpr>)>,
|
||||
filter: Option<Arc<dyn PhysicalExpr>>,
|
||||
}
|
||||
|
||||
impl UnnestPlan {
|
||||
fn new(
|
||||
source: &Dataset,
|
||||
definition: &MaterializedViewDefinition,
|
||||
inputs: &[String],
|
||||
unnest: &super::ViewUnnest,
|
||||
) -> Result<Self> {
|
||||
// Whole root columns: a nested input is projected by the expression.
|
||||
let mut raw_inputs: Vec<String> = inputs
|
||||
.iter()
|
||||
.map(|input| super::root(input).to_string())
|
||||
.chain(std::iter::once(unnest.column.clone()))
|
||||
.collect();
|
||||
raw_inputs.sort();
|
||||
raw_inputs.dedup();
|
||||
let flattened = super::flattened_schema(&ArrowSchema::from(source.schema()), unnest)?;
|
||||
// Physical expressions index columns by position, so the schema is
|
||||
// exactly the scan's output: `raw_inputs` in order, then the row id.
|
||||
let mut read_fields = Vec::with_capacity(raw_inputs.len() + 1);
|
||||
for name in &raw_inputs {
|
||||
let name = if *name == unnest.column {
|
||||
&unnest.alias
|
||||
} else {
|
||||
name
|
||||
};
|
||||
let field = flattened
|
||||
.field_with_name(name)
|
||||
.map_err(|_| Error::Runtime {
|
||||
message: format!("source column '{name}' read by the view is missing"),
|
||||
})?;
|
||||
read_fields.push(field.clone());
|
||||
}
|
||||
read_fields.push(ArrowField::new(ROW_ID, DataType::UInt64, false));
|
||||
let read_schema = Arc::new(ArrowSchema::new(read_fields));
|
||||
let planner = Planner::new(read_schema.clone());
|
||||
let physical = |what: &str, sql: &str| -> Result<Arc<dyn PhysicalExpr>> {
|
||||
let err = |e: lance::Error| Error::Runtime {
|
||||
message: format!("{what}: {e}"),
|
||||
};
|
||||
let parsed = planner.parse_expr(sql).map_err(err)?;
|
||||
let optimized = planner.optimize_expr(parsed).map_err(err)?;
|
||||
planner.create_physical_expr(&optimized).map_err(err)
|
||||
};
|
||||
let mut projections = Vec::with_capacity(definition.projections.len());
|
||||
for projection in &definition.projections {
|
||||
let expr = physical(
|
||||
&format!("view column '{}'", projection.output),
|
||||
&projection.expression,
|
||||
)?;
|
||||
projections.push((projection.output.clone(), expr));
|
||||
}
|
||||
let filter = definition
|
||||
.filter
|
||||
.as_deref()
|
||||
.map(|sql| physical("view filter", sql))
|
||||
.transpose()?;
|
||||
Ok(Self {
|
||||
column: unnest.column.clone(),
|
||||
raw_inputs,
|
||||
read_schema,
|
||||
projections,
|
||||
filter,
|
||||
})
|
||||
}
|
||||
|
||||
fn apply(&self, batch: &RecordBatch) -> datafusion::common::Result<RecordBatch> {
|
||||
let unnested = unnest_batch(batch, &self.column)
|
||||
.map_err(|e| DataFusionError::External(Box::new(e)))?;
|
||||
// Same columns, renamed: the list column is now the element under the alias.
|
||||
let unnested = RecordBatch::try_new(self.read_schema.clone(), unnested.columns().to_vec())?;
|
||||
let unnested = match &self.filter {
|
||||
None => unnested,
|
||||
Some(filter) => {
|
||||
let keep = filter
|
||||
.evaluate(&unnested)?
|
||||
.into_array(unnested.num_rows())?;
|
||||
let keep = keep.as_boolean_opt().ok_or_else(|| {
|
||||
DataFusionError::Internal("view filter did not evaluate to a boolean".into())
|
||||
})?;
|
||||
arrow_select::filter::filter_record_batch(&unnested, keep)?
|
||||
}
|
||||
};
|
||||
let mut columns = Vec::with_capacity(self.projections.len() + 1);
|
||||
for (output, expr) in &self.projections {
|
||||
let value = expr.evaluate(&unnested)?.into_array(unnested.num_rows())?;
|
||||
columns.push((output.clone(), value));
|
||||
}
|
||||
let row_id = unnested
|
||||
.column_by_name(ROW_ID)
|
||||
.expect("scan carries the row id")
|
||||
.clone();
|
||||
columns.push((ROW_ID.to_string(), row_id));
|
||||
Ok(RecordBatch::try_from_iter(columns)?)
|
||||
}
|
||||
}
|
||||
|
||||
/// Expand `list_column` one row per element, repeating every other column
|
||||
/// for each element; an empty or null list contributes no rows. The list
|
||||
/// column is replaced by its element type, so a projection reads the
|
||||
/// element's fields as `alias.field` after this. This is the row-cardinality
|
||||
/// step of an `expanded_select` view, applied per batch on the scan stream.
|
||||
fn unnest_batch(batch: &RecordBatch, list_column: &str) -> Result<RecordBatch> {
|
||||
use arrow_array::{Array, ListArray, UInt32Array};
|
||||
use arrow_select::take::take;
|
||||
|
||||
let (list_index, _) = batch
|
||||
.schema()
|
||||
.column_with_name(list_column)
|
||||
.ok_or_else(|| Error::Runtime {
|
||||
message: format!("expansion column '{list_column}' is not in the batch"),
|
||||
})?;
|
||||
let list = batch
|
||||
.column(list_index)
|
||||
.as_any()
|
||||
.downcast_ref::<ListArray>()
|
||||
.ok_or_else(|| Error::Runtime {
|
||||
message: format!(
|
||||
"expansion column '{list_column}' is {}, not a list",
|
||||
batch.column(list_index).data_type()
|
||||
),
|
||||
})?;
|
||||
|
||||
// One take index per element, naming the source row it came from. A null
|
||||
// list has no elements; its offsets are equal, so it repeats nothing.
|
||||
let offsets = list.value_offsets();
|
||||
let mut repeat = Vec::with_capacity(list.values().len());
|
||||
for row in 0..list.len() {
|
||||
if list.is_valid(row) {
|
||||
let count = (offsets[row + 1] - offsets[row]) as usize;
|
||||
repeat.extend(std::iter::repeat_n(row as u32, count));
|
||||
}
|
||||
}
|
||||
let repeat = UInt32Array::from(repeat);
|
||||
// The flattened elements, in the same order as `repeat`: only the ranges
|
||||
// valid rows cover, so a null row's stale range (if any) is skipped.
|
||||
let elements = {
|
||||
let mut ranges = Vec::new();
|
||||
for row in 0..list.len() {
|
||||
if list.is_valid(row) {
|
||||
ranges.extend((offsets[row] as u32)..(offsets[row + 1] as u32));
|
||||
}
|
||||
}
|
||||
take(list.values().as_ref(), &UInt32Array::from(ranges), None)?
|
||||
};
|
||||
|
||||
let mut fields = Vec::with_capacity(batch.num_columns());
|
||||
let mut columns = Vec::with_capacity(batch.num_columns());
|
||||
for (index, field) in batch.schema().fields().iter().enumerate() {
|
||||
if index == list_index {
|
||||
let element = match field.data_type() {
|
||||
arrow_schema::DataType::List(element) => element.clone(),
|
||||
other => {
|
||||
return Err(Error::Runtime {
|
||||
message: format!("expansion column '{list_column}' is {other}, not a list"),
|
||||
});
|
||||
}
|
||||
};
|
||||
fields.push(Arc::new(
|
||||
arrow_schema::Field::new(field.name(), element.data_type().clone(), true)
|
||||
.with_metadata(field.metadata().clone()),
|
||||
));
|
||||
columns.push(elements.clone());
|
||||
} else {
|
||||
fields.push(field.clone());
|
||||
columns.push(take(batch.column(index).as_ref(), &repeat, None)?);
|
||||
}
|
||||
}
|
||||
let schema = Arc::new(ArrowSchema::new_with_metadata(
|
||||
fields,
|
||||
batch.schema().metadata().clone(),
|
||||
));
|
||||
Ok(RecordBatch::try_new(schema, columns)?)
|
||||
}
|
||||
|
||||
/// Commit the view's removals and additions as one change, on the exact
|
||||
/// generation the refresh planned from. Lance rejects an overlapping
|
||||
/// provenance key, but an unrelated write to the view is not a key conflict,
|
||||
@@ -1672,6 +1956,247 @@ mod tests {
|
||||
(conn, source, view)
|
||||
}
|
||||
|
||||
/// The per-batch expansion behind an `expanded_select` view: every
|
||||
/// element of the list column becomes a row, the other columns repeat
|
||||
/// for each, and an empty or null list contributes no rows at all --
|
||||
/// which is exactly "zero rows out" for a table-valued function.
|
||||
/// Four documents with a `list<struct<chunk, ordinal>>` column named
|
||||
/// `c`: doc 1 has two chunks, doc 2 none, doc 3 a null list, doc 4 one.
|
||||
fn chunked_batch() -> RecordBatch {
|
||||
use arrow_array::builder::{Int32Builder, ListBuilder, StringBuilder, StructBuilder};
|
||||
use arrow_array::{ArrayRef, Int64Array};
|
||||
use arrow_schema::{DataType, Field, Fields};
|
||||
|
||||
let element_fields = Fields::from(vec![
|
||||
Field::new("chunk", DataType::Utf8, true),
|
||||
Field::new("ordinal", DataType::Int32, true),
|
||||
]);
|
||||
let mut list = ListBuilder::new(StructBuilder::new(
|
||||
element_fields,
|
||||
vec![
|
||||
Box::new(StringBuilder::new()),
|
||||
Box::new(Int32Builder::new()),
|
||||
],
|
||||
));
|
||||
for chunks in [Some(vec!["a", "b"]), Some(vec![]), None, Some(vec!["c"])] {
|
||||
match chunks {
|
||||
Some(chunks) => {
|
||||
for (i, c) in chunks.iter().enumerate() {
|
||||
let s = list.values();
|
||||
s.field_builder::<StringBuilder>(0).unwrap().append_value(c);
|
||||
s.field_builder::<Int32Builder>(1)
|
||||
.unwrap()
|
||||
.append_value(i as i32);
|
||||
s.append(true);
|
||||
}
|
||||
list.append(true);
|
||||
}
|
||||
None => list.append(false),
|
||||
}
|
||||
}
|
||||
let meta = arrow_array::StructArray::from(vec![(
|
||||
Arc::new(Field::new("title", DataType::Utf8, true)),
|
||||
Arc::new(arrow_array::StringArray::from(vec!["t1", "t2", "t3", "t4"])) as ArrayRef,
|
||||
)]);
|
||||
RecordBatch::try_from_iter(vec![
|
||||
(
|
||||
"id",
|
||||
Arc::new(Int64Array::from(vec![1, 2, 3, 4])) as ArrayRef,
|
||||
),
|
||||
("meta", Arc::new(meta) as ArrayRef),
|
||||
("c", Arc::new(list.finish()) as ArrayRef),
|
||||
])
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unnest_repeats_siblings_per_element_and_drops_empty_lists() {
|
||||
use arrow_array::{StringArray, StructArray};
|
||||
use arrow_schema::{DataType, Field, Fields};
|
||||
|
||||
let batch = chunked_batch();
|
||||
let element_fields = Fields::from(vec![
|
||||
Field::new("chunk", DataType::Utf8, true),
|
||||
Field::new("ordinal", DataType::Int32, true),
|
||||
]);
|
||||
let out = unnest_batch(&batch, "c").unwrap();
|
||||
assert_eq!(out.num_rows(), 3, "{out:?}");
|
||||
let ids: Vec<i64> = out["id"]
|
||||
.as_primitive::<arrow_array::types::Int64Type>()
|
||||
.values()
|
||||
.to_vec();
|
||||
assert_eq!(ids, [1, 1, 4]);
|
||||
let element = out["c"].as_any().downcast_ref::<StructArray>().unwrap();
|
||||
let chunks: Vec<&str> = element
|
||||
.column(0)
|
||||
.as_any()
|
||||
.downcast_ref::<StringArray>()
|
||||
.unwrap()
|
||||
.iter()
|
||||
.flatten()
|
||||
.collect();
|
||||
assert_eq!(chunks, ["a", "b", "c"]);
|
||||
let ordinals: Vec<i32> = element
|
||||
.column(1)
|
||||
.as_primitive::<arrow_array::types::Int32Type>()
|
||||
.values()
|
||||
.to_vec();
|
||||
assert_eq!(ordinals, [0, 1, 0]);
|
||||
// the element column is now the struct itself, not a list of it
|
||||
assert_eq!(
|
||||
out.schema().field_with_name("c").unwrap().data_type(),
|
||||
&DataType::Struct(element_fields)
|
||||
);
|
||||
}
|
||||
|
||||
/// A Function in FROM position is refreshed from its staging table: the
|
||||
/// query on the view stays the one the user wrote, the staging binding
|
||||
/// says which table and list column refresh reads, and rows come out one
|
||||
/// per element as with UNNEST. Without a staging, a local database
|
||||
/// refuses the query rather than guess.
|
||||
#[tokio::test]
|
||||
async fn a_function_in_from_position_refreshes_from_its_staging() {
|
||||
use arrow_array::StringArray;
|
||||
|
||||
let conn = connect("memory://").execute().await.unwrap();
|
||||
let staging = conn
|
||||
.create_table("docs__chunk", chunked_batch())
|
||||
.write_options(crate::materialized_view::tests::stable_row_ids())
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
let query =
|
||||
"SELECT id AS doc, e.chunk AS text, e.ordinal FROM docs, chunk(meta.title, 2) AS e";
|
||||
let definition = MaterializedViewDefinition::from_sql(query).unwrap();
|
||||
|
||||
let err = crate::materialized_view::prepare_definition(&staging, definition.clone())
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::InvalidInput { message } if message.contains("reads 'docs'")),
|
||||
"{err:?}"
|
||||
);
|
||||
|
||||
let view = crate::materialized_view::prepare_staged_definition(&staging, definition, "c")
|
||||
.await
|
||||
.unwrap()
|
||||
.create("chunks")
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(view.definition().to_sql(), query);
|
||||
let metadata = view.table().schema().await.unwrap().metadata().clone();
|
||||
assert_eq!(
|
||||
crate::materialized_view::read_staging(&metadata).unwrap(),
|
||||
Some(crate::materialized_view::StagingBinding {
|
||||
table: "docs__chunk".into(),
|
||||
namespace: Vec::new(),
|
||||
column: "c".into(),
|
||||
})
|
||||
);
|
||||
|
||||
view.refresh().execute().await.unwrap();
|
||||
assert_eq!(read(view.table(), "ordinal").await, [0, 0, 1]);
|
||||
let batches = view
|
||||
.table()
|
||||
.query()
|
||||
.select(Select::columns(&["text"]))
|
||||
.execute()
|
||||
.await
|
||||
.unwrap()
|
||||
.try_collect::<Vec<_>>()
|
||||
.await
|
||||
.unwrap();
|
||||
let out = arrow_select::concat::concat_batches(&batches[0].schema(), &batches).unwrap();
|
||||
let texts: Vec<&str> = out["text"]
|
||||
.as_any()
|
||||
.downcast_ref::<StringArray>()
|
||||
.unwrap()
|
||||
.iter()
|
||||
.flatten()
|
||||
.collect();
|
||||
assert_eq!(texts, ["a", "b", "c"]);
|
||||
|
||||
// The reopened view reads the same logical query and refreshes
|
||||
// incrementally from the staging.
|
||||
staging.add(chunked_batch()).execute().await.unwrap();
|
||||
let reopened = conn.open_materialized_view("chunks").await.unwrap();
|
||||
assert_eq!(reopened.definition().to_sql(), query);
|
||||
let result = reopened.refresh().execute().await.unwrap();
|
||||
assert_eq!(result.mode, RefreshMode::Incremental);
|
||||
assert_eq!(read(reopened.table(), "ordinal").await, [0, 0, 0, 0, 1, 1]);
|
||||
}
|
||||
|
||||
/// An expanded view materializes one row per list element, with the
|
||||
/// projections reading the element through the alias and the other
|
||||
/// source columns repeated alongside; sources with no elements yield
|
||||
/// no rows. The lineage is the list column, so a change to it is what
|
||||
/// drives incremental refresh.
|
||||
#[tokio::test]
|
||||
async fn an_expanded_view_materializes_one_row_per_element() {
|
||||
use arrow_array::{Int64Array, StringArray};
|
||||
|
||||
let conn = connect("memory://").execute().await.unwrap();
|
||||
let source = conn
|
||||
.create_table("docs", chunked_batch())
|
||||
.write_options(crate::materialized_view::tests::stable_row_ids())
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
let mut view = crate::materialized_view::prepare_definition(
|
||||
&source,
|
||||
MaterializedViewDefinition::from_sql(
|
||||
"SELECT id AS doc, meta.title AS title, e.ordinal + 1 AS nth \
|
||||
FROM docs, UNNEST(c) AS e WHERE e.ordinal < 5",
|
||||
)
|
||||
.unwrap(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
// A computed column's input read through the alias is an element
|
||||
// field; the recorded source input is the list column.
|
||||
let text = view.input_column("e.chunk").unwrap();
|
||||
assert!(view.definition.lateral.is_some());
|
||||
let view = view.create("chunks").await.unwrap();
|
||||
view.refresh().execute().await.unwrap();
|
||||
|
||||
let batches = view
|
||||
.table()
|
||||
.query()
|
||||
.select(Select::columns(&["doc", &text, "title"]))
|
||||
.execute()
|
||||
.await
|
||||
.unwrap()
|
||||
.try_collect::<Vec<_>>()
|
||||
.await
|
||||
.unwrap();
|
||||
let out = arrow_select::concat::concat_batches(&batches[0].schema(), &batches).unwrap();
|
||||
let docs: Vec<i64> = out["doc"]
|
||||
.as_any()
|
||||
.downcast_ref::<Int64Array>()
|
||||
.unwrap()
|
||||
.values()
|
||||
.to_vec();
|
||||
let strings = |column: &str| -> Vec<String> {
|
||||
out[column]
|
||||
.as_any()
|
||||
.downcast_ref::<StringArray>()
|
||||
.unwrap()
|
||||
.iter()
|
||||
.flatten()
|
||||
.map(str::to_string)
|
||||
.collect()
|
||||
};
|
||||
assert_eq!(docs, [1, 1, 4]);
|
||||
assert_eq!(strings(&text), ["a", "b", "c"]);
|
||||
assert_eq!(strings("title"), ["t1", "t1", "t4"]);
|
||||
assert_eq!(read(view.table(), "nth").await, [1, 1, 2]);
|
||||
|
||||
// Appended documents expand incrementally; the existing rows stay.
|
||||
source.add(chunked_batch()).execute().await.unwrap();
|
||||
view.refresh().execute().await.unwrap();
|
||||
assert_eq!(read(view.table(), "nth").await, [1, 1, 1, 1, 2, 2]);
|
||||
}
|
||||
|
||||
async fn read(table: &Table, column: &str) -> Vec<i32> {
|
||||
let batches = table
|
||||
.query()
|
||||
@@ -1765,13 +2290,62 @@ mod tests {
|
||||
view.definition().filter.as_deref(),
|
||||
Some("`PartyAbbrev` = 'D'")
|
||||
);
|
||||
assert_eq!(view.definition().inputs, ["PartyAbbrev", "id"]);
|
||||
|
||||
let result = view.refresh().execute().await.unwrap();
|
||||
assert_eq!(result.rows_written, 2);
|
||||
assert_eq!(read(view.table(), "id").await, vec![1, 3]);
|
||||
}
|
||||
|
||||
/// A legacy layout whose query means what the canonical one means is
|
||||
/// rewritten in the current layout on the next refresh, without a
|
||||
/// rebuild: the rows it certified are the rows the query produces.
|
||||
#[tokio::test]
|
||||
async fn a_legacy_layout_with_the_same_meaning_is_rewritten_without_a_rebuild() {
|
||||
let (conn, source, view) = refreshed_doubled(vec![1]).await;
|
||||
let legacy = serde_json::json!({
|
||||
"kind": "select",
|
||||
"source_table": "src",
|
||||
"projections": [
|
||||
{"output": "x", "expression": "`x`"},
|
||||
{"output": "twice", "expression": "x*2"},
|
||||
],
|
||||
"inputs": ["x"],
|
||||
})
|
||||
.to_string();
|
||||
let native = view.table().as_native().unwrap();
|
||||
let mut dataset = native.dataset.get().await.unwrap().as_ref().clone();
|
||||
let predicted = dataset.version().version + 1;
|
||||
dataset
|
||||
.update_schema_metadata([
|
||||
(DEFINITION_META_KEY.to_string(), Some(legacy)),
|
||||
(
|
||||
VIEW_VERSION_META_KEY.to_string(),
|
||||
Some(predicted.to_string()),
|
||||
),
|
||||
])
|
||||
.await
|
||||
.unwrap();
|
||||
native.dataset.update(dataset);
|
||||
|
||||
append(&source, vec![2]).await;
|
||||
let reopened = conn.open_materialized_view("doubled").await.unwrap();
|
||||
let result = reopened.refresh().execute().await.unwrap();
|
||||
assert_eq!(result.mode, RefreshMode::Incremental);
|
||||
assert_eq!(read(reopened.table(), "twice").await, vec![2, 4]);
|
||||
let stored: serde_json::Value = serde_json::from_str(
|
||||
&reopened.table().schema().await.unwrap().metadata()[DEFINITION_META_KEY],
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
stored,
|
||||
serde_json::json!({
|
||||
"kind": "query",
|
||||
"format": 1,
|
||||
"query": "SELECT x, x * 2 AS twice FROM src",
|
||||
})
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_legacy_raw_filter_rebuilds_and_persists_canonical_definition() {
|
||||
let conn = connect("memory://").execute().await.unwrap();
|
||||
@@ -1797,18 +2371,20 @@ mod tests {
|
||||
|
||||
// Model a definition and up-to-date watermark written before filter
|
||||
// canonicalization was applied to materialized views.
|
||||
let mut legacy = view.definition().clone();
|
||||
legacy.filter = Some(r#""PartyAbbrev" = 'D'"#.into());
|
||||
legacy.inputs = vec!["id".into()];
|
||||
let legacy = serde_json::json!({
|
||||
"kind": "select",
|
||||
"source_table": "legacy_src",
|
||||
"projections": [{"output": "id", "expression": "id"}],
|
||||
"filter": r#""PartyAbbrev" = 'D'"#,
|
||||
"inputs": ["id"],
|
||||
})
|
||||
.to_string();
|
||||
let native = view.table().as_native().unwrap();
|
||||
let mut dataset = native.dataset.get().await.unwrap().as_ref().clone();
|
||||
let predicted = dataset.version().version + 1;
|
||||
dataset
|
||||
.update_schema_metadata([
|
||||
(
|
||||
DEFINITION_META_KEY.to_string(),
|
||||
Some(definition_to_metadata(&legacy).unwrap()),
|
||||
),
|
||||
(DEFINITION_META_KEY.to_string(), Some(legacy)),
|
||||
(
|
||||
VIEW_VERSION_META_KEY.to_string(),
|
||||
Some(predicted.to_string()),
|
||||
@@ -1831,7 +2407,6 @@ mod tests {
|
||||
migrated.definition().filter.as_deref(),
|
||||
Some("`PartyAbbrev` = 'D'")
|
||||
);
|
||||
assert_eq!(migrated.definition().inputs, ["PartyAbbrev", "id"]);
|
||||
assert_eq!(
|
||||
migrated.refresh().execute().await.unwrap().mode,
|
||||
RefreshMode::NoOp
|
||||
@@ -2758,7 +3333,11 @@ mod tests {
|
||||
source.drop_columns(&["x"]).await.unwrap();
|
||||
|
||||
let err = view.refresh().execute().await.unwrap_err();
|
||||
assert!(matches!(err, Error::Schema { message } if message.contains("'x'")));
|
||||
assert!(
|
||||
matches!(&err, Error::Schema { message }
|
||||
if message.contains("dropped or renamed") && message.contains("No field named x")),
|
||||
"{err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
/// A pinned refresh materializes the source as of `version`; catching up
|
||||
@@ -3063,7 +3642,7 @@ mod tests {
|
||||
],
|
||||
filter: None,
|
||||
limit: None,
|
||||
inputs: vec!["x".into()],
|
||||
lateral: None,
|
||||
};
|
||||
let mut metadata = HashMap::new();
|
||||
metadata.insert(
|
||||
@@ -3097,7 +3676,7 @@ mod tests {
|
||||
}],
|
||||
filter: None,
|
||||
limit: None,
|
||||
inputs: vec!["x".into()],
|
||||
lateral: None,
|
||||
};
|
||||
let mut metadata = HashMap::new();
|
||||
metadata.insert(
|
||||
|
||||
@@ -448,7 +448,7 @@ enum BodyLogging {
|
||||
/// always splits back into the parts that made it. The configuration field
|
||||
/// exists because the identifier grammar comes from the Lance REST catalog
|
||||
/// standard, which carries a delimiter setting for other catalogs to adopt.
|
||||
pub(crate) const ID_DELIMITER: &str = "$";
|
||||
pub const ID_DELIMITER: &str = "$";
|
||||
|
||||
fn validate_id_delimiter(delimiter: &str) -> Result<()> {
|
||||
if delimiter != ID_DELIMITER {
|
||||
|
||||
@@ -856,25 +856,28 @@ impl<S: HttpSend> Database for RemoteDatabase<S> {
|
||||
let response = self.client.check_response(&request_id, response).await?;
|
||||
let status = response.status();
|
||||
let body = response.text().await.err_to_http(request_id.clone())?;
|
||||
if status != StatusCode::ACCEPTED {
|
||||
return Err(Error::Http {
|
||||
source: "materialized-view drop must return 202 Accepted".into(),
|
||||
request_id,
|
||||
status_code: Some(status),
|
||||
});
|
||||
}
|
||||
let job_id = extract_job_id(&body).ok_or_else(|| Error::Http {
|
||||
source: "materialized-view drop response did not contain a valid job_id".into(),
|
||||
request_id,
|
||||
status_code: Some(status),
|
||||
})?;
|
||||
let job = match status {
|
||||
StatusCode::OK => Job::new_done(),
|
||||
StatusCode::ACCEPTED => {
|
||||
let job_id = extract_job_id(&body).ok_or_else(|| Error::Http {
|
||||
source: "materialized-view drop response did not contain a valid job_id".into(),
|
||||
request_id,
|
||||
status_code: Some(status),
|
||||
})?;
|
||||
Job::new(Box::new(RemoteJob::new(self.client.clone(), job_id)))
|
||||
}
|
||||
_ => {
|
||||
return Err(Error::Http {
|
||||
source: "materialized-view drop must return 200 OK or 202 Accepted".into(),
|
||||
request_id,
|
||||
status_code: Some(status),
|
||||
});
|
||||
}
|
||||
};
|
||||
self.table_cache
|
||||
.remove(&build_cache_key(name, namespace_path))
|
||||
.await;
|
||||
Ok(Job::new(Box::new(RemoteJob::new(
|
||||
self.client.clone(),
|
||||
job_id,
|
||||
))))
|
||||
Ok(job)
|
||||
}
|
||||
|
||||
async fn list_materialized_views(&self, namespace_path: &[String]) -> Result<Vec<String>> {
|
||||
@@ -1780,6 +1783,50 @@ mod tests {
|
||||
assert_eq!(job.id(), Some("j1-mv-drop"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_drop_materialized_view_completed_inline() {
|
||||
let db = super::RemoteDatabase::new_mock(|request| {
|
||||
assert_eq!(request.method(), "POST");
|
||||
assert_eq!(request.url().path(), "/v1/materialized_view/adults/drop");
|
||||
http::Response::builder().status(200).body("{}").unwrap()
|
||||
});
|
||||
let job = db
|
||||
.drop_materialized_view_async("adults", &[])
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(job.id(), None);
|
||||
assert_eq!(job.status().await.unwrap(), "finished");
|
||||
job.wait().await.unwrap();
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_drop_materialized_view_rejects_incomplete_acceptance() {
|
||||
for body in ["{}", r#"{"job_id":""}"#, r#"{"job_id":null}"#] {
|
||||
let db = super::RemoteDatabase::new_mock(move |_| {
|
||||
http::Response::builder().status(202).body(body).unwrap()
|
||||
});
|
||||
let error = db
|
||||
.drop_materialized_view_async("adults", &[])
|
||||
.await
|
||||
.err()
|
||||
.unwrap();
|
||||
assert!(error.to_string().contains("valid job_id"));
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_drop_materialized_view_rejects_unexpected_success_status() {
|
||||
let db = super::RemoteDatabase::new_mock(|_| {
|
||||
http::Response::builder().status(204).body("").unwrap()
|
||||
});
|
||||
let error = db
|
||||
.drop_materialized_view_async("adults", &[])
|
||||
.await
|
||||
.err()
|
||||
.unwrap();
|
||||
assert!(error.to_string().contains("200 OK or 202 Accepted"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_list_materialized_views_follows_empty_pages() {
|
||||
let page = Arc::new(AtomicUsize::new(0));
|
||||
|
||||
@@ -2099,8 +2099,10 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
|
||||
filter: Option<String>,
|
||||
#[serde(default)]
|
||||
limit: Option<u64>,
|
||||
/// The defining query, once the server describes a view by it;
|
||||
/// takes precedence over the structured fields.
|
||||
#[serde(default)]
|
||||
inputs: Vec<String>,
|
||||
query: Option<String>,
|
||||
#[serde(default)]
|
||||
incarnation: Option<String>,
|
||||
}
|
||||
@@ -2113,10 +2115,12 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
|
||||
let response = self.check_table_response(&request_id, response).await?;
|
||||
let response: DescribeMaterializedViewResponse =
|
||||
response.json().await.err_to_http(request_id)?;
|
||||
Ok(MaterializedViewInfo {
|
||||
definition: MaterializedViewDefinition {
|
||||
let definition = match response.query {
|
||||
Some(query) => MaterializedViewDefinition::from_sql(&query)?,
|
||||
None => MaterializedViewDefinition {
|
||||
source_table: response.source_table,
|
||||
source_namespace: response.source_namespace,
|
||||
lateral: None,
|
||||
projections: response
|
||||
.projections
|
||||
.into_iter()
|
||||
@@ -2127,8 +2131,10 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
|
||||
.collect(),
|
||||
filter: response.filter,
|
||||
limit: response.limit,
|
||||
inputs: response.inputs,
|
||||
},
|
||||
};
|
||||
Ok(MaterializedViewInfo {
|
||||
definition,
|
||||
incarnation: response.incarnation,
|
||||
})
|
||||
}
|
||||
@@ -3314,9 +3320,10 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
|
||||
_read_columns: Option<Vec<String>>,
|
||||
) -> Result<AddColumnsResult> {
|
||||
self.check_mutable().await?;
|
||||
crate::table::computed_columns::ensure_no_function_bindings_for_mutation(
|
||||
crate::table::computed_columns::ensure_not_function_bound(
|
||||
self.schema().await?.as_ref(),
|
||||
"schema evolution",
|
||||
crate::table::schema_evolution::new_column_names(&transforms),
|
||||
)?;
|
||||
match transforms {
|
||||
NewColumnTransform::SqlExpressions(expressions) => {
|
||||
@@ -3369,9 +3376,10 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
|
||||
|
||||
async fn add_computed_columns(&self, columns: &[(String, String)]) -> Result<AddColumnsResult> {
|
||||
self.check_mutable().await?;
|
||||
crate::table::computed_columns::ensure_no_function_bindings_for_mutation(
|
||||
crate::table::computed_columns::ensure_not_function_bound(
|
||||
self.schema().await?.as_ref(),
|
||||
"schema evolution",
|
||||
columns.iter().map(|(name, _)| name),
|
||||
)?;
|
||||
// The server plans the declaration against its table schema, including
|
||||
// Blob v2 semantics inherited by a direct field projection.
|
||||
@@ -3519,6 +3527,35 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
|
||||
})))
|
||||
}
|
||||
|
||||
async fn function_errors(
|
||||
&self,
|
||||
request: &crate::function::FunctionErrorsRequest,
|
||||
) -> Result<crate::function::FunctionErrors> {
|
||||
let mut body = serde_json::json!({});
|
||||
if let Some(job_id) = &request.job_id {
|
||||
body["job_id"] = serde_json::json!(job_id);
|
||||
}
|
||||
if let Some(column) = &request.column {
|
||||
body["column"] = serde_json::json!(column);
|
||||
}
|
||||
if let Some(limit) = request.limit {
|
||||
body["limit"] = serde_json::json!(limit);
|
||||
}
|
||||
self.apply_branch_body(&mut body);
|
||||
let request = self
|
||||
.client
|
||||
.post(&format!("/v1/table/{}/errors", self.identifier))
|
||||
.json(&body);
|
||||
let (request_id, response) = self.send(request, true).await?;
|
||||
let response = self.check_table_response(&request_id, response).await?;
|
||||
let body = response.text().await.err_to_http(request_id.clone())?;
|
||||
serde_json::from_str(&body).map_err(|e| Error::Http {
|
||||
source: format!("Failed to parse errors response: {}", e).into(),
|
||||
request_id,
|
||||
status_code: None,
|
||||
})
|
||||
}
|
||||
|
||||
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult> {
|
||||
self.check_mutable().await?;
|
||||
let body = alterations
|
||||
@@ -7953,8 +7990,9 @@ mod tests {
|
||||
assert_eq!(result.version, 8);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_add_function_column_allows_an_existing_binding() {
|
||||
/// The fixture binding's table: `title` and `body` bound as inputs, its
|
||||
/// two outputs declared, plus an unbound `spare`.
|
||||
fn fixture_bound_schema() -> Schema {
|
||||
let binding = crate::function::FunctionBinding::from_json(include_str!(
|
||||
"../../tests/fixtures/first_class_functions/v1/remote_function_binding.json"
|
||||
))
|
||||
@@ -7981,13 +8019,78 @@ mod tests {
|
||||
),
|
||||
)
|
||||
}));
|
||||
let schema = Schema::new_with_metadata(
|
||||
fields.push(Field::new("spare", DataType::Int32, true));
|
||||
Schema::new_with_metadata(
|
||||
fields,
|
||||
HashMap::from([(
|
||||
crate::table::computed_columns::FUNCTION_BINDINGS_META_KEY.to_string(),
|
||||
binding_metadata,
|
||||
)]),
|
||||
);
|
||||
)
|
||||
}
|
||||
|
||||
/// Only a column the binding uses is refused, and it is refused before
|
||||
/// any request goes out; the rest reach the server as usual.
|
||||
#[tokio::test]
|
||||
async fn test_add_columns_scopes_to_the_columns_a_binding_uses() {
|
||||
let table = Table::new_with_handler("my_table", |request| match request.url().path() {
|
||||
"/v1/table/my_table/describe/" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(describe_response(&fixture_bound_schema()))
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/add_columns/" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"version":10}"#.to_string())
|
||||
.unwrap(),
|
||||
path => panic!("Unexpected path: {path}"),
|
||||
});
|
||||
table
|
||||
.add_columns()
|
||||
.computed("doubled", "spare * 2")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
table
|
||||
.add_columns()
|
||||
.transform(NewColumnTransform::SqlExpressions(vec![(
|
||||
"eager".into(),
|
||||
"spare + 1".into(),
|
||||
)]))
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let table = Table::new_with_handler("my_table", |request| match request.url().path() {
|
||||
"/v1/table/my_table/describe/" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(describe_response(&fixture_bound_schema()))
|
||||
.unwrap(),
|
||||
path => panic!("mutation request must not be sent: {path}"),
|
||||
});
|
||||
for name in ["title", "search_text"] {
|
||||
let err = table
|
||||
.add_columns()
|
||||
.computed(name, "1")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, Error::InvalidInput { .. }), "{err:?}");
|
||||
let err = table
|
||||
.add_columns()
|
||||
.transform(NewColumnTransform::SqlExpressions(vec![(
|
||||
name.into(),
|
||||
"1".into(),
|
||||
)]))
|
||||
.execute()
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, Error::InvalidInput { .. }), "{err:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_add_function_column_allows_an_existing_binding() {
|
||||
let schema = fixture_bound_schema();
|
||||
let table =
|
||||
Table::new_with_handler("my_table", move |request| match request.url().path() {
|
||||
"/v1/table/my_table/describe/" => http::Response::builder()
|
||||
@@ -8188,6 +8291,88 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
/// The error listing is table-addressed with optional job and column
|
||||
/// filters, mirroring the server's SQL surface, and the two non-record
|
||||
/// signals come back as their own fields rather than as rows.
|
||||
#[tokio::test]
|
||||
async fn test_function_errors_lists_the_rows_a_refresh_skipped() {
|
||||
use crate::function::{FunctionErrorFragment, FunctionErrorRecord, FunctionErrorsRequest};
|
||||
|
||||
let table = Table::new_with_handler("my_table", |request| {
|
||||
assert_eq!(request.method(), "POST");
|
||||
assert_eq!(request.url().path(), "/v1/table/my_table/errors");
|
||||
let body = request.body().unwrap().as_bytes().unwrap();
|
||||
let value: serde_json::Value = serde_json::from_slice(body).unwrap();
|
||||
assert_eq!(
|
||||
value,
|
||||
serde_json::json!({"job_id": "j-7", "column": "embedding", "limit": 2})
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(
|
||||
r#"{"records": [{"job_id": "j-7", "fragment_id": 3, "row_offset": 9,
|
||||
"column": "embedding", "function": "embed", "function_version": "2",
|
||||
"table_version": 11, "error_type": "ValueError",
|
||||
"error_message": "bad input 'x'", "created_at_millis": 1700000000000}],
|
||||
"fragments": [{"job_id": "j-7", "fragment_id": 4, "rows_skipped": 500,
|
||||
"rows_recorded": 100}],
|
||||
"truncated": true}"#,
|
||||
)
|
||||
.unwrap()
|
||||
});
|
||||
|
||||
let errors = table
|
||||
.function_errors(
|
||||
FunctionErrorsRequest::new()
|
||||
.job_id("j-7")
|
||||
.column("embedding")
|
||||
.limit(2),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
errors.records,
|
||||
[FunctionErrorRecord {
|
||||
job_id: "j-7".into(),
|
||||
fragment_id: 3,
|
||||
row_offset: Some(9),
|
||||
column: "embedding".into(),
|
||||
function: "embed".into(),
|
||||
function_version: "2".into(),
|
||||
table_version: 11,
|
||||
error_type: "ValueError".into(),
|
||||
error_message: "bad input 'x'".into(),
|
||||
created_at_millis: 1_700_000_000_000,
|
||||
}]
|
||||
);
|
||||
assert_eq!(
|
||||
errors.fragments,
|
||||
[FunctionErrorFragment {
|
||||
job_id: "j-7".into(),
|
||||
fragment_id: 4,
|
||||
rows_skipped: 500,
|
||||
rows_recorded: 100,
|
||||
}]
|
||||
);
|
||||
assert!(errors.truncated);
|
||||
|
||||
// No filter sends no filter, and an empty listing reads as such.
|
||||
let table = Table::new_with_handler("my_table", |request| {
|
||||
let body = request.body().unwrap().as_bytes().unwrap();
|
||||
let value: serde_json::Value = serde_json::from_slice(body).unwrap();
|
||||
assert_eq!(value, serde_json::json!({}));
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"records": []}"#)
|
||||
.unwrap()
|
||||
});
|
||||
let errors = table
|
||||
.function_errors(FunctionErrorsRequest::new())
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(errors, crate::function::FunctionErrors::default());
|
||||
}
|
||||
|
||||
/// The refresh handle is wrapped for read-freshness tracking, so it has to
|
||||
/// forward the detail APIs too -- this is the job an operator is holding
|
||||
/// when a backfill goes quiet.
|
||||
@@ -12269,7 +12454,6 @@ mod tests {
|
||||
let view = crate::MaterializedView::from_table(table).await.unwrap();
|
||||
assert_eq!(view.definition().source_table, "source");
|
||||
assert_eq!(view.definition().source_namespace, ["analytics"]);
|
||||
assert_eq!(view.definition().inputs, ["x"]);
|
||||
assert_eq!(view.incarnation(), Some("inc-1"));
|
||||
|
||||
let result = view
|
||||
|
||||
@@ -53,6 +53,7 @@ use crate::database::Database;
|
||||
use crate::database::read_freshness::TableFreshness;
|
||||
use crate::embeddings::{EmbeddingDefinition, EmbeddingRegistry, MemoryRegistry};
|
||||
use crate::error::{Error, Result};
|
||||
use crate::function::FunctionErrorsRequest;
|
||||
use crate::index::IndexStatistics;
|
||||
use crate::index::{Index, IndexBuilder};
|
||||
use crate::index::{IndexConfig, IndexStatisticsImpl, IndexType};
|
||||
@@ -823,6 +824,17 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
|
||||
message: "computed columns are supported only on local tables".into(),
|
||||
})
|
||||
}
|
||||
/// The per-row errors Function refreshes recorded on this table; see
|
||||
/// [`Table::function_errors`]. The default returns `NotSupported`.
|
||||
async fn function_errors(
|
||||
&self,
|
||||
_request: &crate::function::FunctionErrorsRequest,
|
||||
) -> Result<crate::function::FunctionErrors> {
|
||||
Err(Error::NotSupported {
|
||||
message: "per-row Function errors are recorded only on LanceDB Cloud and Enterprise"
|
||||
.into(),
|
||||
})
|
||||
}
|
||||
/// Alter columns in the table.
|
||||
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult>;
|
||||
/// Drop columns from the table.
|
||||
@@ -1852,6 +1864,36 @@ impl Table {
|
||||
self.inner.refresh_column_async(column.as_ref()).await
|
||||
}
|
||||
|
||||
/// The per-row errors Function refreshes recorded on this table: the
|
||||
/// rows a refresh skipped under its skip policy, with the failing input
|
||||
/// and the error, plus a summary for any fragment whose detail was
|
||||
/// capped. Filter by job or column through the request; a listing that
|
||||
/// hit its limit reports [`FunctionErrors::truncated`].
|
||||
///
|
||||
/// LanceDB Cloud and Enterprise only, and the caller needs read access
|
||||
/// to the table, since a message carries the value that failed.
|
||||
///
|
||||
/// ```
|
||||
/// # use lancedb::Table;
|
||||
/// use lancedb::function::FunctionErrorsRequest;
|
||||
///
|
||||
/// # async fn list_errors(table: &Table) -> Result<(), Box<dyn std::error::Error>> {
|
||||
/// let errors = table
|
||||
/// .function_errors(FunctionErrorsRequest::new().column("embedding"))
|
||||
/// .await?;
|
||||
/// for record in &errors.records {
|
||||
/// println!("{}: {}", record.error_type, record.error_message);
|
||||
/// }
|
||||
/// # Ok(())
|
||||
/// # }
|
||||
/// ```
|
||||
pub async fn function_errors(
|
||||
&self,
|
||||
request: FunctionErrorsRequest,
|
||||
) -> Result<crate::function::FunctionErrors> {
|
||||
self.inner.function_errors(&request).await
|
||||
}
|
||||
|
||||
/// Change a column's name or nullability.
|
||||
pub async fn alter_columns(
|
||||
&self,
|
||||
@@ -4051,6 +4093,31 @@ mod tests {
|
||||
assert_eq!(table.name, "test")
|
||||
}
|
||||
|
||||
/// The per-row error store is a server feature; a local table says so
|
||||
/// rather than answering with an empty listing.
|
||||
#[tokio::test]
|
||||
async fn test_function_errors_are_remote_only() {
|
||||
let tmp_dir = tempdir().unwrap();
|
||||
let conn = connect(tmp_dir.path().to_str().unwrap())
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
let batch = make_test_batches();
|
||||
let table = conn
|
||||
.create_table("t", batch.clone())
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
let err = table
|
||||
.function_errors(FunctionErrorsRequest::new())
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::NotSupported { message } if message.contains("Cloud and Enterprise")),
|
||||
"{err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_open_not_found() {
|
||||
let tmp_dir = tempdir().unwrap();
|
||||
|
||||
@@ -391,6 +391,9 @@ pub(crate) fn ensure_supported_function_metadata(schema: &ArrowSchema) -> Result
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Refuse `operation` outright on a table with a Function binding. For
|
||||
/// operations that cannot say which columns they touch; the others use
|
||||
/// [`ensure_not_function_bound`].
|
||||
pub(crate) fn ensure_no_function_bindings_for_mutation(
|
||||
schema: &ArrowSchema,
|
||||
operation: &str,
|
||||
@@ -406,6 +409,49 @@ pub(crate) fn ensure_no_function_bindings_for_mutation(
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Refuse `operation` only when a path in `touched` names a column a Function
|
||||
/// binding depends on: an input's root, an output, or the assignment column.
|
||||
/// A binding stores those columns' exact Arrow fields, so editing one strands it.
|
||||
pub(crate) fn ensure_not_function_bound<S: AsRef<str>>(
|
||||
schema: &ArrowSchema,
|
||||
operation: &str,
|
||||
touched: impl IntoIterator<Item = S>,
|
||||
) -> Result<()> {
|
||||
ensure_supported_function_metadata(schema)?;
|
||||
let mut protected = BTreeSet::new();
|
||||
for binding in function_bindings(schema)? {
|
||||
for input in binding.inputs() {
|
||||
protected.insert(field_root(&input.field_path)?);
|
||||
}
|
||||
protected.extend(
|
||||
binding
|
||||
.outputs()
|
||||
.iter()
|
||||
.map(|output| output.output_name.clone()),
|
||||
);
|
||||
protected.extend(
|
||||
binding
|
||||
.assignment()
|
||||
.map(|assignment| assignment.output_name.clone()),
|
||||
);
|
||||
}
|
||||
if protected.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
for path in touched {
|
||||
let column = field_root(path.as_ref())?;
|
||||
if protected.contains(&column) {
|
||||
return Err(Error::InvalidInput {
|
||||
message: format!(
|
||||
"{operation} of '{column}' is not supported: a Function binding reads or \
|
||||
writes it"
|
||||
),
|
||||
});
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Read a field's computed-column declaration, if it carries one.
|
||||
///
|
||||
/// A field flagged computed but carrying no kind, or a SQL one missing its
|
||||
@@ -1330,7 +1376,7 @@ pub(crate) fn ensure_not_written<'a>(
|
||||
.map(|declaration| declaration.name)
|
||||
.collect();
|
||||
for name in written {
|
||||
if declared.iter().any(|declared| declared == root(name)) {
|
||||
if declared.iter().any(|declared| *declared == root(name)) {
|
||||
return Err(Error::InvalidInput {
|
||||
message: format!(
|
||||
"column '{}' is computed; its values come from refresh and cannot be \
|
||||
@@ -1523,9 +1569,24 @@ pub(crate) fn ensure_not_retyped(schema: &ArrowSchema, paths: &[&str]) -> Result
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The top-level column a possibly nested input path reads.
|
||||
pub(crate) fn root(path: &str) -> &str {
|
||||
path.split('.').next().unwrap_or(path)
|
||||
/// The top-level column a path addresses, by the grammar lance resolves it
|
||||
/// with, so a quoted spelling names the same column as a bare one.
|
||||
pub(crate) fn field_root(path: &str) -> Result<String> {
|
||||
parse_field_path(path)
|
||||
.map_err(|e| Error::InvalidInput {
|
||||
message: format!("invalid column path '{path}': {e}"),
|
||||
})?
|
||||
.into_iter()
|
||||
.next()
|
||||
.ok_or_else(|| Error::InvalidInput {
|
||||
message: format!("column path '{path}' is empty"),
|
||||
})
|
||||
}
|
||||
|
||||
/// [`field_root`], falling back to the text before the first dot for a
|
||||
/// spelling lance would not resolve.
|
||||
pub(crate) fn root(path: &str) -> String {
|
||||
field_root(path).unwrap_or_else(|_| path.split('.').next().unwrap_or(path).to_string())
|
||||
}
|
||||
|
||||
/// A declaration's expression bound to a schema, ready to evaluate.
|
||||
@@ -1749,7 +1810,7 @@ pub(crate) fn bind(schema: SchemaRef, column: &str, expression: &str) -> Result<
|
||||
let mut indices = Vec::with_capacity(inputs.len());
|
||||
for input in &inputs {
|
||||
let index = runtime_schema
|
||||
.index_of(root(input))
|
||||
.index_of(&root(input))
|
||||
.map_err(|_| invalid(format!("unknown column '{input}'")))?;
|
||||
if !indices.contains(&index) {
|
||||
indices.push(index);
|
||||
@@ -1884,7 +1945,11 @@ pub(crate) fn plan(schema: SchemaRef, columns: &[(String, String)]) -> Result<Ve
|
||||
/// assert!(validate_declarations(schema, &[("c".into(), "random()".into())]).is_err());
|
||||
/// ```
|
||||
pub fn validate_declarations(schema: SchemaRef, columns: &[(String, String)]) -> Result<()> {
|
||||
ensure_no_function_bindings_for_mutation(schema.as_ref(), "schema evolution")?;
|
||||
ensure_not_function_bound(
|
||||
schema.as_ref(),
|
||||
"schema evolution",
|
||||
columns.iter().map(|(name, _)| name),
|
||||
)?;
|
||||
plan(schema, columns).map(drop)
|
||||
}
|
||||
|
||||
@@ -3197,6 +3262,72 @@ mod tests {
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
/// The scoped guard refuses exactly the columns a binding uses -- input
|
||||
/// roots, outputs and the assignment column -- and nothing else.
|
||||
#[test]
|
||||
fn test_function_bound_columns_are_the_only_ones_refused() {
|
||||
let mut raw_binding: Value = serde_json::from_str(include_str!(
|
||||
"../../tests/fixtures/first_class_functions/v1/remote_function_binding.json"
|
||||
))
|
||||
.unwrap();
|
||||
raw_binding["outputs"][0]["nullable"] = Value::Bool(true);
|
||||
raw_binding["outputs"][1]["nullable"] = Value::Bool(true);
|
||||
raw_binding["assignment"] = serde_json::json!({
|
||||
"output_name": "__function_assignment_fb_01K3TEXT",
|
||||
"output_field_id": -1,
|
||||
});
|
||||
raw_binding["output_schema"]["fields"]
|
||||
.as_array_mut()
|
||||
.unwrap()
|
||||
.push(serde_json::json!({
|
||||
"name": "__function_assignment_fb_01K3TEXT",
|
||||
"nullable": true,
|
||||
"type": {"type": "bool"},
|
||||
}));
|
||||
let binding: FunctionBinding = serde_json::from_value(raw_binding).unwrap();
|
||||
let mut fields = valid_function_binding_schema(true, true, &binding)
|
||||
.fields()
|
||||
.to_vec();
|
||||
fields.push(Arc::new(ArrowField::new("spare", DataType::Int32, true)));
|
||||
let schema = ArrowSchema::new_with_metadata(
|
||||
fields,
|
||||
HashMap::from([(
|
||||
FUNCTION_BINDINGS_META_KEY.to_string(),
|
||||
function_bindings_metadata(std::slice::from_ref(&binding)).unwrap(),
|
||||
)]),
|
||||
);
|
||||
|
||||
ensure_not_function_bound(
|
||||
&schema,
|
||||
"schema evolution",
|
||||
["spare", "spare.nested", "new", "`spare`", "`spare.nested`"],
|
||||
)
|
||||
.unwrap();
|
||||
for path in [
|
||||
"title",
|
||||
"body.nested",
|
||||
"search_text",
|
||||
"search_token_count",
|
||||
"__function_assignment_fb_01K3TEXT",
|
||||
"`title`",
|
||||
"`body`.nested",
|
||||
"`search_text`",
|
||||
] {
|
||||
let err = ensure_not_function_bound(&schema, "schema evolution", [path]).unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::InvalidInput { message }
|
||||
if message.contains("a Function binding reads or writes it")),
|
||||
"{path}: {err:?}"
|
||||
);
|
||||
}
|
||||
// A spelling lance cannot resolve is refused rather than compared as text.
|
||||
let err = ensure_not_function_bound(&schema, "schema evolution", ["`title"]).unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::InvalidInput { message } if message.contains("invalid column path")),
|
||||
"{err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_binding_preserves_all_nullable_outputs_with_an_assignment_column() {
|
||||
let mut raw_binding: Value = serde_json::from_str(include_str!(
|
||||
|
||||
@@ -33,11 +33,23 @@ use crate::{
|
||||
use arrow_schema::{DataType, Field};
|
||||
use lance_index::scalar::FullTextSearchQuery;
|
||||
|
||||
/// Datafusion attempts to maintain batch metadata
|
||||
/// An execution plan that erases Arrow schema-level metadata from its input's batches.
|
||||
///
|
||||
/// This is needless and it triggers bugs in DF. This operator erases metadata from the batches.
|
||||
/// DataFusion attempts to maintain batch metadata. This is needless and it triggers bugs in
|
||||
/// DF, so [`BaseTableAdapter::scan`] wraps every scan it produces in one of these.
|
||||
///
|
||||
/// ```
|
||||
/// use std::sync::Arc;
|
||||
///
|
||||
/// use datafusion_physical_plan::ExecutionPlan;
|
||||
/// use lancedb::table::datafusion::MetadataEraserExec;
|
||||
///
|
||||
/// # fn erase_metadata(scan: Arc<dyn ExecutionPlan>) -> Arc<dyn ExecutionPlan> {
|
||||
/// Arc::new(MetadataEraserExec::new(scan))
|
||||
/// # }
|
||||
/// ```
|
||||
#[derive(Debug)]
|
||||
struct MetadataEraserExec {
|
||||
pub struct MetadataEraserExec {
|
||||
input: Arc<dyn ExecutionPlan>,
|
||||
schema: Arc<ArrowSchema>,
|
||||
properties: Arc<PlanProperties>,
|
||||
@@ -62,7 +74,8 @@ impl MetadataEraserExec {
|
||||
)
|
||||
}
|
||||
|
||||
fn new(input: Arc<dyn ExecutionPlan>) -> Self {
|
||||
/// Wrap `input` in an operator that strips schema-level metadata from its batches.
|
||||
pub fn new(input: Arc<dyn ExecutionPlan>) -> Self {
|
||||
let schema = Arc::new(
|
||||
input
|
||||
.schema()
|
||||
|
||||
@@ -104,7 +104,7 @@ pub(crate) async fn set_lsm_write_spec(table: &NativeTable, spec: LsmWriteSpec)
|
||||
.into(),
|
||||
});
|
||||
}
|
||||
if crate::materialized_view::materialized_view_kind(&dataset.schema().metadata)?.is_some() {
|
||||
if crate::materialized_view::read_definition(&dataset.schema().metadata)?.is_some() {
|
||||
return Err(Error::NotSupported {
|
||||
message: "an LSM write spec cannot be installed on a materialized view: \
|
||||
rows in un-compacted tiers are invisible to refresh"
|
||||
|
||||
@@ -103,9 +103,10 @@ pub(crate) async fn execute_add_columns(
|
||||
transforms: NewColumnTransform,
|
||||
read_columns: Option<Vec<String>>,
|
||||
) -> Result<AddColumnsResult> {
|
||||
computed_columns::ensure_no_function_bindings_for_mutation(
|
||||
computed_columns::ensure_not_function_bound(
|
||||
table.schema().await?.as_ref(),
|
||||
"schema evolution",
|
||||
new_column_names(&transforms),
|
||||
)?;
|
||||
// Declarations are admitted only through [`execute_declare`].
|
||||
match &transforms {
|
||||
@@ -131,9 +132,10 @@ pub(crate) async fn execute_declare(
|
||||
// An LSM write spec keeps visible rows in tiers refresh cannot reach;
|
||||
// checked against latest committed state, not this handle's snapshot.
|
||||
table.checkout_latest().await?;
|
||||
computed_columns::ensure_no_function_bindings_for_mutation(
|
||||
computed_columns::ensure_not_function_bound(
|
||||
table.schema().await?.as_ref(),
|
||||
"schema evolution",
|
||||
columns.iter().map(|(name, _)| name),
|
||||
)?;
|
||||
// Unset drops the MemWAL index, so the spec alone stops describing a table
|
||||
// whose SSTables still hold rows. The shard directories outlive it and are
|
||||
@@ -156,6 +158,26 @@ pub(crate) async fn execute_declare(
|
||||
commit_add_columns(table, transform, None).await
|
||||
}
|
||||
|
||||
/// The top-level columns `transforms` adds.
|
||||
pub(crate) fn new_column_names(transforms: &NewColumnTransform) -> Vec<String> {
|
||||
let names = |schema: &ArrowSchema| {
|
||||
schema
|
||||
.fields()
|
||||
.iter()
|
||||
.map(|field| field.name().clone())
|
||||
.collect::<Vec<_>>()
|
||||
};
|
||||
match transforms {
|
||||
NewColumnTransform::SqlExpressions(expressions) => {
|
||||
expressions.iter().map(|(name, _)| name.clone()).collect()
|
||||
}
|
||||
NewColumnTransform::AllNulls(schema) => names(schema),
|
||||
NewColumnTransform::BatchUDF(udf) => names(&udf.output_schema),
|
||||
NewColumnTransform::Stream(stream) => names(&stream.schema()),
|
||||
NewColumnTransform::Reader(reader) => names(&reader.schema()),
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) async fn commit_add_columns(
|
||||
table: &NativeTable,
|
||||
transforms: NewColumnTransform,
|
||||
@@ -178,13 +200,18 @@ pub(crate) async fn execute_alter_columns(
|
||||
) -> Result<AlterColumnsResult> {
|
||||
table.dataset.ensure_mutable()?;
|
||||
let mut dataset = (*table.dataset.get().await?).clone();
|
||||
// Nullability is not part of what an expression resolves against, so only
|
||||
// a rename or a retype can invalidate a binding.
|
||||
let schema = std::sync::Arc::new(ArrowSchema::from(dataset.schema()));
|
||||
computed_columns::ensure_no_function_bindings_for_mutation(
|
||||
// A Function binding stores its columns' exact fields, nullability
|
||||
// included, so every alteration of one counts, and a rename's target too.
|
||||
computed_columns::ensure_not_function_bound(
|
||||
schema.as_ref(),
|
||||
"schema evolution",
|
||||
alterations.iter().flat_map(|alteration| {
|
||||
std::iter::once(alteration.path.as_str()).chain(alteration.rename.as_deref())
|
||||
}),
|
||||
)?;
|
||||
// Nullability is not part of what an expression resolves against, so only
|
||||
// a rename or a retype can invalidate a binding.
|
||||
let rebinding = alterations
|
||||
.iter()
|
||||
.filter(|alteration| alteration.rename.is_some() || alteration.data_type.is_some())
|
||||
@@ -212,14 +239,9 @@ pub(crate) async fn execute_drop_columns(
|
||||
) -> Result<DropColumnsResult> {
|
||||
table.dataset.ensure_mutable()?;
|
||||
let mut dataset = (*table.dataset.get().await?).clone();
|
||||
computed_columns::ensure_no_function_bindings_for_mutation(
|
||||
&ArrowSchema::from(dataset.schema()),
|
||||
"schema evolution",
|
||||
)?;
|
||||
computed_columns::ensure_not_an_input(
|
||||
&std::sync::Arc::new(ArrowSchema::from(dataset.schema())),
|
||||
columns,
|
||||
)?;
|
||||
let schema = std::sync::Arc::new(ArrowSchema::from(dataset.schema()));
|
||||
computed_columns::ensure_not_function_bound(schema.as_ref(), "schema evolution", columns)?;
|
||||
computed_columns::ensure_not_an_input(&schema, columns)?;
|
||||
dataset.drop_columns(columns).await?;
|
||||
let version = dataset.version().version;
|
||||
table.dataset.update(dataset);
|
||||
@@ -241,7 +263,11 @@ pub(crate) async fn execute_update_field_metadata(
|
||||
// binding out from under a refresh. A replace on a declared column would
|
||||
// silently erase it.
|
||||
let schema = ArrowSchema::from(dataset.schema());
|
||||
computed_columns::ensure_no_function_bindings_for_mutation(&schema, "schema evolution")?;
|
||||
computed_columns::ensure_not_function_bound(
|
||||
&schema,
|
||||
"field metadata update",
|
||||
updates.iter().map(|update| update.path.as_str()),
|
||||
)?;
|
||||
let declared: Vec<String> = computed_columns::computed_columns(&schema)
|
||||
.into_iter()
|
||||
.map(|declaration| declaration.name)
|
||||
@@ -263,7 +289,7 @@ pub(crate) async fn execute_update_field_metadata(
|
||||
if update.replace
|
||||
&& declared
|
||||
.iter()
|
||||
.any(|name| name == computed_columns::root(&update.path))
|
||||
.any(|name| *name == computed_columns::root(&update.path))
|
||||
{
|
||||
return Err(Error::InvalidInput {
|
||||
message: format!(
|
||||
@@ -300,8 +326,202 @@ mod tests {
|
||||
|
||||
use super::FieldMetadataUpdate;
|
||||
use crate::connect;
|
||||
use crate::function::FunctionBinding;
|
||||
use crate::query::{ExecutableQuery, QueryBase, Select};
|
||||
use crate::table::NewColumnTransform;
|
||||
use crate::table::computed_columns::{
|
||||
FUNCTION_BINDINGS_META_KEY, ensure_supported_function_metadata, function_bindings,
|
||||
function_bindings_metadata, function_computed_column_metadata,
|
||||
};
|
||||
use crate::{Error, Table};
|
||||
use std::collections::HashMap;
|
||||
|
||||
/// A table carrying the fixture binding: `title` and `body` are its
|
||||
/// inputs, `search_text` and `search_token_count` its outputs, `spare`
|
||||
/// nobody's. Stamped the way the server does it, since no local path
|
||||
/// declares a binding.
|
||||
async fn bound_table() -> Table {
|
||||
let conn = connect("memory://").execute().await.unwrap();
|
||||
let batch = record_batch!(
|
||||
("title", Utf8, ["a"]),
|
||||
("body", Utf8, ["b"]),
|
||||
("search_text", Utf8, ["a b"]),
|
||||
("search_token_count", Int64, [2]),
|
||||
("spare", Int32, [1])
|
||||
)
|
||||
.unwrap();
|
||||
let table = conn.create_table("bound", batch).execute().await.unwrap();
|
||||
let binding = FunctionBinding::from_json(include_str!(
|
||||
"../../tests/fixtures/first_class_functions/v1/remote_function_binding.json"
|
||||
))
|
||||
.unwrap();
|
||||
let native = table.as_native().unwrap();
|
||||
let mut dataset = native.dataset.get().await.unwrap().as_ref().clone();
|
||||
dataset
|
||||
.update_schema_metadata(vec![(
|
||||
FUNCTION_BINDINGS_META_KEY.to_string(),
|
||||
Some(function_bindings_metadata(std::slice::from_ref(&binding)).unwrap()),
|
||||
)])
|
||||
.await
|
||||
.unwrap();
|
||||
let inputs = ["title".to_string(), "body".to_string()];
|
||||
let outputs = binding
|
||||
.outputs()
|
||||
.iter()
|
||||
.map(|output| {
|
||||
(
|
||||
dataset.schema().field(&output.output_name).unwrap().id as u32,
|
||||
function_computed_column_metadata(
|
||||
binding.binding_id(),
|
||||
output.output_ordinal,
|
||||
&inputs,
|
||||
),
|
||||
)
|
||||
})
|
||||
.collect::<Vec<_>>();
|
||||
dataset.replace_field_metadata(outputs).await.unwrap();
|
||||
native.dataset.update(dataset);
|
||||
ensure_supported_function_metadata(&table.schema().await.unwrap()).unwrap();
|
||||
table
|
||||
}
|
||||
|
||||
fn metadata_update(path: &str) -> FieldMetadataUpdate {
|
||||
FieldMetadataUpdate {
|
||||
path: path.into(),
|
||||
metadata: HashMap::from([("unit".to_string(), Some("label".to_string()))]),
|
||||
replace: false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Columns no binding uses evolve as on any table, and the binding is
|
||||
/// still valid afterwards, which is what every later write checks.
|
||||
#[tokio::test]
|
||||
async fn test_schema_evolution_leaves_unbound_columns_free_on_a_bound_table() {
|
||||
let table = bound_table().await;
|
||||
table
|
||||
.add_columns()
|
||||
.transform(NewColumnTransform::SqlExpressions(vec![(
|
||||
"eager".into(),
|
||||
"1".into(),
|
||||
)]))
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
table
|
||||
.add_columns()
|
||||
.computed("derived", "spare * 2")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
table
|
||||
.update_field_metadata(&[metadata_update("eager")])
|
||||
.await
|
||||
.unwrap();
|
||||
table
|
||||
.alter_columns(&[ColumnAlteration::new("eager".into()).rename("moved".into())])
|
||||
.await
|
||||
.unwrap();
|
||||
table.drop_columns(&["moved"]).await.unwrap();
|
||||
|
||||
let schema = table.schema().await.unwrap();
|
||||
ensure_supported_function_metadata(&schema).unwrap();
|
||||
assert_eq!(function_bindings(&schema).unwrap().len(), 1);
|
||||
assert!(schema.field_with_name("derived").is_ok());
|
||||
assert!(schema.field_with_name("moved").is_err());
|
||||
}
|
||||
|
||||
fn bound(err: Error) {
|
||||
assert!(
|
||||
matches!(&err, Error::InvalidInput { message }
|
||||
if message.contains("a Function binding reads or writes it")),
|
||||
"{err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
/// Every schema-evolution door refuses a column a binding reads or
|
||||
/// writes, including a rename onto one.
|
||||
#[tokio::test]
|
||||
async fn test_schema_evolution_refuses_the_columns_a_function_binding_uses() {
|
||||
let table = bound_table().await;
|
||||
let version = table.version().await.unwrap();
|
||||
for column in ["title", "body", "search_text", "search_token_count"] {
|
||||
bound(table.drop_columns(&[column]).await.unwrap_err());
|
||||
bound(
|
||||
table
|
||||
.alter_columns(&[ColumnAlteration::new(column.into()).rename("moved".into())])
|
||||
.await
|
||||
.unwrap_err(),
|
||||
);
|
||||
bound(
|
||||
table
|
||||
.alter_columns(&[ColumnAlteration::new(column.into()).set_nullable(false)])
|
||||
.await
|
||||
.unwrap_err(),
|
||||
);
|
||||
bound(
|
||||
table
|
||||
.update_field_metadata(&[metadata_update(column)])
|
||||
.await
|
||||
.unwrap_err(),
|
||||
);
|
||||
bound(
|
||||
table
|
||||
.add_columns()
|
||||
.transform(NewColumnTransform::SqlExpressions(vec![(
|
||||
column.into(),
|
||||
"1".into(),
|
||||
)]))
|
||||
.execute()
|
||||
.await
|
||||
.unwrap_err(),
|
||||
);
|
||||
bound(
|
||||
table
|
||||
.add_columns()
|
||||
.computed(column, "1")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap_err(),
|
||||
);
|
||||
}
|
||||
bound(
|
||||
table
|
||||
.alter_columns(&[ColumnAlteration::new("spare".into()).rename("title".into())])
|
||||
.await
|
||||
.unwrap_err(),
|
||||
);
|
||||
assert_eq!(table.version().await.unwrap(), version);
|
||||
}
|
||||
|
||||
/// Lance resolves a quoted spelling to the same field as the bare one,
|
||||
/// so the guard compares identities, not text.
|
||||
#[tokio::test]
|
||||
async fn quoted_function_output_path_is_still_refused() {
|
||||
let table = bound_table().await;
|
||||
let version = table.version().await.unwrap();
|
||||
for path in ["`title`", "`search_text`", "`title`.nested"] {
|
||||
bound(table.drop_columns(&[path]).await.unwrap_err());
|
||||
bound(
|
||||
table
|
||||
.alter_columns(&[ColumnAlteration::new(path.into()).set_nullable(false)])
|
||||
.await
|
||||
.unwrap_err(),
|
||||
);
|
||||
bound(
|
||||
table
|
||||
.update_field_metadata(&[metadata_update(path)])
|
||||
.await
|
||||
.unwrap_err(),
|
||||
);
|
||||
}
|
||||
bound(
|
||||
table
|
||||
.alter_columns(&[ColumnAlteration::new("spare".into()).rename("`title`".into())])
|
||||
.await
|
||||
.unwrap_err(),
|
||||
);
|
||||
assert_eq!(table.version().await.unwrap(), version);
|
||||
}
|
||||
|
||||
// Add Columns Tests
|
||||
|
||||
|
||||
Reference in New Issue
Block a user