mirror of
https://github.com/lancedb/lancedb.git
synced 2026-09-12 16:22:24 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c463ca1503 | ||
|
|
79e9dffd06 | ||
|
|
c3efc320a6 | ||
|
|
8d6dea6313 | ||
|
|
5a2d3f39e2 | ||
|
|
219f41339d | ||
|
|
5982eebbb3 | ||
|
|
5a1c839382 | ||
|
|
e40e073a9d | ||
|
|
a47c22b26e |
Generated
+42
-42
@@ -3455,8 +3455,8 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "fsst"
|
name = "fsst"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
"rand 0.9.5",
|
"rand 0.9.5",
|
||||||
@@ -4815,8 +4815,8 @@ checksum = "e037a2e1d8d5fdbd49b16a4ea09d5d6401c1f29eca5ff29d03d3824dba16256a"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance"
|
name = "lance"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arc-swap",
|
"arc-swap",
|
||||||
"arrow",
|
"arrow",
|
||||||
@@ -4890,8 +4890,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-arrow"
|
name = "lance-arrow"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
"arrow-buffer",
|
"arrow-buffer",
|
||||||
@@ -4913,7 +4913,7 @@ dependencies = [
|
|||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-arrow-scalar"
|
name = "lance-arrow-scalar"
|
||||||
version = "58.0.0"
|
version = "58.0.0"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
"arrow-buffer",
|
"arrow-buffer",
|
||||||
@@ -4927,7 +4927,7 @@ dependencies = [
|
|||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-arrow-stats"
|
name = "lance-arrow-stats"
|
||||||
version = "58.0.0"
|
version = "58.0.0"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
"arrow-schema",
|
"arrow-schema",
|
||||||
@@ -4936,8 +4936,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-bitpacking"
|
name = "lance-bitpacking"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrayref",
|
"arrayref",
|
||||||
"crunchy",
|
"crunchy",
|
||||||
@@ -4947,8 +4947,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-core"
|
name = "lance-core"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
"arrow-buffer",
|
"arrow-buffer",
|
||||||
@@ -4988,8 +4988,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-datafusion"
|
name = "lance-datafusion"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow",
|
"arrow",
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
@@ -5019,8 +5019,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-datagen"
|
name = "lance-datagen"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow",
|
"arrow",
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
@@ -5037,8 +5037,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-derive"
|
name = "lance-derive"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"proc-macro2",
|
"proc-macro2",
|
||||||
"quote",
|
"quote",
|
||||||
@@ -5047,8 +5047,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-encoding"
|
name = "lance-encoding"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow-arith",
|
"arrow-arith",
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
@@ -5082,8 +5082,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-file"
|
name = "lance-file"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow-arith",
|
"arrow-arith",
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
@@ -5114,8 +5114,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-index"
|
name = "lance-index"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arc-swap",
|
"arc-swap",
|
||||||
"arrow",
|
"arrow",
|
||||||
@@ -5182,8 +5182,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-index-core"
|
name = "lance-index-core"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
"arrow-schema",
|
"arrow-schema",
|
||||||
@@ -5205,8 +5205,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-io"
|
name = "lance-io"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow",
|
"arrow",
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
@@ -5242,8 +5242,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-linalg"
|
name = "lance-linalg"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
"arrow-buffer",
|
"arrow-buffer",
|
||||||
@@ -5259,8 +5259,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-namespace"
|
name = "lance-namespace"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow",
|
"arrow",
|
||||||
"async-trait",
|
"async-trait",
|
||||||
@@ -5272,8 +5272,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-namespace-impls"
|
name = "lance-namespace-impls"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow",
|
"arrow",
|
||||||
"arrow-ipc",
|
"arrow-ipc",
|
||||||
@@ -5326,8 +5326,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-select"
|
name = "lance-select"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
"arrow-buffer",
|
"arrow-buffer",
|
||||||
@@ -5342,8 +5342,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-table"
|
name = "lance-table"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow",
|
"arrow",
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
@@ -5383,8 +5383,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-testing"
|
name = "lance-testing"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arrow-array",
|
"arrow-array",
|
||||||
"arrow-schema",
|
"arrow-schema",
|
||||||
@@ -5397,8 +5397,8 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lance-tokenizer"
|
name = "lance-tokenizer"
|
||||||
version = "11.0.0-beta.6"
|
version = "11.0.0-beta.7"
|
||||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.6#5ab688688c411111d94d279bacd037d7c319dc10"
|
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.7#e581c49338bc83baf1ea50c5e235bd702f3fbeea"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"frostem",
|
"frostem",
|
||||||
"icu_segmenter",
|
"icu_segmenter",
|
||||||
|
|||||||
+14
-14
@@ -13,20 +13,20 @@ categories = ["database-implementations"]
|
|||||||
rust-version = "1.91.0"
|
rust-version = "1.91.0"
|
||||||
|
|
||||||
[workspace.dependencies]
|
[workspace.dependencies]
|
||||||
lance = { "version" = "=11.0.0-beta.6", default-features = false, "tag" = "v11.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
lance = { "version" = "=11.0.0-beta.7", default-features = false, "tag" = "v11.0.0-beta.7", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-core = { "version" = "=11.0.0-beta.6", "tag" = "v11.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
lance-core = { "version" = "=11.0.0-beta.7", "tag" = "v11.0.0-beta.7", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-datagen = { "version" = "=11.0.0-beta.6", "tag" = "v11.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
lance-datagen = { "version" = "=11.0.0-beta.7", "tag" = "v11.0.0-beta.7", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-file = { "version" = "=11.0.0-beta.6", "tag" = "v11.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
lance-file = { "version" = "=11.0.0-beta.7", "tag" = "v11.0.0-beta.7", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-io = { "version" = "=11.0.0-beta.6", default-features = false, "tag" = "v11.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
lance-io = { "version" = "=11.0.0-beta.7", default-features = false, "tag" = "v11.0.0-beta.7", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-index = { "version" = "=11.0.0-beta.6", "tag" = "v11.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
lance-index = { "version" = "=11.0.0-beta.7", "tag" = "v11.0.0-beta.7", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-linalg = { "version" = "=11.0.0-beta.6", "tag" = "v11.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
lance-linalg = { "version" = "=11.0.0-beta.7", "tag" = "v11.0.0-beta.7", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-namespace = { "version" = "=11.0.0-beta.6", "tag" = "v11.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
lance-namespace = { "version" = "=11.0.0-beta.7", "tag" = "v11.0.0-beta.7", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-namespace-impls = { "version" = "=11.0.0-beta.6", default-features = false, "tag" = "v11.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
lance-namespace-impls = { "version" = "=11.0.0-beta.7", default-features = false, "tag" = "v11.0.0-beta.7", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-table = { "version" = "=11.0.0-beta.6", "tag" = "v11.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
lance-table = { "version" = "=11.0.0-beta.7", "tag" = "v11.0.0-beta.7", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-testing = { "version" = "=11.0.0-beta.6", "tag" = "v11.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
lance-testing = { "version" = "=11.0.0-beta.7", "tag" = "v11.0.0-beta.7", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-datafusion = { "version" = "=11.0.0-beta.6", "tag" = "v11.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
lance-datafusion = { "version" = "=11.0.0-beta.7", "tag" = "v11.0.0-beta.7", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-encoding = { "version" = "=11.0.0-beta.6", "tag" = "v11.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
lance-encoding = { "version" = "=11.0.0-beta.7", "tag" = "v11.0.0-beta.7", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
lance-arrow = { "version" = "=11.0.0-beta.6", "tag" = "v11.0.0-beta.6", "git" = "https://github.com/lance-format/lance.git" }
|
lance-arrow = { "version" = "=11.0.0-beta.7", "tag" = "v11.0.0-beta.7", "git" = "https://github.com/lance-format/lance.git" }
|
||||||
ahash = "0.8"
|
ahash = "0.8"
|
||||||
# Note that this one does not include pyarrow
|
# Note that this one does not include pyarrow
|
||||||
arrow = { version = "58.0.0", optional = false }
|
arrow = { version = "58.0.0", optional = false }
|
||||||
|
|||||||
@@ -69,14 +69,33 @@ abstract addColumns(newColumnTransforms): Promise<AddColumnsResult>
|
|||||||
|
|
||||||
Add new columns with defined values.
|
Add new columns with defined values.
|
||||||
|
|
||||||
|
The `{ computed }` form stores the expression rather than evaluating it
|
||||||
|
now: the column is committed with no values, and rows get them from
|
||||||
|
[Table#refreshColumn](Table.md#refreshcolumn). Declaring one therefore costs the same on a
|
||||||
|
large table as on an empty one.
|
||||||
|
|
||||||
|
A refresh does not revisit rows it has already filled, so mutating an
|
||||||
|
input leaves the value computed at fill time; recomputing means dropping
|
||||||
|
the column and declaring it again. While a declaration reads a column,
|
||||||
|
that column cannot be renamed, retyped or dropped.
|
||||||
|
|
||||||
|
Computed columns are local-only: LanceDB Cloud and Enterprise reject a
|
||||||
|
declaration.
|
||||||
|
|
||||||
#### Parameters
|
#### Parameters
|
||||||
|
|
||||||
* **newColumnTransforms**: `Field`<`any`> \| `Field`<`any`>[] \| `Schema`<`any`> \| [`AddColumnsSql`](../interfaces/AddColumnsSql.md)[]
|
* **newColumnTransforms**:
|
||||||
|
\| `Field`<`any`>
|
||||||
|
\| `Field`<`any`>[]
|
||||||
|
\| `Schema`<`any`>
|
||||||
|
\| [`AddColumnsSql`](../interfaces/AddColumnsSql.md)[]
|
||||||
|
\| `object`
|
||||||
Either:
|
Either:
|
||||||
- An array of objects with column names and SQL expressions to calculate values
|
- An array of objects with column names and SQL expressions to calculate values
|
||||||
- A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
- A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
||||||
- An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
- An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
||||||
- An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
- An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
||||||
|
- `{ computed }`, declaring columns defined by a SQL expression whose type and inputs are derived from it
|
||||||
|
|
||||||
#### Returns
|
#### Returns
|
||||||
|
|
||||||
@@ -85,6 +104,13 @@ Add new columns with defined values.
|
|||||||
A promise that resolves to an object
|
A promise that resolves to an object
|
||||||
containing the new version number of the table after adding the columns.
|
containing the new version number of the table after adding the columns.
|
||||||
|
|
||||||
|
#### Example
|
||||||
|
|
||||||
|
```ts
|
||||||
|
await table.addColumns({ computed: [{ name: "doubled", valueSql: "x * 2" }] });
|
||||||
|
const { rowsFilled } = await table.refreshColumn("doubled");
|
||||||
|
```
|
||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
### alterColumns()
|
### alterColumns()
|
||||||
@@ -718,6 +744,32 @@ for await (const batch of table.query()) {
|
|||||||
|
|
||||||
***
|
***
|
||||||
|
|
||||||
|
### refreshColumn()
|
||||||
|
|
||||||
|
```ts
|
||||||
|
abstract refreshColumn(column): Promise<RefreshColumnResult>
|
||||||
|
```
|
||||||
|
|
||||||
|
Fill the rows of a computed column that hold no value yet.
|
||||||
|
|
||||||
|
Rows appended since the last refresh are filled by the next one; rows
|
||||||
|
already filled are left as they are, so the call is idempotent and does
|
||||||
|
not observe a mutated input. Local tables only.
|
||||||
|
|
||||||
|
#### Parameters
|
||||||
|
|
||||||
|
* **column**: `string`
|
||||||
|
The name of the computed column to fill.
|
||||||
|
|
||||||
|
#### Returns
|
||||||
|
|
||||||
|
`Promise`<[`RefreshColumnResult`](../interfaces/RefreshColumnResult.md)>
|
||||||
|
|
||||||
|
A promise that resolves to the
|
||||||
|
number of rows filled and the new version number of the table.
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
### restore()
|
### restore()
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
|
|||||||
@@ -105,6 +105,7 @@
|
|||||||
- [OptimizeOptions](interfaces/OptimizeOptions.md)
|
- [OptimizeOptions](interfaces/OptimizeOptions.md)
|
||||||
- [OptimizeStats](interfaces/OptimizeStats.md)
|
- [OptimizeStats](interfaces/OptimizeStats.md)
|
||||||
- [QueryExecutionOptions](interfaces/QueryExecutionOptions.md)
|
- [QueryExecutionOptions](interfaces/QueryExecutionOptions.md)
|
||||||
|
- [RefreshColumnResult](interfaces/RefreshColumnResult.md)
|
||||||
- [RemovalStats](interfaces/RemovalStats.md)
|
- [RemovalStats](interfaces/RemovalStats.md)
|
||||||
- [RenameTableOptions](interfaces/RenameTableOptions.md)
|
- [RenameTableOptions](interfaces/RenameTableOptions.md)
|
||||||
- [RestNamespaceConfig](interfaces/RestNamespaceConfig.md)
|
- [RestNamespaceConfig](interfaces/RestNamespaceConfig.md)
|
||||||
|
|||||||
@@ -0,0 +1,23 @@
|
|||||||
|
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
[@lancedb/lancedb](../globals.md) / RefreshColumnResult
|
||||||
|
|
||||||
|
# Interface: RefreshColumnResult
|
||||||
|
|
||||||
|
## Properties
|
||||||
|
|
||||||
|
### rowsFilled
|
||||||
|
|
||||||
|
```ts
|
||||||
|
rowsFilled: number;
|
||||||
|
```
|
||||||
|
|
||||||
|
***
|
||||||
|
|
||||||
|
### version
|
||||||
|
|
||||||
|
```ts
|
||||||
|
version: number;
|
||||||
|
```
|
||||||
@@ -3340,3 +3340,45 @@ describe("LSM merge insert", () => {
|
|||||||
await expect(table.query().useLsm(true).toArray()).rejects.toThrow();
|
await expect(table.query().useLsm(true).toArray()).rejects.toThrow();
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
describe("computed columns", () => {
|
||||||
|
let tmpDir: tmp.DirResult;
|
||||||
|
beforeEach(() => {
|
||||||
|
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
||||||
|
});
|
||||||
|
afterEach(() => tmpDir.removeCallback());
|
||||||
|
|
||||||
|
it("declares a column and fills it on refresh", async () => {
|
||||||
|
const db = await connect(tmpDir.name);
|
||||||
|
const table = await db.createTable("computed", [{ x: 1 }, { x: 2 }]);
|
||||||
|
|
||||||
|
await table.addColumns({
|
||||||
|
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
||||||
|
});
|
||||||
|
let rows = await table.query().toArray();
|
||||||
|
expect(rows.map((r) => r.doubled)).toEqual([null, null]);
|
||||||
|
|
||||||
|
const result = await table.refreshColumn("doubled");
|
||||||
|
expect(result.rowsFilled).toBe(2);
|
||||||
|
|
||||||
|
rows = await table.query().toArray();
|
||||||
|
expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("fills rows added since the last refresh", async () => {
|
||||||
|
const db = await connect(tmpDir.name);
|
||||||
|
const table = await db.createTable("computed_append", [{ x: 1 }]);
|
||||||
|
|
||||||
|
await table.addColumns({
|
||||||
|
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
||||||
|
});
|
||||||
|
await table.refreshColumn("doubled");
|
||||||
|
await table.add([{ x: 5 }]);
|
||||||
|
|
||||||
|
const result = await table.refreshColumn("doubled");
|
||||||
|
expect(result.rowsFilled).toBe(1);
|
||||||
|
|
||||||
|
const rows = await table.query().toArray();
|
||||||
|
expect(rows.map((r) => r.doubled).sort()).toEqual([10, 2]);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|||||||
@@ -50,6 +50,7 @@ export {
|
|||||||
MergeResult,
|
MergeResult,
|
||||||
AddResult,
|
AddResult,
|
||||||
AddColumnsResult,
|
AddColumnsResult,
|
||||||
|
RefreshColumnResult,
|
||||||
AlterColumnsResult,
|
AlterColumnsResult,
|
||||||
UpdateFieldMetadataResult,
|
UpdateFieldMetadataResult,
|
||||||
DeleteResult,
|
DeleteResult,
|
||||||
|
|||||||
+57
-2
@@ -33,6 +33,7 @@ import {
|
|||||||
Job,
|
Job,
|
||||||
Branches as NativeBranches,
|
Branches as NativeBranches,
|
||||||
OptimizeStats,
|
OptimizeStats,
|
||||||
|
RefreshColumnResult,
|
||||||
TableStatistics,
|
TableStatistics,
|
||||||
Tags,
|
Tags,
|
||||||
UpdateFieldMetadataResult,
|
UpdateFieldMetadataResult,
|
||||||
@@ -525,18 +526,54 @@ export abstract class Table {
|
|||||||
abstract vectorSearch(vector: IntoVector | MultiVector): VectorQuery;
|
abstract vectorSearch(vector: IntoVector | MultiVector): VectorQuery;
|
||||||
/**
|
/**
|
||||||
* Add new columns with defined values.
|
* Add new columns with defined values.
|
||||||
|
*
|
||||||
|
* The `{ computed }` form stores the expression rather than evaluating it
|
||||||
|
* now: the column is committed with no values, and rows get them from
|
||||||
|
* {@link Table#refreshColumn}. Declaring one therefore costs the same on a
|
||||||
|
* large table as on an empty one.
|
||||||
|
*
|
||||||
|
* A refresh does not revisit rows it has already filled, so mutating an
|
||||||
|
* input leaves the value computed at fill time; recomputing means dropping
|
||||||
|
* the column and declaring it again. While a declaration reads a column,
|
||||||
|
* that column cannot be renamed, retyped or dropped.
|
||||||
|
*
|
||||||
|
* Computed columns are local-only: LanceDB Cloud and Enterprise reject a
|
||||||
|
* declaration.
|
||||||
* @param {AddColumnsSql[] | Field | Field[] | Schema} newColumnTransforms Either:
|
* @param {AddColumnsSql[] | Field | Field[] | Schema} newColumnTransforms Either:
|
||||||
* - An array of objects with column names and SQL expressions to calculate values
|
* - An array of objects with column names and SQL expressions to calculate values
|
||||||
* - A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
* - A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
||||||
* - An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
* - An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
||||||
* - An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
* - An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
||||||
|
* - `{ computed }`, declaring columns defined by a SQL expression whose type and inputs are derived from it
|
||||||
* @returns {Promise<AddColumnsResult>} A promise that resolves to an object
|
* @returns {Promise<AddColumnsResult>} A promise that resolves to an object
|
||||||
* containing the new version number of the table after adding the columns.
|
* containing the new version number of the table after adding the columns.
|
||||||
|
* @example
|
||||||
|
* ```ts
|
||||||
|
* await table.addColumns({ computed: [{ name: "doubled", valueSql: "x * 2" }] });
|
||||||
|
* const { rowsFilled } = await table.refreshColumn("doubled");
|
||||||
|
* ```
|
||||||
*/
|
*/
|
||||||
abstract addColumns(
|
abstract addColumns(
|
||||||
newColumnTransforms: AddColumnsSql[] | Field | Field[] | Schema,
|
newColumnTransforms:
|
||||||
|
| AddColumnsSql[]
|
||||||
|
| Field
|
||||||
|
| Field[]
|
||||||
|
| Schema
|
||||||
|
| { computed: AddColumnsSql[] },
|
||||||
): Promise<AddColumnsResult>;
|
): Promise<AddColumnsResult>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Fill the rows of a computed column that hold no value yet.
|
||||||
|
*
|
||||||
|
* Rows appended since the last refresh are filled by the next one; rows
|
||||||
|
* already filled are left as they are, so the call is idempotent and does
|
||||||
|
* not observe a mutated input. Local tables only.
|
||||||
|
* @param {string} column The name of the computed column to fill.
|
||||||
|
* @returns {Promise<RefreshColumnResult>} A promise that resolves to the
|
||||||
|
* number of rows filled and the new version number of the table.
|
||||||
|
*/
|
||||||
|
abstract refreshColumn(column: string): Promise<RefreshColumnResult>;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Alter the name or nullability of columns.
|
* Alter the name or nullability of columns.
|
||||||
* @param {ColumnAlteration[]} columnAlterations One or more alterations to
|
* @param {ColumnAlteration[]} columnAlterations One or more alterations to
|
||||||
@@ -1088,8 +1125,22 @@ export class LocalTable extends Table {
|
|||||||
// TODO: Support BatchUDF
|
// TODO: Support BatchUDF
|
||||||
|
|
||||||
async addColumns(
|
async addColumns(
|
||||||
newColumnTransforms: AddColumnsSql[] | Field | Field[] | Schema,
|
newColumnTransforms:
|
||||||
|
| AddColumnsSql[]
|
||||||
|
| Field
|
||||||
|
| Field[]
|
||||||
|
| Schema
|
||||||
|
| { computed: AddColumnsSql[] },
|
||||||
): Promise<AddColumnsResult> {
|
): Promise<AddColumnsResult> {
|
||||||
|
// Columns defined by an expression are declared, not materialized here.
|
||||||
|
if (
|
||||||
|
typeof newColumnTransforms === "object" &&
|
||||||
|
!Array.isArray(newColumnTransforms) &&
|
||||||
|
"computed" in newColumnTransforms
|
||||||
|
) {
|
||||||
|
return await this.inner.addComputedColumns(newColumnTransforms.computed);
|
||||||
|
}
|
||||||
|
|
||||||
// Handle single Field -> convert to array of Fields
|
// Handle single Field -> convert to array of Fields
|
||||||
if (newColumnTransforms instanceof Field) {
|
if (newColumnTransforms instanceof Field) {
|
||||||
newColumnTransforms = [newColumnTransforms];
|
newColumnTransforms = [newColumnTransforms];
|
||||||
@@ -1124,6 +1175,10 @@ export class LocalTable extends Table {
|
|||||||
throw new Error("Invalid input type for addColumns");
|
throw new Error("Invalid input type for addColumns");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async refreshColumn(column: string): Promise<RefreshColumnResult> {
|
||||||
|
return await this.inner.refreshColumn(column);
|
||||||
|
}
|
||||||
|
|
||||||
async alterColumns(
|
async alterColumns(
|
||||||
columnAlterations: ColumnAlteration[],
|
columnAlterations: ColumnAlteration[],
|
||||||
): Promise<AlterColumnsResult> {
|
): Promise<AlterColumnsResult> {
|
||||||
|
|||||||
@@ -347,6 +347,30 @@ impl Table {
|
|||||||
Ok(res.into())
|
Ok(res.into())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn add_computed_columns(
|
||||||
|
&self,
|
||||||
|
columns: Vec<AddColumnsSql>,
|
||||||
|
) -> napi::Result<AddColumnsResult> {
|
||||||
|
let table = self.inner_ref()?;
|
||||||
|
let mut builder = table.add_columns();
|
||||||
|
for column in columns {
|
||||||
|
builder = builder.computed(column.name, column.value_sql);
|
||||||
|
}
|
||||||
|
let res = builder.execute().await.default_error()?;
|
||||||
|
Ok(res.into())
|
||||||
|
}
|
||||||
|
|
||||||
|
#[napi(catch_unwind)]
|
||||||
|
pub async fn refresh_column(&self, column: String) -> napi::Result<RefreshColumnResult> {
|
||||||
|
let res = self
|
||||||
|
.inner_ref()?
|
||||||
|
.refresh_column(column)
|
||||||
|
.await
|
||||||
|
.default_error()?;
|
||||||
|
Ok(res.into())
|
||||||
|
}
|
||||||
|
|
||||||
#[napi(catch_unwind)]
|
#[napi(catch_unwind)]
|
||||||
pub async fn add_columns_with_schema(
|
pub async fn add_columns_with_schema(
|
||||||
&self,
|
&self,
|
||||||
@@ -1196,6 +1220,21 @@ pub struct AddColumnsResult {
|
|||||||
pub version: i64,
|
pub version: i64,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct RefreshColumnResult {
|
||||||
|
pub rows_filled: i64,
|
||||||
|
pub version: i64,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<lancedb::table::RefreshColumnResult> for RefreshColumnResult {
|
||||||
|
fn from(value: lancedb::table::RefreshColumnResult) -> Self {
|
||||||
|
Self {
|
||||||
|
rows_filled: value.rows_filled as i64,
|
||||||
|
version: value.version as i64,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
impl From<lancedb::table::AddColumnsResult> for AddColumnsResult {
|
impl From<lancedb::table::AddColumnsResult> for AddColumnsResult {
|
||||||
fn from(value: lancedb::table::AddColumnsResult) -> Self {
|
fn from(value: lancedb::table::AddColumnsResult) -> Self {
|
||||||
Self {
|
Self {
|
||||||
|
|||||||
@@ -335,6 +335,10 @@ class Table:
|
|||||||
) -> list[FtsToken]: ...
|
) -> list[FtsToken]: ...
|
||||||
async def delete(self, filter: Union[str, PyExpr]) -> DeleteResult: ...
|
async def delete(self, filter: Union[str, PyExpr]) -> DeleteResult: ...
|
||||||
async def add_columns(self, columns: list[tuple[str, str]]) -> AddColumnsResult: ...
|
async def add_columns(self, columns: list[tuple[str, str]]) -> AddColumnsResult: ...
|
||||||
|
async def add_computed_columns(
|
||||||
|
self, columns: list[tuple[str, str]]
|
||||||
|
) -> AddColumnsResult: ...
|
||||||
|
async def refresh_column(self, column: str) -> RefreshColumnResult: ...
|
||||||
async def add_columns_with_schema(self, schema: pa.Schema) -> AddColumnsResult: ...
|
async def add_columns_with_schema(self, schema: pa.Schema) -> AddColumnsResult: ...
|
||||||
async def alter_columns(
|
async def alter_columns(
|
||||||
self, columns: list[dict[str, Any]]
|
self, columns: list[dict[str, Any]]
|
||||||
@@ -680,6 +684,10 @@ class LsmWriteSpec:
|
|||||||
class AddColumnsResult:
|
class AddColumnsResult:
|
||||||
version: int
|
version: int
|
||||||
|
|
||||||
|
class RefreshColumnResult:
|
||||||
|
rows_filled: int
|
||||||
|
version: int
|
||||||
|
|
||||||
class AlterColumnsResult:
|
class AlterColumnsResult:
|
||||||
version: int
|
version: int
|
||||||
|
|
||||||
|
|||||||
@@ -958,9 +958,21 @@ class RemoteTable(Table):
|
|||||||
def count_rows(self, filter: Optional[str] = None) -> int:
|
def count_rows(self, filter: Optional[str] = None) -> int:
|
||||||
return LOOP.run(self._table.count_rows(filter))
|
return LOOP.run(self._table.count_rows(filter))
|
||||||
|
|
||||||
def add_columns(self, transforms: Dict[str, str]) -> AddColumnsResult:
|
def add_columns(
|
||||||
|
self,
|
||||||
|
transforms: Dict[str, str] | None = None,
|
||||||
|
*,
|
||||||
|
computed: Dict[str, str] | None = None,
|
||||||
|
) -> AddColumnsResult:
|
||||||
|
if computed:
|
||||||
|
raise NotImplementedError(
|
||||||
|
"computed columns are supported only on local tables"
|
||||||
|
)
|
||||||
return LOOP.run(self._table.add_columns(transforms))
|
return LOOP.run(self._table.add_columns(transforms))
|
||||||
|
|
||||||
|
def refresh_column(self, column: str):
|
||||||
|
raise NotImplementedError("computed columns are supported only on local tables")
|
||||||
|
|
||||||
def alter_columns(
|
def alter_columns(
|
||||||
self, *alterations: Iterable[Dict[str, str]]
|
self, *alterations: Iterable[Dict[str, str]]
|
||||||
) -> AlterColumnsResult:
|
) -> AlterColumnsResult:
|
||||||
|
|||||||
@@ -176,6 +176,7 @@ if TYPE_CHECKING:
|
|||||||
CompactionStats,
|
CompactionStats,
|
||||||
Tag,
|
Tag,
|
||||||
AddColumnsResult,
|
AddColumnsResult,
|
||||||
|
RefreshColumnResult,
|
||||||
AddResult,
|
AddResult,
|
||||||
AlterColumnsResult,
|
AlterColumnsResult,
|
||||||
UpdateFieldMetadataResult,
|
UpdateFieldMetadataResult,
|
||||||
@@ -1916,7 +1917,14 @@ class Table(ABC):
|
|||||||
|
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
def add_columns(
|
def add_columns(
|
||||||
self, transforms: Dict[str, str] | pa.Field | List[pa.Field] | pa.Schema
|
self,
|
||||||
|
transforms: Dict[str, str]
|
||||||
|
| pa.Field
|
||||||
|
| List[pa.Field]
|
||||||
|
| pa.Schema
|
||||||
|
| None = None,
|
||||||
|
*,
|
||||||
|
computed: Dict[str, str] | None = None,
|
||||||
):
|
):
|
||||||
"""
|
"""
|
||||||
Add new columns with defined values.
|
Add new columns with defined values.
|
||||||
@@ -1930,11 +1938,68 @@ class Table(ABC):
|
|||||||
Alternatively, a pyarrow Field or Schema can be provided to add
|
Alternatively, a pyarrow Field or Schema can be provided to add
|
||||||
new columns with the specified data types. The new columns will
|
new columns with the specified data types. The new columns will
|
||||||
be initialized with null values.
|
be initialized with null values.
|
||||||
|
computed: Dict[str, str], optional
|
||||||
|
A map of column name to a SQL expression defining the column. The
|
||||||
|
column's type and inputs are derived from the expression, so no
|
||||||
|
data type is supplied.
|
||||||
|
|
||||||
|
Unlike ``transforms``, the expression is stored rather than
|
||||||
|
evaluated now: the column is committed with no values, and rows get
|
||||||
|
them from [`refresh_column`][lancedb.table.Table.refresh_column].
|
||||||
|
Declaring one therefore costs the same on a large table as on an
|
||||||
|
empty one.
|
||||||
|
|
||||||
|
A refresh does not revisit rows it has already filled, so mutating
|
||||||
|
an input leaves the value computed at fill time; recomputing means
|
||||||
|
dropping the column and declaring it again. While a declaration
|
||||||
|
reads a column, that column cannot be renamed, retyped or dropped.
|
||||||
|
|
||||||
|
Local tables only; LanceDB Cloud and Enterprise raise
|
||||||
|
``NotImplementedError``. Cannot be combined with ``transforms``.
|
||||||
|
|
||||||
Returns
|
Returns
|
||||||
-------
|
-------
|
||||||
AddColumnsResult
|
AddColumnsResult
|
||||||
version: the new version number of the table after adding columns.
|
version: the new version number of the table after adding columns.
|
||||||
|
|
||||||
|
Examples
|
||||||
|
--------
|
||||||
|
>>> import lancedb
|
||||||
|
>>> db = lancedb.connect("./.lancedb")
|
||||||
|
>>> table = db.create_table("computed_demo", [{"x": 1}, {"x": 2}])
|
||||||
|
>>> table.add_columns(computed={"doubled": "x * 2"})
|
||||||
|
AddColumnsResult(version=2)
|
||||||
|
>>> table.refresh_column("doubled")
|
||||||
|
RefreshColumnResult(rows_filled=2, version=3)
|
||||||
|
>>> table.to_arrow().sort_by("x").to_pandas()
|
||||||
|
x doubled
|
||||||
|
0 1 2
|
||||||
|
1 2 4
|
||||||
|
"""
|
||||||
|
|
||||||
|
@abstractmethod
|
||||||
|
def refresh_column(self, column: str) -> "RefreshColumnResult":
|
||||||
|
"""
|
||||||
|
Fill the rows of a computed column that hold no value yet.
|
||||||
|
|
||||||
|
Declared with ``add_columns(computed=...)``, a column starts empty and
|
||||||
|
gets its values here. Rows appended since the last refresh are filled
|
||||||
|
by the next one; rows already filled are left as they are, so the call
|
||||||
|
is idempotent and does not observe a mutated input.
|
||||||
|
|
||||||
|
Local tables only; LanceDB Cloud and Enterprise raise
|
||||||
|
``NotImplementedError``.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
column: str
|
||||||
|
The name of the computed column to fill.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
RefreshColumnResult
|
||||||
|
rows_filled: the number of rows given a value.
|
||||||
|
version: the new version number of the table.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
@@ -3939,9 +4004,21 @@ class LanceTable(Table):
|
|||||||
return LOOP.run(self._table.index_stats(index_name))
|
return LOOP.run(self._table.index_stats(index_name))
|
||||||
|
|
||||||
def add_columns(
|
def add_columns(
|
||||||
self, transforms: Dict[str, str] | pa.field | List[pa.field] | pa.Schema
|
self,
|
||||||
|
transforms: Dict[str, str]
|
||||||
|
| pa.field
|
||||||
|
| List[pa.field]
|
||||||
|
| pa.Schema
|
||||||
|
| None = None,
|
||||||
|
*,
|
||||||
|
computed: Dict[str, str] | None = None,
|
||||||
) -> AddColumnsResult:
|
) -> AddColumnsResult:
|
||||||
return LOOP.run(self._table.add_columns(transforms))
|
return LOOP.run(self._table.add_columns(transforms, computed=computed))
|
||||||
|
|
||||||
|
def refresh_column(self, column: str) -> "RefreshColumnResult":
|
||||||
|
"""Fill a computed column's unfilled rows. See
|
||||||
|
[`AsyncTable.refresh_column`][lancedb.AsyncTable.refresh_column]."""
|
||||||
|
return LOOP.run(self._table.refresh_column(column))
|
||||||
|
|
||||||
def alter_columns(
|
def alter_columns(
|
||||||
self, *alterations: Iterable[Dict[str, str]]
|
self, *alterations: Iterable[Dict[str, str]]
|
||||||
@@ -5856,7 +5933,14 @@ class AsyncTable:
|
|||||||
return await self._inner.update(updates_sql, where)
|
return await self._inner.update(updates_sql, where)
|
||||||
|
|
||||||
async def add_columns(
|
async def add_columns(
|
||||||
self, transforms: dict[str, str] | pa.field | List[pa.field] | pa.Schema
|
self,
|
||||||
|
transforms: dict[str, str]
|
||||||
|
| pa.field
|
||||||
|
| List[pa.field]
|
||||||
|
| pa.Schema
|
||||||
|
| None = None,
|
||||||
|
*,
|
||||||
|
computed: dict[str, str] | None = None,
|
||||||
) -> AddColumnsResult:
|
) -> AddColumnsResult:
|
||||||
"""
|
"""
|
||||||
Add new columns with defined values.
|
Add new columns with defined values.
|
||||||
@@ -5869,6 +5953,21 @@ class AsyncTable:
|
|||||||
each row in the table, and can reference existing columns.
|
each row in the table, and can reference existing columns.
|
||||||
Alternatively, you can pass a pyarrow field or schema to add
|
Alternatively, you can pass a pyarrow field or schema to add
|
||||||
new columns with NULLs.
|
new columns with NULLs.
|
||||||
|
computed: Dict[str, str], optional
|
||||||
|
A map of column name to a SQL expression defining the column. The
|
||||||
|
column's type and inputs are derived from the expression.
|
||||||
|
|
||||||
|
Unlike ``transforms``, the expression is stored rather than
|
||||||
|
evaluated now: the column is committed with no values, and rows get
|
||||||
|
them from
|
||||||
|
[`refresh_column`][lancedb.table.AsyncTable.refresh_column].
|
||||||
|
|
||||||
|
A refresh does not revisit rows it has already filled, so mutating
|
||||||
|
an input leaves the value computed at fill time. While a
|
||||||
|
declaration reads a column, that column cannot be renamed, retyped
|
||||||
|
or dropped.
|
||||||
|
|
||||||
|
Local tables only. Cannot be combined with ``transforms``.
|
||||||
|
|
||||||
Returns
|
Returns
|
||||||
-------
|
-------
|
||||||
@@ -5882,11 +5981,43 @@ class AsyncTable:
|
|||||||
{isinstance(f, pa.Field) for f in transforms}
|
{isinstance(f, pa.Field) for f in transforms}
|
||||||
):
|
):
|
||||||
transforms = pa.schema(transforms)
|
transforms = pa.schema(transforms)
|
||||||
|
if computed:
|
||||||
|
if transforms:
|
||||||
|
raise ValueError(
|
||||||
|
"add_columns cannot take both transforms and computed columns"
|
||||||
|
)
|
||||||
|
return await self._inner.add_computed_columns(list(computed.items()))
|
||||||
|
if transforms is None:
|
||||||
|
raise ValueError("add_columns requires transforms or computed columns")
|
||||||
if isinstance(transforms, pa.Schema):
|
if isinstance(transforms, pa.Schema):
|
||||||
return await self._inner.add_columns_with_schema(transforms)
|
return await self._inner.add_columns_with_schema(transforms)
|
||||||
else:
|
else:
|
||||||
return await self._inner.add_columns(list(transforms.items()))
|
return await self._inner.add_columns(list(transforms.items()))
|
||||||
|
|
||||||
|
async def refresh_column(self, column: str) -> RefreshColumnResult:
|
||||||
|
"""
|
||||||
|
Fill the rows of a computed column that hold no value yet.
|
||||||
|
|
||||||
|
Declared with ``add_columns(computed=...)``, a column starts empty and
|
||||||
|
gets its values here. Rows appended since the last refresh are filled
|
||||||
|
by the next one; rows already filled are left as they are, so the call
|
||||||
|
is idempotent and does not observe a mutated input.
|
||||||
|
|
||||||
|
Local tables only; LanceDB Cloud and Enterprise raise
|
||||||
|
``NotImplementedError``.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
column: str
|
||||||
|
The name of the computed column to fill.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
RefreshColumnResult
|
||||||
|
The number of rows filled and the new version of the table.
|
||||||
|
"""
|
||||||
|
return await self._inner.refresh_column(column)
|
||||||
|
|
||||||
async def alter_columns(
|
async def alter_columns(
|
||||||
self, *alterations: Iterable[dict[str, Any]]
|
self, *alterations: Iterable[dict[str, Any]]
|
||||||
) -> AlterColumnsResult:
|
) -> AlterColumnsResult:
|
||||||
|
|||||||
@@ -3854,3 +3854,37 @@ async def test_async_search_runs_embedding_on_dedicated_executor(
|
|||||||
assert all(name.startswith("lancedb-embedding") for name in captured_threads), (
|
assert all(name.startswith("lancedb-embedding") for name in captured_threads), (
|
||||||
f"embedding ran off the dedicated executor: {captured_threads}"
|
f"embedding ran off the dedicated executor: {captured_threads}"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_computed_column_declare_and_refresh(tmp_path):
|
||||||
|
db = lancedb.connect(tmp_path)
|
||||||
|
table = db.create_table("computed", [{"x": 1}, {"x": 2}])
|
||||||
|
|
||||||
|
table.add_columns(computed={"doubled": "x * 2"})
|
||||||
|
assert table.to_arrow()["doubled"].to_pylist() == [None, None]
|
||||||
|
|
||||||
|
result = table.refresh_column("doubled")
|
||||||
|
assert result.rows_filled == 2
|
||||||
|
assert sorted(table.to_arrow()["doubled"].to_pylist()) == [2, 4]
|
||||||
|
|
||||||
|
table.add([{"x": 5}])
|
||||||
|
assert table.refresh_column("doubled").rows_filled == 1
|
||||||
|
assert sorted(table.to_arrow()["doubled"].to_pylist()) == [2, 4, 10]
|
||||||
|
|
||||||
|
|
||||||
|
def test_computed_column_rejects_transforms_and_computed_together(tmp_path):
|
||||||
|
db = lancedb.connect(tmp_path)
|
||||||
|
table = db.create_table("computed_mixed", [{"x": 1}])
|
||||||
|
with pytest.raises(ValueError):
|
||||||
|
table.add_columns({"a": "x + 1"}, computed={"b": "x * 2"})
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_computed_column_async(tmp_path):
|
||||||
|
db = await lancedb.connect_async(tmp_path)
|
||||||
|
table = await db.create_table("computed_async", [{"x": 3}])
|
||||||
|
|
||||||
|
await table.add_columns(computed={"tripled": "x * 3"})
|
||||||
|
await table.refresh_column("tripled")
|
||||||
|
|
||||||
|
assert (await table.to_arrow())["tripled"].to_pylist() == [9]
|
||||||
|
|||||||
+3
-1
@@ -16,7 +16,8 @@ use query::{FTSQuery, HybridQuery, Query, VectorQuery};
|
|||||||
use session::Session;
|
use session::Session;
|
||||||
use table::{
|
use table::{
|
||||||
AddColumnsResult, AddResult, AlterColumnsResult, DeleteResult, DropColumnsResult, FtsToken,
|
AddColumnsResult, AddResult, AlterColumnsResult, DeleteResult, DropColumnsResult, FtsToken,
|
||||||
LsmWriteSpec, MergeResult, PyBlobFile, Table, UpdateFieldMetadataResult, UpdateResult,
|
LsmWriteSpec, MergeResult, PyBlobFile, RefreshColumnResult, Table, UpdateFieldMetadataResult,
|
||||||
|
UpdateResult,
|
||||||
};
|
};
|
||||||
|
|
||||||
pub mod arrow;
|
pub mod arrow;
|
||||||
@@ -57,6 +58,7 @@ pub fn _lancedb(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> {
|
|||||||
m.add_class::<VectorQuery>()?;
|
m.add_class::<VectorQuery>()?;
|
||||||
m.add_class::<RecordBatchStream>()?;
|
m.add_class::<RecordBatchStream>()?;
|
||||||
m.add_class::<AddColumnsResult>()?;
|
m.add_class::<AddColumnsResult>()?;
|
||||||
|
m.add_class::<RefreshColumnResult>()?;
|
||||||
m.add_class::<AlterColumnsResult>()?;
|
m.add_class::<AlterColumnsResult>()?;
|
||||||
m.add_class::<UpdateFieldMetadataResult>()?;
|
m.add_class::<UpdateFieldMetadataResult>()?;
|
||||||
m.add_class::<AddResult>()?;
|
m.add_class::<AddResult>()?;
|
||||||
|
|||||||
@@ -415,6 +415,32 @@ pub struct AddColumnsResult {
|
|||||||
pub version: u64,
|
pub version: u64,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[pyclass(get_all, from_py_object)]
|
||||||
|
#[derive(Clone, Debug)]
|
||||||
|
pub struct RefreshColumnResult {
|
||||||
|
pub rows_filled: u64,
|
||||||
|
pub version: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[pymethods]
|
||||||
|
impl RefreshColumnResult {
|
||||||
|
pub fn __repr__(&self) -> String {
|
||||||
|
format!(
|
||||||
|
"RefreshColumnResult(rows_filled={}, version={})",
|
||||||
|
self.rows_filled, self.version
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<lancedb::table::RefreshColumnResult> for RefreshColumnResult {
|
||||||
|
fn from(result: lancedb::table::RefreshColumnResult) -> Self {
|
||||||
|
Self {
|
||||||
|
rows_filled: result.rows_filled,
|
||||||
|
version: result.version,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[pymethods]
|
#[pymethods]
|
||||||
impl AddColumnsResult {
|
impl AddColumnsResult {
|
||||||
pub fn __repr__(&self) -> String {
|
pub fn __repr__(&self) -> String {
|
||||||
@@ -1510,6 +1536,29 @@ impl Table {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn add_computed_columns(
|
||||||
|
self_: PyRef<'_, Self>,
|
||||||
|
columns: Vec<(String, String)>,
|
||||||
|
) -> PyResult<Bound<'_, PyAny>> {
|
||||||
|
let inner = self_.inner_ref()?.clone();
|
||||||
|
future_into_py(self_.py(), async move {
|
||||||
|
let mut builder = inner.add_columns();
|
||||||
|
for (name, expression) in columns {
|
||||||
|
builder = builder.computed(name, expression);
|
||||||
|
}
|
||||||
|
let result = builder.execute().await.infer_error()?;
|
||||||
|
Ok(AddColumnsResult::from(result))
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn refresh_column(self_: PyRef<'_, Self>, column: String) -> PyResult<Bound<'_, PyAny>> {
|
||||||
|
let inner = self_.inner_ref()?.clone();
|
||||||
|
future_into_py(self_.py(), async move {
|
||||||
|
let result = inner.refresh_column(column).await.infer_error()?;
|
||||||
|
Ok(RefreshColumnResult::from(result))
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
pub fn add_columns_with_schema(
|
pub fn add_columns_with_schema(
|
||||||
self_: PyRef<'_, Self>,
|
self_: PyRef<'_, Self>,
|
||||||
schema: PyArrowType<Schema>,
|
schema: PyArrowType<Schema>,
|
||||||
|
|||||||
@@ -71,6 +71,14 @@ pub enum Error {
|
|||||||
IndexNotFound { name: String },
|
IndexNotFound { name: String },
|
||||||
#[snafu(display("Embedding function '{name}' was not found. : {reason}"))]
|
#[snafu(display("Embedding function '{name}' was not found. : {reason}"))]
|
||||||
EmbeddingFunctionNotFound { name: String, reason: String },
|
EmbeddingFunctionNotFound { name: String, reason: String },
|
||||||
|
#[snafu(display("Column '{name}' was not found"))]
|
||||||
|
ColumnNotFound { name: String },
|
||||||
|
#[snafu(display("Column '{name}' already exists"))]
|
||||||
|
ColumnAlreadyExists { name: String },
|
||||||
|
#[snafu(display("Column '{name}' is not a computed column"))]
|
||||||
|
NotAComputedColumn { name: String },
|
||||||
|
#[snafu(display("Invalid expression for column '{column}': {message}"))]
|
||||||
|
InvalidExpression { column: String, message: String },
|
||||||
|
|
||||||
#[snafu(display("Table '{name}' already exists"))]
|
#[snafu(display("Table '{name}' already exists"))]
|
||||||
TableAlreadyExists { name: String },
|
TableAlreadyExists { name: String },
|
||||||
|
|||||||
@@ -2706,6 +2706,13 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
|
|||||||
|
|
||||||
Ok(result)
|
Ok(result)
|
||||||
}
|
}
|
||||||
|
// A declaration reaches here as AllNulls, which the remote protocol
|
||||||
|
// has no representation for.
|
||||||
|
NewColumnTransform::AllNulls(_) => {
|
||||||
|
return Err(Error::NotSupported {
|
||||||
|
message: "computed columns are supported only on local tables".into(),
|
||||||
|
});
|
||||||
|
}
|
||||||
_ => {
|
_ => {
|
||||||
return Err(Error::NotSupported {
|
return Err(Error::NotSupported {
|
||||||
message: "Only SQL expressions are supported for adding columns".into(),
|
message: "Only SQL expressions are supported for adding columns".into(),
|
||||||
@@ -6455,6 +6462,37 @@ mod tests {
|
|||||||
assert_eq!(result.version, if old_server { 0 } else { 43 });
|
assert_eq!(result.version, if old_server { 0 } else { 43 });
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Computed columns are local-only. Both halves say so here rather than
|
||||||
|
/// reaching the wire and failing somewhere less legible.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_computed_columns_are_refused() {
|
||||||
|
let table = Table::new_with_handler("my_table", |request| -> http::Response<String> {
|
||||||
|
panic!("unexpected request: {}", request.url().path())
|
||||||
|
});
|
||||||
|
|
||||||
|
let declared = Arc::new(Schema::new(vec![Field::new(
|
||||||
|
"doubled",
|
||||||
|
DataType::Int32,
|
||||||
|
true,
|
||||||
|
)]));
|
||||||
|
let err = table
|
||||||
|
.add_columns()
|
||||||
|
.transform(NewColumnTransform::AllNulls(declared))
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap_err();
|
||||||
|
assert!(
|
||||||
|
matches!(&err, Error::NotSupported { message } if message.contains("local tables")),
|
||||||
|
"{err:?}"
|
||||||
|
);
|
||||||
|
|
||||||
|
let err = table.refresh_column("doubled").await.unwrap_err();
|
||||||
|
assert!(
|
||||||
|
matches!(&err, Error::NotSupported { message } if message.contains("local tables")),
|
||||||
|
"{err:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn test_prewarm_index() {
|
async fn test_prewarm_index() {
|
||||||
let table = Table::new_with_handler("my_table", |request| {
|
let table = Table::new_with_handler("my_table", |request| {
|
||||||
|
|||||||
@@ -69,6 +69,7 @@ pub mod add_columns;
|
|||||||
mod add_data;
|
mod add_data;
|
||||||
pub mod branch_merge;
|
pub mod branch_merge;
|
||||||
pub mod checkpoint;
|
pub mod checkpoint;
|
||||||
|
pub mod computed_columns;
|
||||||
mod create_index;
|
mod create_index;
|
||||||
pub mod datafusion;
|
pub mod datafusion;
|
||||||
pub(crate) mod dataset;
|
pub(crate) mod dataset;
|
||||||
@@ -78,6 +79,7 @@ pub mod merge;
|
|||||||
pub mod optimize;
|
pub mod optimize;
|
||||||
mod primary_key;
|
mod primary_key;
|
||||||
pub mod query;
|
pub mod query;
|
||||||
|
pub mod refresh;
|
||||||
pub mod schema_evolution;
|
pub mod schema_evolution;
|
||||||
pub mod update;
|
pub mod update;
|
||||||
pub mod write_progress;
|
pub mod write_progress;
|
||||||
@@ -91,6 +93,9 @@ pub use branch_merge::{
|
|||||||
MergeBranchResult, MergeBranchStatus, MergePreview, RowCountSummary,
|
MergeBranchResult, MergeBranchStatus, MergePreview, RowCountSummary,
|
||||||
};
|
};
|
||||||
pub use chrono::Duration;
|
pub use chrono::Duration;
|
||||||
|
pub use computed_columns::{
|
||||||
|
ComputedColumn, ComputedColumnKind, computed_column_from_field, computed_columns,
|
||||||
|
};
|
||||||
pub use delete::DeleteResult;
|
pub use delete::DeleteResult;
|
||||||
use futures::future::join_all;
|
use futures::future::join_all;
|
||||||
pub use lance::dataset::refs::{BranchContents, Ref, TagContents, Tags as LanceTags};
|
pub use lance::dataset::refs::{BranchContents, Ref, TagContents, Tags as LanceTags};
|
||||||
@@ -98,6 +103,7 @@ pub use lance::dataset::scanner::DatasetRecordBatchStream;
|
|||||||
pub use lance_index::optimize::OptimizeOptions;
|
pub use lance_index::optimize::OptimizeOptions;
|
||||||
pub use lsm_stats::{BucketStats, GenerationStats, LsmStats, MemtableStats};
|
pub use lsm_stats::{BucketStats, GenerationStats, LsmStats, MemtableStats};
|
||||||
pub use optimize::{CompactionOptions, OptimizeAction, OptimizeStats};
|
pub use optimize::{CompactionOptions, OptimizeAction, OptimizeStats};
|
||||||
|
pub use refresh::RefreshColumnResult;
|
||||||
pub use schema_evolution::{
|
pub use schema_evolution::{
|
||||||
AddColumnsResult, AlterColumnsResult, DropColumnsResult, FieldMetadataUpdate,
|
AddColumnsResult, AlterColumnsResult, DropColumnsResult, FieldMetadataUpdate,
|
||||||
UpdateFieldMetadataResult,
|
UpdateFieldMetadataResult,
|
||||||
@@ -782,6 +788,14 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
|
|||||||
transforms: NewColumnTransform,
|
transforms: NewColumnTransform,
|
||||||
read_columns: Option<Vec<String>>,
|
read_columns: Option<Vec<String>>,
|
||||||
) -> Result<AddColumnsResult>;
|
) -> Result<AddColumnsResult>;
|
||||||
|
/// Fill a computed column's unfilled rows.
|
||||||
|
///
|
||||||
|
/// The default returns `NotSupported`; Lance-backed tables override it.
|
||||||
|
async fn refresh_column(&self, _column: &str) -> Result<RefreshColumnResult> {
|
||||||
|
Err(Error::NotSupported {
|
||||||
|
message: "computed columns are supported only on local tables".into(),
|
||||||
|
})
|
||||||
|
}
|
||||||
/// Alter columns in the table.
|
/// Alter columns in the table.
|
||||||
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult>;
|
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult>;
|
||||||
/// Drop columns from the table.
|
/// Drop columns from the table.
|
||||||
@@ -1674,6 +1688,29 @@ impl Table {
|
|||||||
AddColumnsBuilder::new(self.inner.clone())
|
AddColumnsBuilder::new(self.inner.clone())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Fill the fragments of a computed column that hold no values yet.
|
||||||
|
///
|
||||||
|
/// Declared with
|
||||||
|
/// [`AddColumnsBuilder::computed`](add_columns::AddColumnsBuilder::computed),
|
||||||
|
/// a column starts empty and gets its values here. Fragments appended
|
||||||
|
/// since the last refresh are filled by the next one; fragments already
|
||||||
|
/// filled are left as they are, so the call is idempotent and does not
|
||||||
|
/// observe a mutated input.
|
||||||
|
///
|
||||||
|
/// Local tables only.
|
||||||
|
///
|
||||||
|
/// ```
|
||||||
|
/// # use lancedb::Table;
|
||||||
|
/// # async fn refresh(table: &Table) -> Result<(), Box<dyn std::error::Error>> {
|
||||||
|
/// let result = table.refresh_column("doubled").await?;
|
||||||
|
/// println!("filled {} rows at version {}", result.rows_filled, result.version);
|
||||||
|
/// # Ok(())
|
||||||
|
/// # }
|
||||||
|
/// ```
|
||||||
|
pub async fn refresh_column(&self, column: impl AsRef<str>) -> Result<RefreshColumnResult> {
|
||||||
|
self.inner.refresh_column(column.as_ref()).await
|
||||||
|
}
|
||||||
|
|
||||||
/// Change a column's name or nullability.
|
/// Change a column's name or nullability.
|
||||||
pub async fn alter_columns(
|
pub async fn alter_columns(
|
||||||
&self,
|
&self,
|
||||||
@@ -3341,6 +3378,12 @@ impl BaseTable for NativeTable {
|
|||||||
Ok(result)
|
Ok(result)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async fn refresh_column(&self, column: &str) -> Result<RefreshColumnResult> {
|
||||||
|
let result = refresh::execute_refresh_column(self, column).await?;
|
||||||
|
self.bump_freshness();
|
||||||
|
Ok(result)
|
||||||
|
}
|
||||||
|
|
||||||
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult> {
|
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult> {
|
||||||
let result = schema_evolution::execute_alter_columns(self, alterations).await?;
|
let result = schema_evolution::execute_alter_columns(self, alterations).await?;
|
||||||
self.bump_freshness();
|
self.bump_freshness();
|
||||||
|
|||||||
@@ -8,6 +8,7 @@ use std::sync::Arc;
|
|||||||
use lance::dataset::NewColumnTransform;
|
use lance::dataset::NewColumnTransform;
|
||||||
|
|
||||||
use super::BaseTable;
|
use super::BaseTable;
|
||||||
|
use super::computed_columns;
|
||||||
use super::schema_evolution::AddColumnsResult;
|
use super::schema_evolution::AddColumnsResult;
|
||||||
use crate::{Error, Result};
|
use crate::{Error, Result};
|
||||||
|
|
||||||
@@ -15,6 +16,7 @@ use crate::{Error, Result};
|
|||||||
pub struct AddColumnsBuilder {
|
pub struct AddColumnsBuilder {
|
||||||
parent: Arc<dyn BaseTable>,
|
parent: Arc<dyn BaseTable>,
|
||||||
transform: Option<NewColumnTransform>,
|
transform: Option<NewColumnTransform>,
|
||||||
|
computed: Vec<(String, String)>,
|
||||||
read_columns: Option<Vec<String>>,
|
read_columns: Option<Vec<String>>,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -23,6 +25,7 @@ impl std::fmt::Debug for AddColumnsBuilder {
|
|||||||
f.debug_struct("AddColumnsBuilder")
|
f.debug_struct("AddColumnsBuilder")
|
||||||
.field("parent", &self.parent)
|
.field("parent", &self.parent)
|
||||||
.field("has_transform", &self.transform.is_some())
|
.field("has_transform", &self.transform.is_some())
|
||||||
|
.field("computed", &self.computed)
|
||||||
.field("read_columns", &self.read_columns)
|
.field("read_columns", &self.read_columns)
|
||||||
.finish()
|
.finish()
|
||||||
}
|
}
|
||||||
@@ -33,19 +36,57 @@ impl AddColumnsBuilder {
|
|||||||
Self {
|
Self {
|
||||||
parent,
|
parent,
|
||||||
transform: None,
|
transform: None,
|
||||||
|
computed: Vec::new(),
|
||||||
read_columns: None,
|
read_columns: None,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Set how the new columns' values are produced. Required.
|
/// Set how the new columns' values are produced.
|
||||||
pub fn transform(mut self, transform: NewColumnTransform) -> Self {
|
pub fn transform(mut self, transform: NewColumnTransform) -> Self {
|
||||||
self.transform = Some(transform);
|
self.transform = Some(transform);
|
||||||
self
|
self
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Add a column defined by `expression`, evaluated by a later refresh
|
||||||
|
/// rather than by this commit. Its type and inputs are derived from the
|
||||||
|
/// expression.
|
||||||
|
///
|
||||||
|
/// The column is committed with no values, so declaring one costs the same
|
||||||
|
/// on an empty table as on a large one. Rows get values from
|
||||||
|
/// [`Table::refresh_column`](super::Table::refresh_column), which fills
|
||||||
|
/// every fragment that has none -- including fragments appended since the
|
||||||
|
/// last refresh.
|
||||||
|
///
|
||||||
|
/// Refresh does not revisit a fragment it has filled, so mutating an input
|
||||||
|
/// leaves the value computed at fill time; recomputing means dropping the
|
||||||
|
/// column and declaring it again. An input cannot be renamed, retyped or
|
||||||
|
/// dropped while a declaration reads it, since the expression names it.
|
||||||
|
///
|
||||||
|
/// Local tables only: LanceDB Cloud and Enterprise reject a declaration
|
||||||
|
/// with `NotSupported`.
|
||||||
|
///
|
||||||
|
/// ```
|
||||||
|
/// # use lancedb::Table;
|
||||||
|
/// # async fn declare(table: &Table) -> Result<(), Box<dyn std::error::Error>> {
|
||||||
|
/// table
|
||||||
|
/// .add_columns()
|
||||||
|
/// .computed("doubled", "x * 2")
|
||||||
|
/// .execute()
|
||||||
|
/// .await?;
|
||||||
|
/// let filled = table.refresh_column("doubled").await?;
|
||||||
|
/// println!("filled {} rows", filled.rows_filled);
|
||||||
|
/// # Ok(())
|
||||||
|
/// # }
|
||||||
|
/// ```
|
||||||
|
pub fn computed(mut self, name: impl Into<String>, expression: impl Into<String>) -> Self {
|
||||||
|
self.computed.push((name.into(), expression.into()));
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
/// Limit which existing columns a [`NewColumnTransform::BatchUDF`] mapper
|
/// Limit which existing columns a [`NewColumnTransform::BatchUDF`] mapper
|
||||||
/// receives. Every other transform determines what it reads, so setting
|
/// receives. Every other transform, and a computed column, determines what
|
||||||
/// this alongside one is an error rather than a silent no-op.
|
/// it reads, so setting this alongside one is an error rather than a silent
|
||||||
|
/// no-op.
|
||||||
pub fn read_columns(mut self, columns: impl IntoIterator<Item = impl Into<String>>) -> Self {
|
pub fn read_columns(mut self, columns: impl IntoIterator<Item = impl Into<String>>) -> Self {
|
||||||
self.read_columns = Some(columns.into_iter().map(Into::into).collect());
|
self.read_columns = Some(columns.into_iter().map(Into::into).collect());
|
||||||
self
|
self
|
||||||
@@ -56,24 +97,43 @@ impl AddColumnsBuilder {
|
|||||||
let Self {
|
let Self {
|
||||||
parent,
|
parent,
|
||||||
transform,
|
transform,
|
||||||
|
computed,
|
||||||
read_columns,
|
read_columns,
|
||||||
} = self;
|
} = self;
|
||||||
|
|
||||||
let Some(transform) = transform else {
|
match (transform, computed.is_empty()) {
|
||||||
return Err(Error::InvalidInput {
|
(None, true) => Err(Error::InvalidInput {
|
||||||
message: "add_columns requires a transform".into(),
|
message: "add_columns requires a transform or a computed column".into(),
|
||||||
});
|
}),
|
||||||
};
|
// The two commit through different transforms, so one call covering
|
||||||
|
// both would be two commits and could half-apply.
|
||||||
if read_columns.is_some() && !matches!(transform, NewColumnTransform::BatchUDF(_)) {
|
(Some(_), false) => Err(Error::InvalidInput {
|
||||||
return Err(Error::InvalidInput {
|
message: "add_columns cannot mix a transform with computed columns; \
|
||||||
message: "read_columns applies only to a BatchUDF transform; \
|
they cannot be added atomically in one call"
|
||||||
every other transform determines what it reads"
|
|
||||||
.into(),
|
.into(),
|
||||||
});
|
}),
|
||||||
|
(Some(transform), true) => {
|
||||||
|
if read_columns.is_some() && !matches!(transform, NewColumnTransform::BatchUDF(_)) {
|
||||||
|
return Err(Error::InvalidInput {
|
||||||
|
message: "read_columns applies only to a BatchUDF transform; \
|
||||||
|
every other transform determines what it reads"
|
||||||
|
.into(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
parent.add_columns(transform, read_columns).await
|
||||||
|
}
|
||||||
|
(None, false) => {
|
||||||
|
if read_columns.is_some() {
|
||||||
|
return Err(Error::InvalidInput {
|
||||||
|
message: "read_columns applies only to a BatchUDF transform; \
|
||||||
|
a computed column's inputs come from its expression"
|
||||||
|
.into(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
let transform = computed_columns::declare(parent.schema().await?, &computed)?;
|
||||||
|
parent.add_columns(transform, None).await
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
parent.add_columns(transform, read_columns).await
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -85,8 +145,8 @@ mod tests {
|
|||||||
use arrow_schema::{DataType, Field, Schema};
|
use arrow_schema::{DataType, Field, Schema};
|
||||||
use lance::dataset::{BatchUDF, NewColumnTransform};
|
use lance::dataset::{BatchUDF, NewColumnTransform};
|
||||||
|
|
||||||
use crate::Table;
|
|
||||||
use crate::connect;
|
use crate::connect;
|
||||||
|
use crate::{Error, Table};
|
||||||
|
|
||||||
async fn table_with_two_columns(name: &str) -> Table {
|
async fn table_with_two_columns(name: &str) -> Table {
|
||||||
let conn = connect("memory://").execute().await.unwrap();
|
let conn = connect("memory://").execute().await.unwrap();
|
||||||
@@ -98,10 +158,7 @@ mod tests {
|
|||||||
async fn test_requires_a_transform() {
|
async fn test_requires_a_transform() {
|
||||||
let table = table_with_two_columns("no_transform").await;
|
let table = table_with_two_columns("no_transform").await;
|
||||||
let err = table.add_columns().execute().await.unwrap_err();
|
let err = table.add_columns().execute().await.unwrap_err();
|
||||||
assert!(
|
assert!(matches!(err, Error::InvalidInput { .. }));
|
||||||
err.to_string().contains("requires a transform"),
|
|
||||||
"got: {err}"
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
@@ -117,7 +174,7 @@ mod tests {
|
|||||||
.execute()
|
.execute()
|
||||||
.await
|
.await
|
||||||
.unwrap_err();
|
.unwrap_err();
|
||||||
assert!(err.to_string().contains("BatchUDF"), "got: {err}");
|
assert!(matches!(err, Error::InvalidInput { .. }));
|
||||||
|
|
||||||
let schema = table.schema().await.unwrap();
|
let schema = table.schema().await.unwrap();
|
||||||
assert!(
|
assert!(
|
||||||
@@ -126,6 +183,47 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_mixing_transform_and_computed_is_rejected() {
|
||||||
|
let table = table_with_two_columns("mixed_add").await;
|
||||||
|
let err = table
|
||||||
|
.add_columns()
|
||||||
|
.transform(NewColumnTransform::SqlExpressions(vec![(
|
||||||
|
"eager".into(),
|
||||||
|
"x * 2".into(),
|
||||||
|
)]))
|
||||||
|
.computed("lazy", "x * 3")
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap_err();
|
||||||
|
assert!(matches!(err, Error::InvalidInput { .. }));
|
||||||
|
|
||||||
|
let schema = table.schema().await.unwrap();
|
||||||
|
assert!(schema.field_with_name("eager").is_err());
|
||||||
|
assert!(schema.field_with_name("lazy").is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_read_columns_with_computed_is_rejected() {
|
||||||
|
let table = table_with_two_columns("read_cols_computed").await;
|
||||||
|
let err = table
|
||||||
|
.add_columns()
|
||||||
|
.computed("doubled", "x * 2")
|
||||||
|
.read_columns(["x"])
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap_err();
|
||||||
|
assert!(matches!(err, Error::InvalidInput { .. }));
|
||||||
|
assert!(
|
||||||
|
table
|
||||||
|
.schema()
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.field_with_name("doubled")
|
||||||
|
.is_err()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn test_read_columns_limits_what_a_batch_udf_sees() {
|
async fn test_read_columns_limits_what_a_batch_udf_sees() {
|
||||||
let table = table_with_two_columns("read_cols_udf").await;
|
let table = table_with_two_columns("read_cols_udf").await;
|
||||||
|
|||||||
@@ -0,0 +1,705 @@
|
|||||||
|
// SPDX-License-Identifier: Apache-2.0
|
||||||
|
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
//! Computed columns.
|
||||||
|
//!
|
||||||
|
//! A computed column is defined by a rule rather than by values supplied at
|
||||||
|
//! write time. Declaring one commits the column carrying that rule in field
|
||||||
|
//! metadata but no data, so the cost does not scale with the table; a later
|
||||||
|
//! refresh fills the rows.
|
||||||
|
//!
|
||||||
|
//! The rule is tagged by kind ([`ComputedColumnKind`]) because kinds differ in
|
||||||
|
//! where the column's type and inputs come from. A SQL expression is
|
||||||
|
//! self-describing -- both are derived from the expression, so a caller writes
|
||||||
|
//! neither -- while a kind resolved through a registry cannot be typed without
|
||||||
|
//! consulting it. Only SQL exists today; the tag is what lets another kind be
|
||||||
|
//! added without a second reading of the same key.
|
||||||
|
//!
|
||||||
|
//! [`computed_columns`] and [`computed_column_from_field`] read declarations
|
||||||
|
//! back off a schema.
|
||||||
|
|
||||||
|
use std::collections::HashMap;
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
use arrow_schema::{Field as ArrowField, Schema as ArrowSchema, SchemaRef};
|
||||||
|
use lance::dataset::NewColumnTransform;
|
||||||
|
use lance_datafusion::planner::Planner;
|
||||||
|
|
||||||
|
use crate::{Error, Result};
|
||||||
|
|
||||||
|
/// Field metadata key marking a column as computed. The value is `"true"`.
|
||||||
|
pub const COMPUTED_COLUMN_META_KEY: &str = "computed_column";
|
||||||
|
|
||||||
|
/// Field metadata key naming the kind of rule that defines the column.
|
||||||
|
pub const KIND_META_KEY: &str = "computed_column.kind";
|
||||||
|
|
||||||
|
/// Field metadata key holding the SQL expression that defines the column.
|
||||||
|
pub const EXPRESSION_META_KEY: &str = "computed_column.expression";
|
||||||
|
|
||||||
|
/// Field metadata key holding the column's inputs, as a JSON array of names.
|
||||||
|
pub const INPUTS_META_KEY: &str = "computed_column.inputs";
|
||||||
|
|
||||||
|
/// Value of [`KIND_META_KEY`] for a column defined by a SQL expression.
|
||||||
|
pub const SQL_KIND: &str = "sql";
|
||||||
|
|
||||||
|
/// The rule that defines a computed column's values.
|
||||||
|
///
|
||||||
|
/// Non-exhaustive: a kind added later is an additive change, and a caller that
|
||||||
|
/// only handles the kinds it knows keeps compiling.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
#[non_exhaustive]
|
||||||
|
pub enum ComputedColumnKind {
|
||||||
|
/// A SQL expression evaluated by DataFusion. It is the whole definition:
|
||||||
|
/// the column's type and its inputs are both derived from it.
|
||||||
|
Sql {
|
||||||
|
/// The expression.
|
||||||
|
expression: String,
|
||||||
|
},
|
||||||
|
/// A kind this version does not understand, written by a newer one.
|
||||||
|
///
|
||||||
|
/// Reported rather than hidden so a caller can tell a column it cannot
|
||||||
|
/// refresh apart from one that was never computed. Nothing produces this.
|
||||||
|
Unrecognized {
|
||||||
|
/// The kind as it was found in the metadata.
|
||||||
|
kind: String,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A computed column's declaration, as read back from field metadata.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct ComputedColumn {
|
||||||
|
/// Name of the computed column.
|
||||||
|
pub name: String,
|
||||||
|
/// The rule that defines it.
|
||||||
|
pub kind: ComputedColumnKind,
|
||||||
|
/// Columns the rule reads, recorded at declaration time.
|
||||||
|
///
|
||||||
|
/// Outside the kind because every kind has inputs and the consumers that
|
||||||
|
/// use them -- refresh planning, dependency ordering -- do not care which
|
||||||
|
/// kind produced them. Where they come from does differ, and that is
|
||||||
|
/// settled at declaration: derived from a SQL expression, supplied by the
|
||||||
|
/// caller for a kind that cannot be parsed.
|
||||||
|
pub inputs: Vec<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build the field metadata recording a SQL binding.
|
||||||
|
fn computed_column_metadata(expression: &str, inputs: &[String]) -> HashMap<String, String> {
|
||||||
|
HashMap::from([
|
||||||
|
(COMPUTED_COLUMN_META_KEY.to_string(), "true".to_string()),
|
||||||
|
(KIND_META_KEY.to_string(), SQL_KIND.to_string()),
|
||||||
|
(EXPRESSION_META_KEY.to_string(), expression.to_string()),
|
||||||
|
(
|
||||||
|
INPUTS_META_KEY.to_string(),
|
||||||
|
serde_json::to_string(inputs).unwrap_or_else(|_| "[]".to_string()),
|
||||||
|
),
|
||||||
|
])
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a field's computed-column declaration, if it carries one.
|
||||||
|
///
|
||||||
|
/// A field flagged computed but carrying no kind, or a SQL one missing its
|
||||||
|
/// expression, is not a computed column here: without the rule there is
|
||||||
|
/// nothing to refresh from, so it is reported as absent rather than as a
|
||||||
|
/// half-formed declaration. An unrecognized kind is different -- the rule is
|
||||||
|
/// there and intact, this version just cannot act on it -- and comes back as
|
||||||
|
/// [`ComputedColumnKind::Unrecognized`].
|
||||||
|
pub fn computed_column_from_field(field: &ArrowField) -> Option<ComputedColumn> {
|
||||||
|
let metadata = field.metadata();
|
||||||
|
if metadata.get(COMPUTED_COLUMN_META_KEY).map(String::as_str) != Some("true") {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let kind = match metadata.get(KIND_META_KEY)?.as_str() {
|
||||||
|
SQL_KIND => ComputedColumnKind::Sql {
|
||||||
|
expression: metadata.get(EXPRESSION_META_KEY)?.clone(),
|
||||||
|
},
|
||||||
|
other => ComputedColumnKind::Unrecognized {
|
||||||
|
kind: other.to_string(),
|
||||||
|
},
|
||||||
|
};
|
||||||
|
let inputs = metadata
|
||||||
|
.get(INPUTS_META_KEY)
|
||||||
|
.and_then(|raw| serde_json::from_str::<Vec<String>>(raw).ok())
|
||||||
|
.unwrap_or_default();
|
||||||
|
Some(ComputedColumn {
|
||||||
|
name: field.name().clone(),
|
||||||
|
kind,
|
||||||
|
inputs,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read every computed-column declaration carried by `schema`, in field order.
|
||||||
|
///
|
||||||
|
/// Introspection is a pure read of the schema the caller already holds, the
|
||||||
|
/// way a SQL catalog reports a generation expression as another column of
|
||||||
|
/// `information_schema.columns`.
|
||||||
|
pub fn computed_columns(schema: &ArrowSchema) -> Vec<ComputedColumn> {
|
||||||
|
schema
|
||||||
|
.fields()
|
||||||
|
.iter()
|
||||||
|
.filter_map(|field| computed_column_from_field(field))
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Reject a schema change to a column some declaration reads.
|
||||||
|
///
|
||||||
|
/// A binding is SQL text naming its inputs, so renaming, retyping or dropping
|
||||||
|
/// one leaves an expression that no longer resolves. Refusing the change keeps
|
||||||
|
/// a declaration that survived [`plan`] evaluable for as long as it exists.
|
||||||
|
///
|
||||||
|
/// Paths are compared at their root: a declaration reading `metadata` is
|
||||||
|
/// invalidated by a change to `metadata.age` just as surely.
|
||||||
|
pub(crate) fn ensure_not_an_input(schema: &ArrowSchema, paths: &[&str]) -> Result<()> {
|
||||||
|
let root = |path: &str| path.split('.').next().unwrap_or(path).to_string();
|
||||||
|
for declaration in computed_columns(schema) {
|
||||||
|
for path in paths {
|
||||||
|
// A declaration does not read itself, so it is free to be dropped
|
||||||
|
// or renamed along with its binding.
|
||||||
|
if declaration.name == root(path) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if declaration
|
||||||
|
.inputs
|
||||||
|
.iter()
|
||||||
|
.any(|input| root(input) == root(path))
|
||||||
|
{
|
||||||
|
return Err(Error::InvalidInput {
|
||||||
|
message: format!(
|
||||||
|
"column '{}' is read by computed column '{}'; drop that column first",
|
||||||
|
path, declaration.name
|
||||||
|
),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve `(name, expression)` pairs against `schema` into fields carrying
|
||||||
|
/// their bindings.
|
||||||
|
///
|
||||||
|
/// Everything that can be known statically is checked here rather than at
|
||||||
|
/// refresh time: that the expression parses, that every column it reads
|
||||||
|
/// exists, and that the target name is free. A declaration that survives this
|
||||||
|
/// is one a refresh can always act on.
|
||||||
|
pub(crate) fn plan(schema: SchemaRef, columns: &[(String, String)]) -> Result<Vec<ArrowField>> {
|
||||||
|
if columns.is_empty() {
|
||||||
|
return Err(Error::InvalidInput {
|
||||||
|
message: "at least one computed column is required".into(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
let planner = Planner::new(schema.clone());
|
||||||
|
let mut fields = Vec::with_capacity(columns.len());
|
||||||
|
let mut declared: Vec<&str> = Vec::with_capacity(columns.len());
|
||||||
|
|
||||||
|
for (name, expression) in columns {
|
||||||
|
if schema.field_with_name(name).is_ok() || declared.contains(&name.as_str()) {
|
||||||
|
return Err(Error::ColumnAlreadyExists { name: name.clone() });
|
||||||
|
}
|
||||||
|
|
||||||
|
let expr = planner
|
||||||
|
.parse_expr(expression)
|
||||||
|
.and_then(|expr| planner.optimize_expr(expr))
|
||||||
|
.map_err(|e| Error::InvalidExpression {
|
||||||
|
column: name.clone(),
|
||||||
|
message: e.to_string(),
|
||||||
|
})?;
|
||||||
|
|
||||||
|
let mut inputs = Planner::column_names_in_expr(&expr);
|
||||||
|
inputs.sort();
|
||||||
|
inputs.dedup();
|
||||||
|
|
||||||
|
// Resolved here rather than left to the planner so an unknown column
|
||||||
|
// names itself in the error instead of surfacing as a plan failure.
|
||||||
|
let mut indices = Vec::with_capacity(inputs.len());
|
||||||
|
for input in &inputs {
|
||||||
|
let index = schema
|
||||||
|
.index_of(input)
|
||||||
|
.map_err(|_| Error::InvalidExpression {
|
||||||
|
column: name.clone(),
|
||||||
|
message: format!("unknown column '{input}'"),
|
||||||
|
})?;
|
||||||
|
indices.push(index);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Physical expressions address columns by position, so the planner
|
||||||
|
// that types the expression has to be built on the projected schema
|
||||||
|
// the refresh will actually read.
|
||||||
|
let read_schema =
|
||||||
|
Arc::new(
|
||||||
|
schema
|
||||||
|
.project(&indices)
|
||||||
|
.map_err(|e| Error::InvalidExpression {
|
||||||
|
column: name.clone(),
|
||||||
|
message: e.to_string(),
|
||||||
|
})?,
|
||||||
|
);
|
||||||
|
let physical = Planner::new(read_schema.clone())
|
||||||
|
.create_physical_expr(&expr)
|
||||||
|
.map_err(|e| Error::InvalidExpression {
|
||||||
|
column: name.clone(),
|
||||||
|
message: e.to_string(),
|
||||||
|
})?;
|
||||||
|
let data_type =
|
||||||
|
physical
|
||||||
|
.data_type(read_schema.as_ref())
|
||||||
|
.map_err(|e| Error::InvalidExpression {
|
||||||
|
column: name.clone(),
|
||||||
|
message: e.to_string(),
|
||||||
|
})?;
|
||||||
|
|
||||||
|
// Declared columns start entirely null, so nullability is a property
|
||||||
|
// of the declaration rather than of what the expression yields.
|
||||||
|
fields.push(
|
||||||
|
ArrowField::new(name, data_type, true)
|
||||||
|
.with_metadata(computed_column_metadata(expression, &inputs)),
|
||||||
|
);
|
||||||
|
declared.push(name);
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(fields)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build the transform that declares `columns` against `schema`.
|
||||||
|
///
|
||||||
|
/// An all-null column is how a binding with no values yet is carried into a
|
||||||
|
/// commit; that it is spelled `AllNulls` is a detail of the commit, not of the
|
||||||
|
/// column, which is why this is internal and
|
||||||
|
/// [`AddColumnsBuilder::computed`](super::AddColumnsBuilder::computed) is the
|
||||||
|
/// public way in.
|
||||||
|
pub(crate) fn declare(
|
||||||
|
schema: SchemaRef,
|
||||||
|
columns: &[(String, String)],
|
||||||
|
) -> Result<NewColumnTransform> {
|
||||||
|
let fields = plan(schema, columns)?;
|
||||||
|
Ok(NewColumnTransform::AllNulls(Arc::new(ArrowSchema::new(
|
||||||
|
fields,
|
||||||
|
))))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Commit a declaration of a kind this version does not produce, the way a
|
||||||
|
/// newer lancedb would leave one behind. Shared with the refresh tests, which
|
||||||
|
/// need the same column to check that refresh refuses it.
|
||||||
|
#[cfg(test)]
|
||||||
|
pub(super) async fn add_foreign_kind(table: &crate::Table, name: &str, kind: &str) {
|
||||||
|
use arrow_schema::DataType;
|
||||||
|
|
||||||
|
let field = ArrowField::new(name, DataType::Int32, true).with_metadata(HashMap::from([
|
||||||
|
(COMPUTED_COLUMN_META_KEY.to_string(), "true".to_string()),
|
||||||
|
(KIND_META_KEY.to_string(), kind.to_string()),
|
||||||
|
(INPUTS_META_KEY.to_string(), r#"["x"]"#.to_string()),
|
||||||
|
]));
|
||||||
|
table
|
||||||
|
.add_columns()
|
||||||
|
.transform(NewColumnTransform::AllNulls(Arc::new(ArrowSchema::new(
|
||||||
|
vec![field],
|
||||||
|
))))
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use arrow_array::record_batch;
|
||||||
|
use arrow_schema::DataType;
|
||||||
|
use futures::TryStreamExt;
|
||||||
|
use lance::dataset::ColumnAlteration;
|
||||||
|
|
||||||
|
use super::*;
|
||||||
|
use crate::connect;
|
||||||
|
use crate::query::{ExecutableQuery, QueryBase, Select};
|
||||||
|
use crate::{Error, Table};
|
||||||
|
|
||||||
|
async fn table_with_ints(name: &str) -> Table {
|
||||||
|
let conn = connect("memory://").execute().await.unwrap();
|
||||||
|
let batch = record_batch!(("x", Int32, [1, 2, 3])).unwrap();
|
||||||
|
conn.create_table(name, batch).execute().await.unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Declare `columns` the way a caller would: plan the expressions, then
|
||||||
|
/// add them through the ordinary column API.
|
||||||
|
async fn add_computed(table: &Table, columns: &[(String, String)]) -> Result<u64> {
|
||||||
|
let mut builder = table.add_columns();
|
||||||
|
for (name, expression) in columns {
|
||||||
|
builder = builder.computed(name, expression);
|
||||||
|
}
|
||||||
|
Ok(builder.execute().await?.version)
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn declared(table: &Table) -> Vec<ComputedColumn> {
|
||||||
|
computed_columns(table.schema().await.unwrap().as_ref())
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_declare_infers_type_and_inputs() {
|
||||||
|
let table = table_with_ints("declare_infers").await;
|
||||||
|
let initial = table.version().await.unwrap();
|
||||||
|
|
||||||
|
let version = add_computed(&table, &[("doubled".into(), "x * 2".into())])
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
assert!(version > initial);
|
||||||
|
|
||||||
|
let schema = table.schema().await.unwrap();
|
||||||
|
let field = schema.field_with_name("doubled").unwrap();
|
||||||
|
assert_eq!(field.data_type(), &DataType::Int32);
|
||||||
|
assert!(field.is_nullable());
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
declared(&table).await,
|
||||||
|
vec![ComputedColumn {
|
||||||
|
name: "doubled".into(),
|
||||||
|
kind: ComputedColumnKind::Sql {
|
||||||
|
expression: "x * 2".into()
|
||||||
|
},
|
||||||
|
inputs: vec!["x".into()],
|
||||||
|
}]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The binding reaches the schema only if `AllNulls` carries per-field
|
||||||
|
/// metadata through the commit. The whole representation rests on it.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_all_nulls_preserves_field_metadata() {
|
||||||
|
let table = table_with_ints("metadata_survives").await;
|
||||||
|
add_computed(&table, &[("doubled".into(), "x * 2".into())])
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let schema = table.schema().await.unwrap();
|
||||||
|
let metadata = schema.field_with_name("doubled").unwrap().metadata();
|
||||||
|
assert_eq!(
|
||||||
|
metadata.get(COMPUTED_COLUMN_META_KEY).map(String::as_str),
|
||||||
|
Some("true")
|
||||||
|
);
|
||||||
|
assert_eq!(metadata.get(KIND_META_KEY).map(String::as_str), Some("sql"));
|
||||||
|
assert_eq!(
|
||||||
|
metadata.get(EXPRESSION_META_KEY).map(String::as_str),
|
||||||
|
Some("x * 2")
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
metadata.get(INPUTS_META_KEY).map(String::as_str),
|
||||||
|
Some(r#"["x"]"#)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_declared_column_is_all_null() {
|
||||||
|
let table = table_with_ints("declare_is_null").await;
|
||||||
|
add_computed(&table, &[("doubled".into(), "x * 2".into())])
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let batches = table
|
||||||
|
.query()
|
||||||
|
.select(Select::columns(&["doubled"]))
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.try_collect::<Vec<_>>()
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let total: usize = batches.iter().map(|b| b.num_rows()).sum();
|
||||||
|
assert_eq!(total, 3);
|
||||||
|
for batch in &batches {
|
||||||
|
assert_eq!(batch["doubled"].null_count(), batch.num_rows());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_unknown_column_fails_at_declare_time() {
|
||||||
|
let table = table_with_ints("unknown_input").await;
|
||||||
|
let err = add_computed(&table, &[("bad".into(), "missing + 1".into())])
|
||||||
|
.await
|
||||||
|
.unwrap_err();
|
||||||
|
assert!(matches!(err, Error::InvalidExpression { column, .. } if column == "bad"));
|
||||||
|
|
||||||
|
let schema = table.schema().await.unwrap();
|
||||||
|
assert!(schema.field_with_name("bad").is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_unparsable_expression_fails_at_declare_time() {
|
||||||
|
let table = table_with_ints("bad_syntax").await;
|
||||||
|
let err = add_computed(&table, &[("bad".into(), "x *".into())])
|
||||||
|
.await
|
||||||
|
.unwrap_err();
|
||||||
|
assert!(matches!(err, Error::InvalidExpression { column, .. } if column == "bad"));
|
||||||
|
assert!(
|
||||||
|
table
|
||||||
|
.schema()
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.field_with_name("bad")
|
||||||
|
.is_err()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A user-defined function is an expression like any other; only its
|
||||||
|
/// resolution is missing. When a registry-aware planner exists this
|
||||||
|
/// becomes a supported declaration rather than a new API.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_unregistered_function_is_rejected_for_now() {
|
||||||
|
let table = table_with_ints("udf_not_yet").await;
|
||||||
|
let err = add_computed(&table, &[("vec".into(), "embed(x)".into())])
|
||||||
|
.await
|
||||||
|
.unwrap_err();
|
||||||
|
assert!(matches!(err, Error::InvalidExpression { column, .. } if column == "vec"));
|
||||||
|
assert!(
|
||||||
|
table
|
||||||
|
.schema()
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.field_with_name("vec")
|
||||||
|
.is_err()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_existing_column_name_is_rejected() {
|
||||||
|
let table = table_with_ints("name_taken").await;
|
||||||
|
let err = add_computed(&table, &[("x".into(), "x * 2".into())])
|
||||||
|
.await
|
||||||
|
.unwrap_err();
|
||||||
|
assert!(matches!(err, Error::ColumnAlreadyExists { name } if name == "x"));
|
||||||
|
assert!(declared(&table).await.is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_constant_expression_needs_no_inputs() {
|
||||||
|
let table = table_with_ints("constant").await;
|
||||||
|
add_computed(&table, &[("answer".into(), "42".into())])
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let declared = declared(&table).await;
|
||||||
|
assert_eq!(declared.len(), 1);
|
||||||
|
assert!(declared[0].inputs.is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_multiple_columns_in_one_commit() {
|
||||||
|
let table = table_with_ints("multi").await;
|
||||||
|
let initial = table.version().await.unwrap();
|
||||||
|
|
||||||
|
add_computed(
|
||||||
|
&table,
|
||||||
|
&[
|
||||||
|
("plus".into(), "x + 1".into()),
|
||||||
|
("squared".into(), "x * x".into()),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
assert_eq!(table.version().await.unwrap(), initial + 1);
|
||||||
|
let declared = declared(&table).await;
|
||||||
|
assert_eq!(declared.len(), 2);
|
||||||
|
assert_eq!(declared[0].name, "plus");
|
||||||
|
assert_eq!(declared[1].name, "squared");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_duplicate_declaration_in_one_call_is_rejected() {
|
||||||
|
let table = table_with_ints("dupe").await;
|
||||||
|
let err = add_computed(
|
||||||
|
&table,
|
||||||
|
&[
|
||||||
|
("dup".into(), "x + 1".into()),
|
||||||
|
("dup".into(), "x + 2".into()),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.unwrap_err();
|
||||||
|
assert!(matches!(err, Error::ColumnAlreadyExists { name } if name == "dup"));
|
||||||
|
assert!(declared(&table).await.is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A column added by an ordinary transform is materialized, not bound, so
|
||||||
|
/// it carries no declaration to report.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_ordinary_columns_are_not_reported_as_computed() {
|
||||||
|
let table = table_with_ints("plain").await;
|
||||||
|
assert!(declared(&table).await.is_empty());
|
||||||
|
|
||||||
|
table
|
||||||
|
.add_columns()
|
||||||
|
.transform(NewColumnTransform::SqlExpressions(vec![(
|
||||||
|
"eager".into(),
|
||||||
|
"x * 2".into(),
|
||||||
|
)]))
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
assert!(declared(&table).await.is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Built-in functions type the column the same way an operator does.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_builtin_function_inference() {
|
||||||
|
let conn = connect("memory://").execute().await.unwrap();
|
||||||
|
let batch = record_batch!(("name", Utf8, ["ada", "grace"]), ("n", Int32, [-1, 2])).unwrap();
|
||||||
|
let table = conn
|
||||||
|
.create_table("builtins", batch)
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
add_computed(
|
||||||
|
&table,
|
||||||
|
&[
|
||||||
|
("shout".into(), "upper(name)".into()),
|
||||||
|
("width".into(), "length(name)".into()),
|
||||||
|
("magnitude".into(), "abs(n)".into()),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let schema = table.schema().await.unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
schema.field_with_name("shout").unwrap().data_type(),
|
||||||
|
&DataType::Utf8
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
schema.field_with_name("magnitude").unwrap().data_type(),
|
||||||
|
&DataType::Int32
|
||||||
|
);
|
||||||
|
// length() returns a width-dependent integer type; assert it is one
|
||||||
|
// rather than pinning which.
|
||||||
|
assert!(
|
||||||
|
schema
|
||||||
|
.field_with_name("width")
|
||||||
|
.unwrap()
|
||||||
|
.data_type()
|
||||||
|
.is_integer()
|
||||||
|
);
|
||||||
|
|
||||||
|
let declared = declared(&table).await;
|
||||||
|
assert_eq!(declared.len(), 3);
|
||||||
|
assert_eq!(declared[0].inputs, vec!["name".to_string()]);
|
||||||
|
assert_eq!(declared[2].inputs, vec!["n".to_string()]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The reason the kind is tagged: a declaration written by a newer version
|
||||||
|
/// has to read back as a computed column this one cannot evaluate, not as
|
||||||
|
/// an ordinary column. Reported as absent it would be refreshable by
|
||||||
|
/// nothing and redeclarable over, silently.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_unrecognized_kind_is_reported_rather_than_hidden() {
|
||||||
|
let table = table_with_ints("foreign_kind").await;
|
||||||
|
super::add_foreign_kind(&table, "embedding", "udf").await;
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
declared(&table).await,
|
||||||
|
vec![ComputedColumn {
|
||||||
|
name: "embedding".into(),
|
||||||
|
kind: ComputedColumnKind::Unrecognized { kind: "udf".into() },
|
||||||
|
inputs: vec!["x".into()],
|
||||||
|
}]
|
||||||
|
);
|
||||||
|
|
||||||
|
let err = add_computed(&table, &[("embedding".into(), "x * 2".into())])
|
||||||
|
.await
|
||||||
|
.unwrap_err();
|
||||||
|
assert!(matches!(err, Error::ColumnAlreadyExists { name } if name == "embedding"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A kind is what makes a declaration readable at all, so the flag alone
|
||||||
|
/// is half-formed in the same way a missing expression is.
|
||||||
|
#[test]
|
||||||
|
fn test_flag_without_a_kind_is_not_a_declaration() {
|
||||||
|
let field =
|
||||||
|
ArrowField::new("half", DataType::Int32, true).with_metadata(HashMap::from([(
|
||||||
|
COMPUTED_COLUMN_META_KEY.to_string(),
|
||||||
|
"true".to_string(),
|
||||||
|
)]));
|
||||||
|
assert_eq!(computed_column_from_field(&field), None);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A SQL declaration is its expression; without one there is nothing to
|
||||||
|
/// refresh from.
|
||||||
|
#[test]
|
||||||
|
fn test_sql_kind_without_an_expression_is_not_a_declaration() {
|
||||||
|
let field = ArrowField::new("half", DataType::Int32, true).with_metadata(HashMap::from([
|
||||||
|
(COMPUTED_COLUMN_META_KEY.to_string(), "true".to_string()),
|
||||||
|
(KIND_META_KEY.to_string(), SQL_KIND.to_string()),
|
||||||
|
]));
|
||||||
|
assert_eq!(computed_column_from_field(&field), None);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_inputs_are_deduplicated_and_sorted() {
|
||||||
|
let conn = connect("memory://").execute().await.unwrap();
|
||||||
|
let batch = record_batch!(("b", Int32, [1, 2]), ("a", Int32, [3, 4])).unwrap();
|
||||||
|
let table = conn.create_table("dedupe", batch).execute().await.unwrap();
|
||||||
|
|
||||||
|
add_computed(&table, &[("total".into(), "b + a + b".into())])
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
declared(&table).await[0].inputs,
|
||||||
|
vec!["a".to_string(), "b".to_string()]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_dropping_an_input_is_refused() {
|
||||||
|
let table = table_with_ints("drop_input").await;
|
||||||
|
add_computed(&table, &[("doubled".into(), "x * 2".into())])
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let err = table.drop_columns(&["x"]).await.unwrap_err();
|
||||||
|
assert!(
|
||||||
|
matches!(&err, Error::InvalidInput { message } if message.contains("doubled")),
|
||||||
|
"{err:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_renaming_an_input_is_refused() {
|
||||||
|
let table = table_with_ints("rename_input").await;
|
||||||
|
add_computed(&table, &[("doubled".into(), "x * 2".into())])
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let err = table
|
||||||
|
.alter_columns(&[ColumnAlteration::new("x".into()).rename("y".into())])
|
||||||
|
.await
|
||||||
|
.unwrap_err();
|
||||||
|
assert!(
|
||||||
|
matches!(&err, Error::InvalidInput { message } if message.contains("doubled")),
|
||||||
|
"{err:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Nothing resolves against nullability, so it is not a rebinding.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_altering_an_input_nullability_is_allowed() {
|
||||||
|
let table = table_with_ints("nullable_input").await;
|
||||||
|
add_computed(&table, &[("doubled".into(), "x * 2".into())])
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
table
|
||||||
|
.alter_columns(&[ColumnAlteration::new("x".into()).set_nullable(true)])
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A declaration does not read itself, so it travels with its binding.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_dropping_the_computed_column_is_allowed() {
|
||||||
|
let table = table_with_ints("drop_computed").await;
|
||||||
|
add_computed(&table, &[("doubled".into(), "x * 2".into())])
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
table.drop_columns(&["doubled"]).await.unwrap();
|
||||||
|
assert!(declared(&table).await.is_empty());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,523 @@
|
|||||||
|
// SPDX-License-Identifier: Apache-2.0
|
||||||
|
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||||
|
|
||||||
|
//! Filling computed columns.
|
||||||
|
//!
|
||||||
|
//! A row without a value gets one; a row that has one keeps it. Refresh is
|
||||||
|
//! therefore idempotent and does not observe input mutation -- once a row is
|
||||||
|
//! filled, changing what the expression reads leaves the stored result alone.
|
||||||
|
//!
|
||||||
|
//! Convergence comes from staging nothing when nothing would change, so an
|
||||||
|
//! expression yielding null settles after one pass rather than re-selecting
|
||||||
|
//! the same rows forever. Fragments that already cover the column and hold no
|
||||||
|
//! nulls are skipped without evaluating it at all.
|
||||||
|
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
use arrow_array::RecordBatch;
|
||||||
|
use arrow_schema::Schema as ArrowSchema;
|
||||||
|
use futures::{TryStreamExt, stream};
|
||||||
|
use lance::Dataset;
|
||||||
|
use lance::dataset::WriteDestination;
|
||||||
|
use lance::dataset::fragment::FileFragment;
|
||||||
|
use lance::dataset::transaction::Operation;
|
||||||
|
use lance_core::ROW_ID;
|
||||||
|
use lance_core::datatypes::Schema as LanceSchema;
|
||||||
|
use serde::{Deserialize, Serialize};
|
||||||
|
|
||||||
|
use super::NativeTable;
|
||||||
|
use super::computed_columns::{ComputedColumnKind, computed_column_from_field};
|
||||||
|
use crate::{Error, Result};
|
||||||
|
|
||||||
|
/// Alias the expression is projected under, so its result and the column's
|
||||||
|
/// current values can be read side by side.
|
||||||
|
const COMPUTED_ALIAS: &str = "__lancedb_computed";
|
||||||
|
|
||||||
|
/// The result of refreshing a computed column.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
|
||||||
|
pub struct RefreshColumnResult {
|
||||||
|
/// Rows that had a value computed.
|
||||||
|
#[serde(default)]
|
||||||
|
pub rows_filled: u64,
|
||||||
|
/// The commit version associated with the operation.
|
||||||
|
#[serde(default)]
|
||||||
|
pub version: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Internal implementation of the refresh logic.
|
||||||
|
pub(crate) async fn execute_refresh_column(
|
||||||
|
table: &NativeTable,
|
||||||
|
column: &str,
|
||||||
|
) -> Result<RefreshColumnResult> {
|
||||||
|
table.dataset.ensure_mutable()?;
|
||||||
|
let dataset = table.dataset.get().await?;
|
||||||
|
|
||||||
|
let expression = declared_expression(&dataset, column)?;
|
||||||
|
let field = dataset
|
||||||
|
.schema()
|
||||||
|
.field(column)
|
||||||
|
.ok_or_else(|| Error::ColumnNotFound {
|
||||||
|
name: column.to_string(),
|
||||||
|
})?;
|
||||||
|
// The dataset's own field, so the identity write_column checks against the
|
||||||
|
// manifest holds by construction.
|
||||||
|
let column_schema = LanceSchema {
|
||||||
|
fields: vec![field.clone()],
|
||||||
|
metadata: Default::default(),
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut rows_filled = 0u64;
|
||||||
|
let mut replacements = Vec::new();
|
||||||
|
for fragment in fragments_to_consider(&dataset, column, field.id).await? {
|
||||||
|
let Some((filled, values)) =
|
||||||
|
fill_fragment(&dataset, &fragment, column, &expression).await?
|
||||||
|
else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
rows_filled += filled;
|
||||||
|
replacements.push(
|
||||||
|
fragment
|
||||||
|
.write_column(stream::iter(values.into_iter().map(Ok)), &column_schema)
|
||||||
|
.await?,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
if replacements.is_empty() {
|
||||||
|
return Ok(RefreshColumnResult {
|
||||||
|
rows_filled: 0,
|
||||||
|
version: dataset.version().version,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
let read_version = dataset.version().version;
|
||||||
|
let new_dataset = Dataset::commit(
|
||||||
|
WriteDestination::Dataset(dataset.clone()),
|
||||||
|
Operation::DataReplacement { replacements },
|
||||||
|
Some(read_version),
|
||||||
|
None,
|
||||||
|
None,
|
||||||
|
Arc::new(Default::default()),
|
||||||
|
false,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let version = new_dataset.version().version;
|
||||||
|
table.dataset.update(new_dataset);
|
||||||
|
Ok(RefreshColumnResult {
|
||||||
|
rows_filled,
|
||||||
|
version,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The SQL expression `column` is declared with.
|
||||||
|
fn declared_expression(dataset: &Dataset, column: &str) -> Result<String> {
|
||||||
|
let schema = ArrowSchema::from(dataset.schema());
|
||||||
|
let field = schema
|
||||||
|
.field_with_name(column)
|
||||||
|
.map_err(|_| Error::ColumnNotFound {
|
||||||
|
name: column.to_string(),
|
||||||
|
})?;
|
||||||
|
let declaration =
|
||||||
|
computed_column_from_field(field).ok_or_else(|| Error::NotAComputedColumn {
|
||||||
|
name: column.to_string(),
|
||||||
|
})?;
|
||||||
|
match declaration.kind {
|
||||||
|
ComputedColumnKind::Sql { expression } => Ok(expression),
|
||||||
|
ComputedColumnKind::Unrecognized { kind } => Err(Error::NotSupported {
|
||||||
|
message: format!(
|
||||||
|
"computed column '{column}' is defined by '{kind}', which this version of \
|
||||||
|
lancedb cannot evaluate"
|
||||||
|
),
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Quote `name` as a lance SQL identifier.
|
||||||
|
///
|
||||||
|
/// Lance's dialect delimits with backticks, so a double-quoted name would
|
||||||
|
/// parse as a string literal rather than a column.
|
||||||
|
fn quote_identifier(name: &str) -> String {
|
||||||
|
format!("`{}`", name.replace('`', "``"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Fragments that could hold a row needing a value.
|
||||||
|
///
|
||||||
|
/// A fragment whose data files do not carry the field cannot hold one that
|
||||||
|
/// does. One that carries it is asked, since a row rewrite -- an update, or a
|
||||||
|
/// compaction folding an unfilled fragment into a filled one -- can leave
|
||||||
|
/// nulls behind a covering file.
|
||||||
|
async fn fragments_to_consider(
|
||||||
|
dataset: &Dataset,
|
||||||
|
column: &str,
|
||||||
|
field_id: i32,
|
||||||
|
) -> Result<Vec<FileFragment>> {
|
||||||
|
let unfilled = format!("{} IS NULL", quote_identifier(column));
|
||||||
|
let mut considered = Vec::new();
|
||||||
|
for fragment in dataset.get_fragments() {
|
||||||
|
let covered = fragment
|
||||||
|
.metadata()
|
||||||
|
.files
|
||||||
|
.iter()
|
||||||
|
.any(|file| file.fields.contains(&field_id));
|
||||||
|
if !covered || fragment.count_rows(Some(unfilled.clone())).await? > 0 {
|
||||||
|
considered.push(fragment);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(considered)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Compute one fragment's column, keeping every value it already holds.
|
||||||
|
///
|
||||||
|
/// `Ok(None)` when no live row gained a value, which is what keeps a refresh
|
||||||
|
/// from restaging a fragment whose expression yields null. Deleted rows are
|
||||||
|
/// carried through so the values line up positionally with the fragment's data
|
||||||
|
/// files; they are never read back, but the column file has to cover them.
|
||||||
|
async fn fill_fragment(
|
||||||
|
dataset: &Dataset,
|
||||||
|
fragment: &FileFragment,
|
||||||
|
column: &str,
|
||||||
|
expression: &str,
|
||||||
|
) -> Result<Option<(u64, Vec<RecordBatch>)>> {
|
||||||
|
let mut scanner = dataset.scan();
|
||||||
|
scanner
|
||||||
|
.with_fragments(vec![fragment.metadata().clone()])
|
||||||
|
.with_row_id()
|
||||||
|
.include_deleted_rows()
|
||||||
|
.project_with_transform(&[
|
||||||
|
(column, quote_identifier(column).as_str()),
|
||||||
|
(COMPUTED_ALIAS, expression),
|
||||||
|
])?;
|
||||||
|
|
||||||
|
let projected = Arc::new(ArrowSchema::new(vec![
|
||||||
|
ArrowSchema::from(dataset.schema())
|
||||||
|
.field_with_name(column)
|
||||||
|
.map_err(|_| Error::ColumnNotFound {
|
||||||
|
name: column.to_string(),
|
||||||
|
})?
|
||||||
|
.clone(),
|
||||||
|
]));
|
||||||
|
|
||||||
|
let missing = |name: &str| Error::Runtime {
|
||||||
|
message: format!("refreshing {column} produced no {name} column"),
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut filled = 0u64;
|
||||||
|
let mut values = Vec::new();
|
||||||
|
let mut batches = scanner.try_into_stream().await?;
|
||||||
|
while let Some(batch) = batches.try_next().await? {
|
||||||
|
let existing = batch
|
||||||
|
.column_by_name(column)
|
||||||
|
.ok_or_else(|| missing(column))?;
|
||||||
|
let computed = batch
|
||||||
|
.column_by_name(COMPUTED_ALIAS)
|
||||||
|
.ok_or_else(|| missing("expression"))?;
|
||||||
|
let row_ids = batch
|
||||||
|
.column_by_name(ROW_ID)
|
||||||
|
.ok_or_else(|| missing(ROW_ID))?;
|
||||||
|
|
||||||
|
// A row is filled only if it gains a value: an expression yielding null
|
||||||
|
// leaves it as unfilled as it was, which is what lets a refresh settle.
|
||||||
|
// A deleted row has a null row id; its value is written but not counted.
|
||||||
|
let unfilled = arrow::compute::is_null(existing.as_ref())?;
|
||||||
|
filled += (0..unfilled.len())
|
||||||
|
.filter(|i| unfilled.value(*i) && row_ids.is_valid(*i) && computed.is_valid(*i))
|
||||||
|
.count() as u64;
|
||||||
|
|
||||||
|
let merged = arrow_select::zip::zip(&unfilled, computed, existing)?;
|
||||||
|
values.push(RecordBatch::try_new(projected.clone(), vec![merged])?);
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok((filled > 0).then_some((filled, values)))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use arrow_array::{Int32Array, record_batch};
|
||||||
|
use futures::TryStreamExt;
|
||||||
|
|
||||||
|
use crate::connect;
|
||||||
|
use crate::query::{ExecutableQuery, QueryBase, Select};
|
||||||
|
use crate::{Error, Result, Table};
|
||||||
|
|
||||||
|
async fn table_with(name: &str, values: Vec<i32>) -> Table {
|
||||||
|
let conn = connect("memory://").execute().await.unwrap();
|
||||||
|
let batch = record_batch!(("x", Int32, values)).unwrap();
|
||||||
|
conn.create_table(name, batch).execute().await.unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn declare_doubled(table: &Table) -> Result<u64> {
|
||||||
|
Ok(table
|
||||||
|
.add_columns()
|
||||||
|
.computed("doubled", "x * 2")
|
||||||
|
.execute()
|
||||||
|
.await?
|
||||||
|
.version)
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn read(table: &Table, column: &str) -> Vec<Option<i32>> {
|
||||||
|
let batches = table
|
||||||
|
.query()
|
||||||
|
.select(Select::columns(&[column]))
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.try_collect::<Vec<_>>()
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
let mut values: Vec<Option<i32>> = batches
|
||||||
|
.iter()
|
||||||
|
.flat_map(|batch| {
|
||||||
|
batch[column]
|
||||||
|
.as_any()
|
||||||
|
.downcast_ref::<Int32Array>()
|
||||||
|
.unwrap()
|
||||||
|
.iter()
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
values.sort();
|
||||||
|
values
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn append(table: &Table, values: Vec<i32>) {
|
||||||
|
let batch = record_batch!(("x", Int32, values)).unwrap();
|
||||||
|
table.add(batch).execute().await.unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_refresh_fills_a_declared_column() {
|
||||||
|
let table = table_with("refresh_fills", vec![1, 2, 3]).await;
|
||||||
|
let declared = declare_doubled(&table).await.unwrap();
|
||||||
|
assert_eq!(read(&table, "doubled").await, vec![None, None, None]);
|
||||||
|
|
||||||
|
let result = table.refresh_column("doubled").await.unwrap();
|
||||||
|
assert!(result.version > declared);
|
||||||
|
assert_eq!(result.rows_filled, 3);
|
||||||
|
assert_eq!(
|
||||||
|
read(&table, "doubled").await,
|
||||||
|
vec![Some(2), Some(4), Some(6)]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Values written after the last refresh must be reachable by another one.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_refresh_fills_rows_appended_since_the_last_refresh() {
|
||||||
|
let table = table_with("refresh_appended", vec![1, 2]).await;
|
||||||
|
declare_doubled(&table).await.unwrap();
|
||||||
|
table.refresh_column("doubled").await.unwrap();
|
||||||
|
|
||||||
|
append(&table, vec![5, 6]).await;
|
||||||
|
assert_eq!(
|
||||||
|
read(&table, "doubled").await,
|
||||||
|
vec![None, None, Some(2), Some(4)]
|
||||||
|
);
|
||||||
|
|
||||||
|
let result = table.refresh_column("doubled").await.unwrap();
|
||||||
|
assert_eq!(result.rows_filled, 2);
|
||||||
|
assert_eq!(
|
||||||
|
read(&table, "doubled").await,
|
||||||
|
vec![Some(2), Some(4), Some(10), Some(12)]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_refresh_with_nothing_to_fill() {
|
||||||
|
let table = table_with("refresh_noop", vec![1, 2, 3]).await;
|
||||||
|
declare_doubled(&table).await.unwrap();
|
||||||
|
table.refresh_column("doubled").await.unwrap();
|
||||||
|
|
||||||
|
let again = table.refresh_column("doubled").await.unwrap();
|
||||||
|
assert_eq!(again.rows_filled, 0);
|
||||||
|
assert_eq!(
|
||||||
|
read(&table, "doubled").await,
|
||||||
|
vec![Some(2), Some(4), Some(6)]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A row is filled only by gaining a value, so an expression yielding null
|
||||||
|
/// settles at once instead of re-selecting the same rows forever. Nothing
|
||||||
|
/// is staged, so the version does not move either.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_refresh_converges_on_a_null_result() {
|
||||||
|
let table = table_with("refresh_null_result", vec![1, 2, 3]).await;
|
||||||
|
let declared = table
|
||||||
|
.add_columns()
|
||||||
|
.computed("maybe", "nullif(x, x)")
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.version;
|
||||||
|
|
||||||
|
let first = table.refresh_column("maybe").await.unwrap();
|
||||||
|
assert_eq!(first.rows_filled, 0);
|
||||||
|
assert_eq!(first.version, declared);
|
||||||
|
assert_eq!(read(&table, "maybe").await, vec![None, None, None]);
|
||||||
|
|
||||||
|
let again = table.refresh_column("maybe").await.unwrap();
|
||||||
|
assert_eq!(again.rows_filled, 0);
|
||||||
|
assert_eq!(again.version, declared);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The contract's boundary: a filled fragment is not revisited, so
|
||||||
|
/// mutating an input leaves the value computed at fill time.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_refresh_does_not_observe_input_mutation() {
|
||||||
|
let table = table_with("refresh_mutation", vec![1]).await;
|
||||||
|
declare_doubled(&table).await.unwrap();
|
||||||
|
table.refresh_column("doubled").await.unwrap();
|
||||||
|
assert_eq!(read(&table, "doubled").await, vec![Some(2)]);
|
||||||
|
|
||||||
|
table.update().column("x", "3").execute().await.unwrap();
|
||||||
|
|
||||||
|
let again = table.refresh_column("doubled").await.unwrap();
|
||||||
|
assert_eq!(again.rows_filled, 0);
|
||||||
|
assert_eq!(read(&table, "doubled").await, vec![Some(2)]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A row rewrite before the first refresh materializes the declared
|
||||||
|
/// column as null behind a covering data file. Those rows are still
|
||||||
|
/// unfilled and a later refresh has to reach them.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_update_before_the_first_refresh() {
|
||||||
|
let table = table_with("refresh_update_first", vec![1]).await;
|
||||||
|
declare_doubled(&table).await.unwrap();
|
||||||
|
|
||||||
|
table.update().column("x", "3").execute().await.unwrap();
|
||||||
|
|
||||||
|
let result = table.refresh_column("doubled").await.unwrap();
|
||||||
|
assert_eq!(result.rows_filled, 1);
|
||||||
|
assert_eq!(read(&table, "doubled").await, vec![Some(6)]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The contract holds row by row, not fragment by fragment: revisiting a
|
||||||
|
/// fragment to fill one row must not recompute a filled row sitting beside
|
||||||
|
/// it, even where the input behind it has since changed.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_refresh_does_not_recompute_a_filled_row_beside_an_unfilled_one() {
|
||||||
|
let table = table_with("refresh_mixed", vec![1, 2]).await;
|
||||||
|
declare_doubled(&table).await.unwrap();
|
||||||
|
table.refresh_column("doubled").await.unwrap();
|
||||||
|
|
||||||
|
append(&table, vec![5]).await;
|
||||||
|
table
|
||||||
|
.update()
|
||||||
|
.column("x", "100")
|
||||||
|
.only_if("x = 1")
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
table
|
||||||
|
.optimize(crate::table::OptimizeAction::Compact {
|
||||||
|
options: crate::table::CompactionOptions::default(),
|
||||||
|
remap_options: None,
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let result = table.refresh_column("doubled").await.unwrap();
|
||||||
|
assert_eq!(result.rows_filled, 1);
|
||||||
|
// 2 is the mutated row keeping the value it was filled with, not 200.
|
||||||
|
assert_eq!(
|
||||||
|
read(&table, "doubled").await,
|
||||||
|
vec![Some(2), Some(4), Some(10)]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Filling a fragment must not disturb the values it already holds, which
|
||||||
|
/// is what makes a compaction-mixed fragment safe to revisit.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_refresh_preserves_already_filled_rows() {
|
||||||
|
let table = table_with("refresh_preserves", vec![1, 2]).await;
|
||||||
|
declare_doubled(&table).await.unwrap();
|
||||||
|
table.refresh_column("doubled").await.unwrap();
|
||||||
|
|
||||||
|
append(&table, vec![5]).await;
|
||||||
|
table
|
||||||
|
.optimize(crate::table::OptimizeAction::Compact {
|
||||||
|
options: crate::table::CompactionOptions::default(),
|
||||||
|
remap_options: None,
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let result = table.refresh_column("doubled").await.unwrap();
|
||||||
|
assert_eq!(result.rows_filled, 1);
|
||||||
|
assert_eq!(
|
||||||
|
read(&table, "doubled").await,
|
||||||
|
vec![Some(2), Some(4), Some(10)]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_refresh_leaves_deleted_rows_alone() {
|
||||||
|
let table = table_with("refresh_deleted", vec![1, 2, 3, 4]).await;
|
||||||
|
declare_doubled(&table).await.unwrap();
|
||||||
|
table.delete("x = 2").await.unwrap();
|
||||||
|
|
||||||
|
let result = table.refresh_column("doubled").await.unwrap();
|
||||||
|
assert_eq!(result.rows_filled, 3);
|
||||||
|
assert_eq!(
|
||||||
|
read(&table, "doubled").await,
|
||||||
|
vec![Some(2), Some(6), Some(8)]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_refresh_a_constant_expression() {
|
||||||
|
let table = table_with("refresh_constant", vec![1, 2, 3]).await;
|
||||||
|
table
|
||||||
|
.add_columns()
|
||||||
|
.computed("answer", "42")
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let result = table.refresh_column("answer").await.unwrap();
|
||||||
|
assert_eq!(result.rows_filled, 3);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A name needing quotes reaches the evaluator intact: it is carried as a
|
||||||
|
/// projection alias, never spliced into SQL text.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_refresh_a_column_whose_name_needs_quoting() {
|
||||||
|
let table = table_with("refresh_quoted", vec![1, 2, 3]).await;
|
||||||
|
table
|
||||||
|
.add_columns()
|
||||||
|
.computed("double value", "x * 2")
|
||||||
|
.execute()
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let result = table.refresh_column("double value").await.unwrap();
|
||||||
|
assert_eq!(result.rows_filled, 3);
|
||||||
|
assert_eq!(
|
||||||
|
read(&table, "double value").await,
|
||||||
|
vec![Some(2), Some(4), Some(6)]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_refresh_rejects_a_plain_column() {
|
||||||
|
let table = table_with("refresh_plain", vec![1, 2, 3]).await;
|
||||||
|
let err = table.refresh_column("x").await.unwrap_err();
|
||||||
|
assert!(matches!(err, Error::NotAComputedColumn { name } if name == "x"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_refresh_rejects_an_unknown_column() {
|
||||||
|
let table = table_with("refresh_missing", vec![1, 2, 3]).await;
|
||||||
|
let err = table.refresh_column("nope").await.unwrap_err();
|
||||||
|
assert!(matches!(err, Error::ColumnNotFound { name } if name == "nope"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A declaration of a kind this version cannot evaluate is refused by
|
||||||
|
/// name, rather than mistaken for a plain column or fed to the SQL path.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_refresh_rejects_a_kind_it_cannot_evaluate() {
|
||||||
|
let table = table_with("refresh_foreign", vec![1, 2, 3]).await;
|
||||||
|
super::super::computed_columns::add_foreign_kind(&table, "embedding", "udf").await;
|
||||||
|
|
||||||
|
let err = table.refresh_column("embedding").await.unwrap_err();
|
||||||
|
assert!(matches!(err, Error::NotSupported { message } if message.contains("udf")));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -8,11 +8,13 @@
|
|||||||
//! - [`alter_columns`](execute_alter_columns): Rename columns, change types, or modify nullability
|
//! - [`alter_columns`](execute_alter_columns): Rename columns, change types, or modify nullability
|
||||||
//! - [`drop_columns`](execute_drop_columns): Remove columns from the table
|
//! - [`drop_columns`](execute_drop_columns): Remove columns from the table
|
||||||
|
|
||||||
|
use arrow_schema::Schema as ArrowSchema;
|
||||||
use lance::dataset::{ColumnAlteration, NewColumnTransform};
|
use lance::dataset::{ColumnAlteration, NewColumnTransform};
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
use super::NativeTable;
|
use super::NativeTable;
|
||||||
|
use super::computed_columns;
|
||||||
use crate::Result;
|
use crate::Result;
|
||||||
|
|
||||||
/// The result of an add columns operation.
|
/// The result of an add columns operation.
|
||||||
@@ -116,6 +118,14 @@ pub(crate) async fn execute_alter_columns(
|
|||||||
) -> Result<AlterColumnsResult> {
|
) -> Result<AlterColumnsResult> {
|
||||||
table.dataset.ensure_mutable()?;
|
table.dataset.ensure_mutable()?;
|
||||||
let mut dataset = (*table.dataset.get().await?).clone();
|
let mut dataset = (*table.dataset.get().await?).clone();
|
||||||
|
// Nullability is not part of what an expression resolves against, so only
|
||||||
|
// a rename or a retype can invalidate a binding.
|
||||||
|
let rebinding = alterations
|
||||||
|
.iter()
|
||||||
|
.filter(|alteration| alteration.rename.is_some() || alteration.data_type.is_some())
|
||||||
|
.map(|alteration| alteration.path.as_str())
|
||||||
|
.collect::<Vec<_>>();
|
||||||
|
computed_columns::ensure_not_an_input(&ArrowSchema::from(dataset.schema()), &rebinding)?;
|
||||||
dataset.alter_columns(alterations).await?;
|
dataset.alter_columns(alterations).await?;
|
||||||
let version = dataset.version().version;
|
let version = dataset.version().version;
|
||||||
table.dataset.update(dataset);
|
table.dataset.update(dataset);
|
||||||
@@ -131,6 +141,7 @@ pub(crate) async fn execute_drop_columns(
|
|||||||
) -> Result<DropColumnsResult> {
|
) -> Result<DropColumnsResult> {
|
||||||
table.dataset.ensure_mutable()?;
|
table.dataset.ensure_mutable()?;
|
||||||
let mut dataset = (*table.dataset.get().await?).clone();
|
let mut dataset = (*table.dataset.get().await?).clone();
|
||||||
|
computed_columns::ensure_not_an_input(&ArrowSchema::from(dataset.schema()), columns)?;
|
||||||
dataset.drop_columns(columns).await?;
|
dataset.drop_columns(columns).await?;
|
||||||
let version = dataset.version().version;
|
let version = dataset.version().version;
|
||||||
table.dataset.update(dataset);
|
table.dataset.update(dataset);
|
||||||
|
|||||||
Reference in New Issue
Block a user