mirror of
https://github.com/lancedb/lancedb.git
synced 2026-08-26 07:58:31 +00:00
Compare commits
10 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 047f431837 | |||
| 928c3dde2d | |||
| 980818df26 | |||
| c429863122 | |||
| fc0d917d32 | |||
| def869bb78 | |||
| 9e4d8bd1c7 | |||
| 4148dfef72 | |||
| 0ac70a8b9f | |||
| 91c5f344d2 |
+1
-1
@@ -1,5 +1,5 @@
|
||||
[tool.bumpversion]
|
||||
current_version = "0.37.1-beta.1"
|
||||
current_version = "0.38.0-beta.0"
|
||||
parse = """(?x)
|
||||
(?P<major>0|[1-9]\\d*)\\.
|
||||
(?P<minor>0|[1-9]\\d*)\\.
|
||||
|
||||
Generated
+47
-48
@@ -3455,8 +3455,8 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c"
|
||||
|
||||
[[package]]
|
||||
name = "fsst"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"rand 0.9.5",
|
||||
@@ -4815,8 +4815,8 @@ checksum = "e037a2e1d8d5fdbd49b16a4ea09d5d6401c1f29eca5ff29d03d3824dba16256a"
|
||||
|
||||
[[package]]
|
||||
name = "lance"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"arrow",
|
||||
@@ -4888,8 +4888,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-arrow"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -4911,7 +4911,7 @@ dependencies = [
|
||||
[[package]]
|
||||
name = "lance-arrow-scalar"
|
||||
version = "58.0.0"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -4925,7 +4925,7 @@ dependencies = [
|
||||
[[package]]
|
||||
name = "lance-arrow-stats"
|
||||
version = "58.0.0"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -4934,8 +4934,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-bitpacking"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrayref",
|
||||
"crunchy",
|
||||
@@ -4945,8 +4945,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-core"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -4983,8 +4983,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-datafusion"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5003,7 +5003,6 @@ dependencies = [
|
||||
"jsonb",
|
||||
"lance-arrow",
|
||||
"lance-core",
|
||||
"lance-datagen",
|
||||
"log",
|
||||
"pin-project",
|
||||
"prost",
|
||||
@@ -5014,8 +5013,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-datagen"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5032,8 +5031,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-derive"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -5042,8 +5041,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-encoding"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow-arith",
|
||||
"arrow-array",
|
||||
@@ -5076,8 +5075,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-file"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow-arith",
|
||||
"arrow-array",
|
||||
@@ -5108,8 +5107,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-index"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"arrow",
|
||||
@@ -5173,8 +5172,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-index-core"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -5196,8 +5195,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-io"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5233,8 +5232,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-linalg"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -5248,8 +5247,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-namespace"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"async-trait",
|
||||
@@ -5261,8 +5260,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-namespace-impls"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-ipc",
|
||||
@@ -5301,9 +5300,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-namespace-reqwest-client"
|
||||
version = "0.8.6"
|
||||
version = "0.11.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ba3f0a235e3ed5f8805205649ccc7d7d0f3df23ce1294242c9265ad488d7f19d"
|
||||
checksum = "0a030196da1c994b63a96a4f0bf5b0cfa459fe6dadc9e962320246ca328da22a"
|
||||
dependencies = [
|
||||
"reqwest 0.12.28",
|
||||
"serde",
|
||||
@@ -5315,8 +5314,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-select"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -5330,8 +5329,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-table"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5371,8 +5370,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-testing"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -5385,8 +5384,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-tokenizer"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
dependencies = [
|
||||
"frostem",
|
||||
"icu_segmenter",
|
||||
@@ -5399,7 +5398,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lancedb"
|
||||
version = "0.37.1-beta.1"
|
||||
version = "0.38.0-beta.0"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"anyhow",
|
||||
@@ -5487,7 +5486,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lancedb-nodejs"
|
||||
version = "0.37.1-beta.1"
|
||||
version = "0.38.0-beta.0"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -5512,7 +5511,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lancedb-python"
|
||||
version = "0.37.1-beta.1"
|
||||
version = "0.38.0-beta.0"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"async-trait",
|
||||
|
||||
+14
-14
@@ -13,20 +13,20 @@ categories = ["database-implementations"]
|
||||
rust-version = "1.91.0"
|
||||
|
||||
[workspace.dependencies]
|
||||
lance = { "version" = "=11.0.0-beta.8", default-features = false, "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-core = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datagen = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-file = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-io = { "version" = "=11.0.0-beta.8", default-features = false, "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-index = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-linalg = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace-impls = { "version" = "=11.0.0-beta.8", default-features = false, "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-table = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-testing = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datafusion = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-encoding = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-arrow = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-core = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datagen = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-file = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-io = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-index = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-linalg = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace-impls = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-table = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-testing = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datafusion = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-encoding = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-arrow = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
ahash = "0.8"
|
||||
# Note that this one does not include pyarrow
|
||||
arrow = { version = "58.0.0", optional = false }
|
||||
|
||||
@@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`:
|
||||
<dependency>
|
||||
<groupId>com.lancedb</groupId>
|
||||
<artifactId>lancedb-core</artifactId>
|
||||
<version>0.37.1-beta.1</version>
|
||||
<version>0.38.0-beta.0</version>
|
||||
</dependency>
|
||||
```
|
||||
|
||||
|
||||
@@ -69,14 +69,34 @@ abstract addColumns(newColumnTransforms): Promise<AddColumnsResult>
|
||||
|
||||
Add new columns with defined values.
|
||||
|
||||
The `{ computed }` form stores the expression rather than evaluating it
|
||||
now: the column is committed with no values, and rows get them from
|
||||
[Table#refreshColumn](Table.md#refreshcolumn). Declaring one therefore costs the same on a
|
||||
large table as on an empty one.
|
||||
|
||||
A refresh does not revisit rows it has already filled, so mutating an
|
||||
input leaves the value computed at fill time; recomputing means dropping
|
||||
the column and declaring it again. While a declaration reads a column,
|
||||
that column cannot be renamed, retyped or dropped.
|
||||
|
||||
On LanceDB Cloud and Enterprise the expression is planned by the
|
||||
server, and the refresh runs as a server job -- see
|
||||
[Table#refreshColumnAsync](Table.md#refreshcolumnasync).
|
||||
|
||||
#### Parameters
|
||||
|
||||
* **newColumnTransforms**: `Field`<`any`> \| `Field`<`any`>[] \| `Schema`<`any`> \| [`AddColumnsSql`](../interfaces/AddColumnsSql.md)[]
|
||||
* **newColumnTransforms**:
|
||||
\| `Field`<`any`>
|
||||
\| `Field`<`any`>[]
|
||||
\| `Schema`<`any`>
|
||||
\| [`AddColumnsSql`](../interfaces/AddColumnsSql.md)[]
|
||||
\| `object`
|
||||
Either:
|
||||
- An array of objects with column names and SQL expressions to calculate values
|
||||
- A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
||||
- An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
||||
- An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
||||
- `{ computed }`, declaring columns defined by a SQL expression whose type and inputs are derived from it
|
||||
|
||||
#### Returns
|
||||
|
||||
@@ -85,6 +105,13 @@ Add new columns with defined values.
|
||||
A promise that resolves to an object
|
||||
containing the new version number of the table after adding the columns.
|
||||
|
||||
#### Example
|
||||
|
||||
```ts
|
||||
await table.addColumns({ computed: [{ name: "doubled", valueSql: "x * 2" }] });
|
||||
const { rowsFilled } = await table.refreshColumn("doubled");
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### alterColumns()
|
||||
@@ -718,6 +745,67 @@ for await (const batch of table.query()) {
|
||||
|
||||
***
|
||||
|
||||
### refreshColumn()
|
||||
|
||||
```ts
|
||||
abstract refreshColumn(column): Promise<RefreshColumnResult>
|
||||
```
|
||||
|
||||
Fill the rows of a computed column that hold no value yet.
|
||||
|
||||
Rows appended since the last refresh are filled by the next one; rows
|
||||
already filled are left as they are, so the call is idempotent and does
|
||||
not observe a mutated input. Local tables only: a remote refresh runs
|
||||
as a server job, through [Table#refreshColumnAsync](Table.md#refreshcolumnasync).
|
||||
|
||||
#### Parameters
|
||||
|
||||
* **column**: `string`
|
||||
The name of the computed column to fill.
|
||||
|
||||
#### Returns
|
||||
|
||||
`Promise`<[`RefreshColumnResult`](../interfaces/RefreshColumnResult.md)>
|
||||
|
||||
A promise that resolves to the
|
||||
number of rows filled and the new version number of the table.
|
||||
|
||||
***
|
||||
|
||||
### refreshColumnAsync()
|
||||
|
||||
```ts
|
||||
abstract refreshColumnAsync(column): Promise<Job>
|
||||
```
|
||||
|
||||
Like [Table#refreshColumn](Table.md#refreshcolumn), but returns a handle to the refresh
|
||||
job instead of blocking until it completes.
|
||||
|
||||
The job may already be complete when returned; callers must not assume
|
||||
the column is filled until [Job.wait](Job.md#wait) resolves. Invalid input --
|
||||
an unknown column, or one that is not computed -- rejects here rather
|
||||
than failing the job. On local tables the job runs in-process; on
|
||||
LanceDB Cloud and Enterprise it is the server's backfill job.
|
||||
|
||||
#### Parameters
|
||||
|
||||
* **column**: `string`
|
||||
The name of the computed column to fill.
|
||||
|
||||
#### Returns
|
||||
|
||||
`Promise`<[`Job`](Job.md)>
|
||||
|
||||
#### Example
|
||||
|
||||
```ts
|
||||
const job = await table.refreshColumnAsync("doubled");
|
||||
await job.wait();
|
||||
console.log(await job.status()); // "finished"
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### restore()
|
||||
|
||||
```ts
|
||||
|
||||
@@ -105,6 +105,7 @@
|
||||
- [OptimizeOptions](interfaces/OptimizeOptions.md)
|
||||
- [OptimizeStats](interfaces/OptimizeStats.md)
|
||||
- [QueryExecutionOptions](interfaces/QueryExecutionOptions.md)
|
||||
- [RefreshColumnResult](interfaces/RefreshColumnResult.md)
|
||||
- [RemovalStats](interfaces/RemovalStats.md)
|
||||
- [RenameTableOptions](interfaces/RenameTableOptions.md)
|
||||
- [RestNamespaceConfig](interfaces/RestNamespaceConfig.md)
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||
|
||||
***
|
||||
|
||||
[@lancedb/lancedb](../globals.md) / RefreshColumnResult
|
||||
|
||||
# Interface: RefreshColumnResult
|
||||
|
||||
## Properties
|
||||
|
||||
### rowsFilled
|
||||
|
||||
```ts
|
||||
rowsFilled: number;
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### version
|
||||
|
||||
```ts
|
||||
version: number;
|
||||
```
|
||||
@@ -8,7 +8,7 @@
|
||||
<parent>
|
||||
<groupId>com.lancedb</groupId>
|
||||
<artifactId>lancedb-parent</artifactId>
|
||||
<version>0.37.1-beta.1</version>
|
||||
<version>0.38.0-beta.0</version>
|
||||
<relativePath>../pom.xml</relativePath>
|
||||
</parent>
|
||||
|
||||
|
||||
+2
-2
@@ -6,7 +6,7 @@
|
||||
|
||||
<groupId>com.lancedb</groupId>
|
||||
<artifactId>lancedb-parent</artifactId>
|
||||
<version>0.37.1-beta.1</version>
|
||||
<version>0.38.0-beta.0</version>
|
||||
<packaging>pom</packaging>
|
||||
<name>${project.artifactId}</name>
|
||||
<description>LanceDB Java SDK Parent POM</description>
|
||||
@@ -28,7 +28,7 @@
|
||||
<properties>
|
||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||
<arrow.version>15.0.0</arrow.version>
|
||||
<lance-core.version>11.0.0-beta.8</lance-core.version>
|
||||
<lance-core.version>11.0.0-beta.13</lance-core.version>
|
||||
<spotless.skip>false</spotless.skip>
|
||||
<spotless.version>2.30.0</spotless.version>
|
||||
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
[package]
|
||||
name = "lancedb-nodejs"
|
||||
edition.workspace = true
|
||||
version = "0.37.1-beta.1"
|
||||
version = "0.38.0-beta.0"
|
||||
publish = false
|
||||
license.workspace = true
|
||||
description.workspace = true
|
||||
|
||||
@@ -1001,4 +1001,49 @@ describe("remote connection jobs surface", () => {
|
||||
},
|
||||
);
|
||||
});
|
||||
|
||||
it("addBases posts the bases array", async () => {
|
||||
const postedBodies: unknown[] = [];
|
||||
await withMockDatabase(
|
||||
(req, res) => {
|
||||
const path = req.url ?? "";
|
||||
if (path.endsWith("/describe/")) {
|
||||
res.writeHead(200, { "Content-Type": "application/json" }).end(
|
||||
JSON.stringify({
|
||||
name: "photos",
|
||||
version: 1,
|
||||
schema: { fields: [] },
|
||||
}),
|
||||
);
|
||||
return;
|
||||
}
|
||||
if (path.endsWith("/bases/")) {
|
||||
const chunks: Buffer[] = [];
|
||||
req.on("data", (chunk) => chunks.push(chunk));
|
||||
req.on("end", () => {
|
||||
postedBodies.push(JSON.parse(Buffer.concat(chunks).toString()));
|
||||
res
|
||||
.writeHead(200, { "Content-Type": "application/json" })
|
||||
.end(JSON.stringify({ version: 2 }));
|
||||
});
|
||||
return;
|
||||
}
|
||||
res.writeHead(404).end();
|
||||
},
|
||||
async (db) => {
|
||||
const table = await db.openTable("photos");
|
||||
await table.addBases({ path: "s3://bucket/media/" });
|
||||
},
|
||||
);
|
||||
expect(postedBodies).toEqual([
|
||||
{
|
||||
bases: [
|
||||
{
|
||||
path: "s3://bucket/media/",
|
||||
isDatasetRoot: false,
|
||||
},
|
||||
],
|
||||
},
|
||||
]);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
import * as fs from "fs";
|
||||
import * as path from "path";
|
||||
import * as tmp from "tmp";
|
||||
import { pathToFileURL } from "url";
|
||||
|
||||
import * as arrow15 from "apache-arrow-15";
|
||||
import * as arrow16 from "apache-arrow-16";
|
||||
@@ -3340,3 +3341,86 @@ describe("LSM merge insert", () => {
|
||||
await expect(table.query().useLsm(true).toArray()).rejects.toThrow();
|
||||
});
|
||||
});
|
||||
|
||||
describe("computed columns", () => {
|
||||
let tmpDir: tmp.DirResult;
|
||||
beforeEach(() => {
|
||||
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
||||
});
|
||||
afterEach(() => tmpDir.removeCallback());
|
||||
|
||||
it("declares a column and fills it on refresh", async () => {
|
||||
const db = await connect(tmpDir.name);
|
||||
const table = await db.createTable("computed", [{ x: 1 }, { x: 2 }]);
|
||||
|
||||
await table.addColumns({
|
||||
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
||||
});
|
||||
let rows = await table.query().toArray();
|
||||
expect(rows.map((r) => r.doubled)).toEqual([null, null]);
|
||||
|
||||
const result = await table.refreshColumn("doubled");
|
||||
expect(result.rowsFilled).toBe(2);
|
||||
|
||||
rows = await table.query().toArray();
|
||||
expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]);
|
||||
});
|
||||
|
||||
it("returns a job handle from refreshColumnAsync", async () => {
|
||||
const db = await connect(tmpDir.name);
|
||||
const table = await db.createTable("computed_job", [{ x: 1 }, { x: 2 }]);
|
||||
|
||||
await table.addColumns({
|
||||
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
||||
});
|
||||
|
||||
const job = await table.refreshColumnAsync("doubled");
|
||||
expect(job.id).toBeNull();
|
||||
await job.wait();
|
||||
expect(await job.status()).toBe("finished");
|
||||
|
||||
const rows = await table.query().toArray();
|
||||
expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]);
|
||||
|
||||
// Bad input rejects at the call, not through the job.
|
||||
await expect(table.refreshColumnAsync("x")).rejects.toThrow(
|
||||
"not a computed column",
|
||||
);
|
||||
});
|
||||
|
||||
it("fills rows added since the last refresh", async () => {
|
||||
const db = await connect(tmpDir.name);
|
||||
const table = await db.createTable("computed_append", [{ x: 1 }]);
|
||||
|
||||
await table.addColumns({
|
||||
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
||||
});
|
||||
await table.refreshColumn("doubled");
|
||||
await table.add([{ x: 5 }]);
|
||||
|
||||
const result = await table.refreshColumn("doubled");
|
||||
expect(result.rowsFilled).toBe(1);
|
||||
|
||||
const rows = await table.query().toArray();
|
||||
expect(rows.map((r) => r.doubled).sort()).toEqual([10, 2]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("table bases", () => {
|
||||
let tmpDir: tmp.DirResult;
|
||||
beforeEach(() => {
|
||||
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
||||
});
|
||||
afterEach(() => tmpDir.removeCallback());
|
||||
|
||||
it("addBases accepts a file uri", async () => {
|
||||
const conn = await connect(tmpDir.name);
|
||||
const table = await conn.createEmptyTable(
|
||||
"photos",
|
||||
new arrow.Schema([new arrow.Field("id", new arrow.Int64(), false)]),
|
||||
);
|
||||
const media = path.join(tmpDir.name, "media");
|
||||
fs.mkdirSync(media);
|
||||
await table.addBases(pathToFileURL(media).toString());
|
||||
});
|
||||
});
|
||||
|
||||
@@ -50,6 +50,7 @@ export {
|
||||
MergeResult,
|
||||
AddResult,
|
||||
AddColumnsResult,
|
||||
RefreshColumnResult,
|
||||
AlterColumnsResult,
|
||||
UpdateFieldMetadataResult,
|
||||
DeleteResult,
|
||||
@@ -129,6 +130,7 @@ export {
|
||||
|
||||
export {
|
||||
Table,
|
||||
TableBase,
|
||||
Branches,
|
||||
BranchColumnSummary,
|
||||
BranchColumnChange,
|
||||
|
||||
+131
-2
@@ -33,6 +33,7 @@ import {
|
||||
Job,
|
||||
Branches as NativeBranches,
|
||||
OptimizeStats,
|
||||
RefreshColumnResult,
|
||||
TableStatistics,
|
||||
Tags,
|
||||
UpdateFieldMetadataResult,
|
||||
@@ -77,6 +78,25 @@ export interface WriteProgress {
|
||||
done: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* An extra storage prefix registered on a table.
|
||||
*
|
||||
* `path` is an object-store URI. `name` is an optional alias. `isDatasetRoot`
|
||||
* is true when `path` points to a Lance dataset root. When false, `path`
|
||||
* points directly to the directory containing the referenced files.
|
||||
*/
|
||||
export interface TableBase {
|
||||
/** Object store URI such as `s3://bucket/media/`. */
|
||||
path: string;
|
||||
/** Optional alias. */
|
||||
name?: string;
|
||||
/**
|
||||
* True when `path` is a Lance dataset root. When false, `path` is the
|
||||
* directory containing the referenced files.
|
||||
*/
|
||||
isDatasetRoot?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Options for adding data to a table.
|
||||
*/
|
||||
@@ -525,18 +545,84 @@ export abstract class Table {
|
||||
abstract vectorSearch(vector: IntoVector | MultiVector): VectorQuery;
|
||||
/**
|
||||
* Add new columns with defined values.
|
||||
*
|
||||
* The `{ computed }` form stores the expression rather than evaluating it
|
||||
* now: the column is committed with no values, and rows get them from
|
||||
* {@link Table#refreshColumn}. Declaring one therefore costs the same on a
|
||||
* large table as on an empty one.
|
||||
*
|
||||
* A refresh does not revisit rows it has already filled, so mutating an
|
||||
* input leaves the value computed at fill time; recomputing means dropping
|
||||
* the column and declaring it again. While a declaration reads a column,
|
||||
* that column cannot be renamed, retyped or dropped.
|
||||
*
|
||||
* On LanceDB Cloud and Enterprise the expression is planned by the
|
||||
* server, and the refresh runs as a server job -- see
|
||||
* {@link Table#refreshColumnAsync}.
|
||||
* @param {AddColumnsSql[] | Field | Field[] | Schema} newColumnTransforms Either:
|
||||
* - An array of objects with column names and SQL expressions to calculate values
|
||||
* - A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
||||
* - An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
||||
* - An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
||||
* - `{ computed }`, declaring columns defined by a SQL expression whose type and inputs are derived from it
|
||||
* @returns {Promise<AddColumnsResult>} A promise that resolves to an object
|
||||
* containing the new version number of the table after adding the columns.
|
||||
* @example
|
||||
* ```ts
|
||||
* await table.addColumns({ computed: [{ name: "doubled", valueSql: "x * 2" }] });
|
||||
* const { rowsFilled } = await table.refreshColumn("doubled");
|
||||
* ```
|
||||
*/
|
||||
abstract addColumns(
|
||||
newColumnTransforms: AddColumnsSql[] | Field | Field[] | Schema,
|
||||
newColumnTransforms:
|
||||
| AddColumnsSql[]
|
||||
| Field
|
||||
| Field[]
|
||||
| Schema
|
||||
| { computed: AddColumnsSql[] },
|
||||
): Promise<AddColumnsResult>;
|
||||
|
||||
/**
|
||||
* Register additional storage bases for this table.
|
||||
*
|
||||
* A URI string is a non-root base with no alias.
|
||||
*/
|
||||
abstract addBases(
|
||||
bases: string | TableBase | Array<string | TableBase>,
|
||||
): Promise<void>;
|
||||
|
||||
/**
|
||||
* Fill the rows of a computed column that hold no value yet.
|
||||
*
|
||||
* Rows appended since the last refresh are filled by the next one; rows
|
||||
* already filled are left as they are, so the call is idempotent and does
|
||||
* not observe a mutated input. Local tables only: a remote refresh runs
|
||||
* as a server job, through {@link Table#refreshColumnAsync}.
|
||||
* @param {string} column The name of the computed column to fill.
|
||||
* @returns {Promise<RefreshColumnResult>} A promise that resolves to the
|
||||
* number of rows filled and the new version number of the table.
|
||||
*/
|
||||
abstract refreshColumn(column: string): Promise<RefreshColumnResult>;
|
||||
|
||||
/**
|
||||
* Like {@link Table#refreshColumn}, but returns a handle to the refresh
|
||||
* job instead of blocking until it completes.
|
||||
*
|
||||
* The job may already be complete when returned; callers must not assume
|
||||
* the column is filled until {@link Job.wait} resolves. Invalid input --
|
||||
* an unknown column, or one that is not computed -- rejects here rather
|
||||
* than failing the job. On local tables the job runs in-process; on
|
||||
* LanceDB Cloud and Enterprise it is the server's backfill job.
|
||||
* @param {string} column The name of the computed column to fill.
|
||||
* @example
|
||||
* ```ts
|
||||
* const job = await table.refreshColumnAsync("doubled");
|
||||
* await job.wait();
|
||||
* console.log(await job.status()); // "finished"
|
||||
* ```
|
||||
*/
|
||||
abstract refreshColumnAsync(column: string): Promise<Job>;
|
||||
|
||||
/**
|
||||
* Alter the name or nullability of columns.
|
||||
* @param {ColumnAlteration[]} columnAlterations One or more alterations to
|
||||
@@ -1088,8 +1174,22 @@ export class LocalTable extends Table {
|
||||
// TODO: Support BatchUDF
|
||||
|
||||
async addColumns(
|
||||
newColumnTransforms: AddColumnsSql[] | Field | Field[] | Schema,
|
||||
newColumnTransforms:
|
||||
| AddColumnsSql[]
|
||||
| Field
|
||||
| Field[]
|
||||
| Schema
|
||||
| { computed: AddColumnsSql[] },
|
||||
): Promise<AddColumnsResult> {
|
||||
// Columns defined by an expression are declared, not materialized here.
|
||||
if (
|
||||
typeof newColumnTransforms === "object" &&
|
||||
!Array.isArray(newColumnTransforms) &&
|
||||
"computed" in newColumnTransforms
|
||||
) {
|
||||
return await this.inner.addComputedColumns(newColumnTransforms.computed);
|
||||
}
|
||||
|
||||
// Handle single Field -> convert to array of Fields
|
||||
if (newColumnTransforms instanceof Field) {
|
||||
newColumnTransforms = [newColumnTransforms];
|
||||
@@ -1124,6 +1224,20 @@ export class LocalTable extends Table {
|
||||
throw new Error("Invalid input type for addColumns");
|
||||
}
|
||||
|
||||
async addBases(
|
||||
bases: string | TableBase | Array<string | TableBase>,
|
||||
): Promise<void> {
|
||||
await this.inner.addBases(normalizeBases(bases));
|
||||
}
|
||||
|
||||
async refreshColumn(column: string): Promise<RefreshColumnResult> {
|
||||
return await this.inner.refreshColumn(column);
|
||||
}
|
||||
|
||||
async refreshColumnAsync(column: string): Promise<Job> {
|
||||
return await this.inner.refreshColumnAsync(column);
|
||||
}
|
||||
|
||||
async alterColumns(
|
||||
columnAlterations: ColumnAlteration[],
|
||||
): Promise<AlterColumnsResult> {
|
||||
@@ -1316,6 +1430,21 @@ export class LocalTable extends Table {
|
||||
}
|
||||
}
|
||||
|
||||
function normalizeBases(
|
||||
bases: string | TableBase | Array<string | TableBase>,
|
||||
): TableBase[] {
|
||||
const baseInputs = Array.isArray(bases) ? bases : [bases];
|
||||
return baseInputs.map((base) =>
|
||||
typeof base === "string"
|
||||
? { path: base, isDatasetRoot: false }
|
||||
: {
|
||||
path: base.path,
|
||||
name: base.name,
|
||||
isDatasetRoot: base.isDatasetRoot ?? false,
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* A definition of a column alteration. The alteration changes the column at
|
||||
* `path` to have the new name `name`, to be nullable if `nullable` is true,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-darwin-arm64",
|
||||
"version": "0.37.1-beta.1",
|
||||
"version": "0.38.0-beta.0",
|
||||
"os": ["darwin"],
|
||||
"cpu": ["arm64"],
|
||||
"main": "lancedb.darwin-arm64.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
||||
"version": "0.37.1-beta.1",
|
||||
"version": "0.38.0-beta.0",
|
||||
"os": ["linux"],
|
||||
"cpu": ["arm64"],
|
||||
"main": "lancedb.linux-arm64-gnu.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-arm64-musl",
|
||||
"version": "0.37.1-beta.1",
|
||||
"version": "0.38.0-beta.0",
|
||||
"os": ["linux"],
|
||||
"cpu": ["arm64"],
|
||||
"main": "lancedb.linux-arm64-musl.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-x64-gnu",
|
||||
"version": "0.37.1-beta.1",
|
||||
"version": "0.38.0-beta.0",
|
||||
"os": ["linux"],
|
||||
"cpu": ["x64"],
|
||||
"main": "lancedb.linux-x64-gnu.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-x64-musl",
|
||||
"version": "0.37.1-beta.1",
|
||||
"version": "0.38.0-beta.0",
|
||||
"os": ["linux"],
|
||||
"cpu": ["x64"],
|
||||
"main": "lancedb.linux-x64-musl.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-win32-arm64-msvc",
|
||||
"version": "0.37.1-beta.1",
|
||||
"version": "0.38.0-beta.0",
|
||||
"os": [
|
||||
"win32"
|
||||
],
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-win32-x64-msvc",
|
||||
"version": "0.37.1-beta.1",
|
||||
"version": "0.38.0-beta.0",
|
||||
"os": ["win32"],
|
||||
"cpu": ["x64"],
|
||||
"main": "lancedb.win32-x64-msvc.node",
|
||||
|
||||
Generated
+2
-2
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb",
|
||||
"version": "0.37.1-beta.1",
|
||||
"version": "0.38.0-beta.0",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "@lancedb/lancedb",
|
||||
"version": "0.37.1-beta.1",
|
||||
"version": "0.38.0-beta.0",
|
||||
"cpu": [
|
||||
"x64",
|
||||
"arm64"
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@
|
||||
"ann"
|
||||
],
|
||||
"private": false,
|
||||
"version": "0.37.1-beta.1",
|
||||
"version": "0.38.0-beta.0",
|
||||
"main": "dist/index.js",
|
||||
"exports": {
|
||||
".": "./dist/index.js",
|
||||
|
||||
@@ -10,6 +10,7 @@ use lancedb::table::{
|
||||
AddDataMode, ColumnAlteration as LanceColumnAlteration, Duration,
|
||||
FieldMetadataUpdate as LanceFieldMetadataUpdate, FtsToken as LanceDbFtsToken,
|
||||
NewColumnTransform, OptimizeAction, OptimizeOptions, Ref, Table as LanceDbTable,
|
||||
TableBase as LanceTableBase,
|
||||
};
|
||||
use napi::bindgen_prelude::*;
|
||||
use napi::threadsafe_function::{ThreadsafeFunction, ThreadsafeFunctionCallMode};
|
||||
@@ -347,6 +348,40 @@ impl Table {
|
||||
Ok(res.into())
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn add_computed_columns(
|
||||
&self,
|
||||
columns: Vec<AddColumnsSql>,
|
||||
) -> napi::Result<AddColumnsResult> {
|
||||
let table = self.inner_ref()?;
|
||||
let mut builder = table.add_columns();
|
||||
for column in columns {
|
||||
builder = builder.computed(column.name, column.value_sql);
|
||||
}
|
||||
let res = builder.execute().await.default_error()?;
|
||||
Ok(res.into())
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn refresh_column(&self, column: String) -> napi::Result<RefreshColumnResult> {
|
||||
let res = self
|
||||
.inner_ref()?
|
||||
.refresh_column(column)
|
||||
.await
|
||||
.default_error()?;
|
||||
Ok(res.into())
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn refresh_column_async(&self, column: String) -> napi::Result<crate::job::Job> {
|
||||
let job = self
|
||||
.inner_ref()?
|
||||
.refresh_column_async(column)
|
||||
.await
|
||||
.default_error()?;
|
||||
Ok(crate::job::Job::new(job))
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn add_columns_with_schema(
|
||||
&self,
|
||||
@@ -412,6 +447,18 @@ impl Table {
|
||||
Ok(res.into())
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn add_bases(&self, bases: Vec<TableBase>) -> napi::Result<()> {
|
||||
self.inner_ref()?
|
||||
.add_bases(bases.into_iter().map(|base| LanceTableBase {
|
||||
path: base.path,
|
||||
name: base.name,
|
||||
is_dataset_root: base.is_dataset_root,
|
||||
}))
|
||||
.await
|
||||
.default_error()
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn drop_columns(&self, columns: Vec<String>) -> napi::Result<DropColumnsResult> {
|
||||
let col_refs = columns.iter().map(String::as_str).collect::<Vec<_>>();
|
||||
@@ -666,6 +713,18 @@ impl Table {
|
||||
}
|
||||
}
|
||||
|
||||
#[napi(object)]
|
||||
/// An extra storage prefix registered on a table.
|
||||
pub struct TableBase {
|
||||
/// Object store URI such as `s3://bucket/media/`.
|
||||
pub path: String,
|
||||
/// Optional alias.
|
||||
pub name: Option<String>,
|
||||
/// True when `path` is a Lance dataset root. When false, `path` is the
|
||||
/// directory containing the referenced files.
|
||||
pub is_dataset_root: bool,
|
||||
}
|
||||
|
||||
#[napi(object)]
|
||||
/// A description of an index currently configured on a column
|
||||
pub struct IndexConfig {
|
||||
@@ -1196,6 +1255,21 @@ pub struct AddColumnsResult {
|
||||
pub version: i64,
|
||||
}
|
||||
|
||||
#[napi(object)]
|
||||
pub struct RefreshColumnResult {
|
||||
pub rows_filled: i64,
|
||||
pub version: i64,
|
||||
}
|
||||
|
||||
impl From<lancedb::table::RefreshColumnResult> for RefreshColumnResult {
|
||||
fn from(value: lancedb::table::RefreshColumnResult) -> Self {
|
||||
Self {
|
||||
rows_filled: value.rows_filled as i64,
|
||||
version: value.version as i64,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<lancedb::table::AddColumnsResult> for AddColumnsResult {
|
||||
fn from(value: lancedb::table::AddColumnsResult) -> Self {
|
||||
Self {
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "lancedb-python"
|
||||
version = "0.37.1-beta.1"
|
||||
version = "0.38.0-beta.0"
|
||||
publish = false
|
||||
edition.workspace = true
|
||||
description = "Python bindings for LanceDB"
|
||||
|
||||
@@ -21,7 +21,7 @@ from .remote.db import RemoteDBConnection
|
||||
from .expr import Expr, col, lit, func
|
||||
from .schema import blob, vector, BlobType
|
||||
from .job import AsyncJob, Job
|
||||
from .table import AsyncTable, Table
|
||||
from .table import AsyncTable, Table, TableBase
|
||||
from .types import BaseTokenizerType
|
||||
from ._lancedb import Session
|
||||
from .namespace import (
|
||||
@@ -521,5 +521,6 @@ __all__ = [
|
||||
"RemoteDBConnection",
|
||||
"Session",
|
||||
"Table",
|
||||
"TableBase",
|
||||
"__version__",
|
||||
]
|
||||
|
||||
@@ -338,6 +338,11 @@ class Table:
|
||||
) -> list[FtsToken]: ...
|
||||
async def delete(self, filter: Union[str, PyExpr]) -> DeleteResult: ...
|
||||
async def add_columns(self, columns: list[tuple[str, str]]) -> AddColumnsResult: ...
|
||||
async def add_computed_columns(
|
||||
self, columns: list[tuple[str, str]]
|
||||
) -> AddColumnsResult: ...
|
||||
async def refresh_column(self, column: str) -> RefreshColumnResult: ...
|
||||
async def refresh_column_async(self, column: str) -> Job: ...
|
||||
async def add_columns_with_schema(self, schema: pa.Schema) -> AddColumnsResult: ...
|
||||
async def alter_columns(
|
||||
self, columns: list[dict[str, Any]]
|
||||
@@ -372,6 +377,7 @@ class Table:
|
||||
def take_offsets(self, offsets: list[int]) -> TakeQuery: ...
|
||||
def take_row_ids(self, row_ids: list[int]) -> TakeQuery: ...
|
||||
async def blob_columns(self) -> list[str]: ...
|
||||
async def add_bases(self, bases: list[Any]) -> None: ...
|
||||
async def fetch_blobs(
|
||||
self, column: str, row_ids: list[int]
|
||||
) -> pa.LargeBinaryArray: ...
|
||||
@@ -683,6 +689,10 @@ class LsmWriteSpec:
|
||||
class AddColumnsResult:
|
||||
version: int
|
||||
|
||||
class RefreshColumnResult:
|
||||
rows_filled: int
|
||||
version: int
|
||||
|
||||
class AlterColumnsResult:
|
||||
version: int
|
||||
|
||||
|
||||
@@ -50,7 +50,7 @@ from lancedb.index import (
|
||||
)
|
||||
from lancedb.job import Job
|
||||
from lancedb.remote.db import LOOP
|
||||
from lancedb.table import IndexConfigType, KNOWN_METRICS
|
||||
from lancedb.table import IndexConfigType, KNOWN_METRICS, TableBase
|
||||
import pyarrow as pa
|
||||
|
||||
from lancedb.common import DATA, VEC, VECTOR_COLUMN_NAME
|
||||
@@ -958,8 +958,19 @@ class RemoteTable(Table):
|
||||
def count_rows(self, filter: Optional[str] = None) -> int:
|
||||
return LOOP.run(self._table.count_rows(filter))
|
||||
|
||||
def add_columns(self, transforms: Dict[str, str]) -> AddColumnsResult:
|
||||
return LOOP.run(self._table.add_columns(transforms))
|
||||
def add_columns(
|
||||
self,
|
||||
transforms: Dict[str, str] | None = None,
|
||||
*,
|
||||
computed: Dict[str, str] | None = None,
|
||||
) -> AddColumnsResult:
|
||||
return LOOP.run(self._table.add_columns(transforms, computed=computed))
|
||||
|
||||
def refresh_column(self, column: str):
|
||||
return LOOP.run(self._table.refresh_column(column))
|
||||
|
||||
def refresh_column_async(self, column: str) -> Job:
|
||||
return Job(LOOP.run(self._table.refresh_column_async(column)))
|
||||
|
||||
def alter_columns(
|
||||
self, *alterations: Iterable[Dict[str, str]]
|
||||
@@ -1071,6 +1082,13 @@ class RemoteTable(Table):
|
||||
def blob_columns(self) -> list[str]:
|
||||
return LOOP.run(self._table.blob_columns())
|
||||
|
||||
def add_bases(
|
||||
self,
|
||||
bases: Union[str, TableBase, Iterable[Union[str, TableBase]]],
|
||||
) -> None:
|
||||
"""Register additional storage bases for this table."""
|
||||
LOOP.run(self._table.add_bases(bases))
|
||||
|
||||
def fetch_blobs(
|
||||
self, column: str, row_ids: Union[list[int], pa.Table]
|
||||
) -> pa.LargeBinaryArray:
|
||||
|
||||
@@ -19,6 +19,7 @@ from typing import (
|
||||
Iterable,
|
||||
List,
|
||||
Literal,
|
||||
Mapping,
|
||||
Optional,
|
||||
Sequence,
|
||||
Tuple,
|
||||
@@ -176,6 +177,7 @@ if TYPE_CHECKING:
|
||||
CompactionStats,
|
||||
Tag,
|
||||
AddColumnsResult,
|
||||
RefreshColumnResult,
|
||||
AddResult,
|
||||
AlterColumnsResult,
|
||||
UpdateFieldMetadataResult,
|
||||
@@ -709,6 +711,21 @@ def _normalize_progress(progress):
|
||||
return progress, False
|
||||
|
||||
|
||||
@dataclass
|
||||
class TableBase:
|
||||
"""An extra storage prefix registered on a table.
|
||||
|
||||
``path`` is an object-store URI. ``name`` is an optional alias.
|
||||
``is_dataset_root`` is true when ``path`` points to a Lance dataset
|
||||
root. When false, ``path`` points directly to the directory containing
|
||||
the referenced files.
|
||||
"""
|
||||
|
||||
path: str
|
||||
name: Optional[str] = None
|
||||
is_dataset_root: bool = False
|
||||
|
||||
|
||||
class Table(ABC):
|
||||
"""
|
||||
A Table is a collection of Records in a LanceDB Database.
|
||||
@@ -1567,6 +1584,18 @@ class Table(ABC):
|
||||
def blob_columns(self) -> list[str]:
|
||||
"""Names of the blob v2 columns declared on this table."""
|
||||
|
||||
def add_bases(
|
||||
self,
|
||||
bases: Union[str, TableBase, Iterable[Union[str, TableBase]]],
|
||||
) -> None:
|
||||
"""Register additional storage bases for this table.
|
||||
|
||||
A URI string is a non-root base with no alias::
|
||||
|
||||
table.add_bases("s3://bucket/media/")
|
||||
"""
|
||||
raise NotImplementedError
|
||||
|
||||
@abstractmethod
|
||||
def fetch_blobs(
|
||||
self, column: str, row_ids: Union[list[int], pa.Table]
|
||||
@@ -1916,7 +1945,14 @@ class Table(ABC):
|
||||
|
||||
@abstractmethod
|
||||
def add_columns(
|
||||
self, transforms: Dict[str, str] | pa.Field | List[pa.Field] | pa.Schema
|
||||
self,
|
||||
transforms: Dict[str, str]
|
||||
| pa.Field
|
||||
| List[pa.Field]
|
||||
| pa.Schema
|
||||
| None = None,
|
||||
*,
|
||||
computed: Dict[str, str] | None = None,
|
||||
):
|
||||
"""
|
||||
Add new columns with defined values.
|
||||
@@ -1930,11 +1966,95 @@ class Table(ABC):
|
||||
Alternatively, a pyarrow Field or Schema can be provided to add
|
||||
new columns with the specified data types. The new columns will
|
||||
be initialized with null values.
|
||||
computed: Dict[str, str], optional
|
||||
A map of column name to a SQL expression defining the column. The
|
||||
column's type and inputs are derived from the expression, so no
|
||||
data type is supplied.
|
||||
|
||||
Unlike ``transforms``, the expression is stored rather than
|
||||
evaluated now: the column is committed with no values, and rows get
|
||||
them from [`refresh_column`][lancedb.table.Table.refresh_column].
|
||||
Declaring one therefore costs the same on a large table as on an
|
||||
empty one.
|
||||
|
||||
A refresh does not revisit rows it has already filled, so mutating
|
||||
an input leaves the value computed at fill time; recomputing means
|
||||
dropping the column and declaring it again. While a declaration
|
||||
reads a column, that column cannot be renamed, retyped or dropped.
|
||||
|
||||
On LanceDB Cloud and Enterprise the expression is planned by the
|
||||
server, and the refresh runs as a server job -- see
|
||||
[`refresh_column_async`][lancedb.table.Table.refresh_column_async].
|
||||
Cannot be combined with ``transforms``.
|
||||
|
||||
Returns
|
||||
-------
|
||||
AddColumnsResult
|
||||
version: the new version number of the table after adding columns.
|
||||
|
||||
Examples
|
||||
--------
|
||||
>>> import lancedb
|
||||
>>> db = lancedb.connect("./.lancedb")
|
||||
>>> table = db.create_table("computed_demo", [{"x": 1}, {"x": 2}])
|
||||
>>> table.add_columns(computed={"doubled": "x * 2"})
|
||||
AddColumnsResult(version=2)
|
||||
>>> table.refresh_column("doubled")
|
||||
RefreshColumnResult(rows_filled=2, version=3)
|
||||
>>> table.to_arrow().sort_by("x").to_pandas()
|
||||
x doubled
|
||||
0 1 2
|
||||
1 2 4
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
def refresh_column(self, column: str) -> "RefreshColumnResult":
|
||||
"""
|
||||
Fill the rows of a computed column that hold no value yet.
|
||||
|
||||
Declared with ``add_columns(computed=...)``, a column starts empty and
|
||||
gets its values here. Rows appended since the last refresh are filled
|
||||
by the next one; rows already filled are left as they are, so the call
|
||||
is idempotent and does not observe a mutated input.
|
||||
|
||||
Local tables only: a remote refresh runs as a server job, through
|
||||
[`refresh_column_async`][lancedb.table.Table.refresh_column_async].
|
||||
|
||||
Parameters
|
||||
----------
|
||||
column: str
|
||||
The name of the computed column to fill.
|
||||
|
||||
Returns
|
||||
-------
|
||||
RefreshColumnResult
|
||||
rows_filled: the number of rows given a value.
|
||||
version: the new version number of the table.
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
def refresh_column_async(self, column: str) -> Job:
|
||||
"""
|
||||
Like :meth:`refresh_column`, but returns a handle to the refresh job
|
||||
instead of blocking until it completes.
|
||||
|
||||
The job may already be complete when returned; callers must not assume
|
||||
the column is filled until :meth:`Job.wait` returns. Invalid input --
|
||||
an unknown column, or one that is not computed -- raises here rather
|
||||
than failing the job. On local tables the job runs in-process; on
|
||||
LanceDB Cloud and Enterprise it is the server's backfill job.
|
||||
|
||||
Examples
|
||||
--------
|
||||
>>> import lancedb
|
||||
>>> db = lancedb.connect("./.lancedb")
|
||||
>>> table = db.create_table("computed_job_demo", [{"x": 1}, {"x": 2}])
|
||||
>>> table.add_columns(computed={"doubled": "x * 2"})
|
||||
AddColumnsResult(version=2)
|
||||
>>> job = table.refresh_column_async("doubled")
|
||||
>>> job.wait()
|
||||
>>> job.status()
|
||||
'finished'
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
@@ -2322,6 +2442,12 @@ class LanceTable(Table):
|
||||
def blob_columns(self) -> list[str]:
|
||||
return LOOP.run(self._table.blob_columns())
|
||||
|
||||
def add_bases(
|
||||
self,
|
||||
bases: Union[str, TableBase, Iterable[Union[str, TableBase]]],
|
||||
) -> None:
|
||||
LOOP.run(self._table.add_bases(bases))
|
||||
|
||||
def fetch_blobs(
|
||||
self, column: str, row_ids: Union[list[int], pa.Table]
|
||||
) -> pa.LargeBinaryArray:
|
||||
@@ -3939,9 +4065,28 @@ class LanceTable(Table):
|
||||
return LOOP.run(self._table.index_stats(index_name))
|
||||
|
||||
def add_columns(
|
||||
self, transforms: Dict[str, str] | pa.field | List[pa.field] | pa.Schema
|
||||
self,
|
||||
transforms: Dict[str, str]
|
||||
| pa.field
|
||||
| List[pa.field]
|
||||
| pa.Schema
|
||||
| None = None,
|
||||
*,
|
||||
computed: Dict[str, str] | None = None,
|
||||
) -> AddColumnsResult:
|
||||
return LOOP.run(self._table.add_columns(transforms))
|
||||
return LOOP.run(self._table.add_columns(transforms, computed=computed))
|
||||
|
||||
def refresh_column(self, column: str) -> "RefreshColumnResult":
|
||||
"""Fill a computed column's unfilled rows. See
|
||||
[`AsyncTable.refresh_column`][lancedb.AsyncTable.refresh_column]."""
|
||||
return LOOP.run(self._table.refresh_column(column))
|
||||
|
||||
def refresh_column_async(self, column: str) -> Job:
|
||||
"""Fill a computed column's unfilled rows, returning a handle to the
|
||||
refresh job. See
|
||||
[`Table.refresh_column_async`][lancedb.table.Table.refresh_column_async].
|
||||
"""
|
||||
return Job(LOOP.run(self._table.refresh_column_async(column)))
|
||||
|
||||
def alter_columns(
|
||||
self, *alterations: Iterable[Dict[str, str]]
|
||||
@@ -5856,7 +6001,14 @@ class AsyncTable:
|
||||
return await self._inner.update(updates_sql, where)
|
||||
|
||||
async def add_columns(
|
||||
self, transforms: dict[str, str] | pa.field | List[pa.field] | pa.Schema
|
||||
self,
|
||||
transforms: dict[str, str]
|
||||
| pa.field
|
||||
| List[pa.field]
|
||||
| pa.Schema
|
||||
| None = None,
|
||||
*,
|
||||
computed: dict[str, str] | None = None,
|
||||
) -> AddColumnsResult:
|
||||
"""
|
||||
Add new columns with defined values.
|
||||
@@ -5869,6 +6021,22 @@ class AsyncTable:
|
||||
each row in the table, and can reference existing columns.
|
||||
Alternatively, you can pass a pyarrow field or schema to add
|
||||
new columns with NULLs.
|
||||
computed: Dict[str, str], optional
|
||||
A map of column name to a SQL expression defining the column. The
|
||||
column's type and inputs are derived from the expression.
|
||||
|
||||
Unlike ``transforms``, the expression is stored rather than
|
||||
evaluated now: the column is committed with no values, and rows get
|
||||
them from
|
||||
[`refresh_column`][lancedb.table.AsyncTable.refresh_column].
|
||||
|
||||
A refresh does not revisit rows it has already filled, so mutating
|
||||
an input leaves the value computed at fill time. While a
|
||||
declaration reads a column, that column cannot be renamed, retyped
|
||||
or dropped.
|
||||
|
||||
On LanceDB Cloud and Enterprise the expression is planned by
|
||||
the server. Cannot be combined with ``transforms``.
|
||||
|
||||
Returns
|
||||
-------
|
||||
@@ -5882,11 +6050,71 @@ class AsyncTable:
|
||||
{isinstance(f, pa.Field) for f in transforms}
|
||||
):
|
||||
transforms = pa.schema(transforms)
|
||||
if computed:
|
||||
if transforms:
|
||||
raise ValueError(
|
||||
"add_columns cannot take both transforms and computed columns"
|
||||
)
|
||||
return await self._inner.add_computed_columns(list(computed.items()))
|
||||
if transforms is None:
|
||||
raise ValueError("add_columns requires transforms or computed columns")
|
||||
if isinstance(transforms, pa.Schema):
|
||||
return await self._inner.add_columns_with_schema(transforms)
|
||||
else:
|
||||
return await self._inner.add_columns(list(transforms.items()))
|
||||
|
||||
async def refresh_column(self, column: str) -> RefreshColumnResult:
|
||||
"""
|
||||
Fill the rows of a computed column that hold no value yet.
|
||||
|
||||
Declared with ``add_columns(computed=...)``, a column starts empty and
|
||||
gets its values here. Rows appended since the last refresh are filled
|
||||
by the next one; rows already filled are left as they are, so the call
|
||||
is idempotent and does not observe a mutated input.
|
||||
|
||||
Local tables only: a remote refresh runs as a server job, through
|
||||
[`refresh_column_async`][lancedb.table.Table.refresh_column_async].
|
||||
|
||||
Parameters
|
||||
----------
|
||||
column: str
|
||||
The name of the computed column to fill.
|
||||
|
||||
Returns
|
||||
-------
|
||||
RefreshColumnResult
|
||||
The number of rows filled and the new version of the table.
|
||||
"""
|
||||
return await self._inner.refresh_column(column)
|
||||
|
||||
async def refresh_column_async(self, column: str) -> AsyncJob:
|
||||
"""
|
||||
Like :meth:`refresh_column`, but returns a handle to the refresh job
|
||||
instead of blocking until it completes.
|
||||
|
||||
The job may already be complete when returned; callers must not assume
|
||||
the column is filled until :meth:`AsyncJob.wait` resolves. Invalid
|
||||
input -- an unknown column, or one that is not computed -- raises here
|
||||
rather than failing the job. On local tables the job runs
|
||||
in-process; on LanceDB Cloud and Enterprise it is the server's
|
||||
backfill job.
|
||||
|
||||
Examples
|
||||
--------
|
||||
>>> import asyncio
|
||||
>>> import lancedb
|
||||
>>> async def refresh_in_background():
|
||||
... db = await lancedb.connect_async("./.lancedb")
|
||||
... table = await db.create_table("computed_job_async_demo", [{"x": 1}])
|
||||
... await table.add_columns(computed={"doubled": "x * 2"})
|
||||
... job = await table.refresh_column_async("doubled")
|
||||
... await job.wait()
|
||||
... return await job.status()
|
||||
>>> asyncio.run(refresh_in_background())
|
||||
'finished'
|
||||
"""
|
||||
return AsyncJob(await self._inner.refresh_column_async(column))
|
||||
|
||||
async def alter_columns(
|
||||
self, *alterations: Iterable[dict[str, Any]]
|
||||
) -> AlterColumnsResult:
|
||||
@@ -6072,6 +6300,18 @@ class AsyncTable:
|
||||
async def blob_columns(self) -> list[str]:
|
||||
return await self._inner.blob_columns()
|
||||
|
||||
async def add_bases(
|
||||
self,
|
||||
bases: Union[str, TableBase, Iterable[Union[str, TableBase]]],
|
||||
) -> None:
|
||||
"""Register additional storage bases for this table.
|
||||
|
||||
A URI string is a non-root base with no alias::
|
||||
|
||||
await table.add_bases("s3://bucket/media/")
|
||||
"""
|
||||
await self._inner.add_bases(_normalize_bases(bases))
|
||||
|
||||
async def fetch_blobs(
|
||||
self, column: str, row_ids: Union[list[int], pa.Table]
|
||||
) -> pa.LargeBinaryArray:
|
||||
@@ -6290,6 +6530,30 @@ class AsyncTable:
|
||||
await self._inner.replace_field_metadata(field_name, new_metadata)
|
||||
|
||||
|
||||
def _normalize_bases(
|
||||
base_inputs: Union[str, TableBase, Iterable[Union[str, TableBase]]],
|
||||
) -> list[TableBase]:
|
||||
if isinstance(base_inputs, (str, TableBase)):
|
||||
items: Iterable[Union[str, TableBase]] = [base_inputs]
|
||||
elif isinstance(base_inputs, Mapping):
|
||||
raise TypeError(
|
||||
"Expected a URI string, TableBase, or an iterable of those values"
|
||||
)
|
||||
else:
|
||||
items = base_inputs
|
||||
normalized_bases: list[TableBase] = []
|
||||
for base in items:
|
||||
if isinstance(base, str):
|
||||
normalized_bases.append(TableBase(path=base))
|
||||
elif isinstance(base, TableBase):
|
||||
normalized_bases.append(base)
|
||||
else:
|
||||
raise TypeError(
|
||||
f"Expected a URI string or TableBase, got {type(base).__name__}"
|
||||
)
|
||||
return normalized_bases
|
||||
|
||||
|
||||
@dataclass
|
||||
class IndexStatistics:
|
||||
"""
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||
|
||||
import pyarrow as pa
|
||||
import pytest
|
||||
|
||||
import lancedb
|
||||
|
||||
|
||||
def test_add_bases_accepts_named_and_dataset_root(tmp_path):
|
||||
media = tmp_path / "media"
|
||||
parent = tmp_path / "parent"
|
||||
media.mkdir()
|
||||
parent.mkdir()
|
||||
db = lancedb.connect(tmp_path / "db")
|
||||
schema = pa.schema([pa.field("id", pa.int64())])
|
||||
table = db.create_table("photos", schema=schema)
|
||||
table.add_bases(
|
||||
[
|
||||
lancedb.TableBase(path=media.as_uri(), name="media", is_dataset_root=False),
|
||||
lancedb.TableBase(
|
||||
path=parent.as_uri(), name="parent", is_dataset_root=True
|
||||
),
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
def test_add_bases_accepts_two_unnamed_paths(tmp_path):
|
||||
media = tmp_path / "media"
|
||||
other = tmp_path / "other"
|
||||
media.mkdir()
|
||||
other.mkdir()
|
||||
db = lancedb.connect(tmp_path / "db")
|
||||
schema = pa.schema([pa.field("id", pa.int64())])
|
||||
table = db.create_table("photos", schema=schema)
|
||||
table.add_bases([media.as_uri(), other.as_uri()])
|
||||
|
||||
|
||||
def test_add_bases_rejects_dict_input(tmp_path):
|
||||
db = lancedb.connect(tmp_path / "db")
|
||||
schema = pa.schema([pa.field("id", pa.int64())])
|
||||
table = db.create_table("photos", schema=schema)
|
||||
with pytest.raises(TypeError, match="TableBase"):
|
||||
table.add_bases({"path": "s3://bucket/media/"})
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_add_bases_accepts_file_uri(tmp_path):
|
||||
media = tmp_path / "media"
|
||||
media.mkdir()
|
||||
db = await lancedb.connect_async(tmp_path / "db")
|
||||
schema = pa.schema([pa.field("id", pa.int64())])
|
||||
table = await db.create_table("photos", schema=schema)
|
||||
await table.add_bases(media.as_uri())
|
||||
|
||||
|
||||
def test_memory_add_bases_accepts_file_uri(tmp_path):
|
||||
media = tmp_path / "media"
|
||||
media.mkdir()
|
||||
db = lancedb.connect("memory:///")
|
||||
schema = pa.schema([pa.field("id", pa.int64())])
|
||||
table = db.create_table("photos", schema=schema)
|
||||
table.add_bases(media.as_uri())
|
||||
|
||||
|
||||
def test_namespace_add_bases_accepts_file_uri(tmp_path):
|
||||
media = tmp_path / "media"
|
||||
media.mkdir()
|
||||
db = lancedb.connect_namespace("dir", {"root": str(tmp_path / "ns")})
|
||||
schema = pa.schema([pa.field("id", pa.int64())])
|
||||
table = db.create_table("photos", schema=schema)
|
||||
table.add_bases(media.as_uri())
|
||||
@@ -2306,3 +2306,36 @@ def test_remote_connection_jobs_surface():
|
||||
assert job.status() == "failed"
|
||||
with pytest.raises(JobFailedError, match="worker died"):
|
||||
job.wait(timeout=timedelta(seconds=5))
|
||||
|
||||
|
||||
def test_remote_add_bases_posts_the_bases_array():
|
||||
captured_body = {}
|
||||
|
||||
def handler(request):
|
||||
if request.path == "/v1/table/test/describe/":
|
||||
request.send_response(200)
|
||||
request.send_header("Content-Type", "application/json")
|
||||
request.end_headers()
|
||||
request.wfile.write(json.dumps(BLOB_DESCRIBE_RESPONSE).encode())
|
||||
elif request.path == "/v1/table/test/bases/":
|
||||
content_len = int(request.headers.get("Content-Length", 0))
|
||||
captured_body.update(json.loads(request.rfile.read(content_len)))
|
||||
request.send_response(200)
|
||||
request.send_header("Content-Type", "application/json")
|
||||
request.end_headers()
|
||||
request.wfile.write(b'{"version": 2}')
|
||||
else:
|
||||
request.send_response(404)
|
||||
request.end_headers()
|
||||
|
||||
with mock_lancedb_connection(handler) as db:
|
||||
table = db.open_table("test")
|
||||
table.add_bases(lancedb.TableBase(path="s3://bucket/media/"))
|
||||
|
||||
assert captured_body["bases"] == [
|
||||
{
|
||||
"path": "s3://bucket/media/",
|
||||
"isDatasetRoot": False,
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
@@ -3854,3 +3854,65 @@ async def test_async_search_runs_embedding_on_dedicated_executor(
|
||||
assert all(name.startswith("lancedb-embedding") for name in captured_threads), (
|
||||
f"embedding ran off the dedicated executor: {captured_threads}"
|
||||
)
|
||||
|
||||
|
||||
def test_computed_column_declare_and_refresh(tmp_path):
|
||||
db = lancedb.connect(tmp_path)
|
||||
table = db.create_table("computed", [{"x": 1}, {"x": 2}])
|
||||
|
||||
table.add_columns(computed={"doubled": "x * 2"})
|
||||
assert table.to_arrow()["doubled"].to_pylist() == [None, None]
|
||||
|
||||
result = table.refresh_column("doubled")
|
||||
assert result.rows_filled == 2
|
||||
assert sorted(table.to_arrow()["doubled"].to_pylist()) == [2, 4]
|
||||
|
||||
table.add([{"x": 5}])
|
||||
assert table.refresh_column("doubled").rows_filled == 1
|
||||
assert sorted(table.to_arrow()["doubled"].to_pylist()) == [2, 4, 10]
|
||||
|
||||
|
||||
def test_computed_column_rejects_transforms_and_computed_together(tmp_path):
|
||||
db = lancedb.connect(tmp_path)
|
||||
table = db.create_table("computed_mixed", [{"x": 1}])
|
||||
with pytest.raises(ValueError):
|
||||
table.add_columns({"a": "x + 1"}, computed={"b": "x * 2"})
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_computed_column_async(tmp_path):
|
||||
db = await lancedb.connect_async(tmp_path)
|
||||
table = await db.create_table("computed_async", [{"x": 3}])
|
||||
|
||||
await table.add_columns(computed={"tripled": "x * 3"})
|
||||
await table.refresh_column("tripled")
|
||||
|
||||
assert (await table.to_arrow())["tripled"].to_pylist() == [9]
|
||||
|
||||
|
||||
def test_refresh_column_async_returns_job(tmp_path):
|
||||
db = lancedb.connect(tmp_path)
|
||||
table = db.create_table("computed_job", [{"x": 1}, {"x": 2}])
|
||||
table.add_columns(computed={"doubled": "x * 2"})
|
||||
|
||||
job = table.refresh_column_async("doubled")
|
||||
assert job.id is None # in-process jobs have no server id
|
||||
job.wait()
|
||||
assert job.status() == "finished"
|
||||
assert sorted(table.to_arrow()["doubled"].to_pylist()) == [2, 4]
|
||||
|
||||
# Bad input raises at the call, not through the job.
|
||||
with pytest.raises(Exception, match="not a computed column"):
|
||||
table.refresh_column_async("x")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_refresh_column_async_job_async_table(tmp_path):
|
||||
db = await lancedb.connect_async(tmp_path)
|
||||
table = await db.create_table("computed_job_async", [{"x": 3}])
|
||||
await table.add_columns(computed={"tripled": "x * 3"})
|
||||
|
||||
job = await table.refresh_column_async("tripled")
|
||||
await job.wait()
|
||||
assert await job.status() == "finished"
|
||||
assert (await table.to_arrow())["tripled"].to_pylist() == [9]
|
||||
|
||||
+3
-1
@@ -16,7 +16,8 @@ use query::{FTSQuery, HybridQuery, Query, VectorQuery};
|
||||
use session::Session;
|
||||
use table::{
|
||||
AddColumnsResult, AddResult, AlterColumnsResult, DeleteResult, DropColumnsResult, FtsToken,
|
||||
LsmWriteSpec, MergeResult, PyBlobFile, Table, UpdateFieldMetadataResult, UpdateResult,
|
||||
LsmWriteSpec, MergeResult, PyBlobFile, RefreshColumnResult, Table, UpdateFieldMetadataResult,
|
||||
UpdateResult,
|
||||
};
|
||||
|
||||
pub mod arrow;
|
||||
@@ -57,6 +58,7 @@ pub fn _lancedb(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_class::<VectorQuery>()?;
|
||||
m.add_class::<RecordBatchStream>()?;
|
||||
m.add_class::<AddColumnsResult>()?;
|
||||
m.add_class::<RefreshColumnResult>()?;
|
||||
m.add_class::<AlterColumnsResult>()?;
|
||||
m.add_class::<UpdateFieldMetadataResult>()?;
|
||||
m.add_class::<AddResult>()?;
|
||||
|
||||
@@ -22,6 +22,7 @@ use lancedb::index::scalar::FtsIndexBuilder;
|
||||
use lancedb::table::{
|
||||
AddDataMode, ColumnAlteration, Duration, FieldMetadataUpdate, FtsToken as LanceDbFtsToken,
|
||||
NewColumnTransform, OptimizeAction, OptimizeOptions, Ref, Table as LanceDbTable,
|
||||
TableBase as LanceTableBase,
|
||||
};
|
||||
use lancedb::tokenize as lancedb_tokenize;
|
||||
use pyo3::{
|
||||
@@ -94,6 +95,13 @@ fn lsm_stats_to_py(py: Python<'_>, stats: &lancedb::table::LsmStats) -> PyResult
|
||||
Ok(out.unbind())
|
||||
}
|
||||
|
||||
#[derive(FromPyObject)]
|
||||
pub(crate) struct PyTableBase {
|
||||
path: String,
|
||||
name: Option<String>,
|
||||
is_dataset_root: bool,
|
||||
}
|
||||
|
||||
#[derive(FromPyObject)]
|
||||
enum PredicateArg {
|
||||
Expr(PyExpr),
|
||||
@@ -415,6 +423,32 @@ pub struct AddColumnsResult {
|
||||
pub version: u64,
|
||||
}
|
||||
|
||||
#[pyclass(get_all, from_py_object)]
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct RefreshColumnResult {
|
||||
pub rows_filled: u64,
|
||||
pub version: u64,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl RefreshColumnResult {
|
||||
pub fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"RefreshColumnResult(rows_filled={}, version={})",
|
||||
self.rows_filled, self.version
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
impl From<lancedb::table::RefreshColumnResult> for RefreshColumnResult {
|
||||
fn from(result: lancedb::table::RefreshColumnResult) -> Self {
|
||||
Self {
|
||||
rows_filled: result.rows_filled,
|
||||
version: result.version,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl AddColumnsResult {
|
||||
pub fn __repr__(&self) -> String {
|
||||
@@ -1212,6 +1246,25 @@ impl Table {
|
||||
})
|
||||
}
|
||||
|
||||
#[pyo3(signature = (bases))]
|
||||
pub fn add_bases(
|
||||
self_: PyRef<'_, Self>,
|
||||
bases: Vec<PyTableBase>,
|
||||
) -> PyResult<Bound<'_, PyAny>> {
|
||||
let inner = self_.inner_ref()?.clone();
|
||||
let bases: Vec<LanceTableBase> = bases
|
||||
.into_iter()
|
||||
.map(|base| LanceTableBase {
|
||||
path: base.path,
|
||||
name: base.name,
|
||||
is_dataset_root: base.is_dataset_root,
|
||||
})
|
||||
.collect();
|
||||
future_into_py(self_.py(), async move {
|
||||
inner.add_bases(bases).await.infer_error()
|
||||
})
|
||||
}
|
||||
|
||||
/// Read blob bytes for `row_ids` from blob v2 column `column`.
|
||||
#[pyo3(signature = (column, row_ids))]
|
||||
pub fn fetch_blobs(
|
||||
@@ -1510,6 +1563,40 @@ impl Table {
|
||||
})
|
||||
}
|
||||
|
||||
pub fn add_computed_columns(
|
||||
self_: PyRef<'_, Self>,
|
||||
columns: Vec<(String, String)>,
|
||||
) -> PyResult<Bound<'_, PyAny>> {
|
||||
let inner = self_.inner_ref()?.clone();
|
||||
future_into_py(self_.py(), async move {
|
||||
let mut builder = inner.add_columns();
|
||||
for (name, expression) in columns {
|
||||
builder = builder.computed(name, expression);
|
||||
}
|
||||
let result = builder.execute().await.infer_error()?;
|
||||
Ok(AddColumnsResult::from(result))
|
||||
})
|
||||
}
|
||||
|
||||
pub fn refresh_column(self_: PyRef<'_, Self>, column: String) -> PyResult<Bound<'_, PyAny>> {
|
||||
let inner = self_.inner_ref()?.clone();
|
||||
future_into_py(self_.py(), async move {
|
||||
let result = inner.refresh_column(column).await.infer_error()?;
|
||||
Ok(RefreshColumnResult::from(result))
|
||||
})
|
||||
}
|
||||
|
||||
pub fn refresh_column_async(
|
||||
self_: PyRef<'_, Self>,
|
||||
column: String,
|
||||
) -> PyResult<Bound<'_, PyAny>> {
|
||||
let inner = self_.inner_ref()?.clone();
|
||||
future_into_py(self_.py(), async move {
|
||||
let job = inner.refresh_column_async(column).await.infer_error()?;
|
||||
Ok(crate::job::Job::new(job))
|
||||
})
|
||||
}
|
||||
|
||||
pub fn add_columns_with_schema(
|
||||
self_: PyRef<'_, Self>,
|
||||
schema: PyArrowType<Schema>,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "lancedb"
|
||||
version = "0.37.1-beta.1"
|
||||
version = "0.38.0-beta.0"
|
||||
edition.workspace = true
|
||||
description = "LanceDB: A serverless, low-latency vector database for AI applications"
|
||||
license.workspace = true
|
||||
|
||||
@@ -1032,6 +1032,7 @@ impl Database for ListingDatabase {
|
||||
};
|
||||
|
||||
Ok(ListTablesResponse {
|
||||
context: None,
|
||||
tables: f,
|
||||
page_token: next_page_token,
|
||||
})
|
||||
|
||||
@@ -71,6 +71,14 @@ pub enum Error {
|
||||
IndexNotFound { name: String },
|
||||
#[snafu(display("Embedding function '{name}' was not found. : {reason}"))]
|
||||
EmbeddingFunctionNotFound { name: String, reason: String },
|
||||
#[snafu(display("Column '{name}' was not found"))]
|
||||
ColumnNotFound { name: String },
|
||||
#[snafu(display("Column '{name}' already exists"))]
|
||||
ColumnAlreadyExists { name: String },
|
||||
#[snafu(display("Column '{name}' is not a computed column"))]
|
||||
NotAComputedColumn { name: String },
|
||||
#[snafu(display("Invalid expression for column '{column}': {message}"))]
|
||||
InvalidExpression { column: String, message: String },
|
||||
|
||||
#[snafu(display("Table '{name}' already exists"))]
|
||||
TableAlreadyExists { name: String },
|
||||
|
||||
@@ -141,7 +141,7 @@ impl SpawnedJob {
|
||||
Ok(Err(err)) => Outcome::Failed(Arc::new(err)),
|
||||
Err(err) if err.is_cancelled() => Outcome::Cancelled,
|
||||
Err(err) => Outcome::Failed(Arc::new(Error::Runtime {
|
||||
message: format!("index job task failed: {err}"),
|
||||
message: format!("job task failed: {err}"),
|
||||
})),
|
||||
};
|
||||
let _ = tx.send(Some(outcome));
|
||||
|
||||
@@ -214,7 +214,7 @@ use lance_linalg::distance::DistanceType as LanceDistanceType;
|
||||
/// a built-in pull-based adapter.
|
||||
#[cfg(feature = "metrics")]
|
||||
pub use metrics;
|
||||
pub use table::{FtsToken, Table};
|
||||
pub use table::{FtsToken, Table, TableBase};
|
||||
|
||||
/// Tokenize a full-text search query using an explicit FTS tokenizer configuration.
|
||||
///
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||
|
||||
pub mod bases;
|
||||
pub mod blobs;
|
||||
pub mod insert;
|
||||
|
||||
@@ -33,7 +34,9 @@ use crate::table::lsm_stats::GetLsmStatsResponse;
|
||||
use crate::table::merge::MergeFilter;
|
||||
use crate::table::query::create_multi_vector_plan;
|
||||
use crate::table::write_progress::FinishOnDrop;
|
||||
use crate::table::{AlterColumnsResult, FieldMetadataUpdate, UpdateFieldMetadataResult};
|
||||
use crate::table::{
|
||||
AlterColumnsResult, FieldMetadataUpdate, RefreshColumnResult, UpdateFieldMetadataResult,
|
||||
};
|
||||
use crate::table::{AnyQuery, Filter, Predicate, PreprocessingOutput, TableStatistics};
|
||||
use crate::utils::background_cache::BackgroundCache;
|
||||
use crate::utils::{
|
||||
@@ -140,6 +143,40 @@ impl FreshnessHeaders {
|
||||
}
|
||||
}
|
||||
|
||||
/// A backfill job whose successful wait establishes a read-freshness
|
||||
/// baseline on the submitting handle, so a later read cannot be served
|
||||
/// from a cache older than the completed fill. A handle pinned by checkout
|
||||
/// at completion keeps its time-travel view instead.
|
||||
struct FreshnessJob<S: HttpSend> {
|
||||
inner: RemoteJob<S>,
|
||||
freshness: Arc<Mutex<FreshnessState>>,
|
||||
version: Arc<RwLock<Option<u64>>>,
|
||||
}
|
||||
|
||||
#[async_trait]
|
||||
impl<S: HttpSend> crate::job::JobHandle for FreshnessJob<S> {
|
||||
fn id(&self) -> Option<&str> {
|
||||
crate::job::JobHandle::id(&self.inner)
|
||||
}
|
||||
|
||||
async fn status(&self) -> Result<String> {
|
||||
crate::job::JobHandle::status(&self.inner).await
|
||||
}
|
||||
|
||||
async fn wait(&self) -> Result<()> {
|
||||
crate::job::JobHandle::wait(&self.inner).await?;
|
||||
let version = self.version.read().await;
|
||||
if version.is_none() {
|
||||
self.freshness.lock().unwrap().checkout_baseline = Some(SystemTime::now());
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn cancel(&self) -> Result<()> {
|
||||
crate::job::JobHandle::cancel(&self.inner).await
|
||||
}
|
||||
}
|
||||
|
||||
fn compute_min_timestamp(
|
||||
state: &FreshnessState,
|
||||
interval: Option<Duration>,
|
||||
@@ -274,10 +311,10 @@ pub struct RemoteTable<S: HttpSend = Sender> {
|
||||
identifier: String,
|
||||
server_version: ServerVersion,
|
||||
|
||||
version: RwLock<Option<u64>>,
|
||||
version: Arc<RwLock<Option<u64>>>,
|
||||
location: RwLock<Option<String>>,
|
||||
schema_cache: BackgroundCache<SchemaRef, Error>,
|
||||
freshness: Mutex<FreshnessState>,
|
||||
freshness: Arc<Mutex<FreshnessState>>,
|
||||
/// The branch this handle is scoped to, or `None` for the main branch.
|
||||
/// Stamped onto every branch-accepting request so reads and writes resolve
|
||||
/// on the branch's own version chain rather than main's.
|
||||
@@ -415,10 +452,10 @@ impl<S: HttpSend> RemoteTable<S> {
|
||||
namespace,
|
||||
identifier,
|
||||
server_version,
|
||||
version: RwLock::new(None),
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
branch: None,
|
||||
}
|
||||
}
|
||||
@@ -447,10 +484,10 @@ impl<S: HttpSend> RemoteTable<S> {
|
||||
namespace: self.namespace.clone(),
|
||||
identifier: self.identifier.clone(),
|
||||
server_version: self.server_version.clone(),
|
||||
version: RwLock::new(None),
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
branch,
|
||||
}
|
||||
}
|
||||
@@ -1268,10 +1305,10 @@ mod test_utils {
|
||||
namespace: vec![],
|
||||
identifier: name,
|
||||
server_version: version.map(ServerVersion).unwrap_or_default(),
|
||||
version: RwLock::new(None),
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
branch: None,
|
||||
}
|
||||
}
|
||||
@@ -1292,10 +1329,10 @@ mod test_utils {
|
||||
namespace: vec![],
|
||||
identifier: name,
|
||||
server_version: ServerVersion::default(),
|
||||
version: RwLock::new(None),
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
branch: None,
|
||||
}
|
||||
}
|
||||
@@ -1325,10 +1362,10 @@ mod test_utils {
|
||||
namespace: vec![],
|
||||
identifier: name,
|
||||
server_version: version.map(ServerVersion).unwrap_or_default(),
|
||||
version: RwLock::new(None),
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
branch: None,
|
||||
}
|
||||
}
|
||||
@@ -2208,6 +2245,10 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
|
||||
self.blob_columns_impl().await
|
||||
}
|
||||
|
||||
async fn add_bases(&self, bases: &[crate::table::TableBase]) -> Result<()> {
|
||||
self.add_bases_impl(bases).await
|
||||
}
|
||||
|
||||
async fn fetch_blobs(&self, column: &str, row_ids: &[u64]) -> Result<LargeBinaryArray> {
|
||||
self.fetch_blobs_impl(column, row_ids).await
|
||||
}
|
||||
@@ -2708,6 +2749,86 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
|
||||
}
|
||||
}
|
||||
|
||||
async fn add_computed_columns(&self, columns: &[(String, String)]) -> Result<AddColumnsResult> {
|
||||
self.check_mutable().await?;
|
||||
// The server plans the declaration: expression validation, type
|
||||
// inference and the persisted binding all happen there.
|
||||
let entries = columns
|
||||
.iter()
|
||||
.map(
|
||||
|(name, expression)| lance_namespace::models::AddColumnsEntry {
|
||||
name: name.clone(),
|
||||
computed: Some(Some(expression.clone())),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.collect::<Vec<_>>();
|
||||
let mut body = serde_json::json!({ "new_columns": entries });
|
||||
self.apply_branch_body(&mut body);
|
||||
let request = self
|
||||
.client
|
||||
.post(&format!("/v1/table/{}/add_columns/", self.identifier))
|
||||
.json(&body);
|
||||
let (request_id, response) = self.send(request, true).await?;
|
||||
let response = self.check_table_response(&request_id, response).await?;
|
||||
let body = response.text().await.err_to_http(request_id.clone())?;
|
||||
|
||||
if body.trim().is_empty() {
|
||||
// Backward compatible with old servers
|
||||
return Ok(AddColumnsResult { version: 0 });
|
||||
}
|
||||
|
||||
let result: AddColumnsResult = serde_json::from_str(&body).map_err(|e| Error::Http {
|
||||
source: format!("Failed to parse add_columns response: {}", e).into(),
|
||||
request_id,
|
||||
status_code: None,
|
||||
})?;
|
||||
|
||||
self.invalidate_schema_cache();
|
||||
self.track_write_version(result.version);
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
async fn refresh_column(&self, _column: &str) -> Result<RefreshColumnResult> {
|
||||
// The server runs a refresh as a job and does not report a fill
|
||||
// count, so the blocking form has no honest result to return.
|
||||
Err(Error::NotSupported {
|
||||
message: "a remote refresh runs as a server job; use refresh_column_async and \
|
||||
wait on the returned handle"
|
||||
.into(),
|
||||
})
|
||||
}
|
||||
|
||||
async fn refresh_column_async(&self, column: &str) -> Result<Job> {
|
||||
self.check_mutable().await?;
|
||||
let mut body = serde_json::json!({ "column": column });
|
||||
self.apply_branch_body(&mut body);
|
||||
let request = self
|
||||
.client
|
||||
.post(&format!("/v1/table/{}/backfill_column", self.identifier))
|
||||
.json(&body);
|
||||
let (request_id, response) = self.send(request, true).await?;
|
||||
let response = self.check_table_response(&request_id, response).await?;
|
||||
let body = response.text().await.err_to_http(request_id.clone())?;
|
||||
|
||||
#[derive(serde::Deserialize)]
|
||||
struct BackfillResponse {
|
||||
job_id: String,
|
||||
}
|
||||
let response: BackfillResponse = serde_json::from_str(&body).map_err(|e| Error::Http {
|
||||
source: format!("Failed to parse backfill_column response: {}", e).into(),
|
||||
request_id,
|
||||
status_code: None,
|
||||
})?;
|
||||
|
||||
Ok(Job::new(Box::new(FreshnessJob {
|
||||
inner: RemoteJob::new(self.client.clone(), response.job_id),
|
||||
freshness: self.freshness.clone(),
|
||||
version: self.version.clone(),
|
||||
})))
|
||||
}
|
||||
|
||||
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult> {
|
||||
self.check_mutable().await?;
|
||||
let body = alterations
|
||||
@@ -3973,6 +4094,42 @@ mod tests {
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_add_bases_posts_the_bases_array() {
|
||||
let table = Table::new_with_handler("my_table", |request| {
|
||||
assert_eq!(request.method(), "POST");
|
||||
assert_eq!(request.url().path(), "/v1/table/my_table/bases/");
|
||||
let body: serde_json::Value =
|
||||
serde_json::from_slice(request.body().unwrap().as_bytes().unwrap()).unwrap();
|
||||
assert_eq!(
|
||||
body["bases"],
|
||||
serde_json::json!([{
|
||||
"path": "s3://bucket/media/",
|
||||
"isDatasetRoot": false
|
||||
}])
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"version": 4}"#)
|
||||
.unwrap()
|
||||
});
|
||||
|
||||
table.add_bases(["s3://bucket/media/"]).await.unwrap();
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_add_bases_rejects_empty_response() {
|
||||
let table = Table::new_with_handler("my_table", |_request| {
|
||||
http::Response::builder().status(200).body("").unwrap()
|
||||
});
|
||||
let err = table.add_bases(["s3://bucket/media/"]).await.unwrap_err();
|
||||
assert!(
|
||||
err.to_string()
|
||||
.contains("invalid response while registering table bases"),
|
||||
"{err}"
|
||||
);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case(semver::Version::new(0, 1, 0))]
|
||||
#[case(semver::Version::new(0, 5, 0))]
|
||||
@@ -6449,6 +6606,346 @@ mod tests {
|
||||
assert_eq!(result.version, if old_server { 0 } else { 43 });
|
||||
}
|
||||
|
||||
/// A declaration is sent as `{name, computed}` entries for the server to
|
||||
/// plan; the client never types the expression itself.
|
||||
#[tokio::test]
|
||||
async fn test_add_computed_columns_sends_the_expression() {
|
||||
let table = Table::new_with_handler("my_table", |request| {
|
||||
assert_eq!(request.method(), "POST");
|
||||
assert_eq!(request.url().path(), "/v1/table/my_table/add_columns/");
|
||||
let body = request.body().unwrap().as_bytes().unwrap();
|
||||
let value: serde_json::Value = serde_json::from_slice(body).unwrap();
|
||||
assert_eq!(
|
||||
value["new_columns"],
|
||||
serde_json::json!([{"name": "doubled", "computed": "x * 2"}])
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"version": 7}"#)
|
||||
.unwrap()
|
||||
});
|
||||
|
||||
let result = table
|
||||
.add_columns()
|
||||
.computed("doubled", "x * 2")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(result.version, 7);
|
||||
}
|
||||
|
||||
/// A remote refresh is a server job: the async form returns its handle,
|
||||
/// and the blocking form refuses rather than invent a fill count.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_column_async_submits_a_backfill_job() {
|
||||
let table = Table::new_with_handler("my_table", |request| {
|
||||
assert_eq!(request.method(), "POST");
|
||||
assert_eq!(request.url().path(), "/v1/table/my_table/backfill_column");
|
||||
let body = request.body().unwrap().as_bytes().unwrap();
|
||||
let value: serde_json::Value = serde_json::from_slice(body).unwrap();
|
||||
assert_eq!(value["column"], "doubled");
|
||||
http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-42"}"#)
|
||||
.unwrap()
|
||||
});
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
assert_eq!(job.id(), Some("j-42"));
|
||||
|
||||
let err = table.refresh_column("doubled").await.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::NotSupported { message }
|
||||
if message.contains("refresh_column_async")),
|
||||
"{err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
/// The gate's reproducer: after a successful wait, a same-handle read
|
||||
/// must carry a freshness baseline so a stale server cache cannot serve
|
||||
/// the pre-backfill snapshot.
|
||||
#[tokio::test]
|
||||
async fn test_backfill_wait_establishes_read_freshness() {
|
||||
let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
let saw = saw_min_timestamp.clone();
|
||||
let table =
|
||||
Table::new_with_handler("my_table", move |request| match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-7"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-7", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
saw.store(
|
||||
request.headers().contains_key("x-lancedb-min-timestamp"),
|
||||
std::sync::atomic::Ordering::SeqCst,
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
});
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
assert!(
|
||||
saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst),
|
||||
"read after wait carried no freshness baseline"
|
||||
);
|
||||
}
|
||||
|
||||
/// A checkout after submission wins over the completion fence: the
|
||||
/// pinned view must not regain a timestamp floor from the job.
|
||||
#[tokio::test]
|
||||
async fn test_checkout_after_submit_beats_the_completion_fence() {
|
||||
let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
let saw = saw_min_timestamp.clone();
|
||||
let table =
|
||||
Table::new_with_handler("my_table", move |request| match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-8"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-8", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/describe/" => {
|
||||
let schema = Schema::new(vec![Field::new("x", DataType::Int32, true)]);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(describe_response(&schema))
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
saw.store(
|
||||
request.headers().contains_key("x-lancedb-min-timestamp"),
|
||||
std::sync::atomic::Ordering::SeqCst,
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
});
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
table.checkout(3).await.unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
assert!(
|
||||
!saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst),
|
||||
"completion fence overrode an explicit checkout"
|
||||
);
|
||||
}
|
||||
|
||||
/// Tag checkout resets freshness state wholesale; the fence must not
|
||||
/// survive it.
|
||||
#[tokio::test]
|
||||
async fn test_tag_checkout_after_submit_beats_the_completion_fence() {
|
||||
let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
let saw = saw_min_timestamp.clone();
|
||||
let table =
|
||||
Table::new_with_handler("my_table", move |request| match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-9"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-9", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/tags/version/" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"version": 5}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/describe/" => {
|
||||
let schema = Schema::new(vec![Field::new("x", DataType::Int32, true)]);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(describe_response(&schema))
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
saw.store(
|
||||
request.headers().contains_key("x-lancedb-min-timestamp"),
|
||||
std::sync::atomic::Ordering::SeqCst,
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
});
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
table.checkout_tag("v1").await.unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
assert!(
|
||||
!saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst),
|
||||
"completion fence overrode a tag checkout"
|
||||
);
|
||||
}
|
||||
|
||||
/// A checkout landing while the submission request is in flight advances
|
||||
/// the epoch past the token captured at submit.
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn test_checkout_during_submission_beats_the_completion_fence() {
|
||||
let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
let saw = saw_min_timestamp.clone();
|
||||
let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
|
||||
let release_rx = Arc::new(std::sync::Mutex::new(release_rx));
|
||||
let (arrived_tx, arrived_rx) = std::sync::mpsc::channel::<()>();
|
||||
let arrived_tx = Arc::new(std::sync::Mutex::new(arrived_tx));
|
||||
let table = Table::new_with_handler("my_table", move |request| {
|
||||
match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => {
|
||||
// Signal arrival, then hold the response until the
|
||||
// test's checkout completes.
|
||||
arrived_tx.lock().unwrap().send(()).unwrap();
|
||||
release_rx
|
||||
.lock()
|
||||
.unwrap()
|
||||
.recv_timeout(std::time::Duration::from_secs(10))
|
||||
.unwrap();
|
||||
http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-10"}"#.to_string())
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-10", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/describe/" => {
|
||||
let schema = Schema::new(vec![Field::new("x", DataType::Int32, true)]);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(describe_response(&schema))
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
saw.store(
|
||||
request.headers().contains_key("x-lancedb-min-timestamp"),
|
||||
std::sync::atomic::Ordering::SeqCst,
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
}
|
||||
});
|
||||
|
||||
let submit = tokio::spawn({
|
||||
let table = table.clone();
|
||||
async move { table.refresh_column_async("doubled").await }
|
||||
});
|
||||
tokio::task::spawn_blocking(move || {
|
||||
arrived_rx
|
||||
.recv_timeout(std::time::Duration::from_secs(10))
|
||||
.unwrap()
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
table.checkout(7).await.unwrap();
|
||||
release_tx.send(()).unwrap();
|
||||
|
||||
let job = submit.await.unwrap().unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
assert!(
|
||||
!saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst),
|
||||
"completion fence overrode a checkout that landed mid-submission"
|
||||
);
|
||||
}
|
||||
|
||||
/// checkout_latest keeps the handle on latest, so a completed backfill
|
||||
/// must still establish its post-fill baseline -- strictly later than the
|
||||
/// checkout's own, or a pre-fill cache could still serve.
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn test_checkout_latest_during_submission_keeps_the_fence() {
|
||||
let seen_min_timestamp = Arc::new(std::sync::Mutex::new(None::<String>));
|
||||
let saw = seen_min_timestamp.clone();
|
||||
let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
|
||||
let release_rx = Arc::new(std::sync::Mutex::new(release_rx));
|
||||
let (arrived_tx, arrived_rx) = std::sync::mpsc::channel::<()>();
|
||||
let arrived_tx = Arc::new(std::sync::Mutex::new(arrived_tx));
|
||||
let table =
|
||||
Table::new_with_handler("my_table", move |request| match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => {
|
||||
arrived_tx.lock().unwrap().send(()).unwrap();
|
||||
release_rx
|
||||
.lock()
|
||||
.unwrap()
|
||||
.recv_timeout(std::time::Duration::from_secs(10))
|
||||
.unwrap();
|
||||
http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-11"}"#.to_string())
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-11", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
*saw.lock().unwrap() = request
|
||||
.headers()
|
||||
.get("x-lancedb-min-timestamp")
|
||||
.map(|v| v.to_str().unwrap().to_string());
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
});
|
||||
|
||||
let submit = tokio::spawn({
|
||||
let table = table.clone();
|
||||
async move { table.refresh_column_async("doubled").await }
|
||||
});
|
||||
tokio::task::spawn_blocking(move || {
|
||||
arrived_rx
|
||||
.recv_timeout(std::time::Duration::from_secs(10))
|
||||
.unwrap()
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
table.checkout_latest().await.unwrap();
|
||||
let after_checkout = SystemTime::now();
|
||||
// Real separation between the checkout baseline and completion.
|
||||
tokio::time::sleep(std::time::Duration::from_millis(50)).await;
|
||||
release_tx.send(()).unwrap();
|
||||
|
||||
let job = submit.await.unwrap().unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
let header = seen_min_timestamp
|
||||
.lock()
|
||||
.unwrap()
|
||||
.clone()
|
||||
.expect("no baseline");
|
||||
let sent: SystemTime = chrono::DateTime::parse_from_rfc3339(&header)
|
||||
.unwrap()
|
||||
.into();
|
||||
assert!(
|
||||
sent > after_checkout,
|
||||
"baseline {header} did not advance past the checkout"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_prewarm_index() {
|
||||
let table = Table::new_with_handler("my_table", |request| {
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||
|
||||
//! Cloud HTTP for registering extra table storage bases.
|
||||
|
||||
use serde::Deserialize;
|
||||
|
||||
use crate::Error;
|
||||
use crate::error::Result;
|
||||
use crate::remote::client::{HttpSend, RequestResultExt};
|
||||
|
||||
use super::RemoteTable;
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
struct AddBasesResponse {
|
||||
version: u64,
|
||||
}
|
||||
|
||||
impl<S: HttpSend> RemoteTable<S> {
|
||||
pub(super) async fn add_bases_impl(&self, bases: &[crate::table::TableBase]) -> Result<()> {
|
||||
self.check_mutable().await?;
|
||||
let mut body = serde_json::json!({ "bases": bases });
|
||||
self.apply_branch_body(&mut body);
|
||||
let request = self
|
||||
.client
|
||||
.post(&format!("/v1/table/{}/bases/", self.identifier))
|
||||
.json(&body);
|
||||
let (request_id, response) = self.send(request, true).await?;
|
||||
let response = self.check_table_response(&request_id, response).await?;
|
||||
let body = response.text().await.err_to_http(request_id.clone())?;
|
||||
let parsed: AddBasesResponse = serde_json::from_str(&body).map_err(|e| Error::Http {
|
||||
source: format!(
|
||||
"The server returned an invalid response while registering table bases: {e}"
|
||||
)
|
||||
.into(),
|
||||
request_id,
|
||||
status_code: None,
|
||||
})?;
|
||||
self.track_write_version(parsed.version);
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
+230
-1
@@ -34,7 +34,7 @@ use lance_index::scalar::inverted::query::collect_query_tokens;
|
||||
use lance_namespace::LanceNamespace;
|
||||
use lance_namespace::error::NamespaceError;
|
||||
use lance_namespace::models::DescribeTableRequest;
|
||||
use lance_table::format::Manifest;
|
||||
use lance_table::format::{BasePath, Manifest};
|
||||
use lance_table::io::commit::CommitHandler;
|
||||
use lance_table::io::commit::ManifestNamingScheme;
|
||||
use lance_table::io::commit::external_manifest::ExternalManifestCommitHandler;
|
||||
@@ -68,6 +68,7 @@ pub mod add_columns;
|
||||
mod add_data;
|
||||
pub mod branch_merge;
|
||||
pub mod checkpoint;
|
||||
pub mod computed_columns;
|
||||
mod create_index;
|
||||
pub mod datafusion;
|
||||
pub(crate) mod dataset;
|
||||
@@ -77,6 +78,7 @@ pub mod merge;
|
||||
pub mod optimize;
|
||||
mod primary_key;
|
||||
pub mod query;
|
||||
pub mod refresh;
|
||||
pub mod schema_evolution;
|
||||
pub mod update;
|
||||
pub mod write_progress;
|
||||
@@ -90,6 +92,9 @@ pub use branch_merge::{
|
||||
MergeBranchResult, MergeBranchStatus, MergePreview, RowCountSummary,
|
||||
};
|
||||
pub use chrono::Duration;
|
||||
pub use computed_columns::{
|
||||
ComputedColumn, ComputedColumnKind, computed_column_from_field, computed_columns,
|
||||
};
|
||||
pub use delete::DeleteResult;
|
||||
use futures::future::join_all;
|
||||
pub use lance::dataset::refs::{BranchContents, Ref, TagContents, Tags as LanceTags};
|
||||
@@ -97,6 +102,7 @@ pub use lance::dataset::scanner::DatasetRecordBatchStream;
|
||||
pub use lance_index::optimize::OptimizeOptions;
|
||||
pub use lsm_stats::{BucketStats, GenerationStats, LsmStats, MemtableStats};
|
||||
pub use optimize::{CompactionOptions, OptimizeAction, OptimizeStats};
|
||||
pub use refresh::RefreshColumnResult;
|
||||
pub use schema_evolution::{
|
||||
AddColumnsResult, AlterColumnsResult, DropColumnsResult, FieldMetadataUpdate,
|
||||
UpdateFieldMetadataResult,
|
||||
@@ -637,6 +643,15 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
|
||||
message: "set_lsm_write_spec is not supported on this table type".into(),
|
||||
})
|
||||
}
|
||||
/// Switch this table to required index catch-up, one way.
|
||||
///
|
||||
/// The default implementation returns `NotSupported`. Implementations
|
||||
/// that support the MemWAL LSM write path must override this.
|
||||
async fn require_mem_wal_index_catchup(&self) -> Result<()> {
|
||||
Err(Error::NotSupported {
|
||||
message: "require_mem_wal_index_catchup is not supported on this table type".into(),
|
||||
})
|
||||
}
|
||||
/// Remove the [`LsmWriteSpec`] from this table.
|
||||
///
|
||||
/// This is a no-op if no spec is currently set.
|
||||
@@ -696,6 +711,12 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
|
||||
message: "blob_columns is not supported on this table type".into(),
|
||||
})
|
||||
}
|
||||
/// Register additional storage bases for this table.
|
||||
async fn add_bases(&self, _bases: &[TableBase]) -> Result<()> {
|
||||
Err(Error::NotSupported {
|
||||
message: "Registering table bases is not supported for this table type.".into(),
|
||||
})
|
||||
}
|
||||
/// Materialize blob bytes for the given row ids. See [`Table::fetch_blobs`].
|
||||
async fn fetch_blobs(&self, _column: &str, _row_ids: &[u64]) -> Result<LargeBinaryArray> {
|
||||
Err(Error::NotSupported {
|
||||
@@ -732,6 +753,34 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
|
||||
transforms: NewColumnTransform,
|
||||
read_columns: Option<Vec<String>>,
|
||||
) -> Result<AddColumnsResult>;
|
||||
/// Declare computed columns, each defined by a SQL expression.
|
||||
///
|
||||
/// Where the declaration is planned depends on the backend: a local table
|
||||
/// validates and types the expression itself, a remote one sends the text
|
||||
/// for the server to plan.
|
||||
async fn add_computed_columns(
|
||||
&self,
|
||||
_columns: &[(String, String)],
|
||||
) -> Result<AddColumnsResult> {
|
||||
Err(Error::NotSupported {
|
||||
message: "computed columns are not supported on this table type".into(),
|
||||
})
|
||||
}
|
||||
/// Fill a computed column's unfilled rows.
|
||||
///
|
||||
/// The default returns `NotSupported`; Lance-backed tables override it.
|
||||
async fn refresh_column(&self, _column: &str) -> Result<RefreshColumnResult> {
|
||||
Err(Error::NotSupported {
|
||||
message: "computed columns are supported only on local tables".into(),
|
||||
})
|
||||
}
|
||||
/// Fill a computed column's unfilled rows, returning a [`Job`] tracking
|
||||
/// the operation.
|
||||
async fn refresh_column_async(&self, _column: &str) -> Result<Job> {
|
||||
Err(Error::NotSupported {
|
||||
message: "computed columns are supported only on local tables".into(),
|
||||
})
|
||||
}
|
||||
/// Alter columns in the table.
|
||||
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult>;
|
||||
/// Drop columns from the table.
|
||||
@@ -846,6 +895,54 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
|
||||
}
|
||||
}
|
||||
|
||||
/// An extra storage prefix registered on a table.
|
||||
///
|
||||
/// `path` is an object-store URI. `name` is an optional alias. `is_dataset_root`
|
||||
/// is true when `path` points to a Lance dataset root. When false, `path`
|
||||
/// points directly to the directory containing the referenced files.
|
||||
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub struct TableBase {
|
||||
/// Object store URI such as `s3://bucket/media/`.
|
||||
pub path: String,
|
||||
/// Optional alias.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub name: Option<String>,
|
||||
/// True when `path` is a Lance dataset root. When false, `path` is the
|
||||
/// directory containing the referenced files.
|
||||
#[serde(default)]
|
||||
pub is_dataset_root: bool,
|
||||
}
|
||||
|
||||
impl TableBase {
|
||||
/// A non-root base with no alias.
|
||||
pub fn new(path: impl Into<String>) -> Self {
|
||||
Self {
|
||||
path: path.into(),
|
||||
name: None,
|
||||
is_dataset_root: false,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<&str> for TableBase {
|
||||
fn from(path: &str) -> Self {
|
||||
Self::new(path)
|
||||
}
|
||||
}
|
||||
|
||||
impl From<&String> for TableBase {
|
||||
fn from(path: &String) -> Self {
|
||||
Self::new(path.as_str())
|
||||
}
|
||||
}
|
||||
|
||||
impl From<String> for TableBase {
|
||||
fn from(path: String) -> Self {
|
||||
Self::new(path)
|
||||
}
|
||||
}
|
||||
|
||||
/// A Table is a collection of strong typed Rows.
|
||||
///
|
||||
/// The type of the each row is defined in Apache Arrow [Schema].
|
||||
@@ -1083,6 +1180,25 @@ impl Table {
|
||||
self.inner.blob_columns().await
|
||||
}
|
||||
|
||||
/// Register additional storage bases for this table.
|
||||
///
|
||||
/// A URI string is a non-root base with no alias.
|
||||
///
|
||||
/// ```
|
||||
/// # use lancedb::Table;
|
||||
/// # async fn register(table: &Table) -> Result<(), Box<dyn std::error::Error>> {
|
||||
/// table.add_bases(["s3://bucket/media/"]).await?;
|
||||
/// # Ok(())
|
||||
/// # }
|
||||
/// ```
|
||||
pub async fn add_bases(
|
||||
&self,
|
||||
bases: impl IntoIterator<Item = impl Into<TableBase>>,
|
||||
) -> Result<()> {
|
||||
let bases: Vec<TableBase> = bases.into_iter().map(Into::into).collect();
|
||||
self.inner.add_bases(&bases).await
|
||||
}
|
||||
|
||||
/// Materialize blob bytes for the given row ids.
|
||||
///
|
||||
/// Output matches `row_ids` in length and order. Null blobs are null;
|
||||
@@ -1624,6 +1740,53 @@ impl Table {
|
||||
AddColumnsBuilder::new(self.inner.clone())
|
||||
}
|
||||
|
||||
/// Fill the fragments of a computed column that hold no values yet.
|
||||
///
|
||||
/// Declared with
|
||||
/// [`AddColumnsBuilder::computed`](add_columns::AddColumnsBuilder::computed),
|
||||
/// a column starts empty and gets its values here. Fragments appended
|
||||
/// since the last refresh are filled by the next one; fragments already
|
||||
/// filled are left as they are, so the call is idempotent and does not
|
||||
/// observe a mutated input.
|
||||
///
|
||||
/// Local tables only: a remote refresh runs as a server job, through
|
||||
/// [`Table::refresh_column_async`].
|
||||
///
|
||||
/// ```
|
||||
/// # use lancedb::Table;
|
||||
/// # async fn refresh(table: &Table) -> Result<(), Box<dyn std::error::Error>> {
|
||||
/// let result = table.refresh_column("doubled").await?;
|
||||
/// println!("filled {} rows at version {}", result.rows_filled, result.version);
|
||||
/// # Ok(())
|
||||
/// # }
|
||||
/// ```
|
||||
pub async fn refresh_column(&self, column: impl AsRef<str>) -> Result<RefreshColumnResult> {
|
||||
self.inner.refresh_column(column.as_ref()).await
|
||||
}
|
||||
|
||||
/// Like [`Table::refresh_column`], but returns a [`Job`] tracking the
|
||||
/// operation instead of blocking until it completes.
|
||||
///
|
||||
/// The job may already be complete when returned, and callers must not
|
||||
/// assume the column is filled until [`Job::wait`] returns. Invalid input
|
||||
/// -- an unknown column, or one that is not computed -- is reported by
|
||||
/// this call rather than by the job. On local tables the job runs as an
|
||||
/// in-process task; on LanceDB Cloud and Enterprise it is the server's
|
||||
/// backfill job.
|
||||
///
|
||||
/// ```
|
||||
/// # use lancedb::Table;
|
||||
/// # async fn refresh_in_background(table: &Table) -> Result<(), Box<dyn std::error::Error>> {
|
||||
/// let job = table.refresh_column_async("doubled").await?;
|
||||
/// println!("refresh running: {:?}", job.status().await?);
|
||||
/// job.wait().await?;
|
||||
/// # Ok(())
|
||||
/// # }
|
||||
/// ```
|
||||
pub async fn refresh_column_async(&self, column: impl AsRef<str>) -> Result<Job> {
|
||||
self.inner.refresh_column_async(column.as_ref()).await
|
||||
}
|
||||
|
||||
/// Change a column's name or nullability.
|
||||
pub async fn alter_columns(
|
||||
&self,
|
||||
@@ -1693,6 +1856,20 @@ impl Table {
|
||||
self.inner.set_lsm_write_spec(spec).await
|
||||
}
|
||||
|
||||
/// Switch this table to required index catch-up, one way.
|
||||
///
|
||||
/// Separate from [`Self::set_lsm_write_spec`] on purpose: a table carrying
|
||||
/// the bit retains its SSTables until an index records that it holds the
|
||||
/// compacted rows, so turn it on only once something can repair coverage.
|
||||
/// A writer that already holds the dataset can call the equivalent on
|
||||
/// `DatasetMemWalExt` instead; this is the table-level entry point.
|
||||
///
|
||||
/// Errors if no spec is set, or if the table already records SSTable
|
||||
/// compaction progress from before this protocol.
|
||||
pub async fn require_mem_wal_index_catchup(&self) -> Result<()> {
|
||||
self.inner.require_mem_wal_index_catchup().await
|
||||
}
|
||||
|
||||
/// Remove the [`LsmWriteSpec`] from this table, reverting to the standard
|
||||
/// `merge_insert` write path.
|
||||
///
|
||||
@@ -2605,6 +2782,7 @@ impl NativeTable {
|
||||
namespace_client: Option<Arc<dyn LanceNamespace>>,
|
||||
pushdown_operations: HashSet<NamespaceClientPushdownOperation>,
|
||||
) -> Result<Self> {
|
||||
computed_columns::ensure_no_foreign_declarations(batches.arrow_schema().fields())?;
|
||||
// Default params uses format v1.
|
||||
let params = params.unwrap_or(WriteParams {
|
||||
..Default::default()
|
||||
@@ -3053,6 +3231,13 @@ impl BaseTable for NativeTable {
|
||||
let ds = self.dataset.get().await?;
|
||||
|
||||
let table_schema = Schema::from(&ds.schema().clone());
|
||||
computed_columns::ensure_not_written(
|
||||
&table_schema,
|
||||
add.data.schema().fields().iter().map(|f| f.name().as_str()),
|
||||
)?;
|
||||
if matches!(add.mode, AddDataMode::Overwrite) {
|
||||
computed_columns::ensure_no_foreign_declarations(add.data.schema().fields())?;
|
||||
}
|
||||
|
||||
let num_partitions = if let Some(parallelism) = add.write_parallelism {
|
||||
parallelism
|
||||
@@ -3213,6 +3398,11 @@ impl BaseTable for NativeTable {
|
||||
params: MergeInsertBuilder,
|
||||
new_data: Box<dyn RecordBatchReader + Send>,
|
||||
) -> Result<MergeResult> {
|
||||
let source_schema = arrow_array::RecordBatchReader::schema(&new_data);
|
||||
computed_columns::ensure_not_written(
|
||||
&Schema::from(self.dataset.get().await?.schema()),
|
||||
source_schema.fields().iter().map(|f| f.name().as_str()),
|
||||
)?;
|
||||
let result = merge::execute_merge_insert(self, params, new_data).await?;
|
||||
self.bump_freshness();
|
||||
Ok(result)
|
||||
@@ -3226,6 +3416,10 @@ impl BaseTable for NativeTable {
|
||||
merge::lsm::set_lsm_write_spec(self, spec).await
|
||||
}
|
||||
|
||||
async fn require_mem_wal_index_catchup(&self) -> Result<()> {
|
||||
merge::lsm::require_mem_wal_index_catchup(self).await
|
||||
}
|
||||
|
||||
async fn unset_lsm_write_spec(&self) -> Result<()> {
|
||||
merge::lsm::unset_lsm_write_spec(self).await
|
||||
}
|
||||
@@ -3243,6 +3437,25 @@ impl BaseTable for NativeTable {
|
||||
Ok(crate::blob::blob_column_names(schema.as_ref()))
|
||||
}
|
||||
|
||||
async fn add_bases(&self, bases: &[TableBase]) -> Result<()> {
|
||||
self.dataset.ensure_mutable()?;
|
||||
let dataset = self.dataset.get().await?;
|
||||
let new_bases = bases
|
||||
.iter()
|
||||
.map(|base| {
|
||||
BasePath::new(
|
||||
0,
|
||||
base.path.clone(),
|
||||
base.name.clone(),
|
||||
base.is_dataset_root,
|
||||
)
|
||||
})
|
||||
.collect();
|
||||
let dataset = dataset.add_bases(new_bases, None).await?;
|
||||
self.dataset.update(dataset);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn fetch_blobs(&self, column: &str, row_ids: &[u64]) -> Result<LargeBinaryArray> {
|
||||
let dataset = self.dataset.get().await?;
|
||||
crate::blob::take_blobs_aligned(&dataset, column, row_ids).await
|
||||
@@ -3294,6 +3507,22 @@ impl BaseTable for NativeTable {
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
async fn add_computed_columns(&self, columns: &[(String, String)]) -> Result<AddColumnsResult> {
|
||||
let result = schema_evolution::execute_declare(self, columns).await?;
|
||||
self.bump_freshness();
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
async fn refresh_column(&self, column: &str) -> Result<RefreshColumnResult> {
|
||||
let result = refresh::execute_refresh_column(self, column).await?;
|
||||
self.bump_freshness();
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
async fn refresh_column_async(&self, column: &str) -> Result<Job> {
|
||||
refresh::execute_refresh_column_async(self, column).await
|
||||
}
|
||||
|
||||
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult> {
|
||||
let result = schema_evolution::execute_alter_columns(self, alterations).await?;
|
||||
self.bump_freshness();
|
||||
|
||||
@@ -15,6 +15,7 @@ use crate::{Error, Result};
|
||||
pub struct AddColumnsBuilder {
|
||||
parent: Arc<dyn BaseTable>,
|
||||
transform: Option<NewColumnTransform>,
|
||||
computed: Vec<(String, String)>,
|
||||
read_columns: Option<Vec<String>>,
|
||||
}
|
||||
|
||||
@@ -23,6 +24,7 @@ impl std::fmt::Debug for AddColumnsBuilder {
|
||||
f.debug_struct("AddColumnsBuilder")
|
||||
.field("parent", &self.parent)
|
||||
.field("has_transform", &self.transform.is_some())
|
||||
.field("computed", &self.computed)
|
||||
.field("read_columns", &self.read_columns)
|
||||
.finish()
|
||||
}
|
||||
@@ -33,19 +35,58 @@ impl AddColumnsBuilder {
|
||||
Self {
|
||||
parent,
|
||||
transform: None,
|
||||
computed: Vec::new(),
|
||||
read_columns: None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Set how the new columns' values are produced. Required.
|
||||
/// Set how the new columns' values are produced.
|
||||
pub fn transform(mut self, transform: NewColumnTransform) -> Self {
|
||||
self.transform = Some(transform);
|
||||
self
|
||||
}
|
||||
|
||||
/// Add a column defined by `expression`, evaluated by a later refresh
|
||||
/// rather than by this commit. Its type and inputs are derived from the
|
||||
/// expression.
|
||||
///
|
||||
/// The column is committed with no values, so declaring one costs the same
|
||||
/// on an empty table as on a large one. Rows get values from
|
||||
/// [`Table::refresh_column`](super::Table::refresh_column), which fills
|
||||
/// every fragment that has none -- including fragments appended since the
|
||||
/// last refresh.
|
||||
///
|
||||
/// Refresh does not revisit a fragment it has filled, so mutating an input
|
||||
/// leaves the value computed at fill time; recomputing means dropping the
|
||||
/// column and declaring it again. An input cannot be renamed, retyped or
|
||||
/// dropped while a declaration reads it, since the expression names it.
|
||||
///
|
||||
/// On LanceDB Cloud and Enterprise the expression is planned by the
|
||||
/// server, and the refresh runs as a server job -- see
|
||||
/// [`Table::refresh_column_async`](super::Table::refresh_column_async).
|
||||
///
|
||||
/// ```
|
||||
/// # use lancedb::Table;
|
||||
/// # async fn declare(table: &Table) -> Result<(), Box<dyn std::error::Error>> {
|
||||
/// table
|
||||
/// .add_columns()
|
||||
/// .computed("doubled", "x * 2")
|
||||
/// .execute()
|
||||
/// .await?;
|
||||
/// let filled = table.refresh_column("doubled").await?;
|
||||
/// println!("filled {} rows", filled.rows_filled);
|
||||
/// # Ok(())
|
||||
/// # }
|
||||
/// ```
|
||||
pub fn computed(mut self, name: impl Into<String>, expression: impl Into<String>) -> Self {
|
||||
self.computed.push((name.into(), expression.into()));
|
||||
self
|
||||
}
|
||||
|
||||
/// Limit which existing columns a [`NewColumnTransform::BatchUDF`] mapper
|
||||
/// receives. Every other transform determines what it reads, so setting
|
||||
/// this alongside one is an error rather than a silent no-op.
|
||||
/// receives. Every other transform, and a computed column, determines what
|
||||
/// it reads, so setting this alongside one is an error rather than a silent
|
||||
/// no-op.
|
||||
pub fn read_columns(mut self, columns: impl IntoIterator<Item = impl Into<String>>) -> Self {
|
||||
self.read_columns = Some(columns.into_iter().map(Into::into).collect());
|
||||
self
|
||||
@@ -56,24 +97,42 @@ impl AddColumnsBuilder {
|
||||
let Self {
|
||||
parent,
|
||||
transform,
|
||||
computed,
|
||||
read_columns,
|
||||
} = self;
|
||||
|
||||
let Some(transform) = transform else {
|
||||
return Err(Error::InvalidInput {
|
||||
message: "add_columns requires a transform".into(),
|
||||
});
|
||||
};
|
||||
|
||||
if read_columns.is_some() && !matches!(transform, NewColumnTransform::BatchUDF(_)) {
|
||||
return Err(Error::InvalidInput {
|
||||
message: "read_columns applies only to a BatchUDF transform; \
|
||||
every other transform determines what it reads"
|
||||
match (transform, computed.is_empty()) {
|
||||
(None, true) => Err(Error::InvalidInput {
|
||||
message: "add_columns requires a transform or a computed column".into(),
|
||||
}),
|
||||
// The two commit through different transforms, so one call covering
|
||||
// both would be two commits and could half-apply.
|
||||
(Some(_), false) => Err(Error::InvalidInput {
|
||||
message: "add_columns cannot mix a transform with computed columns; \
|
||||
they cannot be added atomically in one call"
|
||||
.into(),
|
||||
});
|
||||
}),
|
||||
(Some(transform), true) => {
|
||||
if read_columns.is_some() && !matches!(transform, NewColumnTransform::BatchUDF(_)) {
|
||||
return Err(Error::InvalidInput {
|
||||
message: "read_columns applies only to a BatchUDF transform; \
|
||||
every other transform determines what it reads"
|
||||
.into(),
|
||||
});
|
||||
}
|
||||
parent.add_columns(transform, read_columns).await
|
||||
}
|
||||
(None, false) => {
|
||||
if read_columns.is_some() {
|
||||
return Err(Error::InvalidInput {
|
||||
message: "read_columns applies only to a BatchUDF transform; \
|
||||
a computed column's inputs come from its expression"
|
||||
.into(),
|
||||
});
|
||||
}
|
||||
parent.add_computed_columns(&computed).await
|
||||
}
|
||||
}
|
||||
|
||||
parent.add_columns(transform, read_columns).await
|
||||
}
|
||||
}
|
||||
|
||||
@@ -85,8 +144,8 @@ mod tests {
|
||||
use arrow_schema::{DataType, Field, Schema};
|
||||
use lance::dataset::{BatchUDF, NewColumnTransform};
|
||||
|
||||
use crate::Table;
|
||||
use crate::connect;
|
||||
use crate::{Error, Table};
|
||||
|
||||
async fn table_with_two_columns(name: &str) -> Table {
|
||||
let conn = connect("memory://").execute().await.unwrap();
|
||||
@@ -98,10 +157,7 @@ mod tests {
|
||||
async fn test_requires_a_transform() {
|
||||
let table = table_with_two_columns("no_transform").await;
|
||||
let err = table.add_columns().execute().await.unwrap_err();
|
||||
assert!(
|
||||
err.to_string().contains("requires a transform"),
|
||||
"got: {err}"
|
||||
);
|
||||
assert!(matches!(err, Error::InvalidInput { .. }));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
@@ -117,7 +173,7 @@ mod tests {
|
||||
.execute()
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(err.to_string().contains("BatchUDF"), "got: {err}");
|
||||
assert!(matches!(err, Error::InvalidInput { .. }));
|
||||
|
||||
let schema = table.schema().await.unwrap();
|
||||
assert!(
|
||||
@@ -126,6 +182,47 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_mixing_transform_and_computed_is_rejected() {
|
||||
let table = table_with_two_columns("mixed_add").await;
|
||||
let err = table
|
||||
.add_columns()
|
||||
.transform(NewColumnTransform::SqlExpressions(vec![(
|
||||
"eager".into(),
|
||||
"x * 2".into(),
|
||||
)]))
|
||||
.computed("lazy", "x * 3")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, Error::InvalidInput { .. }));
|
||||
|
||||
let schema = table.schema().await.unwrap();
|
||||
assert!(schema.field_with_name("eager").is_err());
|
||||
assert!(schema.field_with_name("lazy").is_err());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_read_columns_with_computed_is_rejected() {
|
||||
let table = table_with_two_columns("read_cols_computed").await;
|
||||
let err = table
|
||||
.add_columns()
|
||||
.computed("doubled", "x * 2")
|
||||
.read_columns(["x"])
|
||||
.execute()
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, Error::InvalidInput { .. }));
|
||||
assert!(
|
||||
table
|
||||
.schema()
|
||||
.await
|
||||
.unwrap()
|
||||
.field_with_name("doubled")
|
||||
.is_err()
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_read_columns_limits_what_a_batch_udf_sees() {
|
||||
let table = table_with_two_columns("read_cols_udf").await;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -17,7 +17,7 @@ use datafusion_physical_plan::stream::RecordBatchStreamAdapter;
|
||||
use datafusion_physical_plan::{
|
||||
DisplayAs, DisplayFormatType, ExecutionPlan, ExecutionPlanProperties, PlanProperties,
|
||||
};
|
||||
use futures::TryStreamExt;
|
||||
use futures::StreamExt;
|
||||
use lance::Dataset;
|
||||
use lance::dataset::transaction::{Operation, Transaction};
|
||||
use lance::dataset::{CommitBuilder, InsertBuilder, WriteParams, WriteProgressFn};
|
||||
@@ -194,12 +194,23 @@ impl ExecutionPlan for InsertExec {
|
||||
|
||||
let output_bytes = MetricBuilder::new(&self.metrics).output_bytes(partition);
|
||||
let input_schema = input_stream.schema();
|
||||
let declared: Vec<String> = crate::table::computed_columns::computed_columns(
|
||||
&arrow_schema::Schema::from(self.dataset.schema()),
|
||||
)
|
||||
.into_iter()
|
||||
.map(|declaration| declaration.name)
|
||||
.collect();
|
||||
let input_stream: SendableRecordBatchStream =
|
||||
Box::pin(InstrumentedRecordBatchStreamAdapter::new(
|
||||
input_schema,
|
||||
input_stream.map_ok(move |batch| {
|
||||
input_stream.map(move |batch| {
|
||||
let batch = batch?;
|
||||
crate::table::computed_columns::ensure_batch_writes_no_computed_values(
|
||||
&declared, &batch,
|
||||
)
|
||||
.map_err(|e| datafusion::error::DataFusionError::External(Box::new(e)))?;
|
||||
output_bytes.add(batch.get_array_memory_size());
|
||||
batch
|
||||
Ok(batch)
|
||||
}),
|
||||
partition,
|
||||
&self.metrics,
|
||||
|
||||
@@ -94,7 +94,16 @@ pub(crate) async fn set_lsm_write_spec(table: &NativeTable, spec: LsmWriteSpec)
|
||||
.await?
|
||||
};
|
||||
|
||||
table.checkout_latest().await?;
|
||||
let mut dataset = (*table.dataset.get().await?).clone();
|
||||
let schema = arrow_schema::Schema::from(dataset.schema());
|
||||
if !crate::table::computed_columns::computed_columns(&schema).is_empty() {
|
||||
return Err(Error::NotSupported {
|
||||
message: "an LSM write spec cannot be installed on a table with computed \
|
||||
columns: rows in un-compacted tiers are invisible to refresh"
|
||||
.into(),
|
||||
});
|
||||
}
|
||||
let mut builder = dataset.initialize_mem_wal();
|
||||
let writer_config_defaults = match spec {
|
||||
LsmWriteSpec::Bucket {
|
||||
@@ -183,6 +192,36 @@ fn index_name_list(indices: &[IndexConfig]) -> String {
|
||||
format!("[{}]", names.join(", "))
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// require_mem_wal_index_catchup
|
||||
// =============================================================================
|
||||
|
||||
/// Switch this table to required index catch-up, one way.
|
||||
///
|
||||
/// Deliberately **not** part of installing the write spec. Until something can
|
||||
/// actually repair coverage, a table carrying the bit reports every index as
|
||||
/// not known to hold the compacted rows, so its SSTables are retained
|
||||
/// indefinitely -- and the WAL pod trims on the legacy rule meanwhile, leaving
|
||||
/// readers pointed at files that are gone. Turn this on only once remote
|
||||
/// maintenance owns the merge and the repair for the table.
|
||||
///
|
||||
/// Lance refuses the activation if the table already records SSTable
|
||||
/// compaction progress: those numbers predate this protocol and cannot be
|
||||
/// validated, so such a table must be drained rather than activated.
|
||||
#[allow(clippy::redundant_pub_crate)]
|
||||
pub(crate) async fn require_mem_wal_index_catchup(table: &NativeTable) -> Result<()> {
|
||||
table.dataset.ensure_mutable()?;
|
||||
let mut dataset = (*table.dataset.get().await?).clone();
|
||||
if dataset.mem_wal_index_details().await?.is_none() {
|
||||
return Err(Error::InvalidInput {
|
||||
message: "require_mem_wal_index_catchup: no LSM write spec is set on this table".into(),
|
||||
});
|
||||
}
|
||||
dataset.require_mem_wal_index_catchup().await?;
|
||||
table.dataset.update(dataset);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// unset_lsm_write_spec
|
||||
// =============================================================================
|
||||
|
||||
@@ -36,6 +36,7 @@ use lance::dataset::mem_wal::{
|
||||
DatasetMemWalExt, LsmScanner, ShardManifestStore, ShardSnapshot, ShardWriterConfig,
|
||||
};
|
||||
use lance_index::mem_wal::{MemWalIndexDetails, ShardManifest};
|
||||
use lance_table::feature_flags::FLAG_MEM_WAL_INDEX_CATCHUP;
|
||||
use uuid::Uuid;
|
||||
|
||||
use super::NativeTable;
|
||||
@@ -248,18 +249,26 @@ fn pk_columns(dataset: &Dataset) -> Result<Vec<String>> {
|
||||
fn exclusion_watermarks(
|
||||
details: &MemWalIndexDetails,
|
||||
index_names: &[String],
|
||||
catchup_required: bool,
|
||||
) -> HashMap<Uuid, u64> {
|
||||
let mut exclude: HashMap<Uuid, u64> = HashMap::new();
|
||||
for entry in &details.compacted_sstables {
|
||||
let mut watermark = entry.generation;
|
||||
for name in index_names {
|
||||
if let Some(caught_up) = details
|
||||
match details
|
||||
.index_catchup
|
||||
.iter()
|
||||
.find(|icp| icp.index_name == *name)
|
||||
.and_then(|icp| icp.caught_up_generation_for_shard(&entry.shard_id))
|
||||
{
|
||||
watermark = watermark.min(caught_up);
|
||||
Some(caught_up) => watermark = watermark.min(caught_up),
|
||||
// No entry. On a table that requires catch-up this means the
|
||||
// index is *not* known to hold these rows, and the base arm is
|
||||
// index-only -- so every generation stays readable from its
|
||||
// SSTable. Without the bit the field is not maintained at all,
|
||||
// and absence carries no information.
|
||||
None if catchup_required => watermark = 0,
|
||||
None => {}
|
||||
}
|
||||
}
|
||||
exclude.entry(entry.shard_id).or_insert(watermark);
|
||||
@@ -274,13 +283,26 @@ fn exclusion_watermarks(
|
||||
/// with a live cached `ShardWriter` (this session's in-flight writes) the
|
||||
/// writer's authoritative in-memory manifest and memtables override the
|
||||
/// on-disk view so a read sees data not yet flushed.
|
||||
/// Whether this table reads a missing `index_catchup` entry as "not caught up".
|
||||
///
|
||||
/// Both words must be set. A reader honouring the bit while a writer does not
|
||||
/// would retain SSTables the writer had already trimmed, and the reverse would
|
||||
/// serve rows from files the writer still expects to be excluded -- so a
|
||||
/// half-set manifest is treated as legacy, which is the conservative side.
|
||||
fn requires_index_catchup(dataset: &Dataset) -> bool {
|
||||
let manifest = dataset.manifest();
|
||||
manifest.reader_feature_flags & FLAG_MEM_WAL_INDEX_CATCHUP != 0
|
||||
&& manifest.writer_feature_flags & FLAG_MEM_WAL_INDEX_CATCHUP != 0
|
||||
}
|
||||
|
||||
async fn build_read_context(
|
||||
table: &NativeTable,
|
||||
dataset: &Dataset,
|
||||
details: &MemWalIndexDetails,
|
||||
index_names: &[String],
|
||||
) -> Result<(Vec<ShardSnapshot>, HashMap<Uuid, InMemoryMemTables>)> {
|
||||
let exclude = exclusion_watermarks(details, index_names);
|
||||
let catchup_required = requires_index_catchup(dataset);
|
||||
let exclude = exclusion_watermarks(details, index_names, catchup_required);
|
||||
|
||||
let shard_ids = dataset.list_mem_wal_latest_shard_ids().await?;
|
||||
// Use the dataset's own object store (not `ObjectStore::from_uri`, which
|
||||
@@ -767,22 +789,50 @@ mod tests {
|
||||
};
|
||||
|
||||
// Plain scan: drop every compacted generation (through 5).
|
||||
assert_eq!(exclusion_watermarks(&details, &[]).get(&shard), Some(&5));
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &[], false).get(&shard),
|
||||
Some(&5)
|
||||
);
|
||||
|
||||
// FTS arm with a lagging index: exclusion is capped at the index catch-up
|
||||
// (2), so SSTable generations 3..=5 are retained until the index covers
|
||||
// them — otherwise those documents would silently vanish from FTS results.
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &["fts_idx".to_string()]).get(&shard),
|
||||
exclusion_watermarks(&details, &["fts_idx".to_string()], false).get(&shard),
|
||||
Some(&2)
|
||||
);
|
||||
|
||||
// A caught-up index — or one untracked in index_catchup — falls back to the
|
||||
// compaction watermark.
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &["caught_up_idx".to_string()]).get(&shard),
|
||||
exclusion_watermarks(&details, &["caught_up_idx".to_string()], false).get(&shard),
|
||||
Some(&5)
|
||||
);
|
||||
|
||||
// The same missing entry, once the table requires catch-up: absence now
|
||||
// means "not known to hold these rows", so nothing may be excluded and
|
||||
// every generation stays readable from its SSTable. This is the whole
|
||||
// point of the protocol -- an indexed query against a table whose index
|
||||
// has not caught up must not silently lose rows.
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &["untracked_idx".to_string()], true).get(&shard),
|
||||
Some(&0)
|
||||
);
|
||||
|
||||
// A tracked index is unaffected by the mode: the recorded position is
|
||||
// information either way, and it still caps the exclusion.
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &["fts_idx".to_string()], true).get(&shard),
|
||||
Some(&2)
|
||||
);
|
||||
|
||||
// One missing entry is enough to hold everything back, even alongside an
|
||||
// index that has caught up.
|
||||
let mixed = vec!["fts_idx".to_string(), "untracked_idx".to_string()];
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &mixed, true).get(&shard),
|
||||
Some(&0)
|
||||
);
|
||||
}
|
||||
|
||||
/// A hybrid search reads a vector and a full-text index, and either may lag.
|
||||
@@ -809,20 +859,23 @@ mod tests {
|
||||
|
||||
// Each index alone stops at its own catch-up.
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &["vec_idx".to_string()]).get(&shard),
|
||||
exclusion_watermarks(&details, &["vec_idx".to_string()], false).get(&shard),
|
||||
Some(&7)
|
||||
);
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &["fts_idx".to_string()]).get(&shard),
|
||||
exclusion_watermarks(&details, &["fts_idx".to_string()], false).get(&shard),
|
||||
Some(&4)
|
||||
);
|
||||
|
||||
// Used together, the lower one governs regardless of order.
|
||||
let both = ["vec_idx".to_string(), "fts_idx".to_string()];
|
||||
assert_eq!(exclusion_watermarks(&details, &both).get(&shard), Some(&4));
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &both, false).get(&shard),
|
||||
Some(&4)
|
||||
);
|
||||
let reversed = ["fts_idx".to_string(), "vec_idx".to_string()];
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &reversed).get(&shard),
|
||||
exclusion_watermarks(&details, &reversed, false).get(&shard),
|
||||
Some(&4)
|
||||
);
|
||||
}
|
||||
@@ -843,7 +896,10 @@ mod tests {
|
||||
};
|
||||
|
||||
let both = ["fts_idx".to_string(), "untracked_idx".to_string()];
|
||||
assert_eq!(exclusion_watermarks(&details, &both).get(&shard), Some(&4));
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &both, false).get(&shard),
|
||||
Some(&4)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -0,0 +1,954 @@
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||
|
||||
//! Filling computed columns.
|
||||
//!
|
||||
//! A row without a value gets one; a row that has one keeps it. Refresh is
|
||||
//! therefore idempotent and does not observe input mutation -- once a row is
|
||||
//! filled, changing what the expression reads leaves the stored result alone.
|
||||
//!
|
||||
//! Two passes per fragment. The first scans only the unfilled live rows and
|
||||
//! evaluates the expression over them, which yields the exact fill count and
|
||||
//! decides whether the fragment is staged at all -- a fragment where nothing
|
||||
//! would change stages nothing, which is what lets an expression yielding
|
||||
//! null settle instead of restaging forever. The second streams the
|
||||
//! fragment's physical rows into `write_column` a batch at a time, so peak
|
||||
//! memory is bounded by a scan batch. The expression is evaluated by this
|
||||
//! module, never through a projection alias, and only over rows being
|
||||
//! filled: every other row -- deleted, or already holding a value -- has its
|
||||
//! inputs masked to null first, so a poison value in a row nobody is filling
|
||||
//! cannot fail the refresh.
|
||||
|
||||
use std::sync::Arc;
|
||||
|
||||
use arrow_array::{ArrayRef, BooleanArray, RecordBatch, RecordBatchOptions};
|
||||
use arrow_schema::Schema as ArrowSchema;
|
||||
use datafusion_expr::ColumnarValue;
|
||||
use futures::{Stream, StreamExt, TryStreamExt};
|
||||
use lance::Dataset;
|
||||
use lance::dataset::WriteDestination;
|
||||
use lance::dataset::fragment::FileFragment;
|
||||
use lance::dataset::transaction::Operation;
|
||||
use lance_core::ROW_ID;
|
||||
use lance_core::datatypes::Schema as LanceSchema;
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use super::computed_columns::{BoundExpression, ComputedColumnKind, computed_column_from_field};
|
||||
use super::{BaseTable, NativeTable};
|
||||
use crate::job::Job;
|
||||
use crate::{Error, Result};
|
||||
|
||||
/// The result of refreshing a computed column.
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
|
||||
pub struct RefreshColumnResult {
|
||||
/// Rows that had a value computed.
|
||||
#[serde(default)]
|
||||
pub rows_filled: u64,
|
||||
/// The commit version associated with the operation.
|
||||
#[serde(default)]
|
||||
pub version: u64,
|
||||
}
|
||||
|
||||
/// Internal implementation of the refresh logic.
|
||||
pub(crate) async fn execute_refresh_column(
|
||||
table: &NativeTable,
|
||||
column: &str,
|
||||
) -> Result<RefreshColumnResult> {
|
||||
table.dataset.ensure_mutable()?;
|
||||
ensure_no_lsm_write_spec(table).await?;
|
||||
let dataset = table.dataset.get().await?;
|
||||
|
||||
let expression = declared_expression(&dataset, column)?;
|
||||
let schema = Arc::new(ArrowSchema::from(dataset.schema()));
|
||||
let bound = Arc::new(super::computed_columns::bind(schema, column, &expression)?);
|
||||
let field = dataset
|
||||
.schema()
|
||||
.field(column)
|
||||
.ok_or_else(|| Error::ColumnNotFound {
|
||||
name: column.to_string(),
|
||||
})?;
|
||||
// The dataset's own field, so the identity write_column checks against the
|
||||
// manifest holds by construction.
|
||||
let column_schema = LanceSchema {
|
||||
fields: vec![field.clone()],
|
||||
metadata: Default::default(),
|
||||
};
|
||||
|
||||
let mut rows_filled = 0u64;
|
||||
let mut replacements = Vec::new();
|
||||
for fragment in dataset.get_fragments() {
|
||||
let gained = count_fragment_gains(&dataset, &fragment, &bound, column).await?;
|
||||
if gained == 0 {
|
||||
continue;
|
||||
}
|
||||
rows_filled += gained;
|
||||
let values = fill_stream(&dataset, &fragment, bound.clone(), column).await?;
|
||||
replacements.push(fragment.write_column(values, &column_schema).await?);
|
||||
}
|
||||
|
||||
if replacements.is_empty() {
|
||||
return Ok(RefreshColumnResult {
|
||||
rows_filled: 0,
|
||||
version: dataset.version().version,
|
||||
});
|
||||
}
|
||||
|
||||
let read_version = dataset.version().version;
|
||||
// The dataset's own session, so registrations and caches survive the
|
||||
// commit being installed on the handle.
|
||||
let session = dataset.session();
|
||||
let new_dataset = Dataset::commit(
|
||||
WriteDestination::Dataset(dataset.clone()),
|
||||
Operation::DataReplacement { replacements },
|
||||
Some(read_version),
|
||||
None,
|
||||
None,
|
||||
session,
|
||||
false,
|
||||
)
|
||||
.await?;
|
||||
|
||||
let version = new_dataset.version().version;
|
||||
table.dataset.update(new_dataset);
|
||||
Ok(RefreshColumnResult {
|
||||
rows_filled,
|
||||
version,
|
||||
})
|
||||
}
|
||||
|
||||
/// Run the refresh as a [`Job`] in this process.
|
||||
pub(crate) async fn execute_refresh_column_async(table: &NativeTable, column: &str) -> Result<Job> {
|
||||
// Validate before spawning so bad input is reported by this call rather
|
||||
// than only by the job.
|
||||
table.dataset.ensure_mutable()?;
|
||||
ensure_no_lsm_write_spec(table).await?;
|
||||
let dataset = table.dataset.get().await?;
|
||||
declared_expression(&dataset, column)?;
|
||||
drop(dataset);
|
||||
|
||||
let table = table.clone();
|
||||
let column = column.to_string();
|
||||
Ok(Job::spawned(tokio::spawn(async move {
|
||||
execute_refresh_column(&table, &column).await?;
|
||||
table.bump_freshness();
|
||||
Ok(())
|
||||
})))
|
||||
}
|
||||
|
||||
/// Refuse to refresh under an LSM write spec.
|
||||
///
|
||||
/// Refresh enumerates base fragments, and a write spec keeps visible rows in
|
||||
/// un-compacted MemWAL tiers it cannot reach -- success would silently omit
|
||||
/// readable rows.
|
||||
async fn ensure_no_lsm_write_spec(table: &NativeTable) -> Result<()> {
|
||||
// The catch-up flag outlives unset and marks retained SSTable rows.
|
||||
let catchup = table.dataset.get().await?.manifest().reader_feature_flags
|
||||
& lance_table::feature_flags::FLAG_MEM_WAL_INDEX_CATCHUP
|
||||
!= 0;
|
||||
if catchup || table.get_lsm_write_spec().await?.is_some() {
|
||||
return Err(Error::NotSupported {
|
||||
message: "refresh_column is not supported on a table with an LSM write \
|
||||
spec: rows in un-compacted tiers are invisible to refresh"
|
||||
.into(),
|
||||
});
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The SQL expression `column` is declared with.
|
||||
fn declared_expression(dataset: &Dataset, column: &str) -> Result<String> {
|
||||
let schema = ArrowSchema::from(dataset.schema());
|
||||
let field = schema
|
||||
.field_with_name(column)
|
||||
.map_err(|_| Error::ColumnNotFound {
|
||||
name: column.to_string(),
|
||||
})?;
|
||||
let declaration =
|
||||
computed_column_from_field(field).ok_or_else(|| Error::NotAComputedColumn {
|
||||
name: column.to_string(),
|
||||
})?;
|
||||
match declaration.kind {
|
||||
ComputedColumnKind::Sql { expression } => Ok(expression),
|
||||
ComputedColumnKind::Unrecognized { kind } => Err(Error::NotSupported {
|
||||
message: format!(
|
||||
"computed column '{column}' is defined by '{kind}', which this version of \
|
||||
lancedb cannot evaluate"
|
||||
),
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
/// Quote `name` as a lance SQL identifier.
|
||||
///
|
||||
/// Lance's dialect delimits with backticks, so a double-quoted name would
|
||||
/// parse as a string literal rather than a column.
|
||||
fn quote_identifier(name: &str) -> String {
|
||||
format!("`{}`", name.replace('`', "``"))
|
||||
}
|
||||
|
||||
/// Assemble the batch evaluation runs against: the bound roots, in read-schema
|
||||
/// order. Built by name so scan-side column order never matters.
|
||||
fn evaluation_batch(
|
||||
batch: &RecordBatch,
|
||||
bound: &BoundExpression,
|
||||
mask_out: Option<&BooleanArray>,
|
||||
) -> lance_core::Result<RecordBatch> {
|
||||
let mut columns = Vec::with_capacity(bound.roots.len());
|
||||
for name in &bound.roots {
|
||||
let column = batch.column_by_name(name).ok_or_else(|| {
|
||||
lance_core::Error::invalid_input(format!(
|
||||
"refreshing a computed column read no {name} column"
|
||||
))
|
||||
})?;
|
||||
// Rows outside the mask must not reach the expression: a value in a
|
||||
// deleted or already-filled row can be one it would choke on.
|
||||
columns.push(match mask_out {
|
||||
Some(mask) => arrow::compute::nullif(column, mask)?,
|
||||
None => column.clone(),
|
||||
});
|
||||
}
|
||||
Ok(RecordBatch::try_new_with_options(
|
||||
bound.read_schema.clone(),
|
||||
columns,
|
||||
&RecordBatchOptions::new().with_row_count(Some(batch.num_rows())),
|
||||
)?)
|
||||
}
|
||||
|
||||
/// Evaluate the expression over `batch`, materializing a constant result to
|
||||
/// the batch's length.
|
||||
fn evaluate(bound: &BoundExpression, batch: &RecordBatch) -> lance_core::Result<ArrayRef> {
|
||||
let value = bound
|
||||
.physical
|
||||
.evaluate(batch)
|
||||
.map_err(lance_core::Error::from)?;
|
||||
match value {
|
||||
ColumnarValue::Array(array) => Ok(array),
|
||||
scalar => scalar
|
||||
.into_array(batch.num_rows())
|
||||
.map_err(lance_core::Error::from),
|
||||
}
|
||||
}
|
||||
|
||||
/// How many rows of one fragment would gain a value.
|
||||
///
|
||||
/// Scans only the unfilled live rows -- deleted rows never reach the
|
||||
/// expression here, the filter having already excluded them -- and counts the
|
||||
/// non-null results. Exact, so it is both the staging decision and the
|
||||
/// fragment's contribution to `rows_filled`.
|
||||
async fn count_fragment_gains(
|
||||
dataset: &Dataset,
|
||||
fragment: &FileFragment,
|
||||
bound: &BoundExpression,
|
||||
column: &str,
|
||||
) -> Result<u64> {
|
||||
let mut scanner = dataset.scan();
|
||||
scanner
|
||||
.with_fragments(vec![fragment.metadata().clone()])
|
||||
.with_row_id()
|
||||
.filter(&format!("{} IS NULL", quote_identifier(column)))?
|
||||
.project(&bound.roots)?;
|
||||
|
||||
let mut gained = 0u64;
|
||||
let mut batches = scanner.try_into_stream().await?;
|
||||
while let Some(batch) = batches.try_next().await? {
|
||||
let evaluated = evaluate(bound, &evaluation_batch(&batch, bound, None)?)?;
|
||||
gained += (batch.num_rows() - evaluated.null_count()) as u64;
|
||||
}
|
||||
Ok(gained)
|
||||
}
|
||||
|
||||
/// Stream one fragment's column in physical order, filling the unfilled live
|
||||
/// rows and keeping every other value.
|
||||
///
|
||||
/// Deleted rows are carried through so the values line up positionally with
|
||||
/// the fragment's data files; they are never read back, but the column file
|
||||
/// has to cover them.
|
||||
async fn fill_stream(
|
||||
dataset: &Dataset,
|
||||
fragment: &FileFragment,
|
||||
bound: Arc<BoundExpression>,
|
||||
column: &str,
|
||||
) -> Result<impl Stream<Item = lance_core::Result<RecordBatch>> + Send + use<>> {
|
||||
let mut projection: Vec<String> = bound.roots.clone();
|
||||
projection.push(column.to_string());
|
||||
let mut scanner = dataset.scan();
|
||||
scanner
|
||||
.with_fragments(vec![fragment.metadata().clone()])
|
||||
.with_row_id()
|
||||
.include_deleted_rows()
|
||||
.project(&projection)?;
|
||||
|
||||
let projected = Arc::new(ArrowSchema::new(vec![
|
||||
ArrowSchema::from(dataset.schema())
|
||||
.field_with_name(column)
|
||||
.map_err(|_| Error::ColumnNotFound {
|
||||
name: column.to_string(),
|
||||
})?
|
||||
.clone(),
|
||||
]));
|
||||
|
||||
let column = column.to_string();
|
||||
let batches = scanner.try_into_stream().await?;
|
||||
Ok(batches.map(move |batch| {
|
||||
let batch = batch?;
|
||||
let missing = |name: &str| {
|
||||
lance_core::Error::invalid_input(format!(
|
||||
"refreshing a computed column read no {name} column"
|
||||
))
|
||||
};
|
||||
let existing = batch
|
||||
.column_by_name(&column)
|
||||
.ok_or_else(|| missing(&column))?;
|
||||
let row_ids = batch
|
||||
.column_by_name(ROW_ID)
|
||||
.ok_or_else(|| missing(ROW_ID))?;
|
||||
|
||||
// Only an unfilled live row gains a value; a deleted row has a null
|
||||
// row id and keeps its (null) slot.
|
||||
let unfilled = arrow::compute::is_null(existing.as_ref())?;
|
||||
let live = arrow::compute::is_not_null(row_ids.as_ref())?;
|
||||
let fill = arrow::compute::and(&unfilled, &live)?;
|
||||
let keep = arrow::compute::not(&fill)?;
|
||||
|
||||
let computed = evaluate(&bound, &evaluation_batch(&batch, &bound, Some(&keep))?)?;
|
||||
let merged = arrow_select::zip::zip(&fill, &computed, existing)?;
|
||||
Ok(RecordBatch::try_new(projected.clone(), vec![merged])?)
|
||||
}))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use std::sync::Arc;
|
||||
|
||||
use arrow_array::{Int32Array, record_batch};
|
||||
use futures::TryStreamExt;
|
||||
|
||||
use crate::connect;
|
||||
use crate::query::{ExecutableQuery, QueryBase, Select};
|
||||
use crate::{Error, Result, Table};
|
||||
|
||||
async fn table_with(name: &str, values: Vec<i32>) -> Table {
|
||||
let conn = connect("memory://").execute().await.unwrap();
|
||||
let batch = record_batch!(("x", Int32, values)).unwrap();
|
||||
conn.create_table(name, batch).execute().await.unwrap()
|
||||
}
|
||||
|
||||
async fn declare_doubled(table: &Table) -> Result<u64> {
|
||||
Ok(table
|
||||
.add_columns()
|
||||
.computed("doubled", "x * 2")
|
||||
.execute()
|
||||
.await?
|
||||
.version)
|
||||
}
|
||||
|
||||
async fn read(table: &Table, column: &str) -> Vec<Option<i32>> {
|
||||
let batches = table
|
||||
.query()
|
||||
.select(Select::columns(&[column]))
|
||||
.execute()
|
||||
.await
|
||||
.unwrap()
|
||||
.try_collect::<Vec<_>>()
|
||||
.await
|
||||
.unwrap();
|
||||
let mut values: Vec<Option<i32>> = batches
|
||||
.iter()
|
||||
.flat_map(|batch| {
|
||||
batch[column]
|
||||
.as_any()
|
||||
.downcast_ref::<Int32Array>()
|
||||
.unwrap()
|
||||
.iter()
|
||||
.collect::<Vec<_>>()
|
||||
})
|
||||
.collect();
|
||||
values.sort();
|
||||
values
|
||||
}
|
||||
|
||||
async fn append(table: &Table, values: Vec<i32>) {
|
||||
let batch = record_batch!(("x", Int32, values)).unwrap();
|
||||
table.add(batch).execute().await.unwrap();
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_fills_a_declared_column() {
|
||||
let table = table_with("refresh_fills", vec![1, 2, 3]).await;
|
||||
let declared = declare_doubled(&table).await.unwrap();
|
||||
assert_eq!(read(&table, "doubled").await, vec![None, None, None]);
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert!(result.version > declared);
|
||||
assert_eq!(result.rows_filled, 3);
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(4), Some(6)]
|
||||
);
|
||||
}
|
||||
|
||||
/// Values written after the last refresh must be reachable by another one.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_fills_rows_appended_since_the_last_refresh() {
|
||||
let table = table_with("refresh_appended", vec![1, 2]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
|
||||
append(&table, vec![5, 6]).await;
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![None, None, Some(2), Some(4)]
|
||||
);
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 2);
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(4), Some(10), Some(12)]
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_with_nothing_to_fill() {
|
||||
let table = table_with("refresh_noop", vec![1, 2, 3]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
|
||||
let again = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(again.rows_filled, 0);
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(4), Some(6)]
|
||||
);
|
||||
}
|
||||
|
||||
/// A row is filled only by gaining a value, so an expression yielding null
|
||||
/// settles at once instead of re-selecting the same rows forever. Nothing
|
||||
/// is staged, so the version does not move either.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_converges_on_a_null_result() {
|
||||
let table = table_with("refresh_null_result", vec![1, 2, 3]).await;
|
||||
let declared = table
|
||||
.add_columns()
|
||||
.computed("maybe", "nullif(x, x)")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap()
|
||||
.version;
|
||||
|
||||
let first = table.refresh_column("maybe").await.unwrap();
|
||||
assert_eq!(first.rows_filled, 0);
|
||||
assert_eq!(first.version, declared);
|
||||
assert_eq!(read(&table, "maybe").await, vec![None, None, None]);
|
||||
|
||||
let again = table.refresh_column("maybe").await.unwrap();
|
||||
assert_eq!(again.rows_filled, 0);
|
||||
assert_eq!(again.version, declared);
|
||||
}
|
||||
|
||||
/// The contract's boundary: a filled fragment is not revisited, so
|
||||
/// mutating an input leaves the value computed at fill time.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_does_not_observe_input_mutation() {
|
||||
let table = table_with("refresh_mutation", vec![1]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(read(&table, "doubled").await, vec![Some(2)]);
|
||||
|
||||
table.update().column("x", "3").execute().await.unwrap();
|
||||
|
||||
let again = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(again.rows_filled, 0);
|
||||
assert_eq!(read(&table, "doubled").await, vec![Some(2)]);
|
||||
}
|
||||
|
||||
/// A row rewrite before the first refresh materializes the declared
|
||||
/// column as null behind a covering data file. Those rows are still
|
||||
/// unfilled and a later refresh has to reach them.
|
||||
#[tokio::test]
|
||||
async fn test_update_before_the_first_refresh() {
|
||||
let table = table_with("refresh_update_first", vec![1]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
|
||||
table.update().column("x", "3").execute().await.unwrap();
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
assert_eq!(read(&table, "doubled").await, vec![Some(6)]);
|
||||
}
|
||||
|
||||
/// The contract holds row by row, not fragment by fragment: revisiting a
|
||||
/// fragment to fill one row must not recompute a filled row sitting beside
|
||||
/// it, even where the input behind it has since changed.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_does_not_recompute_a_filled_row_beside_an_unfilled_one() {
|
||||
let table = table_with("refresh_mixed", vec![1, 2]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
|
||||
append(&table, vec![5]).await;
|
||||
table
|
||||
.update()
|
||||
.column("x", "100")
|
||||
.only_if("x = 1")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
table
|
||||
.optimize(crate::table::OptimizeAction::Compact {
|
||||
options: crate::table::CompactionOptions::default(),
|
||||
remap_options: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
// 2 is the mutated row keeping the value it was filled with, not 200.
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(4), Some(10)]
|
||||
);
|
||||
}
|
||||
|
||||
/// Filling a fragment must not disturb the values it already holds, which
|
||||
/// is what makes a compaction-mixed fragment safe to revisit.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_preserves_already_filled_rows() {
|
||||
let table = table_with("refresh_preserves", vec![1, 2]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
|
||||
append(&table, vec![5]).await;
|
||||
table
|
||||
.optimize(crate::table::OptimizeAction::Compact {
|
||||
options: crate::table::CompactionOptions::default(),
|
||||
remap_options: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(4), Some(10)]
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_leaves_deleted_rows_alone() {
|
||||
let table = table_with("refresh_deleted", vec![1, 2, 3, 4]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.delete("x = 2").await.unwrap();
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 3);
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(6), Some(8)]
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_a_constant_expression() {
|
||||
let table = table_with("refresh_constant", vec![1, 2, 3]).await;
|
||||
table
|
||||
.add_columns()
|
||||
.computed("answer", "42")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let result = table.refresh_column("answer").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 3);
|
||||
}
|
||||
|
||||
/// A name needing quotes reaches the evaluator intact: it is carried as a
|
||||
/// projection alias, never spliced into SQL text.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_a_column_whose_name_needs_quoting() {
|
||||
let table = table_with("refresh_quoted", vec![1, 2, 3]).await;
|
||||
table
|
||||
.add_columns()
|
||||
.computed("double value", "x * 2")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let result = table.refresh_column("double value").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 3);
|
||||
assert_eq!(
|
||||
read(&table, "double value").await,
|
||||
vec![Some(2), Some(4), Some(6)]
|
||||
);
|
||||
}
|
||||
|
||||
/// A fragment spanning several scan batches exercises the streamed fill:
|
||||
/// the probe buffers only until the first gained value and the rest flows
|
||||
/// through write_column a batch at a time.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_streams_a_multi_batch_fragment() {
|
||||
let values: Vec<i32> = (0..20_000).collect();
|
||||
let table = table_with("refresh_multi_batch", values.clone()).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 20_000);
|
||||
|
||||
let read_back = read(&table, "doubled").await;
|
||||
assert_eq!(read_back.len(), 20_000);
|
||||
let mut expected: Vec<Option<i32>> = values.iter().map(|v| Some(v * 2)).collect();
|
||||
expected.sort();
|
||||
assert_eq!(read_back, expected);
|
||||
}
|
||||
|
||||
/// The gate's reproducer: the commit must reuse the configured session,
|
||||
/// or registrations and caches vanish from the handle after a refresh.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_preserves_the_configured_session() {
|
||||
let session = Arc::new(lance::session::Session::default());
|
||||
let conn = crate::connect("memory://")
|
||||
.session(session.clone())
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
let batch = record_batch!(("x", Int32, [1, 2])).unwrap();
|
||||
let table = conn
|
||||
.create_table("session_kept", batch)
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
|
||||
let dataset = table.as_native().unwrap().dataset.get().await.unwrap();
|
||||
assert!(Arc::ptr_eq(&dataset.session(), &session));
|
||||
}
|
||||
|
||||
/// The async form's job settles with the fill visible, like
|
||||
/// create_index's execute_async.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_async_job_waits_for_the_fill() {
|
||||
let table = table_with("refresh_async", vec![1, 2, 3]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
assert!(job.id().is_none(), "in-process jobs have no server id");
|
||||
job.wait().await.unwrap();
|
||||
assert_eq!(job.status().await.unwrap(), "finished");
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(4), Some(6)]
|
||||
);
|
||||
}
|
||||
|
||||
/// Bad input is reported by the call, not by the job.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_async_rejects_bad_input_before_spawning() {
|
||||
let table = table_with("refresh_async_bad", vec![1, 2, 3]).await;
|
||||
|
||||
let err = table.refresh_column_async("x").await.unwrap_err();
|
||||
assert!(matches!(err, Error::NotAComputedColumn { name } if name == "x"));
|
||||
|
||||
let err = table.refresh_column_async("nope").await.unwrap_err();
|
||||
assert!(matches!(err, Error::ColumnNotFound { name } if name == "nope"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_async_job_reports_success_to_every_waiter() {
|
||||
let table = table_with("refresh_async_waiters", vec![1, 2]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
job.wait().await.unwrap();
|
||||
// A second wait after completion observes the same outcome.
|
||||
job.wait().await.unwrap();
|
||||
assert_eq!(job.status().await.unwrap(), "finished");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_rejects_a_plain_column() {
|
||||
let table = table_with("refresh_plain", vec![1, 2, 3]).await;
|
||||
let err = table.refresh_column("x").await.unwrap_err();
|
||||
assert!(matches!(err, Error::NotAComputedColumn { name } if name == "x"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_rejects_an_unknown_column() {
|
||||
let table = table_with("refresh_missing", vec![1, 2, 3]).await;
|
||||
let err = table.refresh_column("nope").await.unwrap_err();
|
||||
assert!(matches!(err, Error::ColumnNotFound { name } if name == "nope"));
|
||||
}
|
||||
|
||||
/// The gate's reproducer: a poison value in a deleted row must not
|
||||
/// abort filling the live rows, since nobody can read it.
|
||||
#[tokio::test]
|
||||
async fn test_a_deleted_rows_value_is_never_evaluated() {
|
||||
let table = table_with("refresh_deleted_poison", vec![1, 0]).await;
|
||||
table
|
||||
.add_columns()
|
||||
.computed("quotient", "10 / x")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
table.delete("x = 0").await.unwrap();
|
||||
|
||||
let result = table.refresh_column("quotient").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
assert_eq!(read(&table, "quotient").await, vec![Some(10)]);
|
||||
}
|
||||
|
||||
/// The gate's reproducer: an already-filled row's value must not be
|
||||
/// re-evaluated either -- its input may have mutated into one the
|
||||
/// expression chokes on.
|
||||
#[tokio::test]
|
||||
async fn test_a_filled_rows_value_is_never_evaluated() {
|
||||
let table = table_with("refresh_filled_poison", vec![1, 2]).await;
|
||||
table
|
||||
.add_columns()
|
||||
.computed("quotient", "10 / x")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
table.refresh_column("quotient").await.unwrap();
|
||||
|
||||
table
|
||||
.update()
|
||||
.column("x", "0")
|
||||
.only_if("x = 1")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
append(&table, vec![5]).await;
|
||||
|
||||
let result = table.refresh_column("quotient").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
assert_eq!(
|
||||
read(&table, "quotient").await,
|
||||
vec![Some(2), Some(5), Some(10)]
|
||||
);
|
||||
}
|
||||
|
||||
/// The gate's reproducer: the old internal projection alias is an
|
||||
/// ordinary column name; a computed column may use it.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_a_column_named_like_the_old_alias() {
|
||||
let table = table_with("refresh_alias_name", vec![1, 2]).await;
|
||||
table
|
||||
.add_columns()
|
||||
.computed("__lancedb_computed", "x * 2")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let result = table.refresh_column("__lancedb_computed").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 2);
|
||||
assert_eq!(
|
||||
read(&table, "__lancedb_computed").await,
|
||||
vec![Some(2), Some(4)]
|
||||
);
|
||||
}
|
||||
|
||||
/// The gate's reproducer: a late-gain fragment (filled, then one null row
|
||||
/// compacted onto the end) fills without the old probe's buffering, which
|
||||
/// this pins behaviorally; the memory bound is structural -- the fill
|
||||
/// stream retains no batches at all.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_fills_a_late_gain_fragment() {
|
||||
let values: Vec<i32> = (0..20_000).collect();
|
||||
let table = table_with("refresh_late_gain", values).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
|
||||
append(&table, vec![2_000_000]).await;
|
||||
table
|
||||
.optimize(crate::table::OptimizeAction::Compact {
|
||||
options: crate::table::CompactionOptions::default(),
|
||||
remap_options: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
let read_back = read(&table, "doubled").await;
|
||||
assert_eq!(read_back.len(), 20_001);
|
||||
assert_eq!(read_back.last().unwrap(), &Some(4_000_000));
|
||||
}
|
||||
|
||||
/// The gate's reproducer: a nested input declares, refreshes, and guards
|
||||
/// its root against invalidating schema changes.
|
||||
#[tokio::test]
|
||||
async fn test_a_nested_input_declares_and_refreshes() {
|
||||
use arrow_array::{Int32Array, StructArray};
|
||||
use arrow_schema::{DataType, Field, Fields};
|
||||
|
||||
let conn = connect("memory://").execute().await.unwrap();
|
||||
let age = Arc::new(Int32Array::from(vec![30, 40]));
|
||||
let fields = Fields::from(vec![Field::new("age", DataType::Int32, true)]);
|
||||
let metadata = StructArray::new(fields.clone(), vec![age as _], None);
|
||||
let schema = Arc::new(arrow_schema::Schema::new(vec![Field::new(
|
||||
"metadata",
|
||||
DataType::Struct(fields),
|
||||
true,
|
||||
)]));
|
||||
let batch =
|
||||
arrow_array::RecordBatch::try_new(schema, vec![Arc::new(metadata) as _]).unwrap();
|
||||
let table = conn
|
||||
.create_table("refresh_nested", batch)
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
table
|
||||
.add_columns()
|
||||
.computed("next_age", "metadata.age + 1")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
let declaration =
|
||||
&crate::table::computed_columns(table.schema().await.unwrap().as_ref())[0];
|
||||
assert_eq!(declaration.inputs, vec!["metadata.age".to_string()]);
|
||||
|
||||
let result = table.refresh_column("next_age").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 2);
|
||||
assert_eq!(read(&table, "next_age").await, vec![Some(31), Some(41)]);
|
||||
|
||||
// The dotted input guards its root.
|
||||
let err = table.drop_columns(&["metadata"]).await.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::InvalidInput { message } if message.contains("next_age")),
|
||||
"{err:?}"
|
||||
);
|
||||
|
||||
// Masking a struct input for a deleted row goes through the same
|
||||
// nullif path as a primitive; a nested input plus deletions must not
|
||||
// be the combination that breaks it.
|
||||
table.delete("next_age = 31").await.unwrap();
|
||||
append_struct_row(&table, 50).await;
|
||||
let result = table.refresh_column("next_age").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
assert_eq!(read(&table, "next_age").await, vec![Some(41), Some(51)]);
|
||||
}
|
||||
|
||||
/// Append one `metadata: {age}` row to the nested-input table.
|
||||
async fn append_struct_row(table: &Table, age: i32) {
|
||||
use arrow_array::{Int32Array, StructArray};
|
||||
use arrow_schema::{DataType, Field, Fields};
|
||||
|
||||
let ages = Arc::new(Int32Array::from(vec![age]));
|
||||
let fields = Fields::from(vec![Field::new("age", DataType::Int32, true)]);
|
||||
let metadata = StructArray::new(fields.clone(), vec![ages as _], None);
|
||||
let schema = Arc::new(arrow_schema::Schema::new(vec![Field::new(
|
||||
"metadata",
|
||||
DataType::Struct(fields),
|
||||
true,
|
||||
)]));
|
||||
let batch =
|
||||
arrow_array::RecordBatch::try_new(schema, vec![Arc::new(metadata) as _]).unwrap();
|
||||
table.add(batch).execute().await.unwrap();
|
||||
}
|
||||
|
||||
/// Both orders of declare+spec are refused at the source (see the
|
||||
/// schema_evolution tests); refresh's own check covers a dataset another
|
||||
/// writer left in that state.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_refuses_a_foreign_lsm_state() {
|
||||
use crate::table::LsmWriteSpec;
|
||||
|
||||
let tmp_dir = tempfile::tempdir().unwrap();
|
||||
let conn = connect(tmp_dir.path().to_str().unwrap())
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
let schema = Arc::new(arrow_schema::Schema::new(vec![arrow_schema::Field::new(
|
||||
"x",
|
||||
arrow_schema::DataType::Int32,
|
||||
false,
|
||||
)]));
|
||||
let batch =
|
||||
arrow_array::RecordBatch::try_new(schema, vec![Arc::new(Int32Array::from(vec![1]))])
|
||||
.unwrap();
|
||||
let table = conn.create_table("lsm", batch).execute().await.unwrap();
|
||||
table.set_unenforced_primary_key(["x"]).await.unwrap();
|
||||
table
|
||||
.set_lsm_write_spec(LsmWriteSpec::unsharded())
|
||||
.await
|
||||
.unwrap();
|
||||
super::super::computed_columns::add_foreign_kind(&table, "doubled", "sql").await;
|
||||
|
||||
let err = table.refresh_column("doubled").await.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::NotSupported { message } if message.contains("LSM")),
|
||||
"{err:?}"
|
||||
);
|
||||
let err = table.refresh_column_async("doubled").await.unwrap_err();
|
||||
assert!(matches!(err, Error::NotSupported { .. }));
|
||||
}
|
||||
|
||||
/// After catch-up activation and unset, no spec remains but the catch-up
|
||||
/// flag still marks retained SSTable rows; refresh refuses on the flag.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_refuses_retained_catchup_state() {
|
||||
use crate::table::LsmWriteSpec;
|
||||
|
||||
let tmp_dir = tempfile::tempdir().unwrap();
|
||||
let conn = connect(tmp_dir.path().to_str().unwrap())
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
let schema = Arc::new(arrow_schema::Schema::new(vec![arrow_schema::Field::new(
|
||||
"x",
|
||||
arrow_schema::DataType::Int32,
|
||||
false,
|
||||
)]));
|
||||
let batch = arrow_array::RecordBatch::try_new(
|
||||
schema.clone(),
|
||||
vec![Arc::new(Int32Array::from(vec![1]))],
|
||||
)
|
||||
.unwrap();
|
||||
let table = conn
|
||||
.create_table("catchup", batch.clone())
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
table.set_unenforced_primary_key(["x"]).await.unwrap();
|
||||
table
|
||||
.set_lsm_write_spec(LsmWriteSpec::unsharded())
|
||||
.await
|
||||
.unwrap();
|
||||
table.require_mem_wal_index_catchup().await.unwrap();
|
||||
let mut merge = table.merge_insert(&["x"]);
|
||||
merge
|
||||
.when_matched_update_all(None)
|
||||
.when_not_matched_insert_all()
|
||||
.use_lsm(true);
|
||||
merge
|
||||
.execute(Box::new(arrow_array::RecordBatchIterator::new(
|
||||
vec![Ok(batch)],
|
||||
schema,
|
||||
)))
|
||||
.await
|
||||
.unwrap();
|
||||
table.unset_lsm_write_spec().await.unwrap();
|
||||
super::super::computed_columns::add_foreign_kind(&table, "doubled", "sql").await;
|
||||
|
||||
let err = table.refresh_column("doubled").await.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::NotSupported { message } if message.contains("LSM")),
|
||||
"{err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
/// A declaration of a kind this version cannot evaluate is refused by
|
||||
/// name, rather than mistaken for a plain column or fed to the SQL path.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_rejects_a_kind_it_cannot_evaluate() {
|
||||
let table = table_with("refresh_foreign", vec![1, 2, 3]).await;
|
||||
super::super::computed_columns::add_foreign_kind(&table, "embedding", "udf").await;
|
||||
|
||||
let err = table.refresh_column("embedding").await.unwrap_err();
|
||||
assert!(matches!(err, Error::NotSupported { message } if message.contains("udf")));
|
||||
}
|
||||
}
|
||||
@@ -8,12 +8,14 @@
|
||||
//! - [`alter_columns`](execute_alter_columns): Rename columns, change types, or modify nullability
|
||||
//! - [`drop_columns`](execute_drop_columns): Remove columns from the table
|
||||
|
||||
use arrow_schema::Schema as ArrowSchema;
|
||||
use lance::dataset::{ColumnAlteration, NewColumnTransform};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::collections::HashMap;
|
||||
|
||||
use super::NativeTable;
|
||||
use crate::Result;
|
||||
use super::computed_columns;
|
||||
use super::{BaseTable, NativeTable};
|
||||
use crate::{Error, Result};
|
||||
|
||||
/// The result of an add columns operation.
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
|
||||
@@ -98,6 +100,48 @@ pub(crate) async fn execute_add_columns(
|
||||
table: &NativeTable,
|
||||
transforms: NewColumnTransform,
|
||||
read_columns: Option<Vec<String>>,
|
||||
) -> Result<AddColumnsResult> {
|
||||
// Declarations are admitted only through [`execute_declare`].
|
||||
match &transforms {
|
||||
NewColumnTransform::AllNulls(schema) => {
|
||||
computed_columns::ensure_no_foreign_declarations(schema.fields())?
|
||||
}
|
||||
NewColumnTransform::BatchUDF(udf) => {
|
||||
computed_columns::ensure_no_foreign_declarations(udf.output_schema.fields())?
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
commit_add_columns(table, transforms, read_columns).await
|
||||
}
|
||||
|
||||
/// Declare validated computed columns. The only admission path for
|
||||
/// declaration metadata.
|
||||
pub(crate) async fn execute_declare(
|
||||
table: &NativeTable,
|
||||
columns: &[(String, String)],
|
||||
) -> Result<AddColumnsResult> {
|
||||
// An LSM write spec keeps visible rows in tiers refresh cannot reach;
|
||||
// checked against latest committed state, not this handle's snapshot.
|
||||
// The catch-up flag outlives unset and marks retained SSTable rows.
|
||||
table.checkout_latest().await?;
|
||||
let catchup = table.dataset.get().await?.manifest().reader_feature_flags
|
||||
& lance_table::feature_flags::FLAG_MEM_WAL_INDEX_CATCHUP
|
||||
!= 0;
|
||||
if catchup || table.get_lsm_write_spec().await?.is_some() {
|
||||
return Err(Error::NotSupported {
|
||||
message: "computed columns are not supported on a table with an LSM write \
|
||||
spec: rows in un-compacted tiers are invisible to refresh"
|
||||
.into(),
|
||||
});
|
||||
}
|
||||
let transform = computed_columns::declare(table.schema().await?, columns)?;
|
||||
commit_add_columns(table, transform, None).await
|
||||
}
|
||||
|
||||
pub(crate) async fn commit_add_columns(
|
||||
table: &NativeTable,
|
||||
transforms: NewColumnTransform,
|
||||
read_columns: Option<Vec<String>>,
|
||||
) -> Result<AddColumnsResult> {
|
||||
table.dataset.ensure_mutable()?;
|
||||
let mut dataset = (*table.dataset.get().await?).clone();
|
||||
@@ -116,6 +160,21 @@ pub(crate) async fn execute_alter_columns(
|
||||
) -> Result<AlterColumnsResult> {
|
||||
table.dataset.ensure_mutable()?;
|
||||
let mut dataset = (*table.dataset.get().await?).clone();
|
||||
// Nullability is not part of what an expression resolves against, so only
|
||||
// a rename or a retype can invalidate a binding.
|
||||
let schema = std::sync::Arc::new(ArrowSchema::from(dataset.schema()));
|
||||
let rebinding = alterations
|
||||
.iter()
|
||||
.filter(|alteration| alteration.rename.is_some() || alteration.data_type.is_some())
|
||||
.map(|alteration| alteration.path.as_str())
|
||||
.collect::<Vec<_>>();
|
||||
computed_columns::ensure_not_an_input(&schema, &rebinding)?;
|
||||
let retyped = alterations
|
||||
.iter()
|
||||
.filter(|alteration| alteration.data_type.is_some())
|
||||
.map(|alteration| alteration.path.as_str())
|
||||
.collect::<Vec<_>>();
|
||||
computed_columns::ensure_not_retyped(schema.as_ref(), &retyped)?;
|
||||
dataset.alter_columns(alterations).await?;
|
||||
let version = dataset.version().version;
|
||||
table.dataset.update(dataset);
|
||||
@@ -131,6 +190,10 @@ pub(crate) async fn execute_drop_columns(
|
||||
) -> Result<DropColumnsResult> {
|
||||
table.dataset.ensure_mutable()?;
|
||||
let mut dataset = (*table.dataset.get().await?).clone();
|
||||
computed_columns::ensure_not_an_input(
|
||||
&std::sync::Arc::new(ArrowSchema::from(dataset.schema())),
|
||||
columns,
|
||||
)?;
|
||||
dataset.drop_columns(columns).await?;
|
||||
let version = dataset.version().version;
|
||||
table.dataset.update(dataset);
|
||||
@@ -147,6 +210,44 @@ pub(crate) async fn execute_update_field_metadata(
|
||||
table.dataset.ensure_mutable()?;
|
||||
let mut dataset = (*table.dataset.get().await?).clone();
|
||||
|
||||
// A declaration is validated as a whole at declare time; editing its keys
|
||||
// here would bypass that, fabricate one on a plain column, or move a
|
||||
// binding out from under a refresh. A replace on a declared column would
|
||||
// silently erase it.
|
||||
let schema = ArrowSchema::from(dataset.schema());
|
||||
let declared: Vec<String> = computed_columns::computed_columns(&schema)
|
||||
.into_iter()
|
||||
.map(|declaration| declaration.name)
|
||||
.collect();
|
||||
for update in updates {
|
||||
if update
|
||||
.metadata
|
||||
.keys()
|
||||
.any(|key| computed_columns::is_declaration_key(key))
|
||||
{
|
||||
return Err(Error::InvalidInput {
|
||||
message: format!(
|
||||
"metadata keys of a computed-column declaration cannot be edited \
|
||||
(path '{}'); drop the column and declare it again",
|
||||
update.path
|
||||
),
|
||||
});
|
||||
}
|
||||
if update.replace
|
||||
&& declared
|
||||
.iter()
|
||||
.any(|name| name == computed_columns::root(&update.path))
|
||||
{
|
||||
return Err(Error::InvalidInput {
|
||||
message: format!(
|
||||
"replacing all metadata of computed column '{}' would erase its \
|
||||
declaration; drop the column and declare it again",
|
||||
update.path
|
||||
),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
let mut builder = dataset.update_field_metadata();
|
||||
for update in updates {
|
||||
let entries = update.metadata.iter().map(|(k, v)| (k.clone(), v.clone()));
|
||||
|
||||
@@ -82,6 +82,10 @@ pub(crate) async fn execute_update(
|
||||
|
||||
// 1. Snapshot the current dataset
|
||||
let dataset = table.dataset.get().await?;
|
||||
super::computed_columns::ensure_not_written(
|
||||
&arrow_schema::Schema::from(dataset.schema()),
|
||||
update.columns.iter().map(|(name, _)| name.as_str()),
|
||||
)?;
|
||||
|
||||
// 2. Initialize the Lance Core builder
|
||||
let mut builder = LanceUpdateBuilder::new(dataset);
|
||||
|
||||
@@ -0,0 +1,147 @@
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||
|
||||
use std::sync::Arc;
|
||||
|
||||
use arrow_array::{Int64Array, RecordBatch};
|
||||
use arrow_schema::{DataType, Field, Schema};
|
||||
use lance::dataset::{WriteMode, WriteParams};
|
||||
use lancedb::{Result, TableBase, connect, connect_namespace, table::WriteOptions};
|
||||
use tempfile::tempdir;
|
||||
use url::Url;
|
||||
|
||||
fn empty_schema() -> Arc<Schema> {
|
||||
Arc::new(Schema::new(vec![Field::new("id", DataType::Int64, false)]))
|
||||
}
|
||||
|
||||
fn file_uri(path: &std::path::Path) -> String {
|
||||
Url::from_file_path(path)
|
||||
.unwrap_or_else(|_| panic!("not an absolute path: {}", path.display()))
|
||||
.to_string()
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_add_bases_accepts_named_and_dataset_root_entries() -> Result<()> {
|
||||
let tmp = tempdir().unwrap();
|
||||
let db = connect(tmp.path().join("db").to_str().unwrap())
|
||||
.execute()
|
||||
.await?;
|
||||
let table = db.create_empty_table("t", empty_schema()).execute().await?;
|
||||
let media = tmp.path().join("media");
|
||||
let parent = tmp.path().join("parent");
|
||||
std::fs::create_dir_all(&media).unwrap();
|
||||
std::fs::create_dir_all(&parent).unwrap();
|
||||
|
||||
table
|
||||
.add_bases([
|
||||
TableBase {
|
||||
path: file_uri(&media),
|
||||
name: Some("media".into()),
|
||||
is_dataset_root: false,
|
||||
},
|
||||
TableBase {
|
||||
path: file_uri(&parent),
|
||||
name: Some("parent".into()),
|
||||
is_dataset_root: true,
|
||||
},
|
||||
])
|
||||
.await
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_add_bases_accepts_two_unnamed_paths() -> Result<()> {
|
||||
let tmp = tempdir().unwrap();
|
||||
let db = connect(tmp.path().join("db").to_str().unwrap())
|
||||
.execute()
|
||||
.await?;
|
||||
let table = db.create_empty_table("t", empty_schema()).execute().await?;
|
||||
let media = tmp.path().join("media");
|
||||
let other = tmp.path().join("other");
|
||||
std::fs::create_dir_all(&media).unwrap();
|
||||
std::fs::create_dir_all(&other).unwrap();
|
||||
table
|
||||
.add_bases([&file_uri(&media), &file_uri(&other)])
|
||||
.await
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_add_bases_write_and_read_through_registered_base() -> Result<()> {
|
||||
let tmp = tempdir().unwrap();
|
||||
let db = connect(tmp.path().join("db").to_str().unwrap())
|
||||
.execute()
|
||||
.await?;
|
||||
let table = db.create_empty_table("t", empty_schema()).execute().await?;
|
||||
let media = tmp.path().join("media");
|
||||
std::fs::create_dir_all(&media).unwrap();
|
||||
let media_uri = file_uri(&media);
|
||||
table.add_bases([&media_uri]).await?;
|
||||
|
||||
let batch = RecordBatch::try_new(
|
||||
empty_schema(),
|
||||
vec![Arc::new(Int64Array::from(vec![1, 2, 3]))],
|
||||
)
|
||||
.unwrap();
|
||||
table
|
||||
.add(batch)
|
||||
.write_options(WriteOptions {
|
||||
lance_write_params: Some(WriteParams {
|
||||
mode: WriteMode::Append,
|
||||
target_base_names_or_paths: Some(vec![media_uri.clone()]),
|
||||
..Default::default()
|
||||
}),
|
||||
})
|
||||
.execute()
|
||||
.await?;
|
||||
|
||||
assert_eq!(table.count_rows(None).await?, 3);
|
||||
|
||||
let dataset = table.dataset().unwrap().get().await?;
|
||||
let registered = dataset
|
||||
.manifest()
|
||||
.base_paths
|
||||
.values()
|
||||
.find(|base| base.path == media_uri)
|
||||
.expect("registered base");
|
||||
assert_ne!(registered.id, 0);
|
||||
assert!(registered.name.is_none());
|
||||
assert!(
|
||||
dataset.get_fragments().iter().any(|fragment| {
|
||||
fragment
|
||||
.metadata()
|
||||
.files
|
||||
.iter()
|
||||
.any(|file| file.base_id == Some(registered.id))
|
||||
}),
|
||||
"written fragment should reference the registered base"
|
||||
);
|
||||
assert!(
|
||||
std::fs::read_dir(&media)
|
||||
.unwrap()
|
||||
.filter_map(|entry| entry.ok())
|
||||
.any(|entry| entry.path().extension().is_some_and(|ext| ext == "lance")),
|
||||
"data file should land under the registered base"
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_memory_add_bases_accepts_a_file_uri() -> Result<()> {
|
||||
let tmp = tempdir().unwrap();
|
||||
let db = connect("memory://").execute().await?;
|
||||
let table = db.create_empty_table("t", empty_schema()).execute().await?;
|
||||
let media = tmp.path().join("media");
|
||||
std::fs::create_dir_all(&media).unwrap();
|
||||
table.add_bases([file_uri(&media)]).await
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_namespace_add_bases_accepts_a_file_uri() -> Result<()> {
|
||||
let tmp = tempdir().unwrap();
|
||||
let mut properties = std::collections::HashMap::new();
|
||||
properties.insert("root".to_string(), tmp.path().to_str().unwrap().to_string());
|
||||
let db = connect_namespace("dir", properties).execute().await?;
|
||||
let table = db.create_empty_table("t", empty_schema()).execute().await?;
|
||||
let media = tmp.path().join("media");
|
||||
std::fs::create_dir_all(&media).unwrap();
|
||||
table.add_bases([file_uri(&media)]).await
|
||||
}
|
||||
Reference in New Issue
Block a user