mirror of
https://github.com/lancedb/lancedb.git
synced 2026-08-28 00:48:40 +00:00
Compare commits
12 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 9ba03810e7 | |||
| b4053059bf | |||
| 7a7b7a3941 | |||
| d52940cdab | |||
| fafc297675 | |||
| 82ebddbc10 | |||
| f114bba752 | |||
| 78024a30ce | |||
| 72500192e6 | |||
| 8165857a50 | |||
| 5fa98b9af8 | |||
| b525cbbe6a |
+1
-1
@@ -1,5 +1,5 @@
|
||||
[tool.bumpversion]
|
||||
current_version = "0.38.0-beta.0"
|
||||
current_version = "0.37.1-beta.1"
|
||||
parse = """(?x)
|
||||
(?P<major>0|[1-9]\\d*)\\.
|
||||
(?P<minor>0|[1-9]\\d*)\\.
|
||||
|
||||
@@ -4,14 +4,14 @@ on:
|
||||
workflow_call:
|
||||
inputs:
|
||||
tag:
|
||||
description: "Tag name from Lance (e.g. `v7.2.0-beta.1`). If omitted, the newest release is resolved automatically — stable releases are preferred over pre-releases — and the run is skipped if it is not newer than the version currently pinned in Cargo.toml."
|
||||
description: "Tag name from Lance. If omitted, the skill will use the latest Lance release that needs an update."
|
||||
required: false
|
||||
default: ""
|
||||
type: string
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
tag:
|
||||
description: "Tag name from Lance (e.g. `v7.2.0-beta.1`). Leave empty to resolve the newest release automatically — stable releases are preferred over pre-releases — and skip the run if it is not newer than the version currently pinned in Cargo.toml."
|
||||
description: "Tag name from Lance. Leave empty to use the latest Lance release that needs an update."
|
||||
required: false
|
||||
default: ""
|
||||
type: string
|
||||
|
||||
Generated
+48
-47
@@ -3455,8 +3455,8 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c"
|
||||
|
||||
[[package]]
|
||||
name = "fsst"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"rand 0.9.5",
|
||||
@@ -4815,8 +4815,8 @@ checksum = "e037a2e1d8d5fdbd49b16a4ea09d5d6401c1f29eca5ff29d03d3824dba16256a"
|
||||
|
||||
[[package]]
|
||||
name = "lance"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"arrow",
|
||||
@@ -4888,8 +4888,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-arrow"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -4911,7 +4911,7 @@ dependencies = [
|
||||
[[package]]
|
||||
name = "lance-arrow-scalar"
|
||||
version = "58.0.0"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -4925,7 +4925,7 @@ dependencies = [
|
||||
[[package]]
|
||||
name = "lance-arrow-stats"
|
||||
version = "58.0.0"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -4934,8 +4934,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-bitpacking"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrayref",
|
||||
"crunchy",
|
||||
@@ -4945,8 +4945,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-core"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -4983,8 +4983,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-datafusion"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5003,6 +5003,7 @@ dependencies = [
|
||||
"jsonb",
|
||||
"lance-arrow",
|
||||
"lance-core",
|
||||
"lance-datagen",
|
||||
"log",
|
||||
"pin-project",
|
||||
"prost",
|
||||
@@ -5013,8 +5014,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-datagen"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5031,8 +5032,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-derive"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -5041,8 +5042,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-encoding"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow-arith",
|
||||
"arrow-array",
|
||||
@@ -5075,8 +5076,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-file"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow-arith",
|
||||
"arrow-array",
|
||||
@@ -5107,8 +5108,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-index"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"arrow",
|
||||
@@ -5172,8 +5173,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-index-core"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -5195,8 +5196,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-io"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5232,8 +5233,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-linalg"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -5247,8 +5248,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-namespace"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"async-trait",
|
||||
@@ -5260,8 +5261,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-namespace-impls"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-ipc",
|
||||
@@ -5300,9 +5301,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-namespace-reqwest-client"
|
||||
version = "0.11.0"
|
||||
version = "0.8.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0a030196da1c994b63a96a4f0bf5b0cfa459fe6dadc9e962320246ca328da22a"
|
||||
checksum = "ba3f0a235e3ed5f8805205649ccc7d7d0f3df23ce1294242c9265ad488d7f19d"
|
||||
dependencies = [
|
||||
"reqwest 0.12.28",
|
||||
"serde",
|
||||
@@ -5314,8 +5315,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-select"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -5329,8 +5330,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-table"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5370,8 +5371,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-testing"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -5384,8 +5385,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-tokenizer"
|
||||
version = "11.0.0-beta.13"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059"
|
||||
version = "11.0.0-beta.8"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.8#9acbac748e8a7d616146dac8b06da42d8e7c6b62"
|
||||
dependencies = [
|
||||
"frostem",
|
||||
"icu_segmenter",
|
||||
@@ -5398,7 +5399,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lancedb"
|
||||
version = "0.38.0-beta.0"
|
||||
version = "0.37.1-beta.1"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"anyhow",
|
||||
@@ -5486,7 +5487,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lancedb-nodejs"
|
||||
version = "0.38.0-beta.0"
|
||||
version = "0.37.1-beta.1"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -5511,7 +5512,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lancedb-python"
|
||||
version = "0.38.0-beta.0"
|
||||
version = "0.37.1-beta.1"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"async-trait",
|
||||
|
||||
+14
-14
@@ -13,20 +13,20 @@ categories = ["database-implementations"]
|
||||
rust-version = "1.91.0"
|
||||
|
||||
[workspace.dependencies]
|
||||
lance = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-core = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datagen = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-file = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-io = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-index = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-linalg = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace-impls = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-table = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-testing = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datafusion = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-encoding = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-arrow = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance = { "version" = "=11.0.0-beta.8", default-features = false, "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-core = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datagen = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-file = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-io = { "version" = "=11.0.0-beta.8", default-features = false, "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-index = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-linalg = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace-impls = { "version" = "=11.0.0-beta.8", default-features = false, "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-table = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-testing = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datafusion = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-encoding = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-arrow = { "version" = "=11.0.0-beta.8", "tag" = "v11.0.0-beta.8", "git" = "https://github.com/lance-format/lance.git" }
|
||||
ahash = "0.8"
|
||||
# Note that this one does not include pyarrow
|
||||
arrow = { version = "58.0.0", optional = false }
|
||||
|
||||
@@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`:
|
||||
<dependency>
|
||||
<groupId>com.lancedb</groupId>
|
||||
<artifactId>lancedb-core</artifactId>
|
||||
<version>0.38.0-beta.0</version>
|
||||
<version>0.37.1-beta.1</version>
|
||||
</dependency>
|
||||
```
|
||||
|
||||
|
||||
@@ -386,29 +386,6 @@ Drop an existing table.
|
||||
|
||||
***
|
||||
|
||||
### dropTableAsync()
|
||||
|
||||
```ts
|
||||
abstract dropTableAsync(name, namespacePath?): Promise<Job>
|
||||
```
|
||||
|
||||
Start dropping a table and return its cleanup job.
|
||||
|
||||
The table may become unavailable before its data files are removed. Wait
|
||||
on the returned job to know when cleanup has finished.
|
||||
|
||||
#### Parameters
|
||||
|
||||
* **name**: `string`
|
||||
|
||||
* **namespacePath?**: `string`[]
|
||||
|
||||
#### Returns
|
||||
|
||||
`Promise`<[`Job`](Job.md)>
|
||||
|
||||
***
|
||||
|
||||
### getJob()
|
||||
|
||||
```ts
|
||||
|
||||
@@ -69,34 +69,14 @@ abstract addColumns(newColumnTransforms): Promise<AddColumnsResult>
|
||||
|
||||
Add new columns with defined values.
|
||||
|
||||
The `{ computed }` form stores the expression rather than evaluating it
|
||||
now: the column is committed with no values, and rows get them from
|
||||
[Table#refreshColumn](Table.md#refreshcolumn). Declaring one therefore costs the same on a
|
||||
large table as on an empty one.
|
||||
|
||||
A refresh does not revisit rows it has already filled, so mutating an
|
||||
input leaves the value computed at fill time; recomputing means dropping
|
||||
the column and declaring it again. While a declaration reads a column,
|
||||
that column cannot be renamed, retyped or dropped.
|
||||
|
||||
On LanceDB Cloud and Enterprise the expression is planned by the
|
||||
server, and the refresh runs as a server job -- see
|
||||
[Table#refreshColumnAsync](Table.md#refreshcolumnasync).
|
||||
|
||||
#### Parameters
|
||||
|
||||
* **newColumnTransforms**:
|
||||
\| `Field`<`any`>
|
||||
\| `Field`<`any`>[]
|
||||
\| `Schema`<`any`>
|
||||
\| [`AddColumnsSql`](../interfaces/AddColumnsSql.md)[]
|
||||
\| `object`
|
||||
* **newColumnTransforms**: `Field`<`any`> \| `Field`<`any`>[] \| `Schema`<`any`> \| [`AddColumnsSql`](../interfaces/AddColumnsSql.md)[]
|
||||
Either:
|
||||
- An array of objects with column names and SQL expressions to calculate values
|
||||
- A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
||||
- An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
||||
- An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
||||
- `{ computed }`, declaring columns defined by a SQL expression whose type and inputs are derived from it
|
||||
|
||||
#### Returns
|
||||
|
||||
@@ -105,13 +85,6 @@ server, and the refresh runs as a server job -- see
|
||||
A promise that resolves to an object
|
||||
containing the new version number of the table after adding the columns.
|
||||
|
||||
#### Example
|
||||
|
||||
```ts
|
||||
await table.addColumns({ computed: [{ name: "doubled", valueSql: "x * 2" }] });
|
||||
const { rowsFilled } = await table.refreshColumn("doubled");
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### alterColumns()
|
||||
@@ -745,67 +718,6 @@ for await (const batch of table.query()) {
|
||||
|
||||
***
|
||||
|
||||
### refreshColumn()
|
||||
|
||||
```ts
|
||||
abstract refreshColumn(column): Promise<RefreshColumnResult>
|
||||
```
|
||||
|
||||
Fill the rows of a computed column that hold no value yet.
|
||||
|
||||
Rows appended since the last refresh are filled by the next one; rows
|
||||
already filled are left as they are, so the call is idempotent and does
|
||||
not observe a mutated input. Local tables only: a remote refresh runs
|
||||
as a server job, through [Table#refreshColumnAsync](Table.md#refreshcolumnasync).
|
||||
|
||||
#### Parameters
|
||||
|
||||
* **column**: `string`
|
||||
The name of the computed column to fill.
|
||||
|
||||
#### Returns
|
||||
|
||||
`Promise`<[`RefreshColumnResult`](../interfaces/RefreshColumnResult.md)>
|
||||
|
||||
A promise that resolves to the
|
||||
number of rows filled and the new version number of the table.
|
||||
|
||||
***
|
||||
|
||||
### refreshColumnAsync()
|
||||
|
||||
```ts
|
||||
abstract refreshColumnAsync(column): Promise<Job>
|
||||
```
|
||||
|
||||
Like [Table#refreshColumn](Table.md#refreshcolumn), but returns a handle to the refresh
|
||||
job instead of blocking until it completes.
|
||||
|
||||
The job may already be complete when returned; callers must not assume
|
||||
the column is filled until [Job.wait](Job.md#wait) resolves. Invalid input --
|
||||
an unknown column, or one that is not computed -- rejects here rather
|
||||
than failing the job. On local tables the job runs in-process; on
|
||||
LanceDB Cloud and Enterprise it is the server's backfill job.
|
||||
|
||||
#### Parameters
|
||||
|
||||
* **column**: `string`
|
||||
The name of the computed column to fill.
|
||||
|
||||
#### Returns
|
||||
|
||||
`Promise`<[`Job`](Job.md)>
|
||||
|
||||
#### Example
|
||||
|
||||
```ts
|
||||
const job = await table.refreshColumnAsync("doubled");
|
||||
await job.wait();
|
||||
console.log(await job.status()); // "finished"
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### restore()
|
||||
|
||||
```ts
|
||||
|
||||
@@ -105,7 +105,6 @@
|
||||
- [OptimizeOptions](interfaces/OptimizeOptions.md)
|
||||
- [OptimizeStats](interfaces/OptimizeStats.md)
|
||||
- [QueryExecutionOptions](interfaces/QueryExecutionOptions.md)
|
||||
- [RefreshColumnResult](interfaces/RefreshColumnResult.md)
|
||||
- [RemovalStats](interfaces/RemovalStats.md)
|
||||
- [RenameTableOptions](interfaces/RenameTableOptions.md)
|
||||
- [RestNamespaceConfig](interfaces/RestNamespaceConfig.md)
|
||||
|
||||
@@ -1,23 +0,0 @@
|
||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||
|
||||
***
|
||||
|
||||
[@lancedb/lancedb](../globals.md) / RefreshColumnResult
|
||||
|
||||
# Interface: RefreshColumnResult
|
||||
|
||||
## Properties
|
||||
|
||||
### rowsFilled
|
||||
|
||||
```ts
|
||||
rowsFilled: number;
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### version
|
||||
|
||||
```ts
|
||||
version: number;
|
||||
```
|
||||
@@ -8,7 +8,7 @@
|
||||
<parent>
|
||||
<groupId>com.lancedb</groupId>
|
||||
<artifactId>lancedb-parent</artifactId>
|
||||
<version>0.38.0-beta.0</version>
|
||||
<version>0.37.1-beta.1</version>
|
||||
<relativePath>../pom.xml</relativePath>
|
||||
</parent>
|
||||
|
||||
|
||||
+2
-2
@@ -6,7 +6,7 @@
|
||||
|
||||
<groupId>com.lancedb</groupId>
|
||||
<artifactId>lancedb-parent</artifactId>
|
||||
<version>0.38.0-beta.0</version>
|
||||
<version>0.37.1-beta.1</version>
|
||||
<packaging>pom</packaging>
|
||||
<name>${project.artifactId}</name>
|
||||
<description>LanceDB Java SDK Parent POM</description>
|
||||
@@ -28,7 +28,7 @@
|
||||
<properties>
|
||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||
<arrow.version>15.0.0</arrow.version>
|
||||
<lance-core.version>11.0.0-beta.13</lance-core.version>
|
||||
<lance-core.version>11.0.0-beta.8</lance-core.version>
|
||||
<spotless.skip>false</spotless.skip>
|
||||
<spotless.version>2.30.0</spotless.version>
|
||||
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
[package]
|
||||
name = "lancedb-nodejs"
|
||||
edition.workspace = true
|
||||
version = "0.38.0-beta.0"
|
||||
version = "0.37.1-beta.1"
|
||||
publish = false
|
||||
license.workspace = true
|
||||
description.workspace = true
|
||||
|
||||
@@ -89,16 +89,6 @@ describe("given a connection", () => {
|
||||
await db.createTable("test4", [{ id: 1 }, { id: 2 }]);
|
||||
});
|
||||
|
||||
it("should return a completed job when dropping a local table", async () => {
|
||||
await db.createTable("async-drop", [{ id: 1 }]);
|
||||
|
||||
const job = await db.dropTableAsync("async-drop");
|
||||
expect(job.id).toBeNull();
|
||||
await expect(job.status()).resolves.toBe("finished");
|
||||
await job.wait();
|
||||
await expect(db.tableNames()).resolves.toEqual([]);
|
||||
});
|
||||
|
||||
it("should fail if creating table twice, unless overwrite is true", async () => {
|
||||
let tbl = await db.createTable("test", [{ id: 1 }, { id: 2 }]);
|
||||
await expect(tbl.countRows()).resolves.toBe(2);
|
||||
|
||||
@@ -1001,49 +1001,4 @@ describe("remote connection jobs surface", () => {
|
||||
},
|
||||
);
|
||||
});
|
||||
|
||||
it("addBases posts the bases array", async () => {
|
||||
const postedBodies: unknown[] = [];
|
||||
await withMockDatabase(
|
||||
(req, res) => {
|
||||
const path = req.url ?? "";
|
||||
if (path.endsWith("/describe/")) {
|
||||
res.writeHead(200, { "Content-Type": "application/json" }).end(
|
||||
JSON.stringify({
|
||||
name: "photos",
|
||||
version: 1,
|
||||
schema: { fields: [] },
|
||||
}),
|
||||
);
|
||||
return;
|
||||
}
|
||||
if (path.endsWith("/bases/")) {
|
||||
const chunks: Buffer[] = [];
|
||||
req.on("data", (chunk) => chunks.push(chunk));
|
||||
req.on("end", () => {
|
||||
postedBodies.push(JSON.parse(Buffer.concat(chunks).toString()));
|
||||
res
|
||||
.writeHead(200, { "Content-Type": "application/json" })
|
||||
.end(JSON.stringify({ version: 2 }));
|
||||
});
|
||||
return;
|
||||
}
|
||||
res.writeHead(404).end();
|
||||
},
|
||||
async (db) => {
|
||||
const table = await db.openTable("photos");
|
||||
await table.addBases({ path: "s3://bucket/media/" });
|
||||
},
|
||||
);
|
||||
expect(postedBodies).toEqual([
|
||||
{
|
||||
bases: [
|
||||
{
|
||||
path: "s3://bucket/media/",
|
||||
isDatasetRoot: false,
|
||||
},
|
||||
],
|
||||
},
|
||||
]);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
import * as fs from "fs";
|
||||
import * as path from "path";
|
||||
import * as tmp from "tmp";
|
||||
import { pathToFileURL } from "url";
|
||||
|
||||
import * as arrow15 from "apache-arrow-15";
|
||||
import * as arrow16 from "apache-arrow-16";
|
||||
@@ -3341,86 +3340,3 @@ describe("LSM merge insert", () => {
|
||||
await expect(table.query().useLsm(true).toArray()).rejects.toThrow();
|
||||
});
|
||||
});
|
||||
|
||||
describe("computed columns", () => {
|
||||
let tmpDir: tmp.DirResult;
|
||||
beforeEach(() => {
|
||||
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
||||
});
|
||||
afterEach(() => tmpDir.removeCallback());
|
||||
|
||||
it("declares a column and fills it on refresh", async () => {
|
||||
const db = await connect(tmpDir.name);
|
||||
const table = await db.createTable("computed", [{ x: 1 }, { x: 2 }]);
|
||||
|
||||
await table.addColumns({
|
||||
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
||||
});
|
||||
let rows = await table.query().toArray();
|
||||
expect(rows.map((r) => r.doubled)).toEqual([null, null]);
|
||||
|
||||
const result = await table.refreshColumn("doubled");
|
||||
expect(result.rowsFilled).toBe(2);
|
||||
|
||||
rows = await table.query().toArray();
|
||||
expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]);
|
||||
});
|
||||
|
||||
it("returns a job handle from refreshColumnAsync", async () => {
|
||||
const db = await connect(tmpDir.name);
|
||||
const table = await db.createTable("computed_job", [{ x: 1 }, { x: 2 }]);
|
||||
|
||||
await table.addColumns({
|
||||
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
||||
});
|
||||
|
||||
const job = await table.refreshColumnAsync("doubled");
|
||||
expect(job.id).toBeNull();
|
||||
await job.wait();
|
||||
expect(await job.status()).toBe("finished");
|
||||
|
||||
const rows = await table.query().toArray();
|
||||
expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]);
|
||||
|
||||
// Bad input rejects at the call, not through the job.
|
||||
await expect(table.refreshColumnAsync("x")).rejects.toThrow(
|
||||
"not a computed column",
|
||||
);
|
||||
});
|
||||
|
||||
it("fills rows added since the last refresh", async () => {
|
||||
const db = await connect(tmpDir.name);
|
||||
const table = await db.createTable("computed_append", [{ x: 1 }]);
|
||||
|
||||
await table.addColumns({
|
||||
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
||||
});
|
||||
await table.refreshColumn("doubled");
|
||||
await table.add([{ x: 5 }]);
|
||||
|
||||
const result = await table.refreshColumn("doubled");
|
||||
expect(result.rowsFilled).toBe(1);
|
||||
|
||||
const rows = await table.query().toArray();
|
||||
expect(rows.map((r) => r.doubled).sort()).toEqual([10, 2]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("table bases", () => {
|
||||
let tmpDir: tmp.DirResult;
|
||||
beforeEach(() => {
|
||||
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
||||
});
|
||||
afterEach(() => tmpDir.removeCallback());
|
||||
|
||||
it("addBases accepts a file uri", async () => {
|
||||
const conn = await connect(tmpDir.name);
|
||||
const table = await conn.createEmptyTable(
|
||||
"photos",
|
||||
new arrow.Schema([new arrow.Field("id", new arrow.Int64(), false)]),
|
||||
);
|
||||
const media = path.join(tmpDir.name, "media");
|
||||
fs.mkdirSync(media);
|
||||
await table.addBases(pathToFileURL(media).toString());
|
||||
});
|
||||
});
|
||||
|
||||
@@ -327,14 +327,6 @@ export abstract class Connection {
|
||||
*/
|
||||
abstract dropTable(name: string, namespacePath?: string[]): Promise<void>;
|
||||
|
||||
/**
|
||||
* Start dropping a table and return its cleanup job.
|
||||
*
|
||||
* The table may become unavailable before its data files are removed. Wait
|
||||
* on the returned job to know when cleanup has finished.
|
||||
*/
|
||||
abstract dropTableAsync(name: string, namespacePath?: string[]): Promise<Job>;
|
||||
|
||||
/**
|
||||
* Drop all tables in the database.
|
||||
* @param {string[]} namespacePath The namespace path to drop tables from (defaults to root namespace).
|
||||
@@ -713,10 +705,6 @@ export class LocalConnection extends Connection {
|
||||
return this.inner.dropTable(name, namespacePath ?? []);
|
||||
}
|
||||
|
||||
async dropTableAsync(name: string, namespacePath?: string[]): Promise<Job> {
|
||||
return this.inner.dropTableAsync(name, namespacePath ?? []);
|
||||
}
|
||||
|
||||
async dropAllTables(namespacePath?: string[]): Promise<void> {
|
||||
return this.inner.dropAllTables(namespacePath ?? []);
|
||||
}
|
||||
|
||||
@@ -50,7 +50,6 @@ export {
|
||||
MergeResult,
|
||||
AddResult,
|
||||
AddColumnsResult,
|
||||
RefreshColumnResult,
|
||||
AlterColumnsResult,
|
||||
UpdateFieldMetadataResult,
|
||||
DeleteResult,
|
||||
@@ -130,7 +129,6 @@ export {
|
||||
|
||||
export {
|
||||
Table,
|
||||
TableBase,
|
||||
Branches,
|
||||
BranchColumnSummary,
|
||||
BranchColumnChange,
|
||||
|
||||
+2
-131
@@ -33,7 +33,6 @@ import {
|
||||
Job,
|
||||
Branches as NativeBranches,
|
||||
OptimizeStats,
|
||||
RefreshColumnResult,
|
||||
TableStatistics,
|
||||
Tags,
|
||||
UpdateFieldMetadataResult,
|
||||
@@ -78,25 +77,6 @@ export interface WriteProgress {
|
||||
done: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* An extra storage prefix registered on a table.
|
||||
*
|
||||
* `path` is an object-store URI. `name` is an optional alias. `isDatasetRoot`
|
||||
* is true when `path` points to a Lance dataset root. When false, `path`
|
||||
* points directly to the directory containing the referenced files.
|
||||
*/
|
||||
export interface TableBase {
|
||||
/** Object store URI such as `s3://bucket/media/`. */
|
||||
path: string;
|
||||
/** Optional alias. */
|
||||
name?: string;
|
||||
/**
|
||||
* True when `path` is a Lance dataset root. When false, `path` is the
|
||||
* directory containing the referenced files.
|
||||
*/
|
||||
isDatasetRoot?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Options for adding data to a table.
|
||||
*/
|
||||
@@ -545,84 +525,18 @@ export abstract class Table {
|
||||
abstract vectorSearch(vector: IntoVector | MultiVector): VectorQuery;
|
||||
/**
|
||||
* Add new columns with defined values.
|
||||
*
|
||||
* The `{ computed }` form stores the expression rather than evaluating it
|
||||
* now: the column is committed with no values, and rows get them from
|
||||
* {@link Table#refreshColumn}. Declaring one therefore costs the same on a
|
||||
* large table as on an empty one.
|
||||
*
|
||||
* A refresh does not revisit rows it has already filled, so mutating an
|
||||
* input leaves the value computed at fill time; recomputing means dropping
|
||||
* the column and declaring it again. While a declaration reads a column,
|
||||
* that column cannot be renamed, retyped or dropped.
|
||||
*
|
||||
* On LanceDB Cloud and Enterprise the expression is planned by the
|
||||
* server, and the refresh runs as a server job -- see
|
||||
* {@link Table#refreshColumnAsync}.
|
||||
* @param {AddColumnsSql[] | Field | Field[] | Schema} newColumnTransforms Either:
|
||||
* - An array of objects with column names and SQL expressions to calculate values
|
||||
* - A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
||||
* - An array of Arrow Fields defining columns with their data types (columns will be initialized with null values)
|
||||
* - An Arrow Schema defining columns with their data types (columns will be initialized with null values)
|
||||
* - `{ computed }`, declaring columns defined by a SQL expression whose type and inputs are derived from it
|
||||
* @returns {Promise<AddColumnsResult>} A promise that resolves to an object
|
||||
* containing the new version number of the table after adding the columns.
|
||||
* @example
|
||||
* ```ts
|
||||
* await table.addColumns({ computed: [{ name: "doubled", valueSql: "x * 2" }] });
|
||||
* const { rowsFilled } = await table.refreshColumn("doubled");
|
||||
* ```
|
||||
*/
|
||||
abstract addColumns(
|
||||
newColumnTransforms:
|
||||
| AddColumnsSql[]
|
||||
| Field
|
||||
| Field[]
|
||||
| Schema
|
||||
| { computed: AddColumnsSql[] },
|
||||
newColumnTransforms: AddColumnsSql[] | Field | Field[] | Schema,
|
||||
): Promise<AddColumnsResult>;
|
||||
|
||||
/**
|
||||
* Register additional storage bases for this table.
|
||||
*
|
||||
* A URI string is a non-root base with no alias.
|
||||
*/
|
||||
abstract addBases(
|
||||
bases: string | TableBase | Array<string | TableBase>,
|
||||
): Promise<void>;
|
||||
|
||||
/**
|
||||
* Fill the rows of a computed column that hold no value yet.
|
||||
*
|
||||
* Rows appended since the last refresh are filled by the next one; rows
|
||||
* already filled are left as they are, so the call is idempotent and does
|
||||
* not observe a mutated input. Local tables only: a remote refresh runs
|
||||
* as a server job, through {@link Table#refreshColumnAsync}.
|
||||
* @param {string} column The name of the computed column to fill.
|
||||
* @returns {Promise<RefreshColumnResult>} A promise that resolves to the
|
||||
* number of rows filled and the new version number of the table.
|
||||
*/
|
||||
abstract refreshColumn(column: string): Promise<RefreshColumnResult>;
|
||||
|
||||
/**
|
||||
* Like {@link Table#refreshColumn}, but returns a handle to the refresh
|
||||
* job instead of blocking until it completes.
|
||||
*
|
||||
* The job may already be complete when returned; callers must not assume
|
||||
* the column is filled until {@link Job.wait} resolves. Invalid input --
|
||||
* an unknown column, or one that is not computed -- rejects here rather
|
||||
* than failing the job. On local tables the job runs in-process; on
|
||||
* LanceDB Cloud and Enterprise it is the server's backfill job.
|
||||
* @param {string} column The name of the computed column to fill.
|
||||
* @example
|
||||
* ```ts
|
||||
* const job = await table.refreshColumnAsync("doubled");
|
||||
* await job.wait();
|
||||
* console.log(await job.status()); // "finished"
|
||||
* ```
|
||||
*/
|
||||
abstract refreshColumnAsync(column: string): Promise<Job>;
|
||||
|
||||
/**
|
||||
* Alter the name or nullability of columns.
|
||||
* @param {ColumnAlteration[]} columnAlterations One or more alterations to
|
||||
@@ -1174,22 +1088,8 @@ export class LocalTable extends Table {
|
||||
// TODO: Support BatchUDF
|
||||
|
||||
async addColumns(
|
||||
newColumnTransforms:
|
||||
| AddColumnsSql[]
|
||||
| Field
|
||||
| Field[]
|
||||
| Schema
|
||||
| { computed: AddColumnsSql[] },
|
||||
newColumnTransforms: AddColumnsSql[] | Field | Field[] | Schema,
|
||||
): Promise<AddColumnsResult> {
|
||||
// Columns defined by an expression are declared, not materialized here.
|
||||
if (
|
||||
typeof newColumnTransforms === "object" &&
|
||||
!Array.isArray(newColumnTransforms) &&
|
||||
"computed" in newColumnTransforms
|
||||
) {
|
||||
return await this.inner.addComputedColumns(newColumnTransforms.computed);
|
||||
}
|
||||
|
||||
// Handle single Field -> convert to array of Fields
|
||||
if (newColumnTransforms instanceof Field) {
|
||||
newColumnTransforms = [newColumnTransforms];
|
||||
@@ -1224,20 +1124,6 @@ export class LocalTable extends Table {
|
||||
throw new Error("Invalid input type for addColumns");
|
||||
}
|
||||
|
||||
async addBases(
|
||||
bases: string | TableBase | Array<string | TableBase>,
|
||||
): Promise<void> {
|
||||
await this.inner.addBases(normalizeBases(bases));
|
||||
}
|
||||
|
||||
async refreshColumn(column: string): Promise<RefreshColumnResult> {
|
||||
return await this.inner.refreshColumn(column);
|
||||
}
|
||||
|
||||
async refreshColumnAsync(column: string): Promise<Job> {
|
||||
return await this.inner.refreshColumnAsync(column);
|
||||
}
|
||||
|
||||
async alterColumns(
|
||||
columnAlterations: ColumnAlteration[],
|
||||
): Promise<AlterColumnsResult> {
|
||||
@@ -1430,21 +1316,6 @@ export class LocalTable extends Table {
|
||||
}
|
||||
}
|
||||
|
||||
function normalizeBases(
|
||||
bases: string | TableBase | Array<string | TableBase>,
|
||||
): TableBase[] {
|
||||
const baseInputs = Array.isArray(bases) ? bases : [bases];
|
||||
return baseInputs.map((base) =>
|
||||
typeof base === "string"
|
||||
? { path: base, isDatasetRoot: false }
|
||||
: {
|
||||
path: base.path,
|
||||
name: base.name,
|
||||
isDatasetRoot: base.isDatasetRoot ?? false,
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* A definition of a column alteration. The alteration changes the column at
|
||||
* `path` to have the new name `name`, to be nullable if `nullable` is true,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-darwin-arm64",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.37.1-beta.1",
|
||||
"os": ["darwin"],
|
||||
"cpu": ["arm64"],
|
||||
"main": "lancedb.darwin-arm64.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.37.1-beta.1",
|
||||
"os": ["linux"],
|
||||
"cpu": ["arm64"],
|
||||
"main": "lancedb.linux-arm64-gnu.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-arm64-musl",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.37.1-beta.1",
|
||||
"os": ["linux"],
|
||||
"cpu": ["arm64"],
|
||||
"main": "lancedb.linux-arm64-musl.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-x64-gnu",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.37.1-beta.1",
|
||||
"os": ["linux"],
|
||||
"cpu": ["x64"],
|
||||
"main": "lancedb.linux-x64-gnu.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-x64-musl",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.37.1-beta.1",
|
||||
"os": ["linux"],
|
||||
"cpu": ["x64"],
|
||||
"main": "lancedb.linux-x64-musl.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-win32-arm64-msvc",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.37.1-beta.1",
|
||||
"os": [
|
||||
"win32"
|
||||
],
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-win32-x64-msvc",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.37.1-beta.1",
|
||||
"os": ["win32"],
|
||||
"cpu": ["x64"],
|
||||
"main": "lancedb.win32-x64-msvc.node",
|
||||
|
||||
Generated
+2
-2
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.37.1-beta.1",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "@lancedb/lancedb",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.37.1-beta.1",
|
||||
"cpu": [
|
||||
"x64",
|
||||
"arm64"
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@
|
||||
"ann"
|
||||
],
|
||||
"private": false,
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.37.1-beta.1",
|
||||
"main": "dist/index.js",
|
||||
"exports": {
|
||||
".": "./dist/index.js",
|
||||
|
||||
@@ -334,22 +334,6 @@ impl Connection {
|
||||
.default_error()
|
||||
}
|
||||
|
||||
/// Start dropping a table and return its cleanup job.
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn drop_table_async(
|
||||
&self,
|
||||
name: String,
|
||||
namespace_path: Option<Vec<String>>,
|
||||
) -> napi::Result<crate::job::Job> {
|
||||
let ns = namespace_path.unwrap_or_default();
|
||||
let job = self
|
||||
.get_inner()?
|
||||
.drop_table_async(&name, &ns)
|
||||
.await
|
||||
.default_error()?;
|
||||
Ok(crate::job::Job::new(job))
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn drop_all_tables(&self, namespace_path: Option<Vec<String>>) -> napi::Result<()> {
|
||||
let ns = namespace_path.unwrap_or_default();
|
||||
|
||||
@@ -10,7 +10,6 @@ use lancedb::table::{
|
||||
AddDataMode, ColumnAlteration as LanceColumnAlteration, Duration,
|
||||
FieldMetadataUpdate as LanceFieldMetadataUpdate, FtsToken as LanceDbFtsToken,
|
||||
NewColumnTransform, OptimizeAction, OptimizeOptions, Ref, Table as LanceDbTable,
|
||||
TableBase as LanceTableBase,
|
||||
};
|
||||
use napi::bindgen_prelude::*;
|
||||
use napi::threadsafe_function::{ThreadsafeFunction, ThreadsafeFunctionCallMode};
|
||||
@@ -348,40 +347,6 @@ impl Table {
|
||||
Ok(res.into())
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn add_computed_columns(
|
||||
&self,
|
||||
columns: Vec<AddColumnsSql>,
|
||||
) -> napi::Result<AddColumnsResult> {
|
||||
let table = self.inner_ref()?;
|
||||
let mut builder = table.add_columns();
|
||||
for column in columns {
|
||||
builder = builder.computed(column.name, column.value_sql);
|
||||
}
|
||||
let res = builder.execute().await.default_error()?;
|
||||
Ok(res.into())
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn refresh_column(&self, column: String) -> napi::Result<RefreshColumnResult> {
|
||||
let res = self
|
||||
.inner_ref()?
|
||||
.refresh_column(column)
|
||||
.await
|
||||
.default_error()?;
|
||||
Ok(res.into())
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn refresh_column_async(&self, column: String) -> napi::Result<crate::job::Job> {
|
||||
let job = self
|
||||
.inner_ref()?
|
||||
.refresh_column_async(column)
|
||||
.await
|
||||
.default_error()?;
|
||||
Ok(crate::job::Job::new(job))
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn add_columns_with_schema(
|
||||
&self,
|
||||
@@ -447,18 +412,6 @@ impl Table {
|
||||
Ok(res.into())
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn add_bases(&self, bases: Vec<TableBase>) -> napi::Result<()> {
|
||||
self.inner_ref()?
|
||||
.add_bases(bases.into_iter().map(|base| LanceTableBase {
|
||||
path: base.path,
|
||||
name: base.name,
|
||||
is_dataset_root: base.is_dataset_root,
|
||||
}))
|
||||
.await
|
||||
.default_error()
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn drop_columns(&self, columns: Vec<String>) -> napi::Result<DropColumnsResult> {
|
||||
let col_refs = columns.iter().map(String::as_str).collect::<Vec<_>>();
|
||||
@@ -713,18 +666,6 @@ impl Table {
|
||||
}
|
||||
}
|
||||
|
||||
#[napi(object)]
|
||||
/// An extra storage prefix registered on a table.
|
||||
pub struct TableBase {
|
||||
/// Object store URI such as `s3://bucket/media/`.
|
||||
pub path: String,
|
||||
/// Optional alias.
|
||||
pub name: Option<String>,
|
||||
/// True when `path` is a Lance dataset root. When false, `path` is the
|
||||
/// directory containing the referenced files.
|
||||
pub is_dataset_root: bool,
|
||||
}
|
||||
|
||||
#[napi(object)]
|
||||
/// A description of an index currently configured on a column
|
||||
pub struct IndexConfig {
|
||||
@@ -1255,21 +1196,6 @@ pub struct AddColumnsResult {
|
||||
pub version: i64,
|
||||
}
|
||||
|
||||
#[napi(object)]
|
||||
pub struct RefreshColumnResult {
|
||||
pub rows_filled: i64,
|
||||
pub version: i64,
|
||||
}
|
||||
|
||||
impl From<lancedb::table::RefreshColumnResult> for RefreshColumnResult {
|
||||
fn from(value: lancedb::table::RefreshColumnResult) -> Self {
|
||||
Self {
|
||||
rows_filled: value.rows_filled as i64,
|
||||
version: value.version as i64,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<lancedb::table::AddColumnsResult> for AddColumnsResult {
|
||||
fn from(value: lancedb::table::AddColumnsResult) -> Self {
|
||||
Self {
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "lancedb-python"
|
||||
version = "0.38.0-beta.0"
|
||||
version = "0.37.1-beta.1"
|
||||
publish = false
|
||||
edition.workspace = true
|
||||
description = "Python bindings for LanceDB"
|
||||
|
||||
@@ -21,7 +21,7 @@ from .remote.db import RemoteDBConnection
|
||||
from .expr import Expr, col, lit, func
|
||||
from .schema import blob, vector, BlobType
|
||||
from .job import AsyncJob, Job
|
||||
from .table import AsyncTable, Table, TableBase
|
||||
from .table import AsyncTable, Table
|
||||
from .types import BaseTokenizerType
|
||||
from ._lancedb import Session
|
||||
from .namespace import (
|
||||
@@ -521,6 +521,5 @@ __all__ = [
|
||||
"RemoteDBConnection",
|
||||
"Session",
|
||||
"Table",
|
||||
"TableBase",
|
||||
"__version__",
|
||||
]
|
||||
|
||||
@@ -198,9 +198,6 @@ class Connection(object):
|
||||
async def drop_table(
|
||||
self, name: str, namespace_path: Optional[List[str]] = None
|
||||
) -> None: ...
|
||||
async def drop_table_async(
|
||||
self, name: str, namespace_path: Optional[List[str]] = None
|
||||
) -> Job: ...
|
||||
async def drop_all_tables(
|
||||
self, namespace_path: Optional[List[str]] = None
|
||||
) -> None: ...
|
||||
@@ -338,11 +335,6 @@ class Table:
|
||||
) -> list[FtsToken]: ...
|
||||
async def delete(self, filter: Union[str, PyExpr]) -> DeleteResult: ...
|
||||
async def add_columns(self, columns: list[tuple[str, str]]) -> AddColumnsResult: ...
|
||||
async def add_computed_columns(
|
||||
self, columns: list[tuple[str, str]]
|
||||
) -> AddColumnsResult: ...
|
||||
async def refresh_column(self, column: str) -> RefreshColumnResult: ...
|
||||
async def refresh_column_async(self, column: str) -> Job: ...
|
||||
async def add_columns_with_schema(self, schema: pa.Schema) -> AddColumnsResult: ...
|
||||
async def alter_columns(
|
||||
self, columns: list[dict[str, Any]]
|
||||
@@ -377,7 +369,6 @@ class Table:
|
||||
def take_offsets(self, offsets: list[int]) -> TakeQuery: ...
|
||||
def take_row_ids(self, row_ids: list[int]) -> TakeQuery: ...
|
||||
async def blob_columns(self) -> list[str]: ...
|
||||
async def add_bases(self, bases: list[Any]) -> None: ...
|
||||
async def fetch_blobs(
|
||||
self, column: str, row_ids: list[int]
|
||||
) -> pa.LargeBinaryArray: ...
|
||||
@@ -689,10 +680,6 @@ class LsmWriteSpec:
|
||||
class AddColumnsResult:
|
||||
version: int
|
||||
|
||||
class RefreshColumnResult:
|
||||
rows_filled: int
|
||||
version: int
|
||||
|
||||
class AlterColumnsResult:
|
||||
version: int
|
||||
|
||||
|
||||
@@ -524,12 +524,6 @@ class DBConnection(EnforceOverrides):
|
||||
namespace_path = []
|
||||
raise NotImplementedError
|
||||
|
||||
def drop_table_async(
|
||||
self, name: str, namespace_path: Optional[List[str]] = None
|
||||
) -> Job:
|
||||
"""Start dropping a table and return its cleanup job."""
|
||||
raise NotImplementedError
|
||||
|
||||
def rename_table(
|
||||
self,
|
||||
cur_name: str,
|
||||
@@ -722,20 +716,9 @@ class LanceDBConnection(DBConnection):
|
||||
if not isinstance(uri, Path):
|
||||
scheme = get_uri_scheme(uri)
|
||||
is_local = isinstance(uri, Path) or scheme == "file"
|
||||
if is_local:
|
||||
is_file_uri = isinstance(uri, str) and uri.lower().startswith("file:")
|
||||
if is_local and not is_file_uri:
|
||||
if isinstance(uri, str):
|
||||
# Strip file:// or file:/ scheme if present
|
||||
# file:///path becomes file:/path after URL normalization
|
||||
if uri.startswith("file://"):
|
||||
uri = uri[7:] # Remove "file://"
|
||||
elif uri.startswith("file:/"):
|
||||
uri = uri[5:] # Remove "file:"
|
||||
|
||||
if sys.platform == "win32":
|
||||
# On Windows, a path like /C:/path should become C:/path
|
||||
if len(uri) >= 3 and uri[0] == "/" and uri[2] == ":":
|
||||
uri = uri[1:]
|
||||
|
||||
uri = Path(uri)
|
||||
uri = uri.expanduser().absolute()
|
||||
Path(uri).mkdir(parents=True, exist_ok=True)
|
||||
@@ -1192,20 +1175,6 @@ class LanceDBConnection(DBConnection):
|
||||
)
|
||||
)
|
||||
|
||||
@override
|
||||
def drop_table_async(
|
||||
self, name: str, namespace_path: Optional[List[str]] = None
|
||||
) -> Job:
|
||||
"""Start dropping a table and return its cleanup job.
|
||||
|
||||
The table may become unavailable before its data files are removed.
|
||||
Call :meth:`Job.wait` to wait for cleanup to finish.
|
||||
"""
|
||||
if namespace_path is None:
|
||||
namespace_path = []
|
||||
job = LOOP.run(self._conn.drop_table_async(name, namespace_path=namespace_path))
|
||||
return Job(job if isinstance(job, AsyncJob) else AsyncJob(job))
|
||||
|
||||
@override
|
||||
def drop_all_tables(self, namespace_path: Optional[List[str]] = None):
|
||||
if namespace_path is None:
|
||||
@@ -1983,23 +1952,6 @@ class AsyncConnection(object):
|
||||
if f"Table '{name}' was not found" not in str(e):
|
||||
raise e
|
||||
|
||||
async def drop_table_async(
|
||||
self,
|
||||
name: str,
|
||||
*,
|
||||
namespace_path: Optional[List[str]] = None,
|
||||
) -> AsyncJob:
|
||||
"""Start dropping a table and return its cleanup job.
|
||||
|
||||
The table may become unavailable before its data files are removed.
|
||||
Await :meth:`AsyncJob.wait` to wait for cleanup to finish.
|
||||
"""
|
||||
if namespace_path is None:
|
||||
namespace_path = []
|
||||
return AsyncJob(
|
||||
await self._inner.drop_table_async(name, namespace_path=namespace_path)
|
||||
)
|
||||
|
||||
async def drop_all_tables(self, namespace_path: Optional[List[str]] = None):
|
||||
"""Drop all tables from the database.
|
||||
|
||||
|
||||
@@ -49,7 +49,6 @@ from lancedb._lancedb import (
|
||||
)
|
||||
from lancedb.background_loop import LOOP
|
||||
from lancedb.db import AsyncConnection, DBConnection
|
||||
from lancedb.job import AsyncJob, Job
|
||||
from lance_namespace import (
|
||||
LanceNamespace,
|
||||
connect as namespace_connect,
|
||||
@@ -625,18 +624,6 @@ class LanceNamespaceDBConnection(DBConnection):
|
||||
namespace_path = []
|
||||
LOOP.run(self._inner.drop_table(name, namespace_path=namespace_path))
|
||||
|
||||
@override
|
||||
def drop_table_async(
|
||||
self, name: str, namespace_path: Optional[List[str]] = None
|
||||
) -> Job:
|
||||
"""Start dropping a table and return its cleanup job."""
|
||||
if namespace_path is None:
|
||||
namespace_path = []
|
||||
job = LOOP.run(
|
||||
self._inner.drop_table_async(name, namespace_path=namespace_path)
|
||||
)
|
||||
return Job(job if isinstance(job, AsyncJob) else AsyncJob(job))
|
||||
|
||||
@override
|
||||
def rename_table(
|
||||
self,
|
||||
@@ -932,17 +919,7 @@ class LanceNamespaceDBConnection(DBConnection):
|
||||
The namespace client for this connection.
|
||||
"""
|
||||
if self._namespace_client is None:
|
||||
if (
|
||||
self._namespace_client_impl is None
|
||||
or self._namespace_client_properties is None
|
||||
):
|
||||
raise ValueError(
|
||||
"Cannot construct a Python namespace client without "
|
||||
"namespace implementation properties"
|
||||
)
|
||||
self._namespace_client = namespace_connect(
|
||||
self._namespace_client_impl, self._namespace_client_properties
|
||||
)
|
||||
self._namespace_client = LOOP.run(self._inner.namespace_client())
|
||||
return self._namespace_client
|
||||
|
||||
|
||||
@@ -1147,14 +1124,6 @@ class AsyncLanceNamespaceDBConnection:
|
||||
namespace_path = []
|
||||
await self._inner.drop_table(name, namespace_path=namespace_path)
|
||||
|
||||
async def drop_table_async(
|
||||
self, name: str, namespace_path: Optional[List[str]] = None
|
||||
) -> AsyncJob:
|
||||
"""Start dropping a table and return its cleanup job."""
|
||||
if namespace_path is None:
|
||||
namespace_path = []
|
||||
return await self._inner.drop_table_async(name, namespace_path=namespace_path)
|
||||
|
||||
async def rename_table(
|
||||
self,
|
||||
cur_name: str,
|
||||
@@ -1391,17 +1360,7 @@ class AsyncLanceNamespaceDBConnection:
|
||||
The namespace client for this connection.
|
||||
"""
|
||||
if self._namespace_client is None:
|
||||
if (
|
||||
self._namespace_client_impl is None
|
||||
or self._namespace_client_properties is None
|
||||
):
|
||||
raise ValueError(
|
||||
"Cannot construct a Python namespace client without "
|
||||
"namespace implementation properties"
|
||||
)
|
||||
self._namespace_client = namespace_connect(
|
||||
self._namespace_client_impl, self._namespace_client_properties
|
||||
)
|
||||
self._namespace_client = await self._inner.namespace_client()
|
||||
return self._namespace_client
|
||||
|
||||
|
||||
|
||||
@@ -23,7 +23,7 @@ import pyarrow as pa
|
||||
|
||||
from ..common import DATA
|
||||
from ..db import DBConnection, LOOP
|
||||
from ..job import AsyncJob, Job
|
||||
from ..job import Job
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from .._lancedb import JobDescription, JobInfo
|
||||
@@ -663,16 +663,6 @@ class RemoteDBConnection(DBConnection):
|
||||
namespace_path = []
|
||||
LOOP.run(self._conn.drop_table(name, namespace_path=namespace_path))
|
||||
|
||||
@override
|
||||
def drop_table_async(
|
||||
self, name: str, namespace_path: Optional[List[str]] = None
|
||||
) -> Job:
|
||||
"""Start dropping a table and return its cleanup job."""
|
||||
if namespace_path is None:
|
||||
namespace_path = []
|
||||
job = LOOP.run(self._conn.drop_table_async(name, namespace_path=namespace_path))
|
||||
return Job(job if isinstance(job, AsyncJob) else AsyncJob(job))
|
||||
|
||||
@override
|
||||
def rename_table(
|
||||
self,
|
||||
|
||||
@@ -50,7 +50,7 @@ from lancedb.index import (
|
||||
)
|
||||
from lancedb.job import Job
|
||||
from lancedb.remote.db import LOOP
|
||||
from lancedb.table import IndexConfigType, KNOWN_METRICS, TableBase
|
||||
from lancedb.table import IndexConfigType, KNOWN_METRICS
|
||||
import pyarrow as pa
|
||||
|
||||
from lancedb.common import DATA, VEC, VECTOR_COLUMN_NAME
|
||||
@@ -958,19 +958,8 @@ class RemoteTable(Table):
|
||||
def count_rows(self, filter: Optional[str] = None) -> int:
|
||||
return LOOP.run(self._table.count_rows(filter))
|
||||
|
||||
def add_columns(
|
||||
self,
|
||||
transforms: Dict[str, str] | None = None,
|
||||
*,
|
||||
computed: Dict[str, str] | None = None,
|
||||
) -> AddColumnsResult:
|
||||
return LOOP.run(self._table.add_columns(transforms, computed=computed))
|
||||
|
||||
def refresh_column(self, column: str):
|
||||
return LOOP.run(self._table.refresh_column(column))
|
||||
|
||||
def refresh_column_async(self, column: str) -> Job:
|
||||
return Job(LOOP.run(self._table.refresh_column_async(column)))
|
||||
def add_columns(self, transforms: Dict[str, str]) -> AddColumnsResult:
|
||||
return LOOP.run(self._table.add_columns(transforms))
|
||||
|
||||
def alter_columns(
|
||||
self, *alterations: Iterable[Dict[str, str]]
|
||||
@@ -1082,13 +1071,6 @@ class RemoteTable(Table):
|
||||
def blob_columns(self) -> list[str]:
|
||||
return LOOP.run(self._table.blob_columns())
|
||||
|
||||
def add_bases(
|
||||
self,
|
||||
bases: Union[str, TableBase, Iterable[Union[str, TableBase]]],
|
||||
) -> None:
|
||||
"""Register additional storage bases for this table."""
|
||||
LOOP.run(self._table.add_bases(bases))
|
||||
|
||||
def fetch_blobs(
|
||||
self, column: str, row_ids: Union[list[int], pa.Table]
|
||||
) -> pa.LargeBinaryArray:
|
||||
|
||||
@@ -19,7 +19,6 @@ from typing import (
|
||||
Iterable,
|
||||
List,
|
||||
Literal,
|
||||
Mapping,
|
||||
Optional,
|
||||
Sequence,
|
||||
Tuple,
|
||||
@@ -177,7 +176,6 @@ if TYPE_CHECKING:
|
||||
CompactionStats,
|
||||
Tag,
|
||||
AddColumnsResult,
|
||||
RefreshColumnResult,
|
||||
AddResult,
|
||||
AlterColumnsResult,
|
||||
UpdateFieldMetadataResult,
|
||||
@@ -711,21 +709,6 @@ def _normalize_progress(progress):
|
||||
return progress, False
|
||||
|
||||
|
||||
@dataclass
|
||||
class TableBase:
|
||||
"""An extra storage prefix registered on a table.
|
||||
|
||||
``path`` is an object-store URI. ``name`` is an optional alias.
|
||||
``is_dataset_root`` is true when ``path`` points to a Lance dataset
|
||||
root. When false, ``path`` points directly to the directory containing
|
||||
the referenced files.
|
||||
"""
|
||||
|
||||
path: str
|
||||
name: Optional[str] = None
|
||||
is_dataset_root: bool = False
|
||||
|
||||
|
||||
class Table(ABC):
|
||||
"""
|
||||
A Table is a collection of Records in a LanceDB Database.
|
||||
@@ -1584,18 +1567,6 @@ class Table(ABC):
|
||||
def blob_columns(self) -> list[str]:
|
||||
"""Names of the blob v2 columns declared on this table."""
|
||||
|
||||
def add_bases(
|
||||
self,
|
||||
bases: Union[str, TableBase, Iterable[Union[str, TableBase]]],
|
||||
) -> None:
|
||||
"""Register additional storage bases for this table.
|
||||
|
||||
A URI string is a non-root base with no alias::
|
||||
|
||||
table.add_bases("s3://bucket/media/")
|
||||
"""
|
||||
raise NotImplementedError
|
||||
|
||||
@abstractmethod
|
||||
def fetch_blobs(
|
||||
self, column: str, row_ids: Union[list[int], pa.Table]
|
||||
@@ -1945,14 +1916,7 @@ class Table(ABC):
|
||||
|
||||
@abstractmethod
|
||||
def add_columns(
|
||||
self,
|
||||
transforms: Dict[str, str]
|
||||
| pa.Field
|
||||
| List[pa.Field]
|
||||
| pa.Schema
|
||||
| None = None,
|
||||
*,
|
||||
computed: Dict[str, str] | None = None,
|
||||
self, transforms: Dict[str, str] | pa.Field | List[pa.Field] | pa.Schema
|
||||
):
|
||||
"""
|
||||
Add new columns with defined values.
|
||||
@@ -1966,95 +1930,11 @@ class Table(ABC):
|
||||
Alternatively, a pyarrow Field or Schema can be provided to add
|
||||
new columns with the specified data types. The new columns will
|
||||
be initialized with null values.
|
||||
computed: Dict[str, str], optional
|
||||
A map of column name to a SQL expression defining the column. The
|
||||
column's type and inputs are derived from the expression, so no
|
||||
data type is supplied.
|
||||
|
||||
Unlike ``transforms``, the expression is stored rather than
|
||||
evaluated now: the column is committed with no values, and rows get
|
||||
them from [`refresh_column`][lancedb.table.Table.refresh_column].
|
||||
Declaring one therefore costs the same on a large table as on an
|
||||
empty one.
|
||||
|
||||
A refresh does not revisit rows it has already filled, so mutating
|
||||
an input leaves the value computed at fill time; recomputing means
|
||||
dropping the column and declaring it again. While a declaration
|
||||
reads a column, that column cannot be renamed, retyped or dropped.
|
||||
|
||||
On LanceDB Cloud and Enterprise the expression is planned by the
|
||||
server, and the refresh runs as a server job -- see
|
||||
[`refresh_column_async`][lancedb.table.Table.refresh_column_async].
|
||||
Cannot be combined with ``transforms``.
|
||||
|
||||
Returns
|
||||
-------
|
||||
AddColumnsResult
|
||||
version: the new version number of the table after adding columns.
|
||||
|
||||
Examples
|
||||
--------
|
||||
>>> import lancedb
|
||||
>>> db = lancedb.connect("./.lancedb")
|
||||
>>> table = db.create_table("computed_demo", [{"x": 1}, {"x": 2}])
|
||||
>>> table.add_columns(computed={"doubled": "x * 2"})
|
||||
AddColumnsResult(version=2)
|
||||
>>> table.refresh_column("doubled")
|
||||
RefreshColumnResult(rows_filled=2, version=3)
|
||||
>>> table.to_arrow().sort_by("x").to_pandas()
|
||||
x doubled
|
||||
0 1 2
|
||||
1 2 4
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
def refresh_column(self, column: str) -> "RefreshColumnResult":
|
||||
"""
|
||||
Fill the rows of a computed column that hold no value yet.
|
||||
|
||||
Declared with ``add_columns(computed=...)``, a column starts empty and
|
||||
gets its values here. Rows appended since the last refresh are filled
|
||||
by the next one; rows already filled are left as they are, so the call
|
||||
is idempotent and does not observe a mutated input.
|
||||
|
||||
Local tables only: a remote refresh runs as a server job, through
|
||||
[`refresh_column_async`][lancedb.table.Table.refresh_column_async].
|
||||
|
||||
Parameters
|
||||
----------
|
||||
column: str
|
||||
The name of the computed column to fill.
|
||||
|
||||
Returns
|
||||
-------
|
||||
RefreshColumnResult
|
||||
rows_filled: the number of rows given a value.
|
||||
version: the new version number of the table.
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
def refresh_column_async(self, column: str) -> Job:
|
||||
"""
|
||||
Like :meth:`refresh_column`, but returns a handle to the refresh job
|
||||
instead of blocking until it completes.
|
||||
|
||||
The job may already be complete when returned; callers must not assume
|
||||
the column is filled until :meth:`Job.wait` returns. Invalid input --
|
||||
an unknown column, or one that is not computed -- raises here rather
|
||||
than failing the job. On local tables the job runs in-process; on
|
||||
LanceDB Cloud and Enterprise it is the server's backfill job.
|
||||
|
||||
Examples
|
||||
--------
|
||||
>>> import lancedb
|
||||
>>> db = lancedb.connect("./.lancedb")
|
||||
>>> table = db.create_table("computed_job_demo", [{"x": 1}, {"x": 2}])
|
||||
>>> table.add_columns(computed={"doubled": "x * 2"})
|
||||
AddColumnsResult(version=2)
|
||||
>>> job = table.refresh_column_async("doubled")
|
||||
>>> job.wait()
|
||||
>>> job.status()
|
||||
'finished'
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
@@ -2442,12 +2322,6 @@ class LanceTable(Table):
|
||||
def blob_columns(self) -> list[str]:
|
||||
return LOOP.run(self._table.blob_columns())
|
||||
|
||||
def add_bases(
|
||||
self,
|
||||
bases: Union[str, TableBase, Iterable[Union[str, TableBase]]],
|
||||
) -> None:
|
||||
LOOP.run(self._table.add_bases(bases))
|
||||
|
||||
def fetch_blobs(
|
||||
self, column: str, row_ids: Union[list[int], pa.Table]
|
||||
) -> pa.LargeBinaryArray:
|
||||
@@ -4065,28 +3939,9 @@ class LanceTable(Table):
|
||||
return LOOP.run(self._table.index_stats(index_name))
|
||||
|
||||
def add_columns(
|
||||
self,
|
||||
transforms: Dict[str, str]
|
||||
| pa.field
|
||||
| List[pa.field]
|
||||
| pa.Schema
|
||||
| None = None,
|
||||
*,
|
||||
computed: Dict[str, str] | None = None,
|
||||
self, transforms: Dict[str, str] | pa.field | List[pa.field] | pa.Schema
|
||||
) -> AddColumnsResult:
|
||||
return LOOP.run(self._table.add_columns(transforms, computed=computed))
|
||||
|
||||
def refresh_column(self, column: str) -> "RefreshColumnResult":
|
||||
"""Fill a computed column's unfilled rows. See
|
||||
[`AsyncTable.refresh_column`][lancedb.AsyncTable.refresh_column]."""
|
||||
return LOOP.run(self._table.refresh_column(column))
|
||||
|
||||
def refresh_column_async(self, column: str) -> Job:
|
||||
"""Fill a computed column's unfilled rows, returning a handle to the
|
||||
refresh job. See
|
||||
[`Table.refresh_column_async`][lancedb.table.Table.refresh_column_async].
|
||||
"""
|
||||
return Job(LOOP.run(self._table.refresh_column_async(column)))
|
||||
return LOOP.run(self._table.add_columns(transforms))
|
||||
|
||||
def alter_columns(
|
||||
self, *alterations: Iterable[Dict[str, str]]
|
||||
@@ -6001,14 +5856,7 @@ class AsyncTable:
|
||||
return await self._inner.update(updates_sql, where)
|
||||
|
||||
async def add_columns(
|
||||
self,
|
||||
transforms: dict[str, str]
|
||||
| pa.field
|
||||
| List[pa.field]
|
||||
| pa.Schema
|
||||
| None = None,
|
||||
*,
|
||||
computed: dict[str, str] | None = None,
|
||||
self, transforms: dict[str, str] | pa.field | List[pa.field] | pa.Schema
|
||||
) -> AddColumnsResult:
|
||||
"""
|
||||
Add new columns with defined values.
|
||||
@@ -6021,22 +5869,6 @@ class AsyncTable:
|
||||
each row in the table, and can reference existing columns.
|
||||
Alternatively, you can pass a pyarrow field or schema to add
|
||||
new columns with NULLs.
|
||||
computed: Dict[str, str], optional
|
||||
A map of column name to a SQL expression defining the column. The
|
||||
column's type and inputs are derived from the expression.
|
||||
|
||||
Unlike ``transforms``, the expression is stored rather than
|
||||
evaluated now: the column is committed with no values, and rows get
|
||||
them from
|
||||
[`refresh_column`][lancedb.table.AsyncTable.refresh_column].
|
||||
|
||||
A refresh does not revisit rows it has already filled, so mutating
|
||||
an input leaves the value computed at fill time. While a
|
||||
declaration reads a column, that column cannot be renamed, retyped
|
||||
or dropped.
|
||||
|
||||
On LanceDB Cloud and Enterprise the expression is planned by
|
||||
the server. Cannot be combined with ``transforms``.
|
||||
|
||||
Returns
|
||||
-------
|
||||
@@ -6050,71 +5882,11 @@ class AsyncTable:
|
||||
{isinstance(f, pa.Field) for f in transforms}
|
||||
):
|
||||
transforms = pa.schema(transforms)
|
||||
if computed:
|
||||
if transforms:
|
||||
raise ValueError(
|
||||
"add_columns cannot take both transforms and computed columns"
|
||||
)
|
||||
return await self._inner.add_computed_columns(list(computed.items()))
|
||||
if transforms is None:
|
||||
raise ValueError("add_columns requires transforms or computed columns")
|
||||
if isinstance(transforms, pa.Schema):
|
||||
return await self._inner.add_columns_with_schema(transforms)
|
||||
else:
|
||||
return await self._inner.add_columns(list(transforms.items()))
|
||||
|
||||
async def refresh_column(self, column: str) -> RefreshColumnResult:
|
||||
"""
|
||||
Fill the rows of a computed column that hold no value yet.
|
||||
|
||||
Declared with ``add_columns(computed=...)``, a column starts empty and
|
||||
gets its values here. Rows appended since the last refresh are filled
|
||||
by the next one; rows already filled are left as they are, so the call
|
||||
is idempotent and does not observe a mutated input.
|
||||
|
||||
Local tables only: a remote refresh runs as a server job, through
|
||||
[`refresh_column_async`][lancedb.table.Table.refresh_column_async].
|
||||
|
||||
Parameters
|
||||
----------
|
||||
column: str
|
||||
The name of the computed column to fill.
|
||||
|
||||
Returns
|
||||
-------
|
||||
RefreshColumnResult
|
||||
The number of rows filled and the new version of the table.
|
||||
"""
|
||||
return await self._inner.refresh_column(column)
|
||||
|
||||
async def refresh_column_async(self, column: str) -> AsyncJob:
|
||||
"""
|
||||
Like :meth:`refresh_column`, but returns a handle to the refresh job
|
||||
instead of blocking until it completes.
|
||||
|
||||
The job may already be complete when returned; callers must not assume
|
||||
the column is filled until :meth:`AsyncJob.wait` resolves. Invalid
|
||||
input -- an unknown column, or one that is not computed -- raises here
|
||||
rather than failing the job. On local tables the job runs
|
||||
in-process; on LanceDB Cloud and Enterprise it is the server's
|
||||
backfill job.
|
||||
|
||||
Examples
|
||||
--------
|
||||
>>> import asyncio
|
||||
>>> import lancedb
|
||||
>>> async def refresh_in_background():
|
||||
... db = await lancedb.connect_async("./.lancedb")
|
||||
... table = await db.create_table("computed_job_async_demo", [{"x": 1}])
|
||||
... await table.add_columns(computed={"doubled": "x * 2"})
|
||||
... job = await table.refresh_column_async("doubled")
|
||||
... await job.wait()
|
||||
... return await job.status()
|
||||
>>> asyncio.run(refresh_in_background())
|
||||
'finished'
|
||||
"""
|
||||
return AsyncJob(await self._inner.refresh_column_async(column))
|
||||
|
||||
async def alter_columns(
|
||||
self, *alterations: Iterable[dict[str, Any]]
|
||||
) -> AlterColumnsResult:
|
||||
@@ -6300,18 +6072,6 @@ class AsyncTable:
|
||||
async def blob_columns(self) -> list[str]:
|
||||
return await self._inner.blob_columns()
|
||||
|
||||
async def add_bases(
|
||||
self,
|
||||
bases: Union[str, TableBase, Iterable[Union[str, TableBase]]],
|
||||
) -> None:
|
||||
"""Register additional storage bases for this table.
|
||||
|
||||
A URI string is a non-root base with no alias::
|
||||
|
||||
await table.add_bases("s3://bucket/media/")
|
||||
"""
|
||||
await self._inner.add_bases(_normalize_bases(bases))
|
||||
|
||||
async def fetch_blobs(
|
||||
self, column: str, row_ids: Union[list[int], pa.Table]
|
||||
) -> pa.LargeBinaryArray:
|
||||
@@ -6530,30 +6290,6 @@ class AsyncTable:
|
||||
await self._inner.replace_field_metadata(field_name, new_metadata)
|
||||
|
||||
|
||||
def _normalize_bases(
|
||||
base_inputs: Union[str, TableBase, Iterable[Union[str, TableBase]]],
|
||||
) -> list[TableBase]:
|
||||
if isinstance(base_inputs, (str, TableBase)):
|
||||
items: Iterable[Union[str, TableBase]] = [base_inputs]
|
||||
elif isinstance(base_inputs, Mapping):
|
||||
raise TypeError(
|
||||
"Expected a URI string, TableBase, or an iterable of those values"
|
||||
)
|
||||
else:
|
||||
items = base_inputs
|
||||
normalized_bases: list[TableBase] = []
|
||||
for base in items:
|
||||
if isinstance(base, str):
|
||||
normalized_bases.append(TableBase(path=base))
|
||||
elif isinstance(base, TableBase):
|
||||
normalized_bases.append(base)
|
||||
else:
|
||||
raise TypeError(
|
||||
f"Expected a URI string or TableBase, got {type(base).__name__}"
|
||||
)
|
||||
return normalized_bases
|
||||
|
||||
|
||||
@dataclass
|
||||
class IndexStatistics:
|
||||
"""
|
||||
|
||||
@@ -1,72 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||
|
||||
import pyarrow as pa
|
||||
import pytest
|
||||
|
||||
import lancedb
|
||||
|
||||
|
||||
def test_add_bases_accepts_named_and_dataset_root(tmp_path):
|
||||
media = tmp_path / "media"
|
||||
parent = tmp_path / "parent"
|
||||
media.mkdir()
|
||||
parent.mkdir()
|
||||
db = lancedb.connect(tmp_path / "db")
|
||||
schema = pa.schema([pa.field("id", pa.int64())])
|
||||
table = db.create_table("photos", schema=schema)
|
||||
table.add_bases(
|
||||
[
|
||||
lancedb.TableBase(path=media.as_uri(), name="media", is_dataset_root=False),
|
||||
lancedb.TableBase(
|
||||
path=parent.as_uri(), name="parent", is_dataset_root=True
|
||||
),
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
def test_add_bases_accepts_two_unnamed_paths(tmp_path):
|
||||
media = tmp_path / "media"
|
||||
other = tmp_path / "other"
|
||||
media.mkdir()
|
||||
other.mkdir()
|
||||
db = lancedb.connect(tmp_path / "db")
|
||||
schema = pa.schema([pa.field("id", pa.int64())])
|
||||
table = db.create_table("photos", schema=schema)
|
||||
table.add_bases([media.as_uri(), other.as_uri()])
|
||||
|
||||
|
||||
def test_add_bases_rejects_dict_input(tmp_path):
|
||||
db = lancedb.connect(tmp_path / "db")
|
||||
schema = pa.schema([pa.field("id", pa.int64())])
|
||||
table = db.create_table("photos", schema=schema)
|
||||
with pytest.raises(TypeError, match="TableBase"):
|
||||
table.add_bases({"path": "s3://bucket/media/"})
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_add_bases_accepts_file_uri(tmp_path):
|
||||
media = tmp_path / "media"
|
||||
media.mkdir()
|
||||
db = await lancedb.connect_async(tmp_path / "db")
|
||||
schema = pa.schema([pa.field("id", pa.int64())])
|
||||
table = await db.create_table("photos", schema=schema)
|
||||
await table.add_bases(media.as_uri())
|
||||
|
||||
|
||||
def test_memory_add_bases_accepts_file_uri(tmp_path):
|
||||
media = tmp_path / "media"
|
||||
media.mkdir()
|
||||
db = lancedb.connect("memory:///")
|
||||
schema = pa.schema([pa.field("id", pa.int64())])
|
||||
table = db.create_table("photos", schema=schema)
|
||||
table.add_bases(media.as_uri())
|
||||
|
||||
|
||||
def test_namespace_add_bases_accepts_file_uri(tmp_path):
|
||||
media = tmp_path / "media"
|
||||
media.mkdir()
|
||||
db = lancedb.connect_namespace("dir", {"root": str(tmp_path / "ns")})
|
||||
schema = pa.schema([pa.field("id", pa.int64())])
|
||||
table = db.create_table("photos", schema=schema)
|
||||
table.add_bases(media.as_uri())
|
||||
@@ -89,6 +89,32 @@ def test_sync_debugger_inspection_does_not_use_background_loop(tmp_path, monkeyp
|
||||
assert repr(table) == f"LanceTable(name='test', _conn={db!r})"
|
||||
|
||||
|
||||
def test_connect_preserves_file_uri_authority(monkeypatch):
|
||||
uri = "file://server/share/database"
|
||||
received = []
|
||||
|
||||
async def fake_connect(passed_uri, *_args):
|
||||
received.append(passed_uri)
|
||||
return SimpleNamespace(uri=passed_uri)
|
||||
|
||||
monkeypatch.setattr("lancedb.db.lancedb_connect", fake_connect)
|
||||
db = lancedb.connect(uri)
|
||||
|
||||
assert received == [uri]
|
||||
assert db.uri == uri
|
||||
|
||||
|
||||
def test_connect_file_uri_lifecycle(tmp_path):
|
||||
uri = (tmp_path / "sync").as_uri()
|
||||
db = lancedb.connect(uri)
|
||||
|
||||
db.create_table("test", data=[{"id": 1}])
|
||||
assert db.table_names() == ["test"]
|
||||
assert db.open_table("test").count_rows() == 1
|
||||
db.drop_table("test")
|
||||
assert db.table_names() == []
|
||||
|
||||
|
||||
def test_read_consistency_interval_does_not_use_background_loop(tmp_path, monkeypatch):
|
||||
from lancedb.background_loop import LOOP
|
||||
from lancedb.db import LanceDBConnection
|
||||
@@ -405,6 +431,35 @@ async def test_connect(tmp_path):
|
||||
assert str(db) == f"ListingDatabase(uri={tmp_path}, read_consistency_interval=5s)"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_connect_async_preserves_file_uri_authority(monkeypatch):
|
||||
uri = "file://server/share/database"
|
||||
received = []
|
||||
|
||||
async def fake_connect(passed_uri, *_args):
|
||||
received.append(passed_uri)
|
||||
return SimpleNamespace(uri=passed_uri)
|
||||
|
||||
monkeypatch.setattr(lancedb, "lancedb_connect", fake_connect)
|
||||
db = await lancedb.connect_async(uri)
|
||||
|
||||
assert received == [uri]
|
||||
assert db.uri == uri
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_connect_async_file_uri_lifecycle(tmp_path):
|
||||
uri = (tmp_path / "async").as_uri()
|
||||
db = await lancedb.connect_async(uri)
|
||||
|
||||
await db.create_table("test", data=[{"id": 1}])
|
||||
assert await db.table_names() == ["test"]
|
||||
table = await db.open_table("test")
|
||||
assert await table.count_rows() == 1
|
||||
await db.drop_table("test")
|
||||
assert await db.table_names() == []
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_close(mem_db_async: lancedb.AsyncConnection):
|
||||
assert mem_db_async.is_open()
|
||||
@@ -755,7 +810,8 @@ def test_delete_table(tmp_db: lancedb.DBConnection):
|
||||
assert tmp_db.table_names() == []
|
||||
|
||||
|
||||
def test_drop_table_async(tmp_db: lancedb.DBConnection):
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_table_async(tmp_db: lancedb.DBConnection):
|
||||
data = pd.DataFrame(
|
||||
{
|
||||
"vector": [[3.1, 4.1], [5.9, 26.5]],
|
||||
@@ -771,10 +827,7 @@ def test_drop_table_async(tmp_db: lancedb.DBConnection):
|
||||
|
||||
assert tmp_db.table_names() == ["test"]
|
||||
|
||||
job = tmp_db.drop_table_async("test")
|
||||
assert job.id is None
|
||||
assert job.status() == "finished"
|
||||
job.wait()
|
||||
tmp_db.drop_table("test")
|
||||
assert tmp_db.table_names() == []
|
||||
|
||||
tmp_db.create_table("test", data=data)
|
||||
@@ -783,17 +836,6 @@ def test_drop_table_async(tmp_db: lancedb.DBConnection):
|
||||
tmp_db.drop_table("does_not_exist", ignore_missing=True)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_drop_table_async_connection(tmp_db_async: lancedb.AsyncConnection):
|
||||
await tmp_db_async.create_table("test", data=pa.table({"id": [1, 2]}))
|
||||
|
||||
job = await tmp_db_async.drop_table_async("test")
|
||||
assert job.id is None
|
||||
assert await job.status() == "finished"
|
||||
await job.wait()
|
||||
assert await tmp_db_async.table_names() == []
|
||||
|
||||
|
||||
def test_drop_database(tmp_db: lancedb.DBConnection):
|
||||
data = pd.DataFrame(
|
||||
{
|
||||
@@ -1193,6 +1235,40 @@ def test_clone_table_deep_clone_fails(tmp_path):
|
||||
db.clone_table("cloned", source_uri, is_shallow=False)
|
||||
|
||||
|
||||
class _UnsupportedNamespaceConfig:
|
||||
async def namespace_client_config(self):
|
||||
raise RuntimeError("UNC namespace client export is not supported")
|
||||
|
||||
|
||||
def test_sync_namespace_client_propagates_export_guard(monkeypatch):
|
||||
from lancedb.db import AsyncConnection, LanceDBConnection
|
||||
|
||||
monkeypatch.setattr(
|
||||
"lancedb.db.namespace_connect",
|
||||
lambda *_args, **_kwargs: pytest.fail("guarded config was reconstructed"),
|
||||
)
|
||||
db = LanceDBConnection.__new__(LanceDBConnection)
|
||||
db._conn = AsyncConnection(_UnsupportedNamespaceConfig())
|
||||
db._cached_namespace_client = None
|
||||
|
||||
with pytest.raises(RuntimeError, match="UNC namespace client export"):
|
||||
db.namespace_client()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_namespace_client_propagates_export_guard(monkeypatch):
|
||||
from lancedb.db import AsyncConnection
|
||||
|
||||
monkeypatch.setattr(
|
||||
"lancedb.db.namespace_connect",
|
||||
lambda *_args, **_kwargs: pytest.fail("guarded config was reconstructed"),
|
||||
)
|
||||
db = AsyncConnection(_UnsupportedNamespaceConfig())
|
||||
|
||||
with pytest.raises(RuntimeError, match="UNC namespace client export"):
|
||||
await db.namespace_client()
|
||||
|
||||
|
||||
@pytest.mark.skipif(sys.platform == "win32", reason="Namespace client issues")
|
||||
def test_namespace_client_native_storage(tmp_path):
|
||||
"""Test namespace_client() returns DirectoryNamespace for native storage."""
|
||||
|
||||
@@ -60,6 +60,11 @@ class _NamespaceClient:
|
||||
return _ipc_file()
|
||||
|
||||
|
||||
class _UnsupportedNamespaceConfig:
|
||||
async def namespace_client_config(self):
|
||||
raise RuntimeError("UNC namespace client export is not supported")
|
||||
|
||||
|
||||
def _namespace_lance_table(namespace_client: _NamespaceClient) -> LanceTable:
|
||||
table = LanceTable.__new__(LanceTable)
|
||||
table._table = _FailingSyncInner()
|
||||
@@ -138,6 +143,24 @@ class TestNamespaceConnection:
|
||||
db.drop_namespace(["test_ns"])
|
||||
assert "test_ns" not in db.list_namespaces().namespaces
|
||||
|
||||
def test_sync_namespace_client_propagates_export_guard(self, monkeypatch):
|
||||
from lancedb.db import AsyncConnection
|
||||
|
||||
monkeypatch.setattr(
|
||||
"lancedb.namespace.namespace_connect",
|
||||
lambda *_args, **_kwargs: pytest.fail("guarded config was reconstructed"),
|
||||
)
|
||||
db = lancedb.LanceNamespaceDBConnection.__new__(
|
||||
lancedb.LanceNamespaceDBConnection
|
||||
)
|
||||
db._namespace_client = None
|
||||
db._namespace_client_impl = "dir"
|
||||
db._namespace_client_properties = {"root": "file://server/share/database"}
|
||||
db._inner = AsyncConnection(_UnsupportedNamespaceConfig())
|
||||
|
||||
with pytest.raises(RuntimeError, match="UNC namespace client export"):
|
||||
db.namespace_client()
|
||||
|
||||
def test_create_table_through_namespace(self):
|
||||
"""Test creating a table through namespace."""
|
||||
db = lancedb.connect_namespace("dir", {"root": self.temp_dir})
|
||||
@@ -639,6 +662,24 @@ class TestAsyncNamespaceConnection:
|
||||
await db.drop_namespace(["test_ns"])
|
||||
assert "test_ns" not in (await db.list_namespaces()).namespaces
|
||||
|
||||
async def test_async_namespace_client_propagates_export_guard(self, monkeypatch):
|
||||
from lancedb.db import AsyncConnection
|
||||
|
||||
monkeypatch.setattr(
|
||||
"lancedb.namespace.namespace_connect",
|
||||
lambda *_args, **_kwargs: pytest.fail("guarded config was reconstructed"),
|
||||
)
|
||||
db = lancedb.AsyncLanceNamespaceDBConnection.__new__(
|
||||
lancedb.AsyncLanceNamespaceDBConnection
|
||||
)
|
||||
db._namespace_client = None
|
||||
db._namespace_client_impl = "dir"
|
||||
db._namespace_client_properties = {"root": "file://server/share/database"}
|
||||
db._inner = AsyncConnection(_UnsupportedNamespaceConfig())
|
||||
|
||||
with pytest.raises(RuntimeError, match="UNC namespace client export"):
|
||||
await db.namespace_client()
|
||||
|
||||
async def test_async_namespace_client_is_lazy(self):
|
||||
"""namespace_client() should still return the backing client on demand."""
|
||||
pytest.importorskip("lance")
|
||||
|
||||
@@ -2306,36 +2306,3 @@ def test_remote_connection_jobs_surface():
|
||||
assert job.status() == "failed"
|
||||
with pytest.raises(JobFailedError, match="worker died"):
|
||||
job.wait(timeout=timedelta(seconds=5))
|
||||
|
||||
|
||||
def test_remote_add_bases_posts_the_bases_array():
|
||||
captured_body = {}
|
||||
|
||||
def handler(request):
|
||||
if request.path == "/v1/table/test/describe/":
|
||||
request.send_response(200)
|
||||
request.send_header("Content-Type", "application/json")
|
||||
request.end_headers()
|
||||
request.wfile.write(json.dumps(BLOB_DESCRIBE_RESPONSE).encode())
|
||||
elif request.path == "/v1/table/test/bases/":
|
||||
content_len = int(request.headers.get("Content-Length", 0))
|
||||
captured_body.update(json.loads(request.rfile.read(content_len)))
|
||||
request.send_response(200)
|
||||
request.send_header("Content-Type", "application/json")
|
||||
request.end_headers()
|
||||
request.wfile.write(b'{"version": 2}')
|
||||
else:
|
||||
request.send_response(404)
|
||||
request.end_headers()
|
||||
|
||||
with mock_lancedb_connection(handler) as db:
|
||||
table = db.open_table("test")
|
||||
table.add_bases(lancedb.TableBase(path="s3://bucket/media/"))
|
||||
|
||||
assert captured_body["bases"] == [
|
||||
{
|
||||
"path": "s3://bucket/media/",
|
||||
"isDatasetRoot": False,
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
@@ -3854,65 +3854,3 @@ async def test_async_search_runs_embedding_on_dedicated_executor(
|
||||
assert all(name.startswith("lancedb-embedding") for name in captured_threads), (
|
||||
f"embedding ran off the dedicated executor: {captured_threads}"
|
||||
)
|
||||
|
||||
|
||||
def test_computed_column_declare_and_refresh(tmp_path):
|
||||
db = lancedb.connect(tmp_path)
|
||||
table = db.create_table("computed", [{"x": 1}, {"x": 2}])
|
||||
|
||||
table.add_columns(computed={"doubled": "x * 2"})
|
||||
assert table.to_arrow()["doubled"].to_pylist() == [None, None]
|
||||
|
||||
result = table.refresh_column("doubled")
|
||||
assert result.rows_filled == 2
|
||||
assert sorted(table.to_arrow()["doubled"].to_pylist()) == [2, 4]
|
||||
|
||||
table.add([{"x": 5}])
|
||||
assert table.refresh_column("doubled").rows_filled == 1
|
||||
assert sorted(table.to_arrow()["doubled"].to_pylist()) == [2, 4, 10]
|
||||
|
||||
|
||||
def test_computed_column_rejects_transforms_and_computed_together(tmp_path):
|
||||
db = lancedb.connect(tmp_path)
|
||||
table = db.create_table("computed_mixed", [{"x": 1}])
|
||||
with pytest.raises(ValueError):
|
||||
table.add_columns({"a": "x + 1"}, computed={"b": "x * 2"})
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_computed_column_async(tmp_path):
|
||||
db = await lancedb.connect_async(tmp_path)
|
||||
table = await db.create_table("computed_async", [{"x": 3}])
|
||||
|
||||
await table.add_columns(computed={"tripled": "x * 3"})
|
||||
await table.refresh_column("tripled")
|
||||
|
||||
assert (await table.to_arrow())["tripled"].to_pylist() == [9]
|
||||
|
||||
|
||||
def test_refresh_column_async_returns_job(tmp_path):
|
||||
db = lancedb.connect(tmp_path)
|
||||
table = db.create_table("computed_job", [{"x": 1}, {"x": 2}])
|
||||
table.add_columns(computed={"doubled": "x * 2"})
|
||||
|
||||
job = table.refresh_column_async("doubled")
|
||||
assert job.id is None # in-process jobs have no server id
|
||||
job.wait()
|
||||
assert job.status() == "finished"
|
||||
assert sorted(table.to_arrow()["doubled"].to_pylist()) == [2, 4]
|
||||
|
||||
# Bad input raises at the call, not through the job.
|
||||
with pytest.raises(Exception, match="not a computed column"):
|
||||
table.refresh_column_async("x")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_refresh_column_async_job_async_table(tmp_path):
|
||||
db = await lancedb.connect_async(tmp_path)
|
||||
table = await db.create_table("computed_job_async", [{"x": 3}])
|
||||
await table.add_columns(computed={"tripled": "x * 3"})
|
||||
|
||||
job = await table.refresh_column_async("tripled")
|
||||
await job.wait()
|
||||
assert await job.status() == "finished"
|
||||
assert (await table.to_arrow())["tripled"].to_pylist() == [9]
|
||||
|
||||
@@ -346,23 +346,6 @@ impl Connection {
|
||||
})
|
||||
}
|
||||
|
||||
#[pyo3(signature = (name, namespace_path=None))]
|
||||
pub fn drop_table_async(
|
||||
self_: PyRef<'_, Self>,
|
||||
name: String,
|
||||
namespace_path: Option<Vec<String>>,
|
||||
) -> PyResult<Bound<'_, PyAny>> {
|
||||
let inner = self_.get_inner()?.clone();
|
||||
let ns_path = namespace_path.unwrap_or_default();
|
||||
future_into_py(self_.py(), async move {
|
||||
inner
|
||||
.drop_table_async(name, &ns_path)
|
||||
.await
|
||||
.infer_error()
|
||||
.map(crate::job::Job::new)
|
||||
})
|
||||
}
|
||||
|
||||
#[pyo3(signature = (namespace_path=None,))]
|
||||
pub fn drop_all_tables(
|
||||
self_: PyRef<'_, Self>,
|
||||
|
||||
+1
-3
@@ -16,8 +16,7 @@ use query::{FTSQuery, HybridQuery, Query, VectorQuery};
|
||||
use session::Session;
|
||||
use table::{
|
||||
AddColumnsResult, AddResult, AlterColumnsResult, DeleteResult, DropColumnsResult, FtsToken,
|
||||
LsmWriteSpec, MergeResult, PyBlobFile, RefreshColumnResult, Table, UpdateFieldMetadataResult,
|
||||
UpdateResult,
|
||||
LsmWriteSpec, MergeResult, PyBlobFile, Table, UpdateFieldMetadataResult, UpdateResult,
|
||||
};
|
||||
|
||||
pub mod arrow;
|
||||
@@ -58,7 +57,6 @@ pub fn _lancedb(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_class::<VectorQuery>()?;
|
||||
m.add_class::<RecordBatchStream>()?;
|
||||
m.add_class::<AddColumnsResult>()?;
|
||||
m.add_class::<RefreshColumnResult>()?;
|
||||
m.add_class::<AlterColumnsResult>()?;
|
||||
m.add_class::<UpdateFieldMetadataResult>()?;
|
||||
m.add_class::<AddResult>()?;
|
||||
|
||||
@@ -22,7 +22,6 @@ use lancedb::index::scalar::FtsIndexBuilder;
|
||||
use lancedb::table::{
|
||||
AddDataMode, ColumnAlteration, Duration, FieldMetadataUpdate, FtsToken as LanceDbFtsToken,
|
||||
NewColumnTransform, OptimizeAction, OptimizeOptions, Ref, Table as LanceDbTable,
|
||||
TableBase as LanceTableBase,
|
||||
};
|
||||
use lancedb::tokenize as lancedb_tokenize;
|
||||
use pyo3::{
|
||||
@@ -95,13 +94,6 @@ fn lsm_stats_to_py(py: Python<'_>, stats: &lancedb::table::LsmStats) -> PyResult
|
||||
Ok(out.unbind())
|
||||
}
|
||||
|
||||
#[derive(FromPyObject)]
|
||||
pub(crate) struct PyTableBase {
|
||||
path: String,
|
||||
name: Option<String>,
|
||||
is_dataset_root: bool,
|
||||
}
|
||||
|
||||
#[derive(FromPyObject)]
|
||||
enum PredicateArg {
|
||||
Expr(PyExpr),
|
||||
@@ -423,32 +415,6 @@ pub struct AddColumnsResult {
|
||||
pub version: u64,
|
||||
}
|
||||
|
||||
#[pyclass(get_all, from_py_object)]
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct RefreshColumnResult {
|
||||
pub rows_filled: u64,
|
||||
pub version: u64,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl RefreshColumnResult {
|
||||
pub fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"RefreshColumnResult(rows_filled={}, version={})",
|
||||
self.rows_filled, self.version
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
impl From<lancedb::table::RefreshColumnResult> for RefreshColumnResult {
|
||||
fn from(result: lancedb::table::RefreshColumnResult) -> Self {
|
||||
Self {
|
||||
rows_filled: result.rows_filled,
|
||||
version: result.version,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl AddColumnsResult {
|
||||
pub fn __repr__(&self) -> String {
|
||||
@@ -1246,25 +1212,6 @@ impl Table {
|
||||
})
|
||||
}
|
||||
|
||||
#[pyo3(signature = (bases))]
|
||||
pub fn add_bases(
|
||||
self_: PyRef<'_, Self>,
|
||||
bases: Vec<PyTableBase>,
|
||||
) -> PyResult<Bound<'_, PyAny>> {
|
||||
let inner = self_.inner_ref()?.clone();
|
||||
let bases: Vec<LanceTableBase> = bases
|
||||
.into_iter()
|
||||
.map(|base| LanceTableBase {
|
||||
path: base.path,
|
||||
name: base.name,
|
||||
is_dataset_root: base.is_dataset_root,
|
||||
})
|
||||
.collect();
|
||||
future_into_py(self_.py(), async move {
|
||||
inner.add_bases(bases).await.infer_error()
|
||||
})
|
||||
}
|
||||
|
||||
/// Read blob bytes for `row_ids` from blob v2 column `column`.
|
||||
#[pyo3(signature = (column, row_ids))]
|
||||
pub fn fetch_blobs(
|
||||
@@ -1563,40 +1510,6 @@ impl Table {
|
||||
})
|
||||
}
|
||||
|
||||
pub fn add_computed_columns(
|
||||
self_: PyRef<'_, Self>,
|
||||
columns: Vec<(String, String)>,
|
||||
) -> PyResult<Bound<'_, PyAny>> {
|
||||
let inner = self_.inner_ref()?.clone();
|
||||
future_into_py(self_.py(), async move {
|
||||
let mut builder = inner.add_columns();
|
||||
for (name, expression) in columns {
|
||||
builder = builder.computed(name, expression);
|
||||
}
|
||||
let result = builder.execute().await.infer_error()?;
|
||||
Ok(AddColumnsResult::from(result))
|
||||
})
|
||||
}
|
||||
|
||||
pub fn refresh_column(self_: PyRef<'_, Self>, column: String) -> PyResult<Bound<'_, PyAny>> {
|
||||
let inner = self_.inner_ref()?.clone();
|
||||
future_into_py(self_.py(), async move {
|
||||
let result = inner.refresh_column(column).await.infer_error()?;
|
||||
Ok(RefreshColumnResult::from(result))
|
||||
})
|
||||
}
|
||||
|
||||
pub fn refresh_column_async(
|
||||
self_: PyRef<'_, Self>,
|
||||
column: String,
|
||||
) -> PyResult<Bound<'_, PyAny>> {
|
||||
let inner = self_.inner_ref()?.clone();
|
||||
future_into_py(self_.py(), async move {
|
||||
let job = inner.refresh_column_async(column).await.infer_error()?;
|
||||
Ok(crate::job::Job::new(job))
|
||||
})
|
||||
}
|
||||
|
||||
pub fn add_columns_with_schema(
|
||||
self_: PyRef<'_, Self>,
|
||||
schema: PyArrowType<Schema>,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "lancedb"
|
||||
version = "0.38.0-beta.0"
|
||||
version = "0.37.1-beta.1"
|
||||
edition.workspace = true
|
||||
description = "LanceDB: A serverless, low-latency vector database for AI applications"
|
||||
license.workspace = true
|
||||
|
||||
@@ -565,21 +565,6 @@ impl Connection {
|
||||
.await
|
||||
}
|
||||
|
||||
/// Start dropping a table and return a handle to the cleanup job.
|
||||
///
|
||||
/// The table may become unavailable before its physical data is removed.
|
||||
/// Call [`crate::job::Job::wait`] to wait for cleanup to finish. Local
|
||||
/// backends may complete the drop before returning the handle.
|
||||
pub async fn drop_table_async(
|
||||
&self,
|
||||
name: impl AsRef<str>,
|
||||
namespace_path: &[String],
|
||||
) -> Result<crate::job::Job> {
|
||||
self.internal
|
||||
.drop_table_async(name.as_ref(), namespace_path)
|
||||
.await
|
||||
}
|
||||
|
||||
/// Drop the database
|
||||
///
|
||||
/// This is the same as dropping all of the tables
|
||||
|
||||
@@ -323,18 +323,6 @@ pub trait Database:
|
||||
) -> Result<()>;
|
||||
/// Drop a table in the database
|
||||
async fn drop_table(&self, name: &str, namespace_path: &[String]) -> Result<()>;
|
||||
/// Start dropping a table and return a handle to the cleanup job.
|
||||
///
|
||||
/// Backends without asynchronous cleanup complete the drop before
|
||||
/// returning an already-finished job.
|
||||
async fn drop_table_async(
|
||||
&self,
|
||||
name: &str,
|
||||
namespace_path: &[String],
|
||||
) -> Result<crate::job::Job> {
|
||||
self.drop_table(name, namespace_path).await?;
|
||||
Ok(crate::job::Job::new_done())
|
||||
}
|
||||
/// Drop all tables in the database
|
||||
async fn drop_all_tables(&self, namespace_path: &[String]) -> Result<()>;
|
||||
fn as_any(&self) -> &dyn std::any::Any;
|
||||
|
||||
@@ -8,14 +8,23 @@ use std::fs::create_dir_all;
|
||||
use std::path::Path;
|
||||
use std::{collections::HashMap, sync::Arc};
|
||||
|
||||
use lance::dataset::refs::Ref;
|
||||
use lance::dataset::{ReadParams, WriteMode, builder::DatasetBuilder};
|
||||
use lance::dataset::transaction::{Operation, Transaction};
|
||||
use lance::dataset::{
|
||||
CommitBuilder, ReadParams, WriteDestination, WriteMode, builder::DatasetBuilder,
|
||||
};
|
||||
use lance::io::{ObjectStore, ObjectStoreParams, WrappingObjectStore};
|
||||
use lance_datafusion::utils::StreamingWriteSource;
|
||||
use lance_file::version::LanceFileVersion;
|
||||
use lance_io::object_store::{StorageOptionsAccessor, StorageOptionsProvider};
|
||||
use lance_table::io::commit::commit_handler_from_url;
|
||||
use lance_table::{
|
||||
format::{IndexMetadata, Manifest, Transaction as LanceTableTransaction},
|
||||
io::commit::{
|
||||
CommitError, CommitHandler, ManifestLocation, ManifestNamingScheme, ManifestWriter,
|
||||
commit_handler_from_url,
|
||||
},
|
||||
};
|
||||
use object_store::local::LocalFileSystem;
|
||||
use object_store::path::Path as ObjectPath;
|
||||
use snafu::ResultExt;
|
||||
|
||||
use crate::blob::{ensure_blob_storage_version, has_blob_columns};
|
||||
@@ -23,7 +32,9 @@ use crate::connection::ConnectRequest;
|
||||
use crate::database::ReadConsistency;
|
||||
use crate::database::namespace::LanceNamespaceDatabase;
|
||||
use crate::error::{CreateDirSnafu, Error, Result};
|
||||
use crate::io::object_store::MirroringObjectStoreWrapper;
|
||||
#[cfg(windows)]
|
||||
use crate::io::object_store::rooted_file_commit_handler;
|
||||
use crate::io::object_store::{MirroringObjectStoreWrapper, new_default_session};
|
||||
use crate::table::NativeTable;
|
||||
use crate::utils::validate_table_name;
|
||||
|
||||
@@ -45,6 +56,101 @@ pub const OPT_NEW_TABLE_STORAGE_VERSION: &str = "new_table_data_storage_version"
|
||||
pub const OPT_NEW_TABLE_V2_MANIFEST_PATHS: &str = "new_table_enable_v2_manifest_paths";
|
||||
pub const OPT_NEW_TABLE_ENABLE_STABLE_ROW_IDS: &str = "new_table_enable_stable_row_ids";
|
||||
|
||||
fn session_or_default(
|
||||
session: Option<Arc<lance::session::Session>>,
|
||||
) -> (Arc<lance::session::Session>, bool) {
|
||||
match session {
|
||||
Some(session) => (session, false),
|
||||
None => (new_default_session(), true),
|
||||
}
|
||||
}
|
||||
|
||||
fn clone_target_read_params(
|
||||
store_options: ObjectStoreParams,
|
||||
session: Arc<lance::session::Session>,
|
||||
commit_handler: Arc<dyn CommitHandler>,
|
||||
) -> ReadParams {
|
||||
ReadParams {
|
||||
store_options: Some(store_options),
|
||||
session: Some(session),
|
||||
commit_handler: Some(commit_handler),
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
/// Routes clone source metadata through the source commit handler while all
|
||||
/// destination operations continue to use the target handler.
|
||||
#[derive(Debug)]
|
||||
struct CloneCommitHandler {
|
||||
source_base: ObjectPath,
|
||||
source: Arc<dyn CommitHandler>,
|
||||
target: Arc<dyn CommitHandler>,
|
||||
}
|
||||
|
||||
#[async_trait::async_trait]
|
||||
impl CommitHandler for CloneCommitHandler {
|
||||
fn is_version_not_found_definitive(&self) -> bool {
|
||||
self.target.is_version_not_found_definitive()
|
||||
}
|
||||
|
||||
fn propagate_commit_error_after_success(&self) -> bool {
|
||||
self.target.propagate_commit_error_after_success()
|
||||
}
|
||||
|
||||
async fn resolve_latest_location(
|
||||
&self,
|
||||
base_path: &ObjectPath,
|
||||
object_store: &ObjectStore,
|
||||
) -> lance_core::Result<ManifestLocation> {
|
||||
self.target
|
||||
.resolve_latest_location(base_path, object_store)
|
||||
.await
|
||||
}
|
||||
|
||||
async fn resolve_version_location(
|
||||
&self,
|
||||
base_path: &ObjectPath,
|
||||
version: u64,
|
||||
object_store: &dyn object_store::ObjectStore,
|
||||
) -> lance_core::Result<ManifestLocation> {
|
||||
let handler = if base_path == &self.source_base {
|
||||
&self.source
|
||||
} else {
|
||||
&self.target
|
||||
};
|
||||
handler
|
||||
.resolve_version_location(base_path, version, object_store)
|
||||
.await
|
||||
}
|
||||
|
||||
async fn commit(
|
||||
&self,
|
||||
manifest: &mut Manifest,
|
||||
indices: Option<Vec<IndexMetadata>>,
|
||||
base_path: &ObjectPath,
|
||||
object_store: &ObjectStore,
|
||||
manifest_writer: ManifestWriter,
|
||||
naming_scheme: ManifestNamingScheme,
|
||||
transaction: Option<LanceTableTransaction>,
|
||||
) -> std::result::Result<ManifestLocation, CommitError> {
|
||||
self.target
|
||||
.commit(
|
||||
manifest,
|
||||
indices,
|
||||
base_path,
|
||||
object_store,
|
||||
manifest_writer,
|
||||
naming_scheme,
|
||||
transaction,
|
||||
)
|
||||
.await
|
||||
}
|
||||
|
||||
async fn delete(&self, base_path: &ObjectPath) -> lance_core::Result<()> {
|
||||
self.target.delete(base_path).await
|
||||
}
|
||||
}
|
||||
|
||||
/// Controls how new tables should be created
|
||||
#[derive(Clone, Debug, Default)]
|
||||
pub struct NewTableConfig {
|
||||
@@ -355,20 +461,45 @@ impl ListingDatabase {
|
||||
url.to_string()
|
||||
}
|
||||
|
||||
fn rooted_commit_handler_for_uri(uri: &str) -> Option<Arc<dyn CommitHandler>> {
|
||||
#[cfg(windows)]
|
||||
{
|
||||
let is_file = match url::Url::parse(uri) {
|
||||
Ok(url) => url.scheme() == "file" || url.scheme().len() == 1,
|
||||
Err(_) => true,
|
||||
};
|
||||
return is_file.then(rooted_file_commit_handler);
|
||||
}
|
||||
|
||||
#[cfg(not(windows))]
|
||||
{
|
||||
let _ = uri;
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
fn rooted_commit_handler(&self) -> Option<Arc<dyn CommitHandler>> {
|
||||
Self::rooted_commit_handler_for_uri(&self.uri)
|
||||
}
|
||||
|
||||
async fn prepare_namespace_root(
|
||||
uri: &str,
|
||||
storage_options: &HashMap<String, String>,
|
||||
session: Arc<lance::session::Session>,
|
||||
prepare_native_directory: bool,
|
||||
) -> Result<String> {
|
||||
match url::Url::parse(uri) {
|
||||
Ok(url) if url.scheme().len() == 1 && cfg!(windows) => {
|
||||
if prepare_native_directory {
|
||||
Self::try_create_dir(uri).context(CreateDirSnafu { path: uri })?;
|
||||
}
|
||||
let (object_store, _) = ObjectStore::from_uri_and_params(
|
||||
session.store_registry(),
|
||||
uri,
|
||||
&ObjectStoreParams::default(),
|
||||
)
|
||||
.await?;
|
||||
if object_store.is_local() {
|
||||
if object_store.is_local() && prepare_native_directory {
|
||||
Self::try_create_dir(uri).context(CreateDirSnafu { path: uri })?;
|
||||
}
|
||||
Ok(uri.to_string())
|
||||
@@ -399,6 +530,13 @@ impl ListingDatabase {
|
||||
url.set_query(None);
|
||||
let plain_uri = url.to_string();
|
||||
|
||||
#[cfg(windows)]
|
||||
if url.scheme() == "file" && prepare_native_directory {
|
||||
Self::try_create_dir(&plain_uri).context(CreateDirSnafu {
|
||||
path: plain_uri.clone(),
|
||||
})?;
|
||||
}
|
||||
|
||||
let os_params = ObjectStoreParams {
|
||||
storage_options_accessor: if storage_options.is_empty() {
|
||||
None
|
||||
@@ -415,7 +553,7 @@ impl ListingDatabase {
|
||||
&os_params,
|
||||
)
|
||||
.await?;
|
||||
if object_store.is_local() {
|
||||
if object_store.is_local() && prepare_native_directory {
|
||||
Self::try_create_dir(&plain_uri).context(CreateDirSnafu {
|
||||
path: plain_uri.clone(),
|
||||
})?;
|
||||
@@ -424,13 +562,16 @@ impl ListingDatabase {
|
||||
Ok(plain_uri)
|
||||
}
|
||||
Err(_) => {
|
||||
if prepare_native_directory {
|
||||
Self::try_create_dir(uri).context(CreateDirSnafu { path: uri })?;
|
||||
}
|
||||
let (object_store, _) = ObjectStore::from_uri_and_params(
|
||||
session.store_registry(),
|
||||
uri,
|
||||
&ObjectStoreParams::default(),
|
||||
)
|
||||
.await?;
|
||||
if object_store.is_local() {
|
||||
if object_store.is_local() && prepare_native_directory {
|
||||
Self::try_create_dir(uri).context(CreateDirSnafu { path: uri })?;
|
||||
}
|
||||
Ok(uri.to_string())
|
||||
@@ -442,13 +583,14 @@ impl ListingDatabase {
|
||||
request: &ConnectRequest,
|
||||
) -> Result<LanceNamespaceDatabase> {
|
||||
let options = ListingDatabaseOptions::parse_from_map(&request.options)?;
|
||||
let session = request
|
||||
.session
|
||||
.clone()
|
||||
.unwrap_or_else(|| Arc::new(lance::session::Session::default()));
|
||||
let namespace_root =
|
||||
Self::prepare_namespace_root(&request.uri, &options.storage_options, session.clone())
|
||||
.await?;
|
||||
let (session, owns_session) = session_or_default(request.session.clone());
|
||||
let namespace_root = Self::prepare_namespace_root(
|
||||
&request.uri,
|
||||
&options.storage_options,
|
||||
session.clone(),
|
||||
owns_session || !cfg!(windows),
|
||||
)
|
||||
.await?;
|
||||
let ns_properties = Self::build_manifest_enabled_namespace_client_properties(
|
||||
&namespace_root,
|
||||
&options.storage_options,
|
||||
@@ -547,10 +689,14 @@ impl ListingDatabase {
|
||||
url.to_string()
|
||||
};
|
||||
|
||||
let session = request
|
||||
.session
|
||||
.clone()
|
||||
.unwrap_or_else(|| Arc::new(lance::session::Session::default()));
|
||||
let (session, owns_session) = session_or_default(request.session.clone());
|
||||
let prepare_native_directory = owns_session || !cfg!(windows);
|
||||
#[cfg(windows)]
|
||||
if url.scheme() == "file" && prepare_native_directory {
|
||||
Self::try_create_dir(&storage_base_uri).context(CreateDirSnafu {
|
||||
path: storage_base_uri.clone(),
|
||||
})?;
|
||||
}
|
||||
let os_params = ObjectStoreParams {
|
||||
storage_options_accessor: if options.storage_options.is_empty() {
|
||||
None
|
||||
@@ -567,7 +713,7 @@ impl ListingDatabase {
|
||||
&os_params,
|
||||
)
|
||||
.await?;
|
||||
if object_store.is_local() {
|
||||
if object_store.is_local() && prepare_native_directory {
|
||||
Self::try_create_dir(&storage_base_uri).context(CreateDirSnafu {
|
||||
path: storage_base_uri.clone(),
|
||||
})?;
|
||||
@@ -625,14 +771,18 @@ impl ListingDatabase {
|
||||
namespace_client_properties: HashMap<String, String>,
|
||||
session: Option<Arc<lance::session::Session>>,
|
||||
) -> Result<Self> {
|
||||
let session = session.unwrap_or_else(|| Arc::new(lance::session::Session::default()));
|
||||
let (session, owns_session) = session_or_default(session);
|
||||
let prepare_native_directory = owns_session || !cfg!(windows);
|
||||
if prepare_native_directory {
|
||||
Self::try_create_dir(path).context(CreateDirSnafu { path })?;
|
||||
}
|
||||
let (object_store, base_path) = ObjectStore::from_uri_and_params(
|
||||
session.store_registry(),
|
||||
path,
|
||||
&ObjectStoreParams::default(),
|
||||
)
|
||||
.await?;
|
||||
if object_store.is_local() {
|
||||
if object_store.is_local() && prepare_native_directory {
|
||||
Self::try_create_dir(path).context(CreateDirSnafu { path })?;
|
||||
}
|
||||
|
||||
@@ -662,20 +812,16 @@ impl ListingDatabase {
|
||||
|
||||
/// Try to create a local directory to store the lancedb dataset
|
||||
fn try_create_dir(path: &str) -> core::result::Result<(), std::io::Error> {
|
||||
// Strip file:// or file:/ scheme if present to get the actual filesystem path
|
||||
// Note: file:///path becomes file:/path after url.to_string(), so we need to handle both
|
||||
let fs_path = if let Some(stripped) = path.strip_prefix("file://") {
|
||||
// file:///path or file://host/path format
|
||||
stripped
|
||||
} else if let Some(stripped) = path.strip_prefix("file:") {
|
||||
// file:/path format (from url.to_string() on file:///path)
|
||||
// The path after "file:" should already start with "/" for absolute paths
|
||||
stripped
|
||||
} else {
|
||||
path
|
||||
let filesystem_path = match url::Url::parse(path) {
|
||||
Ok(url) if url.scheme() == "file" => url.to_file_path().map_err(|_| {
|
||||
std::io::Error::new(
|
||||
std::io::ErrorKind::InvalidInput,
|
||||
format!("Unable to convert URL '{url}' to a local path"),
|
||||
)
|
||||
})?,
|
||||
_ => Path::new(path).to_path_buf(),
|
||||
};
|
||||
|
||||
let path = Path::new(fs_path);
|
||||
let path = filesystem_path.as_path();
|
||||
if !path.try_exists()? {
|
||||
create_dir_all(path)?;
|
||||
}
|
||||
@@ -733,7 +879,10 @@ impl ListingDatabase {
|
||||
if let Some(query_string) = &self.query_string {
|
||||
uri.push_str(&format!("?{}", query_string));
|
||||
}
|
||||
let commit_handler = commit_handler_from_url(&uri, &Some(object_store_params)).await?;
|
||||
let commit_handler = match self.rooted_commit_handler() {
|
||||
Some(handler) => handler,
|
||||
None => commit_handler_from_url(&uri, &Some(object_store_params)).await?,
|
||||
};
|
||||
for name in names {
|
||||
let dir_name = format!("{}.{}", name, LANCE_EXTENSION);
|
||||
let full_path = self.base_path.clone().join(dir_name.clone());
|
||||
@@ -866,6 +1015,9 @@ impl ListingDatabase {
|
||||
write_params.mode = WriteMode::Overwrite;
|
||||
}
|
||||
|
||||
if write_params.commit_handler.is_none() {
|
||||
write_params.commit_handler = self.rooted_commit_handler();
|
||||
}
|
||||
write_params.session = Some(self.session.clone());
|
||||
|
||||
write_params
|
||||
@@ -1032,7 +1184,6 @@ impl Database for ListingDatabase {
|
||||
};
|
||||
|
||||
Ok(ListTablesResponse {
|
||||
context: None,
|
||||
tables: f,
|
||||
page_token: next_page_token,
|
||||
})
|
||||
@@ -1113,30 +1264,90 @@ impl Database for ListingDatabase {
|
||||
},
|
||||
..Default::default()
|
||||
};
|
||||
let source_commit_handler = Self::rooted_commit_handler_for_uri(&request.source_uri);
|
||||
let read_params = ReadParams {
|
||||
store_options: Some(storage_params.clone()),
|
||||
session: Some(self.session.clone()),
|
||||
commit_handler: source_commit_handler.clone(),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let mut source_dataset = DatasetBuilder::from_uri(&request.source_uri)
|
||||
let source_dataset = DatasetBuilder::from_uri(&request.source_uri)
|
||||
.with_read_params(read_params.clone())
|
||||
.load()
|
||||
.await
|
||||
.map_err(|e| -> Error { e.into() })?;
|
||||
|
||||
let version_ref = match (request.source_version, request.source_tag) {
|
||||
(Some(v), None) => Ok(Ref::Version(None, Some(v))),
|
||||
(None, Some(tag)) => Ok(Ref::Tag(tag)),
|
||||
(None, None) => Ok(Ref::Version(None, Some(source_dataset.version().version))),
|
||||
let (ref_name, version_number) = match (request.source_version, request.source_tag) {
|
||||
(Some(version), None) => Ok((None, version)),
|
||||
(None, Some(tag)) => {
|
||||
let tag = source_dataset.tags().get(&tag).await?;
|
||||
Ok((tag.branch, tag.version))
|
||||
}
|
||||
(None, None) => Ok((None, source_dataset.version().version)),
|
||||
_ => Err(Error::InvalidInput {
|
||||
message: "Cannot specify both source_version and source_tag".to_string(),
|
||||
}),
|
||||
}?;
|
||||
|
||||
let target_uri = self.table_uri(&request.target_table_name)?;
|
||||
source_dataset
|
||||
.shallow_clone(&target_uri, version_ref, Some(storage_params))
|
||||
let source_location = source_dataset
|
||||
.branch_location()
|
||||
.find_branch(ref_name.as_deref())?;
|
||||
let (source_store, source_base) = ObjectStore::from_uri_and_params(
|
||||
self.session.store_registry(),
|
||||
&source_location.uri,
|
||||
&storage_params,
|
||||
)
|
||||
.await?;
|
||||
let (target_store, _) = ObjectStore::from_uri_and_params(
|
||||
self.session.store_registry(),
|
||||
&target_uri,
|
||||
&storage_params,
|
||||
)
|
||||
.await?;
|
||||
let source_commit_handler = match source_commit_handler {
|
||||
Some(handler) => handler,
|
||||
None => {
|
||||
commit_handler_from_url(&source_location.uri, &Some(storage_params.clone())).await?
|
||||
}
|
||||
};
|
||||
let target_commit_handler = match self.rooted_commit_handler() {
|
||||
Some(handler) => handler,
|
||||
None => commit_handler_from_url(&target_uri, &Some(storage_params.clone())).await?,
|
||||
};
|
||||
let target_read_params = clone_target_read_params(
|
||||
storage_params.clone(),
|
||||
self.session.clone(),
|
||||
target_commit_handler.clone(),
|
||||
);
|
||||
let clone_commit_handler = Arc::new(CloneCommitHandler {
|
||||
source_base,
|
||||
source: source_commit_handler,
|
||||
target: target_commit_handler,
|
||||
});
|
||||
let clone_op = Operation::Clone {
|
||||
is_shallow: true,
|
||||
ref_name,
|
||||
ref_version: version_number,
|
||||
ref_path: source_location.uri,
|
||||
branch_name: None,
|
||||
};
|
||||
let transaction = Transaction::new(version_number, clone_op, None);
|
||||
CommitBuilder::new(WriteDestination::Uri(&target_uri))
|
||||
.with_store_params(storage_params)
|
||||
.with_object_store(target_store)
|
||||
.with_source_store(source_store)
|
||||
.with_commit_handler(clone_commit_handler)
|
||||
.with_session(self.session.clone())
|
||||
.with_storage_format(
|
||||
source_dataset
|
||||
.manifest
|
||||
.data_storage_format
|
||||
.lance_file_format()
|
||||
.to_selector(),
|
||||
)
|
||||
.execute(transaction)
|
||||
.await
|
||||
.map_err(|e| -> Error { e.into() })?;
|
||||
|
||||
@@ -1145,7 +1356,7 @@ impl Database for ListingDatabase {
|
||||
&request.target_table_name,
|
||||
request.target_namespace_path,
|
||||
self.store_wrapper.clone(),
|
||||
None,
|
||||
Some(target_read_params),
|
||||
self.read_consistency_interval,
|
||||
request.namespace_client,
|
||||
HashSet::new(), // listing database doesn't support server-side queries
|
||||
@@ -1211,6 +1422,9 @@ impl Database for ListingDatabase {
|
||||
}
|
||||
default_params
|
||||
});
|
||||
if read_params.commit_handler.is_none() {
|
||||
read_params.commit_handler = self.rooted_commit_handler();
|
||||
}
|
||||
read_params.session(self.session.clone());
|
||||
|
||||
let native_table = Arc::new(
|
||||
@@ -1308,6 +1522,60 @@ mod tests {
|
||||
use tokio::sync::Barrier;
|
||||
use tokio::time::timeout;
|
||||
|
||||
#[derive(Debug)]
|
||||
struct TargetOnlyCommitHandler;
|
||||
|
||||
#[async_trait::async_trait]
|
||||
impl CommitHandler for TargetOnlyCommitHandler {
|
||||
async fn resolve_version_location(
|
||||
&self,
|
||||
_base_path: &ObjectPath,
|
||||
_version: u64,
|
||||
_object_store: &dyn object_store::ObjectStore,
|
||||
) -> lance_core::Result<ManifestLocation> {
|
||||
Err(lance_core::Error::invalid_input(
|
||||
"target commit handler was used for the source manifest",
|
||||
))
|
||||
}
|
||||
|
||||
async fn commit(
|
||||
&self,
|
||||
manifest: &mut Manifest,
|
||||
indices: Option<Vec<IndexMetadata>>,
|
||||
base_path: &ObjectPath,
|
||||
object_store: &ObjectStore,
|
||||
manifest_writer: ManifestWriter,
|
||||
naming_scheme: ManifestNamingScheme,
|
||||
transaction: Option<LanceTableTransaction>,
|
||||
) -> std::result::Result<ManifestLocation, CommitError> {
|
||||
lance_table::io::commit::ConditionalPutCommitHandler
|
||||
.commit(
|
||||
manifest,
|
||||
indices,
|
||||
base_path,
|
||||
object_store,
|
||||
manifest_writer,
|
||||
naming_scheme,
|
||||
transaction,
|
||||
)
|
||||
.await
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn commit_handler_forwards_commit_outcome_capabilities() {
|
||||
let target: Arc<dyn CommitHandler> =
|
||||
Arc::new(lance_table::io::commit::ConditionalPutCommitHandler);
|
||||
let handler = CloneCommitHandler {
|
||||
source_base: ObjectPath::from("source"),
|
||||
source: target.clone(),
|
||||
target,
|
||||
};
|
||||
|
||||
assert!(handler.is_version_not_found_definitive());
|
||||
assert!(!handler.propagate_commit_error_after_success());
|
||||
}
|
||||
|
||||
async fn setup_database() -> (tempfile::TempDir, ListingDatabase) {
|
||||
let tempdir = tempdir().unwrap();
|
||||
let uri = tempdir.path().to_str().unwrap();
|
||||
@@ -1330,6 +1598,199 @@ mod tests {
|
||||
(tempdir, db)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn preserves_caller_owned_session_provider() {
|
||||
let registry = Arc::new(lance_io::object_store::ObjectStoreRegistry::default());
|
||||
let provider = registry.get_provider("file").unwrap();
|
||||
let session = Arc::new(lance::session::Session::new(16, 16, registry.clone()));
|
||||
|
||||
let (selected, owns_session) = session_or_default(Some(session.clone()));
|
||||
|
||||
assert!(!owns_session);
|
||||
assert!(Arc::ptr_eq(&selected, &session));
|
||||
assert!(Arc::ptr_eq(
|
||||
®istry.get_provider("file").unwrap(),
|
||||
&provider
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn clone_target_read_params_use_the_target_commit_handler() {
|
||||
let session = Arc::new(lance::session::Session::default());
|
||||
let target_handler: Arc<dyn CommitHandler> = Arc::new(TargetOnlyCommitHandler);
|
||||
|
||||
let params = clone_target_read_params(
|
||||
ObjectStoreParams::default(),
|
||||
session.clone(),
|
||||
target_handler.clone(),
|
||||
);
|
||||
|
||||
assert!(Arc::ptr_eq(params.session.as_ref().unwrap(), &session));
|
||||
assert!(Arc::ptr_eq(
|
||||
params.commit_handler.as_ref().unwrap(),
|
||||
&target_handler
|
||||
));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_listing_database_with_prefixed_file_store() {
|
||||
let tempdir = tempdir().unwrap();
|
||||
let uri = tempdir.path().to_str().unwrap();
|
||||
let session = crate::io::object_store::new_prefixed_file_session();
|
||||
|
||||
let request = ConnectRequest {
|
||||
uri: uri.to_string(),
|
||||
#[cfg(feature = "remote")]
|
||||
client_config: Default::default(),
|
||||
options: Default::default(),
|
||||
namespace_client_properties: Default::default(),
|
||||
manifest_enabled: false,
|
||||
read_consistency_interval: None,
|
||||
session: Some(session),
|
||||
};
|
||||
let db = ListingDatabase::connect_with_options(&request)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let schema = Arc::new(Schema::new(vec![Field::new("id", DataType::Int32, false)]));
|
||||
let batch =
|
||||
RecordBatch::try_new(schema, vec![Arc::new(Int32Array::from(vec![1, 2, 3]))]).unwrap();
|
||||
db.create_table(CreateTableRequest {
|
||||
name: "test".to_string(),
|
||||
namespace_path: vec![],
|
||||
data: Box::new(batch),
|
||||
mode: CreateTableMode::Create,
|
||||
write_options: Default::default(),
|
||||
location: None,
|
||||
namespace_client: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
#[allow(deprecated)]
|
||||
let table_names = db.table_names(TableNamesRequest::default()).await.unwrap();
|
||||
assert_eq!(table_names, vec!["test"]);
|
||||
|
||||
let table = db
|
||||
.open_table(OpenTableRequest {
|
||||
name: "test".to_string(),
|
||||
namespace_path: vec![],
|
||||
index_cache_size: None,
|
||||
lance_read_params: None,
|
||||
location: None,
|
||||
namespace_client: None,
|
||||
managed_versioning: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(table.count_rows(None).await.unwrap(), 3);
|
||||
drop(table);
|
||||
|
||||
db.drop_table("test", &[]).await.unwrap();
|
||||
assert!(!tempdir.path().join("test.lance").exists());
|
||||
assert!(matches!(
|
||||
db.drop_table("test", &[]).await,
|
||||
Err(Error::TableNotFound { .. })
|
||||
));
|
||||
#[allow(deprecated)]
|
||||
let table_names = db.table_names(TableNamesRequest::default()).await.unwrap();
|
||||
assert!(table_names.is_empty());
|
||||
let reopened = db
|
||||
.open_table(OpenTableRequest {
|
||||
name: "test".to_string(),
|
||||
namespace_path: vec![],
|
||||
index_cache_size: None,
|
||||
lance_read_params: None,
|
||||
location: None,
|
||||
namespace_client: None,
|
||||
managed_versioning: None,
|
||||
})
|
||||
.await;
|
||||
assert!(matches!(reopened, Err(Error::TableNotFound { .. })));
|
||||
}
|
||||
|
||||
#[cfg(windows)]
|
||||
#[tokio::test]
|
||||
async fn reopens_a_table_through_a_unc_authority() {
|
||||
use std::path::{Component, Prefix};
|
||||
|
||||
let tempdir = tempdir().unwrap();
|
||||
let drive = match tempdir.path().components().next() {
|
||||
Some(Component::Prefix(prefix)) => match prefix.kind() {
|
||||
Prefix::Disk(drive) | Prefix::VerbatimDisk(drive) => drive,
|
||||
other => panic!("expected a drive-backed temporary directory, got {other:?}"),
|
||||
},
|
||||
other => panic!("expected an absolute Windows temporary directory, got {other:?}"),
|
||||
};
|
||||
let drive_root = PathBuf::from(format!("{}:\\", drive as char));
|
||||
let relative = tempdir.path().strip_prefix(&drive_root).unwrap();
|
||||
// `url` treats `localhost` as an empty file-URL authority, turning
|
||||
// `file://localhost/C$/...` into the invalid drive-relative
|
||||
// `file:///C$/...`. Use the machine's actual network name so this
|
||||
// remains a genuine UNC URL throughout the connection lifecycle.
|
||||
let Some(computer_name) = std::env::var_os("COMPUTERNAME") else {
|
||||
eprintln!("skipping UNC lifecycle test because COMPUTERNAME is unavailable");
|
||||
return;
|
||||
};
|
||||
let unc_path = PathBuf::from(format!(
|
||||
r"\\{}\{}$",
|
||||
computer_name.to_string_lossy(),
|
||||
drive as char
|
||||
))
|
||||
.join(relative);
|
||||
if !unc_path.try_exists().unwrap_or(false) {
|
||||
eprintln!("skipping UNC lifecycle test because the local admin share is unavailable");
|
||||
return;
|
||||
}
|
||||
let uri = url::Url::from_directory_path(&unc_path)
|
||||
.unwrap()
|
||||
.to_string();
|
||||
let request = ConnectRequest {
|
||||
uri,
|
||||
#[cfg(feature = "remote")]
|
||||
client_config: Default::default(),
|
||||
options: Default::default(),
|
||||
namespace_client_properties: Default::default(),
|
||||
manifest_enabled: false,
|
||||
read_consistency_interval: None,
|
||||
session: None,
|
||||
};
|
||||
let db = ListingDatabase::connect_with_options(&request)
|
||||
.await
|
||||
.unwrap();
|
||||
let schema = Arc::new(Schema::new(vec![Field::new("id", DataType::Int32, false)]));
|
||||
let table = db
|
||||
.create_table(CreateTableRequest {
|
||||
name: "unc_reopen".to_string(),
|
||||
namespace_path: vec![],
|
||||
data: Box::new(
|
||||
RecordBatch::try_new(schema, vec![Arc::new(Int32Array::from(vec![1, 2, 3]))])
|
||||
.unwrap(),
|
||||
),
|
||||
mode: CreateTableMode::Create,
|
||||
write_options: Default::default(),
|
||||
location: None,
|
||||
namespace_client: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
drop(table);
|
||||
|
||||
let reopened = db
|
||||
.open_table(OpenTableRequest {
|
||||
name: "unc_reopen".to_string(),
|
||||
namespace_path: vec![],
|
||||
index_cache_size: None,
|
||||
lance_read_params: None,
|
||||
location: None,
|
||||
namespace_client: None,
|
||||
managed_versioning: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(reopened.count_rows(None).await.unwrap(), 3);
|
||||
}
|
||||
|
||||
struct BarrierScannable {
|
||||
batch: RecordBatch,
|
||||
barrier: Arc<Barrier>,
|
||||
@@ -1750,6 +2211,173 @@ mod tests {
|
||||
assert_eq!(source_count, 3);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_clone_table_across_databases_uses_target_store() {
|
||||
let source_dir = tempdir().unwrap();
|
||||
let target_dir = tempdir().unwrap();
|
||||
let source_request = ConnectRequest {
|
||||
uri: source_dir.path().to_string_lossy().into_owned(),
|
||||
#[cfg(feature = "remote")]
|
||||
client_config: Default::default(),
|
||||
options: Default::default(),
|
||||
namespace_client_properties: Default::default(),
|
||||
manifest_enabled: false,
|
||||
read_consistency_interval: None,
|
||||
session: None,
|
||||
};
|
||||
let target_request = ConnectRequest {
|
||||
uri: target_dir.path().to_string_lossy().into_owned(),
|
||||
..source_request.clone()
|
||||
};
|
||||
let source_db = ListingDatabase::connect_with_options(&source_request)
|
||||
.await
|
||||
.unwrap();
|
||||
let target_db = ListingDatabase::connect_with_options(&target_request)
|
||||
.await
|
||||
.unwrap();
|
||||
let schema = Arc::new(Schema::new(vec![Field::new("id", DataType::Int32, false)]));
|
||||
source_db
|
||||
.create_table(CreateTableRequest {
|
||||
name: "source".to_string(),
|
||||
namespace_path: vec![],
|
||||
data: Box::new(
|
||||
RecordBatch::try_new(schema, vec![Arc::new(Int32Array::from(vec![1, 2, 3]))])
|
||||
.unwrap(),
|
||||
),
|
||||
mode: CreateTableMode::Create,
|
||||
write_options: Default::default(),
|
||||
location: None,
|
||||
namespace_client: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let target = target_db
|
||||
.clone_table(CloneTableRequest {
|
||||
target_table_name: "target".to_string(),
|
||||
target_namespace_path: vec![],
|
||||
source_uri: source_db.table_uri("source").unwrap(),
|
||||
source_version: None,
|
||||
source_tag: None,
|
||||
is_shallow: true,
|
||||
namespace_client: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(target.count_rows(None).await.unwrap(), 3);
|
||||
assert!(target_dir.path().join("target.lance").exists());
|
||||
assert!(!source_dir.path().join("target.lance").exists());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn clone_routes_source_manifest_through_source_commit_handler() {
|
||||
let source_dir = tempdir().unwrap();
|
||||
let target_dir = tempdir().unwrap();
|
||||
let source_request = ConnectRequest {
|
||||
uri: source_dir.path().to_string_lossy().into_owned(),
|
||||
#[cfg(feature = "remote")]
|
||||
client_config: Default::default(),
|
||||
options: Default::default(),
|
||||
namespace_client_properties: Default::default(),
|
||||
manifest_enabled: false,
|
||||
read_consistency_interval: None,
|
||||
session: None,
|
||||
};
|
||||
let target_request = ConnectRequest {
|
||||
uri: target_dir.path().to_string_lossy().into_owned(),
|
||||
..source_request.clone()
|
||||
};
|
||||
let source_db = ListingDatabase::connect_with_options(&source_request)
|
||||
.await
|
||||
.unwrap();
|
||||
let target_db = ListingDatabase::connect_with_options(&target_request)
|
||||
.await
|
||||
.unwrap();
|
||||
let schema = Arc::new(Schema::new(vec![Field::new("id", DataType::Int32, false)]));
|
||||
source_db
|
||||
.create_table(CreateTableRequest {
|
||||
name: "source".to_string(),
|
||||
namespace_path: vec![],
|
||||
data: Box::new(
|
||||
RecordBatch::try_new(schema, vec![Arc::new(Int32Array::from(vec![1, 2, 3]))])
|
||||
.unwrap(),
|
||||
),
|
||||
mode: CreateTableMode::Create,
|
||||
write_options: Default::default(),
|
||||
location: None,
|
||||
namespace_client: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let source_uri = source_db.table_uri("source").unwrap();
|
||||
let target_uri = target_db.table_uri("target").unwrap();
|
||||
let storage_params = ObjectStoreParams::default();
|
||||
let source_dataset = DatasetBuilder::from_uri(&source_uri)
|
||||
.with_read_params(ReadParams {
|
||||
session: Some(target_db.session.clone()),
|
||||
..Default::default()
|
||||
})
|
||||
.load()
|
||||
.await
|
||||
.unwrap();
|
||||
let source_location = source_dataset.branch_location().find_branch(None).unwrap();
|
||||
let (source_store, source_base) = ObjectStore::from_uri_and_params(
|
||||
target_db.session.store_registry(),
|
||||
&source_location.uri,
|
||||
&storage_params,
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
let (target_store, _) = ObjectStore::from_uri_and_params(
|
||||
target_db.session.store_registry(),
|
||||
&target_uri,
|
||||
&storage_params,
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
let source_handler = commit_handler_from_url(&source_location.uri, &None)
|
||||
.await
|
||||
.unwrap();
|
||||
let clone_handler = Arc::new(CloneCommitHandler {
|
||||
source_base,
|
||||
source: source_handler,
|
||||
target: Arc::new(TargetOnlyCommitHandler),
|
||||
});
|
||||
let version = source_dataset.version().version;
|
||||
let transaction = Transaction::new(
|
||||
version,
|
||||
Operation::Clone {
|
||||
is_shallow: true,
|
||||
ref_name: None,
|
||||
ref_version: version,
|
||||
ref_path: source_location.uri,
|
||||
branch_name: None,
|
||||
},
|
||||
None,
|
||||
);
|
||||
|
||||
CommitBuilder::new(WriteDestination::Uri(&target_uri))
|
||||
.with_store_params(storage_params)
|
||||
.with_object_store(target_store)
|
||||
.with_source_store(source_store)
|
||||
.with_commit_handler(clone_handler)
|
||||
.with_session(target_db.session.clone())
|
||||
.with_storage_format(
|
||||
source_dataset
|
||||
.manifest
|
||||
.data_storage_format
|
||||
.lance_file_format()
|
||||
.to_selector(),
|
||||
)
|
||||
.execute(transaction)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
assert!(target_dir.path().join("target.lance").exists());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_clone_table_with_storage_options() {
|
||||
let tempdir = tempdir().unwrap();
|
||||
@@ -2654,18 +3282,14 @@ mod tests {
|
||||
/// as `path=<base table>` + `query=/_mem_wal/...`, causing
|
||||
/// `Dataset::write` to find the base table dataset and falsely
|
||||
/// report `Dataset already exists`.
|
||||
///
|
||||
/// Skipped on Windows: `try_create_dir` does not understand
|
||||
/// `file:///C:/…` paths so `connect_with_options` fails before
|
||||
/// even reaching the URL-mutation logic. The pure URL-mutation
|
||||
/// invariant is covered by
|
||||
/// `test_capture_query_treats_empty_as_none` above, which runs
|
||||
/// on all platforms.
|
||||
#[cfg(not(windows))]
|
||||
#[tokio::test]
|
||||
async fn test_table_uri_url_path_has_no_trailing_question_mark() {
|
||||
let tempdir = tempdir().unwrap();
|
||||
let uri = format!("file://{}", tempdir.path().to_str().unwrap());
|
||||
let uri = url::Url::from_directory_path(tempdir.path())
|
||||
.unwrap()
|
||||
.to_string()
|
||||
.trim_end_matches('/')
|
||||
.to_string();
|
||||
|
||||
let request = ConnectRequest {
|
||||
uri: uri.clone(),
|
||||
|
||||
@@ -59,6 +59,46 @@ fn is_table_already_exists_namespace_error(err: &lance::Error) -> bool {
|
||||
/// via the `delimiter` property.
|
||||
const DEFAULT_NAMESPACE_DELIMITER: &str = "$";
|
||||
|
||||
#[cfg(any(windows, test))]
|
||||
fn is_unc_root(root: &str) -> bool {
|
||||
let normalized = root.replace('\\', "/");
|
||||
let lowercase = normalized.to_ascii_lowercase();
|
||||
|
||||
if lowercase.starts_with("//?/unc/") {
|
||||
return true;
|
||||
}
|
||||
if lowercase.starts_with("//?/") || lowercase.starts_with("//./") {
|
||||
return false;
|
||||
}
|
||||
if let Some(authority) = normalized.strip_prefix("//") {
|
||||
return !authority.is_empty() && !authority.starts_with('/');
|
||||
}
|
||||
|
||||
normalized.split_once("://").is_some_and(|(scheme, path)| {
|
||||
scheme.eq_ignore_ascii_case("file") && !path.is_empty() && !path.starts_with('/')
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(any(windows, test))]
|
||||
fn is_directory_unc_root(ns_impl: &str, ns_properties: &HashMap<String, String>) -> bool {
|
||||
ns_impl.eq_ignore_ascii_case("dir")
|
||||
&& ns_properties
|
||||
.get("root")
|
||||
.is_some_and(|root| is_unc_root(root))
|
||||
}
|
||||
|
||||
#[cfg(any(windows, test))]
|
||||
#[derive(Debug)]
|
||||
struct UnavailableUncDirectoryNamespace;
|
||||
|
||||
#[cfg(any(windows, test))]
|
||||
#[async_trait]
|
||||
impl LanceNamespace for UnavailableUncDirectoryNamespace {
|
||||
fn namespace_id(&self) -> String {
|
||||
"unavailable-unc-directory".to_string()
|
||||
}
|
||||
}
|
||||
|
||||
/// A database implementation that uses lance-namespace for table management
|
||||
pub struct LanceNamespaceDatabase {
|
||||
namespace: Arc<dyn LanceNamespace>,
|
||||
@@ -92,6 +132,17 @@ fn resolve_delimiter(ns_properties: &HashMap<String, String>) -> String {
|
||||
}
|
||||
|
||||
impl LanceNamespaceDatabase {
|
||||
fn ensure_storage_supported(&self) -> Result<()> {
|
||||
#[cfg(any(windows, test))]
|
||||
if is_directory_unc_root(&self.ns_impl, &self.ns_properties) {
|
||||
return Err(Error::NotSupported {
|
||||
message: "Directory namespace operations are not supported for UNC roots because the namespace manifest resolver does not preserve file URI authorities; use flat root tables or a non-UNC namespace root"
|
||||
.to_string(),
|
||||
});
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn from_namespace_client(
|
||||
namespace_client: Arc<dyn LanceNamespace>,
|
||||
namespace_client_impl: String,
|
||||
@@ -153,6 +204,25 @@ impl LanceNamespaceDatabase {
|
||||
pushdown_operations: HashSet<NamespaceClientPushdownOperation>,
|
||||
new_table_config: NewTableConfig,
|
||||
) -> Result<Self> {
|
||||
#[cfg(any(windows, test))]
|
||||
if is_directory_unc_root(ns_impl, &ns_properties) {
|
||||
let freshness_baselines: FreshnessBaselines = Arc::new(Mutex::new(HashMap::new()));
|
||||
let delimiter = resolve_delimiter(&ns_properties);
|
||||
return Ok(Self {
|
||||
namespace: Arc::new(UnavailableUncDirectoryNamespace),
|
||||
storage_options,
|
||||
read_consistency_interval,
|
||||
session,
|
||||
uri: format!("namespace://{}", ns_impl),
|
||||
pushdown_operations,
|
||||
ns_impl: ns_impl.to_string(),
|
||||
ns_properties,
|
||||
new_table_config,
|
||||
freshness_baselines,
|
||||
delimiter,
|
||||
});
|
||||
}
|
||||
|
||||
let mut builder = ConnectBuilder::new(ns_impl);
|
||||
for (key, value) in ns_properties.clone() {
|
||||
builder = builder.property(key, value);
|
||||
@@ -310,6 +380,7 @@ impl Database for LanceNamespaceDatabase {
|
||||
&self,
|
||||
request: ListNamespacesRequest,
|
||||
) -> Result<ListNamespacesResponse> {
|
||||
self.ensure_storage_supported()?;
|
||||
Ok(self.namespace.list_namespaces(request).await?)
|
||||
}
|
||||
|
||||
@@ -317,10 +388,12 @@ impl Database for LanceNamespaceDatabase {
|
||||
&self,
|
||||
request: CreateNamespaceRequest,
|
||||
) -> Result<CreateNamespaceResponse> {
|
||||
self.ensure_storage_supported()?;
|
||||
Ok(self.namespace.create_namespace(request).await?)
|
||||
}
|
||||
|
||||
async fn drop_namespace(&self, request: DropNamespaceRequest) -> Result<DropNamespaceResponse> {
|
||||
self.ensure_storage_supported()?;
|
||||
Ok(self.namespace.drop_namespace(request).await?)
|
||||
}
|
||||
|
||||
@@ -328,10 +401,12 @@ impl Database for LanceNamespaceDatabase {
|
||||
&self,
|
||||
request: DescribeNamespaceRequest,
|
||||
) -> Result<DescribeNamespaceResponse> {
|
||||
self.ensure_storage_supported()?;
|
||||
Ok(self.namespace.describe_namespace(request).await?)
|
||||
}
|
||||
|
||||
async fn table_names(&self, request: TableNamesRequest) -> Result<Vec<String>> {
|
||||
self.ensure_storage_supported()?;
|
||||
let ns_request = ListTablesRequest {
|
||||
id: Some(request.namespace_path),
|
||||
page_token: request.start_after,
|
||||
@@ -345,10 +420,12 @@ impl Database for LanceNamespaceDatabase {
|
||||
}
|
||||
|
||||
async fn list_tables(&self, request: ListTablesRequest) -> Result<ListTablesResponse> {
|
||||
self.ensure_storage_supported()?;
|
||||
Ok(self.namespace.list_tables(request).await?)
|
||||
}
|
||||
|
||||
async fn create_table(&self, request: DbCreateTableRequest) -> Result<Arc<dyn BaseTable>> {
|
||||
self.ensure_storage_supported()?;
|
||||
let mut table_id = request.namespace_path.clone();
|
||||
table_id.push(request.name.clone());
|
||||
let mut existing_table = None;
|
||||
@@ -518,6 +595,7 @@ impl Database for LanceNamespaceDatabase {
|
||||
}
|
||||
|
||||
async fn open_table(&self, request: OpenTableRequest) -> Result<Arc<dyn BaseTable>> {
|
||||
self.ensure_storage_supported()?;
|
||||
let native_table = NativeTable::open_from_namespace(
|
||||
self.namespace.clone(),
|
||||
&request.name,
|
||||
@@ -547,6 +625,7 @@ impl Database for LanceNamespaceDatabase {
|
||||
cur_namespace_path: &[String],
|
||||
new_namespace_path: &[String],
|
||||
) -> Result<()> {
|
||||
self.ensure_storage_supported()?;
|
||||
let mut cur_table_id = cur_namespace_path.to_vec();
|
||||
cur_table_id.push(cur_name.to_string());
|
||||
|
||||
@@ -573,6 +652,7 @@ impl Database for LanceNamespaceDatabase {
|
||||
}
|
||||
|
||||
async fn drop_table(&self, name: &str, namespace_path: &[String]) -> Result<()> {
|
||||
self.ensure_storage_supported()?;
|
||||
let mut table_id = namespace_path.to_vec();
|
||||
table_id.push(name.to_string());
|
||||
|
||||
@@ -612,10 +692,12 @@ impl Database for LanceNamespaceDatabase {
|
||||
}
|
||||
|
||||
async fn namespace_client(&self) -> Result<Arc<dyn LanceNamespace>> {
|
||||
self.ensure_storage_supported()?;
|
||||
Ok(self.namespace.clone())
|
||||
}
|
||||
|
||||
async fn namespace_client_config(&self) -> Result<(String, HashMap<String, String>)> {
|
||||
self.ensure_storage_supported()?;
|
||||
Ok((self.ns_impl.clone(), self.ns_properties.clone()))
|
||||
}
|
||||
}
|
||||
@@ -644,6 +726,56 @@ mod tests {
|
||||
RecordBatch::try_new(schema, vec![Arc::new(id_array), Arc::new(name_array)]).unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn detects_unc_namespace_roots_without_misclassifying_drives() {
|
||||
for root in [
|
||||
"file://server/share/database",
|
||||
"FILE://server/share/database",
|
||||
r"\\server\share\database",
|
||||
r"\\?\UNC\server\share\database",
|
||||
] {
|
||||
assert!(is_unc_root(root), "expected an UNC root: {root}");
|
||||
}
|
||||
for root in [
|
||||
"file:///C:/database",
|
||||
r"C:\database",
|
||||
r"\\?\C:\database",
|
||||
"/var/lib/database",
|
||||
] {
|
||||
assert!(!is_unc_root(root), "expected a non-UNC root: {root}");
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn directory_namespace_fails_closed_before_accessing_an_unc_root() {
|
||||
let mut properties = HashMap::new();
|
||||
properties.insert(
|
||||
"root".to_string(),
|
||||
"file://server/share/database".to_string(),
|
||||
);
|
||||
let db = LanceNamespaceDatabase::connect(
|
||||
"dir",
|
||||
properties,
|
||||
HashMap::new(),
|
||||
None,
|
||||
None,
|
||||
HashSet::new(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let error = db
|
||||
.list_tables(ListTablesRequest::default())
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(matches!(error, Error::NotSupported { .. }));
|
||||
assert!(error.to_string().contains("UNC roots"));
|
||||
|
||||
let error = db.namespace_client_config().await.unwrap_err();
|
||||
assert!(matches!(error, Error::NotSupported { .. }));
|
||||
assert!(error.to_string().contains("UNC roots"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_namespace_connection_simple() {
|
||||
// Test that namespace connections work with simple connect_namespace(impl_type, properties)
|
||||
|
||||
@@ -71,14 +71,6 @@ pub enum Error {
|
||||
IndexNotFound { name: String },
|
||||
#[snafu(display("Embedding function '{name}' was not found. : {reason}"))]
|
||||
EmbeddingFunctionNotFound { name: String, reason: String },
|
||||
#[snafu(display("Column '{name}' was not found"))]
|
||||
ColumnNotFound { name: String },
|
||||
#[snafu(display("Column '{name}' already exists"))]
|
||||
ColumnAlreadyExists { name: String },
|
||||
#[snafu(display("Column '{name}' is not a computed column"))]
|
||||
NotAComputedColumn { name: String },
|
||||
#[snafu(display("Invalid expression for column '{column}': {message}"))]
|
||||
InvalidExpression { column: String, message: String },
|
||||
|
||||
#[snafu(display("Table '{name}' already exists"))]
|
||||
TableAlreadyExists { name: String },
|
||||
|
||||
@@ -1,12 +1,22 @@
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||
|
||||
//! A mirroring object store that mirror writes to a secondary object store
|
||||
//! Object-store providers and adapters used by LanceDB.
|
||||
|
||||
use std::{fmt::Formatter, sync::Arc};
|
||||
|
||||
#[cfg(any(windows, test))]
|
||||
use futures::TryStreamExt;
|
||||
use futures::{StreamExt, TryFutureExt, stream::BoxStream};
|
||||
use lance::io::WrappingObjectStore;
|
||||
#[cfg(any(windows, test))]
|
||||
use lance_table::{
|
||||
format::{IndexMetadata, Manifest, Transaction},
|
||||
io::commit::{
|
||||
CommitError, CommitHandler, ManifestLocation, ManifestNamingScheme, ManifestWriter,
|
||||
RenameCommitHandler,
|
||||
},
|
||||
};
|
||||
use object_store::{
|
||||
CopyOptions, Error, GetOptions, GetResult, ListResult, MultipartUpload, ObjectMeta,
|
||||
ObjectStore, ObjectStoreExt, PutMultipartOptions, PutOptions, PutPayload, PutResult, Result,
|
||||
@@ -15,9 +25,797 @@ use object_store::{
|
||||
|
||||
use async_trait::async_trait;
|
||||
|
||||
#[cfg(any(windows, test))]
|
||||
use lance_core::{Error as LanceError, Result as LanceResult};
|
||||
#[cfg(test)]
|
||||
use lance_io::object_store::ObjectStoreRegistry;
|
||||
#[cfg(any(windows, test))]
|
||||
use lance_io::object_store::{
|
||||
DEFAULT_LOCAL_IO_PARALLELISM, ObjectStoreParams, ObjectStoreProvider, StorageOptions,
|
||||
};
|
||||
#[cfg(any(windows, test))]
|
||||
use object_store::local::LocalFileSystem;
|
||||
#[cfg(any(windows, test))]
|
||||
use url::Url;
|
||||
|
||||
#[cfg(test)]
|
||||
pub mod io_tracking;
|
||||
|
||||
/// A local commit handler that resolves the latest manifest through the object store.
|
||||
///
|
||||
/// Lance's native local shortcut reconstructs the selected manifest from a
|
||||
/// filesystem path. On Windows that conversion drops the authority from a UNC
|
||||
/// path. Listing through the already-rooted object store preserves the structural
|
||||
/// server/share prefix while retaining the normal atomic-rename commit behavior.
|
||||
#[derive(Debug)]
|
||||
#[cfg(any(windows, test))]
|
||||
struct RootedFileCommitHandler;
|
||||
|
||||
#[cfg(any(windows, test))]
|
||||
#[async_trait]
|
||||
impl CommitHandler for RootedFileCommitHandler {
|
||||
fn is_version_not_found_definitive(&self) -> bool {
|
||||
RenameCommitHandler.is_version_not_found_definitive()
|
||||
}
|
||||
|
||||
fn propagate_commit_error_after_success(&self) -> bool {
|
||||
RenameCommitHandler.propagate_commit_error_after_success()
|
||||
}
|
||||
|
||||
async fn resolve_latest_location(
|
||||
&self,
|
||||
base_path: &Path,
|
||||
object_store: &lance::io::ObjectStore,
|
||||
) -> LanceResult<ManifestLocation> {
|
||||
self.list_manifest_locations(base_path, object_store, true)
|
||||
.try_next()
|
||||
.await?
|
||||
.ok_or_else(|| LanceError::not_found(base_path.to_string()))
|
||||
}
|
||||
|
||||
async fn commit(
|
||||
&self,
|
||||
manifest: &mut Manifest,
|
||||
indices: Option<Vec<IndexMetadata>>,
|
||||
base_path: &Path,
|
||||
object_store: &lance::io::ObjectStore,
|
||||
manifest_writer: ManifestWriter,
|
||||
naming_scheme: ManifestNamingScheme,
|
||||
transaction: Option<Transaction>,
|
||||
) -> std::result::Result<ManifestLocation, CommitError> {
|
||||
RenameCommitHandler
|
||||
.commit(
|
||||
manifest,
|
||||
indices,
|
||||
base_path,
|
||||
object_store,
|
||||
manifest_writer,
|
||||
naming_scheme,
|
||||
transaction,
|
||||
)
|
||||
.await
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(any(windows, test))]
|
||||
pub(crate) fn rooted_file_commit_handler() -> Arc<dyn CommitHandler> {
|
||||
Arc::new(RootedFileCommitHandler)
|
||||
}
|
||||
|
||||
/// A file-store provider that anchors each request at its filesystem root.
|
||||
///
|
||||
/// On Windows, an unprefixed [`LocalFileSystem`] cannot service UNC paths. Its
|
||||
/// conversion to an object-store [`Path`] drops the UNC host, so subsequent I/O
|
||||
/// is directed at a different local path. Anchoring the store at the drive or
|
||||
/// UNC-share root keeps the UNC authority in the filesystem prefix and exposes
|
||||
/// only paths relative to that prefix to `object_store`.
|
||||
///
|
||||
/// Extracted paths retain the native drive or UNC-share root as a structural
|
||||
/// first component. This keeps Lance's local classification and optimized I/O
|
||||
/// safe without recovering absolute path provenance from ambiguous path text.
|
||||
#[cfg(any(windows, test))]
|
||||
#[derive(Debug)]
|
||||
struct PrefixedFileStoreProvider;
|
||||
|
||||
#[cfg(any(windows, test))]
|
||||
impl PrefixedFileStoreProvider {
|
||||
fn root_and_relative_path(url: &Url) -> LanceResult<(std::path::PathBuf, Path)> {
|
||||
let filesystem_path = url.to_file_path().map_err(|_| {
|
||||
LanceError::invalid_input(format!("Unable to convert URL '{url}' to a local path"))
|
||||
})?;
|
||||
|
||||
let mut root = std::path::PathBuf::new();
|
||||
for component in filesystem_path.components() {
|
||||
match component {
|
||||
std::path::Component::Prefix(_) | std::path::Component::RootDir => {
|
||||
root.push(component.as_os_str());
|
||||
}
|
||||
_ => break,
|
||||
}
|
||||
}
|
||||
if root.as_os_str().is_empty() {
|
||||
return Err(LanceError::invalid_input(format!(
|
||||
"Local path '{}' has no filesystem root",
|
||||
filesystem_path.display()
|
||||
)));
|
||||
}
|
||||
|
||||
let relative = filesystem_path.strip_prefix(&root).map_err(|_| {
|
||||
LanceError::invalid_input(format!(
|
||||
"Local path '{}' is not beneath store root '{}'",
|
||||
filesystem_path.display(),
|
||||
root.display()
|
||||
))
|
||||
})?;
|
||||
let relative = relative
|
||||
.components()
|
||||
.filter_map(|component| match component {
|
||||
std::path::Component::Normal(part) => Some(part),
|
||||
_ => None,
|
||||
})
|
||||
.map(|part| {
|
||||
part.to_str().ok_or_else(|| {
|
||||
LanceError::invalid_input(format!(
|
||||
"Local path '{}' is not valid UTF-8",
|
||||
filesystem_path.display()
|
||||
))
|
||||
})
|
||||
})
|
||||
.collect::<LanceResult<Vec<_>>>()?
|
||||
.join("/");
|
||||
|
||||
Ok((root, Path::parse(relative)?))
|
||||
}
|
||||
|
||||
/// Preserve the native filesystem root as a structural path component.
|
||||
///
|
||||
/// `object_store::Path::from_absolute_path` converts a UNC path to its URL
|
||||
/// path and loses the server. Keeping `C:` or `\\server\share` as the first
|
||||
/// component distinguishes an absolute path from every relative path while
|
||||
/// remaining directly usable by Windows filesystem APIs.
|
||||
fn rooted_path(root: &std::path::Path, relative: &Path) -> LanceResult<Path> {
|
||||
let root = root.to_string_lossy();
|
||||
let root = root.trim_end_matches(['/', '\\']);
|
||||
if root.is_empty() {
|
||||
return Ok(relative.clone());
|
||||
}
|
||||
Ok(relative
|
||||
.parts()
|
||||
.fold(Path::parse(root)?, |path, part| path.join(part)))
|
||||
}
|
||||
}
|
||||
|
||||
/// A local store rooted at a Windows drive or UNC share.
|
||||
///
|
||||
/// Most calls use paths returned by [`PrefixedFileStoreProvider`], which are
|
||||
/// relative to `root`. Some Lance operations retain an existing object store
|
||||
/// while independently re-extracting a Windows file URI with the default file
|
||||
/// provider. Those paths include the drive (`C:/...`) or UNC share
|
||||
/// (`share/...`) again. Normalize that absolute alias before delegating so the
|
||||
/// filesystem prefix is never applied twice.
|
||||
#[cfg(any(windows, test))]
|
||||
#[derive(Debug, Clone)]
|
||||
struct RootedLocalFileSystem {
|
||||
inner: Arc<LocalFileSystem>,
|
||||
root: std::path::PathBuf,
|
||||
absolute_alias: Path,
|
||||
}
|
||||
|
||||
#[cfg(any(windows, test))]
|
||||
impl RootedLocalFileSystem {
|
||||
fn new(root: std::path::PathBuf) -> LanceResult<Self> {
|
||||
let absolute_alias = PrefixedFileStoreProvider::rooted_path(&root, &Path::default())?;
|
||||
Ok(Self {
|
||||
inner: Arc::new(LocalFileSystem::new_with_prefix(&root)?),
|
||||
root,
|
||||
absolute_alias,
|
||||
})
|
||||
}
|
||||
|
||||
fn path_from_parts<'a>(parts: impl Iterator<Item = object_store::path::PathPart<'a>>) -> Path {
|
||||
parts.fold(Path::default(), |path, part| path.join(part))
|
||||
}
|
||||
|
||||
fn normalize(&self, path: &Path) -> Path {
|
||||
if self.absolute_alias.as_ref().is_empty() {
|
||||
return path.clone();
|
||||
}
|
||||
let Some(suffix) = path.prefix_match(&self.absolute_alias) else {
|
||||
return path.clone();
|
||||
};
|
||||
Self::path_from_parts(suffix)
|
||||
}
|
||||
|
||||
fn restore_prefix(&self, path: Path, requested: &Path, normalized: &Path) -> Path {
|
||||
if requested == normalized {
|
||||
return path;
|
||||
}
|
||||
path.prefix_match(normalized)
|
||||
.map(|suffix| suffix.fold(requested.clone(), |path, part| path.join(part)))
|
||||
.unwrap_or(path)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(any(windows, test))]
|
||||
impl std::fmt::Display for RootedLocalFileSystem {
|
||||
fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result {
|
||||
write!(f, "RootedLocalFileSystem({})", self.root.display())
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(any(windows, test))]
|
||||
#[async_trait]
|
||||
impl ObjectStore for RootedLocalFileSystem {
|
||||
async fn put_opts(
|
||||
&self,
|
||||
location: &Path,
|
||||
payload: PutPayload,
|
||||
options: PutOptions,
|
||||
) -> Result<PutResult> {
|
||||
self.inner
|
||||
.put_opts(&self.normalize(location), payload, options)
|
||||
.await
|
||||
}
|
||||
|
||||
async fn put_multipart_opts(
|
||||
&self,
|
||||
location: &Path,
|
||||
options: PutMultipartOptions,
|
||||
) -> Result<Box<dyn MultipartUpload>> {
|
||||
self.inner
|
||||
.put_multipart_opts(&self.normalize(location), options)
|
||||
.await
|
||||
}
|
||||
|
||||
async fn get_opts(&self, location: &Path, options: GetOptions) -> Result<GetResult> {
|
||||
let normalized = self.normalize(location);
|
||||
let mut result = self.inner.get_opts(&normalized, options).await?;
|
||||
result.meta.location = location.clone();
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn delete_stream(
|
||||
&self,
|
||||
locations: BoxStream<'static, Result<Path>>,
|
||||
) -> BoxStream<'static, Result<Path>> {
|
||||
let store = self.clone();
|
||||
locations
|
||||
.map(move |location| {
|
||||
let store = store.clone();
|
||||
async move {
|
||||
let location = location?;
|
||||
let normalized = store.normalize(&location);
|
||||
store.inner.delete(&normalized).await?;
|
||||
Ok(location)
|
||||
}
|
||||
})
|
||||
.buffered(10)
|
||||
.boxed()
|
||||
}
|
||||
|
||||
fn list(&self, prefix: Option<&Path>) -> BoxStream<'static, Result<ObjectMeta>> {
|
||||
let requested = prefix.cloned().unwrap_or_default();
|
||||
let normalized = self.normalize(&requested);
|
||||
let store = self.clone();
|
||||
self.inner
|
||||
.list(prefix.map(|_| &normalized))
|
||||
.map(move |result| {
|
||||
result.map(|mut meta| {
|
||||
meta.location = store.restore_prefix(meta.location, &requested, &normalized);
|
||||
meta
|
||||
})
|
||||
})
|
||||
.boxed()
|
||||
}
|
||||
|
||||
async fn list_with_delimiter(&self, prefix: Option<&Path>) -> Result<ListResult> {
|
||||
let requested = prefix.cloned().unwrap_or_default();
|
||||
let normalized = self.normalize(&requested);
|
||||
let mut result = self
|
||||
.inner
|
||||
.list_with_delimiter(prefix.map(|_| &normalized))
|
||||
.await?;
|
||||
for meta in &mut result.objects {
|
||||
meta.location = self.restore_prefix(meta.location.clone(), &requested, &normalized);
|
||||
}
|
||||
for path in &mut result.common_prefixes {
|
||||
*path = self.restore_prefix(path.clone(), &requested, &normalized);
|
||||
}
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
async fn copy_opts(&self, from: &Path, to: &Path, options: CopyOptions) -> Result<()> {
|
||||
self.inner
|
||||
.copy_opts(&self.normalize(from), &self.normalize(to), options)
|
||||
.await
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(any(windows, test))]
|
||||
#[async_trait]
|
||||
impl ObjectStoreProvider for PrefixedFileStoreProvider {
|
||||
async fn new_store(
|
||||
&self,
|
||||
base_path: Url,
|
||||
params: &ObjectStoreParams,
|
||||
) -> LanceResult<lance::io::ObjectStore> {
|
||||
let (root, _) = Self::root_and_relative_path(&base_path)?;
|
||||
let raw_store: Arc<dyn ObjectStore> = Arc::new(RootedLocalFileSystem::new(root)?);
|
||||
// Native local shortcuts are safe only when the rooted filesystem is
|
||||
// the final store. An arbitrary wrapper can redirect I/O, so use
|
||||
// Lance's non-cloud object-store route in that case. This keeps local
|
||||
// scan planning without enabling native copy/delete or the io_uring
|
||||
// scheduler, all of which would bypass the wrapper.
|
||||
let location = if params.object_store_wrapper.is_some() {
|
||||
Url::parse("memory:///").expect("static URL must be valid")
|
||||
} else {
|
||||
Url::parse("file:///").expect("static URL must be valid")
|
||||
};
|
||||
let storage_options =
|
||||
StorageOptions::new(params.storage_options().cloned().unwrap_or_default());
|
||||
|
||||
// ObjectStore::new initializes the private local-store fields. The
|
||||
// registry owns tracing, custom wrapper, and I/O tracker installation,
|
||||
// so restore the raw store before returning to avoid applying them twice.
|
||||
let mut store = lance::io::ObjectStore::new(
|
||||
raw_store.clone(),
|
||||
location,
|
||||
Some(params.block_size.unwrap_or(4 * 1024)),
|
||||
None,
|
||||
false,
|
||||
false,
|
||||
DEFAULT_LOCAL_IO_PARALLELISM,
|
||||
storage_options.download_retry_count(),
|
||||
params.storage_options(),
|
||||
);
|
||||
store.inner = raw_store;
|
||||
store.store_prefix =
|
||||
self.calculate_object_store_prefix(&base_path, params.storage_options())?;
|
||||
Ok(store)
|
||||
}
|
||||
|
||||
fn extract_path(&self, url: &Url) -> LanceResult<Path> {
|
||||
let (root, relative) = Self::root_and_relative_path(url)?;
|
||||
Self::rooted_path(&root, &relative)
|
||||
}
|
||||
|
||||
fn calculate_object_store_prefix(
|
||||
&self,
|
||||
url: &Url,
|
||||
_storage_options: Option<&std::collections::HashMap<String, String>>,
|
||||
) -> LanceResult<String> {
|
||||
let (root, _) = Self::root_and_relative_path(url)?;
|
||||
let root = root.canonicalize()?;
|
||||
// One store per drive or UNC share keeps registry and metrics
|
||||
// cardinality bounded. Absolute-vs-relative provenance is carried by
|
||||
// the extracted path instead of the cache key.
|
||||
Ok(format!("file${}", root.display()))
|
||||
}
|
||||
}
|
||||
|
||||
/// Build LanceDB's default session with the Windows file fallback installed.
|
||||
///
|
||||
/// Callers that provide a Session retain its registry unchanged.
|
||||
pub(crate) fn new_default_session() -> Arc<lance::session::Session> {
|
||||
let session = Arc::new(lance::session::Session::default());
|
||||
#[cfg(windows)]
|
||||
session
|
||||
.store_registry()
|
||||
.insert("file", Arc::new(PrefixedFileStoreProvider));
|
||||
session
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) fn new_prefixed_file_session() -> Arc<lance::session::Session> {
|
||||
let session = Arc::new(lance::session::Session::default());
|
||||
session
|
||||
.store_registry()
|
||||
.insert("file", Arc::new(PrefixedFileStoreProvider));
|
||||
session
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod prefixed_file_store_test {
|
||||
use super::*;
|
||||
use object_store::memory::InMemory;
|
||||
use std::sync::atomic::{AtomicUsize, Ordering};
|
||||
|
||||
#[derive(Debug, Default)]
|
||||
struct CountingWrapper {
|
||||
calls: AtomicUsize,
|
||||
}
|
||||
|
||||
impl WrappingObjectStore for CountingWrapper {
|
||||
fn wrap(
|
||||
&self,
|
||||
_store_prefix: &str,
|
||||
original: Arc<dyn ObjectStore>,
|
||||
) -> Arc<dyn ObjectStore> {
|
||||
self.calls.fetch_add(1, Ordering::Relaxed);
|
||||
original
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Default)]
|
||||
struct MemoryRedirectWrapper {
|
||||
store: Arc<InMemory>,
|
||||
}
|
||||
|
||||
impl WrappingObjectStore for MemoryRedirectWrapper {
|
||||
fn wrap(
|
||||
&self,
|
||||
_store_prefix: &str,
|
||||
_original: Arc<dyn ObjectStore>,
|
||||
) -> Arc<dyn ObjectStore> {
|
||||
self.store.clone()
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn commit_handler_forwards_commit_outcome_capabilities() {
|
||||
let handler = RootedFileCommitHandler;
|
||||
|
||||
assert!(handler.is_version_not_found_definitive());
|
||||
assert!(!handler.propagate_commit_error_after_success());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn anchors_new_and_existing_directories_at_a_filesystem_prefix() {
|
||||
let tempdir = tempfile::tempdir().unwrap();
|
||||
let database_path = tempdir.path().join("database");
|
||||
std::fs::create_dir(&database_path).unwrap();
|
||||
let table_path = database_path.join("test.lance");
|
||||
let table_url = Url::from_directory_path(&table_path).unwrap();
|
||||
|
||||
let registry = Arc::new(ObjectStoreRegistry::default());
|
||||
registry.insert("file", Arc::new(PrefixedFileStoreProvider));
|
||||
|
||||
// The extracted path remains absolute for Lance's local fast paths,
|
||||
// while the inner store strips the structural root before delegation.
|
||||
let (store, base_path) = lance::io::ObjectStore::from_uri_and_params(
|
||||
registry.clone(),
|
||||
table_url.as_str(),
|
||||
&ObjectStoreParams::default(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(store.scheme(), "file");
|
||||
assert!(store.is_local());
|
||||
assert!(!store.is_cloud());
|
||||
assert_eq!(store.block_size(), 4 * 1024);
|
||||
assert_eq!(store.io_parallelism(), DEFAULT_LOCAL_IO_PARALLELISM);
|
||||
assert_eq!(base_path.filename(), Some("test.lance"));
|
||||
let initial_base_path = base_path.clone();
|
||||
|
||||
let marker = base_path.join("marker");
|
||||
store
|
||||
.inner
|
||||
.put(&marker, bytes::Bytes::from_static(b"new").into())
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(std::fs::read(table_path.join("marker")).unwrap(), b"new");
|
||||
|
||||
// Once the table directory exists, a fresh store uses the same stable
|
||||
// root and object-store path.
|
||||
drop(store);
|
||||
let (store, base_path) = lance::io::ObjectStore::from_uri_and_params(
|
||||
registry,
|
||||
table_url.as_str(),
|
||||
&ObjectStoreParams::default(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(base_path, initial_base_path);
|
||||
let contents = store
|
||||
.inner
|
||||
.get(&base_path.join("marker"))
|
||||
.await
|
||||
.unwrap()
|
||||
.bytes()
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(contents.as_ref(), b"new");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn applies_wrapper_and_io_tracking_once() {
|
||||
let tempdir = tempfile::tempdir().unwrap();
|
||||
let table_url = Url::from_directory_path(tempdir.path().join("test.lance")).unwrap();
|
||||
let registry = Arc::new(ObjectStoreRegistry::default());
|
||||
registry.insert("file", Arc::new(PrefixedFileStoreProvider));
|
||||
let wrapper = Arc::new(CountingWrapper::default());
|
||||
let params = ObjectStoreParams {
|
||||
object_store_wrapper: Some(wrapper.clone()),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let (store, base_path) =
|
||||
lance::io::ObjectStore::from_uri_and_params(registry, table_url.as_str(), ¶ms)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(wrapper.calls.load(Ordering::Relaxed), 1);
|
||||
assert_eq!(store.scheme(), "memory");
|
||||
assert!(!store.is_local());
|
||||
assert!(!store.is_cloud());
|
||||
assert!(!store.prefers_lite_scheduler());
|
||||
|
||||
store
|
||||
.inner
|
||||
.put(
|
||||
&base_path.join("marker"),
|
||||
bytes::Bytes::from_static(b"tracked").into(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
let stats = store.io_tracker().stats();
|
||||
assert_eq!(stats.write_iops, 1);
|
||||
assert_eq!(stats.written_bytes, 7);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn wrapped_store_cleanup_does_not_bypass_the_wrapper() {
|
||||
let tempdir = tempfile::tempdir().unwrap();
|
||||
let table_path = tempdir.path().join("test.lance");
|
||||
std::fs::create_dir(&table_path).unwrap();
|
||||
std::fs::write(table_path.join("native-marker"), b"native").unwrap();
|
||||
|
||||
let table_url = Url::from_directory_path(&table_path).unwrap();
|
||||
let registry = Arc::new(ObjectStoreRegistry::default());
|
||||
registry.insert("file", Arc::new(PrefixedFileStoreProvider));
|
||||
let wrapper = Arc::new(MemoryRedirectWrapper::default());
|
||||
let params = ObjectStoreParams {
|
||||
object_store_wrapper: Some(wrapper.clone()),
|
||||
..Default::default()
|
||||
};
|
||||
let (store, base_path) =
|
||||
lance::io::ObjectStore::from_uri_and_params(registry, table_url.as_str(), ¶ms)
|
||||
.await
|
||||
.unwrap();
|
||||
let wrapped_marker = base_path.clone().join("wrapped-marker");
|
||||
store
|
||||
.inner
|
||||
.put(
|
||||
&wrapped_marker,
|
||||
bytes::Bytes::from_static(b"wrapped").into(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
store.remove_dir_all(base_path).await.unwrap();
|
||||
|
||||
assert!(table_path.join("native-marker").exists());
|
||||
assert!(matches!(
|
||||
wrapper.store.head(&wrapped_marker).await,
|
||||
Err(object_store::Error::NotFound { .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn rooted_commit_handler_uses_store_paths_for_latest_manifest() {
|
||||
let tempdir = tempfile::tempdir().unwrap();
|
||||
let versions = tempdir.path().join("_versions");
|
||||
std::fs::create_dir(&versions).unwrap();
|
||||
std::fs::write(versions.join("2.manifest"), b"native").unwrap();
|
||||
|
||||
let base_path = Path::from_absolute_path(tempdir.path()).unwrap();
|
||||
let raw_store: Arc<dyn ObjectStore> = Arc::new(InMemory::new());
|
||||
raw_store
|
||||
.put(
|
||||
&base_path.clone().join("_versions").join("1.manifest"),
|
||||
bytes::Bytes::from_static(b"wrapped").into(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
let mut store = lance::io::ObjectStore::new(
|
||||
raw_store.clone(),
|
||||
Url::parse("file:///").unwrap(),
|
||||
Some(4 * 1024),
|
||||
None,
|
||||
false,
|
||||
false,
|
||||
DEFAULT_LOCAL_IO_PARALLELISM,
|
||||
0,
|
||||
None,
|
||||
);
|
||||
store.inner = raw_store;
|
||||
|
||||
let native = RenameCommitHandler
|
||||
.resolve_latest_location(&base_path, &store)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(native.version, 2);
|
||||
|
||||
let rooted = rooted_file_commit_handler()
|
||||
.resolve_latest_location(&base_path, &store)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(rooted.version, 1);
|
||||
assert_eq!(
|
||||
rooted.path,
|
||||
base_path.clone().join("_versions").join("1.manifest")
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn reuses_store_cache_across_database_paths() {
|
||||
let tempdir = tempfile::tempdir().unwrap();
|
||||
let first_url = Url::from_directory_path(tempdir.path().join("database")).unwrap();
|
||||
let second_url = Url::from_directory_path(tempdir.path().join("share/database")).unwrap();
|
||||
let registry = Arc::new(ObjectStoreRegistry::default());
|
||||
registry.insert("file", Arc::new(PrefixedFileStoreProvider));
|
||||
|
||||
let (first, _) = lance::io::ObjectStore::from_uri_and_params(
|
||||
registry.clone(),
|
||||
first_url.as_str(),
|
||||
&ObjectStoreParams::default(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
let (second, _) = lance::io::ObjectStore::from_uri_and_params(
|
||||
registry.clone(),
|
||||
second_url.as_str(),
|
||||
&ObjectStoreParams::default(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
assert!(Arc::ptr_eq(&first, &second));
|
||||
assert_eq!(registry.stats().misses, 1);
|
||||
assert_eq!(registry.stats().hits, 1);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn reuses_database_store_for_table_and_clone_targets() {
|
||||
let tempdir = tempfile::tempdir().unwrap();
|
||||
let database_url = Url::from_directory_path(tempdir.path().join("database")).unwrap();
|
||||
let source_url =
|
||||
Url::from_directory_path(tempdir.path().join("database/source.lance")).unwrap();
|
||||
let target_url =
|
||||
Url::from_directory_path(tempdir.path().join("database/target.lance")).unwrap();
|
||||
let registry = Arc::new(ObjectStoreRegistry::default());
|
||||
registry.insert("file", Arc::new(PrefixedFileStoreProvider));
|
||||
|
||||
let (database, _) = lance::io::ObjectStore::from_uri_and_params(
|
||||
registry.clone(),
|
||||
database_url.as_str(),
|
||||
&ObjectStoreParams::default(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
let (source, _) = lance::io::ObjectStore::from_uri_and_params(
|
||||
registry.clone(),
|
||||
source_url.as_str(),
|
||||
&ObjectStoreParams::default(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
let (target, _) = lance::io::ObjectStore::from_uri_and_params(
|
||||
registry.clone(),
|
||||
target_url.as_str(),
|
||||
&ObjectStoreParams::default(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
assert!(Arc::ptr_eq(&database, &source));
|
||||
assert!(Arc::ptr_eq(&database, &target));
|
||||
assert_eq!(registry.stats().misses, 1);
|
||||
assert_eq!(registry.stats().hits, 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn normalizes_absolute_drive_and_unc_aliases() {
|
||||
let tempdir = tempfile::tempdir().unwrap();
|
||||
let mut drive_store = RootedLocalFileSystem::new(tempdir.path().to_path_buf()).unwrap();
|
||||
drive_store.absolute_alias = Path::from("C:");
|
||||
assert_eq!(
|
||||
drive_store.normalize(&Path::from("C:/Users/db/table.lance")),
|
||||
Path::from("Users/db/table.lance")
|
||||
);
|
||||
|
||||
let mut unc_store = RootedLocalFileSystem::new(tempdir.path().to_path_buf()).unwrap();
|
||||
unc_store.absolute_alias = Path::parse(r"\\server\share").unwrap();
|
||||
assert_eq!(
|
||||
unc_store.normalize(&Path::parse(r"\\server\share/share/db/table.lance").unwrap()),
|
||||
Path::from("share/db/table.lance")
|
||||
);
|
||||
assert_eq!(
|
||||
unc_store.normalize(&Path::from("share/db/table.lance")),
|
||||
Path::from("share/db/table.lance")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn relative_unc_alias_does_not_cross_database_roots() {
|
||||
let tempdir = tempfile::tempdir().unwrap();
|
||||
let mut store = RootedLocalFileSystem::new(tempdir.path().to_path_buf()).unwrap();
|
||||
store.absolute_alias = Path::parse(r"\\server\share").unwrap();
|
||||
|
||||
let relative = Path::from("share/database/table.lance/marker");
|
||||
assert_eq!(store.normalize(&relative), relative);
|
||||
assert_eq!(
|
||||
store.normalize(
|
||||
&Path::parse(r"\\server\share/share/database/table.lance/marker").unwrap()
|
||||
),
|
||||
relative
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn routes_unc_alias_lifecycle_through_the_prefix() {
|
||||
let tempdir = tempfile::tempdir().unwrap();
|
||||
let mut store = RootedLocalFileSystem::new(tempdir.path().to_path_buf()).unwrap();
|
||||
store.absolute_alias = Path::parse(r"\\server\share").unwrap();
|
||||
let table = Path::parse(r"\\server\share/share/db/test.lance").unwrap();
|
||||
let marker = table.clone().join("marker");
|
||||
|
||||
store
|
||||
.put(&marker, bytes::Bytes::from_static(b"unc").into())
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
std::fs::read(tempdir.path().join("share/db/test.lance/marker")).unwrap(),
|
||||
b"unc"
|
||||
);
|
||||
|
||||
let listed = store.list(Some(&table)).collect::<Vec<_>>().await;
|
||||
assert_eq!(listed.len(), 1);
|
||||
assert_eq!(listed[0].as_ref().unwrap().location, marker);
|
||||
assert_eq!(
|
||||
store
|
||||
.get(&marker)
|
||||
.await
|
||||
.unwrap()
|
||||
.bytes()
|
||||
.await
|
||||
.unwrap()
|
||||
.as_ref(),
|
||||
b"unc"
|
||||
);
|
||||
|
||||
store.delete(&marker).await.unwrap();
|
||||
assert!(!tempdir.path().join("share/db/test.lance/marker").exists());
|
||||
}
|
||||
|
||||
#[cfg(windows)]
|
||||
#[test]
|
||||
fn extracts_drive_and_unc_share_roots() {
|
||||
let (root, path) = PrefixedFileStoreProvider::root_and_relative_path(
|
||||
&Url::parse("file:///C:/database").unwrap(),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(root, std::path::PathBuf::from(r"C:\"));
|
||||
assert_eq!(path, Path::from("database"));
|
||||
|
||||
let (root, path) = PrefixedFileStoreProvider::root_and_relative_path(
|
||||
&Url::parse("file://server/share/database").unwrap(),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(root, std::path::PathBuf::from(r"\\server\share\"));
|
||||
assert_eq!(path, Path::from("database"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn preserves_drive_and_unc_roots_in_extracted_paths() {
|
||||
assert_eq!(
|
||||
PrefixedFileStoreProvider::rooted_path(
|
||||
std::path::Path::new(r"C:\"),
|
||||
&Path::from("database/table.lance")
|
||||
)
|
||||
.unwrap(),
|
||||
Path::from("C:/database/table.lance")
|
||||
);
|
||||
assert_eq!(
|
||||
PrefixedFileStoreProvider::rooted_path(
|
||||
std::path::Path::new(r"\\server\share\"),
|
||||
&Path::from("database/table.lance")
|
||||
)
|
||||
.unwrap(),
|
||||
Path::parse(r"\\server\share/database/table.lance").unwrap()
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
struct MirroringObjectStore {
|
||||
primary: Arc<dyn ObjectStore>,
|
||||
|
||||
@@ -141,7 +141,7 @@ impl SpawnedJob {
|
||||
Ok(Err(err)) => Outcome::Failed(Arc::new(err)),
|
||||
Err(err) if err.is_cancelled() => Outcome::Cancelled,
|
||||
Err(err) => Outcome::Failed(Arc::new(Error::Runtime {
|
||||
message: format!("job task failed: {err}"),
|
||||
message: format!("index job task failed: {err}"),
|
||||
})),
|
||||
};
|
||||
let _ = tx.send(Some(outcome));
|
||||
|
||||
@@ -214,7 +214,7 @@ use lance_linalg::distance::DistanceType as LanceDistanceType;
|
||||
/// a built-in pull-based adapter.
|
||||
#[cfg(feature = "metrics")]
|
||||
pub use metrics;
|
||||
pub use table::{FtsToken, Table, TableBase};
|
||||
pub use table::{FtsToken, Table};
|
||||
|
||||
/// Tokenize a full-text search query using an explicit FTS tokenizer configuration.
|
||||
///
|
||||
|
||||
@@ -19,15 +19,6 @@ const ARROW_FILE_CONTENT_TYPE: &str = "application/vnd.apache.arrow.file";
|
||||
#[cfg(test)]
|
||||
const JSON_CONTENT_TYPE: &str = "application/json";
|
||||
|
||||
fn extract_job_id(body: &str) -> Option<String> {
|
||||
serde_json::from_str::<serde_json::Value>(body)
|
||||
.ok()?
|
||||
.get("job_id")?
|
||||
.as_str()
|
||||
.filter(|job_id| !job_id.is_empty())
|
||||
.map(str::to_string)
|
||||
}
|
||||
|
||||
pub use client::{ClientConfig, HeaderProvider, RetryConfig, TimeoutConfig, TlsConfig};
|
||||
pub use db::{RemoteDatabaseOptions, RemoteDatabaseOptionsBuilder};
|
||||
pub use oauth::{OAuthConfig, OAuthFlow, OAuthHeaderProvider};
|
||||
|
||||
@@ -9,7 +9,6 @@ use http::StatusCode;
|
||||
use lance_io::object_store::StorageOptions;
|
||||
use lance_namespace_impls::{DynamicContextProvider, OperationInfo};
|
||||
use moka::future::Cache;
|
||||
use reqwest::Response;
|
||||
use reqwest::header::CONTENT_TYPE;
|
||||
|
||||
use lance_namespace::models::{
|
||||
@@ -24,17 +23,15 @@ use crate::database::{
|
||||
JobDescription, JobInfo, OpenTableRequest, ReadConsistency, TableNamesRequest,
|
||||
};
|
||||
use crate::error::Result;
|
||||
use crate::job::Job;
|
||||
use crate::remote::job::RemoteJob;
|
||||
use crate::remote::util::stream_as_body;
|
||||
use crate::table::BaseTable;
|
||||
|
||||
use super::ARROW_STREAM_CONTENT_TYPE;
|
||||
use super::client::{
|
||||
ClientConfig, HeaderProvider, HttpSend, RequestResultExt, RestfulLanceDbClient, Sender,
|
||||
};
|
||||
use super::table::RemoteTable;
|
||||
use super::util::parse_server_version;
|
||||
use super::{ARROW_STREAM_CONTENT_TYPE, extract_job_id};
|
||||
|
||||
// Request structure for the remote clone table API
|
||||
#[derive(serde::Serialize)]
|
||||
@@ -329,22 +326,6 @@ impl RemoteDatabase {
|
||||
}
|
||||
}
|
||||
|
||||
impl<S: HttpSend> RemoteDatabase<S> {
|
||||
async fn submit_drop_table(
|
||||
&self,
|
||||
name: &str,
|
||||
namespace_path: &[String],
|
||||
) -> Result<(String, Response)> {
|
||||
let identifier = build_table_identifier(name, namespace_path, &self.client.id_delimiter);
|
||||
let cache_key = build_cache_key(name, namespace_path);
|
||||
let req = self.client.post(&format!("/v1/table/{}/drop/", identifier));
|
||||
let (request_id, resp) = self.client.send(req).await?;
|
||||
let resp = self.client.check_response(&request_id, resp).await?;
|
||||
self.table_cache.remove(&cache_key).await;
|
||||
Ok((request_id, resp))
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "remote"))]
|
||||
mod test_utils {
|
||||
use super::*;
|
||||
@@ -913,28 +894,13 @@ impl<S: HttpSend> Database for RemoteDatabase<S> {
|
||||
}
|
||||
|
||||
async fn drop_table(&self, name: &str, namespace_path: &[String]) -> Result<()> {
|
||||
self.submit_drop_table(name, namespace_path)
|
||||
.await
|
||||
.map(|_| ())
|
||||
}
|
||||
|
||||
async fn drop_table_async(&self, name: &str, namespace_path: &[String]) -> Result<Job> {
|
||||
let (request_id, response) = self.submit_drop_table(name, namespace_path).await?;
|
||||
let status = response.status();
|
||||
let body = response.text().await.err_to_http(request_id.clone())?;
|
||||
let job_id = extract_job_id(&body);
|
||||
Ok(match job_id {
|
||||
Some(job_id) => Job::new(Box::new(RemoteJob::new(self.client.clone(), job_id))),
|
||||
None if status == StatusCode::ACCEPTED => {
|
||||
return Err(Error::Http {
|
||||
source: "asynchronous drop-table response did not contain a valid job_id"
|
||||
.into(),
|
||||
request_id,
|
||||
status_code: Some(status),
|
||||
});
|
||||
}
|
||||
None => Job::new_done(),
|
||||
})
|
||||
let identifier = build_table_identifier(name, namespace_path, &self.client.id_delimiter);
|
||||
let cache_key = build_cache_key(name, namespace_path);
|
||||
let req = self.client.post(&format!("/v1/table/{}/drop/", identifier));
|
||||
let (request_id, resp) = self.client.send(req).await?;
|
||||
self.client.check_response(&request_id, resp).await?;
|
||||
self.table_cache.remove(&cache_key).await;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn drop_all_tables(&self, namespace_path: &[String]) -> Result<()> {
|
||||
@@ -1526,67 +1492,6 @@ mod tests {
|
||||
// NOTE: the API will return 200 even if the table does not exist. So we shouldn't expect 404.
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_drop_table_does_not_read_response_body() {
|
||||
let conn = Connection::new_with_handler(|_| {
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(vec![0xff])
|
||||
.unwrap()
|
||||
});
|
||||
|
||||
conn.drop_table("table1", &[]).await.unwrap();
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_drop_table_async_returns_job() {
|
||||
let conn = Connection::new_with_handler(|request| {
|
||||
assert_eq!(request.method(), &reqwest::Method::POST);
|
||||
assert_eq!(request.url().path(), "/v1/table/table1/drop/");
|
||||
http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id":"drop-job-123"}"#)
|
||||
.unwrap()
|
||||
});
|
||||
|
||||
let job = conn.drop_table_async("table1", &[]).await.unwrap();
|
||||
assert_eq!(job.id(), Some("drop-job-123"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_drop_table_async_old_server_returns_done_job() {
|
||||
let conn = Connection::new_with_handler(|_| {
|
||||
http::Response::builder().status(200).body("").unwrap()
|
||||
});
|
||||
|
||||
let job = conn.drop_table_async("table1", &[]).await.unwrap();
|
||||
assert_eq!(job.id(), None);
|
||||
assert_eq!(job.status().await.unwrap(), "finished");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_drop_table_async_rejects_accepted_response_without_job_id() {
|
||||
let conn = Connection::new_with_handler(|_| {
|
||||
http::Response::builder().status(202).body("{}").unwrap()
|
||||
});
|
||||
|
||||
let error = conn.drop_table_async("table1", &[]).await.err().unwrap();
|
||||
assert!(error.to_string().contains("valid job_id"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_drop_table_async_rejects_empty_job_id() {
|
||||
let conn = Connection::new_with_handler(|_| {
|
||||
http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id":""}"#)
|
||||
.unwrap()
|
||||
});
|
||||
|
||||
let error = conn.drop_table_async("table1", &[]).await.err().unwrap();
|
||||
assert!(error.to_string().contains("valid job_id"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_rename_table() {
|
||||
let conn = Connection::new_with_handler(|request| {
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||
|
||||
pub mod bases;
|
||||
pub mod blobs;
|
||||
pub mod insert;
|
||||
|
||||
@@ -9,7 +8,7 @@ use self::insert::{RemoteWriteExec, WriteOp};
|
||||
use super::client::RequestResultExt;
|
||||
use super::client::{HttpSend, RestfulLanceDbClient, Sender};
|
||||
use super::db::ServerVersion;
|
||||
use super::{ARROW_FILE_CONTENT_TYPE, ARROW_STREAM_CONTENT_TYPE, extract_job_id};
|
||||
use super::{ARROW_FILE_CONTENT_TYPE, ARROW_STREAM_CONTENT_TYPE};
|
||||
use crate::blob::BlobFile;
|
||||
use crate::data::scannable::{PeekedScannable, Scannable, estimate_write_partitions};
|
||||
use crate::expr::expr_to_sql_string;
|
||||
@@ -34,9 +33,7 @@ use crate::table::lsm_stats::GetLsmStatsResponse;
|
||||
use crate::table::merge::MergeFilter;
|
||||
use crate::table::query::create_multi_vector_plan;
|
||||
use crate::table::write_progress::FinishOnDrop;
|
||||
use crate::table::{
|
||||
AlterColumnsResult, FieldMetadataUpdate, RefreshColumnResult, UpdateFieldMetadataResult,
|
||||
};
|
||||
use crate::table::{AlterColumnsResult, FieldMetadataUpdate, UpdateFieldMetadataResult};
|
||||
use crate::table::{AnyQuery, Filter, Predicate, PreprocessingOutput, TableStatistics};
|
||||
use crate::utils::background_cache::BackgroundCache;
|
||||
use crate::utils::{
|
||||
@@ -143,40 +140,6 @@ impl FreshnessHeaders {
|
||||
}
|
||||
}
|
||||
|
||||
/// A backfill job whose successful wait establishes a read-freshness
|
||||
/// baseline on the submitting handle, so a later read cannot be served
|
||||
/// from a cache older than the completed fill. A handle pinned by checkout
|
||||
/// at completion keeps its time-travel view instead.
|
||||
struct FreshnessJob<S: HttpSend> {
|
||||
inner: RemoteJob<S>,
|
||||
freshness: Arc<Mutex<FreshnessState>>,
|
||||
version: Arc<RwLock<Option<u64>>>,
|
||||
}
|
||||
|
||||
#[async_trait]
|
||||
impl<S: HttpSend> crate::job::JobHandle for FreshnessJob<S> {
|
||||
fn id(&self) -> Option<&str> {
|
||||
crate::job::JobHandle::id(&self.inner)
|
||||
}
|
||||
|
||||
async fn status(&self) -> Result<String> {
|
||||
crate::job::JobHandle::status(&self.inner).await
|
||||
}
|
||||
|
||||
async fn wait(&self) -> Result<()> {
|
||||
crate::job::JobHandle::wait(&self.inner).await?;
|
||||
let version = self.version.read().await;
|
||||
if version.is_none() {
|
||||
self.freshness.lock().unwrap().checkout_baseline = Some(SystemTime::now());
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn cancel(&self) -> Result<()> {
|
||||
crate::job::JobHandle::cancel(&self.inner).await
|
||||
}
|
||||
}
|
||||
|
||||
fn compute_min_timestamp(
|
||||
state: &FreshnessState,
|
||||
interval: Option<Duration>,
|
||||
@@ -311,10 +274,10 @@ pub struct RemoteTable<S: HttpSend = Sender> {
|
||||
identifier: String,
|
||||
server_version: ServerVersion,
|
||||
|
||||
version: Arc<RwLock<Option<u64>>>,
|
||||
version: RwLock<Option<u64>>,
|
||||
location: RwLock<Option<String>>,
|
||||
schema_cache: BackgroundCache<SchemaRef, Error>,
|
||||
freshness: Arc<Mutex<FreshnessState>>,
|
||||
freshness: Mutex<FreshnessState>,
|
||||
/// The branch this handle is scoped to, or `None` for the main branch.
|
||||
/// Stamped onto every branch-accepting request so reads and writes resolve
|
||||
/// on the branch's own version chain rather than main's.
|
||||
@@ -429,7 +392,13 @@ impl<S: HttpSend> RemoteTable<S> {
|
||||
.text()
|
||||
.await
|
||||
.ok()
|
||||
.and_then(|body| extract_job_id(&body));
|
||||
.and_then(|body| serde_json::from_str::<serde_json::Value>(&body).ok())
|
||||
.and_then(|value| {
|
||||
value
|
||||
.get("job_id")
|
||||
.and_then(|id| id.as_str())
|
||||
.map(str::to_string)
|
||||
});
|
||||
|
||||
if let Some(wait_timeout) = index.wait_timeout {
|
||||
let index_name = index.name.unwrap_or_else(|| format!("{}_idx", column));
|
||||
@@ -452,10 +421,10 @@ impl<S: HttpSend> RemoteTable<S> {
|
||||
namespace,
|
||||
identifier,
|
||||
server_version,
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
version: RwLock::new(None),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
branch: None,
|
||||
}
|
||||
}
|
||||
@@ -484,10 +453,10 @@ impl<S: HttpSend> RemoteTable<S> {
|
||||
namespace: self.namespace.clone(),
|
||||
identifier: self.identifier.clone(),
|
||||
server_version: self.server_version.clone(),
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
version: RwLock::new(None),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
branch,
|
||||
}
|
||||
}
|
||||
@@ -1305,10 +1274,10 @@ mod test_utils {
|
||||
namespace: vec![],
|
||||
identifier: name,
|
||||
server_version: version.map(ServerVersion).unwrap_or_default(),
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
version: RwLock::new(None),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
branch: None,
|
||||
}
|
||||
}
|
||||
@@ -1329,10 +1298,10 @@ mod test_utils {
|
||||
namespace: vec![],
|
||||
identifier: name,
|
||||
server_version: ServerVersion::default(),
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
version: RwLock::new(None),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
branch: None,
|
||||
}
|
||||
}
|
||||
@@ -1362,10 +1331,10 @@ mod test_utils {
|
||||
namespace: vec![],
|
||||
identifier: name,
|
||||
server_version: version.map(ServerVersion).unwrap_or_default(),
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
version: RwLock::new(None),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
branch: None,
|
||||
}
|
||||
}
|
||||
@@ -2245,10 +2214,6 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
|
||||
self.blob_columns_impl().await
|
||||
}
|
||||
|
||||
async fn add_bases(&self, bases: &[crate::table::TableBase]) -> Result<()> {
|
||||
self.add_bases_impl(bases).await
|
||||
}
|
||||
|
||||
async fn fetch_blobs(&self, column: &str, row_ids: &[u64]) -> Result<LargeBinaryArray> {
|
||||
self.fetch_blobs_impl(column, row_ids).await
|
||||
}
|
||||
@@ -2749,86 +2714,6 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
|
||||
}
|
||||
}
|
||||
|
||||
async fn add_computed_columns(&self, columns: &[(String, String)]) -> Result<AddColumnsResult> {
|
||||
self.check_mutable().await?;
|
||||
// The server plans the declaration: expression validation, type
|
||||
// inference and the persisted binding all happen there.
|
||||
let entries = columns
|
||||
.iter()
|
||||
.map(
|
||||
|(name, expression)| lance_namespace::models::AddColumnsEntry {
|
||||
name: name.clone(),
|
||||
computed: Some(Some(expression.clone())),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.collect::<Vec<_>>();
|
||||
let mut body = serde_json::json!({ "new_columns": entries });
|
||||
self.apply_branch_body(&mut body);
|
||||
let request = self
|
||||
.client
|
||||
.post(&format!("/v1/table/{}/add_columns/", self.identifier))
|
||||
.json(&body);
|
||||
let (request_id, response) = self.send(request, true).await?;
|
||||
let response = self.check_table_response(&request_id, response).await?;
|
||||
let body = response.text().await.err_to_http(request_id.clone())?;
|
||||
|
||||
if body.trim().is_empty() {
|
||||
// Backward compatible with old servers
|
||||
return Ok(AddColumnsResult { version: 0 });
|
||||
}
|
||||
|
||||
let result: AddColumnsResult = serde_json::from_str(&body).map_err(|e| Error::Http {
|
||||
source: format!("Failed to parse add_columns response: {}", e).into(),
|
||||
request_id,
|
||||
status_code: None,
|
||||
})?;
|
||||
|
||||
self.invalidate_schema_cache();
|
||||
self.track_write_version(result.version);
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
async fn refresh_column(&self, _column: &str) -> Result<RefreshColumnResult> {
|
||||
// The server runs a refresh as a job and does not report a fill
|
||||
// count, so the blocking form has no honest result to return.
|
||||
Err(Error::NotSupported {
|
||||
message: "a remote refresh runs as a server job; use refresh_column_async and \
|
||||
wait on the returned handle"
|
||||
.into(),
|
||||
})
|
||||
}
|
||||
|
||||
async fn refresh_column_async(&self, column: &str) -> Result<Job> {
|
||||
self.check_mutable().await?;
|
||||
let mut body = serde_json::json!({ "column": column });
|
||||
self.apply_branch_body(&mut body);
|
||||
let request = self
|
||||
.client
|
||||
.post(&format!("/v1/table/{}/backfill_column", self.identifier))
|
||||
.json(&body);
|
||||
let (request_id, response) = self.send(request, true).await?;
|
||||
let response = self.check_table_response(&request_id, response).await?;
|
||||
let body = response.text().await.err_to_http(request_id.clone())?;
|
||||
|
||||
#[derive(serde::Deserialize)]
|
||||
struct BackfillResponse {
|
||||
job_id: String,
|
||||
}
|
||||
let response: BackfillResponse = serde_json::from_str(&body).map_err(|e| Error::Http {
|
||||
source: format!("Failed to parse backfill_column response: {}", e).into(),
|
||||
request_id,
|
||||
status_code: None,
|
||||
})?;
|
||||
|
||||
Ok(Job::new(Box::new(FreshnessJob {
|
||||
inner: RemoteJob::new(self.client.clone(), response.job_id),
|
||||
freshness: self.freshness.clone(),
|
||||
version: self.version.clone(),
|
||||
})))
|
||||
}
|
||||
|
||||
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult> {
|
||||
self.check_mutable().await?;
|
||||
let body = alterations
|
||||
@@ -4094,42 +3979,6 @@ mod tests {
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_add_bases_posts_the_bases_array() {
|
||||
let table = Table::new_with_handler("my_table", |request| {
|
||||
assert_eq!(request.method(), "POST");
|
||||
assert_eq!(request.url().path(), "/v1/table/my_table/bases/");
|
||||
let body: serde_json::Value =
|
||||
serde_json::from_slice(request.body().unwrap().as_bytes().unwrap()).unwrap();
|
||||
assert_eq!(
|
||||
body["bases"],
|
||||
serde_json::json!([{
|
||||
"path": "s3://bucket/media/",
|
||||
"isDatasetRoot": false
|
||||
}])
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"version": 4}"#)
|
||||
.unwrap()
|
||||
});
|
||||
|
||||
table.add_bases(["s3://bucket/media/"]).await.unwrap();
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_add_bases_rejects_empty_response() {
|
||||
let table = Table::new_with_handler("my_table", |_request| {
|
||||
http::Response::builder().status(200).body("").unwrap()
|
||||
});
|
||||
let err = table.add_bases(["s3://bucket/media/"]).await.unwrap_err();
|
||||
assert!(
|
||||
err.to_string()
|
||||
.contains("invalid response while registering table bases"),
|
||||
"{err}"
|
||||
);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case(semver::Version::new(0, 1, 0))]
|
||||
#[case(semver::Version::new(0, 5, 0))]
|
||||
@@ -6606,346 +6455,6 @@ mod tests {
|
||||
assert_eq!(result.version, if old_server { 0 } else { 43 });
|
||||
}
|
||||
|
||||
/// A declaration is sent as `{name, computed}` entries for the server to
|
||||
/// plan; the client never types the expression itself.
|
||||
#[tokio::test]
|
||||
async fn test_add_computed_columns_sends_the_expression() {
|
||||
let table = Table::new_with_handler("my_table", |request| {
|
||||
assert_eq!(request.method(), "POST");
|
||||
assert_eq!(request.url().path(), "/v1/table/my_table/add_columns/");
|
||||
let body = request.body().unwrap().as_bytes().unwrap();
|
||||
let value: serde_json::Value = serde_json::from_slice(body).unwrap();
|
||||
assert_eq!(
|
||||
value["new_columns"],
|
||||
serde_json::json!([{"name": "doubled", "computed": "x * 2"}])
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"version": 7}"#)
|
||||
.unwrap()
|
||||
});
|
||||
|
||||
let result = table
|
||||
.add_columns()
|
||||
.computed("doubled", "x * 2")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(result.version, 7);
|
||||
}
|
||||
|
||||
/// A remote refresh is a server job: the async form returns its handle,
|
||||
/// and the blocking form refuses rather than invent a fill count.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_column_async_submits_a_backfill_job() {
|
||||
let table = Table::new_with_handler("my_table", |request| {
|
||||
assert_eq!(request.method(), "POST");
|
||||
assert_eq!(request.url().path(), "/v1/table/my_table/backfill_column");
|
||||
let body = request.body().unwrap().as_bytes().unwrap();
|
||||
let value: serde_json::Value = serde_json::from_slice(body).unwrap();
|
||||
assert_eq!(value["column"], "doubled");
|
||||
http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-42"}"#)
|
||||
.unwrap()
|
||||
});
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
assert_eq!(job.id(), Some("j-42"));
|
||||
|
||||
let err = table.refresh_column("doubled").await.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::NotSupported { message }
|
||||
if message.contains("refresh_column_async")),
|
||||
"{err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
/// The gate's reproducer: after a successful wait, a same-handle read
|
||||
/// must carry a freshness baseline so a stale server cache cannot serve
|
||||
/// the pre-backfill snapshot.
|
||||
#[tokio::test]
|
||||
async fn test_backfill_wait_establishes_read_freshness() {
|
||||
let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
let saw = saw_min_timestamp.clone();
|
||||
let table =
|
||||
Table::new_with_handler("my_table", move |request| match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-7"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-7", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
saw.store(
|
||||
request.headers().contains_key("x-lancedb-min-timestamp"),
|
||||
std::sync::atomic::Ordering::SeqCst,
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
});
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
assert!(
|
||||
saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst),
|
||||
"read after wait carried no freshness baseline"
|
||||
);
|
||||
}
|
||||
|
||||
/// A checkout after submission wins over the completion fence: the
|
||||
/// pinned view must not regain a timestamp floor from the job.
|
||||
#[tokio::test]
|
||||
async fn test_checkout_after_submit_beats_the_completion_fence() {
|
||||
let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
let saw = saw_min_timestamp.clone();
|
||||
let table =
|
||||
Table::new_with_handler("my_table", move |request| match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-8"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-8", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/describe/" => {
|
||||
let schema = Schema::new(vec![Field::new("x", DataType::Int32, true)]);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(describe_response(&schema))
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
saw.store(
|
||||
request.headers().contains_key("x-lancedb-min-timestamp"),
|
||||
std::sync::atomic::Ordering::SeqCst,
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
});
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
table.checkout(3).await.unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
assert!(
|
||||
!saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst),
|
||||
"completion fence overrode an explicit checkout"
|
||||
);
|
||||
}
|
||||
|
||||
/// Tag checkout resets freshness state wholesale; the fence must not
|
||||
/// survive it.
|
||||
#[tokio::test]
|
||||
async fn test_tag_checkout_after_submit_beats_the_completion_fence() {
|
||||
let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
let saw = saw_min_timestamp.clone();
|
||||
let table =
|
||||
Table::new_with_handler("my_table", move |request| match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-9"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-9", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/tags/version/" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"version": 5}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/describe/" => {
|
||||
let schema = Schema::new(vec![Field::new("x", DataType::Int32, true)]);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(describe_response(&schema))
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
saw.store(
|
||||
request.headers().contains_key("x-lancedb-min-timestamp"),
|
||||
std::sync::atomic::Ordering::SeqCst,
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
});
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
table.checkout_tag("v1").await.unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
assert!(
|
||||
!saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst),
|
||||
"completion fence overrode a tag checkout"
|
||||
);
|
||||
}
|
||||
|
||||
/// A checkout landing while the submission request is in flight advances
|
||||
/// the epoch past the token captured at submit.
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn test_checkout_during_submission_beats_the_completion_fence() {
|
||||
let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
let saw = saw_min_timestamp.clone();
|
||||
let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
|
||||
let release_rx = Arc::new(std::sync::Mutex::new(release_rx));
|
||||
let (arrived_tx, arrived_rx) = std::sync::mpsc::channel::<()>();
|
||||
let arrived_tx = Arc::new(std::sync::Mutex::new(arrived_tx));
|
||||
let table = Table::new_with_handler("my_table", move |request| {
|
||||
match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => {
|
||||
// Signal arrival, then hold the response until the
|
||||
// test's checkout completes.
|
||||
arrived_tx.lock().unwrap().send(()).unwrap();
|
||||
release_rx
|
||||
.lock()
|
||||
.unwrap()
|
||||
.recv_timeout(std::time::Duration::from_secs(10))
|
||||
.unwrap();
|
||||
http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-10"}"#.to_string())
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-10", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/describe/" => {
|
||||
let schema = Schema::new(vec![Field::new("x", DataType::Int32, true)]);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(describe_response(&schema))
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
saw.store(
|
||||
request.headers().contains_key("x-lancedb-min-timestamp"),
|
||||
std::sync::atomic::Ordering::SeqCst,
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
}
|
||||
});
|
||||
|
||||
let submit = tokio::spawn({
|
||||
let table = table.clone();
|
||||
async move { table.refresh_column_async("doubled").await }
|
||||
});
|
||||
tokio::task::spawn_blocking(move || {
|
||||
arrived_rx
|
||||
.recv_timeout(std::time::Duration::from_secs(10))
|
||||
.unwrap()
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
table.checkout(7).await.unwrap();
|
||||
release_tx.send(()).unwrap();
|
||||
|
||||
let job = submit.await.unwrap().unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
assert!(
|
||||
!saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst),
|
||||
"completion fence overrode a checkout that landed mid-submission"
|
||||
);
|
||||
}
|
||||
|
||||
/// checkout_latest keeps the handle on latest, so a completed backfill
|
||||
/// must still establish its post-fill baseline -- strictly later than the
|
||||
/// checkout's own, or a pre-fill cache could still serve.
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn test_checkout_latest_during_submission_keeps_the_fence() {
|
||||
let seen_min_timestamp = Arc::new(std::sync::Mutex::new(None::<String>));
|
||||
let saw = seen_min_timestamp.clone();
|
||||
let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
|
||||
let release_rx = Arc::new(std::sync::Mutex::new(release_rx));
|
||||
let (arrived_tx, arrived_rx) = std::sync::mpsc::channel::<()>();
|
||||
let arrived_tx = Arc::new(std::sync::Mutex::new(arrived_tx));
|
||||
let table =
|
||||
Table::new_with_handler("my_table", move |request| match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => {
|
||||
arrived_tx.lock().unwrap().send(()).unwrap();
|
||||
release_rx
|
||||
.lock()
|
||||
.unwrap()
|
||||
.recv_timeout(std::time::Duration::from_secs(10))
|
||||
.unwrap();
|
||||
http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-11"}"#.to_string())
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-11", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
*saw.lock().unwrap() = request
|
||||
.headers()
|
||||
.get("x-lancedb-min-timestamp")
|
||||
.map(|v| v.to_str().unwrap().to_string());
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
});
|
||||
|
||||
let submit = tokio::spawn({
|
||||
let table = table.clone();
|
||||
async move { table.refresh_column_async("doubled").await }
|
||||
});
|
||||
tokio::task::spawn_blocking(move || {
|
||||
arrived_rx
|
||||
.recv_timeout(std::time::Duration::from_secs(10))
|
||||
.unwrap()
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
table.checkout_latest().await.unwrap();
|
||||
let after_checkout = SystemTime::now();
|
||||
// Real separation between the checkout baseline and completion.
|
||||
tokio::time::sleep(std::time::Duration::from_millis(50)).await;
|
||||
release_tx.send(()).unwrap();
|
||||
|
||||
let job = submit.await.unwrap().unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
let header = seen_min_timestamp
|
||||
.lock()
|
||||
.unwrap()
|
||||
.clone()
|
||||
.expect("no baseline");
|
||||
let sent: SystemTime = chrono::DateTime::parse_from_rfc3339(&header)
|
||||
.unwrap()
|
||||
.into();
|
||||
assert!(
|
||||
sent > after_checkout,
|
||||
"baseline {header} did not advance past the checkout"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_prewarm_index() {
|
||||
let table = Table::new_with_handler("my_table", |request| {
|
||||
|
||||
@@ -1,42 +0,0 @@
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||
|
||||
//! Cloud HTTP for registering extra table storage bases.
|
||||
|
||||
use serde::Deserialize;
|
||||
|
||||
use crate::Error;
|
||||
use crate::error::Result;
|
||||
use crate::remote::client::{HttpSend, RequestResultExt};
|
||||
|
||||
use super::RemoteTable;
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
struct AddBasesResponse {
|
||||
version: u64,
|
||||
}
|
||||
|
||||
impl<S: HttpSend> RemoteTable<S> {
|
||||
pub(super) async fn add_bases_impl(&self, bases: &[crate::table::TableBase]) -> Result<()> {
|
||||
self.check_mutable().await?;
|
||||
let mut body = serde_json::json!({ "bases": bases });
|
||||
self.apply_branch_body(&mut body);
|
||||
let request = self
|
||||
.client
|
||||
.post(&format!("/v1/table/{}/bases/", self.identifier))
|
||||
.json(&body);
|
||||
let (request_id, response) = self.send(request, true).await?;
|
||||
let response = self.check_table_response(&request_id, response).await?;
|
||||
let body = response.text().await.err_to_http(request_id.clone())?;
|
||||
let parsed: AddBasesResponse = serde_json::from_str(&body).map_err(|e| Error::Http {
|
||||
source: format!(
|
||||
"The server returned an invalid response while registering table bases: {e}"
|
||||
)
|
||||
.into(),
|
||||
request_id,
|
||||
status_code: None,
|
||||
})?;
|
||||
self.track_write_version(parsed.version);
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
+1
-230
@@ -34,7 +34,7 @@ use lance_index::scalar::inverted::query::collect_query_tokens;
|
||||
use lance_namespace::LanceNamespace;
|
||||
use lance_namespace::error::NamespaceError;
|
||||
use lance_namespace::models::DescribeTableRequest;
|
||||
use lance_table::format::{BasePath, Manifest};
|
||||
use lance_table::format::Manifest;
|
||||
use lance_table::io::commit::CommitHandler;
|
||||
use lance_table::io::commit::ManifestNamingScheme;
|
||||
use lance_table::io::commit::external_manifest::ExternalManifestCommitHandler;
|
||||
@@ -68,7 +68,6 @@ pub mod add_columns;
|
||||
mod add_data;
|
||||
pub mod branch_merge;
|
||||
pub mod checkpoint;
|
||||
pub mod computed_columns;
|
||||
mod create_index;
|
||||
pub mod datafusion;
|
||||
pub(crate) mod dataset;
|
||||
@@ -78,7 +77,6 @@ pub mod merge;
|
||||
pub mod optimize;
|
||||
mod primary_key;
|
||||
pub mod query;
|
||||
pub mod refresh;
|
||||
pub mod schema_evolution;
|
||||
pub mod update;
|
||||
pub mod write_progress;
|
||||
@@ -92,9 +90,6 @@ pub use branch_merge::{
|
||||
MergeBranchResult, MergeBranchStatus, MergePreview, RowCountSummary,
|
||||
};
|
||||
pub use chrono::Duration;
|
||||
pub use computed_columns::{
|
||||
ComputedColumn, ComputedColumnKind, computed_column_from_field, computed_columns,
|
||||
};
|
||||
pub use delete::DeleteResult;
|
||||
use futures::future::join_all;
|
||||
pub use lance::dataset::refs::{BranchContents, Ref, TagContents, Tags as LanceTags};
|
||||
@@ -102,7 +97,6 @@ pub use lance::dataset::scanner::DatasetRecordBatchStream;
|
||||
pub use lance_index::optimize::OptimizeOptions;
|
||||
pub use lsm_stats::{BucketStats, GenerationStats, LsmStats, MemtableStats};
|
||||
pub use optimize::{CompactionOptions, OptimizeAction, OptimizeStats};
|
||||
pub use refresh::RefreshColumnResult;
|
||||
pub use schema_evolution::{
|
||||
AddColumnsResult, AlterColumnsResult, DropColumnsResult, FieldMetadataUpdate,
|
||||
UpdateFieldMetadataResult,
|
||||
@@ -643,15 +637,6 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
|
||||
message: "set_lsm_write_spec is not supported on this table type".into(),
|
||||
})
|
||||
}
|
||||
/// Switch this table to required index catch-up, one way.
|
||||
///
|
||||
/// The default implementation returns `NotSupported`. Implementations
|
||||
/// that support the MemWAL LSM write path must override this.
|
||||
async fn require_mem_wal_index_catchup(&self) -> Result<()> {
|
||||
Err(Error::NotSupported {
|
||||
message: "require_mem_wal_index_catchup is not supported on this table type".into(),
|
||||
})
|
||||
}
|
||||
/// Remove the [`LsmWriteSpec`] from this table.
|
||||
///
|
||||
/// This is a no-op if no spec is currently set.
|
||||
@@ -711,12 +696,6 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
|
||||
message: "blob_columns is not supported on this table type".into(),
|
||||
})
|
||||
}
|
||||
/// Register additional storage bases for this table.
|
||||
async fn add_bases(&self, _bases: &[TableBase]) -> Result<()> {
|
||||
Err(Error::NotSupported {
|
||||
message: "Registering table bases is not supported for this table type.".into(),
|
||||
})
|
||||
}
|
||||
/// Materialize blob bytes for the given row ids. See [`Table::fetch_blobs`].
|
||||
async fn fetch_blobs(&self, _column: &str, _row_ids: &[u64]) -> Result<LargeBinaryArray> {
|
||||
Err(Error::NotSupported {
|
||||
@@ -753,34 +732,6 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
|
||||
transforms: NewColumnTransform,
|
||||
read_columns: Option<Vec<String>>,
|
||||
) -> Result<AddColumnsResult>;
|
||||
/// Declare computed columns, each defined by a SQL expression.
|
||||
///
|
||||
/// Where the declaration is planned depends on the backend: a local table
|
||||
/// validates and types the expression itself, a remote one sends the text
|
||||
/// for the server to plan.
|
||||
async fn add_computed_columns(
|
||||
&self,
|
||||
_columns: &[(String, String)],
|
||||
) -> Result<AddColumnsResult> {
|
||||
Err(Error::NotSupported {
|
||||
message: "computed columns are not supported on this table type".into(),
|
||||
})
|
||||
}
|
||||
/// Fill a computed column's unfilled rows.
|
||||
///
|
||||
/// The default returns `NotSupported`; Lance-backed tables override it.
|
||||
async fn refresh_column(&self, _column: &str) -> Result<RefreshColumnResult> {
|
||||
Err(Error::NotSupported {
|
||||
message: "computed columns are supported only on local tables".into(),
|
||||
})
|
||||
}
|
||||
/// Fill a computed column's unfilled rows, returning a [`Job`] tracking
|
||||
/// the operation.
|
||||
async fn refresh_column_async(&self, _column: &str) -> Result<Job> {
|
||||
Err(Error::NotSupported {
|
||||
message: "computed columns are supported only on local tables".into(),
|
||||
})
|
||||
}
|
||||
/// Alter columns in the table.
|
||||
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult>;
|
||||
/// Drop columns from the table.
|
||||
@@ -895,54 +846,6 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
|
||||
}
|
||||
}
|
||||
|
||||
/// An extra storage prefix registered on a table.
|
||||
///
|
||||
/// `path` is an object-store URI. `name` is an optional alias. `is_dataset_root`
|
||||
/// is true when `path` points to a Lance dataset root. When false, `path`
|
||||
/// points directly to the directory containing the referenced files.
|
||||
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub struct TableBase {
|
||||
/// Object store URI such as `s3://bucket/media/`.
|
||||
pub path: String,
|
||||
/// Optional alias.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub name: Option<String>,
|
||||
/// True when `path` is a Lance dataset root. When false, `path` is the
|
||||
/// directory containing the referenced files.
|
||||
#[serde(default)]
|
||||
pub is_dataset_root: bool,
|
||||
}
|
||||
|
||||
impl TableBase {
|
||||
/// A non-root base with no alias.
|
||||
pub fn new(path: impl Into<String>) -> Self {
|
||||
Self {
|
||||
path: path.into(),
|
||||
name: None,
|
||||
is_dataset_root: false,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl From<&str> for TableBase {
|
||||
fn from(path: &str) -> Self {
|
||||
Self::new(path)
|
||||
}
|
||||
}
|
||||
|
||||
impl From<&String> for TableBase {
|
||||
fn from(path: &String) -> Self {
|
||||
Self::new(path.as_str())
|
||||
}
|
||||
}
|
||||
|
||||
impl From<String> for TableBase {
|
||||
fn from(path: String) -> Self {
|
||||
Self::new(path)
|
||||
}
|
||||
}
|
||||
|
||||
/// A Table is a collection of strong typed Rows.
|
||||
///
|
||||
/// The type of the each row is defined in Apache Arrow [Schema].
|
||||
@@ -1180,25 +1083,6 @@ impl Table {
|
||||
self.inner.blob_columns().await
|
||||
}
|
||||
|
||||
/// Register additional storage bases for this table.
|
||||
///
|
||||
/// A URI string is a non-root base with no alias.
|
||||
///
|
||||
/// ```
|
||||
/// # use lancedb::Table;
|
||||
/// # async fn register(table: &Table) -> Result<(), Box<dyn std::error::Error>> {
|
||||
/// table.add_bases(["s3://bucket/media/"]).await?;
|
||||
/// # Ok(())
|
||||
/// # }
|
||||
/// ```
|
||||
pub async fn add_bases(
|
||||
&self,
|
||||
bases: impl IntoIterator<Item = impl Into<TableBase>>,
|
||||
) -> Result<()> {
|
||||
let bases: Vec<TableBase> = bases.into_iter().map(Into::into).collect();
|
||||
self.inner.add_bases(&bases).await
|
||||
}
|
||||
|
||||
/// Materialize blob bytes for the given row ids.
|
||||
///
|
||||
/// Output matches `row_ids` in length and order. Null blobs are null;
|
||||
@@ -1740,53 +1624,6 @@ impl Table {
|
||||
AddColumnsBuilder::new(self.inner.clone())
|
||||
}
|
||||
|
||||
/// Fill the fragments of a computed column that hold no values yet.
|
||||
///
|
||||
/// Declared with
|
||||
/// [`AddColumnsBuilder::computed`](add_columns::AddColumnsBuilder::computed),
|
||||
/// a column starts empty and gets its values here. Fragments appended
|
||||
/// since the last refresh are filled by the next one; fragments already
|
||||
/// filled are left as they are, so the call is idempotent and does not
|
||||
/// observe a mutated input.
|
||||
///
|
||||
/// Local tables only: a remote refresh runs as a server job, through
|
||||
/// [`Table::refresh_column_async`].
|
||||
///
|
||||
/// ```
|
||||
/// # use lancedb::Table;
|
||||
/// # async fn refresh(table: &Table) -> Result<(), Box<dyn std::error::Error>> {
|
||||
/// let result = table.refresh_column("doubled").await?;
|
||||
/// println!("filled {} rows at version {}", result.rows_filled, result.version);
|
||||
/// # Ok(())
|
||||
/// # }
|
||||
/// ```
|
||||
pub async fn refresh_column(&self, column: impl AsRef<str>) -> Result<RefreshColumnResult> {
|
||||
self.inner.refresh_column(column.as_ref()).await
|
||||
}
|
||||
|
||||
/// Like [`Table::refresh_column`], but returns a [`Job`] tracking the
|
||||
/// operation instead of blocking until it completes.
|
||||
///
|
||||
/// The job may already be complete when returned, and callers must not
|
||||
/// assume the column is filled until [`Job::wait`] returns. Invalid input
|
||||
/// -- an unknown column, or one that is not computed -- is reported by
|
||||
/// this call rather than by the job. On local tables the job runs as an
|
||||
/// in-process task; on LanceDB Cloud and Enterprise it is the server's
|
||||
/// backfill job.
|
||||
///
|
||||
/// ```
|
||||
/// # use lancedb::Table;
|
||||
/// # async fn refresh_in_background(table: &Table) -> Result<(), Box<dyn std::error::Error>> {
|
||||
/// let job = table.refresh_column_async("doubled").await?;
|
||||
/// println!("refresh running: {:?}", job.status().await?);
|
||||
/// job.wait().await?;
|
||||
/// # Ok(())
|
||||
/// # }
|
||||
/// ```
|
||||
pub async fn refresh_column_async(&self, column: impl AsRef<str>) -> Result<Job> {
|
||||
self.inner.refresh_column_async(column.as_ref()).await
|
||||
}
|
||||
|
||||
/// Change a column's name or nullability.
|
||||
pub async fn alter_columns(
|
||||
&self,
|
||||
@@ -1856,20 +1693,6 @@ impl Table {
|
||||
self.inner.set_lsm_write_spec(spec).await
|
||||
}
|
||||
|
||||
/// Switch this table to required index catch-up, one way.
|
||||
///
|
||||
/// Separate from [`Self::set_lsm_write_spec`] on purpose: a table carrying
|
||||
/// the bit retains its SSTables until an index records that it holds the
|
||||
/// compacted rows, so turn it on only once something can repair coverage.
|
||||
/// A writer that already holds the dataset can call the equivalent on
|
||||
/// `DatasetMemWalExt` instead; this is the table-level entry point.
|
||||
///
|
||||
/// Errors if no spec is set, or if the table already records SSTable
|
||||
/// compaction progress from before this protocol.
|
||||
pub async fn require_mem_wal_index_catchup(&self) -> Result<()> {
|
||||
self.inner.require_mem_wal_index_catchup().await
|
||||
}
|
||||
|
||||
/// Remove the [`LsmWriteSpec`] from this table, reverting to the standard
|
||||
/// `merge_insert` write path.
|
||||
///
|
||||
@@ -2782,7 +2605,6 @@ impl NativeTable {
|
||||
namespace_client: Option<Arc<dyn LanceNamespace>>,
|
||||
pushdown_operations: HashSet<NamespaceClientPushdownOperation>,
|
||||
) -> Result<Self> {
|
||||
computed_columns::ensure_no_foreign_declarations(batches.arrow_schema().fields())?;
|
||||
// Default params uses format v1.
|
||||
let params = params.unwrap_or(WriteParams {
|
||||
..Default::default()
|
||||
@@ -3231,13 +3053,6 @@ impl BaseTable for NativeTable {
|
||||
let ds = self.dataset.get().await?;
|
||||
|
||||
let table_schema = Schema::from(&ds.schema().clone());
|
||||
computed_columns::ensure_not_written(
|
||||
&table_schema,
|
||||
add.data.schema().fields().iter().map(|f| f.name().as_str()),
|
||||
)?;
|
||||
if matches!(add.mode, AddDataMode::Overwrite) {
|
||||
computed_columns::ensure_no_foreign_declarations(add.data.schema().fields())?;
|
||||
}
|
||||
|
||||
let num_partitions = if let Some(parallelism) = add.write_parallelism {
|
||||
parallelism
|
||||
@@ -3398,11 +3213,6 @@ impl BaseTable for NativeTable {
|
||||
params: MergeInsertBuilder,
|
||||
new_data: Box<dyn RecordBatchReader + Send>,
|
||||
) -> Result<MergeResult> {
|
||||
let source_schema = arrow_array::RecordBatchReader::schema(&new_data);
|
||||
computed_columns::ensure_not_written(
|
||||
&Schema::from(self.dataset.get().await?.schema()),
|
||||
source_schema.fields().iter().map(|f| f.name().as_str()),
|
||||
)?;
|
||||
let result = merge::execute_merge_insert(self, params, new_data).await?;
|
||||
self.bump_freshness();
|
||||
Ok(result)
|
||||
@@ -3416,10 +3226,6 @@ impl BaseTable for NativeTable {
|
||||
merge::lsm::set_lsm_write_spec(self, spec).await
|
||||
}
|
||||
|
||||
async fn require_mem_wal_index_catchup(&self) -> Result<()> {
|
||||
merge::lsm::require_mem_wal_index_catchup(self).await
|
||||
}
|
||||
|
||||
async fn unset_lsm_write_spec(&self) -> Result<()> {
|
||||
merge::lsm::unset_lsm_write_spec(self).await
|
||||
}
|
||||
@@ -3437,25 +3243,6 @@ impl BaseTable for NativeTable {
|
||||
Ok(crate::blob::blob_column_names(schema.as_ref()))
|
||||
}
|
||||
|
||||
async fn add_bases(&self, bases: &[TableBase]) -> Result<()> {
|
||||
self.dataset.ensure_mutable()?;
|
||||
let dataset = self.dataset.get().await?;
|
||||
let new_bases = bases
|
||||
.iter()
|
||||
.map(|base| {
|
||||
BasePath::new(
|
||||
0,
|
||||
base.path.clone(),
|
||||
base.name.clone(),
|
||||
base.is_dataset_root,
|
||||
)
|
||||
})
|
||||
.collect();
|
||||
let dataset = dataset.add_bases(new_bases, None).await?;
|
||||
self.dataset.update(dataset);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn fetch_blobs(&self, column: &str, row_ids: &[u64]) -> Result<LargeBinaryArray> {
|
||||
let dataset = self.dataset.get().await?;
|
||||
crate::blob::take_blobs_aligned(&dataset, column, row_ids).await
|
||||
@@ -3507,22 +3294,6 @@ impl BaseTable for NativeTable {
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
async fn add_computed_columns(&self, columns: &[(String, String)]) -> Result<AddColumnsResult> {
|
||||
let result = schema_evolution::execute_declare(self, columns).await?;
|
||||
self.bump_freshness();
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
async fn refresh_column(&self, column: &str) -> Result<RefreshColumnResult> {
|
||||
let result = refresh::execute_refresh_column(self, column).await?;
|
||||
self.bump_freshness();
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
async fn refresh_column_async(&self, column: &str) -> Result<Job> {
|
||||
refresh::execute_refresh_column_async(self, column).await
|
||||
}
|
||||
|
||||
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult> {
|
||||
let result = schema_evolution::execute_alter_columns(self, alterations).await?;
|
||||
self.bump_freshness();
|
||||
|
||||
@@ -15,7 +15,6 @@ use crate::{Error, Result};
|
||||
pub struct AddColumnsBuilder {
|
||||
parent: Arc<dyn BaseTable>,
|
||||
transform: Option<NewColumnTransform>,
|
||||
computed: Vec<(String, String)>,
|
||||
read_columns: Option<Vec<String>>,
|
||||
}
|
||||
|
||||
@@ -24,7 +23,6 @@ impl std::fmt::Debug for AddColumnsBuilder {
|
||||
f.debug_struct("AddColumnsBuilder")
|
||||
.field("parent", &self.parent)
|
||||
.field("has_transform", &self.transform.is_some())
|
||||
.field("computed", &self.computed)
|
||||
.field("read_columns", &self.read_columns)
|
||||
.finish()
|
||||
}
|
||||
@@ -35,58 +33,19 @@ impl AddColumnsBuilder {
|
||||
Self {
|
||||
parent,
|
||||
transform: None,
|
||||
computed: Vec::new(),
|
||||
read_columns: None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Set how the new columns' values are produced.
|
||||
/// Set how the new columns' values are produced. Required.
|
||||
pub fn transform(mut self, transform: NewColumnTransform) -> Self {
|
||||
self.transform = Some(transform);
|
||||
self
|
||||
}
|
||||
|
||||
/// Add a column defined by `expression`, evaluated by a later refresh
|
||||
/// rather than by this commit. Its type and inputs are derived from the
|
||||
/// expression.
|
||||
///
|
||||
/// The column is committed with no values, so declaring one costs the same
|
||||
/// on an empty table as on a large one. Rows get values from
|
||||
/// [`Table::refresh_column`](super::Table::refresh_column), which fills
|
||||
/// every fragment that has none -- including fragments appended since the
|
||||
/// last refresh.
|
||||
///
|
||||
/// Refresh does not revisit a fragment it has filled, so mutating an input
|
||||
/// leaves the value computed at fill time; recomputing means dropping the
|
||||
/// column and declaring it again. An input cannot be renamed, retyped or
|
||||
/// dropped while a declaration reads it, since the expression names it.
|
||||
///
|
||||
/// On LanceDB Cloud and Enterprise the expression is planned by the
|
||||
/// server, and the refresh runs as a server job -- see
|
||||
/// [`Table::refresh_column_async`](super::Table::refresh_column_async).
|
||||
///
|
||||
/// ```
|
||||
/// # use lancedb::Table;
|
||||
/// # async fn declare(table: &Table) -> Result<(), Box<dyn std::error::Error>> {
|
||||
/// table
|
||||
/// .add_columns()
|
||||
/// .computed("doubled", "x * 2")
|
||||
/// .execute()
|
||||
/// .await?;
|
||||
/// let filled = table.refresh_column("doubled").await?;
|
||||
/// println!("filled {} rows", filled.rows_filled);
|
||||
/// # Ok(())
|
||||
/// # }
|
||||
/// ```
|
||||
pub fn computed(mut self, name: impl Into<String>, expression: impl Into<String>) -> Self {
|
||||
self.computed.push((name.into(), expression.into()));
|
||||
self
|
||||
}
|
||||
|
||||
/// Limit which existing columns a [`NewColumnTransform::BatchUDF`] mapper
|
||||
/// receives. Every other transform, and a computed column, determines what
|
||||
/// it reads, so setting this alongside one is an error rather than a silent
|
||||
/// no-op.
|
||||
/// receives. Every other transform determines what it reads, so setting
|
||||
/// this alongside one is an error rather than a silent no-op.
|
||||
pub fn read_columns(mut self, columns: impl IntoIterator<Item = impl Into<String>>) -> Self {
|
||||
self.read_columns = Some(columns.into_iter().map(Into::into).collect());
|
||||
self
|
||||
@@ -97,42 +56,24 @@ impl AddColumnsBuilder {
|
||||
let Self {
|
||||
parent,
|
||||
transform,
|
||||
computed,
|
||||
read_columns,
|
||||
} = self;
|
||||
|
||||
match (transform, computed.is_empty()) {
|
||||
(None, true) => Err(Error::InvalidInput {
|
||||
message: "add_columns requires a transform or a computed column".into(),
|
||||
}),
|
||||
// The two commit through different transforms, so one call covering
|
||||
// both would be two commits and could half-apply.
|
||||
(Some(_), false) => Err(Error::InvalidInput {
|
||||
message: "add_columns cannot mix a transform with computed columns; \
|
||||
they cannot be added atomically in one call"
|
||||
let Some(transform) = transform else {
|
||||
return Err(Error::InvalidInput {
|
||||
message: "add_columns requires a transform".into(),
|
||||
});
|
||||
};
|
||||
|
||||
if read_columns.is_some() && !matches!(transform, NewColumnTransform::BatchUDF(_)) {
|
||||
return Err(Error::InvalidInput {
|
||||
message: "read_columns applies only to a BatchUDF transform; \
|
||||
every other transform determines what it reads"
|
||||
.into(),
|
||||
}),
|
||||
(Some(transform), true) => {
|
||||
if read_columns.is_some() && !matches!(transform, NewColumnTransform::BatchUDF(_)) {
|
||||
return Err(Error::InvalidInput {
|
||||
message: "read_columns applies only to a BatchUDF transform; \
|
||||
every other transform determines what it reads"
|
||||
.into(),
|
||||
});
|
||||
}
|
||||
parent.add_columns(transform, read_columns).await
|
||||
}
|
||||
(None, false) => {
|
||||
if read_columns.is_some() {
|
||||
return Err(Error::InvalidInput {
|
||||
message: "read_columns applies only to a BatchUDF transform; \
|
||||
a computed column's inputs come from its expression"
|
||||
.into(),
|
||||
});
|
||||
}
|
||||
parent.add_computed_columns(&computed).await
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
parent.add_columns(transform, read_columns).await
|
||||
}
|
||||
}
|
||||
|
||||
@@ -144,8 +85,8 @@ mod tests {
|
||||
use arrow_schema::{DataType, Field, Schema};
|
||||
use lance::dataset::{BatchUDF, NewColumnTransform};
|
||||
|
||||
use crate::Table;
|
||||
use crate::connect;
|
||||
use crate::{Error, Table};
|
||||
|
||||
async fn table_with_two_columns(name: &str) -> Table {
|
||||
let conn = connect("memory://").execute().await.unwrap();
|
||||
@@ -157,7 +98,10 @@ mod tests {
|
||||
async fn test_requires_a_transform() {
|
||||
let table = table_with_two_columns("no_transform").await;
|
||||
let err = table.add_columns().execute().await.unwrap_err();
|
||||
assert!(matches!(err, Error::InvalidInput { .. }));
|
||||
assert!(
|
||||
err.to_string().contains("requires a transform"),
|
||||
"got: {err}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
@@ -173,7 +117,7 @@ mod tests {
|
||||
.execute()
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, Error::InvalidInput { .. }));
|
||||
assert!(err.to_string().contains("BatchUDF"), "got: {err}");
|
||||
|
||||
let schema = table.schema().await.unwrap();
|
||||
assert!(
|
||||
@@ -182,47 +126,6 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_mixing_transform_and_computed_is_rejected() {
|
||||
let table = table_with_two_columns("mixed_add").await;
|
||||
let err = table
|
||||
.add_columns()
|
||||
.transform(NewColumnTransform::SqlExpressions(vec![(
|
||||
"eager".into(),
|
||||
"x * 2".into(),
|
||||
)]))
|
||||
.computed("lazy", "x * 3")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, Error::InvalidInput { .. }));
|
||||
|
||||
let schema = table.schema().await.unwrap();
|
||||
assert!(schema.field_with_name("eager").is_err());
|
||||
assert!(schema.field_with_name("lazy").is_err());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_read_columns_with_computed_is_rejected() {
|
||||
let table = table_with_two_columns("read_cols_computed").await;
|
||||
let err = table
|
||||
.add_columns()
|
||||
.computed("doubled", "x * 2")
|
||||
.read_columns(["x"])
|
||||
.execute()
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, Error::InvalidInput { .. }));
|
||||
assert!(
|
||||
table
|
||||
.schema()
|
||||
.await
|
||||
.unwrap()
|
||||
.field_with_name("doubled")
|
||||
.is_err()
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_read_columns_limits_what_a_batch_udf_sees() {
|
||||
let table = table_with_two_columns("read_cols_udf").await;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -17,7 +17,7 @@ use datafusion_physical_plan::stream::RecordBatchStreamAdapter;
|
||||
use datafusion_physical_plan::{
|
||||
DisplayAs, DisplayFormatType, ExecutionPlan, ExecutionPlanProperties, PlanProperties,
|
||||
};
|
||||
use futures::StreamExt;
|
||||
use futures::TryStreamExt;
|
||||
use lance::Dataset;
|
||||
use lance::dataset::transaction::{Operation, Transaction};
|
||||
use lance::dataset::{CommitBuilder, InsertBuilder, WriteParams, WriteProgressFn};
|
||||
@@ -194,23 +194,12 @@ impl ExecutionPlan for InsertExec {
|
||||
|
||||
let output_bytes = MetricBuilder::new(&self.metrics).output_bytes(partition);
|
||||
let input_schema = input_stream.schema();
|
||||
let declared: Vec<String> = crate::table::computed_columns::computed_columns(
|
||||
&arrow_schema::Schema::from(self.dataset.schema()),
|
||||
)
|
||||
.into_iter()
|
||||
.map(|declaration| declaration.name)
|
||||
.collect();
|
||||
let input_stream: SendableRecordBatchStream =
|
||||
Box::pin(InstrumentedRecordBatchStreamAdapter::new(
|
||||
input_schema,
|
||||
input_stream.map(move |batch| {
|
||||
let batch = batch?;
|
||||
crate::table::computed_columns::ensure_batch_writes_no_computed_values(
|
||||
&declared, &batch,
|
||||
)
|
||||
.map_err(|e| datafusion::error::DataFusionError::External(Box::new(e)))?;
|
||||
input_stream.map_ok(move |batch| {
|
||||
output_bytes.add(batch.get_array_memory_size());
|
||||
Ok(batch)
|
||||
batch
|
||||
}),
|
||||
partition,
|
||||
&self.metrics,
|
||||
|
||||
@@ -94,16 +94,7 @@ pub(crate) async fn set_lsm_write_spec(table: &NativeTable, spec: LsmWriteSpec)
|
||||
.await?
|
||||
};
|
||||
|
||||
table.checkout_latest().await?;
|
||||
let mut dataset = (*table.dataset.get().await?).clone();
|
||||
let schema = arrow_schema::Schema::from(dataset.schema());
|
||||
if !crate::table::computed_columns::computed_columns(&schema).is_empty() {
|
||||
return Err(Error::NotSupported {
|
||||
message: "an LSM write spec cannot be installed on a table with computed \
|
||||
columns: rows in un-compacted tiers are invisible to refresh"
|
||||
.into(),
|
||||
});
|
||||
}
|
||||
let mut builder = dataset.initialize_mem_wal();
|
||||
let writer_config_defaults = match spec {
|
||||
LsmWriteSpec::Bucket {
|
||||
@@ -192,36 +183,6 @@ fn index_name_list(indices: &[IndexConfig]) -> String {
|
||||
format!("[{}]", names.join(", "))
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// require_mem_wal_index_catchup
|
||||
// =============================================================================
|
||||
|
||||
/// Switch this table to required index catch-up, one way.
|
||||
///
|
||||
/// Deliberately **not** part of installing the write spec. Until something can
|
||||
/// actually repair coverage, a table carrying the bit reports every index as
|
||||
/// not known to hold the compacted rows, so its SSTables are retained
|
||||
/// indefinitely -- and the WAL pod trims on the legacy rule meanwhile, leaving
|
||||
/// readers pointed at files that are gone. Turn this on only once remote
|
||||
/// maintenance owns the merge and the repair for the table.
|
||||
///
|
||||
/// Lance refuses the activation if the table already records SSTable
|
||||
/// compaction progress: those numbers predate this protocol and cannot be
|
||||
/// validated, so such a table must be drained rather than activated.
|
||||
#[allow(clippy::redundant_pub_crate)]
|
||||
pub(crate) async fn require_mem_wal_index_catchup(table: &NativeTable) -> Result<()> {
|
||||
table.dataset.ensure_mutable()?;
|
||||
let mut dataset = (*table.dataset.get().await?).clone();
|
||||
if dataset.mem_wal_index_details().await?.is_none() {
|
||||
return Err(Error::InvalidInput {
|
||||
message: "require_mem_wal_index_catchup: no LSM write spec is set on this table".into(),
|
||||
});
|
||||
}
|
||||
dataset.require_mem_wal_index_catchup().await?;
|
||||
table.dataset.update(dataset);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// unset_lsm_write_spec
|
||||
// =============================================================================
|
||||
|
||||
@@ -36,7 +36,6 @@ use lance::dataset::mem_wal::{
|
||||
DatasetMemWalExt, LsmScanner, ShardManifestStore, ShardSnapshot, ShardWriterConfig,
|
||||
};
|
||||
use lance_index::mem_wal::{MemWalIndexDetails, ShardManifest};
|
||||
use lance_table::feature_flags::FLAG_MEM_WAL_INDEX_CATCHUP;
|
||||
use uuid::Uuid;
|
||||
|
||||
use super::NativeTable;
|
||||
@@ -85,8 +84,9 @@ pub(super) async fn create_lsm_plan(
|
||||
let pk_columns = pk_columns(&ds_ref)?;
|
||||
// The base index an indexed arm relies on may lag compaction; resolve it so the
|
||||
// snapshot retains SSTables the index has not yet caught up to.
|
||||
let arm_indexes = arm_maintained_index_names(&ds_ref, &query, &details).await?;
|
||||
let (snapshots, in_memory) = build_read_context(table, &ds_ref, &details, &arm_indexes).await?;
|
||||
let arm_index = arm_maintained_index_name(&ds_ref, &query, &details).await?;
|
||||
let (snapshots, in_memory) =
|
||||
build_read_context(table, &ds_ref, &details, arm_index.as_deref()).await?;
|
||||
|
||||
let limit = query.base.limit;
|
||||
let offset = query.base.offset;
|
||||
@@ -232,44 +232,28 @@ fn pk_columns(dataset: &Dataset) -> Result<Vec<String>> {
|
||||
Ok(pk)
|
||||
}
|
||||
|
||||
/// Per-shard SSTable exclusion watermark: the generation at or below which
|
||||
/// SSTables are safe to drop for this query.
|
||||
///
|
||||
/// A generation is droppable only once it is compacted into the base table AND
|
||||
/// covered by the catch-up of every index the query relies on, so the watermark
|
||||
/// is the minimum across `index_names`. Gating on fewer than all of them would
|
||||
/// drop SSTables holding rows an uncounted index has not yet indexed, and that
|
||||
/// arm would silently return fewer rows.
|
||||
///
|
||||
/// See [`arm_maintained_index_names`] for which indexes are collected today: a
|
||||
/// vector search with a scalar prefilter is not yet among them.
|
||||
///
|
||||
/// An empty `index_names` (a plain scan) uses the compaction watermark alone.
|
||||
/// First occurrence per shard mirrors Lance's `compacted_generation_for_shard`.
|
||||
/// Per-shard SSTable exclusion watermark: the generation at or below which SSTables
|
||||
/// are safe to drop for this arm. A generation is droppable only once it is
|
||||
/// compacted into the base table AND covered by `index_name`'s catch-up (for an
|
||||
/// indexed arm); a plain scan (`index_name == None`) uses the compaction watermark
|
||||
/// alone. Capping at the index catch-up keeps rows the base index has not yet
|
||||
/// indexed visible through their SSTable. First occurrence per shard mirrors Lance's
|
||||
/// `compacted_generation_for_shard`.
|
||||
fn exclusion_watermarks(
|
||||
details: &MemWalIndexDetails,
|
||||
index_names: &[String],
|
||||
catchup_required: bool,
|
||||
index_name: Option<&str>,
|
||||
) -> HashMap<Uuid, u64> {
|
||||
let mut exclude: HashMap<Uuid, u64> = HashMap::new();
|
||||
for entry in &details.compacted_sstables {
|
||||
let mut watermark = entry.generation;
|
||||
for name in index_names {
|
||||
match details
|
||||
if let Some(name) = index_name
|
||||
&& let Some(caught_up) = details
|
||||
.index_catchup
|
||||
.iter()
|
||||
.find(|icp| icp.index_name == *name)
|
||||
.find(|icp| icp.index_name == name)
|
||||
.and_then(|icp| icp.caught_up_generation_for_shard(&entry.shard_id))
|
||||
{
|
||||
Some(caught_up) => watermark = watermark.min(caught_up),
|
||||
// No entry. On a table that requires catch-up this means the
|
||||
// index is *not* known to hold these rows, and the base arm is
|
||||
// index-only -- so every generation stays readable from its
|
||||
// SSTable. Without the bit the field is not maintained at all,
|
||||
// and absence carries no information.
|
||||
None if catchup_required => watermark = 0,
|
||||
None => {}
|
||||
}
|
||||
{
|
||||
watermark = watermark.min(caught_up);
|
||||
}
|
||||
exclude.entry(entry.shard_id).or_insert(watermark);
|
||||
}
|
||||
@@ -283,26 +267,13 @@ fn exclusion_watermarks(
|
||||
/// with a live cached `ShardWriter` (this session's in-flight writes) the
|
||||
/// writer's authoritative in-memory manifest and memtables override the
|
||||
/// on-disk view so a read sees data not yet flushed.
|
||||
/// Whether this table reads a missing `index_catchup` entry as "not caught up".
|
||||
///
|
||||
/// Both words must be set. A reader honouring the bit while a writer does not
|
||||
/// would retain SSTables the writer had already trimmed, and the reverse would
|
||||
/// serve rows from files the writer still expects to be excluded -- so a
|
||||
/// half-set manifest is treated as legacy, which is the conservative side.
|
||||
fn requires_index_catchup(dataset: &Dataset) -> bool {
|
||||
let manifest = dataset.manifest();
|
||||
manifest.reader_feature_flags & FLAG_MEM_WAL_INDEX_CATCHUP != 0
|
||||
&& manifest.writer_feature_flags & FLAG_MEM_WAL_INDEX_CATCHUP != 0
|
||||
}
|
||||
|
||||
async fn build_read_context(
|
||||
table: &NativeTable,
|
||||
dataset: &Dataset,
|
||||
details: &MemWalIndexDetails,
|
||||
index_names: &[String],
|
||||
index_name: Option<&str>,
|
||||
) -> Result<(Vec<ShardSnapshot>, HashMap<Uuid, InMemoryMemTables>)> {
|
||||
let catchup_required = requires_index_catchup(dataset);
|
||||
let exclude = exclusion_watermarks(details, index_names, catchup_required);
|
||||
let exclude = exclusion_watermarks(details, index_name);
|
||||
|
||||
let shard_ids = dataset.list_mem_wal_latest_shard_ids().await?;
|
||||
// Use the dataset's own object store (not `ObjectStore::from_uri`, which
|
||||
@@ -516,33 +487,19 @@ async fn index_maintained(
|
||||
}))
|
||||
}
|
||||
|
||||
/// Every maintained base index this query relies on, used to gate SSTable
|
||||
/// exclusion by index catch-up.
|
||||
///
|
||||
/// Returns a list because the watermark must be the lowest across every index a
|
||||
/// query relies on. Today it never holds more than one: `reject_unsupported`
|
||||
/// refuses hybrid search, so the vector and full-text arms are mutually
|
||||
/// exclusive.
|
||||
///
|
||||
/// The case that is genuinely multi-index -- a vector search with a scalar or
|
||||
/// bitmap prefilter -- is **not collected yet**. Identifying those needs the
|
||||
/// planner's chosen indexes, not the columns the filter names, and no Lance API
|
||||
/// exposes them. Until it does, such a query is gated on its vector index alone.
|
||||
///
|
||||
/// Empty for a plain scan, or when no maintained index covers the searched
|
||||
/// column.
|
||||
async fn arm_maintained_index_names(
|
||||
/// The maintained base index the query's arm relies on (vector index for ANN, FTS
|
||||
/// index for full-text), used to gate SSTable compaction exclusion by index catch-up.
|
||||
/// `None` for a plain scan or when no maintained index covers the searched column.
|
||||
async fn arm_maintained_index_name(
|
||||
dataset: &Dataset,
|
||||
query: &VectorQueryRequest,
|
||||
details: &MemWalIndexDetails,
|
||||
) -> Result<Vec<String>> {
|
||||
) -> Result<Option<String>> {
|
||||
use lance::index::DatasetIndexExt;
|
||||
|
||||
// Each arm's searched column, the index-detail type it relies on, and a
|
||||
// Resolve the arm's searched column, the index-detail type it relies on, and a
|
||||
// label for diagnostics — catch-up is taken from the vector/FTS index
|
||||
// specifically, not a BTree on the same column.
|
||||
let mut arms: Vec<(String, &str, &str)> = Vec::new();
|
||||
if !query.query_vector.is_empty() {
|
||||
let (column, type_url_suffix, arm) = if !query.query_vector.is_empty() {
|
||||
let arrow_schema = ArrowSchema::from(dataset.schema());
|
||||
let column = match &query.column {
|
||||
Some(column) => column.clone(),
|
||||
@@ -551,43 +508,31 @@ async fn arm_maintained_index_names(
|
||||
default_vector_column(&arrow_schema, dim)?
|
||||
}
|
||||
};
|
||||
arms.push((column, "VectorIndexDetails", "vector"));
|
||||
}
|
||||
if let Some(fts) = &query.base.full_text_search
|
||||
&& let Some(column) = fts.columns().into_iter().next()
|
||||
{
|
||||
arms.push((column, "InvertedIndexDetails", "full-text"));
|
||||
}
|
||||
if arms.is_empty() {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
|
||||
let indices = dataset.load_indices().await?;
|
||||
let mut names = Vec::with_capacity(arms.len());
|
||||
for (column, type_url_suffix, arm) in arms {
|
||||
let Some(field) = dataset.schema().field(&column) else {
|
||||
continue;
|
||||
};
|
||||
let segment_names: Vec<String> = indices
|
||||
.iter()
|
||||
.filter(|idx| {
|
||||
idx.fields.contains(&field.id)
|
||||
&& idx
|
||||
.index_details
|
||||
.as_ref()
|
||||
.is_some_and(|d| d.type_url.ends_with(type_url_suffix))
|
||||
})
|
||||
.map(|idx| idx.name.clone())
|
||||
.collect();
|
||||
if let Some(name) =
|
||||
resolve_single_index(segment_names, &details.maintained_indexes, arm, &column)?
|
||||
{
|
||||
names.push(name);
|
||||
(column, "VectorIndexDetails", "vector")
|
||||
} else if let Some(fts) = &query.base.full_text_search {
|
||||
match fts.columns().into_iter().next() {
|
||||
Some(column) => (column, "InvertedIndexDetails", "full-text"),
|
||||
None => return Ok(None),
|
||||
}
|
||||
}
|
||||
names.sort();
|
||||
names.dedup();
|
||||
Ok(names)
|
||||
} else {
|
||||
return Ok(None);
|
||||
};
|
||||
let Some(field) = dataset.schema().field(&column) else {
|
||||
return Ok(None);
|
||||
};
|
||||
let indices = dataset.load_indices().await?;
|
||||
let segment_names: Vec<String> = indices
|
||||
.iter()
|
||||
.filter(|idx| {
|
||||
idx.fields.contains(&field.id)
|
||||
&& idx
|
||||
.index_details
|
||||
.as_ref()
|
||||
.is_some_and(|d| d.type_url.ends_with(type_url_suffix))
|
||||
})
|
||||
.map(|idx| idx.name.clone())
|
||||
.collect();
|
||||
resolve_single_index(segment_names, &details.maintained_indexes, arm, &column)
|
||||
}
|
||||
|
||||
/// Resolve the single logical index from the names of its matching physical
|
||||
@@ -789,117 +734,22 @@ mod tests {
|
||||
};
|
||||
|
||||
// Plain scan: drop every compacted generation (through 5).
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &[], false).get(&shard),
|
||||
Some(&5)
|
||||
);
|
||||
assert_eq!(exclusion_watermarks(&details, None).get(&shard), Some(&5));
|
||||
|
||||
// FTS arm with a lagging index: exclusion is capped at the index catch-up
|
||||
// (2), so SSTable generations 3..=5 are retained until the index covers
|
||||
// them — otherwise those documents would silently vanish from FTS results.
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &["fts_idx".to_string()], false).get(&shard),
|
||||
exclusion_watermarks(&details, Some("fts_idx")).get(&shard),
|
||||
Some(&2)
|
||||
);
|
||||
|
||||
// A caught-up index — or one untracked in index_catchup — falls back to the
|
||||
// compaction watermark.
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &["caught_up_idx".to_string()], false).get(&shard),
|
||||
exclusion_watermarks(&details, Some("caught_up_idx")).get(&shard),
|
||||
Some(&5)
|
||||
);
|
||||
|
||||
// The same missing entry, once the table requires catch-up: absence now
|
||||
// means "not known to hold these rows", so nothing may be excluded and
|
||||
// every generation stays readable from its SSTable. This is the whole
|
||||
// point of the protocol -- an indexed query against a table whose index
|
||||
// has not caught up must not silently lose rows.
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &["untracked_idx".to_string()], true).get(&shard),
|
||||
Some(&0)
|
||||
);
|
||||
|
||||
// A tracked index is unaffected by the mode: the recorded position is
|
||||
// information either way, and it still caps the exclusion.
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &["fts_idx".to_string()], true).get(&shard),
|
||||
Some(&2)
|
||||
);
|
||||
|
||||
// One missing entry is enough to hold everything back, even alongside an
|
||||
// index that has caught up.
|
||||
let mixed = vec!["fts_idx".to_string(), "untracked_idx".to_string()];
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &mixed, true).get(&shard),
|
||||
Some(&0)
|
||||
);
|
||||
}
|
||||
|
||||
/// A hybrid search reads a vector and a full-text index, and either may lag.
|
||||
/// Retaining to the lower of the two is what keeps both arms complete;
|
||||
/// gating on one alone would drop SSTables the other has not indexed.
|
||||
#[test]
|
||||
fn exclusion_watermark_takes_the_minimum_across_every_index_used() {
|
||||
let shard = Uuid::from_u128(1);
|
||||
let details = MemWalIndexDetails {
|
||||
compacted_sstables: vec![CompactedSsTable::new(shard, 9)],
|
||||
index_catchup: vec![
|
||||
IndexCatchupProgress::new(
|
||||
"vec_idx".to_string(),
|
||||
vec![CompactedSsTable::new(shard, 7)],
|
||||
),
|
||||
IndexCatchupProgress::new(
|
||||
"fts_idx".to_string(),
|
||||
vec![CompactedSsTable::new(shard, 4)],
|
||||
),
|
||||
],
|
||||
maintained_indexes: vec!["vec_idx".to_string(), "fts_idx".to_string()],
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
// Each index alone stops at its own catch-up.
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &["vec_idx".to_string()], false).get(&shard),
|
||||
Some(&7)
|
||||
);
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &["fts_idx".to_string()], false).get(&shard),
|
||||
Some(&4)
|
||||
);
|
||||
|
||||
// Used together, the lower one governs regardless of order.
|
||||
let both = ["vec_idx".to_string(), "fts_idx".to_string()];
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &both, false).get(&shard),
|
||||
Some(&4)
|
||||
);
|
||||
let reversed = ["fts_idx".to_string(), "vec_idx".to_string()];
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &reversed, false).get(&shard),
|
||||
Some(&4)
|
||||
);
|
||||
}
|
||||
|
||||
/// An index with no catch-up entry contributes no cap today, so a lagging
|
||||
/// sibling must still govern rather than being widened by the untracked one.
|
||||
#[test]
|
||||
fn an_untracked_index_does_not_widen_a_lagging_sibling() {
|
||||
let shard = Uuid::from_u128(1);
|
||||
let details = MemWalIndexDetails {
|
||||
compacted_sstables: vec![CompactedSsTable::new(shard, 9)],
|
||||
index_catchup: vec![IndexCatchupProgress::new(
|
||||
"fts_idx".to_string(),
|
||||
vec![CompactedSsTable::new(shard, 4)],
|
||||
)],
|
||||
maintained_indexes: vec!["fts_idx".to_string(), "untracked_idx".to_string()],
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let both = ["fts_idx".to_string(), "untracked_idx".to_string()];
|
||||
assert_eq!(
|
||||
exclusion_watermarks(&details, &both, false).get(&shard),
|
||||
Some(&4)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -1,954 +0,0 @@
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||
|
||||
//! Filling computed columns.
|
||||
//!
|
||||
//! A row without a value gets one; a row that has one keeps it. Refresh is
|
||||
//! therefore idempotent and does not observe input mutation -- once a row is
|
||||
//! filled, changing what the expression reads leaves the stored result alone.
|
||||
//!
|
||||
//! Two passes per fragment. The first scans only the unfilled live rows and
|
||||
//! evaluates the expression over them, which yields the exact fill count and
|
||||
//! decides whether the fragment is staged at all -- a fragment where nothing
|
||||
//! would change stages nothing, which is what lets an expression yielding
|
||||
//! null settle instead of restaging forever. The second streams the
|
||||
//! fragment's physical rows into `write_column` a batch at a time, so peak
|
||||
//! memory is bounded by a scan batch. The expression is evaluated by this
|
||||
//! module, never through a projection alias, and only over rows being
|
||||
//! filled: every other row -- deleted, or already holding a value -- has its
|
||||
//! inputs masked to null first, so a poison value in a row nobody is filling
|
||||
//! cannot fail the refresh.
|
||||
|
||||
use std::sync::Arc;
|
||||
|
||||
use arrow_array::{ArrayRef, BooleanArray, RecordBatch, RecordBatchOptions};
|
||||
use arrow_schema::Schema as ArrowSchema;
|
||||
use datafusion_expr::ColumnarValue;
|
||||
use futures::{Stream, StreamExt, TryStreamExt};
|
||||
use lance::Dataset;
|
||||
use lance::dataset::WriteDestination;
|
||||
use lance::dataset::fragment::FileFragment;
|
||||
use lance::dataset::transaction::Operation;
|
||||
use lance_core::ROW_ID;
|
||||
use lance_core::datatypes::Schema as LanceSchema;
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use super::computed_columns::{BoundExpression, ComputedColumnKind, computed_column_from_field};
|
||||
use super::{BaseTable, NativeTable};
|
||||
use crate::job::Job;
|
||||
use crate::{Error, Result};
|
||||
|
||||
/// The result of refreshing a computed column.
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
|
||||
pub struct RefreshColumnResult {
|
||||
/// Rows that had a value computed.
|
||||
#[serde(default)]
|
||||
pub rows_filled: u64,
|
||||
/// The commit version associated with the operation.
|
||||
#[serde(default)]
|
||||
pub version: u64,
|
||||
}
|
||||
|
||||
/// Internal implementation of the refresh logic.
|
||||
pub(crate) async fn execute_refresh_column(
|
||||
table: &NativeTable,
|
||||
column: &str,
|
||||
) -> Result<RefreshColumnResult> {
|
||||
table.dataset.ensure_mutable()?;
|
||||
ensure_no_lsm_write_spec(table).await?;
|
||||
let dataset = table.dataset.get().await?;
|
||||
|
||||
let expression = declared_expression(&dataset, column)?;
|
||||
let schema = Arc::new(ArrowSchema::from(dataset.schema()));
|
||||
let bound = Arc::new(super::computed_columns::bind(schema, column, &expression)?);
|
||||
let field = dataset
|
||||
.schema()
|
||||
.field(column)
|
||||
.ok_or_else(|| Error::ColumnNotFound {
|
||||
name: column.to_string(),
|
||||
})?;
|
||||
// The dataset's own field, so the identity write_column checks against the
|
||||
// manifest holds by construction.
|
||||
let column_schema = LanceSchema {
|
||||
fields: vec![field.clone()],
|
||||
metadata: Default::default(),
|
||||
};
|
||||
|
||||
let mut rows_filled = 0u64;
|
||||
let mut replacements = Vec::new();
|
||||
for fragment in dataset.get_fragments() {
|
||||
let gained = count_fragment_gains(&dataset, &fragment, &bound, column).await?;
|
||||
if gained == 0 {
|
||||
continue;
|
||||
}
|
||||
rows_filled += gained;
|
||||
let values = fill_stream(&dataset, &fragment, bound.clone(), column).await?;
|
||||
replacements.push(fragment.write_column(values, &column_schema).await?);
|
||||
}
|
||||
|
||||
if replacements.is_empty() {
|
||||
return Ok(RefreshColumnResult {
|
||||
rows_filled: 0,
|
||||
version: dataset.version().version,
|
||||
});
|
||||
}
|
||||
|
||||
let read_version = dataset.version().version;
|
||||
// The dataset's own session, so registrations and caches survive the
|
||||
// commit being installed on the handle.
|
||||
let session = dataset.session();
|
||||
let new_dataset = Dataset::commit(
|
||||
WriteDestination::Dataset(dataset.clone()),
|
||||
Operation::DataReplacement { replacements },
|
||||
Some(read_version),
|
||||
None,
|
||||
None,
|
||||
session,
|
||||
false,
|
||||
)
|
||||
.await?;
|
||||
|
||||
let version = new_dataset.version().version;
|
||||
table.dataset.update(new_dataset);
|
||||
Ok(RefreshColumnResult {
|
||||
rows_filled,
|
||||
version,
|
||||
})
|
||||
}
|
||||
|
||||
/// Run the refresh as a [`Job`] in this process.
|
||||
pub(crate) async fn execute_refresh_column_async(table: &NativeTable, column: &str) -> Result<Job> {
|
||||
// Validate before spawning so bad input is reported by this call rather
|
||||
// than only by the job.
|
||||
table.dataset.ensure_mutable()?;
|
||||
ensure_no_lsm_write_spec(table).await?;
|
||||
let dataset = table.dataset.get().await?;
|
||||
declared_expression(&dataset, column)?;
|
||||
drop(dataset);
|
||||
|
||||
let table = table.clone();
|
||||
let column = column.to_string();
|
||||
Ok(Job::spawned(tokio::spawn(async move {
|
||||
execute_refresh_column(&table, &column).await?;
|
||||
table.bump_freshness();
|
||||
Ok(())
|
||||
})))
|
||||
}
|
||||
|
||||
/// Refuse to refresh under an LSM write spec.
|
||||
///
|
||||
/// Refresh enumerates base fragments, and a write spec keeps visible rows in
|
||||
/// un-compacted MemWAL tiers it cannot reach -- success would silently omit
|
||||
/// readable rows.
|
||||
async fn ensure_no_lsm_write_spec(table: &NativeTable) -> Result<()> {
|
||||
// The catch-up flag outlives unset and marks retained SSTable rows.
|
||||
let catchup = table.dataset.get().await?.manifest().reader_feature_flags
|
||||
& lance_table::feature_flags::FLAG_MEM_WAL_INDEX_CATCHUP
|
||||
!= 0;
|
||||
if catchup || table.get_lsm_write_spec().await?.is_some() {
|
||||
return Err(Error::NotSupported {
|
||||
message: "refresh_column is not supported on a table with an LSM write \
|
||||
spec: rows in un-compacted tiers are invisible to refresh"
|
||||
.into(),
|
||||
});
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The SQL expression `column` is declared with.
|
||||
fn declared_expression(dataset: &Dataset, column: &str) -> Result<String> {
|
||||
let schema = ArrowSchema::from(dataset.schema());
|
||||
let field = schema
|
||||
.field_with_name(column)
|
||||
.map_err(|_| Error::ColumnNotFound {
|
||||
name: column.to_string(),
|
||||
})?;
|
||||
let declaration =
|
||||
computed_column_from_field(field).ok_or_else(|| Error::NotAComputedColumn {
|
||||
name: column.to_string(),
|
||||
})?;
|
||||
match declaration.kind {
|
||||
ComputedColumnKind::Sql { expression } => Ok(expression),
|
||||
ComputedColumnKind::Unrecognized { kind } => Err(Error::NotSupported {
|
||||
message: format!(
|
||||
"computed column '{column}' is defined by '{kind}', which this version of \
|
||||
lancedb cannot evaluate"
|
||||
),
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
/// Quote `name` as a lance SQL identifier.
|
||||
///
|
||||
/// Lance's dialect delimits with backticks, so a double-quoted name would
|
||||
/// parse as a string literal rather than a column.
|
||||
fn quote_identifier(name: &str) -> String {
|
||||
format!("`{}`", name.replace('`', "``"))
|
||||
}
|
||||
|
||||
/// Assemble the batch evaluation runs against: the bound roots, in read-schema
|
||||
/// order. Built by name so scan-side column order never matters.
|
||||
fn evaluation_batch(
|
||||
batch: &RecordBatch,
|
||||
bound: &BoundExpression,
|
||||
mask_out: Option<&BooleanArray>,
|
||||
) -> lance_core::Result<RecordBatch> {
|
||||
let mut columns = Vec::with_capacity(bound.roots.len());
|
||||
for name in &bound.roots {
|
||||
let column = batch.column_by_name(name).ok_or_else(|| {
|
||||
lance_core::Error::invalid_input(format!(
|
||||
"refreshing a computed column read no {name} column"
|
||||
))
|
||||
})?;
|
||||
// Rows outside the mask must not reach the expression: a value in a
|
||||
// deleted or already-filled row can be one it would choke on.
|
||||
columns.push(match mask_out {
|
||||
Some(mask) => arrow::compute::nullif(column, mask)?,
|
||||
None => column.clone(),
|
||||
});
|
||||
}
|
||||
Ok(RecordBatch::try_new_with_options(
|
||||
bound.read_schema.clone(),
|
||||
columns,
|
||||
&RecordBatchOptions::new().with_row_count(Some(batch.num_rows())),
|
||||
)?)
|
||||
}
|
||||
|
||||
/// Evaluate the expression over `batch`, materializing a constant result to
|
||||
/// the batch's length.
|
||||
fn evaluate(bound: &BoundExpression, batch: &RecordBatch) -> lance_core::Result<ArrayRef> {
|
||||
let value = bound
|
||||
.physical
|
||||
.evaluate(batch)
|
||||
.map_err(lance_core::Error::from)?;
|
||||
match value {
|
||||
ColumnarValue::Array(array) => Ok(array),
|
||||
scalar => scalar
|
||||
.into_array(batch.num_rows())
|
||||
.map_err(lance_core::Error::from),
|
||||
}
|
||||
}
|
||||
|
||||
/// How many rows of one fragment would gain a value.
|
||||
///
|
||||
/// Scans only the unfilled live rows -- deleted rows never reach the
|
||||
/// expression here, the filter having already excluded them -- and counts the
|
||||
/// non-null results. Exact, so it is both the staging decision and the
|
||||
/// fragment's contribution to `rows_filled`.
|
||||
async fn count_fragment_gains(
|
||||
dataset: &Dataset,
|
||||
fragment: &FileFragment,
|
||||
bound: &BoundExpression,
|
||||
column: &str,
|
||||
) -> Result<u64> {
|
||||
let mut scanner = dataset.scan();
|
||||
scanner
|
||||
.with_fragments(vec![fragment.metadata().clone()])
|
||||
.with_row_id()
|
||||
.filter(&format!("{} IS NULL", quote_identifier(column)))?
|
||||
.project(&bound.roots)?;
|
||||
|
||||
let mut gained = 0u64;
|
||||
let mut batches = scanner.try_into_stream().await?;
|
||||
while let Some(batch) = batches.try_next().await? {
|
||||
let evaluated = evaluate(bound, &evaluation_batch(&batch, bound, None)?)?;
|
||||
gained += (batch.num_rows() - evaluated.null_count()) as u64;
|
||||
}
|
||||
Ok(gained)
|
||||
}
|
||||
|
||||
/// Stream one fragment's column in physical order, filling the unfilled live
|
||||
/// rows and keeping every other value.
|
||||
///
|
||||
/// Deleted rows are carried through so the values line up positionally with
|
||||
/// the fragment's data files; they are never read back, but the column file
|
||||
/// has to cover them.
|
||||
async fn fill_stream(
|
||||
dataset: &Dataset,
|
||||
fragment: &FileFragment,
|
||||
bound: Arc<BoundExpression>,
|
||||
column: &str,
|
||||
) -> Result<impl Stream<Item = lance_core::Result<RecordBatch>> + Send + use<>> {
|
||||
let mut projection: Vec<String> = bound.roots.clone();
|
||||
projection.push(column.to_string());
|
||||
let mut scanner = dataset.scan();
|
||||
scanner
|
||||
.with_fragments(vec![fragment.metadata().clone()])
|
||||
.with_row_id()
|
||||
.include_deleted_rows()
|
||||
.project(&projection)?;
|
||||
|
||||
let projected = Arc::new(ArrowSchema::new(vec![
|
||||
ArrowSchema::from(dataset.schema())
|
||||
.field_with_name(column)
|
||||
.map_err(|_| Error::ColumnNotFound {
|
||||
name: column.to_string(),
|
||||
})?
|
||||
.clone(),
|
||||
]));
|
||||
|
||||
let column = column.to_string();
|
||||
let batches = scanner.try_into_stream().await?;
|
||||
Ok(batches.map(move |batch| {
|
||||
let batch = batch?;
|
||||
let missing = |name: &str| {
|
||||
lance_core::Error::invalid_input(format!(
|
||||
"refreshing a computed column read no {name} column"
|
||||
))
|
||||
};
|
||||
let existing = batch
|
||||
.column_by_name(&column)
|
||||
.ok_or_else(|| missing(&column))?;
|
||||
let row_ids = batch
|
||||
.column_by_name(ROW_ID)
|
||||
.ok_or_else(|| missing(ROW_ID))?;
|
||||
|
||||
// Only an unfilled live row gains a value; a deleted row has a null
|
||||
// row id and keeps its (null) slot.
|
||||
let unfilled = arrow::compute::is_null(existing.as_ref())?;
|
||||
let live = arrow::compute::is_not_null(row_ids.as_ref())?;
|
||||
let fill = arrow::compute::and(&unfilled, &live)?;
|
||||
let keep = arrow::compute::not(&fill)?;
|
||||
|
||||
let computed = evaluate(&bound, &evaluation_batch(&batch, &bound, Some(&keep))?)?;
|
||||
let merged = arrow_select::zip::zip(&fill, &computed, existing)?;
|
||||
Ok(RecordBatch::try_new(projected.clone(), vec![merged])?)
|
||||
}))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use std::sync::Arc;
|
||||
|
||||
use arrow_array::{Int32Array, record_batch};
|
||||
use futures::TryStreamExt;
|
||||
|
||||
use crate::connect;
|
||||
use crate::query::{ExecutableQuery, QueryBase, Select};
|
||||
use crate::{Error, Result, Table};
|
||||
|
||||
async fn table_with(name: &str, values: Vec<i32>) -> Table {
|
||||
let conn = connect("memory://").execute().await.unwrap();
|
||||
let batch = record_batch!(("x", Int32, values)).unwrap();
|
||||
conn.create_table(name, batch).execute().await.unwrap()
|
||||
}
|
||||
|
||||
async fn declare_doubled(table: &Table) -> Result<u64> {
|
||||
Ok(table
|
||||
.add_columns()
|
||||
.computed("doubled", "x * 2")
|
||||
.execute()
|
||||
.await?
|
||||
.version)
|
||||
}
|
||||
|
||||
async fn read(table: &Table, column: &str) -> Vec<Option<i32>> {
|
||||
let batches = table
|
||||
.query()
|
||||
.select(Select::columns(&[column]))
|
||||
.execute()
|
||||
.await
|
||||
.unwrap()
|
||||
.try_collect::<Vec<_>>()
|
||||
.await
|
||||
.unwrap();
|
||||
let mut values: Vec<Option<i32>> = batches
|
||||
.iter()
|
||||
.flat_map(|batch| {
|
||||
batch[column]
|
||||
.as_any()
|
||||
.downcast_ref::<Int32Array>()
|
||||
.unwrap()
|
||||
.iter()
|
||||
.collect::<Vec<_>>()
|
||||
})
|
||||
.collect();
|
||||
values.sort();
|
||||
values
|
||||
}
|
||||
|
||||
async fn append(table: &Table, values: Vec<i32>) {
|
||||
let batch = record_batch!(("x", Int32, values)).unwrap();
|
||||
table.add(batch).execute().await.unwrap();
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_fills_a_declared_column() {
|
||||
let table = table_with("refresh_fills", vec![1, 2, 3]).await;
|
||||
let declared = declare_doubled(&table).await.unwrap();
|
||||
assert_eq!(read(&table, "doubled").await, vec![None, None, None]);
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert!(result.version > declared);
|
||||
assert_eq!(result.rows_filled, 3);
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(4), Some(6)]
|
||||
);
|
||||
}
|
||||
|
||||
/// Values written after the last refresh must be reachable by another one.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_fills_rows_appended_since_the_last_refresh() {
|
||||
let table = table_with("refresh_appended", vec![1, 2]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
|
||||
append(&table, vec![5, 6]).await;
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![None, None, Some(2), Some(4)]
|
||||
);
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 2);
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(4), Some(10), Some(12)]
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_with_nothing_to_fill() {
|
||||
let table = table_with("refresh_noop", vec![1, 2, 3]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
|
||||
let again = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(again.rows_filled, 0);
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(4), Some(6)]
|
||||
);
|
||||
}
|
||||
|
||||
/// A row is filled only by gaining a value, so an expression yielding null
|
||||
/// settles at once instead of re-selecting the same rows forever. Nothing
|
||||
/// is staged, so the version does not move either.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_converges_on_a_null_result() {
|
||||
let table = table_with("refresh_null_result", vec![1, 2, 3]).await;
|
||||
let declared = table
|
||||
.add_columns()
|
||||
.computed("maybe", "nullif(x, x)")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap()
|
||||
.version;
|
||||
|
||||
let first = table.refresh_column("maybe").await.unwrap();
|
||||
assert_eq!(first.rows_filled, 0);
|
||||
assert_eq!(first.version, declared);
|
||||
assert_eq!(read(&table, "maybe").await, vec![None, None, None]);
|
||||
|
||||
let again = table.refresh_column("maybe").await.unwrap();
|
||||
assert_eq!(again.rows_filled, 0);
|
||||
assert_eq!(again.version, declared);
|
||||
}
|
||||
|
||||
/// The contract's boundary: a filled fragment is not revisited, so
|
||||
/// mutating an input leaves the value computed at fill time.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_does_not_observe_input_mutation() {
|
||||
let table = table_with("refresh_mutation", vec![1]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(read(&table, "doubled").await, vec![Some(2)]);
|
||||
|
||||
table.update().column("x", "3").execute().await.unwrap();
|
||||
|
||||
let again = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(again.rows_filled, 0);
|
||||
assert_eq!(read(&table, "doubled").await, vec![Some(2)]);
|
||||
}
|
||||
|
||||
/// A row rewrite before the first refresh materializes the declared
|
||||
/// column as null behind a covering data file. Those rows are still
|
||||
/// unfilled and a later refresh has to reach them.
|
||||
#[tokio::test]
|
||||
async fn test_update_before_the_first_refresh() {
|
||||
let table = table_with("refresh_update_first", vec![1]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
|
||||
table.update().column("x", "3").execute().await.unwrap();
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
assert_eq!(read(&table, "doubled").await, vec![Some(6)]);
|
||||
}
|
||||
|
||||
/// The contract holds row by row, not fragment by fragment: revisiting a
|
||||
/// fragment to fill one row must not recompute a filled row sitting beside
|
||||
/// it, even where the input behind it has since changed.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_does_not_recompute_a_filled_row_beside_an_unfilled_one() {
|
||||
let table = table_with("refresh_mixed", vec![1, 2]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
|
||||
append(&table, vec![5]).await;
|
||||
table
|
||||
.update()
|
||||
.column("x", "100")
|
||||
.only_if("x = 1")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
table
|
||||
.optimize(crate::table::OptimizeAction::Compact {
|
||||
options: crate::table::CompactionOptions::default(),
|
||||
remap_options: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
// 2 is the mutated row keeping the value it was filled with, not 200.
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(4), Some(10)]
|
||||
);
|
||||
}
|
||||
|
||||
/// Filling a fragment must not disturb the values it already holds, which
|
||||
/// is what makes a compaction-mixed fragment safe to revisit.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_preserves_already_filled_rows() {
|
||||
let table = table_with("refresh_preserves", vec![1, 2]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
|
||||
append(&table, vec![5]).await;
|
||||
table
|
||||
.optimize(crate::table::OptimizeAction::Compact {
|
||||
options: crate::table::CompactionOptions::default(),
|
||||
remap_options: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(4), Some(10)]
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_leaves_deleted_rows_alone() {
|
||||
let table = table_with("refresh_deleted", vec![1, 2, 3, 4]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.delete("x = 2").await.unwrap();
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 3);
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(6), Some(8)]
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_a_constant_expression() {
|
||||
let table = table_with("refresh_constant", vec![1, 2, 3]).await;
|
||||
table
|
||||
.add_columns()
|
||||
.computed("answer", "42")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let result = table.refresh_column("answer").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 3);
|
||||
}
|
||||
|
||||
/// A name needing quotes reaches the evaluator intact: it is carried as a
|
||||
/// projection alias, never spliced into SQL text.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_a_column_whose_name_needs_quoting() {
|
||||
let table = table_with("refresh_quoted", vec![1, 2, 3]).await;
|
||||
table
|
||||
.add_columns()
|
||||
.computed("double value", "x * 2")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let result = table.refresh_column("double value").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 3);
|
||||
assert_eq!(
|
||||
read(&table, "double value").await,
|
||||
vec![Some(2), Some(4), Some(6)]
|
||||
);
|
||||
}
|
||||
|
||||
/// A fragment spanning several scan batches exercises the streamed fill:
|
||||
/// the probe buffers only until the first gained value and the rest flows
|
||||
/// through write_column a batch at a time.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_streams_a_multi_batch_fragment() {
|
||||
let values: Vec<i32> = (0..20_000).collect();
|
||||
let table = table_with("refresh_multi_batch", values.clone()).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 20_000);
|
||||
|
||||
let read_back = read(&table, "doubled").await;
|
||||
assert_eq!(read_back.len(), 20_000);
|
||||
let mut expected: Vec<Option<i32>> = values.iter().map(|v| Some(v * 2)).collect();
|
||||
expected.sort();
|
||||
assert_eq!(read_back, expected);
|
||||
}
|
||||
|
||||
/// The gate's reproducer: the commit must reuse the configured session,
|
||||
/// or registrations and caches vanish from the handle after a refresh.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_preserves_the_configured_session() {
|
||||
let session = Arc::new(lance::session::Session::default());
|
||||
let conn = crate::connect("memory://")
|
||||
.session(session.clone())
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
let batch = record_batch!(("x", Int32, [1, 2])).unwrap();
|
||||
let table = conn
|
||||
.create_table("session_kept", batch)
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
|
||||
let dataset = table.as_native().unwrap().dataset.get().await.unwrap();
|
||||
assert!(Arc::ptr_eq(&dataset.session(), &session));
|
||||
}
|
||||
|
||||
/// The async form's job settles with the fill visible, like
|
||||
/// create_index's execute_async.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_async_job_waits_for_the_fill() {
|
||||
let table = table_with("refresh_async", vec![1, 2, 3]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
assert!(job.id().is_none(), "in-process jobs have no server id");
|
||||
job.wait().await.unwrap();
|
||||
assert_eq!(job.status().await.unwrap(), "finished");
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(4), Some(6)]
|
||||
);
|
||||
}
|
||||
|
||||
/// Bad input is reported by the call, not by the job.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_async_rejects_bad_input_before_spawning() {
|
||||
let table = table_with("refresh_async_bad", vec![1, 2, 3]).await;
|
||||
|
||||
let err = table.refresh_column_async("x").await.unwrap_err();
|
||||
assert!(matches!(err, Error::NotAComputedColumn { name } if name == "x"));
|
||||
|
||||
let err = table.refresh_column_async("nope").await.unwrap_err();
|
||||
assert!(matches!(err, Error::ColumnNotFound { name } if name == "nope"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_async_job_reports_success_to_every_waiter() {
|
||||
let table = table_with("refresh_async_waiters", vec![1, 2]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
job.wait().await.unwrap();
|
||||
// A second wait after completion observes the same outcome.
|
||||
job.wait().await.unwrap();
|
||||
assert_eq!(job.status().await.unwrap(), "finished");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_rejects_a_plain_column() {
|
||||
let table = table_with("refresh_plain", vec![1, 2, 3]).await;
|
||||
let err = table.refresh_column("x").await.unwrap_err();
|
||||
assert!(matches!(err, Error::NotAComputedColumn { name } if name == "x"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_rejects_an_unknown_column() {
|
||||
let table = table_with("refresh_missing", vec![1, 2, 3]).await;
|
||||
let err = table.refresh_column("nope").await.unwrap_err();
|
||||
assert!(matches!(err, Error::ColumnNotFound { name } if name == "nope"));
|
||||
}
|
||||
|
||||
/// The gate's reproducer: a poison value in a deleted row must not
|
||||
/// abort filling the live rows, since nobody can read it.
|
||||
#[tokio::test]
|
||||
async fn test_a_deleted_rows_value_is_never_evaluated() {
|
||||
let table = table_with("refresh_deleted_poison", vec![1, 0]).await;
|
||||
table
|
||||
.add_columns()
|
||||
.computed("quotient", "10 / x")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
table.delete("x = 0").await.unwrap();
|
||||
|
||||
let result = table.refresh_column("quotient").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
assert_eq!(read(&table, "quotient").await, vec![Some(10)]);
|
||||
}
|
||||
|
||||
/// The gate's reproducer: an already-filled row's value must not be
|
||||
/// re-evaluated either -- its input may have mutated into one the
|
||||
/// expression chokes on.
|
||||
#[tokio::test]
|
||||
async fn test_a_filled_rows_value_is_never_evaluated() {
|
||||
let table = table_with("refresh_filled_poison", vec![1, 2]).await;
|
||||
table
|
||||
.add_columns()
|
||||
.computed("quotient", "10 / x")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
table.refresh_column("quotient").await.unwrap();
|
||||
|
||||
table
|
||||
.update()
|
||||
.column("x", "0")
|
||||
.only_if("x = 1")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
append(&table, vec![5]).await;
|
||||
|
||||
let result = table.refresh_column("quotient").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
assert_eq!(
|
||||
read(&table, "quotient").await,
|
||||
vec![Some(2), Some(5), Some(10)]
|
||||
);
|
||||
}
|
||||
|
||||
/// The gate's reproducer: the old internal projection alias is an
|
||||
/// ordinary column name; a computed column may use it.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_a_column_named_like_the_old_alias() {
|
||||
let table = table_with("refresh_alias_name", vec![1, 2]).await;
|
||||
table
|
||||
.add_columns()
|
||||
.computed("__lancedb_computed", "x * 2")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let result = table.refresh_column("__lancedb_computed").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 2);
|
||||
assert_eq!(
|
||||
read(&table, "__lancedb_computed").await,
|
||||
vec![Some(2), Some(4)]
|
||||
);
|
||||
}
|
||||
|
||||
/// The gate's reproducer: a late-gain fragment (filled, then one null row
|
||||
/// compacted onto the end) fills without the old probe's buffering, which
|
||||
/// this pins behaviorally; the memory bound is structural -- the fill
|
||||
/// stream retains no batches at all.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_fills_a_late_gain_fragment() {
|
||||
let values: Vec<i32> = (0..20_000).collect();
|
||||
let table = table_with("refresh_late_gain", values).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
|
||||
append(&table, vec![2_000_000]).await;
|
||||
table
|
||||
.optimize(crate::table::OptimizeAction::Compact {
|
||||
options: crate::table::CompactionOptions::default(),
|
||||
remap_options: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
let read_back = read(&table, "doubled").await;
|
||||
assert_eq!(read_back.len(), 20_001);
|
||||
assert_eq!(read_back.last().unwrap(), &Some(4_000_000));
|
||||
}
|
||||
|
||||
/// The gate's reproducer: a nested input declares, refreshes, and guards
|
||||
/// its root against invalidating schema changes.
|
||||
#[tokio::test]
|
||||
async fn test_a_nested_input_declares_and_refreshes() {
|
||||
use arrow_array::{Int32Array, StructArray};
|
||||
use arrow_schema::{DataType, Field, Fields};
|
||||
|
||||
let conn = connect("memory://").execute().await.unwrap();
|
||||
let age = Arc::new(Int32Array::from(vec![30, 40]));
|
||||
let fields = Fields::from(vec![Field::new("age", DataType::Int32, true)]);
|
||||
let metadata = StructArray::new(fields.clone(), vec![age as _], None);
|
||||
let schema = Arc::new(arrow_schema::Schema::new(vec![Field::new(
|
||||
"metadata",
|
||||
DataType::Struct(fields),
|
||||
true,
|
||||
)]));
|
||||
let batch =
|
||||
arrow_array::RecordBatch::try_new(schema, vec![Arc::new(metadata) as _]).unwrap();
|
||||
let table = conn
|
||||
.create_table("refresh_nested", batch)
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
table
|
||||
.add_columns()
|
||||
.computed("next_age", "metadata.age + 1")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
let declaration =
|
||||
&crate::table::computed_columns(table.schema().await.unwrap().as_ref())[0];
|
||||
assert_eq!(declaration.inputs, vec!["metadata.age".to_string()]);
|
||||
|
||||
let result = table.refresh_column("next_age").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 2);
|
||||
assert_eq!(read(&table, "next_age").await, vec![Some(31), Some(41)]);
|
||||
|
||||
// The dotted input guards its root.
|
||||
let err = table.drop_columns(&["metadata"]).await.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::InvalidInput { message } if message.contains("next_age")),
|
||||
"{err:?}"
|
||||
);
|
||||
|
||||
// Masking a struct input for a deleted row goes through the same
|
||||
// nullif path as a primitive; a nested input plus deletions must not
|
||||
// be the combination that breaks it.
|
||||
table.delete("next_age = 31").await.unwrap();
|
||||
append_struct_row(&table, 50).await;
|
||||
let result = table.refresh_column("next_age").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
assert_eq!(read(&table, "next_age").await, vec![Some(41), Some(51)]);
|
||||
}
|
||||
|
||||
/// Append one `metadata: {age}` row to the nested-input table.
|
||||
async fn append_struct_row(table: &Table, age: i32) {
|
||||
use arrow_array::{Int32Array, StructArray};
|
||||
use arrow_schema::{DataType, Field, Fields};
|
||||
|
||||
let ages = Arc::new(Int32Array::from(vec![age]));
|
||||
let fields = Fields::from(vec![Field::new("age", DataType::Int32, true)]);
|
||||
let metadata = StructArray::new(fields.clone(), vec![ages as _], None);
|
||||
let schema = Arc::new(arrow_schema::Schema::new(vec![Field::new(
|
||||
"metadata",
|
||||
DataType::Struct(fields),
|
||||
true,
|
||||
)]));
|
||||
let batch =
|
||||
arrow_array::RecordBatch::try_new(schema, vec![Arc::new(metadata) as _]).unwrap();
|
||||
table.add(batch).execute().await.unwrap();
|
||||
}
|
||||
|
||||
/// Both orders of declare+spec are refused at the source (see the
|
||||
/// schema_evolution tests); refresh's own check covers a dataset another
|
||||
/// writer left in that state.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_refuses_a_foreign_lsm_state() {
|
||||
use crate::table::LsmWriteSpec;
|
||||
|
||||
let tmp_dir = tempfile::tempdir().unwrap();
|
||||
let conn = connect(tmp_dir.path().to_str().unwrap())
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
let schema = Arc::new(arrow_schema::Schema::new(vec![arrow_schema::Field::new(
|
||||
"x",
|
||||
arrow_schema::DataType::Int32,
|
||||
false,
|
||||
)]));
|
||||
let batch =
|
||||
arrow_array::RecordBatch::try_new(schema, vec![Arc::new(Int32Array::from(vec![1]))])
|
||||
.unwrap();
|
||||
let table = conn.create_table("lsm", batch).execute().await.unwrap();
|
||||
table.set_unenforced_primary_key(["x"]).await.unwrap();
|
||||
table
|
||||
.set_lsm_write_spec(LsmWriteSpec::unsharded())
|
||||
.await
|
||||
.unwrap();
|
||||
super::super::computed_columns::add_foreign_kind(&table, "doubled", "sql").await;
|
||||
|
||||
let err = table.refresh_column("doubled").await.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::NotSupported { message } if message.contains("LSM")),
|
||||
"{err:?}"
|
||||
);
|
||||
let err = table.refresh_column_async("doubled").await.unwrap_err();
|
||||
assert!(matches!(err, Error::NotSupported { .. }));
|
||||
}
|
||||
|
||||
/// After catch-up activation and unset, no spec remains but the catch-up
|
||||
/// flag still marks retained SSTable rows; refresh refuses on the flag.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_refuses_retained_catchup_state() {
|
||||
use crate::table::LsmWriteSpec;
|
||||
|
||||
let tmp_dir = tempfile::tempdir().unwrap();
|
||||
let conn = connect(tmp_dir.path().to_str().unwrap())
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
let schema = Arc::new(arrow_schema::Schema::new(vec![arrow_schema::Field::new(
|
||||
"x",
|
||||
arrow_schema::DataType::Int32,
|
||||
false,
|
||||
)]));
|
||||
let batch = arrow_array::RecordBatch::try_new(
|
||||
schema.clone(),
|
||||
vec![Arc::new(Int32Array::from(vec![1]))],
|
||||
)
|
||||
.unwrap();
|
||||
let table = conn
|
||||
.create_table("catchup", batch.clone())
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
table.set_unenforced_primary_key(["x"]).await.unwrap();
|
||||
table
|
||||
.set_lsm_write_spec(LsmWriteSpec::unsharded())
|
||||
.await
|
||||
.unwrap();
|
||||
table.require_mem_wal_index_catchup().await.unwrap();
|
||||
let mut merge = table.merge_insert(&["x"]);
|
||||
merge
|
||||
.when_matched_update_all(None)
|
||||
.when_not_matched_insert_all()
|
||||
.use_lsm(true);
|
||||
merge
|
||||
.execute(Box::new(arrow_array::RecordBatchIterator::new(
|
||||
vec![Ok(batch)],
|
||||
schema,
|
||||
)))
|
||||
.await
|
||||
.unwrap();
|
||||
table.unset_lsm_write_spec().await.unwrap();
|
||||
super::super::computed_columns::add_foreign_kind(&table, "doubled", "sql").await;
|
||||
|
||||
let err = table.refresh_column("doubled").await.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::NotSupported { message } if message.contains("LSM")),
|
||||
"{err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
/// A declaration of a kind this version cannot evaluate is refused by
|
||||
/// name, rather than mistaken for a plain column or fed to the SQL path.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_rejects_a_kind_it_cannot_evaluate() {
|
||||
let table = table_with("refresh_foreign", vec![1, 2, 3]).await;
|
||||
super::super::computed_columns::add_foreign_kind(&table, "embedding", "udf").await;
|
||||
|
||||
let err = table.refresh_column("embedding").await.unwrap_err();
|
||||
assert!(matches!(err, Error::NotSupported { message } if message.contains("udf")));
|
||||
}
|
||||
}
|
||||
@@ -8,14 +8,12 @@
|
||||
//! - [`alter_columns`](execute_alter_columns): Rename columns, change types, or modify nullability
|
||||
//! - [`drop_columns`](execute_drop_columns): Remove columns from the table
|
||||
|
||||
use arrow_schema::Schema as ArrowSchema;
|
||||
use lance::dataset::{ColumnAlteration, NewColumnTransform};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::collections::HashMap;
|
||||
|
||||
use super::computed_columns;
|
||||
use super::{BaseTable, NativeTable};
|
||||
use crate::{Error, Result};
|
||||
use super::NativeTable;
|
||||
use crate::Result;
|
||||
|
||||
/// The result of an add columns operation.
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
|
||||
@@ -100,48 +98,6 @@ pub(crate) async fn execute_add_columns(
|
||||
table: &NativeTable,
|
||||
transforms: NewColumnTransform,
|
||||
read_columns: Option<Vec<String>>,
|
||||
) -> Result<AddColumnsResult> {
|
||||
// Declarations are admitted only through [`execute_declare`].
|
||||
match &transforms {
|
||||
NewColumnTransform::AllNulls(schema) => {
|
||||
computed_columns::ensure_no_foreign_declarations(schema.fields())?
|
||||
}
|
||||
NewColumnTransform::BatchUDF(udf) => {
|
||||
computed_columns::ensure_no_foreign_declarations(udf.output_schema.fields())?
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
commit_add_columns(table, transforms, read_columns).await
|
||||
}
|
||||
|
||||
/// Declare validated computed columns. The only admission path for
|
||||
/// declaration metadata.
|
||||
pub(crate) async fn execute_declare(
|
||||
table: &NativeTable,
|
||||
columns: &[(String, String)],
|
||||
) -> Result<AddColumnsResult> {
|
||||
// An LSM write spec keeps visible rows in tiers refresh cannot reach;
|
||||
// checked against latest committed state, not this handle's snapshot.
|
||||
// The catch-up flag outlives unset and marks retained SSTable rows.
|
||||
table.checkout_latest().await?;
|
||||
let catchup = table.dataset.get().await?.manifest().reader_feature_flags
|
||||
& lance_table::feature_flags::FLAG_MEM_WAL_INDEX_CATCHUP
|
||||
!= 0;
|
||||
if catchup || table.get_lsm_write_spec().await?.is_some() {
|
||||
return Err(Error::NotSupported {
|
||||
message: "computed columns are not supported on a table with an LSM write \
|
||||
spec: rows in un-compacted tiers are invisible to refresh"
|
||||
.into(),
|
||||
});
|
||||
}
|
||||
let transform = computed_columns::declare(table.schema().await?, columns)?;
|
||||
commit_add_columns(table, transform, None).await
|
||||
}
|
||||
|
||||
pub(crate) async fn commit_add_columns(
|
||||
table: &NativeTable,
|
||||
transforms: NewColumnTransform,
|
||||
read_columns: Option<Vec<String>>,
|
||||
) -> Result<AddColumnsResult> {
|
||||
table.dataset.ensure_mutable()?;
|
||||
let mut dataset = (*table.dataset.get().await?).clone();
|
||||
@@ -160,21 +116,6 @@ pub(crate) async fn execute_alter_columns(
|
||||
) -> Result<AlterColumnsResult> {
|
||||
table.dataset.ensure_mutable()?;
|
||||
let mut dataset = (*table.dataset.get().await?).clone();
|
||||
// Nullability is not part of what an expression resolves against, so only
|
||||
// a rename or a retype can invalidate a binding.
|
||||
let schema = std::sync::Arc::new(ArrowSchema::from(dataset.schema()));
|
||||
let rebinding = alterations
|
||||
.iter()
|
||||
.filter(|alteration| alteration.rename.is_some() || alteration.data_type.is_some())
|
||||
.map(|alteration| alteration.path.as_str())
|
||||
.collect::<Vec<_>>();
|
||||
computed_columns::ensure_not_an_input(&schema, &rebinding)?;
|
||||
let retyped = alterations
|
||||
.iter()
|
||||
.filter(|alteration| alteration.data_type.is_some())
|
||||
.map(|alteration| alteration.path.as_str())
|
||||
.collect::<Vec<_>>();
|
||||
computed_columns::ensure_not_retyped(schema.as_ref(), &retyped)?;
|
||||
dataset.alter_columns(alterations).await?;
|
||||
let version = dataset.version().version;
|
||||
table.dataset.update(dataset);
|
||||
@@ -190,10 +131,6 @@ pub(crate) async fn execute_drop_columns(
|
||||
) -> Result<DropColumnsResult> {
|
||||
table.dataset.ensure_mutable()?;
|
||||
let mut dataset = (*table.dataset.get().await?).clone();
|
||||
computed_columns::ensure_not_an_input(
|
||||
&std::sync::Arc::new(ArrowSchema::from(dataset.schema())),
|
||||
columns,
|
||||
)?;
|
||||
dataset.drop_columns(columns).await?;
|
||||
let version = dataset.version().version;
|
||||
table.dataset.update(dataset);
|
||||
@@ -210,44 +147,6 @@ pub(crate) async fn execute_update_field_metadata(
|
||||
table.dataset.ensure_mutable()?;
|
||||
let mut dataset = (*table.dataset.get().await?).clone();
|
||||
|
||||
// A declaration is validated as a whole at declare time; editing its keys
|
||||
// here would bypass that, fabricate one on a plain column, or move a
|
||||
// binding out from under a refresh. A replace on a declared column would
|
||||
// silently erase it.
|
||||
let schema = ArrowSchema::from(dataset.schema());
|
||||
let declared: Vec<String> = computed_columns::computed_columns(&schema)
|
||||
.into_iter()
|
||||
.map(|declaration| declaration.name)
|
||||
.collect();
|
||||
for update in updates {
|
||||
if update
|
||||
.metadata
|
||||
.keys()
|
||||
.any(|key| computed_columns::is_declaration_key(key))
|
||||
{
|
||||
return Err(Error::InvalidInput {
|
||||
message: format!(
|
||||
"metadata keys of a computed-column declaration cannot be edited \
|
||||
(path '{}'); drop the column and declare it again",
|
||||
update.path
|
||||
),
|
||||
});
|
||||
}
|
||||
if update.replace
|
||||
&& declared
|
||||
.iter()
|
||||
.any(|name| name == computed_columns::root(&update.path))
|
||||
{
|
||||
return Err(Error::InvalidInput {
|
||||
message: format!(
|
||||
"replacing all metadata of computed column '{}' would erase its \
|
||||
declaration; drop the column and declare it again",
|
||||
update.path
|
||||
),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
let mut builder = dataset.update_field_metadata();
|
||||
for update in updates {
|
||||
let entries = update.metadata.iter().map(|(k, v)| (k.clone(), v.clone()));
|
||||
|
||||
@@ -82,10 +82,6 @@ pub(crate) async fn execute_update(
|
||||
|
||||
// 1. Snapshot the current dataset
|
||||
let dataset = table.dataset.get().await?;
|
||||
super::computed_columns::ensure_not_written(
|
||||
&arrow_schema::Schema::from(dataset.schema()),
|
||||
update.columns.iter().map(|(name, _)| name.as_str()),
|
||||
)?;
|
||||
|
||||
// 2. Initialize the Lance Core builder
|
||||
let mut builder = LanceUpdateBuilder::new(dataset);
|
||||
|
||||
@@ -1,147 +0,0 @@
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
||||
|
||||
use std::sync::Arc;
|
||||
|
||||
use arrow_array::{Int64Array, RecordBatch};
|
||||
use arrow_schema::{DataType, Field, Schema};
|
||||
use lance::dataset::{WriteMode, WriteParams};
|
||||
use lancedb::{Result, TableBase, connect, connect_namespace, table::WriteOptions};
|
||||
use tempfile::tempdir;
|
||||
use url::Url;
|
||||
|
||||
fn empty_schema() -> Arc<Schema> {
|
||||
Arc::new(Schema::new(vec![Field::new("id", DataType::Int64, false)]))
|
||||
}
|
||||
|
||||
fn file_uri(path: &std::path::Path) -> String {
|
||||
Url::from_file_path(path)
|
||||
.unwrap_or_else(|_| panic!("not an absolute path: {}", path.display()))
|
||||
.to_string()
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_add_bases_accepts_named_and_dataset_root_entries() -> Result<()> {
|
||||
let tmp = tempdir().unwrap();
|
||||
let db = connect(tmp.path().join("db").to_str().unwrap())
|
||||
.execute()
|
||||
.await?;
|
||||
let table = db.create_empty_table("t", empty_schema()).execute().await?;
|
||||
let media = tmp.path().join("media");
|
||||
let parent = tmp.path().join("parent");
|
||||
std::fs::create_dir_all(&media).unwrap();
|
||||
std::fs::create_dir_all(&parent).unwrap();
|
||||
|
||||
table
|
||||
.add_bases([
|
||||
TableBase {
|
||||
path: file_uri(&media),
|
||||
name: Some("media".into()),
|
||||
is_dataset_root: false,
|
||||
},
|
||||
TableBase {
|
||||
path: file_uri(&parent),
|
||||
name: Some("parent".into()),
|
||||
is_dataset_root: true,
|
||||
},
|
||||
])
|
||||
.await
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_add_bases_accepts_two_unnamed_paths() -> Result<()> {
|
||||
let tmp = tempdir().unwrap();
|
||||
let db = connect(tmp.path().join("db").to_str().unwrap())
|
||||
.execute()
|
||||
.await?;
|
||||
let table = db.create_empty_table("t", empty_schema()).execute().await?;
|
||||
let media = tmp.path().join("media");
|
||||
let other = tmp.path().join("other");
|
||||
std::fs::create_dir_all(&media).unwrap();
|
||||
std::fs::create_dir_all(&other).unwrap();
|
||||
table
|
||||
.add_bases([&file_uri(&media), &file_uri(&other)])
|
||||
.await
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_add_bases_write_and_read_through_registered_base() -> Result<()> {
|
||||
let tmp = tempdir().unwrap();
|
||||
let db = connect(tmp.path().join("db").to_str().unwrap())
|
||||
.execute()
|
||||
.await?;
|
||||
let table = db.create_empty_table("t", empty_schema()).execute().await?;
|
||||
let media = tmp.path().join("media");
|
||||
std::fs::create_dir_all(&media).unwrap();
|
||||
let media_uri = file_uri(&media);
|
||||
table.add_bases([&media_uri]).await?;
|
||||
|
||||
let batch = RecordBatch::try_new(
|
||||
empty_schema(),
|
||||
vec![Arc::new(Int64Array::from(vec![1, 2, 3]))],
|
||||
)
|
||||
.unwrap();
|
||||
table
|
||||
.add(batch)
|
||||
.write_options(WriteOptions {
|
||||
lance_write_params: Some(WriteParams {
|
||||
mode: WriteMode::Append,
|
||||
target_base_names_or_paths: Some(vec![media_uri.clone()]),
|
||||
..Default::default()
|
||||
}),
|
||||
})
|
||||
.execute()
|
||||
.await?;
|
||||
|
||||
assert_eq!(table.count_rows(None).await?, 3);
|
||||
|
||||
let dataset = table.dataset().unwrap().get().await?;
|
||||
let registered = dataset
|
||||
.manifest()
|
||||
.base_paths
|
||||
.values()
|
||||
.find(|base| base.path == media_uri)
|
||||
.expect("registered base");
|
||||
assert_ne!(registered.id, 0);
|
||||
assert!(registered.name.is_none());
|
||||
assert!(
|
||||
dataset.get_fragments().iter().any(|fragment| {
|
||||
fragment
|
||||
.metadata()
|
||||
.files
|
||||
.iter()
|
||||
.any(|file| file.base_id == Some(registered.id))
|
||||
}),
|
||||
"written fragment should reference the registered base"
|
||||
);
|
||||
assert!(
|
||||
std::fs::read_dir(&media)
|
||||
.unwrap()
|
||||
.filter_map(|entry| entry.ok())
|
||||
.any(|entry| entry.path().extension().is_some_and(|ext| ext == "lance")),
|
||||
"data file should land under the registered base"
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_memory_add_bases_accepts_a_file_uri() -> Result<()> {
|
||||
let tmp = tempdir().unwrap();
|
||||
let db = connect("memory://").execute().await?;
|
||||
let table = db.create_empty_table("t", empty_schema()).execute().await?;
|
||||
let media = tmp.path().join("media");
|
||||
std::fs::create_dir_all(&media).unwrap();
|
||||
table.add_bases([file_uri(&media)]).await
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_namespace_add_bases_accepts_a_file_uri() -> Result<()> {
|
||||
let tmp = tempdir().unwrap();
|
||||
let mut properties = std::collections::HashMap::new();
|
||||
properties.insert("root".to_string(), tmp.path().to_str().unwrap().to_string());
|
||||
let db = connect_namespace("dir", properties).execute().await?;
|
||||
let table = db.create_empty_table("t", empty_schema()).execute().await?;
|
||||
let media = tmp.path().join("media");
|
||||
std::fs::create_dir_all(&media).unwrap();
|
||||
table.add_bases([file_uri(&media)]).await
|
||||
}
|
||||
Reference in New Issue
Block a user