mirror of
https://github.com/lancedb/lancedb.git
synced 2026-08-25 15:38:38 +00:00
Merge remote-tracking branch 'origin/main' into gatekeeper/fix-2325-1
This commit is contained in:
+1
-1
@@ -1,5 +1,5 @@
|
||||
[tool.bumpversion]
|
||||
current_version = "0.38.0-beta.0"
|
||||
current_version = "0.38.0-beta.2"
|
||||
parse = """(?x)
|
||||
(?P<major>0|[1-9]\\d*)\\.
|
||||
(?P<minor>0|[1-9]\\d*)\\.
|
||||
|
||||
Generated
+53
-53
@@ -959,7 +959,7 @@ dependencies = [
|
||||
"aws-smithy-runtime-api",
|
||||
"aws-smithy-types",
|
||||
"h2 0.3.27",
|
||||
"h2 0.4.14",
|
||||
"h2 0.4.16",
|
||||
"http 0.2.12",
|
||||
"http 1.5.0",
|
||||
"http-body 0.4.6",
|
||||
@@ -3455,8 +3455,8 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c"
|
||||
|
||||
[[package]]
|
||||
name = "fsst"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"rand 0.9.5",
|
||||
@@ -3877,9 +3877,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "h2"
|
||||
version = "0.4.14"
|
||||
version = "0.4.16"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "171fefbc92fe4a4de27e0698d6a5b392d6a0e333506bc49133760b3bcf948733"
|
||||
checksum = "a9f37a958b41b3b19ee2707c06439c0e9e547e847223eb791ecb0cb821c65e27"
|
||||
dependencies = [
|
||||
"atomic-waker",
|
||||
"bytes",
|
||||
@@ -4188,7 +4188,7 @@ dependencies = [
|
||||
"bytes",
|
||||
"futures-channel",
|
||||
"futures-core",
|
||||
"h2 0.4.14",
|
||||
"h2 0.4.16",
|
||||
"http 1.5.0",
|
||||
"http-body 1.1.0",
|
||||
"httparse",
|
||||
@@ -4815,8 +4815,8 @@ checksum = "e037a2e1d8d5fdbd49b16a4ea09d5d6401c1f29eca5ff29d03d3824dba16256a"
|
||||
|
||||
[[package]]
|
||||
name = "lance"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"arrow",
|
||||
@@ -4888,8 +4888,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-arrow"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -4911,7 +4911,7 @@ dependencies = [
|
||||
[[package]]
|
||||
name = "lance-arrow-scalar"
|
||||
version = "58.0.0"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -4925,7 +4925,7 @@ dependencies = [
|
||||
[[package]]
|
||||
name = "lance-arrow-stats"
|
||||
version = "58.0.0"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -4934,8 +4934,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-bitpacking"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrayref",
|
||||
"crunchy",
|
||||
@@ -4945,8 +4945,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-core"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -4983,8 +4983,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-datafusion"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5013,8 +5013,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-datagen"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5031,8 +5031,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-derive"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -5041,8 +5041,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-encoding"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow-arith",
|
||||
"arrow-array",
|
||||
@@ -5075,8 +5075,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-file"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow-arith",
|
||||
"arrow-array",
|
||||
@@ -5107,8 +5107,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-index"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"arrow",
|
||||
@@ -5172,8 +5172,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-index-core"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -5195,8 +5195,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-io"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5232,8 +5232,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-linalg"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -5247,8 +5247,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-namespace"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"async-trait",
|
||||
@@ -5260,8 +5260,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-namespace-impls"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-ipc",
|
||||
@@ -5300,9 +5300,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-namespace-reqwest-client"
|
||||
version = "0.8.6"
|
||||
version = "0.11.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ba3f0a235e3ed5f8805205649ccc7d7d0f3df23ce1294242c9265ad488d7f19d"
|
||||
checksum = "0a030196da1c994b63a96a4f0bf5b0cfa459fe6dadc9e962320246ca328da22a"
|
||||
dependencies = [
|
||||
"reqwest 0.12.28",
|
||||
"serde",
|
||||
@@ -5314,8 +5314,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-select"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -5329,8 +5329,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-table"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"arrow-array",
|
||||
@@ -5370,8 +5370,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-testing"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-schema",
|
||||
@@ -5384,8 +5384,8 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lance-tokenizer"
|
||||
version = "11.0.0-beta.11"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3"
|
||||
version = "11.0.0-beta.15"
|
||||
source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb"
|
||||
dependencies = [
|
||||
"frostem",
|
||||
"icu_segmenter",
|
||||
@@ -5398,7 +5398,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lancedb"
|
||||
version = "0.38.0-beta.0"
|
||||
version = "0.38.0-beta.2"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"anyhow",
|
||||
@@ -5486,7 +5486,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lancedb-nodejs"
|
||||
version = "0.38.0-beta.0"
|
||||
version = "0.38.0-beta.2"
|
||||
dependencies = [
|
||||
"arrow-array",
|
||||
"arrow-buffer",
|
||||
@@ -5511,7 +5511,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "lancedb-python"
|
||||
version = "0.38.0-beta.0"
|
||||
version = "0.38.0-beta.2"
|
||||
dependencies = [
|
||||
"arrow",
|
||||
"async-trait",
|
||||
@@ -8426,7 +8426,7 @@ dependencies = [
|
||||
"encoding_rs",
|
||||
"futures-core",
|
||||
"futures-util",
|
||||
"h2 0.4.14",
|
||||
"h2 0.4.16",
|
||||
"http 1.5.0",
|
||||
"http-body 1.1.0",
|
||||
"http-body-util",
|
||||
@@ -10082,7 +10082,7 @@ dependencies = [
|
||||
"async-trait",
|
||||
"base64 0.22.1",
|
||||
"bytes",
|
||||
"h2 0.4.14",
|
||||
"h2 0.4.16",
|
||||
"http 1.5.0",
|
||||
"http-body 1.1.0",
|
||||
"http-body-util",
|
||||
|
||||
+14
-14
@@ -13,20 +13,20 @@ categories = ["database-implementations"]
|
||||
rust-version = "1.91.0"
|
||||
|
||||
[workspace.dependencies]
|
||||
lance = { "version" = "=11.0.0-beta.11", default-features = false, "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-core = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datagen = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-file = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-io = { "version" = "=11.0.0-beta.11", default-features = false, "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-index = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-linalg = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace-impls = { "version" = "=11.0.0-beta.11", default-features = false, "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-table = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-testing = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datafusion = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-encoding = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-arrow = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance = { "version" = "=11.0.0-beta.15", default-features = false, "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-core = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datagen = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-file = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-io = { "version" = "=11.0.0-beta.15", default-features = false, "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-index = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-linalg = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-namespace-impls = { "version" = "=11.0.0-beta.15", default-features = false, "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-table = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-testing = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-datafusion = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-encoding = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" }
|
||||
lance-arrow = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" }
|
||||
ahash = "0.8"
|
||||
# Note that this one does not include pyarrow
|
||||
arrow = { version = "58.0.0", optional = false }
|
||||
|
||||
@@ -108,6 +108,12 @@ ignore = [
|
||||
# compact_str/smol_str, so clearing this requires polars to migrate.
|
||||
# https://rustsec.org/advisories/RUSTSEC-2026-0249
|
||||
{ id = "RUSTSEC-2026-0249", reason = "smartstring unmaintained via polars; no fixed upstream release" },
|
||||
|
||||
# h2 0.3: empty DATA frames can be queued without limit. The patched
|
||||
# h2 0.4 line is locked to 0.4.16, but no patched 0.3 release exists.
|
||||
# The old copy is pulled in by aws-smithy's legacy hyper 0.14 client.
|
||||
# https://rustsec.org/advisories/RUSTSEC-2026-0258
|
||||
{ id = "RUSTSEC-2026-0258", reason = "h2 0.3 via legacy aws-smithy/hyper 0.14; no patched 0.3 release" },
|
||||
]
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
+33
-1
@@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`:
|
||||
<dependency>
|
||||
<groupId>com.lancedb</groupId>
|
||||
<artifactId>lancedb-core</artifactId>
|
||||
<version>0.38.0-beta.0</version>
|
||||
<version>0.38.0-beta.2</version>
|
||||
</dependency>
|
||||
```
|
||||
|
||||
@@ -55,6 +55,38 @@ LanceNamespace namespaceClient = LanceDbNamespaceClientBuilder.newBuilder()
|
||||
| `region(String)` | AWS region (default: "us-east-1") | No |
|
||||
| `config(String, String)` | Additional configuration parameters | No |
|
||||
|
||||
### Opening a Table with Vended Credentials
|
||||
|
||||
When the catalog vends temporary object store credentials, open the table through the
|
||||
namespace client. The Lance dataset builder fetches the table location and storage options
|
||||
from the catalog and refreshes the credentials when they expire.
|
||||
|
||||
```java
|
||||
import com.lancedb.LanceDbNamespaceClientBuilder;
|
||||
import org.lance.Dataset;
|
||||
import org.lance.namespace.LanceNamespace;
|
||||
|
||||
import java.util.Arrays;
|
||||
|
||||
LanceNamespace namespaceClient = LanceDbNamespaceClientBuilder.newBuilder()
|
||||
.apiKey(System.getenv("LANCEDB_API_KEY"))
|
||||
.database(System.getenv("LANCEDB_DATABASE"))
|
||||
// Set the endpoint for a LanceDB Enterprise deployment.
|
||||
// .endpoint("https://your-enterprise-endpoint")
|
||||
.build();
|
||||
|
||||
try (Dataset dataset = Dataset.open()
|
||||
.namespaceClient(namespaceClient)
|
||||
.tableId(Arrays.asList("my_namespace", "my_table"))
|
||||
.build()) {
|
||||
System.out.println("Rows: " + dataset.countRows());
|
||||
}
|
||||
```
|
||||
|
||||
Do not call `describeTable()` and then open the returned location with `Dataset.open(uri)`.
|
||||
Opening through `namespaceClient()` is what applies the vended storage options and enables
|
||||
automatic credential refresh. No object store credentials need to be passed by the application.
|
||||
|
||||
## Metadata Operations
|
||||
|
||||
### Creating a Namespace Path
|
||||
|
||||
@@ -79,8 +79,9 @@ input leaves the value computed at fill time; recomputing means dropping
|
||||
the column and declaring it again. While a declaration reads a column,
|
||||
that column cannot be renamed, retyped or dropped.
|
||||
|
||||
Computed columns are local-only: LanceDB Cloud and Enterprise reject a
|
||||
declaration.
|
||||
On LanceDB Cloud and Enterprise the expression is planned by the
|
||||
server, and the refresh runs as a server job -- see
|
||||
[Table#refreshColumnAsync](Table.md#refreshcolumnasync).
|
||||
|
||||
#### Parameters
|
||||
|
||||
@@ -212,6 +213,39 @@ version of the table.
|
||||
|
||||
***
|
||||
|
||||
### checkpointLsm()
|
||||
|
||||
```ts
|
||||
abstract checkpointLsm(): Promise<void>
|
||||
```
|
||||
|
||||
Converge this table's LSM write path into its base table.
|
||||
|
||||
Seals once, then triggers compaction and polls until the L0 that existed
|
||||
at the start is gone. The target set is fixed at the start, so
|
||||
generations created *during* the checkpoint are ignored — that is what
|
||||
lets it terminate under write load, and what makes it best-effort: it
|
||||
converges the fresh tier as of some instant. Idempotent, abandonable at
|
||||
any point, and safe to run on a cadence.
|
||||
|
||||
There is no liveness bound — the compactor pool is shared across tables,
|
||||
so a checkpoint queued behind unrelated work looks exactly like one that
|
||||
is merging. The caller owns the deadline.
|
||||
|
||||
#### Returns
|
||||
|
||||
`Promise`<`void`>
|
||||
|
||||
#### Example
|
||||
|
||||
```ts
|
||||
const before = await table.getLsmStats();
|
||||
await table.checkpointLsm();
|
||||
const after = await table.getLsmStats();
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### close()
|
||||
|
||||
```ts
|
||||
@@ -249,6 +283,24 @@ It is a no-op when no writers are cached.
|
||||
|
||||
***
|
||||
|
||||
### compactLsm()
|
||||
|
||||
```ts
|
||||
abstract compactLsm(): Promise<void>
|
||||
```
|
||||
|
||||
Trigger a background L0 → base compaction pass per bucket.
|
||||
|
||||
Returns once the passes are *dispatched*, not once they finish — watch
|
||||
[Table#getLsmStats](Table.md#getlsmstats) for progress, or use
|
||||
[Table#checkpointLsm](Table.md#checkpointlsm) to wait for convergence.
|
||||
|
||||
#### Returns
|
||||
|
||||
`Promise`<`void`>
|
||||
|
||||
***
|
||||
|
||||
### countRows()
|
||||
|
||||
```ts
|
||||
@@ -447,6 +499,48 @@ Drop an index from the table.
|
||||
|
||||
***
|
||||
|
||||
### flushLsm()
|
||||
|
||||
```ts
|
||||
abstract flushLsm(): Promise<void>
|
||||
```
|
||||
|
||||
Seal every bucket's active memtable into a new L0 generation.
|
||||
|
||||
Returns once the seal is committed. Sealing an empty memtable is a no-op,
|
||||
so this is safe to call repeatedly.
|
||||
|
||||
#### Returns
|
||||
|
||||
`Promise`<`void`>
|
||||
|
||||
***
|
||||
|
||||
### getLsmStats()
|
||||
|
||||
```ts
|
||||
abstract getLsmStats(includeGenerationRows?): Promise<undefined | LsmStats>
|
||||
```
|
||||
|
||||
Read live per-bucket LSM state.
|
||||
|
||||
Answers "how far behind is my fresh tier", "which bucket is hot", and
|
||||
"why is my fresh-tier vector search brute-force". Mutates no table state.
|
||||
|
||||
Resolves to `undefined` only when the LSM write path is not enabled.
|
||||
|
||||
#### Parameters
|
||||
|
||||
* **includeGenerationRows?**: `boolean`
|
||||
Also count rows per L0 generation.
|
||||
Off by default because each count opens an uncached Lance dataset.
|
||||
|
||||
#### Returns
|
||||
|
||||
`Promise`<`undefined` \| [`LsmStats`](../interfaces/LsmStats.md)>
|
||||
|
||||
***
|
||||
|
||||
### getLsmWriteSpec()
|
||||
|
||||
```ts
|
||||
@@ -754,7 +848,8 @@ Fill the rows of a computed column that hold no value yet.
|
||||
|
||||
Rows appended since the last refresh are filled by the next one; rows
|
||||
already filled are left as they are, so the call is idempotent and does
|
||||
not observe a mutated input. Local tables only.
|
||||
not observe a mutated input. Local tables only: a remote refresh runs
|
||||
as a server job, through [Table#refreshColumnAsync](Table.md#refreshcolumnasync).
|
||||
|
||||
#### Parameters
|
||||
|
||||
@@ -770,6 +865,40 @@ number of rows filled and the new version number of the table.
|
||||
|
||||
***
|
||||
|
||||
### refreshColumnAsync()
|
||||
|
||||
```ts
|
||||
abstract refreshColumnAsync(column): Promise<Job>
|
||||
```
|
||||
|
||||
Like [Table#refreshColumn](Table.md#refreshcolumn), but returns a handle to the refresh
|
||||
job instead of blocking until it completes.
|
||||
|
||||
The job may already be complete when returned; callers must not assume
|
||||
the column is filled until [Job.wait](Job.md#wait) resolves. Invalid input --
|
||||
an unknown column, or one that is not computed -- rejects here rather
|
||||
than failing the job. On local tables the job runs in-process; on
|
||||
LanceDB Cloud and Enterprise it is the server's backfill job.
|
||||
|
||||
#### Parameters
|
||||
|
||||
* **column**: `string`
|
||||
The name of the computed column to fill.
|
||||
|
||||
#### Returns
|
||||
|
||||
`Promise`<[`Job`](Job.md)>
|
||||
|
||||
#### Example
|
||||
|
||||
```ts
|
||||
const job = await table.refreshColumnAsync("doubled");
|
||||
await job.wait();
|
||||
console.log(await job.status()); // "finished"
|
||||
```
|
||||
|
||||
***
|
||||
|
||||
### restore()
|
||||
|
||||
```ts
|
||||
|
||||
@@ -58,6 +58,7 @@
|
||||
- [BranchDiff](interfaces/BranchDiff.md)
|
||||
- [BranchIndexSummary](interfaces/BranchIndexSummary.md)
|
||||
- [BranchRowCountSummary](interfaces/BranchRowCountSummary.md)
|
||||
- [BucketStats](interfaces/BucketStats.md)
|
||||
- [ClientConfig](interfaces/ClientConfig.md)
|
||||
- [ColumnAlteration](interfaces/ColumnAlteration.md)
|
||||
- [ColumnOrdering](interfaces/ColumnOrdering.md)
|
||||
@@ -81,6 +82,7 @@
|
||||
- [FtsToken](interfaces/FtsToken.md)
|
||||
- [FullTextQuery](interfaces/FullTextQuery.md)
|
||||
- [FullTextSearchOptions](interfaces/FullTextSearchOptions.md)
|
||||
- [GenerationStats](interfaces/GenerationStats.md)
|
||||
- [HnswPqOptions](interfaces/HnswPqOptions.md)
|
||||
- [HnswSqOptions](interfaces/HnswSqOptions.md)
|
||||
- [IndexConfig](interfaces/IndexConfig.md)
|
||||
@@ -94,7 +96,9 @@
|
||||
- [JobInfo](interfaces/JobInfo.md)
|
||||
- [ListNamespacesOptions](interfaces/ListNamespacesOptions.md)
|
||||
- [ListNamespacesResponse](interfaces/ListNamespacesResponse.md)
|
||||
- [LsmStats](interfaces/LsmStats.md)
|
||||
- [LsmWriteSpec](interfaces/LsmWriteSpec.md)
|
||||
- [MemtableStats](interfaces/MemtableStats.md)
|
||||
- [MergeBlocker](interfaces/MergeBlocker.md)
|
||||
- [MergeBranchResult](interfaces/MergeBranchResult.md)
|
||||
- [MergePreview](interfaces/MergePreview.md)
|
||||
|
||||
@@ -0,0 +1,116 @@
|
||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||
|
||||
***
|
||||
|
||||
[@lancedb/lancedb](../globals.md) / BucketStats
|
||||
|
||||
# Interface: BucketStats
|
||||
|
||||
Live state of one bucket. A table is N buckets on one node; flattening to a
|
||||
single number hides the one hot bucket that is usually why someone opened
|
||||
this endpoint.
|
||||
|
||||
## Properties
|
||||
|
||||
### compacting
|
||||
|
||||
```ts
|
||||
compacting: boolean;
|
||||
```
|
||||
|
||||
Whether a pass owns this bucket's compaction latch right now. Says *a*
|
||||
driver is running, not *whose*, and the latch is held from dispatch —
|
||||
including while the pass queues for a pod-wide compactor permit. Read it
|
||||
as "do not pile on", never as "mine is progressing".
|
||||
|
||||
***
|
||||
|
||||
### currentGeneration
|
||||
|
||||
```ts
|
||||
currentGeneration: number;
|
||||
```
|
||||
|
||||
The generation the active memtable will become.
|
||||
|
||||
***
|
||||
|
||||
### generations
|
||||
|
||||
```ts
|
||||
generations: GenerationStats[];
|
||||
```
|
||||
|
||||
Flushed L0 generations not yet merged into the base table.
|
||||
|
||||
***
|
||||
|
||||
### manifestVersion
|
||||
|
||||
```ts
|
||||
manifestVersion: number;
|
||||
```
|
||||
|
||||
Version of the shard manifest these numbers were read from.
|
||||
|
||||
***
|
||||
|
||||
### memtables?
|
||||
|
||||
```ts
|
||||
optional memtables: MemtableStats[];
|
||||
```
|
||||
|
||||
Oldest first, active last. Absent for a `"Sealed"` bucket, whose
|
||||
in-memory state is torn down.
|
||||
|
||||
***
|
||||
|
||||
### replayAfterWalEntryPosition
|
||||
|
||||
```ts
|
||||
replayAfterWalEntryPosition: number;
|
||||
```
|
||||
|
||||
WAL position replay resumes from.
|
||||
|
||||
***
|
||||
|
||||
### shardId
|
||||
|
||||
```ts
|
||||
shardId: string;
|
||||
```
|
||||
|
||||
The shard this bucket writes.
|
||||
|
||||
***
|
||||
|
||||
### status
|
||||
|
||||
```ts
|
||||
status: string;
|
||||
```
|
||||
|
||||
`"Active"` or `"Sealed"` (drop-table 2PC in flight).
|
||||
|
||||
***
|
||||
|
||||
### walEntryPositionLastSeen
|
||||
|
||||
```ts
|
||||
walEntryPositionLastSeen: number;
|
||||
```
|
||||
|
||||
Highest WAL position the writer has seen. The difference against
|
||||
`replayAfterWalEntryPosition` is the WAL lag.
|
||||
|
||||
***
|
||||
|
||||
### writerEpoch
|
||||
|
||||
```ts
|
||||
writerEpoch: number;
|
||||
```
|
||||
|
||||
Epoch of the writer that currently owns the shard.
|
||||
@@ -0,0 +1,40 @@
|
||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||
|
||||
***
|
||||
|
||||
[@lancedb/lancedb](../globals.md) / GenerationStats
|
||||
|
||||
# Interface: GenerationStats
|
||||
|
||||
One flushed L0 generation.
|
||||
|
||||
## Properties
|
||||
|
||||
### bytes
|
||||
|
||||
```ts
|
||||
bytes: number;
|
||||
```
|
||||
|
||||
On-disk size of the generation.
|
||||
|
||||
***
|
||||
|
||||
### generation
|
||||
|
||||
```ts
|
||||
generation: number;
|
||||
```
|
||||
|
||||
The generation number. Increases as memtables are sealed into L0.
|
||||
|
||||
***
|
||||
|
||||
### rows?
|
||||
|
||||
```ts
|
||||
optional rows: number;
|
||||
```
|
||||
|
||||
Present only when `includeGenerationRows` was requested. Off by default
|
||||
because each count opens an uncached Lance dataset.
|
||||
@@ -0,0 +1,22 @@
|
||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||
|
||||
***
|
||||
|
||||
[@lancedb/lancedb](../globals.md) / LsmStats
|
||||
|
||||
# Interface: LsmStats
|
||||
|
||||
Live per-bucket LSM state, as returned by `Table#getLsmStats`.
|
||||
|
||||
Nothing here is derived: sums and differences (total L0 bytes, WAL lag) are
|
||||
the caller's to compute.
|
||||
|
||||
## Properties
|
||||
|
||||
### buckets
|
||||
|
||||
```ts
|
||||
buckets: BucketStats[];
|
||||
```
|
||||
|
||||
One entry per bucket backing this table.
|
||||
@@ -0,0 +1,60 @@
|
||||
[**@lancedb/lancedb**](../README.md) • **Docs**
|
||||
|
||||
***
|
||||
|
||||
[@lancedb/lancedb](../globals.md) / MemtableStats
|
||||
|
||||
# Interface: MemtableStats
|
||||
|
||||
One in-memory memtable.
|
||||
|
||||
## Properties
|
||||
|
||||
### batches
|
||||
|
||||
```ts
|
||||
batches: number;
|
||||
```
|
||||
|
||||
Record batches currently buffered.
|
||||
|
||||
***
|
||||
|
||||
### bytes
|
||||
|
||||
```ts
|
||||
bytes: number;
|
||||
```
|
||||
|
||||
Estimated in-memory size.
|
||||
|
||||
***
|
||||
|
||||
### generation
|
||||
|
||||
```ts
|
||||
generation: number;
|
||||
```
|
||||
|
||||
The generation this memtable will become once sealed.
|
||||
|
||||
***
|
||||
|
||||
### indexes
|
||||
|
||||
```ts
|
||||
indexes: string[];
|
||||
```
|
||||
|
||||
Names of the indexes this memtable carries. An absent name is the whole
|
||||
answer to "why is my fresh-tier search on that column brute-force".
|
||||
|
||||
***
|
||||
|
||||
### rows
|
||||
|
||||
```ts
|
||||
rows: number;
|
||||
```
|
||||
|
||||
Rows currently buffered.
|
||||
@@ -54,6 +54,8 @@ listing a storage directory.
|
||||
|
||||
::: lancedb.table.Branches
|
||||
|
||||
::: lancedb.LsmWriteSpec
|
||||
|
||||
## Expressions
|
||||
|
||||
Type-safe expression builder for filters and projections. Use these instead
|
||||
|
||||
@@ -29,6 +29,48 @@ LanceNamespace namespaceClient = LanceDbNamespaceClientBuilder.newBuilder()
|
||||
.build();
|
||||
```
|
||||
|
||||
## MemWAL LSM write path
|
||||
|
||||
Most table operations reach LanceDB through the `LanceNamespace` above, which is
|
||||
generated from the Lance Namespace specification. The MemWAL LSM routes are not part
|
||||
of that specification, so they are issued through a separate client:
|
||||
|
||||
```java
|
||||
import com.lancedb.LanceDbRestClient;
|
||||
import com.lancedb.LanceDbTableLsm;
|
||||
import com.lancedb.LsmWriteSpec;
|
||||
|
||||
LanceDbRestClient client = LanceDbNamespaceClientBuilder.newBuilder()
|
||||
.apiKey("your_lancedb_cloud_api_key")
|
||||
.database("your_database_name")
|
||||
.buildRestClient();
|
||||
|
||||
LanceDbTableLsm lsm = new LanceDbTableLsm(client, "my_table");
|
||||
|
||||
// Route future merge_insert upserts through the MemWAL, hash-bucketed by `id`.
|
||||
lsm.setLsmWriteSpec(LsmWriteSpec.bucket("id", 16));
|
||||
|
||||
// ... merge_insert traffic ...
|
||||
|
||||
// Converge the fresh tier into the base table.
|
||||
lsm.checkpointLsm();
|
||||
|
||||
// Inspect live per-bucket state.
|
||||
lsm.getLsmStats().ifPresent(stats -> stats.buckets().forEach(bucket ->
|
||||
System.out.println(bucket.shardId() + ": " + bucket.generations().size() + " L0 generations")));
|
||||
|
||||
client.close();
|
||||
```
|
||||
|
||||
`maintainedIndexes` is tri-state, and the null default is the opposite of what a Java
|
||||
reader usually expects:
|
||||
|
||||
| Value | Meaning |
|
||||
| --- | --- |
|
||||
| unset (null) | Maintain **every** index the MemWAL can, resolved on install |
|
||||
| `Collections.emptyList()` | Maintain **none** |
|
||||
| `Arrays.asList("id_idx")` | Maintain exactly those |
|
||||
|
||||
## Development
|
||||
|
||||
Build:
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
<parent>
|
||||
<groupId>com.lancedb</groupId>
|
||||
<artifactId>lancedb-parent</artifactId>
|
||||
<version>0.38.0-beta.0</version>
|
||||
<version>0.38.0-beta.2</version>
|
||||
<relativePath>../pom.xml</relativePath>
|
||||
</parent>
|
||||
|
||||
@@ -33,6 +33,20 @@
|
||||
<artifactId>arrow-memory-netty</artifactId>
|
||||
</dependency>
|
||||
|
||||
<!-- Transport for the LanceDB routes outside the Lance Namespace spec.
|
||||
Versions match what lance-namespace-apache-client resolves to. -->
|
||||
<dependency>
|
||||
<groupId>org.apache.httpcomponents.client5</groupId>
|
||||
<artifactId>httpclient5</artifactId>
|
||||
<version>5.2.1</version>
|
||||
</dependency>
|
||||
|
||||
<dependency>
|
||||
<groupId>com.fasterxml.jackson.core</groupId>
|
||||
<artifactId>jackson-databind</artifactId>
|
||||
<version>2.17.1</version>
|
||||
</dependency>
|
||||
|
||||
<dependency>
|
||||
<groupId>org.junit.jupiter</groupId>
|
||||
<artifactId>junit-jupiter</artifactId>
|
||||
|
||||
@@ -0,0 +1,194 @@
|
||||
/*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package com.lancedb;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.List;
|
||||
import java.util.Optional;
|
||||
import java.util.OptionalLong;
|
||||
|
||||
/**
|
||||
* Live state of one bucket. A table is N buckets on one node; flattening to a single number hides
|
||||
* the one hot bucket that is usually why someone opened this endpoint.
|
||||
*/
|
||||
public class BucketStats {
|
||||
private static final String CONTEXT = "bucket stats";
|
||||
|
||||
private final String shardId;
|
||||
private final String status;
|
||||
private final long writerEpoch;
|
||||
private final long manifestVersion;
|
||||
private final long currentGeneration;
|
||||
private final long replayAfterWalEntryPosition;
|
||||
private final long walEntryPositionLastSeen;
|
||||
private final List<GenerationStats> generations;
|
||||
private final boolean compacting;
|
||||
private final List<MemtableStats> memtables;
|
||||
|
||||
BucketStats(
|
||||
String shardId,
|
||||
String status,
|
||||
long writerEpoch,
|
||||
long manifestVersion,
|
||||
long currentGeneration,
|
||||
long replayAfterWalEntryPosition,
|
||||
long walEntryPositionLastSeen,
|
||||
List<GenerationStats> generations,
|
||||
boolean compacting,
|
||||
List<MemtableStats> memtables) {
|
||||
this.shardId = shardId;
|
||||
this.status = status;
|
||||
this.writerEpoch = writerEpoch;
|
||||
this.manifestVersion = manifestVersion;
|
||||
this.currentGeneration = currentGeneration;
|
||||
this.replayAfterWalEntryPosition = replayAfterWalEntryPosition;
|
||||
this.walEntryPositionLastSeen = walEntryPositionLastSeen;
|
||||
this.generations = Collections.unmodifiableList(generations);
|
||||
this.compacting = compacting;
|
||||
this.memtables = memtables == null ? null : Collections.unmodifiableList(memtables);
|
||||
}
|
||||
|
||||
/** The shard this bucket writes. */
|
||||
public String shardId() {
|
||||
return shardId;
|
||||
}
|
||||
|
||||
/** {@code "Active"} or {@code "Sealed"} (drop-table 2PC in flight). */
|
||||
public String status() {
|
||||
return status;
|
||||
}
|
||||
|
||||
/** Epoch of the writer that currently owns the shard. */
|
||||
public long writerEpoch() {
|
||||
return writerEpoch;
|
||||
}
|
||||
|
||||
/** Version of the shard manifest these numbers were read from. */
|
||||
public long manifestVersion() {
|
||||
return manifestVersion;
|
||||
}
|
||||
|
||||
/** The generation the active memtable will become. */
|
||||
public long currentGeneration() {
|
||||
return currentGeneration;
|
||||
}
|
||||
|
||||
/** WAL position replay resumes from. */
|
||||
public long replayAfterWalEntryPosition() {
|
||||
return replayAfterWalEntryPosition;
|
||||
}
|
||||
|
||||
/**
|
||||
* Highest WAL position the writer has seen. The difference against {@link
|
||||
* #replayAfterWalEntryPosition()} is the WAL lag.
|
||||
*/
|
||||
public long walEntryPositionLastSeen() {
|
||||
return walEntryPositionLastSeen;
|
||||
}
|
||||
|
||||
/** Flushed L0 generations not yet merged into the base table. */
|
||||
public List<GenerationStats> generations() {
|
||||
return generations;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether a pass owns this bucket's compaction latch right now. Says <em>a</em> driver is
|
||||
* running, not <em>whose</em>, and the latch is held from dispatch — including while the pass
|
||||
* queues for a pod-wide compactor permit. Read it as "do not pile on", never as "mine is
|
||||
* progressing".
|
||||
*/
|
||||
public boolean compacting() {
|
||||
return compacting;
|
||||
}
|
||||
|
||||
/** Oldest first, active last. Empty for a {@code "Sealed"} bucket, whose state is torn down. */
|
||||
public Optional<List<MemtableStats>> memtables() {
|
||||
return Optional.ofNullable(memtables);
|
||||
}
|
||||
|
||||
/** The newest flushed generation, or empty when L0 is empty. */
|
||||
OptionalLong newestGeneration() {
|
||||
OptionalLong newest = OptionalLong.empty();
|
||||
for (GenerationStats generation : generations) {
|
||||
if (!newest.isPresent() || generation.generation() > newest.getAsLong()) {
|
||||
newest = OptionalLong.of(generation.generation());
|
||||
}
|
||||
}
|
||||
return newest;
|
||||
}
|
||||
|
||||
/**
|
||||
* How many generations at or below {@code target} are still in L0.
|
||||
*
|
||||
* <p>A count, not a boolean: one pass drains a bounded prefix rather than the whole target set,
|
||||
* so a boolean would read as "no progress" for every pass but the last. Compaction drains
|
||||
* oldest-first, so this decreases monotonically.
|
||||
*/
|
||||
long outstandingGenerations(long target) {
|
||||
long count = 0;
|
||||
for (GenerationStats generation : generations) {
|
||||
if (generation.generation() <= target) {
|
||||
count++;
|
||||
}
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
static BucketStats fromJson(JsonNode node) {
|
||||
JsonFields.requiredObject(node, CONTEXT);
|
||||
List<GenerationStats> generations = new ArrayList<GenerationStats>();
|
||||
for (JsonNode generation : JsonFields.requiredArray(node, "generations", CONTEXT)) {
|
||||
generations.add(GenerationStats.fromJson(generation));
|
||||
}
|
||||
|
||||
JsonNode memtablesNode = JsonFields.optionalArray(node, "memtables", CONTEXT);
|
||||
List<MemtableStats> memtables = null;
|
||||
if (memtablesNode != null) {
|
||||
memtables = new ArrayList<MemtableStats>();
|
||||
for (JsonNode memtable : memtablesNode) {
|
||||
memtables.add(MemtableStats.fromJson(memtable));
|
||||
}
|
||||
}
|
||||
|
||||
return new BucketStats(
|
||||
JsonFields.requiredText(node, "shard_id", CONTEXT),
|
||||
JsonFields.requiredText(node, "status", CONTEXT),
|
||||
JsonFields.requiredLong(node, "writer_epoch", CONTEXT),
|
||||
JsonFields.requiredLong(node, "manifest_version", CONTEXT),
|
||||
JsonFields.requiredLong(node, "current_generation", CONTEXT),
|
||||
JsonFields.requiredLong(node, "replay_after_wal_entry_position", CONTEXT),
|
||||
JsonFields.requiredLong(node, "wal_entry_position_last_seen", CONTEXT),
|
||||
generations,
|
||||
JsonFields.requiredBoolean(node, "compacting", CONTEXT),
|
||||
memtables);
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
return "BucketStats{shardId="
|
||||
+ shardId
|
||||
+ ", status="
|
||||
+ status
|
||||
+ ", currentGeneration="
|
||||
+ currentGeneration
|
||||
+ ", generations="
|
||||
+ generations
|
||||
+ ", compacting="
|
||||
+ compacting
|
||||
+ "}";
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
/*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package com.lancedb;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.OptionalLong;
|
||||
|
||||
/** One flushed L0 generation. */
|
||||
public class GenerationStats {
|
||||
private static final String CONTEXT = "generation stats";
|
||||
|
||||
private final long generation;
|
||||
private final long bytes;
|
||||
private final Long rows;
|
||||
|
||||
GenerationStats(long generation, long bytes, Long rows) {
|
||||
this.generation = generation;
|
||||
this.bytes = bytes;
|
||||
this.rows = rows;
|
||||
}
|
||||
|
||||
/** The generation number. Increases as memtables are sealed into L0. */
|
||||
public long generation() {
|
||||
return generation;
|
||||
}
|
||||
|
||||
/** On-disk size of the generation. */
|
||||
public long bytes() {
|
||||
return bytes;
|
||||
}
|
||||
|
||||
/**
|
||||
* Rows in this generation, present only when {@code includeGenerationRows} was requested. Off by
|
||||
* default because each count opens an uncached Lance dataset.
|
||||
*/
|
||||
public OptionalLong rows() {
|
||||
return rows == null ? OptionalLong.empty() : OptionalLong.of(rows);
|
||||
}
|
||||
|
||||
static GenerationStats fromJson(JsonNode node) {
|
||||
JsonFields.requiredObject(node, CONTEXT);
|
||||
return new GenerationStats(
|
||||
JsonFields.requiredLong(node, "generation", CONTEXT),
|
||||
JsonFields.requiredLong(node, "bytes", CONTEXT),
|
||||
JsonFields.optionalLong(node, "rows", CONTEXT));
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
return "GenerationStats{generation=" + generation + ", bytes=" + bytes + ", rows=" + rows + "}";
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,109 @@
|
||||
/*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package com.lancedb;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
/**
|
||||
* Strict readers for decoding LanceDB JSON responses.
|
||||
*
|
||||
* <p>Every reader fails closed: a missing, null, or wrong-typed field throws rather than
|
||||
* defaulting. That mirrors the serde decoding the Rust client applies to the same payloads in
|
||||
* {@code rust/lancedb/src/table/lsm_stats.rs}, where a required field has no default and a
|
||||
* malformed response is an error rather than a zero.
|
||||
*
|
||||
* <p>The alternative — Jackson's {@code path()}, which yields a missing node that reads as an empty
|
||||
* array or a zero — is unsafe here because {@link LanceDbTableLsm#checkpointLsm()} decides
|
||||
* convergence from these numbers. A defaulted {@code generations} array is indistinguishable from a
|
||||
* drained one, so a malformed response would report a checkpoint that never happened.
|
||||
*/
|
||||
final class JsonFields {
|
||||
private JsonFields() {}
|
||||
|
||||
/** The node itself, once confirmed to be a JSON object. */
|
||||
static JsonNode requiredObject(JsonNode node, String context) {
|
||||
if (node == null || !node.isObject()) {
|
||||
throw new IllegalStateException(context + " is not a JSON object: " + node);
|
||||
}
|
||||
return node;
|
||||
}
|
||||
|
||||
static String requiredText(JsonNode owner, String field, String context) {
|
||||
JsonNode value = required(owner, field, context);
|
||||
if (!value.isTextual()) {
|
||||
throw new IllegalStateException(fieldIs(context, field, "a string", value));
|
||||
}
|
||||
return value.asText();
|
||||
}
|
||||
|
||||
static long requiredLong(JsonNode owner, String field, String context) {
|
||||
JsonNode value = required(owner, field, context);
|
||||
if (!value.isIntegralNumber()) {
|
||||
throw new IllegalStateException(fieldIs(context, field, "an integer", value));
|
||||
}
|
||||
return value.asLong();
|
||||
}
|
||||
|
||||
static boolean requiredBoolean(JsonNode owner, String field, String context) {
|
||||
JsonNode value = required(owner, field, context);
|
||||
if (!value.isBoolean()) {
|
||||
throw new IllegalStateException(fieldIs(context, field, "a boolean", value));
|
||||
}
|
||||
return value.asBoolean();
|
||||
}
|
||||
|
||||
static JsonNode requiredArray(JsonNode owner, String field, String context) {
|
||||
JsonNode value = required(owner, field, context);
|
||||
if (!value.isArray()) {
|
||||
throw new IllegalStateException(fieldIs(context, field, "an array", value));
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
/** Null when the field is absent or JSON null, mirroring a serde {@code Option}. */
|
||||
static Long optionalLong(JsonNode owner, String field, String context) {
|
||||
JsonNode value = owner.get(field);
|
||||
if (value == null || value.isNull()) {
|
||||
return null;
|
||||
}
|
||||
if (!value.isIntegralNumber()) {
|
||||
throw new IllegalStateException(fieldIs(context, field, "an integer", value));
|
||||
}
|
||||
return value.asLong();
|
||||
}
|
||||
|
||||
/** Null when the field is absent or JSON null, mirroring a serde {@code Option}. */
|
||||
static JsonNode optionalArray(JsonNode owner, String field, String context) {
|
||||
JsonNode value = owner.get(field);
|
||||
if (value == null || value.isNull()) {
|
||||
return null;
|
||||
}
|
||||
if (!value.isArray()) {
|
||||
throw new IllegalStateException(fieldIs(context, field, "an array", value));
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
private static JsonNode required(JsonNode owner, String field, String context) {
|
||||
JsonNode value = owner.get(field);
|
||||
if (value == null || value.isNull()) {
|
||||
throw new IllegalStateException(context + " is missing required field '" + field + "'");
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
private static String fieldIs(String context, String field, String expected, JsonNode value) {
|
||||
return context + " field '" + field + "' is not " + expected + ": " + value;
|
||||
}
|
||||
}
|
||||
@@ -136,29 +136,48 @@ public class LanceDbNamespaceClientBuilder {
|
||||
* @throws IllegalStateException if required parameters are missing
|
||||
*/
|
||||
public LanceNamespace build() {
|
||||
// Validate required fields
|
||||
validate();
|
||||
|
||||
// Build configuration map
|
||||
Map<String, String> config = new HashMap<>(additionalConfig);
|
||||
config.put("header.x-lancedb-database", database);
|
||||
config.put("header.x-api-key", apiKey);
|
||||
config.put("uri", resolveUri());
|
||||
|
||||
return LanceNamespace.connect("rest", config, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a {@link LanceDbRestClient} for the same endpoint.
|
||||
*
|
||||
* <p>Needed only for LanceDB routes that the Lance Namespace specification does not cover — the
|
||||
* MemWAL LSM write path, reached through {@link LanceDbTableLsm}. Every other table operation
|
||||
* belongs on the {@link LanceNamespace} from {@link #build()}.
|
||||
*
|
||||
* <p>The returned client owns an HTTP connection pool; close it when you are done with it.
|
||||
*
|
||||
* @return A configured LanceDbRestClient
|
||||
* @throws IllegalStateException if required parameters are missing
|
||||
*/
|
||||
public LanceDbRestClient buildRestClient() {
|
||||
validate();
|
||||
return new LanceDbRestClient(resolveUri(), apiKey, database);
|
||||
}
|
||||
|
||||
private void validate() {
|
||||
if (apiKey == null) {
|
||||
throw new IllegalStateException("API key is required");
|
||||
}
|
||||
if (database == null) {
|
||||
throw new IllegalStateException("Database is required");
|
||||
}
|
||||
}
|
||||
|
||||
// Build configuration map
|
||||
Map<String, String> config = new HashMap<>(additionalConfig);
|
||||
config.put("header.x-lancedb-database", database);
|
||||
config.put("header.x-api-key", apiKey);
|
||||
|
||||
// Determine base URL
|
||||
String uri;
|
||||
/** The custom endpoint when set, else the LanceDB Cloud URL for this database and region. */
|
||||
private String resolveUri() {
|
||||
if (endpoint.isPresent()) {
|
||||
uri = endpoint.get();
|
||||
} else {
|
||||
String effectiveRegion = region.orElse(DEFAULT_REGION);
|
||||
uri = String.format(CLOUD_URL_PATTERN, database, effectiveRegion);
|
||||
return endpoint.get();
|
||||
}
|
||||
config.put("uri", uri);
|
||||
|
||||
return LanceNamespace.connect("rest", config, null);
|
||||
return String.format(CLOUD_URL_PATTERN, database, region.orElse(DEFAULT_REGION));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,119 @@
|
||||
/*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package com.lancedb;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import org.apache.hc.client5.http.classic.methods.HttpPost;
|
||||
import org.apache.hc.client5.http.impl.classic.CloseableHttpClient;
|
||||
import org.apache.hc.client5.http.impl.classic.HttpClients;
|
||||
import org.apache.hc.core5.http.ContentType;
|
||||
import org.apache.hc.core5.http.io.entity.EntityUtils;
|
||||
import org.apache.hc.core5.http.io.entity.StringEntity;
|
||||
|
||||
import java.io.Closeable;
|
||||
import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
|
||||
/**
|
||||
* Minimal HTTP client for LanceDB Cloud and Enterprise routes that the Lance Namespace
|
||||
* specification does not cover.
|
||||
*
|
||||
* <p>Most table operations reach LanceDB through {@link org.lance.namespace.LanceNamespace}, which
|
||||
* is generated from the namespace spec. A handful of routes — the MemWAL LSM write path in
|
||||
* particular — are served by the same endpoint but are not part of that spec, so they are issued
|
||||
* directly here. See {@link LanceDbTableLsm}.
|
||||
*
|
||||
* <p>Obtain one from {@link LanceDbNamespaceClientBuilder#buildRestClient()}.
|
||||
*/
|
||||
public class LanceDbRestClient implements Closeable {
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
|
||||
private final String baseUri;
|
||||
private final String apiKey;
|
||||
private final String database;
|
||||
private final CloseableHttpClient http;
|
||||
|
||||
LanceDbRestClient(String baseUri, String apiKey, String database) {
|
||||
this.baseUri = baseUri.endsWith("/") ? baseUri.substring(0, baseUri.length() - 1) : baseUri;
|
||||
this.apiKey = apiKey;
|
||||
this.database = database;
|
||||
// Automatic retries off, deliberately. The default strategy retries 429 and 503 —
|
||||
// exactly the two statuses LanceDbTableLsm.checkpointLsm() acts on — which would
|
||||
// silently double its explicit retry budget and would also retry compact_lsm in
|
||||
// place, where the loop is designed to fall through to a fresh stats poll instead.
|
||||
// The checkpoint loop owns the 421/429/503 transitions; the transport must not.
|
||||
this.http = HttpClients.custom().disableAutomaticRetries().build();
|
||||
}
|
||||
|
||||
/**
|
||||
* POST {@code path}, sending {@code body} as JSON when it is non-null.
|
||||
*
|
||||
* @param path Absolute request path, beginning with {@code /}.
|
||||
* @param body Object to serialize as the request body, or null to send no body.
|
||||
* @return The parsed response body, or null when the response carried no content.
|
||||
* @throws HttpException if the server returned a non-2xx status.
|
||||
*/
|
||||
public JsonNode post(String path, Object body) {
|
||||
HttpPost request = new HttpPost(baseUri + path);
|
||||
request.setHeader("x-api-key", apiKey);
|
||||
request.setHeader("x-lancedb-database", database);
|
||||
try {
|
||||
if (body != null) {
|
||||
request.setEntity(
|
||||
new StringEntity(MAPPER.writeValueAsString(body), ContentType.APPLICATION_JSON));
|
||||
}
|
||||
return http.execute(
|
||||
request,
|
||||
response -> {
|
||||
String text =
|
||||
response.getEntity() == null ? "" : EntityUtils.toString(response.getEntity());
|
||||
int status = response.getCode();
|
||||
if (status < 200 || status >= 300) {
|
||||
throw new HttpException(status, "LanceDB request to " + path + " failed: " + text);
|
||||
}
|
||||
return text.isEmpty() ? null : MAPPER.readTree(text);
|
||||
});
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("LanceDB request to " + path + " failed", e);
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() throws IOException {
|
||||
http.close();
|
||||
}
|
||||
|
||||
/**
|
||||
* A non-2xx response.
|
||||
*
|
||||
* <p>The status is exposed because callers act on it: {@link LanceDbTableLsm#checkpointLsm()}
|
||||
* treats 429 and 503 as retryable and 421 as a lost node claim.
|
||||
*/
|
||||
public static class HttpException extends RuntimeException {
|
||||
private static final long serialVersionUID = 1L;
|
||||
|
||||
private final int statusCode;
|
||||
|
||||
public HttpException(int statusCode, String message) {
|
||||
super(message);
|
||||
this.statusCode = statusCode;
|
||||
}
|
||||
|
||||
/** The HTTP status the failed response carried. */
|
||||
public int statusCode() {
|
||||
return statusCode;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,394 @@
|
||||
/*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package com.lancedb;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.OptionalLong;
|
||||
|
||||
/**
|
||||
* The MemWAL LSM write path for one LanceDB Cloud or Enterprise table.
|
||||
*
|
||||
* <p>Installing an {@link LsmWriteSpec} routes {@code mergeInsert} upserts through Lance's MemWAL —
|
||||
* an LSM-style append — instead of the standard merge path. Rows land in an in-memory memtable,
|
||||
* seal into L0 generations, and are merged into the base table by compaction.
|
||||
*
|
||||
* <p>These routes are not part of the Lance Namespace specification, so they are issued directly
|
||||
* rather than through {@link org.lance.namespace.LanceNamespace}.
|
||||
*
|
||||
* <pre>{@code
|
||||
* LanceDbRestClient client = LanceDbNamespaceClientBuilder.newBuilder()
|
||||
* .apiKey("your_lancedb_cloud_api_key")
|
||||
* .database("your_database_name")
|
||||
* .buildRestClient();
|
||||
*
|
||||
* LanceDbTableLsm lsm = new LanceDbTableLsm(client, "my_table");
|
||||
* lsm.setLsmWriteSpec(LsmWriteSpec.bucket("id", 16));
|
||||
* // ... merge_insert traffic ...
|
||||
* lsm.checkpointLsm();
|
||||
* }</pre>
|
||||
*/
|
||||
public class LanceDbTableLsm {
|
||||
|
||||
/**
|
||||
* Interval between {@code get_lsm_stats} polls during a checkpoint. One interval is roughly one
|
||||
* compaction pass, the granularity at which the answer can change.
|
||||
*/
|
||||
private static final long POLL_INTERVAL_MS = 5_000L;
|
||||
|
||||
/**
|
||||
* Cap on re-issues from {@code flushLsm} after a 421, so a crash-looping node cannot turn flush →
|
||||
* compact → 421 → flush into a spin.
|
||||
*
|
||||
* <p>Deliberately not shared with {@link #MAX_RETRIES}: a claim that keeps evaporating is a
|
||||
* broken node, while contention is routine and wants a real budget.
|
||||
*/
|
||||
private static final int MAX_REISSUES = 3;
|
||||
|
||||
/**
|
||||
* Retryable faults tolerated on a <em>single</em> request, reset on every success — scattered
|
||||
* contention across a long checkpoint must not accumulate toward a cap.
|
||||
*/
|
||||
private static final int MAX_RETRIES = 8;
|
||||
|
||||
private static final long RETRY_BACKOFF_BASE_MS = 100L;
|
||||
private static final long RETRY_BACKOFF_MAX_MS = 5_000L;
|
||||
|
||||
private final LanceDbRestClient client;
|
||||
private final String tableIdentifier;
|
||||
|
||||
/**
|
||||
* Bind the LSM routes for one table.
|
||||
*
|
||||
* @param client Transport for the LanceDB endpoint.
|
||||
* @param tableIdentifier The table's full identifier, {@code $}-delimited when it sits inside a
|
||||
* namespace, such as {@code analytics$events}.
|
||||
*/
|
||||
public LanceDbTableLsm(LanceDbRestClient client, String tableIdentifier) {
|
||||
if (client == null) {
|
||||
throw new IllegalArgumentException("Client cannot be null");
|
||||
}
|
||||
if (tableIdentifier == null || tableIdentifier.trim().isEmpty()) {
|
||||
throw new IllegalArgumentException("Table identifier cannot be null or empty");
|
||||
}
|
||||
this.client = client;
|
||||
this.tableIdentifier = tableIdentifier;
|
||||
}
|
||||
|
||||
/**
|
||||
* Install an {@link LsmWriteSpec} on this table, selecting the MemWAL LSM write path for future
|
||||
* {@code mergeInsert} calls.
|
||||
*
|
||||
* <p>All variants require the table to have an unenforced primary key; bucket sharding
|
||||
* additionally requires it to be the single column being bucketed.
|
||||
*/
|
||||
public void setLsmWriteSpec(LsmWriteSpec spec) {
|
||||
if (spec == null) {
|
||||
throw new IllegalArgumentException("Spec cannot be null");
|
||||
}
|
||||
client.post(route("set_lsm_write_spec"), spec.toRequestBody());
|
||||
}
|
||||
|
||||
/**
|
||||
* Remove the {@link LsmWriteSpec} from this table, reverting to the standard {@code mergeInsert}
|
||||
* write path.
|
||||
*
|
||||
* <p>Errors if no spec is currently set.
|
||||
*/
|
||||
public void unsetLsmWriteSpec() {
|
||||
client.post(route("unset_lsm_write_spec"), null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Read the {@link LsmWriteSpec} currently installed on this table.
|
||||
*
|
||||
* <p>Empty when the LSM write path is not enabled. The returned spec mirrors what was installed,
|
||||
* except that {@link LsmWriteSpec#maintainedIndexes()} always reports the concrete list resolved
|
||||
* when the spec was set — a null selection never round-trips.
|
||||
*/
|
||||
public Optional<LsmWriteSpec> getLsmWriteSpec() {
|
||||
JsonNode response = client.post(route("get_lsm_write_spec"), null);
|
||||
if (response == null || !response.hasNonNull("lsm_write_spec")) {
|
||||
return Optional.empty();
|
||||
}
|
||||
return Optional.of(LsmWriteSpec.fromJson(response.get("lsm_write_spec")));
|
||||
}
|
||||
|
||||
/**
|
||||
* Seal every bucket's active memtable into a new L0 generation.
|
||||
*
|
||||
* <p>Returns once the seal is committed. Sealing an empty memtable is a no-op, so this is safe to
|
||||
* call repeatedly.
|
||||
*/
|
||||
public void flushLsm() {
|
||||
client.post(route("flush_lsm"), null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Trigger a background L0 → base compaction pass per bucket.
|
||||
*
|
||||
* <p>Returns once the passes are <em>dispatched</em>, not once they finish — watch {@link
|
||||
* #getLsmStats}, or use {@link #checkpointLsm} to wait for convergence.
|
||||
*/
|
||||
public void compactLsm() {
|
||||
client.post(route("compact_lsm"), null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Read live per-bucket LSM state.
|
||||
*
|
||||
* <p>Answers "how far behind is my fresh tier", "which bucket is hot", and "why is my fresh-tier
|
||||
* vector search brute-force". Mutates no table state.
|
||||
*
|
||||
* <p>Empty only when the LSM write path is not enabled — that is, when the server sends an absent
|
||||
* or null {@code lsm_stats}. A stats object that is present is decoded strictly, and a malformed
|
||||
* one throws rather than decoding to something empty, because {@link #checkpointLsm} reads
|
||||
* convergence out of these numbers and cannot tell a defaulted array from a drained one.
|
||||
*
|
||||
* @param includeGenerationRows Also count rows per L0 generation. Off by default because each
|
||||
* count opens an uncached Lance dataset.
|
||||
* @throws IllegalStateException if the response is absent or does not decode.
|
||||
*/
|
||||
public Optional<LsmStats> getLsmStats(boolean includeGenerationRows) {
|
||||
Map<String, Object> body = new LinkedHashMap<String, Object>();
|
||||
body.put("include_generation_rows", includeGenerationRows);
|
||||
JsonNode response = client.post(route("get_lsm_stats"), body);
|
||||
if (response == null) {
|
||||
throw new IllegalStateException("get_lsm_stats returned an empty response body");
|
||||
}
|
||||
JsonNode stats = response.get("lsm_stats");
|
||||
if (stats == null || stats.isNull()) {
|
||||
return Optional.empty();
|
||||
}
|
||||
return Optional.of(LsmStats.fromJson(stats));
|
||||
}
|
||||
|
||||
/** Equivalent to {@code getLsmStats(false)}. */
|
||||
public Optional<LsmStats> getLsmStats() {
|
||||
return getLsmStats(false);
|
||||
}
|
||||
|
||||
/**
|
||||
* Converge this table's LSM write path into its base table.
|
||||
*
|
||||
* <p>Seals once, fixes a target watermark from the resulting L0, then triggers compaction and
|
||||
* polls until that L0 is gone. The target set is fixed at the start, so generations created
|
||||
* <em>during</em> the checkpoint are ignored — that is what lets it terminate under write load,
|
||||
* and what makes it best-effort: it converges the fresh tier as of some instant. Idempotent,
|
||||
* abandonable at any point, safe on a cadence.
|
||||
*
|
||||
* <p>The loop runs here, not on the server: {@link #compactLsm} dispatches a pass and returns, so
|
||||
* nothing holds a socket and a client can vanish mid-operation with nothing to reconcile.
|
||||
* Completion is read from generation numbers in the shard manifest — durable state, unlike a
|
||||
* count in a compact response, which a concurrent write invalidates.
|
||||
*
|
||||
* <p>No liveness bound — the caller owns the deadline. The compactor pool is shared across
|
||||
* tables, so a checkpoint queued behind unrelated work looks exactly like one that is merging.
|
||||
*/
|
||||
public void checkpointLsm() {
|
||||
for (int reissue = 0; reissue <= MAX_REISSUES; reissue++) {
|
||||
// The seal turns everything written before this call into a generation, so the
|
||||
// watermark has to be read after it. Idempotent: sealing an empty memtable is a
|
||||
// no-op, so a re-issue does not churn empty generations.
|
||||
if (issueVoid(this::flushLsm)) {
|
||||
backoff(reissue);
|
||||
continue;
|
||||
}
|
||||
|
||||
Attempt<Optional<LsmStats>> stats = issue(() -> getLsmStats(false));
|
||||
if (stats.lostClaim) {
|
||||
backoff(reissue);
|
||||
continue;
|
||||
}
|
||||
if (!stats.value.isPresent()) {
|
||||
// Not WAL-backed; flushLsm would have errored first but for a race.
|
||||
return;
|
||||
}
|
||||
|
||||
Map<String, Long> targets = newestGenerations(stats.value.get());
|
||||
if (targets.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (drainToTargets(targets)) {
|
||||
return;
|
||||
}
|
||||
backoff(reissue);
|
||||
}
|
||||
throw new IllegalStateException(
|
||||
"checkpointLsm: the owning node kept losing its claim; re-issued from flush the maximum "
|
||||
+ "number of times");
|
||||
}
|
||||
|
||||
/**
|
||||
* Trigger and poll until no bucket holds a generation at or below its target.
|
||||
*
|
||||
* @return true when the drain finished, false when the table needs re-claiming from flush.
|
||||
*/
|
||||
private boolean drainToTargets(Map<String, Long> targets) {
|
||||
while (true) {
|
||||
Attempt<Optional<LsmStats>> stats = issue(() -> getLsmStats(false));
|
||||
if (stats.lostClaim) {
|
||||
return false;
|
||||
}
|
||||
if (!stats.value.isPresent()) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// `compacting` is the bucket's compaction latch, held from dispatch until the pass
|
||||
// ends — including while it waits on a pod-wide permit. So it answers one question
|
||||
// only: do not pile on. Buckets with nothing outstanding are skipped, not counted
|
||||
// as idle.
|
||||
long outstanding = 0;
|
||||
boolean allCompacting = true;
|
||||
for (BucketStats bucket : stats.value.get().buckets()) {
|
||||
Long target = targets.get(bucket.shardId());
|
||||
if (target == null) {
|
||||
continue;
|
||||
}
|
||||
long remaining = bucket.outstandingGenerations(target);
|
||||
if (remaining > 0) {
|
||||
outstanding += remaining;
|
||||
allCompacting &= bucket.compacting();
|
||||
}
|
||||
}
|
||||
if (outstanding == 0) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (!allCompacting) {
|
||||
try {
|
||||
compactLsm();
|
||||
} catch (LanceDbRestClient.HttpException e) {
|
||||
if (isLostClaim(e)) {
|
||||
return false;
|
||||
}
|
||||
if (!isRetryable(e)) {
|
||||
throw e;
|
||||
}
|
||||
// A 429 here means the server could latch no bucket at all, which the poll
|
||||
// above already handles. Not retried in place: the latch it would contend for
|
||||
// is the one doing the work, so fall through and re-read — POLL_INTERVAL_MS is
|
||||
// the backoff.
|
||||
}
|
||||
}
|
||||
sleep(POLL_INTERVAL_MS);
|
||||
}
|
||||
}
|
||||
|
||||
/** The newest generation held by each bucket, skipping buckets holding none. */
|
||||
private static Map<String, Long> newestGenerations(LsmStats stats) {
|
||||
Map<String, Long> targets = new HashMap<String, Long>();
|
||||
for (BucketStats bucket : stats.buckets()) {
|
||||
OptionalLong newest = bucket.newestGeneration();
|
||||
if (newest.isPresent()) {
|
||||
targets.put(bucket.shardId(), newest.getAsLong());
|
||||
}
|
||||
}
|
||||
return targets;
|
||||
}
|
||||
|
||||
/**
|
||||
* 429 (latch held, pool saturated, or the pod replaying its WAL) and 503 (a draining node, or a
|
||||
* proxy between here and it).
|
||||
*/
|
||||
private static boolean isRetryable(LanceDbRestClient.HttpException e) {
|
||||
return e.statusCode() == 429 || e.statusCode() == 503;
|
||||
}
|
||||
|
||||
/**
|
||||
* 421: the owning node holds no claim. Only {@code flush} re-claims and replays, so this cannot
|
||||
* be retried in place — the caller has to start over.
|
||||
*/
|
||||
private static boolean isLostClaim(LanceDbRestClient.HttpException e) {
|
||||
return e.statusCode() == 421;
|
||||
}
|
||||
|
||||
/**
|
||||
* Issue one LSM request, retrying in place while the fault is retryable.
|
||||
*
|
||||
* <p>The two recoverable faults have separate budgets: contention clears on its own and retries
|
||||
* here against {@link #MAX_RETRIES}, while a 421 needs {@code flush} to re-claim, which only the
|
||||
* caller can drive.
|
||||
*
|
||||
* <p>An exhausted budget propagates the last error as itself rather than a synthesized one — "429
|
||||
* after nine tries" beats "checkpoint failed".
|
||||
*/
|
||||
private static <T> Attempt<T> issue(Call<T> call) {
|
||||
int retries = 0;
|
||||
while (true) {
|
||||
try {
|
||||
return new Attempt<T>(call.run(), false);
|
||||
} catch (LanceDbRestClient.HttpException e) {
|
||||
if (isLostClaim(e)) {
|
||||
return new Attempt<T>(null, true);
|
||||
}
|
||||
if (!isRetryable(e) || retries >= MAX_RETRIES) {
|
||||
throw e;
|
||||
}
|
||||
backoff(retries);
|
||||
retries++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** {@link #issue} for a call with no return value. Returns true when the claim was lost. */
|
||||
private static boolean issueVoid(Runnable call) {
|
||||
return issue(
|
||||
() -> {
|
||||
call.run();
|
||||
return Boolean.TRUE;
|
||||
})
|
||||
.lostClaim;
|
||||
}
|
||||
|
||||
/** Sleep before re-issuing a retryable request. Doubles up to {@link #RETRY_BACKOFF_MAX_MS}. */
|
||||
private static void backoff(int attempt) {
|
||||
long delay = RETRY_BACKOFF_BASE_MS << Math.min(attempt, 8);
|
||||
sleep(Math.min(delay, RETRY_BACKOFF_MAX_MS));
|
||||
}
|
||||
|
||||
private static void sleep(long millis) {
|
||||
try {
|
||||
Thread.sleep(millis);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("Interrupted while waiting on the LSM checkpoint", e);
|
||||
}
|
||||
}
|
||||
|
||||
private String route(String operation) {
|
||||
return "/v1/table/" + tableIdentifier + "/" + operation + "/";
|
||||
}
|
||||
|
||||
/** What one LSM request produced: its value, or word that the owning node holds no claim. */
|
||||
private static final class Attempt<T> {
|
||||
private final T value;
|
||||
private final boolean lostClaim;
|
||||
|
||||
private Attempt(T value, boolean lostClaim) {
|
||||
this.value = value;
|
||||
this.lostClaim = lostClaim;
|
||||
}
|
||||
}
|
||||
|
||||
@FunctionalInterface
|
||||
private interface Call<T> {
|
||||
T run();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,56 @@
|
||||
/*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package com.lancedb;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Live per-bucket LSM state, as returned by {@link LanceDbTableLsm#getLsmStats()}.
|
||||
*
|
||||
* <p>Nothing here is derived: sums and differences (total L0 bytes, WAL lag) are the caller's to
|
||||
* compute. There is no "LSM is off" shape — that case is an empty {@link java.util.Optional},
|
||||
* because a stats object of zeros would read as measurements.
|
||||
*/
|
||||
public class LsmStats {
|
||||
private static final String CONTEXT = "lsm stats";
|
||||
|
||||
private final List<BucketStats> buckets;
|
||||
|
||||
LsmStats(List<BucketStats> buckets) {
|
||||
this.buckets = Collections.unmodifiableList(buckets);
|
||||
}
|
||||
|
||||
/** One entry per bucket. */
|
||||
public List<BucketStats> buckets() {
|
||||
return buckets;
|
||||
}
|
||||
|
||||
static LsmStats fromJson(JsonNode node) {
|
||||
JsonFields.requiredObject(node, CONTEXT);
|
||||
List<BucketStats> buckets = new ArrayList<BucketStats>();
|
||||
for (JsonNode bucket : JsonFields.requiredArray(node, "buckets", CONTEXT)) {
|
||||
buckets.add(BucketStats.fromJson(bucket));
|
||||
}
|
||||
return new LsmStats(buckets);
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
return "LsmStats{buckets=" + buckets + "}";
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,260 @@
|
||||
/*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package com.lancedb;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* Specification selecting Lance's MemWAL LSM-style write path for {@code mergeInsert}.
|
||||
*
|
||||
* <p>Construct via {@link #bucket}, {@link #identity}, or {@link #unsharded}, then optionally chain
|
||||
* {@link #withMaintainedIndexes} and {@link #withWriterConfigDefaults}. Install it with {@link
|
||||
* LanceDbTableLsm#setLsmWriteSpec} and remove it with {@link LanceDbTableLsm#unsetLsmWriteSpec}.
|
||||
*
|
||||
* <p>This is deliberately not {@code org.lance.memwal.InitializeMemWalParams}. That type is Lance's
|
||||
* own, and its maintained-index default is the opposite of this one: it defaults to maintaining
|
||||
* <em>nothing</em>, while a fresh spec here maintains <em>every</em> index. It also cannot express
|
||||
* the null that asks the server to resolve the set.
|
||||
*/
|
||||
public class LsmWriteSpec {
|
||||
|
||||
/** How writes are routed to MemWAL shards. */
|
||||
public enum Sharding {
|
||||
/** Hash-bucket writes by a scalar column. */
|
||||
BUCKET("bucket"),
|
||||
/** Shard by the raw value of a scalar column. */
|
||||
IDENTITY("identity"),
|
||||
/** Route every write to a single shard. */
|
||||
UNSHARDED("unsharded");
|
||||
|
||||
private final String wireName;
|
||||
|
||||
Sharding(String wireName) {
|
||||
this.wireName = wireName;
|
||||
}
|
||||
|
||||
String wireName() {
|
||||
return wireName;
|
||||
}
|
||||
|
||||
static Sharding fromWireName(String name) {
|
||||
for (Sharding s : values()) {
|
||||
if (s.wireName.equals(name)) {
|
||||
return s;
|
||||
}
|
||||
}
|
||||
throw new IllegalArgumentException("Unknown sharding mode: " + name);
|
||||
}
|
||||
}
|
||||
|
||||
private final Sharding sharding;
|
||||
private final String column;
|
||||
private final Integer numBuckets;
|
||||
private final List<String> maintainedIndexes;
|
||||
private final Map<String, String> writerConfigDefaults;
|
||||
|
||||
private LsmWriteSpec(
|
||||
Sharding sharding,
|
||||
String column,
|
||||
Integer numBuckets,
|
||||
List<String> maintainedIndexes,
|
||||
Map<String, String> writerConfigDefaults) {
|
||||
this.sharding = sharding;
|
||||
this.column = column;
|
||||
this.numBuckets = numBuckets;
|
||||
this.maintainedIndexes = maintainedIndexes;
|
||||
this.writerConfigDefaults = writerConfigDefaults;
|
||||
}
|
||||
|
||||
/**
|
||||
* Hash-bucket sharding by a scalar column, maintaining every index on the table.
|
||||
*
|
||||
* <p>Iceberg-compatible Murmur3-x86-32 (seed 0) is used, so each row's {@code bucket(column,
|
||||
* numBuckets)} value is stable across processes.
|
||||
*
|
||||
* @param column A non-nested column with a supported scalar type.
|
||||
* @param numBuckets The number of buckets, in {@code [1, 1024]}.
|
||||
*/
|
||||
public static LsmWriteSpec bucket(String column, int numBuckets) {
|
||||
if (column == null || column.trim().isEmpty()) {
|
||||
throw new IllegalArgumentException("Column cannot be null or empty");
|
||||
}
|
||||
return new LsmWriteSpec(
|
||||
Sharding.BUCKET, column, numBuckets, null, new HashMap<String, String>());
|
||||
}
|
||||
|
||||
/**
|
||||
* Identity sharding — shard by the raw value of {@code column} — maintaining every index on the
|
||||
* table.
|
||||
*
|
||||
* <p>{@code column} must be a deterministic function of the unenforced primary key: every row
|
||||
* with a given primary key must always produce the same {@code column} value, or upserts of that
|
||||
* key can land in different shards and a stale version can win.
|
||||
*/
|
||||
public static LsmWriteSpec identity(String column) {
|
||||
if (column == null || column.trim().isEmpty()) {
|
||||
throw new IllegalArgumentException("Column cannot be null or empty");
|
||||
}
|
||||
return new LsmWriteSpec(Sharding.IDENTITY, column, null, null, new HashMap<String, String>());
|
||||
}
|
||||
|
||||
/** No sharding — every write goes to a single MemWAL shard — maintaining every index. */
|
||||
public static LsmWriteSpec unsharded() {
|
||||
return new LsmWriteSpec(Sharding.UNSHARDED, null, null, null, new HashMap<String, String>());
|
||||
}
|
||||
|
||||
/**
|
||||
* Set the indexes the MemWAL keeps up to date as rows are appended.
|
||||
*
|
||||
* <p>Pass {@code null} — the default for a fresh spec — to maintain every index the MemWAL can,
|
||||
* resolved when the spec is installed. That is a snapshot: indexes created later are not
|
||||
* maintained until the spec is unset and set again. Pass an empty list to maintain none.
|
||||
*
|
||||
* <p>Note that {@code null} and the empty list mean opposite things here.
|
||||
*/
|
||||
public LsmWriteSpec withMaintainedIndexes(List<String> maintainedIndexes) {
|
||||
return new LsmWriteSpec(
|
||||
sharding,
|
||||
column,
|
||||
numBuckets,
|
||||
maintainedIndexes == null ? null : new ArrayList<String>(maintainedIndexes),
|
||||
writerConfigDefaults);
|
||||
}
|
||||
|
||||
/**
|
||||
* Set default {@code ShardWriter} configuration recorded in the MemWAL index.
|
||||
*
|
||||
* <p>A sparse override map — only the keys you set are recorded. Recognized keys include {@code
|
||||
* durable_write}, {@code max_wal_buffer_size}, {@code max_memtable_size}, {@code
|
||||
* max_memtable_rows}, {@code max_memtable_batches}, {@code manifest_scan_batch_size}, {@code
|
||||
* max_unflushed_memtable_bytes}, and {@code enable_memtable}. Duration knobs carry an {@code _ms}
|
||||
* suffix, such as {@code max_wal_flush_interval_ms}.
|
||||
*/
|
||||
public LsmWriteSpec withWriterConfigDefaults(Map<String, String> writerConfigDefaults) {
|
||||
if (writerConfigDefaults == null) {
|
||||
throw new IllegalArgumentException("writerConfigDefaults cannot be null");
|
||||
}
|
||||
return new LsmWriteSpec(
|
||||
sharding,
|
||||
column,
|
||||
numBuckets,
|
||||
maintainedIndexes,
|
||||
new HashMap<String, String>(writerConfigDefaults));
|
||||
}
|
||||
|
||||
/** How writes are routed to shards. */
|
||||
public Sharding sharding() {
|
||||
return sharding;
|
||||
}
|
||||
|
||||
/** The sharding column for {@link Sharding#BUCKET} and {@link Sharding#IDENTITY}, else null. */
|
||||
public String column() {
|
||||
return column;
|
||||
}
|
||||
|
||||
/** The bucket count for {@link Sharding#BUCKET}, else null. */
|
||||
public Integer numBuckets() {
|
||||
return numBuckets;
|
||||
}
|
||||
|
||||
/**
|
||||
* The indexes the MemWAL maintains, or null to have the server resolve every maintainable index
|
||||
* on install. An empty list means none.
|
||||
*/
|
||||
public List<String> maintainedIndexes() {
|
||||
return maintainedIndexes == null ? null : Collections.unmodifiableList(maintainedIndexes);
|
||||
}
|
||||
|
||||
/** Default {@code ShardWriter} configuration recorded in the MemWAL index. */
|
||||
public Map<String, String> writerConfigDefaults() {
|
||||
return Collections.unmodifiableMap(writerConfigDefaults);
|
||||
}
|
||||
|
||||
/** Render this spec as the {@code set_lsm_write_spec} request body. */
|
||||
Map<String, Object> toRequestBody() {
|
||||
Map<String, Object> shardingBody = new LinkedHashMap<String, Object>();
|
||||
shardingBody.put("mode", sharding.wireName());
|
||||
if (column != null) {
|
||||
shardingBody.put("column", column);
|
||||
}
|
||||
if (numBuckets != null) {
|
||||
shardingBody.put("num_buckets", numBuckets);
|
||||
}
|
||||
|
||||
Map<String, Object> body = new LinkedHashMap<String, Object>();
|
||||
body.put("sharding", shardingBody);
|
||||
// Null is meaningful: it asks the server to resolve every maintainable index.
|
||||
body.put("maintained_indexes", maintainedIndexes);
|
||||
body.put("writer_config_defaults", writerConfigDefaults);
|
||||
return body;
|
||||
}
|
||||
|
||||
/**
|
||||
* Rebuild a spec from a {@code get_lsm_write_spec} response body.
|
||||
*
|
||||
* <p>The server always reports a concrete maintained-index list, so a null selection never
|
||||
* round-trips.
|
||||
*/
|
||||
static LsmWriteSpec fromJson(JsonNode node) {
|
||||
JsonNode shardingNode = node.get("sharding");
|
||||
if (shardingNode == null || shardingNode.get("mode") == null) {
|
||||
throw new IllegalStateException("get_lsm_write_spec response has no sharding mode");
|
||||
}
|
||||
Sharding sharding = Sharding.fromWireName(shardingNode.get("mode").asText());
|
||||
|
||||
String column = shardingNode.hasNonNull("column") ? shardingNode.get("column").asText() : null;
|
||||
Integer numBuckets =
|
||||
shardingNode.hasNonNull("num_buckets") ? shardingNode.get("num_buckets").asInt() : null;
|
||||
|
||||
List<String> maintainedIndexes = new ArrayList<String>();
|
||||
JsonNode indexesNode = node.get("maintained_indexes");
|
||||
if (indexesNode != null && indexesNode.isArray()) {
|
||||
for (JsonNode index : indexesNode) {
|
||||
maintainedIndexes.add(index.asText());
|
||||
}
|
||||
}
|
||||
|
||||
Map<String, String> defaults = new HashMap<String, String>();
|
||||
JsonNode defaultsNode = node.get("writer_config_defaults");
|
||||
if (defaultsNode != null && defaultsNode.isObject()) {
|
||||
defaultsNode
|
||||
.fieldNames()
|
||||
.forEachRemaining(name -> defaults.put(name, defaultsNode.get(name).asText()));
|
||||
}
|
||||
|
||||
return new LsmWriteSpec(sharding, column, numBuckets, maintainedIndexes, defaults);
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
return "LsmWriteSpec{sharding="
|
||||
+ sharding
|
||||
+ ", column="
|
||||
+ column
|
||||
+ ", numBuckets="
|
||||
+ numBuckets
|
||||
+ ", maintainedIndexes="
|
||||
+ maintainedIndexes
|
||||
+ ", writerConfigDefaults="
|
||||
+ writerConfigDefaults
|
||||
+ "}";
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,99 @@
|
||||
/*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package com.lancedb;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.List;
|
||||
|
||||
/** One in-memory memtable. */
|
||||
public class MemtableStats {
|
||||
private static final String CONTEXT = "memtable stats";
|
||||
|
||||
private final long generation;
|
||||
private final long rows;
|
||||
private final long bytes;
|
||||
private final long batches;
|
||||
private final List<String> indexes;
|
||||
|
||||
MemtableStats(long generation, long rows, long bytes, long batches, List<String> indexes) {
|
||||
this.generation = generation;
|
||||
this.rows = rows;
|
||||
this.bytes = bytes;
|
||||
this.batches = batches;
|
||||
this.indexes = Collections.unmodifiableList(indexes);
|
||||
}
|
||||
|
||||
/** The generation this memtable will become once sealed. */
|
||||
public long generation() {
|
||||
return generation;
|
||||
}
|
||||
|
||||
/** Rows currently buffered. */
|
||||
public long rows() {
|
||||
return rows;
|
||||
}
|
||||
|
||||
/** Estimated in-memory size. */
|
||||
public long bytes() {
|
||||
return bytes;
|
||||
}
|
||||
|
||||
/** Record batches currently buffered. */
|
||||
public long batches() {
|
||||
return batches;
|
||||
}
|
||||
|
||||
/**
|
||||
* Names of the indexes this memtable carries. An absent name is the whole answer to "why is my
|
||||
* fresh-tier search on that column brute-force".
|
||||
*/
|
||||
public List<String> indexes() {
|
||||
return indexes;
|
||||
}
|
||||
|
||||
static MemtableStats fromJson(JsonNode node) {
|
||||
JsonFields.requiredObject(node, CONTEXT);
|
||||
List<String> indexes = new ArrayList<String>();
|
||||
for (JsonNode index : JsonFields.requiredArray(node, "indexes", CONTEXT)) {
|
||||
if (!index.isTextual()) {
|
||||
throw new IllegalStateException(CONTEXT + " has a non-string index name: " + index);
|
||||
}
|
||||
indexes.add(index.asText());
|
||||
}
|
||||
return new MemtableStats(
|
||||
JsonFields.requiredLong(node, "generation", CONTEXT),
|
||||
JsonFields.requiredLong(node, "rows", CONTEXT),
|
||||
JsonFields.requiredLong(node, "bytes", CONTEXT),
|
||||
JsonFields.requiredLong(node, "batches", CONTEXT),
|
||||
indexes);
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
return "MemtableStats{generation="
|
||||
+ generation
|
||||
+ ", rows="
|
||||
+ rows
|
||||
+ ", bytes="
|
||||
+ bytes
|
||||
+ ", batches="
|
||||
+ batches
|
||||
+ ", indexes="
|
||||
+ indexes
|
||||
+ "}";
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,570 @@
|
||||
/*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package com.lancedb;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.sun.net.httpserver.HttpServer;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.net.InetSocketAddress;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.ArrayDeque;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Arrays;
|
||||
import java.util.Collections;
|
||||
import java.util.Deque;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* Unit tests for the MemWAL LSM routes, run against a scripted local HTTP server.
|
||||
*
|
||||
* <p>The wire assertions mirror the Rust mocked-endpoint tests in {@code
|
||||
* rust/lancedb/src/remote/table.rs}, which are the contract these routes have to match.
|
||||
*/
|
||||
public class LanceDbTableLsmTest {
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
|
||||
private HttpServer server;
|
||||
private LanceDbRestClient client;
|
||||
private LanceDbTableLsm lsm;
|
||||
|
||||
private final List<String> requestPaths = Collections.synchronizedList(new ArrayList<String>());
|
||||
private final List<String> requestBodies = Collections.synchronizedList(new ArrayList<String>());
|
||||
private final Map<String, Deque<Reply>> replies = new ConcurrentHashMap<String, Deque<Reply>>();
|
||||
|
||||
@BeforeEach
|
||||
public void setUp() throws IOException {
|
||||
start();
|
||||
}
|
||||
|
||||
/** Tear down and restart the scripted server, for a test that scripts several exchanges. */
|
||||
private void setUpFresh() {
|
||||
try {
|
||||
client.close();
|
||||
server.stop(0);
|
||||
requestPaths.clear();
|
||||
requestBodies.clear();
|
||||
replies.clear();
|
||||
start();
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException(e);
|
||||
}
|
||||
}
|
||||
|
||||
private void start() throws IOException {
|
||||
server = HttpServer.create(new InetSocketAddress("127.0.0.1", 0), 0);
|
||||
server.createContext(
|
||||
"/",
|
||||
exchange -> {
|
||||
String path = exchange.getRequestURI().getPath();
|
||||
requestPaths.add(path);
|
||||
requestBodies.add(readAll(exchange.getRequestBody()));
|
||||
|
||||
Reply reply = nextReply(path);
|
||||
byte[] out = reply.body.getBytes(StandardCharsets.UTF_8);
|
||||
exchange.sendResponseHeaders(reply.status, out.length == 0 ? -1 : out.length);
|
||||
if (out.length > 0) {
|
||||
exchange.getResponseBody().write(out);
|
||||
}
|
||||
exchange.close();
|
||||
});
|
||||
server.start();
|
||||
|
||||
client =
|
||||
LanceDbNamespaceClientBuilder.newBuilder()
|
||||
.apiKey("test-key")
|
||||
.database("test-db")
|
||||
.endpoint("http://127.0.0.1:" + server.getAddress().getPort())
|
||||
.buildRestClient();
|
||||
lsm = new LanceDbTableLsm(client, "my_table");
|
||||
}
|
||||
|
||||
@AfterEach
|
||||
public void tearDown() throws IOException {
|
||||
client.close();
|
||||
server.stop(0);
|
||||
}
|
||||
|
||||
// ===========================================================================
|
||||
// set / unset / get spec
|
||||
// ===========================================================================
|
||||
|
||||
@Test
|
||||
public void testSetLsmWriteSpecUnsharded() throws Exception {
|
||||
enqueue("set_lsm_write_spec", 200, "");
|
||||
|
||||
lsm.setLsmWriteSpec(LsmWriteSpec.unsharded());
|
||||
|
||||
assertEquals("/v1/table/my_table/set_lsm_write_spec/", requestPaths.get(0));
|
||||
JsonNode body = MAPPER.readTree(requestBodies.get(0));
|
||||
assertEquals("unsharded", body.get("sharding").get("mode").asText());
|
||||
assertFalse(body.get("sharding").has("column"));
|
||||
assertFalse(body.get("sharding").has("num_buckets"));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testSetLsmWriteSpecBucket() throws Exception {
|
||||
enqueue("set_lsm_write_spec", 200, "");
|
||||
|
||||
lsm.setLsmWriteSpec(
|
||||
LsmWriteSpec.bucket("id", 16).withMaintainedIndexes(Arrays.asList("id_idx")));
|
||||
|
||||
JsonNode body = MAPPER.readTree(requestBodies.get(0));
|
||||
assertEquals("bucket", body.get("sharding").get("mode").asText());
|
||||
assertEquals("id", body.get("sharding").get("column").asText());
|
||||
assertEquals(16, body.get("sharding").get("num_buckets").asInt());
|
||||
assertEquals(1, body.get("maintained_indexes").size());
|
||||
assertEquals("id_idx", body.get("maintained_indexes").get(0).asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testSetLsmWriteSpecIdentity() throws Exception {
|
||||
enqueue("set_lsm_write_spec", 200, "");
|
||||
|
||||
lsm.setLsmWriteSpec(LsmWriteSpec.identity("tenant"));
|
||||
|
||||
JsonNode body = MAPPER.readTree(requestBodies.get(0));
|
||||
assertEquals("identity", body.get("sharding").get("mode").asText());
|
||||
assertEquals("tenant", body.get("sharding").get("column").asText());
|
||||
assertFalse(body.get("sharding").has("num_buckets"));
|
||||
}
|
||||
|
||||
/**
|
||||
* The tri-state that motivated a LanceDB-owned spec type: a null selection asks the server to
|
||||
* resolve every maintainable index, while an empty list asks for none. They must not collapse.
|
||||
*/
|
||||
@Test
|
||||
public void testMaintainedIndexesNullAndEmptyAreDistinctOnTheWire() throws Exception {
|
||||
enqueue("set_lsm_write_spec", 200, "");
|
||||
|
||||
lsm.setLsmWriteSpec(LsmWriteSpec.unsharded());
|
||||
JsonNode fresh = MAPPER.readTree(requestBodies.get(0));
|
||||
assertTrue(fresh.has("maintained_indexes"), "the key must be present");
|
||||
assertTrue(fresh.get("maintained_indexes").isNull(), "a fresh spec sends null, not []");
|
||||
|
||||
lsm.setLsmWriteSpec(
|
||||
LsmWriteSpec.unsharded().withMaintainedIndexes(Collections.<String>emptyList()));
|
||||
JsonNode none = MAPPER.readTree(requestBodies.get(1));
|
||||
assertTrue(none.get("maintained_indexes").isArray());
|
||||
assertEquals(0, none.get("maintained_indexes").size());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testSetLsmWriteSpecWriterConfigDefaults() throws Exception {
|
||||
enqueue("set_lsm_write_spec", 200, "");
|
||||
|
||||
Map<String, String> defaults = new HashMap<String, String>();
|
||||
defaults.put("max_memtable_rows", "50000");
|
||||
lsm.setLsmWriteSpec(LsmWriteSpec.unsharded().withWriterConfigDefaults(defaults));
|
||||
|
||||
JsonNode body = MAPPER.readTree(requestBodies.get(0));
|
||||
assertEquals("50000", body.get("writer_config_defaults").get("max_memtable_rows").asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testUnsetLsmWriteSpec() {
|
||||
enqueue("unset_lsm_write_spec", 200, "");
|
||||
|
||||
lsm.unsetLsmWriteSpec();
|
||||
|
||||
assertEquals("/v1/table/my_table/unset_lsm_write_spec/", requestPaths.get(0));
|
||||
assertEquals("", requestBodies.get(0));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testGetLsmWriteSpec() {
|
||||
enqueue(
|
||||
"get_lsm_write_spec",
|
||||
200,
|
||||
"{\"lsm_write_spec\":{\"sharding\":{\"mode\":\"bucket\",\"column\":\"id\","
|
||||
+ "\"num_buckets\":16},\"maintained_indexes\":[\"id_idx\"],"
|
||||
+ "\"writer_config_defaults\":{\"durable_write\":\"true\"}}}");
|
||||
|
||||
Optional<LsmWriteSpec> spec = lsm.getLsmWriteSpec();
|
||||
|
||||
assertTrue(spec.isPresent());
|
||||
assertEquals(LsmWriteSpec.Sharding.BUCKET, spec.get().sharding());
|
||||
assertEquals("id", spec.get().column());
|
||||
assertEquals(Integer.valueOf(16), spec.get().numBuckets());
|
||||
assertEquals(Arrays.asList("id_idx"), spec.get().maintainedIndexes());
|
||||
assertEquals("true", spec.get().writerConfigDefaults().get("durable_write"));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testGetLsmWriteSpecAbsent() {
|
||||
enqueue("get_lsm_write_spec", 200, "{\"lsm_write_spec\":null}");
|
||||
|
||||
assertFalse(lsm.getLsmWriteSpec().isPresent());
|
||||
}
|
||||
|
||||
// ===========================================================================
|
||||
// stats
|
||||
// ===========================================================================
|
||||
|
||||
@Test
|
||||
public void testGetLsmStats() throws Exception {
|
||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 7L, 8L)));
|
||||
|
||||
Optional<LsmStats> got = lsm.getLsmStats(true);
|
||||
|
||||
assertEquals("/v1/table/my_table/get_lsm_stats/", requestPaths.get(0));
|
||||
assertTrue(MAPPER.readTree(requestBodies.get(0)).get("include_generation_rows").asBoolean());
|
||||
assertTrue(got.isPresent());
|
||||
BucketStats decoded = got.get().buckets().get(0);
|
||||
assertEquals("shard-0", decoded.shardId());
|
||||
assertEquals("Active", decoded.status());
|
||||
assertEquals(1, decoded.writerEpoch());
|
||||
assertEquals(2, decoded.manifestVersion());
|
||||
assertEquals(9, decoded.currentGeneration());
|
||||
assertFalse(decoded.compacting());
|
||||
assertEquals(Arrays.asList(7L, 8L), generationNumbers(decoded));
|
||||
assertEquals(1024, decoded.generations().get(0).bytes());
|
||||
assertFalse(decoded.generations().get(0).rows().isPresent(), "rows absent unless requested");
|
||||
assertFalse(decoded.memtables().isPresent(), "absent memtables stay absent");
|
||||
}
|
||||
|
||||
/** The optional fields decode when the server does send them. */
|
||||
@Test
|
||||
public void testGetLsmStatsDecodesOptionalFields() {
|
||||
enqueue(
|
||||
"get_lsm_stats",
|
||||
200,
|
||||
"{\"lsm_stats\":{\"buckets\":[{\"shard_id\":\"shard-0\",\"status\":\"Active\","
|
||||
+ "\"writer_epoch\":1,\"manifest_version\":2,\"current_generation\":9,"
|
||||
+ "\"replay_after_wal_entry_position\":3,\"wal_entry_position_last_seen\":11,"
|
||||
+ "\"generations\":[{\"generation\":7,\"bytes\":1024,\"rows\":42}],"
|
||||
+ "\"compacting\":true,\"memtables\":[{\"generation\":8,\"rows\":5,"
|
||||
+ "\"bytes\":64,\"batches\":2,\"indexes\":[\"id_idx\"]}]}]}}");
|
||||
|
||||
BucketStats decoded = lsm.getLsmStats(true).get().buckets().get(0);
|
||||
|
||||
assertEquals(3, decoded.replayAfterWalEntryPosition());
|
||||
assertEquals(11, decoded.walEntryPositionLastSeen());
|
||||
assertTrue(decoded.compacting());
|
||||
assertEquals(42, decoded.generations().get(0).rows().getAsLong());
|
||||
assertTrue(decoded.memtables().isPresent());
|
||||
MemtableStats memtable = decoded.memtables().get().get(0);
|
||||
assertEquals(8, memtable.generation());
|
||||
assertEquals(5, memtable.rows());
|
||||
assertEquals(64, memtable.bytes());
|
||||
assertEquals(2, memtable.batches());
|
||||
assertEquals(Arrays.asList("id_idx"), memtable.indexes());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testGetLsmStatsAbsentWhenLsmDisabled() {
|
||||
enqueue("get_lsm_stats", 200, "{\"lsm_stats\":null}");
|
||||
|
||||
assertFalse(lsm.getLsmStats().isPresent());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testGetLsmStatsDefaultsToExcludingGenerationRows() throws Exception {
|
||||
enqueue("get_lsm_stats", 200, stats());
|
||||
|
||||
lsm.getLsmStats();
|
||||
|
||||
assertFalse(MAPPER.readTree(requestBodies.get(0)).get("include_generation_rows").asBoolean());
|
||||
}
|
||||
|
||||
// ===========================================================================
|
||||
// flush / compact
|
||||
// ===========================================================================
|
||||
|
||||
@Test
|
||||
public void testFlushAndCompactRoutes() {
|
||||
enqueue("flush_lsm", 200, "");
|
||||
enqueue("compact_lsm", 200, "");
|
||||
|
||||
lsm.flushLsm();
|
||||
lsm.compactLsm();
|
||||
|
||||
assertEquals("/v1/table/my_table/flush_lsm/", requestPaths.get(0));
|
||||
assertEquals("/v1/table/my_table/compact_lsm/", requestPaths.get(1));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testHttpErrorCarriesStatus() {
|
||||
enqueue("flush_lsm", 404, "no such table");
|
||||
|
||||
LanceDbRestClient.HttpException e =
|
||||
assertThrows(LanceDbRestClient.HttpException.class, () -> lsm.flushLsm());
|
||||
assertEquals(404, e.statusCode());
|
||||
}
|
||||
|
||||
// ===========================================================================
|
||||
// checkpoint
|
||||
// ===========================================================================
|
||||
|
||||
@Test
|
||||
public void testCheckpointReturnsWhenLsmDisabled() {
|
||||
enqueue("flush_lsm", 200, "");
|
||||
enqueue("get_lsm_stats", 200, "{\"lsm_stats\":null}");
|
||||
|
||||
lsm.checkpointLsm();
|
||||
|
||||
assertEquals(0, countCalls("compact_lsm"), "nothing to compact when the LSM path is off");
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testCheckpointReturnsWhenNoGenerationsOutstanding() {
|
||||
enqueue("flush_lsm", 200, "");
|
||||
// A bucket with no L0 generations yields no target, so the drain never starts.
|
||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false)));
|
||||
|
||||
lsm.checkpointLsm();
|
||||
|
||||
assertEquals(0, countCalls("compact_lsm"));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testCheckpointConvergesOnceTargetGenerationsAreGone() {
|
||||
enqueue("flush_lsm", 200, "");
|
||||
// Watermark read: shard-0 holds generations 7 and 8, so target = 8.
|
||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 7L, 8L)));
|
||||
// First drain poll: both still outstanding, nothing compacting -> dispatch a pass.
|
||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 7L, 8L)));
|
||||
// Second drain poll: drained past the target -> done.
|
||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 9L)));
|
||||
enqueue("compact_lsm", 200, "");
|
||||
|
||||
lsm.checkpointLsm();
|
||||
|
||||
assertEquals(1, countCalls("compact_lsm"), "one pass dispatched");
|
||||
assertEquals(3, countCalls("get_lsm_stats"), "watermark read plus two drain polls");
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testCheckpointDoesNotPileOnWhileEveryTargetBucketIsCompacting() {
|
||||
enqueue("flush_lsm", 200, "");
|
||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", true, 4L)));
|
||||
// Still compacting on the first poll, so no pass is dispatched; then it drains.
|
||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", true, 4L)));
|
||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 5L)));
|
||||
|
||||
lsm.checkpointLsm();
|
||||
|
||||
assertEquals(0, countCalls("compact_lsm"), "a latched bucket is left alone");
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testCheckpointRetriesFromFlushAfterLostClaim() {
|
||||
// 421 on the watermark read: the node lost its claim, so the whole thing restarts
|
||||
// from flush rather than retrying the read in place.
|
||||
enqueue("flush_lsm", 200, "");
|
||||
enqueue("get_lsm_stats", 421, "no claim");
|
||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false)));
|
||||
|
||||
lsm.checkpointLsm();
|
||||
|
||||
assertEquals(2, countCalls("flush_lsm"), "re-issued from flush");
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testCheckpointRetriesRetryableStatusInPlace() {
|
||||
enqueue("flush_lsm", 429, "latch held");
|
||||
enqueue("flush_lsm", 200, "");
|
||||
enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false)));
|
||||
|
||||
lsm.checkpointLsm();
|
||||
|
||||
assertEquals(2, countCalls("flush_lsm"), "429 retried in place, not re-issued");
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testCheckpointPropagatesTerminalStatus() {
|
||||
enqueue("flush_lsm", 400, "bad request");
|
||||
|
||||
LanceDbRestClient.HttpException e =
|
||||
assertThrows(LanceDbRestClient.HttpException.class, () -> lsm.checkpointLsm());
|
||||
assertEquals(400, e.statusCode());
|
||||
assertEquals(1, countCalls("flush_lsm"), "a terminal status is not retried");
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testCheckpointGivesUpAfterRepeatedLostClaims() {
|
||||
enqueue("flush_lsm", 421, "no claim");
|
||||
|
||||
IllegalStateException e = assertThrows(IllegalStateException.class, () -> lsm.checkpointLsm());
|
||||
assertTrue(e.getMessage().contains("kept losing its claim"), e.getMessage());
|
||||
assertEquals(4, countCalls("flush_lsm"), "the initial attempt plus MAX_REISSUES");
|
||||
}
|
||||
|
||||
// ===========================================================================
|
||||
// strict decoding
|
||||
// ===========================================================================
|
||||
|
||||
/**
|
||||
* A stats payload that does not decode must fail closed. Every one of these bodies used to be
|
||||
* read as "no buckets", which is indistinguishable from a drained table, so {@code checkpointLsm}
|
||||
* reported convergence for a checkpoint that never ran.
|
||||
*/
|
||||
@Test
|
||||
public void testCheckpointRejectsMalformedStats() {
|
||||
Map<String, String> malformed = new LinkedHashMap<String, String>();
|
||||
malformed.put("no response body at all", "");
|
||||
malformed.put("stats object with no buckets", "{\"lsm_stats\":{}}");
|
||||
malformed.put("bucket missing its required fields", "{\"lsm_stats\":{\"buckets\":[{}]}}");
|
||||
malformed.put(
|
||||
"bucket missing generations",
|
||||
"{\"lsm_stats\":{\"buckets\":[{\"shard_id\":\"shard-0\",\"status\":\"Active\","
|
||||
+ "\"writer_epoch\":1,\"manifest_version\":2,\"current_generation\":9,"
|
||||
+ "\"replay_after_wal_entry_position\":0,\"wal_entry_position_last_seen\":0,"
|
||||
+ "\"compacting\":false}]}}");
|
||||
malformed.put(
|
||||
"generation with a non-numeric generation number",
|
||||
"{\"lsm_stats\":{\"buckets\":[{\"shard_id\":\"shard-0\",\"status\":\"Active\","
|
||||
+ "\"writer_epoch\":1,\"manifest_version\":2,\"current_generation\":9,"
|
||||
+ "\"replay_after_wal_entry_position\":0,\"wal_entry_position_last_seen\":0,"
|
||||
+ "\"generations\":[{\"generation\":\"7\",\"bytes\":1024}],"
|
||||
+ "\"compacting\":false}]}}");
|
||||
|
||||
for (Map.Entry<String, String> each : malformed.entrySet()) {
|
||||
setUpFresh();
|
||||
enqueue("flush_lsm", 200, "");
|
||||
enqueue("get_lsm_stats", 200, each.getValue());
|
||||
|
||||
assertThrows(
|
||||
IllegalStateException.class,
|
||||
() -> lsm.checkpointLsm(),
|
||||
each.getKey() + " must not report convergence");
|
||||
}
|
||||
}
|
||||
|
||||
/** The one shape that legitimately means "this table has no LSM write path". */
|
||||
@Test
|
||||
public void testCheckpointTreatsNullStatsAsNotWalBacked() {
|
||||
enqueue("flush_lsm", 200, "");
|
||||
enqueue("get_lsm_stats", 200, "{\"lsm_stats\":null}");
|
||||
|
||||
lsm.checkpointLsm();
|
||||
|
||||
assertEquals(1, countCalls("get_lsm_stats"));
|
||||
}
|
||||
|
||||
// ===========================================================================
|
||||
// retry budget
|
||||
// ===========================================================================
|
||||
|
||||
/**
|
||||
* The transport must not retry on the checkpoint loop's behalf. Apache HttpClient's default
|
||||
* strategy retries exactly 429 and 503 — the two statuses {@code isRetryable} owns — which
|
||||
* doubled every budget here and also retried {@code compact_lsm} in place, where the loop is
|
||||
* built to fall through to a fresh stats poll instead.
|
||||
*/
|
||||
@Test
|
||||
public void testCheckpointRetryBudgetIsNotDoubledByTheTransport() {
|
||||
enqueue("flush_lsm", 429, "latch held");
|
||||
|
||||
LanceDbRestClient.HttpException e =
|
||||
assertThrows(LanceDbRestClient.HttpException.class, () -> lsm.checkpointLsm());
|
||||
|
||||
assertEquals(429, e.statusCode(), "the exhausted budget propagates the last error as itself");
|
||||
assertEquals(9, countCalls("flush_lsm"), "the initial request plus MAX_RETRIES, and no more");
|
||||
}
|
||||
|
||||
// ===========================================================================
|
||||
// harness
|
||||
// ===========================================================================
|
||||
|
||||
private static List<Long> generationNumbers(BucketStats bucket) {
|
||||
List<Long> numbers = new ArrayList<Long>();
|
||||
for (GenerationStats generation : bucket.generations()) {
|
||||
numbers.add(generation.generation());
|
||||
}
|
||||
return numbers;
|
||||
}
|
||||
|
||||
/** Build an {@code lsm_stats} response body from bucket fragments. */
|
||||
private static String stats(String... buckets) {
|
||||
return "{\"lsm_stats\":{\"buckets\":[" + String.join(",", buckets) + "]}}";
|
||||
}
|
||||
|
||||
private static String bucket(String shardId, boolean compacting, Long... generations) {
|
||||
StringBuilder gens = new StringBuilder();
|
||||
for (Long generation : generations) {
|
||||
if (gens.length() > 0) {
|
||||
gens.append(",");
|
||||
}
|
||||
gens.append("{\"generation\":").append(generation).append(",\"bytes\":1024}");
|
||||
}
|
||||
return "{\"shard_id\":\""
|
||||
+ shardId
|
||||
+ "\",\"status\":\"Active\",\"writer_epoch\":1,\"manifest_version\":2,"
|
||||
+ "\"current_generation\":9,\"replay_after_wal_entry_position\":0,"
|
||||
+ "\"wal_entry_position_last_seen\":0,\"generations\":["
|
||||
+ gens
|
||||
+ "],\"compacting\":"
|
||||
+ compacting
|
||||
+ "}";
|
||||
}
|
||||
|
||||
/** Queue a reply for an operation. The last queued reply repeats once the queue drains. */
|
||||
private void enqueue(String operation, int status, String body) {
|
||||
replies.computeIfAbsent(operation, key -> new ArrayDeque<Reply>()).add(new Reply(status, body));
|
||||
}
|
||||
|
||||
private Reply nextReply(String path) {
|
||||
String operation = operationOf(path);
|
||||
Deque<Reply> queued = replies.get(operation);
|
||||
if (queued == null || queued.isEmpty()) {
|
||||
return new Reply(200, "");
|
||||
}
|
||||
return queued.size() > 1 ? queued.poll() : queued.peek();
|
||||
}
|
||||
|
||||
private long countCalls(String operation) {
|
||||
return requestPaths.stream().filter(path -> operationOf(path).equals(operation)).count();
|
||||
}
|
||||
|
||||
/** {@code /v1/table/my_table/flush_lsm/} -> {@code flush_lsm}. */
|
||||
private static String operationOf(String path) {
|
||||
String[] segments = path.split("/");
|
||||
return segments.length == 0 ? "" : segments[segments.length - 1];
|
||||
}
|
||||
|
||||
private static String readAll(InputStream in) throws IOException {
|
||||
ByteArrayOutputStream out = new ByteArrayOutputStream();
|
||||
byte[] buffer = new byte[4096];
|
||||
int read;
|
||||
while ((read = in.read(buffer)) != -1) {
|
||||
out.write(buffer, 0, read);
|
||||
}
|
||||
return new String(out.toByteArray(), StandardCharsets.UTF_8);
|
||||
}
|
||||
|
||||
private static final class Reply {
|
||||
private final int status;
|
||||
private final String body;
|
||||
|
||||
private Reply(int status, String body) {
|
||||
this.status = status;
|
||||
this.body = body;
|
||||
}
|
||||
}
|
||||
}
|
||||
+2
-2
@@ -6,7 +6,7 @@
|
||||
|
||||
<groupId>com.lancedb</groupId>
|
||||
<artifactId>lancedb-parent</artifactId>
|
||||
<version>0.38.0-beta.0</version>
|
||||
<version>0.38.0-beta.2</version>
|
||||
<packaging>pom</packaging>
|
||||
<name>${project.artifactId}</name>
|
||||
<description>LanceDB Java SDK Parent POM</description>
|
||||
@@ -28,7 +28,7 @@
|
||||
<properties>
|
||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||
<arrow.version>15.0.0</arrow.version>
|
||||
<lance-core.version>11.0.0-beta.11</lance-core.version>
|
||||
<lance-core.version>11.0.0-beta.15</lance-core.version>
|
||||
<spotless.skip>false</spotless.skip>
|
||||
<spotless.version>2.30.0</spotless.version>
|
||||
<spotless.java.googlejavaformat.version>1.7</spotless.java.googlejavaformat.version>
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
[package]
|
||||
name = "lancedb-nodejs"
|
||||
edition.workspace = true
|
||||
version = "0.38.0-beta.0"
|
||||
version = "0.38.0-beta.2"
|
||||
publish = false
|
||||
license.workspace = true
|
||||
description.workspace = true
|
||||
|
||||
@@ -3341,6 +3341,59 @@ describe("LSM merge insert", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("LSM convergence and stats", () => {
|
||||
let tmpDir: tmp.DirResult;
|
||||
|
||||
beforeEach(() => {
|
||||
tmpDir = tmp.dirSync({ unsafeCleanup: true });
|
||||
});
|
||||
afterEach(() => tmpDir.removeCallback());
|
||||
|
||||
async function lsmTable(conn: Connection): Promise<Table> {
|
||||
const table = await conn.createEmptyTable(
|
||||
"t",
|
||||
new arrow.Schema([new arrow.Field("id", new arrow.Utf8(), false)]),
|
||||
);
|
||||
await table.setUnenforcedPrimaryKey("id");
|
||||
await table.setLsmWriteSpec({ specType: "unsharded" });
|
||||
return table;
|
||||
}
|
||||
|
||||
// These four route through the server that owns the MemWAL, so a local table
|
||||
// rejects them rather than answering. What is asserted here is that the
|
||||
// bindings reach the core at all; the behavior against a real endpoint is
|
||||
// covered by the mocked endpoint tests in rust/lancedb/src/remote/table.rs.
|
||||
it("rejects flushLsm on a local table", async () => {
|
||||
const conn = await connect(tmpDir.name);
|
||||
const table = await lsmTable(conn);
|
||||
|
||||
await expect(table.flushLsm()).rejects.toThrow(/not supported/i);
|
||||
});
|
||||
|
||||
it("rejects compactLsm on a local table", async () => {
|
||||
const conn = await connect(tmpDir.name);
|
||||
const table = await lsmTable(conn);
|
||||
|
||||
await expect(table.compactLsm()).rejects.toThrow(/not supported/i);
|
||||
});
|
||||
|
||||
it("rejects getLsmStats on a local table", async () => {
|
||||
const conn = await connect(tmpDir.name);
|
||||
const table = await lsmTable(conn);
|
||||
|
||||
await expect(table.getLsmStats()).rejects.toThrow(/not supported/i);
|
||||
await expect(table.getLsmStats(true)).rejects.toThrow(/not supported/i);
|
||||
});
|
||||
|
||||
it("rejects checkpointLsm on a local table", async () => {
|
||||
const conn = await connect(tmpDir.name);
|
||||
const table = await lsmTable(conn);
|
||||
|
||||
// checkpointLsm seals first, so it surfaces flushLsm's rejection.
|
||||
await expect(table.checkpointLsm()).rejects.toThrow(/not supported/i);
|
||||
});
|
||||
});
|
||||
|
||||
describe("computed columns", () => {
|
||||
let tmpDir: tmp.DirResult;
|
||||
beforeEach(() => {
|
||||
@@ -3365,6 +3418,28 @@ describe("computed columns", () => {
|
||||
expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]);
|
||||
});
|
||||
|
||||
it("returns a job handle from refreshColumnAsync", async () => {
|
||||
const db = await connect(tmpDir.name);
|
||||
const table = await db.createTable("computed_job", [{ x: 1 }, { x: 2 }]);
|
||||
|
||||
await table.addColumns({
|
||||
computed: [{ name: "doubled", valueSql: "x * 2" }],
|
||||
});
|
||||
|
||||
const job = await table.refreshColumnAsync("doubled");
|
||||
expect(job.id).toBeNull();
|
||||
await job.wait();
|
||||
expect(await job.status()).toBe("finished");
|
||||
|
||||
const rows = await table.query().toArray();
|
||||
expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]);
|
||||
|
||||
// Bad input rejects at the call, not through the job.
|
||||
await expect(table.refreshColumnAsync("x")).rejects.toThrow(
|
||||
"not a computed column",
|
||||
);
|
||||
});
|
||||
|
||||
it("fills rows added since the last refresh", async () => {
|
||||
const db = await connect(tmpDir.name);
|
||||
const table = await db.createTable("computed_append", [{ x: 1 }]);
|
||||
|
||||
@@ -147,6 +147,10 @@ export {
|
||||
FtsToken,
|
||||
TokenizeTableOptions,
|
||||
LsmWriteSpec,
|
||||
LsmStats,
|
||||
BucketStats,
|
||||
GenerationStats,
|
||||
MemtableStats,
|
||||
ColumnAlteration,
|
||||
FieldMetadataUpdate,
|
||||
} from "./table";
|
||||
|
||||
+106
-3
@@ -31,6 +31,7 @@ import {
|
||||
IndexConfig,
|
||||
IndexStatistics,
|
||||
Job,
|
||||
LsmStats,
|
||||
Branches as NativeBranches,
|
||||
OptimizeStats,
|
||||
RefreshColumnResult,
|
||||
@@ -50,6 +51,12 @@ import {
|
||||
import { sanitizeType } from "./sanitize";
|
||||
import { IntoSql, toSQL } from "./util";
|
||||
export { IndexConfig } from "./native";
|
||||
export {
|
||||
BucketStats,
|
||||
GenerationStats,
|
||||
LsmStats,
|
||||
MemtableStats,
|
||||
} from "./native";
|
||||
|
||||
/**
|
||||
* Progress snapshot for a write operation, delivered to the `progress`
|
||||
@@ -537,8 +544,9 @@ export abstract class Table {
|
||||
* the column and declaring it again. While a declaration reads a column,
|
||||
* that column cannot be renamed, retyped or dropped.
|
||||
*
|
||||
* Computed columns are local-only: LanceDB Cloud and Enterprise reject a
|
||||
* declaration.
|
||||
* On LanceDB Cloud and Enterprise the expression is planned by the
|
||||
* server, and the refresh runs as a server job -- see
|
||||
* {@link Table#refreshColumnAsync}.
|
||||
* @param {AddColumnsSql[] | Field | Field[] | Schema} newColumnTransforms Either:
|
||||
* - An array of objects with column names and SQL expressions to calculate values
|
||||
* - A single Arrow Field defining one column with its data type (column will be initialized with null values)
|
||||
@@ -567,13 +575,33 @@ export abstract class Table {
|
||||
*
|
||||
* Rows appended since the last refresh are filled by the next one; rows
|
||||
* already filled are left as they are, so the call is idempotent and does
|
||||
* not observe a mutated input. Local tables only.
|
||||
* not observe a mutated input. Local tables only: a remote refresh runs
|
||||
* as a server job, through {@link Table#refreshColumnAsync}.
|
||||
* @param {string} column The name of the computed column to fill.
|
||||
* @returns {Promise<RefreshColumnResult>} A promise that resolves to the
|
||||
* number of rows filled and the new version number of the table.
|
||||
*/
|
||||
abstract refreshColumn(column: string): Promise<RefreshColumnResult>;
|
||||
|
||||
/**
|
||||
* Like {@link Table#refreshColumn}, but returns a handle to the refresh
|
||||
* job instead of blocking until it completes.
|
||||
*
|
||||
* The job may already be complete when returned; callers must not assume
|
||||
* the column is filled until {@link Job.wait} resolves. Invalid input --
|
||||
* an unknown column, or one that is not computed -- rejects here rather
|
||||
* than failing the job. On local tables the job runs in-process; on
|
||||
* LanceDB Cloud and Enterprise it is the server's backfill job.
|
||||
* @param {string} column The name of the computed column to fill.
|
||||
* @example
|
||||
* ```ts
|
||||
* const job = await table.refreshColumnAsync("doubled");
|
||||
* await job.wait();
|
||||
* console.log(await job.status()); // "finished"
|
||||
* ```
|
||||
*/
|
||||
abstract refreshColumnAsync(column: string): Promise<Job>;
|
||||
|
||||
/**
|
||||
* Alter the name or nullability of columns.
|
||||
* @param {ColumnAlteration[]} columnAlterations One or more alterations to
|
||||
@@ -685,6 +713,59 @@ export abstract class Table {
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
abstract closeLsmWriters(): Promise<void>;
|
||||
/**
|
||||
* Seal every bucket's active memtable into a new L0 generation.
|
||||
*
|
||||
* Returns once the seal is committed. Sealing an empty memtable is a no-op,
|
||||
* so this is safe to call repeatedly.
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
abstract flushLsm(): Promise<void>;
|
||||
/**
|
||||
* Trigger a background L0 → base compaction pass per bucket.
|
||||
*
|
||||
* Returns once the passes are *dispatched*, not once they finish — watch
|
||||
* {@link Table#getLsmStats} for progress, or use
|
||||
* {@link Table#checkpointLsm} to wait for convergence.
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
abstract compactLsm(): Promise<void>;
|
||||
/**
|
||||
* Converge this table's LSM write path into its base table.
|
||||
*
|
||||
* Seals once, then triggers compaction and polls until the L0 that existed
|
||||
* at the start is gone. The target set is fixed at the start, so
|
||||
* generations created *during* the checkpoint are ignored — that is what
|
||||
* lets it terminate under write load, and what makes it best-effort: it
|
||||
* converges the fresh tier as of some instant. Idempotent, abandonable at
|
||||
* any point, and safe to run on a cadence.
|
||||
*
|
||||
* There is no liveness bound — the compactor pool is shared across tables,
|
||||
* so a checkpoint queued behind unrelated work looks exactly like one that
|
||||
* is merging. The caller owns the deadline.
|
||||
* @returns {Promise<void>}
|
||||
* @example
|
||||
* ```ts
|
||||
* const before = await table.getLsmStats();
|
||||
* await table.checkpointLsm();
|
||||
* const after = await table.getLsmStats();
|
||||
* ```
|
||||
*/
|
||||
abstract checkpointLsm(): Promise<void>;
|
||||
/**
|
||||
* Read live per-bucket LSM state.
|
||||
*
|
||||
* Answers "how far behind is my fresh tier", "which bucket is hot", and
|
||||
* "why is my fresh-tier vector search brute-force". Mutates no table state.
|
||||
*
|
||||
* Resolves to `undefined` only when the LSM write path is not enabled.
|
||||
* @param {boolean} includeGenerationRows Also count rows per L0 generation.
|
||||
* Off by default because each count opens an uncached Lance dataset.
|
||||
* @returns {Promise<LsmStats | undefined>}
|
||||
*/
|
||||
abstract getLsmStats(
|
||||
includeGenerationRows?: boolean,
|
||||
): Promise<LsmStats | undefined>;
|
||||
/** Retrieve the version of the table */
|
||||
|
||||
abstract version(): Promise<number>;
|
||||
@@ -1179,6 +1260,10 @@ export class LocalTable extends Table {
|
||||
return await this.inner.refreshColumn(column);
|
||||
}
|
||||
|
||||
async refreshColumnAsync(column: string): Promise<Job> {
|
||||
return await this.inner.refreshColumnAsync(column);
|
||||
}
|
||||
|
||||
async alterColumns(
|
||||
columnAlterations: ColumnAlteration[],
|
||||
): Promise<AlterColumnsResult> {
|
||||
@@ -1241,6 +1326,24 @@ export class LocalTable extends Table {
|
||||
return await this.inner.closeLsmWriters();
|
||||
}
|
||||
|
||||
async flushLsm(): Promise<void> {
|
||||
return await this.inner.flushLsm();
|
||||
}
|
||||
|
||||
async compactLsm(): Promise<void> {
|
||||
return await this.inner.compactLsm();
|
||||
}
|
||||
|
||||
async checkpointLsm(): Promise<void> {
|
||||
return await this.inner.checkpointLsm();
|
||||
}
|
||||
|
||||
async getLsmStats(
|
||||
includeGenerationRows: boolean = false,
|
||||
): Promise<LsmStats | undefined> {
|
||||
return (await this.inner.getLsmStats(includeGenerationRows)) ?? undefined;
|
||||
}
|
||||
|
||||
async version(): Promise<number> {
|
||||
return await this.inner.version();
|
||||
}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-darwin-arm64",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.38.0-beta.2",
|
||||
"os": ["darwin"],
|
||||
"cpu": ["arm64"],
|
||||
"main": "lancedb.darwin-arm64.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-arm64-gnu",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.38.0-beta.2",
|
||||
"os": ["linux"],
|
||||
"cpu": ["arm64"],
|
||||
"main": "lancedb.linux-arm64-gnu.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-arm64-musl",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.38.0-beta.2",
|
||||
"os": ["linux"],
|
||||
"cpu": ["arm64"],
|
||||
"main": "lancedb.linux-arm64-musl.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-x64-gnu",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.38.0-beta.2",
|
||||
"os": ["linux"],
|
||||
"cpu": ["x64"],
|
||||
"main": "lancedb.linux-x64-gnu.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-linux-x64-musl",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.38.0-beta.2",
|
||||
"os": ["linux"],
|
||||
"cpu": ["x64"],
|
||||
"main": "lancedb.linux-x64-musl.node",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-win32-arm64-msvc",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.38.0-beta.2",
|
||||
"os": [
|
||||
"win32"
|
||||
],
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb-win32-x64-msvc",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.38.0-beta.2",
|
||||
"os": ["win32"],
|
||||
"cpu": ["x64"],
|
||||
"main": "lancedb.win32-x64-msvc.node",
|
||||
|
||||
Generated
+2
-2
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "@lancedb/lancedb",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.38.0-beta.2",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "@lancedb/lancedb",
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.38.0-beta.2",
|
||||
"cpu": [
|
||||
"x64",
|
||||
"arm64"
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@
|
||||
"ann"
|
||||
],
|
||||
"private": false,
|
||||
"version": "0.38.0-beta.0",
|
||||
"version": "0.38.0-beta.2",
|
||||
"main": "dist/index.js",
|
||||
"exports": {
|
||||
".": "./dist/index.js",
|
||||
|
||||
@@ -371,6 +371,16 @@ impl Table {
|
||||
Ok(res.into())
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn refresh_column_async(&self, column: String) -> napi::Result<crate::job::Job> {
|
||||
let job = self
|
||||
.inner_ref()?
|
||||
.refresh_column_async(column)
|
||||
.await
|
||||
.default_error()?;
|
||||
Ok(crate::job::Job::new(job))
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn add_columns_with_schema(
|
||||
&self,
|
||||
@@ -487,6 +497,34 @@ impl Table {
|
||||
self.inner_ref()?.close_lsm_writers().await.default_error()
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn flush_lsm(&self) -> napi::Result<()> {
|
||||
self.inner_ref()?.flush_lsm().await.default_error()
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn compact_lsm(&self) -> napi::Result<()> {
|
||||
self.inner_ref()?.compact_lsm().await.default_error()
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn checkpoint_lsm(&self) -> napi::Result<()> {
|
||||
self.inner_ref()?.checkpoint_lsm().await.default_error()
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn get_lsm_stats(
|
||||
&self,
|
||||
include_generation_rows: bool,
|
||||
) -> napi::Result<Option<LsmStats>> {
|
||||
let stats = self
|
||||
.inner_ref()?
|
||||
.get_lsm_stats(include_generation_rows)
|
||||
.await
|
||||
.default_error()?;
|
||||
Ok(stats.map(LsmStats::from))
|
||||
}
|
||||
|
||||
#[napi(catch_unwind)]
|
||||
pub async fn version(&self) -> napi::Result<i64> {
|
||||
self.inner_ref()?
|
||||
@@ -879,6 +917,129 @@ impl From<lancedb::table::LsmWriteSpec> for LsmWriteSpec {
|
||||
}
|
||||
}
|
||||
|
||||
/// One flushed L0 generation.
|
||||
#[napi(object)]
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct GenerationStats {
|
||||
/// The generation number. Increases as memtables are sealed into L0.
|
||||
pub generation: i64,
|
||||
/// On-disk size of the generation.
|
||||
pub bytes: i64,
|
||||
/// Present only when `includeGenerationRows` was requested. Off by default
|
||||
/// because each count opens an uncached Lance dataset.
|
||||
pub rows: Option<i64>,
|
||||
}
|
||||
|
||||
impl From<lancedb::table::GenerationStats> for GenerationStats {
|
||||
fn from(g: lancedb::table::GenerationStats) -> Self {
|
||||
Self {
|
||||
generation: g.generation as i64,
|
||||
bytes: g.bytes as i64,
|
||||
rows: g.rows.map(|r| r as i64),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// One in-memory memtable.
|
||||
#[napi(object)]
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct MemtableStats {
|
||||
/// The generation this memtable will become once sealed.
|
||||
pub generation: i64,
|
||||
/// Rows currently buffered.
|
||||
pub rows: i64,
|
||||
/// Estimated in-memory size.
|
||||
pub bytes: i64,
|
||||
/// Record batches currently buffered.
|
||||
pub batches: i64,
|
||||
/// Names of the indexes this memtable carries. An absent name is the whole
|
||||
/// answer to "why is my fresh-tier search on that column brute-force".
|
||||
pub indexes: Vec<String>,
|
||||
}
|
||||
|
||||
impl From<lancedb::table::MemtableStats> for MemtableStats {
|
||||
fn from(m: lancedb::table::MemtableStats) -> Self {
|
||||
Self {
|
||||
generation: m.generation as i64,
|
||||
rows: m.rows as i64,
|
||||
bytes: m.bytes as i64,
|
||||
batches: m.batches as i64,
|
||||
indexes: m.indexes,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Live state of one bucket. A table is N buckets on one node; flattening to a
|
||||
/// single number hides the one hot bucket that is usually why someone opened
|
||||
/// this endpoint.
|
||||
#[napi(object)]
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct BucketStats {
|
||||
/// The shard this bucket writes.
|
||||
pub shard_id: String,
|
||||
/// `"Active"` or `"Sealed"` (drop-table 2PC in flight).
|
||||
pub status: String,
|
||||
/// Epoch of the writer that currently owns the shard.
|
||||
pub writer_epoch: i64,
|
||||
/// Version of the shard manifest these numbers were read from.
|
||||
pub manifest_version: i64,
|
||||
/// The generation the active memtable will become.
|
||||
pub current_generation: i64,
|
||||
/// WAL position replay resumes from.
|
||||
pub replay_after_wal_entry_position: i64,
|
||||
/// Highest WAL position the writer has seen. The difference against
|
||||
/// `replayAfterWalEntryPosition` is the WAL lag.
|
||||
pub wal_entry_position_last_seen: i64,
|
||||
/// Flushed L0 generations not yet merged into the base table.
|
||||
pub generations: Vec<GenerationStats>,
|
||||
/// Whether a pass owns this bucket's compaction latch right now. Says *a*
|
||||
/// driver is running, not *whose*, and the latch is held from dispatch —
|
||||
/// including while the pass queues for a pod-wide compactor permit. Read it
|
||||
/// as "do not pile on", never as "mine is progressing".
|
||||
pub compacting: bool,
|
||||
/// Oldest first, active last. Absent for a `"Sealed"` bucket, whose
|
||||
/// in-memory state is torn down.
|
||||
pub memtables: Option<Vec<MemtableStats>>,
|
||||
}
|
||||
|
||||
impl From<lancedb::table::BucketStats> for BucketStats {
|
||||
fn from(b: lancedb::table::BucketStats) -> Self {
|
||||
Self {
|
||||
shard_id: b.shard_id,
|
||||
status: b.status,
|
||||
writer_epoch: b.writer_epoch as i64,
|
||||
manifest_version: b.manifest_version as i64,
|
||||
current_generation: b.current_generation as i64,
|
||||
replay_after_wal_entry_position: b.replay_after_wal_entry_position as i64,
|
||||
wal_entry_position_last_seen: b.wal_entry_position_last_seen as i64,
|
||||
generations: b.generations.into_iter().map(Into::into).collect(),
|
||||
compacting: b.compacting,
|
||||
memtables: b
|
||||
.memtables
|
||||
.map(|ms| ms.into_iter().map(Into::into).collect()),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Live per-bucket LSM state, as returned by `Table#getLsmStats`.
|
||||
///
|
||||
/// Nothing here is derived: sums and differences (total L0 bytes, WAL lag) are
|
||||
/// the caller's to compute.
|
||||
#[napi(object)]
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct LsmStats {
|
||||
/// One entry per bucket backing this table.
|
||||
pub buckets: Vec<BucketStats>,
|
||||
}
|
||||
|
||||
impl From<lancedb::table::LsmStats> for LsmStats {
|
||||
fn from(stats: lancedb::table::LsmStats) -> Self {
|
||||
Self {
|
||||
buckets: stats.buckets.into_iter().map(Into::into).collect(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Statistics about a compaction operation.
|
||||
#[napi(object)]
|
||||
#[derive(Clone, Debug)]
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "lancedb-python"
|
||||
version = "0.38.0-beta.0"
|
||||
version = "0.38.0-beta.2"
|
||||
publish = false
|
||||
edition.workspace = true
|
||||
description = "Python bindings for LanceDB"
|
||||
|
||||
@@ -12,6 +12,7 @@ __version__ = importlib.metadata.version("lancedb")
|
||||
|
||||
from ._lancedb import connect as lancedb_connect
|
||||
from ._lancedb import FtsToken
|
||||
from ._lancedb import LsmWriteSpec
|
||||
from ._lancedb import tokenize as _tokenize
|
||||
from .common import URI, sanitize_uri
|
||||
from urllib.parse import urlparse
|
||||
@@ -519,6 +520,7 @@ __all__ = [
|
||||
"Job",
|
||||
"LanceDBConnection",
|
||||
"LanceNamespaceDBConnection",
|
||||
"LsmWriteSpec",
|
||||
"RemoteDBConnection",
|
||||
"Session",
|
||||
"Table",
|
||||
|
||||
@@ -342,6 +342,7 @@ class Table:
|
||||
self, columns: list[tuple[str, str]]
|
||||
) -> AddColumnsResult: ...
|
||||
async def refresh_column(self, column: str) -> RefreshColumnResult: ...
|
||||
async def refresh_column_async(self, column: str) -> Job: ...
|
||||
async def add_columns_with_schema(self, schema: pa.Schema) -> AddColumnsResult: ...
|
||||
async def alter_columns(
|
||||
self, columns: list[dict[str, Any]]
|
||||
|
||||
@@ -2235,6 +2235,7 @@ class LanceHybridQueryBuilder(LanceQueryBuilder):
|
||||
reranker=self._reranker,
|
||||
limit=self._limit,
|
||||
with_row_ids=True,
|
||||
offset=self._offset,
|
||||
)
|
||||
return self._finish_hybrid_results(results)
|
||||
|
||||
@@ -2256,6 +2257,7 @@ class LanceHybridQueryBuilder(LanceQueryBuilder):
|
||||
reranker,
|
||||
limit: int,
|
||||
with_row_ids: bool,
|
||||
offset: Optional[int] = None,
|
||||
) -> pa.Table:
|
||||
if norm == "rank":
|
||||
vector_results = LanceHybridQueryBuilder._rank(vector_results, "_distance")
|
||||
@@ -2332,7 +2334,7 @@ class LanceHybridQueryBuilder(LanceQueryBuilder):
|
||||
score_i = results.column_names.index("_score")
|
||||
results = results.set_column(score_i, "_score", original_scores)
|
||||
|
||||
results = results.slice(length=limit)
|
||||
results = results.slice(offset=offset or 0, length=limit)
|
||||
|
||||
if not with_row_ids:
|
||||
results = results.drop(["_rowid"])
|
||||
@@ -2679,8 +2681,12 @@ class LanceHybridQueryBuilder(LanceQueryBuilder):
|
||||
|
||||
# Apply common configurations
|
||||
if self._limit:
|
||||
self._vector_query.limit(self._limit)
|
||||
self._fts_query.limit(self._limit)
|
||||
# The final offset/limit window is sliced out of the combined,
|
||||
# reranked results, so each sub-query must fetch enough rows to
|
||||
# cover the skipped prefix as well as the window itself.
|
||||
sub_query_limit = self._limit + (self._offset or 0)
|
||||
self._vector_query.limit(sub_query_limit)
|
||||
self._fts_query.limit(sub_query_limit)
|
||||
if self._columns:
|
||||
self._vector_query.select(self._columns)
|
||||
self._fts_query.select(self._columns)
|
||||
|
||||
@@ -974,14 +974,13 @@ class RemoteTable(Table):
|
||||
*,
|
||||
computed: Dict[str, str] | None = None,
|
||||
) -> AddColumnsResult:
|
||||
if computed:
|
||||
raise NotImplementedError(
|
||||
"computed columns are supported only on local tables"
|
||||
)
|
||||
return LOOP.run(self._table.add_columns(transforms))
|
||||
return LOOP.run(self._table.add_columns(transforms, computed=computed))
|
||||
|
||||
def refresh_column(self, column: str):
|
||||
raise NotImplementedError("computed columns are supported only on local tables")
|
||||
return LOOP.run(self._table.refresh_column(column))
|
||||
|
||||
def refresh_column_async(self, column: str) -> Job:
|
||||
return Job(LOOP.run(self._table.refresh_column_async(column)))
|
||||
|
||||
def alter_columns(
|
||||
self, *alterations: Iterable[Dict[str, str]]
|
||||
@@ -1001,17 +1000,39 @@ class RemoteTable(Table):
|
||||
return LOOP.run(self._table.set_unenforced_primary_key(columns))
|
||||
|
||||
def set_lsm_write_spec(self, spec: "LsmWriteSpec") -> None:
|
||||
"""Not supported on LanceDB Cloud."""
|
||||
"""Install an LsmWriteSpec."""
|
||||
return LOOP.run(self._table.set_lsm_write_spec(spec))
|
||||
|
||||
def unset_lsm_write_spec(self) -> None:
|
||||
"""Not supported on LanceDB Cloud."""
|
||||
"""Remove the LsmWriteSpec."""
|
||||
return LOOP.run(self._table.unset_lsm_write_spec())
|
||||
|
||||
def get_lsm_write_spec(self) -> Optional["LsmWriteSpec"]:
|
||||
"""Read the installed LsmWriteSpec, or ``None``."""
|
||||
return LOOP.run(self._table.get_lsm_write_spec())
|
||||
|
||||
def checkpoint_lsm(self) -> None:
|
||||
"""Synchronous version of
|
||||
[`AsyncTable.checkpoint_lsm`][lancedb.AsyncTable.checkpoint_lsm]."""
|
||||
return LOOP.run(self._table.checkpoint_lsm())
|
||||
|
||||
def flush_lsm(self) -> None:
|
||||
"""Synchronous version of
|
||||
[`AsyncTable.flush_lsm`][lancedb.AsyncTable.flush_lsm]."""
|
||||
return LOOP.run(self._table.flush_lsm())
|
||||
|
||||
def compact_lsm(self) -> None:
|
||||
"""Synchronous version of
|
||||
[`AsyncTable.compact_lsm`][lancedb.AsyncTable.compact_lsm]."""
|
||||
return LOOP.run(self._table.compact_lsm())
|
||||
|
||||
def get_lsm_stats(self, *, include_generation_rows: bool = False) -> Optional[dict]:
|
||||
"""Synchronous version of
|
||||
[`AsyncTable.get_lsm_stats`][lancedb.AsyncTable.get_lsm_stats]."""
|
||||
return LOOP.run(
|
||||
self._table.get_lsm_stats(include_generation_rows=include_generation_rows)
|
||||
)
|
||||
|
||||
def close_lsm_writers(self) -> None:
|
||||
"""No-op on LanceDB Cloud (no local shard writers)."""
|
||||
return LOOP.run(self._table.close_lsm_writers())
|
||||
|
||||
@@ -2029,8 +2029,10 @@ class Table(ABC):
|
||||
dropping the column and declaring it again. While a declaration
|
||||
reads a column, that column cannot be renamed, retyped or dropped.
|
||||
|
||||
Local tables only; LanceDB Cloud and Enterprise raise
|
||||
``NotImplementedError``. Cannot be combined with ``transforms``.
|
||||
On LanceDB Cloud and Enterprise the expression is planned by the
|
||||
server, and the refresh runs as a server job -- see
|
||||
[`refresh_column_async`][lancedb.table.Table.refresh_column_async].
|
||||
Cannot be combined with ``transforms``.
|
||||
|
||||
Returns
|
||||
-------
|
||||
@@ -2062,8 +2064,8 @@ class Table(ABC):
|
||||
by the next one; rows already filled are left as they are, so the call
|
||||
is idempotent and does not observe a mutated input.
|
||||
|
||||
Local tables only; LanceDB Cloud and Enterprise raise
|
||||
``NotImplementedError``.
|
||||
Local tables only: a remote refresh runs as a server job, through
|
||||
[`refresh_column_async`][lancedb.table.Table.refresh_column_async].
|
||||
|
||||
Parameters
|
||||
----------
|
||||
@@ -2077,6 +2079,31 @@ class Table(ABC):
|
||||
version: the new version number of the table.
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
def refresh_column_async(self, column: str) -> Job:
|
||||
"""
|
||||
Like :meth:`refresh_column`, but returns a handle to the refresh job
|
||||
instead of blocking until it completes.
|
||||
|
||||
The job may already be complete when returned; callers must not assume
|
||||
the column is filled until :meth:`Job.wait` returns. Invalid input --
|
||||
an unknown column, or one that is not computed -- raises here rather
|
||||
than failing the job. On local tables the job runs in-process; on
|
||||
LanceDB Cloud and Enterprise it is the server's backfill job.
|
||||
|
||||
Examples
|
||||
--------
|
||||
>>> import lancedb
|
||||
>>> db = lancedb.connect("./.lancedb")
|
||||
>>> table = db.create_table("computed_job_demo", [{"x": 1}, {"x": 2}])
|
||||
>>> table.add_columns(computed={"doubled": "x * 2"})
|
||||
AddColumnsResult(version=2)
|
||||
>>> job = table.refresh_column_async("doubled")
|
||||
>>> job.wait()
|
||||
>>> job.status()
|
||||
'finished'
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
def alter_columns(self, *alterations: Iterable[Dict[str, str]]):
|
||||
"""
|
||||
@@ -4101,6 +4128,13 @@ class LanceTable(Table):
|
||||
[`AsyncTable.refresh_column`][lancedb.AsyncTable.refresh_column]."""
|
||||
return LOOP.run(self._table.refresh_column(column))
|
||||
|
||||
def refresh_column_async(self, column: str) -> Job:
|
||||
"""Fill a computed column's unfilled rows, returning a handle to the
|
||||
refresh job. See
|
||||
[`Table.refresh_column_async`][lancedb.table.Table.refresh_column_async].
|
||||
"""
|
||||
return Job(LOOP.run(self._table.refresh_column_async(column)))
|
||||
|
||||
def alter_columns(
|
||||
self, *alterations: Iterable[Dict[str, str]]
|
||||
) -> AlterColumnsResult:
|
||||
@@ -4848,7 +4882,7 @@ class AsyncTable:
|
||||
|
||||
Examples
|
||||
--------
|
||||
>>> from lancedb._lancedb import LsmWriteSpec
|
||||
>>> from lancedb import LsmWriteSpec
|
||||
>>> # table.set_unenforced_primary_key("id")
|
||||
>>> # table.set_lsm_write_spec(LsmWriteSpec.bucket("id", 16))
|
||||
"""
|
||||
@@ -4893,7 +4927,7 @@ class AsyncTable:
|
||||
``asyncio.wait_for`` for a wall-clock bound; abandoning it partway
|
||||
costs nothing.
|
||||
"""
|
||||
return await self._inner.checkpoint_lsm()
|
||||
await self._inner.checkpoint_lsm()
|
||||
|
||||
async def flush_lsm(self) -> None:
|
||||
"""Seal every bucket's active memtable into L0.
|
||||
@@ -4902,7 +4936,7 @@ class AsyncTable:
|
||||
`compact_lsm`. On a node that has not claimed this table, this claims
|
||||
it and replays its WAL log first.
|
||||
"""
|
||||
return await self._inner.flush_lsm()
|
||||
await self._inner.flush_lsm()
|
||||
|
||||
async def compact_lsm(self) -> None:
|
||||
"""Trigger a background L0 to base compaction pass per bucket.
|
||||
@@ -4911,7 +4945,7 @@ class AsyncTable:
|
||||
``get_lsm_stats`` for progress, or use ``checkpoint_lsm`` to loop
|
||||
until the current L0 has reached base.
|
||||
"""
|
||||
return await self._inner.compact_lsm()
|
||||
await self._inner.compact_lsm()
|
||||
|
||||
async def get_lsm_stats(
|
||||
self, *, include_generation_rows: bool = False
|
||||
@@ -6048,7 +6082,8 @@ class AsyncTable:
|
||||
declaration reads a column, that column cannot be renamed, retyped
|
||||
or dropped.
|
||||
|
||||
Local tables only. Cannot be combined with ``transforms``.
|
||||
On LanceDB Cloud and Enterprise the expression is planned by
|
||||
the server. Cannot be combined with ``transforms``.
|
||||
|
||||
Returns
|
||||
-------
|
||||
@@ -6084,8 +6119,8 @@ class AsyncTable:
|
||||
by the next one; rows already filled are left as they are, so the call
|
||||
is idempotent and does not observe a mutated input.
|
||||
|
||||
Local tables only; LanceDB Cloud and Enterprise raise
|
||||
``NotImplementedError``.
|
||||
Local tables only: a remote refresh runs as a server job, through
|
||||
[`refresh_column_async`][lancedb.table.Table.refresh_column_async].
|
||||
|
||||
Parameters
|
||||
----------
|
||||
@@ -6099,6 +6134,34 @@ class AsyncTable:
|
||||
"""
|
||||
return await self._inner.refresh_column(column)
|
||||
|
||||
async def refresh_column_async(self, column: str) -> AsyncJob:
|
||||
"""
|
||||
Like :meth:`refresh_column`, but returns a handle to the refresh job
|
||||
instead of blocking until it completes.
|
||||
|
||||
The job may already be complete when returned; callers must not assume
|
||||
the column is filled until :meth:`AsyncJob.wait` resolves. Invalid
|
||||
input -- an unknown column, or one that is not computed -- raises here
|
||||
rather than failing the job. On local tables the job runs
|
||||
in-process; on LanceDB Cloud and Enterprise it is the server's
|
||||
backfill job.
|
||||
|
||||
Examples
|
||||
--------
|
||||
>>> import asyncio
|
||||
>>> import lancedb
|
||||
>>> async def refresh_in_background():
|
||||
... db = await lancedb.connect_async("./.lancedb")
|
||||
... table = await db.create_table("computed_job_async_demo", [{"x": 1}])
|
||||
... await table.add_columns(computed={"doubled": "x * 2"})
|
||||
... job = await table.refresh_column_async("doubled")
|
||||
... await job.wait()
|
||||
... return await job.status()
|
||||
>>> asyncio.run(refresh_in_background())
|
||||
'finished'
|
||||
"""
|
||||
return AsyncJob(await self._inner.refresh_column_async(column))
|
||||
|
||||
async def alter_columns(
|
||||
self, *alterations: Iterable[dict[str, Any]]
|
||||
) -> AlterColumnsResult:
|
||||
|
||||
@@ -632,3 +632,101 @@ class TestExprBytesIntegration:
|
||||
.to_arrow()
|
||||
)
|
||||
assert result.num_rows == 2
|
||||
|
||||
|
||||
# ── datetime / timezone integration for lit() (issue #3262) ──────────────────
|
||||
|
||||
|
||||
class TestExprDatetimeTimezoneIntegration:
|
||||
"""Integration coverage for lit(datetime) against table timestamp columns.
|
||||
|
||||
PyArrow stores naive timestamps as UTC wall-clock microseconds. Python's
|
||||
datetime.timestamp() treats naive values as *local* time, which used to
|
||||
shift lit(naive) by the host UTC offset and break equality filters on
|
||||
non-UTC machines. These cases lock the expected semantics.
|
||||
"""
|
||||
|
||||
def test_both_naive_match(self, tmp_path):
|
||||
"""Table naive + lit naive with the same wall clock must match."""
|
||||
db = lancedb.connect(str(tmp_path / "naive"))
|
||||
ts = datetime(2024, 7, 1, 10, 0, 0)
|
||||
table = db.create_table(
|
||||
"t", [{"id": 1, "ts": ts}, {"id": 2, "ts": datetime(2024, 7, 2, 10, 0, 0)}]
|
||||
)
|
||||
result = table.search().where(col("ts") == lit(ts)).to_list()
|
||||
assert len(result) == 1
|
||||
assert result[0]["id"] == 1
|
||||
|
||||
def test_both_same_timezone_match(self, tmp_path):
|
||||
"""Table UTC + lit UTC for the same instant must match."""
|
||||
db = lancedb.connect(str(tmp_path / "utc"))
|
||||
ts = datetime(2024, 7, 1, 10, 0, 0, tzinfo=timezone.utc)
|
||||
table = db.create_table(
|
||||
"t",
|
||||
pa.table(
|
||||
{
|
||||
"id": [1, 2],
|
||||
"ts": pa.array(
|
||||
[ts, datetime(2024, 7, 2, 10, 0, 0, tzinfo=timezone.utc)],
|
||||
type=pa.timestamp("us", tz="UTC"),
|
||||
),
|
||||
}
|
||||
),
|
||||
)
|
||||
result = table.search().where(col("ts") == lit(ts)).to_list()
|
||||
assert len(result) == 1
|
||||
assert result[0]["id"] == 1
|
||||
|
||||
def test_different_timezones_same_instant(self, tmp_path):
|
||||
"""UTC table row equals lit of the same instant in a different zone."""
|
||||
db = lancedb.connect(str(tmp_path / "diff_tz"))
|
||||
ts_utc = datetime(2024, 7, 1, 10, 0, 0, tzinfo=timezone.utc)
|
||||
# Same instant as 06:00 in UTC-4
|
||||
ts_est = datetime(2024, 7, 1, 6, 0, 0, tzinfo=timezone(timedelta(hours=-4)))
|
||||
table = db.create_table(
|
||||
"t",
|
||||
pa.table(
|
||||
{
|
||||
"id": [1],
|
||||
"ts": pa.array([ts_utc], type=pa.timestamp("us", tz="UTC")),
|
||||
}
|
||||
),
|
||||
)
|
||||
result = table.search().where(col("ts") == lit(ts_est)).to_list()
|
||||
assert len(result) == 1
|
||||
assert result[0]["id"] == 1
|
||||
|
||||
def test_table_tz_literal_naive(self, tmp_path):
|
||||
"""UTC table + naive lit uses wall-clock equality (10:00 == 10:00 UTC)."""
|
||||
db = lancedb.connect(str(tmp_path / "tz_naive"))
|
||||
ts_utc = datetime(2024, 7, 1, 10, 0, 0, tzinfo=timezone.utc)
|
||||
ts_naive = datetime(2024, 7, 1, 10, 0, 0)
|
||||
table = db.create_table(
|
||||
"t",
|
||||
pa.table(
|
||||
{
|
||||
"id": [1],
|
||||
"ts": pa.array([ts_utc], type=pa.timestamp("us", tz="UTC")),
|
||||
}
|
||||
),
|
||||
)
|
||||
result = table.search().where(col("ts") == lit(ts_naive)).to_list()
|
||||
assert len(result) == 1
|
||||
assert result[0]["id"] == 1
|
||||
|
||||
def test_table_naive_literal_aware(self, tmp_path):
|
||||
"""Naive table + UTC lit with the same wall clock must match."""
|
||||
db = lancedb.connect(str(tmp_path / "naive_aware"))
|
||||
ts_naive = datetime(2024, 7, 1, 10, 0, 0)
|
||||
ts_utc = datetime(2024, 7, 1, 10, 0, 0, tzinfo=timezone.utc)
|
||||
table = db.create_table("t", [{"id": 1, "ts": ts_naive}])
|
||||
result = table.search().where(col("ts") == lit(ts_utc)).to_list()
|
||||
assert len(result) == 1
|
||||
assert result[0]["id"] == 1
|
||||
|
||||
def test_naive_lit_sql_is_wall_clock_not_local_shifted(self):
|
||||
"""Regression: naive lit must not apply the host local UTC offset."""
|
||||
ts = datetime(2024, 7, 1, 10, 0, 0)
|
||||
sql = lit(ts).to_sql()
|
||||
# Must encode 10:00 wall clock, not 10:00+local_offset.
|
||||
assert "2024-07-01 10:00:00" in sql
|
||||
|
||||
@@ -203,6 +203,31 @@ async def test_async_hybrid_query_default_limit(table: AsyncTable):
|
||||
assert texts.count("a") == 1
|
||||
|
||||
|
||||
def test_hybrid_query_offset(sync_table: Table):
|
||||
# The offset window of a hybrid query must be a suffix of the same query
|
||||
# run without an offset -- it must not be silently ignored.
|
||||
full = (
|
||||
sync_table.search(query_type="hybrid")
|
||||
.vector([0.0, 0.4])
|
||||
.text("dog")
|
||||
.limit(4)
|
||||
.with_row_id(True)
|
||||
.to_arrow()
|
||||
)
|
||||
assert len(full) == 4
|
||||
|
||||
offset_result = (
|
||||
sync_table.search(query_type="hybrid")
|
||||
.vector([0.0, 0.4])
|
||||
.text("dog")
|
||||
.offset(2)
|
||||
.limit(2)
|
||||
.with_row_id(True)
|
||||
.to_arrow()
|
||||
)
|
||||
assert offset_result["_rowid"].to_pylist() == full["_rowid"].to_pylist()[2:]
|
||||
|
||||
|
||||
def test_hybrid_query_minimum_nprobes_zero_raises(sync_table: Table):
|
||||
# minimum_nprobes(0) must raise the same validation error a plain vector
|
||||
# query raises, not silently no-op because 0 is falsy.
|
||||
|
||||
@@ -1133,6 +1133,131 @@ def test_stats():
|
||||
assert res == stats
|
||||
|
||||
|
||||
@contextlib.contextmanager
|
||||
def lsm_test_table(lsm_handler):
|
||||
"""A remote table whose LSM routes are served by ``lsm_handler``.
|
||||
|
||||
``lsm_handler(request, route)`` is called for ``/v1/table/test/<route>/``
|
||||
where route is one of flush_lsm, compact_lsm, get_lsm_stats, and is
|
||||
responsible for writing the response.
|
||||
"""
|
||||
routes = ("flush_lsm", "compact_lsm", "get_lsm_stats")
|
||||
|
||||
def handler(request):
|
||||
match = re.fullmatch(r"/v1/table/test/(\w+)/", request.path)
|
||||
route = match.group(1) if match else None
|
||||
if route in routes:
|
||||
lsm_handler(request, route)
|
||||
elif route == "describe":
|
||||
request.send_response(200)
|
||||
request.send_header("Content-Type", "application/json")
|
||||
request.end_headers()
|
||||
request.wfile.write(b'{"version": 1, "schema": {"fields": []}}')
|
||||
else:
|
||||
request.send_response(404)
|
||||
request.end_headers()
|
||||
|
||||
with mock_lancedb_connection(handler) as db:
|
||||
yield db.open_table("test")
|
||||
|
||||
|
||||
def read_json_body(request):
|
||||
content_len = int(request.headers.get("Content-Length"))
|
||||
return json.loads(request.rfile.read(content_len))
|
||||
|
||||
|
||||
def send_json(request, payload, status=200):
|
||||
request.send_response(status)
|
||||
request.send_header("Content-Type", "application/json")
|
||||
request.end_headers()
|
||||
request.wfile.write(json.dumps(payload).encode())
|
||||
|
||||
|
||||
def test_get_lsm_stats_sync():
|
||||
"""The sync wrapper round-trips the server payload into a dict."""
|
||||
bucket = {
|
||||
"shard_id": "b0",
|
||||
"status": "Active",
|
||||
"writer_epoch": 3,
|
||||
"manifest_version": 12,
|
||||
"current_generation": 6,
|
||||
"replay_after_wal_entry_position": 40,
|
||||
"wal_entry_position_last_seen": 42,
|
||||
"generations": [{"generation": 5, "bytes": 1024, "rows": 7}],
|
||||
"compacting": False,
|
||||
"memtables": [
|
||||
{
|
||||
"generation": 6,
|
||||
"rows": 2,
|
||||
"bytes": 64,
|
||||
"batches": 1,
|
||||
"indexes": ["vec_idx"],
|
||||
}
|
||||
],
|
||||
}
|
||||
seen_bodies = []
|
||||
|
||||
def lsm_handler(request, route):
|
||||
assert route == "get_lsm_stats"
|
||||
seen_bodies.append(read_json_body(request))
|
||||
send_json(request, {"lsm_stats": {"buckets": [bucket]}})
|
||||
|
||||
with lsm_test_table(lsm_handler) as table:
|
||||
assert table.get_lsm_stats() == {"buckets": [bucket]}
|
||||
# Off by default, and forwarded when asked for.
|
||||
assert seen_bodies == [{"include_generation_rows": False}]
|
||||
table.get_lsm_stats(include_generation_rows=True)
|
||||
assert seen_bodies[-1] == {"include_generation_rows": True}
|
||||
|
||||
|
||||
def test_get_lsm_stats_sync_returns_none_when_lsm_disabled():
|
||||
"""A null envelope means the LSM write path is not enabled, not an error."""
|
||||
|
||||
def lsm_handler(request, route):
|
||||
send_json(request, {"lsm_stats": None})
|
||||
|
||||
with lsm_test_table(lsm_handler) as table:
|
||||
assert table.get_lsm_stats() is None
|
||||
|
||||
|
||||
def test_flush_and_compact_lsm_sync():
|
||||
"""Both are one-shot POSTs answered 202 with no body."""
|
||||
called = []
|
||||
|
||||
def lsm_handler(request, route):
|
||||
called.append(route)
|
||||
request.send_response(202)
|
||||
request.end_headers()
|
||||
|
||||
with lsm_test_table(lsm_handler) as table:
|
||||
assert table.flush_lsm() is None
|
||||
assert table.compact_lsm() is None
|
||||
assert called == ["flush_lsm", "compact_lsm"]
|
||||
|
||||
|
||||
def test_checkpoint_lsm_sync():
|
||||
"""Seal, read the watermark, and return once L0 holds nothing.
|
||||
|
||||
The convergence loop itself is covered in Rust; this pins the sync
|
||||
binding to the endpoints it drives.
|
||||
"""
|
||||
called = []
|
||||
|
||||
def lsm_handler(request, route):
|
||||
called.append(route)
|
||||
if route == "get_lsm_stats":
|
||||
# An empty L0 yields no target watermark, so the loop is done
|
||||
# after the seal without ever polling compaction.
|
||||
send_json(request, {"lsm_stats": {"buckets": []}})
|
||||
else:
|
||||
request.send_response(202)
|
||||
request.end_headers()
|
||||
|
||||
with lsm_test_table(lsm_handler) as table:
|
||||
assert table.checkpoint_lsm() is None
|
||||
assert called == ["flush_lsm", "get_lsm_stats"]
|
||||
|
||||
|
||||
@contextlib.contextmanager
|
||||
def query_test_table(query_handler, *, server_version=Version("0.1.0")):
|
||||
def handler(request):
|
||||
|
||||
@@ -3953,3 +3953,31 @@ async def test_computed_column_async(tmp_path):
|
||||
await table.refresh_column("tripled")
|
||||
|
||||
assert (await table.to_arrow())["tripled"].to_pylist() == [9]
|
||||
|
||||
|
||||
def test_refresh_column_async_returns_job(tmp_path):
|
||||
db = lancedb.connect(tmp_path)
|
||||
table = db.create_table("computed_job", [{"x": 1}, {"x": 2}])
|
||||
table.add_columns(computed={"doubled": "x * 2"})
|
||||
|
||||
job = table.refresh_column_async("doubled")
|
||||
assert job.id is None # in-process jobs have no server id
|
||||
job.wait()
|
||||
assert job.status() == "finished"
|
||||
assert sorted(table.to_arrow()["doubled"].to_pylist()) == [2, 4]
|
||||
|
||||
# Bad input raises at the call, not through the job.
|
||||
with pytest.raises(Exception, match="not a computed column"):
|
||||
table.refresh_column_async("x")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_refresh_column_async_job_async_table(tmp_path):
|
||||
db = await lancedb.connect_async(tmp_path)
|
||||
table = await db.create_table("computed_job_async", [{"x": 3}])
|
||||
await table.add_columns(computed={"tripled": "x * 3"})
|
||||
|
||||
job = await table.refresh_column_async("tripled")
|
||||
await job.wait()
|
||||
assert await job.status() == "finished"
|
||||
assert (await table.to_arrow())["tripled"].to_pylist() == [9]
|
||||
|
||||
+20
-1
@@ -191,8 +191,27 @@ pub fn expr_lit(value: Bound<'_, PyAny>) -> PyResult<PyExpr> {
|
||||
}
|
||||
|
||||
// datetime.datetime is a subclass of datetime.date, so it must be checked first.
|
||||
//
|
||||
// Python's datetime.timestamp() treats *naive* datetimes as local wall time.
|
||||
// PyArrow (and therefore Lance table storage) encodes naive timestamps as
|
||||
// UTC wall-clock microseconds. Using .timestamp() for naive values therefore
|
||||
// shifts the literal by the local UTC offset on non-UTC machines, so
|
||||
// `col("ts") == lit(naive_dt)` fails against a table that holds the same
|
||||
// naive value. Fix: treat naive datetimes as UTC wall clock (match Arrow);
|
||||
// keep aware datetimes on the real .timestamp() path (correct epoch).
|
||||
if let Ok(dt) = value.cast::<PyDateTime>() {
|
||||
let ts: f64 = dt.call_method0("timestamp")?.extract()?;
|
||||
let ts: f64 = if dt.getattr("tzinfo")?.is_none() {
|
||||
// Force UTC interpretation of the naive wall clock.
|
||||
let utc = pyo3::types::PyModule::import(value.py(), "datetime")?
|
||||
.getattr("timezone")?
|
||||
.getattr("utc")?;
|
||||
let kwargs = pyo3::types::PyDict::new(value.py());
|
||||
kwargs.set_item("tzinfo", utc)?;
|
||||
let aware = dt.call_method("replace", (), Some(&kwargs))?;
|
||||
aware.call_method0("timestamp")?.extract()?
|
||||
} else {
|
||||
dt.call_method0("timestamp")?.extract()?
|
||||
};
|
||||
let micros = (ts * 1_000_000.0).round() as i64;
|
||||
return Ok(PyExpr(df_lit(ScalarValue::TimestampMicrosecond(
|
||||
Some(micros),
|
||||
|
||||
@@ -1668,6 +1668,17 @@ impl Table {
|
||||
})
|
||||
}
|
||||
|
||||
pub fn refresh_column_async(
|
||||
self_: PyRef<'_, Self>,
|
||||
column: String,
|
||||
) -> PyResult<Bound<'_, PyAny>> {
|
||||
let inner = self_.inner_ref()?.clone();
|
||||
future_into_py(self_.py(), async move {
|
||||
let job = inner.refresh_column_async(column).await.infer_error()?;
|
||||
Ok(crate::job::Job::new(job))
|
||||
})
|
||||
}
|
||||
|
||||
pub fn add_columns_with_schema(
|
||||
self_: PyRef<'_, Self>,
|
||||
schema: PyArrowType<Schema>,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "lancedb"
|
||||
version = "0.38.0-beta.0"
|
||||
version = "0.38.0-beta.2"
|
||||
edition.workspace = true
|
||||
description = "LanceDB: A serverless, low-latency vector database for AI applications"
|
||||
license.workspace = true
|
||||
|
||||
@@ -1032,6 +1032,7 @@ impl Database for ListingDatabase {
|
||||
};
|
||||
|
||||
Ok(ListTablesResponse {
|
||||
context: None,
|
||||
tables: f,
|
||||
page_token: next_page_token,
|
||||
})
|
||||
|
||||
@@ -141,7 +141,7 @@ impl SpawnedJob {
|
||||
Ok(Err(err)) => Outcome::Failed(Arc::new(err)),
|
||||
Err(err) if err.is_cancelled() => Outcome::Cancelled,
|
||||
Err(err) => Outcome::Failed(Arc::new(Error::Runtime {
|
||||
message: format!("index job task failed: {err}"),
|
||||
message: format!("job task failed: {err}"),
|
||||
})),
|
||||
};
|
||||
let _ = tx.send(Some(outcome));
|
||||
|
||||
@@ -33,7 +33,9 @@ use crate::table::lsm_stats::GetLsmStatsResponse;
|
||||
use crate::table::merge::MergeFilter;
|
||||
use crate::table::query::create_multi_vector_plan;
|
||||
use crate::table::write_progress::FinishOnDrop;
|
||||
use crate::table::{AlterColumnsResult, FieldMetadataUpdate, UpdateFieldMetadataResult};
|
||||
use crate::table::{
|
||||
AlterColumnsResult, FieldMetadataUpdate, RefreshColumnResult, UpdateFieldMetadataResult,
|
||||
};
|
||||
use crate::table::{AnyQuery, Filter, Predicate, PreprocessingOutput, TableStatistics};
|
||||
use crate::utils::background_cache::BackgroundCache;
|
||||
use crate::utils::{
|
||||
@@ -140,6 +142,40 @@ impl FreshnessHeaders {
|
||||
}
|
||||
}
|
||||
|
||||
/// A backfill job whose successful wait establishes a read-freshness
|
||||
/// baseline on the submitting handle, so a later read cannot be served
|
||||
/// from a cache older than the completed fill. A handle pinned by checkout
|
||||
/// at completion keeps its time-travel view instead.
|
||||
struct FreshnessJob<S: HttpSend> {
|
||||
inner: RemoteJob<S>,
|
||||
freshness: Arc<Mutex<FreshnessState>>,
|
||||
version: Arc<RwLock<Option<u64>>>,
|
||||
}
|
||||
|
||||
#[async_trait]
|
||||
impl<S: HttpSend> crate::job::JobHandle for FreshnessJob<S> {
|
||||
fn id(&self) -> Option<&str> {
|
||||
crate::job::JobHandle::id(&self.inner)
|
||||
}
|
||||
|
||||
async fn status(&self) -> Result<String> {
|
||||
crate::job::JobHandle::status(&self.inner).await
|
||||
}
|
||||
|
||||
async fn wait(&self) -> Result<()> {
|
||||
crate::job::JobHandle::wait(&self.inner).await?;
|
||||
let version = self.version.read().await;
|
||||
if version.is_none() {
|
||||
self.freshness.lock().unwrap().checkout_baseline = Some(SystemTime::now());
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn cancel(&self) -> Result<()> {
|
||||
crate::job::JobHandle::cancel(&self.inner).await
|
||||
}
|
||||
}
|
||||
|
||||
fn compute_min_timestamp(
|
||||
state: &FreshnessState,
|
||||
interval: Option<Duration>,
|
||||
@@ -274,10 +310,10 @@ pub struct RemoteTable<S: HttpSend = Sender> {
|
||||
identifier: String,
|
||||
server_version: ServerVersion,
|
||||
|
||||
version: RwLock<Option<u64>>,
|
||||
version: Arc<RwLock<Option<u64>>>,
|
||||
location: RwLock<Option<String>>,
|
||||
schema_cache: BackgroundCache<SchemaRef, Error>,
|
||||
freshness: Mutex<FreshnessState>,
|
||||
freshness: Arc<Mutex<FreshnessState>>,
|
||||
/// The branch this handle is scoped to, or `None` for the main branch.
|
||||
/// Stamped onto every branch-accepting request so reads and writes resolve
|
||||
/// on the branch's own version chain rather than main's.
|
||||
@@ -415,10 +451,10 @@ impl<S: HttpSend> RemoteTable<S> {
|
||||
namespace,
|
||||
identifier,
|
||||
server_version,
|
||||
version: RwLock::new(None),
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
branch: None,
|
||||
}
|
||||
}
|
||||
@@ -447,10 +483,10 @@ impl<S: HttpSend> RemoteTable<S> {
|
||||
namespace: self.namespace.clone(),
|
||||
identifier: self.identifier.clone(),
|
||||
server_version: self.server_version.clone(),
|
||||
version: RwLock::new(None),
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
branch,
|
||||
}
|
||||
}
|
||||
@@ -1268,10 +1304,10 @@ mod test_utils {
|
||||
namespace: vec![],
|
||||
identifier: name,
|
||||
server_version: version.map(ServerVersion).unwrap_or_default(),
|
||||
version: RwLock::new(None),
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
branch: None,
|
||||
}
|
||||
}
|
||||
@@ -1292,10 +1328,10 @@ mod test_utils {
|
||||
namespace: vec![],
|
||||
identifier: name,
|
||||
server_version: ServerVersion::default(),
|
||||
version: RwLock::new(None),
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
branch: None,
|
||||
}
|
||||
}
|
||||
@@ -1325,10 +1361,10 @@ mod test_utils {
|
||||
namespace: vec![],
|
||||
identifier: name,
|
||||
server_version: version.map(ServerVersion).unwrap_or_default(),
|
||||
version: RwLock::new(None),
|
||||
version: Arc::new(RwLock::new(None)),
|
||||
location: RwLock::new(None),
|
||||
schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW),
|
||||
freshness: Mutex::new(FreshnessState::default()),
|
||||
freshness: Arc::new(Mutex::new(FreshnessState::default())),
|
||||
branch: None,
|
||||
}
|
||||
}
|
||||
@@ -2700,13 +2736,6 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
// A declaration reaches here as AllNulls, which the remote protocol
|
||||
// has no representation for.
|
||||
NewColumnTransform::AllNulls(_) => {
|
||||
return Err(Error::NotSupported {
|
||||
message: "computed columns are supported only on local tables".into(),
|
||||
});
|
||||
}
|
||||
_ => {
|
||||
return Err(Error::NotSupported {
|
||||
message: "Only SQL expressions are supported for adding columns".into(),
|
||||
@@ -2715,6 +2744,86 @@ impl<S: HttpSend> BaseTable for RemoteTable<S> {
|
||||
}
|
||||
}
|
||||
|
||||
async fn add_computed_columns(&self, columns: &[(String, String)]) -> Result<AddColumnsResult> {
|
||||
self.check_mutable().await?;
|
||||
// The server plans the declaration: expression validation, type
|
||||
// inference and the persisted binding all happen there.
|
||||
let entries = columns
|
||||
.iter()
|
||||
.map(
|
||||
|(name, expression)| lance_namespace::models::AddColumnsEntry {
|
||||
name: name.clone(),
|
||||
computed: Some(Some(expression.clone())),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.collect::<Vec<_>>();
|
||||
let mut body = serde_json::json!({ "new_columns": entries });
|
||||
self.apply_branch_body(&mut body);
|
||||
let request = self
|
||||
.client
|
||||
.post(&format!("/v1/table/{}/add_columns/", self.identifier))
|
||||
.json(&body);
|
||||
let (request_id, response) = self.send(request, true).await?;
|
||||
let response = self.check_table_response(&request_id, response).await?;
|
||||
let body = response.text().await.err_to_http(request_id.clone())?;
|
||||
|
||||
if body.trim().is_empty() {
|
||||
// Backward compatible with old servers
|
||||
return Ok(AddColumnsResult { version: 0 });
|
||||
}
|
||||
|
||||
let result: AddColumnsResult = serde_json::from_str(&body).map_err(|e| Error::Http {
|
||||
source: format!("Failed to parse add_columns response: {}", e).into(),
|
||||
request_id,
|
||||
status_code: None,
|
||||
})?;
|
||||
|
||||
self.invalidate_schema_cache();
|
||||
self.track_write_version(result.version);
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
async fn refresh_column(&self, _column: &str) -> Result<RefreshColumnResult> {
|
||||
// The server runs a refresh as a job and does not report a fill
|
||||
// count, so the blocking form has no honest result to return.
|
||||
Err(Error::NotSupported {
|
||||
message: "a remote refresh runs as a server job; use refresh_column_async and \
|
||||
wait on the returned handle"
|
||||
.into(),
|
||||
})
|
||||
}
|
||||
|
||||
async fn refresh_column_async(&self, column: &str) -> Result<Job> {
|
||||
self.check_mutable().await?;
|
||||
let mut body = serde_json::json!({ "column": column });
|
||||
self.apply_branch_body(&mut body);
|
||||
let request = self
|
||||
.client
|
||||
.post(&format!("/v1/table/{}/backfill_column", self.identifier))
|
||||
.json(&body);
|
||||
let (request_id, response) = self.send(request, true).await?;
|
||||
let response = self.check_table_response(&request_id, response).await?;
|
||||
let body = response.text().await.err_to_http(request_id.clone())?;
|
||||
|
||||
#[derive(serde::Deserialize)]
|
||||
struct BackfillResponse {
|
||||
job_id: String,
|
||||
}
|
||||
let response: BackfillResponse = serde_json::from_str(&body).map_err(|e| Error::Http {
|
||||
source: format!("Failed to parse backfill_column response: {}", e).into(),
|
||||
request_id,
|
||||
status_code: None,
|
||||
})?;
|
||||
|
||||
Ok(Job::new(Box::new(FreshnessJob {
|
||||
inner: RemoteJob::new(self.client.clone(), response.job_id),
|
||||
freshness: self.freshness.clone(),
|
||||
version: self.version.clone(),
|
||||
})))
|
||||
}
|
||||
|
||||
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult> {
|
||||
self.check_mutable().await?;
|
||||
let body = alterations
|
||||
@@ -6456,37 +6565,346 @@ mod tests {
|
||||
assert_eq!(result.version, if old_server { 0 } else { 43 });
|
||||
}
|
||||
|
||||
/// Computed columns are local-only. Both halves say so here rather than
|
||||
/// reaching the wire and failing somewhere less legible.
|
||||
/// A declaration is sent as `{name, computed}` entries for the server to
|
||||
/// plan; the client never types the expression itself.
|
||||
#[tokio::test]
|
||||
async fn test_computed_columns_are_refused() {
|
||||
let table = Table::new_with_handler("my_table", |request| -> http::Response<String> {
|
||||
panic!("unexpected request: {}", request.url().path())
|
||||
async fn test_add_computed_columns_sends_the_expression() {
|
||||
let table = Table::new_with_handler("my_table", |request| {
|
||||
assert_eq!(request.method(), "POST");
|
||||
assert_eq!(request.url().path(), "/v1/table/my_table/add_columns/");
|
||||
let body = request.body().unwrap().as_bytes().unwrap();
|
||||
let value: serde_json::Value = serde_json::from_slice(body).unwrap();
|
||||
assert_eq!(
|
||||
value["new_columns"],
|
||||
serde_json::json!([{"name": "doubled", "computed": "x * 2"}])
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"version": 7}"#)
|
||||
.unwrap()
|
||||
});
|
||||
|
||||
let declared = Arc::new(Schema::new(vec![Field::new(
|
||||
"doubled",
|
||||
DataType::Int32,
|
||||
true,
|
||||
)]));
|
||||
let err = table
|
||||
let result = table
|
||||
.add_columns()
|
||||
.transform(NewColumnTransform::AllNulls(declared))
|
||||
.computed("doubled", "x * 2")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::NotSupported { message } if message.contains("local tables")),
|
||||
"{err:?}"
|
||||
);
|
||||
.unwrap();
|
||||
assert_eq!(result.version, 7);
|
||||
}
|
||||
|
||||
/// A remote refresh is a server job: the async form returns its handle,
|
||||
/// and the blocking form refuses rather than invent a fill count.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_column_async_submits_a_backfill_job() {
|
||||
let table = Table::new_with_handler("my_table", |request| {
|
||||
assert_eq!(request.method(), "POST");
|
||||
assert_eq!(request.url().path(), "/v1/table/my_table/backfill_column");
|
||||
let body = request.body().unwrap().as_bytes().unwrap();
|
||||
let value: serde_json::Value = serde_json::from_slice(body).unwrap();
|
||||
assert_eq!(value["column"], "doubled");
|
||||
http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-42"}"#)
|
||||
.unwrap()
|
||||
});
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
assert_eq!(job.id(), Some("j-42"));
|
||||
|
||||
let err = table.refresh_column("doubled").await.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::NotSupported { message } if message.contains("local tables")),
|
||||
matches!(&err, Error::NotSupported { message }
|
||||
if message.contains("refresh_column_async")),
|
||||
"{err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
/// The gate's reproducer: after a successful wait, a same-handle read
|
||||
/// must carry a freshness baseline so a stale server cache cannot serve
|
||||
/// the pre-backfill snapshot.
|
||||
#[tokio::test]
|
||||
async fn test_backfill_wait_establishes_read_freshness() {
|
||||
let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
let saw = saw_min_timestamp.clone();
|
||||
let table =
|
||||
Table::new_with_handler("my_table", move |request| match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-7"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-7", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
saw.store(
|
||||
request.headers().contains_key("x-lancedb-min-timestamp"),
|
||||
std::sync::atomic::Ordering::SeqCst,
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
});
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
assert!(
|
||||
saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst),
|
||||
"read after wait carried no freshness baseline"
|
||||
);
|
||||
}
|
||||
|
||||
/// A checkout after submission wins over the completion fence: the
|
||||
/// pinned view must not regain a timestamp floor from the job.
|
||||
#[tokio::test]
|
||||
async fn test_checkout_after_submit_beats_the_completion_fence() {
|
||||
let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
let saw = saw_min_timestamp.clone();
|
||||
let table =
|
||||
Table::new_with_handler("my_table", move |request| match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-8"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-8", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/describe/" => {
|
||||
let schema = Schema::new(vec![Field::new("x", DataType::Int32, true)]);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(describe_response(&schema))
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
saw.store(
|
||||
request.headers().contains_key("x-lancedb-min-timestamp"),
|
||||
std::sync::atomic::Ordering::SeqCst,
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
});
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
table.checkout(3).await.unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
assert!(
|
||||
!saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst),
|
||||
"completion fence overrode an explicit checkout"
|
||||
);
|
||||
}
|
||||
|
||||
/// Tag checkout resets freshness state wholesale; the fence must not
|
||||
/// survive it.
|
||||
#[tokio::test]
|
||||
async fn test_tag_checkout_after_submit_beats_the_completion_fence() {
|
||||
let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
let saw = saw_min_timestamp.clone();
|
||||
let table =
|
||||
Table::new_with_handler("my_table", move |request| match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-9"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-9", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/tags/version/" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"version": 5}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/describe/" => {
|
||||
let schema = Schema::new(vec![Field::new("x", DataType::Int32, true)]);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(describe_response(&schema))
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
saw.store(
|
||||
request.headers().contains_key("x-lancedb-min-timestamp"),
|
||||
std::sync::atomic::Ordering::SeqCst,
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
});
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
table.checkout_tag("v1").await.unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
assert!(
|
||||
!saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst),
|
||||
"completion fence overrode a tag checkout"
|
||||
);
|
||||
}
|
||||
|
||||
/// A checkout landing while the submission request is in flight advances
|
||||
/// the epoch past the token captured at submit.
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn test_checkout_during_submission_beats_the_completion_fence() {
|
||||
let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
let saw = saw_min_timestamp.clone();
|
||||
let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
|
||||
let release_rx = Arc::new(std::sync::Mutex::new(release_rx));
|
||||
let (arrived_tx, arrived_rx) = std::sync::mpsc::channel::<()>();
|
||||
let arrived_tx = Arc::new(std::sync::Mutex::new(arrived_tx));
|
||||
let table = Table::new_with_handler("my_table", move |request| {
|
||||
match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => {
|
||||
// Signal arrival, then hold the response until the
|
||||
// test's checkout completes.
|
||||
arrived_tx.lock().unwrap().send(()).unwrap();
|
||||
release_rx
|
||||
.lock()
|
||||
.unwrap()
|
||||
.recv_timeout(std::time::Duration::from_secs(10))
|
||||
.unwrap();
|
||||
http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-10"}"#.to_string())
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-10", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/describe/" => {
|
||||
let schema = Schema::new(vec![Field::new("x", DataType::Int32, true)]);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body(describe_response(&schema))
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
saw.store(
|
||||
request.headers().contains_key("x-lancedb-min-timestamp"),
|
||||
std::sync::atomic::Ordering::SeqCst,
|
||||
);
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
}
|
||||
});
|
||||
|
||||
let submit = tokio::spawn({
|
||||
let table = table.clone();
|
||||
async move { table.refresh_column_async("doubled").await }
|
||||
});
|
||||
tokio::task::spawn_blocking(move || {
|
||||
arrived_rx
|
||||
.recv_timeout(std::time::Duration::from_secs(10))
|
||||
.unwrap()
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
table.checkout(7).await.unwrap();
|
||||
release_tx.send(()).unwrap();
|
||||
|
||||
let job = submit.await.unwrap().unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
assert!(
|
||||
!saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst),
|
||||
"completion fence overrode a checkout that landed mid-submission"
|
||||
);
|
||||
}
|
||||
|
||||
/// checkout_latest keeps the handle on latest, so a completed backfill
|
||||
/// must still establish its post-fill baseline -- strictly later than the
|
||||
/// checkout's own, or a pre-fill cache could still serve.
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn test_checkout_latest_during_submission_keeps_the_fence() {
|
||||
let seen_min_timestamp = Arc::new(std::sync::Mutex::new(None::<String>));
|
||||
let saw = seen_min_timestamp.clone();
|
||||
let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
|
||||
let release_rx = Arc::new(std::sync::Mutex::new(release_rx));
|
||||
let (arrived_tx, arrived_rx) = std::sync::mpsc::channel::<()>();
|
||||
let arrived_tx = Arc::new(std::sync::Mutex::new(arrived_tx));
|
||||
let table =
|
||||
Table::new_with_handler("my_table", move |request| match request.url().path() {
|
||||
"/v1/table/my_table/backfill_column" => {
|
||||
arrived_tx.lock().unwrap().send(()).unwrap();
|
||||
release_rx
|
||||
.lock()
|
||||
.unwrap()
|
||||
.recv_timeout(std::time::Duration::from_secs(10))
|
||||
.unwrap();
|
||||
http::Response::builder()
|
||||
.status(202)
|
||||
.body(r#"{"job_id": "j-11"}"#.to_string())
|
||||
.unwrap()
|
||||
}
|
||||
"/v1/jobs/describe" => http::Response::builder()
|
||||
.status(200)
|
||||
.body(r#"{"job_id": "j-11", "job_state": "DONE"}"#.to_string())
|
||||
.unwrap(),
|
||||
"/v1/table/my_table/count_rows/" => {
|
||||
*saw.lock().unwrap() = request
|
||||
.headers()
|
||||
.get("x-lancedb-min-timestamp")
|
||||
.map(|v| v.to_str().unwrap().to_string());
|
||||
http::Response::builder()
|
||||
.status(200)
|
||||
.body("1".to_string())
|
||||
.unwrap()
|
||||
}
|
||||
path => panic!("unexpected request: {path}"),
|
||||
});
|
||||
|
||||
let submit = tokio::spawn({
|
||||
let table = table.clone();
|
||||
async move { table.refresh_column_async("doubled").await }
|
||||
});
|
||||
tokio::task::spawn_blocking(move || {
|
||||
arrived_rx
|
||||
.recv_timeout(std::time::Duration::from_secs(10))
|
||||
.unwrap()
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
table.checkout_latest().await.unwrap();
|
||||
let after_checkout = SystemTime::now();
|
||||
// Real separation between the checkout baseline and completion.
|
||||
tokio::time::sleep(std::time::Duration::from_millis(50)).await;
|
||||
release_tx.send(()).unwrap();
|
||||
|
||||
let job = submit.await.unwrap().unwrap();
|
||||
job.wait().await.unwrap();
|
||||
table.count_rows(None).await.unwrap();
|
||||
let header = seen_min_timestamp
|
||||
.lock()
|
||||
.unwrap()
|
||||
.clone()
|
||||
.expect("no baseline");
|
||||
let sent: SystemTime = chrono::DateTime::parse_from_rfc3339(&header)
|
||||
.unwrap()
|
||||
.into();
|
||||
assert!(
|
||||
sent > after_checkout,
|
||||
"baseline {header} did not advance past the checkout"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_prewarm_index() {
|
||||
let table = Table::new_with_handler("my_table", |request| {
|
||||
|
||||
@@ -750,6 +750,10 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
|
||||
read_columns: Option<Vec<String>>,
|
||||
) -> Result<AddColumnsResult>;
|
||||
/// Declare computed columns, each defined by a SQL expression.
|
||||
///
|
||||
/// Where the declaration is planned depends on the backend: a local table
|
||||
/// validates and types the expression itself, a remote one sends the text
|
||||
/// for the server to plan.
|
||||
async fn add_computed_columns(
|
||||
&self,
|
||||
_columns: &[(String, String)],
|
||||
@@ -766,6 +770,13 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync {
|
||||
message: "computed columns are supported only on local tables".into(),
|
||||
})
|
||||
}
|
||||
/// Fill a computed column's unfilled rows, returning a [`Job`] tracking
|
||||
/// the operation.
|
||||
async fn refresh_column_async(&self, _column: &str) -> Result<Job> {
|
||||
Err(Error::NotSupported {
|
||||
message: "computed columns are supported only on local tables".into(),
|
||||
})
|
||||
}
|
||||
/// Alter columns in the table.
|
||||
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult>;
|
||||
/// Drop columns from the table.
|
||||
@@ -1667,7 +1678,8 @@ impl Table {
|
||||
/// filled are left as they are, so the call is idempotent and does not
|
||||
/// observe a mutated input.
|
||||
///
|
||||
/// Local tables only.
|
||||
/// Local tables only: a remote refresh runs as a server job, through
|
||||
/// [`Table::refresh_column_async`].
|
||||
///
|
||||
/// ```
|
||||
/// # use lancedb::Table;
|
||||
@@ -1681,6 +1693,29 @@ impl Table {
|
||||
self.inner.refresh_column(column.as_ref()).await
|
||||
}
|
||||
|
||||
/// Like [`Table::refresh_column`], but returns a [`Job`] tracking the
|
||||
/// operation instead of blocking until it completes.
|
||||
///
|
||||
/// The job may already be complete when returned, and callers must not
|
||||
/// assume the column is filled until [`Job::wait`] returns. Invalid input
|
||||
/// -- an unknown column, or one that is not computed -- is reported by
|
||||
/// this call rather than by the job. On local tables the job runs as an
|
||||
/// in-process task; on LanceDB Cloud and Enterprise it is the server's
|
||||
/// backfill job.
|
||||
///
|
||||
/// ```
|
||||
/// # use lancedb::Table;
|
||||
/// # async fn refresh_in_background(table: &Table) -> Result<(), Box<dyn std::error::Error>> {
|
||||
/// let job = table.refresh_column_async("doubled").await?;
|
||||
/// println!("refresh running: {:?}", job.status().await?);
|
||||
/// job.wait().await?;
|
||||
/// # Ok(())
|
||||
/// # }
|
||||
/// ```
|
||||
pub async fn refresh_column_async(&self, column: impl AsRef<str>) -> Result<Job> {
|
||||
self.inner.refresh_column_async(column.as_ref()).await
|
||||
}
|
||||
|
||||
/// Change a column's name or nullability.
|
||||
pub async fn alter_columns(
|
||||
&self,
|
||||
@@ -3394,6 +3429,10 @@ impl BaseTable for NativeTable {
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
async fn refresh_column_async(&self, column: &str) -> Result<Job> {
|
||||
refresh::execute_refresh_column_async(self, column).await
|
||||
}
|
||||
|
||||
async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result<AlterColumnsResult> {
|
||||
let result = schema_evolution::execute_alter_columns(self, alterations).await?;
|
||||
self.bump_freshness();
|
||||
|
||||
@@ -61,8 +61,9 @@ impl AddColumnsBuilder {
|
||||
/// column and declaring it again. An input cannot be renamed, retyped or
|
||||
/// dropped while a declaration reads it, since the expression names it.
|
||||
///
|
||||
/// Local tables only: LanceDB Cloud and Enterprise reject a declaration
|
||||
/// with `NotSupported`.
|
||||
/// On LanceDB Cloud and Enterprise the expression is planned by the
|
||||
/// server, and the refresh runs as a server job -- see
|
||||
/// [`Table::refresh_column_async`](super::Table::refresh_column_async).
|
||||
///
|
||||
/// ```
|
||||
/// # use lancedb::Table;
|
||||
|
||||
@@ -12,7 +12,7 @@
|
||||
//! decides whether the fragment is staged at all -- a fragment where nothing
|
||||
//! would change stages nothing, which is what lets an expression yielding
|
||||
//! null settle instead of restaging forever. The second streams the
|
||||
//! fragment's physical rows into `write_column` a batch at a time, so peak
|
||||
//! fragment's physical rows into `write_columns` a batch at a time, so peak
|
||||
//! memory is bounded by a scan batch. The expression is evaluated by this
|
||||
//! module, never through a projection alias, and only over rows being
|
||||
//! filled: every other row -- deleted, or already holding a value -- has its
|
||||
@@ -35,6 +35,7 @@ use serde::{Deserialize, Serialize};
|
||||
|
||||
use super::computed_columns::{BoundExpression, ComputedColumnKind, computed_column_from_field};
|
||||
use super::{BaseTable, NativeTable};
|
||||
use crate::job::Job;
|
||||
use crate::{Error, Result};
|
||||
|
||||
/// The result of refreshing a computed column.
|
||||
@@ -66,7 +67,7 @@ pub(crate) async fn execute_refresh_column(
|
||||
.ok_or_else(|| Error::ColumnNotFound {
|
||||
name: column.to_string(),
|
||||
})?;
|
||||
// The dataset's own field, so the identity write_column checks against the
|
||||
// The dataset's own field, so the identity write_columns checks against the
|
||||
// manifest holds by construction.
|
||||
let column_schema = LanceSchema {
|
||||
fields: vec![field.clone()],
|
||||
@@ -82,7 +83,7 @@ pub(crate) async fn execute_refresh_column(
|
||||
}
|
||||
rows_filled += gained;
|
||||
let values = fill_stream(&dataset, &fragment, bound.clone(), column).await?;
|
||||
replacements.push(fragment.write_column(values, &column_schema).await?);
|
||||
replacements.push(fragment.write_columns(values, &column_schema).await?);
|
||||
}
|
||||
|
||||
if replacements.is_empty() {
|
||||
@@ -115,6 +116,25 @@ pub(crate) async fn execute_refresh_column(
|
||||
})
|
||||
}
|
||||
|
||||
/// Run the refresh as a [`Job`] in this process.
|
||||
pub(crate) async fn execute_refresh_column_async(table: &NativeTable, column: &str) -> Result<Job> {
|
||||
// Validate before spawning so bad input is reported by this call rather
|
||||
// than only by the job.
|
||||
table.dataset.ensure_mutable()?;
|
||||
ensure_no_lsm_write_spec(table).await?;
|
||||
let dataset = table.dataset.get().await?;
|
||||
declared_expression(&dataset, column)?;
|
||||
drop(dataset);
|
||||
|
||||
let table = table.clone();
|
||||
let column = column.to_string();
|
||||
Ok(Job::spawned(tokio::spawn(async move {
|
||||
execute_refresh_column(&table, &column).await?;
|
||||
table.bump_freshness();
|
||||
Ok(())
|
||||
})))
|
||||
}
|
||||
|
||||
/// Refuse to refresh under an LSM write spec.
|
||||
///
|
||||
/// Refresh enumerates base fragments, and a write spec keeps visible rows in
|
||||
@@ -606,6 +626,230 @@ mod tests {
|
||||
assert!(Arc::ptr_eq(&dataset.session(), &session));
|
||||
}
|
||||
|
||||
/// The async form's job settles with the fill visible, like
|
||||
/// create_index's execute_async.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_async_job_waits_for_the_fill() {
|
||||
let table = table_with("refresh_async", vec![1, 2, 3]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
assert!(job.id().is_none(), "in-process jobs have no server id");
|
||||
job.wait().await.unwrap();
|
||||
assert_eq!(job.status().await.unwrap(), "finished");
|
||||
assert_eq!(
|
||||
read(&table, "doubled").await,
|
||||
vec![Some(2), Some(4), Some(6)]
|
||||
);
|
||||
}
|
||||
|
||||
/// Bad input is reported by the call, not by the job.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_async_rejects_bad_input_before_spawning() {
|
||||
let table = table_with("refresh_async_bad", vec![1, 2, 3]).await;
|
||||
|
||||
let err = table.refresh_column_async("x").await.unwrap_err();
|
||||
assert!(matches!(err, Error::NotAComputedColumn { name } if name == "x"));
|
||||
|
||||
let err = table.refresh_column_async("nope").await.unwrap_err();
|
||||
assert!(matches!(err, Error::ColumnNotFound { name } if name == "nope"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_async_job_reports_success_to_every_waiter() {
|
||||
let table = table_with("refresh_async_waiters", vec![1, 2]).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
|
||||
let job = table.refresh_column_async("doubled").await.unwrap();
|
||||
job.wait().await.unwrap();
|
||||
// A second wait after completion observes the same outcome.
|
||||
job.wait().await.unwrap();
|
||||
assert_eq!(job.status().await.unwrap(), "finished");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_rejects_a_plain_column() {
|
||||
let table = table_with("refresh_plain", vec![1, 2, 3]).await;
|
||||
let err = table.refresh_column("x").await.unwrap_err();
|
||||
assert!(matches!(err, Error::NotAComputedColumn { name } if name == "x"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_refresh_rejects_an_unknown_column() {
|
||||
let table = table_with("refresh_missing", vec![1, 2, 3]).await;
|
||||
let err = table.refresh_column("nope").await.unwrap_err();
|
||||
assert!(matches!(err, Error::ColumnNotFound { name } if name == "nope"));
|
||||
}
|
||||
|
||||
/// The gate's reproducer: a poison value in a deleted row must not
|
||||
/// abort filling the live rows, since nobody can read it.
|
||||
#[tokio::test]
|
||||
async fn test_a_deleted_rows_value_is_never_evaluated() {
|
||||
let table = table_with("refresh_deleted_poison", vec![1, 0]).await;
|
||||
table
|
||||
.add_columns()
|
||||
.computed("quotient", "10 / x")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
table.delete("x = 0").await.unwrap();
|
||||
|
||||
let result = table.refresh_column("quotient").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
assert_eq!(read(&table, "quotient").await, vec![Some(10)]);
|
||||
}
|
||||
|
||||
/// The gate's reproducer: an already-filled row's value must not be
|
||||
/// re-evaluated either -- its input may have mutated into one the
|
||||
/// expression chokes on.
|
||||
#[tokio::test]
|
||||
async fn test_a_filled_rows_value_is_never_evaluated() {
|
||||
let table = table_with("refresh_filled_poison", vec![1, 2]).await;
|
||||
table
|
||||
.add_columns()
|
||||
.computed("quotient", "10 / x")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
table.refresh_column("quotient").await.unwrap();
|
||||
|
||||
table
|
||||
.update()
|
||||
.column("x", "0")
|
||||
.only_if("x = 1")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
append(&table, vec![5]).await;
|
||||
|
||||
let result = table.refresh_column("quotient").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
assert_eq!(
|
||||
read(&table, "quotient").await,
|
||||
vec![Some(2), Some(5), Some(10)]
|
||||
);
|
||||
}
|
||||
|
||||
/// The gate's reproducer: the old internal projection alias is an
|
||||
/// ordinary column name; a computed column may use it.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_a_column_named_like_the_old_alias() {
|
||||
let table = table_with("refresh_alias_name", vec![1, 2]).await;
|
||||
table
|
||||
.add_columns()
|
||||
.computed("__lancedb_computed", "x * 2")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let result = table.refresh_column("__lancedb_computed").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 2);
|
||||
assert_eq!(
|
||||
read(&table, "__lancedb_computed").await,
|
||||
vec![Some(2), Some(4)]
|
||||
);
|
||||
}
|
||||
|
||||
/// The gate's reproducer: a late-gain fragment (filled, then one null row
|
||||
/// compacted onto the end) fills without the old probe's buffering, which
|
||||
/// this pins behaviorally; the memory bound is structural -- the fill
|
||||
/// stream retains no batches at all.
|
||||
#[tokio::test]
|
||||
async fn test_refresh_fills_a_late_gain_fragment() {
|
||||
let values: Vec<i32> = (0..20_000).collect();
|
||||
let table = table_with("refresh_late_gain", values).await;
|
||||
declare_doubled(&table).await.unwrap();
|
||||
table.refresh_column("doubled").await.unwrap();
|
||||
|
||||
append(&table, vec![2_000_000]).await;
|
||||
table
|
||||
.optimize(crate::table::OptimizeAction::Compact {
|
||||
options: crate::table::CompactionOptions::default(),
|
||||
remap_options: None,
|
||||
})
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
let result = table.refresh_column("doubled").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
let read_back = read(&table, "doubled").await;
|
||||
assert_eq!(read_back.len(), 20_001);
|
||||
assert_eq!(read_back.last().unwrap(), &Some(4_000_000));
|
||||
}
|
||||
|
||||
/// The gate's reproducer: a nested input declares, refreshes, and guards
|
||||
/// its root against invalidating schema changes.
|
||||
#[tokio::test]
|
||||
async fn test_a_nested_input_declares_and_refreshes() {
|
||||
use arrow_array::{Int32Array, StructArray};
|
||||
use arrow_schema::{DataType, Field, Fields};
|
||||
|
||||
let conn = connect("memory://").execute().await.unwrap();
|
||||
let age = Arc::new(Int32Array::from(vec![30, 40]));
|
||||
let fields = Fields::from(vec![Field::new("age", DataType::Int32, true)]);
|
||||
let metadata = StructArray::new(fields.clone(), vec![age as _], None);
|
||||
let schema = Arc::new(arrow_schema::Schema::new(vec![Field::new(
|
||||
"metadata",
|
||||
DataType::Struct(fields),
|
||||
true,
|
||||
)]));
|
||||
let batch =
|
||||
arrow_array::RecordBatch::try_new(schema, vec![Arc::new(metadata) as _]).unwrap();
|
||||
let table = conn
|
||||
.create_table("refresh_nested", batch)
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
table
|
||||
.add_columns()
|
||||
.computed("next_age", "metadata.age + 1")
|
||||
.execute()
|
||||
.await
|
||||
.unwrap();
|
||||
let declaration =
|
||||
&crate::table::computed_columns(table.schema().await.unwrap().as_ref())[0];
|
||||
assert_eq!(declaration.inputs, vec!["metadata.age".to_string()]);
|
||||
|
||||
let result = table.refresh_column("next_age").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 2);
|
||||
assert_eq!(read(&table, "next_age").await, vec![Some(31), Some(41)]);
|
||||
|
||||
// The dotted input guards its root.
|
||||
let err = table.drop_columns(&["metadata"]).await.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, Error::InvalidInput { message } if message.contains("next_age")),
|
||||
"{err:?}"
|
||||
);
|
||||
|
||||
// Masking a struct input for a deleted row goes through the same
|
||||
// nullif path as a primitive; a nested input plus deletions must not
|
||||
// be the combination that breaks it.
|
||||
table.delete("next_age = 31").await.unwrap();
|
||||
append_struct_row(&table, 50).await;
|
||||
let result = table.refresh_column("next_age").await.unwrap();
|
||||
assert_eq!(result.rows_filled, 1);
|
||||
assert_eq!(read(&table, "next_age").await, vec![Some(41), Some(51)]);
|
||||
}
|
||||
|
||||
/// Append one `metadata: {age}` row to the nested-input table.
|
||||
async fn append_struct_row(table: &Table, age: i32) {
|
||||
use arrow_array::{Int32Array, StructArray};
|
||||
use arrow_schema::{DataType, Field, Fields};
|
||||
|
||||
let ages = Arc::new(Int32Array::from(vec![age]));
|
||||
let fields = Fields::from(vec![Field::new("age", DataType::Int32, true)]);
|
||||
let metadata = StructArray::new(fields.clone(), vec![ages as _], None);
|
||||
let schema = Arc::new(arrow_schema::Schema::new(vec![Field::new(
|
||||
"metadata",
|
||||
DataType::Struct(fields),
|
||||
true,
|
||||
)]));
|
||||
let batch =
|
||||
arrow_array::RecordBatch::try_new(schema, vec![Arc::new(metadata) as _]).unwrap();
|
||||
table.add(batch).execute().await.unwrap();
|
||||
}
|
||||
|
||||
/// Both orders of declare+spec are refused at the source (see the
|
||||
/// schema_evolution tests); refresh's own check covers a dataset another
|
||||
/// writer left in that state.
|
||||
@@ -639,6 +883,8 @@ mod tests {
|
||||
matches!(&err, Error::NotSupported { message } if message.contains("LSM")),
|
||||
"{err:?}"
|
||||
);
|
||||
let err = table.refresh_column_async("doubled").await.unwrap_err();
|
||||
assert!(matches!(err, Error::NotSupported { .. }));
|
||||
}
|
||||
|
||||
/// After catch-up activation and unset, no spec remains but the catch-up
|
||||
|
||||
Reference in New Issue
Block a user