diff --git a/Cargo.lock b/Cargo.lock index 78d50a1c4d..ad401333c8 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3740,7 +3740,7 @@ checksum = "a4ae5f15dda3c708c0ade84bfee31ccab44a3da4f88015ed22f63732abe300c8" [[package]] name = "datafusion" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "arrow-schema 58.3.0", @@ -3794,7 +3794,7 @@ dependencies = [ [[package]] name = "datafusion-catalog" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "async-trait", @@ -3818,7 +3818,7 @@ dependencies = [ [[package]] name = "datafusion-catalog-listing" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "async-trait", @@ -3840,7 +3840,7 @@ dependencies = [ [[package]] name = "datafusion-common" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "ahash 0.8.12", "arrow 58.3.0", @@ -3864,7 +3864,7 @@ dependencies = [ [[package]] name = "datafusion-common-runtime" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "futures", "log", @@ -3874,7 +3874,7 @@ dependencies = [ [[package]] name = "datafusion-datasource" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "async-compression", @@ -3908,7 +3908,7 @@ dependencies = [ [[package]] name = "datafusion-datasource-arrow" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "arrow-ipc 58.3.0", @@ -3931,7 +3931,7 @@ dependencies = [ [[package]] name = "datafusion-datasource-csv" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "async-trait", @@ -3953,7 +3953,7 @@ dependencies = [ [[package]] name = "datafusion-datasource-json" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "async-trait", @@ -3976,7 +3976,7 @@ dependencies = [ [[package]] name = "datafusion-datasource-parquet" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "async-trait", @@ -4005,12 +4005,12 @@ dependencies = [ [[package]] name = "datafusion-doc" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" [[package]] name = "datafusion-execution" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "arrow-buffer 58.3.0", @@ -4032,7 +4032,7 @@ dependencies = [ [[package]] name = "datafusion-expr" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "async-trait", @@ -4054,7 +4054,7 @@ dependencies = [ [[package]] name = "datafusion-expr-common" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "datafusion-common", @@ -4066,7 +4066,7 @@ dependencies = [ [[package]] name = "datafusion-functions" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "arrow-buffer 58.3.0", @@ -4097,7 +4097,7 @@ dependencies = [ [[package]] name = "datafusion-functions-aggregate" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "ahash 0.8.12", "arrow 58.3.0", @@ -4118,7 +4118,7 @@ dependencies = [ [[package]] name = "datafusion-functions-aggregate-common" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "ahash 0.8.12", "arrow 58.3.0", @@ -4130,7 +4130,7 @@ dependencies = [ [[package]] name = "datafusion-functions-nested" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "arrow-ord 58.3.0", @@ -4154,7 +4154,7 @@ dependencies = [ [[package]] name = "datafusion-functions-table" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "async-trait", @@ -4169,7 +4169,7 @@ dependencies = [ [[package]] name = "datafusion-functions-window" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "datafusion-common", @@ -4186,7 +4186,7 @@ dependencies = [ [[package]] name = "datafusion-functions-window-common" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "datafusion-common", "datafusion-physical-expr-common", @@ -4195,7 +4195,7 @@ dependencies = [ [[package]] name = "datafusion-macros" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "datafusion-doc", "quote", @@ -4205,7 +4205,7 @@ dependencies = [ [[package]] name = "datafusion-optimizer" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "chrono", @@ -4254,7 +4254,7 @@ dependencies = [ [[package]] name = "datafusion-physical-expr" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "ahash 0.8.12", "arrow 58.3.0", @@ -4277,7 +4277,7 @@ dependencies = [ [[package]] name = "datafusion-physical-expr-adapter" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "datafusion-common", @@ -4291,7 +4291,7 @@ dependencies = [ [[package]] name = "datafusion-physical-expr-common" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "ahash 0.8.12", "arrow 58.3.0", @@ -4307,7 +4307,7 @@ dependencies = [ [[package]] name = "datafusion-physical-optimizer" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "datafusion-common", @@ -4325,7 +4325,7 @@ dependencies = [ [[package]] name = "datafusion-physical-plan" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "ahash 0.8.12", "arrow 58.3.0", @@ -4356,7 +4356,7 @@ dependencies = [ [[package]] name = "datafusion-proto" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "chrono", @@ -4383,7 +4383,7 @@ dependencies = [ [[package]] name = "datafusion-proto-common" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "datafusion-common", @@ -4393,7 +4393,7 @@ dependencies = [ [[package]] name = "datafusion-pruning" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "datafusion-common", @@ -4409,7 +4409,7 @@ dependencies = [ [[package]] name = "datafusion-session" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "async-trait", "datafusion-common", @@ -4422,7 +4422,7 @@ dependencies = [ [[package]] name = "datafusion-sql" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "arrow 58.3.0", "bigdecimal 0.4.8", @@ -4440,7 +4440,7 @@ dependencies = [ [[package]] name = "datafusion-substrait" version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=4c8a6bf28347b346348136247eb67dfb933f1147#4c8a6bf28347b346348136247eb67dfb933f1147" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a#a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" dependencies = [ "async-recursion", "async-trait", @@ -4630,10 +4630,23 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e7c1832837b905bbfb5101e07cc24c8deddf52f93225eee6ead5f4d63d53ddcb" dependencies = [ "const-oid 0.9.6", + "der_derive", + "flagset", "pem-rfc7468", "zeroize", ] +[[package]] +name = "der_derive" +version = "0.7.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8034092389675178f570469e6c3b0465d3d30b4505c294a6550db47f3c17ad18" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.117", +] + [[package]] name = "deranged" version = "0.5.8" @@ -5337,6 +5350,12 @@ version = "0.5.7" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1d674e81391d1e1ab681a28d99df07927c6d4aa5b027d7da16ba32d1d21ecd99" +[[package]] +name = "flagset" +version = "0.4.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b7ac824320a75a52197e8f2d787f6a38b6718bb6897a35142d749af3c0e8f4fe" + [[package]] name = "flatbuffers" version = "25.2.10" @@ -5979,7 +5998,7 @@ dependencies = [ "cfg-if", "js-sys", "libc", - "wasi", + "wasi 0.11.1+wasi-snapshot-preview1", "wasm-bindgen", ] @@ -7870,13 +7889,14 @@ checksum = "f9fbbcab51052fe104eb5e5d351cf728d30a5be1fe14d9be8a3b097481fb97de" [[package]] name = "libredox" -version = "0.1.4" +version = "0.1.23" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1580801010e535496706ba011c15f8532df6b42297d2e471fec38ceadd8c0638" +checksum = "8d8f1ea3f21fd3405dcaf6c9b5c1630af9afc422d9073ea39c5f6d6c772e08ed" dependencies = [ "bitflags 2.12.1", "libc", - "redox_syscall 0.5.13", + "plain", + "redox_syscall 0.9.3", ] [[package]] @@ -8546,7 +8566,7 @@ checksum = "02bd0af71c67b473010cbbc60715ee815645a4dc942899111f494b4b737d6fda" dependencies = [ "libc", "log", - "wasi", + "wasi 0.11.1+wasi-snapshot-preview1", "windows-sys 0.61.2", ] @@ -9331,6 +9351,15 @@ dependencies = [ "objc2-encode", ] +[[package]] +name = "objc2-core-foundation" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2a180dd8642fa45cdb7dd721cd4c11b1cadd4929ce112ebd8b9f5803cc79d536" +dependencies = [ + "bitflags 2.12.1", +] + [[package]] name = "objc2-encode" version = "4.1.0" @@ -9347,6 +9376,15 @@ dependencies = [ "objc2", ] +[[package]] +name = "objc2-system-configuration" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7216bd11cbda54ccabcab84d523dc93b858ec75ecfb3a7d89513fa22464da396" +dependencies = [ + "objc2-core-foundation", +] + [[package]] name = "object" version = "0.36.7" @@ -10259,6 +10297,9 @@ dependencies = [ "num-integer", "num-traits", "object_store", + "parquet-variant", + "parquet-variant-compute", + "parquet-variant-json", "paste", "seq-macro", "simdutf8", @@ -10576,16 +10617,7 @@ dependencies = [ "tokio", "tokio-rustls", "tokio-util", - "x509-certificate 0.25.0", -] - -[[package]] -name = "phf" -version = "0.11.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1fd6780a80ae0c52cc120a26a1a42c1ae51b247a253e4e06113d23d2c2edd078" -dependencies = [ - "phf_shared 0.11.3", + "x509-certificate", ] [[package]] @@ -10838,6 +10870,12 @@ version = "0.3.32" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7edddbd0b52d732b21ad9a5fab5c704c14cd949e5e9a1ec5929a24fded1b904c" +[[package]] +name = "plain" +version = "0.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b4596b6d070b27117e987119b4dac604f3c58cfb0b191112e24771b2faeac1a6" + [[package]] name = "plist" version = "1.7.2" @@ -10930,34 +10968,34 @@ dependencies = [ [[package]] name = "postgres-protocol" -version = "0.6.8" +version = "0.6.12" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "76ff0abab4a9b844b93ef7b81f1efc0a366062aaef2cd702c76256b5dc075c54" +checksum = "08808e3c483c46e999108051c78334f473d5adb59d78bb80a1268c7e6aa6c514" dependencies = [ "base64 0.22.1", "byteorder", "bytes", "fallible-iterator", - "hmac 0.12.1", - "md-5 0.10.6", + "hmac 0.13.0", + "md-5 0.11.0", "memchr", - "rand 0.9.4", - "sha2 0.10.9", + "rand 0.10.1", + "sha2 0.11.0", "stringprep", ] [[package]] name = "postgres-types" -version = "0.2.9" +version = "0.2.14" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "613283563cd90e1dfc3518d548caee47e0e725455ed619881f5cf21f36de4b48" +checksum = "851ca9db4932932d69f3ea811b1abe63087a0f740a47692619dd40d4899b68be" dependencies = [ "array-init", "bytes", "chrono", "fallible-iterator", "postgres-protocol", - "serde", + "serde_core", "serde_json", ] @@ -12047,6 +12085,15 @@ dependencies = [ "bitflags 2.12.1", ] +[[package]] +name = "redox_syscall" +version = "0.9.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d678d17679829e73d371e96880897e98fee2ded7acc0a50bdf8af2affa4b2fe5" +dependencies = [ + "bitflags 2.12.1", +] + [[package]] name = "redox_users" version = "0.4.6" @@ -14076,7 +14123,7 @@ dependencies = [ "stringprep", "thiserror 2.0.17", "tracing", - "whoami", + "whoami 1.6.0", ] [[package]] @@ -14115,7 +14162,7 @@ dependencies = [ "stringprep", "thiserror 2.0.17", "tracing", - "whoami", + "whoami 1.6.0", ] [[package]] @@ -15132,6 +15179,27 @@ version = "0.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" +[[package]] +name = "tls_codec" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0de2e01245e2bb89d6f05801c564fa27624dbd7b1846859876c7dad82e90bf6b" +dependencies = [ + "tls_codec_derive", + "zeroize", +] + +[[package]] +name = "tls_codec_derive" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2d2e76690929402faae40aebdda620a2c0e25dd6d3b9afe48867dfd95991f4bd" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.117", +] + [[package]] name = "tokio" version = "1.52.3" @@ -15198,9 +15266,9 @@ dependencies = [ [[package]] name = "tokio-postgres" -version = "0.7.13" +version = "0.7.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6c95d533c83082bb6490e0189acaa0bbeef9084e60471b696ca6988cd0541fb0" +checksum = "a528f7d280f6d5b9cd149635c8705b0dd049754bc67d81d31fa25169a93809d3" dependencies = [ "async-trait", "byteorder", @@ -15211,29 +15279,29 @@ dependencies = [ "log", "parking_lot 0.12.4", "percent-encoding", - "phf 0.11.3", + "phf 0.13.1", "pin-project-lite", "postgres-protocol", "postgres-types", - "rand 0.9.4", - "socket2 0.5.10", + "rand 0.10.1", + "socket2 0.6.4", "tokio", "tokio-util", - "whoami", + "whoami 2.1.3", ] [[package]] name = "tokio-postgres-rustls" -version = "0.12.0" +version = "0.14.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "04fb792ccd6bbcd4bba408eb8a292f70fc4a3589e5d793626f45190e6454b6ab" +checksum = "4c2ad44aa0ae96db89c4742212ed41645b2f597311ff6e1945542a4d9fadc2fb" dependencies = [ - "ring", "rustls", + "sha2 0.11.0", "tokio", "tokio-postgres", "tokio-rustls", - "x509-certificate 0.23.1", + "x509-cert", ] [[package]] @@ -16250,6 +16318,15 @@ version = "0.11.1+wasi-snapshot-preview1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" +[[package]] +name = "wasi" +version = "0.14.7+wasi-0.2.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "883478de20367e224c0090af9cf5f9fa85bed63a95c1abf3afc5c083ebc06e8c" +dependencies = [ + "wasip2", +] + [[package]] name = "wasip2" version = "1.0.2+wasi-0.2.9" @@ -16274,6 +16351,15 @@ version = "0.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b8dad83b4f25e74f184f64c43b150b91efe7647395b42289f38e50566d82855b" +[[package]] +name = "wasite" +version = "1.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "66fe902b4a6b8028a753d5424909b764ccf79b7a209eac9bf97e59cda9f71a42" +dependencies = [ + "wasi 0.14.7+wasi-0.2.4", +] + [[package]] name = "wasm-bindgen" version = "0.2.122" @@ -16484,7 +16570,19 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "6994d13118ab492c3c80c1f81928718159254c53c472bf9ce36f8dae4add02a7" dependencies = [ "redox_syscall 0.5.13", - "wasite", + "wasite 0.1.0", +] + +[[package]] +name = "whoami" +version = "2.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "626c4bac6755d76ffc12cb01b2eac751db1996b9e0041de9aa02c8c211ddc82c" +dependencies = [ + "libc", + "libredox", + "objc2-system-configuration", + "wasite 1.0.2", "web-sys", ] @@ -17093,22 +17191,15 @@ dependencies = [ ] [[package]] -name = "x509-certificate" -version = "0.23.1" +name = "x509-cert" +version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "66534846dec7a11d7c50a74b7cdb208b9a581cad890b7866430d438455847c85" +checksum = "1301e935010a701ae5f8655edc0ad17c44bad3ac5ce8c39185f75453b720ae94" dependencies = [ - "bcder", - "bytes", - "chrono", + "const-oid 0.9.6", "der", - "hex", - "pem", - "ring", - "signature", "spki", - "thiserror 1.0.69", - "zeroize", + "tls_codec", ] [[package]] diff --git a/Cargo.toml b/Cargo.toml index beb558d1e7..e38af7f227 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -251,7 +251,7 @@ strum = { version = "0.27", features = ["derive"] } sysinfo = "0.33" tempfile = "3" tokio = { version = "1.47", features = ["full"] } -tokio-postgres = "0.7" +tokio-postgres = "0.7.18" tokio-rustls = { version = "0.26.2", default-features = false } tokio-stream = "0.1" tokio-util = { version = "0.7", features = ["io-util", "compat"] } @@ -346,21 +346,21 @@ git = "https://github.com/GreptimeTeam/greptime-meter.git" rev = "5618e779cf2bb4755b499c630fba4c35e91898cb" [patch.crates-io] -datafusion = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } -datafusion-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } -datafusion-datasource = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } -datafusion-expr = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } -datafusion-expr-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } -datafusion-functions = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } -datafusion-functions-aggregate-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } -datafusion-functions-window-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } -datafusion-optimizer = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } -datafusion-physical-expr = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } -datafusion-physical-expr-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } -datafusion-physical-plan = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } -datafusion-proto = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } -datafusion-sql = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } -datafusion-substrait = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "4c8a6bf28347b346348136247eb67dfb933f1147" } +datafusion = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } +datafusion-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } +datafusion-datasource = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } +datafusion-expr = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } +datafusion-expr-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } +datafusion-functions = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } +datafusion-functions-aggregate-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } +datafusion-functions-window-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } +datafusion-optimizer = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } +datafusion-physical-expr = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } +datafusion-physical-expr-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } +datafusion-physical-plan = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } +datafusion-proto = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } +datafusion-sql = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } +datafusion-substrait = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "a281d9f0ea4b3fb2ec88bc3553d8dcef307e072a" } sqlparser = { git = "https://github.com/GreptimeTeam/sqlparser-rs.git", rev = "2aefa08a8d69c96eec2d6d6703598a009bba6e4c" } # on branch v0.61.x [profile.release] diff --git a/src/client/src/region.rs b/src/client/src/region.rs index f1730a93c8..2bb5f7dd0a 100644 --- a/src/client/src/region.rs +++ b/src/client/src/region.rs @@ -13,6 +13,7 @@ // limitations under the License. use std::sync::Arc; +use std::time::Duration; use api::region::RegionResponse; use api::v1::ResponseHeader; @@ -24,7 +25,7 @@ use arc_swap::ArcSwapOption; use arrow_flight::Ticket; use async_stream::stream; use async_trait::async_trait; -use common_error::ext::BoxedError; +use common_error::ext::{BoxedError, ErrorExt}; use common_error::status_code::StatusCode; use common_grpc::flight::{FlightDecoder, FlightMessage}; use common_meta::error::{self as meta_error, Result as MetaResult}; @@ -49,6 +50,8 @@ use crate::error::{ use crate::flight::decode_flight_data; use crate::{Client, metrics}; +const FLIGHT_DO_GET_TIMEOUT: Duration = Duration::from_secs(10); + #[derive(Debug)] pub struct RegionRequester { client: Client, @@ -109,20 +112,24 @@ impl RegionRequester { let mut flight_client = self .client .make_flight_client(self.send_compression, self.accept_compression)?; + // Limit Flight DoGet response time without limiting query stream execution. + let addr = flight_client.addr().to_string(); + let mut request = tonic::Request::new(ticket); + request.set_timeout(FLIGHT_DO_GET_TIMEOUT); let response = flight_client .mut_inner() - .do_get(ticket) + .do_get(request) .await .or_else(|e| { let tonic_code = e.code(); let e: error::Error = e.into(); error!( e; "Failed to do Flight get, addr: {}, code: {}", - flight_client.addr(), + addr, tonic_code ); Err(BoxedError::new(e)).with_context(|_| FlightGetSnafu { - addr: flight_client.addr().to_string(), + addr: addr.clone(), tonic_code, }) })?; @@ -133,7 +140,7 @@ impl RegionRequester { let flight_message_stream = flight_data_stream .filter_map(move |flight_data| decode_flight_data(&mut decoder, flight_data)); - recordbatches_from_flight_message_stream(flight_message_stream).await + recordbatches_from_flight_message_stream(addr, flight_message_stream).await } async fn handle_inner(&self, request: RegionRequest) -> Result { @@ -195,6 +202,7 @@ impl RegionRequester { } async fn recordbatches_from_flight_message_stream( + addr: String, mut flight_message_stream: S, ) -> Result where @@ -204,13 +212,17 @@ where return IllegalFlightMessagesSnafu { reason: "Expect the response not to be empty", } - .fail(); + .fail() + .map_err(|error| flight_stream_error(&addr, error)); }; - let FlightMessage::Schema(schema) = first_flight_message? else { + let FlightMessage::Schema(schema) = + first_flight_message.map_err(|e| flight_stream_error(&addr, e))? + else { return IllegalFlightMessagesSnafu { reason: "Expect schema to be the first flight message", } - .fail(); + .fail() + .map_err(|error| flight_stream_error(&addr, error)); }; let metrics = Arc::new(ArcSwapOption::from(None)); @@ -221,6 +233,7 @@ where let schema = Arc::new(datatypes::schema::Schema::try_from(schema).context(error::ConvertSchemaSnafu)?); let schema_cloned = schema.clone(); + let stream_addr = addr.clone(); let stream = Box::pin(stream!({ let _span = tracing_context.attach(common_telemetry::tracing::info_span!( "poll_flight_data_stream" @@ -240,7 +253,8 @@ where let flight_message = match flight_message_item { Some(Ok(message)) => message, Some(Err(e)) => { - yield Err(BoxedError::new(e)).context(ExternalSnafu); + yield Err(BoxedError::new(flight_stream_error(&stream_addr, e))) + .context(ExternalSnafu); break; } None => break, @@ -273,7 +287,8 @@ where break; } Err(e) => { - yield Err(BoxedError::new(e)).context(ExternalSnafu); + yield Err(BoxedError::new(flight_stream_error(&stream_addr, e))) + .context(ExternalSnafu); break; } } @@ -312,6 +327,23 @@ where Ok(Box::pin(record_batch_stream)) } +fn flight_stream_error(addr: &str, error: error::Error) -> error::Error { + let tonic_code = error.tonic_code().unwrap_or(tonic::Code::Unknown); + if error.status_code().should_log_error() { + error!( + error; "Failed to receive Flight data, addr: {}, code: {}", + addr, + tonic_code + ); + } + + error::Error::FlightGet { + addr: addr.to_string(), + tonic_code, + source: BoxedError::new(error), + } +} + pub fn build_remote_dyn_filter_update_request( query_id: impl Into, update: RemoteDynFilterUpdate, @@ -389,6 +421,44 @@ mod test { use super::*; use crate::Error::{self, IllegalDatabaseResponse, Server}; + #[test] + fn test_flight_stream_error_preserves_peer_address() { + let error = flight_stream_error( + "127.0.0.1:4001", + tonic::Status::unavailable("datanode unavailable").into(), + ); + + assert!(matches!( + error, + error::Error::FlightGet { + addr, + tonic_code: tonic::Code::Unavailable, + .. + } if addr == "127.0.0.1:4001" + )); + } + + #[tokio::test] + async fn test_empty_flight_stream_preserves_peer_address() { + let Err(error) = recordbatches_from_flight_message_stream( + "127.0.0.1:4001".to_string(), + stream::empty::>(), + ) + .await + else { + panic!("expected empty Flight stream to fail"); + }; + + assert!(matches!( + error, + error::Error::FlightGet { + addr, + tonic_code: tonic::Code::Unknown, + .. + } if addr == "127.0.0.1:4001" + )); + } + fn test_schema() -> Arc { Arc::new(Schema::new(vec![ColumnSchema::new( "v", @@ -509,11 +579,14 @@ mod test { ) .unwrap(); - let mut recordbatches = recordbatches_from_flight_message_stream(stream::iter(vec![ - Ok(FlightMessage::Schema(schema.arrow_schema().clone())), - Ok(FlightMessage::Metrics(test_metrics_json())), - Ok(FlightMessage::RecordBatch(batch.into_df_record_batch())), - ])) + let mut recordbatches = recordbatches_from_flight_message_stream( + "test-peer".to_string(), + stream::iter(vec![ + Ok(FlightMessage::Schema(schema.arrow_schema().clone())), + Ok(FlightMessage::Metrics(test_metrics_json())), + Ok(FlightMessage::RecordBatch(batch.into_df_record_batch())), + ]), + ) .await .unwrap(); @@ -528,11 +601,14 @@ mod test { #[tokio::test] async fn test_record_batch_stream_exposes_error_after_pre_batch_metrics() { let schema = test_schema(); - let mut recordbatches = recordbatches_from_flight_message_stream(stream::iter(vec![ - Ok(FlightMessage::Schema(schema.arrow_schema().clone())), - Ok(FlightMessage::Metrics(test_metrics_json())), - Err(Error::from(Status::internal("boom after metrics"))), - ])) + let mut recordbatches = recordbatches_from_flight_message_stream( + "test-peer".to_string(), + stream::iter(vec![ + Ok(FlightMessage::Schema(schema.arrow_schema().clone())), + Ok(FlightMessage::Metrics(test_metrics_json())), + Err(Error::from(Status::internal("boom after metrics"))), + ]), + ) .await .unwrap(); diff --git a/src/cmd/src/frontend.rs b/src/cmd/src/frontend.rs index 9d5d0bab1e..3058df6bea 100644 --- a/src/cmd/src/frontend.rs +++ b/src/cmd/src/frontend.rs @@ -434,6 +434,8 @@ impl StartCommand { // Some queries are expected to take long time. let mut channel_config = opts.datanode.client.channel_config(); channel_config.timeout = None; + // Source Flight streams and sink unary responses share pooled connections. + channel_config.http2_adaptive_window = Some(true); if opts.grpc.flight_compression.transport_compression() { channel_config.accept_compression = true; channel_config.send_compression = true; diff --git a/src/common/function/src/scalars/json/json_get.rs b/src/common/function/src/scalars/json/json_get.rs index ce49f9d59f..2357bc88f9 100644 --- a/src/common/function/src/scalars/json/json_get.rs +++ b/src/common/function/src/scalars/json/json_get.rs @@ -12,6 +12,7 @@ // See the License for the specific language governing permissions and // limitations under the License. +use std::borrow::Cow; use std::sync::Arc; use arrow::array::{ArrayRef, BinaryViewArray, new_null_array}; @@ -54,12 +55,17 @@ trait JsonGetResultBuilder { fn build(&mut self) -> ArrayRef; } -fn result_builder(len: usize, with_type: &DataType) -> Result> { +fn result_builder( + len: usize, + with_type: &DataType, + is_json2: bool, +) -> Result> { let builder = match with_type { - DataType::Utf8 | DataType::LargeUtf8 | DataType::Utf8View => { - Box::new(StringResultBuilder(StringViewBuilder::with_capacity(len))) - as Box - } + DataType::Utf8 | DataType::LargeUtf8 | DataType::Utf8View => Box::new(StringResultBuilder { + inner: StringViewBuilder::with_capacity(len), + is_json2, + }) + as Box, DataType::Int64 => Box::new(IntResultBuilder(Int64Builder::with_capacity(len))), DataType::Float64 => Box::new(FloatResultBuilder(Float64Builder::with_capacity(len))), DataType::Boolean => Box::new(BoolResultBuilder(BooleanBuilder::with_capacity(len))), @@ -71,20 +77,30 @@ fn result_builder(len: usize, with_type: &DataType) -> Result Result<()> { - self.0.append_option(jsonb::to_str(value).ok()); + // Scalar casts stay unquoted and map JSON null to SQL NULL; only containers + // use `to_string` to preserve their JSON representation. + let value = if self.is_json2 && (jsonb::is_array(value) || jsonb::is_object(value)) { + Some(jsonb::to_string(value)) + } else { + jsonb::to_str(value).ok() + }; + self.inner.append_option(value); Ok(()) } fn append_null(&mut self) { - self.0.append_null(); + self.inner.append_null(); } fn build(&mut self) -> ArrayRef { - Arc::new(self.0.finish()) + Arc::new(self.inner.finish()) } } @@ -404,18 +420,21 @@ impl Function for JsonGetWithType { let result = match arg0.data_type() { DataType::Binary | DataType::LargeBinary | DataType::BinaryView => { let arg0 = compute::cast(&arg0, &DataType::BinaryView)?; + let is_json2 = args.arg_fields.first().is_some_and(is_json2_extension_type); - if args.arg_fields.first().is_some_and(is_json2_extension_type) { - // Query concretization projects nested JSON2 paths as Struct arrays. A binary - // JSON2 argument is therefore an already-selected scalar or root value that - // only needs conversion from its JSONB representation to the requested type. + if is_json2 && path.trim_start_matches('$').split('.').all(str::is_empty) { JsonArray::from(&arg0) .project_to(&with_type) .map_err(|e| exec_datafusion_err!("{e:?}"))? } else { let jsons = arg0.as_binary_view(); - let mut builder = result_builder(len, &with_type)?; - jsonb_get(jsons, path, builder.as_mut())?; + let path = if is_json2 && !path.starts_with('$') { + Cow::Owned(format!("$.{path}")) + } else { + Cow::Borrowed(path) + }; + let mut builder = result_builder(len, &with_type, is_json2)?; + jsonb_get(jsons, &path, builder.as_mut())?; builder.build() } } @@ -511,6 +530,7 @@ mod tests { use datafusion_common::ScalarValue; use datafusion_common::arrow::array::{BinaryArray, BinaryViewArray, StringArray}; use datafusion_common::arrow::datatypes::{Float64Type, Int64Type}; + use datatypes::extension::json::Json2ExtensionType; use datatypes::types::parse_string_to_jsonb; use serde_json::json; @@ -566,6 +586,15 @@ mod tests { )) } + fn test_json_field(json: &ArrayRef, is_json2: bool) -> Arc { + let field = Field::new("json", json.data_type().clone(), true); + Arc::new(if is_json2 { + field.with_extension_type(Json2ExtensionType::default()) + } else { + field + }) + } + #[test] fn test_json_get_int() { let json_get_int = JsonGetInt::default(); @@ -850,7 +879,10 @@ mod tests { ColumnarValue::Array(json.clone()), ColumnarValue::Scalar(path.into()), ], - arg_fields: vec![], + arg_fields: vec![ + test_json_field(json, i >= json_strings.len()), + Arc::new(Field::new("path", DataType::Utf8, false)), + ], number_rows: 1, return_field: Arc::new(Field::new("x", DataType::Utf8View, false)), config_options: Arc::new(Default::default()), @@ -984,7 +1016,11 @@ mod tests { ColumnarValue::Scalar(path.into()), ColumnarValue::Scalar(ScalarValue::Utf8View(None)), ], - arg_fields: vec![], + arg_fields: vec![ + test_json_field(json, i >= json_strings.len()), + Arc::new(Field::new("path", DataType::Utf8, false)), + Arc::new(Field::new("with_type", DataType::Utf8View, true)), + ], number_rows: 1, return_field: Arc::new(Field::new("x", DataType::Utf8View, false)), config_options: Arc::new(Default::default()), diff --git a/src/common/meta/Cargo.toml b/src/common/meta/Cargo.toml index 8e0cbbdf35..8deb3d0081 100644 --- a/src/common/meta/Cargo.toml +++ b/src/common/meta/Cargo.toml @@ -89,7 +89,7 @@ strum.workspace = true table = { workspace = true, features = ["testing"] } tokio.workspace = true tokio-postgres = { workspace = true, optional = true } -tokio-postgres-rustls = { version = "0.12", optional = true } +tokio-postgres-rustls = { version = "0.14", optional = true } tonic.workspace = true tracing.workspace = true typetag.workspace = true diff --git a/src/common/recordbatch/src/cursor.rs b/src/common/recordbatch/src/cursor.rs index a741953ccc..125a68f4f1 100644 --- a/src/common/recordbatch/src/cursor.rs +++ b/src/common/recordbatch/src/cursor.rs @@ -12,6 +12,7 @@ // See the License for the specific language governing permissions and // limitations under the License. +use datatypes::schema::SchemaRef; use futures::StreamExt; use tokio::sync::Mutex; @@ -28,12 +29,15 @@ struct Inner { /// A cursor on RecordBatchStream that fetches data batch by batch pub struct RecordBatchStreamCursor { + schema: SchemaRef, inner: Mutex, } impl RecordBatchStreamCursor { pub fn new(stream: SendableRecordBatchStream) -> RecordBatchStreamCursor { + let schema = stream.schema(); Self { + schema, inner: Mutex::new(Inner { stream, current_row_index: 0, @@ -43,6 +47,10 @@ impl RecordBatchStreamCursor { } } + pub fn schema(&self) -> SchemaRef { + self.schema.clone() + } + /// Take `size` of row from the `RecordBatchStream` and create a new /// `RecordBatch` for these rows. pub async fn take(&self, size: usize) -> Result { diff --git a/src/common/time/src/timestamp.rs b/src/common/time/src/timestamp.rs index 7daa793baf..88cac7f4fc 100644 --- a/src/common/time/src/timestamp.rs +++ b/src/common/time/src/timestamp.rs @@ -483,6 +483,31 @@ impl Timestamp { ParseTimestampSnafu { raw: s }.fail() } + /// Interprets a timezone-less [`NaiveDateTime`] in the given timezone and + /// returns the corresponding timestamp. + /// + /// Datetimes that fall into a DST gap (a local time that does not exist) + /// are rejected, while ambiguous datetimes (from a repeated local time) + /// are resolved to the earlier instant. This policy is shared with + /// [`Timestamp::from_str`] so that the text and binary protocols interpret + /// datetimes consistently. + pub fn from_naive_datetime( + datetime: NaiveDateTime, + timezone: &Timezone, + ) -> crate::error::Result { + match datetime_to_utc(&datetime, timezone) { + LocalResult::Single(utc) | LocalResult::Ambiguous(utc, _) => { + Timestamp::from_chrono_datetime(utc).context(ParseTimestampSnafu { + raw: format!("{datetime} (timezone {timezone})"), + }) + } + LocalResult::None => ParseTimestampSnafu { + raw: format!("{datetime} (timezone {timezone})"), + } + .fail(), + } + } + pub fn negative(mut self) -> Self { self.value = -self.value; self @@ -531,12 +556,8 @@ fn naive_datetime_to_timestamp( .context(ParseTimestampSnafu { raw: s }); }; - match datetime_to_utc(&datetime, timezone) { - LocalResult::None => ParseTimestampSnafu { raw: s }.fail(), - LocalResult::Single(utc) | LocalResult::Ambiguous(utc, _) => { - Timestamp::from_chrono_datetime(utc).context(ParseTimestampSnafu { raw: s }) - } - } + Timestamp::from_naive_datetime(datetime, timezone) + .map_err(|_| ParseTimestampSnafu { raw: s }.build()) } impl From for Timestamp { @@ -919,6 +940,48 @@ mod tests { ); } + #[test] + fn test_from_naive_datetime() { + let datetime = NaiveDate::from_ymd_opt(2026, 8, 13) + .unwrap() + .and_hms_opt(8, 0, 0) + .unwrap(); + + // A fixed-offset timezone shifts the datetime by a constant amount. + let shanghai = Timezone::from_tz_string("Asia/Shanghai").unwrap(); + assert_eq!( + "2026-08-13 00:00:00", + Timestamp::from_naive_datetime(datetime, &shanghai) + .unwrap() + .to_chrono_datetime() + .unwrap() + .to_string() + ); + + // 2026-03-08 02:30 does not exist in America/New_York (DST gap). + let new_york = Timezone::from_tz_string("America/New_York").unwrap(); + let gap = NaiveDate::from_ymd_opt(2026, 3, 8) + .unwrap() + .and_hms_opt(2, 30, 0) + .unwrap(); + assert!(Timestamp::from_naive_datetime(gap, &new_york).is_err()); + + // 2026-11-01 01:30 is ambiguous in America/New_York; picks the first + // instant (EDT, UTC-4). + let ambiguous = NaiveDate::from_ymd_opt(2026, 11, 1) + .unwrap() + .and_hms_opt(1, 30, 0) + .unwrap(); + assert_eq!( + "2026-11-01 05:30:00", + Timestamp::from_naive_datetime(ambiguous, &new_york) + .unwrap() + .to_chrono_datetime() + .unwrap() + .to_string() + ); + } + #[test] fn test_to_iso8601_string() { set_default_timezone(Some("Asia/Shanghai")).unwrap(); diff --git a/src/common/time/src/timezone.rs b/src/common/time/src/timezone.rs index 41cc1f7842..dbd8dffcd4 100644 --- a/src/common/time/src/timezone.rs +++ b/src/common/time/src/timezone.rs @@ -108,6 +108,15 @@ impl Timezone { } } + /// A named zone merely sitting at +00:00 today is not UTC: it may have been + /// elsewhere at the timestamp being converted. + pub fn is_utc(&self) -> bool { + match self { + Self::Offset(offset) => offset.local_minus_utc() == 0, + Self::Named(tz) => matches!(tz, Tz::UTC), + } + } + /// Returns the number of seconds to add to convert from UTC to the local time. pub fn local_minus_utc(&self) -> i64 { match self { diff --git a/src/datanode/src/heartbeat/handler/open_region.rs b/src/datanode/src/heartbeat/handler/open_region.rs index 89ce857ce4..608fab0d90 100644 --- a/src/datanode/src/heartbeat/handler/open_region.rs +++ b/src/datanode/src/heartbeat/handler/open_region.rs @@ -98,6 +98,8 @@ mod tests { use std::collections::HashMap; use std::sync::Arc; + use common_error::ext::{BoxedError, RetryHint}; + use common_error::status_code::StatusCode; use common_meta::RegionIdent; use common_meta::heartbeat::handler::{HandleControl, HeartbeatResponseHandler}; use common_meta::heartbeat::mailbox::MessageMeta; @@ -105,14 +107,21 @@ mod tests { use common_meta::kv_backend::memory::MemoryKvBackend; use mito2::config::MitoConfig; use mito2::engine::MITO_ENGINE_NAME; + use mito2::error::ManifestDeltaNotFoundSnafu; use mito2::test_util::{CreateRequestBuilder, TestEnv}; + use object_store::{Error as ObjectStoreError, ErrorKind}; + use snafu::IntoError; use store_api::path_utils::table_dir; use store_api::region_request::{RegionCloseRequest, RegionRequest, RegionRequirements}; use store_api::storage::RegionId; - use crate::heartbeat::handler::RegionHeartbeatResponseHandler; + use super::OpenRegionsHandler; + use crate::error::{self, HandleRegionRequestSnafu}; use crate::heartbeat::handler::tests::HeartbeatResponseTestEnv; - use crate::tests::mock_region_server; + use crate::heartbeat::handler::{ + HandlerContext, InstructionHandler, RegionHeartbeatResponseHandler, + }; + use crate::tests::{MockRegionEngine, mock_region_server}; fn open_regions_instruction( region_ids: impl IntoIterator, @@ -198,4 +207,52 @@ mod tests { assert!(engine.is_region_exists(region_id)); assert!(engine.is_region_exists(region_id1)); } + + #[tokio::test] + async fn test_open_regions_preserves_manifest_delta_not_found_retry_hint() { + let retryable_region = RegionId::new(1024, 1); + let non_retryable_region = RegionId::new(1024, 2); + let (engine, _) = MockRegionEngine::with_mock_fn( + MITO_ENGINE_NAME, + Box::new(move |region_id, _request| { + if region_id == retryable_region { + let manifest_error = ManifestDeltaNotFoundSnafu { + version: 1_u64, + path: "manifest/00000000000000000001.json", + } + .into_error(ObjectStoreError::new( + ErrorKind::NotFound, + "mock listed manifest delta not found", + )); + return Err(HandleRegionRequestSnafu { region_id } + .into_error(BoxedError::new(manifest_error))); + } + + error::RegionNotFoundSnafu { region_id }.fail() + }), + ); + let mut region_server = mock_region_server(); + region_server.register_engine(engine); + let ctx = HandlerContext::new_for_test(region_server, Arc::new(MemoryKvBackend::new())); + let Instruction::OpenRegions(open_regions) = + open_regions_instruction([retryable_region, non_retryable_region], "test") + else { + unreachable!() + }; + + // Serial execution makes the first error selection deterministic. + let reply = OpenRegionsHandler { + open_region_parallelism: 1, + } + .handle(&ctx, open_regions) + .await + .unwrap() + .expect_open_regions_reply(); + + assert!(!reply.result); + let error = reply.error.unwrap(); + assert_eq!(StatusCode::StorageUnavailable, error.code); + assert_eq!(RetryHint::Retryable, error.retry_hint); + assert!(error.message.contains("00000000000000000001.json")); + } } diff --git a/src/datanode/src/heartbeat/handler/sync_region.rs b/src/datanode/src/heartbeat/handler/sync_region.rs index 4868754113..f8685f2847 100644 --- a/src/datanode/src/heartbeat/handler/sync_region.rs +++ b/src/datanode/src/heartbeat/handler/sync_region.rs @@ -103,11 +103,18 @@ impl SyncRegionHandler { mod tests { use std::sync::Arc; + use common_error::ext::{BoxedError, RetryHint}; + use common_error::status_code::StatusCode; use common_meta::kv_backend::memory::MemoryKvBackend; + use mito2::engine::MITO_ENGINE_NAME; + use mito2::error::ManifestDeltaNotFoundSnafu; + use object_store::{Error as ObjectStoreError, ErrorKind}; + use snafu::IntoError; use store_api::metric_engine_consts::METRIC_ENGINE_NAME; use store_api::region_engine::{RegionRole, SyncRegionFromRequest}; use store_api::storage::RegionId; + use crate::error::HandleRegionRequestSnafu; use crate::heartbeat::handler::sync_region::SyncRegionHandler; use crate::heartbeat::handler::{HandlerContext, InstructionHandler}; use crate::tests::{MockRegionEngine, mock_region_server}; @@ -201,4 +208,47 @@ mod tests { assert!(reply[0].ready); assert!(reply[0].error.is_none()); } + + #[tokio::test] + async fn test_handle_sync_region_preserves_manifest_delta_not_found_retry_hint() { + let mock_region_server = mock_region_server(); + let region_id = RegionId::new(1024, 1); + let (mock_engine, _) = MockRegionEngine::with_custom_apply_fn(MITO_ENGINE_NAME, |engine| { + engine.mock_role = Some(Some(RegionRole::Leader)); + engine.handle_sync_region_mock_fn = Some(Box::new(|region_id, _request| { + let manifest_error = ManifestDeltaNotFoundSnafu { + version: 1_u64, + path: "manifest/00000000000000000001.json", + } + .into_error(ObjectStoreError::new( + ErrorKind::NotFound, + "mock listed manifest delta not found", + )); + Err(HandleRegionRequestSnafu { region_id } + .into_error(BoxedError::new(manifest_error))) + })); + }); + mock_region_server.register_test_region(region_id, mock_engine); + + let handler_context = + HandlerContext::new_for_test(mock_region_server, Arc::new(MemoryKvBackend::new())); + let sync_region = common_meta::instruction::SyncRegion { + region_id, + request: SyncRegionFromRequest::from_manifest(Default::default()), + }; + + let reply = SyncRegionHandler + .handle(&handler_context, vec![sync_region]) + .await + .unwrap() + .expect_sync_regions_reply(); + + assert_eq!(1, reply.len()); + assert!(reply[0].exists); + assert!(!reply[0].ready); + let error = reply[0].error.as_ref().unwrap(); + assert_eq!(StatusCode::StorageUnavailable, error.code); + assert_eq!(RetryHint::Retryable, error.retry_hint); + assert!(error.message.contains("00000000000000000001.json")); + } } diff --git a/src/datanode/src/region_server.rs b/src/datanode/src/region_server.rs index 55dc6a670a..0d9398b088 100644 --- a/src/datanode/src/region_server.rs +++ b/src/datanode/src/region_server.rs @@ -66,7 +66,9 @@ use servers::error::{ self as servers_error, ExecuteGrpcRequestSnafu, Result as ServerResult, SuspendedSnafu, }; use servers::grpc::FlightCompression; -use servers::grpc::flight::{FlightCraft, FlightRecordBatchStream, TonicStream}; +use servers::grpc::flight::{ + FlightCraft, FlightRecordBatchSource, FlightRecordBatchStream, TonicStream, +}; use servers::grpc::region_server::RegionServerHandler; use session::context::{ FLIGHT_METRICS_HEARTBEAT_INTERVAL, QueryContext, QueryContextBuilder, QueryContextRef, @@ -976,13 +978,19 @@ impl FlightCraft for RegionServer { .map(|h| Arc::new(QueryContext::from(h))) .unwrap_or(QueryContext::arc()); - let result = self - .handle_remote_read(request, query_ctx.clone()) - .trace(tracing_context.attach(info_span!("RegionServer::handle_read"))) - .await?; + let region_server = self.clone(); + let initializer_query_ctx = query_ctx.clone(); + let initializer_tracing_context = tracing_context.clone(); + let initializer = async move { + region_server + .handle_remote_read(request, initializer_query_ctx) + .trace(initializer_tracing_context.attach(info_span!("RegionServer::handle_read"))) + .await + .map_err(Into::into) + }; let stream = Box::pin(FlightRecordBatchStream::new( - result, + FlightRecordBatchSource::initializer(initializer), tracing_context, self.flight_compression, query_ctx, @@ -1330,11 +1338,8 @@ impl RegionServerInner { } if !errors.is_empty() { - return error::UnexpectedSnafu { - // Returns the first error. - violated: format!("Failed to open batch regions: {:?}", errors[0]), - } - .fail(); + // Preserve the first region error so callers can honor its status code and retry hint. + return Err(errors.swap_remove(0)).context(HandleBatchOpenRequestSnafu); } Ok(open_regions) @@ -1979,7 +1984,7 @@ mod tests { use std::sync::Arc; use api::v1::{Rows, SemanticType}; - use common_error::ext::ErrorExt; + use common_error::ext::{ErrorExt, RetryHint}; use common_recordbatch::RecordBatches; use common_recordbatch::adapter::{RecordBatchMetrics, RegionWatermarkEntry}; use datatypes::prelude::{ConcreteDataType, VectorRef}; @@ -2514,7 +2519,9 @@ mod tests { ) .await .unwrap_err(); - assert_eq!(err.status_code(), StatusCode::Unexpected); + assert_matches!(&err, error::Error::HandleBatchOpenRequest { .. }); + assert_eq!(err.status_code(), StatusCode::RegionNotFound); + assert_eq!(err.retry_hint(), RetryHint::NonRetryable); } struct CurrentEngineTest { diff --git a/src/datatypes/src/error.rs b/src/datatypes/src/error.rs index b34a5ffa9b..48ced6a4ac 100644 --- a/src/datatypes/src/error.rs +++ b/src/datatypes/src/error.rs @@ -77,6 +77,13 @@ pub enum Error { location: Location, }, + #[snafu(display("Unimplemented: {feat}"))] + Unimplemented { + feat: String, + #[snafu(implicit)] + location: Location, + }, + #[snafu(display("Failed to parse version in schema meta, value: {}", value))] ParseSchemaVersion { value: String, @@ -203,6 +210,13 @@ pub enum Error { location: Location, }, + #[snafu(display("Invalid JSON2 settings: {reason}"))] + InvalidJson2Settings { + reason: String, + #[snafu(implicit)] + location: Location, + }, + #[snafu(display("Invalid Vector: {}", msg))] InvalidVector { msg: String, @@ -312,6 +326,7 @@ impl ErrorExt for Error { use Error::*; match self { UnsupportedOperation { .. } + | Unimplemented { .. } | UnsupportedArrowType { .. } | UnsupportedJsonType { .. } | UnsupportedDefaultExpr { .. } => StatusCode::Unsupported, @@ -326,6 +341,7 @@ impl ErrorExt for Error { | InvalidPrecisionOrScale { .. } | InvalidJson { .. } | InvalidJson2Layout { .. } + | InvalidJson2Settings { .. } | InvalidJsonb { .. } | InvalidVector { .. } | InvalidFulltextOption { .. } diff --git a/src/datatypes/src/extension/json.rs b/src/datatypes/src/extension/json.rs index ab4a138420..b912a0a3f4 100644 --- a/src/datatypes/src/extension/json.rs +++ b/src/datatypes/src/extension/json.rs @@ -129,27 +129,32 @@ pub struct JsonMetadata { } impl JsonMetadata { - /// Creates metadata for the legacy JSON2 layout. + /// Creates metadata for the JSON2 layout version 2. pub fn new(json_settings: JsonSettings) -> Self { - Self { - json_settings, - layout_version: None, - } - } - - /// Creates metadata for the new JSON2 physical v2 layout. - pub fn new_v2(json_settings: JsonSettings) -> Self { Self { json_settings, layout_version: Some(JSON2_LAYOUT_V2), } } + /// Creates metadata for the legacy JSON2 layout. + pub fn new_v1(json_settings: JsonSettings) -> Self { + Self { + json_settings, + layout_version: None, + } + } + /// Returns the JSON2 settings. pub fn json_settings(&self) -> &JsonSettings { &self.json_settings } + /// Consumes the metadata and returns its JSON2 settings. + pub fn into_json_settings(self) -> JsonSettings { + self.json_settings + } + /// Returns whether this metadata describes JSON2 layout version 2. pub fn is_version_2(&self) -> bool { self.layout_version == Some(JSON2_LAYOUT_V2) @@ -226,7 +231,7 @@ impl ExtensionType for Json2ExtensionType { })?; Ok(Arc::new(metadata)) } else { - Ok(Arc::new(JsonMetadata::default())) + Ok(Arc::new(JsonMetadata::new_v1(JsonSettings::default()))) } } @@ -406,7 +411,7 @@ mod tests { let legacy: JsonMetadata = serde_json::from_str(r#"{"json_settings":{}}"#)?; assert!(!legacy.is_version_2()); - let metadata = JsonMetadata::new_v2(JsonSettings::default()); + let metadata = JsonMetadata::new(JsonSettings::default()); assert!(metadata.is_version_2()); let serialized = serde_json::to_string(&metadata)?; let deserialized: JsonMetadata = serde_json::from_str(&serialized)?; @@ -418,7 +423,9 @@ mod tests { #[test] fn test_parse_json2_physical_layout() -> crate::error::Result<()> { let legacy = Field::new("data", DataType::Struct(Fields::empty()), true) - .with_extension_type(Json2ExtensionType::default()); + .with_extension_type(Json2ExtensionType::new(Arc::new(JsonMetadata::new_v1( + JsonSettings::default(), + )))); assert!(!Json2PhysicalLayout::try_from_root(&legacy)?.is_version_2()); let v2 = Field::new( @@ -432,7 +439,7 @@ mod tests { ), true, ) - .with_extension_type(Json2ExtensionType::new(Arc::new(JsonMetadata::new_v2( + .with_extension_type(Json2ExtensionType::new(Arc::new(JsonMetadata::new( JsonSettings::default(), )))); assert!(Json2PhysicalLayout::try_from_root(&v2)?.is_version_2()); @@ -447,7 +454,7 @@ mod tests { assert!(Json2PhysicalLayout::try_from_root(&field).is_err()); let metadata = - Json2ExtensionType::new(Arc::new(JsonMetadata::new_v2(JsonSettings::default()))); + Json2ExtensionType::new(Arc::new(JsonMetadata::new(JsonSettings::default()))); let missing = Field::new("data", DataType::Struct(Fields::empty()), true) .with_extension_type(metadata.clone()); assert!(json2_remainder_field(&missing).is_ok_and(|x| x.is_none())); diff --git a/src/datatypes/src/json.rs b/src/datatypes/src/json.rs index ec9ca4ac78..0afe563734 100644 --- a/src/datatypes/src/json.rs +++ b/src/datatypes/src/json.rs @@ -29,7 +29,7 @@ use serde_json::{Map, Value as Json}; use snafu::ResultExt; use crate::data_type::ConcreteDataType; -use crate::error::{self, InvalidJson2LayoutSnafu, Result, UnsupportedJsonTypeSnafu}; +use crate::error::{self, InvalidJson2SettingsSnafu, Result, UnsupportedJsonTypeSnafu}; use crate::json::value::{JsonValue, JsonVariant, encode_serde_json_as_jsonb}; use crate::schema::ColumnDefaultConstraint; use crate::types::json_type::{JsonNativeType, JsonObjectType}; @@ -39,6 +39,8 @@ use crate::value::{ListValue, StructValue, Value}; pub const JSON2_MAX_STRUCTURED_DEPTH: usize = 50; /// Reserved physical field containing unexpanded JSON2 paths. pub const JSON2_REMAINDER_FIELD_NAME: &str = "!__remainder__!"; +/// Default maximum number of unhinted JSON leaf paths expanded into Arrow fields. +pub const JSON2_DEFAULT_MAX_AUTO_EXPANDED_PATHS: u32 = 100; /// JSON2 settings stored in column schema metadata and represented through /// Arrow extension metadata. @@ -105,6 +107,14 @@ pub struct JsonContext<'a> { } impl JsonSettings { + /// Creates default v2 settings for newly created JSON2 columns. + pub fn new_v2() -> Self { + Self { + type_hints: vec![], + max_auto_expanded_paths: Some(JSON2_DEFAULT_MAX_AUTO_EXPANDED_PATHS), + } + } + /// Creates and validates JSON2 settings. pub fn try_new( type_hints: Vec, @@ -140,6 +150,16 @@ impl JsonSettings { /// Encode a serde_json::Value into a Value::Json using current settings. pub fn encode(&self, json: Json) -> Result { + if let Json::Object(object) = &json + && object.contains_key(JSON2_REMAINDER_FIELD_NAME) + { + return error::InvalidJsonSnafu { + value: format!( + "root object cannot contain reserved field '{JSON2_REMAINDER_FIELD_NAME}'" + ), + } + .fail(); + } let mut context = JsonContext { path: Vec::new(), settings: self, @@ -149,10 +169,14 @@ impl JsonSettings { } fn validate_type_hints(type_hints: &[JsonTypeHint]) -> Result<()> { + if type_hints.is_empty() { + return Ok(()); + } + let mut object = JsonObjectType::new(); for hint in type_hints { if hint.path.len() > JSON2_MAX_STRUCTURED_DEPTH { - return InvalidJson2LayoutSnafu { + return InvalidJson2SettingsSnafu { reason: format!( "JSON2 type hint path cannot exceed {JSON2_MAX_STRUCTURED_DEPTH} segments" ), @@ -164,7 +188,7 @@ fn validate_type_hints(type_hints: &[JsonTypeHint]) -> Result<()> { .first() .is_some_and(|x| x == JSON2_REMAINDER_FIELD_NAME) { - return InvalidJson2LayoutSnafu { + return InvalidJson2SettingsSnafu { reason: format!( "JSON2 type hint path cannot be rooted at reserved field '{JSON2_REMAINDER_FIELD_NAME}'" ), @@ -183,14 +207,30 @@ fn validate_type_hints(type_hints: &[JsonTypeHint]) -> Result<()> { | ConcreteDataType::Int64(_) | ConcreteDataType::Float32(_) | ConcreteDataType::Float64(_) - | ConcreteDataType::String(_) => (&hint.data_type).into(), + | ConcreteDataType::String(_) + | ConcreteDataType::List(_) + | ConcreteDataType::Struct(_) => (&hint.data_type).into(), data_type => { - return InvalidJson2LayoutSnafu { + return InvalidJson2SettingsSnafu { reason: format!("unsupported JSON2 type hint data type: {data_type}"), } .fail(); } }; + let non_finite_default = match &hint.default_constraint { + Some(ColumnDefaultConstraint::Value(Value::Float32(value))) => !value.0.is_finite(), + Some(ColumnDefaultConstraint::Value(Value::Float64(value))) => !value.0.is_finite(), + _ => false, + }; + if non_finite_default { + return InvalidJson2SettingsSnafu { + reason: format!( + "JSON2 type hint default for '{}' must be finite", + hint.path.join(".") + ), + } + .fail(); + } validate_type_hint(&mut object, &hint.path, data_type)?; } Ok(()) @@ -202,14 +242,14 @@ fn validate_type_hint( data_type: JsonNativeType, ) -> Result<()> { let Some((name, path)) = path.split_first() else { - return InvalidJson2LayoutSnafu { + return InvalidJson2SettingsSnafu { reason: "JSON2 type hint path must not be empty".to_string(), } .fail(); }; if path.is_empty() { if object.insert(name.clone(), data_type).is_some() { - return InvalidJson2LayoutSnafu { + return InvalidJson2SettingsSnafu { reason: format!("duplicate JSON2 type hint path '{name}'"), } .fail(); @@ -221,7 +261,7 @@ fn validate_type_hint( .entry(name.clone()) .or_insert_with(|| JsonNativeType::Object(JsonObjectType::new())); let JsonNativeType::Object(child) = child else { - return InvalidJson2LayoutSnafu { + return InvalidJson2SettingsSnafu { reason: format!("conflicting JSON2 type hint path at '{name}'"), } .fail(); @@ -702,7 +742,7 @@ mod tests { } #[test] - fn test_json_settings_reject_invalid_type_hint_layout() { + fn test_json_settings_reject_invalid_type_hints() { for type_hints in [ json!([{"path": [], "type": {"Int64": {}}, "nullable": true, "inverted_index": false}]), json!([ @@ -777,6 +817,21 @@ mod tests { } } + #[test] + fn test_encode_rejects_reserved_remainder_field() -> Result<()> { + let settings = JsonSettings::default(); + let err = settings + .encode(json!({"!__remainder__!": "user-value"})) + .unwrap_err(); + assert!( + err.to_string() + .contains("root object cannot contain reserved field") + ); + + settings.encode(json!({"nested": {"!__remainder__!": "user-value"}}))?; + Ok(()) + } + #[test] fn test_encode_json_object() { let json = json!({ diff --git a/src/datatypes/src/json/value.rs b/src/datatypes/src/json/value.rs index 728aadeb28..7bcb1ecc2b 100644 --- a/src/datatypes/src/json/value.rs +++ b/src/datatypes/src/json/value.rs @@ -30,6 +30,8 @@ use crate::types::json_type::{JsonNativeType, JsonNumberType, is_include}; use crate::types::{StructField, StructType}; use crate::value::{ListValue, StructValue, Value}; +pub type JsonObjectVariant = BTreeMap; + /// Number in json, can be a positive integer, a negative integer, or a floating number. /// Each of which is represented as `u64`, `i64` and `f64`. /// @@ -124,7 +126,7 @@ pub enum JsonVariant { Number(JsonNumber), String(String), Array(Vec), - Object(BTreeMap), + Object(JsonObjectVariant), /// A special "variant" value of JSON, to represent a union result of conflict JSON type values. Variant(Vec), } @@ -160,7 +162,8 @@ impl JsonVariant { } } - fn contains_empty_object(&self) -> bool { + /// Returns whether this value recursively contains an empty object. + pub(crate) fn contains_empty_object(&self) -> bool { match self { JsonVariant::Array(array) => array.iter().any(JsonVariant::contains_empty_object), JsonVariant::Object(object) => { @@ -721,12 +724,7 @@ where I: IntoIterator, K: Into, { - let mut fields = fields.into_iter().peekable(); - if fields.peek().is_none() { - JsonNativeType::Null - } else { - JsonNativeType::Object(fields.map(|(k, v)| (k.into(), v)).collect()) - } + JsonNativeType::Object(fields.into_iter().map(|(k, v)| (k.into(), v)).collect()) } impl From<()> for JsonVariantRef<'_> { @@ -972,11 +970,10 @@ mod tests { ]))) ); - // Empty objects have native type Null, but the value still needs alignment - // before converting into a typed struct value. + // Empty objects have an empty Object type and remain distinct from Null. let expected = JsonNativeType::Object(JsonObjectType::from([( "empty".to_string(), - JsonNativeType::Null, + JsonNativeType::Object(JsonObjectType::default()), )])); let mut value = parse_json_value(r#"{"empty":{}}"#); assert_eq!(value.json_type(), &expected); @@ -985,7 +982,7 @@ mod tests { value, JsonValue::from(JsonVariant::Object(BTreeMap::from([( "empty".to_string(), - JsonVariant::Null, + JsonVariant::Object(BTreeMap::default()), )]))) ); diff --git a/src/datatypes/src/schema/column_schema.rs b/src/datatypes/src/schema/column_schema.rs index 6efab380e0..a71605afdb 100644 --- a/src/datatypes/src/schema/column_schema.rs +++ b/src/datatypes/src/schema/column_schema.rs @@ -28,10 +28,12 @@ use crate::data_type::{ConcreteDataType, DataType}; use crate::error::{ self, ArrowMetadataSnafu, Error, InvalidFulltextOptionSnafu, ParseExtendedTypeSnafu, Result, }; +use crate::extension::json::Json2ExtensionType; use crate::schema::TYPE_KEY; use crate::schema::constraint::ColumnDefaultConstraint; use crate::value::Value; -use crate::vectors::VectorRef; +use crate::vectors::json::builder::JsonVectorBuilder; +use crate::vectors::{MutableVector, VectorRef}; pub type Metadata = HashMap; @@ -128,6 +130,21 @@ impl ColumnSchema { } } + /// Creates a mutable vector using this column's extension metadata. + pub fn create_mutable_vector(&self, capacity: usize) -> Box { + if self.data_type.is_json2() + && let Some(extension) = self.extension_type::().ok().flatten() + && extension.metadata().is_version_2() + { + Box::new(JsonVectorBuilder::with_settings( + extension.metadata().json_settings(), + capacity, + )) + } else { + self.data_type.create_mutable_vector(capacity) + } + } + #[inline] pub fn is_time_index(&self) -> bool { self.is_time_index diff --git a/src/datatypes/src/types/json_type.rs b/src/datatypes/src/types/json_type.rs index 7905fce76b..77e67d9c6a 100644 --- a/src/datatypes/src/types/json_type.rs +++ b/src/datatypes/src/types/json_type.rs @@ -50,6 +50,11 @@ pub enum JsonNumberType { #[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize, Default)] pub enum JsonNativeType { + /// JSON null value type. + /// + /// This variant may also appear as the initial state while merging inferred + /// types, but it does not represent an empty object. Empty objects are + /// represented as `Object({})`. #[default] Null, Bool, @@ -80,7 +85,7 @@ impl JsonNativeType { Self::Number(JsonNumberType::F64) } - fn object() -> Self { + pub fn object() -> Self { Self::Object(JsonObjectType::new()) } @@ -143,6 +148,14 @@ impl JsonNativeType { JsonNativeType::Variant => ArrowDataType::Binary, } } + + /// Returns whether this type is a boolean, number, or string scalar. + pub fn is_primitive(&self) -> bool { + matches!( + self, + JsonNativeType::Bool | JsonNativeType::Number(_) | JsonNativeType::String + ) + } } impl From<&ConcreteDataType> for JsonNativeType { @@ -360,6 +373,7 @@ impl DataType for JsonType { fn create_mutable_vector(&self, capacity: usize) -> Box { match &self.format { JsonFormat::Jsonb => Box::new(BinaryVectorBuilder::with_capacity(capacity)), + // TODO(LFC): Carry JsonSettings in JsonFormat::Json2 and use with_settings here. JsonFormat::Json2(x) => Box::new(JsonVectorBuilder::new(x.as_ref().clone(), capacity)), } } diff --git a/src/datatypes/src/vectors/helper.rs b/src/datatypes/src/vectors/helper.rs index b2689eee2c..685b07419f 100644 --- a/src/datatypes/src/vectors/helper.rs +++ b/src/datatypes/src/vectors/helper.rs @@ -87,22 +87,6 @@ impl Helper { }) } - pub fn check_get_mutable_vector( - vector: &mut dyn MutableVector, - ) -> Result<&mut T> { - let ty = vector.data_type(); - vector - .as_mut_any() - .downcast_mut() - .with_context(|| error::UnknownVectorSnafu { - msg: format!( - "downcast vector error, vector type: {:?}, expected vector: {:?}", - ty, - std::any::type_name::(), - ), - }) - } - pub fn check_get_scalar_vector( vector: &VectorRef, ) -> Result<&::VectorType> { diff --git a/src/datatypes/src/vectors/json.rs b/src/datatypes/src/vectors/json.rs index c68e6cf3d9..958a98d9eb 100644 --- a/src/datatypes/src/vectors/json.rs +++ b/src/datatypes/src/vectors/json.rs @@ -15,3 +15,5 @@ pub mod array; pub(crate) mod builder; pub mod variant; + +pub use builder::json2_physical_data_type; diff --git a/src/datatypes/src/vectors/json/array.rs b/src/datatypes/src/vectors/json/array.rs index 0426334c3d..153d6b490e 100644 --- a/src/datatypes/src/vectors/json/array.rs +++ b/src/datatypes/src/vectors/json/array.rs @@ -33,9 +33,12 @@ use crate::error::{ AlignJsonArraySnafu, ArrowComputeSnafu, InvalidJsonSnafu, InvalidJsonbSnafu, Result, }; use crate::extension::json::{JSON2_REMAINDER_FIELD_NAME, json2_remainder_field}; +use crate::json::JsonSettings; use crate::json::value::{decode_json_variant, encode_serde_json_as_jsonb}; use crate::prelude::{DataType as _, Value as GreptimeValue}; use crate::value::{ListValue, StructValue}; +use crate::vectors::MutableVector; +use crate::vectors::json::builder::{JsonVectorBuilder, json2_physical_data_type}; use crate::vectors::json::variant::variant_to_json_values; pub struct JsonArray<'a> { @@ -115,6 +118,38 @@ impl JsonArray<'_> { } } + /// Rewrites a JSON2 array from the current physical layout into the specified + /// v2 physical layout. + pub fn rewrite_to_v2( + &self, + field: &Field, + logical_settings: &JsonSettings, + target_layout: &JsonSettings, + ) -> Result { + let is_v2 = json2_remainder_field(field)?.is_some(); + if is_v2 && self.inner.data_type() == &json2_physical_data_type(target_layout) { + return Ok(self.inner.clone()); + } + + let values = if is_v2 { + self.json2_values()? + } else { + (0..self.inner.len()) + .map(|i| self.try_get_value(i)) + .collect::>>()? + }; + let mut builder = JsonVectorBuilder::with_settings(target_layout, values.len()); + for value in values { + if value.is_null() { + builder.push_null(); + } else { + let value = logical_settings.encode(value)?; + builder.try_push_value_ref(&value.as_value_ref())?; + } + } + Ok(builder.to_vector().to_arrow_array()) + } + fn json2_values(&self) -> Result> { let structs = self.inner.as_struct_opt().context(AlignJsonArraySnafu { reason: "JSON2 layout v2 root array must be a struct", @@ -149,7 +184,14 @@ impl JsonArray<'_> { if child.name() == JSON2_REMAINDER_FIELD_NAME { continue; } - let value = JsonArray::from(column).try_get_value(i)?; + let mut value = JsonArray::from(column).try_get_value(i)?; + // Arrow child nulls cannot distinguish a missing path from an explicit JSON + // null. Builders preserve explicit null presence in the remainder, so nulls + // from the explicit branch must be discarded before merging both branches. + remove_null_object_fields(&mut value); + if value.is_null() { + continue; + } merge_explicit_value(&mut object, child.name().clone(), value, &mut path)?; } values.push(Value::Object(object)); @@ -420,6 +462,16 @@ fn merge_explicit_value( Ok(()) } +fn remove_null_object_fields(value: &mut Value) { + let Value::Object(object) = value else { + return; + }; + object.retain(|_, value| { + remove_null_object_fields(value); + !value.is_null() + }); +} + /// Returns whether Arrow can cast between the types without JSON-aware projection. /// Binary and nested types require JSONB decoding or recursive projection. fn can_fast_cast_types(from_type: &DataType, to_type: &DataType) -> bool { @@ -545,6 +597,8 @@ impl<'a> From<&'a ArrayRef> for JsonArray<'a> { #[cfg(test)] mod test { + use std::sync::Arc; + use arrow_array::types::Int64Type; use arrow_array::{ BinaryArray, BooleanArray, Float32Array, Float64Array, Int8Array, Int16Array, Int32Array, @@ -555,7 +609,7 @@ mod test { use super::*; use crate::extension::json::{Json2ExtensionType, JsonMetadata}; - use crate::json::JsonSettings; + use crate::json::{JsonSettings, JsonTypeHint}; use crate::vectors::json::variant::{json_values_to_variant, variant_field}; #[test] @@ -1049,7 +1103,7 @@ mod test { None, )); let field = Field::new("data", DataType::Struct(fields), true).with_extension_type( - Json2ExtensionType::new(Arc::new(JsonMetadata::new_v2(JsonSettings::default()))), + Json2ExtensionType::new(Arc::new(JsonMetadata::new(JsonSettings::default()))), ); assert_eq!( @@ -1063,12 +1117,10 @@ mod test { assert_eq!( json!({ "!__remainder__!": "user value", - "count": null, - "nested": {"left": null} + "nested": {} }), JsonArray::from(&array).json2_values()?[1] ); - let target = DataType::Struct( vec![ Arc::new(Field::new("cold", DataType::UInt64, true)), @@ -1088,6 +1140,38 @@ mod test { Ok(()) } + #[test] + fn test_rewrite_to_v2_reuses_matching_layout() -> Result<()> { + let settings = JsonSettings::try_new( + vec![JsonTypeHint { + path: vec!["kind".to_string()], + data_type: ConcreteDataType::string_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }], + Some(0), + )?; + let value = settings.encode(json!({"kind": "access", "cold": 1}))?; + let mut builder = JsonVectorBuilder::with_settings(&settings, 1); + builder.try_push_value_ref(&value.as_value_ref())?; + let array = builder.to_vector().to_arrow_array(); + let structs = array.as_struct(); + assert!(structs.column_by_name("kind").is_some()); + assert_eq!( + vec![Some(json!({"cold": 1}))], + variant_to_json_values(structs.column_by_name(JSON2_REMAINDER_FIELD_NAME).unwrap())? + ); + let field = Field::new("data", array.data_type().clone(), true).with_extension_type( + Json2ExtensionType::new(Arc::new(JsonMetadata::new(settings.clone()))), + ); + + let rewritten = JsonArray::from(&array).rewrite_to_v2(&field, &settings, &settings)?; + + assert!(Arc::ptr_eq(&array, &rewritten)); + Ok(()) + } + #[test] fn test_project_partial_json2_v2_without_remainder() -> Result<()> { let fields = Fields::from(vec![Arc::new(Field::new("hot", DataType::Int64, true))]); @@ -1097,7 +1181,7 @@ mod test { None, )); let field = Field::new("data", DataType::Struct(fields), true).with_extension_type( - Json2ExtensionType::new(Arc::new(JsonMetadata::new_v2(JsonSettings::default()))), + Json2ExtensionType::new(Arc::new(JsonMetadata::new(JsonSettings::default()))), ); let projected = JsonArray::from(&array).project_to_v2(&field, field.data_type())?; @@ -1124,6 +1208,18 @@ mod test { ) ); + let Value::Object(mut remainder) = json!({"count": 1}) else { + unreachable!(); + }; + let error = merge_explicit_value( + &mut remainder, + "count".to_string(), + json!(1), + &mut Vec::new(), + ) + .unwrap_err(); + assert!(error.to_string().contains("cannot merge 'count'")); + let Value::Object(mut remainder) = json!({"nested": {"count": 1}}) else { unreachable!(); }; diff --git a/src/datatypes/src/vectors/json/builder.rs b/src/datatypes/src/vectors/json/builder.rs index e6ceb6d6a5..3a9c158796 100644 --- a/src/datatypes/src/vectors/json/builder.rs +++ b/src/datatypes/src/vectors/json/builder.rs @@ -13,72 +13,762 @@ // limitations under the License. use std::any::Any; +use std::collections::{BTreeMap, HashMap}; use std::sync::Arc; +use arrow_array::cast::AsArray; +use arrow_array::{Array, ArrayRef, StructArray}; use arrow_schema::DataType; +use parquet_variant_compute::VariantArrayBuilder; +use snafu::{ResultExt, ensure}; use crate::data_type::ConcreteDataType; -use crate::error::{Result, TryFromValueSnafu, UnexpectedSnafu, UnsupportedOperationSnafu}; -use crate::json::value::{JsonNumber, JsonVariant, encode_json_variant}; +use crate::error::{ + ArrowComputeSnafu, Result, TryFromValueSnafu, UnexpectedSnafu, UnimplementedSnafu, + UnsupportedOperationSnafu, +}; +use crate::extension::json::JSON2_REMAINDER_FIELD_NAME; +use crate::json::value::{JsonNumber, JsonVariant, JsonVariantRef, encode_json_variant}; +use crate::json::{ + JSON2_DEFAULT_MAX_AUTO_EXPANDED_PATHS, JSON2_MAX_STRUCTURED_DEPTH, JsonSettings, +}; use crate::prelude::{ValueRef, Vector, VectorRef}; use crate::types::StructType; use crate::types::json_type::{JsonNativeType, is_include}; -use crate::value::{ListValue, StructValue, StructValueRef, Value}; -use crate::vectors::{MutableVector, StructVectorBuilder}; +use crate::value::{ListValue, ListValueRef, StructValue, StructValueRef, Value}; +use crate::vectors::json::variant::{append_json_variant, append_json_variant_ref, variant_field}; +use crate::vectors::{Helper, MutableVector, NullVector, StructVectorBuilder}; -#[derive(Clone)] +type JsonObjectValue = BTreeMap; + +/// Builds JSON2 vectors from object values. +/// +/// Legacy mode merges all observed paths into the explicit Struct schema. +/// Auto-expanding mode always materializes type-hinted paths, selects up to +/// `max_auto_expanded_paths` compatible unhinted leaf paths by frequency, and stores +/// conflicting or unselected paths in the Variant remainder field. pub(crate) struct JsonVectorBuilder { - merged_type: JsonNativeType, - values: Vec, + state: JsonVectorBuilderState, +} + +enum JsonVectorBuilderState { + Legacy { + merged_type: JsonNativeType, + values: Vec, + }, + ExplicitOnly { + /// Paths declared by type hints and stored as dedicated Struct fields. + explicit_type: JsonNativeType, + /// Concrete Struct type used to append explicit values without buffering rows. + struct_type: StructType, + /// Builder for values selected by the explicit type hints. + explicit: StructVectorBuilder, + /// Builder for all values outside the explicit type hints. + remainder: VariantArrayBuilder, + }, + AutoExpanding { + /// Paths declared by type hints and always stored as dedicated Struct fields. + explicit_type: JsonNativeType, + /// Maximum number of additional paths selected from buffered values. + max_auto_expanded_paths: u32, + /// Buffered values used to infer auto-expanded paths before building the vector. + values: Vec, + }, +} + +impl JsonVectorBuilderState { + fn native_type(&self) -> JsonNativeType { + match self { + Self::Legacy { merged_type, .. } => merged_type.clone(), + Self::ExplicitOnly { explicit_type, .. } => explicit_type.clone(), + Self::AutoExpanding { + explicit_type, + max_auto_expanded_paths, + values, + } => infer_expanded_type(explicit_type, *max_auto_expanded_paths, values), + } + } + + fn len(&self) -> usize { + match self { + Self::Legacy { values, .. } | Self::AutoExpanding { values, .. } => values.len(), + Self::ExplicitOnly { explicit, .. } => explicit.len(), + } + } + + fn try_build(&mut self) -> Result { + match self { + Self::Legacy { + merged_type, + values, + } => build_legacy(values, merged_type), + Self::ExplicitOnly { + explicit, + remainder, + .. + } => { + let remainder = std::mem::replace(remainder, VariantArrayBuilder::new(0)).build(); + finish_vector(explicit.to_vector(), ArrayRef::from(remainder)) + } + Self::AutoExpanding { + explicit_type, + max_auto_expanded_paths, + values, + } => { + let expanded_type = + infer_expanded_type(explicit_type, *max_auto_expanded_paths, values); + build_with_remainder(values, &expanded_type) + } + } + } + + fn try_build_cloned(&self) -> Result { + let mut state = match self { + Self::Legacy { + merged_type, + values, + } => Self::Legacy { + merged_type: merged_type.clone(), + values: values.clone(), + }, + Self::AutoExpanding { + explicit_type, + max_auto_expanded_paths, + values, + } => Self::AutoExpanding { + explicit_type: explicit_type.clone(), + max_auto_expanded_paths: *max_auto_expanded_paths, + values: values.clone(), + }, + // Only TimeSeriesMemtable requires a non-consuming snapshot, while JSON2 targets + // BulkMemtable. We've tried our best to support it above, but if this match arm does + // not, it's OK. The only reason it doesn't is because of `VariantArrayBuilder`. We'll + // track the upstream and see. + Self::ExplicitOnly { .. } => { + return UnimplementedSnafu { + feat: "no auto expanded JSON2 array builder", + } + .fail(); + } + }; + state.try_build() + } + + fn try_push_value_ref(&mut self, value: &ValueRef) -> Result<()> { + if matches!(value, ValueRef::Null) { + self.push_null(); + return Ok(()); + } + let ValueRef::Json(value) = value else { + return TryFromValueSnafu { + reason: format!("expected JSON value, got {value:?}"), + } + .fail(); + }; + ensure!( + value.is_object() || value.is_null(), + TryFromValueSnafu { + reason: format!("expected JSON object value, got {value:?}"), + } + ); + match self { + Self::Legacy { + merged_type, + values, + } => { + let json_type = value.json_type(); + if !is_include(merged_type, json_type.as_ref()) { + merged_type.merge(json_type.as_ref()); + } + values.push(JsonVariant::from(value.variant())); + } + Self::ExplicitOnly { + explicit_type, + struct_type, + explicit, + remainder, + } => { + if value.is_null() { + explicit.push_null(); + remainder.append_null(); + } else { + let (value, rest) = + split_to_explicit_ref(value.variant(), explicit_type, struct_type)?; + explicit.push_struct_value_ref(value)?; + append_json_variant_ref(remainder, &rest).context(ArrowComputeSnafu)?; + } + } + Self::AutoExpanding { values, .. } => { + values.push(JsonVariant::from(value.variant())); + } + } + Ok(()) + } + + fn push_null(&mut self) { + match self { + Self::Legacy { values, .. } | Self::AutoExpanding { values, .. } => { + values.push(JsonVariant::Null) + } + Self::ExplicitOnly { + explicit, + remainder, + .. + } => { + explicit.push_null(); + remainder.append_null(); + } + } + } +} + +/// Returns the fixed v2 Arrow physical type produced from `settings`. +pub fn json2_physical_data_type(settings: &JsonSettings) -> DataType { + let DataType::Struct(fields) = explicit_type(settings).as_arrow_type() else { + unreachable!("JSON2 explicit type must map to Arrow Struct") + }; + let mut fields = fields + .iter() + .cloned() + .chain(std::iter::once(Arc::new(variant_field( + JSON2_REMAINDER_FIELD_NAME, + true, + )))) + .collect::>(); + fields.sort_unstable_by(|x, y| x.name().cmp(y.name())); + DataType::Struct(fields.into()) +} + +fn explicit_type(settings: &JsonSettings) -> JsonNativeType { + let mut explicit_type = JsonNativeType::Object(Default::default()); + for hint in settings.type_hints() { + insert_dynamic_type(&mut explicit_type, &hint.path, (&hint.data_type).into()); + } + explicit_type } impl JsonVectorBuilder { + /// Creates a builder that merges all observed paths into the explicit schema. pub(crate) fn new(initial_native_type: JsonNativeType, capacity: usize) -> Self { debug_assert!(matches!( initial_native_type, JsonNativeType::Object(_) | JsonNativeType::Null )); Self { - merged_type: initial_native_type, - values: Vec::with_capacity(capacity), + state: JsonVectorBuilderState::Legacy { + merged_type: initial_native_type, + values: Vec::with_capacity(capacity), + }, } } + /// Creates a builder bounded by the JSON settings and their type hints. + pub(crate) fn with_settings(settings: &JsonSettings, capacity: usize) -> Self { + let explicit_type = explicit_type(settings); + let state = if settings.max_auto_expanded_paths() == Some(0) { + let DataType::Struct(fields) = explicit_type.as_arrow_type() else { + unreachable!("JSON2 explicit type must map to Arrow Struct") + }; + let struct_type = StructType::from(&fields); + JsonVectorBuilderState::ExplicitOnly { + explicit_type, + explicit: StructVectorBuilder::with_type_and_capacity( + struct_type.clone(), + capacity, + ), + struct_type, + remainder: VariantArrayBuilder::new(capacity), + } + } else { + JsonVectorBuilderState::AutoExpanding { + explicit_type, + max_auto_expanded_paths: settings + .max_auto_expanded_paths() + .unwrap_or(JSON2_DEFAULT_MAX_AUTO_EXPANDED_PATHS), + values: Vec::with_capacity(capacity), + } + }; + Self { state } + } + fn try_build(&mut self) -> Result { - let DataType::Struct(fields) = self.merged_type.as_arrow_type() else { - return UnexpectedSnafu { - reason: "merged JSON2 type must map to Arrow Struct in JsonVectorBuilder", - } - .fail(); - }; - // TODO(LFC): Direct use Arrow's Struct datatype here. - let struct_type = StructType::from(&fields); - - let mut builder = - StructVectorBuilder::with_type_and_capacity(struct_type.clone(), self.values.len()); - for value in std::mem::take(&mut self.values) { - if matches!(&value, JsonVariant::Null) { - builder.push_null(); - continue; - } - let value = json_variant_into_struct_value(value, struct_type.clone())?; - builder.push_struct_value_ref(StructValueRef::Ref(&value))?; - } - Ok(builder.to_vector()) + self.state.try_build() } } -fn json_variant_into_struct_value( - value: JsonVariant, - struct_type: StructType, -) -> Result { - let JsonVariant::Object(object) = value else { - return TryFromValueSnafu { - reason: format!("expected json object value, got {value:?}"), +fn build_legacy(values: &mut Vec, merged_type: &JsonNativeType) -> Result { + build_explicit(values, merged_type, false, |value| match value { + JsonVariant::Null => Ok(None), + JsonVariant::Object(value) => Ok(Some(value)), + _ => TryFromValueSnafu { + reason: "expected json object value".to_string(), + } + .fail(), + }) +} + +fn build_with_remainder( + values: &mut Vec, + expanded_type: &JsonNativeType, +) -> Result { + let mut remainder = VariantArrayBuilder::new(values.len()); + let explicit = build_explicit(values, expanded_type, true, |value| { + if matches!(value, JsonVariant::Null) { + remainder.append_null(); + return Ok(None); + } + let (value, rest) = split_to_explicit(value, expanded_type)?; + append_json_variant(&mut remainder, &JsonVariant::Object(rest)) + .context(ArrowComputeSnafu)?; + Ok(Some(value)) + })?; + finish_vector(explicit, ArrayRef::from(remainder.build())) +} + +fn build_explicit( + values: &mut Vec, + explicit_type: &JsonNativeType, + // Temporary compatibility switch for the legacy storage layout. Once JSON2 fully switches to + // the v2 storage layout, empty objects should always be preserved instead of treated as null. + preserve_empty_structs: bool, + mut project: impl FnMut(JsonVariant) -> Result>, +) -> Result { + let DataType::Struct(fields) = explicit_type.as_arrow_type() else { + return UnexpectedSnafu { + reason: "merged JSON2 type must map to Arrow Struct in JsonVectorBuilder", } .fail(); }; + // TODO(LFC): Direct use Arrow's Struct datatype here. + let struct_type = StructType::from(&fields); + let mut builder = + StructVectorBuilder::with_type_and_capacity(struct_type.clone(), values.len()); + for value in std::mem::take(values) { + let Some(value) = project(value)? else { + builder.push_null(); + continue; + }; + let value = + json_variant_into_struct_value(value, struct_type.clone(), preserve_empty_structs)?; + builder.push_struct_value_ref(StructValueRef::Ref(&value))?; + } + Ok(builder.to_vector()) +} + +fn finish_vector(explicit: VectorRef, remainder: ArrayRef) -> Result { + let explicit = explicit.to_arrow_array(); + let explicit = explicit.as_struct(); + let mut children = explicit + .fields() + .iter() + .cloned() + .zip(explicit.columns().iter().cloned()) + .chain(std::iter::once(( + Arc::new(variant_field(JSON2_REMAINDER_FIELD_NAME, true)), + remainder, + ))) + .collect::>(); + children.sort_unstable_by(|(x, _), (y, _)| x.name().cmp(y.name())); + let (fields, columns): (Vec<_>, Vec) = children.into_iter().unzip(); + let array: ArrayRef = Arc::new(StructArray::new( + fields.into(), + columns, + explicit.nulls().cloned(), + )); + Helper::try_into_vector(array) +} + +fn infer_expanded_type( + explicit_type: &JsonNativeType, + max_auto_expanded_paths: u32, + values: &[JsonVariant], +) -> JsonNativeType { + if max_auto_expanded_paths == 0 { + return explicit_type.clone(); + } + + let mut stats = HashMap::new(); + let mut path = Vec::new(); + init_explicit_path_stats(explicit_type, &mut path, &mut stats); + for value in values { + count_dynamic_paths(value, &mut path, &mut stats); + } + let mut candidates = stats + .iter() + // Explicit paths are already in the output schema and do not consume the dynamic + // expansion budget. + .filter(|(_, stat)| !stat.is_explicit && stat.is_leaf) + // Parquet cannot store empty structs, while widening an empty object to a non-empty struct + // loses its shape. Keep paths containing empty objects in the Variant remainder. + .filter(|(_, stat)| !stat.contains_empty_object) + // A leaf is eligible only when both itself and every object prefix have one stable + // role and type across all observed values. + .filter(|(path, _)| { + !(1..=path.len()) + .any(|len| stats.get(&path[..len]).is_some_and(|stats| stats.conflicts)) + }) + .collect::>(); + candidates.sort_unstable_by(|(x_path, x), (y_path, y)| { + y.seen_count + .cmp(&x.seen_count) + .then_with(|| x_path.cmp(y_path)) + }); + + let mut expanded_type = explicit_type.clone(); + for (path, candidate) in candidates + .into_iter() + .take(max_auto_expanded_paths as usize) + { + insert_dynamic_type( + &mut expanded_type, + path, + candidate.expected_leaf_type.clone(), + ); + } + expanded_type +} + +/// Aggregated observations for one JSON path. +/// +/// Schema inference first seeds the map with explicit paths, then walks all input values once. +/// Objects, including empty objects, are non-leaf paths; every other non-null value is a leaf. +/// The first dynamic observation fixes the path role and exact leaf type. A later role or type +/// mismatch sets [`PathStats::conflicts`] permanently. Missing paths and null values do not affect +/// the statistics. +/// +/// After collection, dynamic leaves are ranked by [`PathStats::seen_count`]. A candidate is +/// rejected when it or any parent path conflicts, so candidate selection never rescans the input +/// values. +struct PathStats { + /// Whether the path came from a type hint and is already part of the output schema. + is_explicit: bool, + /// Whether the path is a non-object value rather than an object prefix. + is_leaf: bool, + /// Exact type required for a leaf; unused non-leaf paths keep [`JsonNativeType::Null`]. + expected_leaf_type: JsonNativeType, + /// Number of compatible observations used to rank dynamic leaves. + seen_count: usize, + /// Whether any observed value contains an empty object that requires lossless Variant storage. + contains_empty_object: bool, + /// Whether the path has ever had inconsistent roles or leaf types. + conflicts: bool, +} + +/// Seeds path statistics from the configured explicit JSON shape. +fn init_explicit_path_stats<'a>( + explicit_type: &'a JsonNativeType, + path: &mut Vec<&'a str>, + stats: &mut HashMap, PathStats>, +) { + let JsonNativeType::Object(fields) = explicit_type else { + return; + }; + for (name, data_type) in fields { + path.push(name); + let is_leaf = !matches!(data_type, JsonNativeType::Object(_)); + let expected_leaf_type = if is_leaf { + data_type.clone() + } else { + JsonNativeType::default() + }; + stats.insert( + path.clone(), + PathStats { + is_explicit: true, + is_leaf, + expected_leaf_type, + seen_count: 0, + contains_empty_object: false, + conflicts: false, + }, + ); + init_explicit_path_stats(data_type, path, stats); + path.pop(); + } +} + +/// Collects dynamic path statistics while traversing each input value once. +fn count_dynamic_paths<'a>( + value: &'a JsonVariant, + path: &mut Vec<&'a str>, + stats: &mut HashMap, PathStats>, +) { + if matches!(value, JsonVariant::Null) || path.len() > JSON2_MAX_STRUCTURED_DEPTH { + return; + } + + if !path.is_empty() { + let is_leaf = !matches!(value, JsonVariant::Object(_)); + let contains_empty_object = is_leaf && value.contains_empty_object(); + let conflicts = if let Some(stats) = stats.get_mut(path.as_slice()) { + stats.contains_empty_object |= contains_empty_object; + if !stats.conflicts { + let role_conflict = stats.is_leaf != is_leaf; + let type_conflict = || match (&stats.expected_leaf_type, value) { + // If both objects, they are compatible. + (JsonNativeType::Null | JsonNativeType::Object(_), JsonVariant::Object(_)) => { + false + } + _ => stats.expected_leaf_type != value.native_type(), + }; + if role_conflict || type_conflict() { + stats.conflicts = true; + } else { + stats.seen_count += 1; + } + } + stats.conflicts + } else { + let expected_leaf_type = if is_leaf { + value.native_type() + } else { + JsonNativeType::default() + }; + stats.insert( + path.clone(), + PathStats { + is_explicit: false, + is_leaf, + expected_leaf_type, + seen_count: 1, + contains_empty_object, + conflicts: false, + }, + ); + false + }; + if conflicts { + return; + } + } + + if let JsonVariant::Object(object) = value + && !object.is_empty() + { + for (name, value) in object { + path.push(name); + count_dynamic_paths(value, path, stats); + path.pop(); + } + } +} + +fn insert_dynamic_type>( + explicit_type: &mut JsonNativeType, + path: &[S], + data_type: JsonNativeType, +) { + let JsonNativeType::Object(fields) = explicit_type else { + return; + }; + let Some((name, path)) = path.split_first() else { + return; + }; + let name = name.as_ref().to_string(); + if path.is_empty() { + fields.insert(name, data_type); + return; + } + insert_dynamic_type( + fields + .entry(name) + .or_insert_with(|| JsonNativeType::Object(Default::default())), + path, + data_type, + ) +} + +fn split_to_explicit_ref<'a>( + value: &JsonVariantRef<'a>, + explicit_type: &JsonNativeType, + struct_type: &StructType, +) -> Result<(StructValueRef<'a>, JsonVariantRef<'a>)> { + let JsonVariantRef::Object(object) = value else { + return TryFromValueSnafu { + reason: "expected json object value".to_string(), + } + .fail(); + }; + let explicit = json_object_ref_into_struct_value_ref(object, struct_type)?; + let remainder = remainder_ref(object, explicit_type)?; + Ok((explicit, JsonVariantRef::Object(remainder))) +} + +fn json_object_ref_into_struct_value_ref<'a>( + object: &BTreeMap<&'a str, JsonVariantRef<'a>>, + struct_type: &StructType, +) -> Result> { + let mut values = Vec::with_capacity(struct_type.fields().len()); + for field in struct_type.fields().iter() { + let value = match object.get(field.name()) { + Some(value) => json_variant_ref_into_value_ref(value, field.data_type())?, + None => ValueRef::Null, + }; + values.push(value); + } + Ok(StructValueRef::RefList { + val: values, + fields: struct_type.clone(), + }) +} + +fn json_variant_ref_into_value_ref<'a>( + value: &JsonVariantRef<'a>, + expected_type: &ConcreteDataType, +) -> Result> { + let value = match (value, expected_type) { + (JsonVariantRef::Null, _) | (_, ConcreteDataType::Null(_)) => ValueRef::Null, + (JsonVariantRef::Object(object), ConcreteDataType::Struct(struct_type)) => { + ValueRef::Struct(json_object_ref_into_struct_value_ref(object, struct_type)?) + } + (JsonVariantRef::Bool(x), ConcreteDataType::Boolean(_)) => ValueRef::Boolean(*x), + (JsonVariantRef::Number(x), ConcreteDataType::UInt64(_)) => { + let Some(x) = x.as_u64() else { + return TryFromValueSnafu { + reason: format!("unable to convert {x:?} to UInt64"), + } + .fail(); + }; + ValueRef::UInt64(x) + } + (JsonVariantRef::Number(x), ConcreteDataType::Int64(_)) => { + let x = match x { + JsonNumber::PosInt(x) => i64::try_from(*x).ok(), + JsonNumber::NegInt(x) => Some(*x), + JsonNumber::Float(_) => None, + }; + let Some(x) = x else { + return TryFromValueSnafu { + reason: format!("unable to convert {x:?} to Int64"), + } + .fail(); + }; + ValueRef::Int64(x) + } + (JsonVariantRef::Number(JsonNumber::PosInt(x)), ConcreteDataType::Float64(_)) => { + ValueRef::Float64((*x as f64).into()) + } + (JsonVariantRef::Number(JsonNumber::NegInt(x)), ConcreteDataType::Float64(_)) => { + ValueRef::Float64((*x as f64).into()) + } + (JsonVariantRef::Number(JsonNumber::Float(x)), ConcreteDataType::Float64(_)) => { + ValueRef::Float64(*x) + } + (JsonVariantRef::String(x), ConcreteDataType::String(_)) => ValueRef::String(x), + (JsonVariantRef::Array(array), ConcreteDataType::List(list_type)) => { + let item_type = list_type.item_type().clone(); + let values = array + .iter() + .map(|x| json_variant_ref_into_value_ref(x, &item_type)) + .collect::>>()?; + ValueRef::List(ListValueRef::RefList { + val: values, + item_datatype: Arc::new(item_type), + }) + } + (value, expected_type) => { + return TryFromValueSnafu { + reason: format!("unable to convert json value {value:?} to {expected_type}"), + } + .fail(); + } + }; + Ok(value) +} + +fn remainder_ref<'a>( + object: &BTreeMap<&'a str, JsonVariantRef<'a>>, + explicit_type: &JsonNativeType, +) -> Result>> { + let JsonNativeType::Object(fields) = explicit_type else { + return UnexpectedSnafu { + reason: "JSON2 explicit type must be an object", + } + .fail(); + }; + let mut remainder = BTreeMap::new(); + for (&name, value) in object { + // Preserve explicit JSON nulls in the remainder because Arrow child nulls cannot + // distinguish a present JSON null from a missing path. + if *value == JsonVariantRef::Null { + remainder.insert(name, JsonVariantRef::Null); + continue; + } + + match fields.get(name) { + Some(data_type @ JsonNativeType::Object(_)) => match value { + JsonVariantRef::Object(object) => { + let child = remainder_ref(object, data_type)?; + if !child.is_empty() { + remainder.insert(name, JsonVariantRef::Object(child)); + } + } + _ => { + return TryFromValueSnafu { + reason: "expected json object value".to_string(), + } + .fail(); + } + }, + // A non-object entry in the explicit type tree is an explicit leaf and is + // already written to the Struct builder. So here does nothing. + Some(_) => {} + None => { + remainder.insert(name, value.clone()); + } + } + } + Ok(remainder) +} + +fn split_to_explicit( + value: JsonVariant, + explicit_type: &JsonNativeType, +) -> Result<(JsonObjectValue, JsonObjectValue)> { + let JsonVariant::Object(mut remainder) = value else { + return TryFromValueSnafu { + reason: "expected json object value".to_string(), + } + .fail(); + }; + let JsonNativeType::Object(fields) = explicit_type else { + return UnexpectedSnafu { + reason: "JSON2 explicit type must be an object", + } + .fail(); + }; + let mut explicit = JsonObjectValue::new(); + + for (name, data_type) in fields { + let Some(value) = remainder.remove(name) else { + continue; + }; + if value == JsonVariant::Null { + explicit.insert(name.clone(), JsonVariant::Null); + // Preserve explicit JSON nulls in the remainder because Arrow child nulls cannot + // distinguish a present JSON null from a missing path. + remainder.insert(name.clone(), JsonVariant::Null); + continue; + } + if matches!(data_type, JsonNativeType::Object(_)) { + let (child_explicit, child_remainder) = split_to_explicit(value, data_type)?; + explicit.insert(name.clone(), JsonVariant::Object(child_explicit)); + if !child_remainder.is_empty() { + remainder.insert(name.clone(), JsonVariant::Object(child_remainder)); + } + } else { + explicit.insert(name.clone(), value); + } + } + Ok((explicit, remainder)) +} + +fn json_variant_into_struct_value( + object: JsonObjectValue, + struct_type: StructType, + preserve_empty_structs: bool, +) -> Result { let mut entries = object.into_iter(); let mut entry = entries.next(); let mut values = Vec::with_capacity(struct_type.fields().len()); @@ -86,7 +776,7 @@ fn json_variant_into_struct_value( let value = match entry.take() { Some((name, value)) if name == field.name() => { entry = entries.next(); - json_variant_into_value(value, field.data_type())? + json_variant_into_value(value, field.data_type(), preserve_empty_structs)? } Some((name, _)) if name.as_str() < field.name() => { return TryFromValueSnafu { @@ -111,10 +801,19 @@ fn json_variant_into_struct_value( Ok(StructValue::new(values, struct_type)) } -fn json_variant_into_value(value: JsonVariant, expected_type: &ConcreteDataType) -> Result { +fn json_variant_into_value( + value: JsonVariant, + expected_type: &ConcreteDataType, + preserve_empty_structs: bool, +) -> Result { let value = match (value, expected_type) { (JsonVariant::Null, _) | (_, ConcreteDataType::Null(_)) => Value::Null, - (JsonVariant::Object(object), _) if object.is_empty() => Value::Null, + (JsonVariant::Object(object), _) if object.is_empty() && !preserve_empty_structs => { + Value::Null + } + (JsonVariant::Object(object), ConcreteDataType::Struct(struct_type)) => Value::Struct( + json_variant_into_struct_value(object, struct_type.clone(), preserve_empty_structs)?, + ), (JsonVariant::Bool(x), ConcreteDataType::Boolean(_)) => Value::Boolean(x), (JsonVariant::Number(x), ConcreteDataType::UInt64(_)) => { let Some(x) = x.as_u64() else { @@ -153,13 +852,10 @@ fn json_variant_into_value(value: JsonVariant, expected_type: &ConcreteDataType) let item_type = list_type.item_type().clone(); let values = array .into_iter() - .map(|v| json_variant_into_value(v, &item_type)) + .map(|v| json_variant_into_value(v, &item_type, preserve_empty_structs)) .collect::>>()?; Value::List(ListValue::new(values, Arc::new(item_type))) } - (value @ JsonVariant::Object(_), ConcreteDataType::Struct(struct_type)) => { - Value::Struct(json_variant_into_struct_value(value, struct_type.clone())?) - } (value, ConcreteDataType::Binary(_)) => Value::from(encode_json_variant(value)?), (value, expected_type) => { return TryFromValueSnafu { @@ -173,11 +869,11 @@ fn json_variant_into_value(value: JsonVariant, expected_type: &ConcreteDataType) impl MutableVector for JsonVectorBuilder { fn data_type(&self) -> ConcreteDataType { - ConcreteDataType::json2(self.merged_type.clone()) + ConcreteDataType::json2(self.state.native_type()) } fn len(&self) -> usize { - self.values.len() + self.state.len() } fn as_any(&self) -> &dyn Any { @@ -189,38 +885,27 @@ impl MutableVector for JsonVectorBuilder { } fn to_vector(&mut self) -> VectorRef { - self.try_build().unwrap_or_else(|e| panic!("{:?}", e)) + self.try_build().unwrap_or_else(|e| { + // Just try to avoid panicking here. + common_telemetry::error!(e; "Unable to build JSON2 vector"); + Arc::new(NullVector::new(self.len())) + }) } fn to_vector_cloned(&self) -> VectorRef { - self.clone().to_vector() + self.state.try_build_cloned().unwrap_or_else(|e| { + // Just try to avoid panicking here. + common_telemetry::error!(e; "Unable to build JSON2 vector"); + Arc::new(NullVector::new(self.len())) + }) } fn try_push_value_ref(&mut self, value: &ValueRef) -> Result<()> { - let ValueRef::Json(value) = value else { - return TryFromValueSnafu { - reason: format!("expected json value, got {value:?}"), - } - .fail(); - }; - let json_type = value.json_type(); - let json_type = json_type.as_ref(); - if !matches!(json_type, JsonNativeType::Object(_) | JsonNativeType::Null) { - return TryFromValueSnafu { - reason: format!("expected json object value, got {value:?}"), - } - .fail(); - } - if !is_include(&self.merged_type, json_type) { - self.merged_type.merge(json_type); - } - - self.values.push(JsonVariant::from(value.variant())); - Ok(()) + self.state.try_push_value_ref(value) } fn push_null(&mut self) { - self.values.push(JsonVariant::Null) + self.state.push_null() } fn extend_slice_of(&mut self, _: &dyn Vector, _: usize, _: usize) -> Result<()> { @@ -236,13 +921,21 @@ impl MutableVector for JsonVectorBuilder { mod tests { use std::sync::Arc; + use arrow_array::cast::AsArray; + use arrow_schema::Field; use common_base::bytes::Bytes; + use serde_json::json; use super::*; use crate::data_type::ConcreteDataType; + use crate::extension::json::{Json2ExtensionType, JsonMetadata}; + use crate::json::JsonTypeHint; + use crate::json::value::decode_json_variant; use crate::types::StructField; use crate::types::json_type::JsonObjectType; use crate::value::{ListValue, StructValue, Value, ValueRef}; + use crate::vectors::json::array::JsonArray; + use crate::vectors::json::variant::variant_to_json_values; #[test] fn test_json_vector_builder() -> Result<()> { @@ -346,7 +1039,7 @@ mod tests { let err = object_builder .try_push_value_ref(&boolean.as_value_ref()) .unwrap_err(); - assert!(err.to_string().contains("expected json object value")); + assert!(err.to_string().contains("expected JSON object value")); object_builder.try_push_value_ref(&object.as_value_ref())?; // Non-JSON values should be rejected at push time. @@ -355,20 +1048,376 @@ mod tests { let err = invalid_builder .try_push_value_ref(&ValueRef::Boolean(true)) .unwrap_err(); - assert!(err.to_string().contains("expected json value")); + assert!(err.to_string().contains("expected JSON value")); + + Ok(()) + } + + #[test] + fn test_zero_budget_builder_uses_explicit_only_schema_and_remainder() -> Result<()> { + let settings = JsonSettings::try_new( + vec![ + JsonTypeHint { + path: vec!["kind".to_string()], + data_type: ConcreteDataType::string_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }, + JsonTypeHint { + path: vec!["commit".to_string(), "operation".to_string()], + data_type: ConcreteDataType::string_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }, + JsonTypeHint { + path: vec!["time_us".to_string()], + data_type: ConcreteDataType::int64_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }, + ], + Some(0), + )?; + let mut builder = JsonVectorBuilder::with_settings(&settings, 2); + assert!(matches!( + &builder.state, + JsonVectorBuilderState::ExplicitOnly { .. } + )); + let values = [ + json!({ + "kind": "record", + "commit": {"operation": "create", "collection": "post"}, + "extra": 1, + "time_us": 1 + }), + json!({"kind": "other", "dynamic": true, "time_us": 2}), + ]; + for value in values.clone() { + let value = settings.encode(value)?; + builder.try_push_value_ref(&value.as_value_ref())?; + } + let array = builder.to_vector().to_arrow_array(); + assert_eq!(&json2_physical_data_type(&settings), array.data_type()); + assert_eq!(0, builder.len()); + let structs = array.as_struct(); + assert_eq!( + vec![JSON2_REMAINDER_FIELD_NAME, "commit", "kind", "time_us"], + structs + .fields() + .iter() + .map(|x| x.name().as_str()) + .collect::>() + ); + assert_eq!( + vec![ + Some(json!({"commit": {"collection": "post"}, "extra": 1})), + Some(json!({"commit": {"operation": null}, "dynamic": true})), + ], + variant_to_json_values(structs.column_by_name(JSON2_REMAINDER_FIELD_NAME).unwrap())? + ); + + let field = Field::new("data", array.data_type().clone(), true).with_extension_type( + Json2ExtensionType::new(Arc::new(JsonMetadata::new(settings))), + ); + let reconstructed = JsonArray::from(&array).project_to_v2(&field, &DataType::Binary)?; + let reconstructed = reconstructed.as_binary::(); + assert_eq!( + values[0], + decode_json_variant(reconstructed.value(0)).unwrap() + ); + assert_eq!( + json!({ + "kind": "other", + "commit": {"operation": null}, + "dynamic": true, + "time_us": 2 + }), + decode_json_variant(reconstructed.value(1)).unwrap() + ); + + let settings = JsonSettings::try_new(vec![], Some(0))?; + let mut builder = JsonVectorBuilder::with_settings(&settings, 3); + for value in [json!({}), json!({"x": 1})] { + let value = settings.encode(value)?; + builder.try_push_value_ref(&value.as_value_ref())?; + } + builder.push_null(); + let array = builder.to_vector().to_arrow_array(); + let structs = array.as_struct(); + assert_eq!(1, structs.num_columns()); + assert_eq!( + vec![Some(json!({})), Some(json!({"x": 1})), None], + variant_to_json_values(structs.column(0))? + ); + + Ok(()) + } + + #[test] + fn test_finite_budget_selects_dynamic_paths() -> Result<()> { + let builder = JsonVectorBuilder::with_settings(&JsonSettings::default(), 0); + assert!(matches!( + builder.state, + JsonVectorBuilderState::AutoExpanding { + max_auto_expanded_paths: JSON2_DEFAULT_MAX_AUTO_EXPANDED_PATHS, + .. + } + )); + + let settings = JsonSettings::try_new( + vec![JsonTypeHint { + path: vec!["hint".to_string()], + data_type: ConcreteDataType::string_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }], + Some(2), + )?; + let values = [ + json!({ + "hint": "first", + "conflict": 1, + "popular": {"nested": 1}, + "tie_a": "a", + "tie_b": true, + "rare": 1 + }), + json!({ + "hint": "second", + "conflict": "string", + "popular": {"nested": 2}, + "tie_a": "b", + "tie_b": false + }), + json!({"hint": "third", "popular": "scalar"}), + ]; + let mut builder = JsonVectorBuilder::with_settings(&settings, values.len()); + for value in values.clone() { + let value = settings.encode(value)?; + builder.try_push_value_ref(&value.as_value_ref())?; + } + + let array = builder.to_vector().to_arrow_array(); + let structs = array.as_struct(); + assert_eq!( + vec![JSON2_REMAINDER_FIELD_NAME, "hint", "tie_a", "tie_b"], + structs + .fields() + .iter() + .map(|x| x.name().as_str()) + .collect::>() + ); + assert!(structs.column_by_name("popular").is_none()); + assert_eq!( + vec![ + Some(json!({"conflict": 1, "popular": {"nested": 1}, "rare": 1})), + Some(json!({"conflict": "string", "popular": {"nested": 2}})), + Some(json!({"popular": "scalar"})), + ], + variant_to_json_values(structs.column_by_name(JSON2_REMAINDER_FIELD_NAME).unwrap())? + ); + + let field = Field::new("data", array.data_type().clone(), true).with_extension_type( + Json2ExtensionType::new(Arc::new(JsonMetadata::new(settings))), + ); + let reconstructed = JsonArray::from(&array).project_to_v2(&field, &DataType::Binary)?; + let reconstructed = reconstructed.as_binary::(); + assert_eq!( + values[0], + decode_json_variant(reconstructed.value(0)).unwrap() + ); + assert_eq!( + values[1], + decode_json_variant(reconstructed.value(1)).unwrap() + ); + assert_eq!( + json!({ + "hint": "third", + "popular": "scalar" + }), + decode_json_variant(reconstructed.value(2)).unwrap() + ); + + Ok(()) + } + + #[test] + fn test_v2_builder_preserves_explicit_null_presence() -> Result<()> { + let settings = JsonSettings::try_new(vec![], Some(1))?; + let values = [json!({"value": 1}), json!({"value": null}), json!({})]; + let mut builder = JsonVectorBuilder::with_settings(&settings, values.len()); + for value in values.clone() { + let value = settings.encode(value)?; + builder.try_push_value_ref(&value.as_value_ref())?; + } + let array = builder.to_vector().to_arrow_array(); + let structs = array.as_struct(); + assert_eq!( + vec![ + Some(json!({})), + Some(json!({"value": null})), + Some(json!({})) + ], + variant_to_json_values(structs.column_by_name(JSON2_REMAINDER_FIELD_NAME).unwrap())? + ); + let field = Field::new("data", array.data_type().clone(), true).with_extension_type( + Json2ExtensionType::new(Arc::new(JsonMetadata::new(settings))), + ); + let reconstructed = JsonArray::from(&array).project_to_v2(&field, &DataType::Binary)?; + let reconstructed = reconstructed.as_binary::(); + + assert_eq!( + values[0], + decode_json_variant(reconstructed.value(0)).unwrap() + ); + assert_eq!( + values[1], + decode_json_variant(reconstructed.value(1)).unwrap() + ); + assert_eq!( + values[2], + decode_json_variant(reconstructed.value(2)).unwrap() + ); + Ok(()) + } + + #[test] + fn test_v2_builder_accepts_sql_null() -> Result<()> { + let settings = JsonSettings::try_new(vec![], Some(1))?; + let mut builder = JsonVectorBuilder::with_settings(&settings, 2); + builder.try_push_value_ref(&ValueRef::Null)?; + let value = settings.encode(json!({}))?; + builder.try_push_value_ref(&value.as_value_ref())?; + + let array = builder.to_vector().to_arrow_array(); + let structs = array.as_struct(); + assert_eq!( + vec![None, Some(json!({}))], + variant_to_json_values(structs.column_by_name(JSON2_REMAINDER_FIELD_NAME).unwrap())? + ); + Ok(()) + } + + #[test] + fn test_reconstruct_nested_remainder_only_value() -> Result<()> { + let settings = JsonSettings::try_new(vec![], Some(1))?; + let values = [ + json!({"a": {"hot": 1}}), + json!({"a": {"hot": 2}}), + json!({"a": {"cold": 3}}), + ]; + let mut builder = JsonVectorBuilder::with_settings(&settings, values.len()); + for value in values.clone() { + let value = settings.encode(value)?; + builder.try_push_value_ref(&value.as_value_ref())?; + } + + let array = builder.to_vector().to_arrow_array(); + let field = Field::new("data", array.data_type().clone(), true).with_extension_type( + Json2ExtensionType::new(Arc::new(JsonMetadata::new(settings))), + ); + let reconstructed = JsonArray::from(&array).project_to_v2(&field, &DataType::Binary)?; + let reconstructed = reconstructed.as_binary::(); + assert_eq!( + vec![values[0].clone(), values[1].clone(), values[2].clone()], + (0..reconstructed.len()) + .map(|i| decode_json_variant(reconstructed.value(i)).unwrap()) + .collect::>() + ); + + Ok(()) + } + + #[test] + fn test_dynamic_paths_require_the_same_leaf_type() -> Result<()> { + let settings = JsonSettings::try_new( + vec![JsonTypeHint { + path: vec!["nested".to_string(), "hinted".to_string()], + data_type: ConcreteDataType::string_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }], + Some(8), + )?; + let values = [ + json!({ + "branch": {}, + "different": [1], + "empty": {}, + "nested": {}, + "reverse": "scalar", + "same": [1] + }), + json!({ + "branch": {"leaf": 1}, + "different": ["x"], + "empty": {}, + "nested": {"hinted": "x", "leaf": 1}, + "same": [2] + }), + json!({ + "branch": {}, + "different": [2], + "empty": {}, + "nested": {}, + "reverse": {"leaf": 1}, + "same": [3] + }), + ]; + let mut builder = JsonVectorBuilder::with_settings(&settings, values.len()); + for value in values { + let value = settings.encode(value)?; + builder.try_push_value_ref(&value.as_value_ref())?; + } + + let JsonNativeType::Object(fields) = builder.state.native_type() else { + unreachable!(); + }; + assert!(!fields.contains_key("different")); + assert!(!fields.contains_key("empty")); + assert!(!fields.contains_key("reverse")); + assert!(matches!(fields.get("same"), Some(JsonNativeType::Array(_)))); + assert!(matches!( + fields.get("branch"), + Some(JsonNativeType::Object(fields)) if fields.contains_key("leaf") + )); + assert!(matches!( + fields.get("nested"), + Some(JsonNativeType::Object(fields)) + if fields.contains_key("hinted") && fields.contains_key("leaf") + )); Ok(()) } #[test] fn test_json_variant_into_struct_value() -> Result<()> { + let struct_type = StructType::new(Arc::new(vec![StructField::new( + "value".to_string(), + ConcreteDataType::string_datatype(), + true, + )])); assert_eq!( json_variant_into_value( JsonVariant::Object(Default::default()), - &ConcreteDataType::string_datatype(), + &ConcreteDataType::struct_datatype(struct_type.clone()), + false, )?, Value::Null ); + assert_eq!( + json_variant_into_value( + JsonVariant::Object(Default::default()), + &ConcreteDataType::struct_datatype(struct_type.clone()), + true, + )?, + Value::Struct(StructValue::new(vec![Value::Null], struct_type)) + ); let item_type = ConcreteDataType::struct_datatype(StructType::new(Arc::new(vec![StructField::new( @@ -394,22 +1443,23 @@ mod tests { true, ), ])); - let variant = JsonVariant::from([ + let variant = JsonObjectValue::from([ ( - "items", + "items".to_string(), JsonVariant::Array(vec![ JsonVariant::from([("id", JsonVariant::from(1i64))]), JsonVariant::from([("id", JsonVariant::from(2i64))]), ]), ), ( - "meta", + "meta".to_string(), JsonVariant::from([("name", JsonVariant::from("foo"))]), ), ]); let value = Value::Struct(json_variant_into_struct_value( variant, struct_type.clone(), + true, )?); assert_eq!( diff --git a/src/datatypes/src/vectors/json/variant.rs b/src/datatypes/src/vectors/json/variant.rs index 1a4b8a861d..79b3b56165 100644 --- a/src/datatypes/src/vectors/json/variant.rs +++ b/src/datatypes/src/vectors/json/variant.rs @@ -15,10 +15,7 @@ use std::sync::Arc; use arrow_array::ArrayRef; -#[cfg(test)] -use arrow_schema::ArrowError; -use arrow_schema::{DataType, Field}; -#[cfg(test)] +use arrow_schema::{ArrowError, DataType, Field}; use parquet_variant::{ObjectFieldBuilder, Variant, VariantBuilderExt, VariantDecimal16}; #[cfg(test)] use parquet_variant_compute::VariantArrayBuilder; @@ -27,8 +24,7 @@ use parquet_variant_json::VariantToJson; use snafu::ResultExt; use crate::error::{ArrowComputeSnafu, Result}; -#[cfg(test)] -use crate::json::value::{JsonNumber, JsonVariant, decode_json_variant}; +use crate::json::value::{JsonNumber, JsonVariant, JsonVariantRef, decode_json_variant}; /// Returns the canonical Arrow field for an unshredded Parquet Variant array. pub fn variant_field(name: impl Into, nullable: bool) -> Field { @@ -43,6 +39,10 @@ pub fn variant_field(name: impl Into, nullable: bool) -> Field { .with_extension_type(VariantType) } +/// Encodes JSON values as an unshredded Parquet Variant array. +/// +/// `None` represents an Arrow null while `Some(Value::Null)` represents a JSON +/// null, preserving the distinction required by JSON2. #[cfg(test)] pub(crate) fn json_values_to_variant(values: &[Option]) -> Result { let mut builder = VariantArrayBuilder::new(values.len()); @@ -57,7 +57,7 @@ pub(crate) fn json_values_to_variant(values: &[Option]) -> Re /// Encodes JSON variants as an unshredded Parquet Variant array. #[cfg(test)] -fn json_variants_to_variant(values: &[Option]) -> Result { +pub(crate) fn json_variants_to_variant(values: &[Option]) -> Result { let mut builder = VariantArrayBuilder::new(values.len()); for value in values { match value { @@ -68,8 +68,7 @@ fn json_variants_to_variant(values: &[Option]) -> Result Ok(ArrayRef::from(builder.build())) } -#[cfg(test)] -fn append_json_variant( +pub(super) fn append_json_variant( builder: &mut impl VariantBuilderExt, value: &JsonVariant, ) -> std::result::Result<(), ArrowError> { @@ -115,7 +114,52 @@ fn append_json_variant( Ok(()) } -#[cfg(test)] +pub(super) fn append_json_variant_ref( + builder: &mut impl VariantBuilderExt, + value: &JsonVariantRef<'_>, +) -> std::result::Result<(), ArrowError> { + match value { + JsonVariantRef::Null => builder.append_value(Variant::Null), + JsonVariantRef::Bool(value) => builder.append_value(*value), + JsonVariantRef::Number(JsonNumber::PosInt(value)) => { + if let Ok(value) = i64::try_from(*value) { + builder.append_value(value); + } else { + append_large_u64(builder, *value)?; + } + } + JsonVariantRef::Number(JsonNumber::NegInt(value)) => builder.append_value(*value), + JsonVariantRef::Number(JsonNumber::Float(value)) => { + if value.0.is_finite() { + builder.append_value(value.0) + } else { + builder.append_value("NaN") + } + } + JsonVariantRef::String(value) => builder.append_value(*value), + JsonVariantRef::Array(values) => { + let mut list = builder.try_new_list()?; + for value in values { + append_json_variant_ref(&mut list, value)?; + } + list.finish(); + } + JsonVariantRef::Object(values) => { + let mut object = builder.try_new_object()?; + for (name, value) in values { + append_json_variant_ref(&mut ObjectFieldBuilder::new(name, &mut object), value)?; + } + object.finish(); + } + JsonVariantRef::Variant(value) => { + let value = decode_json_variant(value) + .map_err(|e| ArrowError::JsonError(format!("Failed to decode JSONB: {e}")))?; + append_json_value(builder, &value)?; + } + } + Ok(()) +} + fn append_json_value( builder: &mut impl VariantBuilderExt, value: &serde_json::Value, @@ -157,11 +201,11 @@ fn append_json_value( /// Parquet Variant has no unsigned integer primitive. Treat u64 as i64 first, then use Decimal16 /// to represent large (larger than i64::MAX) u64. -#[cfg(test)] fn append_large_u64( builder: &mut impl VariantBuilderExt, value: u64, ) -> std::result::Result<(), ArrowError> { + // Parquet Variant has no unsigned integer primitive. Decimal16 preserves the full u64 range. let value = VariantDecimal16::try_new(value as i128, 0).map_err(|e| { ArrowError::InvalidArgumentError(format!( "Failed to encode JSON large integer as Variant Decimal16: {e}" diff --git a/src/frontend/src/instance.rs b/src/frontend/src/instance.rs index 6a653b2b1d..2319f1e883 100644 --- a/src/frontend/src/instance.rs +++ b/src/frontend/src/instance.rs @@ -59,6 +59,7 @@ use common_recordbatch::error::StreamTimeoutSnafu; use common_telemetry::logging::SlowQueryOptions; use common_telemetry::{debug, error, tracing}; use dashmap::DashMap; +use datafusion::dataframe::DataFrame; use datafusion::physical_plan::ExecutionPlan; use datafusion_expr::LogicalPlan; use futures::{Stream, StreamExt, future}; @@ -847,6 +848,22 @@ impl Instance { query_interceptor.pre_execute(stmt.as_ref(), Some(&plan), query_ctx.clone())?; + // TQL EXPLAIN/ANALYZE formats are consumed from the query context at + // execution time (see `optimize_physical_plan`); re-apply the side + // effect of `plan_tql` that was lost when the plan was built during + // Describe. `explain_format` is per-query state, so this never + // overwrites anything. + if let Some(Statement::Tql(tql)) = &stmt { + let format = match tql { + Tql::Explain(explain) => explain.format.as_ref(), + Tql::Analyze(analyze) => analyze.format.as_ref(), + Tql::Eval(_) => None, + }; + if let Some(format) = format { + query_ctx.set_explain_format(format.to_string()); + } + } + let query = stmt .as_ref() .map(|s| s.to_string()) @@ -928,6 +945,59 @@ impl Instance { vec![result] } + /// Builds the [`DataFrame`] for an information-schema-backed `SHOW` + /// statement; `None` for other statements. The future is boxed to keep + /// `do_describe_inner`'s state machine small. + fn show_statement_dataframe<'a>( + &'a self, + stmt: &'a Statement, + query_ctx: &'a QueryContextRef, + ) -> Pin>> + Send + 'a>> { + Box::pin(async move { + let engine = &self.query_engine; + let catalog_manager = self.catalog_manager(); + let ctx = query_ctx.clone(); + let dataframe = match stmt { + Statement::ShowDatabases(show) => { + query::sql::show_databases_dataframe(show, engine, catalog_manager, ctx).await + } + Statement::ShowTables(show) => { + query::sql::show_tables_dataframe(show, engine, catalog_manager, ctx).await + } + Statement::ShowViews(show) => { + query::sql::show_views_dataframe(show, engine, catalog_manager, ctx).await + } + Statement::ShowFlows(show) => { + query::sql::show_flows_dataframe(show, engine, catalog_manager, ctx).await + } + Statement::ShowColumns(show) => { + query::sql::show_columns_dataframe(show, engine, catalog_manager, ctx).await + } + Statement::ShowTableStatus(show) => { + query::sql::show_table_status_dataframe(show, engine, catalog_manager, ctx) + .await + } + Statement::ShowCharset(kind) => { + query::sql::show_charsets_dataframe(kind, engine, catalog_manager, ctx).await + } + Statement::ShowCollation(kind) => { + query::sql::show_collations_dataframe(kind, engine, catalog_manager, ctx).await + } + Statement::ShowIndex(show) => { + query::sql::show_index_dataframe(show, engine, catalog_manager, ctx).await + } + Statement::ShowRegion(show) => { + query::sql::show_region_dataframe(show, engine, catalog_manager, ctx).await + } + Statement::ShowProcesslist(show) => { + query::sql::show_processlist_dataframe(show, engine, catalog_manager, ctx).await + } + _ => return None, + }; + Some(dataframe) + }) + } + async fn do_describe_inner( &self, stmt: Statement, @@ -947,6 +1017,37 @@ impl Instance { let plannable = is_inner_plannable(&stmt) || matches!(&stmt, Statement::Explain(explain) if is_inner_plannable(explain.statement.as_ref())); + if let Statement::Tql(tql) = stmt { + // TQL produces a logical plan; describe it from the plan so the + // extended-protocol RowDescription matches the executed DataRows. + self.check_sql_permission(&Statement::Tql(tql.clone()), &query_ctx) + .await?; + let plan = self.statement_executor.plan_tql(tql, &query_ctx).await?; + return self + .query_engine + .describe(plan, query_ctx) + .await + .map(Some) + .context(error::DescribeStatementSnafu); + } + + // Describe SHOW statements from the same projection the executor builds. + if let Some(dataframe) = self + .show_statement_dataframe(&stmt, &query_ctx) + .await + .transpose() + .context(PlanStatementSnafu)? + { + self.check_sql_permission(&stmt, &query_ctx).await?; + let plan = dataframe.into_unoptimized_plan(); + return self + .query_engine + .describe(plan, query_ctx) + .await + .map(Some) + .context(error::DescribeStatementSnafu); + } + if plannable { self.check_sql_permission(&stmt, &query_ctx).await?; diff --git a/src/frontend/src/lib.rs b/src/frontend/src/lib.rs index c170236073..1468fe7c0b 100644 --- a/src/frontend/src/lib.rs +++ b/src/frontend/src/lib.rs @@ -12,6 +12,8 @@ // See the License for the specific language governing permissions and // limitations under the License. +#![recursion_limit = "256"] + pub mod error; pub mod events; pub mod frontend; diff --git a/src/meta-srv/src/procedure/region_migration/open_candidate_region.rs b/src/meta-srv/src/procedure/region_migration/open_candidate_region.rs index 0cb1131e54..bc21746a04 100644 --- a/src/meta-srv/src/procedure/region_migration/open_candidate_region.rs +++ b/src/meta-srv/src/procedure/region_migration/open_candidate_region.rs @@ -487,10 +487,14 @@ mod tests { .await; send_mock_reply(mailbox, rx, |id| { - Ok(new_open_region_reply( + Ok(new_open_region_reply_with_error( id, false, - Some("test mocked".to_string()), + Some(InstructionError { + code: StatusCode::StorageUnavailable, + message: "test mocked".to_string(), + retry_hint: RetryHint::Retryable, + }), )) }); diff --git a/src/meta-srv/src/procedure/repartition/group/sync_region.rs b/src/meta-srv/src/procedure/repartition/group/sync_region.rs index 8ce5ea3dec..dd55db73bc 100644 --- a/src/meta-srv/src/procedure/repartition/group/sync_region.rs +++ b/src/meta-srv/src/procedure/repartition/group/sync_region.rs @@ -350,6 +350,9 @@ impl SyncRegion { mod tests { use std::assert_matches; + use common_error::ext::RetryHint; + use common_error::status_code::StatusCode; + use common_meta::instruction::InstructionError; use common_meta::peer::Peer; use common_meta::rpc::router::{Region, RegionRoute}; use store_api::region_engine::SyncRegionFromRequest; @@ -359,7 +362,9 @@ mod tests { use crate::procedure::repartition::group::GroupPrepareResult; use crate::procedure::repartition::group::sync_region::SyncRegion; use crate::procedure::repartition::test_util::{TestingEnv, new_persistent_context}; - use crate::procedure::test_util::{new_sync_region_reply, send_mock_reply}; + use crate::procedure::test_util::{ + new_sync_region_reply, new_sync_region_reply_with_error, send_mock_reply, + }; use crate::service::mailbox::Channel; #[test] @@ -455,4 +460,46 @@ mod tests { let err = sync_region.sync_regions(&mut ctx).await.unwrap_err(); assert_matches!(err, Error::RetryLater { .. }); } + + #[tokio::test] + async fn test_sync_regions_retryable_instruction_error() { + let mut env = TestingEnv::new(); + let table_id = 1024; + let region_id = RegionId::new(table_id, 3); + let mut persistent_context = new_persistent_context(table_id, vec![], vec![]); + persistent_context.group_prepare_result = Some(test_prepare_result(table_id)); + + let (tx, rx) = tokio::sync::mpsc::channel(1); + env.mailbox_ctx + .insert_heartbeat_response_receiver(Channel::Datanode(1), tx) + .await; + send_mock_reply(env.mailbox_ctx.mailbox().clone(), rx, move |id| { + Ok(new_sync_region_reply_with_error( + id, + region_id, + false, + true, + Some(InstructionError { + code: StatusCode::StorageUnavailable, + message: "manifest delta disappeared".to_string(), + retry_hint: RetryHint::Retryable, + }), + )) + }); + + let mut ctx = env.create_context(persistent_context); + let sync_region = SyncRegion { + region_routes: vec![RegionRoute { + region: Region { + id: region_id, + ..Default::default() + }, + leader_peer: Some(Peer::empty(1)), + ..Default::default() + }], + }; + + let err = sync_region.sync_regions(&mut ctx).await.unwrap_err(); + assert_matches!(err, Error::RetryLater { .. }); + } } diff --git a/src/meta-srv/src/procedure/test_util.rs b/src/meta-srv/src/procedure/test_util.rs index ea8c3d4aec..4208c6e52e 100644 --- a/src/meta-srv/src/procedure/test_util.rs +++ b/src/meta-srv/src/procedure/test_util.rs @@ -288,6 +288,23 @@ pub fn new_sync_region_reply( ready: bool, exists: bool, error: Option, +) -> MailboxMessage { + new_sync_region_reply_with_error( + id, + region_id, + ready, + exists, + legacy_instruction_error(error), + ) +} + +/// Generates a [InstructionReply::SyncRegions] reply with a structured error. +pub fn new_sync_region_reply_with_error( + id: u64, + region_id: RegionId, + ready: bool, + exists: bool, + error: Option, ) -> MailboxMessage { MailboxMessage { id, @@ -301,7 +318,7 @@ pub fn new_sync_region_reply( region_id, ready, exists, - error: legacy_instruction_error(error), + error, }, ]))) .unwrap(), diff --git a/src/meta-srv/src/service/procedure.rs b/src/meta-srv/src/service/procedure.rs index 7b3f98f0fc..7c406081a4 100644 --- a/src/meta-srv/src/service/procedure.rs +++ b/src/meta-srv/src/service/procedure.rs @@ -26,6 +26,7 @@ use api::v1::meta::{ use common_event_recorder::{PersistentEventContext, ProcedureEventInput}; use common_meta::key::TableMetadataManagerRef; use common_meta::key::table_name::TableNameKey; +use common_meta::peer::Peer; use common_meta::procedure_executor::ExecutorContext; use common_meta::rpc::ddl::{ CREATE_DATABASE_CREATOR_EXTENSION_KEY, CREATE_DATABASE_CREATOR_METADATA_KEY, @@ -195,7 +196,7 @@ impl procedure_service_server::ProcedureService for Metasrv { let from_peer = self .lookup_datanode_peer(from_peer) .await? - .context(error::PeerUnavailableSnafu { peer_id: from_peer })?; + .unwrap_or_else(|| Peer::empty(from_peer)); let to_peer = self .lookup_datanode_peer(to_peer) .await? diff --git a/src/mito2/Cargo.toml b/src/mito2/Cargo.toml index e48a5758ea..bf8e1cdaec 100644 --- a/src/mito2/Cargo.toml +++ b/src/mito2/Cargo.toml @@ -65,7 +65,7 @@ log-store = { workspace = true } mito-codec.workspace = true moka = { workspace = true, features = ["sync", "future"] } object-store = { workspace = true, features = ["testing"] } -parquet = { workspace = true, features = ["async"] } +parquet = { workspace = true, features = ["async", "variant_experimental"] } paste.workspace = true pin-project.workspace = true prometheus.workspace = true diff --git a/src/mito2/src/cache.rs b/src/mito2/src/cache.rs index ac1f5b6691..7935b3a55b 100644 --- a/src/mito2/src/cache.rs +++ b/src/mito2/src/cache.rs @@ -34,6 +34,7 @@ use common_datasource::compression::CompressionType; use common_telemetry::warn; use datatypes::arrow::buffer::BooleanBuffer; use datatypes::arrow::record_batch::RecordBatch; +use datatypes::types::json_type::JsonNativeType; use datatypes::value::Value; use datatypes::vectors::VectorRef; use index::bloom_filter_index::{BloomFilterIndexCache, BloomFilterIndexCacheRef}; @@ -2065,8 +2066,7 @@ impl SelectorResultValue { SelectorResult::Flat(batches) => batches.iter().map(record_batch_estimated_size).sum(), }; result_size - + self.json_target_types.len() - * (mem::size_of::() + mem::size_of::()) + + self.json_target_types.len() * (size_of::() + size_of::()) } } diff --git a/src/mito2/src/compaction.rs b/src/mito2/src/compaction.rs index 97d2804b41..3a2cb2afe3 100644 --- a/src/mito2/src/compaction.rs +++ b/src/mito2/src/compaction.rs @@ -14,6 +14,7 @@ mod buckets; pub mod compactor; +mod json2; pub mod memory_manager; pub mod picker; mod reader; @@ -31,6 +32,9 @@ use common_meta::key::SchemaMetadataManagerRef; use common_telemetry::{debug, error}; use common_time::TimeToLive; use common_time::range::TimestampRange; +pub(crate) use json2::{ + Json2RewritePlans, collect_json2_rewrite_plans, rewrite_json2_batch, rewrite_json2_schema, +}; pub use scheduler::CompactionRequest; pub(crate) use scheduler::{CompactionExecution, CompactionPickFinished, CompactionScheduler}; use serde::{Deserialize, Serialize}; diff --git a/src/mito2/src/compaction/json2.rs b/src/mito2/src/compaction/json2.rs new file mode 100644 index 0000000000..fa7ec52f71 --- /dev/null +++ b/src/mito2/src/compaction/json2.rs @@ -0,0 +1,661 @@ +// Copyright 2023 Greptime Team +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use std::collections::{HashMap, HashSet}; +use std::sync::Arc; + +use arrow_schema::extension::ExtensionType; +use datatypes::arrow::datatypes::{DataType as ArrowDataType, Field, Schema, SchemaRef}; +use datatypes::arrow::record_batch::RecordBatch; +use datatypes::extension::json::{JSON2_REMAINDER_FIELD_NAME, Json2ExtensionType, JsonMetadata}; +use datatypes::json::{JSON2_DEFAULT_MAX_AUTO_EXPANDED_PATHS, JsonSettings, JsonTypeHint}; +use datatypes::prelude::ConcreteDataType; +use datatypes::types::json_type::JsonNativeType; +use datatypes::vectors::json::array::JsonArray; +use datatypes::vectors::json::json2_physical_data_type; +use parquet::arrow::parquet_to_arrow_schema; +use parquet::file::metadata::ParquetMetaData; +use snafu::{OptionExt, ResultExt, ensure}; +use store_api::metadata::RegionMetadataRef; + +use crate::error::{ + ConvertValueSnafu, DataTypeMismatchSnafu, InvalidRecordBatchSnafu, NewRecordBatchSnafu, Result, +}; + +/// Plan for rewriting one JSON2 column to a fixed compaction layout. +/// +/// A compaction input may contain v1 and v2 SSTs with different physical schemas. This plan is +/// derived only from the current region metadata and remains fixed while all input batches are +/// decoded and rewritten. It therefore prevents source-only paths from expanding the output +/// schema without a bound. +pub(crate) struct Json2RewritePlan { + /// User-defined settings used to encode logical values. + logical_settings: JsonSettings, + /// Fixed settings used to build the target physical layout. + pub(super) target_layout: JsonSettings, +} + +/// JSON2 rewrite plans keyed by logical column name. +pub(crate) type Json2RewritePlans = HashMap; + +#[derive(Clone)] +struct Json2LeafPathStats { + rows: u64, + data_type: JsonNativeType, + is_type_conflicted: bool, +} + +/// Builds the JSON2 rewrite plans for a compaction. +/// +/// Type hints from current region metadata are always retained. Existing explicit dynamic paths +/// from all input SST schemas are ranked once to produce a fixed layout; paths found only in a v2 +/// remainder are deliberately not promoted. [`rewrite_json2_batch`] decodes inputs and rewrites +/// them according to these plans. Non-JSON2 columns are omitted from the returned map. +/// +/// Returns an error when a JSON2 column has invalid or missing extension metadata, or when its +/// output layout is not v2. Legacy region metadata is upgraded in memory by the region opener +/// before compaction reaches this function. +pub(super) fn collect_json2_rewrite_plans_from_parquet( + metadata: &RegionMetadataRef, + parquet_metadata: &[Arc], +) -> Result { + let schemas = parquet_metadata + .iter() + .map(|metadata| { + let file = metadata.file_metadata(); + let schema = parquet_to_arrow_schema(file.schema_descr(), file.key_value_metadata()) + .map_err(|error| { + InvalidRecordBatchSnafu { + reason: format!("Failed to read compaction input Arrow schema: {error}"), + } + .build() + })?; + let rows = metadata + .row_groups() + .iter() + .map(|x| x.num_rows()) + .sum::() as u64; + Ok((Arc::new(schema), rows)) + }) + .collect::>>()?; + + collect_json2_rewrite_plans(metadata, &schemas) +} + +/// Builds JSON2 rewrite plans from existing physical schemas. +/// +/// Each source row count weights all of its existing explicit leaves. Ordinary paths are never +/// discovered from the v2 remainder, so this operation cannot unexpectedly promote opaque data. +pub(crate) fn collect_json2_rewrite_plans( + metadata: &RegionMetadataRef, + schemas: &[(SchemaRef, u64)], +) -> Result { + let json2_columns = metadata + .column_metadatas + .iter() + .filter_map(|x| { + x.column_schema + .data_type + .is_json2() + .then_some(&x.column_schema) + }) + .collect::>(); + + let mut plans = HashMap::with_capacity(json2_columns.len()); + for column in json2_columns { + let extension = column + .extension_type::() + .context(DataTypeMismatchSnafu)? + .with_context(|| InvalidRecordBatchSnafu { + reason: format!("JSON2 column '{}' has no extension metadata", column.name), + })?; + // Source SSTs may use v1, but current region metadata is copied to the rewritten output. + // Since the target physical layout is always v2, v1 metadata would produce an + // inconsistent persisted field. The region opener normally upgraded this metadata already. + ensure!( + extension.metadata().is_version_2(), + InvalidRecordBatchSnafu { + reason: format!("JSON2 column '{}' is not layout v2", column.name), + } + ); + + let settings = extension.metadata().json_settings(); + let hint_paths = settings + .type_hints() + .iter() + .map(|hint| hint.path.iter().map(String::as_str).collect::>()) + .collect::>(); + let mut stats = HashMap::new(); + for (schema, rows) in schemas { + let Some((_, field)) = schema.fields().find(&column.name) else { + continue; + }; + collect_json2_path_stats(field, *rows, &hint_paths, &mut stats)?; + } + + let mut hints = settings.type_hints().to_vec(); + hints.extend(select_dynamic_hints(settings, &hint_paths, &stats)); + let target_layout = JsonSettings::try_new(hints, Some(0)).context(DataTypeMismatchSnafu)?; + plans.insert( + column.name.clone(), + Json2RewritePlan { + logical_settings: settings.clone(), + target_layout, + }, + ); + } + Ok(plans) +} + +fn collect_json2_path_stats<'a>( + field: &'a Field, + rows: u64, + hint_paths: &HashSet>, + stats: &mut HashMap, Json2LeafPathStats>, +) -> Result<()> { + let ArrowDataType::Struct(fields) = field.data_type() else { + return InvalidRecordBatchSnafu { + reason: format!("JSON2 column '{}' is not a struct", field.name()), + } + .fail(); + }; + let mut paths = Vec::new(); + for field in fields { + if field.name() == JSON2_REMAINDER_FIELD_NAME { + continue; + } + collect_leaf_path_types(field, &mut Vec::new(), &mut paths)?; + } + + for (path, data_type) in paths { + if hint_paths.contains(path.as_slice()) { + continue; + } + let Some(stat) = stats.get_mut(&path) else { + stats.insert( + path, + Json2LeafPathStats { + rows, + data_type, + is_type_conflicted: false, + }, + ); + continue; + }; + if stat.data_type != data_type { + stat.is_type_conflicted = true; + } else { + stat.rows += rows; + } + } + Ok(()) +} + +fn collect_leaf_path_types<'a>( + field: &'a Field, + path: &mut Vec<&'a str>, + paths: &mut Vec<(Vec<&'a str>, JsonNativeType)>, +) -> Result<()> { + path.push(field.name()); + if let ArrowDataType::Struct(fields) = field.data_type() + && !fields.is_empty() + { + for field in fields { + collect_leaf_path_types(field, path, paths)?; + } + } else { + let json_type = + JsonNativeType::try_from(field.data_type()).context(DataTypeMismatchSnafu)?; + paths.push((path.clone(), json_type)); + } + path.pop(); + Ok(()) +} + +fn select_dynamic_hints( + settings: &JsonSettings, + hint_paths: &HashSet>, + stats: &HashMap, Json2LeafPathStats>, +) -> Vec { + let all_paths = stats + .keys() + .map(Vec::as_slice) + .chain(hint_paths.iter().map(Vec::as_slice)) + .collect::>(); + let has_ancestor_path = + |path: &[&str]| (1..path.len()).any(|len| all_paths.contains(&path[..len])); + + let prefixes = all_paths + .iter() + .copied() + .flat_map(|path| (1..path.len()).map(|len| &path[..len])) + .collect::>(); + let has_descendant_path = |path: &[&str]| prefixes.contains(path); + + let mut candidates = stats + .iter() + .filter(|(path, stat)| { + !stat.is_type_conflicted + // TODO(LFC): Instead of "primitive only", consider retaining stable compound types + // that are safe to write to Parquet, as flush does. Or better, unite the two + // selection process. + && stat.data_type.is_primitive() + && !has_ancestor_path(path) + && !has_descendant_path(path) + }) + .collect::>(); + candidates.sort_unstable_by(|(x_path, x), (y_path, y)| { + y.rows.cmp(&x.rows).then_with(|| x_path.cmp(y_path)) + }); + candidates + .into_iter() + .take( + settings + .max_auto_expanded_paths() + .unwrap_or(JSON2_DEFAULT_MAX_AUTO_EXPANDED_PATHS) as usize, + ) + .map(|(path, stat)| JsonTypeHint { + path: path.iter().map(|x| (*x).to_owned()).collect(), + data_type: ConcreteDataType::from_arrow_type(&stat.data_type.as_arrow_type()), + nullable: true, + default_constraint: None, + inverted_index: false, + }) + .collect() +} + +/// Replaces JSON2 physical field types according to the computed rewrite plans. +pub(crate) fn rewrite_json2_schema(schema: &SchemaRef, plans: &Json2RewritePlans) -> SchemaRef { + if plans.is_empty() { + return schema.clone(); + } + let fields = schema + .fields() + .iter() + .map(|field| { + let Some(plan) = plans.get(field.name()) else { + return field.clone(); + }; + let mut field = Field::clone(field); + field.set_data_type(json2_physical_data_type(&plan.target_layout)); + field = field.with_extension_type(Json2ExtensionType::new(Arc::new( + JsonMetadata::new(plan.logical_settings.clone()), + ))); + Arc::new(field) + }) + .collect::>(); + Arc::new(Schema::new_with_metadata(fields, schema.metadata().clone())) +} + +/// Rewrites JSON2 columns in `batch` according to the computed plans. +pub(crate) fn rewrite_json2_batch( + batch: RecordBatch, + plans: &Json2RewritePlans, +) -> Result { + if plans.is_empty() { + return Ok(batch); + } + let mut fields = Vec::with_capacity(batch.num_columns()); + let mut columns = Vec::with_capacity(batch.num_columns()); + + for (field, array) in batch.schema_ref().fields().iter().zip(batch.columns()) { + let Some(plan) = plans.get(field.name()) else { + fields.push(field.clone()); + columns.push(array.clone()); + continue; + }; + + let array = JsonArray::from(array) + .rewrite_to_v2(field, &plan.logical_settings, &plan.target_layout) + .context(ConvertValueSnafu)?; + debug_assert_eq!( + &json2_physical_data_type(&plan.target_layout), + array.data_type() + ); + + let mut field = Field::clone(field); + field.set_data_type(array.data_type().clone()); + field = field.with_extension_type(Json2ExtensionType::new(Arc::new(JsonMetadata::new( + plan.logical_settings.clone(), + )))); + fields.push(Arc::new(field)); + columns.push(array); + } + + let schema = Arc::new(Schema::new_with_metadata( + fields, + batch.schema_ref().metadata().clone(), + )); + RecordBatch::try_new(schema, columns).context(NewRecordBatchSnafu) +} + +#[cfg(test)] +mod tests { + use datatypes::extension::json::{ + JSON2_REMAINDER_FIELD_NAME, Json2PhysicalLayout, JsonMetadata, + }; + use datatypes::json::JsonTypeHint; + use datatypes::prelude::{ConcreteDataType, DataType}; + use datatypes::schema::ColumnSchema; + use datatypes::types::json_type::{JsonNativeType, JsonObjectType}; + use serde_json::json; + + use super::*; + + #[test] + fn test_select_dynamic_hints_rejects_type_and_prefix_conflicts() + -> Result<(), Box> { + let settings = JsonSettings::try_new( + vec![JsonTypeHint { + path: vec!["hint".to_string()], + data_type: ConcreteDataType::string_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }], + Some(2), + )?; + let stat = |rows, data_type, is_type_conflicted| Json2LeafPathStats { + rows, + data_type, + is_type_conflicted, + }; + let stats = HashMap::from([ + ( + vec!["hint", "nested"], + stat(10, JsonNativeType::String, false), + ), + (vec!["popular"], stat(9, JsonNativeType::String, false)), + ( + vec!["popular", "nested"], + stat(8, JsonNativeType::u64(), false), + ), + ( + vec!["type_conflicted"], + stat(7, JsonNativeType::String, true), + ), + ( + vec!["array"], + stat( + 6, + JsonNativeType::Array(Box::new(JsonNativeType::String)), + false, + ), + ), + (vec!["variant"], stat(5, JsonNativeType::Variant, false)), + (vec!["tie_a"], stat(2, JsonNativeType::String, false)), + (vec!["tie_b"], stat(2, JsonNativeType::Bool, false)), + ]); + + let hint_paths = settings + .type_hints() + .iter() + .map(|hint| hint.path.iter().map(String::as_str).collect::>()) + .collect::>(); + let hints = select_dynamic_hints(&settings, &hint_paths, &stats); + assert_eq!( + vec![vec!["tie_a".to_string()], vec!["tie_b".to_string()]], + hints.into_iter().map(|x| x.path).collect::>() + ); + Ok(()) + } + + #[test] + fn test_rewrite_json2_v1_batch_to_target_layout() -> Result<(), Box> { + let settings = JsonSettings::try_new( + vec![JsonTypeHint { + path: vec!["kind".to_string()], + data_type: ConcreteDataType::string_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }], + Some(0), + )?; + let target = json2_physical_data_type(&settings); + let plans = HashMap::from([( + "j".to_string(), + Json2RewritePlan { + logical_settings: settings.clone(), + target_layout: settings, + }, + )]); + let values = [ + json!({"kind": "a", "extra": {"x": 1}}), + json!({"kind": "b", "extra": {"x": 2}}), + ]; + let source_settings = JsonSettings::default(); + let source_extension = + Json2ExtensionType::new(Arc::new(JsonMetadata::new_v1(source_settings.clone()))); + let mut source_column = ColumnSchema::new( + "j", + ConcreteDataType::json2(JsonNativeType::Object(JsonObjectType::new())), + true, + ); + source_column.with_extension_type(&source_extension); + let mut source_builder = source_column.data_type.create_mutable_vector(values.len()); + for value in &values { + let value = source_settings.encode(value.clone())?; + source_builder.try_push_value_ref(&value.as_value_ref())?; + } + let source = source_builder.to_vector().to_arrow_array(); + let field = + Field::new("j", source.data_type().clone(), true).with_extension_type(source_extension); + let batch = RecordBatch::try_new(Arc::new(Schema::new(vec![field])), vec![source])?; + + let batch = rewrite_json2_batch(batch, &plans)?; + let field = batch.schema_ref().field(0); + assert!(Json2PhysicalLayout::try_from_root(field)?.is_version_2()); + assert_eq!(&target, field.data_type()); + let projected = + JsonArray::from(batch.column(0)).project_to_v2(field, &ArrowDataType::Binary)?; + let projected = JsonArray::from(&projected); + for (i, expected) in values.into_iter().enumerate() { + assert_eq!(expected, projected.try_get_value(i)?); + } + Ok(()) + } + + #[test] + fn test_rewrite_json2_v2_source_to_narrower_target() -> Result<(), Box> { + let target_settings = JsonSettings::try_new( + vec![JsonTypeHint { + path: vec!["kind".to_string()], + data_type: ConcreteDataType::string_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }], + Some(0), + )?; + let target_extension = + Json2ExtensionType::new(Arc::new(JsonMetadata::new(target_settings.clone()))); + let target_type = json2_physical_data_type(&target_settings); + let plans = HashMap::from([( + "j".to_string(), + Json2RewritePlan { + logical_settings: target_settings.clone(), + target_layout: target_settings, + }, + )]); + + let source_settings = JsonSettings::try_new( + vec![ + JsonTypeHint { + path: vec!["kind".to_string()], + data_type: ConcreteDataType::string_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }, + JsonTypeHint { + path: vec!["source_only".to_string()], + data_type: ConcreteDataType::int64_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }, + ], + None, + )?; + let source_extension = + Json2ExtensionType::new(Arc::new(JsonMetadata::new(source_settings.clone()))); + let mut source_column = ColumnSchema::new( + "j", + ConcreteDataType::json2(JsonNativeType::Object(JsonObjectType::new())), + true, + ); + source_column.with_extension_type(&source_extension); + let expected = json!({ + "kind": "a", + "source_only": 7, + "dynamic": {"nested": true} + }); + let mut builder = source_column.create_mutable_vector(1); + let value = source_settings.encode(expected.clone())?; + builder.try_push_value_ref(&value.as_value_ref())?; + let source = builder.to_vector().to_arrow_array(); + let source_field = + Field::new("j", source.data_type().clone(), true).with_extension_type(source_extension); + let source = + JsonArray::from(&source).project_to_v2(&source_field, &ArrowDataType::Binary)?; + let field = Field::new("j", ArrowDataType::Binary, true) + .with_extension_type(target_extension.clone()); + let batch = RecordBatch::try_new(Arc::new(Schema::new(vec![field])), vec![source])?; + + let batch = rewrite_json2_batch(batch, &plans)?; + let field = batch.schema_ref().field(0); + assert!(Json2PhysicalLayout::try_from_root(field)?.is_version_2()); + assert_eq!(&target_type, field.data_type()); + let ArrowDataType::Struct(fields) = field.data_type() else { + unreachable!() + }; + assert_eq!( + vec![JSON2_REMAINDER_FIELD_NAME, "kind"], + fields.iter().map(|x| x.name().as_str()).collect::>() + ); + + let projected = + JsonArray::from(batch.column(0)).project_to_v2(field, &ArrowDataType::Binary)?; + assert_eq!(expected, JsonArray::from(&projected).try_get_value(0)?); + + let first_schema = batch.schema(); + let field = + Field::new("j", ArrowDataType::Binary, true).with_extension_type(target_extension); + let batch = RecordBatch::try_new(Arc::new(Schema::new(vec![field])), vec![projected])?; + let batch = rewrite_json2_batch(batch, &plans)?; + assert_eq!(first_schema, batch.schema()); + + let field = batch.schema_ref().field(0); + let projected = + JsonArray::from(batch.column(0)).project_to_v2(field, &ArrowDataType::Binary)?; + assert_eq!(expected, JsonArray::from(&projected).try_get_value(0)?); + Ok(()) + } + + #[test] + fn test_rewrite_json2_v2_source_to_wider_target() -> Result<(), Box> { + let logical_settings = JsonSettings::try_new( + vec![JsonTypeHint { + path: vec!["kind".to_string()], + data_type: ConcreteDataType::string_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }], + Some(0), + )?; + let target_layout = JsonSettings::try_new( + vec![ + JsonTypeHint { + path: vec!["kind".to_string()], + data_type: ConcreteDataType::string_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }, + JsonTypeHint { + path: vec!["promoted".to_string()], + data_type: ConcreteDataType::int64_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }, + ], + Some(0), + )?; + let target_type = json2_physical_data_type(&target_layout); + let plans = HashMap::from([( + "j".to_string(), + Json2RewritePlan { + logical_settings: logical_settings.clone(), + target_layout, + }, + )]); + + let extension = + Json2ExtensionType::new(Arc::new(JsonMetadata::new(logical_settings.clone()))); + let mut column = ColumnSchema::new( + "j", + ConcreteDataType::json2(JsonNativeType::Object(JsonObjectType::new())), + true, + ); + column.with_extension_type(&extension); + let expected = json!({ + "kind": "a", + "promoted": 7, + "dynamic": {"nested": true} + }); + let mut builder = column.create_mutable_vector(1); + let value = logical_settings.encode(expected.clone())?; + builder.try_push_value_ref(&value.as_value_ref())?; + let source = builder.to_vector().to_arrow_array(); + assert_eq!( + &json2_physical_data_type(&logical_settings), + source.data_type() + ); + let field = + Field::new("j", source.data_type().clone(), true).with_extension_type(extension); + let batch = RecordBatch::try_new(Arc::new(Schema::new(vec![field])), vec![source])?; + + let batch = rewrite_json2_batch(batch, &plans)?; + let field = batch.schema_ref().field(0); + assert_eq!(&target_type, field.data_type()); + let ArrowDataType::Struct(fields) = field.data_type() else { + unreachable!() + }; + assert_eq!( + vec![JSON2_REMAINDER_FIELD_NAME, "kind", "promoted"], + fields.iter().map(|x| x.name().as_str()).collect::>() + ); + + let array = batch + .column(0) + .as_any() + .downcast_ref::() + .unwrap(); + let promoted = array + .column_by_name("promoted") + .unwrap() + .as_any() + .downcast_ref::() + .unwrap(); + assert_eq!(7, promoted.value(0)); + + let projected = + JsonArray::from(batch.column(0)).project_to_v2(field, &ArrowDataType::Binary)?; + assert_eq!(expected, JsonArray::from(&projected).try_get_value(0)?); + Ok(()) + } +} diff --git a/src/mito2/src/compaction/reader.rs b/src/mito2/src/compaction/reader.rs index f862d11e1a..2f604799cf 100644 --- a/src/mito2/src/compaction/reader.rs +++ b/src/mito2/src/compaction/reader.rs @@ -12,26 +12,25 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::collections::{BTreeMap, HashMap}; +use std::collections::BTreeMap; use std::sync::Arc; +use arrow_schema::extension::EXTENSION_TYPE_METADATA_KEY; use common_time::Timestamp; use common_time::range::TimestampRange; use common_time::timestamp::TimeUnit; use datafusion_common::ScalarValue; use datafusion_expr::Expr; -use datatypes::extension::json::is_json2_extension_type; -use datatypes::types::json_type::JsonNativeType; -use parquet::arrow::parquet_to_arrow_schema; use parquet::file::metadata::{PageIndexPolicy, ParquetMetaData}; -use snafu::{OptionExt, ResultExt}; +use snafu::OptionExt; use store_api::metadata::RegionMetadataRef; use crate::access_layer::AccessLayerRef; use crate::cache::{CacheManagerRef, CacheStrategy}; -use crate::error::{ - DataTypeMismatchSnafu, ParquetToArrowSchemaSnafu, Result, TimeRangePredicateOverflowSnafu, +use crate::compaction::json2::{ + Json2RewritePlans, collect_json2_rewrite_plans_from_parquet, rewrite_json2_schema, }; +use crate::error::{InvalidRecordBatchSnafu, Result, TimeRangePredicateOverflowSnafu}; use crate::read::FlatSource; use crate::read::flat_projection::FlatProjectionMapper; use crate::read::read_columns::ReadColumns; @@ -40,6 +39,7 @@ use crate::read::seq_scan::SeqScan; use crate::region::options::MergeMode; use crate::sst::file::FileHandle; use crate::sst::parquet::reader::MetadataCacheMetrics; +use crate::sst::parquet::{Json2RewriteTargets, Json2TargetLayout}; /// Builders to create [BoxedRecordBatchStream] for compaction. pub(crate) struct CompactionSstReaderBuilder<'a> { @@ -57,20 +57,24 @@ impl CompactionSstReaderBuilder<'_> { /// Build a [FlatSource] that yields Arrow `RecordBatch`s from reading all the input SST files, /// for compaction. The schema of the [FlatSource] is unified. pub(crate) async fn build_flat_sst_reader(self) -> Result { - let scan_input = self.build_scan_input().await?; + let parquet_metadata = self.collect_parquet_metadata().await?; + let plans = collect_json2_rewrite_plans_from_parquet(&self.metadata, &parquet_metadata)?; + let scan_input = self.build_scan_input(&parquet_metadata, &plans)?; let schema = scan_input.mapper.output_schema(); - let schema = schema.arrow_schema(); + let schema = rewrite_json2_schema(schema.arrow_schema(), &plans); let stream = SeqScan::new(scan_input) .build_flat_reader_for_compaction() .await?; - Ok(FlatSource::new_stream(schema.clone(), stream)) + Ok(FlatSource::new_stream(schema, stream)) } - async fn build_scan_input(self) -> Result { - let schema = self.metadata.schema.arrow_schema(); - let parquet_metadata = self.collect_parquet_metadata().await?; + fn build_scan_input( + self, + parquet_metadata: &[Arc], + plans: &Json2RewritePlans, + ) -> Result { let batch_size = crate::batch_size::estimate_batch_size( parquet_metadata .iter() @@ -84,38 +88,6 @@ impl CompactionSstReaderBuilder<'_> { (row_group.num_rows() as u64, uncompressed_bytes) }), ); - let json_type_hint = if schema.fields().iter().any(is_json2_extension_type) { - let mut json_type_hint = schema - .fields() - .iter() - .filter(|&field| is_json2_extension_type(field)) - .map(|field| (field.name().clone(), JsonNativeType::Null)) - .collect::>(); - - for metadata in &parquet_metadata { - let file_metadata = metadata.file_metadata(); - let schema = parquet_to_arrow_schema( - file_metadata.schema_descr(), - file_metadata.key_value_metadata(), - ) - .context(ParquetToArrowSchemaSnafu { - file: "compaction input", - })?; - for field in schema.fields() { - let Some(merged) = json_type_hint.get_mut(field.name()) else { - continue; - }; - - let json_type = JsonNativeType::try_from(field.data_type()) - .context(DataTypeMismatchSnafu)?; - merged.merge(&json_type); - } - } - - Some(json_type_hint) - } else { - None - }; let projection = (0..self.metadata.column_metadatas.len()).collect(); let read_column_ids = self @@ -124,24 +96,39 @@ impl CompactionSstReaderBuilder<'_> { .iter() .map(|x| x.column_id) .collect::>(); - let json_target_types = json_type_hint - .as_ref() - .map(|hint| { - hint.iter() - .filter_map(|(col_name, json_type)| { - self.metadata - .column_by_name(col_name) - .map(|col| (col.column_id, json_type.clone())) - }) - .collect::>() - }) - .unwrap_or_default(); - let read_columns = - ReadColumns::new(read_column_ids).with_json_target_types(json_target_types); - let mapper = - FlatProjectionMapper::new_with_read_columns(&self.metadata, projection, read_columns)?; + + let mut json2_target_layouts = BTreeMap::new(); + for (name, plan) in plans { + let Some(column) = self.metadata.column_by_name(name) else { + continue; + }; + let extension_metadata = column + .column_schema + .metadata() + .get(EXTENSION_TYPE_METADATA_KEY) + .cloned() + .with_context(|| InvalidRecordBatchSnafu { + reason: format!("JSON2 target column '{name}' has no extension metadata"), + })?; + json2_target_layouts.insert( + column.column_id, + Json2TargetLayout { + extension_metadata, + target_layout: plan.target_layout.clone(), + }, + ); + } + let read_columns = ReadColumns::new(read_column_ids); + let targets: Json2RewriteTargets = Arc::new(json2_target_layouts); + let mapper = FlatProjectionMapper::new_with_json2_rewrite_targets( + &self.metadata, + projection, + read_columns, + &targets, + )?; let mut scan_input = ScanInput::new(self.sst_layer, mapper) + .with_json2_rewrite_targets(targets) .with_files(self.inputs.to_vec()) .with_compaction(true) .with_batch_size(batch_size) diff --git a/src/mito2/src/engine/apply_staging_manifest_test.rs b/src/mito2/src/engine/apply_staging_manifest_test.rs index e10ed0cbf0..49cc97c7d5 100644 --- a/src/mito2/src/engine/apply_staging_manifest_test.rs +++ b/src/mito2/src/engine/apply_staging_manifest_test.rs @@ -12,20 +12,24 @@ // See the License for the specific language governing permissions and // limitations under the License. +use std::collections::HashSet; use std::sync::Arc; use std::{assert_matches, fs}; use api::v1::Rows; +use api::v1::region::{StrictWindow, compact_request}; use common_function::utils::partition_expr_version; use common_recordbatch::RecordBatches; +use datatypes::arrow::array::AsArray; +use datatypes::arrow::datatypes::Float64Type; use datatypes::value::Value; use partition::expr::{PartitionExpr, col}; use store_api::region_engine::{ RegionEngine, RegionRole, RemapManifestsRequest, SettableRegionRoleState, }; use store_api::region_request::{ - ApplyStagingManifestRequest, EnterStagingRequest, RegionFlushRequest, RegionPutRequest, - RegionRequest, StagingPartitionDirective, + ApplyStagingManifestRequest, EnterStagingRequest, RegionCompactRequest, RegionFlushRequest, + RegionPutRequest, RegionRequest, StagingPartitionDirective, }; use store_api::storage::{FileId, RegionId}; @@ -37,7 +41,9 @@ use crate::manifest::action::{ }; use crate::sst::FormatType; use crate::sst::file::FileMeta; -use crate::test_util::{CreateRequestBuilder, TestEnv, build_rows, put_rows, rows_schema}; +use crate::test_util::{ + CreateRequestBuilder, TestEnv, build_rows, build_rows_for_key, put_rows, rows_schema, +}; fn range_expr(col_name: &str, start: i64, end: i64) -> PartitionExpr { col(col_name) @@ -45,6 +51,209 @@ fn range_expr(col_name: &str, start: i64, end: i64) -> PartitionExpr { .and(col(col_name).lt(Value::Int64(end))) } +#[tokio::test] +async fn test_apply_staging_manifest_sequence_domain() { + common_telemetry::init_default_ut_logging(); + test_apply_staging_manifest_sequence_domain_with_format(false).await; + test_apply_staging_manifest_sequence_domain_with_format(true).await; +} + +async fn test_apply_staging_manifest_sequence_domain_with_format(flat_format: bool) { + let mut env = TestEnv::with_prefix("apply-staging-sequence-domain").await; + let engine = env + .create_engine(MitoConfig { + default_flat_format: flat_format, + ..Default::default() + }) + .await; + let source = RegionId::new(1, 1); + let target = RegionId::new(1, 2); + let request = CreateRequestBuilder::new().build(); + let schema = rows_schema(&request); + + engine + .handle_request(source, RegionRequest::Create(request.clone())) + .await + .unwrap(); + for value in 0..3 { + put_rows( + &engine, + source, + Rows { + schema: schema.clone(), + rows: build_rows_for_key("0", 0, 1, value), + }, + ) + .await; + } + engine + .handle_request(source, RegionRequest::Flush(RegionFlushRequest::default())) + .await + .unwrap(); + let source_manifest = engine + .get_region(source) + .unwrap() + .manifest_ctx + .manifest() + .await; + assert_eq!(source_manifest.files.len(), 1); + assert_eq!( + source_manifest.files.values().next().unwrap().sequence, + Some(std::num::NonZeroU64::new(3).unwrap()) + ); + + engine + .set_region_role_state_gracefully(source, SettableRegionRoleState::StagingLeader) + .await + .unwrap(); + let partition_expr = float_range_expr("field_0", 0.1, 100.1) + .as_json_str() + .unwrap(); + let result = engine + .remap_manifests(RemapManifestsRequest { + region_id: source, + input_regions: vec![source], + region_mapping: [(source, vec![target])].into_iter().collect(), + new_partition_exprs: [(target, partition_expr.clone())].into_iter().collect(), + }) + .await + .unwrap(); + engine + .handle_request(target, RegionRequest::Create(request.clone())) + .await + .unwrap(); + engine + .handle_request( + target, + RegionRequest::EnterStaging(EnterStagingRequest { + partition_directive: StagingPartitionDirective::UpdatePartitionExpr( + partition_expr.clone(), + ), + }), + ) + .await + .unwrap(); + engine + .handle_request( + target, + RegionRequest::ApplyStagingManifest(ApplyStagingManifestRequest { + partition_expr, + central_region_id: source, + manifest_path: result.manifest_paths[&target].clone(), + }), + ) + .await + .unwrap(); + + let manifest = engine + .get_region(target) + .unwrap() + .manifest_ctx + .manifest() + .await; + assert_eq!(manifest.files.len(), 1); + assert_eq!(manifest.committed_sequence, Some(1)); + let imported_file = manifest.files.values().next().unwrap(); + assert_eq!(imported_file.region_id, source); + assert_eq!( + imported_file.sequence, + Some(std::num::NonZeroU64::new(1).unwrap()) + ); + + put_rows( + &engine, + target, + Rows { + schema: schema.clone(), + rows: build_rows_for_key("0", 0, 1, 99), + }, + ) + .await; + assert_target_value(&engine, target, 99.0).await; + + engine + .handle_request(target, RegionRequest::Flush(RegionFlushRequest::default())) + .await + .unwrap(); + let input_ids = current_file_ids(&engine, target); + assert_eq!( + input_ids.len(), + 2, + "imported and target-write SSTs must both exist" + ); + + engine + .handle_request( + target, + RegionRequest::Compact(RegionCompactRequest { + options: compact_request::Options::StrictWindow(StrictWindow { + window_seconds: 60, + }), + parallelism: None, + time_range: None, + }), + ) + .await + .unwrap(); + let output_version = engine.get_region(target).unwrap().version(); + let output_files = output_version + .ssts + .levels() + .iter() + .flat_map(|level| level.files.values()) + .collect::>(); + let output_ids = output_files + .iter() + .map(|file| file.meta_ref().file_id) + .collect::>(); + assert_eq!(output_ids.len(), 1); + assert!( + output_files + .iter() + .all(|file| file.meta_ref().region_id == target) + ); + assert!( + input_ids.iter().all(|id| !output_ids.contains(id)), + "real compaction must replace both input SSTs" + ); + assert_target_value(&engine, target, 99.0).await; +} + +fn current_file_ids(engine: &crate::engine::MitoEngine, region_id: RegionId) -> HashSet { + engine + .get_region(region_id) + .unwrap() + .version() + .ssts + .levels() + .iter() + .flat_map(|level| level.files.values()) + .map(|file| file.meta_ref().file_id) + .collect() +} + +async fn assert_target_value( + engine: &crate::engine::MitoEngine, + region_id: RegionId, + expected: f64, +) { + let scan = engine + .scan_to_stream(region_id, ScanRequest::default()) + .await + .unwrap(); + let batches = RecordBatches::try_collect(scan).await.unwrap(); + assert_eq!( + batches.iter().map(|batch| batch.num_rows()).sum::(), + 1 + ); + let batch = batches.iter().next().unwrap(); + let values = batch + .column_by_name("field_0") + .unwrap() + .as_primitive::(); + assert_eq!(values.value(0), expected); +} + fn float_range_expr(col_name: &str, start: f64, end: f64) -> PartitionExpr { col(col_name) .gt_eq(Value::Float64(start.into())) diff --git a/src/mito2/src/engine/create_test.rs b/src/mito2/src/engine/create_test.rs index 25b6995190..2c04e2e24f 100644 --- a/src/mito2/src/engine/create_test.rs +++ b/src/mito2/src/engine/create_test.rs @@ -15,9 +15,13 @@ use std::time::Duration; use api::v1::Rows; +use common_error::ext::{ErrorExt, WhateverResult}; +use common_error::status_code::StatusCode; use common_recordbatch::RecordBatches; +use datatypes::prelude::ConcreteDataType; +use datatypes::types::json_type::{JsonNativeType, JsonObjectType}; use store_api::region_engine::RegionEngine; -use store_api::region_request::{RegionCloseRequest, RegionRequest}; +use store_api::region_request::{PathType, RegionCloseRequest, RegionOpenRequest, RegionRequest}; use store_api::storage::{RegionId, ScanRequest}; use crate::config::MitoConfig; @@ -32,6 +36,81 @@ async fn test_engine_create_new_region() { test_engine_create_new_region_with_format(true).await; } +#[tokio::test] +async fn test_engine_rejects_json2_with_time_series_memtable_on_create_and_open() +-> WhateverResult<()> { + let mut env = TestEnv::with_prefix("json2-rejects-time-series-memtable").await; + let engine = env + .create_engine(MitoConfig { + default_flat_format: false, + ..Default::default() + }) + .await; + let region_id = RegionId::new(1, 1); + let request = CreateRequestBuilder::new() + .field_datatype(ConcreteDataType::json2(JsonNativeType::Object( + JsonObjectType::new(), + ))) + .insert_option("append_mode", "true") + .insert_option("memtable.type", "time_series") + .build(); + + let err = engine + .handle_request(region_id, RegionRequest::Create(request)) + .await + .unwrap_err(); + assert_eq!(StatusCode::InvalidArguments, err.status_code()); + assert!( + err.to_string() + .contains("JSON2 columns only support BulkMemtable") + ); + + let request = CreateRequestBuilder::new() + .field_datatype(ConcreteDataType::json2(JsonNativeType::Object( + JsonObjectType::new(), + ))) + .insert_option("append_mode", "true") + .insert_option("memtable.type", "bulk") + .build(); + engine + .handle_request(region_id, RegionRequest::Create(request)) + .await?; + engine + .handle_request( + region_id, + RegionRequest::Close(RegionCloseRequest::default()), + ) + .await?; + + let options = [ + ("append_mode".to_string(), "true".to_string()), + ("memtable.type".to_string(), "time_series".to_string()), + ] + .into_iter() + .collect(); + let err = engine + .handle_request( + region_id, + RegionRequest::Open(RegionOpenRequest { + engine: String::new(), + table_dir: "test".to_string(), + path_type: PathType::Bare, + options, + skip_wal_replay: false, + checkpoint: None, + requirements: Default::default(), + }), + ) + .await + .unwrap_err(); + assert_eq!(StatusCode::InvalidArguments, err.status_code()); + assert!( + err.output_msg() + .contains("JSON2 columns only support BulkMemtable") + ); + Ok(()) +} + async fn test_engine_create_new_region_with_format(flat_format: bool) { let mut env = TestEnv::with_prefix("new-region").await; let engine = env diff --git a/src/mito2/src/engine/scan_test.rs b/src/mito2/src/engine/scan_test.rs index 4baa707f4e..2011ef3a20 100644 --- a/src/mito2/src/engine/scan_test.rs +++ b/src/mito2/src/engine/scan_test.rs @@ -13,11 +13,13 @@ // limitations under the License. use std::collections::{BTreeMap, HashMap}; +use std::sync::Arc; use api::helper::encode_json_value; use api::v1::Rows; use api::v1::helper::row; use api::v1::value::ValueData; +use arrow_schema::extension::ExtensionType; use common_base::readable_size::ReadableSize; use common_error::ext::{ErrorExt, WhateverResult}; use common_error::status_code::StatusCode; @@ -27,13 +29,16 @@ use datafusion_common::ScalarValue; use datafusion_expr::{col, lit}; use datatypes::arrow::array::AsArray; use datatypes::arrow::datatypes::{Float64Type, TimestampMillisecondType}; +use datatypes::extension::json::{Json2ExtensionType, JsonMetadata}; +use datatypes::json::JsonSettings; use datatypes::json::value::JsonValue; use datatypes::prelude::ConcreteDataType; use datatypes::types::json_type::{JsonNativeType, JsonObjectType}; +use datatypes::vectors::json::array::JsonArray; use futures::TryStreamExt; use serde_json::json; use store_api::region_engine::{PrepareRequest, RegionEngine, RegionScanner}; -use store_api::region_request::RegionRequest; +use store_api::region_request::{RegionCompactRequest, RegionRequest}; use store_api::storage::{RegionId, ScanRequest, TimeSeriesDistribution}; use crate::config::MitoConfig; @@ -41,7 +46,7 @@ use crate::error::Error; use crate::read::read_columns::ReadColumns; use crate::read::scan_region::Scanner; use crate::test_util; -use crate::test_util::{CreateRequestBuilder, TestEnv}; +use crate::test_util::{CreateRequestBuilder, TestEnv, reopen_region}; #[tokio::test] async fn test_json_type_hint_pushdown_scanner_returns_batches() -> WhateverResult<()> { @@ -73,28 +78,34 @@ async fn test_json_type_hint_pushdown_scanner_returns_batches() -> WhateverResul // Write full JSON objects, then flush them so the scanner has an Parquet file where nested // projection can be pushed down. - let rows = Rows { - schema, - rows: vec![ - row(vec![ - ValueData::StringValue("tag-1".to_string()), - ValueData::JsonValue(encode_json_value(JsonValue::from(json!({ - "a": { "x": 10, "y": "ignored-a" }, - "b": "ignored-b" - })))), - ValueData::TimestampMillisecondValue(1000), - ]), - row(vec![ - ValueData::StringValue("tag-2".to_string()), - ValueData::JsonValue(encode_json_value(JsonValue::from(json!({ - "a": { "x": 20, "y": "ignored-c" }, - "b": "ignored-d" - })))), - ValueData::TimestampMillisecondValue(2000), - ]), + for values in [ + vec![ + ValueData::StringValue("tag-1".to_string()), + ValueData::JsonValue(encode_json_value(JsonValue::from(json!({ + "a": { "x": 10, "y": "ignored-a" }, + "b": "ignored-b" + })))), + ValueData::TimestampMillisecondValue(1000), ], - }; - test_util::put_rows(&engine, region_id, rows).await; + vec![ + ValueData::StringValue("tag-2".to_string()), + ValueData::JsonValue(encode_json_value(JsonValue::from(json!({ + "a": { "x": 20, "y": "ignored-c" }, + "b": "ignored-d" + })))), + ValueData::TimestampMillisecondValue(2000), + ], + ] { + test_util::put_rows( + &engine, + region_id, + Rows { + schema: schema.clone(), + rows: vec![row(values)], + }, + ) + .await; + } test_util::flush_region(&engine, region_id, None).await; // Without a type hint, the scanner reads the whole JSON2 root column. @@ -208,6 +219,231 @@ async fn test_json_type_hint_pushdown_scanner_returns_batches() -> WhateverResul Ok(()) } +#[tokio::test] +async fn test_json2_v1_region_reopen_and_compaction() -> WhateverResult<()> { + let mut request = CreateRequestBuilder::new() + .field_datatype(ConcreteDataType::json2(JsonNativeType::Object( + JsonObjectType::new(), + ))) + .insert_option("memtable.type", "bulk") + .build(); + let settings = JsonSettings::default(); + request.column_metadatas[1] + .column_schema + .with_extension_type(&Json2ExtensionType::new(Arc::new(JsonMetadata::new_v1( + settings, + )))); + let table_dir = request.table_dir.clone(); + let schema = test_util::rows_schema(&request); + let mut env = TestEnv::new().await; + let engine = env.create_engine(MitoConfig::default()).await; + let region_id = RegionId::new(1024, 0); + engine + .handle_request(region_id, RegionRequest::Create(request)) + .await?; + + let values = [ + json!({"route": 1}), + json!({"b": {"c": "x"}}), + json!({"route": 3, "written_after_reopen": true}), + ]; + for (i, value) in values[..2].iter().enumerate() { + test_util::put_rows( + &engine, + region_id, + Rows { + schema: schema.clone(), + rows: vec![row(vec![ + ValueData::StringValue("tag".to_string()), + ValueData::JsonValue(encode_json_value(JsonValue::from(value.clone()))), + ValueData::TimestampMillisecondValue((i as i64 + 1) * 1000), + ])], + }, + ) + .await; + test_util::flush_region(&engine, region_id, None).await; + } + + reopen_region( + &engine, + region_id, + table_dir, + true, + HashMap::from([("memtable.type".to_string(), "bulk".to_string())]), + ) + .await; + let region = engine.get_region(region_id).unwrap(); + let version = region.version(); + let column = &version + .metadata + .column_by_name("field_0") + .unwrap() + .column_schema; + let extension = column.extension_type::()?.unwrap(); + assert!(extension.metadata().is_version_2()); + + test_util::put_rows( + &engine, + region_id, + Rows { + schema: schema.clone(), + rows: vec![row(vec![ + ValueData::StringValue("tag".to_string()), + ValueData::JsonValue(encode_json_value(JsonValue::from(values[2].clone()))), + ValueData::TimestampMillisecondValue(3000), + ])], + }, + ) + .await; + test_util::flush_region(&engine, region_id, None).await; + + let region = engine.get_region(region_id).unwrap(); + let input_files = region + .version() + .ssts + .levels() + .iter() + .map(|level| level.files.len()) + .sum::(); + assert_eq!(3, input_files); + + // The same requested path is stored as a v1 explicit leaf in the first SST, is absent from + // the second SST, and lives in the v2 remainder in the third SST. A per-file route preserves + // those differences while exposing one logical query type to the merge reader. + let scanner = engine + .scanner( + region_id, + ScanRequest { + projection: Some(vec![1]), + json_type_hint: HashMap::from([( + "field_0".to_string(), + JsonNativeType::Object(JsonObjectType::from([( + "route".to_string(), + JsonNativeType::i64(), + )])), + )]), + ..Default::default() + }, + ) + .await?; + let batches = RecordBatches::try_collect(scanner.scan().await?).await?; + let mut routed = Vec::new(); + for batch in batches.iter() { + let array = batch.column_by_name("field_0").unwrap().clone(); + let json = JsonArray::from(&array); + for i in 0..array.len() { + routed.push(json.try_get_value(i)?); + } + } + assert_eq!( + [json!({"route": 1}), json!(null), json!({"route": 3})], + routed.as_slice() + ); + + engine + .handle_request( + region_id, + RegionRequest::Compact(RegionCompactRequest::default()), + ) + .await?; + + let scanner = engine + .scanner( + region_id, + ScanRequest { + projection: Some(vec![1]), + ..Default::default() + }, + ) + .await?; + let batches = RecordBatches::try_collect(scanner.scan().await?).await?; + let mut actual = Vec::new(); + for batch in batches.iter() { + let array = batch.column_by_name("field_0").unwrap().clone(); + let json = JsonArray::from(&array); + for i in 0..array.len() { + actual.push(json.try_get_value(i)?); + } + } + assert_eq!(values, actual.as_slice()); + Ok(()) +} + +#[tokio::test] +async fn test_flush_aligns_different_json2_layouts() -> WhateverResult<()> { + let mut request = CreateRequestBuilder::new() + .field_datatype(ConcreteDataType::json2(JsonNativeType::Object( + JsonObjectType::new(), + ))) + .insert_option("append_mode", "true") + .insert_option("memtable.type", "bulk") + .build(); + let settings = JsonSettings::try_new(vec![], Some(1))?; + request.column_metadatas[1] + .column_schema + .with_extension_type(&Json2ExtensionType::new(Arc::new(JsonMetadata::new( + settings, + )))); + let schema = test_util::rows_schema(&request); + let mut env = TestEnv::new().await; + let engine = env.create_engine(MitoConfig::default()).await; + let region_id = RegionId::new(1025, 0); + engine + .handle_request(region_id, RegionRequest::Create(request)) + .await?; + + for (offset, name) in [(0, "a"), (1024, "b")] { + let rows = (0..1024) + .map(|i| { + row(vec![ + ValueData::StringValue("tag".to_string()), + ValueData::JsonValue(encode_json_value(JsonValue::from(json!({(name): i})))), + ValueData::TimestampMillisecondValue(offset + i), + ]) + }) + .collect(); + test_util::put_rows( + &engine, + region_id, + Rows { + schema: schema.clone(), + rows, + }, + ) + .await; + } + + test_util::flush_region(&engine, region_id, None).await; + + let scanner = engine + .scanner( + region_id, + ScanRequest { + projection: Some(vec![1]), + ..Default::default() + }, + ) + .await?; + let batches = RecordBatches::try_collect(scanner.scan().await?).await?; + let mut counts = HashMap::new(); + for batch in batches.iter() { + let array = batch.column_by_name("field_0").unwrap().clone(); + let json = JsonArray::from(&array); + for i in 0..array.len() { + let value = json.try_get_value(i)?; + let name = match (value.get("a"), value.get("b")) { + (Some(_), None) => "a", + (None, Some(_)) => "b", + _ => panic!("expected exactly one dynamic JSON2 field, got {value}"), + }; + *counts.entry(name).or_insert(0) += 1; + } + } + assert_eq!(Some(&1024), counts.get("a")); + assert_eq!(Some(&1024), counts.get("b")); + Ok(()) +} + #[tokio::test] async fn test_incremental_query_stale_error() { let mut env = TestEnv::with_prefix("test_incremental_query_stale_error").await; diff --git a/src/mito2/src/engine/set_role_state_test.rs b/src/mito2/src/engine/set_role_state_test.rs index 40e03b063a..1d0f3c188b 100644 --- a/src/mito2/src/engine/set_role_state_test.rs +++ b/src/mito2/src/engine/set_role_state_test.rs @@ -12,6 +12,8 @@ // See the License for the specific language governing permissions and // limitations under the License. +use std::time::Duration; + use api::v1::Rows; use common_error::ext::ErrorExt; use common_error::status_code::StatusCode; @@ -20,12 +22,16 @@ use store_api::region_engine::{ SettableRegionRoleState, }; use store_api::region_request::{ - EnterStagingRequest, RegionPutRequest, RegionRequest, StagingPartitionDirective, + EnterStagingRequest, RegionFlushRequest, RegionPutRequest, RegionRequest, + StagingPartitionDirective, }; use store_api::storage::RegionId; use crate::config::MitoConfig; -use crate::test_util::{CreateRequestBuilder, TestEnv, build_rows, put_rows, rows_schema}; +use crate::region::{RegionLeaderState, RegionRoleState}; +use crate::test_util::{ + CheckpointTaskBlocker, CreateRequestBuilder, TestEnv, build_rows, put_rows, rows_schema, +}; /// Helper function to assert a successful response with expected entry id fn assert_success_response(response: &SetRegionRoleStateResponse, expected_entry_id: u64) { @@ -212,6 +218,324 @@ async fn test_write_downgrading_region_with_format(flat_format: bool) { assert_eq!(err.status_code(), StatusCode::RegionNotReady) } +#[tokio::test(flavor = "multi_thread")] +async fn test_downgrading_waits_for_checkpoint_and_stops_new_checkpoints() { + let (blocker, mock_layer) = CheckpointTaskBlocker::block_cleanup(); + let mut env = TestEnv::new().await.with_mock_layer(mock_layer); + let engine = env + .create_engine(MitoConfig { + manifest_checkpoint_distance: 1, + ..Default::default() + }) + .await; + let region_id = RegionId::new(1, 1); + let request = CreateRequestBuilder::new().build(); + let column_schemas = rows_schema(&request); + engine + .handle_request(region_id, RegionRequest::Create(request)) + .await + .unwrap(); + + put_rows( + &engine, + region_id, + Rows { + schema: column_schemas.clone(), + rows: build_rows(0, 1), + }, + ) + .await; + engine + .handle_request( + region_id, + RegionRequest::Flush(RegionFlushRequest::default()), + ) + .await + .unwrap(); + tokio::time::timeout(Duration::from_secs(5), blocker.wait_until_blocked()) + .await + .expect("checkpoint cleanup did not start"); + + // Leave one memtable for the final flush after entering Downgrading. + put_rows( + &engine, + region_id, + Rows { + schema: column_schemas.clone(), + rows: build_rows(1, 2), + }, + ) + .await; + + let region = engine.get_region(region_id).unwrap(); + let wait_started = region + .manifest_ctx + .manifest_manager + .read() + .await + .checkpointer() + .pending_checkpoint_wait_started(); + // Register before starting the transition so the notification cannot be lost. + let wait_started = wait_started.notified(); + let cloned_engine = engine.clone(); + let downgrade = tokio::spawn(async move { + cloned_engine + .set_region_role_state_gracefully(region_id, SettableRegionRoleState::DowngradingLeader) + .await + }); + tokio::time::timeout(Duration::from_secs(5), wait_started) + .await + .expect("downgrade did not start waiting for the checkpoint"); + assert_eq!( + RegionRoleState::Leader(RegionLeaderState::Downgrading), + region.state() + ); + assert!( + !downgrade.is_finished(), + "downgrade returned before checkpoint cleanup finished" + ); + + blocker.release(); + assert_success_response(&downgrade.await.unwrap().unwrap(), 2); + + // Final flush publishes a normal delta, but Downgrading must not start a + // checkpoint after the barrier. + blocker.arm_next_close(); + engine + .handle_request( + region_id, + RegionRequest::Flush(RegionFlushRequest::default()), + ) + .await + .unwrap(); + assert!( + !region + .manifest_ctx + .manifest_manager + .read() + .await + .checkpointer() + .is_doing_checkpoint(), + "final flush started a checkpoint while Downgrading" + ); + assert_eq!(2, region.manifest_ctx.manifest().await.manifest_version); + assert_eq!( + 1, + region + .manifest_ctx + .manifest_manager + .read() + .await + .checkpointer() + .last_checkpoint_version() + ); + + // Leaving Downgrading removes the scheduling restriction. The next normal + // manifest update can checkpoint the accumulated deltas. + engine + .set_region_role(region_id, RegionRole::Leader) + .unwrap(); + put_rows( + &engine, + region_id, + Rows { + schema: column_schemas, + rows: build_rows(2, 3), + }, + ) + .await; + engine + .handle_request( + region_id, + RegionRequest::Flush(RegionFlushRequest::default()), + ) + .await + .unwrap(); + tokio::time::timeout(Duration::from_secs(5), blocker.wait_until_blocked()) + .await + .expect("checkpoint scheduling did not resume after leaving Downgrading"); + blocker.release(); + region + .manifest_ctx + .manifest_manager + .write() + .await + .wait_for_pending_checkpoint() + .await; +} + +#[tokio::test(flavor = "multi_thread")] +async fn test_direct_follower_waits_for_pending_checkpoint() { + let (blocker, mock_layer) = CheckpointTaskBlocker::block_cleanup(); + let mut env = TestEnv::new().await.with_mock_layer(mock_layer); + let engine = env + .create_engine(MitoConfig { + manifest_checkpoint_distance: 1, + ..Default::default() + }) + .await; + let region_id = RegionId::new(1, 1); + let request = CreateRequestBuilder::new().build(); + let column_schemas = rows_schema(&request); + engine + .handle_request(region_id, RegionRequest::Create(request)) + .await + .unwrap(); + put_rows( + &engine, + region_id, + Rows { + schema: column_schemas, + rows: build_rows(0, 1), + }, + ) + .await; + engine + .handle_request( + region_id, + RegionRequest::Flush(RegionFlushRequest::default()), + ) + .await + .unwrap(); + tokio::time::timeout(Duration::from_secs(5), blocker.wait_until_blocked()) + .await + .expect("checkpoint cleanup did not start"); + + let region = engine.get_region(region_id).unwrap(); + let wait_started = region + .manifest_ctx + .manifest_manager + .read() + .await + .checkpointer() + .pending_checkpoint_wait_started(); + // The no-flush downgrade path requests Follower directly. It must still + // wait for checkpoint cleanup before replying to the caller. + let wait_started = wait_started.notified(); + let cloned_engine = engine.clone(); + let set_follower = tokio::spawn(async move { + cloned_engine + .set_region_role_state_gracefully(region_id, SettableRegionRoleState::Follower) + .await + }); + tokio::time::timeout(Duration::from_secs(5), wait_started) + .await + .expect("follower transition did not start waiting for the checkpoint"); + assert_eq!(RegionRoleState::Follower, region.state()); + assert!( + !set_follower.is_finished(), + "follower transition returned before checkpoint cleanup finished" + ); + + blocker.release(); + assert_success_response(&set_follower.await.unwrap().unwrap(), 1); +} + +#[tokio::test(flavor = "multi_thread")] +async fn test_retried_downgrade_waits_after_first_request_is_cancelled() { + let (blocker, mock_layer) = CheckpointTaskBlocker::block_cleanup(); + let mut env = TestEnv::new().await.with_mock_layer(mock_layer); + let engine = env + .create_engine(MitoConfig { + manifest_checkpoint_distance: 1, + ..Default::default() + }) + .await; + let region_id = RegionId::new(1, 1); + let request = CreateRequestBuilder::new().build(); + let column_schemas = rows_schema(&request); + engine + .handle_request(region_id, RegionRequest::Create(request)) + .await + .unwrap(); + put_rows( + &engine, + region_id, + Rows { + schema: column_schemas, + rows: build_rows(0, 1), + }, + ) + .await; + engine + .handle_request( + region_id, + RegionRequest::Flush(RegionFlushRequest::default()), + ) + .await + .unwrap(); + tokio::time::timeout(Duration::from_secs(5), blocker.wait_until_blocked()) + .await + .expect("checkpoint cleanup did not start"); + + let region = engine.get_region(region_id).unwrap(); + let first_wait_started = region + .manifest_ctx + .manifest_manager + .read() + .await + .checkpointer() + .pending_checkpoint_wait_started(); + // Register before starting the transition so the notification cannot be lost. + let first_wait_started = first_wait_started.notified(); + // Call the region transition directly so aborting this task cancels the + // actual checkpoint waiter. The engine API submits the same transition to + // a detached worker task, so aborting its caller only drops the reply receiver. + let first_region = region.clone(); + let first_downgrade = tokio::spawn(async move { + first_region + .set_role_state_gracefully(SettableRegionRoleState::DowngradingLeader) + .await + }); + tokio::time::timeout(Duration::from_secs(5), first_wait_started) + .await + .expect("first downgrade did not start waiting for the checkpoint"); + assert_eq!( + RegionRoleState::Leader(RegionLeaderState::Downgrading), + region.state() + ); + assert!( + !first_downgrade.is_finished(), + "first downgrade did not wait for checkpoint cleanup" + ); + + // Cancelling the first caller must not remove the checkpoint handle from + // the manifest manager. A retried migration request observes Downgrading + // and must wait for the same checkpoint task before it can proceed. + first_downgrade.abort(); + assert!(first_downgrade.await.unwrap_err().is_cancelled()); + assert_eq!( + RegionRoleState::Leader(RegionLeaderState::Downgrading), + region.state() + ); + + let second_wait_started = region + .manifest_ctx + .manifest_manager + .read() + .await + .checkpointer() + .pending_checkpoint_wait_started(); + // Register before spawning the retry so the notification cannot be lost. + let second_wait_started = second_wait_started.notified(); + let retry_region = region.clone(); + let retry_downgrade = tokio::spawn(async move { + retry_region + .set_role_state_gracefully(SettableRegionRoleState::DowngradingLeader) + .await + }); + tokio::time::timeout(Duration::from_secs(5), second_wait_started) + .await + .expect("retried downgrade did not start waiting for the checkpoint"); + assert!( + !retry_downgrade.is_finished(), + "retried downgrade did not wait for checkpoint cleanup" + ); + + blocker.release(); + retry_downgrade.await.unwrap().unwrap(); +} + #[tokio::test] async fn test_unified_state_transitions() { test_unified_state_transitions_with_format(false).await; diff --git a/src/mito2/src/engine/staging_test.rs b/src/mito2/src/engine/staging_test.rs index 71abd67eb2..12263976aa 100644 --- a/src/mito2/src/engine/staging_test.rs +++ b/src/mito2/src/engine/staging_test.rs @@ -48,7 +48,9 @@ use crate::manifest::action::{ use crate::region::{RegionLeaderState, RegionRoleState, parse_partition_expr}; use crate::request::WorkerRequest; use crate::sst::FormatType; -use crate::test_util::{CreateRequestBuilder, TestEnv, build_rows, put_rows, rows_schema}; +use crate::test_util::{ + CheckpointTaskBlocker, CreateRequestBuilder, TestEnv, build_rows, put_rows, rows_schema, +}; fn range_expr(col_name: &str, start: i64, end: i64) -> PartitionExpr { col(col_name) @@ -1101,6 +1103,101 @@ async fn test_write_stall_on_enter_staging_with_format(flat_format: bool) { assert_eq!(expected, batches.pretty_print().unwrap()); } +#[tokio::test(flavor = "multi_thread")] +async fn test_enter_staging_waits_for_pending_checkpoint() { + for block_last_checkpoint_write in [false, true] { + let partition_directive = + StagingPartitionDirective::UpdatePartitionExpr(default_partition_expr()); + let (blocker, mock_layer) = if block_last_checkpoint_write { + CheckpointTaskBlocker::block_last_checkpoint_write() + } else { + CheckpointTaskBlocker::block_cleanup() + }; + let mut env = TestEnv::new().await.with_mock_layer(mock_layer); + let engine = env + .create_engine(MitoConfig { + manifest_checkpoint_distance: 1, + ..Default::default() + }) + .await; + let region_id = RegionId::new(1, 1); + let request = CreateRequestBuilder::new().build(); + let column_schemas = rows_schema(&request); + engine + .handle_request(region_id, RegionRequest::Create(request)) + .await + .unwrap(); + put_rows( + &engine, + region_id, + Rows { + schema: column_schemas, + rows: build_rows(0, 1), + }, + ) + .await; + engine + .handle_request( + region_id, + RegionRequest::Flush(RegionFlushRequest::default()), + ) + .await + .unwrap(); + tokio::time::timeout(Duration::from_secs(5), blocker.wait_until_blocked()) + .await + .expect("checkpoint task did not reach the configured block point"); + + let region = engine.get_region(region_id).unwrap(); + let wait_started = region + .manifest_ctx + .manifest_manager + .read() + .await + .checkpointer() + .pending_checkpoint_wait_started(); + // Register before starting the transition so the notification cannot be lost. + let wait_started = wait_started.notified(); + let cloned_engine = engine.clone(); + let enter_staging = tokio::spawn(async move { + cloned_engine + .handle_request( + region_id, + RegionRequest::EnterStaging(EnterStagingRequest { + partition_directive, + }), + ) + .await + }); + tokio::time::timeout(Duration::from_secs(5), wait_started) + .await + .expect("EnterStaging did not start waiting for the checkpoint"); + assert_eq!( + RegionRoleState::Leader(RegionLeaderState::EnteringStaging), + region.state() + ); + assert!( + !enter_staging.is_finished(), + "EnterStaging returned before the checkpoint task finished" + ); + + blocker.release(); + enter_staging.await.unwrap().unwrap(); + assert_eq!( + RegionRoleState::Leader(RegionLeaderState::Staging), + region.state() + ); + assert!( + !region + .manifest_ctx + .manifest_manager + .read() + .await + .checkpointer() + .is_doing_checkpoint() + ); + } +} + #[tokio::test] async fn test_enter_staging_clean_staging_manifest_error() { common_telemetry::init_default_ut_logging(); diff --git a/src/mito2/src/error.rs b/src/mito2/src/error.rs index fe9cef1d3c..b9a57bf794 100644 --- a/src/mito2/src/error.rs +++ b/src/mito2/src/error.rs @@ -67,6 +67,20 @@ pub enum Error { error: object_store::Error, }, + #[snafu(display( + "Manifest delta {} disappeared after it was listed, path: {}", + version, + path + ))] + ManifestDeltaNotFound { + version: ManifestVersion, + path: String, + #[snafu(source)] + error: object_store::Error, + #[snafu(implicit)] + location: Location, + }, + #[snafu(display("Fail to compress object by {}, path: {}", compress_type, path))] CompressObject { compress_type: CompressionType, @@ -1356,6 +1370,7 @@ impl Error { pub(crate) fn is_object_not_found(&self) -> bool { match self { Error::OpenDal { error, .. } => error.kind() == ErrorKind::NotFound, + Error::ManifestDeltaNotFound { .. } => true, _ => false, } } @@ -1397,7 +1412,9 @@ impl ErrorExt for Error { match self { DataTypeMismatch { source, .. } => source.status_code(), - OpenDal { .. } | ReadParquet { .. } => StatusCode::StorageUnavailable, + OpenDal { .. } | ManifestDeltaNotFound { .. } | ReadParquet { .. } => { + StatusCode::StorageUnavailable + } WriteWal { source, .. } | ReadWal { source, .. } | DeleteWal { source, .. } => { source.status_code() } @@ -1592,7 +1609,8 @@ impl ErrorExt for Error { | UpdateManifest { .. } | RegionStopped { .. } | RegionBusy { .. } - | FlushableRegionState { .. } => RetryHint::Retryable, + | FlushableRegionState { .. } + | ManifestDeltaNotFound { .. } => RetryHint::Retryable, OpenDal { error, .. } | DeleteSsts { error, .. } diff --git a/src/mito2/src/flush.rs b/src/mito2/src/flush.rs index a4a42970e4..7c088dea58 100644 --- a/src/mito2/src/flush.rs +++ b/src/mito2/src/flush.rs @@ -24,10 +24,10 @@ use bytes::Bytes; use common_base::cancellation::CancellableFuture; use common_telemetry::{debug, error, info}; use datatypes::arrow::datatypes::SchemaRef; -use datatypes::extension::json::is_json2_extension_type; use partition::expr::PartitionExpr; use smallvec::{SmallVec, smallvec}; -use snafu::ResultExt; +use snafu::{ResultExt, ensure}; +use store_api::metadata::RegionMetadataRef; use store_api::region_request::RegionFlushReason; use store_api::storage::{RegionId, SequenceNumber}; use strum::IntoStaticStr; @@ -37,15 +37,15 @@ use crate::access_layer::{ AccessLayerRef, Metrics, OperationType, SstInfoArray, SstWriteRequest, WriteType, }; use crate::cache::CacheManagerRef; +use crate::compaction::{collect_json2_rewrite_plans, rewrite_json2_batch, rewrite_json2_schema}; use crate::config::MitoConfig; use crate::engine::region_hook::SstFileInfo; use crate::error::{ Error, FlushCancelledSnafu, FlushRegionSnafu, JoinSnafu, RegionBusySnafu, RegionClosedSnafu, - RegionDroppedSnafu, RegionTruncatedSnafu, Result, + RegionDroppedSnafu, RegionTruncatedSnafu, Result, UnexpectedSnafu, }; use crate::manifest::action::{RegionEdit, RegionMetaAction, RegionMetaActionList}; use crate::memtable::bulk::ENCODE_ROW_THRESHOLD; -use crate::memtable::bulk::json_align::Json2Aligner; use crate::memtable::{BoxedRecordBatchIterator, EncodedRange, MemtableRanges, RangesOptions}; use crate::metrics::{ FLUSH_BYTES_TOTAL, FLUSH_ELAPSED, FLUSH_FAILURE_TOTAL, FLUSH_FILE_TOTAL, FLUSH_REQUESTS_TOTAL, @@ -708,6 +708,7 @@ impl RegionFlushTask { let flat_sources = memtable_flat_sources( batch_schema, mem_ranges, + &version.metadata, &version.options, field_column_start, )?; @@ -901,6 +902,7 @@ struct FlatSources { fn memtable_flat_sources( schema: SchemaRef, mem_ranges: MemtableRanges, + metadata: &RegionMetadataRef, options: &RegionOptions, field_column_start: usize, ) -> Result { @@ -918,6 +920,7 @@ fn memtable_flat_sources( if let Some(encoded) = only_range.encoded() { flat_sources.encoded.push((encoded, max_sequence)); } else { + let schema = only_range.record_batch_schema_hint().unwrap_or(schema); let iter = only_range.build_record_batch_iter(None, None)?; // Dedup according to append mode and merge mode. // Even single range may have duplicate rows. @@ -951,12 +954,23 @@ fn memtable_flat_sources( let mut input_iters = Vec::with_capacity(num_ranges); let mut current_ranges = Vec::new(); - let has_json2 = schema.fields().iter().any(is_json2_extension_type); - let mut json_align_schemas = if has_json2 { - Some(Vec::with_capacity(num_ranges)) - } else { - None - }; + let schemas = ranges + .values() + .filter(|range| range.encoded().is_none()) + .map(|range| { + ( + range + .record_batch_schema_hint() + .unwrap_or_else(|| schema.clone()), + range.num_rows() as u64, + ) + }) + .collect::>(); + let plans = Arc::new(collect_json2_rewrite_plans(metadata, &schemas)?); + let schema = rewrite_json2_schema( + schemas.first().map(|(schema, _)| schema).unwrap_or(&schema), + &plans, + ); for (_range_id, range) in ranges { if let Some(encoded) = range.encoded() { @@ -965,15 +979,26 @@ fn memtable_flat_sources( continue; } - // Collect schemas if has json2 field. - if let Some(schemas) = json_align_schemas.as_mut() { - let schema = range - .record_batch_schema_hint() - .unwrap_or_else(|| schema.clone()); - schemas.push(schema); + if let Some(actual) = range.record_batch_schema_hint() { + let actual = rewrite_json2_schema(&actual, &plans); + ensure!( + actual == schema, + UnexpectedSnafu { + reason: format!( + "Different schemas found in a MemtableRanges, expected: {}, actual: {}", + schema, actual, + ), + } + ) } let iter = range.build_record_batch_iter(None, None)?; + let iter: BoxedRecordBatchIterator = if plans.is_empty() { + iter + } else { + let plans = plans.clone(); + Box::new(iter.map(move |batch| rewrite_json2_batch(batch?, &plans))) + }; input_iters.push(iter); let range_rows = range.num_rows(); last_iter_rows += range_rows; @@ -1007,11 +1032,6 @@ fn memtable_flat_sources( let input_iters = std::mem::replace(&mut input_iters, Vec::with_capacity(num_ranges)); - let (schema, input_iters) = maybe_align_json2_iters( - schema.clone(), - json_align_schemas.take(), - input_iters, - )?; let maybe_dedup = merge_and_dedup_with_batch_size( &schema, @@ -1022,17 +1042,12 @@ fn memtable_flat_sources( batch_size, )?; - flat_sources - .sources - .push((FlatSource::new_iter(schema, maybe_dedup), max_sequence)); + flat_sources.sources.push(( + FlatSource::new_iter(schema.clone(), maybe_dedup), + max_sequence, + )); last_iter_rows = 0; current_ranges.clear(); - - json_align_schemas = if has_json2 { - Some(Vec::with_capacity(num_ranges)) - } else { - None - }; } } @@ -1046,9 +1061,6 @@ fn memtable_flat_sources( rows_remaining ); - let (schema, input_iters) = - maybe_align_json2_iters(schema, json_align_schemas, input_iters)?; - let max_sequence = current_ranges .iter() .map(|r| r.stats().max_sequence()) @@ -1078,24 +1090,6 @@ fn memtable_flat_sources( Ok(flat_sources) } -fn maybe_align_json2_iters( - schema: SchemaRef, - schemas: Option>, - input_iters: Vec, -) -> Result<(SchemaRef, Vec)> { - let Some(schemas) = schemas else { - return Ok((schema, input_iters)); - }; - - let aligner = Json2Aligner::try_new(schemas)?; - let input_iters = input_iters - .into_iter() - .map(|input_iter| aligner.wrap_iter(input_iter)) - .collect(); - - Ok((aligner.schema().clone(), input_iters)) -} - /// Merges multiple record batch iterators and applies deduplication based on the specified mode. /// /// This function is used during the flush process to combine data from multiple memtable ranges @@ -1646,6 +1640,8 @@ mod tests { use api::v1::{OpType, Rows}; use common_error::ext::ErrorExt; use common_error::status_code::StatusCode; + use datatypes::arrow::datatypes::Schema; + use datatypes::arrow::record_batch::RecordBatch; use mito_codec::row_converter::build_primary_key_codec; use tokio::sync::oneshot; @@ -1654,7 +1650,9 @@ mod tests { use crate::error::InvalidSchedulerStateSnafu; use crate::memtable::bulk::part::BulkPartConverter; use crate::memtable::time_series::TimeSeriesMemtableBuilder; - use crate::memtable::{Memtable, RangesOptions}; + use crate::memtable::{ + IterBuilder, Memtable, MemtableRange, MemtableRangeContext, MemtableStats, RangesOptions, + }; use crate::request::WriteRequest; use crate::schedule::scheduler::Scheduler; use crate::sst::{FlatSchemaOptions, to_flat_sst_arrow_schema}; @@ -2243,6 +2241,7 @@ mod tests { let flat_sources = memtable_flat_sources( schema.clone(), mem_ranges, + &metadata, &options, metadata.primary_key.len(), ) @@ -2271,9 +2270,14 @@ mod tests { ..Default::default() }; - let flat_sources = - memtable_flat_sources(schema, mem_ranges, &options, metadata.primary_key.len()) - .unwrap(); + let flat_sources = memtable_flat_sources( + schema, + mem_ranges, + &metadata, + &options, + metadata.primary_key.len(), + ) + .unwrap(); assert!(flat_sources.encoded.is_empty()); assert_eq!(1, flat_sources.sources.len()); @@ -2288,6 +2292,118 @@ mod tests { } } + #[test] + fn test_memtable_flat_sources_uses_non_encoded_schema() -> Result<()> { + struct TestIterBuilder { + schema: SchemaRef, + batch: Option, + } + + impl IterBuilder for TestIterBuilder { + fn build( + &self, + _metrics: Option, + ) -> Result { + unimplemented!() + } + + fn is_record_batch(&self) -> bool { + true + } + + fn build_record_batch( + &self, + _time_range: Option<(common_time::Timestamp, common_time::Timestamp)>, + _metrics: Option, + ) -> Result { + let Some(batch) = self.batch.clone() else { + unimplemented!() + }; + Ok(Box::new(std::iter::once(Ok(batch)))) + } + + fn record_batch_schema_hint(&self) -> Option { + Some(self.schema.clone()) + } + + fn encoded_range(&self) -> Option { + self.batch.is_none().then(|| EncodedRange { + data: Bytes::new(), + sst_info: SstInfo::default(), + }) + } + } + + let metadata = metadata_for_test(); + let schema = to_flat_sst_arrow_schema( + &metadata, + &FlatSchemaOptions::from_encoding(metadata.primary_key_encoding), + ); + let pk_codec = build_primary_key_codec(&metadata); + let mut converter = BulkPartConverter::new(&metadata, schema.clone(), 1, pk_codec, true); + let kvs = build_key_values_with_ts_seq_values( + &metadata, + "key".to_string(), + 1, + std::iter::once(1000), + std::iter::once(Some(1.0)), + 1, + ); + converter.append_key_values(&kvs)?; + let batch = converter.convert()?.batch; + let encoded_schema = Arc::new(Schema::empty()); + + let new_range = |id, builder| { + MemtableRange::new( + Arc::new(MemtableRangeContext::new( + id, + Box::new(builder), + Default::default(), + )), + MemtableStats { + num_rows: 1, + ..Default::default() + }, + ) + }; + let mut ranges = std::collections::BTreeMap::new(); + ranges.insert( + 0, + new_range( + 0, + TestIterBuilder { + schema: encoded_schema.clone(), + batch: None, + }, + ), + ); + ranges.insert( + 1, + new_range( + 0, + TestIterBuilder { + schema: schema.clone(), + batch: Some(batch), + }, + ), + ); + + let sources = memtable_flat_sources( + encoded_schema, + MemtableRanges { ranges }, + &metadata, + &RegionOptions { + append_mode: true, + ..Default::default() + }, + metadata.primary_key.len(), + )?; + assert_eq!(1, sources.encoded.len()); + assert_eq!(1, sources.sources.len()); + assert_eq!(&schema, sources.sources[0].0.schema()); + Ok(()) + } + #[tokio::test] async fn test_schedule_pending_request_on_flush_success() { common_telemetry::init_default_ut_logging(); diff --git a/src/mito2/src/manifest/checkpointer.rs b/src/mito2/src/manifest/checkpointer.rs index 9dd35e189a..232421a262 100644 --- a/src/mito2/src/manifest/checkpointer.rs +++ b/src/mito2/src/manifest/checkpointer.rs @@ -14,11 +14,14 @@ use std::fmt::Debug; use std::sync::Arc; -use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +use std::sync::atomic::{AtomicU64, Ordering}; +use common_runtime::JoinHandle; use common_telemetry::{error, info, warn}; use store_api::storage::RegionId; use store_api::{MIN_VERSION, ManifestVersion}; +#[cfg(test)] +use tokio::sync::Notify; use crate::error::Result; use crate::manifest::action::{RegionCheckpoint, RegionManifest}; @@ -31,6 +34,9 @@ use crate::metrics::MANIFEST_OP_ELAPSED; pub(crate) struct Checkpointer { manifest_options: RegionManifestOptions, inner: Arc, + checkpoint_task: Option>, + #[cfg(test)] + pending_checkpoint_wait_started: Arc, } #[derive(Debug)] @@ -38,15 +44,10 @@ struct Inner { region_id: RegionId, manifest_store: ManifestObjectStore, last_checkpoint_version: AtomicU64, - is_doing_checkpoint: AtomicBool, } impl Inner { async fn do_checkpoint(&self, checkpoint: RegionCheckpoint) { - let _guard = scopeguard::guard(&self.is_doing_checkpoint, |x| { - x.store(false, Ordering::Relaxed); - }); - let _t = MANIFEST_OP_ELAPSED .with_label_values(&["checkpoint"]) .start_timer(); @@ -90,14 +91,6 @@ impl Inner { fn region_id(&self) -> RegionId { self.region_id } - - fn is_doing_checkpoint(&self) -> bool { - self.is_doing_checkpoint.load(Ordering::Relaxed) - } - - fn set_doing_checkpoint(&self) { - self.is_doing_checkpoint.store(true, Ordering::Relaxed); - } } impl Checkpointer { @@ -113,8 +106,10 @@ impl Checkpointer { region_id, manifest_store, last_checkpoint_version: AtomicU64::new(last_checkpoint_version), - is_doing_checkpoint: AtomicBool::new(false), }), + checkpoint_task: None, + #[cfg(test)] + pending_checkpoint_wait_started: Arc::new(Notify::new()), } } @@ -140,7 +135,19 @@ impl Checkpointer { /// Check if it's needed to do checkpoint for the region by the checkpoint distance. /// If needed, and there's no currently running checkpoint task, it will start a new checkpoint /// task running in the background. - pub(crate) fn maybe_do_checkpoint(&self, manifest: &RegionManifest) { + pub(crate) async fn maybe_do_checkpoint(&mut self, manifest: &RegionManifest) { + if self + .checkpoint_task + .as_ref() + .is_some_and(|handle| !handle.is_finished()) + { + return; + } + + // Reap a completed task before checking whether to start the next one. + // This keeps the handle as the single source of truth for task state. + self.wait_for_pending_checkpoint().await; + if self.manifest_options.checkpoint_distance == 0 { return; } @@ -152,12 +159,6 @@ impl Checkpointer { return; } - // We can simply check whether there's a running checkpoint task like this, all because of - // the caller of this function is ran single threaded, inside the lock of RegionManifestManager. - if self.inner.is_doing_checkpoint() { - return; - } - let start_version = if last_checkpoint_version == 0 { // Checkpoint version can't be zero by implementation. // So last checkpoint version is zero means no last checkpoint. @@ -181,17 +182,44 @@ impl Checkpointer { self.do_checkpoint(checkpoint); } - fn do_checkpoint(&self, checkpoint: RegionCheckpoint) { - self.inner.set_doing_checkpoint(); - + fn do_checkpoint(&mut self, checkpoint: RegionCheckpoint) { let inner = self.inner.clone(); - common_runtime::spawn_global(async move { + self.checkpoint_task = Some(common_runtime::spawn_global(async move { inner.do_checkpoint(checkpoint).await; - }); + })); + } + + /// Waits for the current checkpoint task without removing its handle first. + /// + /// Keeping the handle in `self` while awaiting is important. If the caller is + /// cancelled, another lifecycle transition can still wait for the same task. + pub(crate) async fn wait_for_pending_checkpoint(&mut self) { + let Some(handle) = self.checkpoint_task.as_mut() else { + return; + }; + + #[cfg(test)] + self.pending_checkpoint_wait_started.notify_one(); + + let result = (&mut *handle).await; + // There is no cancellation point between observing completion and + // clearing the handle. + self.checkpoint_task = None; + + if let Err(e) = result { + warn!(e; "Failed to join checkpoint task for region {}", self.inner.region_id()); + } } #[cfg(test)] pub(crate) fn is_doing_checkpoint(&self) -> bool { - self.inner.is_doing_checkpoint() + self.checkpoint_task + .as_ref() + .is_some_and(|handle| !handle.is_finished()) + } + + #[cfg(test)] + pub(crate) fn pending_checkpoint_wait_started(&self) -> Arc { + self.pending_checkpoint_wait_started.clone() } } diff --git a/src/mito2/src/manifest/manager.rs b/src/mito2/src/manifest/manager.rs index fcc48395fd..c6a0d3f322 100644 --- a/src/mito2/src/manifest/manager.rs +++ b/src/mito2/src/manifest/manager.rs @@ -550,6 +550,28 @@ impl RegionManifestManager { &mut self, action_list: RegionMetaActionList, is_staging: bool, + ) -> Result { + self.update_inner(action_list, is_staging, !is_staging) + .await + } + + /// Updates the normal manifest without starting a checkpoint. + /// + /// This is only used while a leader is downgrading. The final flush still + /// publishes its manifest edit, but must not start cleanup after the + /// downgrade checkpoint barrier. + pub(crate) async fn update_normal_without_checkpoint( + &mut self, + action_list: RegionMetaActionList, + ) -> Result { + self.update_inner(action_list, false, false).await + } + + async fn update_inner( + &mut self, + action_list: RegionMetaActionList, + is_staging: bool, + allow_checkpoint: bool, ) -> Result { let _t = MANIFEST_OP_ELAPSED .with_label_values(&["update"]) @@ -616,13 +638,21 @@ impl RegionManifestManager { .checkpointer .update_manifest_removed_files(new_manifest)?; self.manifest = Arc::new(updated_manifest); - self.checkpointer - .maybe_do_checkpoint(self.manifest.as_ref()); + if allow_checkpoint { + self.checkpointer + .maybe_do_checkpoint(self.manifest.as_ref()) + .await; + } } Ok(version) } + /// Waits for an in-flight checkpoint to finish, including its cleanup. + pub(crate) async fn wait_for_pending_checkpoint(&mut self) { + self.checkpointer.wait_for_pending_checkpoint().await; + } + /// Clear deleted files from manifest's `removed_files` field without update version. Notice if datanode exit before checkpoint then new manifest by open region may still contain these deleted files, which is acceptable for gc process. pub fn clear_deleted_files(&mut self, deleted_files: Vec) { let mut manifest = (*self.manifest()).clone(); diff --git a/src/mito2/src/manifest/storage/delta.rs b/src/mito2/src/manifest/storage/delta.rs index 594b56ddae..8f9962a3c6 100644 --- a/src/mito2/src/manifest/storage/delta.rs +++ b/src/mito2/src/manifest/storage/delta.rs @@ -26,7 +26,8 @@ use tokio::sync::Semaphore; use crate::cache::manifest_cache::ManifestCache; use crate::error::{ - CompressObjectSnafu, DecompressObjectSnafu, InvalidScanIndexSnafu, OpenDalSnafu, Result, + CompressObjectSnafu, DecompressObjectSnafu, InvalidScanIndexSnafu, ManifestDeltaNotFoundSnafu, + OpenDalSnafu, Result, }; use crate::manifest::storage::size_tracker::Tracker; use crate::manifest::storage::utils::{ @@ -212,11 +213,16 @@ impl DeltaStorage { // Fetch from remote object store let compress_type = file_compress_type(entry.name()); - let bytes = self - .object_store - .read(entry.path()) - .await - .context(OpenDalSnafu)?; + let bytes = match self.object_store.read(entry.path()).await { + Ok(bytes) => bytes, + Err(error) if error.kind() == ErrorKind::NotFound => { + return Err(error).context(ManifestDeltaNotFoundSnafu { + version: *v, + path: entry.path(), + }); + } + Err(error) => return Err(error).context(OpenDalSnafu), + }; let data = compress_type .decode(bytes) .await diff --git a/src/mito2/src/manifest/tests/checkpoint.rs b/src/mito2/src/manifest/tests/checkpoint.rs index 07f879701a..86679630ac 100644 --- a/src/mito2/src/manifest/tests/checkpoint.rs +++ b/src/mito2/src/manifest/tests/checkpoint.rs @@ -18,13 +18,16 @@ use std::sync::atomic::{AtomicUsize, Ordering}; use std::time::Duration; use common_datasource::compression::CompressionType; +use common_error::ext::{ErrorExt, RetryHint}; +use common_error::status_code::StatusCode; use object_store::layers::mock::{ - Error as MockError, ErrorKind, MockLayerBuilder, OpDelete, Result as MockResult, oio, + Buffer, Error as MockError, ErrorKind, MockLayer, MockLayerBuilder, OpDelete, + Result as MockResult, oio, }; use store_api::storage::{FileId, RegionId}; use strum::IntoEnumIterator; -use crate::error::Error::ChecksumMismatch; +use crate::error::Error::{ChecksumMismatch, ManifestDeltaNotFound}; use crate::manifest::action::{ RegionCheckpoint, RegionEdit, RegionMetaAction, RegionMetaActionList, }; @@ -33,7 +36,7 @@ use crate::manifest::storage::checkpoint::CheckpointMetadata; use crate::manifest::storage::is_delta_file; use crate::manifest::tests::utils::basic_region_metadata; use crate::sst::file::FileMeta; -use crate::test_util::TestEnv; +use crate::test_util::{CheckpointTaskBlocker, TestEnv}; async fn build_manager( checkpoint_distance: u64, @@ -85,6 +88,31 @@ fn nop_action() -> RegionMetaActionList { })]) } +struct NotFoundReader; + +impl oio::Read for NotFoundReader { + async fn read(&mut self) -> MockResult { + Err(MockError::new( + ErrorKind::NotFound, + "mock listed manifest delta not found", + )) + } +} + +fn fail_manifest_delta_reads_layer() -> MockLayer { + MockLayerBuilder::default() + .reader_factory(Arc::new(|path, _args, inner| { + let file_name = path.rsplit('/').next().unwrap_or(path); + if is_delta_file(file_name) { + Box::new(NotFoundReader) + } else { + inner + } + })) + .build() + .unwrap() +} + #[tokio::test] async fn manager_without_checkpoint() { let (_env, mut manager) = build_manager(0, CompressionType::Uncompressed).await; @@ -589,3 +617,167 @@ async fn checkpoint_advances_and_recovery_works_when_delete_fails() { .expect("manifest should be recoverable"); assert_eq!(reopened.manifest().manifest_version, 10); } + +#[tokio::test] +async fn open_preserves_listed_delta_not_found_retry_hint() { + let env = TestEnv::new() + .await + .with_mock_layer(fail_manifest_delta_reads_layer()); + let metadata = Arc::new(basic_region_metadata()); + let mut manager = env + .create_manifest_manager(CompressionType::Uncompressed, 0, Some(metadata.clone())) + .await + .unwrap() + .unwrap(); + manager.stop().await; + + let error = env + .create_manifest_manager(CompressionType::Uncompressed, 0, None) + .await + .expect_err("reopen must fail on the mocked delta read"); + assert_matches!( + &error, + ManifestDeltaNotFound { + version: 0, + path, + error, + .. + } if path.ends_with("00000000000000000000.json") + && error.kind() == object_store::ErrorKind::NotFound + ); + assert_eq!(StatusCode::StorageUnavailable, error.status_code()); + assert_eq!(RetryHint::Retryable, error.retry_hint()); +} + +#[tokio::test] +async fn install_preserves_listed_delta_not_found_retry_hint() { + let env = TestEnv::new() + .await + .with_mock_layer(fail_manifest_delta_reads_layer()); + let metadata = Arc::new(basic_region_metadata()); + let mut manager = env + .create_manifest_manager(CompressionType::Uncompressed, 0, Some(metadata)) + .await + .unwrap() + .unwrap(); + let mut store = manager.store(); + store + .save(1, &nop_action().encode().unwrap(), false) + .await + .unwrap(); + + let error = manager.install_manifest_to(1).await.unwrap_err(); + assert_matches!( + &error, + ManifestDeltaNotFound { + version: 1, + path, + .. + } if path.ends_with("00000000000000000001.json") + ); + assert_eq!(RetryHint::Retryable, error.retry_hint()); +} + +#[tokio::test] +async fn cancelled_waiter_keeps_pending_checkpoint_handle() { + let (blocker, mock_layer) = CheckpointTaskBlocker::block_cleanup(); + let env = TestEnv::new().await.with_mock_layer(mock_layer); + let metadata = Arc::new(basic_region_metadata()); + let mut manager = env + .create_manifest_manager(CompressionType::Uncompressed, 1, Some(metadata)) + .await + .unwrap() + .unwrap(); + + manager.update(nop_action(), false).await.unwrap(); + tokio::time::timeout(Duration::from_secs(5), blocker.wait_until_blocked()) + .await + .expect("checkpoint cleanup did not start"); + + let wait_started = manager.checkpointer().pending_checkpoint_wait_started(); + let wait_started = wait_started.notified(); + let mut first_wait = Box::pin(manager.wait_for_pending_checkpoint()); + tokio::select! { + _ = wait_started => {} + _ = &mut first_wait => panic!("checkpoint waiter returned before cleanup finished"), + } + drop(first_wait); + assert!(manager.checkpointer().is_doing_checkpoint()); + + blocker.release(); + manager.wait_for_pending_checkpoint().await; + assert!(!manager.checkpointer().is_doing_checkpoint()); + + let (version, _) = manager + .store() + .load_last_checkpoint() + .await + .unwrap() + .expect("checkpoint must be published"); + assert_eq!(1, version); +} + +#[tokio::test] +async fn running_checkpoint_prevents_scheduling_another_checkpoint() { + let (blocker, mock_layer) = CheckpointTaskBlocker::block_cleanup(); + let env = TestEnv::new().await.with_mock_layer(mock_layer); + let metadata = Arc::new(basic_region_metadata()); + let mut manager = env + .create_manifest_manager(CompressionType::Uncompressed, 1, Some(metadata)) + .await + .unwrap() + .unwrap(); + + manager.update(nop_action(), false).await.unwrap(); + tokio::time::timeout(Duration::from_secs(5), blocker.wait_until_blocked()) + .await + .expect("checkpoint cleanup did not start"); + + // The second update is eligible for checkpointing, but the first task still + // owns the pending handle and must prevent another task from being scheduled. + manager.update(nop_action(), false).await.unwrap(); + blocker.release(); + manager.wait_for_pending_checkpoint().await; + + assert_eq!(2, manager.manifest().manifest_version); + assert_eq!(1, manager.checkpointer().last_checkpoint_version()); +} + +#[tokio::test] +async fn completed_checkpoint_is_reaped_before_scheduling_the_next_one() { + let (blocker, mock_layer) = CheckpointTaskBlocker::block_cleanup(); + let env = TestEnv::new().await.with_mock_layer(mock_layer); + let metadata = Arc::new(basic_region_metadata()); + let mut manager = env + .create_manifest_manager(CompressionType::Uncompressed, 1, Some(metadata)) + .await + .unwrap() + .unwrap(); + + manager.update(nop_action(), false).await.unwrap(); + tokio::time::timeout(Duration::from_secs(5), blocker.wait_until_blocked()) + .await + .expect("first checkpoint cleanup did not start"); + blocker.release(); + + tokio::time::timeout(Duration::from_secs(5), async { + while manager.checkpointer().is_doing_checkpoint() { + tokio::task::yield_now().await; + } + }) + .await + .expect("first checkpoint did not finish"); + assert_eq!(1, manager.checkpointer().last_checkpoint_version()); + + // Do not explicitly wait/reap the completed handle. The next eligible + // update must reap it before scheduling another checkpoint. + blocker.arm_next_close(); + manager.update(nop_action(), false).await.unwrap(); + tokio::time::timeout(Duration::from_secs(5), blocker.wait_until_blocked()) + .await + .expect("second checkpoint cleanup did not start"); + blocker.release(); + manager.wait_for_pending_checkpoint().await; + + assert_eq!(2, manager.checkpointer().last_checkpoint_version()); +} diff --git a/src/mito2/src/memtable.rs b/src/mito2/src/memtable.rs index 6cc6e327ae..8c107d15f7 100644 --- a/src/mito2/src/memtable.rs +++ b/src/mito2/src/memtable.rs @@ -30,11 +30,11 @@ pub use mito_codec::key_values::KeyValues; use mito_codec::row_converter::{PrimaryKeyCodec, build_primary_key_codec}; use snafu::ensure; use store_api::codec::PrimaryKeyEncoding; -use store_api::metadata::RegionMetadataRef; +use store_api::metadata::{RegionMetadata, RegionMetadataRef}; use store_api::storage::{ColumnId, SequenceNumber, SequenceRange}; use crate::config::MitoConfig; -use crate::error::{Result, UnsupportedOperationSnafu}; +use crate::error::{InvalidRegionOptionsSnafu, Result, UnsupportedOperationSnafu}; use crate::flush::WriteBufferManagerRef; use crate::memtable::bulk::{BulkMemtableBuilder, CompactDispatcher}; use crate::memtable::time_series::TimeSeriesMemtableBuilder; @@ -401,6 +401,26 @@ pub(crate) struct MemtableBuilderProvider { compact_dispatcher: Arc, } +/// Ensures JSON2 columns are not used with [`TimeSeriesMemtable`]. +pub(crate) fn ensure_json2_not_use_time_series_memtable( + metadata: &RegionMetadata, + options: &RegionOptions, +) -> Result<()> { + if metadata + .column_metadatas + .iter() + .any(|x| x.column_schema.data_type.is_json2()) + { + ensure!( + !matches!(&options.memtable, Some(MemtableOptions::TimeSeries)), + InvalidRegionOptionsSnafu { + reason: "JSON2 columns only support BulkMemtable", + } + ); + } + Ok(()) +} + impl MemtableBuilderProvider { pub(crate) fn new( write_buffer_manager: Option, @@ -772,9 +792,15 @@ impl MemtableRange { mod tests { use std::sync::Arc; + use common_error::ext::WhateverResult; + use datatypes::prelude::ConcreteDataType; + use datatypes::types::json_type::{JsonNativeType, JsonObjectType}; + use store_api::metadata::RegionMetadataBuilder; + use super::*; use crate::flush::{WriteBufferManager, WriteBufferManagerImpl}; use crate::memtable::bulk::BulkMemtableConfig; + use crate::test_util::sst_util::sst_region_metadata; #[test] fn test_alloc_tracker_without_manager() { @@ -848,4 +874,27 @@ mod tests { assert_eq!(&config, builder.config()); } + + #[test] + fn test_json2_requires_bulk_memtable() -> WhateverResult<()> { + let mut metadata = sst_region_metadata(); + metadata.column_metadatas[2].column_schema.data_type = + ConcreteDataType::json2(JsonNativeType::Object(JsonObjectType::new())); + let metadata = RegionMetadataBuilder::from_existing(metadata).build()?; + let mut options = RegionOptions { + sst_format: Some(FormatType::PrimaryKey), + memtable: Some(MemtableOptions::TimeSeries), + ..Default::default() + }; + + let err = ensure_json2_not_use_time_series_memtable(&metadata, &options).unwrap_err(); + assert!( + err.to_string() + .contains("JSON2 columns only support BulkMemtable") + ); + + options.memtable = Some(MemtableOptions::Bulk(BulkMemtableConfig::default())); + ensure_json2_not_use_time_series_memtable(&metadata, &options)?; + Ok(()) + } } diff --git a/src/mito2/src/memtable/builder.rs b/src/mito2/src/memtable/builder.rs index 7e37077f3e..d072b85679 100644 --- a/src/mito2/src/memtable/builder.rs +++ b/src/mito2/src/memtable/builder.rs @@ -22,8 +22,8 @@ use datatypes::arrow::array::{ }; use datatypes::arrow::buffer::Buffer; use datatypes::arrow_array::StringArray; -use datatypes::data_type::DataType; use datatypes::prelude::{ConcreteDataType, MutableVector, VectorRef}; +use datatypes::schema::ColumnSchema; use datatypes::value::ValueRef; use datatypes::vectors::StringVector; @@ -35,11 +35,11 @@ pub(crate) enum FieldBuilder { impl FieldBuilder { /// Creates a [FieldBuilder] instance with given type and capacity. - pub fn create(data_type: &ConcreteDataType, init_cap: usize) -> Self { - if let ConcreteDataType::String(_) = data_type { + pub(crate) fn create(column_schema: &ColumnSchema, init_cap: usize) -> Self { + if let ConcreteDataType::String(_) = &column_schema.data_type { Self::String(StringBuilder::with_capacity(init_cap / 16, init_cap)) } else { - Self::Other(data_type.create_mutable_vector(init_cap)) + Self::Other(column_schema.create_mutable_vector(init_cap)) } } diff --git a/src/mito2/src/memtable/bulk.rs b/src/mito2/src/memtable/bulk.rs index 596db7aa34..1b67d487a5 100644 --- a/src/mito2/src/memtable/bulk.rs +++ b/src/mito2/src/memtable/bulk.rs @@ -16,7 +16,6 @@ pub(crate) mod chunk_reader; pub mod context; -pub(crate) mod json_align; pub mod part; pub mod part_reader; mod row_group_reader; @@ -44,10 +43,12 @@ use store_api::metadata::RegionMetadataRef; use store_api::storage::{ColumnId, FileId, RegionId, SequenceRange}; use tokio::sync::Semaphore; +use crate::compaction::{ + Json2RewritePlans, collect_json2_rewrite_plans, rewrite_json2_batch, rewrite_json2_schema, +}; use crate::error::{Result, UnsupportedOperationSnafu}; use crate::flush::WriteBufferManagerRef; use crate::memtable::bulk::context::BulkIterContext; -use crate::memtable::bulk::json_align::Json2Aligner; use crate::memtable::bulk::part::{ BulkPart, BulkPartEncodeMetrics, BulkPartEncoder, MultiBulkPart, UnorderedPart, should_prune_bulk_part, @@ -464,7 +465,8 @@ impl Memtable for BulkMemtable { // Compacts unordered_part if the row or byte threshold is exceeded. if bulk_parts.should_compact_unordered_part(self.config.encode_bytes_threshold) - && let Some(bulk_part) = bulk_parts.unordered_part.to_bulk_part()? + && let Some(bulk_part) = + bulk_parts.unordered_part.to_bulk_part(&self.metadata)? { bulk_parts.parts.push(BulkPartWrapper { part: PartToMerge::Bulk { @@ -525,7 +527,8 @@ impl Memtable for BulkMemtable { // Adds range for unordered part if not empty if !bulk_parts.unordered_part.is_empty() - && let Some(unordered_bulk_part) = bulk_parts.unordered_part.to_bulk_part()? + && let Some(unordered_bulk_part) = + bulk_parts.unordered_part.to_bulk_part(&self.metadata)? { let part_stats = unordered_bulk_part.to_memtable_stats(&self.metadata); let range = MemtableRange::new( @@ -1116,8 +1119,9 @@ impl PartToMerge { fn create_iterator( self, context: Arc, + plans: Arc, ) -> Result> { - match self { + let iter = match self { PartToMerge::Bulk { part, .. } => { let series_count = part.estimated_series_count(); let iter = BulkPartBatchIter::from_single( @@ -1127,10 +1131,18 @@ impl PartToMerge { series_count, None, // No metrics for merging ); - Ok(Some(Box::new(iter) as BoxedRecordBatchIterator)) + Some(Box::new(iter) as BoxedRecordBatchIterator) } - PartToMerge::Multi { part, .. } => part.read(context, None, None), - PartToMerge::Encoded { part, .. } => part.read(context, None, None), + PartToMerge::Multi { part, .. } => part.read(context, None, None)?, + PartToMerge::Encoded { part, .. } => part.read(context, None, None)?, + }; + if plans.is_empty() { + Ok(iter) + } else { + Ok(iter.map(|x| { + Box::new(x.map(move |batch| rewrite_json2_batch(batch?, &plans))) + as BoxedRecordBatchIterator + })) } } } @@ -1289,19 +1301,37 @@ impl MemtableCompactor { batch_size, )?); - let aligner = Json2Aligner::try_new(parts_to_merge.iter().map(PartToMerge::arrow_schema))?; + let schemas = parts_to_merge + .iter() + .map(|part| (part.arrow_schema(), part.num_rows() as u64)) + .collect::>(); + let plans = Arc::new(collect_json2_rewrite_plans(metadata, &schemas)?); + + debug_assert!(parts_to_merge.windows(2).all(|w| rewrite_json2_schema( + &w[0].arrow_schema(), + &plans + ) == rewrite_json2_schema( + &w[1].arrow_schema(), + &plans + ))); + // Parts in one merge group may differ only in their JSON2 physical layouts. So every source + // schema is therefore a valid template for producing the final target schema that has + // the union JSON2 types (rewritten). + let schema = rewrite_json2_schema(&parts_to_merge[0].arrow_schema(), &plans); let iterators: Vec = parts_to_merge .into_iter() - .filter_map(|part| part.create_iterator(context.clone()).ok().flatten()) - .map(|iter| aligner.wrap_iter(iter)) + .map(|part| part.create_iterator(context.clone(), plans.clone())) + .collect::>>()? + .into_iter() + .flatten() .collect(); if iterators.is_empty() { return Ok(None); } - let merged_iter = FlatMergeIterator::new(aligner.schema().clone(), iterators, batch_size)?; + let merged_iter = FlatMergeIterator::new(schema.clone(), iterators, batch_size)?; let boxed_iter: BoxedRecordBatchIterator = if dedup { match merge_mode { @@ -1310,8 +1340,7 @@ impl MemtableCompactor { Box::new(dedup_iter) } MergeMode::LastNonNull => { - let field_column_start = - field_column_start(metadata, aligner.schema().fields().len()); + let field_column_start = field_column_start(metadata, schema.fields().len()); let dedup_iter = FlatDedupIterator::new( merged_iter, @@ -1332,7 +1361,7 @@ impl MemtableCompactor { let mut metrics = BulkPartEncodeMetrics::default(); let encoded_part = encoder.encode_record_batch_iter( boxed_iter, - aligner.schema().clone(), + schema, min_timestamp, max_timestamp, max_sequence, @@ -1532,9 +1561,14 @@ mod tests { use api::helper::encode_json_value; use api::v1::value::ValueData; use api::v1::{Mutation, Row, Rows, SemanticType}; + use common_error::ext::WhateverResult; + use datatypes::arrow::datatypes::DataType as ArrowDataType; use datatypes::data_type::ConcreteDataType; - use datatypes::extension::json::Json2ExtensionType; + use datatypes::extension::json::{ + JSON2_REMAINDER_FIELD_NAME, Json2ExtensionType, Json2PhysicalLayout, JsonMetadata, + }; use datatypes::json::value::JsonValue; + use datatypes::json::{JsonSettings, JsonTypeHint}; use datatypes::schema::ColumnSchema; use datatypes::types::json_type::{JsonNativeType, JsonObjectType}; use mito_codec::row_converter::build_primary_key_codec; @@ -1678,7 +1712,7 @@ mod tests { #[test] fn test_bulk_memtable_compact_parts_with_json2() { - let metadata = mock_metadata_with_json2(); + let metadata = mock_metadata_with_json2(JsonSettings::default()); let config = BulkMemtableConfig { merge_threshold: 2, @@ -1717,7 +1751,126 @@ mod tests { assert_eq!(4, total_rows); } - fn mock_metadata_with_json2() -> RegionMetadataRef { + #[test] + fn test_bulk_memtable_merge_bounds_json2_paths() -> WhateverResult<()> { + let metadata = mock_metadata_with_json2(JsonSettings::try_new(vec![], Some(1))?); + let first = mock_bulk_part_with_json2_values( + &metadata, + vec![1000, 2000], + vec![json!({"a": 1}), json!({"a": 2})], + 100, + )?; + let second = mock_bulk_part_with_json2_values( + &metadata, + vec![3000, 4000], + vec![json!({"b": 3}), json!({"b": 4})], + 200, + )?; + let parts = vec![ + PartToMerge::Bulk { + part: first, + file_id: FileId::random(), + }, + PartToMerge::Bulk { + part: second, + file_id: FileId::random(), + }, + ]; + + let merged = MemtableCompactor::merge_parts_group( + parts, + &metadata, + false, + MergeMode::LastRow, + usize::MAX, + usize::MAX, + DEFAULT_ROW_GROUP_SIZE, + )? + .unwrap(); + let MergedPart::Multi(part) = merged else { + unreachable!() + }; + let schema = part.schemas().next().unwrap(); + let ArrowDataType::Struct(fields) = schema.field(0).data_type() else { + unreachable!() + }; + assert_eq!( + vec![JSON2_REMAINDER_FIELD_NAME, "a"], + fields.iter().map(|x| x.name().as_str()).collect::>() + ); + Ok(()) + } + + #[test] + fn test_unordered_parts_align_json2_layouts() -> WhateverResult<()> { + let metadata = mock_metadata_with_json2(JsonSettings::try_new(vec![], Some(2))?); + let memtable = BulkMemtable::new( + 42, + BulkMemtableConfig::default(), + metadata.clone(), + None, + None, + false, + MergeMode::LastRow, + ); + memtable.write_bulk(mock_bulk_part_with_json2_values( + &metadata, + vec![1000, 2000], + vec![json!({"a": 1}), json!({"a": 2})], + 100, + )?)?; + memtable.write_bulk(mock_bulk_part_with_json2_values( + &metadata, + vec![3000, 4000], + vec![json!({"b": 3}), json!({"b": 4})], + 200, + )?)?; + + let predicate = PredicateGroup::new(&metadata, &[])?; + let ranges = memtable.ranges(None, RangesOptions::default().with_predicate(predicate))?; + let range = ranges.ranges.values().next().unwrap(); + let batch = range.build_record_batch_iter(None, None)?.next().unwrap()?; + let schema = batch.schema(); + let ArrowDataType::Struct(fields) = schema.field(0).data_type() else { + unreachable!() + }; + assert_eq!( + vec![JSON2_REMAINDER_FIELD_NAME, "a", "b"], + fields.iter().map(|x| x.name().as_str()).collect::>() + ); + Ok(()) + } + + #[test] + fn test_bulk_part_converter_uses_json2_v2_layout() -> WhateverResult<()> { + let settings = JsonSettings::try_new( + vec![JsonTypeHint { + path: vec!["id".to_string()], + data_type: ConcreteDataType::int64_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }], + Some(0), + )?; + let metadata = mock_metadata_with_json2(settings); + let part = mock_bulk_part_with_json2(&metadata, vec![1000, 2000], 100)?; + let schema = part.batch.schema(); + let field = schema.field(0); + let layout = Json2PhysicalLayout::try_from_root(field)?; + + assert!(layout.is_version_2()); + let ArrowDataType::Struct(fields) = field.data_type() else { + unreachable!() + }; + assert_eq!( + vec![JSON2_REMAINDER_FIELD_NAME, "id"], + fields.iter().map(|x| x.name().as_str()).collect::>() + ); + Ok(()) + } + + fn mock_metadata_with_json2(settings: JsonSettings) -> RegionMetadataRef { let col_meta_1 = ColumnMetadata { column_schema: ColumnSchema::new( "ts", @@ -1730,7 +1883,9 @@ mod tests { let data_type = ConcreteDataType::json2(JsonNativeType::Object(JsonObjectType::new())); let mut col_schema = ColumnSchema::new("data", data_type, true); - col_schema.with_extension_type(&Json2ExtensionType::default()); + col_schema.with_extension_type(&Json2ExtensionType::new(Arc::new(JsonMetadata::new( + settings, + )))); let col_meta_2 = ColumnMetadata { column_schema: col_schema, @@ -1749,8 +1904,29 @@ mod tests { metadata: &RegionMetadataRef, timestamps: Vec, sequence: u64, + ) -> Result { + let values = timestamps + .iter() + .map(|ts| { + json!({ + "id": ts, + "payload": { + "message": format!("row-{ts}"), + }, + }) + }) + .collect(); + mock_bulk_part_with_json2_values(metadata, timestamps, values, sequence) + } + + fn mock_bulk_part_with_json2_values( + metadata: &RegionMetadataRef, + timestamps: Vec, + values: Vec, + sequence: u64, ) -> Result { let capacity = timestamps.len(); + debug_assert_eq!(capacity, values.len()); let primary_key_codec = build_primary_key_codec(metadata); let json_type = JsonNativeType::Object(JsonObjectType::from([ ("id".to_string(), JsonNativeType::i64()), @@ -1773,16 +1949,12 @@ mod tests { let rows = timestamps .into_iter() - .map(|ts| { + .zip(values) + .map(|(ts, value)| { let val1 = api::v1::Value { value_data: Some(ValueData::TimestampMillisecondValue(ts)), }; - let value_data = ValueData::JsonValue(encode_json_value(JsonValue::from(json!({ - "id": ts, - "payload": { - "message": format!("row-{ts}"), - }, - })))); + let value_data = ValueData::JsonValue(encode_json_value(JsonValue::from(value))); let val2 = api::v1::Value { value_data: Some(value_data), }; diff --git a/src/mito2/src/memtable/bulk/json_align.rs b/src/mito2/src/memtable/bulk/json_align.rs deleted file mode 100644 index 3a5c6e7fbc..0000000000 --- a/src/mito2/src/memtable/bulk/json_align.rs +++ /dev/null @@ -1,451 +0,0 @@ -// Copyright 2023 Greptime Team -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -use std::collections::HashMap; -use std::sync::Arc; - -use datatypes::arrow::datatypes::{DataType as ArrowDataType, Schema, SchemaRef}; -use datatypes::arrow::record_batch::RecordBatch; -use datatypes::extension::json::is_json2_extension_type; -use datatypes::types::json_type::JsonNativeType; -use datatypes::vectors::json::array::JsonArray; -use snafu::{OptionExt, ResultExt}; - -use crate::error::{ - ConvertValueSnafu, DataTypeMismatchSnafu, NewRecordBatchSnafu, Result, UnexpectedSnafu, -}; -use crate::memtable::BoxedRecordBatchIterator; - -/// Aligns concrete JSON2 Arrow types across record batches. -/// -/// JSON2 column concrete Arrow types are derived from data. Different memtable -/// parts may therefore have different concrete types for the same JSON2 column. -/// This helper merges those concrete types and aligns batches to the merged schema. -#[derive(Clone)] -pub(crate) struct Json2Aligner { - /// Schema after merging all JSON2 column concrete types. - schema: SchemaRef, - /// JSON2 columns that may need per-batch alignment. - json_columns: Vec<(usize, ArrowDataType)>, -} - -impl Json2Aligner { - /// Builds an aligner from input schemas. - /// - /// Note: except for JSON2 columns, all input schemas must be identical. - pub(crate) fn try_new(input_schemas: I) -> Result - where - I: IntoIterator, - { - let mut input_schemas = input_schemas.into_iter(); - - // Use first schema as base: it defines column order and non-JSON types. - let base_schema = input_schemas.next().context(UnexpectedSnafu { - reason: "Json2Aligner requires at least one input schema", - })?; - - // Init merged types from base schema. - let mut merged_types = base_schema - .fields() - .iter() - .enumerate() - .filter(|&(_idx, field)| is_json2_extension_type(field)) - .map(|(idx, field)| { - let json_type = - JsonNativeType::try_from(field.data_type()).context(DataTypeMismatchSnafu)?; - Ok((idx, json_type)) - }) - .collect::>>()?; - - // No JSON2 columns, no alignment needed. - if merged_types.is_empty() { - return Ok(Self { - schema: base_schema, - json_columns: Vec::new(), - }); - } - - // Merge JSON2 types from remaining schemas. - for schema in input_schemas { - // Input schemas should only differ in JSON2 concrete types. - #[cfg(debug_assertions)] - assert_columns_match_except_json2(&base_schema, &schema); - - for (idx, merged) in &mut merged_types { - if *idx >= schema.fields().len() { - continue; - } - let json_type = JsonNativeType::try_from(schema.field(*idx).data_type()) - .context(DataTypeMismatchSnafu)?; - merged.merge(&json_type); - } - } - - // Build output schema with merged JSON2 types. - let mut json_columns = Vec::with_capacity(merged_types.len()); - let fields: Vec<_> = base_schema - .fields() - .iter() - .enumerate() - .map(|(idx, field)| { - if let Some(merged) = merged_types.get(&idx) { - let data_type = merged.as_arrow_type(); - json_columns.push((idx, data_type.clone())); - let mut field = (**field).clone(); - field.set_data_type(data_type); - Arc::new(field) - } else { - field.clone() - } - }) - .collect(); - - let schema = Arc::new(Schema::new_with_metadata( - fields, - base_schema.metadata().clone(), - )); - - Ok(Self { - schema, - json_columns, - }) - } - - /// Returns the aligned output schema. - pub(crate) fn schema(&self) -> &SchemaRef { - &self.schema - } - - /// Aligns a [`RecordBatch`] to [`Self::schema`]. - pub(crate) fn align_batch(&self, batch: RecordBatch) -> Result { - if self.json_columns.is_empty() { - return Ok(batch); - } - let mut cols = batch.columns().to_vec(); - for (idx, expected_type) in &self.json_columns { - if batch.schema_ref().field(*idx).data_type() != expected_type { - cols[*idx] = JsonArray::from(batch.column(*idx)) - .widen_to(expected_type) - .context(ConvertValueSnafu)?; - } - } - RecordBatch::try_new(self.schema.clone(), cols).context(NewRecordBatchSnafu) - } - - /// Aligns [`RecordBatch`]s to [`Self::schema`]. - pub(crate) fn align_batches(&self, batches: I) -> Result> - where - I: IntoIterator, - { - batches - .into_iter() - .map(|batch| self.align_batch(batch)) - .collect() - } - - /// Wraps an iterator so each yielded [`RecordBatch`] is lazily aligned. - pub(crate) fn wrap_iter(&self, iter: BoxedRecordBatchIterator) -> BoxedRecordBatchIterator { - let aligner = self.clone(); - Box::new(iter.map(move |batch| aligner.align_batch(batch?))) - } -} - -#[cfg(debug_assertions)] -fn assert_columns_match_except_json2(base_schema: &Schema, schema: &Schema) { - debug_assert_eq!( - base_schema.fields().len(), - schema.fields().len(), - "input schemas for Json2Aligner must have the same column count" - ); - for (idx, (base_field, field)) in base_schema.fields().iter().zip(schema.fields()).enumerate() { - let base_is_json2 = is_json2_extension_type(base_field); - let is_json2 = is_json2_extension_type(field); - debug_assert_eq!( - base_is_json2, is_json2, - "column {idx} must be JSON2 in all input schemas or none" - ); - if !base_is_json2 && !is_json2 { - debug_assert_eq!( - base_field, field, - "non-JSON2 column {idx} must be identical across input schemas" - ); - } - } -} - -#[cfg(test)] -mod tests { - use std::sync::Arc; - - use datatypes::arrow::array::{ - Array, ArrayRef, AsArray, Int64Array, StringViewArray, StructArray, UInt64Array, - }; - use datatypes::arrow::datatypes::{DataType, Field, Fields, Schema}; - use datatypes::extension::json::{Json2ExtensionType, JsonExtensionType}; - use serde_json::json; - - use super::*; - - #[test] - fn test_try_new_rejects_empty_input() { - let err = match Json2Aligner::try_new([]) { - Ok(_) => panic!("expected empty input to fail"), - Err(err) => err, - }; - assert!( - err.to_string() - .contains("Json2Aligner requires at least one input schema") - ); - } - - #[test] - fn test_try_new_keeps_non_json_schema_unchanged() { - let schema = Arc::new(Schema::new(vec![ - Arc::new(Field::new("ts", DataType::Int64, false)), - Arc::new(Field::new("value", DataType::UInt64, true)), - ])); - let batch = RecordBatch::try_new( - schema.clone(), - vec![ - Arc::new(Int64Array::from_iter_values([1, 2])) as ArrayRef, - Arc::new(UInt64Array::from(vec![Some(10), None])) as ArrayRef, - ], - ) - .unwrap(); - - let aligner = Json2Aligner::try_new([schema.clone()]).unwrap(); - assert!(Arc::ptr_eq(aligner.schema(), &schema)); - - let aligned = aligner.align_batch(batch).unwrap(); - assert!(Arc::ptr_eq(aligned.schema_ref(), &schema)); - } - - #[test] - fn test_try_new_ignores_legacy_jsonb_extension_field() { - let legacy_jsonb_field = Arc::new( - Field::new("data", DataType::Binary, true).with_extension_type(JsonExtensionType), - ); - let schema = Arc::new(Schema::new(vec![ - Arc::new(Field::new("ts", DataType::Int64, false)), - legacy_jsonb_field, - ])); - - let aligner = Json2Aligner::try_new([schema.clone()]).unwrap(); - - assert!(Arc::ptr_eq(aligner.schema(), &schema)); - assert!(aligner.json_columns.is_empty()); - } - - #[test] - fn test_try_new_merges_json2_object_fields() { - let id_fields = Fields::from(vec![id_field()]); - let name_fields = Fields::from(vec![name_field()]); - let schema_with_id = schema_with_json_field(json_field("data", id_fields)); - let schema_with_name = schema_with_json_field(json_field("data", name_fields)); - - let aligner = Json2Aligner::try_new([schema_with_id, schema_with_name]).unwrap(); - let data_field = aligner.schema().field(1); - let DataType::Struct(fields) = data_field.data_type() else { - panic!("expected JSON2 field to be a struct"); - }; - - assert_eq!(2, fields.len()); - assert_eq!("id", fields[0].name()); - assert_eq!(&DataType::Int64, fields[0].data_type()); - assert_eq!("name", fields[1].name()); - assert_eq!(&DataType::Utf8View, fields[1].data_type()); - assert!(is_json2_extension_type(&aligner.schema().fields()[1])); - } - - #[test] - fn test_align_batch_fills_missing_json2_fields() { - let id_fields = Fields::from(vec![id_field()]); - let name_fields = Fields::from(vec![name_field()]); - let schema_with_id = schema_with_json_field(json_field("data", id_fields.clone())); - let schema_with_name = schema_with_json_field(json_field("data", name_fields.clone())); - - let batch_with_id = RecordBatch::try_new( - schema_with_id.clone(), - vec![ - Arc::new(Int64Array::from_iter_values([1, 2])) as ArrayRef, - struct_array( - id_fields, - vec![Arc::new(Int64Array::from_iter_values([10, 20])) as ArrayRef], - ), - ], - ) - .unwrap(); - let batch_with_name = RecordBatch::try_new( - schema_with_name.clone(), - vec![ - Arc::new(Int64Array::from_iter_values([3, 4])) as ArrayRef, - struct_array( - name_fields, - vec![ - Arc::new(StringViewArray::from(vec![Some("alice"), Some("bob")])) - as ArrayRef, - ], - ), - ], - ) - .unwrap(); - - let aligner = Json2Aligner::try_new([schema_with_id, schema_with_name]).unwrap(); - let aligned_with_id = aligner.align_batch(batch_with_id).unwrap(); - let aligned_with_name = aligner.align_batch(batch_with_name).unwrap(); - - let data_with_id = aligned_with_id - .column(1) - .as_any() - .downcast_ref::() - .unwrap(); - let id_values = data_with_id - .column(0) - .as_any() - .downcast_ref::() - .unwrap(); - let missing_names = data_with_id.column(1); - assert_eq!(10, id_values.value(0)); - assert_eq!(20, id_values.value(1)); - assert!(missing_names.is_null(0)); - assert!(missing_names.is_null(1)); - - let data_with_name = aligned_with_name - .column(1) - .as_any() - .downcast_ref::() - .unwrap(); - let missing_ids = data_with_name.column(0); - let name_values = data_with_name - .column(1) - .as_any() - .downcast_ref::() - .unwrap(); - assert!(missing_ids.is_null(0)); - assert!(missing_ids.is_null(1)); - assert_eq!("alice", name_values.value(0)); - assert_eq!("bob", name_values.value(1)); - } - - #[test] - fn test_align_conflicting_number_types_as_variant() { - let u64_fields = Fields::from(vec![Arc::new(Field::new("value", DataType::UInt64, true))]); - let i64_fields = Fields::from(vec![Arc::new(Field::new("value", DataType::Int64, true))]); - let u64_schema = schema_with_json_field(json_field("data", u64_fields.clone())); - let i64_schema = schema_with_json_field(json_field("data", i64_fields.clone())); - let u64_batch = RecordBatch::try_new( - u64_schema.clone(), - vec![ - Arc::new(Int64Array::from_iter_values([1])) as ArrayRef, - struct_array( - u64_fields, - vec![Arc::new(UInt64Array::from_iter_values([u64::MAX])) as ArrayRef], - ), - ], - ) - .unwrap(); - let i64_batch = RecordBatch::try_new( - i64_schema.clone(), - vec![ - Arc::new(Int64Array::from_iter_values([2])) as ArrayRef, - struct_array( - i64_fields, - vec![Arc::new(Int64Array::from_iter_values([i64::MIN])) as ArrayRef], - ), - ], - ) - .unwrap(); - - let aligner = Json2Aligner::try_new([u64_schema, i64_schema]).unwrap(); - let DataType::Struct(fields) = aligner.schema().field(1).data_type() else { - panic!("expected JSON2 field to be a struct"); - }; - assert_eq!(&DataType::Binary, fields[0].data_type()); - - for (batch, expected) in [(u64_batch, json!(u64::MAX)), (i64_batch, json!(i64::MIN))] { - let aligned = aligner.align_batch(batch).unwrap(); - let data = aligned.column(1).as_struct(); - assert_eq!( - expected, - JsonArray::from(data.column(0)).try_get_value(0).unwrap() - ); - } - } - - #[test] - fn test_wrap_iter_aligns_each_batch() { - let id_fields = Fields::from(vec![id_field()]); - let name_fields = Fields::from(vec![name_field()]); - let schema_with_id = schema_with_json_field(json_field("data", id_fields.clone())); - let schema_with_name = schema_with_json_field(json_field("data", name_fields.clone())); - - let batch_with_id = RecordBatch::try_new( - schema_with_id.clone(), - vec![ - Arc::new(Int64Array::from_iter_values([1])) as ArrayRef, - struct_array( - id_fields, - vec![Arc::new(Int64Array::from_iter_values([10])) as ArrayRef], - ), - ], - ) - .unwrap(); - let batch_with_name = RecordBatch::try_new( - schema_with_name.clone(), - vec![ - Arc::new(Int64Array::from_iter_values([2])) as ArrayRef, - struct_array( - name_fields, - vec![Arc::new(StringViewArray::from(vec![Some("alice")])) as ArrayRef], - ), - ], - ) - .unwrap(); - - let aligner = Json2Aligner::try_new([schema_with_id, schema_with_name]).unwrap(); - let iter: BoxedRecordBatchIterator = - Box::new(vec![Ok(batch_with_id), Ok(batch_with_name)].into_iter()); - let aligned = aligner.wrap_iter(iter).collect::>>().unwrap(); - - assert_eq!(2, aligned.len()); - assert!(Arc::ptr_eq(aligned[0].schema_ref(), aligner.schema())); - assert!(Arc::ptr_eq(aligned[1].schema_ref(), aligner.schema())); - } - - fn json_field(name: &str, fields: Fields) -> Arc { - Arc::new( - Field::new(name, DataType::Struct(fields), true) - .with_extension_type(Json2ExtensionType::default()), - ) - } - - fn schema_with_json_field(json_field: Arc) -> SchemaRef { - Arc::new(Schema::new(vec![ - Arc::new(Field::new("ts", DataType::Int64, false)), - json_field, - ])) - } - - fn id_field() -> Arc { - Arc::new(Field::new("id", DataType::Int64, true)) - } - - fn name_field() -> Arc { - Arc::new(Field::new("name", DataType::Utf8View, true)) - } - - fn struct_array(fields: Fields, columns: Vec) -> ArrayRef { - Arc::new(StructArray::new(fields, columns, None)) - } -} diff --git a/src/mito2/src/memtable/bulk/part.rs b/src/mito2/src/memtable/bulk/part.rs index 2048279407..7bcf667cef 100644 --- a/src/mito2/src/memtable/bulk/part.rs +++ b/src/mito2/src/memtable/bulk/part.rs @@ -54,13 +54,13 @@ use store_api::metadata::{RegionMetadata, RegionMetadataRef}; use store_api::storage::consts::PRIMARY_KEY_COLUMN_NAME; use store_api::storage::{ColumnId, FileId, SequenceNumber, SequenceRange}; +use crate::compaction::{collect_json2_rewrite_plans, rewrite_json2_batch, rewrite_json2_schema}; use crate::error::{ self, ColumnNotFoundSnafu, ComputeArrowSnafu, CreateDefaultSnafu, DataTypeMismatchSnafu, EncodeMemtableSnafu, EncodeSnafu, InvalidMetadataSnafu, InvalidRequestSnafu, NewRecordBatchSnafu, Result, }; use crate::memtable::bulk::context::{BulkIterContext, BulkIterContextRef}; -use crate::memtable::bulk::json_align::Json2Aligner; use crate::memtable::bulk::part_reader::EncodedBulkPartIter; use crate::memtable::time_series::{ValueBuilder, Values}; use crate::memtable::{BoxedRecordBatchIterator, MemScanMetrics, MemtableStats}; @@ -440,7 +440,7 @@ impl UnorderedPart { /// Concatenates and sorts all parts into a single RecordBatch. /// Returns None if the collection is empty. - pub fn concat_and_sort(&self) -> Result> { + pub fn concat_and_sort(&self, metadata: &RegionMetadataRef) -> Result> { if self.parts.is_empty() { return Ok(None); } @@ -450,17 +450,28 @@ impl UnorderedPart { return Ok(Some(self.parts[0].batch.clone())); } - // Get the schema from the first part - let schema = self.parts[0].batch.schema(); - let concatenated = if schema.fields().iter().any(is_json2_extension_type) { - let aligner = Json2Aligner::try_new(self.parts.iter().map(|part| part.batch.schema()))?; - let aligned_batches = - aligner.align_batches(self.parts.iter().map(|part| part.batch.clone()))?; - concat_batches(aligner.schema(), &aligned_batches).context(ComputeArrowSnafu)? - } else { - concat_batches(&schema, self.parts.iter().map(|x| &x.batch)) - .context(ComputeArrowSnafu)? - }; + let schemas = self + .parts + .iter() + .map(|x| (x.batch.schema(), x.num_rows() as u64)) + .collect::>(); + let plans = collect_json2_rewrite_plans(metadata, &schemas)?; + + debug_assert!(self.parts.windows(2).all(|w| rewrite_json2_schema( + &w[0].batch.schema(), + &plans + ) == rewrite_json2_schema( + &w[1].batch.schema(), + &plans + ))); + let schema = rewrite_json2_schema(&self.parts[0].batch.schema(), &plans); + + let batches = self + .parts + .iter() + .map(|x| rewrite_json2_batch(x.batch.clone(), &plans)) + .collect::>>()?; + let concatenated = concat_batches(&schema, &batches).context(ComputeArrowSnafu)?; // Sort the concatenated batch let sorted_batch = sort_primary_key_record_batch(&concatenated)?; @@ -470,8 +481,8 @@ impl UnorderedPart { /// Converts all parts into a single sorted BulkPart. /// Returns None if the collection is empty. - pub fn to_bulk_part(&self) -> Result> { - let Some(sorted_batch) = self.concat_and_sort()? else { + pub fn to_bulk_part(&self, metadata: &RegionMetadataRef) -> Result> { + let Some(sorted_batch) = self.concat_and_sort(metadata)? else { return Ok(None); }; diff --git a/src/mito2/src/memtable/time_series.rs b/src/mito2/src/memtable/time_series.rs index 7e89d569ee..fa9970137c 100644 --- a/src/mito2/src/memtable/time_series.rs +++ b/src/mito2/src/memtable/time_series.rs @@ -29,6 +29,7 @@ use datatypes::arrow::array::ArrayRef; use datatypes::arrow_array::StringArray; use datatypes::data_type::{ConcreteDataType, DataType}; use datatypes::prelude::{ScalarVector, Vector, VectorRef}; +use datatypes::schema::ColumnSchema; use datatypes::types::TimestampType; use datatypes::value::{Value, ValueRef}; use datatypes::vectors::{ @@ -46,7 +47,7 @@ use crate::error::{ self, ComputeArrowSnafu, ConvertVectorSnafu, EncodeSnafu, PrimaryKeyLengthMismatchSnafu, Result, }; use crate::flush::WriteBufferManagerRef; -use crate::memtable::builder::{FieldBuilder, StringBuilder}; +use crate::memtable::builder::FieldBuilder; use crate::memtable::bulk::part::BulkPart; use crate::memtable::simple_bulk_memtable::SimpleBulkMemtable; use crate::memtable::stats::WriteMetrics; @@ -918,7 +919,7 @@ pub(crate) struct ValueBuilder { sequence: Vec, op_type: Vec, fields: Vec>, - field_types: Vec, + field_schemas: Vec, } impl ValueBuilder { @@ -931,18 +932,18 @@ impl ValueBuilder { let sequence = Vec::with_capacity(capacity); let op_type = Vec::with_capacity(capacity); - let field_types = region_metadata + let field_schemas = region_metadata .field_columns() - .map(|c| c.column_schema.data_type.clone()) + .map(|c| c.column_schema.clone()) .collect::>(); - let fields = (0..field_types.len()).map(|_| None).collect(); + let fields = (0..field_schemas.len()).map(|_| None).collect(); Self { timestamp: Vec::with_capacity(capacity), timestamp_type, sequence, op_type, fields, - field_types, + field_schemas, } } @@ -984,15 +985,10 @@ impl ValueBuilder { .push(field_value) .unwrap_or_else(|e| panic!("Failed to push field value: {e:?}")); } else { - let mut mutable_vector = - if let ConcreteDataType::String(_) = &self.field_types[idx] { - FieldBuilder::String(StringBuilder::with_capacity(4, 8)) - } else { - FieldBuilder::Other( - self.field_types[idx] - .create_mutable_vector(num_rows.max(INITIAL_BUILDER_CAPACITY)), - ) - }; + let mut mutable_vector = FieldBuilder::create( + &self.field_schemas[idx], + num_rows.max(INITIAL_BUILDER_CAPACITY), + ); mutable_vector.push_nulls(num_rows - 1); mutable_vector .push(field_value) @@ -1017,7 +1013,12 @@ impl ValueBuilder { /// the Arrow string array offset limit and thus can never be accommodated, not even by an /// empty builder. pub(crate) fn can_accommodate(&self, fields: &[VectorRef]) -> Result { - scan_string_capacity(fields, &self.fields, &self.field_types, i32::MAX) + let data_types = self + .field_schemas + .iter() + .map(|x| x.data_type.clone()) + .collect::>(); + scan_string_capacity(fields, &self.fields, &data_types, i32::MAX) } pub(crate) fn extend( @@ -1082,7 +1083,7 @@ impl ValueBuilder { { let builder = field_dest.get_or_insert_with(|| { let mut field_builder = - FieldBuilder::create(&self.field_types[field_idx], INITIAL_BUILDER_CAPACITY); + FieldBuilder::create(&self.field_schemas[field_idx], INITIAL_BUILDER_CAPACITY); field_builder.push_nulls(num_rows_before); field_builder }); @@ -1140,9 +1141,9 @@ impl ValueBuilder { MEMTABLE_ACTIVE_FIELD_BUILDER_COUNT.dec(); v.finish_cloned() } else { - let mut single_null = self.field_types[i].create_mutable_vector(num_rows); - single_null.push_nulls(num_rows); - single_null.to_vector() + let mut builder = FieldBuilder::create(&self.field_schemas[i], num_rows); + builder.push_nulls(num_rows); + builder.finish() } }) .collect::>(); @@ -1282,9 +1283,9 @@ impl From for Values { MEMTABLE_ACTIVE_FIELD_BUILDER_COUNT.dec(); v.finish() } else { - let mut single_null = value.field_types[i].create_mutable_vector(num_rows); - single_null.push_nulls(num_rows); - single_null.to_vector() + let mut builder = FieldBuilder::create(&value.field_schemas[i], num_rows); + builder.push_nulls(num_rows); + builder.finish() } }) .collect::>(); @@ -1390,6 +1391,7 @@ mod tests { use store_api::storage::RegionId; use super::*; + use crate::memtable::builder::StringBuilder; use crate::test_util::column_metadata_to_column_schema; fn schema_for_test() -> RegionMetadataRef { diff --git a/src/mito2/src/read/compat.rs b/src/mito2/src/read/compat.rs index 3fa625694c..b0c4d52f04 100644 --- a/src/mito2/src/read/compat.rs +++ b/src/mito2/src/read/compat.rs @@ -95,16 +95,11 @@ impl FlatCompatBatch { compaction: bool, ) -> Result> { let actual = read_format.metadata(); - let format_projection = read_format.format_projection(); - let mut actual_schema = flat_projected_columns(actual, format_projection); - for (column_id, target_type) in read_format.json_target_types().iter() { - if let Some(i) = actual_schema - .iter() - .position(|(actual_column_id, _)| actual_column_id == column_id) - { - actual_schema[i].1 = ConcreteDataType::json2(target_type.clone()); - } - } + let actual_schema = flat_projected_columns( + actual, + read_format.format_projection(), + read_format.json_target_types(), + ); let expect_schema = mapper.batch_schema(); if expect_schema == actual_schema @@ -176,7 +171,19 @@ impl FlatCompatBatch { // Same column different type. if expect_data_type != *actual_data_type { - cast_type = Some(expect_data_type.clone()) + ensure!( + !expect_data_type.is_json2() && !actual_data_type.is_json2(), + CompatReaderSnafu { + region_id: expect_metadata.region_id, + reason: format!( + "JSON2 column '{}' must be aligned before FlatCompatBatch, actual: {}, expected: {}", + expect_column.column_schema.name, + actual_data_type, + expect_data_type, + ), + } + ); + cast_type = Some(expect_data_type.clone()); } // Source has this column. index_or_defaults.push(IndexOrDefault::Index { @@ -634,6 +641,7 @@ impl FlatCompatPrimaryKey { #[cfg(test)] mod tests { + use std::collections::BTreeMap; use std::sync::Arc; use api::v1::{OpType, SemanticType}; @@ -645,6 +653,7 @@ mod tests { use datatypes::arrow::record_batch::RecordBatch; use datatypes::prelude::ConcreteDataType; use datatypes::schema::ColumnSchema; + use datatypes::types::json_type::JsonNativeType; use datatypes::value::ValueRef; use mito_codec::row_converter::{ DensePrimaryKeyCodec, PrimaryKeyCodecExt, SparsePrimaryKeyCodec, @@ -814,6 +823,57 @@ mod tests { assert_eq!(expected_batch, result); } + #[test] + fn test_flat_compat_batch_uses_projected_json2_type() -> Result<()> { + let json2 = ConcreteDataType::json2(JsonNativeType::object()); + let actual_metadata = Arc::new(new_metadata( + &[ + ( + 0, + SemanticType::Timestamp, + ConcreteDataType::timestamp_millisecond_datatype(), + ), + (1, SemanticType::Field, json2.clone()), + ], + &[], + )); + let expected_metadata = Arc::new(new_metadata( + &[ + ( + 0, + SemanticType::Timestamp, + ConcreteDataType::timestamp_millisecond_datatype(), + ), + (1, SemanticType::Field, json2), + (2, SemanticType::Field, ConcreteDataType::int64_datatype()), + ], + &[], + )); + let read_columns = ReadColumns::new([0, 1, 2]) + .with_json_target_types(BTreeMap::from([(1, JsonNativeType::Variant)])); + let mapper = FlatProjectionMapper::new_with_read_columns( + &expected_metadata, + vec![0, 1, 2], + read_columns.clone(), + )?; + let read_format = FlatReadFormat::new(actual_metadata, read_columns, None, "test", false)?; + + let compat = FlatCompatBatch::try_new(&mapper, &read_format, false)?.unwrap(); + let json_index = mapper + .batch_schema() + .iter() + .position(|(id, _)| *id == 1) + .unwrap(); + assert!(matches!( + &compat.index_or_defaults[json_index], + IndexOrDefault::Index { + cast_type: None, + .. + } + )); + Ok(()) + } + #[test] fn test_flat_compat_batch_with_read_projection_superset() { let actual_metadata = Arc::new(new_metadata( diff --git a/src/mito2/src/read/flat_projection.rs b/src/mito2/src/read/flat_projection.rs index e631cf45db..19dc962c56 100644 --- a/src/mito2/src/read/flat_projection.rs +++ b/src/mito2/src/read/flat_projection.rs @@ -27,11 +27,10 @@ use datatypes::arrow::datatypes::{DataType as ArrowDataType, Field}; use datatypes::extension::json::is_json2_extension_type; use datatypes::prelude::{ConcreteDataType, DataType}; use datatypes::schema::{Schema, SchemaRef}; -use datatypes::types::JsonType; -use datatypes::types::json_type::JsonNativeType; use datatypes::value::Value; use datatypes::vectors::Helper; use datatypes::vectors::json::array::JsonArray; +use datatypes::vectors::json::json2_physical_data_type; use snafu::{OptionExt, ResultExt}; use store_api::metadata::{RegionMetadata, RegionMetadataRef}; use store_api::storage::ColumnId; @@ -39,7 +38,8 @@ use store_api::storage::ColumnId; use crate::cache::CacheStrategy; use crate::error::{InvalidRequestSnafu, RecordBatchSnafu, Result}; use crate::read::projection::{read_column_ids_from_projection, repeated_vector_with_cache}; -use crate::read::read_columns::ReadColumns; +use crate::read::read_columns::{JsonTargetTypes, ReadColumns}; +use crate::sst::parquet::Json2RewriteTargets; use crate::sst::parquet::flat_format::sst_column_id_indices; use crate::sst::parquet::format::FormatProjection; use crate::sst::{ @@ -94,6 +94,21 @@ impl FlatProjectionMapper { metadata: &RegionMetadataRef, projection: Vec, read_cols: ReadColumns, + ) -> Result { + Self::new_with_json2_rewrite_targets( + metadata, + projection, + read_cols, + &Json2RewriteTargets::default(), + ) + } + + /// Returns a mapper for a compaction read with fixed JSON2 output layouts. + pub(crate) fn new_with_json2_rewrite_targets( + metadata: &RegionMetadataRef, + projection: Vec, + read_cols: ReadColumns, + json2_rewrite_targets: &Json2RewriteTargets, ) -> Result { // If the original projection is empty. let is_empty_projection = projection.is_empty(); @@ -113,10 +128,8 @@ impl FlatProjectionMapper { output_col_ids.push(col.column_id); let mut schema = col.column_schema.clone(); - if let Some(data_type) = - json2_read_datatype(col.column_id, &schema.data_type, &read_cols) - { - schema.data_type = data_type; + if let Some(data_type) = read_cols.json_target_type(col.column_id) { + schema.data_type = ConcreteDataType::json2(data_type.clone()); } col_schemas.push(schema); } @@ -133,16 +146,11 @@ impl FlatProjectionMapper { read_cols.clone(), ); - let mut batch_schema = flat_projected_columns(metadata, &format_projection); + let batch_schema = + flat_projected_columns(metadata, &format_projection, read_cols.json_target_types()); - for (column_id, data_type) in batch_schema.iter_mut() { - if let Some(updated) = json2_read_datatype(*column_id, data_type, &read_cols) { - *data_type = updated; - } - } - - // Safety: We get the column id from the metadata. - let input_arrow_schema = compute_input_arrow_schema(metadata, &batch_schema); + let input_arrow_schema = + compute_input_arrow_schema(metadata, &batch_schema, &read_cols, json2_rewrite_targets); // If projection is empty, we don't output any column. let output_schema = if is_empty_projection { @@ -348,9 +356,9 @@ impl FlatProjectionMapper { } let field = &self.output_schema.arrow_schema().fields()[output_idx]; - if is_json2_extension_type(field) { + if is_json2_extension_type(field) && array.data_type() != field.data_type() { array = JsonArray::from(&array) - .project_to(field.data_type()) + .project_to_v2(batch.schema_ref().field(*index), field.data_type()) .context(DataTypesSnafu)?; } @@ -391,35 +399,6 @@ impl FlatProjectionMapper { } } -fn json2_read_datatype( - column_id: ColumnId, - data_type: &ConcreteDataType, - read_cols: &ReadColumns, -) -> Option { - let json_type = data_type.as_json()?; - if !json_type.is_json2() { - return None; - } - - if let Some(concretized) = read_cols.json_target_type(column_id).cloned() { - return Some(ConcreteDataType::json2(concretized)); - } - - if is_empty_json2_type(json_type) { - return Some(ConcreteDataType::json2(JsonNativeType::Variant)); - } - - None -} - -fn is_empty_json2_type(json_type: &JsonType) -> bool { - match json_type.native_type() { - JsonNativeType::Null => true, - JsonNativeType::Object(fields) if fields.is_empty() => true, - _ => false, - } -} - fn single_value_string_dictionary<'a>( array: &'a Arc, output_type: &ConcreteDataType, @@ -442,12 +421,13 @@ fn single_value_string_dictionary<'a>( (dict_array.values().len() == 1 && dict_array.null_count() == 0).then_some(dict_array) } -/// Returns ids and datatypes of columns of the output batch after applying the `projection`. +/// Returns ids and datatypes of columns after applying the projection and JSON2 target types. /// /// It adds the time index column if it doesn't present in the projection. pub(crate) fn flat_projected_columns( metadata: &RegionMetadata, format_projection: &FormatProjection, + json_target_types: &JsonTargetTypes, ) -> Vec<(ColumnId, ConcreteDataType)> { let time_index = metadata.time_index_column(); let num_columns = if format_projection @@ -460,16 +440,18 @@ pub(crate) fn flat_projected_columns( }; let mut schema = vec![None; num_columns]; for (column_id, index) in &format_projection.column_id_to_projected_index { - // Safety: FormatProjection ensures the id is valid. - schema[*index] = Some(( - *column_id, + let data_type = if let Some(json_type) = json_target_types.get(column_id) { + ConcreteDataType::json2(json_type.clone()) + } else { + // Safety: FormatProjection ensures the id is valid. metadata .column_by_id(*column_id) .unwrap() .column_schema .data_type - .clone(), - )); + .clone() + }; + schema[*index] = Some((*column_id, data_type)); } if num_columns != format_projection.column_id_to_projected_index.len() { schema[num_columns - 1] = Some(( @@ -489,13 +471,25 @@ pub(crate) fn flat_projected_columns( pub(crate) fn compute_input_arrow_schema( metadata: &RegionMetadata, batch_schema: &[(ColumnId, ConcreteDataType)], + read_cols: &ReadColumns, + json2_rewrite_targets: &Json2RewriteTargets, ) -> datatypes::arrow::datatypes::SchemaRef { let mut new_fields = Vec::with_capacity(batch_schema.len() + 3); for (column_id, data_type) in batch_schema { + let data_type = json2_rewrite_targets + .get(column_id) + .map(|x| json2_physical_data_type(&x.target_layout)) + .or_else(|| { + read_cols + .json_target_type(*column_id) + .map(|x| x.as_arrow_type()) + }) + .unwrap_or_else(|| data_type.as_arrow_type()); + let column_metadata = metadata.column_by_id(*column_id).unwrap(); let field = Field::new( &column_metadata.column_schema.name, - data_type.as_arrow_type(), + data_type, column_metadata.column_schema.is_nullable(), ) .with_metadata(column_metadata.column_schema.metadata().clone()); diff --git a/src/mito2/src/read/read_columns.rs b/src/mito2/src/read/read_columns.rs index 42a9f71ddc..89e99fa3ef 100644 --- a/src/mito2/src/read/read_columns.rs +++ b/src/mito2/src/read/read_columns.rs @@ -13,6 +13,7 @@ // limitations under the License. use std::collections::BTreeMap; +use std::hash::Hash; use std::mem; use std::sync::Arc; @@ -46,6 +47,7 @@ impl ReadColumns { } } + /// Attaches query-time JSON2 projection types. pub fn with_json_target_types( mut self, json_target_types: BTreeMap, @@ -66,10 +68,11 @@ impl ReadColumns { self.column_ids_iter().collect() } - pub fn json_target_types(&self) -> &JsonTargetTypes { + pub(crate) fn json_target_types(&self) -> &JsonTargetTypes { &self.json_target_types } + /// Returns the query-time JSON2 projection type for a column. pub fn json_target_type(&self, column_id: ColumnId) -> Option<&JsonNativeType> { self.json_target_types.get(&column_id) } @@ -77,7 +80,6 @@ impl ReadColumns { pub fn estimated_size(&self) -> usize { self.col_ids.capacity() * mem::size_of::() + self.col_ids.len() * mem::size_of::() - + self.json_target_types.len() - * (mem::size_of::() + mem::size_of::()) + + self.json_target_types.len() * (size_of::() + size_of::()) } } diff --git a/src/mito2/src/read/scan_region.rs b/src/mito2/src/read/scan_region.rs index da7d6faeba..375e0059de 100644 --- a/src/mito2/src/read/scan_region.rs +++ b/src/mito2/src/read/scan_region.rs @@ -82,6 +82,7 @@ use crate::sst::index::inverted_index::applier::InvertedIndexApplierRef; use crate::sst::index::inverted_index::applier::builder::InvertedIndexApplierBuilder; #[cfg(feature = "vector_index")] use crate::sst::index::vector_index::applier::{VectorIndexApplier, VectorIndexApplierRef}; +use crate::sst::parquet::Json2RewriteTargets; use crate::sst::parquet::file_range::PreFilterMode; use crate::sst::parquet::reader::ReaderMetrics; @@ -927,6 +928,8 @@ pub struct ScanInput { pub(crate) snapshot_sequence: Option, /// Whether this scan is for compaction. pub(crate) compaction: bool, + /// Compaction-only JSON2 physical rewrite targets. + json2_rewrite_targets: Json2RewriteTargets, /// Counters that should receive query-load metrics. pub(crate) query_stat_counters: Option, #[cfg(feature = "enterprise")] @@ -966,6 +969,7 @@ impl ScanInput { explain_flat_format: false, snapshot_sequence: None, compaction: false, + json2_rewrite_targets: Arc::default(), query_stat_counters: None, #[cfg(feature = "enterprise")] extension_ranges: Vec::new(), @@ -1164,6 +1168,13 @@ impl ScanInput { self } + /// Sets compaction-only JSON2 physical rewrite targets. + #[must_use] + pub(crate) fn with_json2_rewrite_targets(mut self, targets: Json2RewriteTargets) -> Self { + self.json2_rewrite_targets = targets; + self + } + /// Builds memtable ranges to scan by `index`. pub(crate) fn build_mem_ranges(&self, index: RowGroupIndex) -> SmallVec<[MemtableRange; 2]> { let memtable = &self.memtables[index.index]; @@ -1291,6 +1302,7 @@ impl ScanInput { .read_sst(file.clone()) .predicate(predicate) .projection(Some(self.read_cols.clone())) + .json2_rewrite_targets(self.json2_rewrite_targets.clone()) .cache(self.cache_strategy.clone()) .inverted_index_appliers(self.inverted_index_appliers.clone()) .bloom_filter_index_appliers(self.bloom_filter_index_appliers.clone()) diff --git a/src/mito2/src/region.rs b/src/mito2/src/region.rs index c5a9021bde..76204da9cf 100644 --- a/src/mito2/src/region.rs +++ b/src/mito2/src/region.rs @@ -483,6 +483,7 @@ impl MitoRegion { let mut manager: RwLockWriteGuard<'_, RegionManifestManager> = self.manifest_ctx.manifest_manager.write().await; let current_state = self.state(); + let mut wait_for_checkpoint = false; let hook_payload: Option = match state { SettableRegionRoleState::Leader => { @@ -545,10 +546,12 @@ impl MitoRegion { ); self.exit_staging()?; self.set_role(RegionRole::Follower); + wait_for_checkpoint = true; } RegionRoleState::Leader(_) => { info!("Demoting region {} from leader to follower", self.region_id); self.set_role(RegionRole::Follower); + wait_for_checkpoint = true; } RegionRoleState::Follower => { // Already in desired state - no-op @@ -568,14 +571,17 @@ impl MitoRegion { ); self.exit_staging()?; self.set_role(RegionRole::DowngradingLeader); + wait_for_checkpoint = true; } RegionRoleState::Leader(RegionLeaderState::Writable) => { info!("Starting downgrade for region {}", self.region_id); self.set_role(RegionRole::DowngradingLeader); + wait_for_checkpoint = true; } RegionRoleState::Leader(RegionLeaderState::Downgrading) => { // Already in desired state - no-op info!("Region {} already in downgrading mode", self.region_id); + wait_for_checkpoint = true; } _ => { warn!( @@ -588,6 +594,13 @@ impl MitoRegion { } }; + // The state is changed before waiting, so no new writable-leader work + // can race with the barrier. Keep the manager lock while joining to + // serialize the barrier with checkpoint scheduling. + if wait_for_checkpoint { + manager.wait_for_pending_checkpoint().await; + } + // Hack(zhongzc): If we have just become leader (writable), persist any backfilled metadata. let mut backfill_hook_payload: Option = None; if self.state() == RegionRoleState::Leader(RegionLeaderState::Writable) { @@ -1371,10 +1384,14 @@ impl ManifestContext { // Clone before `action_list` is moved into `update` so the hook still // sees what was written. let action_list_for_hook = self.hook.as_ref().map(|_| action_list.clone()); - let version = manager - .update(action_list, is_staging) - .await - .inspect_err(|e| error!(e; "Failed to update manifest, region_id: {}", region_id))?; + let version = if !is_staging + && self.state.load() == RegionRoleState::Leader(RegionLeaderState::Downgrading) + { + manager.update_normal_without_checkpoint(action_list).await + } else { + manager.update(action_list, is_staging).await + } + .inspect_err(|e| error!(e; "Failed to update manifest, region_id: {}", region_id))?; Ok(PendingManifestHook::new( region_id, diff --git a/src/mito2/src/region/opener.rs b/src/mito2/src/region/opener.rs index 5d2d36df10..f4e83272b7 100644 --- a/src/mito2/src/region/opener.rs +++ b/src/mito2/src/region/opener.rs @@ -20,8 +20,10 @@ use std::sync::atomic::{AtomicI64, AtomicU64}; use std::sync::{Arc, LazyLock}; use std::time::Instant; +use arrow_schema::extension::ExtensionType; use common_telemetry::{debug, error, info, warn}; use common_wal::options::WalOptions; +use datatypes::extension::json::{Json2ExtensionType, JsonMetadata}; use futures::StreamExt; use futures::future::BoxFuture; use log_store::kafka::log_store::KafkaLogStore; @@ -50,14 +52,14 @@ use crate::config::MitoConfig; use crate::engine::region_hook::RegionHookRef; use crate::error; use crate::error::{ - EmptyRegionDirSnafu, InvalidMetadataSnafu, InvalidRegionOptionsSnafu, ObjectStoreNotFoundSnafu, - RegionCorruptedSnafu, Result, StaleLogEntrySnafu, + DataTypeMismatchSnafu, EmptyRegionDirSnafu, InvalidMetadataSnafu, InvalidRegionOptionsSnafu, + ObjectStoreNotFoundSnafu, RegionCorruptedSnafu, Result, StaleLogEntrySnafu, }; use crate::manifest::action::RegionManifest; use crate::manifest::manager::{RegionManifestManager, RegionManifestOptions}; -use crate::memtable::MemtableBuilderProvider; use crate::memtable::bulk::part::BulkPart; use crate::memtable::time_partition::{TimePartitions, TimePartitionsRef}; +use crate::memtable::{MemtableBuilderProvider, ensure_json2_not_use_time_series_memtable}; use crate::metrics::{CACHE_FILL_DOWNLOADED_FILES, CACHE_FILL_PENDING_FILES}; use crate::region::options::RegionOptions; use crate::region::version::{VersionBuilder, VersionControl, VersionControlRef}; @@ -93,6 +95,41 @@ fn initial_pruned_entry_id(wal_options: &WalOptions) -> EntryId { } } +fn maybe_upgrade_json2_layout(metadata: RegionMetadataRef) -> Result { + let mut upgrades = Vec::new(); + for (index, column) in metadata.column_metadatas.iter().enumerate() { + if !column.column_schema.data_type.is_json2() { + continue; + } + let Some(extension) = column + .column_schema + .extension_type::() + .context(DataTypeMismatchSnafu)? + else { + continue; + }; + if extension.metadata().is_version_2() { + continue; + } + upgrades.push((index, extension.metadata().json_settings().clone())); + } + + if upgrades.is_empty() { + return Ok(metadata); + } + + let mut upgraded = metadata.as_ref().clone(); + for (index, settings) in upgrades { + let extension = Json2ExtensionType::new(Arc::new(JsonMetadata::new(settings))); + upgraded.column_metadatas[index] + .column_schema + .with_extension_type(&extension); + } + + let builder = RegionMetadataBuilder::from_existing(upgraded); + Ok(Arc::new(builder.build().context(InvalidMetadataSnafu)?)) +} + /// A fetcher to retrieve partition expr for a region. /// /// Compatibility: older regions didn't persist `partition_expr` in engine metadata, @@ -324,6 +361,7 @@ impl RegionOpener { options.sst_format = Some(FormatType::PrimaryKey); FormatType::PrimaryKey }; + ensure_json2_not_use_time_series_memtable(&metadata, &options)?; // Create a manifest manager for this region and writes regions to the manifest file. let mut region_manifest_options = RegionManifestOptions::new(config, ®ion_dir, &object_store); @@ -475,8 +513,10 @@ impl RegionOpener { } else { manifest.metadata.clone() }; + let metadata = maybe_upgrade_json2_layout(metadata)?; // Updates the region options with the manifest. sanitize_region_options(&manifest, &mut region_options); + ensure_json2_not_use_time_series_memtable(&metadata, ®ion_options)?; let region_id = self.region_id; let provider = self.provider::(®ion_options.wal_options)?; @@ -1316,23 +1356,31 @@ mod tests { use std::collections::HashMap; use std::sync::Arc; + use arrow_schema::extension::ExtensionType; use common_base::readable_size::ReadableSize; + use common_error::ext::WhateverResult; use common_test_util::temp_dir::create_temp_dir; use common_time::Timestamp; use common_wal::options::{KafkaWalOptions, WalOptions}; use datatypes::arrow::array::{ArrayRef, BinaryArray, Int64Array}; use datatypes::arrow::record_batch::RecordBatch; + use datatypes::extension::json::{Json2ExtensionType, JsonMetadata}; + use datatypes::json::JsonSettings; + use datatypes::prelude::ConcreteDataType; + use datatypes::schema::ColumnSchema; + use datatypes::types::json_type::{JsonNativeType, JsonObjectType}; use object_store::ObjectStore; use object_store::services::{Fs, Memory, S3}; use parquet::arrow::ArrowWriter; use parquet::file::metadata::{KeyValue, PageIndexPolicy}; use parquet::file::properties::WriterProperties; + use store_api::metadata::RegionMetadataBuilder; use store_api::region_request::PathType; use store_api::storage::{FileId, RegionId}; use super::{ - initial_pruned_entry_id, preload_parquet_meta_cache_for_files, sanitize_region_options, - supports_open_region_object_storage_requirement, + initial_pruned_entry_id, maybe_upgrade_json2_layout, preload_parquet_meta_cache_for_files, + sanitize_region_options, supports_open_region_object_storage_requirement, }; use crate::cache::CacheManager; use crate::cache::file_cache::{FileType, IndexKey}; @@ -1386,6 +1434,38 @@ mod tests { ); } + #[test] + fn test_upgrade_json2_layout() -> WhateverResult<()> { + let settings = JsonSettings::try_new(vec![], Some(3))?; + let extension = Json2ExtensionType::new(Arc::new(JsonMetadata::new_v1(settings.clone()))); + let mut column = ColumnSchema::new( + "field_0", + ConcreteDataType::json2(JsonNativeType::Object(JsonObjectType::new())), + true, + ); + column.with_extension_type(&extension); + + let mut metadata = sst_region_metadata(); + metadata.column_metadatas[2].column_schema = column; + let builder = RegionMetadataBuilder::from_existing(metadata); + let metadata = Arc::new(builder.build()?); + + let upgraded = maybe_upgrade_json2_layout(metadata)?; + let column = &upgraded.column_metadatas[2].column_schema; + let extension = column.extension_type::()?.unwrap(); + assert!(extension.metadata().is_version_2()); + assert_eq!(&settings, extension.metadata().json_settings()); + + let arrow_schema = upgraded.schema.arrow_schema(); + let field = arrow_schema.field_with_name("field_0").unwrap(); + let extension = field.try_extension_type::().unwrap(); + assert!(extension.metadata().is_version_2()); + + let unchanged = maybe_upgrade_json2_layout(upgraded.clone())?; + assert!(Arc::ptr_eq(&upgraded, &unchanged)); + Ok(()) + } + #[test] #[cfg(not(feature = "test-shared-fs-region-migration"))] fn test_open_requirement_rejects_fs_object_store() { diff --git a/src/mito2/src/sst.rs b/src/mito2/src/sst.rs index a1f6e3e7f8..223d936137 100644 --- a/src/mito2/src/sst.rs +++ b/src/mito2/src/sst.rs @@ -505,13 +505,20 @@ impl SeriesEstimator { mod tests { use std::sync::Arc; + use ::parquet::arrow::AsyncArrowWriter; + use ::parquet::arrow::arrow_reader::ParquetRecordBatchReaderBuilder; + use ::parquet::basic::LogicalType; + use ::parquet::variant::{VariantArray, VariantType, json_to_variant}; use common_query::prelude::greptime_native_histogram; use datatypes::arrow::array::{ - BinaryArray, DictionaryArray, TimestampMillisecondArray, UInt8Array, UInt32Array, - UInt64Array, + ArrayRef, BinaryArray, DictionaryArray, Int64Array, StringArray, StructArray, + TimestampMillisecondArray, UInt8Array, UInt32Array, UInt64Array, }; use datatypes::arrow::datatypes::{DataType as ArrowDataType, Field, Schema, TimeUnit}; use datatypes::arrow::record_batch::RecordBatch; + use datatypes::extension::json::{Json2ExtensionType, Json2PhysicalLayout}; + use datatypes::vectors::json::array::JsonArray; + use serde_json::json; use super::*; @@ -1004,4 +1011,125 @@ mod tests { err ); } + + fn json2_v2_test_type() -> ArrowDataType { + ArrowDataType::Struct( + vec![ + Arc::new(Field::new("active", ArrowDataType::Boolean, true)), + Arc::new(Field::new("hot", ArrowDataType::Int64, true)), + Arc::new(Field::new("name", ArrowDataType::Utf8, true)), + ] + .into(), + ) + } + + /// Validates the persisted-format foundation for the JSON2 v2 remainder. + /// + /// JSON2 will store `!__remainder__!` as a nested Variant child of its root + /// Struct. Before enabling that layout in production, this test ensures the + /// SST schema wrapper and Arrow writer preserve the Variant extension, + /// encode the Parquet Variant logical type, and round-trip the values + /// without changing the surrounding Struct. + #[tokio::test] + async fn test_nested_variant_survives_sst_writer_schema_roundtrip() + -> Result<(), Box> { + let json: ArrayRef = Arc::new(StringArray::from(vec![ + Some(r#"{}"#), + Some(r#"{"name":"Alice","active":true}"#), + Some(r#"{"nested":{"count":42},"items":[1,"two",null]}"#), + Some(r#"{"\u5b57\u6bb5":"\u503c"}"#), + None, + ])); + let remainder = json_to_variant(&json)?; + let remainder_field = remainder.field("!__remainder__!"); + let remainder_array = ArrayRef::from(remainder); + let hot_field = Field::new("hot", ArrowDataType::Int64, true); + let data_array = Arc::new(StructArray::new( + vec![remainder_field.clone(), hot_field.clone()].into(), + vec![ + remainder_array, + Arc::new(Int64Array::from(vec![ + Some(1), + Some(2), + Some(3), + Some(4), + None, + ])), + ], + None, + )); + let data_field = Field::new( + "data", + ArrowDataType::Struct(vec![remainder_field, hot_field].into()), + true, + ) + .with_extension_type(Json2ExtensionType::default()); + let schema = Arc::new(Schema::new(vec![data_field])); + let source = RecordBatch::try_new(schema.clone(), vec![data_array])?; + + let wrapped = maybe_wrap_schema(&schema)?; + let mut buffer = Vec::new(); + let mut writer = AsyncArrowWriter::try_new(&mut buffer, wrapped, None)?; + writer.write(&source).await?; + writer.close().await?; + + let builder = ParquetRecordBatchReaderBuilder::try_new(bytes::Bytes::from(buffer))?; + let parquet_remainder = + &builder.parquet_schema().root_schema().get_fields()[0].get_fields()[0]; + assert_eq!( + parquet_remainder.get_basic_info().logical_type_ref(), + Some(&LogicalType::Variant { + specification_version: None, + }) + ); + + let ArrowDataType::Struct(children) = builder.schema().field_with_name("data")?.data_type() + else { + unreachable!(); + }; + assert!(children[0].has_valid_extension_type::()); + + let mut reader = builder.build()?; + let result = reader.next().unwrap()?; + assert_eq!(source, result); + let result_field = result.schema().field(0).clone(); + let result = result + .column(0) + .as_any() + .downcast_ref::() + .unwrap(); + VariantArray::try_new(result.column(0))?; + let result: ArrayRef = Arc::new(result.clone()); + let result = + JsonArray::from(&result).project_to_v2(&result_field, &json2_v2_test_type())?; + assert_eq!( + json!({"active": true, "hot": 2, "name": "Alice"}), + JsonArray::from(&result).try_get_value(1)? + ); + Ok(()) + } + + /// Ensures future readers retain compatibility with the first JSON2 v2 layout. + #[test] + fn test_read_json2_v2_fixture() -> Result<(), Box> { + let bytes = bytes::Bytes::from_static(include_bytes!("../test-data/json2-v2.parquet")); + let builder = ParquetRecordBatchReaderBuilder::try_new(bytes)?; + let field = builder.schema().field(0).clone(); + assert!(Json2PhysicalLayout::try_from_root(&field)?.is_version_2()); + + let batch = builder.build()?.next().unwrap()?; + let data = batch + .column(0) + .as_any() + .downcast_ref::() + .unwrap(); + VariantArray::try_new(data.column(0))?; + let data: ArrayRef = Arc::new(data.clone()); + let data = JsonArray::from(&data).project_to_v2(&field, &json2_v2_test_type())?; + assert_eq!( + json!({"active": true, "hot": 2, "name": "Alice"}), + JsonArray::from(&data).try_get_value(1)? + ); + Ok(()) + } } diff --git a/src/mito2/src/sst/index.rs b/src/mito2/src/sst/index.rs index 581d1edebd..7954af2b92 100644 --- a/src/mito2/src/sst/index.rs +++ b/src/mito2/src/sst/index.rs @@ -706,6 +706,14 @@ pub struct IndexBuildTask { pub region_id: RegionId, /// The SST file handle to build index for. pub file: FileHandle, + /// The target region metadata used to decode rows from the SST. + /// + /// An SST may originate in another region while being visible in the target + /// manifest. This metadata defines the target schema and sequence domain; + /// applying the staging manifest only makes imported files visible. Index + /// rebuild happens later when a flush, compaction, schema change, or manual + /// index build request schedules it. + pub(crate) target_region_metadata: RegionMetadataRef, /// The manifest state this build is based on. pub(crate) source: IndexBuildSource, pub reason: IndexBuildType, @@ -846,6 +854,7 @@ impl IndexBuildTask { let mut parquet_reader = self .access_layer .read_sst(self.file.clone()) // use the latest file handle instead of creating a new one + .expected_metadata(Some(self.target_region_metadata.clone())) .build() .await?; @@ -1494,8 +1503,12 @@ mod tests { use datatypes::schema::{ ColumnSchema, FulltextOptions, SkippingIndexOptions, SkippingIndexType, }; + use datatypes::value::Value; + use index::inverted_index::format::reader::InvertedIndexReader; use object_store::ObjectStore; use object_store::services::Memory; + use partition::expr::col; + use puffin::puffin_manager::{PuffinManager, PuffinReader}; use puffin_manager::PuffinManagerFactory; use store_api::metadata::{ColumnMetadata, RegionMetadataBuilder}; use tokio::sync::mpsc; @@ -2088,6 +2101,7 @@ mod tests { let task = IndexBuildTask { region_id, file, + target_region_metadata: version_control.current().version.metadata.clone(), source: IndexBuildSource::new( file_meta, version_control.current().version.metadata.schema_version, @@ -2120,16 +2134,34 @@ mod tests { } #[tokio::test] - async fn test_index_build_task_increments_legacy_index_version() { + async fn test_index_build_task_foreign_file_uses_target_metadata() { let env = SchedulerEnv::new().await; let mut scheduler = env.mock_index_build_scheduler(4); - let metadata = Arc::new(sst_region_metadata()); - let manifest_ctx = env.mock_manifest_context(metadata.clone()).await; - let region_id = metadata.region_id; + let source_metadata = Arc::new(sst_region_metadata()); + let mut target_metadata = (*source_metadata).clone(); + target_metadata.region_id = RegionId::new(1, 3); + let mut target_builder = RegionMetadataBuilder::new(target_metadata.region_id); + for mut column_metadata in target_metadata.column_metadatas.clone() { + if column_metadata.column_id == 2 { + column_metadata.column_schema = + column_metadata.column_schema.with_inverted_index(true); + } + target_builder.push_column_metadata(column_metadata); + } + let partition_expr = col("field_0") + .gt_eq(Value::UInt64(100)) + .and(col("field_0").lt(Value::UInt64(200))); + target_builder + .primary_key(target_metadata.primary_key.clone()) + .partition_expr_json(Some(partition_expr.as_json_str().unwrap())) + .bump_version(); + let target_metadata = Arc::new(target_builder.build().unwrap()); + let manifest_ctx = env.mock_manifest_context(target_metadata.clone()).await; + let region_id = target_metadata.region_id; let file_purger = Arc::new(NoopFilePurger {}); - let sst_info = mock_sst_file(metadata.clone(), &env, IndexBuildMode::Async).await; + let sst_info = mock_sst_file(source_metadata.clone(), &env, IndexBuildMode::Async).await; let file_meta = FileMeta { - region_id, + region_id: source_metadata.region_id, file_id: sst_info.file_id, file_size: sst_info.file_size, max_row_group_uncompressed_size: sst_info.max_row_group_uncompressed_size, @@ -2144,8 +2176,8 @@ mod tests { seed_manifest_file(&manifest_ctx, &file_meta).await; let files = HashMap::from([(file_meta.file_id, file_meta.clone())]); let version_control = - mock_version_control(metadata.clone(), file_purger.clone(), files).await; - let indexer_builder = mock_indexer_builder(metadata.clone(), &env).await; + mock_version_control(target_metadata.clone(), file_purger.clone(), files).await; + let indexer_builder = mock_indexer_builder(target_metadata.clone(), &env).await; let file = FileHandle::new(file_meta.clone(), file_purger.clone()); @@ -2155,6 +2187,7 @@ mod tests { let task = IndexBuildTask { region_id, file, + target_region_metadata: version_control.current().version.metadata.clone(), source: IndexBuildSource::new( file_meta.clone(), version_control.current().version.metadata.schema_version, @@ -2199,9 +2232,40 @@ mod tests { assert!(updated_meta.index_file_size > 0); assert_eq!(updated_meta.file_id, file_meta.file_id); assert_eq!(updated_meta.index_version, 1); + let field_0_index = updated_meta + .indexes + .iter() + .find(|index| index.column_id == 2) + .expect("field_0 should have an inverted index"); + assert_eq!( + field_0_index.created_indexes.as_slice(), + [IndexType::InvertedIndex] + ); } _ => panic!("Unexpected worker request: {:?}", worker_req), } + + let puffin_reader = env + .access_layer + .build_puffin_manager() + .reader(&RegionIndexId::new( + RegionFileId::new(source_metadata.region_id, file_meta.file_id), + 1, + )) + .await + .unwrap(); + let blob = puffin_reader + .blob(inverted_index::INDEX_BLOB_TYPE) + .await + .unwrap(); + let blob_reader = blob.reader().await.unwrap(); + let index_metadata = + index::inverted_index::format::reader::InvertedIndexBlobReader::new(blob_reader) + .metadata(None) + .await + .unwrap(); + assert!(index_metadata.metas.contains_key("2")); + assert_eq!(index_metadata.total_row_count, 100); } async fn schedule_index_build_task_with_mode(build_mode: IndexBuildMode) { @@ -2236,6 +2300,7 @@ mod tests { let task = IndexBuildTask { region_id, file, + target_region_metadata: version_control.current().version.metadata.clone(), source: IndexBuildSource::new( file_meta.clone(), version_control.current().version.metadata.schema_version, @@ -2346,6 +2411,7 @@ mod tests { let task = IndexBuildTask { region_id, file, + target_region_metadata: version_control.current().version.metadata.clone(), source: IndexBuildSource::new( file_meta.clone(), version_control.current().version.metadata.schema_version, @@ -2444,6 +2510,7 @@ mod tests { let task = IndexBuildTask { region_id, file, + target_region_metadata: version_control.current().version.metadata.clone(), source: IndexBuildSource::new( file_meta.clone(), version_control.current().version.metadata.schema_version, @@ -2505,7 +2572,7 @@ mod tests { let schema_version = metadata.schema_version; let manifest_ctx = env.mock_manifest_context(metadata.clone()).await; let file_purger = Arc::new(NoopFilePurger {}); - let indexer_builder = mock_indexer_builder(metadata, env).await; + let indexer_builder = mock_indexer_builder(metadata.clone(), env).await; let (tx, _rx) = mpsc::channel(4); let (result_tx, result_rx) = mpsc::channel::>(4); @@ -2521,6 +2588,7 @@ mod tests { let task = IndexBuildTask { region_id, file, + target_region_metadata: metadata, source: IndexBuildSource::new(file_meta, schema_version), reason, access_layer: env.access_layer.clone(), diff --git a/src/mito2/src/sst/parquet.rs b/src/mito2/src/sst/parquet.rs index 5c40bc36be..058e416527 100644 --- a/src/mito2/src/sst/parquet.rs +++ b/src/mito2/src/sst/parquet.rs @@ -14,11 +14,13 @@ //! SST in parquet format. +use std::collections::BTreeMap; use std::sync::Arc; use common_base::readable_size::ReadableSize; +use datatypes::json::JsonSettings; use parquet::file::metadata::ParquetMetaData; -use store_api::storage::FileId; +use store_api::storage::{ColumnId, FileId}; use crate::sst::DEFAULT_WRITE_BUFFER_SIZE; use crate::sst::file::FileTimeRange; @@ -48,6 +50,19 @@ pub const PARQUET_METADATA_KEY: &str = "greptime:metadata"; /// default execution batch size to reduce rebatching and concatenation in the /// query pipeline. pub(crate) const DEFAULT_READ_BATCH_SIZE: usize = 8 * 1024; + +/// JSON2 physical layouts requested by a compaction read. +pub(crate) type Json2RewriteTargets = Arc>; + +/// Fixed JSON2 physical layout used while rewriting compaction input. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct Json2TargetLayout { + /// Logical JSON2 extension metadata attached to the rewritten field. + pub(crate) extension_metadata: String, + /// Settings used to build the fixed physical layout. + pub(crate) target_layout: JsonSettings, +} + /// Default row group size for parquet files. /// /// Keep the existing persisted/on-disk default stable. It intentionally stays @@ -118,8 +133,8 @@ mod tests { use datafusion_expr::{BinaryExpr, Expr, Literal, Operator, col, lit}; use datatypes::arrow; use datatypes::arrow::array::{ - ArrayRef, BinaryDictionaryBuilder, RecordBatch, StringArray, StringDictionaryBuilder, - TimestampMillisecondArray, UInt8Array, UInt64Array, + ArrayRef, AsArray, BinaryDictionaryBuilder, RecordBatch, StringArray, + StringDictionaryBuilder, TimestampMillisecondArray, UInt8Array, UInt64Array, }; use datatypes::arrow::datatypes::{DataType, Field, Schema, UInt32Type}; use datatypes::arrow::util::pretty::pretty_format_batches; @@ -152,6 +167,7 @@ mod tests { use crate::sst::index::inverted_index::applier::builder::InvertedIndexApplierBuilder; use crate::sst::index::{IndexBuildType, Indexer, IndexerBuilder, IndexerBuilderImpl}; use crate::sst::parquet::flat_format::FlatWriteFormat; + use crate::sst::parquet::metadata::extract_primary_key_range; use crate::sst::parquet::reader::{ParquetReader, ParquetReaderBuilder, ReaderMetrics}; use crate::sst::parquet::row_selection::RowGroupSelection; use crate::sst::parquet::writer::ParquetWriter; @@ -687,12 +703,13 @@ mod tests { let object_store = env.init_object_store_manager(); let metadata = Arc::new(sst_region_metadata()); let batches = vec![ - new_record_batch_by_range(&["a", "d"], 0, 1000), - new_record_batch_by_range(&["b", "f"], 0, 1000), - new_record_batch_by_range(&["c", "g"], 0, 1000), - new_record_batch_by_range(&["b", "h"], 100, 200), - new_record_batch_by_range(&["b", "h"], 200, 300), - new_record_batch_by_range(&["b", "h"], 300, 1000), + new_record_batch_by_range(&["a", "a"], 0, 1000), + new_record_batch_by_range(&["b", "b"], 0, 1000), + new_record_batch_by_range(&["c", "c"], 0, 1000), + new_record_batch_by_range(&["d", "d"], 100, 200), + new_record_batch_by_range(&["d", "d"], 200, 300), + new_record_batch_by_range(&["d", "d"], 300, 1000), + new_record_batch_by_range(&["e", "e"], 0, 100), ]; let total_rows: usize = batches.iter().map(|batch| batch.num_rows()).sum(); @@ -745,6 +762,159 @@ mod tests { assert_eq!(total_rows, rows_read); } + #[tokio::test] + async fn test_split_file_at_series_boundary_inside_batch() { + let mut env = TestEnv::new().await; + let object_store = env.init_object_store_manager(); + let metadata = Arc::new(sst_region_metadata()); + let first_batch_rows = (0..1000).map(|ts| ("a", "a", ts)).collect::>(); + let second_batch_rows = (0..1000).map(|ts| ("b", "b", ts)).collect::>(); + let mut third_batch_rows = (1000..2000).map(|ts| ("b", "b", ts)).collect::>(); + third_batch_rows.extend((0..1000).map(|ts| ("c", "c", ts))); + let batches = vec![ + new_record_batch_from_rows(&first_batch_rows), + new_record_batch_from_rows(&second_batch_rows), + new_record_batch_from_rows(&third_batch_rows), + ]; + let total_rows = batches.iter().map(RecordBatch::num_rows).sum::(); + let source = new_flat_source_from_record_batches(batches); + let write_opts = WriteOptions { + row_group_size: 50, + max_file_size: Some(1), + ..Default::default() + }; + let path_provider = RegionFilePathFactory { + table_dir: "test_series_boundary".to_string(), + path_type: PathType::Bare, + }; + let mut metrics = Metrics::new(WriteType::Compaction); + let mut writer = ParquetWriter::new_with_object_store( + object_store, + metadata.clone(), + IndexConfig::default(), + NoopIndexBuilder, + path_provider, + &mut metrics, + ) + .await; + + let files = writer + .write_all_flat(source, None, &write_opts) + .await + .unwrap(); + + assert!(files.len() > 1); + assert_eq!( + total_rows, + files.iter().map(|file| file.num_rows).sum::() + ); + let primary_key_ranges = files + .iter() + .map(|file| { + extract_primary_key_range(file.file_metadata.as_ref().unwrap(), metadata.as_ref()) + .unwrap() + }) + .collect::>(); + assert!( + primary_key_ranges + .windows(2) + .all(|ranges| ranges[0].1 < ranges[1].0) + ); + } + + #[tokio::test] + async fn test_oversized_single_series_stays_in_one_file() { + let mut env = TestEnv::new().await; + let object_store = env.init_object_store_manager(); + let metadata = Arc::new(sst_region_metadata()); + let first_batch_rows = (0..1000).map(|ts| ("a", "a", ts)).collect::>(); + let second_batch_rows = (1000..2000).map(|ts| ("a", "a", ts)).collect::>(); + let third_batch_rows = (2000..3000).map(|ts| ("a", "a", ts)).collect::>(); + let batches = vec![ + new_record_batch_from_rows(&first_batch_rows), + new_record_batch_from_rows(&second_batch_rows), + new_record_batch_from_rows(&third_batch_rows), + ]; + let total_rows = batches.iter().map(RecordBatch::num_rows).sum::(); + let source = new_flat_source_from_record_batches(batches); + let write_opts = WriteOptions { + row_group_size: 50, + max_file_size: Some(1), + ..Default::default() + }; + let path_provider = RegionFilePathFactory { + table_dir: "test_oversized_series".to_string(), + path_type: PathType::Bare, + }; + let mut metrics = Metrics::new(WriteType::Compaction); + let mut writer = ParquetWriter::new_with_object_store( + object_store, + metadata, + IndexConfig::default(), + NoopIndexBuilder, + path_provider, + &mut metrics, + ) + .await; + + let files = writer + .write_all_flat_as_primary_key(source, None, &write_opts) + .await + .unwrap(); + + assert_eq!(1, files.len()); + assert_eq!(total_rows, files[0].num_rows); + assert!(files[0].file_size > write_opts.max_file_size.unwrap() as u64); + } + + #[tokio::test] + async fn test_write_multiple_files_without_primary_key() { + let mut env = TestEnv::new().await; + let object_store = env.init_object_store_manager(); + let metadata = Arc::new(sst_region_metadata_without_primary_key()); + let batch_rows = 1000; + let batches = vec![ + new_record_batch_without_primary_key(0, batch_rows), + new_record_batch_without_primary_key(batch_rows, 2 * batch_rows), + new_record_batch_without_primary_key(2 * batch_rows, 3 * batch_rows), + ]; + let total_rows = batches.iter().map(RecordBatch::num_rows).sum::(); + let source = new_flat_source_from_record_batches(batches); + let write_opts = WriteOptions { + row_group_size: 50, + max_file_size: Some(1), + ..Default::default() + }; + let path_provider = RegionFilePathFactory { + table_dir: "test_no_primary_key".to_string(), + path_type: PathType::Bare, + }; + let mut metrics = Metrics::new(WriteType::Compaction); + let mut writer = ParquetWriter::new_with_object_store( + object_store, + metadata, + IndexConfig::default(), + NoopIndexBuilder, + path_provider, + &mut metrics, + ) + .await; + + let files = writer + .write_all_flat(source, None, &write_opts) + .await + .unwrap(); + + // Regions without a primary key keep splitting at batch boundaries: + // the limit produces multiple files and no batch is sliced. + assert!(files.len() > 1); + assert!(files.iter().all(|file| file.num_rows % batch_rows == 0)); + assert_eq!( + total_rows, + files.iter().map(|file| file.num_rows).sum::() + ); + } + #[tokio::test] async fn test_write_read_with_index() { let mut env = TestEnv::new().await; @@ -1145,6 +1315,62 @@ mod tests { assert!(reader.next_record_batch().await.unwrap().is_none()); } + /// Creates a new region metadata without primary key for testing SSTs. + /// + /// Schema: field_0, ts + fn sst_region_metadata_without_primary_key() -> RegionMetadata { + let mut builder = RegionMetadataBuilder::new(REGION_ID); + builder + .push_column_metadata(ColumnMetadata { + column_schema: ColumnSchema::new( + "field_0".to_string(), + ConcreteDataType::uint64_datatype(), + true, + ), + semantic_type: SemanticType::Field, + column_id: 0, + }) + .push_column_metadata(ColumnMetadata { + column_schema: ColumnSchema::new( + "ts".to_string(), + ConcreteDataType::timestamp_millisecond_datatype(), + false, + ), + semantic_type: SemanticType::Timestamp, + column_id: 1, + }) + .primary_key(vec![]); + builder.build().unwrap() + } + + /// Creates a flat format RecordBatch for regions without a primary key. + fn new_record_batch_without_primary_key(start: usize, end: usize) -> RecordBatch { + assert!(end >= start); + let metadata = Arc::new(sst_region_metadata_without_primary_key()); + let flat_schema = to_flat_sst_arrow_schema(&metadata, &FlatSchemaOptions::default()); + + let num_rows = end - start; + let mut pk_builder = BinaryDictionaryBuilder::::new(); + // Regions without a primary key encode it as empty bytes. + for _ in 0..num_rows { + pk_builder.append([]).unwrap(); + } + + RecordBatch::try_new( + flat_schema, + vec![ + Arc::new(UInt64Array::from_iter_values(start as u64..end as u64)) as ArrayRef, + Arc::new(TimestampMillisecondArray::from_iter_values( + start as i64..end as i64, + )) as ArrayRef, + Arc::new(pk_builder.finish()) as ArrayRef, + Arc::new(UInt64Array::from_value(1000, num_rows)) as ArrayRef, + Arc::new(UInt8Array::from_value(OpType::Put as u8, num_rows)) as ArrayRef, + ], + ) + .unwrap() + } + fn new_record_batch_from_rows(rows: &[(&str, &str, i64)]) -> RecordBatch { let metadata = Arc::new(sst_region_metadata()); let flat_schema = to_flat_sst_arrow_schema(&metadata, &FlatSchemaOptions::default()); @@ -1414,91 +1640,178 @@ mod tests { #[tokio::test] async fn test_read_with_override_sequence() { + test_read_with_override_sequence_with_format(false).await; + test_read_with_override_sequence_with_format(true).await; + } + + async fn test_read_with_override_sequence_with_format(flat_format: bool) { let mut env = TestEnv::new().await; let object_store = env.init_object_store_manager(); - let handle = sst_file_handle(0, 1000); - let file_path = FixedPathProvider { - region_file_id: handle.file_id(), - }; let metadata = Arc::new(sst_region_metadata()); - // Create batches with sequence 0 to trigger override functionality. - let source = new_flat_source_from_record_batches(vec![ - new_record_batch_with_custom_sequence(&["a", "d"], 0, 60, 0), - new_record_batch_with_custom_sequence(&["b", "f"], 0, 40, 0), - ]); + async fn read_sequences(builder: ParquetReaderBuilder) -> Vec { + let mut reader = builder.build().await.unwrap().unwrap(); + let mut sequences = Vec::new(); + while let Some(batch) = reader.next_record_batch().await.unwrap() { + let sequence = batch + .column(batch.num_columns() - 2) + .as_primitive::(); + sequences.extend((0..sequence.len()).map(|idx| sequence.value(idx))); + } + sequences + } - let write_opts = WriteOptions { - row_group_size: 50, - ..Default::default() - }; + async fn write_sst( + object_store: ObjectStore, + metadata: Arc, + handle: FileHandle, + flat_format: bool, + sequence: u64, + ) { + let file_path = FixedPathProvider { + region_file_id: handle.file_id(), + }; + let source = new_flat_source_from_record_batches(vec![ + new_record_batch_with_custom_sequence(&["a", "d"], 0, 60, sequence), + new_record_batch_with_custom_sequence(&["b", "f"], 0, 40, sequence), + ]); + let write_opts = WriteOptions { + row_group_size: 50, + ..Default::default() + }; + let mut metrics = Metrics::new(WriteType::Flush); + let mut writer = ParquetWriter::new_with_object_store( + object_store, + metadata, + IndexConfig::default(), + NoopIndexBuilder, + file_path, + &mut metrics, + ) + .await; + if flat_format { + writer + .write_all_flat(source, None, &write_opts) + .await + .unwrap(); + } else { + writer + .write_all_flat_as_primary_key(source, None, &write_opts) + .await + .unwrap(); + } + } - let mut metrics = Metrics::new(WriteType::Flush); - let mut writer = ParquetWriter::new_with_object_store( + let custom_sequence = 12345u64; + let local_zero_handle = sst_file_handle(0, 1000); + write_sst( object_store.clone(), metadata.clone(), - IndexConfig::default(), - NoopIndexBuilder, - file_path, - &mut metrics, + local_zero_handle.clone(), + flat_format, + 0, ) .await; - writer - .write_all_flat_as_primary_key(source, None, &write_opts) - .await - .unwrap() - .remove(0); + // Local all-zero SSTs retain the compatibility override. + let local_zero_none = read_sequences( + ParquetReaderBuilder::new( + FILE_DIR.to_string(), + PathType::Bare, + local_zero_handle.clone(), + object_store.clone(), + ) + .expected_metadata(Some(metadata.clone())), + ) + .await; + assert!(local_zero_none.iter().all(|sequence| *sequence == 0)); - // Read without override sequence (should read sequence 0) - let builder = ParquetReaderBuilder::new( - FILE_DIR.to_string(), - PathType::Bare, - handle.clone(), - object_store.clone(), - ); - let mut reader = builder.build().await.unwrap().unwrap(); - let mut normal_batches = Vec::new(); - while let Some(batch) = reader.next_record_batch().await.unwrap() { - normal_batches.push(batch); - } - - // Read with override sequence using FileMeta.sequence - let custom_sequence = 12345u64; - let file_meta = handle.meta_ref(); - let mut override_file_meta = file_meta.clone(); - override_file_meta.sequence = Some(std::num::NonZero::new(custom_sequence).unwrap()); - let override_handle = FileHandle::new( - override_file_meta, + let mut local_zero_meta = local_zero_handle.meta_ref().clone(); + local_zero_meta.sequence = Some(std::num::NonZeroU64::new(custom_sequence).unwrap()); + let local_zero_override_handle = FileHandle::new( + local_zero_meta, Arc::new(crate::sst::file_purger::NoopFilePurger), ); - - let builder = ParquetReaderBuilder::new( - FILE_DIR.to_string(), - PathType::Bare, - override_handle, - object_store.clone(), + let local_zero_override = read_sequences( + ParquetReaderBuilder::new( + FILE_DIR.to_string(), + PathType::Bare, + local_zero_override_handle, + object_store.clone(), + ) + .expected_metadata(Some(metadata.clone())), + ) + .await; + assert!( + local_zero_override + .iter() + .all(|sequence| *sequence == custom_sequence) ); - let mut reader = builder.build().await.unwrap().unwrap(); - let mut override_batches = Vec::new(); - while let Some(batch) = reader.next_record_batch().await.unwrap() { - override_batches.push(batch); - } - // Compare the results - assert_eq!(normal_batches.len(), override_batches.len()); - for (normal, override_batch) in normal_batches.into_iter().zip(override_batches.iter()) { - let expected_batch = { - let mut columns = normal.columns().to_vec(); - let num_cols = columns.len(); - columns[num_cols - 2] = - Arc::new(UInt64Array::from_value(custom_sequence, normal.num_rows())); - RecordBatch::try_new(normal.schema(), columns).unwrap() - }; + let local_nonzero_handle = sst_file_handle(0, 1000); + write_sst( + object_store.clone(), + metadata.clone(), + local_nonzero_handle.clone(), + flat_format, + 7, + ) + .await; - // Override batch should match expected batch - assert_eq!(*override_batch, expected_batch); - } + // Local nonzero SSTs retain physical per-row sequences, even with FileMeta.sequence. + let mut local_nonzero_meta = local_nonzero_handle.meta_ref().clone(); + local_nonzero_meta.sequence = Some(std::num::NonZeroU64::new(custom_sequence).unwrap()); + let local_nonzero_override_handle = FileHandle::new( + local_nonzero_meta, + Arc::new(crate::sst::file_purger::NoopFilePurger), + ); + let local_nonzero_override = read_sequences( + ParquetReaderBuilder::new( + FILE_DIR.to_string(), + PathType::Bare, + local_nonzero_override_handle, + object_store.clone(), + ) + .expected_metadata(Some(metadata.clone())), + ) + .await; + assert!(local_nonzero_override.iter().all(|sequence| *sequence == 7)); + + // None never overrides a local nonzero physical sequence. + let local_nonzero_none = read_sequences( + ParquetReaderBuilder::new( + FILE_DIR.to_string(), + PathType::Bare, + local_nonzero_handle.clone(), + object_store.clone(), + ) + .expected_metadata(Some(metadata.clone())), + ) + .await; + assert!(local_nonzero_none.iter().all(|sequence| *sequence == 7)); + + // A source-owned handle is foreign when read against target metadata, so the + // target-local manifest barrier is applied even for nonzero physical sequences. + let mut target_metadata = (*metadata).clone(); + target_metadata.region_id = RegionId::new(0, 1); + let target_metadata = Arc::new(target_metadata); + let mut foreign_meta = local_nonzero_handle.meta_ref().clone(); + foreign_meta.sequence = Some(std::num::NonZeroU64::new(custom_sequence).unwrap()); + let foreign_handle = FileHandle::new( + foreign_meta, + Arc::new(crate::sst::file_purger::NoopFilePurger), + ); + let foreign = read_sequences( + ParquetReaderBuilder::new( + FILE_DIR.to_string(), + PathType::Bare, + foreign_handle, + object_store, + ) + .expected_metadata(Some(target_metadata)), + ) + .await; + assert!(foreign.iter().all(|sequence| *sequence == custom_sequence)); } #[tokio::test] diff --git a/src/mito2/src/sst/parquet/flat_format.rs b/src/mito2/src/sst/parquet/flat_format.rs index e381ea76f4..5432c112f4 100644 --- a/src/mito2/src/sst/parquet/flat_format.rs +++ b/src/mito2/src/sst/parquet/flat_format.rs @@ -241,9 +241,8 @@ impl FlatReadFormat { } /// Enables wrapping binary `__primary_key` batches back to a dictionary in [`Self::convert_batch`]. - pub(crate) fn set_pk_as_binary(&mut self) -> Result<()> { - self.pk_dict_wrap_schema = Some(self.output_arrow_schema()?); - Ok(()) + pub(crate) fn set_pk_as_binary(&mut self, output_schema: SchemaRef) { + self.pk_dict_wrap_schema = Some(output_schema); } /// Index of a column in the projected batch by its column id. @@ -306,19 +305,16 @@ impl FlatReadFormat { .project(projection) .context(ComputeArrowSnafu)?; let mut fields = schema.fields().iter().cloned().collect::>(); - for (column_id, target_type) in self.json_target_types().iter() { + for (column_id, target) in self.json_target_types().iter() { let Some(index) = self.parquet_projected_index_by_id(*column_id) else { continue; }; let Some(field) = schema.fields().get(index) else { continue; }; - fields[index] = Arc::new( - field - .as_ref() - .clone() - .with_data_type(ConcreteDataType::json2(target_type.clone()).as_arrow_type()), - ); + let mut field = field.as_ref().clone(); + field.set_data_type(ConcreteDataType::json2(target.clone()).as_arrow_type()); + fields[index] = Arc::new(field); } schema.fields = fields.into(); Ok(Arc::new(schema)) @@ -326,7 +322,7 @@ impl FlatReadFormat { /// Index of a column in the projected schema produced directly by parquet /// reading, before any primary-key-to-flat conversion. - fn parquet_projected_index_by_id(&self, column_id: ColumnId) -> Option { + pub(crate) fn parquet_projected_index_by_id(&self, column_id: ColumnId) -> Option { match &self.parquet_adapter { ParquetAdapter::Flat(p) => p .format_projection @@ -359,7 +355,7 @@ impl FlatReadFormat { } } - /// Gets JSON2 target types keyed by column id. + /// Gets JSON2 read targets. pub(crate) fn json_target_types(&self) -> &JsonTargetTypes { self.read_cols.json_target_types() } @@ -1012,9 +1008,8 @@ mod tests { false, ) .unwrap(); - read_format.set_pk_as_binary().unwrap(); - let output_schema = read_format.output_arrow_schema().unwrap(); + read_format.set_pk_as_binary(output_schema.clone()); let binary_schema = override_pk_field_to_binary(&output_schema); // The __primary_key field must preserve its field_id metadata after diff --git a/src/mito2/src/sst/parquet/format.rs b/src/mito2/src/sst/parquet/format.rs index f3f0a34a68..c0cfc7b981 100644 --- a/src/mito2/src/sst/parquet/format.rs +++ b/src/mito2/src/sst/parquet/format.rs @@ -53,7 +53,7 @@ use store_api::storage::{ColumnId, NestedPath, SequenceNumber}; use crate::error::{ ConvertVectorSnafu, DecodeSnafu, InvalidRecordBatchSnafu, NewRecordBatchSnafu, Result, }; -use crate::read::read_columns::{JsonTargetTypes, ReadColumns}; +use crate::read::read_columns::ReadColumns; use crate::read::{Batch, BatchBuilder, BatchColumn}; use crate::sst::file::{FileMeta, FileTimeRange}; use crate::sst::parquet::read_columns::{ParquetReadColumn, ParquetReadColumns}; @@ -599,14 +599,13 @@ impl FormatProjection { sst_column_num: usize, cols: ReadColumns, ) -> Self { - let json_target_types = cols.json_target_types().clone(); let mut projected_columns: Vec<_> = cols .col_ids - .into_iter() + .iter() + .copied() .filter_map(|col_id| { id_to_index.get(&col_id).copied().map(|index_of_sst| { - let nested_paths = - json_target_nested_paths(metadata, &json_target_types, col_id); + let nested_paths = json_target_nested_paths(metadata, &cols, col_id); (col_id, index_of_sst, nested_paths) }) }) @@ -688,10 +687,10 @@ impl FormatProjection { fn json_target_nested_paths( metadata: &RegionMetadataRef, - json_target_types: &JsonTargetTypes, + read_columns: &ReadColumns, column_id: ColumnId, ) -> Vec { - let Some(target_type) = json_target_types.get(&column_id) else { + let Some(target_type) = read_columns.json_target_type(column_id) else { return Vec::new(); }; let Some(column) = metadata.column_by_id(column_id) else { diff --git a/src/mito2/src/sst/parquet/json_align/stream.rs b/src/mito2/src/sst/parquet/json_align/stream.rs index fabb3c807b..bfd0b82edc 100644 --- a/src/mito2/src/sst/parquet/json_align/stream.rs +++ b/src/mito2/src/sst/parquet/json_align/stream.rs @@ -12,15 +12,17 @@ // See the License for the specific language governing permissions and // limitations under the License. +use std::collections::HashMap; use std::pin::Pin; use std::task::{Context, Poll}; use datafusion_common::cast_column; use datafusion_common::format::DEFAULT_CAST_OPTIONS; use datatypes::arrow::array::{ArrayRef, new_null_array}; -use datatypes::arrow::datatypes::{DataType, FieldRef, SchemaRef}; +use datatypes::arrow::datatypes::{DataType, Field, FieldRef, SchemaRef}; use datatypes::arrow::record_batch::RecordBatch; -use datatypes::extension::json::is_json2_extension_type; +use datatypes::extension::json::{JsonMetadata, is_json2_extension_type}; +use datatypes::json::JsonSettings; use datatypes::vectors::json::array::JsonArray; use futures::Stream; use snafu::{ResultExt, ensure}; @@ -28,6 +30,13 @@ use snafu::{ResultExt, ensure}; use crate::error::{ CastColumnSnafu, DataTypeMismatchSnafu, NewRecordBatchSnafu, Result, UnexpectedSnafu, }; +use crate::sst::parquet::Json2TargetLayout; + +#[derive(Debug)] +struct Json2RewriteSettings { + logical_settings: JsonSettings, + target_layout: JsonSettings, +} /// Aligns projected batches to the expected output schema for nested projections. /// @@ -61,6 +70,8 @@ pub struct NestedSchemaAligner { /// Whether all projected roots are present and the stream can pass batches /// through. all_roots_present: bool, + /// JSON2 columns that require semantic source-to-target layout rewriting. + json2_rewrite_targets: HashMap, /// The cache for whether incoming batches already match output schema. is_schema_matched: Option, } @@ -96,9 +107,39 @@ where projected_root_presence, expected_input_col_num, all_roots_present, + json2_rewrite_targets: HashMap::new(), is_schema_matched: None, }) } + + /// Sets JSON2 columns that must be rewritten into the output field layout. + pub(crate) fn with_json2_rewrite_targets( + mut self, + targets: &HashMap, + ) -> Result { + self.json2_rewrite_targets = targets + .iter() + .map(|(name, layout)| { + let metadata = serde_json::from_str::(&layout.extension_metadata) + .map_err(|e| { + UnexpectedSnafu { + reason: format!( + "invalid JSON2 extension metadata for column '{name}': {e}" + ), + } + .build() + })?; + Ok(( + name.clone(), + Json2RewriteSettings { + logical_settings: metadata.into_json_settings(), + target_layout: layout.target_layout.clone(), + }, + )) + }) + .collect::>()?; + Ok(self) + } } impl Stream for NestedSchemaAligner @@ -125,6 +166,7 @@ where &this.output_schema, &this.projected_root_presence, this.expected_input_col_num, + &this.json2_rewrite_targets, ))) } } @@ -140,6 +182,7 @@ fn align_projected_batch( output_schema: &SchemaRef, projected_root_presence: &[bool], expected_input_col_num: usize, + json2_rewrite_targets: &HashMap, ) -> Result { ensure!( rb.columns().len() == expected_input_col_num, @@ -154,6 +197,7 @@ fn align_projected_batch( let mut cols = Vec::with_capacity(projected_root_presence.len()); let mut idx = 0; + let input_schema = rb.schema_ref(); for (field, present) in output_schema.fields().iter().zip(projected_root_presence) { if !present { @@ -161,19 +205,39 @@ fn align_projected_batch( continue; } - cols.push(align_array(rb.column(idx), field)?); + cols.push(align_array( + rb.column(idx), + input_schema.field(idx), + field, + json2_rewrite_targets.get(field.name()), + )?); idx += 1; } RecordBatch::try_new(output_schema.clone(), cols).context(NewRecordBatchSnafu) } -fn align_array(array: &ArrayRef, field: &FieldRef) -> Result { +fn align_array( + array: &ArrayRef, + source: &Field, + field: &FieldRef, + rewrite_settings: Option<&Json2RewriteSettings>, +) -> Result { + if let Some(settings) = rewrite_settings { + return JsonArray::from(array) + .rewrite_to_v2(source, &settings.logical_settings, &settings.target_layout) + .context(DataTypeMismatchSnafu); + } if array.data_type() == field.data_type() { return Ok(array.clone()); } if is_json2_extension_type(field) { + if is_json2_extension_type(source) { + return JsonArray::from(array) + .project_to_v2(source, field.data_type()) + .context(DataTypeMismatchSnafu); + } return JsonArray::from(array) .project_to(field.data_type()) .context(DataTypeMismatchSnafu); @@ -188,6 +252,7 @@ fn align_array(array: &ArrayRef, field: &FieldRef) -> Result { #[cfg(test)] mod tests { + use std::collections::HashMap; use std::sync::Arc; use datatypes::arrow::array::{ @@ -200,6 +265,33 @@ mod tests { use super::*; + #[test] + fn test_aligner_resolves_json2_rewrite_settings() + -> std::result::Result<(), Box> { + let logical_settings = JsonSettings::default(); + let target_layout = JsonSettings::try_new(vec![], Some(0))?; + let rewrite_targets = HashMap::from([( + "j".to_string(), + Json2TargetLayout { + extension_metadata: serde_json::to_string(&JsonMetadata::new( + logical_settings.clone(), + ))?, + target_layout: target_layout.clone(), + }, + )]); + let aligner = NestedSchemaAligner::new( + stream::empty::>(), + vec![], + schema(Vec::::new()), + )? + .with_json2_rewrite_targets(&rewrite_targets)?; + + let settings = &aligner.json2_rewrite_targets["j"]; + assert_eq!(logical_settings, settings.logical_settings); + assert_eq!(target_layout, settings.target_layout); + Ok(()) + } + #[tokio::test] async fn test_aligner_with_all_projected_roots_match() { let output_schema = schema([ diff --git a/src/mito2/src/sst/parquet/read_columns.rs b/src/mito2/src/sst/parquet/read_columns.rs index 6f632aee30..e6a5ab9fab 100644 --- a/src/mito2/src/sst/parquet/read_columns.rs +++ b/src/mito2/src/sst/parquet/read_columns.rs @@ -14,6 +14,7 @@ use std::collections::{HashMap, HashSet}; +use datatypes::extension::json::JSON2_REMAINDER_FIELD_NAME; use parquet::arrow::ProjectionMask; use parquet::basic::{ConvertedType, Type as PhysicalType}; use parquet::schema::types::{ColumnDescriptor, SchemaDescriptor}; @@ -179,7 +180,7 @@ pub struct ProjectionMaskPlan { /// returned plan keeps `k` in the projection mask and marks `j` as /// not present in the output, so it can be synthesized during /// post-processing. -pub fn build_projection_plan( +pub(crate) fn build_projection_plan( parquet_read_cols: &ParquetReadColumns, parquet_schema_desc: &SchemaDescriptor, ) -> ProjectionMaskPlan { @@ -261,24 +262,38 @@ fn build_parquet_leaves_indices( } } - // Then fallback prefix misses to their nearest variant parent. + // Then include v2 remainder leaves or fallback prefix misses to their nearest variant parent. // TODO(fys): Gate fallback planning on the root being JSON2. A raw Binary // leaf is a JSONB variant only under a JSON2 root; plain struct Binary // children should not enter this fallback path. for col in &projection.cols { - for (path_idx, nested_path) in col.nested_paths.iter().enumerate() { - if prefix_matched[&col.root_index][path_idx] { + let path_matches = &prefix_matched[&col.root_index]; + let mut needs_remainder = false; + for (matched, nested_path) in path_matches.iter().zip(&col.nested_paths) { + if *matched { + if !needs_remainder { + needs_remainder = + path_points_to_struct(parquet_schema_desc, col.root_index, nested_path); + } continue; } - let Some(leaf_idx) = + if let Some(leaf_idx) = find_nearest_variant_parent(parquet_schema_desc, col.root_index, nested_path) - else { - continue; - }; + { + matched_leaves.insert(leaf_idx); + matched_roots.insert(col.root_index); + } else { + needs_remainder = true; + } + } - matched_leaves.insert(leaf_idx); - matched_roots.insert(col.root_index); + if needs_remainder { + let remainder_leaves = find_remainder_leaves(parquet_schema_desc, col.root_index); + if !remainder_leaves.is_empty() { + matched_leaves.extend(remainder_leaves); + matched_roots.insert(col.root_index); + } } } @@ -287,6 +302,49 @@ fn build_parquet_leaves_indices( (matched_leaves, matched_roots) } +/// Returns whether a nested path points to an explicitly materialized object. +/// +/// JSON2 v2 can split an object's children between its Struct field and the remainder, +/// so reading the Struct leaves alone may produce an incomplete object. +fn path_points_to_struct( + parquet_schema_desc: &SchemaDescriptor, + root_idx: usize, + path: &[String], +) -> bool { + let Some(mut field) = parquet_schema_desc.root_schema().get_fields().get(root_idx) else { + return false; + }; + for name in path.iter().skip(1) { + if !field.is_group() { + return false; + } + let Some(child) = field.get_fields().iter().find(|field| field.name() == name) else { + return false; + }; + field = child; + } + field.is_group() +} + +/// Finds the Parquet leaves backing a JSON2 v2 remainder field. +/// +/// The remainder is a sibling of explicitly materialized fields, so prefix matching a +/// requested path cannot find it. These leaves are needed when an explicit path is absent +/// or an explicitly materialized object may have additional children in the remainder. +fn find_remainder_leaves(parquet_schema_desc: &SchemaDescriptor, root_idx: usize) -> Vec { + parquet_schema_desc + .columns() + .iter() + .enumerate() + .filter_map(|(i, column)| { + let path = column.path().parts(); + (parquet_schema_desc.get_column_root_idx(i) == root_idx + && path.get(1).is_some_and(|x| x == JSON2_REMAINDER_FIELD_NAME)) + .then_some(i) + }) + .collect::>() +} + fn find_nearest_variant_parent( parquet_schema_desc: &SchemaDescriptor, root_idx: usize, @@ -329,6 +387,7 @@ mod tests { use std::sync::Arc; use parquet::basic::{ConvertedType, LogicalType, Repetition}; + use parquet::errors::ParquetError; use parquet::schema::types::Type; use super::*; @@ -441,6 +500,76 @@ mod tests { ); } + #[test] + fn test_v2_routes_missing_path_to_remainder() -> Result<(), ParquetError> { + let parquet = build_test_v2_schema()?; + let projection = + ParquetReadColumns::from_deduped(vec![ParquetReadColumn::new(0).with_nested_paths( + vec![ + vec!["j".to_string(), "cold".to_string()], + vec!["j".to_string(), "another".to_string()], + ], + )]); + + let plan = build_projection_plan(&projection, &parquet); + + assert_eq!(vec![true], plan.projected_root_presence); + assert_eq!(ProjectionMask::leaves(&parquet, [0, 1]), plan.mask); + Ok(()) + } + + #[test] + fn test_v2_explicit_path_does_not_read_remainder() -> Result<(), ParquetError> { + let parquet = build_test_v2_schema()?; + let projection = ParquetReadColumns::from_deduped(vec![ + ParquetReadColumn::new(0) + .with_nested_paths(vec![vec!["j".to_string(), "hot".to_string()]]), + ]); + + let plan = build_projection_plan(&projection, &parquet); + + assert_eq!(vec![true], plan.projected_root_presence); + assert_eq!(ProjectionMask::leaves(&parquet, [3]), plan.mask); + Ok(()) + } + + #[test] + fn test_v2_container_path_reads_remainder() -> Result<(), ParquetError> { + let parquet = build_test_v2_schema()?; + let projection = ParquetReadColumns::from_deduped(vec![ + ParquetReadColumn::new(0) + .with_nested_paths(vec![vec!["j".to_string(), "commit".to_string()]]), + ]); + + let plan = build_projection_plan(&projection, &parquet); + + assert_eq!(vec![true], plan.projected_root_presence); + assert_eq!(ProjectionMask::leaves(&parquet, [0, 1, 2]), plan.mask); + Ok(()) + } + + // A nested path under an explicit Variant is stored entirely in that Variant parent. The + // remainder may preserve `opaque: null`, but it cannot contain `opaque.leaf`, so reading the + // nearest Variant parent is sufficient. + #[test] + fn test_v2_variant_parent_path_reads_parent() -> Result<(), ParquetError> { + let parquet = build_test_v2_schema()?; + let projection = + ParquetReadColumns::from_deduped(vec![ParquetReadColumn::new(0).with_nested_paths( + vec![vec![ + "j".to_string(), + "opaque".to_string(), + "leaf".to_string(), + ]], + )]); + + let plan = build_projection_plan(&projection, &parquet); + + assert_eq!(vec![true], plan.projected_root_presence); + assert_eq!(ProjectionMask::leaves(&parquet, [4]), plan.mask); + Ok(()) + } + #[test] fn test_merges_mixed_paths() { let parquet_schema_desc = build_test_nested_parquet_schema(); @@ -673,6 +802,63 @@ mod tests { SchemaDescriptor::new(schema) } + fn build_test_v2_schema() -> Result { + let metadata = Arc::new( + Type::primitive_type_builder("metadata", parquet::basic::Type::BYTE_ARRAY) + .with_repetition(Repetition::REQUIRED) + .build()?, + ); + let value = Arc::new( + Type::primitive_type_builder("value", parquet::basic::Type::BYTE_ARRAY) + .with_repetition(Repetition::REQUIRED) + .build()?, + ); + let remainder = Arc::new( + Type::group_type_builder(JSON2_REMAINDER_FIELD_NAME) + .with_repetition(Repetition::OPTIONAL) + .with_logical_type(Some(LogicalType::Variant { + specification_version: None, + })) + .with_fields(vec![metadata, value]) + .build()?, + ); + let operation = Arc::new( + Type::primitive_type_builder("operation", parquet::basic::Type::INT64) + .with_repetition(Repetition::OPTIONAL) + .build()?, + ); + let commit = Arc::new( + Type::group_type_builder("commit") + .with_repetition(Repetition::OPTIONAL) + .with_fields(vec![operation]) + .build()?, + ); + let hot = Arc::new( + Type::primitive_type_builder("hot", parquet::basic::Type::INT64) + .with_repetition(Repetition::OPTIONAL) + .build()?, + ); + // Normally there are no other explicit Variant fields exist if a remainder field is present. + // However, when structured values reach JSON2_MAX_STRUCTURED_DEPTH, there are. `opaque` + // models such a deep leaf without building a deeply nested test schema. + let opaque = Arc::new( + Type::primitive_type_builder("opaque", parquet::basic::Type::BYTE_ARRAY) + .with_repetition(Repetition::OPTIONAL) + .build()?, + ); + let root = Arc::new( + Type::group_type_builder("j") + .with_repetition(Repetition::OPTIONAL) + .with_fields(vec![remainder, commit, hot, opaque]) + .build()?, + ); + Ok(SchemaDescriptor::new(Arc::new( + Type::group_type_builder("schema") + .with_fields(vec![root]) + .build()?, + ))) + } + // Test schema: // schema // `- j diff --git a/src/mito2/src/sst/parquet/reader.rs b/src/mito2/src/sst/parquet/reader.rs index a189e2b405..e38b5a00e0 100644 --- a/src/mito2/src/sst/parquet/reader.rs +++ b/src/mito2/src/sst/parquet/reader.rs @@ -16,13 +16,16 @@ #[cfg(feature = "vector_index")] use std::collections::BTreeSet; -use std::collections::HashSet; +use std::collections::{HashMap, HashSet}; use std::sync::Arc; use std::time::{Duration, Instant}; use api::v1::SemanticType; +use arrow_schema::extension::{ + EXTENSION_TYPE_METADATA_KEY, EXTENSION_TYPE_NAME_KEY, ExtensionType, +}; use common_recordbatch::filter::SimpleFilterEvaluator; -use common_telemetry::{error, tracing, warn}; +use common_telemetry::{debug, error, tracing, warn}; use datafusion::physical_plan::PhysicalExpr; use datafusion_common::tree_node::{TreeNode, TreeNodeRecursion}; use datafusion_expr::utils::expr_to_columns; @@ -31,8 +34,9 @@ use datatypes::arrow::array::ArrayRef; use datatypes::arrow::datatypes::{Field, Schema as ArrowSchema, SchemaRef}; use datatypes::arrow::record_batch::RecordBatch; use datatypes::data_type::ConcreteDataType; -use datatypes::extension::json::is_json2_extension_type; +use datatypes::extension::json::{Json2ExtensionType, is_json2_extension_type}; use datatypes::prelude::DataType; +use datatypes::vectors::json::json2_physical_data_type; use futures::StreamExt; use mito_codec::row_converter::build_primary_key_codec; use object_store::ObjectStore; @@ -76,7 +80,6 @@ use crate::sst::index::inverted_index::applier::{ }; #[cfg(feature = "vector_index")] use crate::sst::index::vector_index::applier::VectorIndexApplierRef; -use crate::sst::parquet::DEFAULT_READ_BATCH_SIZE; use crate::sst::parquet::file_range::{ FileRangeContext, FileRangeContextRef, PartitionFilterContext, PreFilterMode, RangeBase, }; @@ -94,6 +97,7 @@ use crate::sst::parquet::read_columns::{ProjectionMaskPlan, build_projection_pla use crate::sst::parquet::row_group::ParquetFetchMetrics; use crate::sst::parquet::row_selection::RowGroupSelection; use crate::sst::parquet::stats::RowGroupPruningStats; +use crate::sst::parquet::{DEFAULT_READ_BATCH_SIZE, Json2RewriteTargets, Json2TargetLayout}; use crate::sst::{override_pk_field_to_binary, tag_maybe_to_dictionary_field}; const INDEX_TYPE_FULLTEXT: &str = "fulltext"; @@ -108,6 +112,43 @@ fn should_read_pk_as_binary(parquet_meta: &ParquetMetaData) -> bool { should_read_pk_as_binary_with_limit(parquet_meta, DEFAULT_DICTIONARY_PAGE_SIZE_LIMIT) } +fn apply_json2_rewrite_targets( + read_format: &FlatReadFormat, + targets: &Json2RewriteTargets, +) -> Result { + let schema = read_format.output_arrow_schema()?; + if targets.is_empty() { + return Ok(schema); + } + + let mut schema = schema.as_ref().clone(); + let mut fields = schema.fields().iter().cloned().collect::>(); + for (column_id, layout) in targets.iter() { + let Some(index) = read_format.parquet_projected_index_by_id(*column_id) else { + continue; + }; + let Some(field) = fields.get(index) else { + continue; + }; + let mut field = field.as_ref().clone(); + field.set_data_type(json2_physical_data_type(&layout.target_layout)); + + let mut metadata = field.metadata().clone(); + metadata.insert( + EXTENSION_TYPE_NAME_KEY.to_string(), + Json2ExtensionType::NAME.to_string(), + ); + metadata.insert( + EXTENSION_TYPE_METADATA_KEY.to_string(), + layout.extension_metadata.clone(), + ); + field.set_metadata(metadata); + fields[index] = Arc::new(field); + } + schema.fields = fields.into(); + Ok(Arc::new(schema)) +} + fn should_read_pk_as_binary_with_limit( parquet_meta: &ParquetMetaData, dict_page_size_limit: usize, @@ -163,6 +204,8 @@ pub struct ParquetReaderBuilder { /// `None` reads all columns. Due to schema change, the projection /// can contain columns not in the parquet file. read_cols: Option, + /// Compaction-only JSON2 physical rewrite targets. + json2_rewrite_targets: Json2RewriteTargets, /// Strategy to cache SST data. cache_strategy: CacheStrategy, /// Index appliers. @@ -206,6 +249,7 @@ impl ParquetReaderBuilder { object_store, predicate: None, read_cols: None, + json2_rewrite_targets: Arc::default(), cache_strategy: CacheStrategy::Disabled, inverted_index_appliers: [None, None], bloom_filter_index_appliers: [None, None], @@ -247,6 +291,13 @@ impl ParquetReaderBuilder { self } + /// Attaches fixed JSON2 physical layouts used by compaction readers. + #[must_use] + pub(crate) fn json2_rewrite_targets(mut self, targets: Json2RewriteTargets) -> Self { + self.json2_rewrite_targets = targets; + self + } + /// Attaches the cache to the builder. #[must_use] pub fn cache(mut self, cache: CacheStrategy) -> ParquetReaderBuilder { @@ -461,7 +512,24 @@ impl ParquetReaderBuilder { &file_path, skip_auto_convert, )?; - if need_override_sequence(&parquet_meta) { + // `region_meta` comes from the Parquet/source file and must not be used as the + // target identity. When the caller has no current metadata, the handle is the + // only local identity available and therefore denotes a local read. + let expected_region_id = self + .expected_metadata + .as_ref() + .map(|metadata| metadata.region_id) + .unwrap_or(self.file_handle.region_id()); + let is_foreign = self.file_handle.region_id() != expected_region_id; + if is_foreign { + debug!( + "Reading foreign SST, file_id: {}, source_region_id: {}, expected_region_id: {}", + self.file_handle.file_id().file_id(), + self.file_handle.region_id(), + expected_region_id, + ); + } + if is_foreign || need_override_sequence(&parquet_meta) { read_format .set_override_sequence(self.file_handle.meta_ref().sequence.map(|x| x.get())); } @@ -535,6 +603,8 @@ impl ParquetReaderBuilder { ); } + let output_schema = apply_json2_rewrite_targets(&read_format, &self.json2_rewrite_targets)?; + // Create ArrowReaderMetadata for async stream building. let mut arrow_reader_options = ArrowReaderOptions::new(); if !read_format @@ -546,7 +616,7 @@ impl ParquetReaderBuilder { // Read `__primary_key` as Binary when it's too large for dictionary // encoding; convert_batch wraps it back to a DictionaryArray. let schema_for_reader = if should_read_pk_as_binary(&parquet_meta) { - read_format.set_pk_as_binary()?; + read_format.set_pk_as_binary(output_schema.clone()); override_pk_field_to_binary(read_format.arrow_schema()) } else { read_format.arrow_schema().clone() @@ -557,7 +627,18 @@ impl ParquetReaderBuilder { ArrowReaderMetadata::try_new(parquet_meta.clone(), arrow_reader_options) .context(ReadDataPartSnafu)?; - let output_schema = read_format.output_arrow_schema()?; + let json2_rewrite_targets = self + .json2_rewrite_targets + .iter() + .map(|(column_id, layout)| { + let column = region_meta + .column_by_id(*column_id) + .context(UnexpectedSnafu { + reason: format!("JSON2 target column by id {column_id} does not exist"), + })?; + Ok((column.column_schema.name.clone(), layout.clone())) + }) + .collect::>>()?; let reader_builder = RowGroupReaderBuilder { file_handle: self.file_handle.clone(), @@ -566,6 +647,7 @@ impl ParquetReaderBuilder { parquet_metadata_size, arrow_metadata, output_schema, + json2_rewrite_targets, object_store: self.object_store.clone(), projection: projection_plan, has_nested_projection, @@ -1786,6 +1868,8 @@ pub(crate) struct RowGroupReaderBuilder { arrow_metadata: ArrowReaderMetadata, /// Projected output schema aligned with `projection.projected_root_presence`. output_schema: SchemaRef, + /// JSON2 columns that must be semantically rewritten into the projected target layout. + json2_rewrite_targets: HashMap, /// Object store as an Operator. object_store: ObjectStore, /// Projection mask. @@ -1955,7 +2039,7 @@ impl RowGroupReaderBuilder { &self, stream: ProjectedRecordBatchStream, ) -> Result { - if !self.has_nested_projection { + if !self.has_nested_projection && self.json2_rewrite_targets.is_empty() { return Ok(stream); } @@ -1964,6 +2048,7 @@ impl RowGroupReaderBuilder { self.projection.projected_root_presence.clone(), self.output_schema.clone(), )? + .with_json2_rewrite_targets(&self.json2_rewrite_targets)? .boxed()) } diff --git a/src/mito2/src/sst/parquet/writer.rs b/src/mito2/src/sst/parquet/writer.rs index a6a4cd5cd0..b9d5016173 100644 --- a/src/mito2/src/sst/parquet/writer.rs +++ b/src/mito2/src/sst/parquet/writer.rs @@ -23,11 +23,12 @@ use std::sync::atomic::{AtomicUsize, Ordering}; use std::task::{Context, Poll}; use std::time::Instant; +use bytes::Bytes; use common_telemetry::debug; use common_time::Timestamp; use datatypes::arrow::array::{ - ArrayRef, TimestampMicrosecondArray, TimestampMillisecondArray, TimestampNanosecondArray, - TimestampSecondArray, + ArrayRef, BinaryArray, TimestampMicrosecondArray, TimestampMillisecondArray, + TimestampNanosecondArray, TimestampSecondArray, UInt32Array, }; use datatypes::arrow::compute::{max, min}; use datatypes::arrow::datatypes::{DataType, SchemaRef, TimeUnit}; @@ -40,7 +41,7 @@ use parquet::file::metadata::KeyValue; use parquet::file::properties::{WriterProperties, WriterPropertiesBuilder}; use parquet::schema::types::ColumnPath; use smallvec::smallvec; -use snafu::ResultExt; +use snafu::{OptionExt, ResultExt}; use store_api::metadata::RegionMetadataRef; use store_api::storage::consts::{OP_TYPE_COLUMN_NAME, SEQUENCE_COLUMN_NAME}; use store_api::storage::{FileId, SequenceNumber}; @@ -50,13 +51,16 @@ use tokio_util::compat::{Compat, FuturesAsyncWriteCompatExt}; use crate::access_layer::{FilePathProvider, Metrics, SstInfoArray, TempFileCleaner}; use crate::config::{IndexBuildMode, IndexConfig}; use crate::error::{ - InvalidMetadataSnafu, OpenDalSnafu, Result, UnexpectedSnafu, WriteParquetSnafu, + InvalidMetadataSnafu, InvalidRecordBatchSnafu, OpenDalSnafu, Result, UnexpectedSnafu, + WriteParquetSnafu, }; use crate::read::FlatSource; use crate::sst::file::RegionFileId; use crate::sst::index::{IndexOutput, Indexer, IndexerBuilder}; -use crate::sst::parquet::flat_format::{FlatWriteFormat, time_index_column_index}; -use crate::sst::parquet::format::PrimaryKeyWriteFormat; +use crate::sst::parquet::flat_format::{ + FlatWriteFormat, primary_key_column_index, time_index_column_index, +}; +use crate::sst::parquet::format::{PrimaryKeyArray, PrimaryKeyWriteFormat}; use crate::sst::parquet::{PARQUET_METADATA_KEY, SstInfo, WriteOptions}; use crate::sst::{ DEFAULT_WRITE_BUFFER_SIZE, DEFAULT_WRITE_CONCURRENCY, FlatSchemaOptions, SeriesEstimator, @@ -85,6 +89,71 @@ impl FlatBatchConverter { } } +/// Result of splitting a batch at the next series boundary. +enum SeriesBoundarySplit { + /// The whole batch belongs to the current series and stays in the current file. + Continue(RecordBatch), + /// The batch crosses a series boundary. + Split { + /// Remaining rows of the current series, appended to the current file. + current_file_tail: Option, + /// Rows of the following series that start the next file. + next_file_head: RecordBatch, + }, +} + +/// Returns the dictionary keys and values of the encoded `__primary_key` column. +fn encoded_primary_keys(batch: &RecordBatch) -> Result<(&UInt32Array, &BinaryArray)> { + let column = batch.column(primary_key_column_index(batch.num_columns())); + let primary_keys = column + .as_any() + .downcast_ref::() + .with_context(|| InvalidRecordBatchSnafu { + reason: format!( + "expected dictionary primary key column, got {:?}", + column.data_type() + ), + })?; + let values = primary_keys + .values() + .as_any() + .downcast_ref::() + .with_context(|| InvalidRecordBatchSnafu { + reason: format!( + "expected binary primary key values, got {:?}", + primary_keys.values().data_type() + ), + })?; + Ok((primary_keys.keys(), values)) +} + +/// Splits `batch` at the first row whose encoded primary key differs from +/// `current_primary_key`, the last primary key written to the current file. +/// +/// The writer finishes an oversized file at such a series boundary so that the +/// primary key ranges of output files never overlap, which allows pickers like +/// TWCS to detect overlapping files by their time and primary key ranges. A +/// series is never split in the middle: if no row starts a new series, the whole +/// batch stays in the current file even if it already exceeds the size limit. +fn split_at_next_series( + batch: RecordBatch, + current_primary_key: &[u8], +) -> Result { + let (keys, values) = encoded_primary_keys(&batch)?; + let Some(offset) = (0..batch.num_rows()) + .find(|&row| values.value(keys.value(row) as usize) != current_primary_key) + else { + return Ok(SeriesBoundarySplit::Continue(batch)); + }; + + let next_file_head = batch.slice(offset, batch.num_rows() - offset); + let current_file_tail = (offset > 0).then(|| batch.slice(0, offset)); + Ok(SeriesBoundarySplit::Split { + current_file_tail, + next_file_head, + }) +} + /// Parquet SST writer. pub struct ParquetWriter<'a, F: WriterFactory, I: IndexerBuilder, P: FilePathProvider> { /// Path provider that creates SST and index file paths according to file id. @@ -269,6 +338,13 @@ where /// Iterates FlatSource and writes all RecordBatch in flat format to Parquet file. /// + /// The source must yield batches globally sorted by the encoded primary key. + /// `opts.max_file_size` is a soft limit for regions with a primary key: the + /// current file is finished at the next series boundary after exceeding the + /// limit (see [split_at_next_series]), so a series larger than the limit + /// stays in one file. Regions without a primary key are split at batch + /// boundaries. + /// /// Returns the [SstInfo] if the SST is written. pub async fn write_all_flat( &mut self, @@ -304,6 +380,9 @@ where /// Iterates FlatSource and writes all RecordBatch in primary-key format to Parquet file. /// + /// The source must yield batches globally sorted by the encoded primary key. + /// See [Self::write_all_flat] for the `opts.max_file_size` semantics. + /// /// Returns the [SstInfo] if the SST is written. pub async fn write_all_flat_as_primary_key( &mut self, @@ -336,36 +415,55 @@ where let mut results = smallvec![]; let mut stats = SourceStats::default(); - while let Some(record_batch) = self - .write_next_flat_batch(&mut source, converter, opts) - .await - .transpose() - { - match record_batch { - Ok(batch) => { - stats.update_flat(&batch)?; - if matches!(self.index_config.build_mode, IndexBuildMode::Sync) { - let start = Instant::now(); - // safety: self.current_indexer must be set when first batch has been written. - self.current_indexer - .as_mut() - .unwrap() - .update_flat(&batch) - .await; - self.metrics.update_index += start.elapsed(); - } - if let Some(max_file_size) = opts.max_file_size - && self.bytes_written.load(Ordering::Relaxed) > max_file_size - { - self.finish_current_file(&mut results, &mut stats).await?; - } - } + loop { + let start = Instant::now(); + let batch = match source.next_batch().await { + Ok(Some(batch)) => batch, + Ok(None) => break, Err(e) => { - if let Some(indexer) = &mut self.current_indexer { - indexer.abort().await; - } + self.abort_current_indexer().await; return Err(e); } + }; + self.metrics.iter_source += start.elapsed(); + + if self.metadata.primary_key.is_empty() { + self.append_flat_batch(&batch, converter, opts, &mut stats) + .await?; + if self.exceeds_max_file_size(opts) { + self.finish_current_file(&mut results, &mut stats).await?; + } + } else if self.exceeds_max_file_size(opts) + && let Some(current_primary_key) = stats.last_primary_key.as_deref() + { + let series_split = match split_at_next_series(batch, current_primary_key) { + Ok(series_split) => series_split, + Err(e) => { + self.abort_current_indexer().await; + return Err(e); + } + }; + match series_split { + SeriesBoundarySplit::Continue(batch) => { + self.append_flat_batch(&batch, converter, opts, &mut stats) + .await?; + } + SeriesBoundarySplit::Split { + current_file_tail, + next_file_head, + } => { + if let Some(tail) = current_file_tail { + self.append_flat_batch(&tail, converter, opts, &mut stats) + .await?; + } + self.finish_current_file(&mut results, &mut stats).await?; + self.append_flat_batch(&next_file_head, converter, opts, &mut stats) + .await?; + } + } + } else { + self.append_flat_batch(&batch, converter, opts, &mut stats) + .await?; } } @@ -398,29 +496,53 @@ where .set_column_compression(op_type_col, Compression::UNCOMPRESSED) } - async fn write_next_flat_batch( + async fn append_flat_batch( &mut self, - source: &mut FlatSource, + batch: &RecordBatch, converter: &FlatBatchConverter, opts: &WriteOptions, - ) -> Result> { - let start = Instant::now(); - let Some(record_batch) = source.next_batch().await? else { - return Ok(None); - }; - self.metrics.iter_source += start.elapsed(); + stats: &mut SourceStats, + ) -> Result<()> { + let result = async { + let arrow_batch = converter.convert_batch(batch)?; + let start = Instant::now(); + self.maybe_init_writer(arrow_batch.schema_ref(), opts) + .await? + .write(&arrow_batch) + .await + .context(WriteParquetSnafu)?; + self.metrics.write_batch += start.elapsed(); - let arrow_batch = converter.convert_batch(&record_batch)?; + stats.update_flat(batch)?; + if matches!(self.index_config.build_mode, IndexBuildMode::Sync) { + let start = Instant::now(); + // safety: self.current_indexer must be set when first batch has been written. + self.current_indexer + .as_mut() + .unwrap() + .update_flat(batch) + .await; + self.metrics.update_index += start.elapsed(); + } + Ok(()) + } + .await; - let start = Instant::now(); - self.maybe_init_writer(arrow_batch.schema_ref(), opts) - .await? - .write(&arrow_batch) - .await - .context(WriteParquetSnafu)?; - self.metrics.write_batch += start.elapsed(); - // Return original flat batch for stats/indexer which use flat layout. - Ok(Some(record_batch)) + if result.is_err() { + self.abort_current_indexer().await; + } + result + } + + fn exceeds_max_file_size(&self, opts: &WriteOptions) -> bool { + opts.max_file_size + .is_some_and(|max_size| self.bytes_written.load(Ordering::Relaxed) >= max_size) + } + + async fn abort_current_indexer(&mut self) { + if let Some(indexer) = &mut self.current_indexer { + indexer.abort().await; + } } async fn maybe_init_writer( @@ -481,6 +603,8 @@ struct SourceStats { num_rows: usize, /// Time range of fetched batches. time_range: Option<(Timestamp, Timestamp)>, + /// Last primary key written to the current file. + last_primary_key: Option, /// Series estimator for computing num_series. series_estimator: SeriesEstimator, } @@ -493,6 +617,9 @@ impl SourceStats { self.num_rows += record_batch.num_rows(); self.series_estimator.update_flat(record_batch); + let (keys, values) = encoded_primary_keys(record_batch)?; + let key = keys.value(record_batch.num_rows() - 1); + self.last_primary_key = Some(Bytes::copy_from_slice(values.value(key as usize))); // Get the timestamp column by index let time_index_col_idx = time_index_column_index(record_batch.num_columns()); diff --git a/src/mito2/src/test_util.rs b/src/mito2/src/test_util.rs index 8a63d749b5..5bf55edb07 100644 --- a/src/mito2/src/test_util.rs +++ b/src/mito2/src/test_util.rs @@ -52,7 +52,9 @@ use log_store::raft_engine::log_store::RaftEngineLogStore; use log_store::test_util::log_store_util; use moka::future::CacheBuilder; use object_store::ObjectStore; -use object_store::layers::mock::MockLayer; +use object_store::layers::mock::{ + Buffer, Deleter, Metadata, MockLayer, MockLayerBuilder, OpDelete, Result as MockResult, Writer, +}; use object_store::manager::{ObjectStoreManager, ObjectStoreManagerRef}; use object_store::services::Fs; use rskafka::client::partition::{Compression, UnknownTopicHandling}; @@ -67,6 +69,7 @@ use store_api::region_request::{ RegionOpenRequest, RegionPutRequest, RegionRequest, }; use store_api::storage::{ColumnId, RegionId}; +use tokio::sync::Notify; use crate::cache::write_cache::{WriteCache, WriteCacheRef}; use crate::config::MitoConfig; @@ -75,6 +78,7 @@ use crate::engine::{MITO_ENGINE_NAME, MitoEngine}; use crate::error::Result; use crate::flush::{WriteBufferManager, WriteBufferManagerRef}; use crate::manifest::manager::{RegionManifestManager, RegionManifestOptions}; +use crate::manifest::storage::{is_checkpoint_file, is_delta_file}; use crate::read::{Batch, BatchBuilder, BatchReader}; use crate::region::opener::{PartitionExprFetcher, PartitionExprFetcherRef}; use crate::sst::FormatType; @@ -85,6 +89,133 @@ use crate::sst::index::puffin_manager::PuffinManagerFactory; use crate::time_provider::{StdTimeProvider, TimeProviderRef}; use crate::worker::WorkerGroup; +/// Controls a mock object-store layer that blocks a checkpoint task once. +#[derive(Clone)] +pub(crate) struct CheckpointTaskBlocker { + entered: Arc, + release: Arc, + armed: Arc, +} + +impl CheckpointTaskBlocker { + /// Blocks normal manifest checkpoint cleanup at batch-delete close. + pub(crate) fn block_cleanup() -> (Self, MockLayer) { + let blocker = Self { + entered: Arc::new(Notify::new()), + release: Arc::new(Notify::new()), + armed: Arc::new(AtomicBool::new(true)), + }; + let factory_blocker = blocker.clone(); + let layer = MockLayerBuilder::default() + .deleter_factory(Arc::new(move |inner| { + Box::new(BlockingCheckpointDeleter { + inner, + blocker: factory_blocker.clone(), + has_manifest_cleanup_target: false, + }) + })) + .build() + .unwrap(); + (blocker, layer) + } + + /// Blocks publication of `_last_checkpoint` at writer close. + pub(crate) fn block_last_checkpoint_write() -> (Self, MockLayer) { + let blocker = Self { + entered: Arc::new(Notify::new()), + release: Arc::new(Notify::new()), + armed: Arc::new(AtomicBool::new(true)), + }; + let factory_blocker = blocker.clone(); + let layer = MockLayerBuilder::default() + .writer_factory(Arc::new(move |path, _args, inner| { + Box::new(BlockingCheckpointWriter { + path: path.to_string(), + inner, + blocker: factory_blocker.clone(), + }) + })) + .build() + .unwrap(); + (blocker, layer) + } + + pub(crate) async fn wait_until_blocked(&self) { + self.entered.notified().await; + } + + pub(crate) fn release(&self) { + self.release.notify_one(); + } + + pub(crate) fn arm_next_close(&self) { + self.armed.store(true, Ordering::Release); + } + + async fn block_once(&self) { + if self.armed.swap(false, Ordering::AcqRel) { + self.entered.notify_one(); + self.release.notified().await; + } + } +} + +struct BlockingCheckpointDeleter { + inner: Deleter, + blocker: CheckpointTaskBlocker, + has_manifest_cleanup_target: bool, +} + +impl object_store::layers::mock::Delete for BlockingCheckpointDeleter { + async fn delete(&mut self, path: &str, args: OpDelete) -> MockResult<()> { + self.inner.delete(path, args).await?; + if is_manifest_checkpoint_file(path) { + self.has_manifest_cleanup_target = true; + } + Ok(()) + } + + async fn close(&mut self) -> MockResult<()> { + if self.has_manifest_cleanup_target { + self.blocker.block_once().await; + } + self.inner.close().await + } +} + +fn is_manifest_checkpoint_file(path: &str) -> bool { + // The mock deleter receives paths relative to the listed manifest + // directory, so the normal/staging directory segments are unavailable. + let path = Path::new(path); + let Some(file_name) = path.file_name().and_then(|name| name.to_str()) else { + return false; + }; + is_delta_file(file_name) || is_checkpoint_file(file_name) +} + +struct BlockingCheckpointWriter { + path: String, + inner: Writer, + blocker: CheckpointTaskBlocker, +} + +impl object_store::layers::mock::Write for BlockingCheckpointWriter { + async fn write(&mut self, bs: Buffer) -> MockResult<()> { + self.inner.write(bs).await + } + + async fn close(&mut self) -> MockResult { + if self.path.ends_with("_last_checkpoint") { + self.blocker.block_once().await; + } + self.inner.close().await + } + + async fn abort(&mut self) -> MockResult<()> { + self.inner.abort().await + } +} + pub(crate) fn new_noop_file_purger() -> FilePurgerRef { Arc::new(NoopFilePurger) } diff --git a/src/mito2/src/worker/handle_enter_staging.rs b/src/mito2/src/worker/handle_enter_staging.rs index 000525021d..f9e2454124 100644 --- a/src/mito2/src/worker/handle_enter_staging.rs +++ b/src/mito2/src/worker/handle_enter_staging.rs @@ -126,6 +126,10 @@ impl RegionWorkerLoop { // First step: clear all staging manifest files. { let mut manager = region.manifest_ctx.manifest_manager.write().await; + // `set_entering_staging` has already stopped new normal manifest + // publications. Wait for an older checkpoint to finish cleanup + // before exposing staging to repartition remap readers. + manager.wait_for_pending_checkpoint().await; manager .clear_staging_manifest_and_dir() .await diff --git a/src/mito2/src/worker/handle_rebuild_index.rs b/src/mito2/src/worker/handle_rebuild_index.rs index ad0a09e1d3..3eb45b1ab2 100644 --- a/src/mito2/src/worker/handle_rebuild_index.rs +++ b/src/mito2/src/worker/handle_rebuild_index.rs @@ -79,6 +79,7 @@ impl RegionWorkerLoop { IndexBuildTask { region_id: region.region_id, file: file.clone(), + target_region_metadata: version.metadata.clone(), source: IndexBuildSource::new(file_meta, version.metadata.schema_version), reason: build_type, access_layer: access_layer.clone(), diff --git a/src/mito2/test-data/json2-v2.parquet b/src/mito2/test-data/json2-v2.parquet new file mode 100644 index 0000000000..a3cfe8f185 Binary files /dev/null and b/src/mito2/test-data/json2-v2.parquet differ diff --git a/src/operator/src/req_convert/insert/stmt_to_region.rs b/src/operator/src/req_convert/insert/stmt_to_region.rs index 39a30c61b1..e395584d07 100644 --- a/src/operator/src/req_convert/insert/stmt_to_region.rs +++ b/src/operator/src/req_convert/insert/stmt_to_region.rs @@ -31,8 +31,8 @@ use table::metadata::TableInfoRef; use crate::error::{ CatalogSnafu, ColumnDataTypeSnafu, ColumnDefaultValueSnafu, ColumnNoneDefaultValueSnafu, - ColumnNotFoundSnafu, InvalidInsertRequestSnafu, InvalidSqlSnafu, MissingInsertBodySnafu, - ParseSqlSnafu, Result, SchemaReadOnlySnafu, TableNotFoundSnafu, + ColumnNotFoundSnafu, InvalidSqlSnafu, MissingInsertBodySnafu, ParseSqlSnafu, Result, + SchemaReadOnlySnafu, TableNotFoundSnafu, }; use crate::insert::InstantAndNormalInsertRequests; use crate::req_convert::common::partitioner::Partitioner; @@ -73,7 +73,6 @@ impl<'a> StatementToRegion<'a> { !common_catalog::consts::is_readonly_schema(&schema), SchemaReadOnlySnafu { name: schema } ); - let column_names = column_names(stmt, &table_schema); let column_count = column_names.len(); @@ -280,27 +279,9 @@ fn sql_value_to_value( ) .context(crate::error::SqlCommonSnafu)? }; - validate(&value)?; Ok(value) } -fn validate(value: &Value) -> Result<()> { - match value { - Value::Json(value) => { - // Json object will be stored as Arrow struct in parquet, and it has the restriction: - // "Parquet does not support writing empty structs". - ensure!( - !value.is_empty_object(), - InvalidInsertRequestSnafu { - reason: "empty json object is not supported, consider adding a dummy field" - } - ); - Ok(()) - } - _ => Ok(()), - } -} - fn replace_default(sql_val: &SqlValue) -> bool { matches!(sql_val, SqlValue::Placeholder(s) if s.to_lowercase() == DEFAULT_PLACEHOLDER_VALUE) } diff --git a/src/operator/src/statement.rs b/src/operator/src/statement.rs index 1143b1c68c..9be041435d 100644 --- a/src/operator/src/statement.rs +++ b/src/operator/src/statement.rs @@ -83,7 +83,7 @@ use table::table_reference::TableReference; pub use self::admin::{ AdminEventRecorderHandle, AdminFunctionLayer, AdminFunctionLayerRef, AdminFunctionRecordingLayer, AdminFunctionRequest, AdminFunctionResponse, AdminFunctionService, - AdminFunctionServiceRef, + AdminFunctionServiceRef, admin_output_schema, }; use self::set::{ set_bytea_output, set_datestyle, set_intervalstyle, set_timezone, validate_client_encoding, diff --git a/src/operator/src/statement/admin.rs b/src/operator/src/statement/admin.rs index 2b4a909e3c..ff8b26915c 100644 --- a/src/operator/src/statement/admin.rs +++ b/src/operator/src/statement/admin.rs @@ -19,6 +19,7 @@ use std::sync::Arc; use common_function::function::FunctionContext; use common_function::function_registry::{FUNCTION_REGISTRY, get_admin_function}; +use common_function::state::FunctionState; use common_query::Output; use common_recordbatch::{RecordBatch, RecordBatches}; use common_sql::convert::sql_value_to_value; @@ -77,6 +78,103 @@ struct CoreAdminFunctionService { query_engine: query::QueryEngineRef, } +/// Parts of an `ADMIN` call needed both for execution and schema derivation. +struct ResolvedAdminFunction { + admin_udf: datafusion_expr::ScalarUDF, + fn_name: String, + args: Vec, + arg_types: Vec, + ret_type: ArrowDataType, +} + +/// Resolves the function, parses its literal arguments and derives its +/// return type, without executing it. +fn resolve_admin_function( + stmt: &Admin, + query_ctx: &QueryContextRef, + state: Arc, +) -> Result { + let Admin::Func(func) = stmt; + // the function name should be in lower case. + let func_name = func.name.to_string().to_lowercase(); + let factory = get_admin_function(&func_name) + .or_else(|| FUNCTION_REGISTRY.get_function(&func_name)) + .context(error::AdminFunctionNotFoundSnafu { + name: func_name.clone(), + })?; + + let func_ctx = FunctionContext { + query_ctx: query_ctx.clone(), + state, + }; + + let admin_udf = factory.provide(func_ctx); + admin_udf + .as_async() + .context(error::AdminFunctionNotFoundSnafu { name: func_name })?; + + let fn_name = admin_udf.name().to_string(); + let signature = admin_udf.signature(); + + // Parse function arguments + let FunctionArguments::List(args) = &func.args else { + return error::BuildAdminFunctionArgsSnafu { + msg: format!("unsupported function args {} for {}", func.args, fn_name), + } + .fail(); + }; + let arg_values = args + .args + .iter() + .map(|arg| { + let FunctionArg::Unnamed(FunctionArgExpr::Expr(Expr::Value(value))) = arg else { + return error::BuildAdminFunctionArgsSnafu { + msg: format!("unsupported function arg {arg} for {}", fn_name), + } + .fail(); + }; + Ok(&value.value) + }) + .collect::>>()?; + + let args = args_to_vector(&signature.type_signature, &arg_values, query_ctx)?; + let arg_types = args + .iter() + .map(|arg| arg.data_type().as_arrow_type()) + .collect::>(); + let ret_type = + admin_udf + .return_type(&arg_types) + .map_err(|e| error::Error::BuildAdminFunctionArgs { + msg: format!( + "Failed to get return type of admin function {}: {}", + fn_name, e + ), + })?; + + Ok(ResolvedAdminFunction { + admin_udf, + fn_name, + args, + arg_types, + ret_type, + }) +} + +/// Output schema of an `ADMIN` statement, mirroring what +/// [`CoreAdminFunctionService::execute`] produces. `None` if the function +/// or arguments are unresolvable; execution will then surface the error. +pub fn admin_output_schema(stmt: &Admin, query_ctx: &QueryContextRef) -> Option { + let resolved = + resolve_admin_function(stmt, query_ctx, Arc::new(FunctionState::default())).ok()?; + Some(Schema::new(vec![ColumnSchema::new( + // Use statement as the result column name + stmt.to_string(), + ConcreteDataType::from_arrow_type(&resolved.ret_type), + false, + )])) +} + impl CoreAdminFunctionService { fn new(query_engine: query::QueryEngineRef) -> Self { Self { query_engine } @@ -88,62 +186,23 @@ impl CoreAdminFunctionService { query_ctx, } = request; - let Admin::Func(func) = &stmt; - // the function name should be in lower case. - let func_name = func.name.to_string().to_lowercase(); - let factory = get_admin_function(&func_name) - .or_else(|| FUNCTION_REGISTRY.get_function(&func_name)) - .context(error::AdminFunctionNotFoundSnafu { - name: func_name.clone(), - })?; - - let func_ctx = FunctionContext { - query_ctx: query_ctx.clone(), - state: self.query_engine.engine_state().function_state(), - }; - - let admin_udf = factory.provide(func_ctx); + let resolved = resolve_admin_function( + &stmt, + &query_ctx, + self.query_engine.engine_state().function_state(), + )?; + let ResolvedAdminFunction { + admin_udf, + fn_name, + args, + arg_types, + ret_type, + } = resolved; let admin_async_fn = admin_udf .as_async() - .context(error::AdminFunctionNotFoundSnafu { name: func_name })?; - - let fn_name = admin_udf.name(); - let signature = admin_udf.signature(); - - // Parse function arguments - let FunctionArguments::List(args) = &func.args else { - return error::BuildAdminFunctionArgsSnafu { - msg: format!("unsupported function args {} for {}", func.args, fn_name), - } - .fail(); - }; - let arg_values = args - .args - .iter() - .map(|arg| { - let FunctionArg::Unnamed(FunctionArgExpr::Expr(Expr::Value(value))) = arg else { - return error::BuildAdminFunctionArgsSnafu { - msg: format!("unsupported function arg {arg} for {}", fn_name), - } - .fail(); - }; - Ok(&value.value) - }) - .collect::>>()?; - - let args = args_to_vector(&signature.type_signature, &arg_values, &query_ctx)?; - let arg_types = args - .iter() - .map(|arg| arg.data_type().as_arrow_type()) - .collect::>(); - let ret_type = admin_udf.return_type(&arg_types).map_err(|e| { - error::Error::BuildAdminFunctionArgs { - msg: format!( - "Failed to get return type of admin function {}: {}", - fn_name, e - ), - } - })?; + .context(error::AdminFunctionNotFoundSnafu { + name: fn_name.clone(), + })?; // Convert arguments to DataFusion ColumnarValue format let columnar_args: Vec = args @@ -174,9 +233,7 @@ impl CoreAdminFunctionService { let result_columnar = admin_async_fn .invoke_async_with_args(func_args) .await - .with_context(|_| ExecuteAdminFunctionSnafu { - msg: fn_name.to_string(), - })?; + .with_context(|_| ExecuteAdminFunctionSnafu { msg: fn_name })?; // Convert result back to VectorRef let result_columnar: common_query::prelude::ColumnarValue = diff --git a/src/pipeline/Cargo.toml b/src/pipeline/Cargo.toml index d1a1f88ca8..497ce4542c 100644 --- a/src/pipeline/Cargo.toml +++ b/src/pipeline/Cargo.toml @@ -45,7 +45,7 @@ itertools.workspace = true jsonb.workspace = true jsonpath-rust = "0.7.5" lazy_static.workspace = true -moka = { workspace = true, features = ["sync"] } +moka = { workspace = true, features = ["future"] } once_cell.workspace = true operator.workspace = true ordered-float.workspace = true diff --git a/src/pipeline/src/error.rs b/src/pipeline/src/error.rs index 2da62dda2e..4d3bf453c5 100644 --- a/src/pipeline/src/error.rs +++ b/src/pipeline/src/error.rs @@ -389,6 +389,20 @@ pub enum Error { #[snafu(implicit)] location: Location, }, + #[snafu(display("Invalid JSON2 type hint: {reason}"))] + InvalidJson2TypeHint { + reason: String, + #[snafu(implicit)] + location: Location, + }, + #[snafu(display("Invalid JSON2 type hint path '{path}'"))] + ParseJson2TypeHintPath { + path: String, + #[snafu(source)] + source: sql::error::Error, + #[snafu(implicit)] + location: Location, + }, #[snafu(display("Transform index `type` must be set."))] TransformIndexTypeMustBeSet { #[snafu(implicit)] @@ -693,6 +707,15 @@ pub enum Error { source: common_recordbatch::error::Error, }, + /// `try_get_with` shares one loader across concurrent misses, so its error + /// arrives behind an `Arc`. + #[snafu(display("Failed to load pipeline into cache: {}", error))] + CacheLoad { + error: std::sync::Arc, + #[snafu(implicit)] + location: Location, + }, + #[snafu(display("A valid table suffix template is required for tablesuffix section"))] RequiredTableSuffixTemplate, @@ -881,6 +904,7 @@ impl ErrorExt for Error { fn status_code(&self) -> StatusCode { use Error::*; match self { + CacheLoad { error, .. } => error.status_code(), CastType { .. } => StatusCode::Unexpected, PipelineTableNotFound { .. } => StatusCode::TableNotFound, InsertPipeline { source, .. } => source.status_code(), @@ -952,6 +976,8 @@ impl ErrorExt for Error { | TransformElementMustBeMap { .. } | TransformFieldMustBeSet { .. } | TransformTypeMustBeSet { .. } + | InvalidJson2TypeHint { .. } + | ParseJson2TypeHintPath { .. } | TransformIndexTypeMustBeSet { .. } | TransformIndexUnsupportedField { .. } | TransformIndexOptionMustBeScalar { .. } diff --git a/src/pipeline/src/etl.rs b/src/pipeline/src/etl.rs index 26d3520829..f83b5c2626 100644 --- a/src/pipeline/src/etl.rs +++ b/src/pipeline/src/etl.rs @@ -37,7 +37,9 @@ use crate::error::{ YamlLoadSnafu, YamlParseSnafu, }; use crate::etl::processor::ProcessorKind; -use crate::etl::transform::transformer::greptime::{RowWithTableSuffix, values_to_rows}; +use crate::etl::transform::transformer::greptime::{ + RowWithTableSuffix, values_to_row, values_to_rows, +}; use crate::tablesuffix::TableSuffixTemplate; use crate::{ ContextOpt, GreptimeTransformer, IdentityTimeIndex, PipelineContext, SchemaInfo, @@ -236,6 +238,14 @@ pub enum PipelineExecOutput { Filtered, } +/// The result after processors and dispatcher rules have run. +#[derive(Debug)] +pub enum PipelineProcessOutput { + Processed(VrlValue), + DispatchedTo(DispatchedTo, VrlValue), + Filtered, +} + /// Output from a successful pipeline transformation. /// /// Rows are grouped by their ContextOpt, with each row having its own optional @@ -303,24 +313,42 @@ impl Pipeline { pub fn exec_mut( &self, - mut val: VrlValue, + val: VrlValue, pipeline_ctx: &PipelineContext<'_>, schema_info: &mut SchemaInfo, ) -> Result { - // process + match self.process_mut(val)? { + PipelineProcessOutput::Processed(val) => self + .transform_mut(val, pipeline_ctx, schema_info) + .map(PipelineExecOutput::Transformed), + PipelineProcessOutput::DispatchedTo(dispatched_to, val) => { + Ok(PipelineExecOutput::DispatchedTo(dispatched_to, val)) + } + PipelineProcessOutput::Filtered => Ok(PipelineExecOutput::Filtered), + } + } + + pub fn process_mut(&self, mut val: VrlValue) -> Result { for processor in self.processors.iter() { val = processor.exec_mut(val)?; if val.is_null() { - // line is filtered - return Ok(PipelineExecOutput::Filtered); + return Ok(PipelineProcessOutput::Filtered); } } - // dispatch, fast return if matched if let Some(rule) = self.dispatcher.as_ref().and_then(|d| d.exec(&val)) { - return Ok(PipelineExecOutput::DispatchedTo(rule.into(), val)); + return Ok(PipelineProcessOutput::DispatchedTo(rule.into(), val)); } + Ok(PipelineProcessOutput::Processed(val)) + } + + pub fn transform_mut( + &self, + val: VrlValue, + pipeline_ctx: &PipelineContext<'_>, + schema_info: &mut SchemaInfo, + ) -> Result { let mut val = if val.is_array() { val } else { @@ -356,9 +384,7 @@ impl Pipeline { } }; - Ok(PipelineExecOutput::Transformed(TransformedOutput { - rows_by_context, - })) + Ok(TransformedOutput { rows_by_context }) } pub fn processors(&self) -> &processor::Processors { @@ -369,6 +395,10 @@ impl Pipeline { &self.transformer } + pub fn resolve_table_suffix(&self, value: &VrlValue) -> Option { + ContextOpt::resolve_table_suffix(self.tablesuffix.as_ref(), value) + } + // the method is for test purpose pub fn schemas(&self) -> Option<&Vec> { match &self.transformer { @@ -409,34 +439,43 @@ fn transform_array_elements_by_ctx( ); } - let values = - unwrap_or_continue_if_err!(transformer.transform_mut(element, is_v1), skip_error); + let table_suffix = ContextOpt::resolve_table_suffix(tablesuffix_template, element); + let values = unwrap_or_continue_if_err!( + transformer.transform_mut_with_schema( + element, + is_v1, + schema_info, + table_suffix.as_deref(), + ), + skip_error + ); if is_v1 { // v1 mode: just use transformer output directly - let mut opt = unwrap_or_continue_if_err!( + let opt = unwrap_or_continue_if_err!( ContextOpt::from_pipeline_map_to_opt(element), skip_error ); - let table_suffix = opt.resolve_table_suffix(tablesuffix_template, element); rows_by_context .entry(opt) .or_insert_with(Vec::new) .push((Row { values }, table_suffix)); } else { // v2 mode: combine with auto-transform for remaining fields - let element_rows_map = values_to_rows( - schema_info, - element.clone(), - pipeline_ctx, - Some(values), - false, - tablesuffix_template, - ) - .map_err(Box::new) - .context(TransformArrayElementSnafu { index })?; - for (k, v) in element_rows_map { - rows_by_context.entry(k).or_default().extend(v); - } + let mut value = element.clone(); + let opt = unwrap_or_continue_if_err!( + ContextOpt::from_pipeline_map_to_opt(&mut value), + skip_error + ); + let row = unwrap_or_continue_if_err!( + values_to_row(schema_info, value, pipeline_ctx, Some(values), false,) + .map_err(Box::new) + .context(TransformArrayElementSnafu { index }), + skip_error + ); + rows_by_context + .entry(opt) + .or_default() + .push((row, table_suffix)); } } diff --git a/src/pipeline/src/etl/ctx_req.rs b/src/pipeline/src/etl/ctx_req.rs index 737e6147b2..86d160314f 100644 --- a/src/pipeline/src/etl/ctx_req.rs +++ b/src/pipeline/src/etl/ctx_req.rs @@ -69,10 +69,6 @@ pub struct ContextOpt { // reset the schema in query context schema: Option, - - // pipeline options, not set in query context - // can be removed before the end of the pipeline execution - table_suffix: Option, } impl ContextOpt { @@ -112,9 +108,7 @@ impl ContextOpt { GREPTIME_SKIP_WAL => { opt.skip_wal = Some(v); } - GREPTIME_TABLE_SUFFIX => { - opt.table_suffix = Some(v); - } + GREPTIME_TABLE_SUFFIX => {} _ => {} } } @@ -123,12 +117,13 @@ impl ContextOpt { } pub(crate) fn resolve_table_suffix( - &mut self, table_suffix: Option<&TableSuffixTemplate>, pipeline_map: &VrlValue, ) -> Option { - self.table_suffix - .take() + pipeline_map + .as_object() + .and_then(|map| map.get(GREPTIME_TABLE_SUFFIX)) + .map(|value| value.to_string_lossy().to_string()) .or_else(|| table_suffix.and_then(|s| s.apply(pipeline_map))) } diff --git a/src/pipeline/src/etl/transform.rs b/src/pipeline/src/etl/transform.rs index efcb4765ad..c9a1510735 100644 --- a/src/pipeline/src/etl/transform.rs +++ b/src/pipeline/src/etl/transform.rs @@ -17,17 +17,21 @@ pub mod transformer; use std::collections::HashMap; +use api::helper::ColumnDataTypeWrapper; use api::v1::ColumnDataType; use api::v1::value::ValueData; use chrono::Utc; -use datatypes::schema::{FulltextOptions, SkippingIndexOptions}; +use datatypes::json::{JsonSettings, JsonTypeHint}; +use datatypes::schema::{ColumnDefaultConstraint, FulltextOptions, SkippingIndexOptions}; +use datatypes::value::Value; use snafu::{OptionExt, ResultExt, ensure}; use sql::parsers::utils::{ validate_column_fulltext_create_option, validate_column_skipping_index_create_option, }; use crate::error::{ - Error, FieldMustBeTypeSnafu, KeyMustBeStringSnafu, Result, TransformElementMustBeMapSnafu, + Error, FieldMustBeTypeSnafu, InvalidJson2TypeHintSnafu, KeyMustBeStringSnafu, + ParseJson2TypeHintPathSnafu, Result, TransformElementMustBeMapSnafu, TransformFieldMustBeSetSnafu, TransformIndexOptionMustBeScalarSnafu, TransformIndexOptionSnafu, TransformIndexOptionUnsupportedSnafu, TransformIndexOptionsUnsupportedSnafu, TransformIndexTypeMismatchSnafu, TransformIndexTypeMustBeSetSnafu, @@ -49,6 +53,10 @@ const TRANSFORM_INDEX_OPTIONS_FIELD: &str = "index.options"; const TRANSFORM_TAG: &str = "tag"; const TRANSFORM_DEFAULT: &str = "default"; const TRANSFORM_ON_FAILURE: &str = "on_failure"; +const JSON2_TYPE: &str = "json2"; +const JSON2_TYPE_HINT: &str = "type.json2[]"; +const JSON2_TYPE_HINT_PATH: &str = "path"; +const JSON2_TYPE_HINT_NULLABLE: &str = "nullable"; pub use transformer::greptime::GreptimeTransformer; @@ -141,6 +149,7 @@ impl TryFrom<&Vec> for Transforms { pub struct Transform { pub fields: Fields, pub type_: ColumnDataType, + pub(crate) json_settings: Option, pub default: Option, pub index: Option, pub index_options: Option, @@ -196,7 +205,8 @@ impl TransformIndexOptions { // ColumnDataType::TimestampMicrosecond // ColumnDataType::TimestampMillisecond // ColumnDataType::TimestampSecond -// ColumnDataType::Binary +// ColumnDataType::Binary (JSONB) +// ColumnDataType::Json (JSON2) impl Transform { pub(crate) fn get_default(&self) -> Option<&ValueData> { @@ -260,6 +270,7 @@ fn get_default_for_type(ty: &ColumnDataType) -> Result { ColumnDataType::Float32 => ValueData::F32Value(0.0), ColumnDataType::Float64 => ValueData::F64Value(0.0), ColumnDataType::Binary => ValueData::BinaryValue(jsonb::Value::Null.to_vec()), + ColumnDataType::Json => ValueData::JsonValue(Default::default()), ColumnDataType::String => ValueData::StringValue(String::new()), ColumnDataType::TimestampSecond => ValueData::TimestampSecondValue(0), @@ -419,6 +430,162 @@ fn lower_transform_index_options( } } } + +fn parse_transform_type(value: &yaml_rust::Yaml) -> Result<(ColumnDataType, Option)> { + if let Some(type_name) = value.as_str() { + return Ok((parse_str_type(type_name)?, None)); + } + + let config = value.as_hash().context(FieldMustBeTypeSnafu { + field: TRANSFORM_TYPE, + ty: "string or map", + })?; + ensure!( + config.len() == 1, + InvalidJson2TypeHintSnafu { + reason: "transform type map must contain exactly one `json2` field".to_string() + } + ); + let (type_name, hints) = config.iter().next().context(InvalidJson2TypeHintSnafu { + reason: "transform type map must contain a `json2` field".to_string(), + })?; + let type_name = type_name.as_str().with_context(|| KeyMustBeStringSnafu { + k: type_name.clone(), + })?; + ensure!( + type_name.eq_ignore_ascii_case(JSON2_TYPE), + InvalidJson2TypeHintSnafu { + reason: format!("unsupported transform type map `{type_name}`") + } + ); + + let hints = hints.as_vec().context(FieldMustBeTypeSnafu { + field: JSON2_TYPE, + ty: "list", + })?; + let hints = hints + .iter() + .map(parse_json2_type_hint) + .collect::>>()?; + Ok(( + ColumnDataType::Json, + Some(JsonSettings::try_new(hints, None)?), + )) +} + +fn parse_json2_type_hint(value: &yaml_rust::Yaml) -> Result { + let config = value.as_hash().context(FieldMustBeTypeSnafu { + field: JSON2_TYPE_HINT, + ty: "map", + })?; + let mut path = None; + let mut type_name = None; + let mut nullable = true; + let mut default = None; + let mut index = None; + + for (key, value) in config { + let key = key + .as_str() + .with_context(|| KeyMustBeStringSnafu { k: key.clone() })?; + match key { + JSON2_TYPE_HINT_PATH => path = Some(yaml_string(value, JSON2_TYPE_HINT_PATH)?), + TRANSFORM_TYPE => type_name = Some(yaml_string(value, TRANSFORM_TYPE)?), + JSON2_TYPE_HINT_NULLABLE => { + nullable = yaml_bool(value, JSON2_TYPE_HINT_NULLABLE)?; + } + TRANSFORM_DEFAULT => default = Some(value), + TRANSFORM_INDEX => index = Some(value), + _ => { + return InvalidJson2TypeHintSnafu { + reason: format!("unsupported field `{key}`"), + } + .fail(); + } + } + } + + let path = path.context(InvalidJson2TypeHintSnafu { + reason: "`path` must be set".to_string(), + })?; + let path = sql::parse_json2_type_hint_path(&path) + .with_context(|_| ParseJson2TypeHintPathSnafu { path: path.clone() })?; + let type_name = type_name.context(InvalidJson2TypeHintSnafu { + reason: "`type` must be set".to_string(), + })?; + let type_ = parse_str_type(&type_name)?; + ensure!( + matches!( + type_, + ColumnDataType::String + | ColumnDataType::Int64 + | ColumnDataType::Uint64 + | ColumnDataType::Float64 + | ColumnDataType::Boolean + ), + InvalidJson2TypeHintSnafu { + reason: format!("unsupported type `{type_name}`") + } + ); + let data_type = ColumnDataTypeWrapper::new(type_, None).into(); + let default_constraint = default + .map(|value| parse_json2_type_hint_default(value, &type_)) + .transpose()?; + if let Some(default_constraint) = &default_constraint { + default_constraint.validate(&data_type, nullable)?; + } + + let inverted_index = if let Some(value) = index { + let (index, options) = parse_transform_index(value)?; + ensure!( + index == Index::Inverted, + InvalidJson2TypeHintSnafu { + reason: format!("unsupported index `{index}`") + } + ); + lower_transform_index_options(index, &ColumnDataType::Json, options)?; + true + } else { + false + }; + + Ok(JsonTypeHint { + path, + data_type, + nullable, + default_constraint, + inverted_index, + }) +} + +fn parse_json2_type_hint_default( + value: &yaml_rust::Yaml, + type_: &ColumnDataType, +) -> Result { + if value.is_null() { + return Ok(ColumnDefaultConstraint::Value(Value::Null)); + } + + let value = match value { + yaml_rust::Yaml::Real(value) | yaml_rust::Yaml::String(value) => value.clone(), + yaml_rust::Yaml::Integer(value) => value.to_string(), + yaml_rust::Yaml::Boolean(value) => value.to_string(), + _ => { + return FieldMustBeTypeSnafu { + field: TRANSFORM_DEFAULT, + ty: "scalar", + } + .fail(); + } + }; + let value = api::v1::Value { + value_data: Some(parse_str_value(type_, &value)?), + }; + Ok(ColumnDefaultConstraint::Value( + api::helper::pb_value_to_value_ref(&value, None).into(), + )) +} + impl TryFrom<&yaml_rust::yaml::Hash> for Transform { type Error = Error; @@ -430,6 +597,7 @@ impl TryFrom<&yaml_rust::yaml::Hash> for Transform { let mut on_failure = None; let mut type_ = None; + let mut json_settings = None; for (k, v) in hash { let key = k @@ -445,8 +613,9 @@ impl TryFrom<&yaml_rust::yaml::Hash> for Transform { } TRANSFORM_TYPE => { - let t = yaml_string(v, TRANSFORM_TYPE)?; - type_ = Some(parse_str_type(&t)?); + let (parsed_type, parsed_json_settings) = parse_transform_type(v)?; + type_ = Some(parsed_type); + json_settings = parsed_json_settings; } TRANSFORM_INDEX => { @@ -507,6 +676,7 @@ impl TryFrom<&yaml_rust::yaml::Hash> for Transform { let builder = Transform { fields, type_, + json_settings, default: final_default, index, index_options, @@ -529,6 +699,72 @@ mod tests { docs[0].as_hash().unwrap().try_into() } + #[test] + fn test_transform_parses_json2_type_hints() { + let transform = parse_transform( + r#" +field: payload +type: + json2: + - path: "user.id" + type: int64 + nullable: false + default: 7 + index: + type: inverted + - path: 'attrs."http.status_code"' + type: string +"#, + ) + .unwrap(); + + assert_eq!(transform.type_, ColumnDataType::Json); + let hints = transform.json_settings.as_ref().unwrap().type_hints(); + assert_eq!(hints.len(), 2); + assert_eq!(hints[0].path, ["user", "id"]); + assert_eq!( + hints[0].data_type, + datatypes::prelude::ConcreteDataType::int64_datatype() + ); + assert!(!hints[0].nullable); + assert_eq!( + hints[0].default_constraint, + Some(ColumnDefaultConstraint::Value(Value::Int64(7))) + ); + assert!(hints[0].inverted_index); + assert_eq!(hints[1].path, ["attrs", "http.status_code"]); + assert!(hints[1].nullable); + } + + #[test] + fn test_transform_rejects_non_finite_json2_default() { + for default in ["NaN", "1e9999"] { + let err = parse_transform(&format!( + r#" +field: payload +type: + json2: + - path: score + type: float64 + default: {default} +"#, + )) + .unwrap_err(); + + assert!( + matches!( + &err, + Error::Datatypes { + source: datatypes::error::Error::InvalidJson2Settings { .. }, + .. + } + ), + "{err:?}" + ); + assert!(err.to_string().contains("must be finite"), "{err}"); + } + } + #[test] fn test_transform_parses_legacy_string_index() { let transform = parse_transform( diff --git a/src/pipeline/src/etl/transform/transformer/greptime.rs b/src/pipeline/src/etl/transform/transformer/greptime.rs index 3292de749a..c85510e3e4 100644 --- a/src/pipeline/src/etl/transform/transformer/greptime.rs +++ b/src/pipeline/src/etl/transform/transformer/greptime.rs @@ -150,6 +150,7 @@ impl GreptimeTransformer { let transform = Transform { fields: Fields::one(Field::new(greptime_timestamp().to_string(), None)), type_, + json_settings: None, default, index: Some(Index::Time), index_options: None, @@ -225,6 +226,16 @@ impl GreptimeTransformer { &self, pipeline_map: &mut VrlValue, is_v1: bool, + ) -> Result> { + self.transform_mut_with_schema(pipeline_map, is_v1, &SchemaInfo::default(), None) + } + + pub(crate) fn transform_mut_with_schema( + &self, + pipeline_map: &mut VrlValue, + is_v1: bool, + schema_info: &SchemaInfo, + table_suffix: Option<&str>, ) -> Result> { let mut values = vec![GreptimeValue { value_data: None }; self.schema.len()]; let mut output_index = 0; @@ -236,7 +247,17 @@ impl GreptimeTransformer { // let keep us `get` here to be compatible with v1 match pipeline_map.get(column_name) { Some(v) => { - let value_data = coerce_value(v, transform)?; + let json_settings = if transform.type_ == ColumnDataType::Json + && matches!(v, VrlValue::Array(_) | VrlValue::Object(_)) + { + schema_info.json_settings_for_column( + field.target_or_input_field(), + table_suffix, + )? + } else { + None + }; + let value_data = coerce_value(v, transform, json_settings.as_ref())?; // every transform fields has only one output field values[output_index] = GreptimeValue { value_data }; } @@ -276,6 +297,12 @@ impl GreptimeTransformer { &self.schema } + pub fn has_json_transform(&self) -> bool { + self.transforms + .iter() + .any(|transform| transform.type_ == ColumnDataType::Json) + } + pub fn transforms_mut(&mut self) -> &mut Transforms { &mut self.transforms } @@ -285,6 +312,7 @@ impl GreptimeTransformer { pub struct ColumnMetadata { column_schema: datatypes::schema::ColumnSchema, semantic_type: SemanticType, + json_settings: OnceCell, } impl From for ColumnMetadata { @@ -311,10 +339,23 @@ impl From for ColumnMetadata { Self { column_schema, semantic_type, + json_settings: OnceCell::new(), } } } +impl ColumnMetadata { + fn json_settings(&self) -> Result<&JsonSettings> { + self.json_settings.get_or_try_init(|| { + if let Some(extension) = self.column_schema.extension_type::()? { + Ok(extension.metadata().json_settings().clone()) + } else { + Ok(parse_legacy_json2_settings(self.column_schema.metadata())?.unwrap_or_default()) + } + }) + } +} + impl TryFrom for ColumnSchema { type Error = api::error::Error; @@ -322,6 +363,7 @@ impl TryFrom for ColumnSchema { let ColumnMetadata { column_schema, semantic_type, + .. } = value; let options = options_from_column_schema(&column_schema); @@ -348,8 +390,8 @@ pub struct SchemaInfo { pub schema: Vec, /// index of the column name pub index: HashMap, - /// The pipeline's corresponding table (if already created). Useful to retrieve column schemas. - table: Option>, + /// Tables already looked up, keyed by their resolved suffix. Missing tables are cached as None. + tables: HashMap>>, } impl SchemaInfo { @@ -357,7 +399,7 @@ impl SchemaInfo { Self { schema: Vec::with_capacity(capacity), index: HashMap::with_capacity(capacity), - table: None, + tables: HashMap::new(), } } @@ -369,16 +411,33 @@ impl SchemaInfo { Self { schema: schema_list.into_iter().map(Into::into).collect(), index, - table: None, + tables: HashMap::new(), } } pub fn set_table(&mut self, table: Option>) { - self.table = table; + self.set_table_for_suffix(String::new(), table); + } + + pub fn has_table_for_suffix(&self, table_suffix: &str) -> bool { + self.tables.contains_key(table_suffix) + } + + pub fn set_table_for_suffix(&mut self, table_suffix: String, table: Option>) { + self.tables.insert(table_suffix, table); + } + + fn table_for_suffix(&self, table_suffix: Option<&str>) -> Option<&Arc> { + let table_suffix = table_suffix.unwrap_or_default(); + match self.tables.get(table_suffix) { + Some(table) => table.as_ref(), + None if !table_suffix.is_empty() => self.tables.get("").and_then(Option::as_ref), + None => None, + } } fn find_column_schema_in_table(&self, column_name: &str) -> Option { - if let Some(table) = &self.table + if let Some(table) = self.table_for_suffix(None) && let Some(i) = table.schema_ref().column_index_by_name(column_name) { let column_schema = table.schema_ref().column_schemas()[i].clone(); @@ -394,12 +453,37 @@ impl SchemaInfo { Some(ColumnMetadata { column_schema, semantic_type, + json_settings: OnceCell::new(), }) } else { None } } + fn json_settings_for_column( + &self, + column_name: &str, + table_suffix: Option<&str>, + ) -> Result> { + let Some(column_schema) = self + .table_for_suffix(table_suffix) + .and_then(|table| table.schema_ref().column_schema_by_name(column_name)) + else { + return Ok(None); + }; + if !column_schema.data_type.is_json2() { + return Ok(None); + } + + if let Some(extension) = column_schema.extension_type::()? { + Ok(Some(extension.metadata().json_settings().clone())) + } else { + Ok(Some( + parse_legacy_json2_settings(column_schema.metadata())?.unwrap_or_default(), + )) + } + } + pub fn column_schemas(&self) -> api::error::Result> { self.schema .iter() @@ -444,6 +528,7 @@ fn resolve_schema( ColumnMetadata { column_schema, semantic_type, + json_settings: OnceCell::new(), } }); let key = column.to_string(); @@ -499,12 +584,12 @@ pub(crate) fn values_to_rows( // Single object: extract ContextOpt and table_suffix let mut result = std::collections::HashMap::new(); - let mut opt = match ContextOpt::from_pipeline_map_to_opt(&mut values) { + let table_suffix = ContextOpt::resolve_table_suffix(tablesuffix_template, &values); + let opt = match ContextOpt::from_pipeline_map_to_opt(&mut values) { Ok(r) => r, Err(e) => return if skip_error { Ok(result) } else { Err(e) }, }; - let table_suffix = opt.resolve_table_suffix(tablesuffix_template, &values); let row = match values_to_row(schema_info, values, pipeline_ctx, row, need_calc_ts) { Ok(r) => r, Err(e) => return if skip_error { Ok(result) } else { Err(e) }, @@ -528,11 +613,11 @@ pub(crate) fn values_to_rows( } // Extract ContextOpt and table_suffix for this element - let mut opt = unwrap_or_continue_if_err!( + let table_suffix = ContextOpt::resolve_table_suffix(tablesuffix_template, &value); + let opt = unwrap_or_continue_if_err!( ContextOpt::from_pipeline_map_to_opt(&mut value), skip_error ); - let table_suffix = opt.resolve_table_suffix(tablesuffix_template, &value); let transformed_row = unwrap_or_continue_if_err!( values_to_row(schema_info, value, pipeline_ctx, row.clone(), need_calc_ts), skip_error @@ -691,38 +776,29 @@ fn resolve_value( } VrlValue::Array(_) | VrlValue::Object(_) => { - let is_json2 = schema_info - .find_column_schema_in_table(&column_name) - // TODO(LFC): Default to JSON2 for auto-created tables. - .is_some_and(|x| { - matches!( - &x.column_schema.data_type, - ConcreteDataType::Json(column_type) if column_type.is_json2() - ) - }); + let index = index.or_else(|| { + let column = schema_info.find_column_schema_in_table(&column_name)?; + let index = schema_info.schema.len(); + schema_info.schema.push(column); + schema_info.index.insert(column_name.clone(), index); + Some(index) + }); + // TODO(LFC): Default to JSON2 for auto-created tables. + let json2_index = index.filter(|&index| { + matches!( + &schema_info.schema[index].column_schema.data_type, + ConcreteDataType::Json(column_type) if column_type.is_json2() + ) + }); - let value = if is_json2 { + let value = if let Some(index) = json2_index { let value: serde_json::Value = value.try_into().map_err(|e: StdError| { CoerceIncompatibleTypesSnafu { msg: e.to_string() }.build() })?; - let value = - if let Some(column) = schema_info.find_column_schema_in_table(&column_name) { - if let Some(extension) = column - .column_schema - .extension_type::()? - { - extension.metadata().json_settings().encode(value)? - } else { - parse_legacy_json2_settings(column.column_schema.metadata())? - .unwrap_or_default() - .encode(value)? - } - } else { - JsonSettings::default().encode(value)? - }; + let value = schema_info.schema[index].json_settings()?.encode(value)?; resolve_schema( - index, + Some(index), p_ctx, &column_name, &ConcreteDataType::json2(Default::default()), @@ -810,6 +886,7 @@ fn identity_pipeline_inner( schema_info.schema.push(ColumnMetadata { column_schema, semantic_type: SemanticType::Timestamp, + json_settings: OnceCell::new(), }); let mut opt_map = HashMap::new(); @@ -907,7 +984,7 @@ pub fn flatten_object(object: VrlValue, max_nested_levels: usize) -> Result serde_json_crate::Value { +pub(crate) fn vrl_value_to_serde_json(value: &VrlValue) -> serde_json_crate::Value { match value { VrlValue::Null => serde_json_crate::Value::Null, VrlValue::Boolean(b) => serde_json_crate::Value::Bool(*b), @@ -980,10 +1057,125 @@ fn do_flatten_object( #[cfg(test)] mod tests { use api::v1::SemanticType; + use common_recordbatch::RecordBatch; + use datatypes::extension::json::JsonMetadata; + use datatypes::json::JsonTypeHint; + use datatypes::schema::{ColumnSchema as DatatypeColumnSchema, Schema}; + use table::test_util::MemTable; use super::*; use crate::{PipelineDefinition, identity_pipeline}; + #[test] + fn test_column_metadata_caches_json_settings() -> Result<()> { + let column = ColumnMetadata { + column_schema: datatypes::schema::ColumnSchema::new( + "data", + ConcreteDataType::json2(Default::default()), + true, + ), + semantic_type: SemanticType::Field, + json_settings: OnceCell::new(), + }; + + let first = column.json_settings()?; + let second = column.json_settings()?; + assert!(std::ptr::eq(first, second)); + Ok(()) + } + + #[test] + fn test_transform_json2_uses_destination_table_settings() { + let table = |name: &str, settings: JsonSettings, sample: serde_json::Value| { + let data_type = settings.encode(sample).unwrap().data_type(); + let mut column_schema = DatatypeColumnSchema::new("payload", data_type, true); + column_schema.with_extension_type(&Json2ExtensionType::new(Arc::new( + JsonMetadata::new(settings), + ))); + MemTable::table( + name, + RecordBatch::new_empty(Arc::new(Schema::new(vec![column_schema]))), + ) + }; + let int_settings = JsonSettings::try_new( + vec![JsonTypeHint { + path: vec!["age".to_string()], + data_type: ConcreteDataType::int64_datatype(), + nullable: false, + default_constraint: None, + inverted_index: false, + }], + None, + ) + .unwrap(); + let mut schema_info = SchemaInfo::default(); + schema_info.set_table(Some(table( + "events", + int_settings, + serde_json::json!({"age": 42}), + ))); + let string_settings = JsonSettings::try_new( + vec![JsonTypeHint { + path: vec!["age".to_string()], + data_type: ConcreteDataType::string_datatype(), + nullable: false, + default_constraint: None, + inverted_index: false, + }], + None, + ) + .unwrap(); + schema_info.set_table_for_suffix( + "_mobile".to_string(), + Some(table( + "events_mobile", + string_settings, + serde_json::json!({"age": "42"}), + )), + ); + + let pipeline = crate::parse(&crate::Content::Yaml( + r#" +transform: + - field: source, payload + type: json2 +table_suffix: _${device} +"#, + )) + .unwrap(); + let (pipeline, _, pipeline_definition, pipeline_params) = crate::setup_pipeline!(pipeline); + let pipeline_context = + PipelineContext::new(&pipeline_definition, &pipeline_params, Channel::Unknown); + + let error = pipeline + .exec_mut( + serde_json::json!({"source": {"age": "42"}}).into(), + &pipeline_context, + &mut schema_info, + ) + .unwrap_err(); + assert!( + error.to_string().contains("does not match JSON2 type hint"), + "{error:?}" + ); + + let mut rows = pipeline + .exec_mut( + serde_json::json!({"source": {"age": "42"}, "device": "mobile"}).into(), + &pipeline_context, + &mut schema_info, + ) + .unwrap() + .into_transformed() + .unwrap(); + let (row, table_suffix) = rows.swap_remove(0); + assert_eq!(table_suffix.as_deref(), Some("_mobile")); + assert!(matches!( + &row.values[0].value_data, + Some(ValueData::JsonValue(_)) + )); + } + #[test] fn test_identify_pipeline() { let params = GreptimePipelineParams::default(); diff --git a/src/pipeline/src/etl/transform/transformer/greptime/coerce.rs b/src/pipeline/src/etl/transform/transformer/greptime/coerce.rs index 5eaaa6e691..9f01af708b 100644 --- a/src/pipeline/src/etl/transform/transformer/greptime/coerce.rs +++ b/src/pipeline/src/etl/transform/transformer/greptime/coerce.rs @@ -12,10 +12,18 @@ // See the License for the specific language governing permissions and // limitations under the License. +use std::sync::Arc; + use api::v1::column_data_type_extension::TypeExt; use api::v1::column_def::{options_from_fulltext, options_from_inverted, options_from_skipping}; use api::v1::{ColumnDataTypeExtension, ColumnOptions, JsonTypeExtension}; +use arrow_schema::extension::{ + EXTENSION_TYPE_METADATA_KEY, EXTENSION_TYPE_NAME_KEY, ExtensionType, +}; +use datatypes::extension::json::{Json2ExtensionType, JsonMetadata}; +use datatypes::json::JsonSettings; use datatypes::schema::{FulltextOptions, SkippingIndexOptions}; +use datatypes::value::Value; use greptime_proto::v1::value::ValueData; use greptime_proto::v1::{ColumnDataType, ColumnSchema, SemanticType}; use snafu::{OptionExt, ResultExt, ensure}; @@ -28,7 +36,9 @@ use crate::error::{ UnsupportedTypeInPipelineSnafu, VrlRegexValueSnafu, }; use crate::etl::transform::index::Index; -use crate::etl::transform::transformer::greptime::vrl_value_to_jsonb_value; +use crate::etl::transform::transformer::greptime::{ + vrl_value_to_jsonb_value, vrl_value_to_serde_json, +}; use crate::etl::transform::{OnFailure, Transform, TransformIndexOptions}; pub(crate) fn coerce_columns(transform: &Transform) -> Result> { @@ -128,7 +138,7 @@ fn build_skipping_index_options(transform: &Transform) -> Result Result> { validate_transform_index_state(transform)?; - match transform.index { + let mut options = match transform.index { Some(Index::Fulltext) => { let options = build_fulltext_index_options(transform)?; options_from_fulltext(&options).context(ColumnOptionsSnafu) @@ -139,10 +149,32 @@ fn coerce_options(transform: &Transform) -> Result> { } Some(Index::Inverted) => Ok(Some(options_from_inverted())), _ => Ok(None), + }?; + + if transform.type_ == ColumnDataType::Json { + let extension = Json2ExtensionType::new(Arc::new(JsonMetadata::new( + transform.json_settings.clone().unwrap_or_default(), + ))); + let options = options.get_or_insert_default(); + options.options.insert( + EXTENSION_TYPE_NAME_KEY.to_string(), + Json2ExtensionType::NAME.to_string(), + ); + if let Some(metadata) = extension.serialize_metadata() { + options + .options + .insert(EXTENSION_TYPE_METADATA_KEY.to_string(), metadata); + } } + + Ok(options) } -pub(crate) fn coerce_value(val: &VrlValue, transform: &Transform) -> Result> { +pub(crate) fn coerce_value( + val: &VrlValue, + transform: &Transform, + json_settings: Option<&JsonSettings>, +) -> Result> { match val { VrlValue::Null => Ok(None), VrlValue::Integer(n) => coerce_i64_value(*n, transform), @@ -169,7 +201,9 @@ pub(crate) fn coerce_value(val: &VrlValue, transform: &Transform) -> Result coerce_json_value(val, transform), + VrlValue::Array(_) | VrlValue::Object(_) => { + coerce_json_value(val, transform, json_settings) + } VrlValue::Regex(_) => VrlRegexValueSnafu.fail(), } } @@ -205,7 +239,7 @@ fn coerce_bool_value(b: bool, transform: &Transform) -> Result } }, - ColumnDataType::Binary => { + ColumnDataType::Binary | ColumnDataType::Json => { return CoerceJsonTypeToSnafu { ty: transform.type_.as_str_name(), } @@ -294,7 +328,7 @@ fn coerce_i64_value(n: i64, transform: &Transform) -> Result> ColumnDataType::TimestampMillisecond => ValueData::TimestampMillisecondValue(n), ColumnDataType::TimestampSecond => ValueData::TimestampSecondValue(n), - ColumnDataType::Binary => { + ColumnDataType::Binary | ColumnDataType::Json => { return CoerceJsonTypeToSnafu { ty: transform.type_.as_str_name(), } @@ -363,7 +397,7 @@ fn coerce_u64_value(n: u64, transform: &Transform) -> Result> Err(_) => return integer_out_of_range(n, transform), }, - ColumnDataType::Binary => { + ColumnDataType::Binary | ColumnDataType::Json => { return CoerceJsonTypeToSnafu { ty: transform.type_.as_str_name(), } @@ -407,7 +441,7 @@ fn coerce_f64_value(n: f64, transform: &Transform) -> Result> } }, - ColumnDataType::Binary => { + ColumnDataType::Binary | ColumnDataType::Json => { return CoerceJsonTypeToSnafu { ty: transform.type_.as_str_name(), } @@ -486,7 +520,7 @@ fn coerce_string_value(s: &str, transform: &Transform) -> Result CoerceUnsupportedEpochTypeSnafu { ty: "String" }.fail(), }, - ColumnDataType::Binary => CoerceStringToTypeSnafu { + ColumnDataType::Binary | ColumnDataType::Json => CoerceStringToTypeSnafu { s, ty: transform.type_.as_str_name(), } @@ -496,23 +530,48 @@ fn coerce_string_value(s: &str, transform: &Transform) -> Result Result> { - match &transform.type_ { - ColumnDataType::Binary => (), +fn coerce_json_value( + v: &VrlValue, + transform: &Transform, + json_settings: Option<&JsonSettings>, +) -> Result> { + let value = match transform.type_ { + ColumnDataType::Binary => { + let data: jsonb::Value = vrl_value_to_jsonb_value(v); + ValueData::BinaryValue(data.to_vec()) + } + ColumnDataType::Json => { + let json = vrl_value_to_serde_json(v); + let encoded = if let Some(settings) = json_settings.or(transform.json_settings.as_ref()) + { + settings.encode(json) + } else { + JsonSettings::default().encode(json) + }; + let value = match encoded { + Ok(value) => value, + Err(error) => return handle_coercion_failure(transform, error.into()), + }; + let Value::Json(value) = value else { + unreachable!() + }; + ValueData::JsonValue(api::helper::encode_json_value(*value)) + } t => { return CoerceTypeToJsonSnafu { ty: t.as_str_name(), } .fail(); } - } - let data: jsonb::Value = vrl_value_to_jsonb_value(v); - Ok(Some(ValueData::BinaryValue(data.to_vec()))) + }; + Ok(Some(value)) } #[cfg(test)] mod tests { + use datatypes::data_type::ConcreteDataType; + use datatypes::json::JsonTypeHint; use datatypes::schema::{FulltextAnalyzer, FulltextBackend, SkippingIndexType}; use vrl::prelude::Bytes; @@ -523,6 +582,7 @@ mod tests { Transform { fields: Fields::default(), type_, + json_settings: None, default: None, index: None, index_options: None, @@ -666,6 +726,7 @@ mod tests { let transform = Transform { fields: Fields::default(), type_: ColumnDataType::Int32, + json_settings: None, default: None, index: None, index_options: None, @@ -676,14 +737,14 @@ mod tests { // valid string { let val = VrlValue::Integer(123); - let result = coerce_value(&val, &transform).unwrap(); + let result = coerce_value(&val, &transform, None).unwrap(); assert_eq!(result, Some(ValueData::I32Value(123))); } // invalid string { let val = VrlValue::Bytes(Bytes::from("hello")); - let result = coerce_value(&val, &transform); + let result = coerce_value(&val, &transform, None); assert!(result.is_err()); } } @@ -693,6 +754,7 @@ mod tests { let transform = Transform { fields: Fields::default(), type_: ColumnDataType::Int32, + json_settings: None, default: None, index: None, index_options: None, @@ -701,15 +763,43 @@ mod tests { }; let val = VrlValue::Bytes(Bytes::from("hello")); - let result = coerce_value(&val, &transform).unwrap(); + let result = coerce_value(&val, &transform, None).unwrap(); assert_eq!(result, None); } + #[test] + fn test_coerce_json2_with_on_failure() { + let settings = JsonSettings::try_new( + vec![JsonTypeHint { + path: vec!["age".to_string()], + data_type: ConcreteDataType::int64_datatype(), + nullable: false, + default_constraint: None, + inverted_index: false, + }], + None, + ) + .unwrap(); + let mut transform = transform(ColumnDataType::Json); + transform.json_settings = Some(settings); + transform.on_failure = Some(OnFailure::Ignore); + let value: VrlValue = serde_json::json!({"age": "42"}).into(); + + assert_eq!(coerce_value(&value, &transform, None).unwrap(), None); + + transform.on_failure = Some(OnFailure::Default); + assert_eq!( + coerce_value(&value, &transform, None).unwrap(), + Some(ValueData::JsonValue(Default::default())) + ); + } + #[test] fn test_coerce_string_with_on_failure_default() { let mut transform = Transform { fields: Fields::default(), type_: ColumnDataType::Int32, + json_settings: None, default: None, index: None, index_options: None, @@ -720,7 +810,7 @@ mod tests { // with no explicit default value { let val = VrlValue::Bytes(Bytes::from("hello")); - let result = coerce_value(&val, &transform).unwrap(); + let result = coerce_value(&val, &transform, None).unwrap(); assert_eq!(result, Some(ValueData::I32Value(0))); } @@ -728,7 +818,7 @@ mod tests { { transform.default = Some(ValueData::I32Value(42)); let val = VrlValue::Bytes(Bytes::from("hello")); - let result = coerce_value(&val, &transform).unwrap(); + let result = coerce_value(&val, &transform, None).unwrap(); assert_eq!(result, Some(ValueData::I32Value(42))); } } @@ -738,6 +828,7 @@ mod tests { let transform = Transform { fields: Fields::default(), type_: ColumnDataType::String, + json_settings: None, default: None, index: Some(Index::Fulltext), index_options: Some(TransformIndexOptions::Fulltext( @@ -769,6 +860,7 @@ mod tests { let transform = Transform { fields: Fields::default(), type_: ColumnDataType::Int64, + json_settings: None, default: None, index: Some(Index::Skipping), index_options: Some(TransformIndexOptions::Skipping( @@ -792,6 +884,7 @@ mod tests { let transform = Transform { fields: Fields::default(), type_: ColumnDataType::String, + json_settings: None, default: None, index: Some(Index::Fulltext), index_options: Some(TransformIndexOptions::Skipping( @@ -809,6 +902,7 @@ mod tests { let transform = Transform { fields: Fields::default(), type_: ColumnDataType::String, + json_settings: None, default: None, index: None, index_options: Some(TransformIndexOptions::Fulltext( diff --git a/src/pipeline/src/etl/value.rs b/src/pipeline/src/etl/value.rs index 2f5560b456..09d63058ff 100644 --- a/src/pipeline/src/etl/value.rs +++ b/src/pipeline/src/etl/value.rs @@ -102,6 +102,7 @@ pub fn parse_str_type(t: &str) -> Result { // We only consider object and array to be json types. and use Map to represent json // TODO(qtang): Needs to be defined with better semantics "json" => Ok(ColumnDataType::Binary), + "json2" => Ok(ColumnDataType::Json), _ => ValueParseTypeSnafu { t }.fail(), } diff --git a/src/pipeline/src/lib.rs b/src/pipeline/src/lib.rs index c657f61342..c3a45d4fe4 100644 --- a/src/pipeline/src/lib.rs +++ b/src/pipeline/src/lib.rs @@ -27,7 +27,8 @@ pub use etl::transform::GreptimeTransformer; pub use etl::transform::transformer::greptime::{GreptimePipelineParams, SchemaInfo}; pub use etl::transform::transformer::identity_pipeline; pub use etl::{ - Content, DispatchedTo, Pipeline, PipelineExecOutput, TransformedOutput, TransformerMode, parse, + Content, DispatchedTo, Pipeline, PipelineExecOutput, PipelineProcessOutput, TransformedOutput, + TransformerMode, parse, }; pub use manager::{ GREPTIME_INTERNAL_IDENTITY_PIPELINE_NAME, GREPTIME_INTERNAL_TRACE_PIPELINE_V1_NAME, diff --git a/src/pipeline/src/manager/pipeline_cache.rs b/src/pipeline/src/manager/pipeline_cache.rs index 2abe24e94b..98105d543f 100644 --- a/src/pipeline/src/manager/pipeline_cache.rs +++ b/src/pipeline/src/manager/pipeline_cache.rs @@ -12,14 +12,14 @@ // See the License for the specific language governing permissions and // limitations under the License. +use std::future::Future; use std::sync::Arc; use std::time::Duration; -use common_telemetry::debug; use datatypes::timestamp::TimestampNanosecond; -use moka::sync::Cache; +use moka::future::Cache; -use crate::error::{MultiPipelineWithDiffSchemaSnafu, Result}; +use crate::error::{CacheLoadSnafu, MultiPipelineWithDiffSchemaSnafu, Result}; use crate::etl::Pipeline; use crate::manager::PipelineVersion; use crate::table::EMPTY_SCHEMA_NAME; @@ -32,6 +32,11 @@ const PIPELINES_CACHE_TTL: Duration = Duration::from_secs(10); /// Pipeline cache is located on a separate file on purpose, /// to encapsulate inner cache. Only public methods are exposed. +/// +/// `pipelines` and `original_pipelines` are keyed by the *requested* schema so +/// a lookup is a single key probe, as [`Cache::try_get_with`] requires; +/// resolving it to a stored schema is the loader's job. `failover_cache` has no +/// loader and keeps the stored-schema key. pub(crate) struct PipelineCache { pipelines: Cache>, original_pipelines: Cache, @@ -68,163 +73,242 @@ impl PipelineCache { } } - pub(crate) fn insert_pipeline_cache( + /// Concurrent misses on the same key share one `init` call. + pub(crate) async fn get_pipeline_with( &self, schema: &str, name: &str, version: PipelineVersion, - pipeline: Arc, - with_latest: bool, - ) { - insert_cache_generic( - &self.pipelines, - schema, - name, - version, - pipeline.clone(), - with_latest, - ); + init: impl Future>>, + ) -> Result> { + let key = generate_pipeline_cache_key(schema, name, version); + self.pipelines + .try_get_with(key, init) + .await + .map_err(|error| CacheLoadSnafu { error }.build()) } - pub(crate) fn insert_pipeline_str_cache(&self, pipeline: &PipelineContent, with_latest: bool) { - let schema = pipeline.schema.as_str(); - let name = pipeline.name.as_str(); - let version = pipeline.version; - insert_cache_generic( - &self.original_pipelines, - schema, - name, - Some(version), - pipeline.clone(), - with_latest, - ); - insert_cache_generic( - &self.failover_cache, - schema, - name, - Some(version), - pipeline.clone(), - with_latest, - ); - } - - pub(crate) fn get_pipeline_cache( + /// Concurrent misses on the same key share one `init` call. + pub(crate) async fn get_pipeline_str_with( &self, schema: &str, name: &str, version: PipelineVersion, - ) -> Result>> { - get_cache_generic(&self.pipelines, schema, name, version) + init: impl Future>, + ) -> Result { + let key = generate_pipeline_cache_key(schema, name, version); + self.original_pipelines + .try_get_with(key, init) + .await + .map_err(|error| CacheLoadSnafu { error }.build()) } - pub(crate) fn get_failover_cache( + /// Resolves across schemas, unlike the loaded caches: a pipeline stored + /// under the empty schema is reachable from any schema. + pub(crate) async fn get_failover_cache( &self, schema: &str, name: &str, version: PipelineVersion, ) -> Result> { - get_cache_generic(&self.failover_cache, schema, name, version) - } + for key in [ + generate_pipeline_cache_key(EMPTY_SCHEMA_NAME, name, version), + generate_pipeline_cache_key(schema, name, version), + ] { + if let Some(content) = self.failover_cache.get(&key).await { + return Ok(Some(content)); + } + } - pub(crate) fn get_pipeline_str_cache( - &self, - schema: &str, - name: &str, - version: PipelineVersion, - ) -> Result> { - get_cache_generic(&self.original_pipelines, schema, name, version) - } - - // remove cache with version and latest in all schemas - pub(crate) fn remove_cache(&self, name: &str, version: PipelineVersion) { - let version_suffix = generate_pipeline_cache_key_suffix(name, version); - let latest_suffix = generate_pipeline_cache_key_suffix(name, None); - - let ks = self - .pipelines + // Stored under some other schema; unambiguous only if exactly one has it. + let suffix = generate_pipeline_cache_key_suffix(name, version); + let mut found = self + .failover_cache .iter() - .filter_map(|(k, _)| { - if k.ends_with(&version_suffix) || k.ends_with(&latest_suffix) { - Some(k.clone()) - } else { - None - } - }) + .filter(|(k, _)| k.ends_with(&suffix)) .collect::>(); - for k in ks { - let k = k.as_str(); - self.pipelines.remove(k); - self.original_pipelines.remove(k); - self.failover_cache.remove(k); - } - } -} - -fn insert_cache_generic( - cache: &Cache, - schema: &str, - name: &str, - version: PipelineVersion, - value: T, - with_latest: bool, -) { - let k = generate_pipeline_cache_key(schema, name, version); - cache.insert(k, value.clone()); - if with_latest { - let k = generate_pipeline_cache_key(schema, name, None); - cache.insert(k, value); - } -} - -fn get_cache_generic( - cache: &Cache, - schema: &str, - name: &str, - version: PipelineVersion, -) -> Result> { - // lets try empty schema first - let emp_key = generate_pipeline_cache_key(EMPTY_SCHEMA_NAME, name, version); - if let Some(value) = cache.get(&emp_key) { - return Ok(Some(value)); - } - // use input schema - let schema_k = generate_pipeline_cache_key(schema, name, version); - if let Some(value) = cache.get(&schema_k) { - return Ok(Some(value)); - } - - // try all schemas - let suffix_key = generate_pipeline_cache_key_suffix(name, version); - let mut ks = cache - .iter() - .filter(|e| e.0.ends_with(&suffix_key)) - .collect::>(); - - match ks.len() { - 0 => Ok(None), - 1 => { - let (_, value) = ks.remove(0); - Ok(Some(value)) - } - _ => { - debug!( - "caches keys: {:?}, emp key: {:?}, schema key: {:?}, suffix key: {:?}", - cache.iter().map(|e| e.0).collect::>(), - emp_key, - schema_k, - suffix_key - ); - MultiPipelineWithDiffSchemaSnafu { + match found.len() { + 0 => Ok(None), + 1 => Ok(Some(found.remove(0).1)), + _ => MultiPipelineWithDiffSchemaSnafu { name: name.to_string(), current_schema: schema.to_string(), - schemas: ks + schemas: found .iter() .filter_map(|(k, _)| k.split_once('/').map(|k| k.0)) .collect::>() .join(","), } - .fail()? + .fail(), + } + } + + pub(crate) async fn insert_failover_cache(&self, content: PipelineContent, with_latest: bool) { + let versioned = + generate_pipeline_cache_key(&content.schema, &content.name, Some(content.version)); + let latest = generate_pipeline_cache_key(&content.schema, &content.name, None); + + self.failover_cache.insert(versioned, content.clone()).await; + if with_latest { + self.failover_cache.insert(latest, content).await; + } + } + + /// Dropping the stale `latest` aliases also clears the failover entries, so + /// the new version is written back: an outage before the first read-back + /// would otherwise have nothing to fall back on. + pub(crate) async fn on_pipeline_created(&self, content: PipelineContent) { + self.invalidate(&content.name, None).await; + self.insert_failover_cache(content, true).await; + } + + /// Sweeps every schema and all three caches: the `latest` alias always, + /// plus `version` when given. + pub(crate) async fn invalidate(&self, name: &str, version: PipelineVersion) { + let mut suffixes = vec![generate_pipeline_cache_key_suffix(name, None)]; + if version.is_some() { + suffixes.push(generate_pipeline_cache_key_suffix(name, version)); + } + + let ks = self + .pipelines + .iter() + .map(|(k, _)| k) + .chain(self.original_pipelines.iter().map(|(k, _)| k)) + .chain(self.failover_cache.iter().map(|(k, _)| k)) + .filter(|k| suffixes.iter().any(|suffix| k.ends_with(suffix))) + .collect::>(); + + for k in ks { + let k = k.as_str(); + self.pipelines.invalidate(k).await; + self.original_pipelines.invalidate(k).await; + self.failover_cache.invalidate(k).await; } } } + +#[cfg(test)] +mod tests { + use std::sync::atomic::{AtomicUsize, Ordering}; + + use tokio::sync::Barrier; + + use super::*; + + /// Stored under the empty schema, i.e. visible from every schema. + fn content_at(version: i64) -> PipelineContent { + PipelineContent { + name: "p".to_string(), + content: "transform:".to_string(), + version: TimestampNanosecond::new(version), + schema: EMPTY_SCHEMA_NAME.to_string(), + } + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] + async fn test_concurrent_misses_run_one_loader() { + const CONCURRENCY: usize = 8; + + let cache = Arc::new(PipelineCache::new()); + let loads = Arc::new(AtomicUsize::new(0)); + let barrier = Arc::new(Barrier::new(CONCURRENCY)); + + let handles = (0..CONCURRENCY) + .map(|_| { + let (cache, loads, barrier) = (cache.clone(), loads.clone(), barrier.clone()); + tokio::spawn(async move { + barrier.wait().await; + cache + .get_pipeline_str_with("db", "p", None, async { + loads.fetch_add(1, Ordering::SeqCst); + // Hold the loader open so every caller is waiting on it. + tokio::time::sleep(Duration::from_millis(100)).await; + Ok(content_at(1)) + }) + .await + .unwrap() + }) + }) + .collect::>(); + + for handle in handles { + assert_eq!(handle.await.unwrap(), content_at(1)); + } + assert_eq!(loads.load(Ordering::SeqCst), 1); + } + + #[tokio::test] + async fn test_delete_drops_version_pinned_entry() { + let cache = PipelineCache::new(); + let content = content_at(1); + let version = Some(content.version); + + cache + .get_pipeline_str_with("db", "p", version, async { Ok(content.clone()) }) + .await + .unwrap(); + + cache.invalidate("p", version).await; + + let loads = AtomicUsize::new(0); + cache + .get_pipeline_str_with("db", "p", version, async { + loads.fetch_add(1, Ordering::SeqCst); + Ok(content.clone()) + }) + .await + .unwrap(); + assert_eq!(loads.load(Ordering::SeqCst), 1); + } + + #[tokio::test] + async fn test_create_drops_stale_latest_and_primes_failover() { + let cache = PipelineCache::new(); + let v2 = content_at(2); + + cache + .get_pipeline_str_with("a", "p", None, async { Ok(content_at(1)) }) + .await + .unwrap(); + cache.insert_failover_cache(content_at(1), true).await; + + cache.on_pipeline_created(v2.clone()).await; + + let loads = AtomicUsize::new(0); + let cached = cache + .get_pipeline_str_with("a", "p", None, async { + loads.fetch_add(1, Ordering::SeqCst); + Ok(v2.clone()) + }) + .await + .unwrap(); + assert_eq!(loads.load(Ordering::SeqCst), 1); + assert_eq!(cached.version, v2.version); + + let failover = cache.get_failover_cache("b", "p", None).await.unwrap(); + assert_eq!(failover.map(|c| c.version), Some(v2.version)); + } + + #[tokio::test] + async fn test_failover_serves_global_pipeline_to_unwarmed_schema() { + let cache = PipelineCache::new(); + let content = content_at(1); + + cache.insert_failover_cache(content.clone(), true).await; + + let found = cache.get_failover_cache("b", "p", None).await.unwrap(); + assert_eq!(found, Some(content.clone())); + + // A same-named pipeline under another schema must not shadow the global one. + let schema_local = PipelineContent { + schema: "x".to_string(), + ..content_at(2) + }; + cache.insert_failover_cache(schema_local, true).await; + + let found = cache.get_failover_cache("b", "p", None).await.unwrap(); + assert_eq!(found, Some(content)); + } +} diff --git a/src/pipeline/src/manager/table.rs b/src/pipeline/src/manager/table.rs index 8720446a64..d61856dfb3 100644 --- a/src/pipeline/src/manager/table.rs +++ b/src/pipeline/src/manager/table.rs @@ -262,21 +262,12 @@ impl PipelineTable { name: &str, input_version: PipelineVersion, ) -> Result> { - if let Some(pipeline) = self.cache.get_pipeline_cache(schema, name, input_version)? { - return Ok(pipeline); - } - - let pipeline_content = self.get_pipeline_str(schema, name, input_version).await?; - let compiled_pipeline = Arc::new(Self::compile_pipeline(&pipeline_content.content)?); - - self.cache.insert_pipeline_cache( - &pipeline_content.schema, - name, - Some(pipeline_content.version), - compiled_pipeline.clone(), - input_version.is_none(), - ); - Ok(compiled_pipeline) + self.cache + .get_pipeline_with(schema, name, input_version, async { + let pipeline_content = self.get_pipeline_str(schema, name, input_version).await?; + Ok(Arc::new(Self::compile_pipeline(&pipeline_content.content)?)) + }) + .await } /// Get a original pipeline by name. @@ -287,13 +278,19 @@ impl PipelineTable { name: &str, input_version: PipelineVersion, ) -> Result { - if let Some(pipeline) = self - .cache - .get_pipeline_str_cache(schema, name, input_version)? - { - return Ok(pipeline); - } + self.cache + .get_pipeline_str_with(schema, name, input_version, async { + self.load_pipeline_str(schema, name, input_version).await + }) + .await + } + async fn load_pipeline_str( + &self, + schema: &str, + name: &str, + input_version: PipelineVersion, + ) -> Result { let mut pipeline_vec; match self.find_pipeline(name, input_version).await { Ok(p) => { @@ -312,7 +309,8 @@ impl PipelineTable { .inc(); return self .cache - .get_failover_cache(schema, name, input_version)? + .get_failover_cache(schema, name, input_version) + .await? .context(PipelineNotFoundSnafu { name, version: input_version, @@ -338,7 +336,8 @@ impl PipelineTable { let pipeline_content = pipeline_vec.remove(0); self.cache - .insert_pipeline_str_cache(&pipeline_content, input_version.is_none()); + .insert_failover_cache(pipeline_content.clone(), input_version.is_none()) + .await; return Ok(pipeline_content); } @@ -359,12 +358,12 @@ impl PipelineTable { })?; self.cache - .insert_pipeline_str_cache(&pipeline_content, input_version.is_none()); + .insert_failover_cache(pipeline_content.clone(), input_version.is_none()) + .await; Ok(pipeline_content) } /// Insert a pipeline into the pipeline table and compile it. - /// The compiled pipeline will be inserted into the cache. /// Newly created pipelines will be saved under empty schema. pub async fn insert_and_compile( &self, @@ -378,25 +377,14 @@ impl PipelineTable { .insert_pipeline_to_pipeline_table(name, content_type, pipeline) .await?; - { - self.cache.insert_pipeline_cache( - EMPTY_SCHEMA_NAME, - name, - Some(TimestampNanosecond(version)), - compiled_pipeline.clone(), - true, - ); - - let pipeline_content = PipelineContent { + self.cache + .on_pipeline_created(PipelineContent { name: name.to_string(), content: pipeline.to_string(), version: TimestampNanosecond(version), schema: EMPTY_SCHEMA_NAME.to_string(), - }; - - self.cache - .insert_pipeline_str_cache(&pipeline_content, true); - } + }) + .await; Ok((version, compiled_pipeline)) } @@ -464,8 +452,7 @@ impl PipelineTable { output ); - // remove cache with version and latest - self.cache.remove_cache(name, version); + self.cache.invalidate(name, version).await; Ok(Some(())) } diff --git a/src/pipeline/tests/json_parse.rs b/src/pipeline/tests/json_parse.rs index 37979745ed..c374c846cd 100644 --- a/src/pipeline/tests/json_parse.rs +++ b/src/pipeline/tests/json_parse.rs @@ -17,7 +17,12 @@ mod common; use std::borrow::Cow; use api::v1::ColumnDataType; +use api::v1::json_value::Value as JsonValue; use api::v1::value::ValueData; +use arrow_schema::extension::{ + EXTENSION_TYPE_METADATA_KEY, EXTENSION_TYPE_NAME_KEY, ExtensionType, +}; +use datatypes::extension::json::{Json2ExtensionType, JsonMetadata}; const INPUT_VALUE_OBJ: &str = r#" [ @@ -113,6 +118,52 @@ transform: assert_eq!(v, jsonb::Value::Object(expected)); } +#[test] +fn test_json2_parse() { + let pipeline_yaml = r#" +--- +processors: + - json_parse: + field: commit + +transform: + - field: commit + type: + json2: + - path: "commitAuthor" + type: string + nullable: false +"#; + + let output = common::parse_and_exec(INPUT_VALUE_OBJ, pipeline_yaml); + + assert_eq!(output.schema[0].datatype, ColumnDataType::Json as i32); + assert!(output.schema[0].datatype_extension.is_none()); + assert_eq!( + output.schema[0] + .options + .as_ref() + .and_then(|options| options.options.get(EXTENSION_TYPE_NAME_KEY)) + .map(String::as_str), + Some(Json2ExtensionType::NAME) + ); + let metadata = output.schema[0] + .options + .as_ref() + .and_then(|options| options.options.get(EXTENSION_TYPE_METADATA_KEY)) + .unwrap(); + let metadata: JsonMetadata = serde_json::from_str(metadata).unwrap(); + assert_eq!( + metadata.json_settings().type_hints()[0].path, + ["commitAuthor"] + ); + + let ValueData::JsonValue(value) = output.rows[0].values[0].value_data.as_ref().unwrap() else { + panic!("expect JSON2 value"); + }; + assert!(matches!(value.value, Some(JsonValue::Object(_)))); +} + #[test] fn test_json_parse_with_simple_extractor() { let pipeline_yaml = r#" diff --git a/src/query/src/analyze.rs b/src/query/src/analyze.rs index ec0abf25ba..2ab3ec5176 100644 --- a/src/query/src/analyze.rs +++ b/src/query/src/analyze.rs @@ -46,6 +46,17 @@ const STAGE: &str = "stage"; const NODE: &str = "node"; const PLAN: &str = "plan"; +/// Fixed output schema of [`DistAnalyzeExec`], for `Describe` handlers: +/// execution rewrites the plan in `optimize_physical_plan`, so this schema +/// differs from the logical `Analyze` plan's. +pub fn dist_analyze_output_schema() -> SchemaRef { + SchemaRef::new(Schema::new(vec![ + Field::new(STAGE, DataType::UInt32, true), + Field::new(NODE, DataType::UInt32, true), + Field::new(PLAN, DataType::Utf8, true), + ])) +} + #[derive(Debug)] pub struct DistAnalyzeExec { input: Arc, @@ -58,11 +69,7 @@ pub struct DistAnalyzeExec { impl DistAnalyzeExec { /// Create a new DistAnalyzeExec pub fn new(input: Arc, verbose: bool, format: AnalyzeFormat) -> Self { - let schema = SchemaRef::new(Schema::new(vec![ - Field::new(STAGE, DataType::UInt32, true), - Field::new(NODE, DataType::UInt32, true), - Field::new(PLAN, DataType::Utf8, true), - ])); + let schema = dist_analyze_output_schema(); let properties = Arc::new(Self::compute_properties(&input, schema.clone())); Self { input, diff --git a/src/query/src/datafusion.rs b/src/query/src/datafusion.rs index 2d4f237dd6..e6e879a584 100644 --- a/src/query/src/datafusion.rs +++ b/src/query/src/datafusion.rs @@ -30,7 +30,7 @@ use common_function::function::FunctionContext; use common_function::function_factory::ScalarFunctionFactory; use common_query::{Output, OutputData, OutputMeta}; use common_recordbatch::adapter::{RecordBatchStreamAdapter, RegionQueryStatCounters}; -use common_recordbatch::{EmptyRecordBatchStream, SendableRecordBatchStream}; +use common_recordbatch::{EmptyRecordBatchStream, RecordBatch, SendableRecordBatchStream}; use common_telemetry::tracing; use datafusion::catalog::TableFunction; use datafusion::dataframe::DataFrame; @@ -44,6 +44,7 @@ use datafusion_expr::{ use datatypes::prelude::VectorRef; use datatypes::schema::Schema; use futures_util::StreamExt; +use futures_util::future::try_join; use session::context::QueryContextRef; use snafu::{OptionExt, ResultExt, ensure}; use sqlparser::ast::AnalyzeFormat; @@ -80,6 +81,29 @@ pub const QUERY_PARALLELISM_HINT: &str = "query_parallelism"; /// Whether to fallback to the original plan when failed to push down. pub const QUERY_FALLBACK_HINT: &str = "query_fallback"; +// An unbounded queue keeps draining source RPCs while mutation RPCs on a shared +// HTTP/2 connection are pending, trading bounded memory for request liveness. +async fn forward_record_batches( + mut stream: SendableRecordBatchStream, + batch_tx: tokio::sync::mpsc::UnboundedSender>, +) -> Result<()> { + while let Some(batch) = stream.next().await { + match batch.context(CreateRecordBatchSnafu) { + Ok(batch) => { + if batch_tx.send(Ok(batch)).is_err() { + break; + } + tokio::task::yield_now().await; + } + Err(error) => { + let _ = batch_tx.send(Err(error)); + break; + } + } + } + Ok(()) +} + fn query_load_region_id(plan: &Arc) -> Option { let mut region_id = None; let mut stack = vec![plan.clone()]; @@ -205,7 +229,7 @@ impl DatafusionQueryEngine { let Output { data, meta } = self .exec_query_plan((*dml.input).clone(), query_ctx.clone()) .await?; - let mut stream = match data { + let stream = match data { OutputData::RecordBatches(batches) => batches.as_stream(), OutputData::Stream(stream) => stream, _ => unreachable!(), @@ -214,30 +238,35 @@ impl DatafusionQueryEngine { let mut affected_rows = 0; let mut insert_cost = 0; - while let Some(batch) = stream.next().await { - let batch = batch.context(CreateRecordBatchSnafu)?; - let column_vectors = batch - .column_vectors(&table_name.to_string(), table.schema()) - .map_err(BoxedError::new) - .context(QueryExecutionSnafu)?; - - match dml.op { - WriteOp::Insert(_) => { - // We ignore the insert op. - let output = self - .insert(&table_name, column_vectors, query_ctx.clone()) - .await?; - let (rows, cost) = output.extract_rows_and_cost(); - affected_rows += rows; - insert_cost += cost; - } - WriteOp::Delete => { - affected_rows += self - .delete(&table_name, &table, column_vectors, query_ctx.clone()) - .await?; - } - _ => unreachable!("guarded by the 'ensure!' at the beginning"), + match dml.op { + WriteOp::Insert(_) => { + let (batch_tx, batch_rx) = tokio::sync::mpsc::unbounded_channel(); + let producer = forward_record_batches(stream, batch_tx); + let consumer = self.consume_insert_record_batches( + batch_rx, + &table_name, + table.schema(), + query_ctx.clone(), + ); + let ((), (rows, cost)) = try_join(producer, consumer).await?; + affected_rows += rows; + insert_cost += cost; } + WriteOp::Delete => { + // Keep DELETE on the same producer/consumer schedule as INSERT so the source + // stream can continue draining while mutation RPCs are pending. + let (batch_tx, batch_rx) = tokio::sync::mpsc::unbounded_channel(); + let producer = forward_record_batches(stream, batch_tx); + let consumer = self.consume_delete_record_batches( + batch_rx, + &table_name, + &table, + query_ctx.clone(), + ); + let (rows, ()) = try_join(consumer, producer).await?; + affected_rows += rows; + } + _ => unreachable!("guarded by the 'ensure!' at the beginning"), } Ok(Output::new( OutputData::AffectedRows(affected_rows), @@ -245,6 +274,53 @@ impl DatafusionQueryEngine { )) } + async fn consume_insert_record_batches( + &self, + mut batch_rx: tokio::sync::mpsc::UnboundedReceiver>, + table_name: &ResolvedTableReference, + table_schema: Arc, + query_ctx: QueryContextRef, + ) -> Result<(usize, usize)> { + let mut affected_rows = 0; + let mut insert_cost = 0; + while let Some(batch) = batch_rx.recv().await { + let batch = batch?; + let column_vectors = batch + .column_vectors(&table_name.to_string(), table_schema.clone()) + .map_err(BoxedError::new) + .context(QueryExecutionSnafu)?; + // We ignore the insert op. + let output = self + .insert(table_name, column_vectors, query_ctx.clone()) + .await?; + let (rows, cost) = output.extract_rows_and_cost(); + affected_rows += rows; + insert_cost += cost; + } + Ok((affected_rows, insert_cost)) + } + + async fn consume_delete_record_batches( + &self, + mut batch_rx: tokio::sync::mpsc::UnboundedReceiver>, + table_name: &ResolvedTableReference, + table: &TableRef, + query_ctx: QueryContextRef, + ) -> Result { + let mut affected_rows = 0; + while let Some(batch) = batch_rx.recv().await { + let batch = batch?; + let column_vectors = batch + .column_vectors(&table_name.to_string(), table.schema()) + .map_err(BoxedError::new) + .context(QueryExecutionSnafu)?; + affected_rows += self + .delete(table_name, table, column_vectors, query_ctx.clone()) + .await?; + } + Ok(affected_rows) + } + #[tracing::instrument(skip_all)] async fn delete( &self, diff --git a/src/query/src/datafusion/json_expr_planner.rs b/src/query/src/datafusion/json_expr_planner.rs index e650ac102c..04acb48d19 100644 --- a/src/query/src/datafusion/json_expr_planner.rs +++ b/src/query/src/datafusion/json_expr_planner.rs @@ -18,23 +18,32 @@ use arrow_schema::Field; use common_function::scalars::json::json_get::JsonGetWithType; use common_function::scalars::udf::create_udf; use datafusion_common::arrow::datatypes::DataType; -use datafusion_common::{Column, DFSchema, Result, ScalarValue, TableReference}; +use datafusion_common::{Column, DFSchema, DataFusionError, Result, ScalarValue, TableReference}; use datafusion_expr::expr::{BinaryExpr, ScalarFunction}; -use datafusion_expr::planner::{ExprPlanner, PlannerResult, RawBinaryExpr}; -use datafusion_expr::{Expr, ExprSchemable, Operator, ScalarUDF}; +use datafusion_expr::planner::{ + ExprPlanner, PlannerResult, RawAggregateExpr, RawBinaryExpr, RawFieldAccessExpr, RawScalarExpr, + RawWindowExpr, +}; +use datafusion_expr::type_coercion::functions::{UDFCoercionExt, fields_with_udf}; +use datafusion_expr::{ + Expr, ExprSchemable, GetFieldAccess, Operator, ScalarUDF, WindowFunctionDefinition, +}; use datatypes::extension::json::is_json2_extension_type; -use either::Either; use sqlparser::ast::BinaryOperator; /// Rewrites JSON-aware SQL expressions into DataFusion expressions. /// -/// This planner handles two cases: +/// This planner handles three cases: /// - Rewrites compound identifiers on JSON extension columns into `json_get` function. /// For example, `select a.b.c` => `select json_get(a, "b.c")`. +/// - Extends a JSON path with list indexes and fields following an index. +/// For example, `select a.b[0].c` => `select json_get(a, "b[0][\"c\"]")`. /// - Pushes an "expected type" argument into the `json_get` function when it participates in a /// binary operator. So that `json_get` knows the wanted data type when dealing with variant /// JSON values. /// For example, `select json_get(a, "b.c") + 1` => `select json_get(a, "b.c", NULL::Int64) + 1`. +/// - Infers the expected type from scalar, aggregate, and window function signatures. +/// For example, `select abs(a.b.c)` => `select abs(json_get(a, "b.c", NULL::Float64))`. #[derive(Debug)] pub(crate) struct JsonExprPlanner; @@ -50,9 +59,7 @@ impl ExprPlanner for JsonExprPlanner { mut right, } = expr; - if extract_untyped_json_get(&mut left).is_none() - && extract_untyped_json_get(&mut right).is_none() - { + if !is_untyped_json_get(&left) && !is_untyped_json_get(&right) { return Ok(PlannerResult::Original(RawBinaryExpr { op, left, right })); } @@ -62,20 +69,64 @@ impl ExprPlanner for JsonExprPlanner { let left_type = left.get_type(schema)?; let right_type = right.get_type(schema)?; - let left = push_json_get_type_arg(left, right_type)?; - let right = push_json_get_type_arg(right, left_type)?; - match (left, right) { - (Either::Left(left), Either::Left(right)) => { - Ok(PlannerResult::Original(RawBinaryExpr { op, left, right })) - } - (left, right) => Ok(PlannerResult::Planned(Expr::BinaryExpr(BinaryExpr::new( - Box::new(left.into_inner()), + let left_changed = push_json_get_type_arg(&mut left, &right_type)?; + let right_changed = push_json_get_type_arg(&mut right, &left_type)?; + if left_changed || right_changed { + Ok(PlannerResult::Planned(Expr::BinaryExpr(BinaryExpr::new( + Box::new(left), expr_op, - Box::new(right.into_inner()), - )))), + Box::new(right), + )))) + } else { + Ok(PlannerResult::Original(RawBinaryExpr { op, left, right })) } } + /// Extends the path of an untyped `json_get` with one field access. + /// + /// For `j.o.l[1].inner.l[2]`, `plan_compound_identifier` first produces + /// `json_get(j, "o.l")`. DataFusion then calls this method successively + /// with a list index, two named fields, and another list index, producing + /// the final path `o.l[1]["inner"]["l"][2]`. + fn plan_field_access( + &self, + mut expr: RawFieldAccessExpr, + _schema: &DFSchema, + ) -> Result> { + // See `normalize_field_access_after_subscript` for the reason why we construct the + // "suffix" like this. + let suffix = match &expr.field_access { + GetFieldAccess::ListIndex { key } => { + // DataFusion parses ordinary integer literals within the i64 range as Int64. + let Expr::Literal(ScalarValue::Int64(Some(index)), _) = key.as_ref() else { + return Ok(PlannerResult::Original(expr)); + }; + format!("[{index}]") + } + GetFieldAccess::NamedStructField { name } => { + let Some(name) = name.try_as_str().flatten() else { + return Ok(PlannerResult::Original(expr)); + }; + // Encode the field name as a JSON string before embedding it in the + // bracket accessor. This preserves dots as literal field-name characters + // and escapes quotes, backslashes, and control characters correctly. + let name = serde_json::to_string(name) + .map_err(|e| DataFusionError::External(Box::new(e)))?; + format!("[{name}]") + } + GetFieldAccess::ListRange { .. } => return Ok(PlannerResult::Original(expr)), + }; + let Some(json_get) = extract_untyped_json_get(&mut expr.expr) else { + return Ok(PlannerResult::Original(expr)); + }; + let Some(Expr::Literal(ScalarValue::Utf8(Some(path)), _)) = json_get.args.get_mut(1) else { + return Ok(PlannerResult::Original(expr)); + }; + + path.push_str(&suffix); + Ok(PlannerResult::Planned(expr.expr)) + } + fn plan_compound_identifier( &self, field: &Field, @@ -101,31 +152,235 @@ impl ExprPlanner for JsonExprPlanner { ), ))) } + + /// Rewrites JSON2 arguments without taking over the final function planning. + /// + /// `Original` carries the possibly modified raw expression to subsequent planners and then + /// DataFusion's default function construction. Returning `Planned` would short-circuit both. + fn plan_scalar(&self, mut expr: RawScalarExpr) -> Result> { + push_function_arg_types(expr.func.as_ref(), &mut expr.args)?; + Ok(PlannerResult::Original(expr)) + } + + /// Rewrites JSON2 arguments while preserving subsequent aggregate planning. + fn plan_aggregate( + &self, + mut expr: RawAggregateExpr, + ) -> Result> { + push_function_arg_types(expr.func.as_ref(), &mut expr.args)?; + Ok(PlannerResult::Original(expr)) + } + + /// Rewrites JSON2 arguments while preserving subsequent window planning. + fn plan_window(&self, mut expr: RawWindowExpr) -> Result> { + match &expr.func_def { + WindowFunctionDefinition::AggregateUDF(func) => { + push_function_arg_types(func.as_ref(), &mut expr.args)?; + } + WindowFunctionDefinition::WindowUDF(func) => { + push_function_arg_types(func.as_ref(), &mut expr.args)?; + } + } + Ok(PlannerResult::Original(expr)) + } +} + +enum JsonGetTypeResolution { + Fallback, + Typed(Vec<(usize, DataType)>), +} + +/// Infers static output types for untyped `json_get` arguments from a function signature. +/// +/// DataFusion requires every expression to have one Arrow data type during planning. A JSON path +/// may contain heterogeneous values across rows, but it cannot expose those values as different +/// Arrow types in one result column. Preserving their runtime types would require a single +/// Variant-like data type and Variant-aware functions instead. Maybe we can wait for +/// https://github.com/apache/datafusion/issues/16116 +/// +/// This helper uses the function's coercion rules to select a supported output type, then appends +/// a typed NULL argument to each relevant `json_get`. The typed argument makes `json_get` project +/// compatible JSON values to that type and return NULL for incompatible values. Functions that +/// accept json_get's default `Utf8View` output keep the two-argument form so later rewrites can +/// still push down an outer cast. +fn push_function_arg_types(func: &F, args: &mut [Expr]) -> Result<()> +where + F: UDFCoercionExt, +{ + if !args.iter().any(is_untyped_json_get) { + return Ok(()); + } + + let fields = args.iter().map(function_arg_field).collect::>(); + match infer_json_get_types(func, args, &fields) { + JsonGetTypeResolution::Fallback => { + let Some(data_type) = fallback_json_get_type(func, args, &fields) else { + return Ok(()); + }; + for arg in args.iter_mut() { + if is_untyped_json_get(arg) { + let _ = push_json_get_type_arg(arg, &data_type)?; + } + } + } + JsonGetTypeResolution::Typed(types) => { + for (index, data_type) in types { + let _ = push_json_get_type_arg(&mut args[index], &data_type)?; + } + } + } + Ok(()) +} + +fn infer_json_get_types(func: &F, args: &[Expr], fields: &[Arc]) -> JsonGetTypeResolution +where + F: UDFCoercionExt, +{ + // Only untyped json_get arguments use Null placeholders; preserve every other known argument + // type. fields_with_udf performs contextual coercion rather than reverse inference from a + // signature alone. Numeric signatures may preserve all-Null inputs, while Comparable + // signatures may default them to Utf8. For example, retaining the Float64 peer in + // coalesce(json_get(...), 1.0) lets DataFusion resolve json_get to Float64 instead of Utf8. + // + // This is a best-effort probe: a failure does not mean the actual function call is invalid, so + // try concrete JSON types before leaving final validation to DataFusion's default planner. + let Ok(coerced) = fields_with_udf(fields, func) else { + return JsonGetTypeResolution::Fallback; + }; + + let mut inferred_types = Vec::with_capacity(coerced.len()); + for (index, (arg, field)) in args.iter().zip(coerced).enumerate() { + if !is_untyped_json_get(arg) || field.data_type().is_null() { + continue; + } + let Some(data_type) = json_get_output_type(field.data_type()) else { + return JsonGetTypeResolution::Fallback; + }; + inferred_types.push((index, data_type)); + } + if inferred_types.is_empty() { + JsonGetTypeResolution::Fallback + } else { + JsonGetTypeResolution::Typed(inferred_types) + } +} + +fn fallback_json_get_type(func: &F, args: &[Expr], fields: &[Arc]) -> Option +where + F: UDFCoercionExt, +{ + // Prefer json_get's default Utf8View type. If the function rejects strings but accepts numeric + // values, prefer Float64 so both integers and fractions remain usable. + let mut candidate_fields = fields.to_vec(); + for data_type in [ + DataType::Utf8View, + DataType::Float64, + DataType::Int64, + DataType::Boolean, + ] { + for (index, arg) in args.iter().enumerate() { + if is_untyped_json_get(arg) { + candidate_fields[index] = Arc::new( + fields[index] + .as_ref() + .clone() + .with_data_type(data_type.clone()), + ); + } + } + if fields_with_udf(&candidate_fields, func).is_ok() { + return Some(data_type); + } + } + None +} + +fn function_arg_field(expr: &Expr) -> Arc { + let data_type = if is_untyped_json_get(expr) { + DataType::Null + } else if let Some(data_type) = extract_json_get_type(expr) { + data_type + } else { + // Treat unresolved expressions as untyped NULL. This lets signatures such as `power` + // infer a JSON type, while functions such as `coalesce` can leave it untyped for default + // planning. This is only best-effort: overloaded or user-defined functions may select a + // different signature for NULL than for the expression's actual type. + // TODO(LFC): Use the input schema once DataFusion passes it to ExprPlanner::plan_*(). + expr.get_type(&DFSchema::empty()).unwrap_or(DataType::Null) + }; + Arc::new(Field::new("", data_type, true)) +} + +fn json_get_output_type(data_type: &DataType) -> Option { + let output_type = match data_type { + DataType::Boolean => DataType::Boolean, + data_type if data_type.is_integer() => DataType::Int64, + data_type if data_type.is_floating() => DataType::Float64, + DataType::Decimal128(_, _) | DataType::Decimal256(_, _) => DataType::Float64, + data_type if data_type.is_string() => DataType::Utf8View, + _ => return None, + }; + Some(output_type) +} + +macro_rules! is_untyped_json_get_func { + ($func:expr) => { + $func + .func + .name() + .eq_ignore_ascii_case(JsonGetWithType::NAME) + && $func.args.len() == 2 + }; +} + +macro_rules! is_typed_json_get_func { + ($func:expr) => { + $func + .func + .name() + .eq_ignore_ascii_case(JsonGetWithType::NAME) + && $func.args.len() == 3 + }; } fn extract_untyped_json_get(expr: &mut Expr) -> Option<&mut ScalarFunction> { match expr { - Expr::ScalarFunction(f) - if f.func.name().eq_ignore_ascii_case(JsonGetWithType::NAME) && f.args.len() == 2 => - { - Some(f) - } + Expr::ScalarFunction(f) if is_untyped_json_get_func!(f) => Some(f), _ => None, } } -fn push_json_get_type_arg(mut expr: Expr, mut data_type: DataType) -> Result> { - let Some(json_get) = extract_untyped_json_get(&mut expr) else { - return Ok(Either::Left(expr)); +fn extract_json_get_type(expr: &Expr) -> Option { + match expr { + Expr::ScalarFunction(f) if is_typed_json_get_func!(f) => f + .args + .get(2) + .and_then(|x| x.as_literal()) + .map(|x| x.data_type()), + _ => None, + } +} + +fn is_untyped_json_get(expr: &Expr) -> bool { + matches!( + expr, + Expr::ScalarFunction(f) if is_untyped_json_get_func!(f) + ) +} + +fn push_json_get_type_arg(expr: &mut Expr, data_type: &DataType) -> Result { + let Some(json_get) = extract_untyped_json_get(expr) else { + return Ok(false); }; + // The two-argument form already returns Utf8View. Keep it so JsonGetRewriter can still absorb + // a cast added by subsequent function coercion. if data_type.is_string() { - data_type = DataType::Utf8View; + return Ok(false); } - let with_type = ScalarValue::try_new_null(&data_type).map(|x| Expr::Literal(x, None))?; + let with_type = ScalarValue::try_new_null(data_type).map(|x| Expr::Literal(x, None))?; json_get.args.push(with_type); - - Ok(Either::Right(expr)) + Ok(true) } fn parse_sql_op(op: &BinaryOperator) -> Option { @@ -153,6 +408,11 @@ fn parse_sql_op(op: &BinaryOperator) -> Option { #[cfg(test)] mod tests { use arrow_schema::Fields; + use datafusion::functions_aggregate::count::count_udaf; + use datafusion::functions_aggregate::sum::sum_udaf; + use datafusion_expr::WindowFrame; + use datafusion_functions::core::coalesce; + use datafusion_functions::math::{abs, power}; use datatypes::extension::json::Json2ExtensionType; use super::*; @@ -236,6 +496,55 @@ mod tests { Ok(()) } + #[test] + fn test_plan_list_index() -> Result<()> { + let planner = JsonExprPlanner; + let planned = planner.plan_field_access( + RawFieldAccessExpr { + field_access: GetFieldAccess::ListIndex { + key: Box::new(Expr::Literal(ScalarValue::Int64(Some(0)), None)), + }, + expr: json_get_expr(Expr::Column(Column::new_unqualified("j")), "list"), + }, + &DFSchema::empty(), + )?; + let PlannerResult::Planned(Expr::ScalarFunction(func)) = planned else { + unreachable!() + }; + assert_eq!(func.func.name(), JsonGetWithType::NAME); + assert_eq!(func.args.len(), 2); + assert_eq!( + func.args[1], + Expr::Literal(ScalarValue::Utf8(Some("list[0]".to_string())), None) + ); + Ok(()) + } + + #[test] + fn test_plan_field_after_list_index() -> Result<()> { + let planner = JsonExprPlanner; + let planned = planner.plan_field_access( + RawFieldAccessExpr { + field_access: GetFieldAccess::NamedStructField { + name: ScalarValue::Utf8(Some("a.b".to_string())), + }, + expr: json_get_expr(Expr::Column(Column::new_unqualified("j")), "list[0]"), + }, + &DFSchema::empty(), + )?; + let PlannerResult::Planned(Expr::ScalarFunction(func)) = planned else { + unreachable!() + }; + assert_eq!( + func.args[1], + Expr::Literal( + ScalarValue::Utf8(Some("list[0][\"a.b\"]".to_string())), + None + ) + ); + Ok(()) + } + #[test] fn test_plan_compound_identifier() -> Result<()> { let planner = JsonExprPlanner; @@ -281,4 +590,110 @@ mod tests { Ok(()) } + + #[test] + fn test_plan_functions() -> Result<()> { + let planner = JsonExprPlanner; + let json_get = || json_get_expr(Expr::Column(Column::new_unqualified("j")), "a.b"); + + let PlannerResult::Original(scalar) = planner.plan_scalar(RawScalarExpr { + func: abs(), + args: vec![json_get()], + })? + else { + unreachable!(); + }; + assert_eq!( + Some(DataType::Float64), + extract_json_get_type(&scalar.args[0]) + ); + + let PlannerResult::Original(scalar) = planner.plan_scalar(RawScalarExpr { + func: power(), + args: vec![ + json_get(), + Expr::Column(Column::new_unqualified("exponent")), + ], + })? + else { + unreachable!(); + }; + assert_eq!( + Some(DataType::Float64), + extract_json_get_type(&scalar.args[0]) + ); + + let PlannerResult::Original(aggregate) = planner.plan_aggregate(RawAggregateExpr { + func: sum_udaf(), + args: vec![json_get()], + distinct: false, + filter: None, + order_by: vec![], + null_treatment: None, + })? + else { + unreachable!(); + }; + assert_eq!( + Some(DataType::Float64), + extract_json_get_type(&aggregate.args[0]) + ); + + let PlannerResult::Original(count) = planner.plan_aggregate(RawAggregateExpr { + func: count_udaf(), + args: vec![json_get()], + distinct: false, + filter: None, + order_by: vec![], + null_treatment: None, + })? + else { + unreachable!(); + }; + assert_eq!(None, extract_json_get_type(&count.args[0])); + + let PlannerResult::Original(window) = planner.plan_window(RawWindowExpr { + func_def: WindowFunctionDefinition::AggregateUDF(sum_udaf()), + args: vec![json_get()], + partition_by: vec![], + order_by: vec![], + window_frame: WindowFrame::new(None), + filter: None, + null_treatment: None, + distinct: false, + })? + else { + unreachable!(); + }; + assert_eq!( + Some(DataType::Float64), + extract_json_get_type(&window.args[0]) + ); + Ok(()) + } + + #[test] + fn test_plan_function_with_mixed_json_get_types() -> Result<()> { + let planner = JsonExprPlanner; + let json_get = || json_get_expr(Expr::Column(Column::new_unqualified("j")), "a.b"); + let mut typed = json_get(); + push_json_get_type_arg(&mut typed, &DataType::Float64)?; + + let PlannerResult::Original(scalar) = planner.plan_scalar(RawScalarExpr { + func: coalesce(), + args: vec![json_get(), typed], + })? + else { + unreachable!(); + }; + assert_eq!( + Some(DataType::Float64), + extract_json_get_type(&scalar.args[0]) + ); + assert_eq!( + Some(DataType::Float64), + extract_json_get_type(&scalar.args[1]) + ); + Ok(()) + } } diff --git a/src/query/src/dist_plan/merge_scan.rs b/src/query/src/dist_plan/merge_scan.rs index b3726dca2e..b50ff4ab63 100644 --- a/src/query/src/dist_plan/merge_scan.rs +++ b/src/query/src/dist_plan/merge_scan.rs @@ -20,7 +20,7 @@ use std::time::Duration; use ahash::{HashMap, HashSet}; use arrow_schema::{ - ArrowError, DataType, DataType as ArrowDataType, Field, Schema as ArrowSchema, + ArrowError, DataType as ArrowDataType, Field, Schema as ArrowSchema, SchemaRef as ArrowSchemaRef, SortOptions, }; use async_stream::stream; @@ -42,14 +42,12 @@ use datafusion::physical_plan::{ DisplayAs, DisplayFormatType, ExecutionPlan, Partitioning, PlanProperties, SendableRecordBatchStream, }; -use datafusion_common::{Column as ColumnExpr, DataFusionError, Result}; -use datafusion_expr::{Expr, Extension, LogicalPlan, UserDefinedLogicalNodeCore}; +use datafusion_common::stats::Precision; +use datafusion_common::{Column as ColumnExpr, DFSchemaRef, DataFusionError, Result, Statistics}; +use datafusion_expr::{Expr, Extension, FetchType, LogicalPlan, UserDefinedLogicalNodeCore}; use datafusion_physical_expr::expressions::Column; use datafusion_physical_expr::{Distribution, EquivalenceProperties, PhysicalSortExpr}; -use datatypes::extension::json::{ - Json2ExtensionType, is_any_json_extension_type, is_json2_extension_type, - is_legacy_json2_extension_type, -}; +use datatypes::extension::json::is_any_json_extension_type; use futures_util::StreamExt; use greptime_proto::v1::region::RegionRequestHeader; use meter_core::data::ReadItem; @@ -81,6 +79,50 @@ fn query_engine_state_from_task_context(context: &TaskContext) -> Option Option { + match plan { + LogicalPlan::Limit(limit) => { + let input_bound = remote_plan_row_bound(&limit.input); + match limit.get_fetch_type() { + Ok(FetchType::Literal(Some(fetch))) => { + Some(input_bound.map_or(fetch, |bound| bound.min(fetch))) + } + _ => input_bound, + } + } + LogicalPlan::Sort(sort) => { + let input_bound = remote_plan_row_bound(&sort.input); + sort.fetch + .map(|fetch| input_bound.map_or(fetch, |bound| bound.min(fetch))) + .or(input_bound) + } + LogicalPlan::Projection(projection) => remote_plan_row_bound(&projection.input), + LogicalPlan::Filter(filter) => remote_plan_row_bound(&filter.input), + LogicalPlan::SubqueryAlias(alias) => remote_plan_row_bound(&alias.input), + LogicalPlan::Window(window) => remote_plan_row_bound(&window.input), + LogicalPlan::Repartition(repartition) => remote_plan_row_bound(&repartition.input), + LogicalPlan::Distinct(distinct) => remote_plan_row_bound(distinct.input()), + LogicalPlan::Aggregate(aggregate) => { + if aggregate + .group_expr + .iter() + .any(|expr| matches!(expr, Expr::GroupingSet(_))) + { + None + } else if aggregate.group_expr.is_empty() { + Some(1) + } else { + remote_plan_row_bound(&aggregate.input) + } + } + _ => None, + } +} + fn remote_dyn_filter_enabled(query_ctx: &QueryContextRef) -> Result { remote_dyn_filter_pushdown_enabled_from_extensions(&query_ctx.extensions()) .map_err(|err| DataFusionError::External(Box::new(err))) @@ -232,10 +274,12 @@ fn query_context_for_remote_dyn_filter_region( query_context_with_initial_dyn_filter_regs(query_ctx, region_id, captured_dyn_filters) } -#[derive(Debug, Hash, PartialOrd, PartialEq, Eq, Clone)] +#[derive(Debug, Hash, PartialEq, Eq, Clone)] pub struct MergeScanLogicalPlan { /// In logical plan phase it only contains one input input: LogicalPlan, + /// Schema exposed to the local stage. + output_schema: DFSchemaRef, /// If this plan is a placeholder is_placeholder: bool, partition_cols: AliasMapping, @@ -243,6 +287,42 @@ pub struct MergeScanLogicalPlan { remote_dyn_filter_producer_id: Option, } +impl PartialOrd for MergeScanLogicalPlan { + fn partial_cmp(&self, other: &Self) -> Option { + let Self { + input, + output_schema, + is_placeholder, + partition_cols, + remote_dyn_filter_producer_id, + } = self; + let Self { + input: other_input, + output_schema: other_output_schema, + is_placeholder: other_is_placeholder, + partition_cols: other_partition_cols, + remote_dyn_filter_producer_id: other_remote_dyn_filter_producer_id, + } = other; + + let ordering = ( + input, + is_placeholder, + partition_cols, + remote_dyn_filter_producer_id, + ) + .partial_cmp(&( + other_input, + other_is_placeholder, + other_partition_cols, + other_remote_dyn_filter_producer_id, + )); + match ordering { + Some(std::cmp::Ordering::Equal) if output_schema != other_output_schema => None, + ordering => ordering, + } + } +} + impl UserDefinedLogicalNodeCore for MergeScanLogicalPlan { fn name(&self) -> &str { Self::name() @@ -255,7 +335,7 @@ impl UserDefinedLogicalNodeCore for MergeScanLogicalPlan { } fn schema(&self) -> &datafusion_common::DFSchemaRef { - self.input.schema() + &self.output_schema } // Prevent further optimization @@ -281,8 +361,10 @@ impl UserDefinedLogicalNodeCore for MergeScanLogicalPlan { } impl MergeScanLogicalPlan { + /// Creates a merge scan with the input plan's schema. pub fn new(input: LogicalPlan, is_placeholder: bool, partition_cols: AliasMapping) -> Self { Self { + output_schema: input.schema().clone(), input, is_placeholder, partition_cols, @@ -290,6 +372,12 @@ impl MergeScanLogicalPlan { } } + /// Replaces the schema exposed to the local stage. + pub(crate) fn with_output_schema(mut self, output_schema: DFSchemaRef) -> Self { + self.output_schema = output_schema; + self + } + pub(crate) fn with_remote_dyn_filter_producer_id( mut self, remote_dyn_filter_producer_id: RemoteDynFilterProducerId, @@ -373,7 +461,11 @@ impl MergeScanExec { remote_dyn_filter_producer_id: Option, enable_per_region_metrics: bool, ) -> Result { - let arrow_schema = maybe_amend_json2_field(arrow_schema); + // JSON2 schemas are concretized by the analyzer before physical planning. + // Keep the selected boundary schema unchanged here so the physical plan, + // remote batches, and local consumers share the same contract. + let arrow_schema = Arc::new(arrow_schema.clone()); + let output_partition_count = Self::output_partition_count(regions.len(), target_partition); // States the output ordering of the plan. // @@ -383,7 +475,7 @@ impl MergeScanExec { // // Otherwise, we need to use the default ordering. let eq_properties = if let LogicalPlan::Sort(sort) = &plan - && target_partition >= regions.len() + && output_partition_count >= regions.len() { let lex_ordering = sort .expr @@ -422,7 +514,7 @@ impl MergeScanExec { } }) .collect(); - let partitioning = Partitioning::Hash(partition_exprs, target_partition); + let partitioning = Partitioning::Hash(partition_exprs, output_partition_count); let properties = Arc::new(PlanProperties::new( eq_properties, @@ -449,6 +541,25 @@ impl MergeScanExec { }) } + /// Conservative row-count upper bound for all selected regions. + fn estimated_num_rows(&self) -> Precision { + if self.regions.is_empty() { + return Precision::Inexact(0); + } + + let Some(rows_per_region) = remote_plan_row_bound(&self.plan) else { + return Precision::Absent; + }; + rows_per_region + .checked_mul(self.regions.len()) + .map_or(Precision::Absent, Precision::Inexact) + } + + /// Number of partitions populated by the region striping in [`Self::to_stream`]. + fn output_partition_count(num_regions: usize, target_partition: usize) -> usize { + num_regions.max(1).min(target_partition.max(1)) + } + pub fn to_stream( &self, context: Arc, @@ -463,7 +574,8 @@ impl MergeScanExec { let sub_stage_metrics_moved = self.sub_stage_metrics.clone(); let partition_metrics_moved = self.partition_metrics.clone(); let plan = self.plan.clone(); - let target_partition = self.target_partition; + let target_partition = + Self::output_partition_count(self.regions.len(), self.target_partition); let remote_dyn_filter_enabled = remote_dyn_filter_enabled(&self.query_ctx)?; let captured_remote_dyn_filters = if remote_dyn_filter_enabled { self.captured_remote_dyn_filters() @@ -769,7 +881,7 @@ impl MergeScanExec { metric: self.metric.clone(), properties: Arc::new(PlanProperties::new( self.properties.eq_properties.clone(), - Partitioning::Hash(overlaps, self.target_partition), + Partitioning::Hash(overlaps, self.partition_count()), self.properties.emission_type, self.properties.boundedness, )), @@ -820,7 +932,7 @@ impl MergeScanExec { } pub fn partition_count(&self) -> usize { - self.target_partition + Self::output_partition_count(self.regions.len(), self.target_partition) } pub fn region_count(&self) -> usize { @@ -837,41 +949,6 @@ impl MergeScanExec { } } -// If the schema has JSON2 field, AND the field is of empty Struct datatype, amend it with Binary -// datatype. -// This is a very hacky way to make it possible to query the whole JSON2 column. Because when -// querying a whole JSON2 column, like in the SQL `select * from ...`, we can't concretize the JSON2 -// datatype from the query. Hence, the JSON2 datatype remains what in the column schema, i.e., empty -// Struct. An empty Struct is not alignable like any other concretized JSON2 datatypes, so to make -// the query work, we amend(rewrite) it to Binary datatype. -// Why the Binary datatype? Because underlying the scan and projection stage, the JSON2 data are -// variant shape, will be all converted to bytes. -// Anyway, this is not clean nor elegant. TODO(LFC) Maybe make it into some plan analyzer rule? -fn maybe_amend_json2_field(schema: &ArrowSchema) -> ArrowSchemaRef { - let schema = schema.clone(); - let mut new_fields = Vec::with_capacity(schema.fields().len()); - for field in schema.fields().iter() { - let new_field = if is_json2_extension_type(field) - && matches!(field.data_type(), DataType::Struct(fields) if fields.is_empty()) - { - let is_legacy_json2 = is_legacy_json2_extension_type(field); - let mut new_field = field.as_ref().clone(); - new_field.set_data_type(DataType::Binary); - if is_legacy_json2 { - new_field = new_field.with_extension_type(Json2ExtensionType::default()); - } - Arc::new(new_field) - } else { - field.clone() - }; - new_fields.push(new_field); - } - Arc::new(ArrowSchema::new_with_metadata( - new_fields, - schema.metadata().clone(), - )) -} - #[cfg(test)] impl MergeScanExec { fn remote_dyn_filter_producer_id(&self) -> Option { @@ -1085,6 +1162,16 @@ impl ExecutionPlan for MergeScanExec { Some(self.metric.clone_inner()) } + fn partition_statistics(&self, partition: Option) -> Result { + if partition.is_some() { + return Ok(Statistics::new_unknown(&self.arrow_schema)); + } + + let mut statistics = Statistics::new_unknown(&self.arrow_schema); + statistics.num_rows = self.estimated_num_rows(); + Ok(statistics) + } + fn name(&self) -> &str { "MergeScanExec" } @@ -1231,10 +1318,7 @@ mod tests { use std::pin::Pin; use std::task::{Context, Poll}; - use arrow_schema::extension::{ - EXTENSION_TYPE_METADATA_KEY, EXTENSION_TYPE_NAME_KEY, ExtensionType, - }; - use arrow_schema::{DataType as TestArrowDataType, Field, Fields}; + use arrow_schema::{DataType as TestArrowDataType, Field}; use async_trait::async_trait; use common_query::request::INITIAL_REMOTE_DYN_FILTER_REGISTRATIONS_EXTENSION_KEY; use common_recordbatch::adapter::{PlanMetrics, RecordBatchMetrics}; @@ -1303,6 +1387,30 @@ mod tests { .unwrap() } + fn merge_scan_exec_with_plan( + regions: Vec, + plan: LogicalPlan, + target_partition: usize, + ) -> MergeScanExec { + let session_state = SessionStateBuilder::new().build(); + let schema = plan.schema().as_arrow().clone(); + + MergeScanExec::new( + &session_state, + TableName::new("catalog", "schema", "table"), + regions, + plan, + &schema, + Arc::new(TestRegionQueryHandler::default()), + QueryContext::arc(), + target_partition, + AliasMapping::new(), + None, + false, + ) + .unwrap() + } + async fn collect_merge_scan( exec: MergeScanExec, ) -> datafusion_common::Result> { @@ -1328,43 +1436,6 @@ mod tests { metadata } - #[test] - fn test_amend_legacy_json2_field_preserves_json2_identity() { - let field = Field::new("j", TestArrowDataType::Struct(Fields::empty()), true) - .with_metadata(StdHashMap::from([ - ( - EXTENSION_TYPE_NAME_KEY.to_string(), - "greptime.json".to_string(), - ), - ( - EXTENSION_TYPE_METADATA_KEY.to_string(), - serde_json::json!({ - "json_structure_settings": { "Structured": null } - }) - .to_string(), - ), - ])); - - let legacy_schema = ArrowSchema::new(vec![field]); - let amended = maybe_amend_json2_field(&legacy_schema); - let amended_field = amended.field(0); - assert_eq!(&TestArrowDataType::Binary, amended_field.data_type()); - assert_eq!( - Some(Json2ExtensionType::NAME), - amended_field.extension_type_name() - ); - assert!(is_json2_extension_type(amended_field)); - - // The remote wire schema is Binary, while the legacy advertised schema - // still carries the greptime.json identity. MergeScan must accept the - // amended JSON2 field as the corresponding remote column. - let wire_schema = ArrowSchema::new(vec![ - Field::new("j", TestArrowDataType::Binary, true) - .with_metadata(legacy_schema.field(0).metadata().clone()), - ]); - assert!(validate_remote_schema(&wire_schema, amended.as_ref(), "legacy json2").is_ok()); - } - #[test] fn merge_scan_validates_remote_schema_semantics() { let expected = expected_int64_schema(); @@ -1882,6 +1953,118 @@ mod tests { assert!(exec.properties().output_ordering().is_some()); } + #[test] + fn merge_scan_reports_populated_partition_count() { + let cases = [(0, 10, 1), (1, 10, 1), (3, 2, 2), (5, 10, 5), (3, 0, 1)]; + + for (region_count, target, expected) in cases { + let exec = merge_scan_exec_with_sorted_input(region_count, target); + assert_eq!(exec.partition_count(), expected); + assert_eq!( + exec.properties().output_partitioning().partition_count(), + expected + ); + } + } + + #[test] + fn merge_scan_reports_only_deterministic_plan_bounds() { + use datafusion::functions_aggregate::expr_fn::count; + use datafusion_expr::GroupingSet; + + let regions = vec![RegionId::new(1024, 1), RegionId::new(1024, 2)]; + let limited = LogicalPlanBuilder::empty(true) + .project(vec![lit(1i64).alias("col")]) + .unwrap() + .limit(0, Some(50)) + .unwrap() + .build() + .unwrap(); + assert_eq!( + merge_scan_exec_with_plan(regions.clone(), limited, 10) + .partition_statistics(None) + .unwrap() + .num_rows, + Precision::Inexact(100) + ); + + let large_bound = i32::MAX as usize + 1; + let large_limit = LogicalPlanBuilder::empty(true) + .project(vec![lit(1i64).alias("col")]) + .unwrap() + .limit(0, Some(large_bound)) + .unwrap() + .build() + .unwrap(); + assert_eq!( + merge_scan_exec_with_plan(vec![RegionId::new(1024, 1)], large_limit, 10) + .partition_statistics(None) + .unwrap() + .num_rows, + Precision::Inexact(large_bound) + ); + + let uncapped = LogicalPlanBuilder::empty(true) + .project(vec![lit(1i64).alias("col")]) + .unwrap() + .build() + .unwrap(); + assert_eq!( + merge_scan_exec_with_plan(regions.clone(), uncapped.clone(), 10) + .partition_statistics(None) + .unwrap() + .num_rows, + Precision::Absent + ); + assert_eq!( + merge_scan_exec_with_plan(Vec::new(), uncapped, 10) + .partition_statistics(None) + .unwrap() + .num_rows, + Precision::Inexact(0) + ); + + let global_aggregate = LogicalPlanBuilder::empty(true) + .project(vec![lit(1i64).alias("col")]) + .unwrap() + .limit(0, Some(0)) + .unwrap() + .aggregate(Vec::::new(), vec![count(lit(1))]) + .unwrap() + .build() + .unwrap(); + assert_eq!( + merge_scan_exec_with_plan(regions.clone(), global_aggregate, 10) + .partition_statistics(None) + .unwrap() + .num_rows, + Precision::Inexact(2) + ); + + let grouping_sets = LogicalPlanBuilder::empty(true) + .project(vec![lit(1i64).alias("col")]) + .unwrap() + .limit(0, Some(50)) + .unwrap() + .aggregate( + vec![Expr::GroupingSet(GroupingSet::GroupingSets(vec![ + vec![], + vec![col("col")], + ]))], + Vec::::new(), + ) + .unwrap() + .build() + .unwrap(); + assert_eq!( + merge_scan_exec_with_plan(regions, grouping_sets, 10) + .partition_statistics(None) + .unwrap() + .num_rows, + Precision::Absent + ); + } + #[test] fn sub_stage_metrics_are_sorted_by_region_id() { let exec = merge_scan_exec_with_sorted_input(0, 1); diff --git a/src/query/src/dist_plan/planner.rs b/src/query/src/dist_plan/planner.rs index 999515df73..6b52dc8301 100644 --- a/src/query/src/dist_plan/planner.rs +++ b/src/query/src/dist_plan/planner.rs @@ -195,7 +195,7 @@ impl ExtensionPlanner for DistExtensionPlanner { }; // TODO(ruihang): generate different execution plans for different variant merge operation - let schema = optimized_plan.schema().as_arrow(); + let schema = merge_scan.schema().as_arrow(); let query_ctx = session_state .config() .get_extension() diff --git a/src/query/src/lib.rs b/src/query/src/lib.rs index 68d8ff3a9e..4b2a2d5264 100644 --- a/src/query/src/lib.rs +++ b/src/query/src/lib.rs @@ -45,7 +45,7 @@ pub(crate) mod test_util; #[cfg(test)] mod tests; -pub use crate::analyze::analyze_plan_metrics_to_json_value; +pub use crate::analyze::{analyze_plan_metrics_to_json_value, dist_analyze_output_schema}; pub use crate::datafusion::DfContextProviderAdapter; pub use crate::query_engine::{ QueryEngine, QueryEngineContext, QueryEngineFactory, QueryEngineRef, diff --git a/src/query/src/optimizer.rs b/src/query/src/optimizer.rs index 8827d48ed6..480c5046c2 100644 --- a/src/query/src/optimizer.rs +++ b/src/query/src/optimizer.rs @@ -17,6 +17,8 @@ pub mod constant_term; pub mod count_nest_aggr; pub mod count_wildcard; pub mod global_limit; +pub(crate) mod insert_assignment; +pub(crate) mod json_schema_concretize; pub(crate) mod json_type_concretize; pub mod parallelize_scan; pub mod pass_distribution; diff --git a/src/query/src/optimizer/insert_assignment.rs b/src/query/src/optimizer/insert_assignment.rs new file mode 100644 index 0000000000..ad77736ce0 --- /dev/null +++ b/src/query/src/optimizer/insert_assignment.rs @@ -0,0 +1,307 @@ +// Copyright 2023 Greptime Team +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use std::sync::Arc; + +use common_time::Timezone; +use datafusion::config::ConfigOptions; +use datafusion_common::{DFSchemaRef, Result, ScalarValue}; +use datafusion_expr::expr::{Alias, Cast}; +use datafusion_expr::{Distinct, Expr, ExprSchemable, LogicalPlan, Projection, Values}; +use datafusion_optimizer::analyzer::AnalyzerRule; +use datafusion_optimizer::analyzer::type_coercion::TypeCoercion; +use datatypes::arrow::datatypes::{DataType, TimeUnit}; +use session::context::QueryContextRef; + +use crate::optimizer::type_conversion::cast_string_to_timestamp; + +/// Interprets strings assigned to timestamp columns at an `INSERT` boundary +/// using the session timezone. `plan` is the assignment projection under a +/// `WriteOp::Insert`. +/// +/// DataFusion plans `INSERT` as a projection casting each source column to its +/// target column type, and that cast reads a naive string as UTC. Arrow does +/// apply a timezone when the cast target carries one, so the assignment is +/// routed through `Timestamp(unit, Some(tz))` and back. Stripping the timezone +/// afterwards is value-preserving — arrow only shifts values in the opposite +/// direction. +/// +/// The source query is left untouched. Reinterpreting a value where it is +/// *produced* would change what the source query means: pushing the conversion +/// below a `UNION`'s `DISTINCT`, for instance, moves the dedup key from the raw +/// strings to parsed instants and silently drops rows. +/// +/// # Why `TypeCoercion` runs here +/// +/// The rewrite reads source types, and those are only settled once a `UNION`'s +/// branch types have been reconciled: before coercion a union carries its loose +/// schema (the first branch's types), so a mixed +/// `SELECT 'string' UNION ALL SELECT CAST(.. AS TIMESTAMP)` still looks like a +/// string. Retargeting that cast would leave `Timestamp(None) -> +/// Timestamp(Some(tz))` behind once coercion retypes the union — the one +/// direction in which arrow shifts the value instead of relabelling it. +/// +/// Coercing here rather than deferring to the analyzer is forced by where an +/// INSERT is still identifiable: `exec_dml_statement` strips the `Dml` node and +/// executes its input, so by the time the analyzer runs, an assignment +/// projection is indistinguishable from any other projection. +/// +/// Explicit casts stay out of this: the SQL layer turns a user's +/// `CAST(x AS TIMESTAMP)` into an `arrow_cast` call, which only becomes an +/// `Expr::Cast` in the optimizer's `SimplifyExpressions`. Assignment casts are +/// therefore the only `Expr::Cast` reaching a timestamp column here. +/// +/// # Reach +/// +/// The emitted cast only carries its timezone where the expression is evaluated +/// on this node. Substrait drops the timezone name when a plan is pushed down — +/// it encodes any zoned timestamp as `PrecisionTimestampTz` and decodes it back +/// as UTC — so a source reading from a table falls back to UTC, the behaviour it +/// had before this rule existed. Sources that never leave this node (literals, +/// `VALUES`, and `UNION`s of them) keep the session timezone, and those are what +/// an INSERT's timestamp assignment is in practice. +pub(crate) fn rewrite_insert_assignments( + plan: LogicalPlan, + query_ctx: &QueryContextRef, + config: &ConfigOptions, +) -> Result { + let Some(timezone) = session_timezone(query_ctx) else { + return Ok(plan); + }; + + let plan = TypeCoercion::new().analyze(plan, config)?; + rewrite_assignment(plan, &timezone) +} + +/// Session timezone, in both forms the rewrite needs. +struct SessionTimezone { + /// Parses literals, matching the plain `INSERT ... VALUES` path. + parsed: Timezone, + /// Names the intermediate arrow cast target. + name: Arc, +} + +fn session_timezone(query_ctx: &QueryContextRef) -> Option { + let parsed = query_ctx.timezone(); + + // A UTC session already gets UTC semantics from the plain assignment cast. + if parsed.is_utc() { + return None; + } + + Some(SessionTimezone { + name: Arc::from(parsed.to_string()), + parsed, + }) +} + +fn rewrite_assignment(plan: LogicalPlan, timezone: &SessionTimezone) -> Result { + let LogicalPlan::Projection(assignment) = plan else { + return Ok(plan); + }; + + let mut exprs = assignment.expr.clone(); + let mut changed = false; + for expr in &mut exprs { + changed |= retarget_assignment_cast( + expr, + assignment.input.schema(), + Some(assignment.input.as_ref()), + timezone, + )?; + } + + // The planner types `VALUES` against the target table, so the assignment + // cast lands inside the `Values` rows instead of on the projection above. + let mut input = assignment.input.clone(); + if let LogicalPlan::Values(values) = assignment.input.as_ref() + && let Some(rewritten) = rewrite_values(values, timezone)? + { + input = Arc::new(LogicalPlan::Values(rewritten)); + changed = true; + } + + if !changed { + return Ok(LogicalPlan::Projection(assignment)); + } + Projection::try_new(exprs, input).map(LogicalPlan::Projection) +} + +fn rewrite_values(values: &Values, timezone: &SessionTimezone) -> Result> { + let mut rewritten = values.clone(); + let mut changed = false; + for row in &mut rewritten.values { + for expr in row.iter_mut() { + changed |= retarget_assignment_cast(expr, &values.schema, None, timezone)?; + } + } + + Ok(changed.then_some(rewritten)) +} + +/// Reinterprets one assignment cast, returning whether it was rewritten. +/// +/// `source_plan` is the projection's input, used to resolve a literal behind a +/// column reference; `Values` rows carry their expression inline and pass `None`. +fn retarget_assignment_cast( + expr: &mut Expr, + schema: &DFSchemaRef, + source_plan: Option<&LogicalPlan>, + timezone: &SessionTimezone, +) -> Result { + let expr = unalias_mut(expr); + let Expr::Cast(Cast { + expr: source, + data_type: DataType::Timestamp(unit, None), + }) = expr + else { + return Ok(false); + }; + let unit = *unit; + + if !matches!( + source.get_type(schema)?, + DataType::Utf8 | DataType::LargeUtf8 | DataType::Utf8View + ) { + return Ok(false); + } + + // Fold literals with the same parser the plain `INSERT ... VALUES` path + // uses, so a given string means the same thing however it reaches a column. + // The parsers disagree on ambiguous local times: this one resolves them, + // arrow rejects them. + let folded = source_literal(source.as_ref(), source_plan) + .and_then(|literal| convert_literal(&literal, unit, &timezone.parsed)); + if let Some(folded) = folded { + *expr = folded; + return Ok(true); + } + + let source = source.as_ref().clone(); + *expr = Expr::Cast(Cast::new( + Box::new(Expr::Cast(Cast::new( + Box::new(source), + DataType::Timestamp(unit, Some(timezone.name.clone())), + ))), + DataType::Timestamp(unit, None), + )); + Ok(true) +} + +fn source_literal(source: &Expr, source_plan: Option<&LogicalPlan>) -> Option { + match source { + Expr::Literal(value, _) => Some(value.clone()), + Expr::Column(column) => { + let plan = source_plan?; + let index = plan.schema().maybe_index_of_column(column)?; + lineage_literal(plan, index).cloned() + } + _ => None, + } +} + +/// Resolves a literal when every row carries the same value at `output_idx`. +/// +/// Read-only: the literal is folded into the assignment above, so nodes that +/// drop, reorder or deduplicate rows can be traversed — none of them changes +/// the value a surviving row carries, and folding above them leaves their keys +/// on the original strings. +fn lineage_literal(plan: &LogicalPlan, output_idx: usize) -> Option<&ScalarValue> { + if output_idx >= plan.schema().fields().len() { + return None; + } + + match plan { + LogicalPlan::Projection(projection) => match unalias(&projection.expr[output_idx]) { + Expr::Literal(value, _) => Some(value), + Expr::Column(column) => { + let input_idx = projection.input.schema().maybe_index_of_column(column)?; + lineage_literal(projection.input.as_ref(), input_idx) + } + _ => None, + }, + LogicalPlan::Filter(_) + | LogicalPlan::Sort(_) + | LogicalPlan::Limit(_) + | LogicalPlan::SubqueryAlias(_) + | LogicalPlan::Distinct(Distinct::All(_)) => { + let inputs = plan.inputs(); + let [input] = inputs.as_slice() else { + return None; + }; + lineage_literal(input, output_idx) + } + _ => None, + } +} + +fn convert_literal(value: &ScalarValue, unit: TimeUnit, timezone: &Timezone) -> Option { + let ScalarValue::Utf8(Some(value)) = value else { + return None; + }; + cast_string_to_timestamp(value, &DataType::Timestamp(unit, None), Some(timezone)) + .ok() + .filter(|value| !value.is_null()) + .map(|value| Expr::Literal(value, None)) +} + +fn unalias(expr: &Expr) -> &Expr { + match expr { + Expr::Alias(Alias { expr, .. }) => unalias(expr), + expr => expr, + } +} + +fn unalias_mut(expr: &mut Expr) -> &mut Expr { + match expr { + Expr::Alias(Alias { expr, .. }) => unalias_mut(expr), + expr => expr, + } +} + +#[cfg(test)] +mod tests { + use datafusion_common::DFSchema; + use datafusion_expr::expr::Placeholder; + + use super::*; + + fn shanghai() -> SessionTimezone { + let parsed = Timezone::from_tz_string("Asia/Shanghai").unwrap(); + SessionTimezone { + name: Arc::from(parsed.to_string()), + parsed, + } + } + + /// A prepared `INSERT ... VALUES (?)` arrives here as a cast over an untyped + /// placeholder, which must survive for parameter substitution. + #[test] + fn test_untyped_placeholder_assignment_is_left_alone() { + let schema = Arc::new(DFSchema::empty()); + let mut expr = Expr::Cast(Cast::new( + Box::new(Expr::Placeholder(Placeholder::new_with_field( + "$1".to_string(), + None, + ))), + DataType::Timestamp(TimeUnit::Millisecond, None), + )); + let original = expr.clone(); + + let changed = retarget_assignment_cast(&mut expr, &schema, None, &shanghai()).unwrap(); + + assert!(!changed); + assert_eq!(expr, original); + } +} diff --git a/src/query/src/optimizer/json_schema_concretize.rs b/src/query/src/optimizer/json_schema_concretize.rs new file mode 100644 index 0000000000..12f71d44c3 --- /dev/null +++ b/src/query/src/optimizer/json_schema_concretize.rs @@ -0,0 +1,208 @@ +// Copyright 2023 Greptime Team +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use std::collections::HashMap; +use std::sync::Arc; + +use datafusion::config::ConfigOptions; +use datafusion_common::tree_node::Transformed; +use datafusion_common::{DFSchema, DFSchemaRef, Result}; +use datafusion_expr::{LogicalPlan, UserDefinedLogicalNodeCore}; +use datafusion_optimizer::analyzer::AnalyzerRule; +use datatypes::extension::json::{ + Json2ExtensionType, is_json2_extension_type, is_legacy_json2_extension_type, +}; +use datatypes::types::json_type::JsonNativeType; + +use crate::dist_plan::MergeScanLogicalPlan; +use crate::optimizer::json_type_concretize::deduce_json_types; + +/// Keeps JSON2 schemas consistent across distributed query boundaries. +/// +/// An unresolved JSON2 column is represented as an empty `Struct`, while a remote stage always +/// emits a concrete Arrow array. DataFusion expects the schema declared by a logical node, the +/// schema used to build its physical plan, and the schema of its record batches to agree. This rule +/// gives each [`MergeScanLogicalPlan`] the concrete schema emitted by its remote stage and propagates +/// that schema through the local plan. +/// +/// For example: +/// +/// - `SELECT j FROM t` transfers the complete JSON2 value as `Binary` (`Variant`). A projection or +/// window above the MergeScan must therefore also describe `j` as `Binary`, not an empty `Struct`. +/// - `SELECT j.a FROM t` may transfer the complete `j` as `Binary` and extract `a` locally, or +/// transfer only the extracted scalar when the expression runs remotely. The boundary schema +/// must describe the remote output rather than the local expression that consumes it. +/// - `SELECT l.j, r.j FROM l JOIN r ON l.k = r.k` has two independent boundaries. Each input and the +/// join schema must agree on the concrete type of its JSON2 column. +/// +/// Correcting only the physical MergeScan schema can make simple queries work because many +/// operators access columns by position, but it leaves the logical plan describing a different +/// type. Keeping the schemas consistent lets optimizers, physical planners, validators, and future +/// type-aware operators rely on the normal DataFusion contract. It also keeps the generic physical +/// MergeScan implementation independent of JSON2. +/// +/// The boundary schema and the storage read layout answer different questions. For +/// `SELECT j.a::BIGINT FROM t`, the storage scan may use a structured `{a: Int64}` layout, while the +/// distributed boundary can still emit either the complete `j` as `Binary` or only `a` as `Int64`. +/// Therefore this rule cannot replace the separate JSON2 scan-type inference. +#[derive(Debug)] +pub(crate) struct JsonSchemaConcretizeRule; + +impl AnalyzerRule for JsonSchemaConcretizeRule { + fn analyze(&self, plan: LogicalPlan, _config: &ConfigOptions) -> Result { + let plan = plan.transform_up_with_subqueries(|plan| { + let LogicalPlan::Extension(mut extension) = plan else { + return Ok(Transformed::no(plan)); + }; + let Some(merge_scan) = extension + .node + .as_any() + .downcast_ref::() + else { + return Ok(Transformed::no(LogicalPlan::Extension(extension))); + }; + + // Infer the boundary schema from the hidden remote plan, not its local consumer. + let json_types = deduce_json_types(merge_scan.input())?; + if json_types.is_empty() { + return Ok(Transformed::no(LogicalPlan::Extension(extension))); + } + let schema = concretize_json2_schema(merge_scan.schema(), &json_types)?; + if schema.as_ref() == merge_scan.schema().as_ref() { + return Ok(Transformed::no(LogicalPlan::Extension(extension))); + } + + extension.node = Arc::new(merge_scan.clone().with_output_schema(schema)); + Ok(Transformed::yes(LogicalPlan::Extension(extension))) + })?; + + if plan.transformed { + plan.data + .transform_up_with_subqueries(|plan| { + if matches!(plan, LogicalPlan::Extension(_)) || plan.inputs().is_empty() { + Ok(Transformed::no(plan)) + } else { + plan.recompute_schema().map(Transformed::yes) + } + }) + .map(|x| x.data) + } else { + Ok(plan.data) + } + } + + fn name(&self) -> &str { + "JsonSchemaConcretizeRule" + } +} + +fn concretize_json2_schema( + schema: &DFSchemaRef, + json_types: &HashMap, +) -> Result { + if !schema + .iter() + .any(|(_, field)| json_types.contains_key(field.name()) && is_json2_extension_type(field)) + { + return Ok(schema.clone()); + } + + let mut changed = false; + let fields = schema + .iter() + .map(|(qualifier, field)| { + let Some(json_type) = json_types + .get(field.name()) + .filter(|_| is_json2_extension_type(field)) + else { + return (qualifier.cloned(), field.clone()); + }; + let data_type = json_type.as_arrow_type(); + if field.data_type() == &data_type { + return (qualifier.cloned(), field.clone()); + } + + changed = true; + + // Before type hints, JSON2 used the `greptime.json` marker together with + // `json_structure_settings`. Once concretized to `Binary`, that field no longer + // matches the legacy JSON2 shape and could be mistaken for JSONB, so upgrade its + // marker. Do not replace modern markers because that would discard their JSON + // settings and layout version. + let legacy = is_legacy_json2_extension_type(field); + let mut field = field.as_ref().clone().with_data_type(data_type); + if legacy { + field = field.with_extension_type(Json2ExtensionType::default()); + } + (qualifier.cloned(), Arc::new(field)) + }) + .collect(); + + if changed { + let schema = DFSchema::new_with_metadata(fields, schema.metadata().clone())? + .with_functional_dependencies(schema.functional_dependencies().clone())?; + Ok(Arc::new(schema)) + } else { + Ok(schema.clone()) + } +} + +#[cfg(test)] +mod tests { + use arrow_schema::extension::{ + EXTENSION_TYPE_METADATA_KEY, EXTENSION_TYPE_NAME_KEY, ExtensionType, + }; + use arrow_schema::{DataType, Field, Fields, Schema}; + use datafusion_common::DFSchema; + use datafusion_expr::{LogicalPlanBuilder, col}; + use datatypes::extension::json::{JsonExtensionType, is_json2_extension_type}; + + use super::*; + + #[test] + fn test_json_schema_concretize_rule_updates_merge_scan() -> Result<()> { + let field = Field::new("j", DataType::Struct(Fields::empty()), true).with_metadata( + HashMap::from([ + ( + EXTENSION_TYPE_NAME_KEY.to_string(), + JsonExtensionType::NAME.to_string(), + ), + ( + EXTENSION_TYPE_METADATA_KEY.to_string(), + serde_json::json!({ + "json_structure_settings": { "Structured": null } + }) + .to_string(), + ), + ]), + ); + let schema = Arc::new(DFSchema::try_from(Schema::new(vec![field]))?); + let input = LogicalPlan::EmptyRelation(datafusion_expr::logical_plan::EmptyRelation { + produce_one_row: false, + schema, + }); + let merge_scan = + MergeScanLogicalPlan::new(input, false, Default::default()).into_logical_plan(); + let plan = LogicalPlanBuilder::from(merge_scan) + .project(vec![col("j")])? + .build()?; + + let plan = JsonSchemaConcretizeRule.analyze(plan, &ConfigOptions::default())?; + let field = plan.schema().field(0); + assert_eq!(&DataType::Binary, field.data_type()); + assert_eq!(Some(Json2ExtensionType::NAME), field.extension_type_name()); + assert!(is_json2_extension_type(field)); + Ok(()) + } +} diff --git a/src/query/src/optimizer/json_type_concretize.rs b/src/query/src/optimizer/json_type_concretize.rs index 4625b61978..5ec45cd7f6 100644 --- a/src/query/src/optimizer/json_type_concretize.rs +++ b/src/query/src/optimizer/json_type_concretize.rs @@ -108,16 +108,36 @@ fn apply_json_type_hint( false } -fn deduce_json_types(plan: &LogicalPlan) -> Result> { +pub(crate) fn deduce_json_types(plan: &LogicalPlan) -> Result> { let mut json_types = HashMap::::new(); + // JSON2 columns in the final output must retain their complete values even when + // predicates or other expressions access only specific paths. + // For example, `SELECT j FROM t WHERE json_get(j, 'a') = 1`. + plan.schema() + .fields() + .iter() + .filter(|field| is_json2_extension_type(field)) + .for_each(|field| { + json_types.insert(field.name().clone(), JsonNativeType::Variant); + }); + plan.apply(|plan| { for expr in plan.expressions() { + // Optimizer-generated projections may keep the JSON root only so later json_get + // expressions can access another path. A same-name pass-through does not require the + // complete root by itself; any real whole-column consumer above it is visited + // separately, and a whole root in the final output is captured from the plan schema. + if matches!(plan, LogicalPlan::Projection(_)) && is_same_name_column_projection(&expr) { + continue; + } expr.apply(|expr| { if let Some((column, json_type)) = deduce_json_type(expr)? { json_types.entry(column).or_default().merge(&json_type); + Ok(TreeNodeRecursion::Jump) + } else { + Ok(TreeNodeRecursion::Continue) } - Ok(TreeNodeRecursion::Continue) })?; } Ok(TreeNodeRecursion::Continue) @@ -125,9 +145,20 @@ fn deduce_json_types(plan: &LogicalPlan) -> Result bool { + match expr { + Expr::Column(_) => true, + Expr::Alias(alias) => { + matches!(alias.expr.as_ref(), Expr::Column(column) if column.name == alias.name) + } + _ => false, + } +} + fn deduce_json_type(expr: &Expr) -> Result> { let f = match expr { Expr::ScalarFunction(f) if f.name().eq_ignore_ascii_case(JsonGetWithType::NAME) => f, + Expr::Column(c) => return Ok(Some((c.name.clone(), JsonNativeType::Variant))), _ => return Ok(None), }; @@ -153,6 +184,12 @@ fn deduce_json_type(expr: &Expr) -> Result> { ); }; + // Object-only type deduction cannot represent bracket JSONPath access, so preserve the + // full Variant and let json_get apply the expression. + if path.contains('[') { + return Ok(Some((column.name.clone(), JsonNativeType::Variant))); + } + let with_type = f .args .get(2) @@ -293,6 +330,17 @@ mod tests { Ok(()) } + #[test] + fn test_deduce_json_type_with_list_index() -> Result<()> { + let expr = json_get_expr(col("j"), path_expr("l[0]"), Some(DataType::Int64))?; + + assert_eq!( + Some(("j".to_string(), JsonNativeType::Variant)), + deduce_json_type(&expr)? + ); + Ok(()) + } + #[test] fn test_json_type_concretize_rule_conflict_to_variant() -> Result<()> { let exprs = vec![ @@ -360,25 +408,13 @@ mod tests { .rewrite(plan, &OptimizerContext::default())? .transformed ); - assert!(provider.scan_request().json_type_hint.contains_key("j")); - Ok(()) - } - - #[test] - fn test_allow_json2_passthrough_for_later_projection() -> Result<()> { - let json_get = json_get_expr(col("j"), path_expr("a"), Some(DataType::Int64))?; - let (provider, plan) = build_json2_scan()?; - let plan = plan - .project(vec![json_get.alias("__common_expr"), col("j")])? - .aggregate(Vec::::new(), vec![count(lit(1))])? - .build()?; - - assert!( - JsonTypeConcretizeRule - .rewrite(plan, &OptimizerContext::default())? - .transformed + assert_eq!( + Some(&JsonNativeType::Object(JsonObjectType::from([( + "a".to_string(), + JsonNativeType::i64(), + )]))), + provider.scan_request().json_type_hint.get("j") ); - assert!(provider.scan_request().json_type_hint.contains_key("j")); Ok(()) } @@ -402,6 +438,25 @@ mod tests { Ok(()) } + #[test] + fn test_allow_json2_filter_with_root_projection() -> Result<()> { + let predicate = + json_get_expr(col("j"), path_expr("a"), Some(DataType::Int64))?.eq(lit(1_i64)); + let (provider, plan) = build_json2_scan()?; + let plan = plan.filter(predicate)?.build()?; + + assert!( + JsonTypeConcretizeRule + .rewrite(plan, &OptimizerContext::default())? + .transformed + ); + assert_eq!( + Some(&JsonNativeType::Variant), + provider.scan_request().json_type_hint.get("j") + ); + Ok(()) + } + #[test] fn test_deduce_json_type_with_non_column_base() -> Result<()> { let expr = json_get_expr( diff --git a/src/query/src/optimizer/pass_distribution.rs b/src/query/src/optimizer/pass_distribution.rs index 8ff21e046d..c72975d23b 100644 --- a/src/query/src/optimizer/pass_distribution.rs +++ b/src/query/src/optimizer/pass_distribution.rs @@ -270,8 +270,8 @@ mod tests { panic!("expected right merge scan hash partitioning"); }; - assert_eq!(*left_count, 32); - assert_eq!(*right_count, 32); + assert_eq!(*left_count, 2); + assert_eq!(*right_count, 2); assert_eq!( column_names(left_exprs), vec![DATA_SCHEMA_TSID_COLUMN_NAME, "greptime_timestamp"] diff --git a/src/query/src/optimizer/type_conversion.rs b/src/query/src/optimizer/type_conversion.rs index 3941fedfe8..40bb43e791 100644 --- a/src/query/src/optimizer/type_conversion.rs +++ b/src/query/src/optimizer/type_conversion.rs @@ -12,8 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -mod insert_assignment; - use std::sync::Arc; use common_time::Timezone; @@ -31,7 +29,7 @@ use session::context::QueryContextRef; use crate::QueryEngineContext; use crate::optimizer::ExtensionAnalyzerRule; -use crate::optimizer::type_conversion::insert_assignment::rewrite_insert_assignments; +use crate::optimizer::insert_assignment::rewrite_insert_assignments; use crate::plan::ExtractExpr; /// TypeConversionRule converts some literal values in logical plan to other types according @@ -46,7 +44,7 @@ impl ExtensionAnalyzerRule for TypeConversionRule { &self, plan: LogicalPlan, ctx: &QueryEngineContext, - _config: &ConfigOptions, + config: &ConfigOptions, ) -> Result { plan.transform_up_with_subqueries(|plan| match plan { LogicalPlan::Filter(filter) => { @@ -129,7 +127,8 @@ impl ExtensionAnalyzerRule for TypeConversionRule { LogicalPlan::Dml(mut dml) if matches!(dml.op, WriteOp::Insert(_)) => { dml.input = Arc::new(rewrite_insert_assignments( dml.input.as_ref().clone(), - ctx.query_ctx(), + &ctx.query_ctx(), + config, )?); Ok(Transformed::yes(LogicalPlan::Dml(dml))) } @@ -324,7 +323,7 @@ fn timestamp_to_timestamp_ms_expr(val: i64, unit: TimeUnit) -> Expr { ) } -fn cast_string_to_timestamp( +pub(crate) fn cast_string_to_timestamp( string: &str, target_type: &DataType, timezone: Option<&Timezone>, diff --git a/src/query/src/optimizer/type_conversion/insert_assignment.rs b/src/query/src/optimizer/type_conversion/insert_assignment.rs deleted file mode 100644 index 11ab7e7829..0000000000 --- a/src/query/src/optimizer/type_conversion/insert_assignment.rs +++ /dev/null @@ -1,336 +0,0 @@ -// Copyright 2023 Greptime Team -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -use std::sync::Arc; - -use datafusion_common::{Column, Result, ScalarValue}; -use datafusion_expr::expr::{Alias, Cast}; -use datafusion_expr::{Distinct, Expr, ExprSchemable, LogicalPlan, Projection, Union, Values}; -use datatypes::arrow::datatypes::DataType; -use session::context::QueryContextRef; - -use crate::optimizer::type_conversion::cast_string_to_timestamp; -use crate::plan::ExtractExpr; - -/// Rewrites string literals that feed timestamp columns at an INSERT boundary. -/// Constants are folded at the assignment to avoid changing source types; -/// `VALUES` and `UNION` inputs are rewritten per row or branch. Explicit casts -/// stay on DataFusion's existing path. -pub(super) fn rewrite_insert_assignments( - plan: LogicalPlan, - query_ctx: QueryContextRef, -) -> Result { - let LogicalPlan::Projection(assignment) = plan else { - return Ok(plan); - }; - - let converter = InsertAssignmentConverter { query_ctx }; - let mut exprs = assignment.expr.clone(); - let mut input = assignment.input.as_ref().clone(); - let mut changed = false; - for (output_idx, expr) in assignment.expr.iter().enumerate() { - let target_type = assignment.schema.field(output_idx).data_type(); - if !matches!(target_type, DataType::Timestamp(_, _)) { - continue; - } - - let Some(column) = assignment_input_column(expr) else { - continue; - }; - let Some(input_idx) = input.schema().maybe_index_of_column(column) else { - continue; - }; - - let literal = lineage_literal(&input, input_idx).cloned(); - if let Some(literal) = literal - && let Some(folded) = converter.convert_literal(&literal, target_type) - { - let (qualifier, field) = assignment.schema.qualified_field(output_idx); - exprs[output_idx] = folded.alias_qualified(qualifier.cloned(), field.name()); - changed = true; - continue; - } - - // A hand-built DML plan (e.g. from flow) may share one source column - // between targets; rewriting it in place would retype every reader. - if assignment - .expr - .iter() - .enumerate() - .any(|(idx, other)| idx != output_idx && other.column_refs().contains(column)) - { - continue; - } - if let Some(rewritten) = converter.rewrite_output_column(&input, input_idx, target_type)? { - input = rewritten; - changed = true; - } - } - - if !changed { - return Ok(LogicalPlan::Projection(assignment)); - } - Projection::try_new(exprs, Arc::new(input)).map(LogicalPlan::Projection) -} - -/// Resolves a literal when every row carries the same value at `output_idx`. -fn lineage_literal(plan: &LogicalPlan, output_idx: usize) -> Option<&ScalarValue> { - if output_idx >= plan.schema().fields().len() { - return None; - } - - match plan { - LogicalPlan::Projection(projection) => match unalias(&projection.expr[output_idx]) { - Expr::Literal(value, _) => Some(value), - Expr::Column(column) => { - let input_idx = projection.input.schema().maybe_index_of_column(column)?; - lineage_literal(projection.input.as_ref(), input_idx) - } - _ => None, - }, - // These nodes drop, reorder or deduplicate rows without touching the - // value a surviving row carries, and their schema stays positional. - LogicalPlan::Filter(_) - | LogicalPlan::Sort(_) - | LogicalPlan::Limit(_) - | LogicalPlan::SubqueryAlias(_) - | LogicalPlan::Distinct(Distinct::All(_)) => { - let inputs = plan.inputs(); - let [input] = inputs.as_slice() else { - return None; - }; - lineage_literal(input, output_idx) - } - _ => None, - } -} - -struct InsertAssignmentConverter { - query_ctx: QueryContextRef, -} - -impl InsertAssignmentConverter { - fn rewrite_output_column( - &self, - plan: &LogicalPlan, - output_idx: usize, - target_type: &DataType, - ) -> Result> { - if output_idx >= plan.schema().fields().len() { - return Ok(None); - } - - match plan { - // DataFusion pushes INSERT assignment casts into Values, so inspect - // them even though the Values schema already has the target type. - LogicalPlan::Values(values) => self.rewrite_values(values, output_idx, target_type), - // Once the source query has produced the target type, its casts and - // coercions belong to the source query rather than INSERT assignment. - _ if plan.schema().field(output_idx).data_type() == target_type => Ok(None), - LogicalPlan::Projection(projection) => { - self.rewrite_projection(projection, output_idx, target_type) - } - LogicalPlan::Union(union) => self.rewrite_union(union, output_idx, target_type), - // Filter and Sort are excluded here: their predicates and keys read - // the retyped column, which would change the source query. Constants - // still reach the assignment through `lineage_literal`. Distinct::All - // holds no expressions; allowing it shifts dedup keys from raw - // strings to parsed instants. - LogicalPlan::Limit(_) - | LogicalPlan::SubqueryAlias(_) - | LogicalPlan::Distinct(Distinct::All(_)) => { - self.rewrite_passthrough(plan, output_idx, target_type) - } - _ => Ok(None), - } - } - - fn rewrite_projection( - &self, - projection: &Projection, - output_idx: usize, - target_type: &DataType, - ) -> Result> { - match unalias(&projection.expr[output_idx]) { - Expr::Literal(value, _) => { - let Some(converted) = self.convert_literal(value, target_type) else { - return Ok(None); - }; - let (qualifier, field) = projection.schema.qualified_field(output_idx); - let mut exprs = projection.expr.clone(); - exprs[output_idx] = converted.alias_qualified(qualifier.cloned(), field.name()); - - Projection::try_new(exprs, projection.input.clone()) - .map(LogicalPlan::Projection) - .map(Some) - } - Expr::Column(column) => { - // Retyping the input column affects every output column reading - // it, so only follow lineage with a single consumer. - if projection - .expr - .iter() - .enumerate() - .any(|(idx, other)| idx != output_idx && other.column_refs().contains(column)) - { - return Ok(None); - } - - let input = projection.input.as_ref(); - let Some(input_idx) = input.schema().maybe_index_of_column(column) else { - return Ok(None); - }; - let Some(rewritten) = self.rewrite_output_column(input, input_idx, target_type)? - else { - return Ok(None); - }; - - Projection::try_new(projection.expr.clone(), Arc::new(rewritten)) - .map(LogicalPlan::Projection) - .map(Some) - } - _ => Ok(None), - } - } - - fn rewrite_values( - &self, - values: &Values, - output_idx: usize, - target_type: &DataType, - ) -> Result> { - let mut rewritten = values.clone(); - let mut changed = false; - for row in &mut rewritten.values { - let Some(expr) = row.get_mut(output_idx) else { - return Ok(None); - }; - - if let Expr::Cast(Cast { - expr: inner, - data_type, - }) = expr - && data_type == target_type - && let Expr::Literal(value, _) = inner.as_ref() - && let Some(value) = self.convert_literal(value, target_type) - { - *expr = value; - changed = true; - continue; - } - - if expr.get_type(values.schema.as_ref()).ok().as_ref() != Some(target_type) { - return Ok(None); - } - } - - Ok(changed.then_some(LogicalPlan::Values(rewritten))) - } - - fn rewrite_union( - &self, - union: &Union, - output_idx: usize, - target_type: &DataType, - ) -> Result> { - let mut inputs = Vec::with_capacity(union.inputs.len()); - // A union has one shared schema, so rewrite it only when the same output - // column is an INSERT-assigned literal in every branch. - for input in &union.inputs { - let Some(rewritten) = self.rewrite_output_column(input, output_idx, target_type)? - else { - return Ok(None); - }; - inputs.push(Arc::new(rewritten)); - } - - // Untouched columns may still have mismatched branch types before - // TypeCoercion runs, so rebuild loosely like the SQL planner does. - Union::try_new_with_loose_types(inputs) - .map(LogicalPlan::Union) - .map(Some) - } - - fn rewrite_passthrough( - &self, - plan: &LogicalPlan, - output_idx: usize, - target_type: &DataType, - ) -> Result> { - let inputs = plan.inputs(); - let [input] = inputs.as_slice() else { - return Ok(None); - }; - let Some(rewritten_input) = self.rewrite_output_column(input, output_idx, target_type)? - else { - return Ok(None); - }; - - plan.with_new_exprs(plan.expressions_consider_join(), vec![rewritten_input]) - .map(Some) - } - - fn convert_literal(&self, value: &ScalarValue, target_type: &DataType) -> Option { - let ScalarValue::Utf8(Some(value)) = value else { - return None; - }; - cast_string_to_timestamp(value, target_type, Some(&self.query_ctx.timezone())) - .ok() - .filter(|value| !value.is_null()) - .map(|value| Expr::Literal(value, None)) - } -} - -fn assignment_input_column(expr: &Expr) -> Option<&Column> { - let expr = match unalias(expr) { - Expr::Cast(Cast { expr, .. }) => expr.as_ref(), - expr => expr, - }; - let Expr::Column(column) = expr else { - return None; - }; - Some(column) -} - -fn unalias(expr: &Expr) -> &Expr { - match expr { - Expr::Alias(Alias { expr, .. }) => unalias(expr), - expr => expr, - } -} - -#[cfg(test)] -mod tests { - use datafusion_common::ScalarValue; - use datafusion_common::arrow::datatypes::TimeUnit; - use session::context::QueryContext; - - use super::*; - - #[test] - fn test_convert_literal_falls_back_for_unsupported_literal() { - let converter = InsertAssignmentConverter { - query_ctx: QueryContext::arc(), - }; - let target_type = DataType::Timestamp(TimeUnit::Nanosecond, None); - - for literal in ["1970-01-01", "-8-01-01 00:00:01.5"] { - assert_eq!( - converter - .convert_literal(&ScalarValue::Utf8(Some(literal.to_string())), &target_type,), - None - ); - } - } -} diff --git a/src/query/src/planner.rs b/src/query/src/planner.rs index 09172340af..0851a76d6e 100644 --- a/src/query/src/planner.rs +++ b/src/query/src/planner.rs @@ -15,6 +15,7 @@ use std::any::Any; use std::borrow::Cow; use std::collections::{HashMap, HashSet}; +use std::ops::ControlFlow; use std::str::FromStr; use std::sync::Arc; @@ -34,7 +35,8 @@ use datafusion_expr::{ Analyze, Explain, ExplainFormat, Expr as DfExpr, LogicalPlan, LogicalPlanBuilder, PlanType, ToStringifiedPlan, col, }; -use datafusion_sql::planner::{ParserOptions, SqlToRel}; +use datafusion_sql::parser::Statement as DfStatement; +use datafusion_sql::planner::{IdentNormalizer, ParserOptions, SqlToRel}; use log_query::LogQuery; use promql_parser::parser::EvalStmt; use session::context::QueryContextRef; @@ -45,6 +47,7 @@ use sql::statements::explain::ExplainStatement; use sql::statements::query::Query; use sql::statements::statement::Statement; use sql::statements::tql::Tql; +use sqlparser::ast::{AccessExpr, Value, visit_expressions_mut}; use crate::error::{ CteColumnSchemaMismatchSnafu, PlanSqlSnafu, QueryPlanSnafu, Result, SqlSnafu, @@ -192,6 +195,13 @@ impl DfLogicalPlanner { } let mut df_stmt = stmt.as_ref().try_into().context(SqlSnafu)?; + normalize_field_access_after_subscript( + &mut df_stmt, + self.session_state + .config_options() + .sql_parser + .enable_ident_normalization, + ); // TODO(LFC): Remove this when Datafusion supports **both** the syntax and implementation of "explain with format". if let datafusion::sql::parser::Statement::Statement( @@ -595,6 +605,72 @@ impl DfLogicalPlanner { } } +/// Normalizes dot field accesses that follow a subscript for DataFusion. +/// +/// sqlparser represents `j.o.l[1].inner.l[2]` as a compound field access with +/// the following access chain: +/// +/// ```text +/// Dot(Identifier("o")), +/// Dot(Identifier("l")), +/// Subscript(1), +/// Dot(Identifier("inner")), +/// Dot(Identifier("l")), +/// Subscript(2) +/// ``` +/// +/// DataFusion first resolves the leading `j.o.l` through +/// `JsonExprPlanner::plan_compound_identifier`, which produces an untyped +/// `json_get` with path `o.l`. Before invoking `JsonExprPlanner::plan_field_access`, +/// however, DataFusion eagerly converts every remaining access into a +/// `GetFieldAccess`. It accepts string values but not [`SqlExpr::Identifier`]s +/// in [`AccessExpr::Dot`] after a subscript. Without this normalization, that +/// conversion fails at `.inner`, and `plan_field_access` is never called, even +/// for the preceding `[1]`. +/// +/// This function converts dot identifiers after the first subscript into +/// `Dot(Value(SingleQuotedString(...)))`, applying DataFusion's identifier +/// normalization before discarding whether each identifier was quoted. It +/// changes neither the SQL text nor the dot accesses into subscript nodes: the +/// resulting AST is conceptually `j.o.l[1].'inner'.'l'[2]`. DataFusion converts +/// the string-valued dot accesses into named field accesses, which +/// `plan_field_access` safely encodes as bracket members. It can then extend the +/// JSON path to `o.l[1]["inner"]["l"][2]`. +/// +/// This behavior is unchanged in the latest upstream releases checked here: +/// DataFusion 55.0.0 and sqlparser 0.62.0. +/// +/// TODO(LFC): Remove this workaround after upstream supports dot identifiers after subscripts. +fn normalize_field_access_after_subscript(stmt: &mut DfStatement, normalize_ident: bool) { + let DfStatement::Statement(stmt) = stmt else { + return; + }; + let normalizer = IdentNormalizer::new(normalize_ident); + + let _ = visit_expressions_mut(stmt.as_mut(), |expr| { + let SqlExpr::CompoundFieldAccess { access_chain, .. } = expr else { + return ControlFlow::<()>::Continue(()); + }; + let Some(index) = access_chain + .iter() + .position(|x| matches!(x, AccessExpr::Subscript(_))) + else { + return ControlFlow::Continue(()); + }; + + for access in &mut access_chain[index + 1..] { + let AccessExpr::Dot(SqlExpr::Identifier(ident)) = access else { + continue; + }; + let value = normalizer.normalize(ident.clone()); + *access = AccessExpr::Dot(SqlExpr::Value( + Value::SingleQuotedString(value).with_span(ident.span), + )); + } + ControlFlow::Continue(()) + }); +} + #[async_trait] impl LogicalPlanner for DfLogicalPlanner { #[tracing::instrument(skip_all)] @@ -852,6 +928,30 @@ mod tests { engine.planner().plan(&stmt, query_ctx).await.unwrap() } + /// Plans `sql` and runs the DataFusion analyzer, which is where + /// `InsertAssignmentRule` sits. Planning alone stops short of it, so these + /// assertions would not see the assignment rewrite at all. + async fn analyze_insert( + engine: &QueryEngineRef, + sql: &str, + query_ctx: &QueryContextRef, + ) -> String { + let stmt = QueryLanguageParser::parse_sql(sql, query_ctx).unwrap(); + let plan = engine + .planner() + .plan(&stmt, query_ctx.clone()) + .await + .unwrap(); + let context = engine.engine_context(query_ctx.clone()); + let state = context.state(); + state + .analyzer() + .execute_and_check(plan, state.config_options(), |_, _| {}) + .unwrap() + .display_indent_schema() + .to_string() + } + #[tokio::test] async fn test_insert_timestamp_literals_use_query_timezone() { let query_ctx = Arc::new( @@ -884,20 +984,6 @@ mod tests { ) AS source", &[1_785_902_400_001_i64][..], ), - ( - "INSERT INTO timestamps (ts, st) \ - SELECT '2026-08-06 12:00:00.001', now() \ - UNION ALL \ - SELECT '2026-08-07 12:00:00.001', now()", - &[1_785_988_800_001_i64, 1_786_075_200_001_i64][..], - ), - ( - "INSERT INTO timestamps (ts, st) \ - SELECT '2026-08-16 12:00:00.001', now() \ - UNION \ - SELECT '2026-08-17 12:00:00.001', now()", - &[1_786_852_800_001_i64, 1_786_939_200_001_i64][..], - ), ( "INSERT INTO timestamps (ts, st) \ SELECT '2026-08-18 12:00:00.001', max(st) \ @@ -926,14 +1012,7 @@ mod tests { &[1_786_766_400_001_i64][..], ), ] { - let stmt = QueryLanguageParser::parse_sql(sql, &query_ctx).unwrap(); - let plan = engine - .planner() - .plan(&stmt, query_ctx.clone()) - .await - .unwrap() - .display_indent() - .to_string(); + let plan = analyze_insert(&engine, sql, &query_ctx).await; for expected_timestamp in expected_timestamps { assert!( @@ -954,19 +1033,19 @@ mod tests { let engine = create_timestamp_test_engine().await; let sql = "INSERT INTO timestamps (ts, st) \ VALUES (CAST('2026-08-08 12:00:00.001' AS TIMESTAMP), now())"; - let stmt = QueryLanguageParser::parse_sql(sql, &query_ctx).unwrap(); - let plan = engine - .planner() - .plan(&stmt, query_ctx) - .await - .unwrap() - .display_indent() - .to_string(); + let plan = analyze_insert(&engine, sql, &query_ctx).await; + // An explicit cast reaches the analyzer as an `arrow_cast` call rather + // than an `Expr::Cast`, which is how it stays out of the rewrite. assert!( plan.contains("arrow_cast(Utf8(\"2026-08-08 12:00:00.001\")"), "{plan}" ); + // 12:00:00.001 read as Shanghai local time; the source query keeps UTC. + assert!( + !plan.contains("TimestampMillisecond(1786104000001, None)"), + "{plan}" + ); } #[tokio::test] @@ -1003,14 +1082,7 @@ mod tests { ][..], ), ] { - let stmt = QueryLanguageParser::parse_sql(sql, &query_ctx).unwrap(); - let plan = engine - .planner() - .plan(&stmt, query_ctx.clone()) - .await - .unwrap() - .display_indent_schema() - .to_string(); + let plan = analyze_insert(&engine, sql, &query_ctx).await; for expected in expected { assert!(plan.contains(expected), "{plan}"); @@ -1019,30 +1091,37 @@ mod tests { } #[tokio::test] - async fn test_insert_union_tolerates_uncoerced_untouched_column() { + async fn test_insert_union_converts_via_assignment_cast() { let query_ctx = Arc::new( QueryContextBuilder::default() .timezone(Timezone::from_tz_string("Asia/Shanghai").unwrap()) .build(), ); let engine = create_timestamp_test_engine().await; - // `st` stays Timestamp vs Null across branches until TypeCoercion runs. - let sql = "INSERT INTO timestamps (ts, st) \ - SELECT '2026-08-06 12:00:00.001', now() \ - UNION ALL \ - SELECT '2026-08-07 12:00:00.001', NULL"; - let stmt = QueryLanguageParser::parse_sql(sql, &query_ctx).unwrap(); - let plan = engine - .planner() - .plan(&stmt, query_ctx) - .await - .unwrap() - .display_indent() - .to_string(); + // Branches disagree, so the conversion stays as a cast on the + // assignment instead of folding. One cast covers every branch, which is + // why a NULL branch no longer cancels the conversion for the column and + // why UNION's dedup keys stay on the original strings. + for sql in [ + "INSERT INTO timestamps (ts, st) \ + SELECT '2026-08-06 12:00:00.001', now() \ + UNION ALL \ + SELECT '2026-08-07 12:00:00.001', NULL", + "INSERT INTO timestamps (ts, st) \ + SELECT '2026-08-16 12:00:00.001', now() \ + UNION \ + SELECT '2026-08-17 12:00:00.001', now()", + ] { + let plan = analyze_insert(&engine, sql, &query_ctx).await; - for expected_timestamp in [1_785_988_800_001_i64, 1_786_075_200_001_i64] { assert!( - plan.contains(&format!("TimestampMillisecond({expected_timestamp}, None)")), + plan.contains("AS Timestamp(ms, \"Asia/Shanghai\")"), + "{plan}" + ); + // The branches themselves are untouched. + assert!( + plan.contains("Utf8(\"2026-08-07 12:00:00.001\")") + || plan.contains("Utf8(\"2026-08-17 12:00:00.001\")"), "{plan}" ); } @@ -1060,14 +1139,8 @@ mod tests { SELECT '2026-08-10 12:00:00.001', now() \ UNION ALL \ SELECT CAST('2026-08-11 12:00:00.001' AS TIMESTAMP), now()"; - let stmt = QueryLanguageParser::parse_sql(sql, &query_ctx).unwrap(); - let plan = engine - .planner() - .plan(&stmt, query_ctx) - .await - .unwrap() - .display_indent_schema() - .to_string(); + let plan = analyze_insert(&engine, sql, &query_ctx).await; + assert!( !plan.contains("TimestampMillisecond(1786334400001, None)"), "{plan}" @@ -1076,6 +1149,11 @@ mod tests { plan.contains("arrow_cast(Utf8(\"2026-08-11 12:00:00.001\")"), "{plan}" ); + // TypeCoercion has already settled this union to timestamp, so the + // assignment has nothing left to reinterpret. Retargeting the cast here + // would leave a Timestamp(None) -> Timestamp(Some(tz)) step behind, + // which shifts the value instead of relabelling it. + assert!(!plan.contains("Asia/Shanghai"), "{plan}"); } #[tokio::test] diff --git a/src/query/src/promql/planner.rs b/src/query/src/promql/planner.rs index 3d2514ee9c..2ba73195d1 100644 --- a/src/query/src/promql/planner.rs +++ b/src/query/src/promql/planner.rs @@ -4925,12 +4925,20 @@ impl PromPlanner { F: FnMut(&String) -> Result, { let table_ref = self.ctx.table_name.clone().map(TableReference::bare); + // Derived labels can be unqualified even when the context still names the source table. + let input_schema = input.schema().clone(); let non_field_columns_iter = self .ctx .tag_columns .iter() .chain(self.ctx.time_index_column.iter()) - .map(|col| Ok(DfExpr::Column(Column::new(table_ref.clone(), col)))); + .map(|col| { + input_schema + .qualified_field_with_name(table_ref.as_ref(), col) + .or_else(|_| input_schema.qualified_field_with_unqualified_name(col)) + .map(|field| DfExpr::Column(field.into())) + .context(DataFusionPlanningSnafu) + }); let tsid_iter = Self::optional_tsid_projection(input.schema(), table_ref.as_ref(), self.ctx.use_tsid) .into_iter() @@ -8743,6 +8751,37 @@ Filter: up.field_0 IS NOT NULL [timestamp:Timestamp(ms), field_0:Float64;N, foo: assert_eq!(format!("\n{ret}"), expected, "\n{}", ret); } + #[tokio::test] + async fn label_replace_aggregation_queries_plan_successfully() { + let aggregate = + r#"sum by (foo) (label_replace(some_metric, "foo", "$1", "tag_0", "(.*)"))"#; + let queries = [ + aggregate.to_string(), + format!("{aggregate} <= 10"), + format!("{aggregate} * 0.8"), + format!("0.8 * {aggregate}"), + format!("{aggregate} <= {aggregate} * 0.8"), + ]; + let state = build_query_engine_state(); + let mut failures = Vec::new(); + + for query in queries { + let table_provider = build_test_table_provider( + &[(DEFAULT_SCHEMA_NAME.to_string(), "some_metric".to_string())], + 1, + 1, + ) + .await; + if let Err(error) = + PromPlanner::stmt_to_plan(table_provider, &build_eval_stmt(&query), &state).await + { + failures.push(format!("{query}: {error:?}")); + } + } + + assert!(failures.is_empty(), "{}", failures.join("\n")); + } + #[tokio::test] async fn test_matchers_to_expr() { let mut eval_stmt = EvalStmt { diff --git a/src/query/src/query_engine/state.rs b/src/query/src/query_engine/state.rs index 50898aaa58..e7bb6d6750 100644 --- a/src/query/src/query_engine/state.rs +++ b/src/query/src/query_engine/state.rs @@ -66,6 +66,7 @@ use crate::optimizer::constant_term::MatchesConstantTermOptimizer; use crate::optimizer::count_nest_aggr::CountNestAggrRule; use crate::optimizer::count_wildcard::CountWildcardToTimeIndexRule; use crate::optimizer::global_limit::EnsureGlobalLimitForFetch; +use crate::optimizer::json_schema_concretize::JsonSchemaConcretizeRule; use crate::optimizer::json_type_concretize::JsonTypeConcretizeRule; use crate::optimizer::parallelize_scan::ParallelizeScan; use crate::optimizer::pass_distribution::PassDistribution; @@ -204,6 +205,7 @@ impl QueryEngineState { if with_dist_planner { analyzer.rules.push(Arc::new(DistPlannerAnalyzer)); + analyzer.rules.push(Arc::new(JsonSchemaConcretizeRule)); } analyzer.rules.push(Arc::new(FixStateUdafOrderingAnalyzer)); diff --git a/src/query/src/sql.rs b/src/query/src/sql.rs index b2d979c666..93e74ca33e 100644 --- a/src/query/src/sql.rs +++ b/src/query/src/sql.rs @@ -41,6 +41,7 @@ use common_recordbatch::RecordBatches; use common_recordbatch::adapter::RecordBatchStreamAdapter; use common_time::Timestamp; use common_time::timezone::get_timezone; +use datafusion::dataframe::DataFrame; use datafusion::prelude::SessionContext; use datafusion_expr::{Expr, SortExpr, col, lit}; use datatypes::prelude::*; @@ -109,7 +110,7 @@ const INDEX_KEY_NAME_COLUMN: &str = "Key_name"; const INDEX_SEQ_IN_INDEX_COLUMN: &str = "Seq_in_index"; const INDEX_COLUMN_NAME_COLUMN: &str = "Column_name"; -static DESCRIBE_TABLE_OUTPUT_SCHEMA: Lazy> = Lazy::new(|| { +pub static DESCRIBE_TABLE_OUTPUT_SCHEMA: Lazy> = Lazy::new(|| { Arc::new(Schema::new(vec![ ColumnSchema::new( COLUMN_NAME_COLUMN, @@ -178,6 +179,18 @@ pub async fn show_databases( catalog_manager: &CatalogManagerRef, query_ctx: QueryContextRef, ) -> Result { + let dataframe = + show_databases_dataframe(&stmt, query_engine, catalog_manager, query_ctx).await?; + dataframe_to_output(dataframe).await +} + +/// Builds the [`DataFrame`] for `SHOW DATABASES` without executing it. +pub async fn show_databases_dataframe( + stmt: &ShowDatabases, + query_engine: &QueryEngineRef, + catalog_manager: &CatalogManagerRef, + query_ctx: QueryContextRef, +) -> Result { let projects = if stmt.full { vec![ (schemata::SCHEMA_NAME, SCHEMAS_COLUMN), @@ -191,7 +204,7 @@ pub async fn show_databases( let like_field = Some(schemata::SCHEMA_NAME); let sort = vec![col(schemata::SCHEMA_NAME).sort(true, true)]; - query_from_information_schema_table( + query_from_information_schema_dataframe( query_engine, catalog_manager, query_ctx, @@ -201,7 +214,7 @@ pub async fn show_databases( filters, like_field, sort, - stmt.kind, + &stmt.kind, ) .await } @@ -249,6 +262,45 @@ async fn query_from_information_schema_table( sort: Vec, kind: ShowKind, ) -> Result { + let dataframe = query_from_information_schema_dataframe( + query_engine, + catalog_manager, + query_ctx, + table_name, + select, + projects, + filters, + like_field, + sort, + &kind, + ) + .await?; + dataframe_to_output(dataframe).await +} + +async fn dataframe_to_output(dataframe: DataFrame) -> Result { + let stream = dataframe.execute_stream().await?; + Ok(Output::new_with_stream(Box::pin( + RecordBatchStreamAdapter::try_new(stream).context(error::CreateRecordBatchSnafu)?, + ))) +} + +/// Builds the [`DataFrame`] for a `SHOW` statement without executing it, +/// so `Describe` handlers can derive the output schema from the same +/// projection the executor uses. +#[allow(clippy::too_many_arguments)] +async fn query_from_information_schema_dataframe( + query_engine: &QueryEngineRef, + catalog_manager: &CatalogManagerRef, + query_ctx: QueryContextRef, + table_name: &str, + select: Vec, + projects: Vec<(&str, &str)>, + filters: Vec, + like_field: Option<&str>, + sort: Vec, + kind: &ShowKind, +) -> Result { let table = catalog_manager .table( query_ctx.current_catalog(), @@ -336,7 +388,7 @@ async fn query_from_information_schema_table( .expect("Must be the datafusion planner"); let filter = planner - .sql_to_expr(filter, dataframe.schema(), false, query_ctx) + .sql_to_expr(filter.clone(), dataframe.schema(), false, query_ctx) .await?; // Apply the `where` clause filters @@ -344,11 +396,7 @@ async fn query_from_information_schema_table( } }; - let stream = dataframe.execute_stream().await?; - - Ok(Output::new_with_stream(Box::pin( - RecordBatchStreamAdapter::try_new(stream).context(error::CreateRecordBatchSnafu)?, - ))) + Ok(dataframe) } /// Execute `SHOW COLUMNS` statement. @@ -358,8 +406,19 @@ pub async fn show_columns( catalog_manager: &CatalogManagerRef, query_ctx: QueryContextRef, ) -> Result { - let schema_name = if let Some(database) = stmt.database { - database + let dataframe = show_columns_dataframe(&stmt, query_engine, catalog_manager, query_ctx).await?; + dataframe_to_output(dataframe).await +} + +/// Builds the [`DataFrame`] for `SHOW COLUMNS` without executing it. +pub async fn show_columns_dataframe( + stmt: &ShowColumns, + query_engine: &QueryEngineRef, + catalog_manager: &CatalogManagerRef, + query_ctx: QueryContextRef, +) -> Result { + let schema_name = if let Some(database) = &stmt.database { + database.clone() } else { query_ctx.current_schema() }; @@ -397,7 +456,7 @@ pub async fn show_columns( let like_field = Some(columns::COLUMN_NAME); let sort = vec![col(columns::COLUMN_NAME).sort(true, true)]; - query_from_information_schema_table( + query_from_information_schema_dataframe( query_engine, catalog_manager, query_ctx, @@ -407,7 +466,7 @@ pub async fn show_columns( filters, like_field, sort, - stmt.kind, + &stmt.kind, ) .await } @@ -419,8 +478,19 @@ pub async fn show_index( catalog_manager: &CatalogManagerRef, query_ctx: QueryContextRef, ) -> Result { - let schema_name = if let Some(database) = stmt.database { - database + let dataframe = show_index_dataframe(&stmt, query_engine, catalog_manager, query_ctx).await?; + dataframe_to_output(dataframe).await +} + +/// Builds the [`DataFrame`] for `SHOW INDEX` without executing it. +pub async fn show_index_dataframe( + stmt: &ShowIndex, + query_engine: &QueryEngineRef, + catalog_manager: &CatalogManagerRef, + query_ctx: QueryContextRef, +) -> Result { + let schema_name = if let Some(database) = &stmt.database { + database.clone() } else { query_ctx.current_schema() }; @@ -471,7 +541,7 @@ pub async fn show_index( col(statistics::SEQ_IN_INDEX).sort(true, true), ]; - query_from_information_schema_table( + query_from_information_schema_dataframe( query_engine, catalog_manager, query_ctx, @@ -481,7 +551,7 @@ pub async fn show_index( filters, like_field, sort, - stmt.kind, + &stmt.kind, ) .await } @@ -493,8 +563,19 @@ pub async fn show_region( catalog_manager: &CatalogManagerRef, query_ctx: QueryContextRef, ) -> Result { - let schema_name = if let Some(database) = stmt.database { - database + let dataframe = show_region_dataframe(&stmt, query_engine, catalog_manager, query_ctx).await?; + dataframe_to_output(dataframe).await +} + +/// Builds the [`DataFrame`] for `SHOW REGION` without executing it. +pub async fn show_region_dataframe( + stmt: &ShowRegion, + query_engine: &QueryEngineRef, + catalog_manager: &CatalogManagerRef, + query_ctx: QueryContextRef, +) -> Result { + let schema_name = if let Some(database) = &stmt.database { + database.clone() } else { query_ctx.current_schema() }; @@ -517,7 +598,7 @@ pub async fn show_region( col(columns::PEER_ID).sort(true, true), ]; - query_from_information_schema_table( + query_from_information_schema_dataframe( query_engine, catalog_manager, query_ctx, @@ -527,7 +608,7 @@ pub async fn show_region( filters, like_field, sort, - stmt.kind, + &stmt.kind, ) .await } @@ -539,8 +620,19 @@ pub async fn show_tables( catalog_manager: &CatalogManagerRef, query_ctx: QueryContextRef, ) -> Result { - let schema_name = if let Some(database) = stmt.database { - database + let dataframe = show_tables_dataframe(&stmt, query_engine, catalog_manager, query_ctx).await?; + dataframe_to_output(dataframe).await +} + +/// Builds the [`DataFrame`] for [`ShowTables`] without executing it. +pub async fn show_tables_dataframe( + stmt: &ShowTables, + query_engine: &QueryEngineRef, + catalog_manager: &CatalogManagerRef, + query_ctx: QueryContextRef, +) -> Result { + let schema_name = if let Some(database) = &stmt.database { + database.clone() } else { query_ctx.current_schema() }; @@ -564,15 +656,16 @@ pub async fn show_tables( // Transform the WHERE clause for backward compatibility: // Replace "Tables" with "Tables_in_{schema}" to support old queries - let kind = match stmt.kind { - ShowKind::Where(mut filter) => { + let rewritten_kind = match &stmt.kind { + ShowKind::Where(filter) => { + let mut filter = filter.clone(); replace_column_in_expr(&mut filter, "Tables", &tables_column); ShowKind::Where(filter) } - other => other, + kind => kind.clone(), }; - query_from_information_schema_table( + query_from_information_schema_dataframe( query_engine, catalog_manager, query_ctx, @@ -582,7 +675,7 @@ pub async fn show_tables( filters, like_field, sort, - kind, + &rewritten_kind, ) .await } @@ -594,8 +687,20 @@ pub async fn show_table_status( catalog_manager: &CatalogManagerRef, query_ctx: QueryContextRef, ) -> Result { - let schema_name = if let Some(database) = stmt.database { - database + let dataframe = + show_table_status_dataframe(&stmt, query_engine, catalog_manager, query_ctx).await?; + dataframe_to_output(dataframe).await +} + +/// Builds the [`DataFrame`] for [`ShowTableStatus`] without executing it. +pub async fn show_table_status_dataframe( + stmt: &ShowTableStatus, + query_engine: &QueryEngineRef, + catalog_manager: &CatalogManagerRef, + query_ctx: QueryContextRef, +) -> Result { + let schema_name = if let Some(database) = &stmt.database { + database.clone() } else { query_ctx.current_schema() }; @@ -629,7 +734,7 @@ pub async fn show_table_status( let like_field = Some(tables::TABLE_NAME); let sort = vec![col(tables::TABLE_NAME).sort(true, true)]; - query_from_information_schema_table( + query_from_information_schema_dataframe( query_engine, catalog_manager, query_ctx, @@ -639,7 +744,7 @@ pub async fn show_table_status( filters, like_field, sort, - stmt.kind, + &stmt.kind, ) .await } @@ -651,6 +756,18 @@ pub async fn show_collations( catalog_manager: &CatalogManagerRef, query_ctx: QueryContextRef, ) -> Result { + let dataframe = + show_collations_dataframe(&kind, query_engine, catalog_manager, query_ctx).await?; + dataframe_to_output(dataframe).await +} + +/// Builds the [`DataFrame`] for `SHOW COLLATION` without executing it. +pub async fn show_collations_dataframe( + kind: &ShowKind, + query_engine: &QueryEngineRef, + catalog_manager: &CatalogManagerRef, + query_ctx: QueryContextRef, +) -> Result { // Refer to https://dev.mysql.com/doc/refman/8.0/en/show-collation.html let projects = vec![ ("collation_name", "Collation"), @@ -665,7 +782,7 @@ pub async fn show_collations( let like_field = Some("collation_name"); let sort = vec![]; - query_from_information_schema_table( + query_from_information_schema_dataframe( query_engine, catalog_manager, query_ctx, @@ -687,6 +804,18 @@ pub async fn show_charsets( catalog_manager: &CatalogManagerRef, query_ctx: QueryContextRef, ) -> Result { + let dataframe = + show_charsets_dataframe(&kind, query_engine, catalog_manager, query_ctx).await?; + dataframe_to_output(dataframe).await +} + +/// Builds the [`DataFrame`] for `SHOW CHARSET` without executing it. +pub async fn show_charsets_dataframe( + kind: &ShowKind, + query_engine: &QueryEngineRef, + catalog_manager: &CatalogManagerRef, + query_ctx: QueryContextRef, +) -> Result { // Refer to https://dev.mysql.com/doc/refman/8.0/en/show-character-set.html let projects = vec![ ("character_set_name", "Charset"), @@ -699,7 +828,7 @@ pub async fn show_charsets( let like_field = Some("character_set_name"); let sort = vec![]; - query_from_information_schema_table( + query_from_information_schema_dataframe( query_engine, catalog_manager, query_ctx, @@ -916,8 +1045,19 @@ pub async fn show_views( catalog_manager: &CatalogManagerRef, query_ctx: QueryContextRef, ) -> Result { - let schema_name = if let Some(database) = stmt.database { - database + let dataframe = show_views_dataframe(&stmt, query_engine, catalog_manager, query_ctx).await?; + dataframe_to_output(dataframe).await +} + +/// Builds the [`DataFrame`] for [`ShowViews`] without executing it. +pub async fn show_views_dataframe( + stmt: &ShowViews, + query_engine: &QueryEngineRef, + catalog_manager: &CatalogManagerRef, + query_ctx: QueryContextRef, +) -> Result { + let schema_name = if let Some(database) = &stmt.database { + database.clone() } else { query_ctx.current_schema() }; @@ -930,7 +1070,7 @@ pub async fn show_views( let like_field = Some(tables::TABLE_NAME); let sort = vec![col(tables::TABLE_NAME).sort(true, true)]; - query_from_information_schema_table( + query_from_information_schema_dataframe( query_engine, catalog_manager, query_ctx, @@ -940,7 +1080,7 @@ pub async fn show_views( filters, like_field, sort, - stmt.kind, + &stmt.kind, ) .await } @@ -952,12 +1092,23 @@ pub async fn show_flows( catalog_manager: &CatalogManagerRef, query_ctx: QueryContextRef, ) -> Result { + let dataframe = show_flows_dataframe(&stmt, query_engine, catalog_manager, query_ctx).await?; + dataframe_to_output(dataframe).await +} + +/// Builds the [`DataFrame`] for [`ShowFlows`] without executing it. +pub async fn show_flows_dataframe( + stmt: &ShowFlows, + query_engine: &QueryEngineRef, + catalog_manager: &CatalogManagerRef, + query_ctx: QueryContextRef, +) -> Result { let projects = vec![(flows::FLOW_NAME, FLOWS_COLUMN)]; let filters = vec![col(flows::TABLE_CATALOG).eq(lit(query_ctx.current_catalog()))]; let like_field = Some(flows::FLOW_NAME); let sort = vec![col(flows::FLOW_NAME).sort(true, true)]; - query_from_information_schema_table( + query_from_information_schema_dataframe( query_engine, catalog_manager, query_ctx, @@ -967,7 +1118,7 @@ pub async fn show_flows( filters, like_field, sort, - stmt.kind, + &stmt.kind, ) .await } @@ -1340,6 +1491,18 @@ pub async fn show_processlist( catalog_manager: &CatalogManagerRef, query_ctx: QueryContextRef, ) -> Result { + let dataframe = + show_processlist_dataframe(&stmt, query_engine, catalog_manager, query_ctx).await?; + dataframe_to_output(dataframe).await +} + +/// Builds the [`DataFrame`] for `SHOW PROCESSLIST` without executing it. +pub async fn show_processlist_dataframe( + stmt: &ShowProcessList, + query_engine: &QueryEngineRef, + catalog_manager: &CatalogManagerRef, + query_ctx: QueryContextRef, +) -> Result { let projects = if stmt.full { vec![ (process_list::ID, "Id"), @@ -1367,17 +1530,17 @@ pub async fn show_processlist( }; let like_field = None; let sort = vec![col("id").sort(true, true)]; - query_from_information_schema_table( + query_from_information_schema_dataframe( query_engine, catalog_manager, query_ctx.clone(), "process_list", vec![], - projects.clone(), + projects, filters, like_field, sort, - ShowKind::All, + &ShowKind::All, ) .await } diff --git a/src/servers/Cargo.toml b/src/servers/Cargo.toml index c5e98d789a..3114713e40 100644 --- a/src/servers/Cargo.toml +++ b/src/servers/Cargo.toml @@ -166,8 +166,8 @@ serde_json.workspace = true session = { workspace = true, features = ["testing"] } table.workspace = true tempfile = "3.0.0" -tokio-postgres = "0.7" -tokio-postgres-rustls = "0.12" +tokio-postgres.workspace = true +tokio-postgres-rustls = "0.14" [target.'cfg(unix)'.dev-dependencies] pprof = { version = "0.14", features = ["criterion", "flamegraph"] } diff --git a/src/servers/src/grpc/flight.rs b/src/servers/src/grpc/flight.rs index 97a579d952..e6f35e17b2 100644 --- a/src/servers/src/grpc/flight.rs +++ b/src/servers/src/grpc/flight.rs @@ -53,7 +53,7 @@ use tokio_stream::wrappers::ReceiverStream; use tonic::{Request, Response, Status, Streaming}; use crate::error::{InvalidParameterSnafu, Result, ToJsonSnafu}; -pub use crate::grpc::flight::stream::FlightRecordBatchStream; +pub use crate::grpc::flight::stream::{FlightRecordBatchSource, FlightRecordBatchStream}; use crate::grpc::greptime_handler::{ GreptimeRequestHandler, create_query_context, get_request_type, }; @@ -583,7 +583,7 @@ fn to_flight_data_stream( match output.data { OutputData::Stream(stream) => { let stream = FlightRecordBatchStream::new( - stream, + FlightRecordBatchSource::RecordBatches(stream), tracing_context, flight_compression, query_ctx, @@ -592,7 +592,7 @@ fn to_flight_data_stream( } OutputData::RecordBatches(x) => { let stream = FlightRecordBatchStream::new( - x.as_stream(), + FlightRecordBatchSource::RecordBatches(x.as_stream()), tracing_context, flight_compression, query_ctx, diff --git a/src/servers/src/grpc/flight/stream.rs b/src/servers/src/grpc/flight/stream.rs index c222c72e1c..319a797d9c 100644 --- a/src/servers/src/grpc/flight/stream.rs +++ b/src/servers/src/grpc/flight/stream.rs @@ -13,6 +13,7 @@ // limitations under the License. use std::collections::VecDeque; +use std::future::Future; use std::pin::Pin; use std::task::{Context, Poll}; use std::time::{Duration, Instant}; @@ -40,6 +41,30 @@ use crate::error; use crate::grpc::FlightCompression; use crate::grpc::flight::TonicResult; +pub type FlightRecordBatchStreamInitializer = Pin< + Box< + dyn Future> + + Send + + 'static, + >, +>; + +pub enum FlightRecordBatchSource { + RecordBatches(SendableRecordBatchStream), + Initializer(FlightRecordBatchStreamInitializer), +} + +impl FlightRecordBatchSource { + pub fn initializer(initializer: F) -> Self + where + F: Future> + + Send + + 'static, + { + Self::Initializer(Box::pin(initializer)) + } +} + /// Metrics collector for Flight stream with RAII logging pattern struct StreamMetrics { send_schema_duration: Duration, @@ -136,7 +161,7 @@ impl FlightRecordBatchStream { } pub fn new( - recordbatches: SendableRecordBatchStream, + source: FlightRecordBatchSource, tracing_context: TracingContext, compression: FlightCompression, query_ctx: QueryContextRef, @@ -148,17 +173,43 @@ impl FlightRecordBatchStream { .remote_query_id() .zip(query_ctx.extension(SUPPORT_FLIGHT_METRICS_BEFORE_BATCH_EXTENSION_KEY)) .is_some_and(|(remote_query_id, capability)| capability == remote_query_id); - let (tx, rx) = mpsc::channel::>(1); - let join_handle = common_runtime::spawn_global(async move { - Self::flight_data_stream( - recordbatches, - tx, - should_send_partial_metrics, - can_send_metrics_before_batch, - ) - .trace(tracing_context.attach(info_span!("flight_data_stream"))) - .await - }); + let (mut tx, rx) = mpsc::channel::>(1); + let source_type = match &source { + FlightRecordBatchSource::RecordBatches(_) => "record_batches", + FlightRecordBatchSource::Initializer(_) => "initializer", + }; + let initializer_tracing_context = tracing_context.clone(); + let join_handle = common_runtime::spawn_global( + async move { + let recordbatches = async move { + match source { + FlightRecordBatchSource::RecordBatches(recordbatches) => Ok(recordbatches), + FlightRecordBatchSource::Initializer(initializer) => initializer.await, + } + } + .trace( + initializer_tracing_context + .attach(info_span!("flight_data_stream_init", source_type)), + ) + .await; + + match recordbatches { + Ok(recordbatches) => { + Self::flight_data_stream( + recordbatches, + tx, + should_send_partial_metrics, + can_send_metrics_before_batch, + ) + .await; + } + Err(status) => { + let _ = tx.send(Err(status)).await; + } + } + } + .trace(tracing_context.attach(info_span!("flight_data_stream"))), + ); let encoder = if compression.arrow_compression() { FlightEncoder::default() } else { @@ -434,7 +485,7 @@ mod test { .unwrap() .as_stream(); let mut stream = FlightRecordBatchStream::new( - recordbatches, + FlightRecordBatchSource::RecordBatches(recordbatches), TracingContext::default(), FlightCompression::default(), QueryContext::arc(), @@ -468,6 +519,23 @@ mod test { } } + #[tokio::test] + async fn test_flight_record_batch_stream_forwards_initializer_error() { + let mut stream = FlightRecordBatchStream::new( + FlightRecordBatchSource::initializer(async { + Err(tonic::Status::unavailable( + "remote read initialization failed", + )) + }), + TracingContext::default(), + FlightCompression::default(), + QueryContext::arc(), + ); + + let error = stream.next().await.unwrap().unwrap_err(); + assert_eq!(tonic::Code::Unavailable, error.code()); + assert!(stream.next().await.is_none()); + } #[tokio::test] async fn test_flight_record_batch_stream_emits_metrics_while_pending() { let schema = Arc::new(Schema::new(vec![ColumnSchema::new( @@ -486,7 +554,7 @@ mod test { let query_ctx = query_context_with_live_metrics_and_matching_capability(); query_ctx.set_explain_verbose(true); let mut stream = FlightRecordBatchStream::new( - recordbatches, + FlightRecordBatchSource::RecordBatches(recordbatches), TracingContext::default(), FlightCompression::default(), query_ctx, @@ -541,7 +609,7 @@ mod test { let query_ctx = query_context_with_live_metrics_and_matching_capability(); query_ctx.set_explain_verbose(true); let mut stream = FlightRecordBatchStream::new( - recordbatches, + FlightRecordBatchSource::RecordBatches(recordbatches), TracingContext::default(), FlightCompression::default(), query_ctx, @@ -595,7 +663,7 @@ mod test { let query_ctx = query_context_with_matching_capability(); query_ctx.set_explain_verbose(true); let mut stream = FlightRecordBatchStream::new( - recordbatches, + FlightRecordBatchSource::RecordBatches(recordbatches), TracingContext::default(), FlightCompression::default(), query_ctx, @@ -638,7 +706,7 @@ mod test { let query_ctx = Arc::new(query_ctx); query_ctx.set_explain_verbose(true); let mut stream = FlightRecordBatchStream::new( - recordbatches, + FlightRecordBatchSource::RecordBatches(recordbatches), TracingContext::default(), FlightCompression::default(), query_ctx, @@ -677,7 +745,7 @@ mod test { }); let query_ctx = query_context_with_matching_capability(); let mut stream = FlightRecordBatchStream::new( - recordbatches, + FlightRecordBatchSource::RecordBatches(recordbatches), TracingContext::default(), FlightCompression::default(), query_ctx, diff --git a/src/servers/src/mysql/handler.rs b/src/servers/src/mysql/handler.rs index b0eb33f8fd..30579d8846 100644 --- a/src/servers/src/mysql/handler.rs +++ b/src/servers/src/mysql/handler.rs @@ -25,6 +25,7 @@ use common_catalog::parse_optional_catalog_and_schema_from_db_string; use common_error::ext::ErrorExt; use common_query::Output; use common_telemetry::{debug, error, tracing, warn}; +use common_time::Timezone; use datafusion_common::ParamValues; use datafusion_expr::LogicalPlan; use datatypes::prelude::ConcreteDataType; @@ -289,12 +290,13 @@ impl MysqlInstanceShim { .fail(); } + let timezone = query_ctx.timezone(); let replaced_plan = match params { Params::ProtocolParams(params) => { - replace_params_with_values(&plan, param_types, ¶ms) + replace_params_with_values(&plan, param_types, ¶ms, &timezone) } Params::CliParams(params) => { - replace_params_with_exprs(&plan, param_types, ¶ms) + replace_params_with_exprs(&plan, param_types, ¶ms, &timezone) } }?; @@ -807,6 +809,7 @@ fn replace_params_with_values( plan: &LogicalPlan, param_types: HashMap>, params: &[ParamValue], + timezone: &Timezone, ) -> Result { debug_assert_eq!(param_types.len(), params.len()); @@ -824,7 +827,7 @@ fn replace_params_with_values( for (i, param) in params.iter().enumerate() { if let Some(Some(t)) = param_types.get(&format_placeholder(i + 1)) { - let value = helper::convert_value(param, t)?; + let value = helper::convert_value(param, t, timezone)?; values.push(value.into()); } @@ -839,6 +842,7 @@ fn replace_params_with_exprs( plan: &LogicalPlan, param_types: HashMap>, params: &[sql::ast::Expr], + timezone: &Timezone, ) -> Result { debug_assert_eq!(param_types.len(), params.len()); @@ -853,7 +857,7 @@ fn replace_params_with_exprs( for (i, param) in params.iter().enumerate() { if let Some(Some(t)) = param_types.get(&format_placeholder(i + 1)) { - let value = helper::convert_expr_to_scalar_value(param, t)?; + let value = helper::convert_expr_to_scalar_value(param, t, timezone)?; values.push(value.into()); } diff --git a/src/servers/src/mysql/helper.rs b/src/servers/src/mysql/helper.rs index a4f289cdfb..f9ab532592 100644 --- a/src/servers/src/mysql/helper.rs +++ b/src/servers/src/mysql/helper.rs @@ -15,10 +15,11 @@ use std::ops::ControlFlow; use std::time::Duration; -use chrono::NaiveDate; +use chrono::{NaiveDate, NaiveDateTime}; use common_query::prelude::ScalarValue; use common_sql::convert::sql_value_to_value; -use common_time::{Date, Timestamp}; +use common_time::timestamp::TimeUnit; +use common_time::{Date, Timestamp, Timezone}; use datatypes::prelude::{ConcreteDataType, DataType}; use datatypes::schema::ColumnSchema; use datatypes::types::TimestampType; @@ -137,11 +138,15 @@ where /// Convert [`ParamValue`] into [`Value`] according to param type. /// It will try it's best to do type conversions if possible -pub fn convert_value(param: &ParamValue, t: &ConcreteDataType) -> Result { +pub fn convert_value( + param: &ParamValue, + t: &ConcreteDataType, + timezone: &Timezone, +) -> Result { if let ConcreteDataType::Dictionary(dictionary) = t { return Ok(ScalarValue::Dictionary( Box::new(dictionary.key_type().as_arrow_type()), - Box::new(convert_value(param, dictionary.value_type())?), + Box::new(convert_value(param, dictionary.value_type(), timezone)?), )); } @@ -221,7 +226,9 @@ pub fn convert_value(param: &ParamValue, t: &ConcreteDataType) -> Result Ok(ScalarValue::Binary(Some(b.to_vec()))), - ConcreteDataType::Timestamp(ts_type) => convert_bytes_to_timestamp(b, ts_type), + ConcreteDataType::Timestamp(ts_type) => { + convert_bytes_to_timestamp(b, ts_type, timezone) + } ConcreteDataType::Date(_) => convert_bytes_to_date(b), _ => error::PreparedStmtTypeMismatchSnafu { expected: t, @@ -233,29 +240,26 @@ pub fn convert_value(param: &ParamValue, t: &ConcreteDataType) -> Result { - let timestamp_millis = to_naive_datetime(param.value) - .map_err(|e| { + ValueInner::Datetime(_) => match t { + ConcreteDataType::Timestamp(_) => { + let datetime = to_naive_datetime(param.value).map_err(|e| { error::MysqlValueConversionSnafu { err_msg: e.to_string(), } .build() - })? - .and_utc() - .timestamp_millis(); - - match t { - ConcreteDataType::Timestamp(_) => Ok(ScalarValue::TimestampMillisecond( + })?; + let timestamp_millis = datetime_to_timestamp_millis(datetime, timezone)?; + Ok(ScalarValue::TimestampMillisecond( Some(timestamp_millis), None, - )), - _ => error::PreparedStmtTypeMismatchSnafu { - expected: t, - actual: param.coltype, - } - .fail(), + )) } - } + _ => error::PreparedStmtTypeMismatchSnafu { + expected: t, + actual: param.coltype, + } + .fail(), + }, ValueInner::Time(_) => Ok(ScalarValue::Time64Nanosecond(Some( Duration::from(param.value).as_millis() as i64, ))), @@ -264,13 +268,18 @@ pub fn convert_value(param: &ParamValue, t: &ConcreteDataType) -> Result Result { +pub fn convert_expr_to_scalar_value( + param: &Expr, + t: &ConcreteDataType, + timezone: &Timezone, +) -> Result { if let ConcreteDataType::Dictionary(dictionary) = t { return Ok(ScalarValue::Dictionary( Box::new(dictionary.key_type().as_arrow_type()), Box::new(convert_expr_to_scalar_value( param, dictionary.value_type(), + timezone, )?), )); } @@ -278,7 +287,7 @@ pub fn convert_expr_to_scalar_value(param: &Expr, t: &ConcreteDataType) -> Resul let column_schema = ColumnSchema::new("", t.clone(), true); match param { Expr::Value(v) => { - let v = sql_value_to_value(&column_schema, &v.value, None, None, true); + let v = sql_value_to_value(&column_schema, &v.value, Some(timezone), None, true); match v { Ok(v) => v .try_to_scalar_value(t) @@ -290,7 +299,7 @@ pub fn convert_expr_to_scalar_value(param: &Expr, t: &ConcreteDataType) -> Resul } } Expr::UnaryOp { op, expr } if let Expr::Value(v) = &**expr => { - let v = sql_value_to_value(&column_schema, &v.value, None, Some(*op), true); + let v = sql_value_to_value(&column_schema, &v.value, Some(timezone), Some(*op), true); match v { Ok(v) => v .try_to_scalar_value(t) @@ -308,8 +317,32 @@ pub fn convert_expr_to_scalar_value(param: &Expr, t: &ConcreteDataType) -> Resul } } -fn convert_bytes_to_timestamp(bytes: &[u8], ts_type: &TimestampType) -> Result { - let ts = Timestamp::from_str_utc(&String::from_utf8_lossy(bytes)) +/// Interprets a timezone-less datetime in the given timezone and returns the +/// corresponding epoch timestamp in milliseconds. +fn datetime_to_timestamp_millis(datetime: NaiveDateTime, timezone: &Timezone) -> Result { + let ts = Timestamp::from_naive_datetime(datetime, timezone) + .map_err(|e| { + error::MysqlValueConversionSnafu { + err_msg: e.to_string(), + } + .build() + })? + .convert_to(TimeUnit::Millisecond) + .ok_or_else(|| { + error::MysqlValueConversionSnafu { + err_msg: "Overflow when converting datetime to milliseconds".to_string(), + } + .build() + })?; + Ok(ts.value()) +} + +fn convert_bytes_to_timestamp( + bytes: &[u8], + ts_type: &TimestampType, + timezone: &Timezone, +) -> Result { + let ts = Timestamp::from_str(&String::from_utf8_lossy(bytes), Some(timezone)) .map_err(|e| { error::MysqlValueConversionSnafu { err_msg: e.to_string(), @@ -446,19 +479,20 @@ mod tests { #[test] fn test_convert_expr_to_scalar_value() { + let utc = Timezone::from_tz_string("UTC").unwrap(); let expr = Expr::Value(ValueExpr::Number("123".to_string(), false).into()); let t = ConcreteDataType::int32_datatype(); - let v = convert_expr_to_scalar_value(&expr, &t).unwrap(); + let v = convert_expr_to_scalar_value(&expr, &t, &utc).unwrap(); assert_eq!(ScalarValue::Int32(Some(123)), v); let expr = Expr::Value(ValueExpr::Number("123.456789".to_string(), false).into()); let t = ConcreteDataType::float64_datatype(); - let v = convert_expr_to_scalar_value(&expr, &t).unwrap(); + let v = convert_expr_to_scalar_value(&expr, &t, &utc).unwrap(); assert_eq!(ScalarValue::Float64(Some(123.456789)), v); let expr = Expr::Value(ValueExpr::SingleQuotedString("2001-01-02".to_string()).into()); let t = ConcreteDataType::date_datatype(); - let v = convert_expr_to_scalar_value(&expr, &t).unwrap(); + let v = convert_expr_to_scalar_value(&expr, &t, &utc).unwrap(); let scalar_v = ScalarValue::Utf8(Some("2001-01-02".to_string())) .cast_to(&arrow_schema::DataType::Date32) .unwrap(); @@ -467,7 +501,7 @@ mod tests { let expr = Expr::Value(ValueExpr::SingleQuotedString("2001-01-02 03:04:05".to_string()).into()); let t = ConcreteDataType::timestamp_microsecond_datatype(); - let v = convert_expr_to_scalar_value(&expr, &t).unwrap(); + let v = convert_expr_to_scalar_value(&expr, &t, &utc).unwrap(); let scalar_v = ScalarValue::Utf8(Some("2001-01-02 03:04:05".to_string())) .cast_to(&arrow_schema::DataType::Timestamp( arrow_schema::TimeUnit::Microsecond, @@ -478,14 +512,14 @@ mod tests { let expr = Expr::Value(ValueExpr::SingleQuotedString("hello".to_string()).into()); let t = ConcreteDataType::string_datatype(); - let v = convert_expr_to_scalar_value(&expr, &t).unwrap(); + let v = convert_expr_to_scalar_value(&expr, &t, &utc).unwrap(); assert_eq!(ScalarValue::Utf8(Some("hello".to_string())), v); let t = ConcreteDataType::dictionary_datatype( ConcreteDataType::uint32_datatype(), ConcreteDataType::string_datatype(), ); - let v = convert_expr_to_scalar_value(&expr, &t).unwrap(); + let v = convert_expr_to_scalar_value(&expr, &t, &utc).unwrap(); assert_eq!( ScalarValue::Dictionary( Box::new(arrow_schema::DataType::UInt32), @@ -496,7 +530,7 @@ mod tests { let expr = Expr::Value(ValueExpr::Null.into()); let t = ConcreteDataType::time_microsecond_datatype(); - let v = convert_expr_to_scalar_value(&expr, &t).unwrap(); + let v = convert_expr_to_scalar_value(&expr, &t, &utc).unwrap(); assert_eq!(ScalarValue::Time64Microsecond(None), v); } @@ -577,12 +611,81 @@ mod tests { ), ]; + let utc = Timezone::from_tz_string("UTC").unwrap(); for (input, ts_type, expected) in test_cases { - let result = convert_bytes_to_timestamp(input.as_bytes(), &ts_type).unwrap(); + let result = convert_bytes_to_timestamp(input.as_bytes(), &ts_type, &utc).unwrap(); assert_eq!(result, expected); } } + fn utc_millis(year: i32, month: u32, day: u32, hour: u32, minute: u32, second: u32) -> i64 { + NaiveDate::from_ymd_opt(year, month, day) + .unwrap() + .and_hms_opt(hour, minute, second) + .unwrap() + .and_utc() + .timestamp_millis() + } + + #[test] + fn test_datetime_to_timestamp_millis() { + let datetime = NaiveDate::from_ymd_opt(2026, 8, 13) + .unwrap() + .and_hms_opt(8, 0, 0) + .unwrap(); + + let utc = Timezone::from_tz_string("UTC").unwrap(); + assert_eq!( + datetime_to_timestamp_millis(datetime, &utc).unwrap(), + utc_millis(2026, 8, 13, 8, 0, 0) + ); + + let shanghai = Timezone::from_tz_string("Asia/Shanghai").unwrap(); + assert_eq!( + datetime_to_timestamp_millis(datetime, &shanghai).unwrap(), + utc_millis(2026, 8, 13, 0, 0, 0) + ); + + let minus_7 = Timezone::from_tz_string("-07:00").unwrap(); + assert_eq!( + datetime_to_timestamp_millis(datetime, &minus_7).unwrap(), + utc_millis(2026, 8, 13, 15, 0, 0) + ); + + // 2026-03-08 02:30 does not exist in America/New_York (DST gap). + let new_york = Timezone::from_tz_string("America/New_York").unwrap(); + let gap = NaiveDate::from_ymd_opt(2026, 3, 8) + .unwrap() + .and_hms_opt(2, 30, 0) + .unwrap(); + assert!(datetime_to_timestamp_millis(gap, &new_york).is_err()); + + // 2026-11-01 01:30 is ambiguous in America/New_York; picks the first (EDT, UTC-4). + let ambiguous = NaiveDate::from_ymd_opt(2026, 11, 1) + .unwrap() + .and_hms_opt(1, 30, 0) + .unwrap(); + assert_eq!( + datetime_to_timestamp_millis(ambiguous, &new_york).unwrap(), + utc_millis(2026, 11, 1, 5, 30, 0) + ); + } + + #[test] + fn test_convert_bytes_to_timestamp_with_timezone() { + let shanghai = Timezone::from_tz_string("Asia/Shanghai").unwrap(); + let result = convert_bytes_to_timestamp( + "2026-08-13 08:00:00".as_bytes(), + &TimestampType::Millisecond(TimestampMillisecondType), + &shanghai, + ) + .unwrap(); + assert_eq!( + result, + ScalarValue::TimestampMillisecond(Some(utc_millis(2026, 8, 13, 0, 0, 0)), None) + ); + } + #[test] fn test_convert_bytes_to_date() { let test_cases = vec![ diff --git a/src/servers/src/pipeline.rs b/src/servers/src/pipeline.rs index 7f6fa20f1f..a7081cf89d 100644 --- a/src/servers/src/pipeline.rs +++ b/src/servers/src/pipeline.rs @@ -21,7 +21,7 @@ use api::v1::{ColumnDataType, Row, RowInsertRequest, Rows, Value}; use common_time::timestamp::TimeUnit; use pipeline::{ ContextOpt, ContextReq, DispatchedTo, GREPTIME_INTERNAL_IDENTITY_PIPELINE_NAME, Pipeline, - PipelineContext, PipelineDefinition, PipelineExecOutput, SchemaInfo, TransformedOutput, + PipelineContext, PipelineDefinition, PipelineProcessOutput, SchemaInfo, TransformedOutput, TransformerMode, identity_pipeline, unwrap_or_continue_if_err, }; use session::context::{Channel, QueryContextRef}; @@ -140,10 +140,14 @@ async fn run_custom_pipeline( let table = handler.get_table(&table_name, query_ctx).await?; schema_info.set_table(table); + let needs_json_settings = matches!( + pipeline.transformer(), + TransformerMode::GreptimeTransformer(transformer) if transformer.has_json_transform() + ); for pipeline_map in pipeline_maps { let result = pipeline - .exec_mut(pipeline_map, pipeline_ctx, &mut schema_info) + .process_mut(pipeline_map) .inspect_err(|_| { METRIC_HTTP_LOGS_TRANSFORM_ELAPSED .with_label_values(&[db.as_str(), METRIC_FAILURE_VALUE]) @@ -151,9 +155,36 @@ async fn run_custom_pipeline( }) .context(PipelineSnafu); - let r = unwrap_or_continue_if_err!(result, skip_error); - match r { - PipelineExecOutput::Transformed(TransformedOutput { rows_by_context }) => { + match unwrap_or_continue_if_err!(result, skip_error) { + PipelineProcessOutput::Processed(value) => { + if needs_json_settings { + // JSON2 coercion must use settings from the final routed table. + let values = match &value { + VrlValue::Array(values) => values.as_slice(), + value => std::slice::from_ref(value), + }; + for value in values.iter().filter(|value| value.is_object()) { + let table_suffix = pipeline.resolve_table_suffix(value).unwrap_or_default(); + if !schema_info.has_table_for_suffix(&table_suffix) { + let destination = + table_suffix_to_table_name(&table_name, &table_suffix); + let table = handler.get_table(&destination, query_ctx).await?; + schema_info.set_table_for_suffix(table_suffix, table); + } + } + } + + let result = pipeline + .transform_mut(value, pipeline_ctx, &mut schema_info) + .inspect_err(|_| { + METRIC_HTTP_LOGS_TRANSFORM_ELAPSED + .with_label_values(&[db.as_str(), METRIC_FAILURE_VALUE]) + .observe(transform_timer.elapsed().as_secs_f64()); + }) + .context(PipelineSnafu); + let TransformedOutput { rows_by_context } = + unwrap_or_continue_if_err!(result, skip_error); + // Process each ContextOpt group separately for (opt, rows_with_suffix) in rows_by_context { let rows_by_suffix = transformed_map.entry(opt).or_default(); @@ -166,10 +197,10 @@ async fn run_custom_pipeline( } } } - PipelineExecOutput::DispatchedTo(dispatched_to, val) => { + PipelineProcessOutput::DispatchedTo(dispatched_to, val) => { push_to_map!(dispatched, dispatched_to, val, arr_len); } - PipelineExecOutput::Filtered => { + PipelineProcessOutput::Filtered => { continue; } } diff --git a/src/servers/src/postgres/handler.rs b/src/servers/src/postgres/handler.rs index 484bb6a1f1..aca4362bd3 100644 --- a/src/servers/src/postgres/handler.rs +++ b/src/servers/src/postgres/handler.rs @@ -28,6 +28,7 @@ use datafusion_pg_catalog::sql::PostgresCompatibilityParser; use datatypes::prelude::ConcreteDataType; use datatypes::schema::{Schema, SchemaRef}; use futures::{Sink, SinkExt, Stream, StreamExt, future, stream}; +use operator::statement::admin_output_schema; use pgwire::api::portal::{Format, Portal}; use pgwire::api::query::{ExtendedQueryHandler, SimpleQueryHandler}; use pgwire::api::results::{ @@ -40,8 +41,10 @@ use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; use pgwire::messages::PgWireBackendMessage; use pgwire::messages::copy::CopyData; use pgwire::messages::data::DataRow; +use query::dist_analyze_output_schema; use query::planner::DfLogicalPlanner; use query::query_engine::DescribeResult; +use query::sql::DESCRIBE_TABLE_OUTPUT_SCHEMA; use session::Session; use session::context::QueryContextRef; use snafu::ResultExt; @@ -554,6 +557,13 @@ fn describe_fields( session: &Arc, ) -> PgWireResult> { match sql_plan { + // Execution swaps in DistAnalyzeExec (stage/node/plan), whose schema + // differs from the logical `Analyze` plan's (plan_type/plan). + SqlPlan::Plan(LogicalPlan::Analyze(_), _) => { + let schema: Schema = + Schema::try_from(dist_analyze_output_schema()).map_err(convert_err)?; + schema_to_pg(&schema, format, None).map_err(convert_err) + } // query SqlPlan::Plan(plan, _) if !matches!(plan, LogicalPlan::Dml(_) | LogicalPlan::Ddl(_)) => { let schema: Schema = plan.schema().clone().try_into().map_err(convert_err)?; @@ -647,17 +657,6 @@ fn describe_fields( ), ]), - // single column show statements - SqlPlan::Statement( - Statement::ShowTables(_) | Statement::ShowFlows(_) | Statement::ShowViews(_), - _, - ) => Ok(vec![FieldInfo::new( - "name".to_string(), - None, - None, - Type::TEXT, - format.format_for(0), - )]), #[cfg(feature = "enterprise")] SqlPlan::Statement(Statement::ShowTriggers(_), _) => Ok(vec![FieldInfo::new( "name".to_string(), @@ -679,6 +678,61 @@ fn describe_fields( Ok(vec![]) } } + // Single column named after the variable (see `query::sql::show_variable`). + SqlPlan::Statement(Statement::ShowVariables(show), _) => Ok(vec![FieldInfo::new( + show.variable.to_string().to_uppercase(), + None, + None, + Type::TEXT, + format.format_for(0), + )]), + // Mirrors `query::sql::show_status` (currently always empty). + SqlPlan::Statement(Statement::ShowStatus(_), _) => Ok(vec![ + FieldInfo::new( + "Variable_name".to_string(), + None, + None, + Type::TEXT, + format.format_for(0), + ), + FieldInfo::new( + "Value".to_string(), + None, + None, + Type::TEXT, + format.format_for(1), + ), + ]), + SqlPlan::Statement(Statement::ShowSearchPath(_), _) => Ok(vec![FieldInfo::new( + "search_path".to_string(), + None, + None, + Type::TEXT, + format.format_for(0), + )]), + // Mirrors `query::sql::describe_table`. + SqlPlan::Statement(Statement::DescribeTable(_), _) => { + schema_to_pg(&DESCRIBE_TABLE_OUTPUT_SCHEMA, format, None).map_err(convert_err) + } + // Single column typed with the function's return type (see + // `operator::statement::admin_output_schema`). + SqlPlan::Statement(Statement::Admin(admin), _) => { + let query_ctx = session.new_query_context(); + match admin_output_schema(admin, &query_ctx) { + Some(schema) => schema_to_pg(&schema, format, None).map_err(convert_err), + // Unresolvable; execution will surface the error. + None => Ok(vec![]), + } + } + // Describe from the declared cursor's schema. + SqlPlan::Statement(Statement::FetchCursor(fetch), _) => { + let cursor_name = fetch.cursor_name.to_string(); + match session.get_cursor(&cursor_name) { + Some(cursor) => schema_to_pg(&cursor.schema(), format, None).map_err(convert_err), + // Cursor not declared yet; execution will error. + None => Ok(vec![]), + } + } _ => { // NoData Ok(vec![]) diff --git a/src/session/src/lib.rs b/src/session/src/lib.rs index 110560b324..c1414fc5a3 100644 --- a/src/session/src/lib.rs +++ b/src/session/src/lib.rs @@ -111,6 +111,12 @@ impl Session { .into() } + /// Cursors are shared across query contexts created from this session. + pub fn get_cursor(&self, name: &str) -> Option> { + let guard = self.mutable_inner.read().unwrap(); + guard.cursors.get(name).cloned() + } + pub fn conn_info(&self) -> &ConnInfo { &self.conn_info } diff --git a/src/sql/src/lib.rs b/src/sql/src/lib.rs index e8c6bdf8ef..b017f798f3 100644 --- a/src/sql/src/lib.rs +++ b/src/sql/src/lib.rs @@ -23,6 +23,6 @@ pub mod partition; pub mod statements; pub mod util; -pub use parsers::create_parser::{ENGINE, MAXVALUE}; +pub use parsers::create_parser::{ENGINE, MAXVALUE, parse_json2_type_hint_path}; pub use parsers::tql_parser::TQL; pub use parsers::with_tql_parser::{CteContent, HybridCteWith}; diff --git a/src/sql/src/parsers/create_parser.rs b/src/sql/src/parsers/create_parser.rs index be7ed0670d..0e0ece1b33 100644 --- a/src/sql/src/parsers/create_parser.rs +++ b/src/sql/src/parsers/create_parser.rs @@ -24,6 +24,7 @@ use datafusion_common::ScalarValue; use datatypes::arrow::datatypes::{DataType as ArrowDataType, IntervalUnit}; use datatypes::data_type::ConcreteDataType; use itertools::Itertools; +pub use json::parse_json2_type_hint_path; use snafu::{OptionExt, ResultExt, ensure}; use sqlparser::ast::{ ColumnOption, ColumnOptionDef, DataType, Expr, KeyOrIndexDisplay, NullsDistinctOption, diff --git a/src/sql/src/parsers/create_parser/json.rs b/src/sql/src/parsers/create_parser/json.rs index 63399f81a1..2f8a80c79a 100644 --- a/src/sql/src/parsers/create_parser/json.rs +++ b/src/sql/src/parsers/create_parser/json.rs @@ -21,6 +21,7 @@ use sqlparser::parser::Parser; use sqlparser::tokenizer::Token; use crate::ast::Ident; +use crate::dialect::GreptimeDbDialect; use crate::error::{InvalidSqlSnafu, Result, SyntaxSnafu}; use crate::parsers::create_parser::{INVERTED, SKIPPING}; use crate::statements::create::{Json2Options, JsonTypeHint}; @@ -29,6 +30,25 @@ use crate::statements::transform::type_alias::get_type_by_alias; const JSON2_TYPE_NAME: &str = "JSON2"; const MAX_AUTO_EXPANDED_PATHS: &str = "max_auto_expanded_paths"; +/// Parses a JSON2 type hint path with the same grammar used by `CREATE TABLE`. +pub fn parse_json2_type_hint_path(path: &str) -> Result> { + let dialect = GreptimeDbDialect {}; + let mut parser = Parser::new(&dialect) + .try_with_sql(path) + .context(SyntaxSnafu)?; + let path = parse_json2_path(&mut parser)?; + ensure!( + parser.peek_token().token == Token::EOF, + InvalidSqlSnafu { + msg: format!( + "unexpected token '{}' in JSON2 type hint path", + parser.peek_token() + ) + } + ); + Ok(path) +} + pub(super) fn parse_json2_type_and_options( parser: &mut Parser<'_>, ) -> Result)>> { @@ -322,6 +342,7 @@ fn ensure_no_path_conflict(hints: &[JsonTypeHint], path: &[String]) -> Result<() mod tests { use sqlparser::ast::{DataType, ExactNumberInfo}; + use super::parse_json2_type_hint_path; use crate::dialect::GreptimeDbDialect; use crate::parser::{ParseOptions, ParserContext}; use crate::statements::create::Column; @@ -339,6 +360,15 @@ mod tests { create_table.columns.remove(0) } + #[test] + fn test_parse_json2_type_hint_path() { + assert_eq!( + parse_json2_type_hint_path(r#"attrs."http.status_code""#).unwrap(), + vec!["attrs", "http.status_code"] + ); + assert!(parse_json2_type_hint_path("user.id trailing").is_err()); + } + #[test] fn test_parse_json2_type_hints() { let column = parse_json2_column( diff --git a/src/sql/src/statements.rs b/src/sql/src/statements.rs index 4f96b66f87..cf0b248675 100644 --- a/src/sql/src/statements.rs +++ b/src/sql/src/statements.rs @@ -40,6 +40,7 @@ use api::v1::SemanticType; use common_sql::default_constraint::parse_column_default_constraint; use common_time::timezone::Timezone; use datatypes::extension::json::{Json2ExtensionType, JsonMetadata}; +use datatypes::json::JsonSettings; use datatypes::prelude::ConcreteDataType; use datatypes::schema::{COMMENT_KEY, ColumnDefaultConstraint, ColumnSchema}; use datatypes::types::json_type::JsonNativeType; @@ -163,7 +164,10 @@ pub fn column_to_schema( false }; if is_json2_column { - let settings = column.extensions.build_json_settings()?.unwrap_or_default(); + let settings = column + .extensions + .build_json_settings()? + .unwrap_or_else(JsonSettings::new_v2); let extension = Json2ExtensionType::new(Arc::new(JsonMetadata::new(settings))); column_schema.with_extension_type(&extension); } @@ -643,6 +647,53 @@ mod tests { ); } + #[test] + fn test_new_json2_column_uses_v2_layout() -> std::result::Result<(), Box> + { + let column = Column { + column_def: ColumnDef { + name: "data".into(), + data_type: SqlDataType::Custom( + sqlparser::ast::ObjectName::from(vec!["JSON2".into()]), + vec![], + ), + options: vec![], + }, + extensions: ColumnExtensions::default(), + }; + + let schema = column_to_schema(&column, "ts", None)?; + let metadata: serde_json::Value = + serde_json::from_str(schema.metadata().get("ARROW:extension:metadata").unwrap())?; + assert_eq!(Some(2), metadata["layout_version"].as_u64()); + assert_eq!( + Some(100), + metadata["json_settings"]["max_auto_expanded_paths"].as_u64() + ); + + let mut hinted = column; + hinted + .extensions + .set_json_settings(datatypes::json::JsonSettings::try_new( + vec![datatypes::json::JsonTypeHint { + path: vec!["kind".to_string()], + data_type: ConcreteDataType::string_datatype(), + nullable: true, + default_constraint: None, + inverted_index: false, + }], + None, + )?)?; + let schema = column_to_schema(&hinted, "ts", None)?; + let metadata: serde_json::Value = + serde_json::from_str(schema.metadata().get("ARROW:extension:metadata").unwrap())?; + assert_eq!( + Some(100), + metadata["json_settings"]["max_auto_expanded_paths"].as_u64() + ); + Ok(()) + } + #[test] pub fn test_column_to_schema_timestamp_with_timezone() { let column = Column { diff --git a/src/sql/src/statements/create.rs b/src/sql/src/statements/create.rs index f3fbc90836..0db002c672 100644 --- a/src/sql/src/statements/create.rs +++ b/src/sql/src/statements/create.rs @@ -17,7 +17,7 @@ use std::fmt::{Display, Formatter}; use common_catalog::consts::FILE_ENGINE; use common_sql::default_constraint::parse_column_default_constraint; -use datatypes::json::JsonSettings; +use datatypes::json::{JSON2_DEFAULT_MAX_AUTO_EXPANDED_PATHS, JsonSettings}; use datatypes::prelude::ConcreteDataType; use datatypes::schema::{ ColumnDefaultConstraint, FulltextOptions, SkippingIndexOptions, VectorDistanceMetric, @@ -369,7 +369,12 @@ impl ColumnExtensions { }) }) .collect::>>()?; - let settings = JsonSettings::try_new(type_hints, options.max_auto_expanded_paths)?; + let settings = JsonSettings::try_new( + type_hints, + options + .max_auto_expanded_paths + .or(Some(JSON2_DEFAULT_MAX_AUTO_EXPANDED_PATHS)), + )?; Ok(Some(settings)) } @@ -1017,6 +1022,31 @@ ENGINE=mito } } + #[test] + fn test_parse_json2_max_auto_expanded_paths_option() -> Result<()> { + let sql = r#"CREATE TABLE traces ( + log_json_data JSON2 ( + status_code INT64 NOT NULL, + max_auto_expanded_paths = 1 + ), + ts TIMESTAMP TIME INDEX + )"#; + let result = ParserContext::create_with_dialect( + sql, + &GreptimeDbDialect {}, + ParseOptions::default(), + )?; + let Statement::CreateTable(create_table) = &result[0] else { + unreachable!() + }; + let settings = create_table.columns[0] + .extensions + .build_json_settings()? + .unwrap(); + assert_eq!(settings.max_auto_expanded_paths(), Some(1)); + Ok(()) + } + #[test] fn test_display_json2_type_hints_quotes_numeric_segments() { let sql = r#"CREATE TABLE traces ( diff --git a/tests-integration/fixtures/etcd-tls-certs/ca-key.pem b/tests-integration/fixtures/etcd-tls-certs/ca-key.pem index 2066dfb7b4..07a0a86ee3 100644 --- a/tests-integration/fixtures/etcd-tls-certs/ca-key.pem +++ b/tests-integration/fixtures/etcd-tls-certs/ca-key.pem @@ -1,28 +1,28 @@ -----BEGIN PRIVATE KEY----- -MIIEvQIBADANBgkqhkiG9w0BAQEFAASCBKcwggSjAgEAAoIBAQCfIi0iTllUQyNY -ka+d9Zt30ILA0QBzLKHwez9QJj8bQLpfKR26d794CQn/OO6XO7MWsJ6nvqwRNYM0 -WX+UwUtuT9MrttYWpS5yEAVO79dMtF67gCNafXKjQyK8L1vz6Avq4XJN4VqZEifT -Fdk3s2MmS0twv4sgrK//jh7aZzBQJtw+ROe9RN1dqlKoa3TtpB7iwNTDxGiR0EYj -wHj+GdiNHtDMWvrnTNNX/IKeEwLP/Mjirzr7GnaOjR25ruqcLuwWrufu+EMhqFTU -r+LIQpZD0TFuPX6DPlQZOs/4UpjteoC84m5AnQXNlh0vzjbW1efEYWYQoKDesx9e -oE5jcABpAgMBAAECggEAA3Rw/r6CYaTG1tdBilA47LF/XTihu15mh6Y4BNadESDn -IEYav0p2mDW4sgH7lWyg4ecPaBHo124jfUFM1p8Zsvk9sERwbR5vSIpWKyq5hbtM -Fo2yIXakGgJM9ev19vDR7WqxRMVAkR6GnyZpcwmh0e+ENvGZpNR1n7qSAJOJRibW -Ye8qMF/zMLF9VekBRPFJFIZBvKgHxSGCLbXRzsi5LKMbwfBoJmxEZSOC9zODl43o -Z2i+LvlS0R/Rsf2KG+Hwgj1P7HH1znJeA5bVJFDewljjSwu1mk0JpQL7Jgp2xpqs -F/mRZmtoQSH3EsTm2Kjz1sLGXwUb9JaQ274aV4t7pwKBgQDXYF+qFYip7k+b9FTF -sFbfp3CjOqxXiAxnoE5hlfV/h0w4V1Su42AdejnbIVYFX8bvmezkNWPXhB6P5oHU -4nO+oAuhsRT4317mfRM5NWO6mv2eglSBnS04GLsX142A7jMPr9e+Cqh4dYfw/A1L -bSdsdILemTFQrak27pC11mHZOwKBgQC9JhDhQf5uRh6k8GvgKGu9ZEZmkYCVu2sr -7kpD60UhklnyQo6DKYsa6OcChRQMv9ss3Hqu3Y1G0/sTe1Tjz5zwrWWAZoLlbdOt -vipesvwNTxvJoUWPtrjOoFZtmyg0Q0TqxaxfEWAssDU9IfnIy+zx7sAg4zZIPUyy -6kudaY1SqwKBgQCAQ/71Fjn7qddzc4GA8lHqhJeKPokg3/8zP78uUtaQCo2UCD6A -oR0+sOn/3MyUCsQ5MZxpFHrPgPmKjabIl8yCvGHw+7sXtD+aWOa37VnlaiSc39Vg -E7E4dVIHEvJM1I9ISlrb7REEHErHc/Se9PTDnGfMFcPO3n2mH1HDWVeQvQKBgGGn -BHH3a08tXmbTRS5uT+lwmrQbjKJBJ3x/wtG75m4Fq/BaEk9/JDUZZyKy5/4JEzPf -BGvBME4P5QFS3CndJu5O5ydaRVwDzpRVqHRJvb11SShY3Zvrvw/WUai2wRPyYuM+ -eNaAFwIbWvEb2GSle8gP9htEkuLK2w1HzxAOzYqPAoGAKl5l/WsoLfQP5UIUjSQe -JtVTDVmoz8m8Mg2nn56UdUetG246EVhDx5ngj84kZDM0GDXhfVWV6MWnIZr5NQsr -bORIDNP7Czat8FnlXeAQ+R6A/87o5g5p6ydH9jYqrrLxYjaZyXJptE+ja4zW7Lbl -ogWAc76iZvXLXqKRfepG0zQ= +MIIEvAIBADANBgkqhkiG9w0BAQEFAASCBKYwggSiAgEAAoIBAQC05De2Tk8TMZ0J +ME+7Rp+zvvLkEBPPXCCFcXfo91gYVHMMr1TGQJuVtbx7rIqC1tFhc2luAGfsBmW3 +IfTXZjzY6834ZUzmPL7uuBfoRpd0Paj5X4owpn0eHIgcsqx4juHSTYc3fdnrTLpq +lNMulZEemkNULLCE7pyxg3hhYIx9JEe9nv26fkyr1vW5z1cPlEpyOK+SUIeqHgXi +VtWPU9WGv5sdQeXnD4E1JTX2URihb7C8xali3YVbI3SUxy/gAhM6swQgZ4SLgW8k +f87a+PcP9C1N5Gujfw4fT90dJoutyiTSbcDrjZLjF03TgjcDaACvntb0qN85XD96 +dIvhg5jfAgMBAAECggEAAd3vHe6d+KTMDZ1HwPAAKS151ipnpX3W3aRgi/FC6OMm +r0VUpWC4jCJlif3kYvaID66yFeucJveD75CJw6VaF4bSnTa9GmKM5F2BQhsrhKj+ +69jY+MU8cztGfJMdgczNfi79qDc8zSW6TUFpoMhmD6Og8icRCODbkbRWDDnnQi1d +1CVe/lZ4yIapCGgzHfQxdl9I8LEZWyeAfqMCvOTI2eE4Z4ZYto/EmuRoJN3oqOx1 +MAtpS5mw6RTFVitO4b0j5gwAN2u6Cgo8jDc5UTzHGAiGc/dAn7S+mNGv+IXP1/aL +UA/ZsBwQhZbqtGCkSmCo5gGDL+anDTiIy6nRCwIWWQKBgQDtE7V4pNj5J2MciK+h +KAR/kT36NuGoZSnf585HJVB/xAs/d8F4kU5pCofmrtm2Bl5Lx4sCE3wPmv/HMYcI +QH2zt/YxPA3F/9zoBowfTNB8/iCDlzNb/pUry+qkBhKG2HgWaUkkGEtiO16PeCF2 +bidYA+h+XZBnHKexAKL74ee71wKBgQDDVHJ6HRo1uTAT/8Uz3AF5NRj+qj9QC3UX +On6dtzbtDGsJFkcqW5CV+LwUPN6zQ7MW7U6AEeH99q1UAUDkd5Q8/4bqs+0y2Ss7 +0UUWr1LDJVYo6FIs/f2ON0e9LzcJcG9TgZ6T95TL1kGEG1Ng96Je3hAZWolLmr+x +yfjTSYKqOQKBgEwRWtTO799hx+dL5C5tTKQx0hUKrvT9IKZ7FjC1xFJ6cLF4l1c8 +KFCD1H8r8yb4fCEMcYnE/rVzIkajmZQIPU0A5bl+b1zsb9Dy6NrSJsM0NvKB/TSz +RuG6mBrw59jkdAOc3J78PJPUQM7/2JzLU0xmVJ7XHpI3G4crkSAIp/YZAoGATX6m +gF4ldOUI9xZFleKWTxFK3laLEeXJybJyY34582g22v8UsvBq96UccWcI79RPLCxw +NY1ivNBuSeLJbRsoG99BFsLVu5O/fFb1cx+R1Uxt14L8f08xlofGFX+y9TK/aEEH +uirCxPA3RANXXCRDLiIp/vUVfYJixVWdO65xgbkCgYAVQhvJ3jPXZLXPl9fVC3Ro +qweEZHs0knPc4mqL/7bE+BCE1HcccU7O7+ZKbhW2q11EMrPJ7cHUqweHE3GsTgRC +C5FIvRxInZUgtPNx2UIZAVNjxRdCZJHBc/59eu8wVp9f5ETiSFmdz5xlsJECSxdO +mBakpZMqnESl1/6FMtcqeQ== -----END PRIVATE KEY----- diff --git a/tests-integration/fixtures/etcd-tls-certs/ca.crt b/tests-integration/fixtures/etcd-tls-certs/ca.crt index 8723f7ffed..ecd404c891 100644 --- a/tests-integration/fixtures/etcd-tls-certs/ca.crt +++ b/tests-integration/fixtures/etcd-tls-certs/ca.crt @@ -1,21 +1,21 @@ -----BEGIN CERTIFICATE----- -MIIDeTCCAmGgAwIBAgIUeRD4bMVd0XEzS3UHo2ptnknsWzEwDQYJKoZIhvcNAQEL +MIIDeTCCAmGgAwIBAgIUM4xko7n+KubCgGJnX4EaUzP7ZicwDQYJKoZIhvcNAQEL BQAwTDELMAkGA1UEBhMCVVMxCzAJBgNVBAgMAkNBMQswCQYDVQQHDAJTRjERMA8G -A1UECgwIR3JlcHRpbWUxEDAOBgNVBAMMB2V0Y2QtY2EwHhcNMjUwODI1MTgwNjU1 -WhcNMjYwODI1MTgwNjU1WjBMMQswCQYDVQQGEwJVUzELMAkGA1UECAwCQ0ExCzAJ +A1UECgwIR3JlcHRpbWUxEDAOBgNVBAMMB2V0Y2QtY2EwHhcNMjYwODI2MDQxMzQy +WhcNMjcwODI2MDQxMzQyWjBMMQswCQYDVQQGEwJVUzELMAkGA1UECAwCQ0ExCzAJ BgNVBAcMAlNGMREwDwYDVQQKDAhHcmVwdGltZTEQMA4GA1UEAwwHZXRjZC1jYTCC -ASIwDQYJKoZIhvcNAQEBBQADggEPADCCAQoCggEBAJ8iLSJOWVRDI1iRr531m3fQ -gsDRAHMsofB7P1AmPxtAul8pHbp3v3gJCf847pc7sxawnqe+rBE1gzRZf5TBS25P -0yu21halLnIQBU7v10y0XruAI1p9cqNDIrwvW/PoC+rhck3hWpkSJ9MV2TezYyZL -S3C/iyCsr/+OHtpnMFAm3D5E571E3V2qUqhrdO2kHuLA1MPEaJHQRiPAeP4Z2I0e -0Mxa+udM01f8gp4TAs/8yOKvOvsado6NHbmu6pwu7Bau5+74QyGoVNSv4shClkPR -MW49foM+VBk6z/hSmO16gLzibkCdBc2WHS/ONtbV58RhZhCgoN6zH16gTmNwAGkC -AwEAAaNTMFEwHQYDVR0OBBYEFOCkI3Uyx7F38LtXtSlg3ORE9ur6MB8GA1UdIwQY -MBaAFOCkI3Uyx7F38LtXtSlg3ORE9ur6MA8GA1UdEwEB/wQFMAMBAf8wDQYJKoZI -hvcNAQELBQADggEBAJXbWJJ9b+5nXjRiFXOg+wQnzn7kzMf6sWKIFk+AuKJWWt78 -O00t6vyAz6zel5Cj3ho9yAaMFNy8vYEnJCYngy5pT/2hOncnz/w7IKTeoEhzqAnf -MCEmCgHbTKDoFfMfrrRwtyoePVfx4xbiGVWMQFTPG+WNlE/ivFMRvFwsgPJ+7SUK -mR2FscH6DVo4sqF6s729lTmr6/U7bOD5l2HYGpKJ2cjCI8+HDv55aAT43tsTwBoJ -BgX+RT4w8ryGhYV+hrRTjPlMHWiHsNbzJeAi5bNA0f6vUP6k6zHn/Ur3MNb1avcf -cSsdYU2dhs7PeB6IpoJ0QCeJw9MIFK3XTnmbeVY= +ASIwDQYJKoZIhvcNAQEBBQADggEPADCCAQoCggEBALTkN7ZOTxMxnQkwT7tGn7O+ +8uQQE89cIIVxd+j3WBhUcwyvVMZAm5W1vHusioLW0WFzaW4AZ+wGZbch9NdmPNjr +zfhlTOY8vu64F+hGl3Q9qPlfijCmfR4ciByyrHiO4dJNhzd92etMumqU0y6VkR6a +Q1QssITunLGDeGFgjH0kR72e/bp+TKvW9bnPVw+USnI4r5JQh6oeBeJW1Y9T1Ya/ +mx1B5ecPgTUlNfZRGKFvsLzFqWLdhVsjdJTHL+ACEzqzBCBnhIuBbyR/ztr49w/0 +LU3ka6N/Dh9P3R0mi63KJNJtwOuNkuMXTdOCNwNoAK+e1vSo3zlcP3p0i+GDmN8C +AwEAAaNTMFEwHQYDVR0OBBYEFDjvuYaWEoViLOKyCse6mAwH66mIMB8GA1UdIwQY +MBaAFDjvuYaWEoViLOKyCse6mAwH66mIMA8GA1UdEwEB/wQFMAMBAf8wDQYJKoZI +hvcNAQELBQADggEBAJuGxtprq/GF/fwUJ5hk0cG80VY24G+iKKi7Jw7PWPEMvnlQ +/ZIKgNDupE8mSSyokicKZMrtwT8QF6wlDLVcz4LTPCtKUmYU3FiuNE4V57+yw7zt +XZn6rwiWGxeyfY0ZktLXiSYA5FydOJ9GQQWgxkpbd5QG1L9aoFW49cbz8ob6efxG +p+YRgr0Fr4hXn7ImJWWqeawf2tg4iLwpV4sem5atpI7y9qamxrGGUp2h/CnADqUz +YWYIYP1RCPp2bXzxO40Ni1NYmFsGUU/SL1vtJw5n2Zmx/Phe7u/CLe05dH1O54Qf +vCbk6X0CthFI0HzSepbEVAuniVHkrpWwDLHcQ6s= -----END CERTIFICATE----- diff --git a/tests-integration/fixtures/etcd-tls-certs/ca.srl b/tests-integration/fixtures/etcd-tls-certs/ca.srl index 69ac026da2..ad32d06e16 100644 --- a/tests-integration/fixtures/etcd-tls-certs/ca.srl +++ b/tests-integration/fixtures/etcd-tls-certs/ca.srl @@ -1 +1 @@ -4484A9BBD4F25F32994F3C03D80D294105B119E6 +4484A9BBD4F25F32994F3C03D80D294105B119E8 diff --git a/tests-integration/fixtures/etcd-tls-certs/client-key.pem b/tests-integration/fixtures/etcd-tls-certs/client-key.pem index 13a23c5347..1d64fc66a3 100644 --- a/tests-integration/fixtures/etcd-tls-certs/client-key.pem +++ b/tests-integration/fixtures/etcd-tls-certs/client-key.pem @@ -1,28 +1,28 @@ -----BEGIN PRIVATE KEY----- -MIIEvAIBADANBgkqhkiG9w0BAQEFAASCBKYwggSiAgEAAoIBAQCOcrzRmE3sEGYo -pazDrpkkzbiaBhHClfmBiZAAGV2cvSXY8e9pc21GOgrGBNeyPPieS305rAE02EZl -oIAt97XRFxkZniUy8C5+PT3eicAKPRcrgGhIrYxUs39rdRocqVVpZsiZGzhHhboB -bXNW1l37SDQpuxAQJuhHu2wu9J7c5/TBXWN9ZOxzWTlrIdEQOA46VclE/N1TlaKs -CSWw8qH87YLwkLEXcFLOJfmWNZxOOhdL5MSzKKNcQKbpmkd65B/PHNwBIBZHpI+K -xphK1SybA+ESkN09qZaa1Kj3e0I62UjHNXUCuAod6RhRRmpiFDFgggXhD9T9Ttuf -1Lq38XxDAgMBAAECggEAANNhqQldwHzi9WCxXObLoufBe9lM73+1Df1IQaDHkZR3 -jHliixxF18XQW7qZ5tAJqjsDWLPm4BxSKXozsjsvBdgIUjbqkyx1dsGlO9TjeGu5 -H2/8jlH14boIMWJ9Y0JBWwvU7EnVU5IkUOEiwsxvFlzM5Q0H+841CrSD+Xd7iSZE -ZrpyZgpHNmvc/u7AmseZVccjX21dxwWtcd7cGUmFWC9o2a/hkgG6+baA8zH2ze3J -Smeowa/oOyOVkTM1EHLCBvddE+/W+BeMGPY7A6SztVw0gLqlSOL3cr/g2llUTmFg -T6MOS6FZZ7MD5c2HBL0V3OXIhxNABzD+c0O3CaRbSQKBgQC/+QuhMpDCzxBPaL/K -sqRp4sQgURDCXPA1nAfI+NoHos/CbjwcMGp6LUSiBW0tLTJ+LZoVgZXj6j/uxji9 -DWxQfULvQ7oanm/q/D+40NnupQQTzmMT04z/mV3XjFtzEhbO30YZfqwXy+tqB8vR -g1KajXVqLmCxKDtsFedAlIVxRQKBgQC99TKGcBwsKlYJJThU2eIIy8G0GkjjXleK -ROMO5i0keJtLac1oRvXo4JN5UBLJM6ZpSy0TEa2bxfzy6Uea+41aZmdCpPg0Kcsq -Pb1S3rsPpVdsfxSxsdKvwZmHXOzAdF+fom1+ZRAdo/8ALIPaNQSSaw+KkmmpHPEV -qdzzWJ4b5wKBgFubEtKcF3nuZwENoh+ueUhRvncRV+b3hGSAjTJ4lUn5hhxoj+R/ -sf+VJGAQKNXa8HJHfnRuvsDgYhulmSOViS8rZspXzjGvkwZV0m51stjvA3AUFzE5 -zNmXLLGTt3vEkP+siX3W9XXxh+ezyq2ydbNsdy/w65D9+sUL+qrVdIvlAoGAOOQy -2ajCB0g2tE59bIxE8jV0MiidI9uhhDvVdSTi6EVm3VM2vcBi7fg0suSUe8YIVQi6 -2zc0M688btQHKhek4ipBSuh1ncnWmzQae7NRewIeCNSWshF79D+bZ7sg/RLdgMX4 -3R4PkZEIUlkCtFuknuWJpgrrskaEveQ91HP6BokCgYAX42nKPANIMrTXuJ2vACWu -qKQ9atAC1ngVzYe2PI8rHjA4gCDraGfCaILewrd89rK9ukBeSDIm5Mn448q3ystw -p/QMxmMdHADCD6UwCjlt9N0Xigv+IKbDzDzhartRVIPpZdTViQXW3cSHt9id8zvO -HTu5C6W+N9Vb7j8Ak/mvzw== +MIIEvgIBADANBgkqhkiG9w0BAQEFAASCBKgwggSkAgEAAoIBAQC5F3hVsri2FBzP +dtRe+aJFh9yqhXnJ7zN+xbUmNWsTZVnPwrbVo9H6PtGWzkbZ6YMBzMlQSYP2txaS +ROjE8wwEs38eKbyJrHO2Ccq3iR9NfDxSbHTClngsOjf9fEoYExif84VKWLO6+EKL +KqqOfj3+iuI3hbabr0MqGu4e2Nr6MlQtViBxTs5aSnahc8hmZbujBcWRTsw/GNx5 +Am0a8BrrJRVmVR+BzlYtzbf1ltfLxIhnc8dAGyT9B6GTwcMNST6m4WtTzR/uOr6S +2YEbIqwCHB1VQ+oJUGJvExqaNLoM/x5a8sUNIWx3LBN0/XC4D+MMmh/bbyb2cAOE +gyUYu+hLAgMBAAECggEAEkt+wksVP9toEkecFpEqgss65VimdkPK/WkHDpzxawGL +FDfvq4Py9tRrvLIXRbseObjSCI9jrai2Ne9Sv1jeDhQ8kzDWk8MfItX/8A6Whf8A +KrjRUlneIfz/2Mcxaa9wMSs4YxC2dIYidFqwxhSLWyluJ8Ud4hs7kndb+beUmW3i +DgkLsQ9ytQ0HDQrAERnrbRf19fmVaqQapfxpYT2gg80BMIq+zlzoiXareDz3r/1g +aoponxArWV6yCvQOYYjWcDr4WZPvIA2APEX5isxyvMBvb72Fpd2z+0JX+4LVe6N1 +FqZLNpSeJn6O+Q1vJSScSJcoJYkLWCfZbFLHPttZGQKBgQDdVr5PtgPIVM+67cvv +p3x7NNC10Gp387PgQS/FlG+QtbP87RTBnANT1uFd/FMFqL35or7Y46YKDUdk00OZ +LinpM4ARFRYFqkthM6bcUWatpF6Fg2dfsKuD4H6ARp8YHhlmCnDCQ0swZp1HCmAB +6vltUNTZjQjRdd44GlQ0icX0mQKBgQDWE50TXRszea75S+zUfXeyvxMyptIOS/rY +L50XVuhGVknH4OuBQbGF6UZulIGSpN7gmmbO29Yx1HKmqiJVaWlvB71baK4hWLDd +2grIUy0zuq1B7I1IOLmOpMcL3xau76g7ZWRUprX26hxAuxYyPb9V8gNo1P7HN8si +VLJVE8ZugwKBgQCaxoGmM91JRSVNzeOB3ljJvxEDUo5g+uWZt3u0aivpwWXvQ8nz +6Sjag7RsiHl1x52w5wEVoXsGJGr8Mk9e2k0saXrwdxJDO+YiPoA8KB/o5LvEGTM8 +UspdGarcAIZX0xRnqn1XGr+FRPxOJQ8lyC5LJu7wghLchdOy35Zqdr0aYQKBgQC/ +38yFsonS1VnS8A5RVjOW7lPSrlrPnaIzalmutaJyiJyQnjP3Il5u2+rY6hpIyaVK +QpmrBrcw6m3om80yKMzrS1CZQXXxRYEhF3Fao9J77vGjiNYIyW7nPyF4rneyS/PJ +aNNIXDP0H1k7W3RFi7qW2dfceivxezyChM9iGdtc6QKBgAIfb63+1q98MNlAB7eD +x7K8vuSxHrJ9gX9KcLrG1ScLOxnAC4Op+8xf06ZHt828CajchAOqee2nh52MsFOr +AdPNdhS/8mTq8KosTfgVuawv4yTHpZViMR2KSOcTKhKaBmhufY+k8mJ0YAAOTnzO +JJqjDSJdjWsDG+7UlNgFcvuU -----END PRIVATE KEY----- diff --git a/tests-integration/fixtures/etcd-tls-certs/client.crt b/tests-integration/fixtures/etcd-tls-certs/client.crt index 935351a48b..0dee5915ff 100644 --- a/tests-integration/fixtures/etcd-tls-certs/client.crt +++ b/tests-integration/fixtures/etcd-tls-certs/client.crt @@ -1,20 +1,20 @@ -----BEGIN CERTIFICATE----- -MIIDMjCCAhqgAwIBAgIURISpu9TyXzKZTzwD2A0pQQWxGeYwDQYJKoZIhvcNAQEL +MIIDMjCCAhqgAwIBAgIURISpu9TyXzKZTzwD2A0pQQWxGegwDQYJKoZIhvcNAQEL BQAwTDELMAkGA1UEBhMCVVMxCzAJBgNVBAgMAkNBMQswCQYDVQQHDAJTRjERMA8G -A1UECgwIR3JlcHRpbWUxEDAOBgNVBAMMB2V0Y2QtY2EwHhcNMjUwODI1MTgwNjU1 -WhcNMjYwODI1MTgwNjU1WjAWMRQwEgYDVQQDDAtldGNkLWNsaWVudDCCASIwDQYJ -KoZIhvcNAQEBBQADggEPADCCAQoCggEBAI5yvNGYTewQZiilrMOumSTNuJoGEcKV -+YGJkAAZXZy9Jdjx72lzbUY6CsYE17I8+J5LfTmsATTYRmWggC33tdEXGRmeJTLw -Ln49Pd6JwAo9FyuAaEitjFSzf2t1GhypVWlmyJkbOEeFugFtc1bWXftINCm7EBAm -6Ee7bC70ntzn9MFdY31k7HNZOWsh0RA4DjpVyUT83VOVoqwJJbDyofztgvCQsRdw -Us4l+ZY1nE46F0vkxLMoo1xApumaR3rkH88c3AEgFkekj4rGmErVLJsD4RKQ3T2p -lprUqPd7QjrZSMc1dQK4Ch3pGFFGamIUMWCCBeEP1P1O25/UurfxfEMCAwEAAaNC -MEAwHQYDVR0OBBYEFINcPBk3iqCw1B2sZzcDpSF6ZNGnMB8GA1UdIwQYMBaAFOCk -I3Uyx7F38LtXtSlg3ORE9ur6MA0GCSqGSIb3DQEBCwUAA4IBAQBgyF1XSpAyiQNA -sFCqXrBfwPnv6V+R5jhO6Glmn+Nhj0gbzNBlU8EKsK/WZllDrJjmxubPBqb563Zt -J+QcNLqugZdioXPbxmmi9xj3oK25pUCRBj03nlHpGqIwuCJyn/ZxlfXMpNYE1KjM -nprFNUKgnYPk2AGffAngQdm0yrnmSpxXhNTuhQE8avgRhd5nEUEDyPIvHWwo4iU6 -pw2aP2FpWorESv9LVna+JortThYDQrIFgCtPM6BP1N7Aeqn0H846+JsXlq2ch1Iy -aBRqKNQRc1Q2Qb9/QCDEJ3dEixgFjxztIaqrIcr18ZJ797MhTFS3aVEwE8clSRR5 -9ow9jK8B +A1UECgwIR3JlcHRpbWUxEDAOBgNVBAMMB2V0Y2QtY2EwHhcNMjYwODI2MDQxMzQy +WhcNMjcwODI2MDQxMzQyWjAWMRQwEgYDVQQDDAtldGNkLWNsaWVudDCCASIwDQYJ +KoZIhvcNAQEBBQADggEPADCCAQoCggEBALkXeFWyuLYUHM921F75okWH3KqFecnv +M37FtSY1axNlWc/CttWj0fo+0ZbORtnpgwHMyVBJg/a3FpJE6MTzDASzfx4pvIms +c7YJyreJH018PFJsdMKWeCw6N/18ShgTGJ/zhUpYs7r4Qosqqo5+Pf6K4jeFtpuv +Qyoa7h7Y2voyVC1WIHFOzlpKdqFzyGZlu6MFxZFOzD8Y3HkCbRrwGuslFWZVH4HO +Vi3Nt/WW18vEiGdzx0AbJP0HoZPBww1JPqbha1PNH+46vpLZgRsirAIcHVVD6glQ +Ym8TGpo0ugz/HlryxQ0hbHcsE3T9cLgP4wyaH9tvJvZwA4SDJRi76EsCAwEAAaNC +MEAwHQYDVR0OBBYEFD1aTqJO87ovA2usg/v8UckSsnwhMB8GA1UdIwQYMBaAFDjv +uYaWEoViLOKyCse6mAwH66mIMA0GCSqGSIb3DQEBCwUAA4IBAQArP3mtjdj5a9uT +XTys4DUjp+CJaSmsZ/4ICf/KeddaEyijx6uoiqbNsAtfwlD5A5BtyjJrxg3SpjqD +Fm5mTV0HlwvYHxcFPTUlR56//sOyeRhm+kVPp4xdVq2p6IZXFF/hDkeT/AETWOjj +l4+90gAsJsUHrKUZmzlqHBeOXoAhzo2PwSh8KLcGK0705FGql4QnS731PHo9a9qp +e8uxk1KKoc+RHm2OjJvZ7hvkCXuCm3TJeOHQEtjZ5buQg40BRrt11GVnsPePpnQR +8OtV0DvPT3UcOss88kLDQuRuZeJf8UyaevGGvItv3+Yk5cwUBSaJ0mY6sgrmkXn0 +TstUzYIT -----END CERTIFICATE----- diff --git a/tests-integration/fixtures/etcd-tls-certs/server-key.pem b/tests-integration/fixtures/etcd-tls-certs/server-key.pem index fca8a0122f..a73ac97443 100644 --- a/tests-integration/fixtures/etcd-tls-certs/server-key.pem +++ b/tests-integration/fixtures/etcd-tls-certs/server-key.pem @@ -1,28 +1,28 @@ -----BEGIN PRIVATE KEY----- -MIIEvQIBADANBgkqhkiG9w0BAQEFAASCBKcwggSjAgEAAoIBAQCtYFxH+oet6jTD -0eF55ISO+bwm1NPTluP9zLJrQFNDkqlddyiIasK5q7qN1pPXOw9mJF/EN8Hf1lWj -42an6tr7PgE0NHLcOrqDv4dRjegRi87IWiPkfep8peUE3glI4IOMXfZmAUNp+nRO -JP/ymmR0pzHYJlx+D082OYmeBny7uFQxo+vxgVNo2MVVwy6oTv+jtjLgZ4iOrIvh -iILoXtXcdg/tjB1rS2hSABGPJMNHwKswcrt6xXv+7Llsxroq0cx0E2PrAbc5qB40 -v1Qp7PjFTJ0drgf0g51LkOb7znpB1viPD2BLMxpELTZqa4OruarcDNxShrK8WWMP -1mygLHLxAgMBAAECggEAC4TgtLaQP62VImKGLs1QQliU1+ahiUgX70Ojog0UyyNK -JefWFVQ0ilX+z9AvI+hsYj6t7zE+LBNHPtuLtUHdGT66IUAP1pJ/VGQMB07cmZfW -ngihJFv6UZxLDkr7RnCGRPP0PDw+wKKPiiaaq8F2xapbHS+VSxnUyzdA7bMkI+uj -+gLAFj+kEU0FNbIp/mBzz8sksDzcB29oy3VnZkR1211KkL1BJDvr9oKiUg8NFKaN -NF2WXRYI/m0CQMlX6zlTyv7+h/Ay7y8oNG7cfHavmCrWvYIMfaDggYxbIpNdGxb5 -MAdbrtXJ6daUBa2sezxgQUyJX8+8GGF2zpGvCfhkfQKBgQDiAwA8rlxrBnXclwRK -9s/apILKCxPEGwU6XZi/8na3ugIO14y2WD0zZTjUK/3xwoKN6j6+ObLkUBX2RRYZ -ewwxpqos2cyxi5EBKPhsydqbJQlat68I5+NOG4KOZQkqO92tBtgY3gGFWVntaqtE -UVE/Clh2gFcc/5nR/5C5Mil8LQKBgQDEYXtrl2kl4qcyXuuXKDA/bLSnDuIEMhLO -4oIiqmywXQgjPjGAWy0eztjMq2a353pqTFCqto/cfOif+rjkB7oFfi9gMm0plIKI -+F8QsBB7HpGqWos3zb1yt8X9hbxKuFFwZS/un3Li4lwPlk8EZ1YIvhC44BHAvNkD -VzxzBwoYVQKBgArRe/RroC7bS074x5LTB5X+o+gJ6bNMW860Zjhh4b7fn3OYa7ra -tGs+YB7/0BL/bYJfgQtX9bEqCDMWkX08v5Os1553+m1RMeqtTF7gtp8Qgcce3bj+ -aIn3lSM9wNeNsAm1NyjRj58TbNOJdJM7lTkARMW/VOwla/Z6VjIXLZctAoGBAJKf -xira7eMfi36MaJJ/qyZv36Ir9ozzZh+Z91gyrtwvWfgWY5dWfCXYgv6tqw/8gOYE -/OW5UUhq6rUn2gxHyJh5Up4ciGzXOW9TIoevLV7/v/rVh8SulJimpelYhPG1FPk6 -U8NywbCtGdd5fp3nGdGFN68Rfa/OUKmx5KxtwRfRAoGAWpfzmoI8EmCBZNXYk03k -hnDKmICsKi9seDclcQq2PbyNvE7LDEd8O4pDnwV6RwKVEDTer9Y4IGcjkEBZcJVm -za3l4En9+M27Xlf1I8XeB8o8RYbFjoDQyScFFIgQpvTDsOezXeDR55FGPZN7V6Si -lKmd+8BuzJ7d31KJtRjoS1U= +MIIEvgIBADANBgkqhkiG9w0BAQEFAASCBKgwggSkAgEAAoIBAQDQLtVfdk9l1rdH +/ssuy/BK9LpmNPx7gatp/E3EqpgekBjizMXIxnCHDFNM4Dtq1v3afaxeijXnnW/b +0CVKSeGkOBmb0dO0K21xe86Y5VfNHkGotQyCZ8mPOzep/EQOGJY5+dprErL0YKM+ +2drgiRNGwidmNk/jD3OtzXSMR+8UvDWJSxUoZ1ncvgmqf3zTTiaTiABHrfLdXaGI +23+JftyLgoeU4vEI+x+RQJMAa6Gzjo9vALxX3kTtR/uH34hfm8vb3yo68UHGKVMp +i3yi6YDP4YEao5lOpm7fIyEes10fPkD224++S34KHykAk4q+eTPetWkDYFCYzdSp +AR2Ob3cxAgMBAAECggEAMDgAxO0mw7xBVGQgFJU48WuQtvaj2kl09gXxz1ECDeYr +VXC/iNrpmmYQ7zfqmzrzrkU4hOc3SA/PplamJHhLUpmJ2Oz3P35liYj3F7PbK8/L +vnM81AGNDmdVY8Jh0u//76q+29kHaRHvDbIw/5vQQq3aqVKAG2PrU8DIM2u/5Qmt +0GHJTMmyVADAOLe7j2agN9hrOrjeoWobcELwtwgKXxmU0l+g0XX1OdJpbjA3qDPa +S5a4MzcucGINeB3bohcqbcx2f1HwVUoC19mti3Cz+KvxF2YZV9bS9izW4hSEsAwU +0upNHF3MwTUQg7ftLPAZcfNoXIZMFXwYMS1HW2v7HwKBgQDuHHkFRN2lvor/Ycjr +gJD6LGb9RlQ7NMYBQHB+5g49Y/q8oDZJgyl4CfgYWdVjjOfEX8oMst1WU/Stbva5 +HlHR7yYFyebScQYOPHZaPyVKGIbcl2p7Qy9H4pxJc1ZMNxO8SxoSC68LZEan08FD +/Qsish51fZ2Sqf37b+8Uz/CGdwKBgQDf0sIp88MPGbhtzdYtF9gA4AoIOPo9UxeI +F6wTb05uCqZyK+hcdK37av1xcNpczd/yvlALxRdM9xGedJkcS2cgeExtMRcfhAzL +as0dYPzuhRkpEIfN/y9cqtHj2wigrbu35eDgRRnJDgXgL9dYl6r+Swqn3jBaasP7 +KtQdQb/RlwKBgQC7qe0n3fLi4p4iUStNkPKyebRiAb/5Ocqkyejf2ul2MQo5B/xB +TAKu/QxwBL1NzIwOFYDlKUOQ+nJpDn+dvuu1jcpl3Y7yZOnk5npQ/luhXltMGHpv +06+79DpBGYn2X6JKUNanSlYXoFyfgSFdOF5CZifjabF7Gkd2l+3SdWCYWQKBgQDB +3k4sBGZKaB7ljUscl/CTIXvPD3tBLv3M9aQo2Vp32mW9suZ7Xt1sTonkfrnFdNWr +7shqyXabRc5PD/OnHHDhIRIh6kl7FOf4MjQkZGPxPfxDI3xeI9EkVRmkYY6hjppw +eX9FAtWI3sqcGxROOmD0Do/WQ5BiYOQMZFaCWPcLVQKBgGVT+iUYkYTs+UJeD4RC +QeL6QEJDszvAI/gavUp7sD/LWn7Ig+hJiAl2O+HFYMurh7eda0JnE2KF40UyPiYA +DDl19xRAXrLPOlOj9YKDA9HkROcaVZCQCqyDalKsbw1IqMrAkRmLCIZW3EHaMrUf +UNjg+GaxwvrRj11gbd4IPgai -----END PRIVATE KEY----- diff --git a/tests-integration/fixtures/etcd-tls-certs/server.crt b/tests-integration/fixtures/etcd-tls-certs/server.crt index 6ed8b1054e..8dcb7b13b6 100644 --- a/tests-integration/fixtures/etcd-tls-certs/server.crt +++ b/tests-integration/fixtures/etcd-tls-certs/server.crt @@ -1,21 +1,21 @@ -----BEGIN CERTIFICATE----- -MIIDjDCCAnSgAwIBAgIURISpu9TyXzKZTzwD2A0pQQWxGeUwDQYJKoZIhvcNAQEL +MIIDjDCCAnSgAwIBAgIURISpu9TyXzKZTzwD2A0pQQWxGecwDQYJKoZIhvcNAQEL BQAwTDELMAkGA1UEBhMCVVMxCzAJBgNVBAgMAkNBMQswCQYDVQQHDAJTRjERMA8G -A1UECgwIR3JlcHRpbWUxEDAOBgNVBAMMB2V0Y2QtY2EwHhcNMjUwODI1MTgwNjU1 -WhcNMjYwODI1MTgwNjU1WjATMREwDwYDVQQDDAhldGNkLXRsczCCASIwDQYJKoZI -hvcNAQEBBQADggEPADCCAQoCggEBAK1gXEf6h63qNMPR4XnkhI75vCbU09OW4/3M -smtAU0OSqV13KIhqwrmruo3Wk9c7D2YkX8Q3wd/WVaPjZqfq2vs+ATQ0ctw6uoO/ -h1GN6BGLzshaI+R96nyl5QTeCUjgg4xd9mYBQ2n6dE4k//KaZHSnMdgmXH4PTzY5 -iZ4GfLu4VDGj6/GBU2jYxVXDLqhO/6O2MuBniI6si+GIguhe1dx2D+2MHWtLaFIA -EY8kw0fAqzByu3rFe/7suWzGuirRzHQTY+sBtzmoHjS/VCns+MVMnR2uB/SDnUuQ -5vvOekHW+I8PYEszGkQtNmprg6u5qtwM3FKGsrxZYw/WbKAscvECAwEAAaOBnjCB +A1UECgwIR3JlcHRpbWUxEDAOBgNVBAMMB2V0Y2QtY2EwHhcNMjYwODI2MDQxMzQy +WhcNMjcwODI2MDQxMzQyWjATMREwDwYDVQQDDAhldGNkLXRsczCCASIwDQYJKoZI +hvcNAQEBBQADggEPADCCAQoCggEBANAu1V92T2XWt0f+yy7L8Er0umY0/HuBq2n8 +TcSqmB6QGOLMxcjGcIcMU0zgO2rW/dp9rF6KNeedb9vQJUpJ4aQ4GZvR07QrbXF7 +zpjlV80eQai1DIJnyY87N6n8RA4Yljn52msSsvRgoz7Z2uCJE0bCJ2Y2T+MPc63N +dIxH7xS8NYlLFShnWdy+Cap/fNNOJpOIAEet8t1doYjbf4l+3IuCh5Ti8Qj7H5FA +kwBrobOOj28AvFfeRO1H+4ffiF+by9vfKjrxQcYpUymLfKLpgM/hgRqjmU6mbt8j +IR6zXR8+QPbbj75LfgofKQCTir55M961aQNgUJjN1KkBHY5vdzECAwEAAaOBnjCB mzAJBgNVHRMEAjAAMAsGA1UdDwQEAwIEMDBBBgNVHREEOjA4gglsb2NhbGhvc3SC CGV0Y2QtdGxzggkxMjcuMC4wLjGHBH8AAAGHEAAAAAAAAAAAAAAAAAAAAAEwHQYD -VR0OBBYEFK8y9gWv8Si9AusP280BrgDF6y3jMB8GA1UdIwQYMBaAFOCkI3Uyx7F3 -8LtXtSlg3ORE9ur6MA0GCSqGSIb3DQEBCwUAA4IBAQBMBsAPIlk17I2ioN9xwxwl -yYbSjZlTs+18wSZAoCNLfxwWIYAQ08dPoEUdALfMUPEEe1Ol2IAp/qx2JoDrZWje -maA43hppBUpFFCkTKWUMYxsetN7d7BVxWL43GDoMwoD/k36nhoUKBUjlbF0+nkem -dOMr8SA4GZbEg7qk2cL93g0UHv/Z/dZgKf3epZkR9hEN4/R2jSP6OPmY9XIvQCoZ -8D9jgbQKavAAXmUcP5a81alQMRroGaBCzI1f5OlS3EuVE7ZTEBgxKK1idbouSrPt -4UHEbOGz9zXDjmfut6CTm247+lJm9jzYe2Xx+XGw29l0pzd/8tGFN9zIvW8mCv7a +VR0OBBYEFNqGaSuuVHwEIvQvYxt9H0kdWFPSMB8GA1UdIwQYMBaAFDjvuYaWEoVi +LOKyCse6mAwH66mIMA0GCSqGSIb3DQEBCwUAA4IBAQAHttiuyEZieeTsggAKzAIC +UXsjWCQMAmjRxery6mzM8P33CdCKA4cwGyy6tzoFPiZC/wh+gPq+LiUO9G+vkqNA +NQ/DSJS+fsqsOBWUt5DSYekD25o80mZVE/4H/ynFOgENBqgkVl1ikpkXu9BK8wTm +Wij+p9RWyNup8lr9OFYZRpy7wrxPp7opwp3qG2Bfu/BGY1YEr04CIrLMKnk1zRMm +73X1lcrlrM9ycCmYcfXxyU69G0iFKaPcbS7T0fngm1NciA1qxYYLuMyzDtF7/SyC +J2ynvDBBnc9+/X8piu8p9D0I9mbcbEySv6zZfYMT0kqXbsn2GijZTH5mCJ3P8cwc -----END CERTIFICATE----- diff --git a/tests-integration/src/grpc/flight.rs b/tests-integration/src/grpc/flight.rs index 068cb98142..3dee75e066 100644 --- a/tests-integration/src/grpc/flight.rs +++ b/tests-integration/src/grpc/flight.rs @@ -17,34 +17,132 @@ mod test { use std::collections::HashMap; use std::net::SocketAddr; use std::sync::Arc; + use std::time::{Duration, Instant}; use api::v1::auth_header::AuthScheme; use api::v1::query_request::Query; use api::v1::{Basic, ColumnDataType, ColumnDef, CreateTableExpr, QueryRequest, SemanticType}; - use arrow_flight::FlightDescriptor; + use arrow_flight::flight_service_server::FlightServiceServer; + use arrow_flight::{FlightData, FlightDescriptor, Ticket}; use auth::user_provider_from_option; use client::{Client, Database}; use common_catalog::consts::{DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME}; + use common_grpc::channel_manager::{ChannelConfig, ChannelManager}; use common_grpc::flight::do_put::DoPutMetadata; use common_grpc::flight::{FlightEncoder, FlightMessage}; use common_query::OutputData; - use common_recordbatch::RecordBatch; use common_recordbatch::adapter::RegionWatermarkEntry; + use common_recordbatch::{RecordBatch, RecordBatches, SendableRecordBatchStream}; + use common_telemetry::tracing_context::TracingContext; use datatypes::prelude::{ConcreteDataType, ScalarVector, VectorRef}; use datatypes::schema::{ColumnSchema, Schema}; use datatypes::vectors::{Int32Vector, StringVector, TimestampMillisecondVector}; use futures_util::StreamExt; + use hyper_util::rt::TokioIo; use itertools::Itertools; use servers::grpc::builder::GrpcServerBuilder; + use servers::grpc::flight::{ + FlightCraft, FlightCraftWrapper, FlightRecordBatchSource, FlightRecordBatchStream, + TonicStream, + }; use servers::grpc::greptime_handler::GreptimeRequestHandler; use servers::grpc::{FlightCompression, GrpcServerConfig}; use servers::server::Server; + use tonic::Response; + use tonic::transport::Server as TonicServer; + use tower::service_fn; use crate::cluster::GreptimeDbClusterBuilder; use crate::grpc::query_and_expect; use crate::test_util::{StorageType, setup_grpc_server}; use crate::tests::test_util::MockInstance; + struct SlowFlightCraft; + + fn slow_recordbatch_stream() -> SendableRecordBatchStream { + let schema = Arc::new(Schema::new(vec![ColumnSchema::new( + "value", + ConcreteDataType::int32_datatype(), + false, + )])); + let recordbatch = RecordBatch::new( + schema.clone(), + vec![Arc::new(Int32Vector::from_vec(vec![1])) as VectorRef], + ) + .unwrap(); + + RecordBatches::try_new(schema, vec![recordbatch]) + .unwrap() + .as_stream() + } + + #[async_trait::async_trait] + impl FlightCraft for SlowFlightCraft { + async fn do_get( + &self, + _: tonic::Request, + ) -> std::result::Result>, tonic::Status> { + let stream = FlightRecordBatchStream::new( + FlightRecordBatchSource::initializer(async { + tokio::time::sleep(Duration::from_secs(2)).await; + Ok(slow_recordbatch_stream()) + }), + TracingContext::default(), + FlightCompression::default(), + session::context::QueryContext::arc(), + ); + + Ok(Response::new(Box::pin(stream))) + } + } + + #[tokio::test(flavor = "multi_thread")] + async fn test_do_get_timeout_does_not_cancel_slow_flight_stream() { + let (client_io, server_io) = tokio::io::duplex(1024); + tokio::spawn(async move { + TonicServer::builder() + .add_service(FlightServiceServer::new(FlightCraftWrapper( + SlowFlightCraft, + ))) + .serve_with_incoming(futures::stream::iter(vec![Ok::<_, std::io::Error>( + server_io, + )])) + .await + .unwrap(); + }); + + let channel_manager = ChannelManager::with_config(ChannelConfig::new().timeout(None), None); + let mut client_io = Some(client_io); + channel_manager + .reset_with_connector( + "slow-flight", + service_fn(move |_| { + let client_io = client_io.take(); + + async move { + client_io + .map(TokioIo::new) + .ok_or_else(|| std::io::Error::other("Client already taken")) + } + }), + ) + .unwrap(); + let client = Client::with_manager_and_urls(channel_manager, ["slow-flight"]); + let mut flight_client = client.make_flight_client(false, false).unwrap(); + + let start = Instant::now(); + let mut request = tonic::Request::new(Ticket::default()); + request.set_timeout(Duration::from_secs(1)); + let response = flight_client.mut_inner().do_get(request).await.unwrap(); + assert!(start.elapsed() < Duration::from_secs(1)); + let mut stream = response.into_inner(); + + let start = Instant::now(); + assert!(stream.message().await.unwrap().is_some()); + assert!(start.elapsed() >= Duration::from_secs(1)); + + assert!(stream.message().await.unwrap().is_some()); + } #[tokio::test(flavor = "multi_thread")] async fn test_standalone_flight_do_put() { common_telemetry::init_default_ut_logging(); diff --git a/tests-integration/src/tests/instance_test.rs b/tests-integration/src/tests/instance_test.rs index f21b459c2a..a716284c75 100644 --- a/tests-integration/src/tests/instance_test.rs +++ b/tests-integration/src/tests/instance_test.rs @@ -3360,19 +3360,21 @@ CREATE TABLE b ( let output = execute_sql(&instance, "SHOW CREATE TABLE b").await.data; let expected = r#" -+-------+----------------------------------+ -| Table | Create Table | -+-------+----------------------------------+ -| b | CREATE TABLE IF NOT EXISTS "b" ( | -| | "j" JSON2 NULL, | -| | "ts" TIMESTAMP(3) NOT NULL, | -| | TIME INDEX ("ts") | -| | ) | -| | | -| | ENGINE=mito | -| | WITH( | -| | append_mode = 'true' | -| | ) | -+-------+----------------------------------+"#; ++-------+-----------------------------------+ +| Table | Create Table | ++-------+-----------------------------------+ +| b | CREATE TABLE IF NOT EXISTS "b" ( | +| | "j" JSON2( | +| | max_auto_expanded_paths = 100 | +| | ) NULL, | +| | "ts" TIMESTAMP(3) NOT NULL, | +| | TIME INDEX ("ts") | +| | ) | +| | | +| | ENGINE=mito | +| | WITH( | +| | append_mode = 'true' | +| | ) | ++-------+-----------------------------------+"#; check_output_stream(output, expected).await; } diff --git a/tests-integration/tests/region_migration.rs b/tests-integration/tests/region_migration.rs index 17a58de2ee..0df42ee1e8 100644 --- a/tests-integration/tests/region_migration.rs +++ b/tests-integration/tests/region_migration.rs @@ -25,8 +25,10 @@ use common_event_recorder::{ DEFAULT_EVENTS_TABLE_NAME, DEFAULT_FLUSH_INTERVAL_SECONDS, EVENTS_TABLE_TIMESTAMP_COLUMN_NAME, EVENTS_TABLE_TYPE_COLUMN_NAME, PersistentEventContext, TriggerReason, }; +use common_meta::distributed_time_constants::default_distributed_time_constants; use common_meta::key::{RegionDistribution, RegionRoleSet, TableMetadataManagerRef}; use common_meta::peer::Peer; +use common_meta::rpc::store::BatchDeleteRequest; use common_procedure::ProcedureContext; use common_procedure::event::{ EVENTS_TABLE_PROCEDURE_ID_COLUMN_NAME, EVENTS_TABLE_PROCEDURE_STATE_COLUMN_NAME, @@ -46,6 +48,7 @@ use futures::future::BoxFuture; use meta_srv::error; use meta_srv::error::Result as MetaResult; use meta_srv::event::region_migration::REGION_MIGRATION_EVENT_TYPE; +use meta_srv::key::DatanodeLeaseKey; use meta_srv::metasrv::SelectorContext; use meta_srv::procedure::region_migration::{ RegionMigrationProcedureTask, RegionMigrationTriggerReason, @@ -102,6 +105,7 @@ macro_rules! region_migration_tests { test_region_migration, test_region_migration_by_sql, + test_region_migration_with_offline_source_by_sql, test_region_migration_multiple_regions, test_region_migration_all_regions, test_region_migration_incorrect_from_peer, @@ -419,8 +423,24 @@ pub async fn test_metric_table_region_migration_by_sql( .await; } -/// A naive region migration test by SQL function +/// A naive region migration test by SQL function. pub async fn test_region_migration_by_sql(store_type: StorageType, endpoints: Vec) { + test_region_migration_by_sql_inner(store_type, endpoints, false).await; +} + +/// A region migration test by SQL function with an offline source datanode. +pub async fn test_region_migration_with_offline_source_by_sql( + store_type: StorageType, + endpoints: Vec, +) { + test_region_migration_by_sql_inner(store_type, endpoints, true).await; +} + +async fn test_region_migration_by_sql_inner( + store_type: StorageType, + endpoints: Vec, + simulate_offline_source: bool, +) { let cluster_name = "test_region_migration"; let peer_factory = |id| Peer { id, @@ -437,7 +457,7 @@ pub async fn test_region_migration_by_sql(store_type: StorageType, endpoints: Ve peer_factory(2), peer_factory(3), ])); - let cluster = builder + let mut cluster = builder .with_datanodes(datanodes as u32) .with_store_config(store_config) .with_datanode_wal_config(DatanodeWalConfig::Kafka(DatanodeKafkaConfig { @@ -463,6 +483,7 @@ pub async fn test_region_migration_by_sql(store_type: StorageType, endpoints: Ve .with_meta_selector(const_selector.clone()) .build(true) .await; + let table_metadata_manager = cluster.metasrv.table_metadata_manager().clone(); let (actor_db, _actor_grpc_server) = setup_authenticated_grpc_database( cluster.fe_instance().clone(), PROCEDURE_ACTOR, @@ -487,7 +508,7 @@ pub async fn test_region_migration_by_sql(store_type: StorageType, endpoints: Ve let old_distribution = distribution.clone(); // Selecting target of region migration. - let region_migration_manager = cluster.metasrv.region_migration_manager(); + let region_migration_manager = cluster.metasrv.region_migration_manager().clone(); let (from_peer_id, from_regions) = distribution.pop_first().unwrap(); info!( "Selecting from peer: {from_peer_id}, and regions: {:?}", @@ -545,6 +566,114 @@ pub async fn test_region_migration_by_sql(store_type: StorageType, endpoints: Ve // Asserts the writes. assert_values(cluster.fe_instance()).await; + if simulate_offline_source { + let mut expected_distribution = + find_region_distribution(&table_metadata_manager, table_id).await; + let (offline_from_peer_id, offline_region_number) = expected_distribution + .iter() + .find(|(peer_id, regions)| { + **peer_id != from_peer_id + && **peer_id != to_peer_id + && !regions.leader_regions.is_empty() + }) + .map(|(peer_id, regions)| (*peer_id, regions.leader_regions[0])) + .unwrap(); + let offline_region_id = RegionId::new(table_id, offline_region_number); + + // Simulates scale-in: the source datanode stops and its lease disappears. + cluster + .datanode_instances + .get_mut(&offline_from_peer_id) + .unwrap() + .shutdown() + .await + .unwrap(); + let source_lease_key: Vec = DatanodeLeaseKey { + node_id: offline_from_peer_id, + } + .try_into() + .unwrap(); + cluster + .metasrv + .in_memory() + .batch_delete(BatchDeleteRequest { + keys: vec![source_lease_key], + prev_kv: false, + }) + .await + .unwrap(); + + let offline_procedure_id = trigger_migration_by_sql( + &cluster, + offline_region_id.as_u64(), + offline_from_peer_id, + from_peer_id, + ) + .await; + let frontend = cluster.fe_instance().clone(); + let procedure_id_for_closure = offline_procedure_id.clone(); + wait_condition( + default_distributed_time_constants().region_lease + Duration::from_secs(10), + Box::pin(async move { + loop { + let state = query_procedure_by_sql(&frontend, &procedure_id_for_closure).await; + if state == "{\"status\":\"Done\"}" { + info!("Offline-source migration done: {state}"); + break; + } + info!("Offline-source migration not finished: {state}"); + tokio::time::sleep(Duration::from_millis(200)).await; + } + }), + ) + .await; + + check_region_migration_events_system_table( + cluster.fe_instance(), + &offline_procedure_id, + offline_region_id.as_u64(), + offline_from_peer_id, + from_peer_id, + Some("greptime"), + ) + .await; + + let remove_offline_source = { + let source_regions = expected_distribution + .get_mut(&offline_from_peer_id) + .unwrap(); + source_regions + .leader_regions + .retain(|region_number| *region_number != offline_region_number); + source_regions.leader_regions.is_empty() && source_regions.follower_regions.is_empty() + }; + if remove_offline_source { + expected_distribution.remove(&offline_from_peer_id); + } + let target_regions = expected_distribution.entry(from_peer_id).or_default(); + target_regions.add_leader_region(offline_region_number); + target_regions.sort(); + + let table_metadata_manager = table_metadata_manager.clone(); + wait_condition( + Duration::from_secs(10), + Box::pin(async move { + loop { + let distribution = + find_region_distribution(&table_metadata_manager, table_id).await; + if distribution == expected_distribution { + break; + } + info!("Offline-source migration has unexpected distribution: {distribution:?}"); + tokio::time::sleep(Duration::from_millis(200)).await; + } + }), + ) + .await; + + assert_values(cluster.fe_instance()).await; + } + // Triggers again. let err = region_migration_manager .submit_procedure( diff --git a/tests-integration/tests/sql.rs b/tests-integration/tests/sql.rs index e46a8f58d1..c95adb483e 100644 --- a/tests-integration/tests/sql.rs +++ b/tests-integration/tests/sql.rs @@ -89,7 +89,10 @@ macro_rules! sql_tests { test_postgres_explain_bind_parameter, test_postgres_array_types, test_mysql_prepare_stmt_insert_timestamp, + test_mysql_prepare_stmt_timezone, test_mysql_federated_prepare_stmt, + test_mysql_prepare_tql_and_show, + test_postgres_extended_query_row_returning_statements, test_declare_fetch_close_cursor, test_alter_update_on, ); @@ -1486,6 +1489,280 @@ fn assert_pg_numeric_range_error(error: tokio_postgres::Error) { assert_eq!("numeric_value_out_of_range", error.message()); } +pub async fn test_postgres_extended_query_row_returning_statements(store_type: StorageType) { + // Regression test for the tokio-postgres >= 0.7.14 DataRow/RowDescription + // mismatch: statements answered with NoData at Describe but emitting + // DataRows at Execute must describe their real output schema. + let (mut guard, fe_pg_server) = setup_pg_server(store_type, "test_pg_extended_row_stmts").await; + let addr = fe_pg_server.bind_addr().unwrap().to_string(); + + let (client, connection) = tokio_postgres::connect(&format!("postgres://{addr}/public"), NoTls) + .await + .unwrap(); + + let (tx, rx) = tokio::sync::oneshot::channel(); + tokio::spawn(async move { + connection.await.unwrap(); + tx.send(()).unwrap(); + }); + + client + .execute( + "CREATE TABLE demo_metrics (ts timestamp time index, val double, host string primary key skipping index)", + &[], + ) + .await + .unwrap(); + client + .execute( + "INSERT INTO demo_metrics (ts, host, val) VALUES (1000, 'host-a', 1.0), (2000, 'host-b', 2.0)", + &[], + ) + .await + .unwrap(); + + // ---- SHOW DATABASES: single `Database` column ---- + let rows = client.query("SHOW DATABASES", &[]).await.unwrap(); + assert!(!rows.is_empty()); + assert_eq!(1, rows[0].columns().len()); + assert_eq!("Database", rows[0].columns()[0].name()); + assert!(rows.iter().any(|r| r.get::<_, String>(0) == "public")); + + // ---- SHOW FULL DATABASES: `Database` + `Options` columns ---- + let rows = client.query("SHOW FULL DATABASES", &[]).await.unwrap(); + assert!(!rows.is_empty()); + assert_eq!(2, rows[0].columns().len()); + assert_eq!("Database", rows[0].columns()[0].name()); + assert_eq!("Options", rows[0].columns()[1].name()); + assert!(rows.iter().any(|r| r.get::<_, String>(0) == "public")); + + // ---- SHOW TABLES: single `Tables_in_` column ---- + let rows = client.query("SHOW TABLES", &[]).await.unwrap(); + assert!(!rows.is_empty()); + assert_eq!(1, rows[0].columns().len()); + assert_eq!("Tables_in_public", rows[0].columns()[0].name()); + assert!(rows.iter().any(|r| r.get::<_, String>(0) == "demo_metrics")); + + // ---- SHOW FULL TABLES: `Tables_in_` + `Table_type` columns ---- + let rows = client.query("SHOW FULL TABLES", &[]).await.unwrap(); + assert!(!rows.is_empty()); + assert_eq!(2, rows[0].columns().len()); + assert_eq!("Tables_in_public", rows[0].columns()[0].name()); + assert_eq!("Table_type", rows[0].columns()[1].name()); + + // ---- SHOW VIEWS / SHOW FLOWS: empty results, still described with one column ---- + let rows = client.query("SHOW VIEWS", &[]).await.unwrap(); + assert!(rows.is_empty()); + let stmt = client.prepare("SHOW VIEWS").await.unwrap(); + assert_eq!(1, stmt.columns().len()); + assert_eq!("Views", stmt.columns()[0].name()); + let rows = client.query("SHOW FLOWS", &[]).await.unwrap(); + assert!(rows.is_empty()); + let stmt = client.prepare("SHOW FLOWS").await.unwrap(); + assert_eq!(1, stmt.columns().len()); + assert_eq!("Flows", stmt.columns()[0].name()); + + // ---- SHOW TABLE STATUS: fixed eighteen-column schema ---- + let rows = client.query("SHOW TABLE STATUS", &[]).await.unwrap(); + assert!(!rows.is_empty()); + let names: Vec<&str> = rows[0].columns().iter().map(|c| c.name()).collect(); + assert_eq!( + vec![ + "Name", + "Engine", + "Version", + "Row_format", + "Rows", + "Avg_row_length", + "Data_length", + "Max_data_length", + "Index_length", + "Data_free", + "Auto_increment", + "Create_time", + "Update_time", + "Check_time", + "Collation", + "Checksum", + "Create_options", + "Comment", + ], + names + ); + + // ---- SHOW COLUMNS / SHOW FULL COLUMNS ---- + let rows = client + .query("SHOW COLUMNS FROM demo_metrics", &[]) + .await + .unwrap(); + assert_eq!(3, rows.len()); + let names: Vec<&str> = rows[0].columns().iter().map(|c| c.name()).collect(); + assert_eq!( + vec![ + "Field", + "Type", + "Null", + "Key", + "Default", + "Extra", + "Greptime_type" + ], + names + ); + let rows = client + .query("SHOW FULL COLUMNS FROM demo_metrics", &[]) + .await + .unwrap(); + assert_eq!(10, rows[0].columns().len()); + + // ---- SHOW CHARSET / SHOW COLLATION ---- + let rows = client.query("SHOW CHARSET", &[]).await.unwrap(); + assert!(!rows.is_empty()); + assert_eq!(4, rows[0].columns().len()); + let rows = client.query("SHOW COLLATION", &[]).await.unwrap(); + assert!(!rows.is_empty()); + assert_eq!(6, rows[0].columns().len()); + + // ---- SHOW INDEX: fixed fifteen-column schema ---- + let rows = client + .query("SHOW INDEX IN demo_metrics", &[]) + .await + .unwrap(); + assert!(!rows.is_empty()); + let names: Vec<&str> = rows[0].columns().iter().map(|c| c.name()).collect(); + assert_eq!( + vec![ + "Table", + "Non_unique", + "Key_name", + "Seq_in_index", + "Column_name", + "Collation", + "Cardinality", + "Sub_part", + "Packed", + "Null", + "Index_type", + "Comment", + "Index_comment", + "Visible", + "Expression", + ], + names + ); + + // ---- SHOW REGION ---- + let rows = client + .query("SHOW REGION IN demo_metrics", &[]) + .await + .unwrap(); + assert!(!rows.is_empty()); + assert_eq!(4, rows[0].columns().len()); + + // ---- SHOW SEARCH_PATH: single string column ---- + let rows = client.query("SHOW SEARCH_PATH", &[]).await.unwrap(); + assert_eq!(1, rows.len()); + assert_eq!(1, rows[0].columns().len()); + assert_eq!("search_path", rows[0].columns()[0].name()); + assert_eq!("public", rows[0].get::<_, String>(0)); + + // ---- SHOW VARIABLES: single column named after the variable ---- + let rows = client.query("SHOW VARIABLES timezone", &[]).await.unwrap(); + assert_eq!(1, rows.len()); + assert_eq!(1, rows[0].columns().len()); + assert_eq!("TIMEZONE", rows[0].columns()[0].name()); + let _ = rows[0].get::<_, String>(0); + + // ---- DESCRIBE TABLE: fixed six-column string schema ---- + let rows = client + .query("DESCRIBE TABLE demo_metrics", &[]) + .await + .unwrap(); + assert_eq!(3, rows.len()); + let names: Vec<&str> = rows[0].columns().iter().map(|c| c.name()).collect(); + assert_eq!( + vec!["Column", "Type", "Key", "Null", "Default", "Semantic Type"], + names + ); + // first column of each row is the column name; ensure values decode as TEXT + let columns: Vec = rows.iter().map(|r| r.get::<_, String>(0)).collect(); + assert!(columns.contains(&"ts".to_string())); + assert!(columns.contains(&"val".to_string())); + assert!(columns.contains(&"host".to_string())); + + // ---- ADMIN function: single column named after the statement ---- + let rows = client + .query("ADMIN flush_table('demo_metrics')", &[]) + .await + .unwrap(); + assert_eq!(1, rows.len()); + assert_eq!(1, rows[0].columns().len()); + assert!(rows[0].columns()[0].name().contains("flush_table")); + + // ---- TQL EVAL / EXPLAIN / ANALYZE: described from the planned query ---- + let rows = client + .query("TQL EVAL (0, 3000, '1s') demo_metrics", &[]) + .await + .unwrap(); + assert!(!rows.is_empty()); + let names: Vec<&str> = rows[0].columns().iter().map(|c| c.name()).collect(); + assert_eq!(vec!["ts", "val", "host"], names); + + let rows = client + .query("TQL EXPLAIN (0, 3000, '1s') demo_metrics", &[]) + .await + .unwrap(); + assert!(!rows.is_empty()); + assert_eq!(2, rows[0].columns().len()); + + let rows = client + .query("TQL ANALYZE (0, 3000, '1s') demo_metrics", &[]) + .await + .unwrap(); + assert!(!rows.is_empty()); + assert_eq!(3, rows[0].columns().len()); + + // FORMAT JSON variants keep working through the plan-based execution path + let rows = client + .query("TQL EXPLAIN FORMAT JSON (0, 3000, '1s') demo_metrics", &[]) + .await + .unwrap(); + assert!(!rows.is_empty()); + let rows = client + .query("TQL ANALYZE FORMAT JSON (0, 3000, '1s') demo_metrics", &[]) + .await + .unwrap(); + assert!(!rows.is_empty()); + + // ---- the same statements also work over the simple query protocol ---- + for sql in [ + "SHOW DATABASES", + "SHOW FULL TABLES", + "SHOW TABLE STATUS", + "SHOW COLUMNS FROM demo_metrics", + "SHOW CHARSET", + "SHOW COLLATION", + "SHOW INDEX IN demo_metrics", + "SHOW REGION IN demo_metrics", + "SHOW SEARCH_PATH", + "DESCRIBE TABLE demo_metrics", + "ADMIN flush_table('demo_metrics')", + "TQL EVAL (0, 3000, '1s') demo_metrics", + ] { + let msgs = client.simple_query(sql).await.unwrap(); + assert!( + msgs.iter().any(|m| matches!(m, SimpleQueryMessage::Row(_))), + "simple query {sql} should return rows" + ); + } + + drop(client); + rx.await.unwrap(); + + let _ = fe_pg_server.shutdown().await; + guard.remove_all().await; +} + pub async fn test_postgres_explain_bind_parameter(store_type: StorageType) { // Regression test for #8029: EXPLAIN / EXPLAIN ANALYZE must accept bind // parameters over the Postgres extended query protocol. @@ -1779,6 +2056,77 @@ pub async fn test_mysql_prepare_stmt_insert_timestamp(store_type: StorageType) { guard.remove_all().await; } +pub async fn test_mysql_prepare_stmt_timezone(store_type: StorageType) { + let (mut guard, server) = + setup_mysql_server(store_type, "test_mysql_prepare_stmt_timezone").await; + let addr = server.bind_addr().unwrap().to_string(); + + let mut conn = MySqlConnection::connect(&format!("mysql://{addr}/public")) + .await + .unwrap(); + + conn.execute("create table demo(i bigint, ts timestamp time index)") + .await + .unwrap(); + conn.execute("SET time_zone = 'Asia/Shanghai'") + .await + .unwrap(); + + // Server-side prepared statement: the binary DATETIME parameter must be + // interpreted in the session timezone. + sqlx::query("insert into demo values(?, ?)") + .bind(1) + .bind( + NaiveDate::from_ymd_opt(2026, 8, 13) + .and_then(|x| x.and_hms_opt(8, 0, 0)) + .unwrap(), + ) + .execute(&mut conn) + .await + .unwrap(); + + // Text protocol with an equivalent timezone-less literal one hour later. + sqlx::query("insert into demo values(2, '2026-08-13 09:00:00')") + .execute(&mut conn) + .await + .unwrap(); + + // Timestamps are read back in the session timezone, and sqlx parses the + // timezone-less wire representation as UTC. + let rows = sqlx::query("select i, ts from demo order by i") + .fetch_all(&mut conn) + .await + .unwrap(); + assert_eq!(rows.len(), 2); + let ts: DateTime = rows[0].get("ts"); + assert_eq!(ts.to_string(), "2026-08-13 08:00:00 UTC"); + let ts: DateTime = rows[1].get("ts"); + assert_eq!(ts.to_string(), "2026-08-13 09:00:00 UTC"); + + // The prepared predicate compares the same instant as the text literal. + let rows = sqlx::query("select i from demo where ts = ? order by i") + .bind( + NaiveDate::from_ymd_opt(2026, 8, 13) + .and_then(|x| x.and_hms_opt(8, 0, 0)) + .unwrap(), + ) + .fetch_all(&mut conn) + .await + .unwrap(); + assert_eq!(rows.len(), 1); + assert_eq!(rows[0].get::("i"), 1); + + let rows = sqlx::query("select i from demo where ts = '2026-08-13 08:00:00'") + .fetch_all(&mut conn) + .await + .unwrap(); + assert_eq!(rows.len(), 1); + assert_eq!(rows[0].get::("i"), 1); + + let _ = server.shutdown().await; + guard.remove_all().await; +} + pub async fn test_mysql_federated_prepare_stmt(store_type: StorageType) { common_telemetry::init_default_ut_logging(); @@ -1818,6 +2166,73 @@ pub async fn test_mysql_federated_prepare_stmt(store_type: StorageType) { guard.remove_all().await; } +pub async fn test_mysql_prepare_tql_and_show(store_type: StorageType) { + // `do_describe` now plans TQL and information-schema-backed SHOW + // statements, so MySQL prepared statements derive their column metadata + // from the planned query and execute the plan directly. This exercises + // that path end-to-end. + common_telemetry::init_default_ut_logging(); + + let (mut guard, fe_mysql_server) = + setup_mysql_server(store_type, "test_mysql_prepare_tql_and_show").await; + let addr = fe_mysql_server.bind_addr().unwrap().to_string(); + + let pool = MySqlPoolOptions::new() + .max_connections(2) + .connect(&format!("mysql://{addr}/public")) + .await + .unwrap(); + + sqlx::query( + "CREATE TABLE demo_metrics (ts timestamp time index, val double, host string primary key)", + ) + .execute(&pool) + .await + .unwrap(); + sqlx::query("INSERT INTO demo_metrics (ts, host, val) VALUES (1000, 'host-a', 1.0), (2000, 'host-b', 2.0)") + .execute(&pool) + .await + .unwrap(); + + // sqlx::query uses the binary prepared statement protocol + // (COM_STMT_PREPARE + COM_STMT_EXECUTE). + let rows = sqlx::query("TQL EVAL (0, 3000, '1s') demo_metrics") + .fetch_all(&pool) + .await + .unwrap(); + assert!(!rows.is_empty()); + assert_eq!(3, rows[0].columns().len()); + + let rows = sqlx::query("TQL ANALYZE (0, 3000, '1s') demo_metrics") + .fetch_all(&pool) + .await + .unwrap(); + assert!(!rows.is_empty()); + + // SHOW statements share the same describe path; prepared SHOW FULL + // TABLES must report both columns. + let rows = sqlx::query("SHOW TABLES").fetch_all(&pool).await.unwrap(); + assert!(!rows.is_empty()); + assert_eq!(1, rows[0].columns().len()); + + let rows = sqlx::query("SHOW FULL TABLES") + .fetch_all(&pool) + .await + .unwrap(); + assert!(!rows.is_empty()); + assert_eq!(2, rows[0].columns().len()); + + let rows = sqlx::query("SHOW DATABASES") + .fetch_all(&pool) + .await + .unwrap(); + assert!(!rows.is_empty()); + assert_eq!(1, rows[0].columns().len()); + + let _ = fe_mysql_server.shutdown().await; + guard.remove_all().await; +} + pub async fn test_postgres_array_types(store_type: StorageType) { let (mut guard, fe_pg_server) = setup_pg_server(store_type, "test_postgres_array_types").await; let addr = fe_pg_server.bind_addr().unwrap().to_string(); diff --git a/tests/cases/distributed/explain/step_aggr.result b/tests/cases/distributed/explain/step_aggr.result index 78218d514d..89634a3cc6 100644 --- a/tests/cases/distributed/explain/step_aggr.result +++ b/tests/cases/distributed/explain/step_aggr.result @@ -67,6 +67,7 @@ FROM |_|_AggregateExec: mode=Final, gby=[], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -96,6 +97,7 @@ FROM |_|_|_AggregateExec: mode=Final, gby=[], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_state(integers.i), __sum_state(integers.i), __uddsketch_state_state(Int64(128),Float64(0.01),integers.i), __hll_state(integers.i)] REDACTED @@ -148,6 +150,7 @@ FROM | physical_plan | AggregateExec: mode=Final, gby=[], aggr=[avg(integers.i)]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[avg(integers.i)]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -172,6 +175,7 @@ FROM | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[avg(integers.i)] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[avg(integers.i)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__avg_state(integers.i)] REDACTED @@ -249,6 +253,7 @@ ORDER BY |_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)]_| |_|_RepartitionExec: partitioning=REDACTED |_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -285,6 +290,7 @@ ORDER BY |_|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[__count_state(integers.i), __sum_state(integers.i), __uddsketch_state_state(Int64(128),Float64(0.01),integers.i), __hll_state(integers.i)] REDACTED @@ -362,6 +368,7 @@ ORDER BY |_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("2 seconds"),integers.ts)@0 as date_bin(Utf8("2 seconds"),integers.ts)], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)]_| |_|_RepartitionExec: partitioning=REDACTED |_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("2 seconds"),integers.ts)@0 as date_bin(Utf8("2 seconds"),integers.ts)], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -398,6 +405,7 @@ ORDER BY |_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("2 seconds"),integers.ts)@0 as date_bin(Utf8("2 seconds"),integers.ts)], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("2 seconds"),integers.ts)@0 as date_bin(Utf8("2 seconds"),integers.ts)], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("2 seconds"),integers.ts)@0 as date_bin(Utf8("2 seconds"),integers.ts)], aggr=[__count_state(integers.i), __sum_state(integers.i), __uddsketch_state_state(Int64(128),Float64(0.01),integers.i), __hll_state(integers.i)] REDACTED diff --git a/tests/cases/distributed/explain/step_aggr_advance.result b/tests/cases/distributed/explain/step_aggr_advance.result index a78a177bc3..9fc8dfe5a1 100644 --- a/tests/cases/distributed/explain/step_aggr_advance.result +++ b/tests/cases/distributed/explain/step_aggr_advance.result @@ -108,8 +108,11 @@ tql explain (1752591864, 1752592164, '30s') sum by (a, b) (max_over_time(aggr_op | | ]] | | physical_plan | SortPreservingMergeExec: [a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, greptime_timestamp@2 ASC NULLS LAST] | | | SortExec: expr=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, greptime_timestamp@2 ASC NULLS LAST], preserve_partitioning=[true] | -| | AggregateExec: mode=SinglePartitioned, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[sum(prom_max_over_time(greptime_timestamp_range,greptime_value))] | -| | MergeScanExec: REDACTED +| | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[sum(prom_max_over_time(greptime_timestamp_range,greptime_value))] | +| | RepartitionExec: partitioning=REDACTED +| | AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[sum(prom_max_over_time(greptime_timestamp_range,greptime_value))] | +| | RepartitionExec: partitioning=REDACTED +| | MergeScanExec: REDACTED | | | +---------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ @@ -127,7 +130,10 @@ tql analyze (1752591864, 1752592164, '30s') sum by (a, b) (max_over_time(aggr_op +-+-+-+ | 0_| 0_|_SortPreservingMergeExec: [a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, greptime_timestamp@2 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, greptime_timestamp@2 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=SinglePartitioned, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[sum(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[sum(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[sum(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[__sum_state(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED @@ -177,8 +183,11 @@ tql explain (1752591864, 1752592164, '30s') avg by (a) (max_over_time(aggr_optim | | ]] | | physical_plan | SortPreservingMergeExec: [a@0 ASC NULLS LAST, greptime_timestamp@1 ASC NULLS LAST] | | | SortExec: expr=[a@0 ASC NULLS LAST, greptime_timestamp@1 ASC NULLS LAST], preserve_partitioning=[true] | -| | AggregateExec: mode=SinglePartitioned, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[avg(prom_max_over_time(greptime_timestamp_range,greptime_value))] | -| | MergeScanExec: REDACTED +| | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[avg(prom_max_over_time(greptime_timestamp_range,greptime_value))] | +| | RepartitionExec: partitioning=REDACTED +| | AggregateExec: mode=Partial, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[avg(prom_max_over_time(greptime_timestamp_range,greptime_value))] | +| | RepartitionExec: partitioning=REDACTED +| | MergeScanExec: REDACTED | | | +---------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ @@ -196,7 +205,10 @@ tql analyze (1752591864, 1752592164, '30s') avg by (a) (max_over_time(aggr_optim +-+-+-+ | 0_| 0_|_SortPreservingMergeExec: [a@0 ASC NULLS LAST, greptime_timestamp@1 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[a@0 ASC NULLS LAST, greptime_timestamp@1 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=SinglePartitioned, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[avg(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[avg(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[avg(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[__avg_state(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED @@ -311,8 +323,11 @@ tql explain (1752591864, 1752592164, '30s') min by (b, c, d) (max_over_time(aggr | | ]] | | physical_plan | SortPreservingMergeExec: [b@0 ASC NULLS LAST, c@1 ASC NULLS LAST, d@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST] | | | SortExec: expr=[b@0 ASC NULLS LAST, c@1 ASC NULLS LAST, d@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST], preserve_partitioning=[true] | -| | AggregateExec: mode=SinglePartitioned, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[min(prom_max_over_time(greptime_timestamp_range,greptime_value))] | -| | MergeScanExec: REDACTED +| | AggregateExec: mode=FinalPartitioned, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[min(prom_max_over_time(greptime_timestamp_range,greptime_value))] | +| | RepartitionExec: partitioning=REDACTED +| | AggregateExec: mode=Partial, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[min(prom_max_over_time(greptime_timestamp_range,greptime_value))] | +| | RepartitionExec: partitioning=REDACTED +| | MergeScanExec: REDACTED | | | +---------------+--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ @@ -330,7 +345,10 @@ tql analyze (1752591864, 1752592164, '30s') min by (b, c, d) (max_over_time(aggr +-+-+-+ | 0_| 0_|_SortPreservingMergeExec: [b@0 ASC NULLS LAST, c@1 ASC NULLS LAST, d@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[b@0 ASC NULLS LAST, c@1 ASC NULLS LAST, d@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=SinglePartitioned, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[min(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[min(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[min(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[__min_state(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED @@ -380,7 +398,8 @@ tql explain sum(aggr_optimize_not); | | AggregateExec: mode=FinalPartitioned, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[sum(aggr_optimize_not.greptime_value)] | | | RepartitionExec: partitioning=REDACTED | | AggregateExec: mode=Partial, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[sum(aggr_optimize_not.greptime_value)] | -| | MergeScanExec: REDACTED +| | RepartitionExec: partitioning=REDACTED +| | MergeScanExec: REDACTED | | | +---------------+----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ @@ -401,6 +420,7 @@ tql analyze sum(aggr_optimize_not); |_|_|_AggregateExec: mode=FinalPartitioned, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[sum(aggr_optimize_not.greptime_value)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_AggregateExec: mode=Partial, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[sum(aggr_optimize_not.greptime_value)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[__sum_state(aggr_optimize_not.greptime_value)] REDACTED @@ -477,14 +497,17 @@ tql explain (1752591864, 1752592164, '30s') sum by (a, b, c) (rate(aggr_optimize | physical_plan | ProjectionExec: expr=[a@0 as a, b@1 as b, c@2 as c, greptime_timestamp@3 as greptime_timestamp, sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))@5 / sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))@4 as aggr_optimize_not.sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000))) / aggr_optimize_not_count.sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] | | | REDACTED | | CoalescePartitionsExec | -| | AggregateExec: mode=SinglePartitioned, gby=[a@2 as a, b@3 as b, c@4 as c, greptime_timestamp@0 as greptime_timestamp], aggr=[sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] | -| | FilterExec: prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000))@1 IS NOT NULL | -| | ProjectionExec: expr=[greptime_timestamp@4 as greptime_timestamp, prom_rate(greptime_timestamp_range@6, greptime_value@5, greptime_timestamp@4, 120000) as prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)), a@0 as a, b@1 as b, c@2 as c] | -| | PromRangeManipulateExec: req range=[1752591864000..1752592164000], interval=[30000], eval range=[120000], time index=[greptime_timestamp] | -| | PromSeriesNormalizeExec: offset=[0], time index=[greptime_timestamp], filter NaN: [true] | -| | PromSeriesDivideExec: tags=["a", "b", "c", "d"] | -| | SortExec: expr=[a@0 ASC, b@1 ASC, c@2 ASC, d@3 ASC, greptime_timestamp@4 ASC], preserve_partitioning=[true] | -| | MergeScanExec: REDACTED +| | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, c@2 as c, greptime_timestamp@3 as greptime_timestamp], aggr=[sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] | +| | RepartitionExec: partitioning=REDACTED +| | AggregateExec: mode=Partial, gby=[a@2 as a, b@3 as b, c@4 as c, greptime_timestamp@0 as greptime_timestamp], aggr=[sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] | +| | FilterExec: prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000))@1 IS NOT NULL | +| | ProjectionExec: expr=[greptime_timestamp@4 as greptime_timestamp, prom_rate(greptime_timestamp_range@6, greptime_value@5, greptime_timestamp@4, 120000) as prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)), a@0 as a, b@1 as b, c@2 as c] | +| | PromRangeManipulateExec: req range=[1752591864000..1752592164000], interval=[30000], eval range=[120000], time index=[greptime_timestamp] | +| | PromSeriesNormalizeExec: offset=[0], time index=[greptime_timestamp], filter NaN: [true] | +| | PromSeriesDivideExec: tags=["a", "b", "c", "d"] | +| | SortExec: expr=[a@0 ASC, b@1 ASC, c@2 ASC, d@3 ASC, greptime_timestamp@4 ASC], preserve_partitioning=[true] | +| | RepartitionExec: partitioning=REDACTED +| | MergeScanExec: REDACTED | | RepartitionExec: partitioning=REDACTED | | MergeSortExec: [a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST] | | | SortExec: expr=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST], preserve_partitioning=[true] | @@ -507,13 +530,16 @@ tql analyze (1752591864, 1752592164, '30s') sum by (a, b, c) (rate(aggr_optimize | 0_| 0_|_ProjectionExec: expr=[a@0 as a, b@1 as b, c@2 as c, greptime_timestamp@3 as greptime_timestamp, sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))@5 / sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))@4 as aggr_optimize_not.sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000))) / aggr_optimize_not_count.sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] REDACTED |_|_|_REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=SinglePartitioned, gby=[a@2 as a, b@3 as b, c@4 as c, greptime_timestamp@0 as greptime_timestamp], aggr=[sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, c@2 as c, greptime_timestamp@3 as greptime_timestamp], aggr=[sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[a@2 as a, b@3 as b, c@4 as c, greptime_timestamp@0 as greptime_timestamp], aggr=[sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] REDACTED |_|_|_FilterExec: prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000))@1 IS NOT NULL REDACTED |_|_|_ProjectionExec: expr=[greptime_timestamp@4 as greptime_timestamp, prom_rate(greptime_timestamp_range@6, greptime_value@5, greptime_timestamp@4, 120000) as prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)), a@0 as a, b@1 as b, c@2 as c] REDACTED |_|_|_PromRangeManipulateExec: req range=[1752591864000..1752592164000], interval=[30000], eval range=[120000], time index=[greptime_timestamp] REDACTED |_|_|_PromSeriesNormalizeExec: offset=[0], time index=[greptime_timestamp], filter NaN: [true] REDACTED |_|_|_PromSeriesDivideExec: tags=["a", "b", "c", "d"] REDACTED |_|_|_SortExec: expr=[a@0 ASC, b@1 ASC, c@2 ASC, d@3 ASC, greptime_timestamp@4 ASC], preserve_partitioning=[true] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeSortExec: [a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST] REDACTED @@ -710,9 +736,11 @@ GROUP BY | | TableScan: aggr_optimize_not | | | ]] | | physical_plan | ProjectionExec: expr=[min(aggr_optimize_not.greptime_value)@2 as min(aggr_optimize_not.greptime_value)] | -| | AggregateExec: mode=SinglePartitioned, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value)] | -| | CooperativeExec | -| | MergeScanExec: REDACTED +| | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value)] | +| | RepartitionExec: partitioning=REDACTED +| | AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value)] | +| | RepartitionExec: partitioning=REDACTED +| | MergeScanExec: REDACTED | | | +---------------+---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ @@ -736,8 +764,10 @@ GROUP BY | stage | node | plan_| +-+-+-+ | 0_| 0_|_ProjectionExec: expr=[min(aggr_optimize_not.greptime_value)@2 as min(aggr_optimize_not.greptime_value)] REDACTED -|_|_|_AggregateExec: mode=SinglePartitioned, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value)] REDACTED -|_|_|_CooperativeExec REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[__min_state(aggr_optimize_not.greptime_value)] REDACTED @@ -776,9 +806,11 @@ GROUP BY | | TableScan: aggr_optimize_not | | | ]] | | physical_plan | ProjectionExec: expr=[min(aggr_optimize_not.greptime_value)@2 + max(aggr_optimize_not.greptime_value)@3 as min(aggr_optimize_not.greptime_value) + max(aggr_optimize_not.greptime_value)] | -| | AggregateExec: mode=SinglePartitioned, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value), max(aggr_optimize_not.greptime_value)] | -| | CooperativeExec | -| | MergeScanExec: REDACTED +| | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value), max(aggr_optimize_not.greptime_value)] | +| | RepartitionExec: partitioning=REDACTED +| | AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value), max(aggr_optimize_not.greptime_value)] | +| | RepartitionExec: partitioning=REDACTED +| | MergeScanExec: REDACTED | | | +---------------+--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ @@ -802,8 +834,10 @@ GROUP BY | stage | node | plan_| +-+-+-+ | 0_| 0_|_ProjectionExec: expr=[min(aggr_optimize_not.greptime_value)@2 + max(aggr_optimize_not.greptime_value)@3 as min(aggr_optimize_not.greptime_value) + max(aggr_optimize_not.greptime_value)] REDACTED -|_|_|_AggregateExec: mode=SinglePartitioned, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value), max(aggr_optimize_not.greptime_value)] REDACTED -|_|_|_CooperativeExec REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value), max(aggr_optimize_not.greptime_value)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value), max(aggr_optimize_not.greptime_value)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[__min_state(aggr_optimize_not.greptime_value), __max_state(aggr_optimize_not.greptime_value)] REDACTED @@ -851,9 +885,11 @@ GROUP BY | | Projection: aggr_optimize_not.a, aggr_optimize_not.b, aggr_optimize_not.greptime_value | | | TableScan: aggr_optimize_not | | | ]] | -| physical_plan | AggregateExec: mode=SinglePartitioned, gby=[a@0 as a], aggr=[min(aggr_optimize_not.greptime_value)] | -| | CooperativeExec | -| | MergeScanExec: REDACTED +| physical_plan | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[min(aggr_optimize_not.greptime_value)] | +| | RepartitionExec: partitioning=REDACTED +| | AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[min(aggr_optimize_not.greptime_value)] | +| | RepartitionExec: partitioning=REDACTED +| | MergeScanExec: REDACTED | | | +---------------+----------------------------------------------------------------------------------------------------------------------------------------------------------+ @@ -886,8 +922,10 @@ GROUP BY +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=SinglePartitioned, gby=[a@0 as a], aggr=[min(aggr_optimize_not.greptime_value)] REDACTED -|_|_|_CooperativeExec REDACTED +| 0_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[min(aggr_optimize_not.greptime_value)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[min(aggr_optimize_not.greptime_value)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[__min_state(aggr_optimize_not.greptime_value)] REDACTED @@ -1017,6 +1055,7 @@ EXPLAIN SELECT COUNT(DISTINCT val_col_1) FROM step_aggr_extended; |_|_AggregateExec: mode=FinalPartitioned, gby=[alias1@0 as alias1], aggr=[]_| |_|_RepartitionExec: partitioning=REDACTED |_|_AggregateExec: mode=Partial, gby=[val_col_1@0 as alias1], aggr=[]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_ProjectionExec: expr=[val_col_1@2 as val_col_1]_| |_|_MergeScanExec: REDACTED |_|_| @@ -1060,6 +1099,7 @@ EXPLAIN SELECT pk_col_2, sum(val_col_1) FROM step_aggr_extended GROUP BY pk_col_ |_|_AggregateExec: mode=FinalPartitioned, gby=[pk_col_2@0 as pk_col_2], aggr=[sum(step_aggr_extended.val_col_1)]_| |_|_RepartitionExec: partitioning=REDACTED |_|_AggregateExec: mode=Partial, gby=[pk_col_2@0 as pk_col_2], aggr=[sum(step_aggr_extended.val_col_1)]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -1096,6 +1136,7 @@ EXPLAIN SELECT SUM(val_col_3), COUNT(val_col_2), COUNT(val_col_3), COUNT(*) FROM |_|_AggregateExec: mode=Final, gby=[], aggr=[sum(step_aggr_extended.val_col_3), count(step_aggr_extended.val_col_2), count(step_aggr_extended.val_col_3), count(Int64(1))]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[sum(step_aggr_extended.val_col_3), count(step_aggr_extended.val_col_2), count(step_aggr_extended.val_col_3), count(Int64(1))]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -1130,6 +1171,7 @@ EXPLAIN SELECT MIN(pk_col_1), MAX(val_col_2) FROM step_aggr_extended; | physical_plan | AggregateExec: mode=Final, gby=[], aggr=[min(step_aggr_extended.pk_col_1), max(step_aggr_extended.val_col_2)]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[min(step_aggr_extended.pk_col_1), max(step_aggr_extended.val_col_2)]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -1167,6 +1209,7 @@ EXPLAIN SELECT SUM(val_col_1), COUNT(*) FROM step_aggr_extended WHERE pk_col_1 = |_|_AggregateExec: mode=Final, gby=[], aggr=[sum(step_aggr_extended.val_col_1), count(Int64(1))]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[sum(step_aggr_extended.val_col_1), count(Int64(1))]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ diff --git a/tests/cases/distributed/explain/step_aggr_basic.result b/tests/cases/distributed/explain/step_aggr_basic.result index 6a00211543..cf11b7f9ef 100644 --- a/tests/cases/distributed/explain/step_aggr_basic.result +++ b/tests/cases/distributed/explain/step_aggr_basic.result @@ -60,6 +60,7 @@ FROM | physical_plan | AggregateExec: mode=Final, gby=[], aggr=[count(integers.i)]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[count(integers.i)]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -85,6 +86,7 @@ FROM | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(integers.i)] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(integers.i)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_state(integers.i)] REDACTED @@ -156,6 +158,7 @@ ORDER BY |_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(integers.i)]_| |_|_RepartitionExec: partitioning=REDACTED |_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(integers.i)]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -189,6 +192,7 @@ ORDER BY |_|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(integers.i)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[__count_state(integers.i)] REDACTED @@ -262,6 +266,7 @@ ORDER BY |_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("1 hour"),integers.ts)@0 as date_bin(Utf8("1 hour"),integers.ts)], aggr=[count(integers.i)]_| |_|_RepartitionExec: partitioning=REDACTED |_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("1 hour"),integers.ts)@0 as date_bin(Utf8("1 hour"),integers.ts)], aggr=[count(integers.i)]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -296,6 +301,7 @@ ORDER BY |_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("1 hour"),integers.ts)@0 as date_bin(Utf8("1 hour"),integers.ts)], aggr=[count(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("1 hour"),integers.ts)@0 as date_bin(Utf8("1 hour"),integers.ts)], aggr=[count(integers.i)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("1 hour"),integers.ts)@0 as date_bin(Utf8("1 hour"),integers.ts)], aggr=[__count_state(integers.i)] REDACTED @@ -377,6 +383,7 @@ ORDER BY |_|_AggregateExec: mode=FinalPartitioned, gby=[integers.ts + Int64(1)@0 as integers.ts + Int64(1), integers.i / Int64(2)@1 as integers.i / Int64(2)], aggr=[count(integers.i)] | |_|_RepartitionExec: partitioning=REDACTED |_|_AggregateExec: mode=Partial, gby=[integers.ts + Int64(1)@0 as integers.ts + Int64(1), integers.i / Int64(2)@1 as integers.i / Int64(2)], aggr=[count(integers.i)]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -413,6 +420,7 @@ ORDER BY |_|_|_AggregateExec: mode=FinalPartitioned, gby=[integers.ts + Int64(1)@0 as integers.ts + Int64(1), integers.i / Int64(2)@1 as integers.i / Int64(2)], aggr=[count(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_AggregateExec: mode=Partial, gby=[integers.ts + Int64(1)@0 as integers.ts + Int64(1), integers.i / Int64(2)@1 as integers.i / Int64(2)], aggr=[count(integers.i)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[integers.ts + Int64(1)@0 as integers.ts + Int64(1), integers.i / Int64(2)@1 as integers.i / Int64(2)], aggr=[__count_state(integers.i)] REDACTED @@ -500,6 +508,7 @@ FROM |_|_AggregateExec: mode=Final, gby=[], aggr=[uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state), hll_merge(sink_table.hll_state)]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state), hll_merge(sink_table.hll_state)]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -527,6 +536,7 @@ FROM |_|_|_AggregateExec: mode=Final, gby=[], aggr=[uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state), hll_merge(sink_table.hll_state)] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state), hll_merge(sink_table.hll_state)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__uddsketch_merge_state(Int64(128),Float64(0.01),sink_table.udd_state), __hll_merge_state(sink_table.hll_state)] REDACTED diff --git a/tests/cases/distributed/explain/step_aggr_massive.result b/tests/cases/distributed/explain/step_aggr_massive.result index 1da87d364e..7707f6190c 100644 --- a/tests/cases/distributed/explain/step_aggr_massive.result +++ b/tests/cases/distributed/explain/step_aggr_massive.result @@ -56,7 +56,7 @@ Affected Rows: 0 -- SQLNESS REPLACE (-+) - -- SQLNESS REPLACE (\s\s+) _ --- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED +-- SQLNESS REPLACE (?m)^.*RepartitionExec:\spartitioning=RoundRobinBatch.*\n -- SQLNESS REPLACE (Hash.*) REDACTED -- SQLNESS REPLACE (peers.*) REDACTED EXPLAIN @@ -261,7 +261,7 @@ GROUP BY +-+-+ -- SQLNESS REPLACE (metrics.*) REDACTED --- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED +-- SQLNESS REPLACE (?m)^.*RepartitionExec:\spartitioning=RoundRobinBatch.*\n -- SQLNESS REPLACE (-+) - -- SQLNESS REPLACE (\s\s+) _ -- SQLNESS REPLACE (peers.*) REDACTED @@ -578,7 +578,7 @@ GROUP BY -- SQLNESS REPLACE (-+) - -- SQLNESS REPLACE (\s\s+) _ --- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED +-- SQLNESS REPLACE (?m)^.*RepartitionExec:\spartitioning=RoundRobinBatch.*\n -- SQLNESS REPLACE (Hash.*) REDACTED -- SQLNESS REPLACE (peers.*) REDACTED EXPLAIN @@ -608,7 +608,7 @@ where +-+-+ -- SQLNESS REPLACE (metrics.*) REDACTED --- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED +-- SQLNESS REPLACE (?m)^.*RepartitionExec:\spartitioning=RoundRobinBatch.*\n -- SQLNESS REPLACE (-+) - -- SQLNESS REPLACE (\s\s+) _ -- SQLNESS REPLACE (peers.*) REDACTED diff --git a/tests/cases/distributed/explain/step_aggr_massive.sql b/tests/cases/distributed/explain/step_aggr_massive.sql index e30bf79a67..6ce2ffbb98 100644 --- a/tests/cases/distributed/explain/step_aggr_massive.sql +++ b/tests/cases/distributed/explain/step_aggr_massive.sql @@ -54,7 +54,7 @@ CREATE TABLE IF NOT EXISTS base_table ( -- SQLNESS REPLACE (-+) - -- SQLNESS REPLACE (\s\s+) _ --- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED +-- SQLNESS REPLACE (?m)^.*RepartitionExec:\spartitioning=RoundRobinBatch.*\n -- SQLNESS REPLACE (Hash.*) REDACTED -- SQLNESS REPLACE (peers.*) REDACTED EXPLAIN @@ -241,7 +241,7 @@ GROUP BY date_bin('60 seconds' :: INTERVAL, time) :: TIMESTAMP(0); -- SQLNESS REPLACE (metrics.*) REDACTED --- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED +-- SQLNESS REPLACE (?m)^.*RepartitionExec:\spartitioning=RoundRobinBatch.*\n -- SQLNESS REPLACE (-+) - -- SQLNESS REPLACE (\s\s+) _ -- SQLNESS REPLACE (peers.*) REDACTED @@ -434,7 +434,7 @@ GROUP BY -- SQLNESS REPLACE (-+) - -- SQLNESS REPLACE (\s\s+) _ --- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED +-- SQLNESS REPLACE (?m)^.*RepartitionExec:\spartitioning=RoundRobinBatch.*\n -- SQLNESS REPLACE (Hash.*) REDACTED -- SQLNESS REPLACE (peers.*) REDACTED EXPLAIN @@ -446,7 +446,7 @@ where time >= 0; -- SQLNESS REPLACE (metrics.*) REDACTED --- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED +-- SQLNESS REPLACE (?m)^.*RepartitionExec:\spartitioning=RoundRobinBatch.*\n -- SQLNESS REPLACE (-+) - -- SQLNESS REPLACE (\s\s+) _ -- SQLNESS REPLACE (peers.*) REDACTED diff --git a/tests/cases/distributed/explain/subqueries.result b/tests/cases/distributed/explain/subqueries.result index e6f7286535..d7ad6ed920 100644 --- a/tests/cases/distributed/explain/subqueries.result +++ b/tests/cases/distributed/explain/subqueries.result @@ -101,9 +101,9 @@ order by t.i desc; | physical_plan | SortPreservingMergeExec: [i@0 DESC]_| |_|_SortExec: expr=[i@0 DESC], preserve_partitioning=[true]_| |_|_CrossJoinExec_| -|_|_CoalescePartitionsExec_| |_|_ProjectionExec: expr=[i@0 as i]_| |_|_MergeScanExec: REDACTED +|_|_RepartitionExec: partitioning=REDACTED |_|_ProjectionExec: expr=[]_| |_|_MergeScanExec: REDACTED |_|_| @@ -179,7 +179,9 @@ EXPLAIN SELECT * FROM integers i1 WHERE EXISTS(SELECT i FROM integers WHERE i=i1 | physical_plan | SortPreservingMergeExec: [i@0 ASC NULLS LAST]_| |_|_SortExec: expr=[i@0 ASC NULLS LAST], preserve_partitioning=[true]_| |_|_REDACTED +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -212,7 +214,9 @@ EXPLAIN SELECT * FROM integers i1 WHERE EXISTS(SELECT count(i) FROM integers WHE | physical_plan | SortPreservingMergeExec: [i@0 ASC NULLS LAST]_| |_|_SortExec: expr=[i@0 ASC NULLS LAST], preserve_partitioning=[true]_| |_|_REDACTED +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED +|_|_RepartitionExec: partitioning=REDACTED |_|_ProjectionExec: expr=[i@1 as i]_| |_|_MergeScanExec: REDACTED |_|_| @@ -363,7 +367,10 @@ EXPLAIN SELECT DISTINCT x FROM (SELECT a AS x FROM t) sq ORDER BY x; |_| ]]_| | physical_plan | SortPreservingMergeExec: [x@0 ASC NULLS LAST]_| |_|_SortExec: expr=[x@0 ASC NULLS LAST], preserve_partitioning=[true]_| -|_|_AggregateExec: mode=SinglePartitioned, gby=[x@0 as x], aggr=[]_| +|_|_AggregateExec: mode=FinalPartitioned, gby=[x@0 as x], aggr=[]_| +|_|_RepartitionExec: partitioning=REDACTED +|_|_AggregateExec: mode=Partial, gby=[x@0 as x], aggr=[]_| +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -560,7 +567,9 @@ ORDER BY u1.a; | physical_plan | SortPreservingMergeExec: [a@0 ASC NULLS LAST]_| |_|_SortExec: expr=[a@0 ASC NULLS LAST], preserve_partitioning=[true]_| |_|_REDACTED +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED +|_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ diff --git a/tests/cases/distributed/flow-tql/tsid_on_phy.result b/tests/cases/distributed/flow-tql/tsid_on_phy.result index 857e45d16a..68dd01374c 100644 --- a/tests/cases/distributed/flow-tql/tsid_on_phy.result +++ b/tests/cases/distributed/flow-tql/tsid_on_phy.result @@ -119,8 +119,8 @@ TQL EXPLAIN ( | | TableScan: phy projection=[ts, v, tag1, tag2, le, tag4, tag5, tag6, tag7, tag8, __table_id, __tsid], partial_filters=[phy.ts >= TimestampMillisecond(1769137200001, None), phy.ts <= TimestampMillisecond(1769139900000, None), phy.__table_id=UInt32(REDACTED)] | | | ]] | | physical_plan | HistogramFoldExec: le=@0, field=@4, quantile=0.5 | -| | SortExec: expr=[tag4@1 ASC NULLS LAST, tag5@2 ASC NULLS LAST, ts@3 ASC NULLS LAST, CAST(le@0 AS Float64) ASC NULLS LAST], preserve_partitioning=[true] | -| | RepartitionExec: REDACTED +| | RepartitionExec: REDACTED +| | SortExec: expr=[tag4@1 ASC NULLS LAST, tag5@2 ASC NULLS LAST, ts@3 ASC NULLS LAST, CAST(le@0 AS Float64) ASC NULLS LAST], preserve_partitioning=[false] | | | MergeScanExec: REDACTED | | | +---------------+--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ diff --git a/tests/cases/distributed/optimizer/count.result b/tests/cases/distributed/optimizer/count.result index a339bc3bf4..7368ad6bbd 100644 --- a/tests/cases/distributed/optimizer/count.result +++ b/tests/cases/distributed/optimizer/count.result @@ -281,6 +281,7 @@ select count(1) from count_where_bug; | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_ProjectionExec: expr=[{count[count]:REDACTED} as __count_state(count_where_bug.ts)] REDACTED @@ -316,6 +317,7 @@ select count(1) from count_where_bug where `tag` = 'b'; | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_state(count_where_bug.ts)] REDACTED @@ -342,6 +344,7 @@ select count(1) from count_where_bug where `tag` = 'b'; | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_state(count_where_bug.ts)] REDACTED @@ -378,6 +381,7 @@ select count(1) from count_where_bug where ts > '2024-09-06T06:00:04Z'; | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_state(count_where_bug.ts)] REDACTED @@ -417,6 +421,7 @@ select count(1) from count_where_bug where num != 3; | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_state(count_where_bug.ts)] REDACTED diff --git a/tests/cases/distributed/optimizer/first_value_advance.result b/tests/cases/distributed/optimizer/first_value_advance.result index 08e1b8a411..519ec9d9c1 100644 --- a/tests/cases/distributed/optimizer/first_value_advance.result +++ b/tests/cases/distributed/optimizer/first_value_advance.result @@ -325,6 +325,7 @@ explain select first_value(ts order by ts) from t; | physical_plan | AggregateExec: mode=Final, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -348,6 +349,7 @@ explain analyze | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED @@ -414,6 +416,7 @@ order by time_window, ordered_host; |_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -447,6 +450,7 @@ order by time_window, ordered_host; |_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED @@ -525,6 +529,7 @@ explain | physical_plan | AggregateExec: mode=Final, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -548,6 +553,7 @@ explain analyze | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED @@ -711,6 +717,7 @@ order by time_window, ordered_host; |_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -744,6 +751,7 @@ order by time_window, ordered_host; |_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED diff --git a/tests/cases/distributed/optimizer/last_value_advance.result b/tests/cases/distributed/optimizer/last_value_advance.result index 7b131d44f3..6269c5d397 100644 --- a/tests/cases/distributed/optimizer/last_value_advance.result +++ b/tests/cases/distributed/optimizer/last_value_advance.result @@ -325,6 +325,7 @@ explain select last_value(ts order by ts) from t; | physical_plan | AggregateExec: mode=Final, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -348,6 +349,7 @@ explain analyze | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED @@ -414,6 +416,7 @@ order by time_window, ordered_host; |_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -447,6 +450,7 @@ order by time_window, ordered_host; |_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED @@ -525,6 +529,7 @@ explain | physical_plan | AggregateExec: mode=Final, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -548,6 +553,7 @@ explain analyze | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED @@ -711,6 +717,7 @@ order by time_window, ordered_host; |_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -744,6 +751,7 @@ order by time_window, ordered_host; |_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED diff --git a/tests/cases/distributed/optimizer/range_select_projection.result b/tests/cases/distributed/optimizer/range_select_projection.result index c32878abbf..451e743da9 100644 --- a/tests/cases/distributed/optimizer/range_select_projection.result +++ b/tests/cases/distributed/optimizer/range_select_projection.result @@ -48,7 +48,7 @@ ORDER BY station, "channel", ts; | 0_| 0_|_SortExec: expr=[station@1 ASC NULLS LAST, channel@2 ASC NULLS LAST, ts@0 ASC NULLS LAST], preserve_REDACTED |_|_|_ProjectionExec: expr=[ts@1 as ts, station@2 as station, channel@3 as channel, avg(range_select_projection.value_a + range_select_projection.value_b) RANGE 10s@0 as avg_value] REDACTED |_|_|_RangeSelectExec: range_expr=[avg(range_select_projection.value_a + range_select_projection.value_b) RANGE 10s], align=5000ms, align_to=0ms, align_by=[station@1, channel@2], time_index=ts REDACTED -|_|_|_CoalescePartitionsExec REDACTED +|_|_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_CooperativeExec REDACTED diff --git a/tests/cases/distributed/optimizer/time_index_filter_pushdown.result b/tests/cases/distributed/optimizer/time_index_filter_pushdown.result index 016c49deea..76aa2dd707 100644 --- a/tests/cases/distributed/optimizer/time_index_filter_pushdown.result +++ b/tests/cases/distributed/optimizer/time_index_filter_pushdown.result @@ -91,6 +91,7 @@ group by -- SQLNESS REPLACE (\s\s+) _ -- SQLNESS REPLACE (peers.*) REDACTED -- SQLNESS REPLACE (?m)^\|_\|_SortExec:.*\n +-- SQLNESS REPLACE (?m)^\|_\|_(RepartitionExec:\spartitioning=RoundRobinBatch\(\d+\),\sinput_partitions=\d+|CooperativeExec)_\|\n EXPLAIN SELECT rack, os, @@ -131,7 +132,6 @@ WHERE |_|_MergeSortExec: [greptime_timestamp@0 DESC], fetch=1_| |_|_MergeScanExec: REDACTED |_|_ProjectionExec: expr=[rack@0 as rack, os@1 as os, greptime_timestamp@3 as greptime_timestamp]_| -|_|_CooperativeExec_| |_|_MergeScanExec: REDACTED |_|_| +-+-+ diff --git a/tests/cases/distributed/optimizer/time_index_filter_pushdown.sql b/tests/cases/distributed/optimizer/time_index_filter_pushdown.sql index c9c80c24a6..07f53d8484 100644 --- a/tests/cases/distributed/optimizer/time_index_filter_pushdown.sql +++ b/tests/cases/distributed/optimizer/time_index_filter_pushdown.sql @@ -66,6 +66,7 @@ group by -- SQLNESS REPLACE (\s\s+) _ -- SQLNESS REPLACE (peers.*) REDACTED -- SQLNESS REPLACE (?m)^\|_\|_SortExec:.*\n +-- SQLNESS REPLACE (?m)^\|_\|_(RepartitionExec:\spartitioning=RoundRobinBatch\(\d+\),\sinput_partitions=\d+|CooperativeExec)_\|\n EXPLAIN SELECT rack, os, diff --git a/tests/cases/standalone/common/aggregate/distinct.result b/tests/cases/standalone/common/aggregate/distinct.result index 5a721b7a48..5c10ee2223 100644 --- a/tests/cases/standalone/common/aggregate/distinct.result +++ b/tests/cases/standalone/common/aggregate/distinct.result @@ -169,7 +169,10 @@ EXPLAIN ANALYZE SELECT DISTINCT a FROM test ORDER BY a; +-+-+-+ | 0_| 0_|_SortPreservingMergeExec: [a@0 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[a@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=SinglePartitioned, gby=[a@0 as a], aggr=[] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[] REDACTED @@ -199,7 +202,10 @@ EXPLAIN ANALYZE SELECT DISTINCT a, b FROM test ORDER BY a; +-+-+-+ | 0_| 0_|_SortPreservingMergeExec: [a@0 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[a@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=SinglePartitioned, gby=[a@0 as a, b@1 as b], aggr=[] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[] REDACTED diff --git a/tests/cases/standalone/common/aggregate/multi_regions.result b/tests/cases/standalone/common/aggregate/multi_regions.result index 4b104df849..8627b83c15 100644 --- a/tests/cases/standalone/common/aggregate/multi_regions.result +++ b/tests/cases/standalone/common/aggregate/multi_regions.result @@ -62,6 +62,7 @@ select sum(val) from t; | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[sum(t.val)] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[sum(t.val)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__sum_state(t.val)] REDACTED @@ -96,6 +97,7 @@ select sum(val) from t group by idc; |_|_|_AggregateExec: mode=FinalPartitioned, gby=[idc@0 as idc], aggr=[sum(t.val)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_AggregateExec: mode=Partial, gby=[idc@0 as idc], aggr=[sum(t.val)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[idc@0 as idc], aggr=[__sum_state(t.val)] REDACTED diff --git a/tests/cases/standalone/common/create/metric_engine_partition.result b/tests/cases/standalone/common/create/metric_engine_partition.result index 0a5899bd9b..ff3fd40e94 100644 --- a/tests/cases/standalone/common/create/metric_engine_partition.result +++ b/tests/cases/standalone/common/create/metric_engine_partition.result @@ -148,7 +148,10 @@ select host, count(*) from logical_table_2 GROUP BY host ORDER BY host; | physical_plan | SortPreservingMergeExec: [host@0 ASC NULLS LAST]_| |_|_SortExec: expr=[host@0 ASC NULLS LAST], preserve_partitioning=[true]_| |_|_ProjectionExec: expr=[host@0 as host, count(Int64(1))@1 as count(*)]_| -|_|_AggregateExec: mode=SinglePartitioned, gby=[host@0 as host], aggr=[count(Int64(1))]_| +|_|_AggregateExec: mode=FinalPartitioned, gby=[host@0 as host], aggr=[count(Int64(1))]_| +|_|_RepartitionExec: REDACTED +|_|_AggregateExec: mode=Partial, gby=[host@0 as host], aggr=[count(Int64(1))]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -188,6 +191,7 @@ select ts, count(*) from logical_table_2 GROUP BY ts ORDER BY ts; |_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(Int64(1))]_| |_|_RepartitionExec: REDACTED |_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(Int64(1))]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -283,6 +287,7 @@ select a, count(*) from logical_table_3 GROUP BY a ORDER BY a; |_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count(Int64(1))]_| |_|_RepartitionExec: REDACTED |_|_AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count(Int64(1))]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -351,7 +356,7 @@ EXPLAIN select count(*) from logical_table_4; | plan_type_| plan_| +-+-+ | logical_plan_| Projection: count(Int64(1)) AS count(*)_| -|_|_Aggregate: groupBy=[[]], aggr=[[__count_merge(__count_state(logical_table_4.ts)) AS count(Int64(1))]] | +|_|_Aggregate: groupBy=[[]], aggr=[[__count_merge(__count_state(logical_table_4.ts)) AS count(Int64(1))]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[]], aggr=[[__count_state(logical_table_4.ts)]]_| |_|_TableScan: logical_table_4_| @@ -360,6 +365,7 @@ EXPLAIN select count(*) from logical_table_4; |_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -399,6 +405,7 @@ select ts, count(*) from logical_table_4 GROUP BY ts ORDER BY ts; |_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(Int64(1))]_| |_|_RepartitionExec: REDACTED |_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(Int64(1))]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ diff --git a/tests/cases/standalone/common/filter/hash_join_topk_dyn_filter.result b/tests/cases/standalone/common/filter/hash_join_topk_dyn_filter.result index a81fc4918b..888f611e19 100644 --- a/tests/cases/standalone/common/filter/hash_join_topk_dyn_filter.result +++ b/tests/cases/standalone/common/filter/hash_join_topk_dyn_filter.result @@ -36,6 +36,8 @@ Affected Rows: 3 -- and hash join generates filter on customer_id -- SQLNESS REPLACE (peers.*) REDACTED -- SQLNESS REPLACE region=\d+\(\d+,\s+\d+\) region=REDACTED +-- SQLNESS REPLACE RoundRobinBatch\(\d+\) RoundRobinBatch(REDACTED) +-- SQLNESS REPLACE (RoundRobinBatch\(REDACTED\),\sinput_partitions=\d+)\s+\| $1| -- SQLNESS REPLACE (=Hash.*) =REDACTED EXPLAIN SELECT top_orders."id", top_orders.amount, c."name", c.tier FROM ( @@ -67,11 +69,11 @@ WHERE c.tier IN ('gold', 'bronze'); | | Filter: customers.customer_id IS NOT NULL AND (customers.tier = Utf8("gold") OR customers.tier = Utf8("bronze")) | | | TableScan: customers, partial_filters=[customers.tier = Utf8("gold") OR customers.tier = Utf8("bronze"), customers.customer_id IS NOT NULL] | | | ]] | -| physical_plan | HashJoinExec: mode=Partitioned, join_type=Inner, on=[(customer_id@1, customer_id@0)], projection=[id@0, amount@2, name@4, tier@5] | -| | RepartitionExec: partitioning=REDACTED -| | ProjectionExec: expr=[id@0 as id, customer_id@1 as customer_id, amount@2 as amount] | +| physical_plan | HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(customer_id@1, customer_id@0)], projection=[id@0, amount@2, name@4, tier@5] | +| | ProjectionExec: expr=[id@0 as id, customer_id@1 as customer_id, amount@2 as amount] | +| | CooperativeExec | | | MergeScanExec: REDACTED -| | RepartitionExec: partitioning=REDACTED +| | RepartitionExec: partitioning=RoundRobinBatch(REDACTED), input_partitions=1| | | ProjectionExec: expr=[customer_id@0 as customer_id, name@1 as name, tier@2 as tier] | | | MergeScanExec: REDACTED | | | @@ -107,9 +109,9 @@ WHERE c.tier IN ('gold', 'bronze'); +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(customer_id@1, customer_id@0)], projection=[id@0, amount@2, name@4, tier@5] metrics=REDACTED_| -|_|_|_RepartitionExec: partitioning=REDACTED +| 0_| 0_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(customer_id@1, customer_id@0)], projection=[id@0, amount@2, name@4, tier@5] metrics=REDACTED_| |_|_|_ProjectionExec: expr=[id@0 as id, customer_id@1 as customer_id, amount@2 as amount] metrics=REDACTED_| +|_|_|_CooperativeExec metrics=REDACTED_| |_|_|_MergeScanExec: REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_ProjectionExec: expr=[customer_id@0 as customer_id, name@1 as name, tier@2 as tier] metrics=REDACTED_| diff --git a/tests/cases/standalone/common/filter/hash_join_topk_dyn_filter.sql b/tests/cases/standalone/common/filter/hash_join_topk_dyn_filter.sql index 97f0c1231d..e4f62aa40c 100644 --- a/tests/cases/standalone/common/filter/hash_join_topk_dyn_filter.sql +++ b/tests/cases/standalone/common/filter/hash_join_topk_dyn_filter.sql @@ -28,6 +28,8 @@ INSERT INTO customers VALUES -- and hash join generates filter on customer_id -- SQLNESS REPLACE (peers.*) REDACTED -- SQLNESS REPLACE region=\d+\(\d+,\s+\d+\) region=REDACTED +-- SQLNESS REPLACE RoundRobinBatch\(\d+\) RoundRobinBatch(REDACTED) +-- SQLNESS REPLACE (RoundRobinBatch\(REDACTED\),\sinput_partitions=\d+)\s+\| $1| -- SQLNESS REPLACE (=Hash.*) =REDACTED EXPLAIN SELECT top_orders."id", top_orders.amount, c."name", c.tier FROM ( diff --git a/tests/cases/standalone/common/insert/insert_default_timezone.result b/tests/cases/standalone/common/insert/insert_default_timezone.result index 413eb5d46a..41d8ca6cba 100644 --- a/tests/cases/standalone/common/insert/insert_default_timezone.result +++ b/tests/cases/standalone/common/insert/insert_default_timezone.result @@ -114,6 +114,27 @@ INSERT INTO test3 (ts, st, ts_ns) VALUES ( Affected Rows: 1 +-- a NULL branch must not cancel the conversion for the whole column +INSERT INTO test3 (ts, ts_ns) +SELECT '2026-08-13 12:00:00.001' AS a, '2026-08-13 12:00:00.123456789' AS b +UNION ALL +SELECT '2026-08-14 12:00:00.001', NULL; + +Affected Rows: 2 + +-- NULL in the first branch: the union's schema starts out as Null +INSERT INTO test3 (ts, ts_ns) +SELECT '2026-08-15 12:00:00.001' AS a, NULL AS b +UNION ALL +SELECT '2026-08-19 12:00:00.001', '2026-08-19 12:00:00.123456789'; + +Affected Rows: 2 + +-- the assignment cast also lands on non-literal VALUES expressions +INSERT INTO test3 (ts, st) VALUES (concat('2026-08-20 ', '12:00:00.001'), now()); + +Affected Rows: 1 + SELECT ts, ts_ns FROM test3 ORDER BY ts; +-------------------------+-------------------------------+ @@ -130,10 +151,43 @@ SELECT ts, ts_ns FROM test3 ORDER BY ts; | 2026-08-10T12:00:00.001 | | | 2026-08-11T12:00:00.001 | | | 2026-08-12T04:00:00.123 | 2026-08-12T04:00:00.123456789 | +| 2026-08-13T04:00:00.001 | 2026-08-13T04:00:00.123456789 | +| 2026-08-14T04:00:00.001 | | +| 2026-08-15T04:00:00.001 | | | 2026-08-16T04:00:00.001 | | | 2026-08-17T04:00:00.001 | | +| 2026-08-19T04:00:00.001 | 2026-08-19T04:00:00.123456789 | +| 2026-08-20T04:00:00.001 | | +-------------------------+-------------------------------+ +-- UNION dedup keys must stay on the source strings: these two spell the same +-- instant differently, so the source query yields two rows and both are kept. +CREATE TABLE test4 (ts TIMESTAMP TIME INDEX) WITH ('append_mode'='true'); + +Affected Rows: 0 + +INSERT INTO test4 (ts) +SELECT '2026-08-06 04:00:00' UNION SELECT '2026-08-06 04:00:00.000'; + +Affected Rows: 2 + +SELECT count(*) FROM test4; + ++----------+ +| count(*) | ++----------+ +| 2 | ++----------+ + +SELECT ts FROM test4 ORDER BY ts; + ++---------------------+ +| ts | ++---------------------+ +| 2026-08-05T20:00:00 | +| 2026-08-05T20:00:00 | ++---------------------+ + SET time_zone = 'UTC'; Affected Rows: 0 @@ -150,3 +204,7 @@ DROP TABLE test3; Affected Rows: 0 +DROP TABLE test4; + +Affected Rows: 0 + diff --git a/tests/cases/standalone/common/insert/insert_default_timezone.sql b/tests/cases/standalone/common/insert/insert_default_timezone.sql index 7cbef15d6d..f7c196699b 100644 --- a/tests/cases/standalone/common/insert/insert_default_timezone.sql +++ b/tests/cases/standalone/common/insert/insert_default_timezone.sql @@ -56,8 +56,34 @@ INSERT INTO test3 (ts, st, ts_ns) VALUES ( '2026-08-09 12:00:00.123456789' ); +-- a NULL branch must not cancel the conversion for the whole column +INSERT INTO test3 (ts, ts_ns) +SELECT '2026-08-13 12:00:00.001' AS a, '2026-08-13 12:00:00.123456789' AS b +UNION ALL +SELECT '2026-08-14 12:00:00.001', NULL; + +-- NULL in the first branch: the union's schema starts out as Null +INSERT INTO test3 (ts, ts_ns) +SELECT '2026-08-15 12:00:00.001' AS a, NULL AS b +UNION ALL +SELECT '2026-08-19 12:00:00.001', '2026-08-19 12:00:00.123456789'; + +-- the assignment cast also lands on non-literal VALUES expressions +INSERT INTO test3 (ts, st) VALUES (concat('2026-08-20 ', '12:00:00.001'), now()); + SELECT ts, ts_ns FROM test3 ORDER BY ts; +-- UNION dedup keys must stay on the source strings: these two spell the same +-- instant differently, so the source query yields two rows and both are kept. +CREATE TABLE test4 (ts TIMESTAMP TIME INDEX) WITH ('append_mode'='true'); + +INSERT INTO test4 (ts) +SELECT '2026-08-06 04:00:00' UNION SELECT '2026-08-06 04:00:00.000'; + +SELECT count(*) FROM test4; + +SELECT ts FROM test4 ORDER BY ts; + SET time_zone = 'UTC'; DROP TABLE test1; @@ -65,3 +91,5 @@ DROP TABLE test1; DROP TABLE test2; DROP TABLE test3; + +DROP TABLE test4; diff --git a/tests/cases/standalone/common/order/order_by_exceptions.result b/tests/cases/standalone/common/order/order_by_exceptions.result index bde25ea927..bdf215f8f2 100644 --- a/tests/cases/standalone/common/order/order_by_exceptions.result +++ b/tests/cases/standalone/common/order/order_by_exceptions.result @@ -82,9 +82,10 @@ EXPLAIN SELECT a % 2, b FROM test UNION SELECT a % 2 AS k, b FROM test ORDER BY | | AggregateExec: mode=FinalPartitioned, gby=[test.a % Int64(2)@0 as test.a % Int64(2), b@1 as b], aggr=[] | | | RepartitionExec: REDACTED | | AggregateExec: mode=Partial, gby=[test.a % Int64(2)@0 as test.a % Int64(2), b@1 as b], aggr=[] | -| | InterleaveExec | -| | MergeScanExec: REDACTED -| | MergeScanExec: REDACTED +| | RepartitionExec: REDACTED +| | InterleaveExec | +| | MergeScanExec: REDACTED +| | MergeScanExec: REDACTED | | | +---------------+-----------------------------------------------------------------------------------------------------------+ diff --git a/tests/cases/standalone/common/promql/histogram_multi_partition.result b/tests/cases/standalone/common/promql/histogram_multi_partition.result index 091ed7bfcc..063ee07384 100644 --- a/tests/cases/standalone/common/promql/histogram_multi_partition.result +++ b/tests/cases/standalone/common/promql/histogram_multi_partition.result @@ -30,6 +30,7 @@ Affected Rows: 12 -- Ensure the physical plan keeps the required repartition/order before folding buckets. -- SQLNESS REPLACE (metrics.*) REDACTED +-- SQLNESS REPLACE RoundRobinBatch\(\d+\) RoundRobinBatch(REDACTED) -- SQLNESS REPLACE (peers.*) REDACTED -- SQLNESS REPLACE region=\d+\(\d+,\s+\d+\) region=REDACTED -- SQLNESS REPLACE (-+) - @@ -47,6 +48,7 @@ tql analyze (0, 10, '10s') histogram_quantile(0.5, sum by (le) (histogram_gap_bu |_|_|_AggregateExec: mode=FinalPartitioned, gby=[le@0 as le, ts@1 as ts], aggr=[sum(histogram_gap_bucket.val)] REDACTED |_|_|_RepartitionExec: partitioning=Hash([le@0, ts@1],REDACTED |_|_|_AggregateExec: mode=Partial, gby=[le@0 as le, ts@1 as ts], aggr=[sum(histogram_gap_bucket.val)] REDACTED +|_|_|_RepartitionExec: partitioning=RoundRobinBatch(REDACTED), input_partitions=2 REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[le@0 as le, ts@1 as ts], aggr=[__sum_state(histogram_gap_bucket.val)] REDACTED diff --git a/tests/cases/standalone/common/promql/histogram_multi_partition.sql b/tests/cases/standalone/common/promql/histogram_multi_partition.sql index b360999fcf..6d441bff80 100644 --- a/tests/cases/standalone/common/promql/histogram_multi_partition.sql +++ b/tests/cases/standalone/common/promql/histogram_multi_partition.sql @@ -26,6 +26,7 @@ insert into histogram_gap_bucket values -- Ensure the physical plan keeps the required repartition/order before folding buckets. -- SQLNESS REPLACE (metrics.*) REDACTED +-- SQLNESS REPLACE RoundRobinBatch\(\d+\) RoundRobinBatch(REDACTED) -- SQLNESS REPLACE (peers.*) REDACTED -- SQLNESS REPLACE region=\d+\(\d+,\s+\d+\) region=REDACTED -- SQLNESS REPLACE (-+) - diff --git a/tests/cases/standalone/common/promql/label.result b/tests/cases/standalone/common/promql/label.result index f83bbd38e6..bd9228a324 100644 --- a/tests/cases/standalone/common/promql/label.result +++ b/tests/cases/standalone/common/promql/label.result @@ -314,6 +314,23 @@ TQL EVAL (0, 15, '5s') label_join(test{host="host1"}, "new_host", "-", "idc", "h | 1970-01-01T00:00:15 | 3 | idc2:zone1-host1 | host1 | idc2:zone1 | +---------------------+-----+------------------+-------+------------+ +-- Issue 8969 -- +-- SQLNESS SORT_RESULT 3 1 +TQL EVAL (0, 15, '5s') sum by (foo) (label_replace(test, "foo", "$1", "host", "(.*)")) * 0.8; + ++-------+---------------------+------------------------------+ +| foo | ts | sum(test.val) * Float64(0.8) | ++-------+---------------------+------------------------------+ +| host1 | 1970-01-01T00:00:00 | 0.8 | +| host1 | 1970-01-01T00:00:05 | 3.2 | +| host1 | 1970-01-01T00:00:10 | 7.2 | +| host1 | 1970-01-01T00:00:15 | 12.8 | +| host2 | 1970-01-01T00:00:00 | 1.6 | +| host2 | 1970-01-01T00:00:05 | 4.800000000000001 | +| host2 | 1970-01-01T00:00:10 | 9.600000000000001 | +| host2 | 1970-01-01T00:00:15 | 16.0 | ++-------+---------------------+------------------------------+ + DROP TABLE test; Affected Rows: 0 diff --git a/tests/cases/standalone/common/promql/label.sql b/tests/cases/standalone/common/promql/label.sql index 3fb20792e0..6a77e9b521 100644 --- a/tests/cases/standalone/common/promql/label.sql +++ b/tests/cases/standalone/common/promql/label.sql @@ -96,6 +96,10 @@ TQL EVAL (0, 15, '5s') label_replace(test{host="host1"}, "new_idc", "idc99", "id -- SQLNESS SORT_RESULT 3 1 TQL EVAL (0, 15, '5s') label_join(test{host="host1"}, "new_host", "-", "idc", "host") == 3; +-- Issue 8969 -- +-- SQLNESS SORT_RESULT 3 1 +TQL EVAL (0, 15, '5s') sum by (foo) (label_replace(test, "foo", "$1", "host", "(.*)")) * 0.8; + DROP TABLE test; CREATE TABLE test ( diff --git a/tests/cases/standalone/common/promql/tsid_binary_join_regression.result b/tests/cases/standalone/common/promql/tsid_binary_join_regression.result index 75aaf8176c..72a6dc9c5b 100644 --- a/tests/cases/standalone/common/promql/tsid_binary_join_regression.result +++ b/tests/cases/standalone/common/promql/tsid_binary_join_regression.result @@ -117,11 +117,11 @@ TQL ANALYZE (0, 5, '5s') tsid_binary_join_left / tsid_binary_join_right; +-+-+-+ | 0_| 0_|_ProjectionExec: expr=[host@2 as host, job@3 as job, ts@5 as ts, __tsid@4 as __tsid, greptime_value@0 / greptime_value@1 as tsid_binary_join_left.greptime_value / tsid_binary_join_right.greptime_value] REDACTED |_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@1, __tsid@3), (ts@2, ts@4)], projection=[greptime_value@0, greptime_value@3, host@4, job@5, __tsid@6, ts@7], NullsEqual: true REDACTED -|_|_|_CoalescePartitionsExec REDACTED |_|_|_ProjectionExec: expr=[greptime_value@0 as greptime_value, __tsid@3 as __tsid, ts@4 as ts] REDACTED -|_|_|_MergeScanExec: REDACTED |_|_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_PromInstantManipulateExec: range=[0..5000], lookback=[300000], interval=[5000], time index=[ts] REDACTED |_|_|_PromSeriesDivideExec: tags=["__tsid"] REDACTED @@ -363,11 +363,11 @@ TQL ANALYZE (0, 5, '5s') tsid_binary_join_left > bool tsid_binary_join_right; +-+-+-+ | 0_| 0_|_ProjectionExec: expr=[host@2 as host, job@3 as job, ts@5 as ts, __tsid@4 as __tsid, CAST(greptime_value@1 < greptime_value@0 AS Float64) as tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value] REDACTED |_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@1, __tsid@3), (ts@2, ts@4)], projection=[greptime_value@0, greptime_value@3, host@4, job@5, __tsid@6, ts@7], NullsEqual: true REDACTED -|_|_|_CoalescePartitionsExec REDACTED |_|_|_ProjectionExec: expr=[greptime_value@0 as greptime_value, __tsid@3 as __tsid, ts@4 as ts] REDACTED -|_|_|_MergeScanExec: REDACTED |_|_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_PromInstantManipulateExec: range=[0..5000], lookback=[300000], interval=[5000], time index=[ts] REDACTED |_|_|_PromSeriesDivideExec: tags=["__tsid"] REDACTED @@ -411,7 +411,7 @@ TQL ANALYZE (0, 5, '5s') ((tsid_binary_join_left > tsid_binary_join_right) / tsi |_|_|_RepartitionExec: partitioning=Hash([REDACTED |_|_|_ProjectionExec: expr=[greptime_value@0 as greptime_value, __tsid@3 as __tsid, ts@4 as ts] REDACTED |_|_|_MergeScanExec: REDACTED -|_|_|_CooperativeExec REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_PromInstantManipulateExec: range=[0..5000], lookback=[300000], interval=[5000], time index=[ts] REDACTED @@ -466,7 +466,7 @@ TQL ANALYZE (0, 5, '5s') ((tsid_binary_join_left > bool tsid_binary_join_right) |_|_|_RepartitionExec: partitioning=Hash([REDACTED |_|_|_ProjectionExec: expr=[greptime_value@0 as greptime_value, __tsid@3 as __tsid, ts@4 as ts] REDACTED |_|_|_MergeScanExec: REDACTED -|_|_|_CooperativeExec REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_PromInstantManipulateExec: range=[0..5000], lookback=[300000], interval=[5000], time index=[ts] REDACTED @@ -516,12 +516,12 @@ TQL ANALYZE (0, 5, '5s') (tsid_binary_join_left or tsid_binary_join_right) / tsi |_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@1, __tsid@3), (ts@0, ts@4)], projection=[greptime_value@2, greptime_value@3, host@4, job@5, __tsid@6, ts@7], NullsEqual: true REDACTED |_|_|_ProjectionExec: expr=[ts@0 as ts, __tsid@1 as __tsid, greptime_value@2 as greptime_value] REDACTED |_|_|_UnionDistinctOnExec: on col=[5, 6], ts_col=0 REDACTED -|_|_|_CoalescePartitionsExec REDACTED -|_|_|_MergeScanExec: REDACTED -|_|_|_CoalescePartitionsExec REDACTED +|_|_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_ProjectionExec: expr=[ts@2 as ts, __tsid@3 as __tsid, greptime_value@4 as greptime_value, host@5 as host, job@6 as job, CASE WHEN __common_expr_1@0 IS NOT NULL THEN __common_expr_1@0 ELSE_END as __promql_or_match_0, CASE WHEN __common_expr_2@1 IS NOT NULL THEN __common_expr_2@1 ELSE_END as __promql_or_match_1] REDACTED |_|_|_ProjectionExec: expr=[CAST(host@1 AS Utf8) as __common_expr_1, CAST(job@2 AS Utf8) as __common_expr_2, ts@4 as ts, __tsid@3 as __tsid, greptime_value@0 as greptime_value, host@1 as host, job@2 as job] REDACTED @@ -575,7 +575,7 @@ TQL ANALYZE (0, 5, '5s') (tsid_binary_join_left / ignoring(host) group_left tsid |_|_|_RepartitionExec: partitioning=Hash([REDACTED |_|_|_ProjectionExec: expr=[greptime_value@0 as greptime_value, job@2 as job, __tsid@3 as __tsid, ts@4 as ts] REDACTED |_|_|_MergeScanExec: REDACTED -|_|_|_CooperativeExec REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_PromInstantManipulateExec: range=[0..5000], lookback=[300000], interval=[5000], time index=[ts] REDACTED diff --git a/tests/cases/standalone/common/promql/tsid_histogram_quantile_regression.result b/tests/cases/standalone/common/promql/tsid_histogram_quantile_regression.result index fc883d96e7..ba2e97bcc2 100644 --- a/tests/cases/standalone/common/promql/tsid_histogram_quantile_regression.result +++ b/tests/cases/standalone/common/promql/tsid_histogram_quantile_regression.result @@ -109,8 +109,8 @@ TQL ANALYZE (0, 10, '5s') histogram_quantile(0.5, tsid_no_aggr_histogram_bucket) | stage | node | plan_| +-+-+-+ | 0_| 0_|_HistogramFoldExec: le=@2, field=@0, quantile=0.5 REDACTED -|_|_|_SortExec: expr=[job@1 ASC NULLS LAST, ts@3 ASC NULLS LAST, CAST(le@2 AS Float64) ASC NULLS LAST], preserve_partitioning=[true] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_SortExec: expr=[job@1 ASC NULLS LAST, ts@3 ASC NULLS LAST, CAST(le@2 AS Float64) ASC NULLS LAST], preserve_partitioning=[false] REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_ProjectionExec: expr=[val@0 as val, job@1 as job, le@2 as le, ts@4 as ts] REDACTED diff --git a/tests/cases/standalone/common/range/nest.result b/tests/cases/standalone/common/range/nest.result index 74742d5d86..184b27545d 100644 --- a/tests/cases/standalone/common/range/nest.result +++ b/tests/cases/standalone/common/range/nest.result @@ -138,7 +138,7 @@ EXPLAIN SELECT ts, host, min(val) RANGE '5s' FROM host ALIGN '5s'; |_|_TableScan: host_| |_| ]]_| | physical_plan | RangeSelectExec: range_expr=[min(host.val) RANGE 5s], align=5000ms, align_to=0ms, align_by=[host@1], time_index=ts | -|_|_CoalescePartitionsExec_| +|_|_CooperativeExec_| |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -154,7 +154,7 @@ EXPLAIN ANALYZE SELECT ts, host, min(val) RANGE '5s' FROM host ALIGN '5s'; | stage | node | plan_| +-+-+-+ | 0_| 0_|_RangeSelectExec: range_expr=[min(host.val) RANGE 5s], align=5000ms, align_to=0ms, align_by=[host@1], time_index=ts REDACTED -|_|_|_CoalescePartitionsExec REDACTED +|_|_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_CooperativeExec REDACTED diff --git a/tests/cases/standalone/common/tql-explain-analyze/explain.result b/tests/cases/standalone/common/tql-explain-analyze/explain.result index e3d02226c3..2d6eab49da 100644 --- a/tests/cases/standalone/common/tql-explain-analyze/explain.result +++ b/tests/cases/standalone/common/tql-explain-analyze/explain.result @@ -100,6 +100,7 @@ TQL EXPLAIN VERBOSE (0, 10, '5s') test; |_|_Filter: test.j >= TimestampMillisecond(-299999, None) AND test.j <= TimestampMillisecond(10000, None)_| |_|_TableScan: test, partial_filters=[test.j >= TimestampMillisecond(-299999, None), test.j <= TimestampMillisecond(10000, None)] | |_| ]]_| +| logical_plan after JsonSchemaConcretizeRule_| SAME TEXT AS ABOVE_| | logical_plan after FixStateUdafOrderingAnalyzer_| SAME TEXT AS ABOVE_| | analyzed_logical_plan_| SAME TEXT AS ABOVE_| | logical_plan after rewrite_set_comparison_| SAME TEXT AS ABOVE_| @@ -250,6 +251,7 @@ TQL EXPLAIN VERBOSE (0, 10, '5s') test AS series; |_|_Filter: test.j >= TimestampMillisecond(-299999, None) AND test.j <= TimestampMillisecond(10000, None)_| |_|_TableScan: test, partial_filters=[test.j >= TimestampMillisecond(-299999, None), test.j <= TimestampMillisecond(10000, None)] | |_| ]]_| +| logical_plan after JsonSchemaConcretizeRule_| SAME TEXT AS ABOVE_| | logical_plan after FixStateUdafOrderingAnalyzer_| SAME TEXT AS ABOVE_| | analyzed_logical_plan_| SAME TEXT AS ABOVE_| | logical_plan after rewrite_set_comparison_| SAME TEXT AS ABOVE_| @@ -432,6 +434,7 @@ TQL EXPLAIN VERBOSE (0, 10, '5s') test_nano; |_|_Filter: test_nano.j >= TimestampNanosecond(-299999999999, None) AND test_nano.j < TimestampNanosecond(10001000000, None)_| |_|_TableScan: test_nano, partial_filters=[test_nano.j >= TimestampNanosecond(-299999999999, None), test_nano.j < TimestampNanosecond(10001000000, None)] | |_| ]]_| +| logical_plan after JsonSchemaConcretizeRule_| SAME TEXT AS ABOVE_| | logical_plan after FixStateUdafOrderingAnalyzer_| SAME TEXT AS ABOVE_| | analyzed_logical_plan_| SAME TEXT AS ABOVE_| | logical_plan after rewrite_set_comparison_| SAME TEXT AS ABOVE_| diff --git a/tests/cases/standalone/common/tql/partition.result b/tests/cases/standalone/common/tql/partition.result index 45927f8abb..f1072edbb3 100644 --- a/tests/cases/standalone/common/tql/partition.result +++ b/tests/cases/standalone/common/tql/partition.result @@ -68,7 +68,10 @@ tql analyze (0, 10, '1s') 100 - (avg by (k) (irate(t[1m])) * 100); |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_SortPreservingMergeExec: [k@0 ASC NULLS LAST, j@1 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[k@0 ASC NULLS LAST, j@1 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=SinglePartitioned, gby=[k@0 as k, j@1 as j], aggr=[avg(prom_irate(j_range,i))] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[k@0 as k, j@1 as j], aggr=[avg(prom_irate(j_range,i))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[k@0 as k, j@1 as j], aggr=[avg(prom_irate(j_range,i))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[k@0 as k, j@1 as j], aggr=[__avg_state(prom_irate(j_range,i))] REDACTED diff --git a/tests/cases/standalone/common/tql/tql-cte.result b/tests/cases/standalone/common/tql/tql-cte.result index 0547754b48..8efe983763 100644 --- a/tests/cases/standalone/common/tql/tql-cte.result +++ b/tests/cases/standalone/common/tql/tql-cte.result @@ -487,8 +487,8 @@ ORDER BY ts; | | ]] | | physical_plan | ProjectionExec: expr=[ts@0 as ts, val@1 as val, lag(tql_base.val,Int64(1)) ORDER BY [tql_base.ts ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW@2 as prev_value] | | | BoundedWindowAggExec: wdw=[lag(tql_base.val,Int64(1)) ORDER BY [tql_base.ts ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW: Field { "lag(tql_base.val,Int64(1)) ORDER BY [tql_base.ts ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW": nullable Float64 }, frame: RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW], mode=[Sorted] | -| | SortPreservingMergeExec: [ts@0 ASC NULLS LAST] | -| | SortExec: expr=[ts@0 ASC NULLS LAST], preserve_REDACTED +| | SortExec: expr=[ts@0 ASC NULLS LAST], preserve_REDACTED +| | CooperativeExec | | | MergeScanExec: REDACTED | | | +---------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ @@ -765,11 +765,15 @@ LIMIT 5; | | ProjectionExec: expr=[ts@0 as ts, cpu@1 as avg_value, host@2 as host] | | | HashJoinExec: mode=Partitioned, join_type=Inner, on=[(date_trunc(Utf8("second"),t.ts)@2, date_trunc(Utf8("second"),l.ts)@2)], projection=[ts@0, cpu@1, host@4] | | | RepartitionExec: REDACTED -| | ProjectionExec: expr=[ts@0 as ts, cpu@2 as cpu, date_trunc(second, ts@0) as date_trunc(Utf8("second"),t.ts)] | -| | MergeScanExec: REDACTED +| | ProjectionExec: expr=[ts@0 as ts, cpu@1 as cpu, date_trunc(second, ts@0) as date_trunc(Utf8("second"),t.ts)] | +| | RepartitionExec: REDACTED +| | ProjectionExec: expr=[ts@0 as ts, cpu@2 as cpu] | +| | MergeScanExec: REDACTED | | RepartitionExec: REDACTED | | ProjectionExec: expr=[ts@0 as ts, host@1 as host, date_trunc(second, ts@0) as date_trunc(Utf8("second"),l.ts)] | -| | MergeScanExec: REDACTED +| | RepartitionExec: REDACTED +| | ProjectionExec: expr=[ts@0 as ts, host@1 as host] | +| | MergeScanExec: REDACTED | | | +---------------+----------------------------------------------------------------------------------------------------------------------------------------------------------------------+ @@ -843,11 +847,15 @@ LIMIT 5; | | ProjectionExec: expr=[ts@1 as ts, cpu@0 as avg_value, host@2 as host] | | | HashJoinExec: mode=Partitioned, join_type=Inner, on=[(date_trunc(Utf8("second"),t.ts)@2, date_trunc(Utf8("second"),l.ts)@2)], projection=[cpu@0, ts@1, host@4] | | | RepartitionExec: REDACTED -| | ProjectionExec: expr=[cpu@0 as cpu, ts@2 as ts, date_trunc(second, ts@2) as date_trunc(Utf8("second"),t.ts)] | -| | MergeScanExec: REDACTED +| | ProjectionExec: expr=[cpu@0 as cpu, ts@1 as ts, date_trunc(second, ts@1) as date_trunc(Utf8("second"),t.ts)] | +| | RepartitionExec: REDACTED +| | ProjectionExec: expr=[cpu@0 as cpu, ts@2 as ts] | +| | MergeScanExec: REDACTED | | RepartitionExec: REDACTED | | ProjectionExec: expr=[ts@0 as ts, host@1 as host, date_trunc(second, ts@0) as date_trunc(Utf8("second"),l.ts)] | -| | MergeScanExec: REDACTED +| | RepartitionExec: REDACTED +| | ProjectionExec: expr=[ts@0 as ts, host@1 as host] | +| | MergeScanExec: REDACTED | | | +---------------+----------------------------------------------------------------------------------------------------------------------------------------------------------------------+ diff --git a/tests/cases/standalone/common/types/json/json2.result b/tests/cases/standalone/common/types/json/json2.result index 2e9c9c8ee8..a617849600 100644 --- a/tests/cases/standalone/common/types/json/json2.result +++ b/tests/cases/standalone/common/types/json/json2.result @@ -126,14 +126,39 @@ select j.a, j.a.x from json2_table order by ts; | {"b":-2} | | | {"b":3} | | | {"b":-4} | | -| | | +| {} | | | | | | {"b":"s7"} | | | {"b":8} | | -| {"b":null,"x":true} | true | -| {"b":10,"x":null} | | +| {"x":true} | true | +| {"b":10} | | +-----------------------------------+-------------------------------------+ +select j, j.a from json2_table order by ts; + ++--------------------------------------------------+-----------------------------------+ +| j | json_get(json2_table.j,Utf8("a")) | ++--------------------------------------------------+-----------------------------------+ +| {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1}}]} | {"b":1} | +| {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2}}]} | {"b":-2} | +| {"a":{"b":3},"c":"s3"} | {"b":3} | +| {"a":{"b":-4},"d":[{"e":{"g":-0.4}}]} | {"b":-4} | +| {"a":{},"c":"s5"} | {} | +| {"c":"s6"} | | +| {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | {"b":"s7"} | +| {"a":{"b":8},"c":"s8"} | {"b":8} | +| {"a":{"x":true},"c":"s9","d":[{"e":{"g":-0.9}}]} | {"x":true} | +| {"a":{"b":10},"y":false} | {"b":10} | ++--------------------------------------------------+-----------------------------------+ + +select j from json2_table where j.a.b = 1; + ++----------------------------------------------+ +| j | ++----------------------------------------------+ +| {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1}}]} | ++----------------------------------------------+ + select j.c, j.y from json2_table order by ts; +-----------------------------------+-----------------------------------+ @@ -153,37 +178,37 @@ select j.c, j.y from json2_table order by ts; select j from json2_table order by ts; -+--------------------------------------------------------------------+ -| j | -+--------------------------------------------------------------------+ -| {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1,"g":null}}]} | -| {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2,"g":null}}]} | -| {"a":{"b":3},"c":"s3","d":null} | -| {"a":{"b":-4},"c":null,"d":[{"e":{"f":null,"g":-0.4}}]} | -| {"a":null,"c":"s5","d":null} | -| {"a":null,"c":"s6","d":null} | -| {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | -| {"a":{"b":8},"c":"s8","d":null} | -| {"a":{"b":null,"x":true},"c":"s9","d":[{"e":{"g":-0.9}}],"y":null} | -| {"a":{"b":10,"x":null},"c":null,"d":null,"y":false} | -+--------------------------------------------------------------------+ ++--------------------------------------------------+ +| j | ++--------------------------------------------------+ +| {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1}}]} | +| {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2}}]} | +| {"a":{"b":3},"c":"s3"} | +| {"a":{"b":-4},"d":[{"e":{"g":-0.4}}]} | +| {"a":{},"c":"s5"} | +| {"c":"s6"} | +| {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | +| {"a":{"b":8},"c":"s8"} | +| {"a":{"x":true},"c":"s9","d":[{"e":{"g":-0.9}}]} | +| {"a":{"b":10},"y":false} | ++--------------------------------------------------+ select * from json2_table order by ts; -+-------------------------+--------------------------------------------------------------------+ -| ts | j | -+-------------------------+--------------------------------------------------------------------+ -| 1970-01-01T00:00:00.001 | {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1,"g":null}}]} | -| 1970-01-01T00:00:00.002 | {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2,"g":null}}]} | -| 1970-01-01T00:00:00.003 | {"a":{"b":3},"c":"s3","d":null} | -| 1970-01-01T00:00:00.004 | {"a":{"b":-4},"c":null,"d":[{"e":{"f":null,"g":-0.4}}]} | -| 1970-01-01T00:00:00.005 | {"a":null,"c":"s5","d":null} | -| 1970-01-01T00:00:00.006 | {"a":null,"c":"s6","d":null} | -| 1970-01-01T00:00:00.007 | {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | -| 1970-01-01T00:00:00.008 | {"a":{"b":8},"c":"s8","d":null} | -| 1970-01-01T00:00:00.009 | {"a":{"b":null,"x":true},"c":"s9","d":[{"e":{"g":-0.9}}],"y":null} | -| 1970-01-01T00:00:00.010 | {"a":{"b":10,"x":null},"c":null,"d":null,"y":false} | -+-------------------------+--------------------------------------------------------------------+ ++-------------------------+--------------------------------------------------+ +| ts | j | ++-------------------------+--------------------------------------------------+ +| 1970-01-01T00:00:00.001 | {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1}}]} | +| 1970-01-01T00:00:00.002 | {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2}}]} | +| 1970-01-01T00:00:00.003 | {"a":{"b":3},"c":"s3"} | +| 1970-01-01T00:00:00.004 | {"a":{"b":-4},"d":[{"e":{"g":-0.4}}]} | +| 1970-01-01T00:00:00.005 | {"a":{},"c":"s5"} | +| 1970-01-01T00:00:00.006 | {"c":"s6"} | +| 1970-01-01T00:00:00.007 | {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | +| 1970-01-01T00:00:00.008 | {"a":{"b":8},"c":"s8"} | +| 1970-01-01T00:00:00.009 | {"a":{"x":true},"c":"s9","d":[{"e":{"g":-0.9}}]} | +| 1970-01-01T00:00:00.010 | {"a":{"b":10},"y":false} | ++-------------------------+--------------------------------------------------+ select count(*) from (select j from json2_table group by j); @@ -203,88 +228,110 @@ select count(*) from (select distinct j from json2_table); select ts, j from (select ts, j from json2_table) order by ts; -+-------------------------+--------------------------------------------------------------------+ -| ts | j | -+-------------------------+--------------------------------------------------------------------+ -| 1970-01-01T00:00:00.001 | {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1,"g":null}}]} | -| 1970-01-01T00:00:00.002 | {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2,"g":null}}]} | -| 1970-01-01T00:00:00.003 | {"a":{"b":3},"c":"s3","d":null} | -| 1970-01-01T00:00:00.004 | {"a":{"b":-4},"c":null,"d":[{"e":{"f":null,"g":-0.4}}]} | -| 1970-01-01T00:00:00.005 | {"a":null,"c":"s5","d":null} | -| 1970-01-01T00:00:00.006 | {"a":null,"c":"s6","d":null} | -| 1970-01-01T00:00:00.007 | {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | -| 1970-01-01T00:00:00.008 | {"a":{"b":8},"c":"s8","d":null} | -| 1970-01-01T00:00:00.009 | {"a":{"b":null,"x":true},"c":"s9","d":[{"e":{"g":-0.9}}],"y":null} | -| 1970-01-01T00:00:00.010 | {"a":{"b":10,"x":null},"c":null,"d":null,"y":false} | -+-------------------------+--------------------------------------------------------------------+ ++-------------------------+--------------------------------------------------+ +| ts | j | ++-------------------------+--------------------------------------------------+ +| 1970-01-01T00:00:00.001 | {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1}}]} | +| 1970-01-01T00:00:00.002 | {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2}}]} | +| 1970-01-01T00:00:00.003 | {"a":{"b":3},"c":"s3"} | +| 1970-01-01T00:00:00.004 | {"a":{"b":-4},"d":[{"e":{"g":-0.4}}]} | +| 1970-01-01T00:00:00.005 | {"a":{},"c":"s5"} | +| 1970-01-01T00:00:00.006 | {"c":"s6"} | +| 1970-01-01T00:00:00.007 | {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | +| 1970-01-01T00:00:00.008 | {"a":{"b":8},"c":"s8"} | +| 1970-01-01T00:00:00.009 | {"a":{"x":true},"c":"s9","d":[{"e":{"g":-0.9}}]} | +| 1970-01-01T00:00:00.010 | {"a":{"b":10},"y":false} | ++-------------------------+--------------------------------------------------+ + +select + ts, + j, + row_number() over (order by ts) as row_num +from json2_table +order by ts; + ++-------------------------+--------------------------------------------------+---------+ +| ts | j | row_num | ++-------------------------+--------------------------------------------------+---------+ +| 1970-01-01T00:00:00.001 | {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1}}]} | 1 | +| 1970-01-01T00:00:00.002 | {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2}}]} | 2 | +| 1970-01-01T00:00:00.003 | {"a":{"b":3},"c":"s3"} | 3 | +| 1970-01-01T00:00:00.004 | {"a":{"b":-4},"d":[{"e":{"g":-0.4}}]} | 4 | +| 1970-01-01T00:00:00.005 | {"a":{},"c":"s5"} | 5 | +| 1970-01-01T00:00:00.006 | {"c":"s6"} | 6 | +| 1970-01-01T00:00:00.007 | {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | 7 | +| 1970-01-01T00:00:00.008 | {"a":{"b":8},"c":"s8"} | 8 | +| 1970-01-01T00:00:00.009 | {"a":{"x":true},"c":"s9","d":[{"e":{"g":-0.9}}]} | 9 | +| 1970-01-01T00:00:00.010 | {"a":{"b":10},"y":false} | 10 | ++-------------------------+--------------------------------------------------+---------+ select json_get(j, '') from json2_table order by ts; -+--------------------------------------------------------------------+ -| json_get(json2_table.j,Utf8("")) | -+--------------------------------------------------------------------+ -| {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1,"g":null}}]} | -| {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2,"g":null}}]} | -| {"a":{"b":3},"c":"s3","d":null} | -| {"a":{"b":-4},"c":null,"d":[{"e":{"f":null,"g":-0.4}}]} | -| {"a":null,"c":"s5","d":null} | -| {"a":null,"c":"s6","d":null} | -| {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | -| {"a":{"b":8},"c":"s8","d":null} | -| {"a":{"b":null,"x":true},"c":"s9","d":[{"e":{"g":-0.9}}],"y":null} | -| {"a":{"b":10,"x":null},"c":null,"d":null,"y":false} | -+--------------------------------------------------------------------+ ++--------------------------------------------------+ +| json_get(json2_table.j,Utf8("")) | ++--------------------------------------------------+ +| {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1}}]} | +| {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2}}]} | +| {"a":{"b":3},"c":"s3"} | +| {"a":{"b":-4},"d":[{"e":{"g":-0.4}}]} | +| {"a":{},"c":"s5"} | +| {"c":"s6"} | +| {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | +| {"a":{"b":8},"c":"s8"} | +| {"a":{"x":true},"c":"s9","d":[{"e":{"g":-0.9}}]} | +| {"a":{"b":10},"y":false} | ++--------------------------------------------------+ select json_get(j, '$') from json2_table order by ts; -+--------------------------------------------------------------------+ -| json_get(json2_table.j,Utf8("$")) | -+--------------------------------------------------------------------+ -| {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1,"g":null}}]} | -| {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2,"g":null}}]} | -| {"a":{"b":3},"c":"s3","d":null} | -| {"a":{"b":-4},"c":null,"d":[{"e":{"f":null,"g":-0.4}}]} | -| {"a":null,"c":"s5","d":null} | -| {"a":null,"c":"s6","d":null} | -| {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | -| {"a":{"b":8},"c":"s8","d":null} | -| {"a":{"b":null,"x":true},"c":"s9","d":[{"e":{"g":-0.9}}],"y":null} | -| {"a":{"b":10,"x":null},"c":null,"d":null,"y":false} | -+--------------------------------------------------------------------+ ++--------------------------------------------------+ +| json_get(json2_table.j,Utf8("$")) | ++--------------------------------------------------+ +| {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1}}]} | +| {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2}}]} | +| {"a":{"b":3},"c":"s3"} | +| {"a":{"b":-4},"d":[{"e":{"g":-0.4}}]} | +| {"a":{},"c":"s5"} | +| {"c":"s6"} | +| {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | +| {"a":{"b":8},"c":"s8"} | +| {"a":{"x":true},"c":"s9","d":[{"e":{"g":-0.9}}]} | +| {"a":{"b":10},"y":false} | ++--------------------------------------------------+ select json_get(j, '.') from json2_table order by ts; -+--------------------------------------------------------------------+ -| json_get(json2_table.j,Utf8(".")) | -+--------------------------------------------------------------------+ -| {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1,"g":null}}]} | -| {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2,"g":null}}]} | -| {"a":{"b":3},"c":"s3","d":null} | -| {"a":{"b":-4},"c":null,"d":[{"e":{"f":null,"g":-0.4}}]} | -| {"a":null,"c":"s5","d":null} | -| {"a":null,"c":"s6","d":null} | -| {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | -| {"a":{"b":8},"c":"s8","d":null} | -| {"a":{"b":null,"x":true},"c":"s9","d":[{"e":{"g":-0.9}}],"y":null} | -| {"a":{"b":10,"x":null},"c":null,"d":null,"y":false} | -+--------------------------------------------------------------------+ ++--------------------------------------------------+ +| json_get(json2_table.j,Utf8(".")) | ++--------------------------------------------------+ +| {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1}}]} | +| {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2}}]} | +| {"a":{"b":3},"c":"s3"} | +| {"a":{"b":-4},"d":[{"e":{"g":-0.4}}]} | +| {"a":{},"c":"s5"} | +| {"c":"s6"} | +| {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | +| {"a":{"b":8},"c":"s8"} | +| {"a":{"x":true},"c":"s9","d":[{"e":{"g":-0.9}}]} | +| {"a":{"b":10},"y":false} | ++--------------------------------------------------+ select json_get(j, '$.') from json2_table order by ts; -+--------------------------------------------------------------------+ -| json_get(json2_table.j,Utf8("$.")) | -+--------------------------------------------------------------------+ -| {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1,"g":null}}]} | -| {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2,"g":null}}]} | -| {"a":{"b":3},"c":"s3","d":null} | -| {"a":{"b":-4},"c":null,"d":[{"e":{"f":null,"g":-0.4}}]} | -| {"a":null,"c":"s5","d":null} | -| {"a":null,"c":"s6","d":null} | -| {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | -| {"a":{"b":8},"c":"s8","d":null} | -| {"a":{"b":null,"x":true},"c":"s9","d":[{"e":{"g":-0.9}}],"y":null} | -| {"a":{"b":10,"x":null},"c":null,"d":null,"y":false} | -+--------------------------------------------------------------------+ ++--------------------------------------------------+ +| json_get(json2_table.j,Utf8("$.")) | ++--------------------------------------------------+ +| {"a":{"b":1},"c":"s1","d":[{"e":{"f":0.1}}]} | +| {"a":{"b":-2},"c":"s2","d":[{"e":{"f":0.2}}]} | +| {"a":{"b":3},"c":"s3"} | +| {"a":{"b":-4},"d":[{"e":{"g":-0.4}}]} | +| {"a":{},"c":"s5"} | +| {"c":"s6"} | +| {"a":{"b":"s7"},"c":[1],"d":[{"e":{"g":-0.7}}]} | +| {"a":{"b":8},"c":"s8"} | +| {"a":{"x":true},"c":"s9","d":[{"e":{"g":-0.9}}]} | +| {"a":{"b":10},"y":false} | ++--------------------------------------------------+ select j.a.b + 1 from json2_table order by ts; @@ -303,28 +350,15 @@ select j.a.b + 1 from json2_table order by ts; | 11 | +------------------------------------------------------------+ -select abs(j.a.b) from json2_table order by ts; - -Error: 3000(PlanQuery), Failed to plan SQL: Error during planning: Function 'abs' expects NativeType::Numeric but received NativeType::String No function matches the given name and argument types 'abs(Utf8View)'. You might need to add explicit type casts. - Candidate functions: - abs(Numeric(1)) - --- "j.c" is of type "String", "abs" is expected to be all "null"s. -select abs(j.c) from json2_table order by ts; - -Error: 3000(PlanQuery), Failed to plan SQL: Error during planning: Function 'abs' expects NativeType::Numeric but received NativeType::String No function matches the given name and argument types 'abs(Utf8View)'. You might need to add explicit type casts. - Candidate functions: - abs(Numeric(1)) - select j.d from json2_table order by ts; +-----------------------------------+ | json_get(json2_table.j,Utf8("d")) | +-----------------------------------+ -| [{"e":{"f":0.1,"g":null}}] | -| [{"e":{"f":0.2,"g":null}}] | +| [{"e":{"f":0.1}}] | +| [{"e":{"f":0.2}}] | | | -| [{"e":{"f":null,"g":-0.4}}] | +| [{"e":{"g":-0.4}}] | | | | | | [{"e":{"g":-0.7}}] | @@ -392,3 +426,116 @@ drop table json2_variant_null; Affected Rows: 0 +create table json2_finite_paths ( + ts timestamp time index, + j json2( + max_auto_expanded_paths = 1, + hint string + ) +) +with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +Affected Rows: 0 + +show create table json2_finite_paths; + ++--------------------+---------------------------------------------------+ +| Table | Create Table | ++--------------------+---------------------------------------------------+ +| json2_finite_paths | CREATE TABLE IF NOT EXISTS "json2_finite_paths" ( | +| | "ts" TIMESTAMP(3) NOT NULL, | +| | "j" JSON2( | +| | max_auto_expanded_paths = 1, | +| | "hint" STRING NULL | +| | ) NULL, | +| | TIME INDEX ("ts") | +| | ) | +| | | +| | ENGINE=mito | +| | WITH( | +| | append_mode = 'true', | +| | sst_format = 'flat' | +| | ) | ++--------------------+---------------------------------------------------+ + +insert into json2_finite_paths values + (1, '{"hint":"h1","alpha":1,"conflict":1}'), + (2, '{"hint":"h2","alpha":2,"conflict":"text"}'); + +Affected Rows: 2 + +admin flush_table('json2_finite_paths'); + ++-----------------------------------------+ +| ADMIN flush_table('json2_finite_paths') | ++-----------------------------------------+ +| 0 | ++-----------------------------------------+ + +insert into json2_finite_paths values + (3, '{"hint":"h3","beta":3,"conflict":true}'), + (4, '{"hint":"h4","beta":4,"conflict":"other"}'); + +Affected Rows: 2 + +admin flush_table('json2_finite_paths'); + ++-----------------------------------------+ +| ADMIN flush_table('json2_finite_paths') | ++-----------------------------------------+ +| 0 | ++-----------------------------------------+ + +select + ts, + j, + j.hint, + j.alpha::bigint as alpha, + j.beta::bigint as beta, + j.conflict +from json2_finite_paths +order by ts; + ++-------------------------+-------------------------------------------+---------------------------------------------+-------+------+-------------------------------------------------+ +| ts | j | json_get(json2_finite_paths.j,Utf8("hint")) | alpha | beta | json_get(json2_finite_paths.j,Utf8("conflict")) | ++-------------------------+-------------------------------------------+---------------------------------------------+-------+------+-------------------------------------------------+ +| 1970-01-01T00:00:00.001 | {"alpha":1,"conflict":1,"hint":"h1"} | h1 | 1 | | 1 | +| 1970-01-01T00:00:00.002 | {"alpha":2,"conflict":"text","hint":"h2"} | h2 | 2 | | text | +| 1970-01-01T00:00:00.003 | {"beta":3,"conflict":true,"hint":"h3"} | h3 | | 3 | true | +| 1970-01-01T00:00:00.004 | {"beta":4,"conflict":"other","hint":"h4"} | h4 | | 4 | other | ++-------------------------+-------------------------------------------+---------------------------------------------+-------+------+-------------------------------------------------+ + +admin compact_table('json2_finite_paths'); + ++-------------------------------------------+ +| ADMIN compact_table('json2_finite_paths') | ++-------------------------------------------+ +| 0 | ++-------------------------------------------+ + +select + ts, + j, + j.hint, + j.alpha::bigint as alpha, + j.beta::bigint as beta, + j.conflict +from json2_finite_paths +order by ts; + ++-------------------------+-------------------------------------------+---------------------------------------------+-------+------+-------------------------------------------------+ +| ts | j | json_get(json2_finite_paths.j,Utf8("hint")) | alpha | beta | json_get(json2_finite_paths.j,Utf8("conflict")) | ++-------------------------+-------------------------------------------+---------------------------------------------+-------+------+-------------------------------------------------+ +| 1970-01-01T00:00:00.001 | {"alpha":1,"conflict":1,"hint":"h1"} | h1 | 1 | | 1 | +| 1970-01-01T00:00:00.002 | {"alpha":2,"conflict":"text","hint":"h2"} | h2 | 2 | | text | +| 1970-01-01T00:00:00.003 | {"beta":3,"conflict":true,"hint":"h3"} | h3 | | 3 | true | +| 1970-01-01T00:00:00.004 | {"beta":4,"conflict":"other","hint":"h4"} | h4 | | 4 | other | ++-------------------------+-------------------------------------------+---------------------------------------------+-------+------+-------------------------------------------------+ + +drop table json2_finite_paths; + +Affected Rows: 0 + diff --git a/tests/cases/standalone/common/types/json/json2.sql b/tests/cases/standalone/common/types/json/json2.sql index f0bb5286b4..17248741c3 100644 --- a/tests/cases/standalone/common/types/json/json2.sql +++ b/tests/cases/standalone/common/types/json/json2.sql @@ -44,6 +44,10 @@ select j.a.b from json2_table order by ts; select j.a, j.a.x from json2_table order by ts; +select j, j.a from json2_table order by ts; + +select j from json2_table where j.a.b = 1; + select j.c, j.y from json2_table order by ts; select j from json2_table order by ts; @@ -56,6 +60,13 @@ select count(*) from (select distinct j from json2_table); select ts, j from (select ts, j from json2_table) order by ts; +select + ts, + j, + row_number() over (order by ts) as row_num +from json2_table +order by ts; + select json_get(j, '') from json2_table order by ts; select json_get(j, '$') from json2_table order by ts; @@ -66,11 +77,6 @@ select json_get(j, '$.') from json2_table order by ts; select j.a.b + 1 from json2_table order by ts; -select abs(j.a.b) from json2_table order by ts; - --- "j.c" is of type "String", "abs" is expected to be all "null"s. -select abs(j.c) from json2_table order by ts; - select j.d from json2_table order by ts; drop table json2_table; @@ -101,3 +107,53 @@ from json2_variant_null order by ts; drop table json2_variant_null; + +create table json2_finite_paths ( + ts timestamp time index, + j json2( + max_auto_expanded_paths = 1, + hint string + ) +) +with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +show create table json2_finite_paths; + +insert into json2_finite_paths values + (1, '{"hint":"h1","alpha":1,"conflict":1}'), + (2, '{"hint":"h2","alpha":2,"conflict":"text"}'); + +admin flush_table('json2_finite_paths'); + +insert into json2_finite_paths values + (3, '{"hint":"h3","beta":3,"conflict":true}'), + (4, '{"hint":"h4","beta":4,"conflict":"other"}'); + +admin flush_table('json2_finite_paths'); + +select + ts, + j, + j.hint, + j.alpha::bigint as alpha, + j.beta::bigint as beta, + j.conflict +from json2_finite_paths +order by ts; + +admin compact_table('json2_finite_paths'); + +select + ts, + j, + j.hint, + j.alpha::bigint as alpha, + j.beta::bigint as beta, + j.conflict +from json2_finite_paths +order by ts; + +drop table json2_finite_paths; diff --git a/tests/cases/standalone/common/types/json/json2_empty.result b/tests/cases/standalone/common/types/json/json2_empty.result new file mode 100644 index 0000000000..c9fec51396 --- /dev/null +++ b/tests/cases/standalone/common/types/json/json2_empty.result @@ -0,0 +1,445 @@ +-- Empty-object JSON2 values are stored in the JSONB remainder column and must +-- survive memtable flush and compaction unchanged, both as whole values and +-- through path access. +-- Whole-value empty objects: insert, flush, then mix with expanded documents +-- before compacting. +create table json_empty ( + ts timestamp time index, + j json2 +) with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +Affected Rows: 0 + +insert into json_empty values (1, '{}'); + +Affected Rows: 1 + +select ts, j from json_empty order by ts; + ++-------------------------+----+ +| ts | j | ++-------------------------+----+ +| 1970-01-01T00:00:00.001 | {} | ++-------------------------+----+ + +select ts, j.a from json_empty order by ts; + ++-------------------------+----------------------------------+ +| ts | json_get(json_empty.j,Utf8("a")) | ++-------------------------+----------------------------------+ +| 1970-01-01T00:00:00.001 | | ++-------------------------+----------------------------------+ + +admin flush_table('json_empty'); + ++---------------------------------+ +| ADMIN flush_table('json_empty') | ++---------------------------------+ +| 0 | ++---------------------------------+ + +select ts, j from json_empty order by ts; + ++-------------------------+----+ +| ts | j | ++-------------------------+----+ +| 1970-01-01T00:00:00.001 | {} | ++-------------------------+----+ + +insert into json_empty values (2, '{}'), (3, '{"a": 1}'); + +Affected Rows: 2 + +select ts, j, j.a from json_empty order by ts; + ++-------------------------+---------+----------------------------------+ +| ts | j | json_get(json_empty.j,Utf8("a")) | ++-------------------------+---------+----------------------------------+ +| 1970-01-01T00:00:00.001 | {} | | +| 1970-01-01T00:00:00.002 | {} | | +| 1970-01-01T00:00:00.003 | {"a":1} | 1 | ++-------------------------+---------+----------------------------------+ + +admin flush_table('json_empty'); + ++---------------------------------+ +| ADMIN flush_table('json_empty') | ++---------------------------------+ +| 0 | ++---------------------------------+ + +admin compact_table('json_empty'); + ++-----------------------------------+ +| ADMIN compact_table('json_empty') | ++-----------------------------------+ +| 0 | ++-----------------------------------+ + +select ts, j, j.a from json_empty order by ts; + ++-------------------------+---------+----------------------------------+ +| ts | j | json_get(json_empty.j,Utf8("a")) | ++-------------------------+---------+----------------------------------+ +| 1970-01-01T00:00:00.001 | {} | | +| 1970-01-01T00:00:00.002 | {} | | +| 1970-01-01T00:00:00.003 | {"a":1} | 1 | ++-------------------------+---------+----------------------------------+ + +select count(*) from json_empty; + ++----------+ +| count(*) | ++----------+ +| 3 | ++----------+ + +drop table json_empty; + +Affected Rows: 0 + +-- Nested empty objects and arrays, which also produce only remainder content, +-- must round trip through flush and compaction as whole values. +create table json_empty_nested ( + ts timestamp time index, + j json2 +) with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +Affected Rows: 0 + +insert into json_empty_nested values + (1, '{"a": {}}'), + (2, '{"a": {"b": {}}}'); + +Affected Rows: 2 + +select ts, j from json_empty_nested order by ts; + ++-------------------------+----------------+ +| ts | j | ++-------------------------+----------------+ +| 1970-01-01T00:00:00.001 | {"a":{}} | +| 1970-01-01T00:00:00.002 | {"a":{"b":{}}} | ++-------------------------+----------------+ + +admin flush_table('json_empty_nested'); + ++----------------------------------------+ +| ADMIN flush_table('json_empty_nested') | ++----------------------------------------+ +| 0 | ++----------------------------------------+ + +select ts, j from json_empty_nested order by ts; + ++-------------------------+----------------+ +| ts | j | ++-------------------------+----------------+ +| 1970-01-01T00:00:00.001 | {"a":{}} | +| 1970-01-01T00:00:00.002 | {"a":{"b":{}}} | ++-------------------------+----------------+ + +insert into json_empty_nested values (3, '{"x": []}'); + +Affected Rows: 1 + +admin flush_table('json_empty_nested'); + ++----------------------------------------+ +| ADMIN flush_table('json_empty_nested') | ++----------------------------------------+ +| 0 | ++----------------------------------------+ + +admin compact_table('json_empty_nested'); + ++------------------------------------------+ +| ADMIN compact_table('json_empty_nested') | ++------------------------------------------+ +| 0 | ++------------------------------------------+ + +select ts, j from json_empty_nested order by ts; + ++-------------------------+----------------+ +| ts | j | ++-------------------------+----------------+ +| 1970-01-01T00:00:00.001 | {"a":{}} | +| 1970-01-01T00:00:00.002 | {"a":{"b":{}}} | +| 1970-01-01T00:00:00.003 | {"x":[]} | ++-------------------------+----------------+ + +drop table json_empty_nested; + +Affected Rows: 0 + +-- Lists containing empty objects cannot be materialized as Parquet list fields because that +-- would produce an empty Struct child. They must stay in the remainder and preserve the empty +-- objects through flush and compaction. +create table json_empty_in_lists ( + ts timestamp time index, + j json2 +) with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +Affected Rows: 0 + +insert into json_empty_in_lists values + (1, '{"items":[{}],"mixed":[{"id":1},{}],"nested":[{"meta":{"value":1}},{"meta":{}}]}'); + +Affected Rows: 1 + +admin flush_table('json_empty_in_lists'); + ++------------------------------------------+ +| ADMIN flush_table('json_empty_in_lists') | ++------------------------------------------+ +| 0 | ++------------------------------------------+ + +select ts, j, j.items[0] as item0, j.items[1] as item1, + j.mixed[1] as mixed, j.nested[1].meta as meta +from json_empty_in_lists +order by ts; + ++-------------------------+----------------------------------------------------------------------------------+-------+-------+-------+------+ +| ts | j | item0 | item1 | mixed | meta | ++-------------------------+----------------------------------------------------------------------------------+-------+-------+-------+------+ +| 1970-01-01T00:00:00.001 | {"items":[{}],"mixed":[{"id":1},{}],"nested":[{"meta":{"value":1}},{"meta":{}}]} | {} | | {} | {} | ++-------------------------+----------------------------------------------------------------------------------+-------+-------+-------+------+ + +insert into json_empty_in_lists values + (2, '{"items":[{}],"mixed":[{"id":2},{}],"nested":[{"meta":{"value":2}},{"meta":{}}],"name":"second"}'); + +Affected Rows: 1 + +admin flush_table('json_empty_in_lists'); + ++------------------------------------------+ +| ADMIN flush_table('json_empty_in_lists') | ++------------------------------------------+ +| 0 | ++------------------------------------------+ + +-- Empty objects mixed with non-empty objects in the same list must also stay in +-- the remainder. Otherwise the first item may be reconstructed as {"id":null}. +insert into json_empty_in_lists values + (3, '{"items":[{},{"id":1}]}'); + +Affected Rows: 1 + +admin flush_table('json_empty_in_lists'); + ++------------------------------------------+ +| ADMIN flush_table('json_empty_in_lists') | ++------------------------------------------+ +| 0 | ++------------------------------------------+ + +admin compact_table('json_empty_in_lists'); + ++--------------------------------------------+ +| ADMIN compact_table('json_empty_in_lists') | ++--------------------------------------------+ +| 0 | ++--------------------------------------------+ + +select ts, j, j.items[0] as item0, j.items[1] as item1, + j.mixed[1] as mixed, j.nested[1].meta as meta +from json_empty_in_lists +order by ts; + ++-------------------------+--------------------------------------------------------------------------------------------------+-------+----------+-------+------+ +| ts | j | item0 | item1 | mixed | meta | ++-------------------------+--------------------------------------------------------------------------------------------------+-------+----------+-------+------+ +| 1970-01-01T00:00:00.001 | {"items":[{}],"mixed":[{"id":1},{}],"nested":[{"meta":{"value":1}},{"meta":{}}]} | {} | | {} | {} | +| 1970-01-01T00:00:00.002 | {"items":[{}],"mixed":[{"id":2},{}],"name":"second","nested":[{"meta":{"value":2}},{"meta":{}}]} | {} | | {} | {} | +| 1970-01-01T00:00:00.003 | {"items":[{},{"id":1}]} | {} | {"id":1} | | | ++-------------------------+--------------------------------------------------------------------------------------------------+-------+----------+-------+------+ + +drop table json_empty_in_lists; + +Affected Rows: 0 + +-- Empty objects, SQL NULL values, and JSON null values must remain distinct +-- through memtable flush and compaction. +create table json_empty_null_mixed ( + ts timestamp time index, + j json2 +) with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +Affected Rows: 0 + +insert into json_empty_null_mixed values + (1, '{}'), + (2, NULL), + (3, '{"a": {}}'), + (4, '{"a": null}'); + +Affected Rows: 4 + +select ts, j, j is null from json_empty_null_mixed order by ts; + ++-------------------------+------------+---------------------------------+ +| ts | j | json_empty_null_mixed.j IS NULL | ++-------------------------+------------+---------------------------------+ +| 1970-01-01T00:00:00.001 | {} | false | +| 1970-01-01T00:00:00.002 | | true | +| 1970-01-01T00:00:00.003 | {"a":{}} | false | +| 1970-01-01T00:00:00.004 | {"a":null} | false | ++-------------------------+------------+---------------------------------+ + +admin flush_table('json_empty_null_mixed'); + ++--------------------------------------------+ +| ADMIN flush_table('json_empty_null_mixed') | ++--------------------------------------------+ +| 0 | ++--------------------------------------------+ + +select ts, j, j is null from json_empty_null_mixed order by ts; + ++-------------------------+------------+---------------------------------+ +| ts | j | json_empty_null_mixed.j IS NULL | ++-------------------------+------------+---------------------------------+ +| 1970-01-01T00:00:00.001 | {} | false | +| 1970-01-01T00:00:00.002 | | true | +| 1970-01-01T00:00:00.003 | {"a":{}} | false | +| 1970-01-01T00:00:00.004 | {"a":null} | false | ++-------------------------+------------+---------------------------------+ + +insert into json_empty_null_mixed values + (5, '{"a": 1}'), + (6, NULL); + +Affected Rows: 2 + +select ts, j, j.a, j is null from json_empty_null_mixed order by ts; + ++-------------------------+------------+---------------------------------------------+---------------------------------+ +| ts | j | json_get(json_empty_null_mixed.j,Utf8("a")) | json_empty_null_mixed.j IS NULL | ++-------------------------+------------+---------------------------------------------+---------------------------------+ +| 1970-01-01T00:00:00.001 | {} | | false | +| 1970-01-01T00:00:00.002 | | | true | +| 1970-01-01T00:00:00.003 | {"a":{}} | {} | false | +| 1970-01-01T00:00:00.004 | {"a":null} | | false | +| 1970-01-01T00:00:00.005 | {"a":1} | 1 | false | +| 1970-01-01T00:00:00.006 | | | true | ++-------------------------+------------+---------------------------------------------+---------------------------------+ + +admin flush_table('json_empty_null_mixed'); + ++--------------------------------------------+ +| ADMIN flush_table('json_empty_null_mixed') | ++--------------------------------------------+ +| 0 | ++--------------------------------------------+ + +admin compact_table('json_empty_null_mixed'); + ++----------------------------------------------+ +| ADMIN compact_table('json_empty_null_mixed') | ++----------------------------------------------+ +| 0 | ++----------------------------------------------+ + +select ts, j, j.a, j is null from json_empty_null_mixed order by ts; + ++-------------------------+------------+---------------------------------------------+---------------------------------+ +| ts | j | json_get(json_empty_null_mixed.j,Utf8("a")) | json_empty_null_mixed.j IS NULL | ++-------------------------+------------+---------------------------------------------+---------------------------------+ +| 1970-01-01T00:00:00.001 | {} | | false | +| 1970-01-01T00:00:00.002 | | | true | +| 1970-01-01T00:00:00.003 | {"a":{}} | {} | false | +| 1970-01-01T00:00:00.004 | {"a":null} | | false | +| 1970-01-01T00:00:00.005 | {"a":1} | 1 | false | +| 1970-01-01T00:00:00.006 | | | true | ++-------------------------+------------+---------------------------------------------+---------------------------------+ + +select count(*) from json_empty_null_mixed; + ++----------+ +| count(*) | ++----------+ +| 6 | ++----------+ + +drop table json_empty_null_mixed; + +Affected Rows: 0 + +-- Explicit SQL NULL and omitted-column inserts must be accepted and round trip +-- as NULL (not panic), remaining distinct from empty objects. +create table json_empty_null_forms ( + ts timestamp time index, + j json2 +) with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +Affected Rows: 0 + +insert into json_empty_null_forms (ts, j) values (1, NULL); + +Affected Rows: 1 + +insert into json_empty_null_forms (ts) values (2); + +Affected Rows: 1 + +insert into json_empty_null_forms (ts, j) values (3, '{}'); + +Affected Rows: 1 + +select ts, j, j is null from json_empty_null_forms order by ts; + ++-------------------------+----+---------------------------------+ +| ts | j | json_empty_null_forms.j IS NULL | ++-------------------------+----+---------------------------------+ +| 1970-01-01T00:00:00.001 | | true | +| 1970-01-01T00:00:00.002 | | true | +| 1970-01-01T00:00:00.003 | {} | false | ++-------------------------+----+---------------------------------+ + +admin flush_table('json_empty_null_forms'); + ++--------------------------------------------+ +| ADMIN flush_table('json_empty_null_forms') | ++--------------------------------------------+ +| 0 | ++--------------------------------------------+ + +admin compact_table('json_empty_null_forms'); + ++----------------------------------------------+ +| ADMIN compact_table('json_empty_null_forms') | ++----------------------------------------------+ +| 0 | ++----------------------------------------------+ + +select ts, j, j is null from json_empty_null_forms order by ts; + ++-------------------------+----+---------------------------------+ +| ts | j | json_empty_null_forms.j IS NULL | ++-------------------------+----+---------------------------------+ +| 1970-01-01T00:00:00.001 | | true | +| 1970-01-01T00:00:00.002 | | true | +| 1970-01-01T00:00:00.003 | {} | false | ++-------------------------+----+---------------------------------+ + +drop table json_empty_null_forms; + +Affected Rows: 0 + diff --git a/tests/cases/standalone/common/types/json/json2_empty.sql b/tests/cases/standalone/common/types/json/json2_empty.sql new file mode 100644 index 0000000000..4374df3ab7 --- /dev/null +++ b/tests/cases/standalone/common/types/json/json2_empty.sql @@ -0,0 +1,173 @@ +-- Empty-object JSON2 values are stored in the JSONB remainder column and must +-- survive memtable flush and compaction unchanged, both as whole values and +-- through path access. + +-- Whole-value empty objects: insert, flush, then mix with expanded documents +-- before compacting. +create table json_empty ( + ts timestamp time index, + j json2 +) with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +insert into json_empty values (1, '{}'); + +select ts, j from json_empty order by ts; + +select ts, j.a from json_empty order by ts; + +admin flush_table('json_empty'); + +select ts, j from json_empty order by ts; + +insert into json_empty values (2, '{}'), (3, '{"a": 1}'); + +select ts, j, j.a from json_empty order by ts; + +admin flush_table('json_empty'); + +admin compact_table('json_empty'); + +select ts, j, j.a from json_empty order by ts; + +select count(*) from json_empty; + +drop table json_empty; + +-- Nested empty objects and arrays, which also produce only remainder content, +-- must round trip through flush and compaction as whole values. +create table json_empty_nested ( + ts timestamp time index, + j json2 +) with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +insert into json_empty_nested values + (1, '{"a": {}}'), + (2, '{"a": {"b": {}}}'); + +select ts, j from json_empty_nested order by ts; + +admin flush_table('json_empty_nested'); + +select ts, j from json_empty_nested order by ts; + +insert into json_empty_nested values (3, '{"x": []}'); + +admin flush_table('json_empty_nested'); + +admin compact_table('json_empty_nested'); + +select ts, j from json_empty_nested order by ts; + +drop table json_empty_nested; + +-- Lists containing empty objects cannot be materialized as Parquet list fields because that +-- would produce an empty Struct child. They must stay in the remainder and preserve the empty +-- objects through flush and compaction. +create table json_empty_in_lists ( + ts timestamp time index, + j json2 +) with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +insert into json_empty_in_lists values + (1, '{"items":[{}],"mixed":[{"id":1},{}],"nested":[{"meta":{"value":1}},{"meta":{}}]}'); + +admin flush_table('json_empty_in_lists'); + +select ts, j, j.items[0] as item0, j.items[1] as item1, + j.mixed[1] as mixed, j.nested[1].meta as meta +from json_empty_in_lists +order by ts; + +insert into json_empty_in_lists values + (2, '{"items":[{}],"mixed":[{"id":2},{}],"nested":[{"meta":{"value":2}},{"meta":{}}],"name":"second"}'); + +admin flush_table('json_empty_in_lists'); + +-- Empty objects mixed with non-empty objects in the same list must also stay in +-- the remainder. Otherwise the first item may be reconstructed as {"id":null}. +insert into json_empty_in_lists values + (3, '{"items":[{},{"id":1}]}'); + +admin flush_table('json_empty_in_lists'); + +admin compact_table('json_empty_in_lists'); + +select ts, j, j.items[0] as item0, j.items[1] as item1, + j.mixed[1] as mixed, j.nested[1].meta as meta +from json_empty_in_lists +order by ts; + +drop table json_empty_in_lists; + +-- Empty objects, SQL NULL values, and JSON null values must remain distinct +-- through memtable flush and compaction. +create table json_empty_null_mixed ( + ts timestamp time index, + j json2 +) with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +insert into json_empty_null_mixed values + (1, '{}'), + (2, NULL), + (3, '{"a": {}}'), + (4, '{"a": null}'); + +select ts, j, j is null from json_empty_null_mixed order by ts; + +admin flush_table('json_empty_null_mixed'); + +select ts, j, j is null from json_empty_null_mixed order by ts; + +insert into json_empty_null_mixed values + (5, '{"a": 1}'), + (6, NULL); + +select ts, j, j.a, j is null from json_empty_null_mixed order by ts; + +admin flush_table('json_empty_null_mixed'); + +admin compact_table('json_empty_null_mixed'); + +select ts, j, j.a, j is null from json_empty_null_mixed order by ts; + +select count(*) from json_empty_null_mixed; + +drop table json_empty_null_mixed; + +-- Explicit SQL NULL and omitted-column inserts must be accepted and round trip +-- as NULL (not panic), remaining distinct from empty objects. +create table json_empty_null_forms ( + ts timestamp time index, + j json2 +) with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +insert into json_empty_null_forms (ts, j) values (1, NULL); + +insert into json_empty_null_forms (ts) values (2); + +insert into json_empty_null_forms (ts, j) values (3, '{}'); + +select ts, j, j is null from json_empty_null_forms order by ts; + +admin flush_table('json_empty_null_forms'); + +admin compact_table('json_empty_null_forms'); + +select ts, j, j is null from json_empty_null_forms order by ts; + +drop table json_empty_null_forms; diff --git a/tests/cases/standalone/common/types/json/json2_functions.result b/tests/cases/standalone/common/types/json/json2_functions.result new file mode 100644 index 0000000000..8d8561020f --- /dev/null +++ b/tests/cases/standalone/common/types/json/json2_functions.result @@ -0,0 +1,230 @@ +create table json2_function_table ( + ts timestamp time index, + j json2, + exponent double +) with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +Affected Rows: 0 + +insert into json2_function_table values + (1, '{"metrics":{"value":1,"enabled":true},"label":"one","tags":["a"]}', 2), + (2, '{"metrics":{"value":-2.5,"enabled":false},"label":2,"tags":{"source":"b"}}', 2); + +Affected Rows: 2 + +admin flush_table('json2_function_table'); + ++-------------------------------------------+ +| ADMIN flush_table('json2_function_table') | ++-------------------------------------------+ +| 0 | ++-------------------------------------------+ + +insert into json2_function_table values + (3, '{"metrics":{"value":"three","enabled":"yes"},"label":true,"tags":[1,2]}', 2), + (4, '{"metrics":{"value":true},"label":{"name":"four"},"tags":"four"}', 2), + (5, '{"metrics":{"value":{"nested":5}},"label":["five"],"tags":null}', 2), + (6, '{"metrics":{"value":[6]},"label":null}', 2); + +Affected Rows: 4 + +admin flush_table('json2_function_table'); + ++-------------------------------------------+ +| ADMIN flush_table('json2_function_table') | ++-------------------------------------------+ +| 0 | ++-------------------------------------------+ + +admin compact_table('json2_function_table', 'swcs', '86400'); + ++--------------------------------------------------------------+ +| ADMIN compact_table('json2_function_table', 'swcs', '86400') | ++--------------------------------------------------------------+ +| 0 | ++--------------------------------------------------------------+ + +insert into json2_function_table values + (7, '{"metrics":{"value":null},"label":"seven","extra":7}', 2), + (8, '{"metrics":"opaque","label":8,"extra":false}', 2); + +Affected Rows: 2 + +admin flush_table('json2_function_table'); + ++-------------------------------------------+ +| ADMIN flush_table('json2_function_table') | ++-------------------------------------------+ +| 0 | ++-------------------------------------------+ + +insert into json2_function_table values + (9, '{"other":9,"label":{"name":"nine"}}', 2), + (10, '{"metrics":{"value":10},"label":"ten","extra":[10]}', 2); + +Affected Rows: 2 + +-- Arrow casts Boolean values to Float64 as true = 1.0 and false = 0.0. +select abs(j.metrics.value) as abs_value from json2_function_table order by ts; + ++-----------+ +| abs_value | ++-----------+ +| 1.0 | +| 2.5 | +| | +| 1.0 | +| | +| | +| | +| | +| | +| 10.0 | ++-----------+ + +select power(j.label, 2) * 2 as scaled_label from json2_function_table order by ts; + ++--------------+ +| scaled_label | ++--------------+ +| | +| 8.0 | +| 2.0 | +| | +| | +| | +| | +| 128.0 | +| | +| | ++--------------+ + +select power(j.metrics.value, exponent) as powered_value from json2_function_table order by ts; + ++---------------+ +| powered_value | ++---------------+ +| 1.0 | +| 6.25 | +| | +| 1.0 | +| | +| | +| | +| | +| | +| 100.0 | ++---------------+ + +select coalesce(j.metrics.value, exponent) as value_or_exponent from json2_function_table order by ts; + ++-------------------+ +| value_or_exponent | ++-------------------+ +| 1.0 | +| -2.5 | +| 2.0 | +| 1.0 | +| 2.0 | +| 2.0 | +| 2.0 | +| 2.0 | +| 2.0 | +| 10.0 | ++-------------------+ + +select coalesce( + j.metrics.value, + j.label::double +) as mixed_value from json2_function_table order by ts; + ++-------------+ +| mixed_value | ++-------------+ +| 1.0 | +| -2.5 | +| 1.0 | +| 1.0 | +| | +| | +| | +| 8.0 | +| | +| 10.0 | ++-------------+ + +select ts, j.label as label, signum(j.metrics.value) as value_sign +from json2_function_table +order by ts; + ++-------------------------+-----------------+------------+ +| ts | label | value_sign | ++-------------------------+-----------------+------------+ +| 1970-01-01T00:00:00.001 | one | 1.0 | +| 1970-01-01T00:00:00.002 | 2 | -1.0 | +| 1970-01-01T00:00:00.003 | true | | +| 1970-01-01T00:00:00.004 | {"name":"four"} | 1.0 | +| 1970-01-01T00:00:00.005 | ["five"] | | +| 1970-01-01T00:00:00.006 | | | +| 1970-01-01T00:00:00.007 | seven | | +| 1970-01-01T00:00:00.008 | 8 | | +| 1970-01-01T00:00:00.009 | {"name":"nine"} | | +| 1970-01-01T00:00:00.010 | ten | 1.0 | ++-------------------------+-----------------+------------+ + +select ts, j.label as label +from json2_function_table +where iszero(j.metrics.value) = false +order by ts; + ++-------------------------+-----------------+ +| ts | label | ++-------------------------+-----------------+ +| 1970-01-01T00:00:00.001 | one | +| 1970-01-01T00:00:00.002 | 2 | +| 1970-01-01T00:00:00.004 | {"name":"four"} | +| 1970-01-01T00:00:00.010 | ten | ++-------------------------+-----------------+ + +select count(j.metrics.value) as value_count from json2_function_table; + ++-------------+ +| value_count | ++-------------+ +| 7 | ++-------------+ + +select sum(j.metrics.value) as value_sum from json2_function_table; + ++-----------+ +| value_sum | ++-----------+ +| 9.5 | ++-----------+ + +select lag(j.metrics.value) over (order by ts) as previous_value +from json2_function_table +order by ts; + ++----------------+ +| previous_value | ++----------------+ +| | +| 1 | +| -2.5 | +| three | +| true | +| {"nested":5} | +| [6] | +| | +| | +| | ++----------------+ + +drop table json2_function_table; + +Affected Rows: 0 + diff --git a/tests/cases/standalone/common/types/json/json2_functions.sql b/tests/cases/standalone/common/types/json/json2_functions.sql new file mode 100644 index 0000000000..3dcced0557 --- /dev/null +++ b/tests/cases/standalone/common/types/json/json2_functions.sql @@ -0,0 +1,67 @@ +create table json2_function_table ( + ts timestamp time index, + j json2, + exponent double +) with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +insert into json2_function_table values + (1, '{"metrics":{"value":1,"enabled":true},"label":"one","tags":["a"]}', 2), + (2, '{"metrics":{"value":-2.5,"enabled":false},"label":2,"tags":{"source":"b"}}', 2); + +admin flush_table('json2_function_table'); + +insert into json2_function_table values + (3, '{"metrics":{"value":"three","enabled":"yes"},"label":true,"tags":[1,2]}', 2), + (4, '{"metrics":{"value":true},"label":{"name":"four"},"tags":"four"}', 2), + (5, '{"metrics":{"value":{"nested":5}},"label":["five"],"tags":null}', 2), + (6, '{"metrics":{"value":[6]},"label":null}', 2); + +admin flush_table('json2_function_table'); + +admin compact_table('json2_function_table', 'swcs', '86400'); + +insert into json2_function_table values + (7, '{"metrics":{"value":null},"label":"seven","extra":7}', 2), + (8, '{"metrics":"opaque","label":8,"extra":false}', 2); + +admin flush_table('json2_function_table'); + +insert into json2_function_table values + (9, '{"other":9,"label":{"name":"nine"}}', 2), + (10, '{"metrics":{"value":10},"label":"ten","extra":[10]}', 2); + +-- Arrow casts Boolean values to Float64 as true = 1.0 and false = 0.0. +select abs(j.metrics.value) as abs_value from json2_function_table order by ts; + +select power(j.label, 2) * 2 as scaled_label from json2_function_table order by ts; + +select power(j.metrics.value, exponent) as powered_value from json2_function_table order by ts; + +select coalesce(j.metrics.value, exponent) as value_or_exponent from json2_function_table order by ts; + +select coalesce( + j.metrics.value, + j.label::double +) as mixed_value from json2_function_table order by ts; + +select ts, j.label as label, signum(j.metrics.value) as value_sign +from json2_function_table +order by ts; + +select ts, j.label as label +from json2_function_table +where iszero(j.metrics.value) = false +order by ts; + +select count(j.metrics.value) as value_count from json2_function_table; + +select sum(j.metrics.value) as value_sum from json2_function_table; + +select lag(j.metrics.value) over (order by ts) as previous_value +from json2_function_table +order by ts; + +drop table json2_function_table; diff --git a/tests/cases/standalone/common/types/json/json2_join.result b/tests/cases/standalone/common/types/json/json2_join.result new file mode 100644 index 0000000000..04942f0360 --- /dev/null +++ b/tests/cases/standalone/common/types/json/json2_join.result @@ -0,0 +1,81 @@ +create table json2_join_same_name_left ( + ts timestamp time index, + k string, + j json2 +) +with ( + 'append_mode' = 'true' +); + +Affected Rows: 0 + +create table json2_join_same_name_right ( + ts timestamp time index, + k string, + j json2 +) +with ( + 'append_mode' = 'true' +); + +Affected Rows: 0 + +insert into json2_join_same_name_left values + (1, 'a', '{"a": 1, "left_only": "kept"}'); + +Affected Rows: 1 + +insert into json2_join_same_name_right values + (1, 'a', '{"a": "right", "right_only": "should be kept"}'); + +Affected Rows: 1 + +admin flush_table('json2_join_same_name_left'); + ++------------------------------------------------+ +| ADMIN flush_table('json2_join_same_name_left') | ++------------------------------------------------+ +| 0 | ++------------------------------------------------+ + +admin flush_table('json2_join_same_name_right'); + ++-------------------------------------------------+ +| ADMIN flush_table('json2_join_same_name_right') | ++-------------------------------------------------+ +| 0 | ++-------------------------------------------------+ + +-- Conflicting hints for same-named JSON2 columns preserve both values as Variant. +select json_get(r.j, 'a')::string, json_get(l.j, 'a')::int64 +from json2_join_same_name_left l +join json2_join_same_name_right r +on l.k = r.k; + ++-------------------------+---------------------------------------------------+ +| json_get(r.j,Utf8("a")) | arrow_cast(json_get(l.j,Utf8("a")),Utf8("Int64")) | ++-------------------------+---------------------------------------------------+ +| right | 1 | ++-------------------------+---------------------------------------------------+ + +select + l.j as left_json, + r.j as right_json +from json2_join_same_name_left l +join json2_join_same_name_right r +on l.k = r.k; + ++----------------------------+---------------------------------------------+ +| left_json | right_json | ++----------------------------+---------------------------------------------+ +| {"a":1,"left_only":"kept"} | {"a":"right","right_only":"should be kept"} | ++----------------------------+---------------------------------------------+ + +drop table json2_join_same_name_left; + +Affected Rows: 0 + +drop table json2_join_same_name_right; + +Affected Rows: 0 + diff --git a/tests/cases/standalone/common/types/json/json2_join.sql b/tests/cases/standalone/common/types/json/json2_join.sql new file mode 100644 index 0000000000..88890741fa --- /dev/null +++ b/tests/cases/standalone/common/types/json/json2_join.sql @@ -0,0 +1,44 @@ +create table json2_join_same_name_left ( + ts timestamp time index, + k string, + j json2 +) +with ( + 'append_mode' = 'true' +); + +create table json2_join_same_name_right ( + ts timestamp time index, + k string, + j json2 +) +with ( + 'append_mode' = 'true' +); + +insert into json2_join_same_name_left values + (1, 'a', '{"a": 1, "left_only": "kept"}'); + +insert into json2_join_same_name_right values + (1, 'a', '{"a": "right", "right_only": "should be kept"}'); + +admin flush_table('json2_join_same_name_left'); + +admin flush_table('json2_join_same_name_right'); + +-- Conflicting hints for same-named JSON2 columns preserve both values as Variant. +select json_get(r.j, 'a')::string, json_get(l.j, 'a')::int64 +from json2_join_same_name_left l +join json2_join_same_name_right r +on l.k = r.k; + +select + l.j as left_json, + r.j as right_json +from json2_join_same_name_left l +join json2_join_same_name_right r +on l.k = r.k; + +drop table json2_join_same_name_left; + +drop table json2_join_same_name_right; diff --git a/tests/cases/standalone/common/types/json/json2_limit.result b/tests/cases/standalone/common/types/json/json2_limit.result index 035c4ed07b..555e76f1e5 100644 --- a/tests/cases/standalone/common/types/json/json2_limit.result +++ b/tests/cases/standalone/common/types/json/json2_limit.result @@ -28,10 +28,6 @@ insert into json2_disable_non_object_insert values (5, 'null'); Error: 1001(Unsupported), Non-object json is not supported currently -insert into json2_disable_non_object_insert values (6, '{}'); - -Error: 1004(InvalidArguments), Invalid InsertRequest, reason: empty json object is not supported, consider adding a dummy field - drop table json2_disable_non_object_insert; Affected Rows: 0 @@ -65,87 +61,10 @@ order by json_get(j, 'a.b'); | 2 | 1 | +---------------------------------------------------+----------+ -select j, j.a from json2_whole_and_path_read; - -Error: 3001(EngineExecuteQuery), Invalid argument error: column types must match schema types, expected Binary but found Struct("a": Utf8View) at column index 0 - -select j from json2_whole_and_path_read where j.a.b = 1; - -Error: 3001(EngineExecuteQuery), Invalid argument error: column types must match schema types, expected Binary but found Struct("a": Struct("b": Int64)) at column index 0 - drop table json2_whole_and_path_read; Affected Rows: 0 -create table json2_join_same_name_left ( - ts timestamp time index, - k string, - j json2 -) -with ( - 'append_mode' = 'true' -); - -Affected Rows: 0 - -create table json2_join_same_name_right ( - ts timestamp time index, - k string, - j json2 -) -with ( - 'append_mode' = 'true' -); - -Affected Rows: 0 - -insert into json2_join_same_name_left values - (1, 'a', '{"a": 1, "left_only": "kept"}'); - -Affected Rows: 1 - -insert into json2_join_same_name_right values - (1, 'a', '{"a": "right", "right_only": "should be kept"}'); - -Affected Rows: 1 - -admin flush_table('json2_join_same_name_left'); - -+------------------------------------------------+ -| ADMIN flush_table('json2_join_same_name_left') | -+------------------------------------------------+ -| 0 | -+------------------------------------------------+ - -admin flush_table('json2_join_same_name_right'); - -+-------------------------------------------------+ -| ADMIN flush_table('json2_join_same_name_right') | -+-------------------------------------------------+ -| 0 | -+-------------------------------------------------+ - --- FIXME: This should return `right` and `1`. The current NULL values are caused --- by JSON type hints losing the table qualifier in joins. -select json_get(r.j, 'a')::string, json_get(l.j, 'a')::int64 -from json2_join_same_name_left l -join json2_join_same_name_right r -on l.k = r.k; - -+---------------------------------------------+---------------------------------------------------+ -| json_get(r.j,Utf8("a")) | arrow_cast(json_get(l.j,Utf8("a")),Utf8("Int64")) | -+---------------------------------------------+---------------------------------------------------+ -| {"a":"right","right_only":"should be kept"} | | -+---------------------------------------------+---------------------------------------------------+ - -drop table json2_join_same_name_left; - -Affected Rows: 0 - -drop table json2_join_same_name_right; - -Affected Rows: 0 - create table json2_without_append_mode ( ts timestamp time index, j json2 diff --git a/tests/cases/standalone/common/types/json/json2_limit.sql b/tests/cases/standalone/common/types/json/json2_limit.sql index 986ed58eb2..653fdcec12 100644 --- a/tests/cases/standalone/common/types/json/json2_limit.sql +++ b/tests/cases/standalone/common/types/json/json2_limit.sql @@ -16,8 +16,6 @@ insert into json2_disable_non_object_insert values (4, 'true'); insert into json2_disable_non_object_insert values (5, 'null'); -insert into json2_disable_non_object_insert values (6, '{}'); - drop table json2_disable_non_object_insert; create table json2_whole_and_path_read ( @@ -38,51 +36,8 @@ from json2_whole_and_path_read group by json_get(j, 'a.b') order by json_get(j, 'a.b'); -select j, j.a from json2_whole_and_path_read; - -select j from json2_whole_and_path_read where j.a.b = 1; - drop table json2_whole_and_path_read; -create table json2_join_same_name_left ( - ts timestamp time index, - k string, - j json2 -) -with ( - 'append_mode' = 'true' -); - -create table json2_join_same_name_right ( - ts timestamp time index, - k string, - j json2 -) -with ( - 'append_mode' = 'true' -); - -insert into json2_join_same_name_left values - (1, 'a', '{"a": 1, "left_only": "kept"}'); - -insert into json2_join_same_name_right values - (1, 'a', '{"a": "right", "right_only": "should be kept"}'); - -admin flush_table('json2_join_same_name_left'); - -admin flush_table('json2_join_same_name_right'); - --- FIXME: This should return `right` and `1`. The current NULL values are caused --- by JSON type hints losing the table qualifier in joins. -select json_get(r.j, 'a')::string, json_get(l.j, 'a')::int64 -from json2_join_same_name_left l -join json2_join_same_name_right r -on l.k = r.k; - -drop table json2_join_same_name_left; - -drop table json2_join_same_name_right; - create table json2_without_append_mode ( ts timestamp time index, j json2 diff --git a/tests/cases/standalone/common/types/json/json2_list_index.result b/tests/cases/standalone/common/types/json/json2_list_index.result new file mode 100644 index 0000000000..e7415ce8e2 --- /dev/null +++ b/tests/cases/standalone/common/types/json/json2_list_index.result @@ -0,0 +1,131 @@ +CREATE TABLE json2_list_index ( + ts TIMESTAMP TIME INDEX, + host STRING, + j JSON2 +) WITH ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +Affected Rows: 0 + +INSERT INTO json2_list_index VALUES + (1, 'host1', '{"l":[[10,11],["a","b"]],"o":{"l":[{"inner":{"l":[1,2,3]}},{"inner":{"l":["x","y","z"]},"casekey":"normalized","UPPER":"quoted","a.b":"dotted"}]}}'), + (2, 'host2', '{"l":[[20],[21,22,23]],"o":{"l":[{"inner":{"l":[4,5,6]}},{"inner":{"l":[7,8]}}]}}'), + (3, 'host3', '{"l":[null,[30]],"o":{"l":[null,{"inner":{"l":[true,false,null]}}]}}'); + +Affected Rows: 3 + +ADMIN FLUSH_TABLE('json2_list_index'); + ++---------------------------------------+ +| ADMIN FLUSH_TABLE('json2_list_index') | ++---------------------------------------+ +| 0 | ++---------------------------------------+ + +INSERT INTO json2_list_index VALUES + (4, 'host4', '{"l":"not a list","o":{"l":{"inner":{"l":[40]}}}}'), + (5, 'host5', '{"l":[[50,51],null],"o":{"l":[{},null]}}'), + (6, 'host6', '{"other":"missing paths"}'); + +Affected Rows: 3 + +ADMIN FLUSH_TABLE('json2_list_index'); + ++---------------------------------------+ +| ADMIN FLUSH_TABLE('json2_list_index') | ++---------------------------------------+ +| 0 | ++---------------------------------------+ + +ADMIN COMPACT_TABLE('json2_list_index', 'swcs', '86400'); + ++----------------------------------------------------------+ +| ADMIN COMPACT_TABLE('json2_list_index', 'swcs', '86400') | ++----------------------------------------------------------+ +| 0 | ++----------------------------------------------------------+ + +SELECT ts, host, j.l[0][1] AS nested_list +FROM json2_list_index +ORDER BY ts; + ++-------------------------+-------+-------------+ +| ts | host | nested_list | ++-------------------------+-------+-------------+ +| 1970-01-01T00:00:00.001 | host1 | 11 | +| 1970-01-01T00:00:00.002 | host2 | | +| 1970-01-01T00:00:00.003 | host3 | | +| 1970-01-01T00:00:00.004 | host4 | | +| 1970-01-01T00:00:00.005 | host5 | 51 | +| 1970-01-01T00:00:00.006 | host6 | | ++-------------------------+-------+-------------+ + +SELECT ts, j.l[1][0] AS second_list +FROM json2_list_index +ORDER BY ts; + ++-------------------------+-------------+ +| ts | second_list | ++-------------------------+-------------+ +| 1970-01-01T00:00:00.001 | a | +| 1970-01-01T00:00:00.002 | 21 | +| 1970-01-01T00:00:00.003 | 30 | +| 1970-01-01T00:00:00.004 | | +| 1970-01-01T00:00:00.005 | | +| 1970-01-01T00:00:00.006 | | ++-------------------------+-------------+ + +SELECT ts, host, j.o.l[1].inner.l[2] AS deeply_nested +FROM json2_list_index +ORDER BY ts; + ++-------------------------+-------+---------------+ +| ts | host | deeply_nested | ++-------------------------+-------+---------------+ +| 1970-01-01T00:00:00.001 | host1 | z | +| 1970-01-01T00:00:00.002 | host2 | | +| 1970-01-01T00:00:00.003 | host3 | | +| 1970-01-01T00:00:00.004 | host4 | | +| 1970-01-01T00:00:00.005 | host5 | | +| 1970-01-01T00:00:00.006 | host6 | | ++-------------------------+-------+---------------+ + +SELECT ts, + j.o.l[1].CASEKEY AS normalized, + j.o.l[1]."UPPER" AS quoted_upper, + j.o.l[1]."a.b" AS dotted_key +FROM json2_list_index +ORDER BY ts; + ++-------------------------+------------+--------------+------------+ +| ts | normalized | quoted_upper | dotted_key | ++-------------------------+------------+--------------+------------+ +| 1970-01-01T00:00:00.001 | normalized | quoted | dotted | +| 1970-01-01T00:00:00.002 | | | | +| 1970-01-01T00:00:00.003 | | | | +| 1970-01-01T00:00:00.004 | | | | +| 1970-01-01T00:00:00.005 | | | | +| 1970-01-01T00:00:00.006 | | | | ++-------------------------+------------+--------------+------------+ + +SELECT ts, j.l[0][0]::DOUBLE * 2 AS calculated +FROM json2_list_index +ORDER BY ts; + ++-------------------------+------------+ +| ts | calculated | ++-------------------------+------------+ +| 1970-01-01T00:00:00.001 | 20.0 | +| 1970-01-01T00:00:00.002 | 40.0 | +| 1970-01-01T00:00:00.003 | | +| 1970-01-01T00:00:00.004 | | +| 1970-01-01T00:00:00.005 | 100.0 | +| 1970-01-01T00:00:00.006 | | ++-------------------------+------------+ + +DROP TABLE json2_list_index; + +Affected Rows: 0 + diff --git a/tests/cases/standalone/common/types/json/json2_list_index.sql b/tests/cases/standalone/common/types/json/json2_list_index.sql new file mode 100644 index 0000000000..d14c0a5996 --- /dev/null +++ b/tests/cases/standalone/common/types/json/json2_list_index.sql @@ -0,0 +1,49 @@ +CREATE TABLE json2_list_index ( + ts TIMESTAMP TIME INDEX, + host STRING, + j JSON2 +) WITH ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +INSERT INTO json2_list_index VALUES + (1, 'host1', '{"l":[[10,11],["a","b"]],"o":{"l":[{"inner":{"l":[1,2,3]}},{"inner":{"l":["x","y","z"]},"casekey":"normalized","UPPER":"quoted","a.b":"dotted"}]}}'), + (2, 'host2', '{"l":[[20],[21,22,23]],"o":{"l":[{"inner":{"l":[4,5,6]}},{"inner":{"l":[7,8]}}]}}'), + (3, 'host3', '{"l":[null,[30]],"o":{"l":[null,{"inner":{"l":[true,false,null]}}]}}'); + +ADMIN FLUSH_TABLE('json2_list_index'); + +INSERT INTO json2_list_index VALUES + (4, 'host4', '{"l":"not a list","o":{"l":{"inner":{"l":[40]}}}}'), + (5, 'host5', '{"l":[[50,51],null],"o":{"l":[{},null]}}'), + (6, 'host6', '{"other":"missing paths"}'); + +ADMIN FLUSH_TABLE('json2_list_index'); + +ADMIN COMPACT_TABLE('json2_list_index', 'swcs', '86400'); + +SELECT ts, host, j.l[0][1] AS nested_list +FROM json2_list_index +ORDER BY ts; + +SELECT ts, j.l[1][0] AS second_list +FROM json2_list_index +ORDER BY ts; + +SELECT ts, host, j.o.l[1].inner.l[2] AS deeply_nested +FROM json2_list_index +ORDER BY ts; + +SELECT ts, + j.o.l[1].CASEKEY AS normalized, + j.o.l[1]."UPPER" AS quoted_upper, + j.o.l[1]."a.b" AS dotted_key +FROM json2_list_index +ORDER BY ts; + +SELECT ts, j.l[0][0]::DOUBLE * 2 AS calculated +FROM json2_list_index +ORDER BY ts; + +DROP TABLE json2_list_index; diff --git a/tests/cases/standalone/common/types/json/json2_no_auto_paths.result b/tests/cases/standalone/common/types/json/json2_no_auto_paths.result new file mode 100644 index 0000000000..b11b16fd2e --- /dev/null +++ b/tests/cases/standalone/common/types/json/json2_no_auto_paths.result @@ -0,0 +1,135 @@ +-- With no type hints and automatic path expansion disabled, every JSON value +-- is stored in the remainder column. +create table json2_no_auto_paths ( + ts timestamp time index, + j json2( + max_auto_expanded_paths = 0 + ) +) with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +Affected Rows: 0 + +show create table json2_no_auto_paths; + ++---------------------+----------------------------------------------------+ +| Table | Create Table | ++---------------------+----------------------------------------------------+ +| json2_no_auto_paths | CREATE TABLE IF NOT EXISTS "json2_no_auto_paths" ( | +| | "ts" TIMESTAMP(3) NOT NULL, | +| | "j" JSON2( | +| | max_auto_expanded_paths = 0 | +| | ) NULL, | +| | TIME INDEX ("ts") | +| | ) | +| | | +| | ENGINE=mito | +| | WITH( | +| | append_mode = 'true', | +| | sst_format = 'flat' | +| | ) | ++---------------------+----------------------------------------------------+ + +insert into json2_no_auto_paths values + (1, '{"profile":{"name":"alice","contact":{"address":{"city":"Paris"}}},"groups":[{"members":[{"name":"a0"}]},{"members":[{"name":"a1"},{"name":"a2"}]}],"matrix":[[1,2],[3,4]]}'), + (2, '{"profile":{"name":"bob","contact":{"address":{"city":"Berlin"}}},"groups":[{"members":[]},{"members":[{"name":"b1"}]}],"matrix":[[10],[20,21]]}'); + +Affected Rows: 2 + +admin flush_table('json2_no_auto_paths'); + ++------------------------------------------+ +| ADMIN flush_table('json2_no_auto_paths') | ++------------------------------------------+ +| 0 | ++------------------------------------------+ + +insert into json2_no_auto_paths values + (3, '{}'), + (4, '{"profile":{"name":"carol","contact":{"address":{}}},"groups":[null,{"members":[null,{"name":"c1"}]}],"matrix":[[],[30]]}'); + +Affected Rows: 2 + +admin flush_table('json2_no_auto_paths'); + ++------------------------------------------+ +| ADMIN flush_table('json2_no_auto_paths') | ++------------------------------------------+ +| 0 | ++------------------------------------------+ + +admin compact_table('json2_no_auto_paths', 'swcs', '86400'); + ++-------------------------------------------------------------+ +| ADMIN compact_table('json2_no_auto_paths', 'swcs', '86400') | ++-------------------------------------------------------------+ +| 0 | ++-------------------------------------------------------------+ + +select ts, j from json2_no_auto_paths order by ts; + ++-------------------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ +| ts | j | ++-------------------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ +| 1970-01-01T00:00:00.001 | {"groups":[{"members":[{"name":"a0"}]},{"members":[{"name":"a1"},{"name":"a2"}]}],"matrix":[[1,2],[3,4]],"profile":{"contact":{"address":{"city":"Paris"}},"name":"alice"}} | +| 1970-01-01T00:00:00.002 | {"groups":[{"members":[]},{"members":[{"name":"b1"}]}],"matrix":[[10],[20,21]],"profile":{"contact":{"address":{"city":"Berlin"}},"name":"bob"}} | +| 1970-01-01T00:00:00.003 | {} | +| 1970-01-01T00:00:00.004 | {"groups":[null,{"members":[null,{"name":"c1"}]}],"matrix":[[],[30]],"profile":{"contact":{"address":{}},"name":"carol"}} | ++-------------------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ + +select j as empty_object from json2_no_auto_paths where ts = 3; + ++--------------+ +| empty_object | ++--------------+ +| {} | ++--------------+ + +select ts, upper(j.profile.name) as upper_name +from json2_no_auto_paths +order by ts; + ++-------------------------+------------+ +| ts | upper_name | ++-------------------------+------------+ +| 1970-01-01T00:00:00.001 | ALICE | +| 1970-01-01T00:00:00.002 | BOB | +| 1970-01-01T00:00:00.003 | | +| 1970-01-01T00:00:00.004 | CAROL | ++-------------------------+------------+ + +select ts, j.profile.contact.address.city as city +from json2_no_auto_paths +order by ts; + ++-------------------------+--------+ +| ts | city | ++-------------------------+--------+ +| 1970-01-01T00:00:00.001 | Paris | +| 1970-01-01T00:00:00.002 | Berlin | +| 1970-01-01T00:00:00.003 | | +| 1970-01-01T00:00:00.004 | | ++-------------------------+--------+ + +select + ts, + j.matrix[0][1] as matrix_value, + j.groups[1].members[0].name as member_name +from json2_no_auto_paths +order by ts; + ++-------------------------+--------------+-------------+ +| ts | matrix_value | member_name | ++-------------------------+--------------+-------------+ +| 1970-01-01T00:00:00.001 | 2 | a1 | +| 1970-01-01T00:00:00.002 | | b1 | +| 1970-01-01T00:00:00.003 | | | +| 1970-01-01T00:00:00.004 | | | ++-------------------------+--------------+-------------+ + +drop table json2_no_auto_paths; + +Affected Rows: 0 + diff --git a/tests/cases/standalone/common/types/json/json2_no_auto_paths.sql b/tests/cases/standalone/common/types/json/json2_no_auto_paths.sql new file mode 100644 index 0000000000..163d354598 --- /dev/null +++ b/tests/cases/standalone/common/types/json/json2_no_auto_paths.sql @@ -0,0 +1,48 @@ +-- With no type hints and automatic path expansion disabled, every JSON value +-- is stored in the remainder column. +create table json2_no_auto_paths ( + ts timestamp time index, + j json2( + max_auto_expanded_paths = 0 + ) +) with ( + 'append_mode' = 'true', + 'sst_format' = 'flat' +); + +show create table json2_no_auto_paths; + +insert into json2_no_auto_paths values + (1, '{"profile":{"name":"alice","contact":{"address":{"city":"Paris"}}},"groups":[{"members":[{"name":"a0"}]},{"members":[{"name":"a1"},{"name":"a2"}]}],"matrix":[[1,2],[3,4]]}'), + (2, '{"profile":{"name":"bob","contact":{"address":{"city":"Berlin"}}},"groups":[{"members":[]},{"members":[{"name":"b1"}]}],"matrix":[[10],[20,21]]}'); + +admin flush_table('json2_no_auto_paths'); + +insert into json2_no_auto_paths values + (3, '{}'), + (4, '{"profile":{"name":"carol","contact":{"address":{}}},"groups":[null,{"members":[null,{"name":"c1"}]}],"matrix":[[],[30]]}'); + +admin flush_table('json2_no_auto_paths'); + +admin compact_table('json2_no_auto_paths', 'swcs', '86400'); + +select ts, j from json2_no_auto_paths order by ts; + +select j as empty_object from json2_no_auto_paths where ts = 3; + +select ts, upper(j.profile.name) as upper_name +from json2_no_auto_paths +order by ts; + +select ts, j.profile.contact.address.city as city +from json2_no_auto_paths +order by ts; + +select + ts, + j.matrix[0][1] as matrix_value, + j.groups[1].members[0].name as member_name +from json2_no_auto_paths +order by ts; + +drop table json2_no_auto_paths; diff --git a/tests/cases/standalone/common/types/json/json2_type_hints.result b/tests/cases/standalone/common/types/json/json2_type_hints.result index 785fa76942..d9a384e084 100644 --- a/tests/cases/standalone/common/types/json/json2_type_hints.result +++ b/tests/cases/standalone/common/types/json/json2_type_hints.result @@ -20,6 +20,7 @@ SHOW CREATE TABLE json2_type_hints; | json2_type_hints | CREATE TABLE IF NOT EXISTS "json2_type_hints" ( | | | "ts" TIMESTAMP(3) NOT NULL, | | | "j" JSON2( | +| | max_auto_expanded_paths = 100, | | | "user"."age" BIGINT NOT NULL DEFAULT 18, | | | "user"."name" STRING NULL DEFAULT 'unknown', | | | "user"."active" BOOLEAN NULL, | diff --git a/tests/cases/standalone/optimizer/count.result b/tests/cases/standalone/optimizer/count.result index 99f910136f..8a128a9200 100644 --- a/tests/cases/standalone/optimizer/count.result +++ b/tests/cases/standalone/optimizer/count.result @@ -278,6 +278,7 @@ select count(1) from count_where_bug; | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_ProjectionExec: expr=[{count[count]:REDACTED} as __count_state(count_where_bug.ts)] REDACTED @@ -313,6 +314,7 @@ select count(1) from count_where_bug where `tag` = 'b'; | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_state(count_where_bug.ts)] REDACTED @@ -339,6 +341,7 @@ select count(1) from count_where_bug where `tag` = 'b'; | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_state(count_where_bug.ts)] REDACTED @@ -375,6 +378,7 @@ select count(1) from count_where_bug where ts > '2024-09-06T06:00:04Z'; | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_state(count_where_bug.ts)] REDACTED @@ -414,6 +418,7 @@ select count(1) from count_where_bug where num != 3; | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_state(count_where_bug.ts)] REDACTED diff --git a/tests/cases/standalone/optimizer/first_value_advance.result b/tests/cases/standalone/optimizer/first_value_advance.result index afcefae4cf..c543a8df2a 100644 --- a/tests/cases/standalone/optimizer/first_value_advance.result +++ b/tests/cases/standalone/optimizer/first_value_advance.result @@ -322,6 +322,7 @@ explain select first_value(ts order by ts) from t; | physical_plan | AggregateExec: mode=Final, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -345,6 +346,7 @@ explain analyze | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED @@ -411,6 +413,7 @@ order by time_window, ordered_host; |_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -444,6 +447,7 @@ order by time_window, ordered_host; |_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED @@ -522,6 +526,7 @@ explain | physical_plan | AggregateExec: mode=Final, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -545,6 +550,7 @@ explain analyze | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED @@ -705,6 +711,7 @@ order by time_window, ordered_host; |_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -738,6 +745,7 @@ order by time_window, ordered_host; |_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED diff --git a/tests/cases/standalone/optimizer/last_value_advance.result b/tests/cases/standalone/optimizer/last_value_advance.result index a1882a7588..3199692a73 100644 --- a/tests/cases/standalone/optimizer/last_value_advance.result +++ b/tests/cases/standalone/optimizer/last_value_advance.result @@ -322,6 +322,7 @@ explain select last_value(ts order by ts) from t; | physical_plan | AggregateExec: mode=Final, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -345,6 +346,7 @@ explain analyze | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED @@ -411,6 +413,7 @@ order by time_window, ordered_host; |_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -444,6 +447,7 @@ order by time_window, ordered_host; |_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED @@ -522,6 +526,7 @@ explain | physical_plan | AggregateExec: mode=Final, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| |_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -545,6 +550,7 @@ explain analyze | 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED @@ -705,6 +711,7 @@ order by time_window, ordered_host; |_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -738,6 +745,7 @@ order by time_window, ordered_host; |_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED diff --git a/tests/cases/standalone/optimizer/rewrite_set_comparison.result b/tests/cases/standalone/optimizer/rewrite_set_comparison.result index eb41d5f78e..abd7631736 100644 --- a/tests/cases/standalone/optimizer/rewrite_set_comparison.result +++ b/tests/cases/standalone/optimizer/rewrite_set_comparison.result @@ -47,6 +47,8 @@ ADMIN FLUSH_TABLE('sc_s'); -- SQLNESS REPLACE region=\d+\(\d+,\s+\d+\) region=REDACTED -- SQLNESS REPLACE (peers.*) REDACTED +-- SQLNESS REPLACE RoundRobinBatch\(\d+\) RoundRobinBatch(REDACTED) +-- SQLNESS REPLACE (RoundRobinBatch\(REDACTED\),\sinput_partitions=\d+)\s+\| $1| EXPLAIN SELECT v FROM sc_t WHERE v > ANY(SELECT v FROM sc_s) ORDER BY v; +---------------+---------------------------------------------------------------------------------------------------------------+ @@ -81,11 +83,12 @@ EXPLAIN SELECT v FROM sc_t WHERE v > ANY(SELECT v FROM sc_s) ORDER BY v; | | CoalescePartitionsExec | | | FilterExec: mark@1 OR NOT mark@1 AND NULL | | | NestedLoopJoinExec: join_type=LeftMark, filter=(v@0 > v@1) IS NOT DISTINCT FROM true | -| | CoalescePartitionsExec | -| | ProjectionExec: expr=[v@1 as v] | -| | MergeScanExec: REDACTED -| | MergeScanExec: REDACTED -| | MergeScanExec: REDACTED +| | ProjectionExec: expr=[v@1 as v] | +| | MergeScanExec: REDACTED +| | RepartitionExec: partitioning=RoundRobinBatch(REDACTED), input_partitions=1| +| | MergeScanExec: REDACTED +| | RepartitionExec: partitioning=RoundRobinBatch(REDACTED), input_partitions=1| +| | MergeScanExec: REDACTED | | | +---------------+---------------------------------------------------------------------------------------------------------------+ @@ -100,6 +103,8 @@ SELECT v FROM sc_t WHERE v > ANY(SELECT v FROM sc_s) ORDER BY v; -- SQLNESS REPLACE region=\d+\(\d+,\s+\d+\) region=REDACTED -- SQLNESS REPLACE (peers.*) REDACTED +-- SQLNESS REPLACE RoundRobinBatch\(\d+\) RoundRobinBatch(REDACTED) +-- SQLNESS REPLACE (RoundRobinBatch\(REDACTED\),\sinput_partitions=\d+)\s+\| $1| EXPLAIN SELECT v FROM sc_t WHERE v != ALL(SELECT v FROM sc_s) ORDER BY v; +---------------+----------------------------------------------------------------------------------------------------+ @@ -135,11 +140,12 @@ EXPLAIN SELECT v FROM sc_t WHERE v != ALL(SELECT v FROM sc_s) ORDER BY v; | | CoalescePartitionsExec | | | FilterExec: NOT mark@1, projection=[v@0] | | | NestedLoopJoinExec: join_type=LeftMark, filter=(v@0 != v@1) IS NOT DISTINCT FROM false | -| | CoalescePartitionsExec | -| | ProjectionExec: expr=[v@1 as v] | -| | MergeScanExec: REDACTED -| | MergeScanExec: REDACTED -| | MergeScanExec: REDACTED +| | ProjectionExec: expr=[v@1 as v] | +| | MergeScanExec: REDACTED +| | RepartitionExec: partitioning=RoundRobinBatch(REDACTED), input_partitions=1| +| | MergeScanExec: REDACTED +| | RepartitionExec: partitioning=RoundRobinBatch(REDACTED), input_partitions=1| +| | MergeScanExec: REDACTED | | | +---------------+----------------------------------------------------------------------------------------------------+ diff --git a/tests/cases/standalone/optimizer/rewrite_set_comparison.sql b/tests/cases/standalone/optimizer/rewrite_set_comparison.sql index e622d670cd..d808ac7da5 100644 --- a/tests/cases/standalone/optimizer/rewrite_set_comparison.sql +++ b/tests/cases/standalone/optimizer/rewrite_set_comparison.sql @@ -27,12 +27,16 @@ ADMIN FLUSH_TABLE('sc_s'); -- SQLNESS REPLACE region=\d+\(\d+,\s+\d+\) region=REDACTED -- SQLNESS REPLACE (peers.*) REDACTED +-- SQLNESS REPLACE RoundRobinBatch\(\d+\) RoundRobinBatch(REDACTED) +-- SQLNESS REPLACE (RoundRobinBatch\(REDACTED\),\sinput_partitions=\d+)\s+\| $1| EXPLAIN SELECT v FROM sc_t WHERE v > ANY(SELECT v FROM sc_s) ORDER BY v; SELECT v FROM sc_t WHERE v > ANY(SELECT v FROM sc_s) ORDER BY v; -- SQLNESS REPLACE region=\d+\(\d+,\s+\d+\) region=REDACTED -- SQLNESS REPLACE (peers.*) REDACTED +-- SQLNESS REPLACE RoundRobinBatch\(\d+\) RoundRobinBatch(REDACTED) +-- SQLNESS REPLACE (RoundRobinBatch\(REDACTED\),\sinput_partitions=\d+)\s+\| $1| EXPLAIN SELECT v FROM sc_t WHERE v != ALL(SELECT v FROM sc_s) ORDER BY v; SELECT v FROM sc_t WHERE v != ALL(SELECT v FROM sc_s) ORDER BY v; diff --git a/tests/cases/standalone/tql-explain-analyze/analyze.result b/tests/cases/standalone/tql-explain-analyze/analyze.result index 1753d59da9..203debf90b 100644 --- a/tests/cases/standalone/tql-explain-analyze/analyze.result +++ b/tests/cases/standalone/tql-explain-analyze/analyze.result @@ -288,6 +288,7 @@ TQL ANALYZE sum(test2); |_|_|_AggregateExec: mode=FinalPartitioned, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[sum(test2.greptime_value)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_AggregateExec: mode=Partial, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[sum(test2.greptime_value)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[__sum_state(test2.greptime_value)] REDACTED diff --git a/tests/cases/standalone/tql-explain-analyze/tsid_column.result b/tests/cases/standalone/tql-explain-analyze/tsid_column.result index 4a7a875060..8d27a55ab3 100644 --- a/tests/cases/standalone/tql-explain-analyze/tsid_column.result +++ b/tests/cases/standalone/tql-explain-analyze/tsid_column.result @@ -102,10 +102,10 @@ TQL ANALYZE (0, 10, '5s') sum(irate(tsid_metric[1h])) / scalar(count(count(tsid | 0_| 0_|_ProjectionExec: expr=[ts@1 as ts, sum(prom_irate(ts_range,val))@2 / scalar(count(count(tsid_metric.val)))@0 as lhs.sum(prom_irate(ts_range,val)) / rhs.scalar(count(count(tsid_metric.val)))] REDACTED |_|_|_REDACTED |_|_|_ScalarCalculateExec: tags=[] REDACTED -|_|_|_CoalescePartitionsExec REDACTED -|_|_|_MergeScanExec: REDACTED |_|_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_SortPreservingMergeExec: [ts@0 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[ts@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED @@ -153,10 +153,10 @@ TQL ANALYZE (0, 10, '5s') sum(irate(tsid_metric[1h])) / scalar(count(sum(tsid_m | 0_| 0_|_ProjectionExec: expr=[ts@1 as ts, sum(prom_irate(ts_range,val))@2 / scalar(count(sum(tsid_metric.val)))@0 as lhs.sum(prom_irate(ts_range,val)) / rhs.scalar(count(sum(tsid_metric.val)))] REDACTED |_|_|_REDACTED |_|_|_ScalarCalculateExec: tags=[] REDACTED -|_|_|_CoalescePartitionsExec REDACTED -|_|_|_MergeScanExec: REDACTED |_|_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_SortPreservingMergeExec: [ts@0 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[ts@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED diff --git a/tests/compatibility/cases/legacy_json2_non_append_table/verify.result b/tests/compatibility/cases/legacy_json2_non_append_table/verify.result index 2cc3c230e2..078eefb94f 100644 --- a/tests/compatibility/cases/legacy_json2_non_append_table/verify.result +++ b/tests/compatibility/cases/legacy_json2_non_append_table/verify.result @@ -50,6 +50,22 @@ INSERT INTO t_legacy_json2_non_append_table (ts, j) VALUES Affected Rows: 1 +ADMIN FLUSH_TABLE('t_legacy_json2_non_append_table'); + ++-------------------------------------------------------+ +| ADMIN flush_table('t_legacy_json2_non_append_table') | ++-------------------------------------------------------+ +| 0 | ++-------------------------------------------------------+ + +ADMIN compact_table('t_legacy_json2_non_append_table'); + ++---------------------------------------------------------+ +| ADMIN compact_table('t_legacy_json2_non_append_table') | ++---------------------------------------------------------+ +| 0 | ++---------------------------------------------------------+ + SELECT ts, j.a AS a, j.nested.s AS nested_s FROM t_legacy_json2_non_append_table ORDER BY ts; diff --git a/tests/compatibility/cases/legacy_json2_non_append_table/verify.sql b/tests/compatibility/cases/legacy_json2_non_append_table/verify.sql index 2dbc7fc729..1cdc0015da 100644 --- a/tests/compatibility/cases/legacy_json2_non_append_table/verify.sql +++ b/tests/compatibility/cases/legacy_json2_non_append_table/verify.sql @@ -11,6 +11,10 @@ SHOW CREATE TABLE t_legacy_json2_non_append_table; INSERT INTO t_legacy_json2_non_append_table (ts, j) VALUES ('2026-07-08 00:02:00+0000', '{"a": 3, "nested": {"s": "current-3"}}'); +ADMIN FLUSH_TABLE('t_legacy_json2_non_append_table'); + +ADMIN compact_table('t_legacy_json2_non_append_table'); + SELECT ts, j.a AS a, j.nested.s AS nested_s FROM t_legacy_json2_non_append_table ORDER BY ts;