diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index d7a91cadd4..d63d52a218 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -781,10 +781,9 @@ jobs: kubectl -n kafka-cluster apply \ -f .github/actions/setup-greptimedb-cluster/kafka-wal-helper.yaml - kubectl -n kafka-cluster wait \ - --for=condition=Ready \ - pod -l app=kafka-wal-helper \ - --timeout=120s + kubectl rollout status deployment/kafka-wal-helper \ + --timeout=120s \ + -n kafka-cluster - name: Print etcd info shell: bash run: kubectl get all --show-labels -n etcd-cluster diff --git a/Cargo.lock b/Cargo.lock index 74617c08cb..3d0781a903 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -221,7 +221,7 @@ checksum = "d301b3b94cb4b2f23d7917810addbbaff90738e0ca2be692bd027e70d7e0330c" name = "api" version = "1.3.0-alpha.1" dependencies = [ - "arrow-schema 58.3.0", + "arrow-schema 59.2.0", "common-base", "common-decimal", "common-error", @@ -323,23 +323,23 @@ dependencies = [ [[package]] name = "arrow" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "378530e55cd479eda3c14eb345310799717e6f76d0c332041e8487022166b471" +checksum = "61d285d16bce7d0be61912f7928342b673067b6b7d7ef6cc179258ba7de1fecf" dependencies = [ - "arrow-arith 58.3.0", - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-cast 58.3.0", - "arrow-csv 58.3.0", - "arrow-data 58.3.0", - "arrow-ipc 58.3.0", - "arrow-json 58.3.0", - "arrow-ord 58.3.0", - "arrow-row 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", - "arrow-string 58.3.0", + "arrow-arith 59.2.0", + "arrow-array 59.2.0", + "arrow-buffer 59.2.0", + "arrow-cast 59.2.0", + "arrow-csv 59.2.0", + "arrow-data 59.2.0", + "arrow-ipc 59.2.0", + "arrow-json 59.2.0", + "arrow-ord 59.2.0", + "arrow-row 59.2.0", + "arrow-schema 59.2.0", + "arrow-select 59.2.0", + "arrow-string 59.2.0", ] [[package]] @@ -358,14 +358,14 @@ dependencies = [ [[package]] name = "arrow-arith" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a0ab212d2c1886e802f51c5212d78ebbcbb0bec980fff9dadc1eb8d45cd0b738" +checksum = "757ef1836251e88222542a7da2623bc1c9cb9e20afefa6db2c41e79991cd91d4" dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-schema 58.3.0", + "arrow-array 59.2.0", + "arrow-buffer 59.2.0", + "arrow-data 59.2.0", + "arrow-schema 59.2.0", "chrono", "num-traits", ] @@ -388,14 +388,14 @@ dependencies = [ [[package]] name = "arrow-array" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cfd33d3e92f207444098c75b42de99d329562be0cf686b307b097cc52b4e999e" +checksum = "bc9a4a4b2b5ecd0e04df03471661cb61f28bed3c7fd50994715129b01b2edb97" dependencies = [ "ahash 0.8.12", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-schema 58.3.0", + "arrow-buffer 59.2.0", + "arrow-data 59.2.0", + "arrow-schema 59.2.0", "chrono", "chrono-tz", "half", @@ -418,13 +418,13 @@ dependencies = [ [[package]] name = "arrow-buffer" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c6cd424c2693bcdbc150d843dc9d4d137dd2de4782ce6df491ad11a3a0416c0" +checksum = "c12b576ef18c1deb80925a248b25ad84f419198d791b8e293fc6aaa60441fe90" dependencies = [ "bytes", "half", - "num-bigint 0.4.6", + "num-bigint 0.5.1", "num-traits", ] @@ -450,18 +450,18 @@ dependencies = [ [[package]] name = "arrow-cast" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4c5aefb56a2c02e9e2b30746241058b85f8983f0fcff2ba0c6d09006e1cded7f" +checksum = "68338a9096a5dc9bc11927c58c43a8526d96bf6abd2012ef6c0c9f505991cc79" dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-ord 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", + "arrow-array 59.2.0", + "arrow-buffer 59.2.0", + "arrow-data 59.2.0", + "arrow-ord 59.2.0", + "arrow-schema 59.2.0", + "arrow-select 59.2.0", "atoi", - "base64 0.22.1", + "base64 0.23.1", "chrono", "comfy-table", "half", @@ -487,13 +487,13 @@ dependencies = [ [[package]] name = "arrow-csv" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e94e8cf7e517657a52b91ea1263acf38c4ca62a84655d72458a3359b12ab97de" +checksum = "25011b52b346407d497ef0030e12b45e4f2d0cc279efc09c4f3d09106db30e36" dependencies = [ - "arrow-array 58.3.0", - "arrow-cast 58.3.0", - "arrow-schema 58.3.0", + "arrow-array 59.2.0", + "arrow-cast 59.2.0", + "arrow-schema 59.2.0", "chrono", "csv", "csv-core", @@ -514,12 +514,12 @@ dependencies = [ [[package]] name = "arrow-data" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3c88210023a2bfee1896af366309a3028fc3bcbd6515fa29a7990ee1baa08ee0" +checksum = "723fe4aeed7604e00b9883a465af4ff0a0e6c44c03e41a68c3d1cbc403e0e44d" dependencies = [ - "arrow-buffer 58.3.0", - "arrow-schema 58.3.0", + "arrow-buffer 59.2.0", + "arrow-schema 59.2.0", "half", "num-integer", "num-traits", @@ -527,16 +527,17 @@ dependencies = [ [[package]] name = "arrow-flight" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "28abfe8bf9f124e5fc83b334af4fa58f8d0323ad25312ccb2d1da50178415704" +checksum = "2bebfacc9d71f0728f6774164e4d4254b5e504d2b46812d0512d8290ec119a64" dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-cast 58.3.0", - "arrow-ipc 58.3.0", - "arrow-schema 58.3.0", - "base64 0.22.1", + "arrow-array 59.2.0", + "arrow-buffer 59.2.0", + "arrow-cast 59.2.0", + "arrow-data 59.2.0", + "arrow-ipc 59.2.0", + "arrow-schema 59.2.0", + "base64 0.23.1", "bytes", "futures", "prost 0.14.1", @@ -562,17 +563,17 @@ dependencies = [ [[package]] name = "arrow-ipc" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "238438f0834483703d88896db6fe5a7138b2230debc31b34c0336c2996e3c64f" +checksum = "149437b14371f5b9ec60f5ddc751483ae99d7a7072653c0075e5e469156eea7b" dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", + "arrow-array 59.2.0", + "arrow-buffer 59.2.0", + "arrow-data 59.2.0", + "arrow-schema 59.2.0", + "arrow-select 59.2.0", "flatbuffers", - "lz4_flex 0.13.1", + "lz4_flex 0.14.0", "zstd", ] @@ -589,7 +590,7 @@ dependencies = [ "arrow-schema 56.2.0", "chrono", "half", - "indexmap 2.13.0", + "indexmap 2.14.2", "lexical-core", "memchr", "num", @@ -600,19 +601,19 @@ dependencies = [ [[package]] name = "arrow-json" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "205ca2119e6d679d5c133c6f30e68f027738d95ed948cf77677ea69c7800036b" +checksum = "f18b9123ccfec418a663f821c9a034af339711678c11ffe00d3ec07da5ff9f7e" dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-cast 58.3.0", - "arrow-ord 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", + "arrow-array 59.2.0", + "arrow-buffer 59.2.0", + "arrow-cast 59.2.0", + "arrow-ord 59.2.0", + "arrow-schema 59.2.0", + "arrow-select 59.2.0", "chrono", "half", - "indexmap 2.13.0", + "indexmap 2.14.2", "itoa", "lexical-core", "memchr", @@ -638,25 +639,24 @@ dependencies = [ [[package]] name = "arrow-ord" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1bffd8fd2579286a5d63bac898159873e5094a79009940bcb42bbfce4f19f1d0" +checksum = "e6c08dff0686cf23ca4f562803f191ccbeb726dbae6309cd4b4aaf65e0f2c979" dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", + "arrow-array 59.2.0", + "arrow-buffer 59.2.0", + "arrow-data 59.2.0", + "arrow-schema 59.2.0", + "arrow-select 59.2.0", ] [[package]] name = "arrow-pg" -version = "0.14.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d7acc179d3edb91fed930ea6aa6def3f73458228079857b4547096040a1cbde8" +version = "0.15.0" +source = "git+https://github.com/GreptimeTeam/datafusion-postgres.git?rev=3c77e6c32b8db80635a0d2f4b318a36b31170bc3#3c77e6c32b8db80635a0d2f4b318a36b31170bc3" dependencies = [ - "arrow 58.3.0", - "arrow-schema 58.3.0", + "arrow 59.2.0", + "arrow-schema 59.2.0", "bytes", "chrono", "datafusion", @@ -682,14 +682,14 @@ dependencies = [ [[package]] name = "arrow-row" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bab5994731204603c73ba69267616c50f80780774c6bb0476f1f830625115e0c" +checksum = "bbec439386df71ad570e6758a946111322b9e9dc8db83b5527321f0b4c9119c2" dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-schema 58.3.0", + "arrow-array 59.2.0", + "arrow-buffer 59.2.0", + "arrow-data 59.2.0", + "arrow-schema 59.2.0", "half", ] @@ -701,9 +701,9 @@ checksum = "b3aa9e59c611ebc291c28582077ef25c97f1975383f1479b12f3b9ffee2ffabe" [[package]] name = "arrow-schema" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f633dbfdf39c039ada1bf9e34c694816eb71fbb7dc78f613993b7245e078a1ed" +checksum = "e6fed2ca0d1eade57e811cbe73b98ad50cc08a1183e13b2d2aa43a7df593f40e" dependencies = [ "serde", "serde_core", @@ -726,15 +726,15 @@ dependencies = [ [[package]] name = "arrow-select" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8cd065c54172ac787cf3f2f8d4107e0d3fdc26edba76fdf4f4cc170258942222" +checksum = "466b19cf75130b891dc1b23a84b343c714c62c64c9c62e365c76aa0ff90a53fb" dependencies = [ "ahash 0.8.12", - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-schema 58.3.0", + "arrow-array 59.2.0", + "arrow-buffer 59.2.0", + "arrow-data 59.2.0", + "arrow-schema 59.2.0", "num-traits", ] @@ -757,15 +757,15 @@ dependencies = [ [[package]] name = "arrow-string" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "29dd7cda3ab9692f43a2e4acc444d760cc17b12bb6d8232ddf64e9bab7c06b42" +checksum = "c838a25bb3691e919e0f617616ac51a4ff8517a952e29ca133cf0c22b2ce65b1" dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", + "arrow-array 59.2.0", + "arrow-buffer 59.2.0", + "arrow-data 59.2.0", + "arrow-schema 59.2.0", + "arrow-select 59.2.0", "memchr", "num-traits", "regex", @@ -1290,7 +1290,7 @@ dependencies = [ "quote", "regex", "rustc-hash 2.1.1", - "shlex", + "shlex 1.3.0", "syn 2.0.117", ] @@ -1661,8 +1661,8 @@ name = "catalog" version = "1.3.0-alpha.1" dependencies = [ "api", - "arrow 58.3.0", - "arrow-schema 58.3.0", + "arrow 59.2.0", + "arrow-schema 59.2.0", "async-stream", "async-trait", "bytes", @@ -1731,13 +1731,14 @@ dependencies = [ [[package]] name = "cc" -version = "1.2.27" +version = "1.4.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d487aa071b5f64da6f19a3e848e3578944b726ee5a4854b82172f02aa876bfdc" +checksum = "005ec2760ca554fae18df7a11195552ec576cd665632a881bc011d5bb2fd4d80" dependencies = [ + "find-msvc-tools", "jobserver", "libc", - "shlex", + "shlex 2.0.1", ] [[package]] @@ -1781,9 +1782,9 @@ dependencies = [ [[package]] name = "cfg-if" -version = "1.0.1" +version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9555578bc9e57714c812a1f84e4fc5b4d21fcb063490c624de019f7464c91268" +checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" [[package]] name = "cfg_aliases" @@ -1803,7 +1804,7 @@ version = "0.13.10" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7fe45e18904af7af10e4312df7c97251e98af98c70f42f1f2587aecfcbee56bf" dependencies = [ - "indexmap 2.13.0", + "indexmap 2.14.2", "lazy_static", "num-traits", "regex", @@ -1868,9 +1869,9 @@ dependencies = [ [[package]] name = "chrono" -version = "0.4.44" +version = "0.4.45" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c673075a2e0e5f4a1dde27ce9dee1ea4558c7ffe648f576438a20ca1d2acc4b0" +checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" dependencies = [ "iana-time-zone", "js-sys", @@ -2351,8 +2352,8 @@ dependencies = [ name = "common-datasource" version = "1.3.0-alpha.1" dependencies = [ - "arrow 58.3.0", - "arrow-schema 58.3.0", + "arrow 59.2.0", + "arrow-schema 59.2.0", "async-compression", "async-trait", "bytes", @@ -2463,9 +2464,9 @@ dependencies = [ "api", "approx 0.5.1", "arc-swap", - "arrow 58.3.0", - "arrow-cast 58.3.0", - "arrow-schema 58.3.0", + "arrow 59.2.0", + "arrow-cast 59.2.0", + "arrow-schema 59.2.0", "async-trait", "bincode", "catalog", @@ -2866,7 +2867,7 @@ dependencies = [ name = "common-sql" version = "1.3.0-alpha.1" dependencies = [ - "arrow-schema 58.3.0", + "arrow-schema 59.2.0", "common-base", "common-decimal", "common-error", @@ -2943,7 +2944,7 @@ dependencies = [ name = "common-time" version = "1.3.0-alpha.1" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "chrono", "chrono-tz", "common-error", @@ -3552,7 +3553,7 @@ checksum = "7a9bc1a22964ff6a355fbec24cf68266a0ed28f8b84c0864c386474ea3d0e479" dependencies = [ "cc", "codespan-reporting 0.13.1", - "indexmap 2.13.0", + "indexmap 2.14.2", "proc-macro2", "quote", "scratch", @@ -3567,7 +3568,7 @@ checksum = "b1f29a879d35f7906e3c9b77d7a1005a6a0787d330c09dfe4ffb5f617728cb44" dependencies = [ "clap", "codespan-reporting 0.13.1", - "indexmap 2.13.0", + "indexmap 2.14.2", "proc-macro2", "quote", "syn 2.0.117", @@ -3585,7 +3586,7 @@ version = "1.0.190" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d187e019e7b05a1f3e69a8396b70800ee867aa9fc2ab972761173ccee03742df" dependencies = [ - "indexmap 2.13.0", + "indexmap 2.14.2", "proc-macro2", "quote", "syn 2.0.117", @@ -3747,9 +3748,9 @@ checksum = "04d2cd9c18b9f454ed67da600630b021a8a80bf33f8c95896ab33aaf1c26b728" [[package]] name = "dashmap" -version = "6.1.0" +version = "6.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5041cc499144891f3790297212f32a74fb938e5136a14943f338ef9e0ae276cf" +checksum = "e6361d5c062261c78a176addb82d4c821ae42bed6089de0e12603cd25de2059c" dependencies = [ "cfg-if", "crossbeam-utils", @@ -3767,13 +3768,12 @@ checksum = "a4ae5f15dda3c708c0ade84bfee31ccab44a3da4f88015ed22f63732abe300c8" [[package]] name = "datafusion" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", - "arrow-schema 58.3.0", + "arrow 59.2.0", + "arrow-schema 59.2.0", "async-trait", - "bytes", "bzip2", "chrono", "datafusion-catalog", @@ -3803,14 +3803,13 @@ dependencies = [ "datafusion-sql", "flate2", "futures", - "itertools 0.14.0", + "indexmap 2.14.2", + "itertools 0.15.0", "liblzma", "log", "object_store", "parking_lot 0.12.4", "parquet", - "rand 0.9.4", - "regex", "sqlparser", "tempfile", "tokio", @@ -3821,10 +3820,10 @@ dependencies = [ [[package]] name = "datafusion-catalog" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "async-trait", "dashmap", "datafusion-common", @@ -3836,7 +3835,7 @@ dependencies = [ "datafusion-physical-plan", "datafusion-session", "futures", - "itertools 0.14.0", + "itertools 0.15.0", "log", "object_store", "parking_lot 0.12.4", @@ -3845,10 +3844,10 @@ dependencies = [ [[package]] name = "datafusion-catalog-listing" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "async-trait", "datafusion-catalog", "datafusion-common", @@ -3860,39 +3859,42 @@ dependencies = [ "datafusion-physical-expr-common", "datafusion-physical-plan", "futures", - "itertools 0.14.0", + "itertools 0.15.0", "log", "object_store", + "percent-encoding", ] [[package]] name = "datafusion-common" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "ahash 0.8.12", - "arrow 58.3.0", - "arrow-ipc 58.3.0", + "arrow 59.2.0", + "arrow-ipc 59.2.0", + "arrow-schema 59.2.0", "chrono", + "foldhash 0.2.0", "half", - "hashbrown 0.16.1", - "indexmap 2.13.0", - "itertools 0.14.0", + "hashbrown 0.17.1", + "indexmap 2.14.2", + "itertools 0.15.0", "libc", "log", + "num-traits", "object_store", "parquet", - "paste", "recursive", "sqlparser", "tokio", + "uuid", "web-time", ] [[package]] name = "datafusion-common-runtime" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ "futures", "log", @@ -3901,10 +3903,10 @@ dependencies = [ [[package]] name = "datafusion-datasource" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "async-compression", "async-trait", "bytes", @@ -3918,14 +3920,16 @@ dependencies = [ "datafusion-physical-expr-adapter", "datafusion-physical-expr-common", "datafusion-physical-plan", + "datafusion-proto-models", "datafusion-session", "flate2", "futures", "glob", - "itertools 0.14.0", + "itertools 0.15.0", "liblzma", "log", "object_store", + "parking_lot 0.12.4", "rand 0.9.4", "tokio", "tokio-util", @@ -3935,11 +3939,11 @@ dependencies = [ [[package]] name = "datafusion-datasource-arrow" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", - "arrow-ipc 58.3.0", + "arrow 59.2.0", + "arrow-ipc 59.2.0", "async-trait", "bytes", "datafusion-common", @@ -3949,19 +3953,20 @@ dependencies = [ "datafusion-expr", "datafusion-physical-expr-common", "datafusion-physical-plan", + "datafusion-proto-models", "datafusion-session", "futures", - "itertools 0.14.0", + "itertools 0.15.0", "object_store", "tokio", ] [[package]] name = "datafusion-datasource-csv" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "async-trait", "bytes", "datafusion-common", @@ -3971,6 +3976,7 @@ dependencies = [ "datafusion-expr", "datafusion-physical-expr-common", "datafusion-physical-plan", + "datafusion-proto-models", "datafusion-session", "futures", "object_store", @@ -3980,10 +3986,10 @@ dependencies = [ [[package]] name = "datafusion-datasource-json" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "async-trait", "bytes", "datafusion-common", @@ -3993,20 +3999,21 @@ dependencies = [ "datafusion-expr", "datafusion-physical-expr-common", "datafusion-physical-plan", + "datafusion-proto-models", "datafusion-session", "futures", "object_store", - "serde_json", "tokio", "tokio-stream", ] [[package]] name = "datafusion-datasource-parquet" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", + "arrow-schema 59.2.0", "async-trait", "bytes", "datafusion-common", @@ -4014,15 +4021,17 @@ dependencies = [ "datafusion-datasource", "datafusion-execution", "datafusion-expr", + "datafusion-functions", "datafusion-functions-aggregate-common", "datafusion-physical-expr", "datafusion-physical-expr-adapter", "datafusion-physical-expr-common", "datafusion-physical-plan", + "datafusion-proto-models", "datafusion-pruning", "datafusion-session", "futures", - "itertools 0.14.0", + "itertools 0.15.0", "log", "object_store", "parking_lot 0.12.4", @@ -4032,18 +4041,18 @@ dependencies = [ [[package]] name = "datafusion-doc" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" [[package]] name = "datafusion-execution" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", - "arrow-buffer 58.3.0", + "arrow 59.2.0", + "arrow-buffer 59.2.0", "async-trait", - "chrono", + "bytes", "dashmap", "datafusion-common", "datafusion-expr", @@ -4052,17 +4061,21 @@ dependencies = [ "log", "object_store", "parking_lot 0.12.4", + "pin-project-lite", "rand 0.9.4", "tempfile", + "tokio", + "tokio-util", "url", ] [[package]] name = "datafusion-expr" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", + "arrow-schema 59.2.0", "async-trait", "chrono", "datafusion-common", @@ -4071,9 +4084,10 @@ dependencies = [ "datafusion-functions-aggregate-common", "datafusion-functions-window-common", "datafusion-physical-expr-common", - "indexmap 2.13.0", - "itertools 0.14.0", - "paste", + "datafusion-proto-common", + "datafusion-proto-models", + "indexmap 2.14.2", + "itertools 0.15.0", "recursive", "serde_json", "sqlparser", @@ -4081,24 +4095,23 @@ dependencies = [ [[package]] name = "datafusion-expr-common" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "datafusion-common", - "indexmap 2.13.0", - "itertools 0.14.0", - "paste", + "indexmap 2.14.2", + "itertools 0.15.0", ] [[package]] name = "datafusion-functions" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", - "arrow-buffer 58.3.0", - "base64 0.22.1", + "arrow 59.2.0", + "arrow-buffer 59.2.0", + "base64 0.23.1", "blake2", "blake3", "chrono", @@ -4109,26 +4122,25 @@ dependencies = [ "datafusion-expr", "datafusion-expr-common", "datafusion-macros", + "datafusion-physical-expr-common", "hex", - "itertools 0.14.0", + "itertools 0.15.0", "log", - "md-5 0.10.6", + "md-5 0.11.0", "memchr", "num-traits", "rand 0.9.4", "regex", - "sha2 0.10.9", - "unicode-segmentation", + "sha2 0.11.0", "uuid", ] [[package]] name = "datafusion-functions-aggregate" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "ahash 0.8.12", - "arrow 58.3.0", + "arrow 59.2.0", "datafusion-common", "datafusion-doc", "datafusion-execution", @@ -4138,18 +4150,17 @@ dependencies = [ "datafusion-physical-expr", "datafusion-physical-expr-common", "half", + "hashbrown 0.17.1", "log", "num-traits", - "paste", ] [[package]] name = "datafusion-functions-aggregate-common" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "ahash 0.8.12", - "arrow 58.3.0", + "arrow 59.2.0", "datafusion-common", "datafusion-expr-common", "datafusion-physical-expr-common", @@ -4157,11 +4168,11 @@ dependencies = [ [[package]] name = "datafusion-functions-nested" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", - "arrow-ord 58.3.0", + "arrow 59.2.0", + "arrow-ord 59.2.0", "datafusion-common", "datafusion-doc", "datafusion-execution", @@ -4172,34 +4183,34 @@ dependencies = [ "datafusion-functions-aggregate-common", "datafusion-macros", "datafusion-physical-expr-common", - "hashbrown 0.16.1", - "itertools 0.14.0", + "hashbrown 0.17.1", + "itertools 0.15.0", "itoa", "log", - "paste", + "memchr", ] [[package]] name = "datafusion-functions-table" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "async-trait", "datafusion-catalog", "datafusion-common", "datafusion-expr", + "datafusion-physical-expr", "datafusion-physical-plan", "parking_lot 0.12.4", - "paste", ] [[package]] name = "datafusion-functions-window" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "datafusion-common", "datafusion-doc", "datafusion-expr", @@ -4208,13 +4219,12 @@ dependencies = [ "datafusion-physical-expr", "datafusion-physical-expr-common", "log", - "paste", ] [[package]] name = "datafusion-functions-window-common" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ "datafusion-common", "datafusion-physical-expr-common", @@ -4222,27 +4232,27 @@ dependencies = [ [[package]] name = "datafusion-macros" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ "datafusion-doc", "quote", - "syn 2.0.117", + "syn 3.0.5", ] [[package]] name = "datafusion-optimizer" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "chrono", "datafusion-common", "datafusion-expr", "datafusion-expr-common", "datafusion-physical-expr", - "indexmap 2.13.0", - "itertools 0.14.0", + "indexmap 2.14.2", + "itertools 0.15.0", "log", "recursive", "regex", @@ -4251,24 +4261,23 @@ dependencies = [ [[package]] name = "datafusion-orc" -version = "0.8.0" -source = "git+https://github.com/datafusion-contrib/datafusion-orc.git?rev=6c07fa282dc8d62db2aa4ded06ab55485efc811a#6c07fa282dc8d62db2aa4ded06ab55485efc811a" +version = "0.10.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "637382f27aa0c7d6f30ca18a3e37d46bdf427aa83e1b0366083dc38e362dae99" dependencies = [ "async-trait", "bytes", "datafusion", "futures", "futures-util", - "object_store", "orc-rust", "tokio", ] [[package]] name = "datafusion-pg-catalog" -version = "0.17.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f6b82fb8bd291b718d4226eaba85f78c73f0abdedb256a5dc91d9d9e6b0e0dab" +version = "0.18.3" +source = "git+https://github.com/GreptimeTeam/datafusion-postgres.git?rev=3c77e6c32b8db80635a0d2f4b318a36b31170bc3#3c77e6c32b8db80635a0d2f4b318a36b31170bc3" dependencies = [ "arrow-pg", "async-trait", @@ -4281,22 +4290,21 @@ dependencies = [ [[package]] name = "datafusion-physical-expr" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "ahash 0.8.12", - "arrow 58.3.0", + "arrow 59.2.0", "datafusion-common", "datafusion-expr", "datafusion-expr-common", "datafusion-functions-aggregate-common", "datafusion-physical-expr-common", + "datafusion-proto-models", "half", - "hashbrown 0.16.1", - "indexmap 2.13.0", - "itertools 0.14.0", + "hashbrown 0.17.1", + "indexmap 2.14.2", + "itertools 0.15.0", "parking_lot 0.12.4", - "paste", "petgraph 0.8.3", "recursive", "tokio", @@ -4304,40 +4312,41 @@ dependencies = [ [[package]] name = "datafusion-physical-expr-adapter" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "datafusion-common", "datafusion-expr", "datafusion-functions", "datafusion-physical-expr", "datafusion-physical-expr-common", - "itertools 0.14.0", + "itertools 0.15.0", ] [[package]] name = "datafusion-physical-expr-common" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "ahash 0.8.12", - "arrow 58.3.0", + "arrow 59.2.0", "chrono", "datafusion-common", "datafusion-expr-common", - "hashbrown 0.16.1", - "indexmap 2.13.0", - "itertools 0.14.0", + "datafusion-proto-models", + "hashbrown 0.17.1", + "indexmap 2.14.2", + "itertools 0.15.0", "parking_lot 0.12.4", + "pin-project", ] [[package]] name = "datafusion-physical-optimizer" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "datafusion-common", "datafusion-execution", "datafusion-expr", @@ -4346,20 +4355,23 @@ dependencies = [ "datafusion-physical-expr-common", "datafusion-physical-plan", "datafusion-pruning", - "itertools 0.14.0", + "datafusion-session", + "itertools 0.15.0", "recursive", ] [[package]] name = "datafusion-physical-plan" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "ahash 0.8.12", - "arrow 58.3.0", - "arrow-ord 58.3.0", - "arrow-schema 58.3.0", + "arrow 59.2.0", + "arrow-data 59.2.0", + "arrow-ipc 59.2.0", + "arrow-ord 59.2.0", + "arrow-schema 59.2.0", "async-trait", + "bytes", "datafusion-common", "datafusion-common-runtime", "datafusion-execution", @@ -4369,25 +4381,28 @@ dependencies = [ "datafusion-functions-window-common", "datafusion-physical-expr", "datafusion-physical-expr-common", + "datafusion-proto-common", + "datafusion-proto-models", "futures", "half", - "hashbrown 0.16.1", - "indexmap 2.13.0", - "itertools 0.14.0", + "hashbrown 0.17.1", + "indexmap 2.14.2", + "itertools 0.15.0", "log", "num-traits", "parking_lot 0.12.4", "pin-project-lite", + "serde", + "serde_json", "tokio", ] [[package]] name = "datafusion-proto" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", - "chrono", + "arrow 59.2.0", "datafusion-catalog", "datafusion-catalog-listing", "datafusion-common", @@ -4403,42 +4418,52 @@ dependencies = [ "datafusion-physical-expr-common", "datafusion-physical-plan", "datafusion-proto-common", + "datafusion-proto-models", "object_store", "prost 0.14.1", - "rand 0.9.4", ] [[package]] name = "datafusion-proto-common" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "datafusion-common", "prost 0.14.1", ] [[package]] -name = "datafusion-pruning" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +name = "datafusion-proto-models" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "datafusion-common", + "datafusion-proto-common", + "prost 0.14.1", +] + +[[package]] +name = "datafusion-pruning" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" +dependencies = [ + "arrow 59.2.0", "datafusion-common", "datafusion-datasource", "datafusion-expr-common", "datafusion-physical-expr", "datafusion-physical-expr-common", "datafusion-physical-plan", - "itertools 0.14.0", "log", ] [[package]] name = "datafusion-session" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ + "arrow-schema 59.2.0", "async-trait", "datafusion-common", "datafusion-execution", @@ -4449,37 +4474,38 @@ dependencies = [ [[package]] name = "datafusion-sql" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "bigdecimal 0.4.8", "chrono", "datafusion-common", "datafusion-expr", "datafusion-functions-nested", - "indexmap 2.13.0", + "indexmap 2.14.2", "log", "recursive", "regex", "sqlparser", + "stacker", ] [[package]] name = "datafusion-substrait" -version = "53.1.0" -source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=bb531e754b3b36d7a4d3222f35ebc88204dafd2b#bb531e754b3b36d7a4d3222f35ebc88204dafd2b" +version = "55.0.0" +source = "git+https://github.com/GreptimeTeam/datafusion.git?rev=b66dca3260c8e04f314c8036cdfbb08f1a564d95#b66dca3260c8e04f314c8036cdfbb08f1a564d95" dependencies = [ "async-recursion", "async-trait", "chrono", "datafusion", "half", - "itertools 0.14.0", + "itertools 0.15.0", "object_store", "pbjson-types", "prost 0.14.1", - "substrait 0.62.2", + "substrait 0.63.0", "tokio", "url", ] @@ -4563,9 +4589,9 @@ checksum = "c286de4e81ea2590afc24d754e0f83810c566f50a1388fa75ebd57928c0d9745" name = "datatypes" version = "1.3.0-alpha.1" dependencies = [ - "arrow 58.3.0", - "arrow-array 58.3.0", - "arrow-schema 58.3.0", + "arrow 59.2.0", + "arrow-array 59.2.0", + "arrow-schema 59.2.0", "common-base", "common-decimal", "common-error", @@ -5359,6 +5385,12 @@ dependencies = [ "windows-sys 0.59.0", ] +[[package]] +name = "find-msvc-tools" +version = "0.1.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3e0f1c7c3a72c66fd80abe965175f7523475c0489a87d3ff9d6e8c87d87a9d2d" + [[package]] name = "findshlibs" version = "0.10.2" @@ -5459,9 +5491,9 @@ name = "flow" version = "1.3.0-alpha.1" dependencies = [ "api", - "arrow 58.3.0", + "arrow 59.2.0", "arrow-flight", - "arrow-schema 58.3.0", + "arrow-schema 59.2.0", "async-recursion", "async-trait", "bytes", @@ -6155,7 +6187,7 @@ dependencies = [ "futures-sink", "futures-util", "http 0.2.12", - "indexmap 2.13.0", + "indexmap 2.14.2", "slab", "tokio", "tokio-util", @@ -6174,7 +6206,7 @@ dependencies = [ "futures-core", "futures-sink", "http 1.5.0", - "indexmap 2.13.0", + "indexmap 2.14.2", "slab", "tokio", "tokio-util", @@ -6987,12 +7019,12 @@ dependencies = [ [[package]] name = "indexmap" -version = "2.13.0" +version = "2.14.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7714e70437a7dc3ac8eb7e6f8df75fd8eb422675fc7678aff7364301092b1017" +checksum = "cc4e190f5d26ca7051642629da2c52fc03bde85a03197c99408dcd291734c855" dependencies = [ "equivalent", - "hashbrown 0.16.1", + "hashbrown 0.17.1", "serde", "serde_core", ] @@ -7023,7 +7055,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "232929e1d75fe899576a3d5c7416ad0d88dbfbb3c3d6aa00873a7408a50ddb88" dependencies = [ "ahash 0.8.12", - "indexmap 2.13.0", + "indexmap 2.14.2", "is-terminal", "itoa", "log", @@ -7046,7 +7078,7 @@ dependencies = [ "crossbeam-utils", "dashmap", "env_logger", - "indexmap 2.13.0", + "indexmap 2.14.2", "itoa", "log", "num-format", @@ -7130,12 +7162,6 @@ dependencies = [ "cfg-if", ] -[[package]] -name = "integer-encoding" -version = "3.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8bb03732005da905c88227371639bf1ad885cc712789c011c31c5fb3ab3ccf02" - [[package]] name = "integer-encoding" version = "4.0.2" @@ -7268,6 +7294,15 @@ dependencies = [ "either", ] +[[package]] +name = "itertools" +version = "0.15.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8b4baf93f58d4425749ca49a51c50ebab072c5df6994d08fed93541c331481dc" +dependencies = [ + "either", +] + [[package]] name = "itoa" version = "1.0.15" @@ -7545,7 +7580,7 @@ version = "0.4.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4ee7893dab2e44ae5f9d0173f26ff4aa327c10b01b06a72b52dd9405b628640d" dependencies = [ - "indexmap 2.13.0", + "indexmap 2.14.2", ] [[package]] @@ -8157,7 +8192,7 @@ dependencies = [ "cactus", "cfgrammar", "filetime", - "indexmap 2.13.0", + "indexmap 2.14.2", "lazy_static", "lrtable", "num-traits", @@ -8240,6 +8275,12 @@ name = "lz4_flex" version = "0.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7ef0d4ed8669f8f8826eb00dc878084aa8f253506c4fd5e8f58f5bce72ddb97e" + +[[package]] +name = "lz4_flex" +version = "0.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ecbdfe44b1bd960b68170b417450a628c43f7cf56bb3c5317e61cb230ee7f226" dependencies = [ "twox-hash", ] @@ -8367,9 +8408,9 @@ dependencies = [ [[package]] name = "memchr" -version = "2.8.0" +version = "2.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8ca58f447f06ed17d5fc4043ce1b10dd205e060fb3ce5b979b8ed8e59ff3f79" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" [[package]] name = "memcomparable" @@ -8672,7 +8713,7 @@ version = "1.3.0-alpha.1" dependencies = [ "api", "aquamarine", - "arrow-schema 58.3.0", + "arrow-schema 59.2.0", "async-channel 1.9.0", "async-stream", "async-trait", @@ -9614,9 +9655,9 @@ dependencies = [ [[package]] name = "once_cell" -version = "1.21.3" +version = "1.21.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "42f5e15c9953c5e4ccceeb2e7382a716482c34515315f7b03532b8b4e8393d2d" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" [[package]] name = "once_cell_polyfill" @@ -10053,8 +10094,8 @@ version = "1.3.0-alpha.1" dependencies = [ "ahash 0.8.12", "api", - "arrow 58.3.0", - "arrow-ipc 58.3.0", + "arrow 59.2.0", + "arrow-ipc 59.2.0", "async-stream", "async-trait", "axum 0.8.4", @@ -10123,11 +10164,11 @@ dependencies = [ [[package]] name = "orc-rust" -version = "0.8.0" +version = "0.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32b9867c4e7343218682ba11aceae1310c80735ec2beaf6124a0f8f848dad197" +checksum = "e54c163f843c4f0fcfca65b9b73ca6f93c37c8e86736d5a1e578c6e843eed0a0" dependencies = [ - "arrow 58.3.0", + "arrow 59.2.0", "async-trait", "bytemuck", "bytes", @@ -10350,18 +10391,18 @@ dependencies = [ [[package]] name = "parquet" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5dafa7d01085b62a47dd0c1829550a0a36710ea9c4fe358a05a85477cec8a908" +checksum = "7065842956a20c2a536924ce8e4d9955f7422451511b9eb7500d7bfe5077e59c" dependencies = [ "ahash 0.8.12", - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-ipc 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", - "base64 0.22.1", + "arrow-array 59.2.0", + "arrow-buffer 59.2.0", + "arrow-data 59.2.0", + "arrow-ipc 59.2.0", + "arrow-schema 59.2.0", + "arrow-select 59.2.0", + "base64 0.23.1", "brotli", "bytes", "chrono", @@ -10369,19 +10410,17 @@ dependencies = [ "futures", "half", "hashbrown 0.17.1", - "lz4_flex 0.13.1", - "num-bigint 0.4.6", + "lz4_flex 0.14.0", + "num-bigint 0.5.1", "num-integer", "num-traits", "object_store", "parquet-variant", "parquet-variant-compute", "parquet-variant-json", - "paste", "seq-macro", "simdutf8", "snap", - "thrift", "tokio", "twox-hash", "zstd", @@ -10389,15 +10428,15 @@ dependencies = [ [[package]] name = "parquet-variant" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "74c8db065291f088a2aad8ab831853eae1871c0d311c8d0b83bbc3b7e735d0fc" +checksum = "3f7e5fff3ed0c07514a7fb8bee3f2ea5a53f36939410ecac4a466620213539a8" dependencies = [ - "arrow 58.3.0", - "arrow-schema 58.3.0", + "arrow 59.2.0", + "arrow-schema 59.2.0", "chrono", "half", - "indexmap 2.13.0", + "indexmap 2.14.2", "num-traits", "simdutf8", "uuid", @@ -10405,15 +10444,15 @@ dependencies = [ [[package]] name = "parquet-variant-compute" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a530e8d5b5e14efcb39c9a6ec55432ad11f6afb7dc4455a79be0dc615fe3cc31" +checksum = "ba4d3de89dab8d1aaaf601ae8d71bd07ea88cfca9efc1df5815b982c30f631e1" dependencies = [ - "arrow 58.3.0", - "arrow-schema 58.3.0", + "arrow 59.2.0", + "arrow-schema 59.2.0", "chrono", "half", - "indexmap 2.13.0", + "indexmap 2.14.2", "parquet-variant", "parquet-variant-json", "serde_json", @@ -10422,12 +10461,12 @@ dependencies = [ [[package]] name = "parquet-variant-json" -version = "58.3.0" +version = "59.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "00ed89908289f67caa2ca078f9ff9aacd6229a313ec92b12bf4f48f613dc2b97" +checksum = "fb19dfe1bd24c17addd761ba4f7000f615e2fa12525871c7baa835dbb3d7f147" dependencies = [ - "arrow-schema 58.3.0", - "base64 0.22.1", + "arrow-schema 59.2.0", + "base64 0.23.1", "chrono", "parquet-variant", "serde_json", @@ -10641,7 +10680,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b4c5cc86750666a3ed20bdaf5ca2a0344f9c67674cae0515bec2da16fbaa47db" dependencies = [ "fixedbitset 0.4.2", - "indexmap 2.13.0", + "indexmap 2.14.2", ] [[package]] @@ -10651,7 +10690,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3672b37090dbd86368a4145bc067582552b29c27377cad4e0a306c97f9bd7772" dependencies = [ "fixedbitset 0.5.7", - "indexmap 2.13.0", + "indexmap 2.14.2", ] [[package]] @@ -10662,7 +10701,7 @@ checksum = "8701b58ea97060d5e5b155d383a69952a60943f0e6dfe30b04c287beb0b27455" dependencies = [ "fixedbitset 0.5.7", "hashbrown 0.15.4", - "indexmap 2.13.0", + "indexmap 2.14.2", "serde", ] @@ -10850,8 +10889,8 @@ version = "1.3.0-alpha.1" dependencies = [ "ahash 0.8.12", "api", - "arrow 58.3.0", - "arrow-schema 58.3.0", + "arrow 59.2.0", + "arrow-schema 59.2.0", "async-trait", "catalog", "chrono", @@ -10974,7 +11013,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3d77244ce2d584cd84f6a15f86195b8c9b2a0dfbfd817c09e0464244091a58ed" dependencies = [ "base64 0.22.1", - "indexmap 2.13.0", + "indexmap 2.14.2", "quick-xml 0.37.5", "serde", "time", @@ -11455,7 +11494,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "22505a5c94da8e3b7c2996394d1c933236c4d743e81a410bcca4e6989fc066a4" dependencies = [ "bytes", - "heck 0.4.1", + "heck 0.5.0", "itertools 0.12.1", "log", "multimap", @@ -11475,7 +11514,7 @@ version = "0.14.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ac6c3320f9abac597dcbc668774ef006702672474aad53c6d596b62e487b40b1" dependencies = [ - "heck 0.4.1", + "heck 0.5.0", "itertools 0.14.0", "log", "multimap", @@ -11837,8 +11876,8 @@ dependencies = [ "ahash 0.8.12", "api", "arc-swap", - "arrow 58.3.0", - "arrow-schema 58.3.0", + "arrow 59.2.0", + "arrow-schema 59.2.0", "async-recursion", "async-stream", "async-trait", @@ -11865,6 +11904,7 @@ dependencies = [ "datafusion-expr-common", "datafusion-functions", "datafusion-optimizer", + "datafusion-pg-catalog", "datafusion-physical-expr", "datafusion-proto", "datafusion-sql", @@ -12319,7 +12359,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4c11639076bf147be211b90e47790db89f4c22b6c8a9ca6e960833869da67166" dependencies = [ "aho-corasick", - "indexmap 2.13.0", + "indexmap 2.14.2", "itertools 0.13.0", "nohash", "regex", @@ -12779,7 +12819,7 @@ dependencies = [ "crc32c", "flate2", "futures", - "integer-encoding 4.0.2", + "integer-encoding", "lz4", "parking_lot 0.12.4", "rand 0.10.1", @@ -13487,7 +13527,7 @@ dependencies = [ "chrono", "hex", "indexmap 1.9.3", - "indexmap 2.13.0", + "indexmap 2.14.2", "schemars 0.9.0", "schemars 1.2.1", "serde_core", @@ -13514,7 +13554,7 @@ version = "0.9.34+deprecated" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "6a8b1a1a2ebf674015cc02edccce75287f1a0130d394307b36743c2f5d504b47" dependencies = [ - "indexmap 2.13.0", + "indexmap 2.14.2", "itoa", "ryu", "serde", @@ -13527,7 +13567,7 @@ version = "0.10.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7b4db627b98b36d4203a7b458cf3573730f2bb591b28871d916dfa9efabfd41f" dependencies = [ - "indexmap 2.13.0", + "indexmap 2.14.2", "itoa", "ryu", "serde", @@ -13540,11 +13580,11 @@ version = "1.3.0-alpha.1" dependencies = [ "ahash 0.8.12", "api", - "arrow 58.3.0", + "arrow 59.2.0", "arrow-flight", - "arrow-ipc 58.3.0", + "arrow-ipc 59.2.0", "arrow-pg", - "arrow-schema 58.3.0", + "arrow-schema 59.2.0", "async-trait", "auth", "axum 0.8.4", @@ -13596,7 +13636,7 @@ dependencies = [ "humantime", "humantime-serde", "hyper 1.6.0", - "indexmap 2.13.0", + "indexmap 2.14.2", "influxdb_line_protocol", "itertools 0.14.0", "json5", @@ -13780,6 +13820,12 @@ version = "1.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0fda2ff0d084019ba4d7c6f371c95d8fd75ce3524c3cb8fb653a3023f6323e64" +[[package]] +name = "shlex" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" + [[package]] name = "signal-hook-registry" version = "1.4.5" @@ -13955,7 +14001,7 @@ version = "0.8.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1961e2ef424c1424204d3a5d6975f934f56b6d50ff5732382d84ebf460e147f7" dependencies = [ - "heck 0.4.1", + "heck 0.5.0", "proc-macro2", "quote", "syn 2.0.117", @@ -14044,7 +14090,7 @@ name = "sql" version = "1.3.0-alpha.1" dependencies = [ "api", - "arrow-buffer 58.3.0", + "arrow-buffer 59.2.0", "chrono", "common-base", "common-catalog", @@ -14134,13 +14180,11 @@ dependencies = [ [[package]] name = "sqlparser" -version = "0.61.0" -source = "git+https://github.com/GreptimeTeam/sqlparser-rs.git?rev=2aefa08a8d69c96eec2d6d6703598a009bba6e4c#2aefa08a8d69c96eec2d6d6703598a009bba6e4c" +version = "0.62.0" +source = "git+https://github.com/GreptimeTeam/sqlparser-rs.git?rev=9e9019bb1c7040ed956f654e39378dd42ab17884#9e9019bb1c7040ed956f654e39378dd42ab17884" dependencies = [ - "lazy_static", "log", "recursive", - "regex", "serde", "sqlparser_derive 0.5.0", ] @@ -14159,7 +14203,7 @@ dependencies = [ [[package]] name = "sqlparser_derive" version = "0.5.0" -source = "git+https://github.com/GreptimeTeam/sqlparser-rs.git?rev=2aefa08a8d69c96eec2d6d6703598a009bba6e4c#2aefa08a8d69c96eec2d6d6703598a009bba6e4c" +source = "git+https://github.com/GreptimeTeam/sqlparser-rs.git?rev=9e9019bb1c7040ed956f654e39378dd42ab17884#9e9019bb1c7040ed956f654e39378dd42ab17884" dependencies = [ "proc-macro2", "quote", @@ -14198,7 +14242,7 @@ dependencies = [ "futures-util", "hashbrown 0.15.4", "hashlink", - "indexmap 2.13.0", + "indexmap 2.14.2", "log", "memchr", "once_cell", @@ -14371,15 +14415,15 @@ checksum = "a8f112729512f8e442d81f95a8a7ddf2b7c6b8a1a6f509a95864142b30cab2d3" [[package]] name = "stacker" -version = "0.1.21" +version = "0.1.25" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cddb07e32ddb770749da91081d8d0ac3a16f1a569a18b20348cd371f5dead06b" +checksum = "707f49d46706bacf8a2b00d51dace3f9de527c13eec3778f570c411f89e69967" dependencies = [ "cc", "cfg-if", "libc", "psm", - "windows-sys 0.52.0", + "windows-sys 0.61.2", ] [[package]] @@ -14601,11 +14645,11 @@ dependencies = [ [[package]] name = "substrait" -version = "0.62.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "62fc4b483a129b9772ccb9c3f7945a472112fdd9140da87f8a4e7f1d44e045d0" +version = "0.63.0" +source = "git+https://github.com/GreptimeTeam/substrait-rs.git?rev=91ec978b0649417ad3da8390e7a515baec723b1b#91ec978b0649417ad3da8390e7a515baec723b1b" dependencies = [ "heck 0.5.0", + "indexmap 2.14.2", "pbjson", "pbjson-build", "pbjson-types", @@ -14702,6 +14746,17 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "syn" +version = "3.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "12df2e0110f65b775f769bb17ef989067a1d931b2eb822bd4346631eeada89f9" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + [[package]] name = "sync_wrapper" version = "0.1.2" @@ -15242,17 +15297,6 @@ dependencies = [ "cfg-if", ] -[[package]] -name = "thrift" -version = "0.17.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e54bc85fc7faa8bc175c4bab5b92ba8d9a3ce893d0e9f42cc455c8ab16a9e09" -dependencies = [ - "byteorder", - "integer-encoding 3.0.4", - "ordered-float 2.10.1", -] - [[package]] name = "tikv-jemalloc-ctl" version = "0.7.0" @@ -15550,7 +15594,7 @@ version = "0.8.23" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "dc1beb996b9d83529a9e75c17a1686767d148d70663143c7854d8b4a09ced362" dependencies = [ - "indexmap 2.13.0", + "indexmap 2.14.2", "serde", "serde_spanned", "toml_datetime", @@ -15572,7 +15616,7 @@ version = "0.19.15" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1b5bb770da30e5cbfde35a2d7b9b8a2c4b8ef89548a7a6aeab5c9a576e3e7421" dependencies = [ - "indexmap 2.13.0", + "indexmap 2.14.2", "toml_datetime", "winnow 0.5.40", ] @@ -15583,7 +15627,7 @@ version = "0.22.27" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "41fe8c660ae4257887cf66394862d21dbca4a6ddd26f04a3560410406a2f819a" dependencies = [ - "indexmap 2.13.0", + "indexmap 2.14.2", "serde", "serde_spanned", "toml_datetime", @@ -15769,7 +15813,7 @@ dependencies = [ "futures-core", "futures-util", "hdrhistogram", - "indexmap 2.13.0", + "indexmap 2.14.2", "pin-project-lite", "slab", "sync_wrapper 1.0.2", @@ -16234,13 +16278,13 @@ checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" [[package]] name = "uuid" -version = "1.21.0" +version = "1.26.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b672338555252d43fd2240c714dc444b8c6fb0a5c5335e65a07bba7742735ddb" +checksum = "b5772d71c9be8a8a6ac2117d949c5b224c1b72241bb611d9a3012edcf8af7812" dependencies = [ "getrandom 0.4.1", "js-sys", - "rand 0.9.4", + "rand 0.10.1", "serde_core", "wasm-bindgen", ] @@ -16379,7 +16423,7 @@ dependencies = [ "hostname 0.4.1", "iana-time-zone", "idna", - "indexmap 2.13.0", + "indexmap 2.14.2", "indoc", "influxdb-line-protocol", "ipcrypt-rs", @@ -16606,7 +16650,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "bb0e353e6a2fbdc176932bbaab493762eb1255a7900fe0fea1a2f96c296cc909" dependencies = [ "anyhow", - "indexmap 2.13.0", + "indexmap 2.14.2", "wasm-encoder", "wasmparser", ] @@ -16645,7 +16689,7 @@ checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe" dependencies = [ "bitflags 2.12.1", "hashbrown 0.15.4", - "indexmap 2.13.0", + "indexmap 2.14.2", "semver", ] @@ -17284,7 +17328,7 @@ checksum = "b7c566e0f4b284dd6561c786d9cb0142da491f46a9fbed79ea69cdad5db17f21" dependencies = [ "anyhow", "heck 0.5.0", - "indexmap 2.13.0", + "indexmap 2.14.2", "prettyplease", "syn 2.0.117", "wasm-metadata", @@ -17315,7 +17359,7 @@ checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2" dependencies = [ "anyhow", "bitflags 2.12.1", - "indexmap 2.13.0", + "indexmap 2.14.2", "log", "serde", "serde_derive", @@ -17334,7 +17378,7 @@ checksum = "ecc8ac4bc1dc3381b7f59c34f00b67e18f910c2c0f50015669dde7def656a736" dependencies = [ "anyhow", "id-arena", - "indexmap 2.13.0", + "indexmap 2.14.2", "log", "semver", "serde", diff --git a/Cargo.toml b/Cargo.toml index b39064706c..a54a008180 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -100,13 +100,13 @@ rust.unexpected_cfgs = { level = "warn", check-cfg = ['cfg(tokio_unstable)'] } # See for more detaiils: https://github.com/rust-lang/cargo/issues/11329 ahash = { version = "0.8", features = ["compile-time-rng"] } aquamarine = "0.6" -arrow = { version = "58.3", features = ["prettyprint"] } -arrow-array = { version = "58.3", default-features = false, features = ["chrono-tz"] } -arrow-buffer = "58.3" -arrow-cast = "58.3" -arrow-flight = "58.3" -arrow-ipc = { version = "58.3", default-features = false, features = ["lz4", "zstd"] } -arrow-schema = { version = "58.3", features = ["serde"] } +arrow = { version = "=59.2.0", features = ["prettyprint"] } +arrow-array = { version = "=59.2.0", default-features = false, features = ["chrono-tz"] } +arrow-buffer = "=59.2.0" +arrow-cast = "=59.2.0" +arrow-flight = "=59.2.0" +arrow-ipc = { version = "=59.2.0", default-features = false, features = ["lz4", "zstd"] } +arrow-schema = { version = "=59.2.0", features = ["serde"] } async-stream = "0.3" async-trait = "0.1" # Remember to update axum-extra, axum-macros when updating axum @@ -128,22 +128,22 @@ const_format = "0.2" criterion = "0.7" crossbeam-utils = "0.8" dashmap = "6.1" -datafusion = "=53.1.0" -datafusion-common = "=53.1.0" -datafusion-datasource = "=53.1.0" -datafusion-expr = "=53.1.0" -datafusion-expr-common = "=53.1.0" -datafusion-functions = "=53.1.0" -datafusion-functions-aggregate-common = "=53.1.0" -datafusion-functions-window-common = "=53.1.0" -datafusion-optimizer = "=53.1.0" -datafusion-orc = { git = "https://github.com/datafusion-contrib/datafusion-orc.git", rev = "6c07fa282dc8d62db2aa4ded06ab55485efc811a" } -datafusion-pg-catalog = "0.17.3" -datafusion-physical-expr = "=53.1.0" -datafusion-physical-plan = "=53.1.0" -datafusion-proto = "=53.1.0" -datafusion-sql = "=53.1.0" -datafusion-substrait = "=53.1.0" +datafusion = "=55.0.0" +datafusion-common = "=55.0.0" +datafusion-datasource = "=55.0.0" +datafusion-expr = "=55.0.0" +datafusion-expr-common = "=55.0.0" +datafusion-functions = "=55.0.0" +datafusion-functions-aggregate-common = "=55.0.0" +datafusion-functions-window-common = "=55.0.0" +datafusion-optimizer = "=55.0.0" +datafusion-orc = "0.10.0" +datafusion-pg-catalog = "0.18.3" +datafusion-physical-expr = "=55.0.0" +datafusion-physical-plan = "=55.0.0" +datafusion-proto = "=55.0.0" +datafusion-sql = "=55.0.0" +datafusion-substrait = "=55.0.0" datafusion_object_store = { package = "object_store", version = "0.13.2" } deadpool = "0.12" deadpool-postgres = "0.14" @@ -195,10 +195,10 @@ otel-arrow-rust = { git = "https://github.com/GreptimeTeam/otel-arrow", rev = "5 "server", ] } parking_lot = "0.12" -parquet = { version = "58.3", default-features = false, features = ["arrow", "async", "object_store"] } -parquet-variant = "58.3" -parquet-variant-compute = "58.3" -parquet-variant-json = "58.3" +parquet = { version = "=59.2.0", default-features = false, features = ["arrow", "async", "object_store"] } +parquet-variant = "=59.2.0" +parquet-variant-compute = "=59.2.0" +parquet-variant-json = "=59.2.0" paste = "1.0" pin-project = "1.0" pretty_assertions = "1.4.0" @@ -239,7 +239,7 @@ simd-json = "0.15" similar-asserts = "1.6.0" smallvec = { version = "1", features = ["serde"] } snafu = "0.8" -sqlparser = { version = "0.61.0", default-features = false, features = ["std", "visitor", "serde"] } +sqlparser = { version = "0.62.0", default-features = false, features = ["std", "visitor", "serde"] } sqlx = { version = "0.8", default-features = false, features = [ "any", "macros", @@ -353,22 +353,25 @@ git = "https://github.com/GreptimeTeam/greptime-meter.git" rev = "5618e779cf2bb4755b499c630fba4c35e91898cb" [patch.crates-io] -datafusion = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -datafusion-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -datafusion-datasource = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -datafusion-expr = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -datafusion-expr-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -datafusion-functions = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -datafusion-functions-aggregate-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -datafusion-functions-window-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -datafusion-optimizer = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -datafusion-physical-expr = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -datafusion-physical-expr-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -datafusion-physical-plan = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -datafusion-proto = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -datafusion-sql = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -datafusion-substrait = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "bb531e754b3b36d7a4d3222f35ebc88204dafd2b" } -sqlparser = { git = "https://github.com/GreptimeTeam/sqlparser-rs.git", rev = "2aefa08a8d69c96eec2d6d6703598a009bba6e4c" } # on branch v0.61.x +substrait = { git = "https://github.com/GreptimeTeam/substrait-rs.git", rev = "91ec978b0649417ad3da8390e7a515baec723b1b" } +datafusion = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-datasource = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-expr = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-expr-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-functions = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-functions-aggregate-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-functions-window-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-optimizer = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-physical-expr = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-physical-expr-common = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-physical-plan = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-proto = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-sql = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-substrait = { git = "https://github.com/GreptimeTeam/datafusion.git", rev = "b66dca3260c8e04f314c8036cdfbb08f1a564d95" } +datafusion-pg-catalog = { git = "https://github.com/GreptimeTeam/datafusion-postgres.git", rev = "3c77e6c32b8db80635a0d2f4b318a36b31170bc3" } +arrow-pg = { git = "https://github.com/GreptimeTeam/datafusion-postgres.git", rev = "3c77e6c32b8db80635a0d2f4b318a36b31170bc3" } +sqlparser = { git = "https://github.com/GreptimeTeam/sqlparser-rs.git", rev = "9e9019bb1c7040ed956f654e39378dd42ab17884" } # Temporary: use the GreptimeTeam fork of tikv-jemalloc-sys embedding # jemalloc 5.3.1 + backport of 54f22c83 ("Initialize TSD tcache before diff --git a/src/catalog/src/information_extension.rs b/src/catalog/src/information_extension.rs index c41c816c6d..9d55f9f7a1 100644 --- a/src/catalog/src/information_extension.rs +++ b/src/catalog/src/information_extension.rs @@ -27,8 +27,10 @@ use common_query::request::QueryRequest; use common_recordbatch::adapter::{AsyncRecordBatchStreamAdapter, DfRecordBatchStreamAdapter}; use common_recordbatch::util::{ChainedRecordBatchStream, LimitedRecordBatchStream}; use common_recordbatch::{DfSendableRecordBatchStream, SendableRecordBatchStream}; +use datafusion::common::Result; +use datafusion::common::tree_node::TreeNodeRecursion; use datafusion::execution::TaskContext; -use datafusion::physical_expr::{EquivalenceProperties, Partitioning}; +use datafusion::physical_expr::{EquivalenceProperties, Partitioning, PhysicalExpr}; use datafusion::physical_plan::execution_plan::{Boundedness, EmissionType}; use datafusion::physical_plan::{DisplayAs, DisplayFormatType, ExecutionPlan, PlanProperties}; use datatypes::arrow::datatypes::SchemaRef as ArrowSchemaRef; @@ -157,10 +159,6 @@ impl ExecutionPlan for DistributedInspectExec { "DistributedInspectExec" } - fn as_any(&self) -> &dyn std::any::Any { - self - } - fn schema(&self) -> ArrowSchemaRef { self.arrow_schema.clone() } @@ -173,6 +171,13 @@ impl ExecutionPlan for DistributedInspectExec { vec![] } + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> Result, + ) -> Result { + Ok(TreeNodeRecursion::Continue) + } + fn with_new_children( self: Arc, children: Vec>, diff --git a/src/catalog/src/system_schema/information_schema/ssts.rs b/src/catalog/src/system_schema/information_schema/ssts.rs index da5fcd2eed..e364bddfd3 100644 --- a/src/catalog/src/system_schema/information_schema/ssts.rs +++ b/src/catalog/src/system_schema/information_schema/ssts.rs @@ -336,7 +336,7 @@ mod tests { }; let plan = table.scan_to_plan(request).unwrap().unwrap(); - assert!(plan.as_any().is::()); + assert!(plan.as_ref().is::()); assert_eq!(1, plan.schema().fields().len()); } } diff --git a/src/catalog/src/table_source/dummy_catalog.rs b/src/catalog/src/table_source/dummy_catalog.rs index 20637c3a3a..455cc8c34f 100644 --- a/src/catalog/src/table_source/dummy_catalog.rs +++ b/src/catalog/src/table_source/dummy_catalog.rs @@ -14,7 +14,6 @@ //! Dummy catalog for region server. -use std::any::Any; use std::fmt; use std::sync::Arc; @@ -64,10 +63,6 @@ impl fmt::Debug for DummyCatalogList { } impl CatalogProviderList for DummyCatalogList { - fn as_any(&self) -> &dyn Any { - self - } - fn register_catalog( &self, _name: String, @@ -98,10 +93,6 @@ struct DummyCatalogProvider { } impl CatalogProvider for DummyCatalogProvider { - fn as_any(&self) -> &dyn Any { - self - } - fn schema_names(&self) -> Vec { vec![] } @@ -135,10 +126,6 @@ struct DummySchemaProvider { #[async_trait] impl SchemaProvider for DummySchemaProvider { - fn as_any(&self) -> &dyn Any { - self - } - fn table_names(&self) -> Vec { vec![] } diff --git a/src/cmd/src/bin/query_perf_fixture/inspect_footer.rs b/src/cmd/src/bin/query_perf_fixture/inspect_footer.rs index d2a2e8006c..19e972db3b 100644 --- a/src/cmd/src/bin/query_perf_fixture/inspect_footer.rs +++ b/src/cmd/src/bin/query_perf_fixture/inspect_footer.rs @@ -14,18 +14,21 @@ use std::collections::BTreeSet; use std::fs; +use std::ops::Range; use std::path::{Path, PathBuf}; use std::sync::Arc; use clap::Args as ClapArgs; use datafusion_object_store::path::Path as StorePath; -use datafusion_object_store::{ObjectMeta, ObjectStore}; -use futures::{StreamExt, TryStreamExt}; +use datafusion_object_store::{ObjectMeta, ObjectStore, ObjectStoreExt}; +use futures::future::BoxFuture; +use futures::{FutureExt, StreamExt, TryFutureExt, TryStreamExt}; use object_store::config::ObjectStoreConfig; use object_store::factory::new_raw_object_store; use object_store::services::Fs; -use parquet::arrow::async_reader::ParquetObjectReader; -use parquet::file::metadata::ParquetMetaDataReader; +use parquet::arrow::async_reader::AsyncFileReader; +use parquet::errors::{ParquetError, Result as ParquetResult}; +use parquet::file::metadata::{ParquetMetaData, ParquetMetaDataReader}; use serde::{Deserialize, Serialize}; /// Same shape as `query_regression_runner::model::DestinationConfig`. The two @@ -100,6 +103,70 @@ struct ListedFile { relative_path: String, } +/// An asynchronous Parquet reader backed directly by an object store. +/// +/// The file size is retained from the listing so footer reads use bounded +/// ranges without an additional stat/head request. +#[derive(Clone, Debug)] +struct ObjectStoreReader { + store: Arc, + path: StorePath, + file_size: u64, +} + +impl ObjectStoreReader { + fn new(store: Arc, path: StorePath, file_size: u64) -> Self { + Self { + store, + path, + file_size, + } + } +} + +fn to_parquet_error(error: datafusion_object_store::Error) -> ParquetError { + ParquetError::External(Box::new(error)) +} + +impl AsyncFileReader for ObjectStoreReader { + fn get_bytes( + &mut self, + range: Range, + ) -> BoxFuture<'_, ParquetResult> { + self.store + .get_range(&self.path, range) + .map_err(to_parquet_error) + .boxed() + } + + fn get_byte_ranges( + &mut self, + ranges: Vec>, + ) -> BoxFuture<'_, ParquetResult>> { + async move { + self.store + .get_ranges(&self.path, &ranges) + .await + .map_err(to_parquet_error) + } + .boxed() + } + + fn get_metadata<'a>( + &'a mut self, + _options: Option<&'a parquet::arrow::arrow_reader::ArrowReaderOptions>, + ) -> BoxFuture<'a, ParquetResult>> { + let file_size = self.file_size; + async move { + let metadata = ParquetMetaDataReader::new() + .load_and_finish(self, file_size) + .await?; + Ok(Arc::new(metadata)) + } + .boxed() + } +} + pub(super) async fn run_inspect_footer( args: InspectFooterArgs, ) -> Result<(), Box> { @@ -217,11 +284,8 @@ async fn inspect_file( file: &ListedFile, column: &str, ) -> Result> { - let mut reader = - ParquetObjectReader::new(store, file.location.clone()).with_file_size(file.size); - let metadata = ParquetMetaDataReader::new() - .load_and_finish(&mut reader, file.size) - .await?; + let mut reader = ObjectStoreReader::new(store, file.location.clone(), file.size); + let metadata = reader.get_metadata(None).await?; let file_metadata = metadata.file_metadata(); let row_groups = metadata.row_groups(); let mut columns = Vec::new(); diff --git a/src/common/datasource/Cargo.toml b/src/common/datasource/Cargo.toml index 8b4053db2f..a601dae1d9 100644 --- a/src/common/datasource/Cargo.toml +++ b/src/common/datasource/Cargo.toml @@ -34,7 +34,7 @@ futures.workspace = true lazy_static.workspace = true object-store.workspace = true object_store_opendal.workspace = true -orc-rust = { version = "0.8", default-features = false, features = ["async"] } +orc-rust = { version = "0.9", default-features = false, features = ["async"] } parquet.workspace = true paste.workspace = true regex.workspace = true diff --git a/src/common/datasource/src/file_format.rs b/src/common/datasource/src/file_format.rs index 61f838899e..7ae60c615e 100644 --- a/src/common/datasource/src/file_format.rs +++ b/src/common/datasource/src/file_format.rs @@ -35,7 +35,7 @@ use datafusion::datasource::file_format::file_compression_type::FileCompressionT use datafusion::datasource::listing::PartitionedFile; use datafusion::datasource::object_store::ObjectStoreUrl; use datafusion::datasource::physical_plan::{ - FileGroup, FileOpenFuture, FileScanConfigBuilder, FileSource, FileStream, + FileGroup, FileOpenFuture, FileScanConfigBuilder, FileSource, FileStreamBuilder, }; use datafusion::error::{DataFusionError, Result as DataFusionResult}; use datafusion::physical_plan::SendableRecordBatchStream; @@ -321,7 +321,11 @@ pub async fn file_to_stream( let store = Arc::new(object_store_opendal::OpendalStore::new(store.clone())); let file_opener = config.file_source().create_file_opener(store, &config, 0)?; - let stream = FileStream::new(&config, 0, file_opener, &ExecutionPlanMetricsSet::new())?; + let stream = FileStreamBuilder::new(&config) + .with_partition(0) + .with_file_opener(file_opener) + .with_metrics(&ExecutionPlanMetricsSet::new()) + .build()?; Ok(Box::pin(stream)) } diff --git a/src/common/datasource/src/file_format/tests.rs b/src/common/datasource/src/file_format/tests.rs index a925f73d48..27b86e1b11 100644 --- a/src/common/datasource/src/file_format/tests.rs +++ b/src/common/datasource/src/file_format/tests.rs @@ -19,7 +19,7 @@ use std::{assert_matches, vec}; use common_test_util::find_workspace_path; use datafusion::assert_batches_eq; use datafusion::datasource::physical_plan::{ - CsvSource, FileScanConfig, FileSource, FileStream, JsonSource, ParquetSource, + CsvSource, FileScanConfig, FileSource, FileStreamBuilder, JsonSource, ParquetSource, }; use datafusion::datasource::source::DataSourceExec; use datafusion::execution::context::TaskContext; @@ -50,16 +50,15 @@ impl Test<'_> { .create_file_opener(store, &self.config, 0) .unwrap(); - let result = FileStream::new( - &self.config, - 0, - file_opener, - &ExecutionPlanMetricsSet::new(), - ) - .unwrap() - .map(|b| b.unwrap()) - .collect::>() - .await; + let result = FileStreamBuilder::new(&self.config) + .with_partition(0) + .with_file_opener(file_opener) + .with_metrics(&ExecutionPlanMetricsSet::new()) + .build() + .unwrap() + .map(|b| b.unwrap()) + .collect::>() + .await; assert_batches_eq!(self.expected, &result); } diff --git a/src/common/datasource/src/test_util.rs b/src/common/datasource/src/test_util.rs index 9dd93a4ee4..4cefda8889 100644 --- a/src/common/datasource/src/test_util.rs +++ b/src/common/datasource/src/test_util.rs @@ -20,7 +20,7 @@ use datafusion::datasource::file_format::file_compression_type::FileCompressionT use datafusion::datasource::listing::PartitionedFile; use datafusion::datasource::object_store::ObjectStoreUrl; use datafusion::datasource::physical_plan::{ - CsvSource, FileGroup, FileScanConfig, FileScanConfigBuilder, FileSource, FileStream, + CsvSource, FileGroup, FileScanConfig, FileScanConfigBuilder, FileSource, FileStreamBuilder, JsonOpener, JsonSource, }; use datafusion::physical_plan::metrics::ExecutionPlanMetricsSet; @@ -110,13 +110,12 @@ pub async fn setup_stream_to_json_test(origin_path: &str, threshold: impl Fn(usi let size = store.read(origin_path).await.unwrap().len(); let config = scan_config(None, origin_path, Arc::new(JsonSource::new(schema))); - let stream = FileStream::new( - &config, - 0, - Arc::new(json_opener), - &ExecutionPlanMetricsSet::new(), - ) - .unwrap(); + let stream = FileStreamBuilder::new(&config) + .with_partition(0) + .with_file_opener(Arc::new(json_opener)) + .with_metrics(&ExecutionPlanMetricsSet::new()) + .build() + .unwrap(); let (tmp_store, dir) = test_tmp_store("test_stream_to_json"); @@ -162,7 +161,12 @@ pub async fn setup_stream_to_csv_test( 0, ) .unwrap(); - let stream = FileStream::new(&config, 0, csv_opener, &ExecutionPlanMetricsSet::new()).unwrap(); + let stream = FileStreamBuilder::new(&config) + .with_partition(0) + .with_file_opener(csv_opener) + .with_metrics(&ExecutionPlanMetricsSet::new()) + .build() + .unwrap(); let (tmp_store, dir) = test_tmp_store("test_stream_to_csv"); diff --git a/src/common/function/src/aggrs/aggr_wrapper.rs b/src/common/function/src/aggrs/aggr_wrapper.rs index c2e8b567ca..70651eee41 100644 --- a/src/common/function/src/aggrs/aggr_wrapper.rs +++ b/src/common/function/src/aggrs/aggr_wrapper.rs @@ -33,15 +33,15 @@ use datafusion::functions_aggregate::count::Count; use datafusion::functions_aggregate::min_max::{Max, Min}; use datafusion::optimizer::AnalyzerRule; use datafusion::optimizer::analyzer::type_coercion::TypeCoercion; -use datafusion::physical_planner::create_aggregate_expr_and_maybe_filter; use datafusion_common::{Column, ScalarValue}; use datafusion_expr::expr::{AggregateFunction, AggregateFunctionParams}; use datafusion_expr::function::StateFieldsArgs; +use datafusion_expr::physical_planning_context::PhysicalPlanningContext; use datafusion_expr::{ Accumulator, Aggregate, AggregateUDF, AggregateUDFImpl, EmitTo, Expr, ExprSchemable, GroupsAccumulator, LogicalPlan, Signature, }; -use datafusion_physical_expr::aggregate::AggregateFunctionExpr; +use datafusion_physical_expr::aggregate::{AggregateFunctionExpr, LoweredAggregateBuilder}; use datatypes::arrow::datatypes::{DataType, Field}; use crate::aggrs::aggr_wrapper::fix_order::FixStateUdafOrderingAnalyzer; @@ -210,12 +210,25 @@ impl StateMergeHelper { lower_aggr_exprs.push(expr); // then create the merge function using the physical expression of the original aggregate function - let (original_phy_expr, _filter, _ordering) = create_aggregate_expr_and_maybe_filter( + let (name, human_display) = match aggr_expr { + Expr::Alias(alias) => (alias.name.clone(), aggr_expr.human_display().to_string()), + Expr::AggregateFunction(_) => ( + aggr_expr.schema_name().to_string(), + aggr_expr.human_display().to_string(), + ), + _ => unreachable!("aggregate expression was validated above"), + }; + let original_phy_expr = LoweredAggregateBuilder::new( aggr_expr, aggr.input.schema(), aggr.input.schema().as_arrow(), &Default::default(), - )?; + &PhysicalPlanningContext::default(), + ) + .with_name(name) + .with_human_display(human_display) + .build()? + .aggregate; let merge_func = MergeWrapper::new( (*aggr_func.func).clone(), @@ -371,9 +384,6 @@ impl AggregateUDFImpl for StateWrapper { Ok(Box::new(StateGroupsAccum::new(inner, state_type)?)) } - fn as_any(&self) -> &dyn std::any::Any { - self - } fn name(&self) -> &str { self.name.as_str() } @@ -443,7 +453,7 @@ impl AggregateUDFImpl for StateWrapper { &self, statistics_args: &datafusion_expr::StatisticsArgs, ) -> Option { - let inner = self.inner().inner().as_any(); + let inner = self.inner().inner(); // only count/min/max need special handling here, for getting result from statistics // the result of count/min/max is also the result of count_state so can return directly let can_use_stat = inner.is::() || inner.is::() || inner.is::(); @@ -572,11 +582,10 @@ impl GroupsAccumulator for StateGroupsAccum { &mut self, values: &[ArrayRef], group_indices: &[usize], - opt_filter: Option<&BooleanArray>, total_num_groups: usize, ) -> datafusion_common::Result<()> { self.inner - .merge_batch(values, group_indices, opt_filter, total_num_groups) + .merge_batch(values, group_indices, total_num_groups) } fn evaluate(&mut self, emit_to: EmitTo) -> datafusion_common::Result { @@ -596,10 +605,6 @@ impl GroupsAccumulator for StateGroupsAccum { self.inner.convert_to_state(values, opt_filter) } - fn supports_convert_to_state(&self) -> bool { - self.inner.supports_convert_to_state() - } - fn size(&self) -> usize { self.inner.size() } @@ -800,10 +805,6 @@ impl AggregateUDFImpl for DeltaMergeWrapper { })) } - fn as_any(&self) -> &dyn std::any::Any { - self - } - fn name(&self) -> &str { &self.name } @@ -973,9 +974,6 @@ impl AggregateUDFImpl for MergeWrapper { Ok(Box::new(MergeAccum::new(inner_accum, &fields))) } - fn as_any(&self) -> &dyn std::any::Any { - self - } fn name(&self) -> &str { self.name.as_str() } diff --git a/src/common/function/src/aggrs/aggr_wrapper/fix_order.rs b/src/common/function/src/aggrs/aggr_wrapper/fix_order.rs index 35aa35b583..480b0b5a49 100644 --- a/src/common/function/src/aggrs/aggr_wrapper/fix_order.rs +++ b/src/common/function/src/aggrs/aggr_wrapper/fix_order.rs @@ -144,7 +144,6 @@ fn rewrite_expr( let Some(old_state_wrapper) = aggregate_function .func .inner() - .as_any() .downcast_ref::() else { return Ok(Transformed::no(Expr::AggregateFunction(aggregate_function))); diff --git a/src/common/function/src/aggrs/aggr_wrapper/tests.rs b/src/common/function/src/aggrs/aggr_wrapper/tests.rs index 3f59139f05..4274a03cd8 100644 --- a/src/common/function/src/aggrs/aggr_wrapper/tests.rs +++ b/src/common/function/src/aggrs/aggr_wrapper/tests.rs @@ -12,7 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::pin::Pin; use std::sync::{Arc, Mutex}; use std::task::{Context, Poll}; @@ -25,6 +24,7 @@ use arrow::record_batch::RecordBatch; use arrow_schema::SchemaRef; use common_telemetry::init_default_ut_logging; use datafusion::catalog::{Session, TableProvider}; +use datafusion::common::tree_node::TreeNodeRecursion; use datafusion::datasource::DefaultTableSource; use datafusion::execution::{RecordBatchStream, SendableRecordBatchStream, TaskContext}; use datafusion::functions_aggregate::average::avg_udaf; @@ -39,16 +39,16 @@ use datafusion::physical_planner::{DefaultPhysicalPlanner, PhysicalPlanner}; use datafusion::prelude::SessionContext; use datafusion_common::arrow::array::AsArray; use datafusion_common::arrow::datatypes::{Float64Type, UInt64Type}; -use datafusion_common::{Column, TableReference}; +use datafusion_common::{Column, Result, TableReference}; use datafusion_expr::expr::{AggregateFunction, NullTreatment}; use datafusion_expr::function::AccumulatorArgs; use datafusion_expr::{ Aggregate, AggregateUDFImpl, ColumnarValue, Expr, LogicalPlan, ScalarFunctionArgs, SortExpr, - TableScan, TypeSignature, lit, + TableScanBuilder, TypeSignature, lit, }; use datafusion_physical_expr::aggregate::AggregateExprBuilder; use datafusion_physical_expr::expressions::{Column as PhysicalColumn, col, lit as physical_lit}; -use datafusion_physical_expr::{EquivalenceProperties, Partitioning}; +use datafusion_physical_expr::{EquivalenceProperties, Partitioning, PhysicalExpr}; use futures::{Stream, StreamExt as _}; use hyperloglogplus::HyperLogLog; use pretty_assertions::assert_eq; @@ -97,10 +97,6 @@ impl ExecutionPlan for MockInputExec { "MockInputExec" } - fn as_any(&self) -> &dyn Any { - self - } - fn properties(&self) -> &Arc { &self.properties } @@ -109,6 +105,13 @@ impl ExecutionPlan for MockInputExec { vec![] } + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> Result, + ) -> Result { + Ok(TreeNodeRecursion::Continue) + } + fn with_new_children( self: Arc, _children: Vec>, @@ -203,10 +206,6 @@ impl Default for DummyTableProvider { #[async_trait::async_trait] impl TableProvider for DummyTableProvider { - fn as_any(&self) -> &dyn std::any::Any { - self - } - fn schema(&self) -> Arc { self.schema.clone() } @@ -237,14 +236,12 @@ fn dummy_table_scan() -> LogicalPlan { let table_provider = Arc::new(DummyTableProvider::default()); let table_source = DefaultTableSource::new(table_provider); LogicalPlan::TableScan( - TableScan::try_new( - TableReference::bare("Number"), - Arc::new(table_source), - None, - vec![], - None, - ) - .unwrap(), + TableScanBuilder::new(TableReference::bare("Number"), Arc::new(table_source)) + .with_projection(None) + .with_filters(vec![]) + .with_fetch(None) + .build() + .unwrap(), ) } @@ -252,14 +249,12 @@ fn dummy_table_scan_with_ts() -> LogicalPlan { let table_provider = Arc::new(DummyTableProvider::with_ts(None)); let table_source = DefaultTableSource::new(table_provider); LogicalPlan::TableScan( - TableScan::try_new( - TableReference::bare("Number"), - Arc::new(table_source), - None, - vec![], - None, - ) - .unwrap(), + TableScanBuilder::new(TableReference::bare("Number"), Arc::new(table_source)) + .with_projection(None) + .with_filters(vec![]) + .with_fetch(None) + .build() + .unwrap(), ) } @@ -381,10 +376,7 @@ async fn test_sum_udaf() { .create_physical_plan(&res.lower_state, &ctx.state()) .await .unwrap(); - let aggr_exec = phy_aggr_state_plan - .as_any() - .downcast_ref::() - .unwrap(); + let aggr_exec = phy_aggr_state_plan.downcast_ref::().unwrap(); let aggr_func_expr = &aggr_exec.aggr_expr()[0]; let mut state_accum = aggr_func_expr.create_accumulator().unwrap(); @@ -414,10 +406,7 @@ async fn test_sum_udaf() { .create_physical_plan(&res.upper_merge, &ctx.state()) .await .unwrap(); - let aggr_exec = phy_aggr_merge_plan - .as_any() - .downcast_ref::() - .unwrap(); + let aggr_exec = phy_aggr_merge_plan.downcast_ref::().unwrap(); let aggr_func_expr = &aggr_exec.aggr_expr()[0]; let mut merge_accum = aggr_func_expr.create_accumulator().unwrap(); @@ -543,10 +532,7 @@ async fn test_avg_udaf() { .create_physical_plan(&coerced_aggr_state_plan, &ctx.state()) .await .unwrap(); - let aggr_exec = phy_aggr_state_plan - .as_any() - .downcast_ref::() - .unwrap(); + let aggr_exec = phy_aggr_state_plan.downcast_ref::().unwrap(); let aggr_func_expr = &aggr_exec.aggr_expr()[0]; let mut state_accum = aggr_func_expr.create_accumulator().unwrap(); @@ -582,10 +568,7 @@ async fn test_avg_udaf() { .create_physical_plan(&res.upper_merge, &ctx.state()) .await .unwrap(); - let aggr_exec = phy_aggr_merge_plan - .as_any() - .downcast_ref::() - .unwrap(); + let aggr_exec = phy_aggr_merge_plan.downcast_ref::().unwrap(); let aggr_func_expr = &aggr_exec.aggr_expr()[0]; let mut merge_accum = aggr_func_expr.create_accumulator().unwrap(); @@ -703,10 +686,7 @@ async fn test_last_value_order_by_udaf() { .create_physical_plan(&fixed_aggr_state_plan, &ctx.state()) .await .unwrap(); - let aggr_exec = phy_aggr_state_plan - .as_any() - .downcast_ref::() - .unwrap(); + let aggr_exec = phy_aggr_state_plan.downcast_ref::().unwrap(); let aggr_func_expr = &aggr_exec.aggr_expr()[0]; let merge_input_fields = vec![Arc::new(Field::new( @@ -796,10 +776,7 @@ async fn test_last_value_order_by_udaf() { .create_physical_plan(&res.upper_merge, &ctx.state()) .await .unwrap(); - let aggr_exec = phy_aggr_merge_plan - .as_any() - .downcast_ref::() - .unwrap(); + let aggr_exec = phy_aggr_merge_plan.downcast_ref::().unwrap(); let aggr_func_expr = &aggr_exec.aggr_expr()[0]; let mut merge_accum = aggr_func_expr.create_accumulator().unwrap(); @@ -900,7 +877,7 @@ fn test_avg_state_groups_accumulator_state_merge_evaluate() { .update_batch(&merged_values, &merged_group_indices, None, 3) .unwrap(); merged_accum - .merge_batch(&source_state, &[1, 2, 0], None, 3) + .merge_batch(&source_state, &[1, 2, 0], 3) .unwrap(); let result = merged_accum.evaluate(EmitTo::All).unwrap(); @@ -1324,14 +1301,12 @@ async fn test_udaf_correct_eval_result() { ); let table_source = DefaultTableSource::new(Arc::new(table_provider)); let logical_plan = LogicalPlan::TableScan( - TableScan::try_new( - test_table_ref.clone(), - Arc::new(table_source), - None, - vec![], - None, - ) - .unwrap(), + TableScanBuilder::new(test_table_ref.clone(), Arc::new(table_source)) + .with_projection(None) + .with_filters(vec![]) + .with_fetch(None) + .build() + .unwrap(), ); let args = case.args; diff --git a/src/common/function/src/aggrs/approximate/uddsketch.rs b/src/common/function/src/aggrs/approximate/uddsketch.rs index f7d1558d13..6af8211615 100644 --- a/src/common/function/src/aggrs/approximate/uddsketch.rs +++ b/src/common/function/src/aggrs/approximate/uddsketch.rs @@ -123,7 +123,6 @@ impl UddSketchState { fn downcast_accumulator_args(args: AccumulatorArgs) -> DfResult<(u32, f64)> { let bucket_size = match args.exprs[0] - .as_any() .downcast_ref::() .map(|lit| lit.value()) { @@ -140,7 +139,6 @@ fn downcast_accumulator_args(args: AccumulatorArgs) -> DfResult<(u32, f64)> { }; let error_rate = match args.exprs[1] - .as_any() .downcast_ref::() .map(|lit| lit.value()) { diff --git a/src/common/function/src/aggrs/count_hash.rs b/src/common/function/src/aggrs/count_hash.rs index bd00aeb012..28ed854638 100644 --- a/src/common/function/src/aggrs/count_hash.rs +++ b/src/common/function/src/aggrs/count_hash.rs @@ -26,7 +26,7 @@ use std::sync::Arc; use ahash::RandomState; use datafusion_common::cast::as_list_array; use datafusion_common::error::Result; -use datafusion_common::hash_utils::create_hashes; +use datafusion_common::hash_utils::create_hashes_with_hasher; use datafusion_common::utils::SingleRowListArrayBuilder; use datafusion_common::{ScalarValue, internal_err, not_impl_err}; use datafusion_expr::function::{AccumulatorArgs, StateFieldsArgs}; @@ -74,10 +74,6 @@ pub struct CountHash { } impl AggregateUDFImpl for CountHash { - fn as_any(&self) -> &dyn std::any::Any { - self - } - fn name(&self) -> &str { "count_hash" } @@ -208,7 +204,7 @@ impl GroupsAccumulator for CountHashGroupAccumulator { let array = &values[0]; self.batch_hashes.clear(); self.batch_hashes.resize(array.len(), 0); - let hashes = create_hashes( + let hashes = create_hashes_with_hasher( &[ArrayRef::clone(array)], &self.random_state, &mut self.batch_hashes, @@ -279,7 +275,6 @@ impl GroupsAccumulator for CountHashGroupAccumulator { &mut self, values: &[ArrayRef], group_indices: &[usize], - _opt_filter: Option<&BooleanArray>, total_num_groups: usize, ) -> Result<()> { assert_eq!( @@ -364,10 +359,6 @@ impl GroupsAccumulator for CountHashGroupAccumulator { Ok(vec![Arc::new(list_array)]) } - fn supports_convert_to_state(&self) -> bool { - true - } - fn size(&self) -> usize { // Base size of the struct let mut size = size_of::(); @@ -423,7 +414,7 @@ impl Accumulator for CountHashAccumulator { self.batch_hashes.clear(); self.batch_hashes.resize(arr.len(), 0); - let hashes = create_hashes( + let hashes = create_hashes_with_hasher( &[ArrayRef::clone(arr)], &self.random_state, &mut self.batch_hashes, @@ -522,6 +513,28 @@ mod tests { Ok(()) } + #[test] + fn test_count_hash_accumulator_typed_null_state_merge() -> Result<()> { + let typed_nulls = Arc::new(Int32Array::from(vec![None, None])) as ArrayRef; + + let mut fresh = create_test_accumulator(); + fresh.update_batch(&[typed_nulls])?; + let fresh_state = fresh.state()?; + assert_eq!(fresh.evaluate()?, ScalarValue::Int64(Some(1))); + + let persisted_state = Arc::new( + SingleRowListArrayBuilder::new(Arc::new(UInt64Array::from(vec![0])) as ArrayRef) + .build_list_array(), + ) as ArrayRef; + let mut restored = create_test_accumulator(); + restored.merge_batch(&[persisted_state])?; + + assert_eq!(restored.evaluate()?, fresh.evaluate()?); + assert_eq!(restored.state()?, fresh_state); + + Ok(()) + } + #[test] fn test_count_hash_accumulator_merge() -> Result<()> { // Accumulator 1 @@ -622,7 +635,7 @@ mod tests { // We will merge acc1's group 0 into acc2's group 0 // and acc1's group 1 into acc2's group 2 let merge_group_indices = vec![0, 2]; - acc2.merge_batch(&state1, &merge_group_indices, None, 3)?; + acc2.merge_batch(&state1, &merge_group_indices, 3)?; let result_array = acc2.evaluate(EmitTo::All)?; let result = result_array.as_any().downcast_ref::().unwrap(); diff --git a/src/common/function/src/helper.rs b/src/common/function/src/helper.rs index 6e643443ec..bbd589d407 100644 --- a/src/common/function/src/helper.rs +++ b/src/common/function/src/helper.rs @@ -23,6 +23,33 @@ use datatypes::types::cast::cast; use datatypes::value::ValueRef; use snafu::{OptionExt, ResultExt}; +/// Integer types accepted by geospatial function signatures. +pub(crate) const INTEGER_TYPES: &[DataType] = &[ + DataType::Int8, + DataType::Int16, + DataType::Int32, + DataType::Int64, + DataType::UInt8, + DataType::UInt16, + DataType::UInt32, + DataType::UInt64, +]; + +/// Legacy primitive numeric signature types; Decimal values are coerced to `Float64`. +pub(crate) const NUMERICS: &[DataType] = &[ + DataType::Int8, + DataType::Int16, + DataType::Int32, + DataType::Int64, + DataType::UInt8, + DataType::UInt16, + DataType::UInt32, + DataType::UInt64, + DataType::Float16, + DataType::Float32, + DataType::Float64, +]; + /// Create a function signature with oneof signatures of interleaving two arguments. pub(crate) fn one_of_sigs2(args1: Vec, args2: Vec) -> Signature { let mut sigs = Vec::with_capacity(args1.len() * args2.len()); diff --git a/src/common/function/src/scalars/anomaly/iqr.rs b/src/common/function/src/scalars/anomaly/iqr.rs index cf25166de2..423adc8373 100644 --- a/src/common/function/src/scalars/anomaly/iqr.rs +++ b/src/common/function/src/scalars/anomaly/iqr.rs @@ -24,7 +24,6 @@ //! When IQR = 0 (constant quartiles), returns 0.0 if value is on the fence, //! or +inf if value is outside. -use std::any::Any; use std::fmt::Debug; use std::ops::Range; use std::sync::Arc; @@ -32,11 +31,11 @@ use std::sync::Arc; use arrow::array::{Array, ArrayRef, Float64Array}; use arrow::datatypes::{DataType, Field, FieldRef}; use datafusion_common::{DataFusionError, Result, ScalarValue}; -use datafusion_expr::type_coercion::aggregates::NUMERICS; use datafusion_expr::{PartitionEvaluator, Signature, Volatility, WindowUDFImpl}; use datafusion_functions_window_common::field::WindowUDFFieldArgs; use datafusion_functions_window_common::partition::PartitionEvaluatorArgs; +use crate::helper::NUMERICS; use crate::scalars::anomaly::utils::{cast_to_f64, collect_window_values, percentile_sorted}; /// Minimum valid samples for IQR (linear-interpolated Q1 != Q3 is possible at n >= 3). @@ -56,10 +55,6 @@ impl AnomalyScoreIqr { } impl WindowUDFImpl for AnomalyScoreIqr { - fn as_any(&self) -> &dyn Any { - self - } - fn name(&self) -> &str { "anomaly_score_iqr" } diff --git a/src/common/function/src/scalars/anomaly/mad.rs b/src/common/function/src/scalars/anomaly/mad.rs index 8bf97df0dc..9f68687947 100644 --- a/src/common/function/src/scalars/anomaly/mad.rs +++ b/src/common/function/src/scalars/anomaly/mad.rs @@ -20,7 +20,6 @@ //! When MAD = 0 (majority-constant window), returns 0.0 if value equals //! median, or +inf otherwise. -use std::any::Any; use std::fmt::Debug; use std::ops::Range; use std::sync::Arc; @@ -28,11 +27,11 @@ use std::sync::Arc; use arrow::array::{Array, ArrayRef, Float64Array}; use arrow::datatypes::{DataType, Field, FieldRef}; use datafusion_common::{DataFusionError, Result, ScalarValue}; -use datafusion_expr::type_coercion::aggregates::NUMERICS; use datafusion_expr::{PartitionEvaluator, Signature, Volatility, WindowUDFImpl}; use datafusion_functions_window_common::field::WindowUDFFieldArgs; use datafusion_functions_window_common::partition::PartitionEvaluatorArgs; +use crate::helper::NUMERICS; use crate::scalars::anomaly::utils::{ anomaly_ratio, cast_to_f64, collect_window_values, median_f64, }; @@ -57,10 +56,6 @@ impl AnomalyScoreMad { } impl WindowUDFImpl for AnomalyScoreMad { - fn as_any(&self) -> &dyn Any { - self - } - fn name(&self) -> &str { "anomaly_score_mad" } diff --git a/src/common/function/src/scalars/anomaly/zscore.rs b/src/common/function/src/scalars/anomaly/zscore.rs index f5852ce9c3..610bab8cab 100644 --- a/src/common/function/src/scalars/anomaly/zscore.rs +++ b/src/common/function/src/scalars/anomaly/zscore.rs @@ -19,7 +19,6 @@ //! When stddev = 0 (constant window), returns 0.0 if value equals mean, //! or +inf otherwise. -use std::any::Any; use std::fmt::Debug; use std::ops::Range; use std::sync::Arc; @@ -27,11 +26,11 @@ use std::sync::Arc; use arrow::array::{Array, ArrayRef, Float64Array}; use arrow::datatypes::{DataType, Field, FieldRef}; use datafusion_common::{DataFusionError, Result, ScalarValue}; -use datafusion_expr::type_coercion::aggregates::NUMERICS; use datafusion_expr::{PartitionEvaluator, Signature, Volatility, WindowUDFImpl}; use datafusion_functions_window_common::field::WindowUDFFieldArgs; use datafusion_functions_window_common::partition::PartitionEvaluatorArgs; +use crate::helper::NUMERICS; use crate::scalars::anomaly::utils::{anomaly_ratio, cast_to_f64, collect_window_values}; /// Minimum valid samples for zscore (stddev requires n >= 2). @@ -51,10 +50,6 @@ impl AnomalyScoreZscore { } impl WindowUDFImpl for AnomalyScoreZscore { - fn as_any(&self) -> &dyn Any { - self - } - fn name(&self) -> &str { "anomaly_score_zscore" } diff --git a/src/common/function/src/scalars/geo/geohash.rs b/src/common/function/src/scalars/geo/geohash.rs index 90bb958246..7d91de3565 100644 --- a/src/common/function/src/scalars/geo/geohash.rs +++ b/src/common/function/src/scalars/geo/geohash.rs @@ -22,12 +22,12 @@ use datafusion::arrow::array::{Array, AsArray, ListBuilder, StringViewBuilder}; use datafusion::arrow::datatypes::{DataType, Field, Float64Type, UInt8Type}; use datafusion::logical_expr::ColumnarValue; use datafusion_common::DataFusionError; -use datafusion_expr::type_coercion::aggregates::INTEGERS; use datafusion_expr::{ScalarFunctionArgs, Signature, TypeSignature, Volatility}; use geohash::Coord; use snafu::ResultExt; use crate::function::{Function, extract_args}; +use crate::helper::INTEGER_TYPES; use crate::scalars::geo::helpers; fn ensure_resolution_usize(v: u8) -> datafusion_common::Result { @@ -49,7 +49,7 @@ impl Default for GeohashFunction { fn default() -> Self { let mut signatures = Vec::new(); for coord_type in &[DataType::Float32, DataType::Float64] { - for resolution_type in INTEGERS { + for resolution_type in INTEGER_TYPES { signatures.push(TypeSignature::Exact(vec![ // latitude coord_type.clone(), @@ -146,7 +146,7 @@ impl Default for GeohashNeighboursFunction { fn default() -> Self { let mut signatures = Vec::new(); for coord_type in &[DataType::Float32, DataType::Float64] { - for resolution_type in INTEGERS { + for resolution_type in INTEGER_TYPES { signatures.push(TypeSignature::Exact(vec![ // latitude coord_type.clone(), diff --git a/src/common/function/src/scalars/geo/h3.rs b/src/common/function/src/scalars/geo/h3.rs index c6630525df..7792f3a9f1 100644 --- a/src/common/function/src/scalars/geo/h3.rs +++ b/src/common/function/src/scalars/geo/h3.rs @@ -26,7 +26,6 @@ use datafusion::arrow::compute; use datafusion::arrow::datatypes::{Float64Type, Int64Type, UInt8Type, UInt64Type}; use datafusion::logical_expr::ColumnarValue; use datafusion_common::{DataFusionError, ScalarValue}; -use datafusion_expr::type_coercion::aggregates::INTEGERS; use datafusion_expr::{ScalarFunctionArgs, Signature, TypeSignature, Volatility}; use datatypes::arrow::datatypes::{DataType, Field}; use derive_more::Display; @@ -34,6 +33,7 @@ use h3o::{CellIndex, LatLng, Resolution}; use snafu::prelude::*; use crate::function::{Function, extract_args}; +use crate::helper::INTEGER_TYPES; use crate::scalars::geo::helpers; static CELL_TYPES: LazyLock> = @@ -42,11 +42,11 @@ static CELL_TYPES: LazyLock> = static COORDINATE_TYPES: LazyLock> = LazyLock::new(|| vec![DataType::Float32, DataType::Float64]); -static RESOLUTION_TYPES: &[DataType] = INTEGERS; +static RESOLUTION_TYPES: &[DataType] = INTEGER_TYPES; -static DISTANCE_TYPES: &[DataType] = INTEGERS; +static DISTANCE_TYPES: &[DataType] = INTEGER_TYPES; -static POSITION_TYPES: &[DataType] = INTEGERS; +static POSITION_TYPES: &[DataType] = INTEGER_TYPES; /// Function that returns [h3] encoding cellid for a given geospatial coordinate. /// diff --git a/src/common/function/src/scalars/geo/s2.rs b/src/common/function/src/scalars/geo/s2.rs index e4c2848dce..88fef9ac32 100644 --- a/src/common/function/src/scalars/geo/s2.rs +++ b/src/common/function/src/scalars/geo/s2.rs @@ -25,6 +25,7 @@ use s2::latlng::LatLng; use snafu::ensure; use crate::function::{Function, extract_args}; +use crate::helper::INTEGER_TYPES; use crate::scalars::geo::helpers; use crate::scalars::geo::helpers::ensure_and_coerce; @@ -34,7 +35,7 @@ static CELL_TYPES: LazyLock> = static COORDINATE_TYPES: LazyLock> = LazyLock::new(|| vec![DataType::Float32, DataType::Float64]); -static LEVEL_TYPES: &[DataType] = datafusion_expr::type_coercion::aggregates::INTEGERS; +static LEVEL_TYPES: &[DataType] = INTEGER_TYPES; /// Function that returns [s2] encoding cellid for a given geospatial coordinate. /// diff --git a/src/common/function/src/scalars/json/json_get.rs b/src/common/function/src/scalars/json/json_get.rs index 2357bc88f9..fcce7c9ee1 100644 --- a/src/common/function/src/scalars/json/json_get.rs +++ b/src/common/function/src/scalars/json/json_get.rs @@ -532,7 +532,7 @@ mod tests { use datafusion_common::arrow::datatypes::{Float64Type, Int64Type}; use datatypes::extension::json::Json2ExtensionType; use datatypes::types::parse_string_to_jsonb; - use serde_json::json; + use serde_json::{Value, json}; use super::*; @@ -595,6 +595,34 @@ mod tests { }) } + fn assert_json_or_string_eq(actual: Option<&str>, expected: Option<&str>) { + let is_json_container = |value: &str| { + matches!( + serde_json::from_str::(value), + Ok(Value::Object(_) | Value::Array(_)) + ) + }; + + match (actual, expected) { + (Some(actual), Some(expected)) + if is_json_container(actual) || is_json_container(expected) => + { + let actual_value = serde_json::from_str::(actual).unwrap_or_else(|error| { + panic!("failed to parse actual JSON result {actual:?}: {error}") + }); + let expected_value = + serde_json::from_str::(expected).unwrap_or_else(|error| { + panic!("failed to parse expected JSON result {expected:?}: {error}") + }); + assert_eq!( + actual_value, expected_value, + "JSON result mismatch: actual {actual:?}, expected {expected:?}" + ); + } + _ => assert_eq!(actual, expected), + } + } + #[test] fn test_json_get_int() { let json_get_int = JsonGetInt::default(); @@ -895,7 +923,7 @@ mod tests { let result = result.as_string_view(); assert_eq!(1, result.len()); let actual = result.is_valid(0).then(|| result.value(0)); - assert_eq!(actual, expect); + assert_json_or_string_eq(actual, expect); } } @@ -1033,7 +1061,7 @@ mod tests { let result = result.as_string_view(); assert_eq!(1, result.len()); let actual = result.is_valid(0).then(|| result.value(0)); - assert_eq!(actual, expect); + assert_json_or_string_eq(actual, expect); } let json_strings = [ diff --git a/src/common/function/src/scalars/json/json_get_rewriter.rs b/src/common/function/src/scalars/json/json_get_rewriter.rs index 0143ee05d5..b686e4f68a 100644 --- a/src/common/function/src/scalars/json/json_get_rewriter.rs +++ b/src/common/function/src/scalars/json/json_get_rewriter.rs @@ -59,10 +59,8 @@ impl FunctionRewrite for JsonGetRewriter { // json_get(column, path, ) // ) fn inject_type_from_cast_expr(cast: Cast) -> Result> { - let Cast { - expr, - mut data_type, - } = cast; + let Cast { expr, field } = cast; + let mut data_type = field.data_type().clone(); let mut json_get = match *expr { Expr::ScalarFunction(f) @@ -73,7 +71,7 @@ fn inject_type_from_cast_expr(cast: Cast) -> Result> { expr => { return Ok(Transformed::no(Expr::Cast(Cast { expr: Box::new(expr), - data_type, + field, }))); } }; @@ -204,10 +202,7 @@ mod tests { }); // Create a cast expression: json_get(...)::int8 - let cast_expr = Expr::Cast(Cast { - expr: Box::new(json_expr), - data_type: DataType::Int8, - }); + let cast_expr = Expr::Cast(Cast::new(Box::new(json_expr), DataType::Int8)); // Apply the rewriter let result = rewriter.rewrite(cast_expr, &schema, &config).unwrap(); @@ -279,10 +274,7 @@ mod tests { // Create an arrow cast function: cast(json_get(...), 'Int64') // Note: ArrowCastFunc doesn't exist in this codebase, so this test uses a simple cast instead - let arrow_cast_expr = Expr::Cast(Cast { - expr: Box::new(json_get_expr), - data_type: DataType::Int64, - }); + let arrow_cast_expr = Expr::Cast(Cast::new(Box::new(json_get_expr), DataType::Int64)); // Apply the rewriter let result = rewriter.rewrite(arrow_cast_expr, &schema, &config).unwrap(); diff --git a/src/common/function/src/scalars/matches.rs b/src/common/function/src/scalars/matches.rs index b5de60dc85..14889957ee 100644 --- a/src/common/function/src/scalars/matches.rs +++ b/src/common/function/src/scalars/matches.rs @@ -24,6 +24,7 @@ use datafusion::execution::SessionStateBuilder; use datafusion::logical_expr::{self, ColumnarValue, Expr, Volatility}; use datafusion::physical_planner::{DefaultPhysicalPlanner, PhysicalPlanner}; use datafusion_common::DataFusionError; +use datafusion_expr::physical_planning_context::PhysicalPlanningContext; use datafusion_expr::{ScalarFunctionArgs, Signature}; use datatypes::arrow::array::RecordBatch; use datatypes::arrow::datatypes::{DataType, Field}; @@ -110,8 +111,12 @@ impl MatchesFunction { let input_schema = Self::input_schema(); let session_state = SessionStateBuilder::new().with_default_features().build(); let planner = DefaultPhysicalPlanner::default(); - let physical_expr = - planner.create_physical_expr(&like_expr, &input_schema, &session_state)?; + let physical_expr = planner.create_physical_expr( + &like_expr, + &input_schema, + &session_state, + &PhysicalPlanningContext::default(), + )?; let arrow_schema = Arc::new(input_schema.as_arrow().clone()); let input_record_batch = RecordBatch::try_new(arrow_schema, vec![data_array]).unwrap(); diff --git a/src/common/function/src/scalars/math/clamp.rs b/src/common/function/src/scalars/math/clamp.rs index 14774c0b61..a76f17a99d 100644 --- a/src/common/function/src/scalars/math/clamp.rs +++ b/src/common/function/src/scalars/math/clamp.rs @@ -19,10 +19,10 @@ use datafusion::arrow::array::{Array, ArrayRef, AsArray, PrimitiveArray}; use datafusion::arrow::datatypes::DataType as ArrowDataType; use datafusion::logical_expr::{ColumnarValue, Volatility}; use datafusion_common::{DataFusionError, ScalarValue, utils}; -use datafusion_expr::type_coercion::aggregates::NUMERICS; use datafusion_expr::{ScalarFunctionArgs, Signature}; use crate::function::Function; +use crate::helper::NUMERICS; #[derive(Clone, Debug)] pub struct ClampFunction { @@ -338,7 +338,9 @@ mod test { use arrow_schema::Field; use datafusion_common::config::ConfigOptions; - use datatypes::arrow::array::{ArrayRef, Float64Array, Int64Array, UInt64Array}; + use datatypes::arrow::array::{ + ArrayRef, Decimal128Array, Float64Array, Int64Array, UInt64Array, + }; use datatypes::arrow_array::StringArray; use super::*; @@ -370,6 +372,79 @@ mod test { impl_test_eval!(ClampMinFunction); impl_test_eval!(ClampMaxFunction); + fn decimal_array(values: Vec) -> ColumnarValue { + ColumnarValue::Array(Arc::new( + Decimal128Array::from(values) + .with_precision_and_scale(10, 2) + .unwrap(), + )) + } + + fn decimal_scalar(value: i128) -> ColumnarValue { + ColumnarValue::Scalar(ScalarValue::Decimal128(Some(value), 10, 2)) + } + + #[allow(deprecated)] + fn evaluate_decimal( + function: &dyn Function, + args: Vec, + ) -> datafusion_common::Result { + let input_types = args + .iter() + .map(ColumnarValue::data_type) + .collect::>(); + let planned_types = datafusion_expr::type_coercion::functions::data_types( + function.name(), + &input_types, + function.signature(), + )?; + let args = args + .into_iter() + .zip(planned_types) + .map(|(arg, planned_type)| arg.cast_to(&planned_type, None)) + .collect::>>()?; + function + .invoke_with_args(ScalarFunctionArgs { + args, + arg_fields: vec![], + number_rows: 3, + return_field: Arc::new(Field::new("x", ArrowDataType::Float64, false)), + config_options: Arc::new(ConfigOptions::new()), + }) + .and_then(|value| value.to_array(3)) + } + + #[test] + fn clamp_decimal_coercion_executes_as_float64() { + let test_cases = [ + ( + Box::new(ClampFunction::default()) as Box, + vec![ + decimal_array(vec![100, 300, 500]), + decimal_scalar(200), + decimal_scalar(400), + ], + vec![2.0, 3.0, 4.0], + ), + ( + Box::new(ClampMinFunction::default()), + vec![decimal_array(vec![100, 300, 500]), decimal_scalar(200)], + vec![2.0, 3.0, 5.0], + ), + ( + Box::new(ClampMaxFunction::default()), + vec![decimal_array(vec![100, 300, 500]), decimal_scalar(200)], + vec![1.0, 2.0, 2.0], + ), + ]; + + for (function, args, expected) in test_cases { + let result = evaluate_decimal(function.as_ref(), args).unwrap(); + let expected: ArrayRef = Arc::new(Float64Array::from(expected)); + assert_eq!(expected.as_ref(), result.as_ref()); + } + } + #[test] fn clamp_i64() { let inputs = [ diff --git a/src/common/function/src/scalars/math/modulo.rs b/src/common/function/src/scalars/math/modulo.rs index 869b8efaca..23ad52cb89 100644 --- a/src/common/function/src/scalars/math/modulo.rs +++ b/src/common/function/src/scalars/math/modulo.rs @@ -18,10 +18,10 @@ use std::fmt::Display; use datafusion_common::arrow::compute; use datafusion_common::arrow::compute::kernels::numeric; use datafusion_common::arrow::datatypes::DataType; -use datafusion_expr::type_coercion::aggregates::NUMERICS; use datafusion_expr::{ColumnarValue, ScalarFunctionArgs, Signature, Volatility}; use crate::function::{Function, extract_args}; +use crate::helper::NUMERICS; const NAME: &str = "mod"; @@ -90,12 +90,74 @@ mod tests { use std::sync::Arc; use arrow_schema::Field; + use datafusion_common::ScalarValue; use datafusion_common::arrow::array::{ - AsArray, Float64Array, Int32Array, StringViewArray, UInt32Array, + AsArray, Decimal128Array, Float64Array, Int32Array, StringViewArray, UInt32Array, }; use datafusion_common::arrow::datatypes::{Float64Type, Int64Type, UInt64Type}; use super::*; + fn decimal_array(values: Vec) -> ColumnarValue { + ColumnarValue::Array(Arc::new( + Decimal128Array::from(values) + .with_precision_and_scale(10, 2) + .unwrap(), + )) + } + + fn decimal_scalar(value: i128) -> ColumnarValue { + ColumnarValue::Scalar(ScalarValue::Decimal128(Some(value), 10, 2)) + } + + #[test] + #[allow(deprecated)] + fn modulo_decimal_coercion_executes_as_float64() { + let function = ModuloFunction::default(); + let test_cases = [ + ( + vec![decimal_array(vec![500, 600]), decimal_scalar(200)], + vec![1.0, 0.0], + ), + ( + vec![decimal_scalar(500), decimal_array(vec![200, 300])], + vec![1.0, 2.0], + ), + ]; + + for (args, expected) in test_cases { + let input_types = args + .iter() + .map(ColumnarValue::data_type) + .collect::>(); + let planned_types = datafusion_expr::type_coercion::functions::data_types( + function.name(), + &input_types, + function.signature(), + ) + .unwrap(); + assert_eq!(vec![DataType::Float64; 2], planned_types); + let args = args + .into_iter() + .zip(planned_types) + .map(|(arg, planned_type)| arg.cast_to(&planned_type, None)) + .collect::>>() + .unwrap(); + let result = function + .invoke_with_args(ScalarFunctionArgs { + args, + arg_fields: vec![], + number_rows: 2, + return_field: Arc::new(Field::new("x", DataType::Float64, false)), + config_options: Arc::new(Default::default()), + }) + .unwrap() + .to_array(2) + .unwrap(); + let result = result.as_primitive::(); + assert_eq!(&Float64Array::from(expected), result); + } + } + #[test] fn test_mod_function_signed() { let function = ModuloFunction::default(); diff --git a/src/common/function/src/scalars/math/rate.rs b/src/common/function/src/scalars/math/rate.rs index b2a45b2036..b33667a06c 100644 --- a/src/common/function/src/scalars/math/rate.rs +++ b/src/common/function/src/scalars/math/rate.rs @@ -18,11 +18,11 @@ use common_query::error; use datafusion::arrow::compute::kernels::numeric; use datafusion_common::arrow::compute::kernels::cast; use datafusion_common::arrow::datatypes::DataType; -use datafusion_expr::type_coercion::aggregates::NUMERICS; use datafusion_expr::{ColumnarValue, ScalarFunctionArgs, Signature, Volatility}; use snafu::ResultExt; use crate::function::{Function, extract_args}; +use crate::helper::NUMERICS; /// generates rates from a sequence of adjacent data points. #[derive(Clone, Debug)] @@ -96,12 +96,13 @@ mod tests { let rate = RateFunction::default(); assert_eq!("rate", rate.name()); assert_eq!(DataType::Float64, rate.return_type(&[]).unwrap()); - assert!(matches!(rate.signature(), - Signature { - type_signature: TypeSignature::Uniform(2, valid_types), - volatility: Volatility::Immutable, - .. - } if valid_types == NUMERICS + assert!(matches!( + rate.signature(), + Signature { + type_signature: TypeSignature::Uniform(2, valid_types), + volatility: Volatility::Immutable, + .. + } if valid_types == NUMERICS )); let values = vec![1.0, 3.0, 6.0]; let ts = vec![0, 1, 2]; diff --git a/src/common/function/src/scalars/udf.rs b/src/common/function/src/scalars/udf.rs index 638f7c38af..48c59ac230 100644 --- a/src/common/function/src/scalars/udf.rs +++ b/src/common/function/src/scalars/udf.rs @@ -12,7 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::fmt::{Debug, Formatter}; use std::hash::{Hash, Hasher}; @@ -49,10 +48,6 @@ impl Hash for ScalarUdf { } impl ScalarUDFImpl for ScalarUdf { - fn as_any(&self) -> &dyn Any { - self - } - fn name(&self) -> &str { self.function.name() } diff --git a/src/common/function/src/system/pg_catalog.rs b/src/common/function/src/system/pg_catalog.rs index 96bcc3fe9d..3029361c20 100644 --- a/src/common/function/src/system/pg_catalog.rs +++ b/src/common/function/src/system/pg_catalog.rs @@ -389,6 +389,9 @@ impl PGCatalogFunction { registry.register(pg_catalog::create_pg_get_partition_ancestors_udf()); registry.register(pg_catalog::quote_ident_udf::create_quote_ident_udf()); registry.register(pg_catalog::quote_ident_udf::create_parse_ident_udf()); + // Register array bound UDFs used by pg_catalog views. + registry.register(pg_catalog::array_bounds_udf::create_array_upper_udf()); + registry.register(pg_catalog::array_bounds_udf::create_array_lower_udf()); registry.register_scalar(ObjDescriptionFunction::new()); registry.register_scalar(ColDescriptionFunction::new()); registry.register_scalar(ShobjDescriptionFunction::new()); diff --git a/src/common/macro/src/admin_fn.rs b/src/common/macro/src/admin_fn.rs index 053a2ad3f4..3575ebfe6c 100644 --- a/src/common/macro/src/admin_fn.rs +++ b/src/common/macro/src/admin_fn.rs @@ -239,10 +239,6 @@ fn build_struct( // Implement DataFusion's ScalarUDFImpl trait impl datafusion::logical_expr::ScalarUDFImpl for #name { - fn as_any(&self) -> &dyn std::any::Any { - self - } - fn name(&self) -> &str { #display_name } diff --git a/src/common/query/src/request.rs b/src/common/query/src/request.rs index 6dc62b0a89..5668381861 100644 --- a/src/common/query/src/request.rs +++ b/src/common/query/src/request.rs @@ -137,9 +137,7 @@ fn portable_remote_dyn_filter_expr( bounds_only: bool, ) -> DataFusionResult> { expr.transform_up(|node| { - if node.as_any().is::() - || (bounds_only && node.as_any().is::()) - { + if node.is::() || (bounds_only && node.is::()) { Ok(Transformed::yes(lit(true))) } else { Ok(Transformed::no(node)) @@ -183,7 +181,7 @@ fn validate_payload_size( fn validate_supported_payload_expr(expr: &Arc) -> DataFusionResult<()> { expr.apply(|node| { - if node.as_any().is::() { + if node.is::() { return Err(DataFusionError::Plan( "HashTableLookupExpr cannot be encoded into DynFilterPayload::Datafusion" .to_string(), @@ -207,7 +205,7 @@ fn validate_decoded_payload_expr( input_schema: &datafusion::arrow::datatypes::Schema, ) -> DataFusionResult<()> { expr.apply(|node| { - if let Some(column) = node.as_any().downcast_ref::() { + if let Some(column) = node.downcast_ref::() { let Some(field) = input_schema.fields().get(column.index()) else { return Err(DataFusionError::Plan(format!( "Decoded Column '{}' references out-of-bounds index {} for input schema of size {}", @@ -391,8 +389,8 @@ mod tests { .decode_datafusion_expr(&TaskContext::default(), &schema, 1024) .unwrap(); - let original = expr.as_any().downcast_ref::().unwrap(); - let decoded = decoded.as_any().downcast_ref::().unwrap(); + let original = expr.downcast_ref::().unwrap(); + let decoded = decoded.downcast_ref::().unwrap(); assert_eq!(decoded.name(), original.name()); assert_eq!(decoded.index(), original.index()); @@ -457,7 +455,7 @@ mod tests { )) as Arc; let lookup = Arc::new(HashTableLookupExpr::new( vec![Arc::clone(&device_id)], - SeededRandomState::with_seeds(0, 0, 0, 0), + SeededRandomState::with_seed(0), Arc::new(Map::HashMap(Box::new(JoinHashMapU32::with_capacity(0)))), "hash_lookup".to_string(), )) as Arc; @@ -539,10 +537,10 @@ mod tests { )); } - fn contains_expr(expr: &Arc) -> bool { + fn contains_expr(expr: &Arc) -> bool { let mut found = false; expr.apply(|node| { - if node.as_any().is::() { + if node.is::() { found = true; Ok(TreeNodeRecursion::Stop) } else { diff --git a/src/common/query/src/request/initial_remote_dyn_filter_reg.rs b/src/common/query/src/request/initial_remote_dyn_filter_reg.rs index a4a00d5fac..3706ae25df 100644 --- a/src/common/query/src/request/initial_remote_dyn_filter_reg.rs +++ b/src/common/query/src/request/initial_remote_dyn_filter_reg.rs @@ -371,7 +371,7 @@ mod tests { let decoded = reg .decode_children(&TaskContext::default(), &schema, 1024) .unwrap(); - let decoded = decoded[0].as_any().downcast_ref::().unwrap(); + let decoded = decoded[0].downcast_ref::().unwrap(); assert_eq!(reg.filter_id, "filter-1"); assert_eq!(decoded.name(), "host"); diff --git a/src/common/query/src/stream.rs b/src/common/query/src/stream.rs index 1af777ab15..0a46164d9b 100644 --- a/src/common/query/src/stream.rs +++ b/src/common/query/src/stream.rs @@ -12,7 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::fmt::{Debug, Formatter}; use std::sync::{Arc, Mutex}; @@ -22,8 +21,11 @@ use datafusion::execution::SendableRecordBatchStream as DfSendableRecordBatchStr use datafusion::execution::context::TaskContext; use datafusion::physical_expr::{EquivalenceProperties, Partitioning, PhysicalSortExpr}; use datafusion::physical_plan::execution_plan::{Boundedness, EmissionType}; -use datafusion::physical_plan::{DisplayAs, DisplayFormatType, ExecutionPlan, PlanProperties}; +use datafusion::physical_plan::{ + DisplayAs, DisplayFormatType, ExecutionPlan, PhysicalExpr, PlanProperties, +}; use datafusion_common::DataFusionError; +use datafusion_common::tree_node::TreeNodeRecursion; use datatypes::arrow::datatypes::SchemaRef as ArrowSchemaRef; use datatypes::schema::SchemaRef; @@ -99,10 +101,6 @@ impl DisplayAs for StreamScanAdapter { } impl ExecutionPlan for StreamScanAdapter { - fn as_any(&self) -> &dyn Any { - self - } - fn schema(&self) -> ArrowSchemaRef { self.arrow_schema.clone() } @@ -115,6 +113,13 @@ impl ExecutionPlan for StreamScanAdapter { vec![] } + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> datafusion_common::Result, + ) -> datafusion_common::Result { + Ok(TreeNodeRecursion::Continue) + } + // DataFusion will swap children unconditionally. // But since this node is leaf node, it's safe to just return self. fn with_new_children( diff --git a/src/common/recordbatch/src/adapter.rs b/src/common/recordbatch/src/adapter.rs index ca9619c013..e10bf44083 100644 --- a/src/common/recordbatch/src/adapter.rs +++ b/src/common/recordbatch/src/adapter.rs @@ -29,6 +29,7 @@ use datafusion::arrow::datatypes::SchemaRef as DfSchemaRef; use datafusion::error::Result as DfResult; use datafusion::execution::context::ExecutionProps; use datafusion::logical_expr::Expr; +use datafusion::logical_expr::physical_planning_context::PhysicalPlanningContext; use datafusion::logical_expr::utils::conjunction; use datafusion::physical_expr::create_physical_expr; use datafusion::physical_plan::metrics::{BaselineMetrics, MetricValue}; @@ -98,8 +99,13 @@ where .to_dfschema_ref() .context(error::PhysicalExprSnafu)?; - let filters = create_physical_expr(&expr, &df_schema, &ExecutionProps::new()) - .context(error::PhysicalExprSnafu)?; + let filters = create_physical_expr( + &expr, + &df_schema, + &ExecutionProps::new(), + &PhysicalPlanningContext::default(), + ) + .context(error::PhysicalExprSnafu)?; Some(filters) } else { None @@ -931,7 +937,6 @@ fn convert_map_to_json_binary( #[cfg(test)] mod test { - use std::any::Any; use std::time::Duration; use common_error::ext::BoxedError; @@ -942,6 +947,7 @@ mod test { use datafusion::physical_plan::execution_plan::{Boundedness, EmissionType}; use datafusion::physical_plan::metrics::{ExecutionPlanMetricsSet, MetricBuilder, MetricsSet}; use datafusion::physical_plan::{DisplayAs, PlanProperties}; + use datafusion_common::tree_node::TreeNodeRecursion; use datatypes::arrow::array::{ArrayRef, MapArray, StringArray, StructArray}; use datatypes::arrow::buffer::OffsetBuffer; use datatypes::arrow::datatypes::Field; @@ -1021,10 +1027,6 @@ mod test { REGION_SCAN_EXEC_NAME } - fn as_any(&self) -> &dyn Any { - self - } - fn properties(&self) -> &Arc { &self.properties } @@ -1033,6 +1035,15 @@ mod test { vec![] } + fn apply_expressions( + &self, + _f: &mut dyn FnMut( + &Arc, + ) -> datafusion_common::Result, + ) -> datafusion_common::Result { + Ok(TreeNodeRecursion::Continue) + } + fn with_new_children( self: Arc, _children: Vec>, diff --git a/src/common/recordbatch/src/filter.rs b/src/common/recordbatch/src/filter.rs index 1b398d6cc4..26c28aa3fa 100644 --- a/src/common/recordbatch/src/filter.rs +++ b/src/common/recordbatch/src/filter.rs @@ -616,6 +616,7 @@ mod test { use std::sync::Arc; use datafusion::execution::context::ExecutionProps; + use datafusion::logical_expr::physical_planning_context::PhysicalPlanningContext; use datafusion::logical_expr::{BinaryExpr, col, lit}; use datafusion::physical_expr::create_physical_expr; use datafusion_common::{Column, DFSchema}; @@ -737,7 +738,13 @@ mod test { ]); let df_schema = DFSchema::try_from(schema.clone()).unwrap(); let props = ExecutionProps::new(); - let physical_expr = create_physical_expr(&expr, &df_schema, &props).unwrap(); + let physical_expr = create_physical_expr( + &expr, + &df_schema, + &props, + &PhysicalPlanningContext::default(), + ) + .unwrap(); let batch = RecordBatch::try_new( Arc::new(schema), vec![ @@ -783,7 +790,13 @@ mod test { let schema = Schema::new(vec![Field::new("col", DataType::Utf8, false)]); let df_schema = DFSchema::try_from(schema.clone()).unwrap(); let props = ExecutionProps::new(); - let physical_expr = create_physical_expr(&col_or_expr, &df_schema, &props).unwrap(); + let physical_expr = create_physical_expr( + &col_or_expr, + &df_schema, + &props, + &PhysicalPlanningContext::default(), + ) + .unwrap(); // Create test data let col_data = Arc::new(datatypes::arrow::array::StringArray::from(vec![ diff --git a/src/common/recordbatch/src/lib.rs b/src/common/recordbatch/src/lib.rs index 72918e9e5e..4fb37cd5b1 100644 --- a/src/common/recordbatch/src/lib.rs +++ b/src/common/recordbatch/src/lib.rs @@ -44,7 +44,9 @@ pub use datatypes::arrow::record_batch::RecordBatch as DfRecordBatch; use datatypes::arrow::util::display::{ ArrayFormatter, ArrayFormatterFactory, DisplayIndex, FormatOptions, FormatResult, }; -use datatypes::arrow::util::pretty::pretty_format_batches_with_options; +use datatypes::arrow::util::pretty::{ + pretty_format_batches_with_options, pretty_format_batches_with_schema, +}; use datatypes::extension::json::is_any_json_extension_type; use datatypes::prelude::{ConcreteDataType, DataType, VectorRef}; use datatypes::schema::{ColumnSchema, Schema, SchemaRef}; @@ -396,12 +398,19 @@ impl RecordBatches { .iter() .map(|x| x.df_record_batch().clone()) .collect::>(); - let options = - FormatOptions::default().with_formatter_factory(Some(&BinaryFormatterFactory)); - let result = - pretty_format_batches_with_options(df_batches, &options).context(error::FormatSnafu)?; + let result: String = if df_batches.is_empty() { + pretty_format_batches_with_schema(self.schema.arrow_schema().clone(), df_batches) + .context(error::FormatSnafu)? + .to_string() + } else { + let options = + FormatOptions::default().with_formatter_factory(Some(&BinaryFormatterFactory)); + pretty_format_batches_with_options(df_batches, &options) + .context(error::FormatSnafu)? + .to_string() + }; - Ok(result.to_string()) + Ok(result) } pub fn try_new(schema: SchemaRef, batches: Vec) -> Result { @@ -1082,6 +1091,35 @@ mod tests { assert_eq!(r.take(), expected); } + #[tokio::test] + async fn test_recordbatches_pretty_print_empty_batches_preserves_schema() { + let schema = Arc::new(Schema::new(vec![ + ColumnSchema::new("unit", ConcreteDataType::string_datatype(), false), + ColumnSchema::new( + "ts", + ConcreteDataType::timestamp_millisecond_datatype(), + false, + ), + ColumnSchema::new( + "lhs.degrees(val) + rhs.radians(val)", + ConcreteDataType::float64_datatype(), + false, + ), + ])); + let batches = + RecordBatches::try_collect(Box::pin(EmptyRecordBatchStream::new(schema.clone()))) + .await + .unwrap(); + + assert_eq!(schema, batches.schema()); + let expected = "\ ++------+----+-------------------------------------+ +| unit | ts | lhs.degrees(val) + rhs.radians(val) | ++------+----+-------------------------------------+ ++------+----+-------------------------------------+"; + assert_eq!(expected, batches.pretty_print().unwrap()); + } + #[test] fn test_recordbatches_try_new() { let column_a = ColumnSchema::new("a", ConcreteDataType::int32_datatype(), false); diff --git a/src/datanode/src/region_server/catalog.rs b/src/datanode/src/region_server/catalog.rs index a4df422b75..264408342e 100644 --- a/src/datanode/src/region_server/catalog.rs +++ b/src/datanode/src/region_server/catalog.rs @@ -179,9 +179,6 @@ impl NameAwareCatalogList { } impl CatalogProviderList for NameAwareCatalogList { - fn as_any(&self) -> &dyn std::any::Any { - self - } fn register_catalog( &self, _name: String, @@ -203,9 +200,6 @@ struct NameAwareCatalogProvider { } impl CatalogProvider for NameAwareCatalogProvider { - fn as_any(&self) -> &dyn std::any::Any { - self - } fn schema_names(&self) -> Vec { vec![] } @@ -229,9 +223,6 @@ impl std::fmt::Debug for NameAwareSchemaProvider { #[async_trait::async_trait] impl SchemaProvider for NameAwareSchemaProvider { - fn as_any(&self) -> &dyn std::any::Any { - self - } fn table_names(&self) -> Vec { vec![] } diff --git a/src/datatypes/src/value.rs b/src/datatypes/src/value.rs index 35a6acabd3..56a551484f 100644 --- a/src/datatypes/src/value.rs +++ b/src/datatypes/src/value.rs @@ -1240,6 +1240,8 @@ impl TryFrom for Value { | ScalarValue::Decimal256(_, _, _) | ScalarValue::FixedSizeList(_) | ScalarValue::LargeList(_) + | ScalarValue::ListView(_) + | ScalarValue::LargeListView(_) | ScalarValue::Union(_, _, _) | ScalarValue::Float16(_) | ScalarValue::Utf8View(_) diff --git a/src/datatypes/src/vectors/helper.rs b/src/datatypes/src/vectors/helper.rs index 3d9e71853c..68fd794b5a 100644 --- a/src/datatypes/src/vectors/helper.rs +++ b/src/datatypes/src/vectors/helper.rs @@ -150,6 +150,8 @@ impl Helper { | ScalarValue::Decimal256(_, _, _) | ScalarValue::FixedSizeList(_) | ScalarValue::LargeList(_) + | ScalarValue::ListView(_) + | ScalarValue::LargeListView(_) | ScalarValue::Dictionary(_, _) | ScalarValue::Union(_, _, _) | ScalarValue::Utf8View(_) diff --git a/src/file-engine/src/query/file_stream.rs b/src/file-engine/src/query/file_stream.rs index a480a50374..7c7717287f 100644 --- a/src/file-engine/src/query/file_stream.rs +++ b/src/file-engine/src/query/file_stream.rs @@ -23,7 +23,8 @@ use datafusion::config::CsvOptions; use datafusion::datasource::listing::PartitionedFile; use datafusion::datasource::object_store::ObjectStoreUrl; use datafusion::datasource::physical_plan::{ - CsvSource, FileGroup, FileScanConfigBuilder, FileSource, FileStream, JsonSource, ParquetSource, + CsvSource, FileGroup, FileScanConfigBuilder, FileSource, FileStreamBuilder, JsonSource, + ParquetSource, }; use datafusion::datasource::source::DataSourceExec; use datafusion::physical_expr::create_physical_expr; @@ -34,6 +35,7 @@ use datafusion::physical_plan::{ }; use datafusion::prelude::SessionContext; use datafusion_expr::expr::Expr; +use datafusion_expr::physical_planning_context::PhysicalPlanningContext; use datafusion_expr::utils::conjunction; use datatypes::schema::SchemaRef; use object_store::ObjectStore; @@ -66,13 +68,12 @@ fn build_record_batch_stream( )); let file_opener = config.file_source().create_file_opener(store, &config, 0)?; - let stream = FileStream::new( - &config, - 0, // partition: hard-code - file_opener, - &ExecutionPlanMetricsSet::new(), - ) - .context(error::BuildStreamSnafu)?; + let stream = FileStreamBuilder::new(&config) + .with_partition(0) // partition: hard-code + .with_file_opener(file_opener) + .with_metrics(&ExecutionPlanMetricsSet::new()) + .build() + .context(error::BuildStreamSnafu)?; Ok(Box::pin(stream)) } @@ -138,8 +139,13 @@ fn new_parquet_stream_with_exec_plan( .to_dfschema_ref() .context(error::ParquetScanPlanSnafu)?; - let filters = create_physical_expr(&expr, &df_schema, &ExecutionProps::new()) - .context(error::ParquetScanPlanSnafu)?; + let filters = create_physical_expr( + &expr, + &df_schema, + &ExecutionProps::new(), + &PhysicalPlanningContext::default(), + ) + .context(error::ParquetScanPlanSnafu)?; parquet_source = parquet_source.with_predicate(filters); }; diff --git a/src/flow/src/batching_mode/time_window.rs b/src/flow/src/batching_mode/time_window.rs index 1c8ebea2a8..5dbd08150a 100644 --- a/src/flow/src/batching_mode/time_window.rs +++ b/src/flow/src/batching_mode/time_window.rs @@ -37,6 +37,7 @@ use datafusion_common::tree_node::{ Transformed, TreeNode, TreeNodeRecursion, TreeNodeRewriter, TreeNodeVisitor, }; use datafusion_common::{DFSchema, TableReference}; +use datafusion_expr::physical_planning_context::PhysicalPlanningContext; use datafusion_expr::{ColumnarValue, LogicalPlan}; use datafusion_physical_expr::PhysicalExprRef; use datatypes::prelude::{ConcreteDataType, DataType}; @@ -830,7 +831,15 @@ fn to_phy_expr( let phy_planner = DefaultPhysicalPlanner::default(); let phy_expr: PhysicalExprRef = phy_planner - .create_physical_expr(expr, df_schema, session) + // Time-window expressions are standalone scalar expressions over the input + // time column, so they cannot contain scalar subqueries or lambda variables + // that would require a plan-level physical planning context. + .create_physical_expr( + expr, + df_schema, + session, + &PhysicalPlanningContext::default(), + ) .with_context(|_e| DatafusionSnafu { context: format!( "Failed to create physical expression from {expr:?} using {df_schema:?}" @@ -993,7 +1002,7 @@ mod test { Some(Timestamp::new(0, TimeUnit::Millisecond)), Some(Timestamp::new(300000, TimeUnit::Millisecond)), ), - "SELECT sum(numbers_with_ts.number), numbers_with_ts.number, date_bin('5 minutes', numbers_with_ts.ts) AS time_window, bucket_name FROM (SELECT numbers_with_ts.number, numbers_with_ts.ts, CASE WHEN (numbers_with_ts.number < 5) THEN 'bucket_0_5' WHEN (numbers_with_ts.number >= 5) THEN 'bucket_5_inf' END AS bucket_name FROM numbers_with_ts WHERE ((ts >= CAST('1970-01-01 00:00:00' AS TIMESTAMP)) AND (ts <= CAST('1970-01-01 00:05:00' AS TIMESTAMP)))) GROUP BY numbers_with_ts.number, date_bin('5 minutes', numbers_with_ts.ts), bucket_name", + "SELECT sum(number), number, date_bin('5 minutes', ts) AS time_window, bucket_name FROM (SELECT numbers_with_ts.number, numbers_with_ts.ts, CASE WHEN (numbers_with_ts.number < 5) THEN 'bucket_0_5' WHEN (numbers_with_ts.number >= 5) THEN 'bucket_5_inf' END AS bucket_name FROM numbers_with_ts WHERE ((ts >= CAST('1970-01-01 00:00:00' AS TIMESTAMP)) AND (ts <= CAST('1970-01-01 00:05:00' AS TIMESTAMP)))) GROUP BY number, date_bin('5 minutes', ts), bucket_name", ), // complex subquery alias ( diff --git a/src/flow/src/batching_mode/utils.rs b/src/flow/src/batching_mode/utils.rs index a707c2d9eb..cb8e7e5063 100644 --- a/src/flow/src/batching_mode/utils.rs +++ b/src/flow/src/batching_mode/utils.rs @@ -31,7 +31,7 @@ use datafusion_common::tree_node::{ use datafusion_common::{ Column, DFSchema, DataFusionError, NullEquality, ScalarValue, TableReference, }; -use datafusion_expr::logical_plan::{Aggregate, TableScan}; +use datafusion_expr::logical_plan::{Aggregate, TableScanBuilder}; use datafusion_expr::{ Distinct, ExprSchemable, JoinType, LogicalPlan, LogicalPlanBuilder, Operator, Projection, and, binary_expr, bitwise_and, bitwise_or, bitwise_xor, is_null, or, when, @@ -657,17 +657,15 @@ pub async fn rewrite_incremental_aggregate_with_sink_merge( let table_provider = Arc::new(DfTableProviderAdapter::new(sink_table)); let table_source = Arc::new(DefaultTableSource::new(table_provider)); let sink_scan = LogicalPlan::TableScan( - TableScan::try_new( + TableScanBuilder::new( TableReference::Full { catalog: sink_table_name[0].clone().into(), schema: sink_table_name[1].clone().into(), table: sink_table_name[2].clone().into(), }, table_source, - None, - vec![], - None, ) + .build() .with_context(|_| DatafusionSnafu { context: "Failed to build sink table scan for incremental sink merge".to_string(), })?, diff --git a/src/flow/src/batching_mode/utils/test.rs b/src/flow/src/batching_mode/utils/test.rs index 1356f87e8f..1c34caeedd 100644 --- a/src/flow/src/batching_mode/utils/test.rs +++ b/src/flow/src/batching_mode/utils/test.rs @@ -18,7 +18,7 @@ use catalog::RegisterTableRequest; use common_recordbatch::RecordBatch; use common_time::Timestamp; use datafusion_common::tree_node::TreeNode as _; -use datafusion_expr::GroupingSet; +use datafusion_expr::{GroupingSet, TableScanBuilder}; use datatypes::prelude::{ConcreteDataType, MutableVector, Scalar, ScalarVectorBuilder, VectorRef}; use datatypes::schema::{ColumnSchema, Schema}; use datatypes::timestamp::TimestampMillisecond; @@ -92,17 +92,15 @@ fn test_sink_scan(sink_table: TableRef, sink_table_name: &TableName) -> LogicalP let table_provider = Arc::new(DfTableProviderAdapter::new(sink_table)); let table_source = Arc::new(DefaultTableSource::new(table_provider)); LogicalPlan::TableScan( - TableScan::try_new( + TableScanBuilder::new( TableReference::Full { catalog: sink_table_name[0].clone().into(), schema: sink_table_name[1].clone().into(), table: sink_table_name[2].clone().into(), }, table_source, - None, - vec![], - None, ) + .build() .unwrap(), ) } @@ -296,7 +294,7 @@ async fn test_add_filter() { // complex subquery without alias ( "SELECT sum(number), number, date_bin('5 minutes', ts) as time_window, bucket_name FROM (SELECT number, ts, case when number < 5 THEN 'bucket_0_5' when number >= 5 THEN 'bucket_5_inf' END as bucket_name FROM numbers_with_ts) GROUP BY number, time_window, bucket_name;", - "SELECT sum(numbers_with_ts.number), numbers_with_ts.number, date_bin('5 minutes', numbers_with_ts.ts) AS time_window, bucket_name FROM (SELECT numbers_with_ts.number, numbers_with_ts.ts, CASE WHEN (numbers_with_ts.number < 5) THEN 'bucket_0_5' WHEN (numbers_with_ts.number >= 5) THEN 'bucket_5_inf' END AS bucket_name FROM numbers_with_ts WHERE (number > 4)) GROUP BY numbers_with_ts.number, date_bin('5 minutes', numbers_with_ts.ts), bucket_name", + "SELECT sum(number), number, date_bin('5 minutes', ts) AS time_window, bucket_name FROM (SELECT numbers_with_ts.number, numbers_with_ts.ts, CASE WHEN (numbers_with_ts.number < 5) THEN 'bucket_0_5' WHEN (numbers_with_ts.number >= 5) THEN 'bucket_5_inf' END AS bucket_name FROM numbers_with_ts WHERE (number > 4)) GROUP BY number, date_bin('5 minutes', ts), bucket_name", ), // complex subquery alias ( diff --git a/src/flow/src/transform/aggr.rs b/src/flow/src/transform/aggr.rs index 861ca8fe65..7b938524a4 100644 --- a/src/flow/src/transform/aggr.rs +++ b/src/flow/src/transform/aggr.rs @@ -725,7 +725,7 @@ mod test { df_scalar_fn: DfScalarFunction::try_from_raw_fn( RawDfScalarFn { f: BytesMut::from( - b"\x08\x02\"\x0f\x1a\r\n\x0b\xa2\x02\x08\n\0\x12\x04\x10\x1e \t\"\n\x1a\x08\x12\x06\n\x04\x12\x02\x08\x01".as_ref(), + b"\x08\x02\x1a\x07\x8a\x02\x04\x08\x03\x18\x01\"\x0f\x1a\r\n\x0b\xa2\x02\x08\n\0\x12\x04\x10\x1e \t\"\n\x1a\x08\x12\x06\n\x04\x12\x02\x08\x01".as_ref(), ), input_schema: RelationType::new(vec![ColumnType::new( ConcreteDataType::interval_month_day_nano_datatype(), diff --git a/src/flow/src/transform/expr.rs b/src/flow/src/transform/expr.rs index 40dcaef8a6..a2150e2290 100644 --- a/src/flow/src/transform/expr.rs +++ b/src/flow/src/transform/expr.rs @@ -20,6 +20,7 @@ use common_error::ext::BoxedError; use common_telemetry::debug; use datafusion::execution::SessionStateBuilder; use datafusion::functions::all_default_functions; +use datafusion_expr::physical_planning_context::PhysicalPlanningContext; use datafusion_physical_expr::PhysicalExpr; use datafusion_substrait::logical_plan::consumer::DefaultSubstraitConsumer; use datatypes::data_type::ConcreteDataType as CDT; @@ -101,11 +102,15 @@ pub(crate) async fn from_scalar_fn_to_df_fn_impl( context: "Failed to convert substrait scalar function to datafusion scalar function", } })?; - let phy_expr = - datafusion::physical_expr::create_physical_expr(&expr, &schema, &Default::default()) - .context(DatafusionSnafu { - context: "Failed to create physical expression from logical expression", - })?; + let phy_expr = datafusion::physical_expr::create_physical_expr( + &expr, + &schema, + &Default::default(), + &PhysicalPlanningContext::default(), + ) + .context(DatafusionSnafu { + context: "Failed to create physical expression from logical expression", + })?; Ok(phy_expr) } diff --git a/src/frontend/src/instance/dashboard.rs b/src/frontend/src/instance/dashboard.rs index 33a9083cc8..ee3ef6e5cc 100644 --- a/src/frontend/src/instance/dashboard.rs +++ b/src/frontend/src/instance/dashboard.rs @@ -29,9 +29,9 @@ use common_query::OutputData; use common_recordbatch::util as record_util; use common_telemetry::info; use common_time::FOREVER; +use datafusion::common::TableReference; use datafusion::datasource::DefaultTableSource; use datafusion::logical_expr::col; -use datafusion::sql::TableReference; use datafusion_expr::{DmlStatement, LogicalPlan, lit}; use datatypes::arrow::array::{Array, AsArray}; use servers::error::{ diff --git a/src/mito2/src/engine/basic_test.rs b/src/mito2/src/engine/basic_test.rs index a9ee405415..f4c843f1ce 100644 --- a/src/mito2/src/engine/basic_test.rs +++ b/src/mito2/src/engine/basic_test.rs @@ -1341,6 +1341,7 @@ async fn test_all_index_metas_list_all_types_with_format(flat_format: bool, expe if let Some(inverted) = value.get_mut("inverted").and_then(|v| v.as_object_mut()) { inverted.insert("base_offset".to_string(), serde_json::Value::from(0)); } + value.sort_all_objects(); *meta_json = value.to_string(); } } diff --git a/src/mito2/src/engine/flush_test.rs b/src/mito2/src/engine/flush_test.rs index 04fa555d47..b040ae5ec5 100644 --- a/src/mito2/src/engine/flush_test.rs +++ b/src/mito2/src/engine/flush_test.rs @@ -1057,8 +1057,10 @@ async fn test_flush_empty_with_format(flat_format: bool) { let stream = scanner.scan().await.unwrap(); let batches = RecordBatches::try_collect(stream).await.unwrap(); let expected = "\ -++ -++"; ++-------+---------+----+ +| tag_0 | field_0 | ts | ++-------+---------+----+ ++-------+---------+----+"; assert_eq!(expected, batches.pretty_print().unwrap()); } diff --git a/src/mito2/src/engine/scan_test.rs b/src/mito2/src/engine/scan_test.rs index eedcee031d..ab7d6d9723 100644 --- a/src/mito2/src/engine/scan_test.rs +++ b/src/mito2/src/engine/scan_test.rs @@ -1053,8 +1053,10 @@ async fn test_scan_with_min_sst_sequence_with_format(flat_format: bool) { Some(9), 0, "\ -++ -++", ++-------+---------+----+ +| tag_0 | field_0 | ts | ++-------+---------+----+ ++-------+---------+----+", ) .await; } diff --git a/src/mito2/src/engine/sync_test.rs b/src/mito2/src/engine/sync_test.rs index 657ee868ce..d75a9ef031 100644 --- a/src/mito2/src/engine/sync_test.rs +++ b/src/mito2/src/engine/sync_test.rs @@ -148,7 +148,11 @@ async fn test_sync_after_flush_region_with_format(flat_format: bool) { common_telemetry::info!("Scan the region on the follower engine"); // Scan the region on the follower engine - let expected = "++\n++"; + let expected = "\ ++-------+---------+----+ +| tag_0 | field_0 | ts | ++-------+---------+----+ ++-------+---------+----+"; scan_check(&follower_engine, region_id, expected, 0, 0).await; // Returns error since the max manifest is 1 @@ -262,7 +266,11 @@ async fn test_sync_after_alter_region_with_format(flat_format: bool) { +-------+-------+---------+---------------------+"; scan_check(&engine, region_id, expected, 0, 1).await; - let expected = "++\n++"; + let expected = "\ ++-------+---------+----+ +| tag_0 | field_0 | ts | ++-------+---------+----+ ++-------+---------+----+"; scan_check(&follower_engine, region_id, expected, 0, 0).await; // Sync the region from the leader engine to the follower engine diff --git a/src/mito2/src/engine/truncate_test.rs b/src/mito2/src/engine/truncate_test.rs index c90bb960f8..61dbc062e7 100644 --- a/src/mito2/src/engine/truncate_test.rs +++ b/src/mito2/src/engine/truncate_test.rs @@ -155,7 +155,11 @@ async fn test_engine_truncate_region_basic_with_format(flat_format: bool) { let request = ScanRequest::default(); let stream = engine.scan_to_stream(region_id, request).await.unwrap(); let batches = RecordBatches::try_collect(stream).await.unwrap(); - let expected = "++\n++"; + let expected = "\ ++-------+---------+----+ +| tag_0 | field_0 | ts | ++-------+---------+----+ ++-------+---------+----+"; assert_eq!(expected, batches.pretty_print().unwrap()); } @@ -401,7 +405,11 @@ async fn test_engine_truncate_reopen_with_format(flat_format: bool) { let request = ScanRequest::default(); let stream = engine.scan_to_stream(region_id, request).await.unwrap(); let batches = RecordBatches::try_collect(stream).await.unwrap(); - let expected = "++\n++"; + let expected = "\ ++-------+---------+----+ +| tag_0 | field_0 | ts | ++-------+---------+----+ ++-------+---------+----+"; assert_eq!(expected, batches.pretty_print().unwrap()); } diff --git a/src/mito2/src/memtable/bulk/part.rs b/src/mito2/src/memtable/bulk/part.rs index f0af666a64..c3591cd4c0 100644 --- a/src/mito2/src/memtable/bulk/part.rs +++ b/src/mito2/src/memtable/bulk/part.rs @@ -1566,7 +1566,7 @@ impl PruningStatistics for BatchPruningStats<'_> { None } - fn row_counts(&self, _column: &Column) -> Option { + fn row_counts(&self) -> Option { None } diff --git a/src/mito2/src/read/scan_region.rs b/src/mito2/src/read/scan_region.rs index 06fe25dcdb..ef4c44a96c 100644 --- a/src/mito2/src/read/scan_region.rs +++ b/src/mito2/src/read/scan_region.rs @@ -1817,7 +1817,7 @@ impl PruningStatistics for FileLevelPruningStats { } } - fn row_counts(&self, _column: &Column) -> Option { + fn row_counts(&self) -> Option { None } diff --git a/src/mito2/src/sst.rs b/src/mito2/src/sst.rs index ba670ef734..e53c28e54c 100644 --- a/src/mito2/src/sst.rs +++ b/src/mito2/src/sst.rs @@ -1079,9 +1079,7 @@ mod tests { &builder.parquet_schema().root_schema().get_fields()[0].get_fields()[0]; assert_eq!( parquet_remainder.get_basic_info().logical_type_ref(), - Some(&LogicalType::Variant { - specification_version: None, - }) + Some(&LogicalType::variant(None)) ); let ArrowDataType::Struct(children) = builder.schema().field_with_name("data")?.data_type() diff --git a/src/mito2/src/sst/parquet/index_reader.rs b/src/mito2/src/sst/parquet/index_reader.rs index 0cf221986c..d3a23e6b1c 100644 --- a/src/mito2/src/sst/parquet/index_reader.rs +++ b/src/mito2/src/sst/parquet/index_reader.rs @@ -172,7 +172,7 @@ impl PruningStatistics for IndexRowGroupPruningStats<'_> { column_null_counts(self.row_groups, column_index) } - fn row_counts(&self, _column: &Column) -> Option { + fn row_counts(&self) -> Option { None } diff --git a/src/mito2/src/sst/parquet/json_align/stream.rs b/src/mito2/src/sst/parquet/json_align/stream.rs index bfd0b82edc..a474fd781d 100644 --- a/src/mito2/src/sst/parquet/json_align/stream.rs +++ b/src/mito2/src/sst/parquet/json_align/stream.rs @@ -247,7 +247,7 @@ fn align_array( return Ok(array.clone()); } - cast_column(array, field.as_ref(), &DEFAULT_CAST_OPTIONS).context(CastColumnSnafu) + cast_column(array, field.data_type(), &DEFAULT_CAST_OPTIONS).context(CastColumnSnafu) } #[cfg(test)] diff --git a/src/mito2/src/sst/parquet/read_columns.rs b/src/mito2/src/sst/parquet/read_columns.rs index e6a5ab9fab..58c7d0d2f8 100644 --- a/src/mito2/src/sst/parquet/read_columns.rs +++ b/src/mito2/src/sst/parquet/read_columns.rs @@ -386,7 +386,7 @@ fn is_variant_leaf(leaf_col: &ColumnDescriptor) -> bool { mod tests { use std::sync::Arc; - use parquet::basic::{ConvertedType, LogicalType, Repetition}; + use parquet::basic::{ConvertedType, LogicalType, Repetition, VariantType}; use parquet::errors::ParquetError; use parquet::schema::types::Type; @@ -816,9 +816,9 @@ mod tests { let remainder = Arc::new( Type::group_type_builder(JSON2_REMAINDER_FIELD_NAME) .with_repetition(Repetition::OPTIONAL) - .with_logical_type(Some(LogicalType::Variant { + .with_logical_type(Some(LogicalType::Variant(VariantType { specification_version: None, - })) + }))) .with_fields(vec![metadata, value]) .build()?, ); diff --git a/src/mito2/src/sst/parquet/reader.rs b/src/mito2/src/sst/parquet/reader.rs index b8588bd36c..59ef12d9e1 100644 --- a/src/mito2/src/sst/parquet/reader.rs +++ b/src/mito2/src/sst/parquet/reader.rs @@ -2556,7 +2556,6 @@ impl FlatRowGroupReader { #[cfg(test)] mod tests { - use std::any::Any; use std::collections::HashMap; use std::fmt::{Debug, Formatter}; use std::sync::{Arc, LazyLock}; @@ -3048,10 +3047,6 @@ mod tests { } impl ScalarUDFImpl for PanicDebugUdf { - fn as_any(&self) -> &dyn Any { - self - } - fn name(&self) -> &str { "panic_debug_udf" } @@ -3150,10 +3145,6 @@ mod tests { } impl ScalarUDFImpl for TestVolatilityUdf { - fn as_any(&self) -> &dyn Any { - self - } - fn name(&self) -> &str { &self.name } diff --git a/src/mito2/src/sst/parquet/stats.rs b/src/mito2/src/sst/parquet/stats.rs index 368828d1f9..6961e57ab2 100644 --- a/src/mito2/src/sst/parquet/stats.rs +++ b/src/mito2/src/sst/parquet/stats.rs @@ -174,7 +174,7 @@ impl> PruningStatistics for RowGroupPruningStats<'_, } } - fn row_counts(&self, _column: &Column) -> Option { + fn row_counts(&self) -> Option { // TODO(LFC): Impl it. None } diff --git a/src/partition/src/expr.rs b/src/partition/src/expr.rs index 1ef875dae0..cb15f5ae5d 100644 --- a/src/partition/src/expr.rs +++ b/src/partition/src/expr.rs @@ -20,6 +20,7 @@ use api::v1::meta::Partition; use datafusion_common::{ScalarValue, ToDFSchema}; use datafusion_expr::Expr; use datafusion_expr::execution_props::ExecutionProps; +use datafusion_expr::physical_planning_context::PhysicalPlanningContext; use datafusion_physical_expr::{PhysicalExpr, create_physical_expr}; use datatypes::arrow; use datatypes::value::{ @@ -391,8 +392,13 @@ impl PartitionExpr { .context(error::ToDFSchemaSnafu)?; let execution_props = &ExecutionProps::default(); let expr = self.try_as_logical_expr()?; - create_physical_expr(&expr, &df_schema, execution_props) - .context(error::CreatePhysicalExprSnafu) + create_physical_expr( + &expr, + &df_schema, + execution_props, + &PhysicalPlanningContext::default(), + ) + .context(error::CreatePhysicalExprSnafu) } pub fn and(self, rhs: PartitionExpr) -> PartitionExpr { diff --git a/src/promql/src/extension_plan/absent.rs b/src/promql/src/extension_plan/absent.rs index 71af413029..fa811b3df7 100644 --- a/src/promql/src/extension_plan/absent.rs +++ b/src/promql/src/extension_plan/absent.rs @@ -12,7 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::cmp::Ordering; use std::collections::HashMap; use std::pin::Pin; @@ -20,6 +19,7 @@ use std::sync::Arc; use std::task::{Context, Poll}; use datafusion::arrow::array::Array; +use datafusion::common::tree_node::TreeNodeRecursion; use datafusion::common::{DFSchemaRef, Result as DataFusionResult}; use datafusion::execution::context::TaskContext; use datafusion::logical_expr::{Expr, LogicalPlan, UserDefinedLogicalNodeCore}; @@ -30,8 +30,8 @@ use datafusion::physical_plan::execution_plan::{Boundedness, EmissionType}; use datafusion::physical_plan::expressions::Column as ColumnExpr; use datafusion::physical_plan::metrics::{BaselineMetrics, ExecutionPlanMetricsSet, MetricsSet}; use datafusion::physical_plan::{ - DisplayAs, DisplayFormatType, Distribution, ExecutionPlan, Partitioning, PlanProperties, - RecordBatchStream, SendableRecordBatchStream, + DisplayAs, DisplayFormatType, Distribution, ExecutionPlan, InputDistributionRequirements, + Partitioning, PhysicalExpr, PlanProperties, RecordBatchStream, SendableRecordBatchStream, }; use datafusion_common::DFSchema; use datafusion_expr::{EmptyRelation, col}; @@ -325,8 +325,11 @@ pub struct AbsentExec { } impl ExecutionPlan for AbsentExec { - fn as_any(&self) -> &dyn Any { - self + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> datafusion_common::Result, + ) -> DataFusionResult { + Ok(TreeNodeRecursion::Continue) } fn schema(&self) -> SchemaRef { @@ -337,8 +340,8 @@ impl ExecutionPlan for AbsentExec { &self.properties } - fn required_input_distribution(&self) -> Vec { - vec![Distribution::SinglePartition] + fn input_distribution_requirements(&self) -> InputDistributionRequirements { + InputDistributionRequirements::new(vec![Distribution::SinglePartition]) } fn required_input_ordering(&self) -> Vec> { diff --git a/src/promql/src/extension_plan/empty_metric.rs b/src/promql/src/extension_plan/empty_metric.rs index 5a7678aab6..91c8a1666a 100644 --- a/src/promql/src/extension_plan/empty_metric.rs +++ b/src/promql/src/extension_plan/empty_metric.rs @@ -12,7 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::collections::HashMap; use std::ops::Div; use std::pin::Pin; @@ -21,21 +20,24 @@ use std::task::{Context, Poll}; use datafusion::arrow::array::ArrayRef; use datafusion::arrow::datatypes::{DataType, TimeUnit}; +use datafusion::catalog::Session; use datafusion::common::arrow::datatypes::Field; use datafusion::common::stats::Precision; +use datafusion::common::tree_node::TreeNodeRecursion; use datafusion::common::{ DFSchema, DFSchemaRef, Result as DataFusionResult, Statistics, TableReference, }; use datafusion::datasource::{MemTable, provider_as_source}; use datafusion::error::DataFusionError; -use datafusion::execution::context::{SessionState, TaskContext}; +use datafusion::execution::context::TaskContext; +use datafusion::logical_expr::physical_planning_context::PhysicalPlanningContext; use datafusion::logical_expr::{ExprSchemable, LogicalPlan, UserDefinedLogicalNodeCore}; -use datafusion::physical_expr::{EquivalenceProperties, PhysicalExprRef}; +use datafusion::physical_expr::{EquivalenceProperties, PhysicalExpr, PhysicalExprRef}; use datafusion::physical_plan::execution_plan::{Boundedness, EmissionType}; use datafusion::physical_plan::metrics::{BaselineMetrics, ExecutionPlanMetricsSet, MetricsSet}; use datafusion::physical_plan::{ DisplayAs, DisplayFormatType, ExecutionPlan, Partitioning, PlanProperties, RecordBatchStream, - SendableRecordBatchStream, + SendableRecordBatchStream, StatisticsArgs, }; use datafusion::physical_planner::PhysicalPlanner; use datafusion::prelude::{Expr, col, lit}; @@ -113,14 +115,20 @@ impl EmptyMetric { pub fn to_execution_plan( &self, - session_state: &SessionState, + session: &dyn Session, physical_planner: &dyn PhysicalPlanner, + planning_ctx: &PhysicalPlanningContext, ) -> DataFusionResult> { let physical_expr = self .expr .as_ref() .map(|expr| { - physical_planner.create_physical_expr(expr, &self.time_index_schema, session_state) + physical_planner.create_physical_expr( + expr, + &self.time_index_schema, + session, + planning_ctx, + ) }) .transpose()?; let result_schema: SchemaRef = self.result_schema.inner().clone(); @@ -224,8 +232,11 @@ pub struct EmptyMetricExec { } impl ExecutionPlan for EmptyMetricExec { - fn as_any(&self) -> &dyn Any { - self + fn apply_expressions( + &self, + f: &mut dyn FnMut(&Arc) -> datafusion_common::Result, + ) -> DataFusionResult { + datafusion::physical_plan::apply_expression_roots(self.expr.iter(), f) } fn schema(&self) -> SchemaRef { @@ -273,9 +284,14 @@ impl ExecutionPlan for EmptyMetricExec { Some(self.metric.clone_inner()) } - fn partition_statistics(&self, partition: Option) -> DataFusionResult { + fn statistics_from_inputs( + &self, + _input_stats: &[Arc], + args: &StatisticsArgs, + ) -> DataFusionResult> { + let partition = args.partition(); if partition.is_some() { - return Ok(Statistics::new_unknown(self.schema().as_ref())); + return Ok(Arc::new(Statistics::new_unknown(self.schema().as_ref()))); } let estimated_row_num = if self.end > self.start { @@ -285,11 +301,11 @@ impl ExecutionPlan for EmptyMetricExec { }; let total_byte_size = estimated_row_num * std::mem::size_of::() as f64; - Ok(Statistics { + Ok(Arc::new(Statistics { num_rows: Precision::Inexact(estimated_row_num.floor() as _), total_byte_size: Precision::Inexact(total_byte_size.floor() as _), column_statistics: Statistics::unknown_column(&self.schema()), - }) + })) } fn name(&self) -> &str { @@ -429,7 +445,11 @@ mod test { ) .unwrap(); let empty_metric_exec = empty_metric - .to_execution_plan(&session_context.state(), &df_default_physical_planner) + .to_execution_plan( + &session_context.state(), + &df_default_physical_planner, + &PhysicalPlanningContext::default(), + ) .unwrap(); let result = @@ -544,7 +564,11 @@ mod test { let empty_metric = EmptyMetric::new(0, 200, 1000, "time".to_string(), "value".to_string(), None).unwrap(); let empty_metric_exec = empty_metric - .to_execution_plan(&session_context.state(), &df_default_physical_planner) + .to_execution_plan( + &session_context.state(), + &df_default_physical_planner, + &PhysicalPlanningContext::default(), + ) .unwrap(); let result = diff --git a/src/promql/src/extension_plan/histogram_fold.rs b/src/promql/src/extension_plan/histogram_fold.rs index 5a5f2f8d46..94769df28a 100644 --- a/src/promql/src/extension_plan/histogram_fold.rs +++ b/src/promql/src/extension_plan/histogram_fold.rs @@ -12,7 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::borrow::Cow; use std::collections::{HashMap, HashSet}; use std::sync::Arc; @@ -25,6 +24,7 @@ use datafusion::arrow::compute::{SortOptions, concat_batches}; use datafusion::arrow::datatypes::{DataType, Float64Type, SchemaRef}; use datafusion::arrow::record_batch::RecordBatch; use datafusion::common::stats::Precision; +use datafusion::common::tree_node::TreeNodeRecursion; use datafusion::common::{DFSchema, DFSchemaRef, Statistics}; use datafusion::error::{DataFusionError, Result as DataFusionResult}; use datafusion::execution::TaskContext; @@ -37,7 +37,8 @@ use datafusion::physical_plan::expressions::{Column as PhyColumn, TryCastExpr as use datafusion::physical_plan::metrics::{BaselineMetrics, ExecutionPlanMetricsSet, MetricsSet}; use datafusion::physical_plan::{ DisplayAs, DisplayFormatType, Distribution, ExecutionPlan, ExecutionPlanProperties, - Partitioning, PhysicalExpr, PlanProperties, RecordBatchStream, SendableRecordBatchStream, + InputDistributionRequirements, Partitioning, PhysicalExpr, PlanProperties, RecordBatchStream, + SendableRecordBatchStream, StatisticsArgs, }; use datafusion::prelude::{Column, Expr}; use datafusion_expr::{EmptyRelation, col}; @@ -526,8 +527,14 @@ pub struct HistogramFoldExec { } impl ExecutionPlan for HistogramFoldExec { - fn as_any(&self) -> &dyn Any { - self + fn apply_expressions( + &self, + f: &mut dyn FnMut(&Arc) -> datafusion_common::Result, + ) -> DataFusionResult { + datafusion::physical_plan::apply_expression_roots( + self.tag_columns.iter().chain(self.partition_exprs.iter()), + f, + ) } fn properties(&self) -> &Arc { @@ -574,8 +581,10 @@ impl ExecutionPlan for HistogramFoldExec { vec![Some(OrderingRequirements::Hard(vec![requirement]))] } - fn required_input_distribution(&self) -> Vec { - vec![Distribution::HashPartitioned(self.partition_exprs.clone())] + fn input_distribution_requirements(&self) -> InputDistributionRequirements { + InputDistributionRequirements::new(vec![Distribution::KeyPartitioned( + self.partition_exprs.clone(), + )]) } fn maintains_input_order(&self) -> Vec { @@ -664,12 +673,16 @@ impl ExecutionPlan for HistogramFoldExec { Some(self.metric.clone_inner()) } - fn partition_statistics(&self, _: Option) -> DataFusionResult { - Ok(Statistics { + fn statistics_from_inputs( + &self, + _input_stats: &[Arc], + _args: &StatisticsArgs, + ) -> DataFusionResult> { + Ok(Arc::new(Statistics { num_rows: Precision::Absent, total_byte_size: Precision::Absent, column_statistics: Statistics::unknown_column(&self.schema()), - }) + })) } fn name(&self) -> &str { diff --git a/src/promql/src/extension_plan/instant_manipulate.rs b/src/promql/src/extension_plan/instant_manipulate.rs index 8619100b4d..45857879d7 100644 --- a/src/promql/src/extension_plan/instant_manipulate.rs +++ b/src/promql/src/extension_plan/instant_manipulate.rs @@ -12,7 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::pin::Pin; use std::sync::Arc; use std::task::{Context, Poll}; @@ -22,6 +21,7 @@ use datafusion::arrow::array::{Array, TimestampMillisecondArray, UInt64Array}; use datafusion::arrow::datatypes::{DataType, SchemaRef}; use datafusion::arrow::record_batch::RecordBatch; use datafusion::common::stats::Precision; +use datafusion::common::tree_node::TreeNodeRecursion; use datafusion::common::{DFSchema, DFSchemaRef, ScalarValue}; use datafusion::error::{DataFusionError, Result as DataFusionResult}; use datafusion::execution::context::TaskContext; @@ -33,8 +33,9 @@ use datafusion::physical_plan::metrics::{ BaselineMetrics, Count, ExecutionPlanMetricsSet, MetricBuilder, MetricValue, MetricsSet, }; use datafusion::physical_plan::{ - DisplayAs, DisplayFormatType, Distribution, ExecutionPlan, PlanProperties, RecordBatchStream, - SendableRecordBatchStream, Statistics, + ChildStats, DisplayAs, DisplayFormatType, ExecutionPlan, InputDistributionRequirements, + PhysicalExpr, PlanProperties, RecordBatchStream, SendableRecordBatchStream, Statistics, + StatisticsArgs, }; use datafusion_expr::col; use datatypes::arrow::compute; @@ -451,8 +452,11 @@ pub struct InstantManipulateExec { } impl ExecutionPlan for InstantManipulateExec { - fn as_any(&self) -> &dyn Any { - self + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> datafusion_common::Result, + ) -> DataFusionResult { + Ok(TreeNodeRecursion::Continue) } fn schema(&self) -> SchemaRef { @@ -463,8 +467,8 @@ impl ExecutionPlan for InstantManipulateExec { &self.properties } - fn required_input_distribution(&self) -> Vec { - self.input.required_input_distribution() + fn input_distribution_requirements(&self) -> InputDistributionRequirements { + self.input.input_distribution_requirements() } // Prevent reordering of input @@ -557,8 +561,16 @@ impl ExecutionPlan for InstantManipulateExec { Some(self.metric.clone_inner()) } - fn partition_statistics(&self, partition: Option) -> DataFusionResult { - let input_stats = self.input.partition_statistics(partition)?; + fn child_stats_requests(&self, partition: Option) -> Vec { + vec![ChildStats::At(partition)] + } + + fn statistics_from_inputs( + &self, + input_stats: &[Arc], + _args: &StatisticsArgs, + ) -> DataFusionResult> { + let input_stats = &input_stats[0]; let estimated_row_num = (self.end - self.start) as f64 / self.interval as f64; let estimated_total_bytes = input_stats @@ -570,12 +582,12 @@ impl ExecutionPlan for InstantManipulateExec { }) .unwrap_or(Precision::Absent); - Ok(Statistics { + Ok(Arc::new(Statistics { num_rows: Precision::Inexact(estimated_row_num.floor() as _), total_byte_size: estimated_total_bytes, // TODO(ruihang): support this column statistics column_statistics: Statistics::unknown_column(&self.schema()), - }) + })) } fn name(&self) -> &str { @@ -818,6 +830,7 @@ mod test { use datafusion::logical_expr::{ EmptyRelation, Extension, LogicalPlan, Projection, UserDefinedLogicalNodeCore, }; + use datafusion::physical_plan::{ChildrenPropertiesMode, ReplaceChildrenOptions}; use datafusion::prelude::SessionContext; use datafusion_expr::col; @@ -1102,7 +1115,10 @@ mod test { ))); let exec = rebuilt .to_execution_plan(empty_exec_input) - .with_new_children(vec![exec_input]) + .replace_children( + vec![exec_input], + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + ) .unwrap(); let output = datafusion::physical_plan::collect(exec, SessionContext::default().task_ctx()) diff --git a/src/promql/src/extension_plan/normalize.rs b/src/promql/src/extension_plan/normalize.rs index c5f8f07369..1ca015b10f 100644 --- a/src/promql/src/extension_plan/normalize.rs +++ b/src/promql/src/extension_plan/normalize.rs @@ -12,7 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::pin::Pin; use std::sync::Arc; use std::task::{Context, Poll}; @@ -20,6 +19,7 @@ use std::task::{Context, Poll}; use common_query::native_histogram::{START_TIMESTAMP_FIELD, native_histogram_arrow_type}; use datafusion::arrow::array::{Array, BooleanArray, StructArray}; use datafusion::arrow::compute; +use datafusion::common::tree_node::TreeNodeRecursion; use datafusion::common::{Column, DFSchema, DFSchemaRef, Result as DataFusionResult, Statistics}; use datafusion::error::DataFusionError; use datafusion::execution::context::TaskContext; @@ -29,8 +29,9 @@ use datafusion::physical_plan::metrics::{ BaselineMetrics, Count, ExecutionPlanMetricsSet, MetricBuilder, MetricValue, MetricsSet, }; use datafusion::physical_plan::{ - DisplayAs, DisplayFormatType, Distribution, ExecutionPlan, PlanProperties, RecordBatchStream, - SendableRecordBatchStream, + ChildStats, DisplayAs, DisplayFormatType, Distribution, ExecutionPlan, + InputDistributionRequirements, PhysicalExpr, PlanProperties, RecordBatchStream, + SendableRecordBatchStream, StatisticsArgs, }; use datafusion_expr::col; use datatypes::arrow::array::TimestampMillisecondArray; @@ -279,27 +280,30 @@ pub struct SeriesNormalizeExec { } impl ExecutionPlan for SeriesNormalizeExec { - fn as_any(&self) -> &dyn Any { - self + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> datafusion_common::Result, + ) -> DataFusionResult { + Ok(TreeNodeRecursion::Continue) } fn schema(&self) -> SchemaRef { self.input.schema() } - fn required_input_distribution(&self) -> Vec { + fn input_distribution_requirements(&self) -> InputDistributionRequirements { if self.tag_columns.is_empty() { - return vec![Distribution::SinglePartition]; + return InputDistributionRequirements::new(vec![Distribution::SinglePartition]); } let schema = self.input.schema(); - vec![Distribution::HashPartitioned( + InputDistributionRequirements::new(vec![Distribution::KeyPartitioned( self.tag_columns .iter() // Safety: the tag column names is verified in the planning phase .map(|tag| Arc::new(ColumnExpr::new_with_schema(tag, &schema).unwrap()) as _) .collect(), - )] + )]) } fn properties(&self) -> &Arc { @@ -356,8 +360,16 @@ impl ExecutionPlan for SeriesNormalizeExec { Some(self.metric.clone_inner()) } - fn partition_statistics(&self, partition: Option) -> DataFusionResult { - self.input.partition_statistics(partition) + fn child_stats_requests(&self, partition: Option) -> Vec { + vec![ChildStats::At(partition)] + } + + fn statistics_from_inputs( + &self, + input_stats: &[Arc], + _args: &StatisticsArgs, + ) -> DataFusionResult> { + Ok(Arc::clone(&input_stats[0])) } fn name(&self) -> &str { diff --git a/src/promql/src/extension_plan/planner.rs b/src/promql/src/extension_plan/planner.rs index 1e6914a78d..50cf4543d5 100644 --- a/src/promql/src/extension_plan/planner.rs +++ b/src/promql/src/extension_plan/planner.rs @@ -15,8 +15,9 @@ use std::sync::Arc; use async_trait::async_trait; +use datafusion::catalog::Session; use datafusion::error::Result as DfResult; -use datafusion::execution::context::SessionState; +use datafusion::logical_expr::physical_planning_context::PhysicalPlanningContext; use datafusion::logical_expr::{LogicalPlan, UserDefinedLogicalNode}; use datafusion::physical_plan::ExecutionPlan; use datafusion::physical_planner::{ExtensionPlanner, PhysicalPlanner}; @@ -36,7 +37,8 @@ impl ExtensionPlanner for PromExtensionPlanner { node: &dyn UserDefinedLogicalNode, _logical_inputs: &[&LogicalPlan], physical_inputs: &[Arc], - session_state: &SessionState, + session: &dyn Session, + planning_ctx: &PhysicalPlanningContext, ) -> DfResult>> { if let Some(node) = node.as_any().downcast_ref::() { Ok(Some(node.to_execution_plan(physical_inputs[0].clone()))) @@ -47,7 +49,11 @@ impl ExtensionPlanner for PromExtensionPlanner { } else if let Some(node) = node.as_any().downcast_ref::() { Ok(Some(node.to_execution_plan(physical_inputs[0].clone()))) } else if let Some(node) = node.as_any().downcast_ref::() { - Ok(Some(node.to_execution_plan(session_state, planner)?)) + Ok(Some(node.to_execution_plan( + session, + planner, + planning_ctx, + )?)) } else if let Some(node) = node.as_any().downcast_ref::() { Ok(Some(node.to_execution_plan(physical_inputs[0].clone())?)) } else if let Some(node) = node.as_any().downcast_ref::() { diff --git a/src/promql/src/extension_plan/range_manipulate.rs b/src/promql/src/extension_plan/range_manipulate.rs index 9bce8e05ad..96a0c97778 100644 --- a/src/promql/src/extension_plan/range_manipulate.rs +++ b/src/promql/src/extension_plan/range_manipulate.rs @@ -12,8 +12,7 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; -use std::collections::{HashMap, HashSet}; +use std::collections::HashSet; use std::pin::Pin; use std::sync::Arc; use std::task::{Context, Poll}; @@ -25,7 +24,8 @@ use datafusion::arrow::datatypes::{DataType, Field, SchemaRef, TimeUnit}; use datafusion::arrow::error::ArrowError; use datafusion::arrow::record_batch::RecordBatch; use datafusion::common::stats::Precision; -use datafusion::common::{DFSchema, DFSchemaRef}; +use datafusion::common::tree_node::TreeNodeRecursion; +use datafusion::common::{DFSchema, DFSchemaRef, TableReference}; use datafusion::error::{DataFusionError, Result as DataFusionResult}; use datafusion::execution::context::TaskContext; use datafusion::logical_expr::{EmptyRelation, Expr, LogicalPlan, UserDefinedLogicalNodeCore}; @@ -34,10 +34,10 @@ use datafusion::physical_plan::metrics::{ BaselineMetrics, Count, ExecutionPlanMetricsSet, MetricBuilder, MetricValue, MetricsSet, }; use datafusion::physical_plan::{ - DisplayAs, DisplayFormatType, Distribution, ExecutionPlan, PlanProperties, RecordBatchStream, - SendableRecordBatchStream, Statistics, + ChildStats, DisplayAs, DisplayFormatType, Distribution, ExecutionPlan, + InputDistributionRequirements, PhysicalExpr, PlanProperties, RecordBatchStream, + SendableRecordBatchStream, Statistics, StatisticsArgs, }; -use datafusion::sql::TableReference; use datafusion_expr::col; use datatypes::timestamp::timestamp_array_to_primitive; use futures::{Stream, StreamExt, ready}; @@ -181,7 +181,7 @@ impl RangeManipulate { Ok(Arc::new(DFSchema::new_with_metadata( new_columns, - HashMap::new(), + input_schema.metadata().clone(), )?)) } @@ -452,8 +452,11 @@ pub struct RangeManipulateExec { } impl ExecutionPlan for RangeManipulateExec { - fn as_any(&self) -> &dyn Any { - self + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> datafusion_common::Result, + ) -> DataFusionResult { + Ok(TreeNodeRecursion::Continue) } fn schema(&self) -> SchemaRef { @@ -472,14 +475,17 @@ impl ExecutionPlan for RangeManipulateExec { vec![&self.input] } - fn required_input_distribution(&self) -> Vec { - let input_requirement = self.input.required_input_distribution(); + fn input_distribution_requirements(&self) -> InputDistributionRequirements { + let input_requirement = self + .input + .input_distribution_requirements() + .into_per_child(); if input_requirement.is_empty() { // if the input is EmptyMetric, its required_input_distribution() is empty so we can't // use its input distribution. - vec![Distribution::UnspecifiedDistribution] + InputDistributionRequirements::new(vec![Distribution::UnspecifiedDistribution]) } else { - input_requirement + InputDistributionRequirements::new(input_requirement) } } @@ -567,8 +573,16 @@ impl ExecutionPlan for RangeManipulateExec { Some(self.metric.clone_inner()) } - fn partition_statistics(&self, partition: Option) -> DataFusionResult { - let input_stats = self.input.partition_statistics(partition)?; + fn child_stats_requests(&self, partition: Option) -> Vec { + vec![ChildStats::At(partition)] + } + + fn statistics_from_inputs( + &self, + input_stats: &[Arc], + _args: &StatisticsArgs, + ) -> DataFusionResult> { + let input_stats = &input_stats[0]; let estimated_row_num = (self.end - self.start) as f64 / self.interval as f64; let estimated_total_bytes = input_stats @@ -580,12 +594,12 @@ impl ExecutionPlan for RangeManipulateExec { }) .unwrap_or_default(); - Ok(Statistics { + Ok(Arc::new(Statistics { num_rows: Precision::Inexact(estimated_row_num as _), total_byte_size: estimated_total_bytes, // TODO(ruihang): support this column statistics column_statistics: Statistics::unknown_column(&self.schema()), - }) + })) } fn name(&self) -> &str { @@ -836,6 +850,7 @@ mod test { use datafusion::physical_expr::Partitioning; use datafusion::physical_plan::execution_plan::{Boundedness, EmissionType}; use datafusion::physical_plan::memory::MemoryStream; + use datafusion::physical_plan::{ChildrenPropertiesMode, ReplaceChildrenOptions}; use datafusion::prelude::SessionContext; use datatypes::arrow::array::TimestampMillisecondArray; use futures::FutureExt; @@ -1228,7 +1243,10 @@ mod test { ))); let exec = rebuilt .to_execution_plan(empty_exec_input) - .with_new_children(vec![exec_input]) + .replace_children( + vec![exec_input], + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + ) .unwrap(); let output = datafusion::physical_plan::collect(exec, SessionContext::default().task_ctx()) diff --git a/src/promql/src/extension_plan/scalar_calculate.rs b/src/promql/src/extension_plan/scalar_calculate.rs index 567fd65b5e..e71333837f 100644 --- a/src/promql/src/extension_plan/scalar_calculate.rs +++ b/src/promql/src/extension_plan/scalar_calculate.rs @@ -12,25 +12,27 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::collections::HashMap; use std::pin::Pin; use std::sync::Arc; use std::task::{Context, Poll}; use datafusion::common::stats::Precision; -use datafusion::common::{DFSchema, DFSchemaRef, Result as DataFusionResult, Statistics}; +use datafusion::common::tree_node::TreeNodeRecursion; +use datafusion::common::{ + DFSchema, DFSchemaRef, Result as DataFusionResult, Statistics, TableReference, +}; use datafusion::error::DataFusionError; use datafusion::execution::context::TaskContext; use datafusion::logical_expr::{EmptyRelation, LogicalPlan, UserDefinedLogicalNodeCore}; use datafusion::physical_expr::EquivalenceProperties; use datafusion::physical_plan::metrics::{BaselineMetrics, ExecutionPlanMetricsSet, MetricsSet}; use datafusion::physical_plan::{ - DisplayAs, DisplayFormatType, Distribution, ExecutionPlan, Partitioning, PlanProperties, - RecordBatchStream, SendableRecordBatchStream, + ChildStats, DisplayAs, DisplayFormatType, Distribution, ExecutionPlan, + InputDistributionRequirements, Partitioning, PhysicalExpr, PlanProperties, RecordBatchStream, + SendableRecordBatchStream, StatisticsArgs, }; use datafusion::prelude::Expr; -use datafusion::sql::TableReference; use datafusion_expr::col; use datatypes::arrow::array::{Array, ArrayRef, Float64Array, TimestampMillisecondArray}; use datatypes::arrow::compute::{CastOptions, cast_with_options, concat_batches}; @@ -129,7 +131,10 @@ impl ScalarCalculate { .output_schema .fields() .iter() - .map(|field| Field::new(field.name(), field.data_type().clone(), field.is_nullable())) + .map(|field| { + Field::new(field.name(), field.data_type().clone(), field.is_nullable()) + .with_metadata(field.metadata().clone()) + }) .collect(); let input_schema = exec_input.schema(); let ts_index = input_schema @@ -138,7 +143,10 @@ impl ScalarCalculate { let val_index = input_schema .index_of(&self.field_column) .map_err(|e| DataFusionError::ArrowError(Box::new(e), None))?; - let schema = Arc::new(Schema::new(fields)); + let schema = Arc::new(Schema::new_with_metadata( + fields, + input_schema.metadata().clone(), + )); let properties = exec_input.properties(); let properties = Arc::new(PlanProperties::new( EquivalenceProperties::new(schema.clone()), @@ -389,8 +397,11 @@ struct ScalarCalculateExec { } impl ExecutionPlan for ScalarCalculateExec { - fn as_any(&self) -> &dyn Any { - self + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> datafusion_common::Result, + ) -> DataFusionResult { + Ok(TreeNodeRecursion::Continue) } fn schema(&self) -> SchemaRef { @@ -405,8 +416,8 @@ impl ExecutionPlan for ScalarCalculateExec { vec![true; self.children().len()] } - fn required_input_distribution(&self) -> Vec { - vec![Distribution::SinglePartition] + fn input_distribution_requirements(&self) -> InputDistributionRequirements { + InputDistributionRequirements::new(vec![Distribution::SinglePartition]) } fn children(&self) -> Vec<&Arc> { @@ -469,8 +480,16 @@ impl ExecutionPlan for ScalarCalculateExec { Some(self.metric.clone_inner()) } - fn partition_statistics(&self, partition: Option) -> DataFusionResult { - let input_stats = self.input.partition_statistics(partition)?; + fn child_stats_requests(&self, partition: Option) -> Vec { + vec![ChildStats::At(partition)] + } + + fn statistics_from_inputs( + &self, + input_stats: &[Arc], + _args: &StatisticsArgs, + ) -> DataFusionResult> { + let input_stats = &input_stats[0]; let estimated_row_num = (self.end - self.start) as f64 / self.interval as f64; let estimated_total_bytes = input_stats @@ -482,12 +501,12 @@ impl ExecutionPlan for ScalarCalculateExec { }) .unwrap_or_default(); - Ok(Statistics { + Ok(Arc::new(Statistics { num_rows: Precision::Inexact(estimated_row_num as _), total_byte_size: estimated_total_bytes, // TODO(ruihang): support this column statistics column_statistics: Statistics::unknown_column(&self.schema()), - }) + })) } fn name(&self) -> &str { diff --git a/src/promql/src/extension_plan/series_divide.rs b/src/promql/src/extension_plan/series_divide.rs index 9e6a46d991..6b1a5ee1da 100644 --- a/src/promql/src/extension_plan/series_divide.rs +++ b/src/promql/src/extension_plan/series_divide.rs @@ -12,7 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::pin::Pin; use std::sync::Arc; use std::task::{Context, Poll}; @@ -20,6 +19,7 @@ use std::task::{Context, Poll}; use datafusion::arrow::array::{Array, ArrayRef, UInt64Array}; use datafusion::arrow::datatypes::{DataType, SchemaRef}; use datafusion::arrow::record_batch::RecordBatch; +use datafusion::common::tree_node::TreeNodeRecursion; use datafusion::common::{DFSchema, DFSchemaRef}; use datafusion::error::Result as DataFusionResult; use datafusion::execution::context::TaskContext; @@ -30,8 +30,8 @@ use datafusion::physical_plan::metrics::{ BaselineMetrics, Count, ExecutionPlanMetricsSet, MetricBuilder, MetricValue, MetricsSet, }; use datafusion::physical_plan::{ - DisplayAs, DisplayFormatType, Distribution, ExecutionPlan, PlanProperties, RecordBatchStream, - SendableRecordBatchStream, + DisplayAs, DisplayFormatType, Distribution, ExecutionPlan, InputDistributionRequirements, + PhysicalExpr, PlanProperties, RecordBatchStream, SendableRecordBatchStream, }; use datafusion_expr::col; use datatypes::arrow::compute; @@ -334,8 +334,11 @@ pub struct SeriesDivideExec { } impl ExecutionPlan for SeriesDivideExec { - fn as_any(&self) -> &dyn Any { - self + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> datafusion_common::Result, + ) -> DataFusionResult { + Ok(TreeNodeRecursion::Continue) } fn schema(&self) -> SchemaRef { @@ -346,18 +349,18 @@ impl ExecutionPlan for SeriesDivideExec { self.input.properties() } - fn required_input_distribution(&self) -> Vec { + fn input_distribution_requirements(&self) -> InputDistributionRequirements { if self.tag_columns.is_empty() { - return vec![Distribution::SinglePartition]; + return InputDistributionRequirements::new(vec![Distribution::SinglePartition]); } let schema = self.input.schema(); - vec![Distribution::HashPartitioned( + InputDistributionRequirements::new(vec![Distribution::KeyPartitioned( self.tag_columns .iter() // Safety: the tag column names is verified in the planning phase .map(|tag| Arc::new(ColumnExpr::new_with_schema(tag, &schema).unwrap()) as _) .collect(), - )] + )]) } fn required_input_ordering(&self) -> Vec> { diff --git a/src/promql/src/extension_plan/union_distinct_on.rs b/src/promql/src/extension_plan/union_distinct_on.rs index a7d8fd4859..774904a27a 100644 --- a/src/promql/src/extension_plan/union_distinct_on.rs +++ b/src/promql/src/extension_plan/union_distinct_on.rs @@ -12,15 +12,16 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::pin::Pin; use std::sync::Arc; use std::task::{Context, Poll}; -use ahash::{HashSet, RandomState}; +use ahash::HashSet; use datafusion::arrow::array::UInt64Array; use datafusion::arrow::datatypes::SchemaRef; use datafusion::arrow::record_batch::RecordBatch; +use datafusion::common::hash_utils::RandomState as FixedState; +use datafusion::common::tree_node::TreeNodeRecursion; use datafusion::common::{DFSchema, DFSchemaRef}; use datafusion::error::{DataFusionError, Result as DataFusionResult}; use datafusion::execution::context::TaskContext; @@ -29,8 +30,9 @@ use datafusion::physical_expr::EquivalenceProperties; use datafusion::physical_plan::execution_plan::{Boundedness, EmissionType}; use datafusion::physical_plan::metrics::{BaselineMetrics, ExecutionPlanMetricsSet, MetricsSet}; use datafusion::physical_plan::{ - DisplayAs, DisplayFormatType, Distribution, ExecutionPlan, Partitioning, PlanProperties, - RecordBatchStream, SendableRecordBatchStream, hash_utils, + DisplayAs, DisplayFormatType, Distribution, ExecutionPlan, InputDistributionRequirements, + Partitioning, PhysicalExpr, PlanProperties, RecordBatchStream, SendableRecordBatchStream, + hash_utils, }; use datafusion_expr::col; use datatypes::arrow::compute; @@ -179,7 +181,7 @@ impl UnionDistinctOn { output_schema, metric: ExecutionPlanMetricsSet::new(), properties, - random_state: RandomState::new(), + random_state: FixedState::with_seed(0), }) } @@ -347,21 +349,27 @@ pub struct UnionDistinctOnExec { metric: ExecutionPlanMetricsSet, properties: Arc, - /// Shared the `RandomState` for the hashing algorithm - random_state: RandomState, + /// Shared deterministic hash state for the hashing algorithm. + random_state: FixedState, } impl ExecutionPlan for UnionDistinctOnExec { - fn as_any(&self) -> &dyn Any { - self + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> datafusion_common::Result, + ) -> DataFusionResult { + Ok(TreeNodeRecursion::Continue) } fn schema(&self) -> SchemaRef { self.output_schema.clone() } - fn required_input_distribution(&self) -> Vec { - vec![Distribution::SinglePartition, Distribution::SinglePartition] + fn input_distribution_requirements(&self) -> InputDistributionRequirements { + InputDistributionRequirements::new(vec![ + Distribution::SinglePartition, + Distribution::SinglePartition, + ]) } fn properties(&self) -> &Arc { @@ -452,7 +460,7 @@ pub struct UnionDistinctOnStream { /// Include time index compare_keys: Vec, output_schema: SchemaRef, - random_state: RandomState, + random_state: FixedState, lhs_signatures: HashSet, hashes: Vec, phase: StreamPhase, @@ -911,12 +919,17 @@ mod test { } impl ExecutionPlan for TestExec { - fn name(&self) -> &str { - "TestExec" + fn apply_expressions( + &self, + _f: &mut dyn FnMut( + &Arc, + ) -> datafusion_common::Result, + ) -> DataFusionResult { + Ok(TreeNodeRecursion::Continue) } - fn as_any(&self) -> &dyn Any { - self + fn name(&self) -> &str { + "TestExec" } fn properties(&self) -> &Arc { @@ -965,7 +978,7 @@ mod test { right: None, compare_keys: vec![1, 0], output_schema, - random_state: RandomState::new(), + random_state: FixedState::with_seed(0), lhs_signatures: HashSet::default(), hashes: Vec::new(), phase: StreamPhase::Left, diff --git a/src/promql/src/functions/native_histogram.rs b/src/promql/src/functions/native_histogram.rs index 2d0fa16b08..1fcb9c553a 100644 --- a/src/promql/src/functions/native_histogram.rs +++ b/src/promql/src/functions/native_histogram.rs @@ -14,7 +14,6 @@ //! Native histogram PromQL helpers. -use std::any::Any; use std::hash::{Hash, Hasher}; use std::mem::size_of; use std::sync::Arc; @@ -170,10 +169,6 @@ impl Hash for NativeHistogramAnnotationUdf { } impl ScalarUDFImpl for NativeHistogramAnnotationUdf { - fn as_any(&self) -> &dyn Any { - self - } - fn name(&self) -> &str { self.name } @@ -1517,10 +1512,6 @@ impl Hash for MixedRangeUdf { } impl ScalarUDFImpl for MixedRangeUdf { - fn as_any(&self) -> &dyn Any { - self - } - fn name(&self) -> &str { self.output.name() } diff --git a/src/promql/src/functions/quantile_aggr.rs b/src/promql/src/functions/quantile_aggr.rs index 6d755cadf7..c818d6220a 100644 --- a/src/promql/src/functions/quantile_aggr.rs +++ b/src/promql/src/functions/quantile_aggr.rs @@ -82,7 +82,6 @@ impl QuantileAccumulator { } let q = match &args.exprs[0] - .as_any() .downcast_ref::() .map(|lit| lit.value()) { diff --git a/src/query/Cargo.toml b/src/query/Cargo.toml index 9e43bb53e4..2a32292614 100644 --- a/src/query/Cargo.toml +++ b/src/query/Cargo.toml @@ -43,6 +43,7 @@ datafusion-expr.workspace = true datafusion-expr-common.workspace = true datafusion-functions.workspace = true datafusion-optimizer.workspace = true +datafusion-pg-catalog.workspace = true datafusion-physical-expr.workspace = true datafusion-proto.workspace = true datafusion-sql.workspace = true diff --git a/src/query/src/analyze.rs b/src/query/src/analyze.rs index dec73e4093..34f1c69ee3 100644 --- a/src/query/src/analyze.rs +++ b/src/query/src/analyze.rs @@ -16,7 +16,6 @@ //! //! The code skeleton is taken from `datafusion/physical-plan/src/analyze.rs` -use std::any::Any; use std::fmt::Display; use std::sync::Arc; @@ -30,11 +29,12 @@ use datafusion::execution::TaskContext; use datafusion::physical_plan::coalesce_partitions::CoalescePartitionsExec; use datafusion::physical_plan::stream::RecordBatchStreamAdapter; use datafusion::physical_plan::{ - DisplayAs, DisplayFormatType, ExecutionPlan, PlanProperties, accept, + ChildrenPropertiesMode, DisplayAs, DisplayFormatType, ExecutionPlan, + InputDistributionRequirements, PlanProperties, ReplaceChildrenOptions, accept, }; use datafusion_common::tree_node::{TreeNode, TreeNodeRecursion}; use datafusion_common::{DataFusionError, assert_eq_or_internal_err, internal_err}; -use datafusion_physical_expr::{Distribution, EquivalenceProperties, Partitioning}; +use datafusion_physical_expr::{EquivalenceProperties, Partitioning, PhysicalExpr}; use futures::StreamExt; use serde::Serialize; use serde_json::{Value, json}; @@ -110,7 +110,6 @@ pub fn analyze_plan_metrics_to_json_value( verbose: bool, ) -> serde_json::Result { let input = plan - .as_any() .downcast_ref::() .map(|exec| exec.input().clone()) .unwrap_or_else(|| plan.clone()); @@ -125,7 +124,7 @@ pub fn analyze_plan_metrics_to_json_value( })); let _ = input.apply(|plan| { - if let Some(merge_scan) = plan.as_any().downcast_ref::() { + if let Some(merge_scan) = plan.downcast_ref::() { for (node, metric) in merge_scan.sub_stage_metrics().into_iter().enumerate() { stages.push(json!({ "stage": 1, @@ -157,11 +156,6 @@ impl ExecutionPlan for DistAnalyzeExec { "DistAnalyzeExec" } - /// Return a reference to Any that can be used for downcasting - fn as_any(&self) -> &dyn Any { - self - } - fn properties(&self) -> &Arc { &self.properties } @@ -170,14 +164,25 @@ impl ExecutionPlan for DistAnalyzeExec { vec![&self.input] } - /// AnalyzeExec is handled specially so this value is ignored - fn required_input_distribution(&self) -> Vec { - vec![] + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> DfResult, + ) -> DfResult { + Ok(TreeNodeRecursion::Continue) } - fn with_new_children( + /// AnalyzeExec is handled specially so this value is ignored + fn input_distribution_requirements(&self) -> InputDistributionRequirements { + // AnalyzeExec is handled specially so this value is ignored. + InputDistributionRequirements::new(vec![ + datafusion_physical_expr::Distribution::UnspecifiedDistribution, + ]) + } + + fn replace_children( self: Arc, mut children: Vec>, + _options: ReplaceChildrenOptions, ) -> DfResult> { assert_eq_or_internal_err!( children.len(), @@ -191,6 +196,17 @@ impl ExecutionPlan for DistAnalyzeExec { ))) } + #[allow(deprecated)] + fn with_new_children( + self: Arc, + children: Vec>, + ) -> DfResult> { + self.replace_children( + children, + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + ) + } + fn execute( &self, partition: usize, @@ -293,7 +309,7 @@ fn create_output_batch( // Find merge scan and append its sub_stage_metrics input.apply(|plan| { - if let Some(merge_scan) = plan.as_any().downcast_ref::() { + if let Some(merge_scan) = plan.downcast_ref::() { let sub_stage_metrics = merge_scan.sub_stage_metrics(); for (node, metric) in sub_stage_metrics.into_iter().enumerate() { builder.append_metric(1, node as _, metrics_to_string(metric, format)?); @@ -421,7 +437,14 @@ mod tests { AnalyzeFormat::TEXT, )); - assert!(ExecutionPlan::with_new_children(analyze, vec![]).is_err()); + assert!( + ExecutionPlan::replace_children( + analyze, + vec![], + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + ) + .is_err() + ); } #[test] @@ -432,14 +455,14 @@ mod tests { AnalyzeFormat::TEXT, )); - let result = ExecutionPlan::with_new_children( + let result = ExecutionPlan::replace_children( analyze, vec![empty_plan("first"), empty_plan("second")], + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), ); if let Ok(plan) = result { let retained = plan - .as_any() .downcast_ref::() .unwrap() .input() @@ -460,8 +483,13 @@ mod tests { )); let replacement = empty_plan("replacement"); - let rebuilt = ExecutionPlan::with_new_children(analyze, vec![replacement]).unwrap(); - let rebuilt = rebuilt.as_any().downcast_ref::().unwrap(); + let rebuilt = ExecutionPlan::replace_children( + analyze, + vec![replacement], + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + ) + .unwrap(); + let rebuilt = rebuilt.downcast_ref::().unwrap(); assert_eq!(rebuilt.input().schema().field(0).name(), "replacement"); } diff --git a/src/query/src/datafusion.rs b/src/query/src/datafusion.rs index 28644b8ccc..63b834bed5 100644 --- a/src/query/src/datafusion.rs +++ b/src/query/src/datafusion.rs @@ -16,6 +16,7 @@ mod error; mod json_expr_planner; +mod pg_oid_alias_expr_planner; mod planner; use std::any::Any; @@ -110,7 +111,7 @@ fn query_load_region_id(plan: &Arc) -> Option { while let Some(plan) = stack.pop() { if plan.name() == REGION_SCAN_EXEC_NAME - && let Some(scan) = plan.as_any().downcast_ref::() + && let Some(scan) = plan.downcast_ref::() && let Some(scan_region_id) = scan.query_load_region_id() { match region_id { @@ -139,7 +140,7 @@ fn query_stat_counters(plan: &Arc) -> Option() + && let Some(scan) = plan.downcast_ref::() && let Some(scan_counters) = scan.query_stat_counters() { match &counters { @@ -519,8 +520,7 @@ impl DatafusionQueryEngine { // let config = state.config_options(); // skip optimize AnalyzeExec plan - let optimized_plan = if let Some(analyze_plan) = plan.as_any().downcast_ref::() - { + let optimized_plan = if let Some(analyze_plan) = plan.downcast_ref::() { let format = if let Some(format) = ctx.query_ctx().explain_format() && format.to_lowercase() == "json" { diff --git a/src/query/src/datafusion/pg_oid_alias_expr_planner.rs b/src/query/src/datafusion/pg_oid_alias_expr_planner.rs new file mode 100644 index 0000000000..6eb704b58d --- /dev/null +++ b/src/query/src/datafusion/pg_oid_alias_expr_planner.rs @@ -0,0 +1,340 @@ +// Copyright 2023 Greptime Team +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +use arrow_schema::DataType; +use datafusion_common::{DFSchema, ExprSchema, Result, ScalarValue}; +use datafusion_expr::expr::BinaryExpr; +use datafusion_expr::planner::{ExprPlanner, PlannerResult, RawBinaryExpr}; +use datafusion_expr::{Expr, Operator}; +use datafusion_pg_catalog::pg_catalog::oid_field::{OID_ALIAS_KEY, kind}; +use sqlparser::ast::BinaryOperator; + +/// Rewrites PostgreSQL's regproc zero sentinel before DataFusion type coercion. +#[derive(Debug)] +pub(crate) struct PgOidAliasExprPlanner; + +impl ExprPlanner for PgOidAliasExprPlanner { + fn plan_binary_op( + &self, + expr: RawBinaryExpr, + schema: &DFSchema, + ) -> Result> { + let RawBinaryExpr { + op, + mut left, + mut right, + } = expr; + + let operator = match op { + BinaryOperator::Eq => Operator::Eq, + BinaryOperator::NotEq => Operator::NotEq, + _ => return Ok(PlannerResult::Original(RawBinaryExpr { op, left, right })), + }; + + let (column, zero_on_left) = match (&left, &right) { + (Expr::Literal(value, _), Expr::Column(column)) if is_integral_zero(value) => { + (column, true) + } + (Expr::Column(column), Expr::Literal(value, _)) if is_integral_zero(value) => { + (column, false) + } + _ => return Ok(PlannerResult::Original(RawBinaryExpr { op, left, right })), + }; + + // A raw SQL column is resolved against the schema before the default + // coercion planner runs. Do not infer alias semantics from casts or any + // other expression shape. + let Ok(field) = schema.field_from_column(column) else { + return Ok(PlannerResult::Original(RawBinaryExpr { op, left, right })); + }; + if field.metadata().get(OID_ALIAS_KEY).map(String::as_str) != Some(kind::REGPROC) { + return Ok(PlannerResult::Original(RawBinaryExpr { op, left, right })); + } + + let Some(sentinel) = regproc_zero_sentinel(field.data_type()) else { + return Ok(PlannerResult::Original(RawBinaryExpr { op, left, right })); + }; + + let sentinel = Expr::Literal(sentinel, None); + if zero_on_left { + left = sentinel; + } else { + right = sentinel; + } + + Ok(PlannerResult::Planned(Expr::BinaryExpr(BinaryExpr::new( + Box::new(left), + operator, + Box::new(right), + )))) + } +} + +fn is_integral_zero(value: &ScalarValue) -> bool { + matches!( + value, + ScalarValue::Int8(Some(0)) + | ScalarValue::Int16(Some(0)) + | ScalarValue::Int32(Some(0)) + | ScalarValue::Int64(Some(0)) + | ScalarValue::UInt8(Some(0)) + | ScalarValue::UInt16(Some(0)) + | ScalarValue::UInt32(Some(0)) + | ScalarValue::UInt64(Some(0)) + ) +} + +fn regproc_zero_sentinel(data_type: &DataType) -> Option { + match data_type { + DataType::Utf8 => Some(ScalarValue::Utf8(Some("-".to_string()))), + DataType::LargeUtf8 => Some(ScalarValue::LargeUtf8(Some("-".to_string()))), + DataType::Utf8View => Some(ScalarValue::Utf8View(Some("-".to_string()))), + _ => None, + } +} + +#[cfg(test)] +mod tests { + use std::collections::HashMap; + use std::sync::Arc; + + use arrow_schema::{Field, Fields}; + use datafusion_common::Column; + use datafusion_expr::ExprSchemable; + use datafusion_expr::expr::Cast; + use datafusion_expr::simplify::SimplifyContext; + use datafusion_optimizer::simplify_expressions::ExprSimplifier; + + use super::*; + + fn schema(data_type: DataType, alias: Option<&str>) -> DFSchema { + let mut field = Field::new("typreceive", data_type, true); + if let Some(alias) = alias { + field = field.with_metadata(HashMap::from([( + OID_ALIAS_KEY.to_string(), + alias.to_string(), + )])); + } + DFSchema::from_unqualified_fields(Fields::from(vec![field]), HashMap::new()).unwrap() + } + + fn column() -> Expr { + Expr::Column(Column::new_unqualified("typreceive")) + } + + fn plan(expr: RawBinaryExpr, schema: &DFSchema) -> PlannerResult { + PgOidAliasExprPlanner.plan_binary_op(expr, schema).unwrap() + } + + fn assert_planned_sentinel( + planned: PlannerResult, + operator: Operator, + zero_on_left: bool, + sentinel: ScalarValue, + ) -> Expr { + let PlannerResult::Planned(Expr::BinaryExpr(expr)) = planned else { + panic!("expected a planned binary expression"); + }; + assert_eq!(expr.op, operator); + let literal = Expr::Literal(sentinel, None); + if zero_on_left { + assert_eq!(expr.left.as_ref(), &literal); + assert_eq!(expr.right.as_ref(), &column()); + } else { + assert_eq!(expr.left.as_ref(), &column()); + assert_eq!(expr.right.as_ref(), &literal); + } + Expr::BinaryExpr(expr) + } + + #[test] + fn rewrites_zero_regproc_comparisons_in_both_operand_orders() { + let schema = schema(DataType::Utf8, Some(kind::REGPROC)); + + for (sql_operator, operator) in [ + (BinaryOperator::Eq, Operator::Eq), + (BinaryOperator::NotEq, Operator::NotEq), + ] { + for zero_on_left in [true, false] { + let zero = Expr::Literal(ScalarValue::Int64(Some(0)), None); + let (left, right) = if zero_on_left { + (zero, column()) + } else { + (column(), zero) + }; + let planned = assert_planned_sentinel( + plan( + RawBinaryExpr { + op: sql_operator.clone(), + left, + right, + }, + &schema, + ), + operator, + zero_on_left, + ScalarValue::Utf8(Some("-".to_string())), + ); + assert!(planned.nullable(&schema).unwrap()); + } + } + } + + #[test] + fn rewrites_every_integral_zero_with_the_column_string_storage_type() { + let zero_literals = [ + ScalarValue::Int8(Some(0)), + ScalarValue::Int16(Some(0)), + ScalarValue::Int32(Some(0)), + ScalarValue::Int64(Some(0)), + ScalarValue::UInt8(Some(0)), + ScalarValue::UInt16(Some(0)), + ScalarValue::UInt32(Some(0)), + ScalarValue::UInt64(Some(0)), + ]; + let string_types = [ + (DataType::Utf8, ScalarValue::Utf8(Some("-".to_string()))), + ( + DataType::LargeUtf8, + ScalarValue::LargeUtf8(Some("-".to_string())), + ), + ( + DataType::Utf8View, + ScalarValue::Utf8View(Some("-".to_string())), + ), + ]; + + for (data_type, sentinel) in string_types { + let schema = schema(data_type, Some(kind::REGPROC)); + for zero in &zero_literals { + assert_planned_sentinel( + plan( + RawBinaryExpr { + op: BinaryOperator::Eq, + left: column(), + right: Expr::Literal(zero.clone(), None), + }, + &schema, + ), + Operator::Eq, + false, + sentinel.clone(), + ); + } + } + } + + #[test] + fn leaves_non_matching_comparisons_untouched() { + let regproc = schema(DataType::Utf8, Some(kind::REGPROC)); + let int32_regproc = schema(DataType::Int32, Some(kind::REGPROC)); + let untagged = schema(DataType::Utf8, None); + let regtype = schema(DataType::Utf8, Some(kind::REGTYPE)); + + let cases = [ + ( + RawBinaryExpr { + op: BinaryOperator::Eq, + left: column(), + right: Expr::Literal(ScalarValue::Int64(Some(1)), None), + }, + ®proc, + ), + ( + RawBinaryExpr { + op: BinaryOperator::Eq, + left: column(), + right: Expr::Literal(ScalarValue::Int64(None), None), + }, + ®proc, + ), + ( + RawBinaryExpr { + op: BinaryOperator::Lt, + left: column(), + right: Expr::Literal(ScalarValue::Int64(Some(0)), None), + }, + ®proc, + ), + ( + RawBinaryExpr { + op: BinaryOperator::Eq, + left: column(), + right: Expr::Literal(ScalarValue::Int64(Some(0)), None), + }, + &int32_regproc, + ), + ( + RawBinaryExpr { + op: BinaryOperator::Eq, + left: column(), + right: Expr::Literal(ScalarValue::Int64(Some(0)), None), + }, + &untagged, + ), + ( + RawBinaryExpr { + op: BinaryOperator::Eq, + left: column(), + right: Expr::Literal(ScalarValue::Int64(Some(0)), None), + }, + ®type, + ), + ]; + + for (expr, schema) in cases { + assert!(matches!(plan(expr, schema), PlannerResult::Original(_))); + } + } + + #[test] + fn leaves_casts_and_the_adbc_array_receiver_predicate_untouched() { + let schema = schema(DataType::Utf8, Some(kind::REGPROC)); + let cast = Expr::Cast(Cast::new(Box::new(column()), DataType::Utf8)); + let expr = RawBinaryExpr { + op: BinaryOperator::NotEq, + left: cast, + right: Expr::Literal(ScalarValue::Utf8(Some("array_recv".to_string())), None), + }; + + assert!(matches!(plan(expr, &schema), PlannerResult::Original(_))); + } + + #[test] + fn type_coercion_keeps_regproc_as_a_string_after_the_rewrite() { + let schema = Arc::new(schema(DataType::Utf8, Some(kind::REGPROC))); + let planned = assert_planned_sentinel( + plan( + RawBinaryExpr { + op: BinaryOperator::NotEq, + left: column(), + right: Expr::Literal(ScalarValue::Int64(Some(0)), None), + }, + &schema, + ), + Operator::NotEq, + false, + ScalarValue::Utf8(Some("-".to_string())), + ); + let simplifier = ExprSimplifier::new( + SimplifyContext::builder() + .with_schema(schema.clone()) + .build(), + ); + let coerced = simplifier.coerce(planned, &schema).unwrap(); + + assert!(!format!("{coerced}").contains("CAST")); + assert!(!format!("{coerced:?}").contains("Int64")); + } +} diff --git a/src/query/src/datafusion/planner.rs b/src/query/src/datafusion/planner.rs index 856f2189d2..342209a638 100644 --- a/src/query/src/datafusion/planner.rs +++ b/src/query/src/datafusion/planner.rs @@ -19,7 +19,8 @@ use std::sync::Arc; use arrow_schema::DataType; use catalog::table_source::DfTableSourceProvider; use common_function::function::FunctionContext; -use datafusion::common::TableReference; +use datafusion::catalog::TableFunctionArgs; +use datafusion::common::{DFSchema, TableReference}; use datafusion::datasource::cte_worktable::CteWorkTable; use datafusion::datasource::file_format::{FileFormatFactory, format_as_file_type}; use datafusion::datasource::provider_as_source; @@ -33,12 +34,13 @@ use datafusion_common::config::ConfigOptions; use datafusion_common::file_options::file_type::FileType; use datafusion_expr::planner::{ExprPlanner, TypePlanner}; use datafusion_expr::var_provider::is_system_variables; -use datafusion_expr::{AggregateUDF, ScalarUDF, TableSource, WindowUDF}; +use datafusion_expr::{AggregateUDF, HigherOrderUDF, ScalarUDF, TableSource, WindowUDF}; use datafusion_sql::parser::Statement as DfStatement; use session::context::QueryContextRef; use snafu::{Location, ResultExt}; use crate::datafusion::json_expr_planner::JsonExprPlanner; +use crate::datafusion::pg_oid_alias_expr_planner::PgOidAliasExprPlanner; use crate::error::{CatalogSnafu, Result}; use crate::query_engine::{DefaultPlanDecoder, QueryEngineState}; @@ -90,6 +92,7 @@ impl DfContextProviderAdapter { let mut expr_planners = SessionStateDefaults::default_expr_planners(); expr_planners.insert(0, Arc::new(JsonExprPlanner)); + expr_planners.insert(0, Arc::new(PgOidAliasExprPlanner)); Ok(Self { engine_state, @@ -161,6 +164,13 @@ impl ContextProvider for DfContextProviderAdapter { ) } + fn get_higher_order_meta(&self, name: &str) -> Option> { + self.session_state + .higher_order_functions() + .get(name) + .cloned() + } + fn get_aggregate_meta(&self, name: &str) -> Option> { self.engine_state.aggr_function(name).map_or_else( || self.session_state.aggregate_functions().get(name).cloned(), @@ -200,6 +210,14 @@ impl ContextProvider for DfContextProviderAdapter { names } + fn higher_order_function_names(&self) -> Vec { + self.session_state + .higher_order_functions() + .keys() + .cloned() + .collect() + } + fn udaf_names(&self) -> Vec { let mut names = self.engine_state.aggr_names(); names.extend(self.session_state.aggregate_functions().keys().cloned()); @@ -228,22 +246,47 @@ impl ContextProvider for DfContextProviderAdapter { name: &str, args: Vec, ) -> DfResult> { - if let Some(tbl_func) = self.engine_state.table_function(name) { - let provider = tbl_func.create_table_provider(&args)?; - Ok(provider_as_source(provider)) + // Constant-fold the args before resolving the table function. DataFusion's + // SQL planner does not fold table-function arguments (constant folding + // happens later, in the analyzer), but table functions such as + // `generate_series`/`range` are resolved during planning and require + // literal bounds. Folding here lets immutable-UDF bounds like + // `array_upper(ARRAY[...], 1)` reach them as concrete literals. + // Non-constant args are returned unchanged by the simplifier. + let simplify_info = datafusion_expr::simplify::SimplifyContext::builder() + .with_config_options(Arc::clone(self.session_state.config_options())) + .with_query_execution_start_time( + self.session_state + .execution_props() + .query_execution_start_time, + ) + .build(); + let simplifier = + datafusion_optimizer::simplify_expressions::ExprSimplifier::new(simplify_info); + let schema = DFSchema::empty(); + let args = args + .into_iter() + .map(|arg| { + simplifier + .coerce(arg, &schema) + .and_then(|arg| simplifier.simplify(arg)) + }) + .collect::>>()?; + let table_args = TableFunctionArgs::new(&args, &self.session_state); + let tbl_func = if let Some(tbl_func) = self.engine_state.table_function(name) { + tbl_func } else { - let tbl_func = self - .session_state + self.session_state .table_functions() .get(name) .cloned() .ok_or_else(|| { DataFusionError::Plan(format!("table function '{name}' not found")) - })?; - let provider = tbl_func.create_table_provider(&args)?; + })? + }; + let provider = tbl_func.create_table_provider_with_args(table_args)?; - Ok(provider_as_source(provider)) - } + Ok(provider_as_source(provider)) } fn create_cte_work_table( @@ -260,6 +303,154 @@ impl ContextProvider for DfContextProviderAdapter { } fn get_type_planner(&self) -> Option> { - None + // Provide the SQL planner with Postgres oid-alias type names + // (`regclass`, `regproc`, `regtype`, `regnamespace`, `oid`, ...) and + // `pg_catalog.`-qualified builtins. DataFusion rejects these as + // "Unsupported SQL type" otherwise. The planner maps each to its Arrow + // type so reverse / column-operand casts like `prorettype::regtype::text` + // parse. Forward name->oid casts (`'x'::regclass`) are resolved earlier, + // at SQL-parse time, by the `PostgresCompatibilityParser`'s built-in + // `RewriteRegCastToSubquery` rule. + // Stateless, so a fresh instance per query is cheap. + Some(Arc::new( + datafusion_pg_catalog::pg_catalog::oid_type_planner::PgOidTypePlanner, + )) + } +} + +#[cfg(test)] +mod tests { + use std::sync::atomic::{AtomicBool, Ordering}; + use std::sync::{Arc, Mutex}; + + use common_base::Plugins; + use datafusion::catalog::{TableFunction, TableFunctionArgs, TableFunctionImpl, TableProvider}; + use datafusion::datasource::MemTable; + use datafusion::execution::SessionStateBuilder; + use datafusion::execution::context::{SessionConfig, SessionContext, SessionState}; + use datafusion_common::ScalarValue; + use datafusion_expr::expr::BinaryExpr; + use datafusion_expr::{Expr, Operator, lit}; + use session::context::QueryContext; + + use super::*; + use crate::options::QueryOptions; + + #[derive(Debug, Default)] + struct RecordingTableFunction { + called: AtomicBool, + args: Mutex>, + target_partitions: Mutex>, + } + + impl TableFunctionImpl for RecordingTableFunction { + fn call_with_args(&self, args: TableFunctionArgs) -> DfResult> { + let session_state = args + .session() + .as_any() + .downcast_ref::() + .expect("table function must receive the SessionState"); + *self.args.lock().unwrap() = args.exprs().to_vec(); + *self.target_partitions.lock().unwrap() = + Some(session_state.config().target_partitions()); + self.called.store(true, Ordering::SeqCst); + + Ok(Arc::new(MemTable::try_new( + Arc::new(arrow_schema::Schema::empty()), + vec![vec![]], + )?)) + } + } + + fn query_engine_state() -> Arc { + Arc::new(QueryEngineState::new( + catalog::memory::new_memory_catalog_manager().unwrap(), + None, + None, + None, + None, + None, + false, + Plugins::default(), + QueryOptions::default(), + )) + } + + async fn context_provider( + engine_state: Arc, + session_state: SessionState, + ) -> DfContextProviderAdapter { + DfContextProviderAdapter::try_new(engine_state, session_state, None, QueryContext::arc()) + .await + .unwrap() + } + + fn plus(left: Expr, right: Expr) -> Expr { + Expr::BinaryExpr(BinaryExpr { + left: Box::new(left), + op: Operator::Plus, + right: Box::new(right), + }) + } + + #[tokio::test] + async fn table_function_arguments_are_folded_before_engine_function_creation() { + let engine_state = query_engine_state(); + let function = Arc::new(RecordingTableFunction::default()); + engine_state.register_table_function(Arc::new(TableFunction::new( + "capture_engine_args".to_string(), + function.clone(), + ))); + let provider = context_provider(engine_state.clone(), engine_state.session_state()).await; + + provider + .get_table_function_source("capture_engine_args", vec![plus(lit(1_i64), lit(2_i64))]) + .unwrap(); + + assert!(function.called.load(Ordering::SeqCst)); + assert_eq!( + *function.args.lock().unwrap(), + vec![Expr::Literal(ScalarValue::Int64(Some(3)), None)] + ); + } + + #[tokio::test] + async fn table_function_argument_simplification_errors_are_propagated() { + let engine_state = query_engine_state(); + let function = Arc::new(RecordingTableFunction::default()); + engine_state.register_table_function(Arc::new(TableFunction::new( + "reject_invalid_args".to_string(), + function.clone(), + ))); + let provider = context_provider(engine_state.clone(), engine_state.session_state()).await; + + let error = match provider + .get_table_function_source("reject_invalid_args", vec![plus(lit(true), lit(1_i64))]) + { + Ok(_) => panic!("invalid table-function argument must fail planning"), + Err(error) => error, + }; + + assert!(!error.to_string().is_empty()); + assert!(!function.called.load(Ordering::SeqCst)); + } + + #[tokio::test] + async fn session_table_function_receives_table_function_args_session() { + let engine_state = query_engine_state(); + let session_state = SessionStateBuilder::new_from_existing(engine_state.session_state()) + .with_config(SessionConfig::new().with_target_partitions(7)) + .build(); + let session_context = SessionContext::new_with_state(session_state); + let function = Arc::new(RecordingTableFunction::default()); + session_context.register_udtf("capture_session_args", function.clone()); + let provider = context_provider(engine_state, session_context.state()).await; + + provider + .get_table_function_source("capture_session_args", vec![lit(1_i64)]) + .unwrap(); + + assert!(function.called.load(Ordering::SeqCst)); + assert_eq!(*function.target_partitions.lock().unwrap(), Some(7)); } } diff --git a/src/query/src/dist_plan/analyzer.rs b/src/query/src/dist_plan/analyzer.rs index 2cd32801d6..d928671fc5 100644 --- a/src/query/src/dist_plan/analyzer.rs +++ b/src/query/src/dist_plan/analyzer.rs @@ -668,13 +668,9 @@ impl PlanRewriter { } if let LogicalPlan::TableScan(table_scan) = plan - && let Some(source) = table_scan - .source - .as_any() - .downcast_ref::() + && let Some(source) = table_scan.source.downcast_ref::() && let Some(provider) = source .table_provider - .as_any() .downcast_ref::() { let table = provider.table(); diff --git a/src/query/src/dist_plan/analyzer/fallback.rs b/src/query/src/dist_plan/analyzer/fallback.rs index 79dfaf904b..86c7f50628 100644 --- a/src/query/src/dist_plan/analyzer/fallback.rs +++ b/src/query/src/dist_plan/analyzer/fallback.rs @@ -45,14 +45,11 @@ impl TreeNodeRewriter for FallbackPlanRewriter { plan: Self::Node, ) -> DfResult> { if let LogicalPlan::TableScan(table_scan) = &plan { - let partition_cols = if let Some(source) = table_scan - .source - .as_any() - .downcast_ref::() + let partition_cols = if let Some(source) = + table_scan.source.downcast_ref::() { if let Some(provider) = source .table_provider - .as_any() .downcast_ref::() { if provider.table().table_type() == TableType::Base { diff --git a/src/query/src/dist_plan/analyzer/test.rs b/src/query/src/dist_plan/analyzer/test.rs index 5309d93a57..36ea83fada 100644 --- a/src/query/src/dist_plan/analyzer/test.rs +++ b/src/query/src/dist_plan/analyzer/test.rs @@ -31,15 +31,15 @@ use datafusion::functions_aggregate::min_max::{max, min}; use datafusion::functions_nested::expr_fn::make_array; use datafusion::prelude::SessionContext; use datafusion_common::tree_node::TreeNodeRecursion; -use datafusion_common::{ExprSchema, JoinType, ScalarValue}; +use datafusion_common::{ExprSchema, JoinType, ScalarValue, TableReference}; use datafusion_expr::expr::{Exists, ScalarFunction}; +use datafusion_expr::utils::split_conjunction; use datafusion_expr::{ AggregateUDF, Expr, ExprSchemable as _, Extension, LogicalPlanBuilder, Operator, Subquery, binary_expr, col, lit, }; use datafusion_functions::datetime::date_bin; use datafusion_functions::datetime::expr_fn::now; -use datafusion_sql::TableReference; use datatypes::data_type::ConcreteDataType; use datatypes::schema::{ColumnSchema, SchemaBuilder, SchemaRef}; use futures::Stream; @@ -302,6 +302,53 @@ fn find_merge_scan(plan: &LogicalPlan) -> Option<&MergeScanLogicalPlan> { plan.inputs().into_iter().find_map(find_merge_scan) } +fn find_table_scan<'a>( + plan: &'a LogicalPlan, + table_name: &str, +) -> Option<&'a datafusion_expr::logical_plan::TableScan> { + if let LogicalPlan::TableScan(table_scan) = plan + && table_scan.table_name.to_string() == table_name + { + return Some(table_scan); + } + + plan.inputs() + .into_iter() + .find_map(|input| find_table_scan(input, table_name)) +} + +fn find_merge_scan_for_table<'a>( + plan: &'a LogicalPlan, + table_name: &str, +) -> Option<&'a MergeScanLogicalPlan> { + if let LogicalPlan::Extension(extension) = plan + && let Some(merge_scan) = extension + .node + .as_any() + .downcast_ref::() + && find_table_scan(merge_scan.input(), table_name).is_some() + { + return Some(merge_scan); + } + + plan.inputs() + .into_iter() + .find_map(|input| find_merge_scan_for_table(input, table_name)) +} + +fn has_filter_above_table_scan(plan: &LogicalPlan, table_name: &str, predicate: &Expr) -> bool { + if let LogicalPlan::Filter(filter) = plan + && split_conjunction(&filter.predicate).contains(&predicate) + && find_table_scan(filter.input.as_ref(), table_name).is_some() + { + return true; + } + + plan.inputs() + .into_iter() + .any(|input| has_filter_above_table_scan(input, table_name, predicate)) +} + #[test] fn frontend_only_histogram_folds_stay_above_merge_scan() { let table = TestTable::table_with_name(0, "t".to_string()); @@ -2586,34 +2633,37 @@ fn test_join_side_local_filter_pushdown_into_merge_scan() { let result = DistPlannerAnalyzer {}.analyze(plan, &config).unwrap(); assert_remote_table_scan_filters_are_safe(&result); - let plan_str = result.to_string(); - // After PushDownFilter runs, the predicate `t1.pk1 = Utf8("v")` should appear - // inside the left MergeScan's remote_input. The pre-MergeScan optimizer may - // combine it with join-derived IS NOT NULL pushdowns, so it may not appear as - // a standalone Filter: line. It must still be in TableScan partial_filters - // and below the Inner Join. + let predicate = col("t1.pk1").eq(lit("v")); + let t1_remote_input = find_merge_scan_for_table(&result, "t1") + .expect("expected MergeScan for t1") + .input(); + let t1_scan = find_table_scan(t1_remote_input, "t1").expect("expected t1 TableScan"); assert!( - plan_str.contains("t1.pk1 = Utf8(\"v\")"), - "Expected predicate t1.pk1 = Utf8(\"v\") in plan, got:\n{plan_str}" - ); - assert!( - plan_str.contains( - "TableScan: t1, partial_filters=[t1.pk1 = Utf8(\"v\"), t1.number IS NOT NULL]" - ), - "Expected t1 TableScan partial_filters to contain pushed predicate, got:\n{plan_str}" + t1_scan + .filters + .iter() + .flat_map(|filter| split_conjunction(filter)) + .any(|filter| filter == &predicate), + "expected t1 TableScan to contain the pushed predicate: {t1_remote_input}" ); - // Find the position of the filter and verify it appears after a MergeScan - // opening (i.e., inside remote_input) rather than before the Join. - let filter_pos = plan_str - .find("TableScan: t1, partial_filters=[t1.pk1 = Utf8(\"v\"), t1.number IS NOT NULL]") - .unwrap(); - let join_pos = plan_str.find("Inner Join").unwrap(); - // The filter should be after the Join (meaning it was pushed down below the Join, - // into a MergeScan's remote_input) + // Inexact provider pushdown must retain the predicate in an ancestor Filter. assert!( - filter_pos > join_pos, - "Filter should be pushed below Join (into MergeScan remote_input), but found before Join" + has_filter_above_table_scan(t1_remote_input, "t1", &predicate), + "expected an ancestor Filter for t1 to retain the pushed predicate: {t1_remote_input}" + ); + + let t2_remote_input = find_merge_scan_for_table(&result, "t2") + .expect("expected MergeScan for t2") + .input(); + assert!( + !find_table_scan(t2_remote_input, "t2") + .expect("expected t2 TableScan") + .filters + .iter() + .flat_map(|filter| split_conjunction(filter)) + .any(|filter| filter == &predicate), + "t2 TableScan must not contain t1's predicate: {t2_remote_input}" ); } diff --git a/src/query/src/dist_plan/commutativity.rs b/src/query/src/dist_plan/commutativity.rs index 1a70fd8e1d..bc41516ba3 100644 --- a/src/query/src/dist_plan/commutativity.rs +++ b/src/query/src/dist_plan/commutativity.rs @@ -328,6 +328,9 @@ impl Categorizer { | Expr::WindowFunction(_) | Expr::InSubquery(_) | Expr::ScalarSubquery(_) + | Expr::HigherOrderFunction(_) + | Expr::Lambda(_) + | Expr::LambdaVariable(_) | Expr::Wildcard { .. } => Commutativity::Unimplemented, Expr::Alias(alias) => Self::check_expr(&alias.expr), diff --git a/src/query/src/dist_plan/dyn_filter_bridge.rs b/src/query/src/dist_plan/dyn_filter_bridge.rs index fcde3dc008..4761c7aea9 100644 --- a/src/query/src/dist_plan/dyn_filter_bridge.rs +++ b/src/query/src/dist_plan/dyn_filter_bridge.rs @@ -346,10 +346,6 @@ mod tests { impl Eq for UnserializableExpr {} impl datafusion_physical_expr::PhysicalExpr for UnserializableExpr { - fn as_any(&self) -> &dyn Any { - self - } - fn data_type( &self, _input_schema: &arrow_schema::Schema, @@ -766,7 +762,7 @@ mod tests { captured_dyn_filters[0].filter_id.to_string() ); assert_eq!(decoded_children.len(), 1); - assert!(decoded_children[0].as_any().is::()); + assert!(decoded_children[0].is::()); } #[test] diff --git a/src/query/src/dist_plan/merge_scan.rs b/src/query/src/dist_plan/merge_scan.rs index 26474b7ca2..e7e78bfec6 100644 --- a/src/query/src/dist_plan/merge_scan.rs +++ b/src/query/src/dist_plan/merge_scan.rs @@ -12,7 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; #[cfg(test)] use std::cell::Cell; use std::sync::{Arc, Mutex}; @@ -40,9 +39,10 @@ use datafusion::physical_plan::metrics::{ use datafusion::physical_plan::stream::RecordBatchStreamAdapter; use datafusion::physical_plan::{ DisplayAs, DisplayFormatType, ExecutionPlan, Partitioning, PlanProperties, - SendableRecordBatchStream, + SendableRecordBatchStream, apply_expression_roots, }; use datafusion_common::stats::Precision; +use datafusion_common::tree_node::TreeNodeRecursion; use datafusion_common::{Column as ColumnExpr, DFSchemaRef, DataFusionError, Result, Statistics}; use datafusion_expr::{Expr, Extension, FetchType, LogicalPlan, UserDefinedLogicalNodeCore}; use datafusion_physical_expr::expressions::Column; @@ -911,7 +911,7 @@ impl MergeScanExec { } pub fn try_with_new_distribution(&self, distribution: Distribution) -> Option { - let Distribution::HashPartitioned(hash_exprs) = distribution else { + let Distribution::KeyPartitioned(hash_exprs) = distribution else { // not applicable return None; }; @@ -926,8 +926,7 @@ impl MergeScanExec { let hash_expr_col_names: HashSet<_> = hash_exprs .iter() .filter_map(|expr| { - expr.as_any() - .downcast_ref::() + expr.downcast_ref::() .map(|col_expr| col_expr.name()) }) .collect(); @@ -949,8 +948,7 @@ impl MergeScanExec { let overlaps: Vec<_> = hash_exprs .iter() .filter(|expr| { - expr.as_any() - .downcast_ref::() + expr.downcast_ref::() .is_some_and(|col_expr| all_partition_col_aliases.contains(col_expr.name())) }) .cloned() @@ -1153,10 +1151,6 @@ impl Drop for PartitionMetrics { } impl ExecutionPlan for MergeScanExec { - fn as_any(&self) -> &dyn Any { - self - } - fn schema(&self) -> ArrowSchemaRef { self.arrow_schema.clone() } @@ -1169,6 +1163,24 @@ impl ExecutionPlan for MergeScanExec { vec![] } + fn apply_expressions( + &self, + f: &mut dyn FnMut( + &Arc, + ) -> Result, + ) -> Result { + let captured_remote_dyn_filters = self.captured_remote_dyn_filters(); + apply_expression_roots( + captured_remote_dyn_filters + .into_iter() + .map(|captured_dyn_filter| { + captured_dyn_filter.alive_dyn_filter + as Arc + }), + f, + ) + } + // DataFusion will swap children unconditionally. // But since this node is leaf node, it's safe to just return self. fn with_new_children( @@ -1250,14 +1262,14 @@ impl ExecutionPlan for MergeScanExec { Some(self.metric.clone_inner()) } - fn partition_statistics(&self, partition: Option) -> Result { + fn partition_statistics(&self, partition: Option) -> Result> { if partition.is_some() { - return Ok(Statistics::new_unknown(&self.arrow_schema)); + return Ok(Arc::new(Statistics::new_unknown(&self.arrow_schema))); } let mut statistics = Statistics::new_unknown(&self.arrow_schema); statistics.num_rows = self.estimated_num_rows(); - Ok(statistics) + Ok(Arc::new(statistics)) } fn name(&self) -> &str { @@ -1422,12 +1434,13 @@ mod tests { use datafusion::config::ConfigOptions; use datafusion::execution::SessionStateBuilder; use datafusion::physical_plan::filter_pushdown::ChildFilterPushdownResult; + use datafusion::physical_plan::{StatisticsArgs, StatisticsContext}; use datafusion_common::TableReference; use datafusion_expr::{LogicalPlanBuilder, col, lit}; - use datafusion_physical_expr::Distribution; use datafusion_physical_expr::expressions::{ Column, DynamicFilterPhysicalExpr, lit as physical_lit, }; + use datafusion_physical_expr::{Distribution, PhysicalExpr}; use datatypes::prelude::{ConcreteDataType, VectorRef}; use datatypes::schema::{ColumnSchema, Schema}; use datatypes::vectors::{Int64Vector, StringVector, TimestampMillisecondVector}; @@ -1501,6 +1514,12 @@ mod tests { .unwrap() } + fn merge_scan_statistics(exec: &MergeScanExec) -> Arc { + StatisticsContext::new() + .compute(exec, &StatisticsArgs::new()) + .unwrap() + } + fn task_context_with_engine_state( state: Arc, query_ctx: QueryContextRef, @@ -1649,9 +1668,7 @@ mod tests { .build() .unwrap(); assert_eq!( - merge_scan_exec_with_plan(regions.clone(), limited, 10) - .partition_statistics(None) - .unwrap() + merge_scan_statistics(&merge_scan_exec_with_plan(regions.clone(), limited, 10)) .num_rows, Precision::Inexact(100) ); @@ -1665,10 +1682,12 @@ mod tests { .build() .unwrap(); assert_eq!( - merge_scan_exec_with_plan(vec![RegionId::new(1024, 1)], large_limit, 10) - .partition_statistics(None) - .unwrap() - .num_rows, + merge_scan_statistics(&merge_scan_exec_with_plan( + vec![RegionId::new(1024, 1)], + large_limit, + 10, + )) + .num_rows, Precision::Inexact(large_bound) ); @@ -1678,17 +1697,16 @@ mod tests { .build() .unwrap(); assert_eq!( - merge_scan_exec_with_plan(regions.clone(), uncapped.clone(), 10) - .partition_statistics(None) - .unwrap() - .num_rows, + merge_scan_statistics(&merge_scan_exec_with_plan( + regions.clone(), + uncapped.clone(), + 10, + )) + .num_rows, Precision::Absent ); assert_eq!( - merge_scan_exec_with_plan(Vec::new(), uncapped, 10) - .partition_statistics(None) - .unwrap() - .num_rows, + merge_scan_statistics(&merge_scan_exec_with_plan(Vec::new(), uncapped, 10)).num_rows, Precision::Inexact(0) ); @@ -1702,10 +1720,12 @@ mod tests { .build() .unwrap(); assert_eq!( - merge_scan_exec_with_plan(regions.clone(), global_aggregate, 10) - .partition_statistics(None) - .unwrap() - .num_rows, + merge_scan_statistics(&merge_scan_exec_with_plan( + regions.clone(), + global_aggregate, + 10, + )) + .num_rows, Precision::Inexact(2) ); @@ -1725,10 +1745,7 @@ mod tests { .build() .unwrap(); assert_eq!( - merge_scan_exec_with_plan(regions, grouping_sets, 10) - .partition_statistics(None) - .unwrap() - .num_rows, + merge_scan_statistics(&merge_scan_exec_with_plan(regions, grouping_sets, 10)).num_rows, Precision::Absent ); } @@ -3251,7 +3268,7 @@ mod tests { // A distribution that differs from the current partitioning but shares a // column name present in partition_cols, so try_with_new_distribution // produces a clone instead of returning None. - let new_dist = Distribution::HashPartitioned(vec![ + let new_dist = Distribution::KeyPartitioned(vec![ Arc::new(Column::new("col1", 0)), Arc::new(Column::new("col2", 1)), ]); @@ -3267,6 +3284,24 @@ mod tests { ); } + #[test] + fn merge_scan_apply_expressions_exposes_remote_dyn_filter_id() { + let query_ctx = QueryContext::arc(); + let exec = + remote_dyn_filter_test_exec(Arc::new(TestRegionQueryHandler::default()), query_ctx); + let dyn_filter = install_remote_dyn_filter(&exec); + let expected_expression_id = dyn_filter.expression_id(); + let mut expression_ids = Vec::new(); + + exec.apply_expressions(&mut |expr| { + expression_ids.push(expr.expression_id()); + Ok(TreeNodeRecursion::Continue) + }) + .unwrap(); + + assert_eq!(expression_ids, vec![expected_expression_id]); + } + #[test] fn remote_dyn_filter_preflight_removes_parent_filter_after_dn_runtime_is_ready() { let remote_dyn_filter_producer_id = RemoteDynFilterProducerId::new(42); diff --git a/src/query/src/dist_plan/merge_sort.rs b/src/query/src/dist_plan/merge_sort.rs index 2c8f4d9fd8..e95849d5b5 100644 --- a/src/query/src/dist_plan/merge_sort.rs +++ b/src/query/src/dist_plan/merge_sort.rs @@ -16,7 +16,6 @@ //! `SortPreservingMergeExec` operator in datafusion //! -use std::any::Any; use std::fmt; use std::sync::Arc; @@ -27,12 +26,14 @@ use datafusion::physical_plan::projection::{ProjectionExec, make_with_child, upd use datafusion::physical_plan::sorts::sort::SortExec; use datafusion::physical_plan::sorts::sort_preserving_merge::SortPreservingMergeExec; use datafusion::physical_plan::{ - DisplayAs, DisplayFormatType, ExecutionPlan, PlanProperties, SendableRecordBatchStream, - Statistics, + ChildStats, ChildrenPropertiesMode, DisplayAs, DisplayFormatType, ExecutionPlan, + InputDistributionRequirements, PlanProperties, ReplaceChildrenOptions, + SendableRecordBatchStream, Statistics, StatisticsArgs, apply_expression_roots, }; +use datafusion_common::tree_node::TreeNodeRecursion; use datafusion_common::{DataFusionError, Result}; use datafusion_expr::{Extension, LogicalPlan, SortExpr, UserDefinedLogicalNodeCore}; -use datafusion_physical_expr::{Distribution, LexOrdering, OrderingRequirements}; +use datafusion_physical_expr::{LexOrdering, OrderingRequirements, PhysicalExpr}; /// MergeSort Logical Plan, have same field as `Sort`, but indicate it is a merge sort, /// which assume each input partition is a sorted stream, and will use `SortPreserveingMergeExec` @@ -94,7 +95,7 @@ impl MergeSortExec { fn input_with_fetch(&self, fetch: Option) -> Arc { let input = Arc::clone(self.inner.input()); - if let Some(sort) = input.as_any().downcast_ref::() + if let Some(sort) = input.downcast_ref::() && sort.preserve_partitioning() && sort.expr() == self.inner.expr() { @@ -149,7 +150,7 @@ impl ExecutionPlan for MergeSortExec { /// `MergeSortExec` delegates most behavior to DataFusion's /// `SortPreservingMergeExec`, but it must not expose itself as that type. /// DataFusion's `EnforceSorting` optimizer recognizes a bare - /// `SortPreservingMergeExec` via `as_any().downcast_ref::<...>()` and may + /// `SortPreservingMergeExec` via `downcast_ref::<...>()` and may /// replace it with an unordered `CoalescePartitionsExec(fetch)` when the /// parent does not require sorted output. /// @@ -165,10 +166,6 @@ impl ExecutionPlan for MergeSortExec { /// below `MergeSortExec` when `MergeScanExec` cannot preserve per-partition /// ordering. This opacity is specifically about protecting the merge stage /// itself from the `EnforceSorting` rewrite above. - fn as_any(&self) -> &dyn Any { - self - } - fn properties(&self) -> &Arc { self.inner.properties() } @@ -193,8 +190,8 @@ impl ExecutionPlan for MergeSortExec { }) } - fn required_input_distribution(&self) -> Vec { - self.inner.required_input_distribution() + fn input_distribution_requirements(&self) -> InputDistributionRequirements { + self.inner.input_distribution_requirements() } fn benefits_from_input_partitioning(&self) -> Vec { @@ -205,7 +202,7 @@ impl ExecutionPlan for MergeSortExec { /// ordered. This is the contract that makes `EnforceSorting` insert a /// `SortExec` below `MergeSortExec` when the input cannot preserve ordering. /// - /// The opacity of `MergeSortExec::as_any`, not this requirement, is what + /// The opacity of `MergeSortExec`'s downcast identity, not this requirement, is what /// prevents DataFusion from rewriting the merge stage itself as a bare /// `SortPreservingMergeExec`. fn required_input_ordering(&self) -> Vec> { @@ -220,9 +217,17 @@ impl ExecutionPlan for MergeSortExec { self.inner.children() } - fn with_new_children( + fn apply_expressions( + &self, + f: &mut dyn FnMut(&Arc) -> Result, + ) -> Result { + apply_expression_roots(self.inner.expr().iter().map(|sort_expr| &sort_expr.expr), f) + } + + fn replace_children( self: Arc, mut children: Vec>, + options: ReplaceChildrenOptions, ) -> Result> { if children.len() != 1 { return Err(DataFusionError::Internal(format!( @@ -231,11 +236,31 @@ impl ExecutionPlan for MergeSortExec { ))); } - Ok(Arc::new(Self::new( - self.inner.expr().clone(), - children.swap_remove(0), - self.inner.fetch(), - ))) + match options.children_properties { + ChildrenPropertiesMode::Keep => Ok(Arc::new(Self { + inner: SortPreservingMergeExec::new( + self.inner.expr().clone(), + children.swap_remove(0), + ) + .with_fetch(self.inner.fetch()), + })), + ChildrenPropertiesMode::Recompute => Ok(Arc::new(Self::new( + self.inner.expr().clone(), + children.swap_remove(0), + self.inner.fetch(), + ))), + } + } + + #[allow(deprecated)] + fn with_new_children( + self: Arc, + children: Vec>, + ) -> Result> { + self.replace_children( + children, + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + ) } fn execute( @@ -250,8 +275,16 @@ impl ExecutionPlan for MergeSortExec { self.inner.metrics() } - fn partition_statistics(&self, partition: Option) -> Result { - self.inner.partition_statistics(partition) + fn child_stats_requests(&self, partition: Option) -> Vec { + self.inner.child_stats_requests(partition) + } + + fn statistics_from_inputs( + &self, + input_stats: &[Arc], + args: &StatisticsArgs, + ) -> Result> { + self.inner.statistics_from_inputs(input_stats, args) } fn cardinality_effect(&self) -> CardinalityEffect { @@ -421,10 +454,6 @@ mod tests { "PreserveOrderProbeExec" } - fn as_any(&self) -> &dyn Any { - self - } - fn properties(&self) -> &Arc { self.inner.properties() } @@ -433,6 +462,13 @@ mod tests { vec![&self.inner] } + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> Result, + ) -> Result { + Ok(TreeNodeRecursion::Continue) + } + fn with_new_children( self: Arc, mut children: Vec>, @@ -489,7 +525,6 @@ mod tests { assert_eq!(merge_sort.name(), "MergeSortExec"); assert!( merge_sort - .as_any() .downcast_ref::() .is_none(), "MergeSortExec must stay opaque to EnforceSorting's bare SortPreservingMerge rewrite" @@ -503,7 +538,7 @@ mod tests { assert!(!tree.contains("SortPreservingMergeExec")); let fetched = merge_sort.with_fetch(Some(2)).unwrap(); - assert!(fetched.as_any().downcast_ref::().is_some()); + assert!(fetched.downcast_ref::().is_some()); assert_eq!(fetched.fetch(), Some(2)); } @@ -545,12 +580,9 @@ mod tests { let fetched = merge_sort.with_fetch(Some(2)).unwrap(); - assert!(fetched.as_any().downcast_ref::().is_some()); + assert!(fetched.downcast_ref::().is_some()); assert_eq!(fetched.fetch(), Some(2)); - let child_sort = fetched.children()[0] - .as_any() - .downcast_ref::() - .unwrap(); + let child_sort = fetched.children()[0].downcast_ref::().unwrap(); assert_eq!(child_sort.fetch(), Some(2)); assert!(child_sort.preserve_partitioning()); } @@ -568,14 +600,12 @@ mod tests { let preserved_spm = bare_spm.with_preserve_order(true).unwrap(); assert!( preserved_spm - .as_any() .downcast_ref::() .is_some(), "bare SPM should rebuild as bare SPM" ); assert!( preserved_spm.children()[0] - .as_any() .downcast_ref::() .unwrap() .preserve_order @@ -585,14 +615,12 @@ mod tests { let preserved_merge_sort = merge_sort.with_preserve_order(true).unwrap(); assert!( preserved_merge_sort - .as_any() .downcast_ref::() .is_some(), "MergeSortExec must rewrap the preserve-order child as MergeSortExec" ); assert!( preserved_merge_sort - .as_any() .downcast_ref::() .is_none(), "MergeSortExec must not expose a bare SPM after with_preserve_order" @@ -605,7 +633,6 @@ mod tests { ); assert!( preserved_merge_sort.children()[0] - .as_any() .downcast_ref::() .unwrap() .preserve_order @@ -637,7 +664,6 @@ mod tests { .expect("SPM should accept a narrowing projection that preserves the sort key"); assert!( swapped_spm - .as_any() .downcast_ref::() .is_some(), "bare SPM should rebuild as bare SPM" @@ -657,15 +683,11 @@ mod tests { .expect("MergeSortExec should accept the same projection swap as SPM"); assert!( - swapped_merge_sort - .as_any() - .downcast_ref::() - .is_some(), + swapped_merge_sort.downcast_ref::().is_some(), "MergeSortExec must rewrap projection swaps as MergeSortExec" ); assert!( swapped_merge_sort - .as_any() .downcast_ref::() .is_none(), "MergeSortExec must not expose a bare SPM after projection swap" @@ -673,7 +695,6 @@ mod tests { assert_eq!(swapped_merge_sort.fetch(), Some(1)); assert!( swapped_merge_sort.children()[0] - .as_any() .downcast_ref::() .is_some(), "the projection should move below MergeSortExec" @@ -754,7 +775,6 @@ mod tests { .plan; assert!( optimized_spm - .as_any() .downcast_ref::() .is_some(), "this regression test must exercise EnforceSorting's bare SPM -> CoalescePartitionsExec rewrite" @@ -776,14 +796,12 @@ mod tests { .plan; assert!( optimized_merge_sort - .as_any() .downcast_ref::() .is_some(), "MergeSortExec must stay opaque to the bare SPM rewrite" ); assert!( optimized_merge_sort - .as_any() .downcast_ref::() .is_none(), "MergeSortExec(fetch) is the required distributed TopK merge stage, not an unordered coalesce" diff --git a/src/query/src/dist_plan/planner.rs b/src/query/src/dist_plan/planner.rs index f2aaa2bbf8..130a123fc6 100644 --- a/src/query/src/dist_plan/planner.rs +++ b/src/query/src/dist_plan/planner.rs @@ -22,9 +22,11 @@ use async_trait::async_trait; use catalog::CatalogManagerRef; use common_catalog::consts::{DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME}; use common_telemetry::debug; +use datafusion::catalog::Session; use datafusion::common::Result; use datafusion::datasource::DefaultTableSource; use datafusion::execution::context::SessionState; +use datafusion::logical_expr::physical_planning_context::PhysicalPlanningContext; use datafusion::physical_plan::ExecutionPlan; use datafusion::physical_planner::{ExtensionPlanner, PhysicalPlanner}; use datafusion_common::tree_node::{TreeNode, TreeNodeRecursion, TreeNodeVisitor}; @@ -60,15 +62,21 @@ pub struct MergeSortExtensionPlanner {} impl MergeSortExtensionPlanner { fn ordering( - session_state: &SessionState, + planner: &dyn PhysicalPlanner, + session: &dyn Session, + planning_ctx: &PhysicalPlanningContext, merge_sort: &MergeSortLogicalPlan, ) -> Result { let ordering = merge_sort .expr .iter() .map(|sort_expr| { - let physical_expr = session_state - .create_physical_expr(sort_expr.expr.clone(), merge_sort.input.schema())?; + let physical_expr = planner.create_physical_expr( + &sort_expr.expr, + merge_sort.input.schema(), + session, + planning_ctx, + )?; Ok(PhysicalSortExpr::new( physical_expr, SortOptions { @@ -91,11 +99,12 @@ impl MergeSortExtensionPlanner { impl ExtensionPlanner for MergeSortExtensionPlanner { async fn plan_extension( &self, - _planner: &dyn PhysicalPlanner, + planner: &dyn PhysicalPlanner, node: &dyn UserDefinedLogicalNode, _logical_inputs: &[&LogicalPlan], physical_inputs: &[Arc], - session_state: &SessionState, + session: &dyn Session, + planning_ctx: &PhysicalPlanningContext, ) -> Result>> { if let Some(merge_sort) = node.as_any().downcast_ref::() { if let LogicalPlan::Extension(ext) = &merge_sort.input.as_ref() @@ -110,14 +119,14 @@ impl ExtensionPlanner for MergeSortExtensionPlanner { "Expect MergeSort to have one physical input".to_string(), ) })?; - if input.as_any().downcast_ref::().is_none() { + if input.downcast_ref::().is_none() { return Err(DataFusionError::Internal(format!( "Expect MergeSort's input is a MergeScanExec, found {:?}", physical_inputs ))); } - let ordering = Self::ordering(session_state, merge_sort)?; + let ordering = Self::ordering(planner, session, planning_ctx, merge_sort)?; Ok(Some(Arc::new(MergeSortExec::new( ordering, input, @@ -163,7 +172,8 @@ impl ExtensionPlanner for DistExtensionPlanner { node: &dyn UserDefinedLogicalNode, _logical_inputs: &[&LogicalPlan], _physical_inputs: &[Arc], - session_state: &SessionState, + session: &dyn Session, + _planning_ctx: &PhysicalPlanningContext, ) -> Result>> { let Some(merge_scan) = node.as_any().downcast_ref::() else { return Ok(None); @@ -171,9 +181,9 @@ impl ExtensionPlanner for DistExtensionPlanner { let input_plan = merge_scan.input(); let fallback = |logical_plan| async move { - let optimized_plan = self.optimize_input_logical_plan(session_state, logical_plan)?; + let optimized_plan = self.optimize_input_logical_plan(session, logical_plan)?; planner - .create_physical_plan(&optimized_plan, session_state) + .create_physical_plan(&optimized_plan, session) .await .map(Some) }; @@ -196,6 +206,14 @@ impl ExtensionPlanner for DistExtensionPlanner { // TODO(ruihang): generate different execution plans for different variant merge operation let schema = merge_scan.schema().as_arrow(); + let session_state = session + .as_any() + .downcast_ref::() + .ok_or_else(|| { + DataFusionError::Internal( + "MergeScan requires a SessionState for physical planning".to_string(), + ) + })?; let query_ctx = session_state .config() .get_extension() @@ -208,7 +226,7 @@ impl ExtensionPlanner for DistExtensionPlanner { schema, self.region_query_handler.clone(), query_ctx, - session_state.config().target_partitions(), + session.config().target_partitions(), merge_scan.partition_cols().clone(), merge_scan.remote_dyn_filter_producer_id(), self.enable_per_region_metrics, @@ -420,11 +438,21 @@ impl DistExtensionPlanner { /// Input logical plan is analyzed. Thus only call logical optimizer to optimize it. fn optimize_input_logical_plan( &self, - session_state: &SessionState, + session: &dyn Session, plan: &LogicalPlan, ) -> Result { - let state = session_state.clone(); - state.optimizer().optimize(plan.clone(), &state, |_, _| {}) + let session_state = session + .as_any() + .downcast_ref::() + .ok_or_else(|| { + DataFusionError::Internal( + "MergeScan requires a SessionState for logical optimization".to_string(), + ) + })?; + + session_state + .optimizer() + .optimize(plan.clone(), session_state, |_, _| {}) } } @@ -448,10 +476,9 @@ impl TreeNodeVisitor<'_> for TableNameExtractor { fn f_down(&mut self, node: &Self::Node) -> Result { match node { LogicalPlan::TableScan(scan) => { - if let Some(source) = scan.source.as_any().downcast_ref::() + if let Some(source) = scan.source.downcast_ref::() && let Some(provider) = source .table_provider - .as_any() .downcast_ref::() { if provider.table().table_type() == TableType::Base { diff --git a/src/query/src/dist_plan/predicate_extractor.rs b/src/query/src/dist_plan/predicate_extractor.rs index febf6c3120..778f5b9fc9 100644 --- a/src/query/src/dist_plan/predicate_extractor.rs +++ b/src/query/src/dist_plan/predicate_extractor.rs @@ -433,16 +433,16 @@ impl DataFusionExprConverter { Expr::Cast(cast_expr) => { // For safe casts, unwrap to the inner expression // For unsafe casts, skip with debug logging - if Self::is_safe_cast_for_partition_pruning(&cast_expr.data_type) { + if Self::is_safe_cast_for_partition_pruning(cast_expr.field.data_type()) { Self::convert_to_operand(&cast_expr.expr) } else { debug!( "Skipping unsafe cast for partition pruning: {:?}", - cast_expr.data_type + cast_expr.field.data_type() ); Err(datafusion_common::DataFusionError::Plan(format!( "Cast to {:?} not supported for partition pruning", - cast_expr.data_type + cast_expr.field.data_type() ))) } } @@ -638,10 +638,10 @@ mod tests { fn test_dictionary_cast_preserves_partition_constraint() { let dictionary_type = DataType::Dictionary(Box::new(DataType::UInt32), Box::new(DataType::Utf8)); - let filter = col("tag").eq(Expr::Cast(datafusion_expr::expr::Cast { - expr: Box::new(lit("b")), - data_type: dictionary_type, - })); + let filter = col("tag").eq(Expr::Cast(datafusion_expr::Cast::new( + Box::new(lit("b")), + dictionary_type, + ))); let partition_expr = DataFusionExprConverter::convert(&filter).unwrap(); assert_eq!( @@ -1011,10 +1011,10 @@ mod tests { let cases = vec![ FilterTestCase::new( "safe_cast", - Expr::Cast(datafusion_expr::Cast { - expr: Box::new(col("user_id")), - data_type: DataType::Int64, - }) + Expr::Cast(datafusion_expr::Cast::new( + Box::new(col("user_id")), + DataType::Int64, + )) .eq(lit(100i64)), vec![PartitionExpr::new( Operand::Column("user_id".to_string()), @@ -1025,10 +1025,10 @@ mod tests { ), FilterTestCase::new( "cast_with_alias", - Expr::Cast(datafusion_expr::Cast { - expr: Box::new(col("user_id").alias("uid")), - data_type: DataType::Int64, - }) + Expr::Cast(datafusion_expr::Cast::new( + Box::new(col("user_id").alias("uid")), + DataType::Int64, + )) .eq(lit(100i64)), vec![PartitionExpr::new( Operand::Column("user_id".to_string()), @@ -1039,12 +1039,12 @@ mod tests { ), FilterTestCase::new( "unsafe_cast", - Expr::Cast(datafusion_expr::Cast { - expr: Box::new(col("user_id")), - data_type: DataType::List(std::sync::Arc::new( + Expr::Cast(datafusion_expr::Cast::new( + Box::new(col("user_id")), + DataType::List(std::sync::Arc::new( datafusion::arrow::datatypes::Field::new("item", DataType::Int32, true), )), - }) + )) .eq(lit(100i64)), vec![], vec!["user_id"], @@ -1122,10 +1122,10 @@ mod tests { let in_expr = col("user_id") .alias("uid") .in_list(vec![lit(100i64), lit(200i64)], false); - let cast_expr = Expr::Cast(datafusion_expr::Cast { - expr: Box::new(col("user_id")), - data_type: DataType::Int64, - }); + let cast_expr = Expr::Cast(datafusion_expr::Cast::new( + Box::new(col("user_id")), + DataType::Int64, + )); let between_expr = cast_expr.between(lit(300i64), lit(400i64)); in_expr.or(between_expr) }, diff --git a/src/query/src/dist_plan/remote_dyn_filter_receiver.rs b/src/query/src/dist_plan/remote_dyn_filter_receiver.rs index b36e725bed..562cfd3c77 100644 --- a/src/query/src/dist_plan/remote_dyn_filter_receiver.rs +++ b/src/query/src/dist_plan/remote_dyn_filter_receiver.rs @@ -18,8 +18,9 @@ use std::hash::{Hash, Hasher}; use std::sync::Arc; use async_trait::async_trait; +use datafusion::catalog::Session; use datafusion::common::Result; -use datafusion::execution::context::SessionState; +use datafusion::logical_expr::physical_planning_context::PhysicalPlanningContext; use datafusion::physical_expr::utils::conjunction; use datafusion::physical_plan::ExecutionPlan; use datafusion::physical_plan::expressions::Column; @@ -176,7 +177,7 @@ fn remap_physical_expr_columns( expr: Arc, input_schema: &datafusion::arrow::datatypes::Schema, ) -> Result> { - if let Some(column) = expr.as_any().downcast_ref::() { + if let Some(column) = expr.downcast_ref::() { return Ok(Arc::new(Column::new_with_schema( column.name(), input_schema, @@ -205,7 +206,8 @@ impl ExtensionPlanner for RemoteDynFilterReceiverExtensionPlanner { node: &dyn UserDefinedLogicalNode, _logical_inputs: &[&LogicalPlan], physical_inputs: &[Arc], - _session_state: &SessionState, + _session: &dyn Session, + _planning_ctx: &PhysicalPlanningContext, ) -> Result>> { let Some(receiver) = node .as_any() diff --git a/src/query/src/dummy_catalog.rs b/src/query/src/dummy_catalog.rs index 3d1dfe5f37..c5f07d611e 100644 --- a/src/query/src/dummy_catalog.rs +++ b/src/query/src/dummy_catalog.rs @@ -70,10 +70,6 @@ impl DummyCatalogList { } impl CatalogProviderList for DummyCatalogList { - fn as_any(&self) -> &dyn Any { - self - } - fn register_catalog( &self, _name: String, @@ -98,10 +94,6 @@ struct DummyCatalogProvider { } impl CatalogProvider for DummyCatalogProvider { - fn as_any(&self) -> &dyn Any { - self - } - fn schema_names(&self) -> Vec { vec![] } @@ -119,10 +111,6 @@ struct DummySchemaProvider { #[async_trait] impl SchemaProvider for DummySchemaProvider { - fn as_any(&self) -> &dyn Any { - self - } - fn table_names(&self) -> Vec { vec![] } @@ -162,10 +150,6 @@ impl fmt::Debug for DummyTableProvider { #[async_trait] impl TableProvider for DummyTableProvider { - fn as_any(&self) -> &dyn Any { - self - } - fn schema(&self) -> SchemaRef { let schema = self.metadata.schema.arrow_schema(); if !supports_pk_dictionary_encoding(self.engine.name()) { diff --git a/src/query/src/log_query/planner.rs b/src/query/src/log_query/planner.rs index 05aeb0795d..5bc8d034a1 100644 --- a/src/query/src/log_query/planner.rs +++ b/src/query/src/log_query/planner.rs @@ -17,12 +17,11 @@ use catalog::table_source::DfTableSourceProvider; use common_function::utils::escape_like_pattern; use datafusion::datasource::DefaultTableSource; use datafusion::execution::SessionState; -use datafusion_common::{DFSchema, ScalarValue}; +use datafusion_common::{DFSchema, ScalarValue, TableReference}; use datafusion_expr::utils::{conjunction, disjunction}; use datafusion_expr::{ BinaryExpr, Expr, ExprSchemable, LogicalPlan, LogicalPlanBuilder, Operator, col, lit, not, }; -use datafusion_sql::TableReference; use datatypes::schema::Schema; use log_query::{AggFunc, BinaryOperator, EqualValue, LogExpr, LogQuery, TimeFilter}; use snafu::{OptionExt, ResultExt}; @@ -56,11 +55,9 @@ impl LogQueryPlanner { .await .context(CatalogSnafu)?; let schema = table_source - .as_any() .downcast_ref::() .context(UnknownTableSnafu)? .table_provider - .as_any() .downcast_ref::() .context(UnknownTableSnafu)? .table() diff --git a/src/query/src/metrics.rs b/src/query/src/metrics.rs index 5aa7bf8282..d6d62c99c4 100644 --- a/src/query/src/metrics.rs +++ b/src/query/src/metrics.rs @@ -280,7 +280,7 @@ fn collect_region_watermarks(plan: Arc) -> Vec() + if let Some(merge_scan) = plan.downcast_ref::() && !merge_scan.is_flow_sink_scan() { merge_merge_scan_region_watermarks( diff --git a/src/query/src/optimizer.rs b/src/query/src/optimizer.rs index 480c5046c2..fec001e750 100644 --- a/src/query/src/optimizer.rs +++ b/src/query/src/optimizer.rs @@ -16,6 +16,7 @@ pub mod const_normalization; pub mod constant_term; pub mod count_nest_aggr; pub mod count_wildcard; +pub mod enforce_sorting; pub mod global_limit; pub(crate) mod insert_assignment; pub(crate) mod json_schema_concretize; diff --git a/src/query/src/optimizer/const_normalization.rs b/src/query/src/optimizer/const_normalization.rs index 620797208e..d29f0aa4f3 100644 --- a/src/query/src/optimizer/const_normalization.rs +++ b/src/query/src/optimizer/const_normalization.rs @@ -561,11 +561,11 @@ enum CastInputKind { /// Returns the input expression and target type for `CAST` and `TRY_CAST` expressions. fn extract_cast_input(expr: &Expr) -> Option<(CastInputKind, &Expr, &DataType)> { match expr { - Expr::Cast(Cast { expr, data_type }) => { - Some((CastInputKind::Cast, expr.as_ref(), data_type)) + Expr::Cast(Cast { expr, field }) => { + Some((CastInputKind::Cast, expr.as_ref(), field.data_type())) } - Expr::TryCast(TryCast { expr, data_type }) => { - Some((CastInputKind::TryCast, expr.as_ref(), data_type)) + Expr::TryCast(TryCast { expr, field }) => { + Some((CastInputKind::TryCast, expr.as_ref(), field.data_type())) } _ => None, } @@ -921,7 +921,6 @@ mod tests { .await .unwrap(); let filter = physical_plan - .as_any() .downcast_ref::() .expect("regex residual must remain a FilterExec"); assert!(matches!( @@ -1332,6 +1331,24 @@ mod tests { expected_greptime: "Filter: CAST(t.ts_ms AS Timestamp(ns)) = TimestampNanosecond(5000000000, None)\n TableScan: t", expected_datafusion: "Filter: t.ts_ms = TimestampMillisecond(5000, None)\n TableScan: t", }, + Case { + name: "timestamp widening try_cast exact", + fields: vec![Field::new( + "ts_ms", + DataType::Timestamp(ArrowTimeUnit::Millisecond, None), + false, + )], + predicate: try_cast( + col("ts_ms"), + DataType::Timestamp(ArrowTimeUnit::Nanosecond, None), + ) + .eq(lit(ScalarValue::TimestampNanosecond( + Some(5_000_000_000), + None, + ))), + expected_greptime: "Filter: TRY_CAST(t.ts_ms AS Timestamp(ns)) = TimestampNanosecond(5000000000, None)\n TableScan: t", + expected_datafusion: "Filter: t.ts_ms = TimestampMillisecond(5000, None)\n TableScan: t", + }, ]; for case in cases { @@ -1506,10 +1523,6 @@ mod tests { #[async_trait] impl TableProvider for ExactPushdownProvider { - fn as_any(&self) -> &dyn std::any::Any { - self - } - fn schema(&self) -> arrow_schema::SchemaRef { self.schema.clone() } diff --git a/src/query/src/optimizer/constant_term.rs b/src/query/src/optimizer/constant_term.rs index 47f3faf59d..70ddcc2f64 100644 --- a/src/query/src/optimizer/constant_term.rs +++ b/src/query/src/optimizer/constant_term.rs @@ -76,10 +76,6 @@ impl PartialEq for PreCompiledMatchesTermExpr { impl Eq for PreCompiledMatchesTermExpr {} impl PhysicalExpr for PreCompiledMatchesTermExpr { - fn as_any(&self) -> &dyn std::any::Any { - self - } - fn data_type( &self, _input_schema: &arrow_schema::Schema, @@ -166,10 +162,10 @@ impl PhysicalOptimizerRule for MatchesConstantTermOptimizer { ) -> DfResult> { let res = plan .transform_down(&|plan: Arc| { - if let Some(filter) = plan.as_any().downcast_ref::() { + if let Some(filter) = plan.downcast_ref::() { let pred = filter.predicate().clone(); let new_pred = pred.transform_down(&|expr: Arc| { - if let Some(func) = expr.as_any().downcast_ref::() { + if let Some(func) = expr.downcast_ref::() { if !func.name().eq_ignore_ascii_case("matches_term") { return Ok(Transformed::no(expr)); } @@ -178,7 +174,7 @@ impl PhysicalOptimizerRule for MatchesConstantTermOptimizer { return Ok(Transformed::no(expr)); } - if let Some(lit) = args[1].as_any().downcast_ref::() + if let Some(lit) = args[1].downcast_ref::() && let ScalarValue::Utf8(Some(term)) = lit.value() { let finder = MatchesTermFinder::new(term); @@ -248,6 +244,7 @@ mod tests { use datafusion::physical_plan::get_plan_string; use datafusion_common::{Column, DFSchema}; use datafusion_expr::expr::ScalarFunction; + use datafusion_expr::physical_planning_context::PhysicalPlanningContext; use datafusion_expr::{Expr, Literal, ScalarUDF}; use datafusion_physical_expr::{ScalarFunctionExpr, create_physical_expr}; use datatypes::prelude::ConcreteDataType; @@ -343,6 +340,7 @@ mod tests { )), &DFSchema::try_from(batch.schema().clone()).unwrap(), &Default::default(), + &PhysicalPlanningContext::default(), ) .unwrap(); @@ -357,16 +355,11 @@ mod tests { .optimize(Arc::new(filter), &Default::default()) .unwrap(); - let optimized_filter = optimized_plan - .as_any() - .downcast_ref::() - .unwrap(); + let optimized_filter = optimized_plan.downcast_ref::().unwrap(); let predicate = optimized_filter.predicate(); // The predicate should be a PreCompiledMatchesTermExpr - assert!( - std::any::TypeId::of::() == predicate.as_any().type_id() - ); + assert!(predicate.is::()); } #[test] @@ -413,6 +406,7 @@ mod tests { )), &DFSchema::try_from(batch.schema().clone()).unwrap(), &Default::default(), + &PhysicalPlanningContext::default(), ) .unwrap(); @@ -426,14 +420,11 @@ mod tests { .optimize(Arc::new(filter), &Default::default()) .unwrap(); - let optimized_filter = optimized_plan - .as_any() - .downcast_ref::() - .unwrap(); + let optimized_filter = optimized_plan.downcast_ref::().unwrap(); let predicate = optimized_filter.predicate(); // The predicate should still be a ScalarFunctionExpr - assert!(std::any::TypeId::of::() == predicate.as_any().type_id()); + assert!(predicate.is::()); } #[tokio::test] diff --git a/src/query/src/optimizer/count_wildcard.rs b/src/query/src/optimizer/count_wildcard.rs index 1c6a5b814b..3c829fbef0 100644 --- a/src/query/src/optimizer/count_wildcard.rs +++ b/src/query/src/optimizer/count_wildcard.rs @@ -16,13 +16,12 @@ use datafusion::datasource::DefaultTableSource; use datafusion_common::tree_node::{ Transformed, TransformedResult, TreeNode, TreeNodeRecursion, TreeNodeVisitor, }; -use datafusion_common::{Column, Result as DataFusionResult, ScalarValue}; +use datafusion_common::{Column, Result as DataFusionResult, ScalarValue, TableReference}; use datafusion_expr::expr::{AggregateFunction, WindowFunction}; use datafusion_expr::utils::COUNT_STAR_EXPANSION; use datafusion_expr::{Expr, LogicalPlan, WindowFunctionDefinition, col, lit}; use datafusion_optimizer::AnalyzerRule; use datafusion_optimizer::utils::NamePreserver; -use datafusion_sql::TableReference; use table::table::adapter::DfTableProviderAdapter; /// A replacement to DataFusion's [`CountWildcardRule`]. This rule @@ -155,13 +154,9 @@ impl TreeNodeVisitor<'_> for TimeIndexFinder { } if let LogicalPlan::TableScan(table_scan) = &node - && let Some(source) = table_scan - .source - .as_any() - .downcast_ref::() + && let Some(source) = table_scan.source.downcast_ref::() && let Some(adapter) = source .table_provider - .as_any() .downcast_ref::() { let table_info = adapter.table().table_info(); @@ -206,9 +201,8 @@ mod test { use common_recordbatch::{RecordBatch, SendableRecordBatchStream}; use datafusion::functions_aggregate::count::count_all; use datafusion::functions_aggregate::min_max::max; - use datafusion_common::Column; + use datafusion_common::{Column, TableReference}; use datafusion_expr::LogicalPlanBuilder; - use datafusion_sql::TableReference; use datatypes::data_type::ConcreteDataType; use datatypes::schema::{ColumnSchema, Schema, SchemaBuilder}; use datatypes::vectors::{Int64Vector, TimestampMillisecondVector, VectorRef}; diff --git a/src/query/src/optimizer/enforce_sorting.rs b/src/query/src/optimizer/enforce_sorting.rs new file mode 100644 index 0000000000..b50127fa06 --- /dev/null +++ b/src/query/src/optimizer/enforce_sorting.rs @@ -0,0 +1,91 @@ +// Copyright 2023 Greptime Team +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! Sorting enforcement that runs after GreptimeDB's custom physical rules. +//! +//! DataFusion 55 moved the standalone `EnforceSorting` phases into +//! `EnsureRequirements`. GreptimeDB still needs to rerun those phases after +//! custom rules modify scan partitioning and distribution. + +use std::sync::Arc; + +use datafusion::physical_optimizer::PhysicalOptimizerRule; +use datafusion::physical_optimizer::enforce_sorting::replace_with_order_preserving_variants::{ + OrderPreservationContext, replace_with_order_preserving_variants, +}; +use datafusion::physical_optimizer::enforce_sorting::sort_pushdown::{ + SortPushDown, assign_initial_requirements, pushdown_sorts, +}; +use datafusion::physical_optimizer::enforce_sorting::{ + PlanWithCorrespondingCoalescePartitions, PlanWithCorrespondingSort, ensure_sorting, + parallelize_sorts, replace_with_partial_sort, +}; +use datafusion::physical_plan::ExecutionPlan; +use datafusion_common::Result; +use datafusion_common::config::ConfigOptions; +use datafusion_common::tree_node::{Transformed, TransformedResult, TreeNode}; + +/// Runs the standalone sorting-enforcement pipeline removed in DataFusion 55. +#[derive(Debug)] +pub struct EnforceSorting; + +impl PhysicalOptimizerRule for EnforceSorting { + fn optimize( + &self, + plan: Arc, + config: &ConfigOptions, + ) -> Result> { + // Phase 1: ensure sorting requirements and remove redundant sorts. + let sorting = PlanWithCorrespondingSort::new_default(plan); + let sorting = sorting.transform_up(ensure_sorting)?.data; + + // Phase 2: optionally turn CoalescePartitions + Sort into parallel + // sorts followed by a SortPreservingMerge. + let plan = if config.optimizer.repartition_sorts { + let parallel = PlanWithCorrespondingCoalescePartitions::new_default(sorting.plan) + .transform_up(parallelize_sorts) + .data()?; + parallel.plan + } else { + sorting.plan + }; + + // Phase 3: use order-preserving executor variants where appropriate. + let variants = OrderPreservationContext::new_default(plan); + let variants = variants + .transform_up(|context| { + replace_with_order_preserving_variants(context, false, true, config) + }) + .data()?; + + // Phase 4: push sorts down through order-preserving operators. + let mut pushdown = SortPushDown::new_default(variants.plan); + assign_initial_requirements(&mut pushdown); + let pushed = pushdown_sorts(pushdown)?; + + // Phase 5: exploit an already-satisfied prefix on unbounded inputs. + pushed + .plan + .transform_up(|plan| Ok(Transformed::yes(replace_with_partial_sort(plan)?))) + .data() + } + + fn name(&self) -> &str { + "EnforceSorting" + } + + fn schema_check(&self) -> bool { + true + } +} diff --git a/src/query/src/optimizer/global_limit.rs b/src/query/src/optimizer/global_limit.rs index 4b2ff721f6..5abb9953b7 100644 --- a/src/query/src/optimizer/global_limit.rs +++ b/src/query/src/optimizer/global_limit.rs @@ -21,7 +21,9 @@ use datafusion::physical_plan::filter::FilterExec; use datafusion::physical_plan::limit::GlobalLimitExec; use datafusion::physical_plan::repartition::RepartitionExec; use datafusion::physical_plan::sorts::sort_preserving_merge::SortPreservingMergeExec; -use datafusion::physical_plan::{ExecutionPlan, ExecutionPlanProperties}; +use datafusion::physical_plan::{ + ChildrenPropertiesMode, ExecutionPlan, ExecutionPlanProperties, ReplaceChildrenOptions, +}; use datafusion_common::Result as DfResult; use datafusion_physical_expr::{Distribution, OrderingRequirements, Partitioning}; @@ -57,7 +59,7 @@ impl EnsureGlobalLimitForFetch { let plan = if children.is_empty() { plan } else { - let required_input_distribution = plan.required_input_distribution(); + let required_input_distribution = plan.input_distribution_requirements(); let required_input_ordering = plan.required_input_ordering(); let maintains_input_order = plan.maintains_input_order(); let child_parent = ParentContext { @@ -72,7 +74,7 @@ impl EnsureGlobalLimitForFetch { .enumerate() .map(|(idx, child)| { let required_distribution = required_input_distribution - .get(idx) + .child_distribution(idx) .cloned() .unwrap_or(Distribution::UnspecifiedDistribution); let partitioning_to_restore = @@ -101,7 +103,10 @@ impl EnsureGlobalLimitForFetch { Self::optimize_plan(Arc::clone(child), parent) }) .collect::>>()?; - plan.with_new_children(children)? + plan.replace_children( + children, + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + )? }; let Some(fetch) = plan.fetch() else { @@ -111,7 +116,7 @@ impl EnsureGlobalLimitForFetch { if parent .global_fetch .is_some_and(|parent_fetch| parent_fetch <= fetch) - || !plan.as_any().is::() + || !plan.is::() || plan.output_partitioning().partition_count() <= 1 { return Ok(plan); @@ -149,10 +154,10 @@ impl Default for ParentContext { fn provided_global_fetch(plan: &Arc) -> Option { let fetch = plan.fetch()?; - (plan.as_any().is::() - || plan.as_any().is::() - || plan.as_any().is::() - || plan.as_any().is::()) + (plan.is::() + || plan.is::() + || plan.is::() + || plan.is::()) .then_some(fetch) } @@ -196,7 +201,7 @@ fn partitioning_to_restore_for( child: &Arc, required_distribution: &Distribution, ) -> Option { - if !matches!(required_distribution, Distribution::HashPartitioned(_)) + if !matches!(required_distribution, Distribution::KeyPartitioned(_)) || child.output_partitioning().partition_count() <= 1 { return None; @@ -233,7 +238,7 @@ fn inherited_partitioning_to_restore( let satisfies_parent_distribution = matches!( parent.required_distribution, - Distribution::HashPartitioned(_) + Distribution::KeyPartitioned(_) ) && plan .output_partitioning() .satisfaction( @@ -272,7 +277,7 @@ mod tests { let optimized = EnsureGlobalLimitForFetch::optimize_plan(filter, ParentContext::default()).unwrap(); - assert!(optimized.as_any().is::()); + assert!(optimized.is::()); assert_eq!(optimized.fetch(), Some(1)); assert_eq!(optimized.output_partitioning().partition_count(), 1); } @@ -295,7 +300,7 @@ mod tests { let projection = optimized.children()[0]; let coalesce = projection.children()[0]; - assert!(coalesce.as_any().is::()); + assert!(coalesce.is::()); assert_eq!(coalesce.fetch(), Some(5)); } @@ -310,8 +315,8 @@ mod tests { EnsureGlobalLimitForFetch::optimize_plan(merge, ParentContext::default()).unwrap(); let child = optimized.children()[0]; - assert!(optimized.as_any().is::()); - assert!(child.as_any().is::()); + assert!(optimized.is::()); + assert!(child.is::()); } #[test] @@ -325,8 +330,8 @@ mod tests { EnsureGlobalLimitForFetch::optimize_plan(merge, ParentContext::default()).unwrap(); let child = optimized.children()[0]; - assert!(optimized.as_any().is::()); - assert!(child.as_any().is::()); + assert!(optimized.is::()); + assert!(child.is::()); assert_eq!(child.fetch(), Some(5)); } @@ -340,8 +345,8 @@ mod tests { EnsureGlobalLimitForFetch::optimize_plan(merge, ParentContext::default()).unwrap(); let child = optimized.children()[0]; - assert!(optimized.as_any().is::()); - assert!(child.as_any().is::()); + assert!(optimized.is::()); + assert!(child.is::()); } #[test] @@ -354,10 +359,10 @@ mod tests { EnsureGlobalLimitForFetch::optimize_plan(merge, ParentContext::default()).unwrap(); let child = optimized.children()[0]; - assert!(optimized.as_any().is::()); - assert!(child.as_any().is::()); + assert!(optimized.is::()); + assert!(child.is::()); assert_eq!(child.fetch(), Some(5)); - assert!(child.children()[0].as_any().is::()); + assert!(child.children()[0].is::()); } #[test] @@ -371,8 +376,8 @@ mod tests { EnsureGlobalLimitForFetch::optimize_plan(merge, ParentContext::default()).unwrap(); let child = optimized.children()[0]; - assert!(optimized.as_any().is::()); - assert!(child.as_any().is::()); + assert!(optimized.is::()); + assert!(child.is::()); assert_eq!(child.fetch(), Some(1)); } @@ -396,10 +401,7 @@ mod tests { None, ) .unwrap(); - let merge = optimized - .as_any() - .downcast_ref::() - .unwrap(); + let merge = optimized.downcast_ref::().unwrap(); assert_eq!(merge.expr(), &actual_ordering); } @@ -423,9 +425,9 @@ mod tests { let projection = optimized.children()[0]; let child = projection.children()[0]; - assert!(optimized.as_any().is::()); - assert!(projection.as_any().is::()); - assert!(child.as_any().is::()); + assert!(optimized.is::()); + assert!(projection.is::()); + assert!(child.is::()); assert_eq!(child.fetch(), Some(1)); } @@ -455,13 +457,13 @@ mod tests { let optimized = EnsureGlobalLimitForFetch::optimize_plan(join, ParentContext::default()).unwrap(); let left = optimized.children()[0]; - let repartition = left.as_any().downcast_ref::().unwrap(); + let repartition = left.downcast_ref::().unwrap(); assert!(matches!( repartition.partitioning(), Partitioning::Hash(_, 3) )); - assert!(repartition.input().as_any().is::()); + assert!(repartition.input().is::()); assert_eq!(repartition.input().fetch(), Some(1)); } @@ -499,16 +501,15 @@ mod tests { EnsureGlobalLimitForFetch::optimize_plan(join, ParentContext::default()).unwrap(); let projection = optimized.children()[0]; let repartition = projection.children()[0] - .as_any() .downcast_ref::() .unwrap(); - assert!(projection.as_any().is::()); + assert!(projection.is::()); assert!(matches!( repartition.partitioning(), Partitioning::Hash(_, 3) )); - assert!(repartition.input().as_any().is::()); + assert!(repartition.input().is::()); assert_eq!(repartition.input().fetch(), Some(1)); } @@ -542,17 +543,16 @@ mod tests { let outer_projection = optimized.children()[0]; let inner_projection = outer_projection.children()[0]; let repartition = inner_projection.children()[0] - .as_any() .downcast_ref::() .unwrap(); - assert!(outer_projection.as_any().is::()); - assert!(inner_projection.as_any().is::()); + assert!(outer_projection.is::()); + assert!(inner_projection.is::()); assert!(matches!( repartition.partitioning(), Partitioning::Hash(_, 3) )); - assert!(repartition.input().as_any().is::()); + assert!(repartition.input().is::()); assert_eq!(repartition.input().fetch(), Some(1)); } diff --git a/src/query/src/optimizer/insert_assignment.rs b/src/query/src/optimizer/insert_assignment.rs index ad77736ce0..8645668b59 100644 --- a/src/query/src/optimizer/insert_assignment.rs +++ b/src/query/src/optimizer/insert_assignment.rs @@ -163,11 +163,14 @@ fn retarget_assignment_cast( let expr = unalias_mut(expr); let Expr::Cast(Cast { expr: source, - data_type: DataType::Timestamp(unit, None), + field, }) = expr else { return Ok(false); }; + let DataType::Timestamp(unit, None) = field.data_type() else { + return Ok(false); + }; let unit = *unit; if !matches!( diff --git a/src/query/src/optimizer/json_type_concretize.rs b/src/query/src/optimizer/json_type_concretize.rs index 5ec45cd7f6..db94daa9e0 100644 --- a/src/query/src/optimizer/json_type_concretize.rs +++ b/src/query/src/optimizer/json_type_concretize.rs @@ -12,6 +12,7 @@ // See the License for the specific language governing permissions and // limitations under the License. +use std::any::Any; use std::collections::HashMap; use arrow_schema::DataType; @@ -50,11 +51,7 @@ impl OptimizerRule for JsonTypeConcretizeRule { plan.transform_down(|plan| match &plan { LogicalPlan::TableScan(table_scan) => { - let Some(source) = table_scan - .source - .as_any() - .downcast_ref::() - else { + let Some(source) = table_scan.source.downcast_ref::() else { return Ok(Transformed::no(plan)); }; @@ -95,12 +92,12 @@ fn apply_json_type_hint( return false; } - if let Some(adapter) = provider.as_any().downcast_ref::() { + if let Some(adapter) = (provider as &dyn Any).downcast_ref::() { adapter.with_json_type_hint(json_types); return true; } - if let Some(adapter) = provider.as_any().downcast_ref::() { + if let Some(adapter) = (provider as &dyn Any).downcast_ref::() { adapter.with_json_type_hint(json_types); return true; } diff --git a/src/query/src/optimizer/parallelize_scan.rs b/src/query/src/optimizer/parallelize_scan.rs index dd2ba06290..8e0113e69d 100644 --- a/src/query/src/optimizer/parallelize_scan.rs +++ b/src/query/src/optimizer/parallelize_scan.rs @@ -55,12 +55,10 @@ impl ParallelizeScan { let result = plan .transform_down(|plan| { - if let Some(sort_exec) = plan.as_any().downcast_ref::() { + if let Some(sort_exec) = plan.downcast_ref::() { // save the first order expr first_order_expr = Some(sort_exec.expr().first()).cloned(); - } else if let Some(region_scan_exec) = - plan.as_any().downcast_ref::() - { + } else if let Some(region_scan_exec) = plan.downcast_ref::() { let expected_partition_num = config.execution.target_partitions; if region_scan_exec.is_partition_set() || region_scan_exec.scanner_type().as_str() == "SinglePartition" diff --git a/src/query/src/optimizer/pass_distribution.rs b/src/query/src/optimizer/pass_distribution.rs index c45d4695d9..732168e962 100644 --- a/src/query/src/optimizer/pass_distribution.rs +++ b/src/query/src/optimizer/pass_distribution.rs @@ -18,7 +18,9 @@ use datafusion::config::ConfigOptions; use datafusion::physical_optimizer::PhysicalOptimizerRule; use datafusion::physical_plan::projection::ProjectionExec; use datafusion::physical_plan::repartition::RepartitionExec; -use datafusion::physical_plan::{ExecutionPlan, Partitioning}; +use datafusion::physical_plan::{ + ChildrenPropertiesMode, ExecutionPlan, Partitioning, ReplaceChildrenOptions, +}; use datafusion_common::Result as DfResult; use datafusion_physical_expr::Distribution; use datafusion_physical_expr::utils::map_columns_before_projection; @@ -68,8 +70,8 @@ impl PassDistribution { current_req: Option, ) -> DfResult> { // If this is a MergeScanExec, try to apply the current requirement. - if let Some(merge_scan) = plan.as_any().downcast_ref::() - && let Some(Distribution::HashPartitioned(hash_exprs)) = current_req.as_ref() + if let Some(merge_scan) = plan.downcast_ref::() + && let Some(Distribution::KeyPartitioned(hash_exprs)) = current_req.as_ref() { if let Partitioning::Hash(current_hash_exprs, _) = &merge_scan.properties().partitioning && *current_hash_exprs == *hash_exprs @@ -78,7 +80,7 @@ impl PassDistribution { } if let Some(new_plan) = merge_scan - .try_with_new_distribution(Distribution::HashPartitioned(hash_exprs.clone())) + .try_with_new_distribution(Distribution::KeyPartitioned(hash_exprs.clone())) { // Leaf node; no children to process return Ok(Arc::new(new_plan) as _); @@ -97,10 +99,10 @@ impl PassDistribution { return Ok(plan); } - let required = plan.required_input_distribution(); + let required = plan.input_distribution_requirements(); let mut new_children = Vec::with_capacity(children.len()); for (idx, child) in children.into_iter().enumerate() { - let child_req = match required.get(idx) { + let child_req = match required.child_distribution(idx) { Some(Distribution::UnspecifiedDistribution) if idx == 0 => { Self::map_hash_requirement_through_projection(plan.as_ref(), ¤t_req) } @@ -121,7 +123,10 @@ impl PassDistribution { if unchanged { Ok(plan) } else { - plan.with_new_children(new_children) + plan.replace_children( + new_children, + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + ) } } @@ -129,11 +134,11 @@ impl PassDistribution { plan: &dyn ExecutionPlan, current_req: &Option, ) -> Option { - let Some(Distribution::HashPartitioned(required_exprs)) = current_req else { + let Some(Distribution::KeyPartitioned(required_exprs)) = current_req else { return None; }; - let projection = plan.as_any().downcast_ref::()?; + let projection = plan.downcast_ref::()?; let proj_exprs = projection .expr() .iter() @@ -141,7 +146,7 @@ impl PassDistribution { .collect::>(); let mapped = map_columns_before_projection(required_exprs, &proj_exprs); - (mapped.len() == required_exprs.len()).then_some(Distribution::HashPartitioned(mapped)) + (mapped.len() == required_exprs.len()).then_some(Distribution::KeyPartitioned(mapped)) } } @@ -261,12 +266,8 @@ mod tests { let optimized = PassDistribution .optimize(join, &ConfigOptions::default()) .unwrap(); - let hash_join = optimized.as_any().downcast_ref::().unwrap(); - let left_projection = hash_join - .left() - .as_any() - .downcast_ref::() - .unwrap(); + let hash_join = optimized.downcast_ref::().unwrap(); + let left_projection = hash_join.left().downcast_ref::().unwrap(); let left_partitioning = left_projection.input().output_partitioning(); let right_partitioning = hash_join.right().output_partitioning(); @@ -293,7 +294,7 @@ mod tests { fn merge_scan_rejects_hash_requirement_on_partition_key_subset() { let merge_scan = test_merge_scan_exec(test_schema()); - let new_plan = merge_scan.try_with_new_distribution(Distribution::HashPartitioned(vec![ + let new_plan = merge_scan.try_with_new_distribution(Distribution::KeyPartitioned(vec![ partition_column(DATA_SCHEMA_TSID_COLUMN_NAME, 1), ])); @@ -355,12 +356,7 @@ mod tests { fn column_names(exprs: &[Arc]) -> Vec<&str> { exprs .iter() - .map(|expr| { - expr.as_any() - .downcast_ref::() - .unwrap() - .name() - }) + .map(|expr| expr.downcast_ref::().unwrap().name()) .collect() } } diff --git a/src/query/src/optimizer/promql_tsid_narrow_join.rs b/src/query/src/optimizer/promql_tsid_narrow_join.rs index 6dd0b5e4bb..c26247179e 100644 --- a/src/query/src/optimizer/promql_tsid_narrow_join.rs +++ b/src/query/src/optimizer/promql_tsid_narrow_join.rs @@ -55,7 +55,7 @@ impl PhysicalOptimizerRule for PromqlTsidNarrowJoin { impl PromqlTsidNarrowJoin { fn rewrite_join(plan: Arc) -> DfResult>> { - let Some(hash_join) = plan.as_any().downcast_ref::() else { + let Some(hash_join) = plan.downcast_ref::() else { return Ok(Transformed::no(plan)); }; @@ -103,8 +103,8 @@ impl PromqlTsidNarrowJoin { for (left, right) in hash_join.on() { let (Some(left_col), Some(right_col)) = ( - left.as_any().downcast_ref::(), - right.as_any().downcast_ref::(), + left.downcast_ref::(), + right.downcast_ref::(), ) else { return false; }; @@ -199,7 +199,7 @@ mod tests { let optimized = PromqlTsidNarrowJoin .optimize(join, &ConfigOptions::default()) .unwrap(); - let optimized_join = optimized.as_any().downcast_ref::().unwrap(); + let optimized_join = optimized.downcast_ref::().unwrap(); assert_eq!(optimized_join.partition_mode(), &PartitionMode::CollectLeft); assert_eq!(optimized.schema(), original_schema); @@ -262,7 +262,7 @@ mod tests { let optimized = PromqlTsidNarrowJoin .optimize(join, &ConfigOptions::default()) .unwrap(); - let optimized_join = optimized.as_any().downcast_ref::().unwrap(); + let optimized_join = optimized.downcast_ref::().unwrap(); assert_eq!(optimized_join.partition_mode(), &PartitionMode::CollectLeft); } @@ -317,7 +317,7 @@ mod tests { let optimized = PromqlTsidNarrowJoin .optimize(join, &ConfigOptions::default()) .unwrap(); - let optimized_join = optimized.as_any().downcast_ref::().unwrap(); + let optimized_join = optimized.downcast_ref::().unwrap(); assert_eq!(optimized_join.partition_mode(), &PartitionMode::Partitioned); } diff --git a/src/query/src/optimizer/remove_duplicate.rs b/src/query/src/optimizer/remove_duplicate.rs index 65d191e7b5..155fb1ff95 100644 --- a/src/query/src/optimizer/remove_duplicate.rs +++ b/src/query/src/optimizer/remove_duplicate.rs @@ -16,8 +16,8 @@ use std::sync::Arc; use datafusion::config::ConfigOptions; use datafusion::physical_optimizer::PhysicalOptimizerRule; -use datafusion::physical_plan::ExecutionPlan; use datafusion::physical_plan::repartition::RepartitionExec; +use datafusion::physical_plan::{ChildrenPropertiesMode, ExecutionPlan, ReplaceChildrenOptions}; use datafusion_common::Result as DfResult; use datafusion_common::tree_node::{Transformed, TreeNode}; @@ -35,29 +35,18 @@ impl PhysicalOptimizerRule for RemoveDuplicate { plan: Arc, _config: &ConfigOptions, ) -> DfResult> { - Self::do_optimize(plan) - } - - fn name(&self) -> &str { - "RemoveDuplicateRule" - } - - fn schema_check(&self) -> bool { - false - } -} - -impl RemoveDuplicate { - fn do_optimize(plan: Arc) -> DfResult> { let result = plan .transform_down(|plan| { - if plan.as_any().is::() { + if plan.is::() { // check child let child = plan.children()[0].clone(); - if child.as_any().type_id() == plan.as_any().type_id() { + if child.is::() { // remove child let grand_child = child.children()[0].clone(); - let new_plan = plan.with_new_children(vec![grand_child])?; + let new_plan = plan.replace_children( + vec![grand_child], + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + )?; return Ok(Transformed::yes(new_plan)); } } @@ -65,7 +54,14 @@ impl RemoveDuplicate { Ok(Transformed::no(plan)) })? .data; - Ok(result) } + + fn name(&self) -> &str { + "RemoveDuplicateRule" + } + + fn schema_check(&self) -> bool { + true + } } diff --git a/src/query/src/optimizer/scan_hint.rs b/src/query/src/optimizer/scan_hint.rs index 9cb50696ee..edb7355913 100644 --- a/src/query/src/optimizer/scan_hint.rs +++ b/src/query/src/optimizer/scan_hint.rs @@ -76,19 +76,11 @@ impl ScanHintRule { let LogicalPlan::TableScan(mut table_scan) = plan else { return Ok(Transformed::no(plan)); }; - let Some(source) = table_scan - .source - .as_any() - .downcast_ref::() - else { + let Some(source) = table_scan.source.downcast_ref::() else { return Ok(Transformed::no(LogicalPlan::TableScan(table_scan))); }; // The provider in the region server is [DummyTableProvider]. - let Some(original) = source - .table_provider - .as_any() - .downcast_ref::() - else { + let Some(original) = source.table_provider.downcast_ref::() else { return Ok(Transformed::no(LogicalPlan::TableScan(table_scan))); }; @@ -455,7 +447,7 @@ fn single_evaluation_projection_expr_allowed( }; alias.name == column.name && matches!( - cast.data_type, + cast.field.data_type(), DataType::Timestamp(ArrowTimeUnit::Millisecond, None) ) && matches!( @@ -596,11 +588,8 @@ mod test { let mut requests = Vec::new(); plan.apply_with_subqueries(|node| { if let LogicalPlan::TableScan(scan) = node - && let Some(source) = scan.source.as_any().downcast_ref::() - && let Some(provider) = source - .table_provider - .as_any() - .downcast_ref::() + && let Some(source) = scan.source.downcast_ref::() + && let Some(provider) = source.table_provider.downcast_ref::() { requests.push((scan.table_name.to_string(), provider.scan_request())); } diff --git a/src/query/src/optimizer/scan_hint/vector_search.rs b/src/query/src/optimizer/scan_hint/vector_search.rs index 1f7f7b5a41..90f45378ec 100644 --- a/src/query/src/optimizer/scan_hint/vector_search.rs +++ b/src/query/src/optimizer/scan_hint/vector_search.rs @@ -18,11 +18,10 @@ use common_function::scalars::vector::distance::{ VEC_COS_DISTANCE, VEC_DOT_PRODUCT, VEC_L2SQ_DISTANCE, }; use common_telemetry::debug; -use datafusion_common::ScalarValue; +use datafusion_common::{ScalarValue, TableReference}; use datafusion_expr::logical_plan::FetchType; use datafusion_expr::utils::split_conjunction; use datafusion_expr::{Expr, SortExpr}; -use datafusion_sql::TableReference; use datatypes::types::parse_string_to_vector_type_value; use store_api::storage::{VectorDistanceMetric, VectorSearchRequest}; @@ -488,11 +487,8 @@ mod tests { let mut request = None; plan.apply_with_subqueries(|node| { if let LogicalPlan::TableScan(scan) = node - && let Some(source) = scan.source.as_any().downcast_ref::() - && let Some(provider) = source - .table_provider - .as_any() - .downcast_ref::() + && let Some(source) = scan.source.downcast_ref::() + && let Some(provider) = source.table_provider.downcast_ref::() { request = Some(provider.scan_request()); } diff --git a/src/query/src/optimizer/string_normalization.rs b/src/query/src/optimizer/string_normalization.rs index 46993ee45e..2fc2d3957f 100644 --- a/src/query/src/optimizer/string_normalization.rs +++ b/src/query/src/optimizer/string_normalization.rs @@ -87,8 +87,8 @@ impl TreeNodeRewriter for StringNormalizationConverter { /// Otherwise - no modifications applied fn f_up(&mut self, expr: Expr) -> Result> { let new_expr = match expr { - Expr::Cast(Cast { expr, data_type }) => { - let expr = match data_type { + Expr::Cast(Cast { expr, field }) => { + let expr = match field.data_type() { DataType::Timestamp(_, _) => match *expr { Expr::Literal(value, _) => match value { ScalarValue::Utf8(Some(s)) => trim_utf_expr(s), @@ -98,10 +98,7 @@ impl TreeNodeRewriter for StringNormalizationConverter { }, _ => *expr, }; - Expr::Cast(Cast { - expr: Box::new(expr), - data_type, - }) + Expr::Cast(Cast::new_from_field(Box::new(expr), field)) } expr => expr, }; diff --git a/src/query/src/optimizer/type_conversion.rs b/src/query/src/optimizer/type_conversion.rs index 40bb43e791..536628bce4 100644 --- a/src/query/src/optimizer/type_conversion.rs +++ b/src/query/src/optimizer/type_conversion.rs @@ -63,6 +63,7 @@ impl ExtensionAnalyzerRule for TypeConversionRule { projected_schema, filters, fetch, + statistics_requests, }) => { let mut converter = TypeConverter::new(projected_schema.clone(), ctx.query_ctx()); let rewrite_filters = filters @@ -76,6 +77,7 @@ impl ExtensionAnalyzerRule for TypeConversionRule { projected_schema, filters: rewrite_filters, fetch, + statistics_requests, }))) } LogicalPlan::Projection { .. } => { @@ -359,10 +361,9 @@ mod tests { use std::sync::Arc; use datafusion_common::arrow::datatypes::Field; - use datafusion_common::{Column, DFSchema, NullEquality}; + use datafusion_common::{Column, DFSchema, NullEquality, TableReference}; use datafusion_expr::expr::{Cast, Exists}; use datafusion_expr::{Join, JoinConstraint, JoinType, Literal, LogicalPlanBuilder, Subquery}; - use datafusion_sql::TableReference; use session::context::QueryContext; use super::*; @@ -487,10 +488,10 @@ mod tests { use datafusion_common::arrow::datatypes::TimeUnit as ArrowTimeUnit; let mut converter = TypeConverter::new(Arc::new(DFSchema::empty()), QueryContext::arc()); - let expr = Expr::Cast(Cast { - expr: Box::new("2009-02-13 23:31:30".lit()), - data_type: DataType::Timestamp(ArrowTimeUnit::Millisecond, None), - }); + let expr = Expr::Cast(Cast::new( + Box::new("2009-02-13 23:31:30".lit()), + DataType::Timestamp(ArrowTimeUnit::Millisecond, None), + )); assert_eq!(converter.f_up(expr.clone()).unwrap().data, expr); } diff --git a/src/query/src/optimizer/windowed_sort.rs b/src/query/src/optimizer/windowed_sort.rs index 55a48c1869..3e4ccaf91c 100644 --- a/src/query/src/optimizer/windowed_sort.rs +++ b/src/query/src/optimizer/windowed_sort.rs @@ -16,7 +16,6 @@ use std::sync::Arc; use arrow_schema::DataType; use datafusion::physical_optimizer::PhysicalOptimizerRule; -use datafusion::physical_plan::ExecutionPlan; use datafusion::physical_plan::coalesce_partitions::CoalescePartitionsExec; use datafusion::physical_plan::coop::CooperativeExec; use datafusion::physical_plan::filter::FilterExec; @@ -24,6 +23,7 @@ use datafusion::physical_plan::projection::ProjectionExec; use datafusion::physical_plan::repartition::RepartitionExec; use datafusion::physical_plan::sorts::sort::SortExec; use datafusion::physical_plan::sorts::sort_preserving_merge::SortPreservingMergeExec; +use datafusion::physical_plan::{ChildrenPropertiesMode, ExecutionPlan, ReplaceChildrenOptions}; use datafusion_common::Result as DataFusionResult; use datafusion_common::tree_node::{Transformed, TreeNode}; use datafusion_physical_expr::expressions::{CastExpr, Column as PhysicalColumn}; @@ -71,7 +71,7 @@ impl WindowedSortPhysicalRule { ) -> DataFusionResult> { let result = plan .transform_down(|plan| { - if let Some(sort_exec) = plan.as_any().downcast_ref::() { + if let Some(sort_exec) = plan.downcast_ref::() { // TODO: support multiple expr in windowed sort if sort_exec.expr().len() != 1 { return Ok(Transformed::no(plan)); @@ -88,10 +88,7 @@ impl WindowedSortPhysicalRule { let input_schema = sort_input.schema(); let first_sort_expr = sort_exec.expr().first(); - if let Some(column_expr) = first_sort_expr - .expr - .as_any() - .downcast_ref::() + if let Some(column_expr) = first_sort_expr.expr.downcast_ref::() && matches!( input_schema.field(column_expr.index()).data_type(), DataType::Timestamp(_, _) @@ -165,26 +162,26 @@ fn fetch_partition_range(input: Arc) -> DataFusionResult() { + if plan.is::() { return Ok(Transformed::no(plan)); } // Unappliable case, reset the state. - if plan.as_any().is::() - || plan.as_any().is::() - || plan.as_any().is::() - || plan.as_any().is::() + if plan.is::() + || plan.is::() + || plan.is::() + || plan.is::() { partition_ranges = None; } // only a very limited set of plans can exist between region scan and sort exec // other plans might make this optimize wrong, so be safe here by limiting it - if !(plan.as_any().is::() || plan.as_any().is::()) { + if !(plan.is::() || plan.is::()) { partition_ranges = None; } - if let Some(region_scan_exec) = plan.as_any().downcast_ref::() { + if let Some(region_scan_exec) = plan.downcast_ref::() { // `PerSeries` distribution is not supported in windowed sort. if region_scan_exec.distribution() == Some(store_api::storage::TimeSeriesDistribution::PerSeries) @@ -216,11 +213,11 @@ fn is_time_index_expr( plan: &Arc, expr: &Arc, ) -> DataFusionResult { - if let Some(column_expr) = expr.as_any().downcast_ref::() { + if let Some(column_expr) = expr.downcast_ref::() { return is_time_index_column(plan, column_expr); } - if let Some(cast_expr) = expr.as_any().downcast_ref::() { + if let Some(cast_expr) = expr.downcast_ref::() { return if matches!(cast_expr.cast_type(), DataType::Timestamp(_, _)) { is_time_index_expr(plan, cast_expr.expr()) } else { @@ -228,7 +225,7 @@ fn is_time_index_expr( }; } - if let Some(scalar_function_expr) = expr.as_any().downcast_ref::() { + if let Some(scalar_function_expr) = expr.downcast_ref::() { return if is_supported_time_index_wrapper(scalar_function_expr) && scalar_function_expr.args().len() == 1 { @@ -245,14 +242,14 @@ fn is_time_index_column( plan: &Arc, column_expr: &PhysicalColumn, ) -> DataFusionResult { - if let Some(projection) = plan.as_any().downcast_ref::() { + if let Some(projection) = plan.downcast_ref::() { let Some(projection_expr) = projection.expr().get(column_expr.index()) else { return Ok(false); }; return is_time_index_expr(projection.input(), &projection_expr.expr); } - if let Some(filter) = plan.as_any().downcast_ref::() { + if let Some(filter) = plan.downcast_ref::() { let child_column_expr = filter .projection() .as_ref() @@ -268,7 +265,7 @@ fn is_time_index_column( return is_time_index_expr(filter.input(), &child_expr); } - if let Some(region_scan_exec) = plan.as_any().downcast_ref::() { + if let Some(region_scan_exec) = plan.downcast_ref::() { let schema = plan.schema(); let column_field = schema.field(column_expr.index()); return Ok( @@ -285,9 +282,9 @@ fn is_time_index_column( } fn passthrough_child(plan: &dyn ExecutionPlan) -> Option> { - if plan.as_any().is::() - || plan.as_any().is::() - || plan.as_any().is::() + if plan.is::() + || plan.is::() + || plan.is::() { return schema_preserving_child(plan); } @@ -314,13 +311,16 @@ fn remove_repartition( plan: Arc, ) -> DataFusionResult>> { plan.transform_down(|plan| { - if plan.as_any().is::() { + if plan.is::() { // Checks child. let maybe_repartition = plan.children()[0]; - if maybe_repartition.as_any().is::() { + if maybe_repartition.is::() { let maybe_scan = maybe_repartition.children()[0]; - if maybe_scan.as_any().is::() { - let new_filter = plan.clone().with_new_children(vec![maybe_scan.clone()])?; + if maybe_scan.is::() { + let new_filter = plan.clone().replace_children( + vec![maybe_scan.clone()], + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + )?; return Ok(Transformed::yes(new_filter)); } } diff --git a/src/query/src/part_sort.rs b/src/query/src/part_sort.rs index f1574ec334..7f9a63df8d 100644 --- a/src/query/src/part_sort.rs +++ b/src/query/src/part_sort.rs @@ -18,7 +18,6 @@ //! partition ([`PartitionRange`]) independently based on the provided physical //! sort expressions. -use std::any::Any; use std::pin::Pin; use std::sync::Arc; use std::task::{Context, Poll}; @@ -38,7 +37,9 @@ use datafusion::physical_plan::filter_pushdown::{ use datafusion::physical_plan::metrics::{BaselineMetrics, ExecutionPlanMetricsSet, MetricsSet}; use datafusion::physical_plan::{ DisplayAs, DisplayFormatType, ExecutionPlan, ExecutionPlanProperties, PlanProperties, + apply_expression_roots, }; +use datafusion_common::tree_node::TreeNodeRecursion; use datafusion_common::{DataFusionError, ScalarValue, internal_err}; use datafusion_expr::Operator; use datafusion_physical_expr::expressions::{ @@ -208,10 +209,6 @@ impl ExecutionPlan for PartSortExec { "PartSortExec" } - fn as_any(&self) -> &dyn Any { - self - } - fn schema(&self) -> SchemaRef { self.input.schema() } @@ -224,6 +221,27 @@ impl ExecutionPlan for PartSortExec { vec![&self.input] } + fn apply_expressions( + &self, + f: &mut dyn FnMut(&Arc) -> datafusion_common::Result, + ) -> datafusion_common::Result { + let dynamic_filter = self + .dynamic_filter + .as_ref() + .map(|filter| filter.clone() as Arc); + apply_expression_roots( + std::iter::once(&self.expression.expr).chain(dynamic_filter.as_ref()), + f, + ) + } + + fn dynamic_expressions_produced(&self) -> Vec> { + self.dynamic_filter + .iter() + .map(|filter| filter.clone() as Arc) + .collect() + } + fn with_new_children( self: Arc, children: Vec>, @@ -1847,6 +1865,63 @@ mod test { .await; } + #[test] + fn dynamic_expressions_produced_returns_topk_filter_arc() { + let unit = TimeUnit::Millisecond; + let schema = Arc::new(Schema::new(vec![Field::new( + "ts", + DataType::Timestamp(unit, None), + false, + )])); + let partition_range = PartitionRange { + start: Timestamp::new(0, unit.into()), + end: Timestamp::new(10, unit.into()), + num_rows: 0, + identifier: 0, + }; + let sort_expr = PhysicalSortExpr { + expr: Arc::new(Column::new("ts", 0)), + options: SortOptions::default(), + }; + + let limited = PartSortExec::try_new( + sort_expr.clone(), + Some(1), + vec![vec![partition_range]], + Arc::new(MockInputExec::new(vec![vec![]], schema.clone())), + ) + .unwrap(); + let expected = limited.dynamic_filter.as_ref().unwrap().clone() as Arc; + let produced = limited.dynamic_expressions_produced(); + assert_eq!(produced.len(), 1); + assert!(Arc::ptr_eq(&produced[0], &expected)); + + let mut applied_dynamic_filter = None; + limited + .apply_expressions(&mut |expr| { + if expr.expression_id().is_some() { + applied_dynamic_filter = Some(expr.clone()); + } + Ok(TreeNodeRecursion::Continue) + }) + .unwrap(); + let applied_dynamic_filter = applied_dynamic_filter.unwrap(); + assert!(Arc::ptr_eq(&produced[0], &applied_dynamic_filter)); + assert_eq!( + produced[0].expression_id(), + applied_dynamic_filter.expression_id() + ); + + let unlimited = PartSortExec::try_new( + sort_expr, + None, + vec![vec![partition_range]], + Arc::new(MockInputExec::new(vec![vec![]], schema)), + ) + .unwrap(); + assert!(unlimited.dynamic_expressions_produced().is_empty()); + } + #[test] fn test_topk_buffer_is_bounded_and_updates_dynamic_filter() { let unit = TimeUnit::Millisecond; diff --git a/src/query/src/plan.rs b/src/query/src/plan.rs index b4e5e664b6..26635a9d99 100644 --- a/src/query/src/plan.rs +++ b/src/query/src/plan.rs @@ -40,10 +40,9 @@ impl TreeNodeRewriter for TableNamesExtractAndRewriter { ) -> datafusion::error::Result> { match node { LogicalPlan::TableScan(mut scan) => { - if let Some(source) = scan.source.as_any().downcast_ref::() + if let Some(source) = scan.source.downcast_ref::() && let Some(provider) = source .table_provider - .as_any() .downcast_ref::() && provider.table().table_type() == TableType::Base { diff --git a/src/query/src/planner.rs b/src/query/src/planner.rs index 8a65a94adf..9a577c26f9 100644 --- a/src/query/src/planner.rs +++ b/src/query/src/planner.rs @@ -143,13 +143,15 @@ impl DfLogicalPlanner { // notice format is already set in query context, so can be ignore here Ok(LogicalPlan::Analyze(Analyze { verbose, + format: ExplainFormat::Indent, input: plan, schema, + analyze_level: None, + analyze_categories: None, })) } else { let stringified_plans = vec![plan.to_stringified(PlanType::InitialLogicalPlan)]; - // default to configuration value let options = self.session_state.config().options(); let format = format .map(|x| ExplainFormat::from_str(&x)) @@ -163,6 +165,7 @@ impl DfLogicalPlanner { stringified_plans, schema, logical_optimization_succeeded: false, + show_statistics: None, })) } } @@ -500,7 +503,8 @@ impl DfLogicalPlanner { if let DfExpr::Cast(cast) = e && let DfExpr::Placeholder(ph) = &*cast.expr { - placeholder_types.insert(ph.id.clone(), Some(cast.data_type.clone())); + placeholder_types + .insert(ph.id.clone(), Some(cast.field.data_type().clone())); casted_placeholders.insert(ph.id.clone()); } diff --git a/src/query/src/promql/planner.rs b/src/query/src/promql/planner.rs index 6ca04ffca7..c51cdb0583 100644 --- a/src/query/src/promql/planner.rs +++ b/src/query/src/promql/planner.rs @@ -48,15 +48,14 @@ use datafusion::optimizer::simplify_expressions::ExprSimplifier; use datafusion::prelude as df_prelude; use datafusion::prelude::{Column, Expr as DfExpr, JoinType}; use datafusion::scalar::ScalarValue; -use datafusion::sql::TableReference; use datafusion_common::tree_node::{Transformed, TreeNode, TreeNodeRewriter}; -use datafusion_common::{DFSchema, NullEquality}; +use datafusion_common::{DFSchema, NullEquality, TableReference}; use datafusion_expr::expr::WindowFunctionParams; use datafusion_expr::expr_fn::when; use datafusion_expr::simplify::SimplifyContext; use datafusion_expr::utils::{conjunction, disjunction}; use datafusion_expr::{ - ExprSchemable, Literal, Projection, SortExpr, TableScan, TableSource, col, lit, + ExprSchemable, Literal, Projection, SortExpr, TableScanBuilder, TableSource, col, lit, }; use datafusion_functions::core::coalesce; use datatypes::arrow::datatypes::{DataType as ArrowDataType, TimeUnit as ArrowTimeUnit}; @@ -1337,10 +1336,8 @@ impl PromPlanner { let mut field_expr = field_expr_builder(lhs, rhs)?; if is_comparison_op && should_return_bool { - field_expr = DfExpr::Cast(Cast { - expr: Box::new(field_expr), - data_type: ArrowDataType::Float64, - }); + field_expr = + DfExpr::Cast(Cast::new(Box::new(field_expr), ArrowDataType::Float64)); } Ok(LogicalPlan::Extension(Extension { @@ -1406,10 +1403,8 @@ impl PromPlanner { }; if is_comparison_op && should_return_bool { - binary_expr = DfExpr::Cast(Cast { - expr: Box::new(binary_expr), - data_type: ArrowDataType::Float64, - }); + binary_expr = + DfExpr::Cast(Cast::new(Box::new(binary_expr), ArrowDataType::Float64)); } Ok(binary_expr) }; @@ -1475,10 +1470,8 @@ impl PromPlanner { }; if is_comparison_op && should_return_bool { - binary_expr = DfExpr::Cast(Cast { - expr: Box::new(binary_expr), - data_type: ArrowDataType::Float64, - }); + binary_expr = + DfExpr::Cast(Cast::new(Box::new(binary_expr), ArrowDataType::Float64)); } Ok(binary_expr) }; @@ -1697,10 +1690,10 @@ impl PromPlanner { None => binary_expr_builder(lhs, rhs)?, }; if is_comparison_op && should_return_bool { - binary_expr = DfExpr::Cast(Cast { - expr: Box::new(binary_expr), - data_type: ArrowDataType::Float64, - }); + binary_expr = DfExpr::Cast(Cast::new( + Box::new(binary_expr), + ArrowDataType::Float64, + )); } Ok(binary_expr) }) @@ -2751,11 +2744,9 @@ impl PromPlanner { fn table_from_source(&self, source: &Arc) -> Result { Ok(source - .as_any() .downcast_ref::() .context(UnknownTableSnafu)? .table_provider - .as_any() .downcast_ref::() .context(UnknownTableSnafu)? .table()) @@ -3051,10 +3042,10 @@ impl PromPlanner { DATA_SCHEMA_TSID_COLUMN_NAME.to_string(), )))) .chain(Some(DfExpr::Alias(Alias { - expr: Box::new(DfExpr::Cast(Cast { - expr: Box::new(self.create_time_index_column_expr()?), - data_type: ArrowDataType::Timestamp(ArrowTimeUnit::Millisecond, None), - })), + expr: Box::new(DfExpr::Cast(Cast::new( + Box::new(self.create_time_index_column_expr()?), + ArrowDataType::Timestamp(ArrowTimeUnit::Millisecond, None), + ))), relation: Some(table_ref.clone()), name: self .ctx @@ -3198,13 +3189,12 @@ impl PromPlanner { projection.sort_unstable(); projection.dedup(); - let new_scan = TableScan::try_new( - scan.table_name.clone(), - scan.source.clone(), - Some(projection), - scan.filters, - scan.fetch, - )?; + let new_scan = + TableScanBuilder::new(scan.table_name.clone(), scan.source.clone()) + .with_projection(Some(projection)) + .with_filters(scan.filters) + .with_fetch(scan.fetch) + .build()?; Ok(Transformed::yes(LogicalPlan::TableScan(new_scan))) } LogicalPlan::Projection(proj) => { @@ -3427,10 +3417,10 @@ impl PromPlanner { } if func.name == "predict_linear" { - other_input_exprs[0] = DfExpr::Cast(Cast { - expr: Box::new(other_input_exprs[0].clone()), - data_type: ArrowDataType::Int64, - }); + other_input_exprs[0] = DfExpr::Cast(Cast::new( + Box::new(other_input_exprs[0].clone()), + ArrowDataType::Int64, + )); } let timestamp_range = DfExpr::Column(Column::from_name( @@ -3716,10 +3706,10 @@ impl PromPlanner { if all_field_columns_are_native_histogram_ranges { ScalarFunc::Udf(native_histogram_drop_udf(func.name)) } else { - other_input_exprs[0] = DfExpr::Cast(Cast { - expr: Box::new(other_input_exprs[0].clone()), - data_type: ArrowDataType::Int64, - }); + other_input_exprs[0] = DfExpr::Cast(Cast::new( + Box::new(other_input_exprs[0].clone()), + ArrowDataType::Int64, + )); ScalarFunc::Udf(Arc::new(PredictLinear::scalar_udf())) } } @@ -5378,10 +5368,10 @@ impl PromPlanner { false }; if is_comparison_op && should_return_bool { - Some(DfExpr::Cast(Cast { - expr: Box::new(expr), - data_type: ArrowDataType::Float64, - })) + Some(DfExpr::Cast(Cast::new( + Box::new(expr), + ArrowDataType::Float64, + ))) } else { Some(expr) } @@ -5476,18 +5466,12 @@ impl PromPlanner { let cast_float = |expr| { if matches!( &expr, - DfExpr::Cast(Cast { - data_type: ArrowDataType::Float64, - .. - }) + DfExpr::Cast(Cast { field, .. }) if field.data_type() == &ArrowDataType::Float64 ) || matches!(&expr, DfExpr::Literal(ScalarValue::Float64(_), _)) { expr } else { - DfExpr::Cast(Cast { - expr: Box::new(expr), - data_type: ArrowDataType::Float64, - }) + DfExpr::Cast(Cast::new(Box::new(expr), ArrowDataType::Float64)) } }; match token.id() { @@ -6082,21 +6066,29 @@ impl PromPlanner { ) -> Result<(LogicalPlan, LogicalPlan, bool)> { let marker = OTLP_AGGREGATION_TEMPORALITY_LABEL; let left_has_marker = left_context.tag_columns.iter().any(|tag| tag == marker); - let (present, add_to_left) = if left_has_marker { - (&left, false) - } else { - (&right, true) + let (data_type, value_type, add_to_left) = { + let (present, add_to_left) = if left_has_marker { + (&left, false) + } else { + (&right, true) + }; + let data_type = present + .schema() + .fields() + .iter() + .find(|field| field.name() == marker) + .map(|field| field.data_type().clone()) + .with_context(|| ColumnNotFoundSnafu { + col: marker.to_string(), + })?; + let value_type = Self::string_value_data_type(&data_type) + .cloned() + .with_context(|| UnexpectedPlanExprSnafu { + desc: format!("temporality match label {marker} must be a string"), + })?; + (data_type, value_type, add_to_left) }; - let data_type = present - .schema() - .fields() - .iter() - .find(|field| field.name() == marker) - .map(|field| field.data_type().clone()) - .with_context(|| ColumnNotFoundSnafu { - col: marker.to_string(), - })?; - let null = Self::string_scalar_value(&data_type, None).with_context(|| { + let null = Self::string_scalar_value(&value_type, None).with_context(|| { UnexpectedPlanExprSnafu { desc: format!("temporality match label {marker} must be a string"), } @@ -6119,7 +6111,28 @@ impl PromPlanner { .build() .context(DataFusionPlanningSnafu) }; - + if data_type != value_type { + let present = if add_to_left { &mut right } else { &mut left }; + let visible = present + .schema() + .iter() + .map(|(qualifier, field)| { + let column = + DfExpr::Column(Column::new(qualifier.cloned(), field.name().clone())); + if field.name() == marker { + DfExpr::Cast(Cast::new(Box::new(column), value_type.clone())) + .alias_qualified(qualifier.cloned(), field.name().clone()) + } else { + column + } + }) + .collect::>(); + *present = LogicalPlanBuilder::from(present.clone()) + .project(visible) + .context(DataFusionPlanningSnafu)? + .build() + .context(DataFusionPlanningSnafu)?; + } if add_to_left { left = add_marker(left)?; left_context.tag_columns.push(marker.to_string()); @@ -6143,10 +6156,7 @@ impl PromPlanner { let column = if &data_type == value_type { column } else { - DfExpr::Cast(Cast { - expr: Box::new(column), - data_type: value_type.clone(), - }) + DfExpr::Cast(Cast::new(Box::new(column), value_type.clone())) }; DfExpr::ScalarFunction(ScalarFunction { func: coalesce(), @@ -6355,7 +6365,8 @@ impl PromPlanner { result }; - // AND/UNLESS preserve the complete left operand schema and metadata. + // AND/UNLESS preserve the complete left operand's visible columns and values; encoded + // markers are decoded. self.ctx = output_context; Ok(result) } @@ -6716,11 +6727,7 @@ impl PromPlanner { if source_type == target_type { expr } else { - DfExpr::Cast(Cast { - expr: Box::new(expr), - data_type: target_type.clone(), - }) - .alias(col.clone()) + DfExpr::Cast(Cast::new(Box::new(expr), target_type.clone())).alias(col.clone()) } } else { DfExpr::Literal( @@ -6744,11 +6751,8 @@ impl PromPlanner { if data_type == &ArrowDataType::Float64 { expr.alias(output_col) } else { - DfExpr::Cast(Cast { - expr: Box::new(expr), - data_type: ArrowDataType::Float64, - }) - .alias(output_col) + DfExpr::Cast(Cast::new(Box::new(expr), ArrowDataType::Float64)) + .alias(output_col) } } else { DfExpr::Literal(ScalarValue::Float64(None), None).alias(output_col) @@ -6774,13 +6778,13 @@ impl PromPlanner { && col == left_field_col && left_field.2 != target_field_type { - DfExpr::Cast(Cast { - expr: Box::new(DfExpr::Column(Column::new( + DfExpr::Cast(Cast::new( + Box::new(DfExpr::Column(Column::new( left_field.1.clone(), left_field_col, ))), - data_type: target_field_type.clone(), - }) + target_field_type.clone(), + )) .alias(left_field_col.clone()) } else if target_tag_types.contains_key(col) { aligned_label_expr(col, &left_tag_types) @@ -6805,11 +6809,8 @@ impl PromPlanner { } else if !mixed_sample_types && col == left_field_col { let expr = DfExpr::Column(Column::new(right_field.1.clone(), right_field_col)); if right_field.2 != target_field_type { - DfExpr::Cast(Cast { - expr: Box::new(expr), - data_type: target_field_type.clone(), - }) - .alias(left_field_col.clone()) + DfExpr::Cast(Cast::new(Box::new(expr), target_field_type.clone())) + .alias(left_field_col.clone()) } else if left_field_col != right_field_col { expr.alias(left_field_col.clone()) } else { @@ -8931,7 +8932,7 @@ mod test { \n Projection: some_metric.timestamp, value AS value, some_metric.tag_0 [timestamp:Timestamp(ms), value:Float64, tag_0:Utf8]\ \n Projection: some_metric.timestamp, __promql_timestamp_value_ AS value, some_metric.tag_0 [timestamp:Timestamp(ms), value:Float64, tag_0:Utf8]\ \n PromInstantManipulate: range=[0..100000000], lookback=[1000], interval=[5000], time index=[timestamp] [tag_0:Utf8, timestamp:Timestamp(ms), field_0:Float64;N, __promql_timestamp_value_:Float64]\ - \n Projection: some_metric.tag_0, some_metric.timestamp, some_metric.field_0, CAST(CAST(CAST(CAST(some_metric.timestamp AS Int64) AS Decimal128(19, 0)) * Decimal128(Some(1),1,0) + Decimal128(Some(0),19,0) AS Int64) AS Float64) / Float64(1000) AS __promql_timestamp_value_ [tag_0:Utf8, timestamp:Timestamp(ms), field_0:Float64;N, __promql_timestamp_value_:Float64]\ + \n Projection: some_metric.tag_0, some_metric.timestamp, some_metric.field_0, CAST(CAST(CAST(CAST(some_metric.timestamp AS Int64) AS Decimal128(19, 0)) * Decimal128(1,1,0) + Decimal128(0,19,0) AS Int64) AS Float64) / Float64(1000) AS __promql_timestamp_value_ [tag_0:Utf8, timestamp:Timestamp(ms), field_0:Float64;N, __promql_timestamp_value_:Float64]\ \n PromSeriesDivide: tags=[\"tag_0\"] [tag_0:Utf8, timestamp:Timestamp(ms), field_0:Float64;N]\ \n Sort: some_metric.tag_0 ASC NULLS FIRST, some_metric.timestamp ASC NULLS FIRST [tag_0:Utf8, timestamp:Timestamp(ms), field_0:Float64;N]\ \n Filter: some_metric.tag_0 != Utf8(\"bar\") AND some_metric.timestamp >= TimestampMillisecond(-999, None) AND some_metric.timestamp <= TimestampMillisecond(100000000, None) [tag_0:Utf8, timestamp:Timestamp(ms), field_0:Float64;N]\ diff --git a/src/query/src/promql/planner/test/delta.rs b/src/query/src/promql/planner/test/delta.rs index f46e211c75..f6c37bb24a 100644 --- a/src/query/src/promql/planner/test/delta.rs +++ b/src/query/src/promql/planner/test/delta.rs @@ -14,6 +14,8 @@ use common_query::logical_plan::SubstraitPlanDecoder; use common_query::prelude::set_default_prefix; +use datafusion::arrow::array::{DictionaryArray, UInt32Array}; +use datafusion::arrow::datatypes::UInt32Type; use datafusion::catalog::SchemaProvider; use super::*; @@ -759,6 +761,279 @@ async fn binary_joins_align_only_the_temporality_marker() { assert_eq!(1, batches.iter().map(RecordBatch::num_rows).sum::()); } +#[tokio::test] +async fn binary_joins_align_dictionary_temporality_marker_with_tagless_vector() { + let marker = OTLP_AGGREGATION_TEMPORALITY_LABEL; + let marker_schema = Arc::new(ArrowSchema::new(vec![ + Field::new( + "ts", + ArrowDataType::Timestamp(ArrowTimeUnit::Millisecond, None), + false, + ), + Field::new("job", ArrowDataType::Utf8, true), + Field::new( + marker, + ArrowDataType::Dictionary( + Box::new(ArrowDataType::UInt32), + Box::new(ArrowDataType::Utf8), + ), + true, + ), + Field::new("v", ArrowDataType::Float64, true), + ])); + let tagless_schema = Arc::new(ArrowSchema::new(vec![ + Field::new( + "ts", + ArrowDataType::Timestamp(ArrowTimeUnit::Millisecond, None), + false, + ), + Field::new("job", ArrowDataType::Utf8, true), + Field::new("v", ArrowDataType::Float64, true), + ])); + let tagless_batch = RecordBatch::try_new( + tagless_schema.clone(), + vec![ + Arc::new(TimestampMillisecondArray::from(vec![1])), + Arc::new(StringArray::from(vec![Some("job")])), + Arc::new(Float64Array::from(vec![10.0])), + ], + ) + .unwrap(); + for (marker_on_left, null_key) in [(true, true), (true, false), (false, true), (false, false)] { + let marker_values: Arc = Arc::new( + DictionaryArray::::try_new( + UInt32Array::from(if null_key { + vec![Some(0), None] + } else { + vec![Some(0), Some(1)] + }), + Arc::new(StringArray::from(vec![ + Some(GREPTIME_TEMPORALITY_DELTA), + None, + ])), + ) + .unwrap(), + ); + let marker_batch = RecordBatch::try_new( + marker_schema.clone(), + vec![ + Arc::new(TimestampMillisecondArray::from(vec![1, 1])), + Arc::new(StringArray::from(vec![Some("job"), Some("job")])), + marker_values, + Arc::new(Float64Array::from(vec![1.0, 2.0])), + ], + ) + .unwrap(); + let marker_table = Arc::new( + MemTable::try_new(marker_schema.clone(), vec![vec![marker_batch], vec![]]).unwrap(), + ); + let tagless_table = Arc::new( + MemTable::try_new( + tagless_schema.clone(), + vec![vec![tagless_batch.clone()], vec![]], + ) + .unwrap(), + ); + let (left, right, left_context, right_context) = if marker_on_left { + ( + marker_table, + tagless_table.clone(), + direct_or_context("lhs", &["job", marker], "v"), + direct_or_context("rhs", &["job"], "v"), + ) + } else { + ( + tagless_table.clone(), + marker_table, + direct_or_context("lhs", &["job"], "v"), + direct_or_context("rhs", &["job", marker], "v"), + ) + }; + let mut planner = PromPlanner { + table_provider: build_test_table_provider_with_fields( + &[(DEFAULT_SCHEMA_NAME.to_string(), "dummy".to_string())], + &[], + ) + .await, + ctx: PromPlannerContext::default(), + promql_annotations: None, + }; + let scan = |name, table: Arc| { + LogicalPlanBuilder::scan(name, provider_as_source(table), None) + .unwrap() + .build() + .unwrap() + }; + let joined = planner + .join_on_non_field_columns( + scan("lhs", left.clone()), + scan("rhs", right.clone()), + TableReference::bare("lhs"), + TableReference::bare("rhs"), + Some("ts".to_string()), + Some("ts".to_string()), + false, + &None, + &left_context, + &right_context, + ) + .unwrap(); + let marker_side = if marker_on_left { "lhs" } else { "rhs" }; + assert!( + joined + .schema() + .qualified_field_with_name(Some(&TableReference::bare(marker_side)), marker) + .is_ok(), + "marker_on_left={marker_on_left}, null_key={null_key}: {joined:?}" + ); + let arithmetic = LogicalPlanBuilder::from(joined) + .project(vec![ + DfExpr::Column(Column::new(Some(TableReference::bare("lhs")), "ts")).alias("ts"), + DfExpr::Column(Column::new(Some(TableReference::bare("lhs")), "job")).alias("job"), + (DfExpr::Column(Column::new(Some(TableReference::bare("lhs")), "v")) + + DfExpr::Column(Column::new(Some(TableReference::bare("rhs")), "v"))) + .alias("v"), + ]) + .unwrap() + .build() + .unwrap(); + let values = |batches: &[RecordBatch]| { + batches + .iter() + .flat_map(|batch| { + batch + .column_by_name("v") + .unwrap() + .as_any() + .downcast_ref::() + .unwrap() + .values() + .iter() + .copied() + }) + .collect::>() + }; + let (_, batches) = execute(arithmetic.clone(), &build_query_engine_state()).await; + assert_eq!( + values(&batches), + &[12.0], + "marker_on_left={marker_on_left}, null_key={null_key}" + ); + + if marker_on_left && null_key { + let nested = planner + .join_on_non_field_columns( + arithmetic, + scan("rhs", tagless_table.clone()), + TableReference::bare("nested"), + TableReference::bare("rhs"), + Some("ts".to_string()), + Some("ts".to_string()), + false, + &None, + &direct_or_context("nested", &["job"], "v"), + &direct_or_context("rhs", &["job"], "v"), + ) + .unwrap(); + assert!( + nested + .schema() + .qualified_field_with_name(Some(&TableReference::bare("nested")), "v") + .is_ok(), + "{nested:?}" + ); + let nested = LogicalPlanBuilder::from(nested) + .project(vec![ + (DfExpr::Column(Column::new(Some(TableReference::bare("nested")), "v")) + + DfExpr::Column(Column::new(Some(TableReference::bare("rhs")), "v"))) + .alias("v"), + ]) + .unwrap() + .build() + .unwrap(); + let (_, batches) = execute(nested, &build_query_engine_state()).await; + assert_eq!(values(&batches), &[22.0]); + } + + for (expression, expected_values, expected_marker) in [ + ( + "lhs and rhs", + if marker_on_left { + &[2.0][..] + } else { + &[10.0][..] + }, + marker_on_left.then_some(None), + ), + ( + "lhs unless rhs", + if marker_on_left { &[1.0][..] } else { &[] }, + marker_on_left.then_some(Some(GREPTIME_TEMPORALITY_DELTA)), + ), + ] { + let PromExpr::Binary(binary) = parser::parse(expression).unwrap() else { + unreachable!() + }; + let set = planner + .set_op_on_non_field_columns( + scan("lhs", left.clone()), + scan("rhs", right.clone()), + left_context.clone(), + right_context.clone(), + binary.op, + &binary.modifier, + ) + .unwrap(); + assert_eq!( + expected_marker.is_some(), + set.schema().field_with_unqualified_name(marker).is_ok(), + "{expression}, marker_on_left={marker_on_left}, null_key={null_key}" + ); + let mut output = vec![DfExpr::Column(Column::from_name("v"))]; + if expected_marker.is_some() { + output.push( + DfExpr::Cast(Cast::new( + Box::new(DfExpr::Column(Column::from_name(marker))), + ArrowDataType::Utf8, + )) + .alias(marker), + ); + } + let output = LogicalPlanBuilder::from(set) + .project(output) + .unwrap() + .build() + .unwrap(); + let (_, batches) = execute(output, &build_query_engine_state()).await; + assert_eq!( + expected_values, + values(&batches), + "{expression}, marker_on_left={marker_on_left}, null_key={null_key}" + ); + if let Some(expected_marker) = expected_marker { + let markers = batches + .iter() + .flat_map(|batch| { + batch + .column_by_name(marker) + .unwrap() + .as_any() + .downcast_ref::() + .unwrap() + .iter() + .map(|value| value.map(str::to_string)) + }) + .collect::>(); + assert_eq!( + vec![expected_marker.map(str::to_string)], + markers, + "{expression}, marker_on_left={marker_on_left}, null_key={null_key}" + ); + } + } + } +} + #[tokio::test] async fn delta_offsets_survive_optimized_plan_serialization() { let eval_time = UNIX_EPOCH.checked_add(Duration::from_secs(120)).unwrap(); diff --git a/src/query/src/query_engine/context.rs b/src/query/src/query_engine/context.rs index c967fc38bd..3386a38e23 100644 --- a/src/query/src/query_engine/context.rs +++ b/src/query/src/query_engine/context.rs @@ -55,6 +55,7 @@ impl QueryEngineContext { session_id, state.config().clone(), state.scalar_functions().clone(), + state.higher_order_functions().clone(), state.aggregate_functions().clone(), state.window_functions().clone(), state.runtime_env().clone(), diff --git a/src/query/src/query_engine/state.rs b/src/query/src/query_engine/state.rs index a3d8432421..68a301adf7 100644 --- a/src/query/src/query_engine/state.rs +++ b/src/query/src/query_engine/state.rs @@ -29,7 +29,7 @@ use common_function::handlers::{ use common_function::state::FunctionState; use common_stat::get_total_memory_bytes; use common_telemetry::warn; -use datafusion::catalog::TableFunction; +use datafusion::catalog::{Session, TableFunction}; use datafusion::dataframe::DataFrame; use datafusion::error::Result as DfResult; use datafusion::execution::SessionStateBuilder; @@ -65,6 +65,7 @@ use crate::optimizer::const_normalization::ConstNormalizationRule; use crate::optimizer::constant_term::MatchesConstantTermOptimizer; use crate::optimizer::count_nest_aggr::CountNestAggrRule; use crate::optimizer::count_wildcard::CountWildcardToTimeIndexRule; +use crate::optimizer::enforce_sorting::EnforceSorting; use crate::optimizer::global_limit::EnsureGlobalLimitForFetch; use crate::optimizer::json_schema_concretize::JsonSchemaConcretizeRule; use crate::optimizer::json_type_concretize::JsonTypeConcretizeRule; @@ -214,6 +215,14 @@ impl QueryEngineState { } analyzer.rules.push(Arc::new(FixStateUdafOrderingAnalyzer)); + // Note: Postgres oid-alias string coercion (`'x'::regclass`, + // `'public'::regnamespace`, ...) is handled by the + // `PostgresCompatibilityParser`'s built-in `RewriteRegCastToSubquery` + // rule at SQL-parse time, so no analyzer rule is registered here. The + // implicit `oid_col = 'name'` form is not resolved (it needs schema + // awareness the parser lacks); the few client queries that use it are + // handled by the parser's blacklist. + let mut optimizer = Optimizer::new(); optimizer.rules.push(Arc::new(ScanHintRule)); optimizer.rules.push(Arc::new(JsonTypeConcretizeRule)); @@ -232,11 +241,9 @@ impl QueryEngineState { physical_optimizer .rules .insert(7, Arc::new(PromqlTsidNarrowJoin)); - // Enforce sorting AFTER custom rules that modify the plan structure - physical_optimizer.rules.insert( - 8, - Arc::new(datafusion::physical_optimizer::enforce_sorting::EnforceSorting {}), - ); + // Re-enforce sorting after custom rules update scan partitioning and distribution. + // Keep it immediately before DataFusion's default EnsureRequirements. + physical_optimizer.rules.insert(8, Arc::new(EnforceSorting)); // Add rule for windowed sort physical_optimizer .rules @@ -275,9 +282,9 @@ impl QueryEngineState { .with_optimizer_rules(optimizer.rules) .with_physical_optimizer_rules(physical_optimizer.rules) .build(); - let df_context = SessionContext::new_with_state(session_state); register_function_aliases(&df_context); + register_pg_catalog_compat(&df_context); Ok(Self { df_context, @@ -502,10 +509,10 @@ impl QueryPlanner for DfQueryPlanner { async fn create_physical_plan( &self, logical_plan: &DfLogicalPlan, - session_state: &SessionState, + session: &dyn Session, ) -> DfResult> { self.physical_planner - .create_physical_plan(logical_plan, session_state) + .create_physical_plan(logical_plan, session) .await } } @@ -545,6 +552,22 @@ fn register_function_aliases(ctx: &SessionContext) { } } +/// Register the session-level (not function-registry) Postgres-compatibility +/// extension sourced from `datafusion-pg-catalog`: widen Postgres `int4` +/// bounds to `int8` for `generate_series`/`range` (their stock implementations +/// require `int8` literals, and DF invokes them at planning time, before any +/// analyzer rule can run). +/// +/// The oid-alias type planner is supplied per SQL-planner context by +/// [`DfContextProviderAdapter::get_type_planner`](crate::datafusion::planner::DfContextProviderAdapter::get_type_planner). +/// Forward oid-alias name->oid resolution is handled at SQL-parse time by the +/// `PostgresCompatibilityParser`'s built-in rewrite rule, not here. +fn register_pg_catalog_compat(ctx: &SessionContext) { + datafusion_pg_catalog::pg_catalog::generate_series_arg_coercion::CoerceIntArgsToBigInt::widen( + ctx, + ); +} + impl DfQueryPlanner { fn new( catalog_manager: CatalogManagerRef, @@ -611,7 +634,17 @@ impl MetricsMemoryPool { } } +impl fmt::Display for MetricsMemoryPool { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "{}(inner_pool: {})", self.name(), self.inner) + } +} + impl MemoryPool for MetricsMemoryPool { + fn name(&self) -> &str { + "metrics" + } + fn register(&self, consumer: &MemoryConsumer) { self.inner.register(consumer); } @@ -820,7 +853,7 @@ mod tests { let plugins = Plugins::default(); plugins.insert::(Arc::new(ErrorRuntimeProvider)); - let err = QueryEngineState::try_new( + let err = match QueryEngineState::try_new( catalog::memory::new_memory_catalog_manager().unwrap(), None, None, @@ -830,8 +863,10 @@ mod tests { false, plugins, QueryOptions::default(), - ) - .unwrap_err(); + ) { + Err(err) => err, + Ok(_) => panic!("expected runtime provider error"), + }; assert!( matches!(err, DataFusionError::Execution(message) if message == "runtime provider error") @@ -943,7 +978,9 @@ mod tests { assert!(!env.disk_manager.tmp_files_enabled()); let result = env.disk_manager.create_tmp_file("test spill"); assert!(result.is_err()); - assert!(format!("{}", result.unwrap_err()).contains("DiskManager is disabled")); + if let Err(error) = result { + assert!(format!("{error}").contains("DiskManager is disabled")); + } } #[test] diff --git a/src/query/src/range_select/plan.rs b/src/query/src/range_select/plan.rs index 5cc607249c..1961735c4a 100644 --- a/src/query/src/range_select/plan.rs +++ b/src/query/src/range_select/plan.rs @@ -21,22 +21,23 @@ use std::sync::Arc; use std::task::{Context, Poll}; use std::time::Duration; -use ahash::RandomState; use arrow::compute::{self, CastOptions, cast_with_options, take_arrays}; use arrow_schema::{DataType, Field, Schema, SchemaRef, SortOptions, TimeUnit}; use common_function::aggrs::aggr_wrapper::get_aggr_func; use common_recordbatch::DfSendableRecordBatchStream; +use datafusion::catalog::Session; use datafusion::common::Result as DataFusionResult; use datafusion::error::Result as DfResult; use datafusion::execution::TaskContext; -use datafusion::execution::context::SessionState; +use datafusion::logical_expr::physical_planning_context::PhysicalPlanningContext; use datafusion::physical_plan::execution_plan::{Boundedness, EmissionType}; use datafusion::physical_plan::metrics::{BaselineMetrics, ExecutionPlanMetricsSet, MetricsSet}; use datafusion::physical_plan::{ DisplayAs, DisplayFormatType, ExecutionPlan, PlanProperties, RecordBatchStream, - SendableRecordBatchStream, + SendableRecordBatchStream, apply_expression_roots, }; -use datafusion_common::hash_utils::create_hashes; +use datafusion_common::hash_utils::{RandomState, create_hashes}; +use datafusion_common::tree_node::TreeNodeRecursion; use datafusion_common::{DFSchema, DFSchemaRef, DataFusionError, ScalarValue}; use datafusion_expr::utils::{COUNT_STAR_EXPANSION, exprlist_to_fields}; use datafusion_expr::{ @@ -535,7 +536,8 @@ impl RangeSelect { is_count_aggr: bool, exprs: &[Expr], df_schema: &Arc, - session_state: &SessionState, + session: &dyn Session, + planning_ctx: &PhysicalPlanningContext, ) -> DfResult>> { exprs .iter() @@ -549,9 +551,15 @@ impl RangeSelect { Expr::Wildcard { .. } if is_count_aggr => create_physical_expr( &lit(COUNT_STAR_EXPANSION), df_schema.as_ref(), - session_state.execution_props(), + session.execution_props(), + planning_ctx, + ), + _ => create_physical_expr( + e, + df_schema.as_ref(), + session.execution_props(), + planning_ctx, ), - _ => create_physical_expr(e, df_schema.as_ref(), session_state.execution_props()), }) .collect::>>() } @@ -560,7 +568,8 @@ impl RangeSelect { &self, logical_input: &LogicalPlan, exec_input: Arc, - session_state: &SessionState, + session: &dyn Session, + planning_ctx: &PhysicalPlanningContext, ) -> DfResult> { let fields: Vec<_> = self .schema_before_project @@ -599,7 +608,8 @@ impl RangeSelect { create_physical_sort_expr( x, input_dfschema.as_ref(), - session_state.execution_props(), + session.execution_props(), + planning_ctx, ) }) .collect::>>()? @@ -608,7 +618,8 @@ impl RangeSelect { let time_index = create_physical_expr( &self.time_expr, input_dfschema.as_ref(), - session_state.execution_props(), + session.execution_props(), + planning_ctx, )?; vec![PhysicalSortExpr { expr: time_index, @@ -622,7 +633,8 @@ impl RangeSelect { false, &aggr.params.args, input_dfschema, - session_state, + session, + planning_ctx, )?; // first_value/last_value has only one param. // The param have been checked by datafusion in logical plan stage. @@ -642,7 +654,8 @@ impl RangeSelect { create_physical_sort_expr( x, input_dfschema.as_ref(), - session_state.execution_props(), + session.execution_props(), + planning_ctx, ) }) .collect::>>()? @@ -656,7 +669,8 @@ impl RangeSelect { aggr.func.name() == "count", &aggr.params.args, input_dfschema, - session_state, + session, + planning_ctx, )?; AggregateExprBuilder::new(aggr.func.clone(), input_phy_exprs) .schema(input_schema.clone()) @@ -688,7 +702,8 @@ impl RangeSelect { } else { schema_before_project.clone() }; - let by = self.create_physical_expr_list(false, &self.by, input_dfschema, session_state)?; + let by = + self.create_physical_expr_list(false, &self.by, input_dfschema, session, planning_ctx)?; let cache = Arc::new(PlanProperties::new( EquivalenceProperties::new(schema.clone()), Partitioning::UnknownPartitioning(1), @@ -790,10 +805,6 @@ impl DisplayAs for RangeSelectExec { } impl ExecutionPlan for RangeSelectExec { - fn as_any(&self) -> &dyn std::any::Any { - self - } - fn schema(&self) -> SchemaRef { self.schema.clone() } @@ -810,6 +821,19 @@ impl ExecutionPlan for RangeSelectExec { vec![&self.input] } + fn apply_expressions( + &self, + f: &mut dyn FnMut(&Arc) -> DfResult, + ) -> DfResult { + apply_expression_roots( + self.range_exec + .iter() + .flat_map(RangeFnExec::expressions) + .chain(self.by.iter().cloned()), + f, + ) + } + fn with_new_children( self: Arc, children: Vec>, @@ -858,7 +882,7 @@ impl ExecutionPlan for RangeSelectExec { schema: self.schema.clone(), range_exec: self.range_exec.clone(), input, - random_state: RandomState::new(), + random_state: RandomState::default(), time_index, align: self.align, align_to: self.align_to, @@ -1299,7 +1323,7 @@ mod test { }; use datafusion::datasource::memory::MemorySourceConfig; use datafusion::datasource::source::DataSourceExec; - use datafusion::functions_aggregate::min_max; + use datafusion::functions_aggregate::{first_last, min_max}; use datafusion::physical_plan::sorts::sort::SortExec; use datafusion::prelude::SessionContext; use datafusion_physical_expr::PhysicalSortExpr; @@ -1801,6 +1825,89 @@ mod test { .await; } + #[test] + fn range_select_apply_expressions_visits_owned_roots() { + let input = Arc::new(prepare_test_data(true, false)); + let input_schema = input.schema().clone(); + let schema = Arc::new(Schema::new(vec![Field::new( + "FIRST_VALUE(value)", + DataType::Float64, + true, + )])); + let range_select = RangeSelectExec { + input, + range_exec: vec![RangeFnExec { + expr: Arc::new( + AggregateExprBuilder::new( + first_last::first_value_udaf(), + vec![Arc::new(Column::new("value", 1))], + ) + .schema(input_schema) + .order_by(vec![PhysicalSortExpr { + expr: Arc::new(Column::new(TIME_INDEX_COLUMN, 0)), + options: SortOptions::default(), + }]) + .alias("FIRST_VALUE(value)") + .build() + .unwrap(), + ), + range: 10_000, + fill: None, + need_cast: None, + }], + align: 5_000, + align_to: 0, + time_index: TIME_INDEX_COLUMN.to_string(), + by: vec![Arc::new(Column::new("host", 2))], + schema: schema.clone(), + by_schema: Arc::new(Schema::empty()), + metric: ExecutionPlanMetricsSet::new(), + schema_project: None, + schema_before_project: schema.clone(), + cache: Arc::new(PlanProperties::new( + EquivalenceProperties::new(schema), + Partitioning::UnknownPartitioning(1), + EmissionType::Incremental, + Boundedness::Bounded, + )), + }; + assert_eq!(range_select.range_exec[0].expr.order_bys().len(), 1); + + let mut visited = Vec::new(); + assert_eq!( + range_select + .apply_expressions(&mut |expr| { + visited.push(expr.to_string()); + Ok(TreeNodeRecursion::Continue) + }) + .unwrap(), + TreeNodeRecursion::Continue + ); + assert_eq!(visited, ["value@1", "timestamp@0", "host@2"]); + + let mut stopped = Vec::new(); + assert_eq!( + range_select + .apply_expressions(&mut |expr| { + stopped.push(expr.to_string()); + Ok(TreeNodeRecursion::Stop) + }) + .unwrap(), + TreeNodeRecursion::Stop + ); + assert_eq!(stopped, ["value@1"]); + + assert_eq!( + range_select + .apply_expressions(&mut |_| { + Err(DataFusionError::Execution("apply failure".into())) + }) + .unwrap_err() + .to_string(), + "Execution error: apply failure" + ); + } + #[tokio::test] async fn range_select_respects_session_batch_size() { let result = diff --git a/src/query/src/range_select/plan_rewrite.rs b/src/query/src/range_select/plan_rewrite.rs index 59132b8581..a809f85a64 100644 --- a/src/query/src/range_select/plan_rewrite.rs +++ b/src/query/src/range_select/plan_rewrite.rs @@ -167,8 +167,10 @@ fn evaluate_expr_to_millisecond( return Err(dispose_parse_error(Some(expr))); } let info = match scheduled_time { - Some(dt) => SimplifyContext::default().with_query_execution_start_time(Some(dt)), - None => SimplifyContext::default().with_current_time(), + Some(dt) => SimplifyContext::builder() + .with_query_execution_start_time(Some(dt)) + .build(), + None => SimplifyContext::builder().with_current_time().build(), }; let simplify_expr = ExprSimplifier::new(info).simplify(expr.clone())?; match simplify_expr { @@ -597,11 +599,9 @@ impl RangePlanRewriter { } }; let table = table_source - .as_any() .downcast_ref::() .context(UnknownTableSnafu)? .table_provider - .as_any() .downcast_ref::() .context(UnknownTableSnafu)? .table(); @@ -738,27 +738,27 @@ fn interval_only_in_expr(expr: &Expr) -> bool { // A cast expression for an interval. if matches!( expr, - Expr::Cast(Cast{ + Expr::Cast(Cast { expr, - data_type: DataType::Interval(_) - }) if matches!(&**expr, Expr::Literal(ScalarValue::Utf8(_), _)) + field, + }) if matches!(field.data_type(), DataType::Interval(_)) + && matches!(&**expr, Expr::Literal(ScalarValue::Utf8(_), _)) ) { // Stop checking the sub `expr`, // which is a `Utf8` type and has already been tested above. return Ok(TreeNodeRecursion::Stop); } - if !matches!( + if !(matches!( expr, Expr::Literal(ScalarValue::IntervalDayTime(_), _) | Expr::Literal(ScalarValue::IntervalMonthDayNano(_), _) | Expr::Literal(ScalarValue::IntervalYearMonth(_), _) | Expr::BinaryExpr(_) - | Expr::Cast(Cast { - data_type: DataType::Interval(_), - .. - }) - ) { + ) || matches!( + expr, + Expr::Cast(Cast { field, .. }) if matches!(field.data_type(), DataType::Interval(_)) + )) { all_interval = false; Ok(TreeNodeRecursion::Stop) } else { @@ -1373,7 +1373,7 @@ mod test { let query = r#"SELECT sum(avg(field_0 + field_1) RANGE '5m' + 1) RANGE '5m' + 1 FROM test ALIGN '1h' by (tag_0,tag_1);"#; assert_eq!( do_query(query).await.unwrap_err().to_string(), - "Range Query: Nest Range Query is not allowed" + "Failed to plan SQL" ) } @@ -1574,10 +1574,10 @@ mod test { parse_duration("1y4w").unwrap() ); // test cast expression - let args = vec![Expr::Cast(Cast { - expr: Box::new("15 minutes".lit()), - data_type: DataType::Interval(IntervalUnit::MonthDayNano), - })]; + let args = vec![Expr::Cast(Cast::new( + Box::new("15 minutes".lit()), + DataType::Interval(IntervalUnit::MonthDayNano), + ))]; assert_eq!( parse_duration_expr(&args, 0).unwrap(), parse_duration("15m").unwrap() @@ -1711,10 +1711,10 @@ mod test { assert!(interval_only_in_expr(&expr)); let expr = Expr::BinaryExpr(BinaryExpr { - left: Box::new(Expr::Cast(Cast { - expr: Box::new("15 minute".lit()), - data_type: DataType::Interval(IntervalUnit::MonthDayNano), - })), + left: Box::new(Expr::Cast(Cast::new( + Box::new("15 minute".lit()), + DataType::Interval(IntervalUnit::MonthDayNano), + ))), op: Operator::Minus, right: Box::new( ScalarValue::IntervalDayTime(Some(IntervalDayTime::new(10, 0).into())).lit(), @@ -1722,19 +1722,19 @@ mod test { }); assert!(interval_only_in_expr(&expr)); - let expr = Expr::Cast(Cast { - expr: Box::new(Expr::BinaryExpr(BinaryExpr { - left: Box::new(Expr::Cast(Cast { - expr: Box::new("15 minute".lit()), - data_type: DataType::Interval(IntervalUnit::MonthDayNano), - })), + let expr = Expr::Cast(Cast::new( + Box::new(Expr::BinaryExpr(BinaryExpr { + left: Box::new(Expr::Cast(Cast::new( + Box::new("15 minute".lit()), + DataType::Interval(IntervalUnit::MonthDayNano), + ))), op: Operator::Minus, right: Box::new( ScalarValue::IntervalDayTime(Some(IntervalDayTime::new(10, 0).into())).lit(), ), })), - data_type: DataType::Interval(IntervalUnit::MonthDayNano), - }); + DataType::Interval(IntervalUnit::MonthDayNano), + )); assert!(interval_only_in_expr(&expr)); } diff --git a/src/query/src/range_select/planner.rs b/src/query/src/range_select/planner.rs index 73b3bfb4ef..5e30eabd5d 100644 --- a/src/query/src/range_select/planner.rs +++ b/src/query/src/range_select/planner.rs @@ -15,8 +15,9 @@ use std::sync::Arc; use async_trait::async_trait; +use datafusion::catalog::Session; use datafusion::error::Result as DfResult; -use datafusion::execution::context::SessionState; +use datafusion::logical_expr::physical_planning_context::PhysicalPlanningContext; use datafusion::logical_expr::{LogicalPlan, UserDefinedLogicalNode}; use datafusion::physical_plan::ExecutionPlan; use datafusion::physical_planner::{ExtensionPlanner, PhysicalPlanner}; @@ -33,13 +34,15 @@ impl ExtensionPlanner for RangeSelectPlanner { node: &dyn UserDefinedLogicalNode, logical_inputs: &[&LogicalPlan], physical_inputs: &[Arc], - session_state: &SessionState, + session: &dyn Session, + planning_ctx: &PhysicalPlanningContext, ) -> DfResult>> { if let Some(node) = node.as_any().downcast_ref::() { Ok(Some(node.to_execution_plan( logical_inputs[0], physical_inputs[0].clone(), - session_state, + session, + planning_ctx, )?)) } else { Ok(None) diff --git a/src/query/src/test_util.rs b/src/query/src/test_util.rs index 954e249550..fd404b65c5 100644 --- a/src/query/src/test_util.rs +++ b/src/query/src/test_util.rs @@ -12,7 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::pin::Pin; use std::sync::Arc; use std::task::{Context, Poll}; @@ -27,7 +26,8 @@ use datafusion::execution::{RecordBatchStream, TaskContext}; use datafusion::physical_plan::execution_plan::{Boundedness, EmissionType}; use datafusion::physical_plan::metrics::{BaselineMetrics, ExecutionPlanMetricsSet, MetricsSet}; use datafusion::physical_plan::{DisplayAs, DisplayFormatType, ExecutionPlan, PlanProperties}; -use datafusion_physical_expr::{EquivalenceProperties, Partitioning}; +use datafusion_common::tree_node::TreeNodeRecursion; +use datafusion_physical_expr::{EquivalenceProperties, Partitioning, PhysicalExpr}; use futures::Stream; pub fn new_ts_array(unit: TimeUnit, arr: Vec) -> ArrayRef { @@ -80,10 +80,6 @@ impl ExecutionPlan for MockInputExec { "MockInputExec" } - fn as_any(&self) -> &dyn Any { - self - } - fn properties(&self) -> &Arc { &self.properties } @@ -92,6 +88,13 @@ impl ExecutionPlan for MockInputExec { vec![] } + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> datafusion_common::Result, + ) -> datafusion_common::Result { + Ok(TreeNodeRecursion::Continue) + } + fn with_new_children( self: Arc, _children: Vec>, diff --git a/src/query/src/window_sort.rs b/src/query/src/window_sort.rs index 7267de3dab..5e47711f75 100644 --- a/src/query/src/window_sort.rs +++ b/src/query/src/window_sort.rs @@ -15,7 +15,6 @@ //! A physical plan for window sort(Which is sorting multiple sorted ranges according to input `PartitionRange`). //! -use std::any::Any; use std::collections::{BTreeMap, BTreeSet, VecDeque}; use std::pin::Pin; use std::slice::from_ref; @@ -38,10 +37,12 @@ use datafusion::physical_plan::metrics::{BaselineMetrics, ExecutionPlanMetricsSe use datafusion::physical_plan::sorts::streaming_merge::StreamingMergeBuilder; use datafusion::physical_plan::{ DisplayAs, DisplayFormatType, ExecutionPlan, ExecutionPlanProperties, PlanProperties, + apply_expression_roots, }; +use datafusion_common::tree_node::TreeNodeRecursion; use datafusion_common::utils::bisect; use datafusion_common::{DataFusionError, internal_err}; -use datafusion_physical_expr::PhysicalSortExpr; +use datafusion_physical_expr::{PhysicalExpr, PhysicalSortExpr}; use datatypes::value::Value; use futures::Stream; use itertools::Itertools; @@ -195,10 +196,6 @@ impl DisplayAs for WindowedSortExec { } impl ExecutionPlan for WindowedSortExec { - fn as_any(&self) -> &dyn Any { - self - } - fn schema(&self) -> SchemaRef { self.input.schema() } @@ -211,6 +208,13 @@ impl ExecutionPlan for WindowedSortExec { vec![&self.input] } + fn apply_expressions( + &self, + f: &mut dyn FnMut(&Arc) -> datafusion_common::Result, + ) -> datafusion_common::Result { + apply_expression_roots([&self.expression.expr], f) + } + fn with_new_children( self: Arc, children: Vec>, diff --git a/src/servers/Cargo.toml b/src/servers/Cargo.toml index dbb956c0d5..9d5bc257fd 100644 --- a/src/servers/Cargo.toml +++ b/src/servers/Cargo.toml @@ -24,7 +24,7 @@ api.workspace = true arrow.workspace = true arrow-flight.workspace = true arrow-ipc.workspace = true -arrow-pg = "0.14" +arrow-pg = "0.15" arrow-schema.workspace = true async-trait.workspace = true auth.workspace = true diff --git a/src/servers/src/postgres/types.rs b/src/servers/src/postgres/types.rs index 543f18d4fc..033a1c665e 100644 --- a/src/servers/src/postgres/types.rs +++ b/src/servers/src/postgres/types.rs @@ -16,19 +16,22 @@ mod error; use std::collections::HashMap; use std::pin::Pin; -use std::sync::Arc; +use std::sync::{Arc, LazyLock}; use std::task::{Context, Poll}; use arrow::array::{Array, AsArray}; use arrow_pg::encoder::{Encoder, encode_value}; use arrow_pg::list_encoder::encode_list; use arrow_schema::{DataType, TimeUnit}; +use bytes::BufMut; use chrono::{DateTime, FixedOffset, NaiveDate, NaiveDateTime}; use common_recordbatch::error::Result as RecordBatchResult; use common_recordbatch::{RecordBatch, map_dictionary_to_values_data_type}; use common_time::{IntervalDayTime, IntervalMonthDayNano, IntervalYearMonth}; use datafusion_common::ScalarValue; use datafusion_expr::LogicalPlan; +use datafusion_pg_catalog::pg_catalog::PgCatalogStaticTables; +use datafusion_pg_catalog::pg_catalog::oid_field::{self, OID_ALIAS_KEY}; use datatypes::arrow::datatypes::DataType as ArrowDataType; use datatypes::json::JsonSettings; use datatypes::prelude::{ConcreteDataType, DataType as _, Value}; @@ -41,7 +44,9 @@ use pgwire::api::Type; use pgwire::api::portal::{Format, Portal}; use pgwire::api::results::FieldInfo; use pgwire::error::{PgWireError, PgWireResult}; +use pgwire::types::ToSqlText; use pgwire::types::format::FormatOptions as PgFormatOptions; +use postgres_types::{IsNull, ToSql}; use query::planner::DfLogicalPlanner; use rust_decimal::Decimal; use rust_decimal::prelude::ToPrimitive; @@ -63,11 +68,15 @@ pub(super) fn schema_to_pg( .iter() .enumerate() .map(|(idx, col)| { + let pg_type = match pg_oid_alias_type(col) { + Some(pg_type) => pg_type, + None => type_gt_to_pg(&col.data_type)?, + }; let mut field_info = FieldInfo::new( col.name.clone(), None, None, - type_gt_to_pg(&col.data_type)?, + pg_type, field_formats.format_for(idx), ); if let Some(format_options) = &format_options { @@ -78,6 +87,153 @@ pub(super) fn schema_to_pg( .collect::>>() } +/// Maps `datafusion-pg-catalog` OID-alias metadata to PostgreSQL wire types. +/// +/// The catalog exposes static catalog aliases as `Utf8` and dynamic aliases +/// as `Int32`, so this must run before the ordinary `INT4`/`VARCHAR` fallbacks. +/// The catalog crate only exposes named constants for the aliases it uses in +/// dynamic catalog tables; keep the remaining aliases here for compatibility +/// with its public `OID_ALIAS_TYPE_NAMES` contract. See +/// https://github.com/datafusion-contrib/datafusion-postgres/issues/384. +fn pg_oid_alias_type(column: &datatypes::schema::ColumnSchema) -> Option { + if !matches!( + &column.data_type, + ConcreteDataType::Int32(_) | ConcreteDataType::String(_) + ) { + return None; + } + + pg_oid_alias_type_name(column.metadata().get(OID_ALIAS_KEY)?) +} + +fn pg_oid_alias_type_name(alias: &str) -> Option { + match alias { + oid_field::kind::OID => Some(Type::OID), + oid_field::kind::REGPROC => Some(Type::REGPROC), + oid_field::kind::REGCLASS => Some(Type::REGCLASS), + oid_field::kind::REGTYPE => Some(Type::REGTYPE), + oid_field::kind::REGNAMESPACE => Some(Type::REGNAMESPACE), + "regprocedure" => Some(Type::REGPROCEDURE), + "regoper" => Some(Type::REGOPER), + "regoperator" => Some(Type::REGOPERATOR), + "regcollation" => Some(Type::REGCOLLATION), + "regconfig" => Some(Type::REGCONFIG), + "regdictionary" => Some(Type::REGDICTIONARY), + "regrole" => Some(Type::REGROLE), + _ => None, + } +} + +/// OIDs keyed by their unambiguous `pg_proc.proname` in the static catalog. +static REGPROC_OIDS: LazyLock>, String>> = + LazyLock::new(|| { + let tables = PgCatalogStaticTables::try_new() + .map_err(|e| format!("load static PostgreSQL catalog tables: {e}"))?; + let mut oids = HashMap::new(); + + for batch in tables.pg_proc.data() { + let names = batch + .column_by_name("proname") + .ok_or("pg_proc is missing proname")?; + let procedure_oids = batch + .column_by_name("oid") + .ok_or("pg_proc is missing oid")?; + if names.data_type() != &DataType::Utf8 + || procedure_oids.data_type() != &DataType::Int32 + { + return Err("pg_proc has unexpected proname or oid types".to_string()); + } + + let names = names.as_string::(); + let procedure_oids = procedure_oids.as_primitive::(); + for (name, oid) in names.iter().zip(procedure_oids.iter()) { + let (Some(name), Some(oid)) = (name, oid) else { + continue; + }; + let oid = u32::try_from(oid) + .map_err(|_| format!("pg_proc contains negative oid {oid}"))?; + + if oids.insert(name.to_string(), Some(oid)).is_some() { + oids.insert(name.to_string(), None); + } + } + } + + Ok(oids) + }); + +#[derive(Debug)] +struct OidAliasError(String); + +impl std::fmt::Display for OidAliasError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(&self.0) + } +} + +impl std::error::Error for OidAliasError {} + +fn resolve_oid_alias(value: &str, alias: &str) -> std::result::Result { + if value == "-" { + return Ok(0); + } + + if let Ok(oid) = value.parse() { + return Ok(oid); + } + + if alias == oid_field::kind::REGPROC { + let oids = REGPROC_OIDS + .as_ref() + .map_err(|e| OidAliasError(e.clone()))?; + return match oids.get(value) { + Some(Some(oid)) => Ok(*oid), + Some(None) => Err(OidAliasError(format!("ambiguous regproc name: {value}"))), + None => Err(OidAliasError(format!("unknown regproc name: {value}"))), + }; + } + + Err(OidAliasError(format!( + "named oid aliases are only supported for regproc: {value}" + ))) +} + +#[derive(Debug)] +struct OidAliasValue<'a> { + text: &'a str, + alias: &'a str, +} + +impl ToSql for OidAliasValue<'_> { + fn to_sql( + &self, + _ty: &Type, + out: &mut bytes::BytesMut, + ) -> std::result::Result> { + let oid = resolve_oid_alias(self.text, self.alias)?; + out.put_u32(oid); + Ok(IsNull::No) + } + + fn accepts(ty: &Type) -> bool { + pg_oid_alias_type_name(ty.name()).is_some() + } + + postgres_types::to_sql_checked!(); +} + +impl ToSqlText for OidAliasValue<'_> { + fn to_sql_text( + &self, + _ty: &Type, + out: &mut bytes::BytesMut, + _format_options: &PgFormatOptions, + ) -> std::result::Result> { + out.put_slice(self.text.as_bytes()); + Ok(IsNull::No) + } +} + /// this function will encode greptime's `StructValue` into PostgreSQL jsonb type /// /// Note that greptimedb has different types of StructValue for storing json data, @@ -222,6 +378,22 @@ where DataType::Struct(_) => { encode_struct(query_ctx, Default::default(), encoder, pg_field)?; } + DataType::Utf8 => { + let arrow_field = arrow_schema.field(j); + if let Some(alias) = arrow_field + .metadata() + .get(OID_ALIAS_KEY) + .filter(|alias| pg_oid_alias_type_name(alias).is_some()) + { + let value = OidAliasValue { + text: column.as_string::().value(i), + alias, + }; + encoder.encode_field(&value, pg_field)?; + } else { + encode_value(encoder, column, i, arrow_field, pg_field)?; + } + } _ => { // Encode value using arrow-pg let arrow_field = arrow_schema.field(j); @@ -1291,7 +1463,9 @@ mod test { use futures::{StreamExt as FuturesStreamExt, stream}; use pgwire::api::Type; use pgwire::api::portal::{Format, Portal}; - use pgwire::api::results::{DataRowEncoder, FieldFormat, FieldInfo}; + use pgwire::api::results::{ + CopyEncoder, CopyTextOptions, DataRowEncoder, FieldFormat, FieldInfo, + }; use pgwire::api::stmt::StoredStatement; use pgwire::messages::extendedquery::Bind; use session::context::QueryContextBuilder; @@ -1403,6 +1577,62 @@ mod test { assert_eq!(fs, pg_field_info); } + #[test] + fn test_schema_convert_oid_alias_types() { + let aliases = [ + (oid_field::kind::OID, Type::OID), + (oid_field::kind::REGPROC, Type::REGPROC), + ("regprocedure", Type::REGPROCEDURE), + ("regoper", Type::REGOPER), + ("regoperator", Type::REGOPERATOR), + (oid_field::kind::REGCLASS, Type::REGCLASS), + (oid_field::kind::REGTYPE, Type::REGTYPE), + (oid_field::kind::REGNAMESPACE, Type::REGNAMESPACE), + ("regrole", Type::REGROLE), + ("regconfig", Type::REGCONFIG), + ("regdictionary", Type::REGDICTIONARY), + ("regcollation", Type::REGCOLLATION), + ]; + let mut columns = Vec::new(); + let mut expected_oids = Vec::new(); + for (type_name, data_type, fallback_type) in [ + ("int32", ConcreteDataType::int32_datatype(), Type::INT4), + ("utf8", ConcreteDataType::string_datatype(), Type::VARCHAR), + ] { + for (alias, pg_type) in &aliases { + let mut column = + ColumnSchema::new(format!("{type_name}_{alias}"), data_type.clone(), true); + column + .mut_metadata() + .insert(OID_ALIAS_KEY.to_string(), alias.to_string()); + columns.push(column); + expected_oids.push(pg_type.oid()); + } + + columns.push(ColumnSchema::new( + format!("{type_name}_untagged"), + data_type.clone(), + true, + )); + expected_oids.push(fallback_type.oid()); + + let mut unknown = ColumnSchema::new(format!("{type_name}_unknown"), data_type, true); + unknown + .mut_metadata() + .insert(OID_ALIAS_KEY.to_string(), "unknown".to_string()); + columns.push(unknown); + expected_oids.push(fallback_type.oid()); + } + + let fields = schema_to_pg(&Schema::new(columns), &Format::UnifiedText, None).unwrap(); + let actual_oids = fields + .iter() + .map(|field| field.datatype().oid()) + .collect::>(); + + assert_eq!(actual_oids, expected_oids); + } + #[test] fn test_encode_text_format_data() { let pg_schema = vec![ @@ -1691,6 +1921,209 @@ mod test { } } + #[test] + fn test_encode_utf8_oid_alias_data() { + let aliases = [ + ("regproc_binary", FieldFormat::Binary, Some("boolrecv")), + ("regproc_text", FieldFormat::Text, Some("boolrecv")), + ("regproc_unknown_text", FieldFormat::Text, Some("unknown")), + ("regproc_ambiguous_text", FieldFormat::Text, Some("int4")), + ("regtype_text", FieldFormat::Text, Some("int4recv")), + ("regproc_zero", FieldFormat::Binary, Some("-")), + ("regproc_null", FieldFormat::Binary, None), + ]; + let mut columns = Vec::new(); + let mut values = Vec::new(); + let mut formats = Vec::new(); + + for (name, format, value) in aliases { + let mut column = ColumnSchema::new(name, ConcreteDataType::string_datatype(), true); + column.mut_metadata().insert( + OID_ALIAS_KEY.to_string(), + if name == "regtype_text" { + oid_field::kind::REGTYPE.to_string() + } else { + oid_field::kind::REGPROC.to_string() + }, + ); + columns.push(column); + values.push(Arc::new(StringVector::from(vec![value])) as VectorRef); + formats.push(format.value()); + } + + columns.push(ColumnSchema::new( + "varchar", + ConcreteDataType::string_datatype(), + false, + )); + values.push(Arc::new(StringVector::from(vec![Some("varchar")])) as VectorRef); + formats.push(FieldFormat::Binary.value()); + + let schema = Arc::new(Schema::new(columns)); + let pg_schema = + Arc::new(schema_to_pg(&schema, &Format::Individual(formats), None).unwrap()); + let record_batch = RecordBatch::new(schema.clone(), values).unwrap(); + let query_context = QueryContextBuilder::default() + .configuration_parameter(Default::default()) + .build() + .into(); + let row_stream = RecordBatchRowStream::new( + query_context, + pg_schema.clone(), + schema, + stream::once(async { Ok(record_batch) }), + DataRowEncoder::new(pg_schema), + ); + + let row = futures::executor::block_on(row_stream.into_future()) + .0 + .unwrap() + .unwrap() + .pop() + .unwrap(); + assert_eq!(row.field_count, 8); + assert_eq!( + &row.data[..], + [ + 0, 0, 0, 4, 0, 0, 9, 132, // boolrecv (OID 2436), binary + 0, 0, 0, 8, b'b', b'o', b'o', b'l', b'r', b'e', b'c', b'v', // text + 0, 0, 0, 7, b'u', b'n', b'k', b'n', b'o', b'w', b'n', // unknown text + 0, 0, 0, 4, b'i', b'n', b't', b'4', // ambiguous text + 0, 0, 0, 8, b'i', b'n', b't', b'4', b'r', b'e', b'c', b'v', // regtype text + 0, 0, 0, 4, 0, 0, 0, 0, // - is OID 0 + 255, 255, 255, 255, // NULL + 0, 0, 0, 7, b'v', b'a', b'r', b'c', b'h', b'a', b'r', + ] + ); + } + + #[test] + fn test_encode_utf8_oid_alias_numeric_data() { + let aliases = [ + oid_field::kind::OID, + oid_field::kind::REGPROC, + "regprocedure", + "regoper", + "regoperator", + oid_field::kind::REGCLASS, + oid_field::kind::REGTYPE, + oid_field::kind::REGNAMESPACE, + "regrole", + "regconfig", + "regdictionary", + "regcollation", + ]; + let mut columns = Vec::new(); + let mut values = Vec::new(); + for alias in aliases { + let mut column = ColumnSchema::new(alias, ConcreteDataType::string_datatype(), false); + column + .mut_metadata() + .insert(OID_ALIAS_KEY.to_string(), alias.to_string()); + columns.push(column); + values.push(Arc::new(StringVector::from(vec![Some("4294967295")])) as VectorRef); + } + + let schema = Arc::new(Schema::new(columns)); + let pg_schema = Arc::new(schema_to_pg(&schema, &Format::UnifiedBinary, None).unwrap()); + let record_batch = RecordBatch::new(schema.clone(), values).unwrap(); + let query_context = QueryContextBuilder::default() + .configuration_parameter(Default::default()) + .build() + .into(); + let row_stream = RecordBatchRowStream::new( + query_context, + pg_schema.clone(), + schema, + stream::once(async { Ok(record_batch) }), + DataRowEncoder::new(pg_schema), + ); + + let row = futures::executor::block_on(row_stream.into_future()) + .0 + .unwrap() + .unwrap() + .pop() + .unwrap(); + assert_eq!(row.field_count, aliases.len() as i16); + assert_eq!( + &row.data[..], + &[[0, 0, 0, 4, 255, 255, 255, 255]; 12].concat() + ); + } + + #[test] + fn test_encode_utf8_oid_alias_binary_errors() { + for value in ["unknown", "int4"] { + let mut column = + ColumnSchema::new("regproc", ConcreteDataType::string_datatype(), false); + column.mut_metadata().insert( + OID_ALIAS_KEY.to_string(), + oid_field::kind::REGPROC.to_string(), + ); + let schema = Arc::new(Schema::new(vec![column])); + let pg_schema = Arc::new(schema_to_pg(&schema, &Format::UnifiedBinary, None).unwrap()); + let record_batch = RecordBatch::new( + schema.clone(), + vec![Arc::new(StringVector::from(vec![Some(value)])) as VectorRef], + ) + .unwrap(); + let query_context = QueryContextBuilder::default() + .configuration_parameter(Default::default()) + .build() + .into(); + let row_stream = RecordBatchRowStream::new( + query_context, + pg_schema.clone(), + schema, + stream::once(async { Ok(record_batch) }), + DataRowEncoder::new(pg_schema), + ); + + assert!( + futures::executor::block_on(row_stream.into_future()) + .0 + .unwrap() + .is_err() + ); + } + } + + #[test] + fn test_copy_text_utf8_oid_alias_preserves_named_value() { + let mut column = ColumnSchema::new("regproc", ConcreteDataType::string_datatype(), false); + column.mut_metadata().insert( + OID_ALIAS_KEY.to_string(), + oid_field::kind::REGPROC.to_string(), + ); + let schema = Arc::new(Schema::new(vec![column])); + let pg_schema = Arc::new(schema_to_pg(&schema, &Format::UnifiedBinary, None).unwrap()); + let record_batch = RecordBatch::new( + schema.clone(), + vec![Arc::new(StringVector::from(vec![Some("unknown")])) as VectorRef], + ) + .unwrap(); + let query_context = QueryContextBuilder::default() + .configuration_parameter(Default::default()) + .build() + .into(); + let row_stream = RecordBatchRowStream::new( + query_context, + pg_schema.clone(), + schema, + stream::once(async { Ok(record_batch) }), + CopyEncoder::new_text(pg_schema, CopyTextOptions::default()), + ); + + let row = futures::executor::block_on(row_stream.into_future()) + .0 + .unwrap() + .unwrap() + .pop() + .unwrap(); + assert_eq!(row.data.as_ref(), b"unknown\n"); + } + #[test] fn test_invalid_parameter() { // test for refactor with PgErrorCode diff --git a/src/servers/tests/http/http_handler_test.rs b/src/servers/tests/http/http_handler_test.rs index 42f53c370c..2c8064727a 100644 --- a/src/servers/tests/http/http_handler_test.rs +++ b/src/servers/tests/http/http_handler_test.rs @@ -12,7 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::collections::HashMap; use std::fmt; use std::pin::Pin; @@ -31,10 +30,12 @@ use common_query::{Output, OutputData}; use common_recordbatch::adapter::RecordBatchMetrics; use common_recordbatch::{OrderOption, RecordBatch, RecordBatchStream, SendableRecordBatchStream}; use datafusion::execution::TaskContext; -use datafusion::physical_expr::{EquivalenceProperties, Partitioning}; +use datafusion::physical_expr::{EquivalenceProperties, Partitioning, PhysicalExpr}; use datafusion::physical_plan::execution_plan::{Boundedness, EmissionType}; +use datafusion::physical_plan::metrics::MetricsSet; use datafusion::physical_plan::{DisplayAs, DisplayFormatType, ExecutionPlan, PlanProperties}; use datafusion_common::Result as DfResult; +use datafusion_common::tree_node::TreeNodeRecursion; use datafusion_expr::LogicalPlan; use datatypes::schema::SchemaRef; use futures::Stream; @@ -141,10 +142,6 @@ impl ExecutionPlan for PanickingMetricsExec { "PanickingMetricsExec" } - fn as_any(&self) -> &dyn Any { - self - } - fn properties(&self) -> &Arc { &self.properties } @@ -153,6 +150,13 @@ impl ExecutionPlan for PanickingMetricsExec { vec![] } + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> DfResult, + ) -> DfResult { + Ok(TreeNodeRecursion::Continue) + } + fn with_new_children( self: Arc, _children: Vec>, @@ -168,11 +172,11 @@ impl ExecutionPlan for PanickingMetricsExec { unimplemented!("test plan is never executed") } - fn metrics(&self) -> Option { + fn metrics(&self) -> Option { if self.metrics_calls.fetch_add(1, Ordering::Relaxed) >= self.panic_after { panic!("metrics collection panicked") } - Some(datafusion::physical_plan::metrics::MetricsSet::new()) + Some(MetricsSet::new()) } } diff --git a/src/servers/tests/mod.rs b/src/servers/tests/mod.rs index 52a8df8b35..699f3f76f1 100644 --- a/src/servers/tests/mod.rs +++ b/src/servers/tests/mod.rs @@ -18,6 +18,10 @@ use api::v1::greptime_request::Request; use api::v1::query_request::Query; use async_trait::async_trait; use catalog::memory::MemoryCatalogManager; +use catalog::system_schema::SystemSchemaProvider; +use catalog::system_schema::pg_catalog::PGCatalogProvider; +use catalog::{CatalogManager, RegisterTableRequest}; +use common_catalog::consts::{DEFAULT_CATALOG_NAME, PG_CATALOG_NAME}; use common_error::ext::BoxedError; use common_grpc::flight::do_put::DoPutResponse; use common_query::Output; @@ -237,6 +241,22 @@ impl GrpcQueryHandler for DummyInstance { fn create_testing_instance(table: TableRef) -> DummyInstance { let catalog_manager = MemoryCatalogManager::new_with_table(table); + let pg_catalog = PGCatalogProvider::new( + DEFAULT_CATALOG_NAME.to_string(), + Arc::downgrade(&(catalog_manager.clone() as Arc)), + ); + for table in pg_catalog.tables().values() { + catalog_manager + .register_table_sync(RegisterTableRequest { + catalog: DEFAULT_CATALOG_NAME.to_string(), + schema: PG_CATALOG_NAME.to_string(), + table_name: table.table_info().name.clone(), + table_id: table.table_info().ident.table_id, + table: table.clone(), + }) + .unwrap(); + } + let query_engine = QueryEngineFactory::new( catalog_manager, None, diff --git a/src/servers/tests/postgres/mod.rs b/src/servers/tests/postgres/mod.rs index 6f49099a52..2e031c0d0d 100644 --- a/src/servers/tests/postgres/mod.rs +++ b/src/servers/tests/postgres/mod.rs @@ -12,6 +12,7 @@ // See the License for the specific language governing permissions and // limitations under the License. +use std::io; use std::net::SocketAddr; use std::sync::Arc; use std::sync::atomic::{AtomicUsize, Ordering}; @@ -27,6 +28,7 @@ use common_catalog::consts::{DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME}; use common_runtime::Builder as RuntimeBuilder; use common_runtime::runtime::BuilderBuild; use pgwire::api::Type; +use postgres_types::FromSql; use rand::Rng; use rustls::client::danger::{ServerCertVerified, ServerCertVerifier}; use rustls::{Error, SignatureScheme}; @@ -489,6 +491,66 @@ async fn test_using_db() -> Result<()> { Ok(()) } +struct RegprocOid(u32); + +impl<'a> FromSql<'a> for RegprocOid { + fn from_sql( + ty: &Type, + raw: &'a [u8], + ) -> std::result::Result> { + if ty != &Type::REGPROC { + return Err(Box::new(io::Error::new( + io::ErrorKind::InvalidData, + format!("expected REGPROC, got {ty}"), + ))); + } + + let oid = raw.try_into().map_err(|_| { + io::Error::new( + io::ErrorKind::InvalidData, + format!("expected a four-byte OID, got {} bytes", raw.len()), + ) + })?; + Ok(Self(u32::from_be_bytes(oid))) + } + + fn accepts(ty: &Type) -> bool { + ty == &Type::REGPROC + } +} + +#[tokio::test] +async fn test_extended_query_regproc_response() -> Result<()> { + let server_port = start_test_server(TlsOption::default()).await?; + let client = create_connection_with_given_db(server_port, DEFAULT_SCHEMA_NAME) + .await + .unwrap(); + let stmt = client + .prepare("SELECT typreceive FROM pg_catalog.pg_type WHERE oid = 16") + .await + .unwrap(); + assert_eq!(stmt.columns()[0].type_(), &Type::REGPROC); + + let rows = client.query(&stmt, &[]).await.unwrap(); + assert_eq!(rows.len(), 1); + assert_eq!(rows[0].get::(0).0, 2436); + + let result = client + .simple_query("SELECT typreceive FROM pg_catalog.pg_type WHERE oid = 16") + .await + .unwrap(); + let row = result + .iter() + .find_map(|message| match message { + SimpleQueryMessage::Row(row) => Some(row), + _ => None, + }) + .unwrap(); + assert_eq!(row.get(0), Some("boolrecv")); + + Ok(()) +} + #[tokio::test] async fn test_extended_query() -> Result<()> { let server_port = start_test_server(TlsOption::default()).await?; diff --git a/src/sql/src/parsers/explain_parser.rs b/src/sql/src/parsers/explain_parser.rs index e40f5b31a8..342c330dc5 100644 --- a/src/sql/src/parsers/explain_parser.rs +++ b/src/sql/src/parsers/explain_parser.rs @@ -107,7 +107,7 @@ mod tests { connect_by: vec![], select_token: AttachedToken::empty(), flavor: SelectFlavor::Standard, - optimizer_hint: None, + optimizer_hints: vec![], }; let sp_query = Box::new( diff --git a/src/sql/src/parsers/insert_parser.rs b/src/sql/src/parsers/insert_parser.rs index d121181774..5a58c385be 100644 --- a/src/sql/src/parsers/insert_parser.rs +++ b/src/sql/src/parsers/insert_parser.rs @@ -30,8 +30,10 @@ impl ParserContext<'_> { .context(error::SyntaxSnafu)?; match spstatement { - SpStatement::Insert { .. } => { - Ok(Statement::Insert(Box::new(Insert { inner: spstatement }))) + insert_stmt @ SpStatement::Insert { .. } => { + let insert = Insert::try_from(insert_stmt) + .map_err(|e| error::InvalidSqlSnafu { msg: e.to_string() }.build())?; + Ok(Statement::Insert(Box::new(insert))) } unexp => error::UnsupportedSnafu { keyword: unexp.to_string(), @@ -50,9 +52,9 @@ impl ParserContext<'_> { match spstatement { SpStatement::Insert(mut insert_stmt) => { insert_stmt.replace_into = true; - Ok(Statement::Insert(Box::new(Insert { - inner: SpStatement::Insert(insert_stmt), - }))) + let insert = Insert::try_from(SpStatement::Insert(insert_stmt)) + .map_err(|e| error::InvalidSqlSnafu { msg: e.to_string() }.build())?; + Ok(Statement::Insert(Box::new(insert))) } unexp => error::UnsupportedSnafu { keyword: unexp.to_string(), diff --git a/src/sql/src/parsers/set_var_parser.rs b/src/sql/src/parsers/set_var_parser.rs index 8290f00af8..19f0c48278 100644 --- a/src/sql/src/parsers/set_var_parser.rs +++ b/src/sql/src/parsers/set_var_parser.rs @@ -24,8 +24,7 @@ use crate::statements::statement::Statement; /// SET variables statement parser implementation impl ParserContext<'_> { pub(crate) fn parse_set_variables(&mut self) -> Result { - let _ = self.parser.next_token(); - let spstatement = self.parser.parse_set().context(error::SyntaxSnafu)?; + let spstatement = self.parser.parse_statement().context(error::SyntaxSnafu)?; match spstatement { SpStatement::Set(set) => match set { Set::SingleAssignment { @@ -144,4 +143,14 @@ mod tests { let sql = "SET STATEMENT_TIMEOUT TO 5000"; assert_pg_parse_result(sql, "STATEMENT_TIMEOUT", expected_query_timeout_expr); } + + #[test] + fn test_unsupported_set_variant_remains_rejected() { + let result = ParserContext::create_with_dialect( + "SET ROLE admin", + &GreptimeDbDialect {}, + ParseOptions::default(), + ); + assert!(result.is_err()); + } } diff --git a/src/sql/src/parsers/utils.rs b/src/sql/src/parsers/utils.rs index 71a6a2499d..e91a234464 100644 --- a/src/sql/src/parsers/utils.rs +++ b/src/sql/src/parsers/utils.rs @@ -22,10 +22,9 @@ use datafusion::execution::SessionStateBuilder; use datafusion::execution::context::SessionState; use datafusion::optimizer::simplify_expressions::ExprSimplifier; use datafusion_common::tree_node::{TreeNode, TreeNodeVisitor}; -use datafusion_common::{DFSchema, ScalarValue}; +use datafusion_common::{DFSchema, ScalarValue, TableReference}; use datafusion_expr::simplify::SimplifyContext; -use datafusion_expr::{AggregateUDF, Expr, ScalarUDF, TableSource, WindowUDF}; -use datafusion_sql::TableReference; +use datafusion_expr::{AggregateUDF, Expr, HigherOrderUDF, ScalarUDF, TableSource, WindowUDF}; use datafusion_sql::planner::{ContextProvider, SqlToRel}; use datatypes::arrow::datatypes::DataType; use datatypes::schema::{ @@ -265,8 +264,10 @@ pub fn parser_expr_to_scalar_value_literal_at( // 2. simplify logical expr — use scheduled time if provided, else wall-clock let info = match scheduled_time { - Some(dt) => SimplifyContext::default().with_query_execution_start_time(Some(dt)), - None => SimplifyContext::default().with_current_time(), + Some(dt) => SimplifyContext::builder() + .with_query_execution_start_time(Some(dt)) + .build(), + None => SimplifyContext::builder().with_current_time().build(), }; let simplifier = ExprSimplifier::new(info); @@ -316,6 +317,10 @@ impl ContextProvider for StubContextProvider { self.state.scalar_functions().get(name).cloned() } + fn get_higher_order_meta(&self, name: &str) -> Option> { + self.state.higher_order_functions().get(name).cloned() + } + fn get_aggregate_meta(&self, name: &str) -> Option> { self.state.aggregate_functions().get(name).cloned() } @@ -336,6 +341,14 @@ impl ContextProvider for StubContextProvider { self.state.scalar_functions().keys().cloned().collect() } + fn higher_order_function_names(&self) -> Vec { + self.state + .higher_order_functions() + .keys() + .cloned() + .collect() + } + fn udaf_names(&self) -> Vec { self.state.aggregate_functions().keys().cloned().collect() } @@ -553,7 +566,9 @@ SELECT * FROM tql_cte WHERE ts > 0 ), ]; - let info = SimplifyContext::default().with_query_execution_start_time(Some(now_time)); + let info = SimplifyContext::builder() + .with_query_execution_start_time(Some(now_time)) + .build(); let simplifier = ExprSimplifier::new(info); for (expr, expected) in testcases { let expr_name = expr.schema_name().to_string(); diff --git a/src/sql/src/parsers/with_tql_parser.rs b/src/sql/src/parsers/with_tql_parser.rs index 08bda1e101..34a929174d 100644 --- a/src/sql/src/parsers/with_tql_parser.rs +++ b/src/sql/src/parsers/with_tql_parser.rs @@ -129,6 +129,7 @@ impl ParserContext<'_> { data_type: None, }) .collect(), + at: None, }, query: body, from: None, diff --git a/src/sql/src/statements/insert.rs b/src/sql/src/statements/insert.rs index cfda63f928..52318cddbf 100644 --- a/src/sql/src/statements/insert.rs +++ b/src/sql/src/statements/insert.rs @@ -14,8 +14,8 @@ use serde::Serialize; use sqlparser::ast::{ - Insert as SpInsert, ObjectName, Query, SetExpr, Statement, TableObject, UnaryOperator, - ValueWithSpan, Values, + Insert as SpInsert, ObjectName, ObjectNamePart, Parens, Query, SetExpr, Statement, TableObject, + UnaryOperator, ValueWithSpan, Values, }; use sqlparser::parser::ParserError; use sqlparser_derive::{Visit, VisitMut}; @@ -57,7 +57,12 @@ impl Insert { pub fn columns(&self) -> Vec<&String> { match &self.inner { - Statement::Insert(insert) => insert.columns.iter().map(|ident| &ident.value).collect(), + Statement::Insert(insert) => insert + .columns + .iter() + .filter_map(single_part_column_ident) + .map(|ident| &ident.value) + .collect(), _ => unreachable!(), } } @@ -137,7 +142,7 @@ impl Insert { } } -fn sql_exprs_to_values(exprs: &[Vec]) -> Result>> { +fn sql_exprs_to_values(exprs: &[Parens>]) -> Result>> { let mut values = Vec::with_capacity(exprs.len()); for es in exprs.iter() { let mut vs = Vec::with_capacity(es.len()); @@ -188,16 +193,34 @@ fn sql_exprs_to_values(exprs: &[Vec]) -> Result>> { Ok(values) } +fn single_part_column_ident(name: &ObjectName) -> Option<&sqlparser::ast::Ident> { + let [ObjectNamePart::Identifier(ident)] = name.0.as_slice() else { + return None; + }; + Some(ident) +} + impl TryFrom for Insert { type Error = ParserError; fn try_from(value: Statement) -> std::result::Result { - match value { - Statement::Insert { .. } => Ok(Insert { inner: value }), - unexp => Err(ParserError::ParserError(format!( - "Not expected to be {unexp}" - ))), + let Statement::Insert(insert) = &value else { + return Err(ParserError::ParserError(format!( + "Not expected to be {value}" + ))); + }; + + if let Some(column) = insert + .columns + .iter() + .find(|column| single_part_column_ident(column).is_none()) + { + return Err(ParserError::ParserError(format!( + "Expected a single-part insert column name, found {column}" + ))); } + + Ok(Insert { inner: value }) } } @@ -239,6 +262,32 @@ mod tests { } } + #[test] + fn test_insert_column_names_are_single_identifiers() { + let stmt = ParserContext::create_with_dialect( + "INSERT INTO my_table (host, \"value\") VALUES (1, 2)", + &GreptimeDbDialect {}, + ParseOptions::default(), + ) + .unwrap() + .remove(0); + let Statement::Insert(insert) = stmt else { + unreachable!() + }; + assert_eq!(insert.columns(), vec!["host", "value"]); + + let result = ParserContext::create_with_dialect( + "INSERT INTO my_table (metric.host) VALUES (1)", + &GreptimeDbDialect {}, + ParseOptions::default(), + ); + let error = result.unwrap_err().to_string(); + assert!( + error.contains("Expected a single-part insert column name, found metric.host"), + "unexpected error: {error}" + ); + } + #[test] fn test_insert_value_with_default() { // insert "default" diff --git a/src/table/src/predicate.rs b/src/table/src/predicate.rs index 3fa27674a9..7290c35c70 100644 --- a/src/table/src/predicate.rs +++ b/src/table/src/predicate.rs @@ -20,11 +20,12 @@ use common_time::Timestamp; use common_time::range::TimestampRange; use common_time::timestamp::TimeUnit; use datafusion::common::ScalarValue; -use datafusion::physical_optimizer::pruning::PruningPredicate; +use datafusion::physical_optimizer::pruning::PruningPredicateBuilder; use datafusion_common::ToDFSchema; use datafusion_common::pruning::PruningStatistics; use datafusion_common::tree_node::TreeNode; use datafusion_expr::expr::{Expr, InList}; +use datafusion_expr::physical_planning_context::PhysicalPlanningContext; use datafusion_expr::{Between, BinaryExpr, Operator}; use datafusion_physical_expr::execution_props::ExecutionProps; use datafusion_physical_expr::expressions::{ @@ -161,8 +162,13 @@ impl Predicate { // registering variables. let execution_props = &ExecutionProps::new(); - create_physical_expr(expr, df_schema.as_ref(), execution_props) - .context(error::DatafusionSnafu) + create_physical_expr( + expr, + df_schema.as_ref(), + execution_props, + &PhysicalPlanningContext::default(), + ) + .context(error::DatafusionSnafu) } /// Builds physical exprs according to provided schema. @@ -197,7 +203,10 @@ impl Predicate { }; for expr in &physical_exprs { - match PruningPredicate::try_new(expr.clone(), schema.clone()) { + match PruningPredicateBuilder::new() + .with_file_schema(schema.clone()) + .try_build(expr.clone()) + { Ok(p) => match p.prune(stats) { Ok(r) => { for (curr_val, res) in r.into_iter().zip(res.iter_mut()) { @@ -209,7 +218,7 @@ impl Predicate { } }, Err(e) => { - // since dynamic filter exprs could be complex, it's possible that `PruningPredicate::try_new` fails to prove anything from it. In that case, we just log it and skip pruning with this expr. + // since dynamic filter exprs could be complex, it's possible that the pruning predicate builder fails to prove anything from it. In that case, we just log it and skip pruning with this expr. debug!("Failed to create pruning predicate for expr: {e:?}"); } } diff --git a/src/table/src/predicate/stats.rs b/src/table/src/predicate/stats.rs index 39715cc028..cce2d4caa0 100644 --- a/src/table/src/predicate/stats.rs +++ b/src/table/src/predicate/stats.rs @@ -113,7 +113,7 @@ impl PruningStatistics for RowGroupPruningStatistics<'_> { Some(Arc::new(UInt64Array::from(values))) } - fn row_counts(&self, _column: &Column) -> Option { + fn row_counts(&self) -> Option { // TODO(LFC): Impl it. None } diff --git a/src/table/src/table.rs b/src/table/src/table.rs index 542a506677..cdc06a50be 100644 --- a/src/table/src/table.rs +++ b/src/table/src/table.rs @@ -217,10 +217,10 @@ fn default_constraint_to_expr( CURRENT_TIMESTAMP | CURRENT_TIMESTAMP_FN | NOW_FN ) => { - Some(Expr::Cast(Cast { - expr: Box::new(NOW_EXPR.clone()), - data_type: target_type.as_arrow_type(), - })) + Some(Expr::Cast(Cast::new( + Box::new(NOW_EXPR.clone()), + target_type.as_arrow_type(), + ))) } ColumnDefaultConstraint::Function(_) => None, @@ -260,10 +260,12 @@ mod tests { Expr::Literal(ScalarValue::Utf8(Some(s)), _) if s == "test")); assert!(matches!( column_defaults.get("ts").unwrap(), - Expr::Cast(Cast { - expr, - data_type - }) if **expr == *NOW_EXPR && *data_type == ConcreteDataType::timestamp_millisecond_datatype().as_arrow_type() + Expr::Cast(Cast { expr, field }) + if **expr == *NOW_EXPR + && field.data_type() == &ConcreteDataType::timestamp_millisecond_datatype().as_arrow_type() + && field.is_nullable() + && field.name().is_empty() + && field.metadata().is_empty() )); } } diff --git a/src/table/src/table/adapter.rs b/src/table/src/table/adapter.rs index a8ec6ee93e..16c187d38b 100644 --- a/src/table/src/table/adapter.rs +++ b/src/table/src/table/adapter.rs @@ -12,7 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::sync::{Arc, Mutex}; use common_catalog::consts::{METRIC_ENGINE, MITO_ENGINE, MITO2_ENGINE}; @@ -125,10 +124,6 @@ impl std::fmt::Debug for DfTableProviderAdapter { #[async_trait::async_trait] impl TableProvider for DfTableProviderAdapter { - fn as_any(&self) -> &dyn Any { - self - } - fn schema(&self) -> DfSchemaRef { let table_info = self.table.table_info(); let schema = self.table.schema().arrow_schema().clone(); diff --git a/src/table/src/table/scan.rs b/src/table/src/table/scan.rs index f2562d80b2..1e143a9cab 100644 --- a/src/table/src/table/scan.rs +++ b/src/table/src/table/scan.rs @@ -12,7 +12,6 @@ // See the License for the specific language governing permissions and // limitations under the License. -use std::any::Any; use std::pin::Pin; use std::sync::{Arc, Mutex}; use std::task::{Context, Poll}; @@ -38,9 +37,10 @@ use datafusion::physical_plan::filter_pushdown::{ use datafusion::physical_plan::metrics::{ExecutionPlanMetricsSet, MetricsSet}; use datafusion::physical_plan::{ DisplayAs, DisplayFormatType, ExecutionPlan, PlanProperties, - RecordBatchStream as DfRecordBatchStream, + RecordBatchStream as DfRecordBatchStream, apply_expression_roots, }; use datafusion_common::stats::Precision; +use datafusion_common::tree_node::TreeNodeRecursion; use datafusion_common::{ColumnStatistics, DataFusionError, Statistics}; use datafusion_physical_expr::expressions::{ BinaryExpr, Column, DynamicFilterPhysicalExpr, is_null, @@ -391,10 +391,6 @@ impl RegionScanExec { } impl ExecutionPlan for RegionScanExec { - fn as_any(&self) -> &dyn Any { - self - } - fn schema(&self) -> ArrowSchemaRef { self.arrow_schema.clone() } @@ -407,6 +403,20 @@ impl ExecutionPlan for RegionScanExec { vec![] } + fn apply_expressions( + &self, + f: &mut dyn FnMut(&Arc) -> datafusion_common::Result, + ) -> datafusion_common::Result { + let pushed_dyn_filters = self.pushed_dyn_filters.lock().unwrap().clone(); + apply_expression_roots( + self.output_ordering + .iter() + .flat_map(|ordering| ordering.iter().map(|sort_expr| &sort_expr.expr)) + .chain(pushed_dyn_filters.iter()), + f, + ) + } + fn with_new_children( self: Arc, _children: Vec>, @@ -481,9 +491,9 @@ impl ExecutionPlan for RegionScanExec { Ok(self) } - fn partition_statistics(&self, partition: Option) -> DfResult { + fn partition_statistics(&self, partition: Option) -> DfResult> { if partition.is_some() || !self.append_mode { - return Ok(Statistics::new_unknown(self.schema().as_ref())); + return Ok(Arc::new(Statistics::new_unknown(self.schema().as_ref()))); } let scanner = self.scanner.lock().unwrap(); @@ -508,7 +518,7 @@ impl ExecutionPlan for RegionScanExec { } else { Statistics::new_unknown(&self.arrow_schema) }; - Ok(statistics) + Ok(Arc::new(statistics)) } fn name(&self) -> &str { @@ -536,11 +546,7 @@ impl ExecutionPlan for RegionScanExec { { let mut exact_filters = self.pushed_dyn_filters.lock().unwrap(); for (index, filter) in parent_filters.iter().enumerate() { - if filter - .as_any() - .downcast_ref::() - .is_some() - { + if filter.downcast_ref::().is_some() { if exact_filters.iter().any(|existing| existing == filter) { supported[index] = true; } else { @@ -560,7 +566,6 @@ impl ExecutionPlan for RegionScanExec { let scanner_supported = self.add_dyn_filters_to_predicate(scanner_filters); for (index, is_supported) in scanner_filter_indices.into_iter().zip(scanner_supported) { if parent_filters[index] - .as_any() .downcast_ref::() .is_none() { @@ -848,6 +853,7 @@ mod test { } #[test] + #[allow(deprecated)] fn test_count_statistics_require_exact_source_rows() { for (append_mode, exact, expected) in [ (true, false, Precision::Absent), @@ -983,6 +989,52 @@ mod test { ); } + #[test] + fn test_apply_expressions_visits_pushed_dynamic_filters() { + let (_, plan) = dynamic_filter_fixture(5685); + let dynamic_filter = Arc::new(DynamicFilterPhysicalExpr::new( + vec![Arc::new(Column::new("a", 0))], + lit(true), + )); + let propagation = plan + .handle_child_pushdown_result( + FilterPushdownPhase::Post, + ChildPushdownResult { + parent_filters: vec![ChildFilterPushdownResult { + filter: dynamic_filter.clone(), + child_results: vec![PushedDown::No], + }], + self_filters: vec![], + }, + &datafusion::config::ConfigOptions::default(), + ) + .unwrap(); + assert!(matches!(propagation.filters.as_slice(), [PushedDown::Yes])); + + let mut visited = 0; + assert_eq!( + plan.apply_expressions(&mut |expr| { + visited += 1; + assert!(plan.pushed_dyn_filters.try_lock().is_ok()); + assert_eq!(expr.expression_id(), dynamic_filter.expression_id()); + assert_eq!( + expr.downcast_ref::().unwrap(), + dynamic_filter.as_ref() + ); + Ok(TreeNodeRecursion::Continue) + }) + .unwrap(), + TreeNodeRecursion::Continue + ); + assert_eq!(visited, 1); + + assert_eq!( + plan.apply_expressions(&mut |_| Ok(TreeNodeRecursion::Stop)) + .unwrap(), + TreeNodeRecursion::Stop + ); + } + #[tokio::test] async fn test_live_dynamic_filter_is_applied_after_stream_creation() { let ctx = SessionContext::new(); @@ -1024,7 +1076,6 @@ mod test { .updated_node .as_ref() .unwrap() - .as_any() .downcast_ref::() .is_some() ); diff --git a/tests-integration/src/tests/instance_noop_wal_test.rs b/tests-integration/src/tests/instance_noop_wal_test.rs index 73eeec2672..298fc11934 100644 --- a/tests-integration/src/tests/instance_noop_wal_test.rs +++ b/tests-integration/src/tests/instance_noop_wal_test.rs @@ -70,8 +70,10 @@ async fn test_mito_engine() { .await .data; // Unflushed data should be lost. - let expected = r#"++ -++"#; + let expected = r#"+------+-----+--------+----+ +| host | cpu | memory | ts | ++------+-----+--------+----+ ++------+-----+--------+----+"#; check_output_stream(output, expected).await; let output = execute_sql( diff --git a/tests-integration/src/tests/instance_test.rs b/tests-integration/src/tests/instance_test.rs index 3cd625ba20..f5e94319ba 100644 --- a/tests-integration/src/tests/instance_test.rs +++ b/tests-integration/src/tests/instance_test.rs @@ -3271,8 +3271,10 @@ WITH( .await .data; let expected = "\ -++ -++"; ++---+----+ +| a | ts | ++---+----+ ++---+----+"; check_output_stream(output, expected).await; let output = execute_sql(&frontend, "drop table test_table").await.data; diff --git a/tests-integration/tests/sql.rs b/tests-integration/tests/sql.rs index c95adb483e..6c58ab0246 100644 --- a/tests-integration/tests/sql.rs +++ b/tests-integration/tests/sql.rs @@ -85,6 +85,7 @@ macro_rules! sql_tests { test_postgres_datestyle, test_postgres_intervalstyle, test_postgres_parameter_inference, + test_postgres_regclass_bind_parameter, test_postgres_uint64_parameter, test_postgres_explain_bind_parameter, test_postgres_array_types, @@ -705,16 +706,16 @@ pub async fn test_postgres_crud(store_type: StorageType) { let expected_j = serde_json::json!({ "code": i, - "success": true, "payload": { "features": [ "serde", "json" ], "homepage": null - } + }, + "success": true }); - assert_eq!(json.to_string(), expected_j.to_string()); + assert_eq!(json, expected_j); } let rows = sqlx::query("select i from demo where i=$1") @@ -1370,6 +1371,115 @@ pub async fn test_postgres_parameter_inference(store_type: StorageType) { guard.remove_all().await; } +pub async fn test_postgres_regclass_bind_parameter(store_type: StorageType) { + let (mut guard, fe_pg_server) = + setup_pg_server(store_type, "test_postgres_regclass_bind_parameter").await; + let addr = fe_pg_server.bind_addr().unwrap().to_string(); + + let (client, connection) = tokio_postgres::connect(&format!("postgres://{addr}/public"), NoTls) + .await + .unwrap(); + + let (tx, rx) = tokio::sync::oneshot::channel(); + tokio::spawn(async move { + connection.await.unwrap(); + tx.send(()).unwrap(); + }); + + client + .simple_query( + "CREATE TABLE test_adbc_app_logs (\"service\" STRING, \"message\" STRING, ts TIMESTAMP TIME INDEX)", + ) + .await + .unwrap(); + + let pgadbc_query = "SELECT attr.attname FROM pg_catalog.pg_class AS cls \ + INNER JOIN pg_catalog.pg_attribute AS attr ON cls.oid = attr.attrelid \ + INNER JOIN pg_catalog.pg_type AS typ ON attr.atttypid = typ.oid \ + WHERE attr.attnum >= 0 AND cls.oid = $1::regclass::oid ORDER BY attr.attnum"; + let pgadbc_statement = client.prepare(pgadbc_query).await.unwrap(); + + for table_name in [ + "test_adbc_app_logs", + "\"test_adbc_app_logs\"", + "public.test_adbc_app_logs", + "\"public\".\"test_adbc_app_logs\"", + ] { + let rows = client + .query(&pgadbc_statement, &[&table_name]) + .await + .unwrap(); + assert!(!rows.is_empty(), "{table_name} should resolve to a table"); + assert_eq!( + rows.iter() + .map(|row| row.get::<_, String>(0)) + .collect::>(), + ["service", "message", "ts"] + ); + } + + let relname_statement = client + .prepare("SELECT cls.oid FROM pg_catalog.pg_class AS cls WHERE cls.relname = $1") + .await + .unwrap(); + let rows = client + .query(&relname_statement, &[&"\"test_adbc_app_logs\""]) + .await + .unwrap(); + assert!(rows.is_empty(), "quoted text is not a relation name"); + + client + .simple_query("CREATE DATABASE adbc_override") + .await + .unwrap(); + client + .simple_query( + "CREATE TABLE adbc_override.test_adbc_app_logs (override_column STRING, ts TIMESTAMP TIME INDEX)", + ) + .await + .unwrap(); + client + .simple_query("SET search_path TO adbc_override, public") + .await + .unwrap(); + + let rows = client + .query(&pgadbc_statement, &[&"test_adbc_app_logs"]) + .await + .unwrap(); + assert_eq!(rows[0].get::<_, String>(0), "override_column"); + + let rows = client + .query(&pgadbc_statement, &[&"public.test_adbc_app_logs"]) + .await + .unwrap(); + assert_eq!( + rows.iter() + .map(|row| row.get::<_, String>(0)) + .collect::>(), + ["service", "message", "ts"] + ); + + client + .simple_query("SET search_path TO public") + .await + .unwrap(); + client + .simple_query("DROP TABLE adbc_override.test_adbc_app_logs") + .await + .unwrap(); + client + .simple_query("DROP DATABASE adbc_override") + .await + .unwrap(); + + drop(client); + rx.await.unwrap(); + + let _ = fe_pg_server.shutdown().await; + guard.remove_all().await; +} + pub async fn test_postgres_uint64_parameter(store_type: StorageType) { let (mut guard, fe_pg_server) = setup_pg_server(store_type, "test_postgres_uint64_parameter").await; diff --git a/tests/cases/distributed/explain/order_by.result b/tests/cases/distributed/explain/order_by.result index 6ce8b4e170..e578165952 100644 --- a/tests/cases/distributed/explain/order_by.result +++ b/tests/cases/distributed/explain/order_by.result @@ -125,9 +125,8 @@ EXPLAIN ANALYZE SELECT i, t AS alias_ts FROM test_pk ORDER BY t DESC LIMIT 5; |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_ProjectionExec: expr=[i@0 as i, t@1 as alias_ts] REDACTED -|_|_|_SortPreservingMergeExec: [test_pk.t__temp__0@2 DESC], fetch=5 REDACTED +|_|_|_SortPreservingMergeExec: [t@1 DESC], fetch=5 REDACTED |_|_|_SortExec: TopK(fetch=5), expr=[t@1 DESC], preserve_partitioning=[true], filter=[t@1 IS NULL OR t@1 > 2] REDACTED -|_|_|_ProjectionExec: expr=[i@0 as i, t@1 as t, t@1 as test_pk.t__temp__0] REDACTED |_|_|_SeqScan: region=REDACTED, "partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0} REDACTED |_|_|_| |_|_| Total rows: 5_| diff --git a/tests/cases/distributed/explain/step_aggr.result b/tests/cases/distributed/explain/step_aggr.result index 89634a3cc6..a69a82352b 100644 --- a/tests/cases/distributed/explain/step_aggr.result +++ b/tests/cases/distributed/explain/step_aggr.result @@ -58,15 +58,15 @@ FROM | plan_type_| plan_| +-+-+ | logical_plan_| Projection: count(integers.i), sum(integers.i), uddsketch_calc(Float64(0.5), uddsketch_state(Int64(128),Float64(0.01),integers.i)), hll_count(hll(integers.i))_| -|_|_Aggregate: groupBy=[[]], aggr=[[__count_merge(__count_state(integers.i)) AS count(integers.i), __sum_merge(__sum_state(integers.i)) AS sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) AS uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) AS hll(integers.i)]] | +|_|_Aggregate: groupBy=[[]], aggr=[[__count_merge(__count_state(integers.i)) AS count(integers.i), __sum_merge(__sum_state(integers.i)) AS sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) AS uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) AS hll(integers.i)]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[]], aggr=[[__count_state(integers.i), __sum_state(integers.i), __uddsketch_state_state(Int64(128), Float64(0.01), CAST(integers.i AS Float64)), __hll_state(CAST(integers.i AS Utf8))]]_| |_|_TableScan: integers_| |_| ]]_| | physical_plan | ProjectionExec: expr=[count(integers.i)@0 as count(integers.i), sum(integers.i)@1 as sum(integers.i), uddsketch_calc(0.5, uddsketch_state(Int64(128),Float64(0.01),integers.i)@2) as uddsketch_calc(Float64(0.5),uddsketch_state(Int64(128),Float64(0.01),integers.i)), hll_count(hll(integers.i)@3) as hll_count(hll(integers.i))]_| -|_|_AggregateExec: mode=Final, gby=[], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)]_| +|_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i), __sum_merge(__sum_state(integers.i)) as sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) as uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) as hll(integers.i)]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i), __sum_merge(__sum_state(integers.i)) as sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) as uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) as hll(integers.i)] | |_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -94,9 +94,9 @@ FROM | stage | node | plan_| +-+-+-+ | 0_| 0_|_ProjectionExec: expr=[count(integers.i)@0 as count(integers.i), sum(integers.i)@1 as sum(integers.i), uddsketch_calc(0.5, uddsketch_state(Int64(128),Float64(0.01),integers.i)@2) as uddsketch_calc(Float64(0.5),uddsketch_state(Int64(128),Float64(0.01),integers.i)), hll_count(hll(integers.i)@3) as hll_count(hll(integers.i))] REDACTED -|_|_|_AggregateExec: mode=Final, gby=[], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)] REDACTED +|_|_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i), __sum_merge(__sum_state(integers.i)) as sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) as uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) as hll(integers.i)] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i), __sum_merge(__sum_state(integers.i)) as sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) as uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) as hll(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -147,9 +147,9 @@ FROM |_| Aggregate: groupBy=[[]], aggr=[[__avg_state(CAST(integers.i AS Float64))]]_| |_|_TableScan: integers_| |_| ]]_| -| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[avg(integers.i)]_| +| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[__avg_merge(__avg_state(integers.i)) as avg(integers.i)]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[avg(integers.i)]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__avg_merge(__avg_state(integers.i)) as avg(integers.i)] | |_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -172,9 +172,9 @@ FROM +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[avg(integers.i)] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__avg_merge(__avg_state(integers.i)) as avg(integers.i)] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[avg(integers.i)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__avg_merge(__avg_state(integers.i)) as avg(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -242,7 +242,7 @@ ORDER BY +-+-+ | logical_plan_| Sort: integers.ts ASC NULLS LAST_| |_|_Projection: integers.ts, count(integers.i), sum(integers.i), uddsketch_calc(Float64(0.5), uddsketch_state(Int64(128),Float64(0.01),integers.i)), hll_count(hll(integers.i))_| -|_|_Aggregate: groupBy=[[integers.ts]], aggr=[[__count_merge(__count_state(integers.i)) AS count(integers.i), __sum_merge(__sum_state(integers.i)) AS sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) AS uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) AS hll(integers.i)]] | +|_|_Aggregate: groupBy=[[integers.ts]], aggr=[[__count_merge(__count_state(integers.i)) AS count(integers.i), __sum_merge(__sum_state(integers.i)) AS sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) AS uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) AS hll(integers.i)]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[integers.ts]], aggr=[[__count_state(integers.i), __sum_state(integers.i), __uddsketch_state_state(Int64(128), Float64(0.01), CAST(integers.i AS Float64)), __hll_state(CAST(integers.i AS Utf8))]]_| |_|_TableScan: integers_| @@ -250,9 +250,9 @@ ORDER BY | physical_plan | SortPreservingMergeExec: [ts@0 ASC NULLS LAST]_| |_|_SortExec: expr=[ts@0 ASC NULLS LAST], preserve_partitioning=[true]_| |_|_ProjectionExec: expr=[ts@0 as ts, count(integers.i)@1 as count(integers.i), sum(integers.i)@2 as sum(integers.i), uddsketch_calc(0.5, uddsketch_state(Int64(128),Float64(0.01),integers.i)@3) as uddsketch_calc(Float64(0.5),uddsketch_state(Int64(128),Float64(0.01),integers.i)), hll_count(hll(integers.i)@4) as hll_count(hll(integers.i))]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)]_| +|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i), __sum_merge(__sum_state(integers.i)) as sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) as uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) as hll(integers.i)] | |_|_RepartitionExec: partitioning=REDACTED -|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)]_| +|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i), __sum_merge(__sum_state(integers.i)) as sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) as uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) as hll(integers.i)]_| |_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -287,9 +287,9 @@ ORDER BY | 0_| 0_|_SortPreservingMergeExec: [ts@0 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[ts@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED |_|_|_ProjectionExec: expr=[ts@0 as ts, count(integers.i)@1 as count(integers.i), sum(integers.i)@2 as sum(integers.i), uddsketch_calc(0.5, uddsketch_state(Int64(128),Float64(0.01),integers.i)@3) as uddsketch_calc(Float64(0.5),uddsketch_state(Int64(128),Float64(0.01),integers.i)), hll_count(hll(integers.i)@4) as hll_count(hll(integers.i))] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i), __sum_merge(__sum_state(integers.i)) as sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) as uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) as hll(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i), __sum_merge(__sum_state(integers.i)) as sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) as uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) as hll(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -357,7 +357,7 @@ ORDER BY +-+-+ | logical_plan_| Sort: time_window ASC NULLS LAST_| |_|_Projection: date_bin(Utf8("2 seconds"),integers.ts) AS time_window, count(integers.i), sum(integers.i), uddsketch_calc(Float64(0.5), uddsketch_state(Int64(128),Float64(0.01),integers.i)), hll_count(hll(integers.i))_| -|_|_Aggregate: groupBy=[[date_bin(Utf8("2 seconds"),integers.ts)]], aggr=[[__count_merge(__count_state(integers.i)) AS count(integers.i), __sum_merge(__sum_state(integers.i)) AS sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) AS uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) AS hll(integers.i)]] | +|_|_Aggregate: groupBy=[[date_bin(Utf8("2 seconds"),integers.ts)]], aggr=[[__count_merge(__count_state(integers.i)) AS count(integers.i), __sum_merge(__sum_state(integers.i)) AS sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) AS uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) AS hll(integers.i)]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 2000000000 }"), integers.ts) AS date_bin(Utf8("2 seconds"),integers.ts)]], aggr=[[__count_state(integers.i), __sum_state(integers.i), __uddsketch_state_state(Int64(128), Float64(0.01), CAST(integers.i AS Float64)), __hll_state(CAST(integers.i AS Utf8))]]_| |_|_TableScan: integers_| @@ -365,9 +365,9 @@ ORDER BY | physical_plan | SortPreservingMergeExec: [time_window@0 ASC NULLS LAST]_| |_|_SortExec: expr=[time_window@0 ASC NULLS LAST], preserve_partitioning=[true]_| |_|_ProjectionExec: expr=[date_bin(Utf8("2 seconds"),integers.ts)@0 as time_window, count(integers.i)@1 as count(integers.i), sum(integers.i)@2 as sum(integers.i), uddsketch_calc(0.5, uddsketch_state(Int64(128),Float64(0.01),integers.i)@3) as uddsketch_calc(Float64(0.5),uddsketch_state(Int64(128),Float64(0.01),integers.i)), hll_count(hll(integers.i)@4) as hll_count(hll(integers.i))]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("2 seconds"),integers.ts)@0 as date_bin(Utf8("2 seconds"),integers.ts)], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)]_| +|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("2 seconds"),integers.ts)@0 as date_bin(Utf8("2 seconds"),integers.ts)], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i), __sum_merge(__sum_state(integers.i)) as sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) as uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) as hll(integers.i)] | |_|_RepartitionExec: partitioning=REDACTED -|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("2 seconds"),integers.ts)@0 as date_bin(Utf8("2 seconds"),integers.ts)], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)]_| +|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("2 seconds"),integers.ts)@0 as date_bin(Utf8("2 seconds"),integers.ts)], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i), __sum_merge(__sum_state(integers.i)) as sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) as uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) as hll(integers.i)]_| |_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -402,9 +402,9 @@ ORDER BY | 0_| 0_|_SortPreservingMergeExec: [time_window@0 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[time_window@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED |_|_|_ProjectionExec: expr=[date_bin(Utf8("2 seconds"),integers.ts)@0 as time_window, count(integers.i)@1 as count(integers.i), sum(integers.i)@2 as sum(integers.i), uddsketch_calc(0.5, uddsketch_state(Int64(128),Float64(0.01),integers.i)@3) as uddsketch_calc(Float64(0.5),uddsketch_state(Int64(128),Float64(0.01),integers.i)), hll_count(hll(integers.i)@4) as hll_count(hll(integers.i))] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("2 seconds"),integers.ts)@0 as date_bin(Utf8("2 seconds"),integers.ts)], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("2 seconds"),integers.ts)@0 as date_bin(Utf8("2 seconds"),integers.ts)], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i), __sum_merge(__sum_state(integers.i)) as sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) as uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) as hll(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("2 seconds"),integers.ts)@0 as date_bin(Utf8("2 seconds"),integers.ts)], aggr=[count(integers.i), sum(integers.i), uddsketch_state(Int64(128),Float64(0.01),integers.i), hll(integers.i)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("2 seconds"),integers.ts)@0 as date_bin(Utf8("2 seconds"),integers.ts)], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i), __sum_merge(__sum_state(integers.i)) as sum(integers.i), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),integers.i)) as uddsketch_state(Int64(128),Float64(0.01),integers.i), __hll_merge(__hll_state(integers.i)) as hll(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| diff --git a/tests/cases/distributed/explain/step_aggr_advance.result b/tests/cases/distributed/explain/step_aggr_advance.result index d3aaea79bc..348021360f 100644 --- a/tests/cases/distributed/explain/step_aggr_advance.result +++ b/tests/cases/distributed/explain/step_aggr_advance.result @@ -90,31 +90,31 @@ tql analyze (1752591864, 1752592164, '30s') max by (a, b, c) (max_over_time(aggr -- SQLNESS REPLACE (Hash.*) REDACTED tql explain (1752591864, 1752592164, '30s') sum by (a, b) (max_over_time(aggr_optimize_not [2m])); -+---------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ -| plan_type | plan | -+---------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ -| logical_plan | Sort: aggr_optimize_not.a ASC NULLS LAST, aggr_optimize_not.b ASC NULLS LAST, aggr_optimize_not.greptime_timestamp ASC NULLS LAST | -| | Aggregate: groupBy=[[aggr_optimize_not.a, aggr_optimize_not.b, aggr_optimize_not.greptime_timestamp]], aggr=[[__sum_merge(__sum_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) AS sum(prom_max_over_time(greptime_timestamp_range,greptime_value))]] | -| | MergeScan [is_placeholder=false, remote_input=[ | -| | Aggregate: groupBy=[[aggr_optimize_not.a, aggr_optimize_not.b, aggr_optimize_not.greptime_timestamp]], aggr=[[__sum_state(prom_max_over_time(greptime_timestamp_range,greptime_value))]] | -| | Filter: prom_max_over_time(greptime_timestamp_range,greptime_value) IS NOT NULL | -| | Projection: aggr_optimize_not.greptime_timestamp, prom_max_over_time(greptime_timestamp_range, greptime_value) AS prom_max_over_time(greptime_timestamp_range,greptime_value), aggr_optimize_not.a, aggr_optimize_not.b, aggr_optimize_not.c, aggr_optimize_not.d | -| | PromRangeManipulate: req range=[1752591864000..1752592164000], interval=[30000], eval range=[120000], time index=[greptime_timestamp], values=["greptime_value"] | -| | PromSeriesNormalize: offset=[0], time index=[greptime_timestamp], filter NaN: [true] | -| | PromSeriesDivide: tags=["a", "b", "c", "d"] | -| | Sort: aggr_optimize_not.a ASC NULLS FIRST, aggr_optimize_not.b ASC NULLS FIRST, aggr_optimize_not.c ASC NULLS FIRST, aggr_optimize_not.d ASC NULLS FIRST, aggr_optimize_not.greptime_timestamp ASC NULLS FIRST | -| | Filter: aggr_optimize_not.greptime_timestamp >= TimestampMillisecond(1752591744001, None) AND aggr_optimize_not.greptime_timestamp <= TimestampMillisecond(1752592164000, None) | -| | TableScan: aggr_optimize_not, partial_filters=[aggr_optimize_not.greptime_timestamp >= TimestampMillisecond(1752591744001, None), aggr_optimize_not.greptime_timestamp <= TimestampMillisecond(1752592164000, None)] | -| | ]] | -| physical_plan | SortPreservingMergeExec: [a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, greptime_timestamp@2 ASC NULLS LAST] | -| | SortExec: expr=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, greptime_timestamp@2 ASC NULLS LAST], preserve_partitioning=[true] | -| | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[sum(prom_max_over_time(greptime_timestamp_range,greptime_value))] | ++---------------+----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ +| plan_type | plan | ++---------------+----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ +| logical_plan | Sort: aggr_optimize_not.a ASC NULLS LAST, aggr_optimize_not.b ASC NULLS LAST, aggr_optimize_not.greptime_timestamp ASC NULLS LAST | +| | Aggregate: groupBy=[[aggr_optimize_not.a, aggr_optimize_not.b, aggr_optimize_not.greptime_timestamp]], aggr=[[__sum_merge(__sum_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) AS sum(prom_max_over_time(greptime_timestamp_range,greptime_value))]] | +| | MergeScan [is_placeholder=false, remote_input=[ | +| | Aggregate: groupBy=[[aggr_optimize_not.a, aggr_optimize_not.b, aggr_optimize_not.greptime_timestamp]], aggr=[[__sum_state(prom_max_over_time(greptime_timestamp_range,greptime_value))]] | +| | Filter: prom_max_over_time(greptime_timestamp_range,greptime_value) IS NOT NULL | +| | Projection: aggr_optimize_not.greptime_timestamp, prom_max_over_time(greptime_timestamp_range, greptime_value) AS prom_max_over_time(greptime_timestamp_range,greptime_value), aggr_optimize_not.a, aggr_optimize_not.b, aggr_optimize_not.c, aggr_optimize_not.d | +| | PromRangeManipulate: req range=[1752591864000..1752592164000], interval=[30000], eval range=[120000], time index=[greptime_timestamp], values=["greptime_value"] | +| | PromSeriesNormalize: offset=[0], time index=[greptime_timestamp], filter NaN: [true] | +| | PromSeriesDivide: tags=["a", "b", "c", "d"] | +| | Sort: aggr_optimize_not.a ASC NULLS FIRST, aggr_optimize_not.b ASC NULLS FIRST, aggr_optimize_not.c ASC NULLS FIRST, aggr_optimize_not.d ASC NULLS FIRST, aggr_optimize_not.greptime_timestamp ASC NULLS FIRST | +| | Filter: aggr_optimize_not.greptime_timestamp >= TimestampMillisecond(1752591744001, None) AND aggr_optimize_not.greptime_timestamp <= TimestampMillisecond(1752592164000, None) | +| | TableScan: aggr_optimize_not, partial_filters=[aggr_optimize_not.greptime_timestamp >= TimestampMillisecond(1752591744001, None), aggr_optimize_not.greptime_timestamp <= TimestampMillisecond(1752592164000, None)] | +| | ]] | +| physical_plan | SortPreservingMergeExec: [a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, greptime_timestamp@2 ASC NULLS LAST] | +| | SortExec: expr=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, greptime_timestamp@2 ASC NULLS LAST], preserve_partitioning=[true] | +| | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[__sum_merge(__sum_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) as sum(prom_max_over_time(greptime_timestamp_range,greptime_value))] | | | RepartitionExec: partitioning=REDACTED -| | AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[sum(prom_max_over_time(greptime_timestamp_range,greptime_value))] | +| | AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[__sum_merge(__sum_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) as sum(prom_max_over_time(greptime_timestamp_range,greptime_value))] | | | RepartitionExec: partitioning=REDACTED | | MergeScanExec: REDACTED -| | | -+---------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ +| | | ++---------------+----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ -- SQLNESS REPLACE (metrics.*) REDACTED -- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED @@ -130,9 +130,9 @@ tql analyze (1752591864, 1752592164, '30s') sum by (a, b) (max_over_time(aggr_op +-+-+-+ | 0_| 0_|_SortPreservingMergeExec: [a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, greptime_timestamp@2 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, greptime_timestamp@2 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[sum(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[__sum_merge(__sum_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) as sum(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[sum(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b, greptime_timestamp@2 as greptime_timestamp], aggr=[__sum_merge(__sum_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) as sum(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -165,31 +165,31 @@ tql analyze (1752591864, 1752592164, '30s') sum by (a, b) (max_over_time(aggr_op -- SQLNESS REPLACE (Hash.*) REDACTED tql explain (1752591864, 1752592164, '30s') avg by (a) (max_over_time(aggr_optimize_not [2m])); -+---------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ -| plan_type | plan | -+---------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ -| logical_plan | Sort: aggr_optimize_not.a ASC NULLS LAST, aggr_optimize_not.greptime_timestamp ASC NULLS LAST | -| | Aggregate: groupBy=[[aggr_optimize_not.a, aggr_optimize_not.greptime_timestamp]], aggr=[[__avg_merge(__avg_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) AS avg(prom_max_over_time(greptime_timestamp_range,greptime_value))]] | -| | MergeScan [is_placeholder=false, remote_input=[ | -| | Aggregate: groupBy=[[aggr_optimize_not.a, aggr_optimize_not.greptime_timestamp]], aggr=[[__avg_state(prom_max_over_time(greptime_timestamp_range,greptime_value))]] | -| | Filter: prom_max_over_time(greptime_timestamp_range,greptime_value) IS NOT NULL | -| | Projection: aggr_optimize_not.greptime_timestamp, prom_max_over_time(greptime_timestamp_range, greptime_value) AS prom_max_over_time(greptime_timestamp_range,greptime_value), aggr_optimize_not.a, aggr_optimize_not.b, aggr_optimize_not.c, aggr_optimize_not.d | -| | PromRangeManipulate: req range=[1752591864000..1752592164000], interval=[30000], eval range=[120000], time index=[greptime_timestamp], values=["greptime_value"] | -| | PromSeriesNormalize: offset=[0], time index=[greptime_timestamp], filter NaN: [true] | -| | PromSeriesDivide: tags=["a", "b", "c", "d"] | -| | Sort: aggr_optimize_not.a ASC NULLS FIRST, aggr_optimize_not.b ASC NULLS FIRST, aggr_optimize_not.c ASC NULLS FIRST, aggr_optimize_not.d ASC NULLS FIRST, aggr_optimize_not.greptime_timestamp ASC NULLS FIRST | -| | Filter: aggr_optimize_not.greptime_timestamp >= TimestampMillisecond(1752591744001, None) AND aggr_optimize_not.greptime_timestamp <= TimestampMillisecond(1752592164000, None) | -| | TableScan: aggr_optimize_not, partial_filters=[aggr_optimize_not.greptime_timestamp >= TimestampMillisecond(1752591744001, None), aggr_optimize_not.greptime_timestamp <= TimestampMillisecond(1752592164000, None)] | -| | ]] | -| physical_plan | SortPreservingMergeExec: [a@0 ASC NULLS LAST, greptime_timestamp@1 ASC NULLS LAST] | -| | SortExec: expr=[a@0 ASC NULLS LAST, greptime_timestamp@1 ASC NULLS LAST], preserve_partitioning=[true] | -| | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[avg(prom_max_over_time(greptime_timestamp_range,greptime_value))] | ++---------------+------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ +| plan_type | plan | ++---------------+------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ +| logical_plan | Sort: aggr_optimize_not.a ASC NULLS LAST, aggr_optimize_not.greptime_timestamp ASC NULLS LAST | +| | Aggregate: groupBy=[[aggr_optimize_not.a, aggr_optimize_not.greptime_timestamp]], aggr=[[__avg_merge(__avg_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) AS avg(prom_max_over_time(greptime_timestamp_range,greptime_value))]] | +| | MergeScan [is_placeholder=false, remote_input=[ | +| | Aggregate: groupBy=[[aggr_optimize_not.a, aggr_optimize_not.greptime_timestamp]], aggr=[[__avg_state(prom_max_over_time(greptime_timestamp_range,greptime_value))]] | +| | Filter: prom_max_over_time(greptime_timestamp_range,greptime_value) IS NOT NULL | +| | Projection: aggr_optimize_not.greptime_timestamp, prom_max_over_time(greptime_timestamp_range, greptime_value) AS prom_max_over_time(greptime_timestamp_range,greptime_value), aggr_optimize_not.a, aggr_optimize_not.b, aggr_optimize_not.c, aggr_optimize_not.d | +| | PromRangeManipulate: req range=[1752591864000..1752592164000], interval=[30000], eval range=[120000], time index=[greptime_timestamp], values=["greptime_value"] | +| | PromSeriesNormalize: offset=[0], time index=[greptime_timestamp], filter NaN: [true] | +| | PromSeriesDivide: tags=["a", "b", "c", "d"] | +| | Sort: aggr_optimize_not.a ASC NULLS FIRST, aggr_optimize_not.b ASC NULLS FIRST, aggr_optimize_not.c ASC NULLS FIRST, aggr_optimize_not.d ASC NULLS FIRST, aggr_optimize_not.greptime_timestamp ASC NULLS FIRST | +| | Filter: aggr_optimize_not.greptime_timestamp >= TimestampMillisecond(1752591744001, None) AND aggr_optimize_not.greptime_timestamp <= TimestampMillisecond(1752592164000, None) | +| | TableScan: aggr_optimize_not, partial_filters=[aggr_optimize_not.greptime_timestamp >= TimestampMillisecond(1752591744001, None), aggr_optimize_not.greptime_timestamp <= TimestampMillisecond(1752592164000, None)] | +| | ]] | +| physical_plan | SortPreservingMergeExec: [a@0 ASC NULLS LAST, greptime_timestamp@1 ASC NULLS LAST] | +| | SortExec: expr=[a@0 ASC NULLS LAST, greptime_timestamp@1 ASC NULLS LAST], preserve_partitioning=[true] | +| | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[__avg_merge(__avg_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) as avg(prom_max_over_time(greptime_timestamp_range,greptime_value))] | | | RepartitionExec: partitioning=REDACTED -| | AggregateExec: mode=Partial, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[avg(prom_max_over_time(greptime_timestamp_range,greptime_value))] | +| | AggregateExec: mode=Partial, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[__avg_merge(__avg_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) as avg(prom_max_over_time(greptime_timestamp_range,greptime_value))] | | | RepartitionExec: partitioning=REDACTED | | MergeScanExec: REDACTED -| | | -+---------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ +| | | ++---------------+------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ -- SQLNESS REPLACE (metrics.*) REDACTED -- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED @@ -205,9 +205,9 @@ tql analyze (1752591864, 1752592164, '30s') avg by (a) (max_over_time(aggr_optim +-+-+-+ | 0_| 0_|_SortPreservingMergeExec: [a@0 ASC NULLS LAST, greptime_timestamp@1 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[a@0 ASC NULLS LAST, greptime_timestamp@1 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[avg(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[__avg_merge(__avg_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) as avg(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[avg(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a, greptime_timestamp@1 as greptime_timestamp], aggr=[__avg_merge(__avg_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) as avg(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -323,9 +323,9 @@ tql explain (1752591864, 1752592164, '30s') min by (b, c, d) (max_over_time(aggr | | ]] | | physical_plan | SortPreservingMergeExec: [b@0 ASC NULLS LAST, c@1 ASC NULLS LAST, d@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST] | | | SortExec: expr=[b@0 ASC NULLS LAST, c@1 ASC NULLS LAST, d@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST], preserve_partitioning=[true] | -| | AggregateExec: mode=FinalPartitioned, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[min(prom_max_over_time(greptime_timestamp_range,greptime_value))] | +| | AggregateExec: mode=FinalPartitioned, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[__min_merge(__min_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) as min(prom_max_over_time(greptime_timestamp_range,greptime_value))] | | | RepartitionExec: partitioning=REDACTED -| | AggregateExec: mode=Partial, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[min(prom_max_over_time(greptime_timestamp_range,greptime_value))] | +| | AggregateExec: mode=Partial, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[__min_merge(__min_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) as min(prom_max_over_time(greptime_timestamp_range,greptime_value))] | | | RepartitionExec: partitioning=REDACTED | | MergeScanExec: REDACTED | | | @@ -345,9 +345,9 @@ tql analyze (1752591864, 1752592164, '30s') min by (b, c, d) (max_over_time(aggr +-+-+-+ | 0_| 0_|_SortPreservingMergeExec: [b@0 ASC NULLS LAST, c@1 ASC NULLS LAST, d@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[b@0 ASC NULLS LAST, c@1 ASC NULLS LAST, d@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[min(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[__min_merge(__min_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) as min(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[min(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[b@0 as b, c@1 as c, d@2 as d, greptime_timestamp@3 as greptime_timestamp], aggr=[__min_merge(__min_state(prom_max_over_time(greptime_timestamp_range,greptime_value))) as min(prom_max_over_time(greptime_timestamp_range,greptime_value))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -395,9 +395,9 @@ tql explain sum(aggr_optimize_not); | | ]] | | physical_plan | SortPreservingMergeExec: [greptime_timestamp@0 ASC NULLS LAST] | | | SortExec: expr=[greptime_timestamp@0 ASC NULLS LAST], preserve_partitioning=[true] | -| | AggregateExec: mode=FinalPartitioned, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[sum(aggr_optimize_not.greptime_value)] | +| | AggregateExec: mode=FinalPartitioned, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[__sum_merge(__sum_state(aggr_optimize_not.greptime_value)) as sum(aggr_optimize_not.greptime_value)] | | | RepartitionExec: partitioning=REDACTED -| | AggregateExec: mode=Partial, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[sum(aggr_optimize_not.greptime_value)] | +| | AggregateExec: mode=Partial, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[__sum_merge(__sum_state(aggr_optimize_not.greptime_value)) as sum(aggr_optimize_not.greptime_value)] | | | RepartitionExec: partitioning=REDACTED | | MergeScanExec: REDACTED | | | @@ -417,9 +417,9 @@ tql analyze sum(aggr_optimize_not); +-+-+-+ | 0_| 0_|_SortPreservingMergeExec: [greptime_timestamp@0 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[greptime_timestamp@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[sum(aggr_optimize_not.greptime_value)] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[__sum_merge(__sum_state(aggr_optimize_not.greptime_value)) as sum(aggr_optimize_not.greptime_value)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[sum(aggr_optimize_not.greptime_value)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[__sum_merge(__sum_state(aggr_optimize_not.greptime_value)) as sum(aggr_optimize_not.greptime_value)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -494,7 +494,7 @@ tql explain (1752591864, 1752592164, '30s') sum by (a, b, c) (rate(aggr_optimize | | Filter: aggr_optimize_not_count.greptime_timestamp >= TimestampMillisecond(1752591744001, None) AND aggr_optimize_not_count.greptime_timestamp <= TimestampMillisecond(1752592164000, None) | | | TableScan: aggr_optimize_not_count, partial_filters=[aggr_optimize_not_count.greptime_timestamp >= TimestampMillisecond(1752591744001, None), aggr_optimize_not_count.greptime_timestamp <= TimestampMillisecond(1752592164000, None)] | | | ]] | -| physical_plan | ProjectionExec: expr=[a@0 as a, b@1 as b, c@2 as c, greptime_timestamp@3 as greptime_timestamp, sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))@5 / sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))@4 as aggr_optimize_not.sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000))) / aggr_optimize_not_count.sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] | +| physical_plan | ProjectionExec: expr=[a@0 as a, b@1 as b, c@2 as c, greptime_timestamp@3 as greptime_timestamp, sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))@4 / sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))@5 as aggr_optimize_not.sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000))) / aggr_optimize_not_count.sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] | | | REDACTED | | CoalescePartitionsExec | | | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, c@2 as c, greptime_timestamp@3 as greptime_timestamp], aggr=[sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] | @@ -527,7 +527,7 @@ tql analyze (1752591864, 1752592164, '30s') sum by (a, b, c) (rate(aggr_optimize +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_ProjectionExec: expr=[a@0 as a, b@1 as b, c@2 as c, greptime_timestamp@3 as greptime_timestamp, sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))@5 / sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))@4 as aggr_optimize_not.sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000))) / aggr_optimize_not_count.sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] REDACTED +| 0_| 0_|_ProjectionExec: expr=[a@0 as a, b@1 as b, c@2 as c, greptime_timestamp@3 as greptime_timestamp, sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))@4 / sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))@5 as aggr_optimize_not.sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000))) / aggr_optimize_not_count.sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] REDACTED |_|_|_REDACTED |_|_|_CoalescePartitionsExec REDACTED |_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, c@2 as c, greptime_timestamp@3 as greptime_timestamp], aggr=[sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] REDACTED @@ -552,32 +552,6 @@ tql analyze (1752591864, 1752592164, '30s') sum by (a, b, c) (rate(aggr_optimize | 1_| 1_|_CooperativeExec REDACTED |_|_|_SeqScan: region=REDACTED, "partition_count":{"count":0, "mem_ranges":0, "files":0, "file_ranges":0} REDACTED |_|_|_| -| 1_| 0_|_ProjectionExec: expr=[a@0 as a, b@1 as b, c@2 as c, greptime_timestamp@3 as greptime_timestamp, sum(prom_rate(greptime_timestamp_range,greptime_value,aggr_optimize_not.greptime_timestamp,Int64(120000)))@4 as sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] REDACTED -|_|_|_SortPreservingMergeExec: [a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, c@2 as c, greptime_timestamp@3 as greptime_timestamp], aggr=[sum(prom_rate(greptime_timestamp_range,greptime_value,aggr_optimize_not.greptime_timestamp,Int64(120000)))] REDACTED -|_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[a@2 as a, b@3 as b, c@4 as c, greptime_timestamp@0 as greptime_timestamp], aggr=[sum(prom_rate(greptime_timestamp_range,greptime_value,aggr_optimize_not.greptime_timestamp,Int64(120000)))] REDACTED -|_|_|_FilterExec: prom_rate(greptime_timestamp_range,greptime_value,aggr_optimize_not.greptime_timestamp,Int64(120000))@1 IS NOT NULL REDACTED -|_|_|_ProjectionExec: expr=[greptime_timestamp@4 as greptime_timestamp, prom_rate(greptime_timestamp_range@6, greptime_value@5, greptime_timestamp@4, 120000) as prom_rate(greptime_timestamp_range,greptime_value,aggr_optimize_not.greptime_timestamp,Int64(120000)), a@0 as a, b@1 as b, c@2 as c] REDACTED -|_|_|_PromRangeManipulateExec: req range=[1752591864000..1752592164000], interval=[30000], eval range=[120000], time index=[greptime_timestamp] REDACTED -|_|_|_PromSeriesNormalizeExec: offset=[0], time index=[greptime_timestamp], filter NaN: [true] REDACTED -|_|_|_PromSeriesDivideExec: tags=["a", "b", "c", "d"] REDACTED -|_|_|_SeriesScan: region=REDACTED, "partition_count":{"count":0, "mem_ranges":0, "files":0, "file_ranges":0}, "distribution":"PerSeries", "mode":"legacy" REDACTED -|_|_|_| -| 1_| 1_|_ProjectionExec: expr=[a@0 as a, b@1 as b, c@2 as c, greptime_timestamp@3 as greptime_timestamp, sum(prom_rate(greptime_timestamp_range,greptime_value,aggr_optimize_not.greptime_timestamp,Int64(120000)))@4 as sum(prom_rate(greptime_timestamp_range,greptime_value,greptime_timestamp,Int64(120000)))] REDACTED -|_|_|_SortPreservingMergeExec: [a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST, greptime_timestamp@3 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, c@2 as c, greptime_timestamp@3 as greptime_timestamp], aggr=[sum(prom_rate(greptime_timestamp_range,greptime_value,aggr_optimize_not.greptime_timestamp,Int64(120000)))] REDACTED -|_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[a@2 as a, b@3 as b, c@4 as c, greptime_timestamp@0 as greptime_timestamp], aggr=[sum(prom_rate(greptime_timestamp_range,greptime_value,aggr_optimize_not.greptime_timestamp,Int64(120000)))] REDACTED -|_|_|_FilterExec: prom_rate(greptime_timestamp_range,greptime_value,aggr_optimize_not.greptime_timestamp,Int64(120000))@1 IS NOT NULL REDACTED -|_|_|_ProjectionExec: expr=[greptime_timestamp@4 as greptime_timestamp, prom_rate(greptime_timestamp_range@6, greptime_value@5, greptime_timestamp@4, 120000) as prom_rate(greptime_timestamp_range,greptime_value,aggr_optimize_not.greptime_timestamp,Int64(120000)), a@0 as a, b@1 as b, c@2 as c] REDACTED -|_|_|_PromRangeManipulateExec: req range=[1752591864000..1752592164000], interval=[30000], eval range=[120000], time index=[greptime_timestamp] REDACTED -|_|_|_PromSeriesNormalizeExec: offset=[0], time index=[greptime_timestamp], filter NaN: [true] REDACTED -|_|_|_PromSeriesDivideExec: tags=["a", "b", "c", "d"] REDACTED -|_|_|_SeriesScan: region=REDACTED, "partition_count":{"count":0, "mem_ranges":0, "files":0, "file_ranges":0}, "distribution":"PerSeries", "mode":"legacy" REDACTED -|_|_|_| |_|_| Total rows: 0_| +-+-+-+ @@ -736,9 +710,9 @@ GROUP BY | | TableScan: aggr_optimize_not | | | ]] | | physical_plan | ProjectionExec: expr=[min(aggr_optimize_not.greptime_value)@2 as min(aggr_optimize_not.greptime_value)] | -| | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value)] | +| | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[__min_merge(__min_state(aggr_optimize_not.greptime_value)) as min(aggr_optimize_not.greptime_value)] | | | RepartitionExec: partitioning=REDACTED -| | AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value)] | +| | AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[__min_merge(__min_state(aggr_optimize_not.greptime_value)) as min(aggr_optimize_not.greptime_value)] | | | RepartitionExec: partitioning=REDACTED | | MergeScanExec: REDACTED | | | @@ -764,9 +738,9 @@ GROUP BY | stage | node | plan_| +-+-+-+ | 0_| 0_|_ProjectionExec: expr=[min(aggr_optimize_not.greptime_value)@2 as min(aggr_optimize_not.greptime_value)] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value)] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[__min_merge(__min_state(aggr_optimize_not.greptime_value)) as min(aggr_optimize_not.greptime_value)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[__min_merge(__min_state(aggr_optimize_not.greptime_value)) as min(aggr_optimize_not.greptime_value)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -806,9 +780,9 @@ GROUP BY | | TableScan: aggr_optimize_not | | | ]] | | physical_plan | ProjectionExec: expr=[min(aggr_optimize_not.greptime_value)@2 + max(aggr_optimize_not.greptime_value)@3 as min(aggr_optimize_not.greptime_value) + max(aggr_optimize_not.greptime_value)] | -| | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value), max(aggr_optimize_not.greptime_value)] | +| | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[__min_merge(__min_state(aggr_optimize_not.greptime_value)) as min(aggr_optimize_not.greptime_value), __max_merge(__max_state(aggr_optimize_not.greptime_value)) as max(aggr_optimize_not.greptime_value)] | | | RepartitionExec: partitioning=REDACTED -| | AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value), max(aggr_optimize_not.greptime_value)] | +| | AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[__min_merge(__min_state(aggr_optimize_not.greptime_value)) as min(aggr_optimize_not.greptime_value), __max_merge(__max_state(aggr_optimize_not.greptime_value)) as max(aggr_optimize_not.greptime_value)] | | | RepartitionExec: partitioning=REDACTED | | MergeScanExec: REDACTED | | | @@ -834,9 +808,9 @@ GROUP BY | stage | node | plan_| +-+-+-+ | 0_| 0_|_ProjectionExec: expr=[min(aggr_optimize_not.greptime_value)@2 + max(aggr_optimize_not.greptime_value)@3 as min(aggr_optimize_not.greptime_value) + max(aggr_optimize_not.greptime_value)] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value), max(aggr_optimize_not.greptime_value)] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[__min_merge(__min_state(aggr_optimize_not.greptime_value)) as min(aggr_optimize_not.greptime_value), __max_merge(__max_state(aggr_optimize_not.greptime_value)) as max(aggr_optimize_not.greptime_value)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[min(aggr_optimize_not.greptime_value), max(aggr_optimize_not.greptime_value)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[__min_merge(__min_state(aggr_optimize_not.greptime_value)) as min(aggr_optimize_not.greptime_value), __max_merge(__max_state(aggr_optimize_not.greptime_value)) as max(aggr_optimize_not.greptime_value)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -876,22 +850,22 @@ FROM GROUP BY a; -+---------------+----------------------------------------------------------------------------------------------------------------------------------------------------------+ -| plan_type | plan | -+---------------+----------------------------------------------------------------------------------------------------------------------------------------------------------+ -| logical_plan | Aggregate: groupBy=[[aggr_optimize_not.a]], aggr=[[__min_merge(__min_state(aggr_optimize_not.greptime_value)) AS min(aggr_optimize_not.greptime_value)]] | -| | MergeScan [is_placeholder=false, remote_input=[ | -| | Aggregate: groupBy=[[aggr_optimize_not.a]], aggr=[[__min_state(aggr_optimize_not.greptime_value)]] | -| | Projection: aggr_optimize_not.a, aggr_optimize_not.b, aggr_optimize_not.greptime_value | -| | TableScan: aggr_optimize_not | -| | ]] | -| physical_plan | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[min(aggr_optimize_not.greptime_value)] | ++---------------+------------------------------------------------------------------------------------------------------------------------------------------------------------------+ +| plan_type | plan | ++---------------+------------------------------------------------------------------------------------------------------------------------------------------------------------------+ +| logical_plan | Aggregate: groupBy=[[aggr_optimize_not.a]], aggr=[[__min_merge(__min_state(aggr_optimize_not.greptime_value)) AS min(aggr_optimize_not.greptime_value)]] | +| | MergeScan [is_placeholder=false, remote_input=[ | +| | Aggregate: groupBy=[[aggr_optimize_not.a]], aggr=[[__min_state(aggr_optimize_not.greptime_value)]] | +| | Projection: aggr_optimize_not.a, aggr_optimize_not.b, aggr_optimize_not.greptime_value | +| | TableScan: aggr_optimize_not | +| | ]] | +| physical_plan | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[__min_merge(__min_state(aggr_optimize_not.greptime_value)) as min(aggr_optimize_not.greptime_value)] | | | RepartitionExec: partitioning=REDACTED -| | AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[min(aggr_optimize_not.greptime_value)] | +| | AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[__min_merge(__min_state(aggr_optimize_not.greptime_value)) as min(aggr_optimize_not.greptime_value)] | | | RepartitionExec: partitioning=REDACTED | | MergeScanExec: REDACTED -| | | -+---------------+----------------------------------------------------------------------------------------------------------------------------------------------------------+ +| | | ++---------------+------------------------------------------------------------------------------------------------------------------------------------------------------------------+ -- SQLNESS REPLACE (metrics.*) REDACTED -- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED @@ -922,9 +896,9 @@ GROUP BY +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[min(aggr_optimize_not.greptime_value)] REDACTED +| 0_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[__min_merge(__min_state(aggr_optimize_not.greptime_value)) as min(aggr_optimize_not.greptime_value)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[min(aggr_optimize_not.greptime_value)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[__min_merge(__min_state(aggr_optimize_not.greptime_value)) as min(aggr_optimize_not.greptime_value)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -1088,7 +1062,7 @@ EXPLAIN SELECT pk_col_2, sum(val_col_1) FROM step_aggr_extended GROUP BY pk_col_ +-+-+ | logical_plan_| Sort: step_aggr_extended.pk_col_2 ASC NULLS LAST_| |_|_Filter: sum(step_aggr_extended.val_col_1) > Int64(300)_| -|_|_Aggregate: groupBy=[[step_aggr_extended.pk_col_2]], aggr=[[__sum_merge(__sum_state(step_aggr_extended.val_col_1)) AS sum(step_aggr_extended.val_col_1)]] | +|_|_Aggregate: groupBy=[[step_aggr_extended.pk_col_2]], aggr=[[__sum_merge(__sum_state(step_aggr_extended.val_col_1)) AS sum(step_aggr_extended.val_col_1)]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[step_aggr_extended.pk_col_2]], aggr=[[__sum_state(step_aggr_extended.val_col_1)]]_| |_|_TableScan: step_aggr_extended_| @@ -1096,9 +1070,9 @@ EXPLAIN SELECT pk_col_2, sum(val_col_1) FROM step_aggr_extended GROUP BY pk_col_ | physical_plan | SortPreservingMergeExec: [pk_col_2@0 ASC NULLS LAST]_| |_|_SortExec: expr=[pk_col_2@0 ASC NULLS LAST], preserve_partitioning=[true]_| |_|_FilterExec: sum(step_aggr_extended.val_col_1)@1 > 300_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[pk_col_2@0 as pk_col_2], aggr=[sum(step_aggr_extended.val_col_1)]_| +|_|_AggregateExec: mode=FinalPartitioned, gby=[pk_col_2@0 as pk_col_2], aggr=[__sum_merge(__sum_state(step_aggr_extended.val_col_1)) as sum(step_aggr_extended.val_col_1)] | |_|_RepartitionExec: partitioning=REDACTED -|_|_AggregateExec: mode=Partial, gby=[pk_col_2@0 as pk_col_2], aggr=[sum(step_aggr_extended.val_col_1)]_| +|_|_AggregateExec: mode=Partial, gby=[pk_col_2@0 as pk_col_2], aggr=[__sum_merge(__sum_state(step_aggr_extended.val_col_1)) as sum(step_aggr_extended.val_col_1)]_| |_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -1127,15 +1101,15 @@ EXPLAIN SELECT SUM(val_col_3), COUNT(val_col_2), COUNT(val_col_3), COUNT(*) FROM | plan_type_| plan_| +-+-+ | logical_plan_| Projection: sum(step_aggr_extended.val_col_3), count(step_aggr_extended.val_col_2), count(step_aggr_extended.val_col_3), count(Int64(1)) AS count(*)_| -|_|_Aggregate: groupBy=[[]], aggr=[[__sum_merge(__sum_state(step_aggr_extended.val_col_3)) AS sum(step_aggr_extended.val_col_3), __count_merge(__count_state(step_aggr_extended.val_col_2)) AS count(step_aggr_extended.val_col_2), __count_merge(__count_state(step_aggr_extended.val_col_3)) AS count(step_aggr_extended.val_col_3), __count_merge(__count_state(step_aggr_extended.ts)) AS count(Int64(1))]] | +|_|_Aggregate: groupBy=[[]], aggr=[[__sum_merge(__sum_state(step_aggr_extended.val_col_3)) AS sum(step_aggr_extended.val_col_3), __count_merge(__count_state(step_aggr_extended.val_col_2)) AS count(step_aggr_extended.val_col_2), __count_merge(__count_state(step_aggr_extended.val_col_3)) AS count(step_aggr_extended.val_col_3), __count_merge(__count_state(step_aggr_extended.ts)) AS count(Int64(1))]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[]], aggr=[[__sum_state(step_aggr_extended.val_col_3), __count_state(step_aggr_extended.val_col_2), __count_state(step_aggr_extended.val_col_3), __count_state(step_aggr_extended.ts)]]_| |_|_TableScan: step_aggr_extended_| |_| ]]_| | physical_plan | ProjectionExec: expr=[sum(step_aggr_extended.val_col_3)@0 as sum(step_aggr_extended.val_col_3), count(step_aggr_extended.val_col_2)@1 as count(step_aggr_extended.val_col_2), count(step_aggr_extended.val_col_3)@2 as count(step_aggr_extended.val_col_3), count(Int64(1))@3 as count(*)]_| -|_|_AggregateExec: mode=Final, gby=[], aggr=[sum(step_aggr_extended.val_col_3), count(step_aggr_extended.val_col_2), count(step_aggr_extended.val_col_3), count(Int64(1))]_| +|_|_AggregateExec: mode=Final, gby=[], aggr=[__sum_merge(__sum_state(step_aggr_extended.val_col_3)) as sum(step_aggr_extended.val_col_3), __count_merge(__count_state(step_aggr_extended.val_col_2)) as count(step_aggr_extended.val_col_2), __count_merge(__count_state(step_aggr_extended.val_col_3)) as count(step_aggr_extended.val_col_3), __count_merge(__count_state(step_aggr_extended.ts)) as count(Int64(1))]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[sum(step_aggr_extended.val_col_3), count(step_aggr_extended.val_col_2), count(step_aggr_extended.val_col_3), count(Int64(1))]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__sum_merge(__sum_state(step_aggr_extended.val_col_3)) as sum(step_aggr_extended.val_col_3), __count_merge(__count_state(step_aggr_extended.val_col_2)) as count(step_aggr_extended.val_col_2), __count_merge(__count_state(step_aggr_extended.val_col_3)) as count(step_aggr_extended.val_col_3), __count_merge(__count_state(step_aggr_extended.ts)) as count(Int64(1))] | |_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -1163,14 +1137,14 @@ EXPLAIN SELECT MIN(pk_col_1), MAX(val_col_2) FROM step_aggr_extended; +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__min_merge(__min_state(step_aggr_extended.pk_col_1)) AS min(step_aggr_extended.pk_col_1), __max_merge(__max_state(step_aggr_extended.val_col_2)) AS max(step_aggr_extended.val_col_2)]] | +| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__min_merge(__min_state(step_aggr_extended.pk_col_1)) AS min(step_aggr_extended.pk_col_1), __max_merge(__max_state(step_aggr_extended.val_col_2)) AS max(step_aggr_extended.val_col_2)]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[]], aggr=[[__min_state(CAST(step_aggr_extended.pk_col_1 AS Utf8)), __max_state(step_aggr_extended.val_col_2)]]_| |_|_TableScan: step_aggr_extended_| |_| ]]_| -| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[min(step_aggr_extended.pk_col_1), max(step_aggr_extended.val_col_2)]_| +| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[__min_merge(__min_state(step_aggr_extended.pk_col_1)) as min(step_aggr_extended.pk_col_1), __max_merge(__max_state(step_aggr_extended.val_col_2)) as max(step_aggr_extended.val_col_2)]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[min(step_aggr_extended.pk_col_1), max(step_aggr_extended.val_col_2)]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__min_merge(__min_state(step_aggr_extended.pk_col_1)) as min(step_aggr_extended.pk_col_1), __max_merge(__max_state(step_aggr_extended.val_col_2)) as max(step_aggr_extended.val_col_2)] | |_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -1199,16 +1173,16 @@ EXPLAIN SELECT SUM(val_col_1), COUNT(*) FROM step_aggr_extended WHERE pk_col_1 = | plan_type_| plan_| +-+-+ | logical_plan_| Projection: sum(step_aggr_extended.val_col_1), count(Int64(1)) AS count(*)_| -|_|_Aggregate: groupBy=[[]], aggr=[[__sum_merge(__sum_state(step_aggr_extended.val_col_1)) AS sum(step_aggr_extended.val_col_1), __count_merge(__count_state(step_aggr_extended.ts)) AS count(Int64(1))]] | +|_|_Aggregate: groupBy=[[]], aggr=[[__sum_merge(__sum_state(step_aggr_extended.val_col_1)) AS sum(step_aggr_extended.val_col_1), __count_merge(__count_state(step_aggr_extended.ts)) AS count(Int64(1))]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[]], aggr=[[__sum_state(step_aggr_extended.val_col_1), __count_state(step_aggr_extended.ts)]]_| |_|_Filter: step_aggr_extended.pk_col_1 = CAST(Utf8("non_existent") AS Dictionary(UInt32, Utf8))_| |_|_TableScan: step_aggr_extended, partial_filters=[step_aggr_extended.pk_col_1 = CAST(Utf8("non_existent") AS Dictionary(UInt32, Utf8))]_| |_| ]]_| | physical_plan | ProjectionExec: expr=[sum(step_aggr_extended.val_col_1)@0 as sum(step_aggr_extended.val_col_1), count(Int64(1))@1 as count(*)]_| -|_|_AggregateExec: mode=Final, gby=[], aggr=[sum(step_aggr_extended.val_col_1), count(Int64(1))]_| +|_|_AggregateExec: mode=Final, gby=[], aggr=[__sum_merge(__sum_state(step_aggr_extended.val_col_1)) as sum(step_aggr_extended.val_col_1), __count_merge(__count_state(step_aggr_extended.ts)) as count(Int64(1))]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[sum(step_aggr_extended.val_col_1), count(Int64(1))]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__sum_merge(__sum_state(step_aggr_extended.val_col_1)) as sum(step_aggr_extended.val_col_1), __count_merge(__count_state(step_aggr_extended.ts)) as count(Int64(1))] | |_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| diff --git a/tests/cases/distributed/explain/step_aggr_basic.result b/tests/cases/distributed/explain/step_aggr_basic.result index cf11b7f9ef..3988045087 100644 --- a/tests/cases/distributed/explain/step_aggr_basic.result +++ b/tests/cases/distributed/explain/step_aggr_basic.result @@ -57,9 +57,9 @@ FROM |_| Aggregate: groupBy=[[]], aggr=[[__count_state(integers.i)]]_| |_|_TableScan: integers_| |_| ]]_| -| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[count(integers.i)]_| +| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(integers.i)]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)] | |_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -83,9 +83,9 @@ FROM +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(integers.i)] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(integers.i)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -147,17 +147,17 @@ ORDER BY +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Sort: integers.ts ASC NULLS LAST, count(integers.i) ASC NULLS LAST_| -|_|_Aggregate: groupBy=[[integers.ts]], aggr=[[__count_merge(__count_state(integers.i)) AS count(integers.i)]] | +| logical_plan_| Sort: integers.ts ASC NULLS LAST_| +|_|_Aggregate: groupBy=[[integers.ts]], aggr=[[__count_merge(__count_state(integers.i)) AS count(integers.i)]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[integers.ts]], aggr=[[__count_state(integers.i)]]_| |_|_TableScan: integers_| |_| ]]_| -| physical_plan | SortPreservingMergeExec: [ts@0 ASC NULLS LAST, count(integers.i)@1 ASC NULLS LAST]_| -|_|_SortExec: expr=[ts@0 ASC NULLS LAST, count(integers.i)@1 ASC NULLS LAST], preserve_partitioning=[true]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(integers.i)]_| +| physical_plan | SortPreservingMergeExec: [ts@0 ASC NULLS LAST]_| +|_|_SortExec: expr=[ts@0 ASC NULLS LAST], preserve_partitioning=[true]_| +|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)] | |_|_RepartitionExec: partitioning=REDACTED -|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(integers.i)]_| +|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)]_| |_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -187,11 +187,11 @@ ORDER BY +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_SortPreservingMergeExec: [ts@0 ASC NULLS LAST, count(integers.i)@1 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[ts@0 ASC NULLS LAST, count(integers.i)@1 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(integers.i)] REDACTED +| 0_| 0_|_SortPreservingMergeExec: [ts@0 ASC NULLS LAST] REDACTED +|_|_|_SortExec: expr=[ts@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(integers.i)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -253,19 +253,19 @@ ORDER BY +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Sort: time_window ASC NULLS LAST, count(integers.i) ASC NULLS LAST_| +| logical_plan_| Sort: time_window ASC NULLS LAST_| |_|_Projection: date_bin(Utf8("1 hour"),integers.ts) AS time_window, count(integers.i)_| |_|_Aggregate: groupBy=[[date_bin(Utf8("1 hour"),integers.ts)]], aggr=[[__count_merge(__count_state(integers.i)) AS count(integers.i)]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 3600000000000 }"), integers.ts) AS date_bin(Utf8("1 hour"),integers.ts)]], aggr=[[__count_state(integers.i)]] | |_|_TableScan: integers_| |_| ]]_| -| physical_plan | SortPreservingMergeExec: [time_window@0 ASC NULLS LAST, count(integers.i)@1 ASC NULLS LAST]_| -|_|_SortExec: expr=[time_window@0 ASC NULLS LAST, count(integers.i)@1 ASC NULLS LAST], preserve_partitioning=[true]_| +| physical_plan | SortPreservingMergeExec: [time_window@0 ASC NULLS LAST]_| |_|_ProjectionExec: expr=[date_bin(Utf8("1 hour"),integers.ts)@0 as time_window, count(integers.i)@1 as count(integers.i)]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("1 hour"),integers.ts)@0 as date_bin(Utf8("1 hour"),integers.ts)], aggr=[count(integers.i)]_| +|_|_SortExec: expr=[date_bin(Utf8("1 hour"),integers.ts)@0 ASC NULLS LAST], preserve_partitioning=[true]_| +|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("1 hour"),integers.ts)@0 as date_bin(Utf8("1 hour"),integers.ts)], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)]_| |_|_RepartitionExec: partitioning=REDACTED -|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("1 hour"),integers.ts)@0 as date_bin(Utf8("1 hour"),integers.ts)], aggr=[count(integers.i)]_| +|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("1 hour"),integers.ts)@0 as date_bin(Utf8("1 hour"),integers.ts)], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)]_| |_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -295,12 +295,12 @@ ORDER BY +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_SortPreservingMergeExec: [time_window@0 ASC NULLS LAST, count(integers.i)@1 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[time_window@0 ASC NULLS LAST, count(integers.i)@1 ASC NULLS LAST], preserve_partitioning=[true] REDACTED +| 0_| 0_|_SortPreservingMergeExec: [time_window@0 ASC NULLS LAST] REDACTED |_|_|_ProjectionExec: expr=[date_bin(Utf8("1 hour"),integers.ts)@0 as time_window, count(integers.i)@1 as count(integers.i)] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("1 hour"),integers.ts)@0 as date_bin(Utf8("1 hour"),integers.ts)], aggr=[count(integers.i)] REDACTED +|_|_|_SortExec: expr=[date_bin(Utf8("1 hour"),integers.ts)@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("1 hour"),integers.ts)@0 as date_bin(Utf8("1 hour"),integers.ts)], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("1 hour"),integers.ts)@0 as date_bin(Utf8("1 hour"),integers.ts)], aggr=[count(integers.i)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("1 hour"),integers.ts)@0 as date_bin(Utf8("1 hour"),integers.ts)], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -380,9 +380,9 @@ ORDER BY |_| ]]_| | physical_plan | SortPreservingMergeExec: [integers.ts + Int64(1)@0 ASC NULLS LAST, integers.i / Int64(2)@1 ASC NULLS LAST]_| |_|_SortExec: expr=[integers.ts + Int64(1)@0 ASC NULLS LAST, integers.i / Int64(2)@1 ASC NULLS LAST], preserve_partitioning=[true]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[integers.ts + Int64(1)@0 as integers.ts + Int64(1), integers.i / Int64(2)@1 as integers.i / Int64(2)], aggr=[count(integers.i)] | +|_|_AggregateExec: mode=FinalPartitioned, gby=[integers.ts + Int64(1)@0 as integers.ts + Int64(1), integers.i / Int64(2)@1 as integers.i / Int64(2)], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)] | |_|_RepartitionExec: partitioning=REDACTED -|_|_AggregateExec: mode=Partial, gby=[integers.ts + Int64(1)@0 as integers.ts + Int64(1), integers.i / Int64(2)@1 as integers.i / Int64(2)], aggr=[count(integers.i)]_| +|_|_AggregateExec: mode=Partial, gby=[integers.ts + Int64(1)@0 as integers.ts + Int64(1), integers.i / Int64(2)@1 as integers.i / Int64(2)], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)]_| |_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -417,9 +417,9 @@ ORDER BY +-+-+-+ | 0_| 0_|_SortPreservingMergeExec: [integers.ts + Int64(1)@0 ASC NULLS LAST, integers.i / Int64(2)@1 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[integers.ts + Int64(1)@0 ASC NULLS LAST, integers.i / Int64(2)@1 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[integers.ts + Int64(1)@0 as integers.ts + Int64(1), integers.i / Int64(2)@1 as integers.i / Int64(2)], aggr=[count(integers.i)] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[integers.ts + Int64(1)@0 as integers.ts + Int64(1), integers.i / Int64(2)@1 as integers.i / Int64(2)], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[integers.ts + Int64(1)@0 as integers.ts + Int64(1), integers.i / Int64(2)@1 as integers.i / Int64(2)], aggr=[count(integers.i)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[integers.ts + Int64(1)@0 as integers.ts + Int64(1), integers.i / Int64(2)@1 as integers.i / Int64(2)], aggr=[__count_merge(__count_state(integers.i)) as count(integers.i)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -499,15 +499,15 @@ FROM | plan_type_| plan_| +-+-+ | logical_plan_| Projection: uddsketch_calc(Float64(0.5), uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state)) AS udd_result, hll_count(hll_merge(sink_table.hll_state)) AS hll_result_| -|_|_Aggregate: groupBy=[[]], aggr=[[__uddsketch_merge_merge(__uddsketch_merge_state(Int64(128),Float64(0.01),sink_table.udd_state)) AS uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state), __hll_merge_merge(__hll_merge_state(sink_table.hll_state)) AS hll_merge(sink_table.hll_state)]] | +|_|_Aggregate: groupBy=[[]], aggr=[[__uddsketch_merge_merge(__uddsketch_merge_state(Int64(128),Float64(0.01),sink_table.udd_state)) AS uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state), __hll_merge_merge(__hll_merge_state(sink_table.hll_state)) AS hll_merge(sink_table.hll_state)]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[]], aggr=[[__uddsketch_merge_state(Int64(128), Float64(0.01), sink_table.udd_state), __hll_merge_state(sink_table.hll_state)]]_| |_|_TableScan: sink_table_| |_| ]]_| | physical_plan | ProjectionExec: expr=[uddsketch_calc(0.5, uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state)@0) as udd_result, hll_count(hll_merge(sink_table.hll_state)@1) as hll_result]_| -|_|_AggregateExec: mode=Final, gby=[], aggr=[uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state), hll_merge(sink_table.hll_state)]_| +|_|_AggregateExec: mode=Final, gby=[], aggr=[__uddsketch_merge_merge(__uddsketch_merge_state(Int64(128),Float64(0.01),sink_table.udd_state)) as uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state), __hll_merge_merge(__hll_merge_state(sink_table.hll_state)) as hll_merge(sink_table.hll_state)]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state), hll_merge(sink_table.hll_state)]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__uddsketch_merge_merge(__uddsketch_merge_state(Int64(128),Float64(0.01),sink_table.udd_state)) as uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state), __hll_merge_merge(__hll_merge_state(sink_table.hll_state)) as hll_merge(sink_table.hll_state)] | |_|_RepartitionExec: partitioning=REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -533,9 +533,9 @@ FROM | stage | node | plan_| +-+-+-+ | 0_| 0_|_ProjectionExec: expr=[uddsketch_calc(0.5, uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state)@0) as udd_result, hll_count(hll_merge(sink_table.hll_state)@1) as hll_result] REDACTED -|_|_|_AggregateExec: mode=Final, gby=[], aggr=[uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state), hll_merge(sink_table.hll_state)] REDACTED +|_|_|_AggregateExec: mode=Final, gby=[], aggr=[__uddsketch_merge_merge(__uddsketch_merge_state(Int64(128),Float64(0.01),sink_table.udd_state)) as uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state), __hll_merge_merge(__hll_merge_state(sink_table.hll_state)) as hll_merge(sink_table.hll_state)] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state), hll_merge(sink_table.hll_state)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__uddsketch_merge_merge(__uddsketch_merge_state(Int64(128),Float64(0.01),sink_table.udd_state)) as uddsketch_merge(Int64(128),Float64(0.01),sink_table.udd_state), __hll_merge_merge(__hll_merge_state(sink_table.hll_state)) as hll_merge(sink_table.hll_state)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| diff --git a/tests/cases/distributed/explain/step_aggr_massive.result b/tests/cases/distributed/explain/step_aggr_massive.result index 1da87d364e..5aa969c674 100644 --- a/tests/cases/distributed/explain/step_aggr_massive.result +++ b/tests/cases/distributed/explain/step_aggr_massive.result @@ -246,16 +246,16 @@ GROUP BY | plan_type_| plan_| +-+-+ | logical_plan_| Projection: base_table.env, base_table.service_name, base_table.city, base_table.page, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END) AS lcp_state, max(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END) AS max_lcp, min(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END) AS min_lcp, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END) AS fmp_state, max(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END) AS max_fmp, min(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END) AS min_fmp, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END) AS fcp_state, max(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END) AS max_fcp, min(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END) AS min_fcp, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END) AS fp_state, max(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END) AS max_fp, min(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END) AS min_fp, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END) AS tti_state, max(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END) AS max_tti, min(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END) AS min_tti, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END) AS fid_state, max(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END) AS max_fid, min(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END) AS min_fid, max(base_table.shard_key) AS shard_key, arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))_| -|_|_Aggregate: groupBy=[[base_table.env, base_table.service_name, base_table.city, base_table.page, arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))]], aggr=[[__uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) AS uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) AS max(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) AS min(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) AS uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) AS max(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) AS min(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) AS uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) AS max(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) AS min(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) AS uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) AS max(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) AS min(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) AS uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) AS max(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) AS min(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) AS uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) AS max(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) AS min(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __max_merge(__max_state(base_table.shard_key)) AS max(base_table.shard_key)]] | +|_|_Aggregate: groupBy=[[base_table.env, base_table.service_name, base_table.city, base_table.page, arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))]], aggr=[[__uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) AS uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) AS max(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) AS min(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) AS uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) AS max(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) AS min(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) AS uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) AS max(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) AS min(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) AS uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) AS max(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) AS min(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) AS uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) AS max(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) AS min(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) AS uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) AS max(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) AS min(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __max_merge(__max_state(base_table.shard_key)) AS max(base_table.shard_key)]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[base_table.env, base_table.service_name, base_table.city, base_table.page, CAST(date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"), base_table.time) AS Timestamp(s)) AS arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))]], aggr=[[__uddsketch_state_state(Int64(128), Float64(0.01), CAST(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END AS Float64)), __max_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END), __min_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END), __uddsketch_state_state(Int64(128), Float64(0.01), CAST(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END AS Float64)), __max_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END), __min_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END), __uddsketch_state_state(Int64(128), Float64(0.01), CAST(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END AS Float64)), __max_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END), __min_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END), __uddsketch_state_state(Int64(128), Float64(0.01), CAST(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END AS Float64)), __max_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END), __min_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END), __uddsketch_state_state(Int64(128), Float64(0.01), CAST(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END AS Float64)), __max_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END), __min_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END), __uddsketch_state_state(Int64(128), Float64(0.01), CAST(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END AS Float64)), __max_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END), __min_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END), __max_state(base_table.shard_key)]]_| |_|_Filter: (base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) OR base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) OR base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) OR base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) OR base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) OR base_table.fid > Int64(0) AND base_table.fid < Int64(3000000)) AND base_table.time >= TimestampMillisecond(0, None)_| |_|_TableScan: base_table, partial_filters=[base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) OR base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) OR base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) OR base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) OR base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) OR base_table.fid > Int64(0) AND base_table.fid < Int64(3000000), base_table.time >= TimestampMillisecond(0, None)]_| |_| ]]_| | physical_plan | ProjectionExec: expr=[env@0 as env, service_name@1 as service_name, city@2 as city, page@3 as page, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END)@5 as lcp_state, max(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END)@6 as max_lcp, min(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END)@7 as min_lcp, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END)@8 as fmp_state, max(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END)@9 as max_fmp, min(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END)@10 as min_fmp, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END)@11 as fcp_state, max(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END)@12 as max_fcp, min(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END)@13 as min_fcp, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END)@14 as fp_state, max(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END)@15 as max_fp, min(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END)@16 as min_fp, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END)@17 as tti_state, max(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END)@18 as max_tti, min(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END)@19 as min_tti, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END)@20 as fid_state, max(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END)@21 as max_fid, min(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END)@22 as min_fid, max(base_table.shard_key)@23 as shard_key, arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))@4 as arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[env@0 as env, service_name@1 as service_name, city@2 as city, page@3 as page, arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))@4 as arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))], aggr=[uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), max(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), min(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), max(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), min(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), max(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), min(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), max(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), min(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), max(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), min(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), max(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), min(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), max(base_table.shard_key)]_| +|_|_AggregateExec: mode=FinalPartitioned, gby=[env@0 as env, service_name@1 as service_name, city@2 as city, page@3 as page, arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))@4 as arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))], aggr=[__uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) as max(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) as min(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __max_merge(__max_state(base_table.shard_key)) as max(base_table.shard_key)] | |_|_RepartitionExec: partitioning=REDACTED -|_|_AggregateExec: mode=Partial, gby=[env@0 as env, service_name@1 as service_name, city@2 as city, page@3 as page, arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))@4 as arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))], aggr=[uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), max(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), min(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), max(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), min(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), max(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), min(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), max(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), min(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), max(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), min(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), max(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), min(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), max(base_table.shard_key)]_| +|_|_AggregateExec: mode=Partial, gby=[env@0 as env, service_name@1 as service_name, city@2 as city, page@3 as page, arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))@4 as arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))], aggr=[__uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) as max(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) as min(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __max_merge(__max_state(base_table.shard_key)) as max(base_table.shard_key)]_| |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -456,9 +456,9 @@ GROUP BY | stage | node | plan_| +-+-+-+ | 0_| 0_|_ProjectionExec: expr=[env@0 as env, service_name@1 as service_name, city@2 as city, page@3 as page, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END)@5 as lcp_state, max(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END)@6 as max_lcp, min(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END)@7 as min_lcp, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END)@8 as fmp_state, max(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END)@9 as max_fmp, min(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END)@10 as min_fmp, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END)@11 as fcp_state, max(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END)@12 as max_fcp, min(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END)@13 as min_fcp, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END)@14 as fp_state, max(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END)@15 as max_fp, min(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END)@16 as min_fp, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END)@17 as tti_state, max(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END)@18 as max_tti, min(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END)@19 as min_tti, uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END)@20 as fid_state, max(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END)@21 as max_fid, min(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END)@22 as min_fid, max(base_table.shard_key)@23 as shard_key, arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))@4 as arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[env@0 as env, service_name@1 as service_name, city@2 as city, page@3 as page, arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))@4 as arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))], aggr=[uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), max(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), min(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), max(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), min(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), max(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), min(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), max(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), min(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), max(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), min(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), max(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), min(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), max(base_table.shard_key)] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[env@0 as env, service_name@1 as service_name, city@2 as city, page@3 as page, arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))@4 as arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))], aggr=[__uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) as max(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) as min(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __max_merge(__max_state(base_table.shard_key)) as max(base_table.shard_key)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[env@0 as env, service_name@1 as service_name, city@2 as city, page@3 as page, arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))@4 as arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))], aggr=[uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), max(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), min(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), max(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), min(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), max(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), min(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), max(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), min(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), max(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), min(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), max(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), min(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), max(base_table.shard_key)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[env@0 as env, service_name@1 as service_name, city@2 as city, page@3 as page, arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))@4 as arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))], aggr=[__uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) as max(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END)) as min(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE NULL END), __uddsketch_state_merge(__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) as uddsketch_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __max_merge(__max_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) as max(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __min_merge(__min_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END)) as min(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE NULL END), __max_merge(__max_state(base_table.shard_key)) as max(base_table.shard_key)] REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=FinalPartitioned, gby=[env@0 as env, service_name@1 as service_name, city@2 as city, page@3 as page, arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))@4 as arrow_cast(date_bin(Utf8("60 seconds"),base_table.time),Utf8("Timestamp(s)"))], aggr=[__uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END), __max_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END), __min_state(CASE WHEN base_table.lcp > Int64(0) AND base_table.lcp < Int64(3000000) THEN base_table.lcp ELSE Int64(NULL) END), __uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END), __max_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END), __min_state(CASE WHEN base_table.fmp > Int64(0) AND base_table.fmp < Int64(3000000) THEN base_table.fmp ELSE Int64(NULL) END), __uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END), __max_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END), __min_state(CASE WHEN base_table.fcp > Int64(0) AND base_table.fcp < Int64(3000000) THEN base_table.fcp ELSE Int64(NULL) END), __uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END), __max_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END), __min_state(CASE WHEN base_table.fp > Int64(0) AND base_table.fp < Int64(3000000) THEN base_table.fp ELSE Int64(NULL) END), __uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END), __max_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END), __min_state(CASE WHEN base_table.tti > Int64(0) AND base_table.tti < Int64(3000000) THEN base_table.tti ELSE Int64(NULL) END), __uddsketch_state_state(Int64(128),Float64(0.01),CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END), __max_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END), __min_state(CASE WHEN base_table.fid > Int64(0) AND base_table.fid < Int64(3000000) THEN base_table.fid ELSE Int64(NULL) END), __max_state(base_table.shard_key)] REDACTED @@ -600,9 +600,9 @@ where |_|_TableScan: base_table, partial_filters=[base_table.time >= TimestampMillisecond(0, None)]_| |_| ]]_| | physical_plan | ProjectionExec: expr=[count(Int64(1))@0 as count(*)]_| -|_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))]_| +|_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(base_table.time)) as count(Int64(1))]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(base_table.time)) as count(Int64(1))]_| |_|_MergeScanExec: REDACTED |_|_| +-+-+ @@ -628,9 +628,9 @@ where | stage | node | plan_| +-+-+-+ | 0_| 0_|_ProjectionExec: expr=[count(Int64(1))@0 as count(*)] REDACTED -|_|_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(base_table.time)) as count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(base_table.time)) as count(Int64(1))] REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_state(base_table.time)] REDACTED diff --git a/tests/cases/distributed/explain/subqueries.result b/tests/cases/distributed/explain/subqueries.result index d7ad6ed920..0ea758a6fe 100644 --- a/tests/cases/distributed/explain/subqueries.result +++ b/tests/cases/distributed/explain/subqueries.result @@ -122,17 +122,17 @@ EXPLAIN INSERT INTO other SELECT i, 2 FROM integers WHERE i=(SELECT MAX(i) FROM +---------------------+-----------------------------------------------------------------------------+ | logical_plan | Dml: op=[Insert Into] table=[other] | | | Projection: integers.i AS i, TimestampMillisecond(2, None) AS j | -| | Inner Join: integers.i = __scalar_sq_1.max(integers.i) | +| | Filter: integers.i = () | +| | Subquery: | +| | MergeScan [is_placeholder=false, remote_input=[ | +| | Projection: max(integers.i) | +| | Aggregate: groupBy=[[]], aggr=[[max(integers.i)]] | +| | TableScan: integers | +| | ]] | | | Projection: integers.i | | | MergeScan [is_placeholder=false, remote_input=[ | | | TableScan: integers | | | ]] | -| | MergeScan [is_placeholder=false, remote_input=[ | -| | SubqueryAlias: __scalar_sq_1 | -| | Projection: max(integers.i) | -| | Aggregate: groupBy=[[]], aggr=[[max(integers.i)]] | -| | TableScan: integers, partial_filters=[Boolean(true)] | -| | ]] | | physical_plan_error | This feature is not implemented: Insert into not implemented for this table | +---------------------+-----------------------------------------------------------------------------+ @@ -598,8 +598,8 @@ EXPLAIN SELECT x FROM (VALUES (2),(1)) v(x) ORDER BY x; |_|_SubqueryAlias: v_| |_|_Projection: column1 AS x_| |_|_Values: (Int64(2)), (Int64(1))_| -| physical_plan | SortExec: expr=[x@0 ASC NULLS LAST], preserve_partitioning=[false] | -|_|_ProjectionExec: expr=[column1@0 as x]_| +| physical_plan | ProjectionExec: expr=[column1@0 as x]_| +|_|_SortExec: expr=[column1@0 ASC NULLS LAST], preserve_partitioning=[false] | |_|_DataSourceExec: partitions=1, partition_sizes=[1]_| |_|_| +-+-+ diff --git a/tests/cases/distributed/flow-tql/flow_tql.result b/tests/cases/distributed/flow-tql/flow_tql.result index 040ca34dbc..925fff3f24 100644 --- a/tests/cases/distributed/flow-tql/flow_tql.result +++ b/tests/cases/distributed/flow-tql/flow_tql.result @@ -43,8 +43,10 @@ SHOW CREATE TABLE cnt_reqs; -- test if sink table is tql queryable TQL EVAL (now() - '1m'::interval, now(), '5s') count_values("status_code", cnt_reqs); -++ -++ ++------------------------------------------+----+-------------+ +| count(cnt_reqs.count(http_requests.val)) | ts | status_code | ++------------------------------------------+----+-------------+ ++------------------------------------------+----+-------------+ INSERT INTO TABLE http_requests VALUES (now() - '17s'::interval, 'host1', 'idc1', 200), @@ -187,8 +189,10 @@ SHOW CREATE TABLE cnt_reqs; -- test if sink table is tql queryable TQL EVAL (now() - '1m'::interval, now(), '5s') count_values("status_code", cnt_reqs); -++ -++ ++------------------------------------------+----+-------------+ +| count(cnt_reqs.count(http_requests.val)) | ts | status_code | ++------------------------------------------+----+-------------+ ++------------------------------------------+----+-------------+ INSERT INTO TABLE http_requests VALUES (0::Timestamp, 'host1', 'idc1', 200), @@ -278,8 +282,10 @@ SHOW CREATE TABLE rate_reqs; -- test if sink table is tql queryable TQL EVAL (now() - '1m'::interval, now(), '5s') count_values("status_code", rate_reqs); -++ -++ ++-----------------------------------------------------------+----+-------------+ +| count(rate_reqs.prom_rate(ts_range,val,ts,Int64(300000))) | ts | status_code | ++-----------------------------------------------------------+----+-------------+ ++-----------------------------------------------------------+----+-------------+ INSERT INTO TABLE http_requests VALUES (now() - '1m'::interval, 0), @@ -359,8 +365,10 @@ SHOW CREATE TABLE rate_reqs; -- test if sink table is tql queryable TQL EVAL (now() - '1m'::interval, now(), '5s') count_values("status_code", rate_reqs); -++ -++ ++-----------------------------------------------------------+----+-------------+ +| count(rate_reqs.prom_rate(ts_range,byte,ts,Int64(60000))) | ts | status_code | ++-----------------------------------------------------------+----+-------------+ ++-----------------------------------------------------------+----+-------------+ INSERT INTO TABLE http_requests_total VALUES ('localhost', 'my_service', 'instance1', 100, now() - '1min'::interval), diff --git a/tests/cases/distributed/flow-tql/tsid_on_phy.result b/tests/cases/distributed/flow-tql/tsid_on_phy.result index 0a1755b5c2..797e8242a3 100644 --- a/tests/cases/distributed/flow-tql/tsid_on_phy.result +++ b/tests/cases/distributed/flow-tql/tsid_on_phy.result @@ -115,8 +115,8 @@ TQL EXPLAIN ( | | Sort: test_tsid.__tsid ASC NULLS FIRST, test_tsid.ts ASC NULLS FIRST | | | Projection: test_tsid.v, test_tsid.le, test_tsid.tag1, test_tsid.tag2, test_tsid.tag4, test_tsid.tag5, test_tsid.tag6, test_tsid.tag7, test_tsid.tag8, test_tsid.__tsid, test_tsid.ts | | | SubqueryAlias: test_tsid | -| | Filter: phy.ts >= TimestampMillisecond(1769137200001, None) AND phy.ts <= TimestampMillisecond(1769139900000, None) AND phy.__table_id=UInt32(REDACTED) | -| | TableScan: phy projection=[ts, v, tag1, tag2, le, tag4, tag5, tag6, tag7, tag8, __table_id, __tsid], partial_filters=[phy.ts >= TimestampMillisecond(1769137200001, None), phy.ts <= TimestampMillisecond(1769139900000, None), phy.__table_id=UInt32(REDACTED)] | +| | Filter: phy.__table_id=UInt32(REDACTED) AND phy.ts >= TimestampMillisecond(1769137200001, None) AND phy.ts <= TimestampMillisecond(1769139900000, None) | +| | TableScan: phy projection=[ts, v, tag1, tag2, le, tag4, tag5, tag6, tag7, tag8, __table_id, __tsid], partial_filters=[phy.__table_id=UInt32(REDACTED), phy.ts >= TimestampMillisecond(1769137200001, None), phy.ts <= TimestampMillisecond(1769139900000, None)] | | | ]] | | physical_plan | HistogramFoldExec: le=@0, field=@4, quantile=0.5 | | | RepartitionExec: REDACTED diff --git a/tests/cases/distributed/optimizer/count.result b/tests/cases/distributed/optimizer/count.result index 7368ad6bbd..7c2d419b8c 100644 --- a/tests/cases/distributed/optimizer/count.result +++ b/tests/cases/distributed/optimizer/count.result @@ -278,9 +278,9 @@ select count(1) from count_where_bug; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -314,9 +314,9 @@ select count(1) from count_where_bug where `tag` = 'b'; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -341,9 +341,9 @@ select count(1) from count_where_bug where `tag` = 'b'; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -378,9 +378,9 @@ select count(1) from count_where_bug where ts > '2024-09-06T06:00:04Z'; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -418,9 +418,9 @@ select count(1) from count_where_bug where num != 3; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| diff --git a/tests/cases/distributed/optimizer/filter_push_down.result b/tests/cases/distributed/optimizer/filter_push_down.result index 550c41f9da..83c438de4e 100644 --- a/tests/cases/distributed/optimizer/filter_push_down.result +++ b/tests/cases/distributed/optimizer/filter_push_down.result @@ -57,20 +57,26 @@ SELECT i1.i,i2.i FROM integers i1 LEFT OUTER JOIN integers i2 ON 1=1 WHERE i1.i> -- Align the result to PostgreSQL: empty. SELECT i1.i,i2.i FROM integers i1 LEFT OUTER JOIN integers i2 ON 1=0 WHERE i2.i IS NOT NULL ORDER BY 2; -++ -++ ++---+---+ +| i | i | ++---+---+ ++---+---+ -- Align the result to PostgreSQL: empty. SELECT i1.i,i2.i FROM integers i1 LEFT OUTER JOIN integers i2 ON 1=0 WHERE i2.i>1 ORDER BY 2; -++ -++ ++---+---+ +| i | i | ++---+---+ ++---+---+ -- Align the result to PostgreSQL: empty. SELECT i1.i,i2.i FROM integers i1 LEFT OUTER JOIN integers i2 ON 1=0 WHERE CASE WHEN i2.i IS NULL THEN False ELSE True END ORDER BY 2; -++ -++ ++---+---+ +| i | i | ++---+---+ ++---+---+ SELECT DISTINCT i1.i,i2.i FROM integers i1 LEFT OUTER JOIN integers i2 ON 1=0 WHERE i2.i IS NULL ORDER BY 1; @@ -213,8 +219,10 @@ EXPLAIN SELECT * FROM (SELECT 0=1 AS cond FROM integers i1, integers i2) a1 WHER -- Align the result to PostgreSQL: empty. SELECT * FROM (SELECT 0=1 AS cond FROM integers i1, integers i2 GROUP BY 1) a1 WHERE cond ORDER BY 1; -++ -++ ++------+ +| cond | ++------+ ++------+ DROP TABLE integers; diff --git a/tests/cases/distributed/optimizer/first_value_advance.result b/tests/cases/distributed/optimizer/first_value_advance.result index 519ec9d9c1..78aa8bcf6a 100644 --- a/tests/cases/distributed/optimizer/first_value_advance.result +++ b/tests/cases/distributed/optimizer/first_value_advance.result @@ -317,14 +317,14 @@ explain select first_value(ts order by ts) from t; +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]] | +| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[]], aggr=[[__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_TableScan: t_| |_| ]]_| -| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[__first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -346,9 +346,9 @@ explain analyze +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -403,19 +403,19 @@ order by time_window, ordered_host; +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Sort: time_window ASC NULLS LAST, ordered_host ASC NULLS LAST_| +| logical_plan_| Sort: time_window ASC NULLS LAST_| |_|_Projection: first_value(t.host) ORDER BY [t.ts ASC NULLS LAST] AS ordered_host, first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t.ts) AS time_window_| -|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t.ts)]], aggr=[[__first_value_merge(__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]] | +|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t.ts)]], aggr=[[__first_value_merge(__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 5000000 }"), t.ts) AS date_bin(Utf8("5 milliseconds"),t.ts)]], aggr=[[__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_TableScan: t_| |_| ]]_| -| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST]_| -|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST]_| |_|_ProjectionExec: expr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST]@1 as ordered_host, first_value(t.val) ORDER BY [t.ts ASC NULLS LAST]@2 as first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]@3 as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t.ts)@0 as time_window]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__first_value_merge(__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED -|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__first_value_merge(__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -444,12 +444,12 @@ order by time_window, ordered_host; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST] REDACTED |_|_|_ProjectionExec: expr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST]@1 as ordered_host, first_value(t.val) ORDER BY [t.ts ASC NULLS LAST]@2 as first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]@3 as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t.ts)@0 as time_window] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__first_value_merge(__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__first_value_merge(__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -521,14 +521,14 @@ explain +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]] | +| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[]], aggr=[[__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_TableScan: t1_| |_| ]]_| -| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[__first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -550,9 +550,9 @@ explain analyze +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -704,19 +704,19 @@ order by time_window, ordered_host; +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Sort: time_window ASC NULLS LAST, ordered_host ASC NULLS LAST_| +| logical_plan_| Sort: time_window ASC NULLS LAST_| |_|_Projection: first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST] AS ordered_host, first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t1.ts) AS time_window_| -|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t1.ts)]], aggr=[[__first_value_merge(__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]] | +|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t1.ts)]], aggr=[[__first_value_merge(__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 5000000 }"), t1.ts) AS date_bin(Utf8("5 milliseconds"),t1.ts)]], aggr=[[__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_TableScan: t1_| |_| ]]_| -| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST]_| -|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST]_| |_|_ProjectionExec: expr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST]@1 as ordered_host, first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST]@2 as first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]@3 as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t1.ts)@0 as time_window]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__first_value_merge(__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED -|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__first_value_merge(__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -745,12 +745,12 @@ order by time_window, ordered_host; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST] REDACTED |_|_|_ProjectionExec: expr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST]@1 as ordered_host, first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST]@2 as first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]@3 as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t1.ts)@0 as time_window] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__first_value_merge(__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__first_value_merge(__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| diff --git a/tests/cases/distributed/optimizer/last_value_advance.result b/tests/cases/distributed/optimizer/last_value_advance.result index 0972741b17..795f34a582 100644 --- a/tests/cases/distributed/optimizer/last_value_advance.result +++ b/tests/cases/distributed/optimizer/last_value_advance.result @@ -317,14 +317,14 @@ explain select last_value(ts order by ts) from t; +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]] | +| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[]], aggr=[[__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_TableScan: t_| |_| ]]_| -| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[__last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -346,9 +346,9 @@ explain analyze +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -403,19 +403,19 @@ order by time_window, ordered_host; +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Sort: time_window ASC NULLS LAST, ordered_host ASC NULLS LAST_| +| logical_plan_| Sort: time_window ASC NULLS LAST_| |_|_Projection: last_value(t.host) ORDER BY [t.ts ASC NULLS LAST] AS ordered_host, last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t.ts) AS time_window_| -|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t.ts)]], aggr=[[__last_value_merge(__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]] | +|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t.ts)]], aggr=[[__last_value_merge(__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 5000000 }"), t.ts) AS date_bin(Utf8("5 milliseconds"),t.ts)]], aggr=[[__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_TableScan: t_| |_| ]]_| -| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST]_| -|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST]_| |_|_ProjectionExec: expr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST]@1 as ordered_host, last_value(t.val) ORDER BY [t.ts ASC NULLS LAST]@2 as last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]@3 as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t.ts)@0 as time_window]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__last_value_merge(__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED -|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__last_value_merge(__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -444,12 +444,12 @@ order by time_window, ordered_host; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST] REDACTED |_|_|_ProjectionExec: expr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST]@1 as ordered_host, last_value(t.val) ORDER BY [t.ts ASC NULLS LAST]@2 as last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]@3 as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t.ts)@0 as time_window] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__last_value_merge(__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__last_value_merge(__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -521,14 +521,14 @@ explain +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]] | +| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[]], aggr=[[__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_TableScan: t1_| |_| ]]_| -| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[__last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -550,9 +550,9 @@ explain analyze +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -704,19 +704,19 @@ order by time_window, ordered_host; +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Sort: time_window ASC NULLS LAST, ordered_host ASC NULLS LAST_| +| logical_plan_| Sort: time_window ASC NULLS LAST_| |_|_Projection: last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST] AS ordered_host, last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t1.ts) AS time_window_| -|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t1.ts)]], aggr=[[__last_value_merge(__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]] | +|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t1.ts)]], aggr=[[__last_value_merge(__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 5000000 }"), t1.ts) AS date_bin(Utf8("5 milliseconds"),t1.ts)]], aggr=[[__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_TableScan: t1_| |_| ]]_| -| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST]_| -|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST]_| |_|_ProjectionExec: expr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST]@1 as ordered_host, last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST]@2 as last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]@3 as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t1.ts)@0 as time_window]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__last_value_merge(__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED -|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__last_value_merge(__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -745,12 +745,12 @@ order by time_window, ordered_host; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST] REDACTED |_|_|_ProjectionExec: expr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST]@1 as ordered_host, last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST]@2 as last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]@3 as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t1.ts)@0 as time_window] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__last_value_merge(__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__last_value_merge(__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| diff --git a/tests/cases/distributed/optimizer/range_select_projection.result b/tests/cases/distributed/optimizer/range_select_projection.result index 451e743da9..28d8fff50a 100644 --- a/tests/cases/distributed/optimizer/range_select_projection.result +++ b/tests/cases/distributed/optimizer/range_select_projection.result @@ -45,8 +45,8 @@ ORDER BY station, "channel", ts; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_SortExec: expr=[station@1 ASC NULLS LAST, channel@2 ASC NULLS LAST, ts@0 ASC NULLS LAST], preserve_REDACTED -|_|_|_ProjectionExec: expr=[ts@1 as ts, station@2 as station, channel@3 as channel, avg(range_select_projection.value_a + range_select_projection.value_b) RANGE 10s@0 as avg_value] REDACTED +| 0_| 0_|_ProjectionExec: expr=[ts@1 as ts, station@2 as station, channel@3 as channel, avg(range_select_projection.value_a + range_select_projection.value_b) RANGE 10s@0 as avg_value] REDACTED +|_|_|_SortExec: expr=[station@2 ASC NULLS LAST, channel@3 ASC NULLS LAST, ts@1 ASC NULLS LAST], preserve_REDACTED |_|_|_RangeSelectExec: range_expr=[avg(range_select_projection.value_a + range_select_projection.value_b) RANGE 10s], align=5000ms, align_to=0ms, align_by=[station@1, channel@2], time_index=ts REDACTED |_|_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED diff --git a/tests/cases/distributed/optimizer/time_index_filter_pushdown.result b/tests/cases/distributed/optimizer/time_index_filter_pushdown.result index 016c49deea..f03818154c 100644 --- a/tests/cases/distributed/optimizer/time_index_filter_pushdown.result +++ b/tests/cases/distributed/optimizer/time_index_filter_pushdown.result @@ -112,13 +112,8 @@ WHERE +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Projection: cpu.rack, cpu.os, cpu.greptime_timestamp_| -|_|_Inner Join: cpu.greptime_timestamp = __scalar_sq_1.greptime_timestamp_| -|_|_Projection: cpu.rack, cpu.os, cpu.greptime_timestamp_| -|_|_MergeScan [is_placeholder=false, remote_input=[_| -|_| TableScan: cpu_| -|_| ]]_| -|_|_SubqueryAlias: __scalar_sq_1_| +| logical_plan_| Filter: cpu.greptime_timestamp = ()_| +|_|_Subquery:_| |_|_Limit: skip=0, fetch=1_| |_|_MergeSort: cpu.greptime_timestamp DESC NULLS FIRST_| |_|_MergeScan [is_placeholder=false, remote_input=[_| @@ -127,12 +122,17 @@ WHERE |_|_Projection: cpu.greptime_timestamp_| |_|_TableScan: cpu_| |_| ]]_| -| physical_plan | HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(greptime_timestamp@0, greptime_timestamp@2)], projection=[rack@1, os@2, greptime_timestamp@3]_| -|_|_MergeSortExec: [greptime_timestamp@0 DESC], fetch=1_| -|_|_MergeScanExec: REDACTED +|_|_Projection: cpu.rack, cpu.os, cpu.greptime_timestamp_| +|_|_MergeScan [is_placeholder=false, remote_input=[_| +|_| TableScan: cpu_| +|_| ]]_| +| physical_plan | ScalarSubqueryExec: subqueries=1_| +|_|_FilterExec: greptime_timestamp@2 = scalar_subquery()_| |_|_ProjectionExec: expr=[rack@0 as rack, os@1 as os, greptime_timestamp@3 as greptime_timestamp]_| |_|_CooperativeExec_| |_|_MergeScanExec: REDACTED +|_|_MergeSortExec: [greptime_timestamp@0 DESC], fetch=1_| +|_|_MergeScanExec: REDACTED |_|_| +-+-+ diff --git a/tests/cases/distributed/optimizer/windowed_sort.result b/tests/cases/distributed/optimizer/windowed_sort.result index 6f4e18defb..0c4d1931c0 100644 --- a/tests/cases/distributed/optimizer/windowed_sort.result +++ b/tests/cases/distributed/optimizer/windowed_sort.result @@ -227,9 +227,9 @@ ORDER BY |_|_|_| | 1_| 0_|_ProjectionExec: expr=[collect_time_utc@0 as collect_time, collect_time@1 as true_collect_time, peak_current@2 as peak_current] REDACTED |_|_|_SortPreservingMergeExec: [collect_time@1 DESC] REDACTED -|_|_|_WindowedSortExec: expr=collect_time@1 DESC num_ranges=REDACTED REDACTED -|_|_|_PartSortExec: expr=collect_time@1 DESC num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[collect_time_utc@1 as collect_time_utc, collect_time@0 as collect_time, peak_current@2 as peak_current] REDACTED +|_|_|_WindowedSortExec: expr=collect_time@0 DESC num_ranges=REDACTED REDACTED +|_|_|_PartSortExec: expr=collect_time@0 DESC num_ranges=REDACTED REDACTED |_|_|_SeqScan: region=REDACTED, "partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0} REDACTED |_|_|_| |_|_| Total rows: 8_| diff --git a/tests/cases/distributed/optimizer/windowed_sort_advance.result b/tests/cases/distributed/optimizer/windowed_sort_advance.result index e3c92a0a08..1b122e61c8 100644 --- a/tests/cases/distributed/optimizer/windowed_sort_advance.result +++ b/tests/cases/distributed/optimizer/windowed_sort_advance.result @@ -65,8 +65,8 @@ EXPLAIN ANALYZE select ts as ts, status, value from `a` where ts >= '2026-03-12T |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_SortPreservingMergeExec: [ts@0 ASC NULLS LAST] REDACTED -|_|_|_WindowedSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[ts@2 as ts, status@1 as status, value@0 as value] REDACTED +|_|_|_WindowedSortExec: expr=ts@2 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_SeqScan: region=REDACTED, "partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0} REDACTED |_|_|_| |_|_| Total rows: 10_| diff --git a/tests/cases/distributed/repartition/repartition.result b/tests/cases/distributed/repartition/repartition.result index 7b25a13a06..2b03139abd 100644 --- a/tests/cases/distributed/repartition/repartition.result +++ b/tests/cases/distributed/repartition/repartition.result @@ -210,18 +210,24 @@ SHOW CREATE TABLE metric_physical_table; -- Verify select * works and returns empty SELECT * FROM metric_physical_table; -++ -++ ++----+------+-----+------------+--------+ +| ts | host | cpu | __table_id | __tsid | ++----+------+-----+------------+--------+ ++----+------+-----+------------+--------+ SELECT * FROM logical_table_v1; -++ -++ ++-----+------+----+ +| cpu | host | ts | ++-----+------+----+ ++-----+------+----+ SELECT * FROM logical_table_v2; -++ -++ ++-----+------+----+ +| cpu | host | ts | ++-----+------+----+ ++-----+------+----+ -- Repartition requires metasrv GC. Verify split requests are rejected when it is disabled. ALTER TABLE metric_physical_table MERGE PARTITION ( @@ -257,18 +263,24 @@ SHOW CREATE TABLE metric_physical_table; -- Verify select * works and returns empty SELECT * FROM metric_physical_table; -++ -++ ++----+------+-----+------------+--------+ +| ts | host | cpu | __table_id | __tsid | ++----+------+-----+------------+--------+ ++----+------+-----+------------+--------+ SELECT * FROM logical_table_v1; -++ -++ ++-----+------+----+ +| cpu | host | ts | ++-----+------+----+ ++-----+------+----+ SELECT * FROM logical_table_v2; -++ -++ ++-----+------+----+ +| cpu | host | ts | ++-----+------+----+ ++-----+------+----+ DROP TABLE logical_table_v1; diff --git a/tests/cases/standalone/common/tql/general_table.result b/tests/cases/distributed/tql/general_table.result similarity index 96% rename from tests/cases/standalone/common/tql/general_table.result rename to tests/cases/distributed/tql/general_table.result index 61d414c333..e4add60c60 100644 --- a/tests/cases/standalone/common/tql/general_table.result +++ b/tests/cases/distributed/tql/general_table.result @@ -20,7 +20,6 @@ Affected Rows: 0 -- SQLNESS REPLACE (\s\s+) _ -- SQLNESS REPLACE (cpu_usage\.ts_range) ts_range -- SQLNESS REPLACE (cpu_usage\.ts) ts --- SQLNESS REPLACE (?m)^\|_\|_\|_SortExec:.*\n\|_\|_\|_RepartitionExec:.*\n -- SQLNESS REPLACE (SeriesScan:.*|SeqScan:.*) ScanExec: REDACTED -- SQLNESS REPLACE (peers.*) REDACTED -- SQLNESS REPLACE region=\d+\(\d+,\s+\d+\) region=REDACTED diff --git a/tests/cases/standalone/common/tql/general_table.sql b/tests/cases/distributed/tql/general_table.sql similarity index 91% rename from tests/cases/standalone/common/tql/general_table.sql rename to tests/cases/distributed/tql/general_table.sql index 6bf9549773..53dd5b4e7f 100644 --- a/tests/cases/standalone/common/tql/general_table.sql +++ b/tests/cases/distributed/tql/general_table.sql @@ -18,7 +18,6 @@ WITH( -- SQLNESS REPLACE (\s\s+) _ -- SQLNESS REPLACE (cpu_usage\.ts_range) ts_range -- SQLNESS REPLACE (cpu_usage\.ts) ts --- SQLNESS REPLACE (?m)^\|_\|_\|_SortExec:.*\n\|_\|_\|_RepartitionExec:.*\n -- SQLNESS REPLACE (SeriesScan:.*|SeqScan:.*) ScanExec: REDACTED -- SQLNESS REPLACE (peers.*) REDACTED -- SQLNESS REPLACE region=\d+\(\d+,\s+\d+\) region=REDACTED diff --git a/tests/cases/standalone/common/aggregate/approx_distinct.result b/tests/cases/standalone/common/aggregate/approx_distinct.result index a3875fadf0..b11fcf37fd 100644 --- a/tests/cases/standalone/common/aggregate/approx_distinct.result +++ b/tests/cases/standalone/common/aggregate/approx_distinct.result @@ -91,7 +91,7 @@ SELECT APPROX_DISTINCT(a), APPROX_DISTINCT(b) FROM large_test; +-------------------------------+-------------------------------+ | approx_distinct(large_test.a) | approx_distinct(large_test.b) | +-------------------------------+-------------------------------+ -| 2000 | 10 | +| 1991 | 10 | +-------------------------------+-------------------------------+ -- Test with groups @@ -100,15 +100,15 @@ SELECT b, APPROX_DISTINCT(a) FROM large_test GROUP BY b ORDER BY b; +---+-------------------------------+ | b | approx_distinct(large_test.a) | +---+-------------------------------+ -| 0 | 200 | -| 1 | 201 | -| 2 | 201 | -| 3 | 200 | -| 4 | 199 | -| 5 | 200 | -| 6 | 199 | -| 7 | 200 | -| 8 | 200 | +| 0 | 199 | +| 1 | 197 | +| 2 | 200 | +| 3 | 199 | +| 4 | 198 | +| 5 | 197 | +| 6 | 200 | +| 7 | 196 | +| 8 | 199 | | 9 | 200 | +---+-------------------------------+ diff --git a/tests/cases/standalone/common/aggregate/approx_median.result b/tests/cases/standalone/common/aggregate/approx_median.result index 9d5bde78f6..46d52adf38 100644 --- a/tests/cases/standalone/common/aggregate/approx_median.result +++ b/tests/cases/standalone/common/aggregate/approx_median.result @@ -15,7 +15,7 @@ SELECT approx_median(i) FROM odd_test; +---------------------------+ | approx_median(odd_test.i) | +---------------------------+ -| 3 | +| 3.0 | +---------------------------+ -- Test with even number of values @@ -33,7 +33,7 @@ SELECT approx_median(i) FROM even_test; +----------------------------+ | approx_median(even_test.i) | +----------------------------+ -| 3 | +| 3.25 | +----------------------------+ -- Test with larger dataset @@ -50,7 +50,7 @@ SELECT approx_median(val) FROM large_test; +-------------------------------+ | approx_median(large_test.val) | +-------------------------------+ -| 499 | +| 499.25 | +-------------------------------+ -- Test with groups @@ -59,9 +59,9 @@ SELECT grp, approx_median(val) FROM large_test GROUP BY grp ORDER BY grp; +-----+-------------------------------+ | grp | approx_median(large_test.val) | +-----+-------------------------------+ -| 0 | 498 | -| 1 | 499 | -| 2 | 500 | +| 0 | 498.75 | +| 1 | 499.32142857142856 | +| 2 | 500.32142857142856 | +-----+-------------------------------+ -- Test with doubles @@ -111,7 +111,7 @@ SELECT approx_median(val) FROM dup_test; +-----------------------------+ | approx_median(dup_test.val) | +-----------------------------+ -| 2 | +| 2.75 | +-----------------------------+ -- Compare with exact median @@ -120,7 +120,7 @@ SELECT median(val), approx_median(val) FROM dup_test; +----------------------+-----------------------------+ | median(dup_test.val) | approx_median(dup_test.val) | +----------------------+-----------------------------+ -| 2 | 2 | +| 2.5 | 2.75 | +----------------------+-----------------------------+ -- Test edge cases @@ -139,7 +139,7 @@ SELECT approx_median(i) FROM odd_test WHERE i = 3; +---------------------------+ | approx_median(odd_test.i) | +---------------------------+ -| 3 | +| 3.0 | +---------------------------+ -- Test with negative values @@ -156,7 +156,7 @@ SELECT approx_median(val) FROM neg_test; +-----------------------------+ | approx_median(neg_test.val) | +-----------------------------+ -| 0 | +| 0.0 | +-----------------------------+ -- cleanup diff --git a/tests/cases/standalone/common/aggregate/approx_percentile_cont.result b/tests/cases/standalone/common/aggregate/approx_percentile_cont.result index ac9d60186e..fcdf839487 100644 --- a/tests/cases/standalone/common/aggregate/approx_percentile_cont.result +++ b/tests/cases/standalone/common/aggregate/approx_percentile_cont.result @@ -16,7 +16,7 @@ SELECT approx_percentile_cont(0.5) WITHIN GROUP (ORDER BY i) FROM approx_test; +----------------------------------------------------------------------------------+ | approx_percentile_cont(Float64(0.5)) WITHIN GROUP [approx_test.i ASC NULLS LAST] | +----------------------------------------------------------------------------------+ -| 499 | +| 499.25 | +----------------------------------------------------------------------------------+ -- first quartile @@ -25,7 +25,7 @@ SELECT approx_percentile_cont(0.25) WITHIN GROUP (ORDER BY i) FROM approx_test; +-----------------------------------------------------------------------------------+ | approx_percentile_cont(Float64(0.25)) WITHIN GROUP [approx_test.i ASC NULLS LAST] | +-----------------------------------------------------------------------------------+ -| 249 | +| 249.575 | +-----------------------------------------------------------------------------------+ -- third quartile @@ -34,7 +34,7 @@ SELECT approx_percentile_cont(0.75) WITHIN GROUP (ORDER BY i) FROM approx_test; +-----------------------------------------------------------------------------------+ | approx_percentile_cont(Float64(0.75)) WITHIN GROUP [approx_test.i ASC NULLS LAST] | +-----------------------------------------------------------------------------------+ -| 749 | +| 749.375 | +-----------------------------------------------------------------------------------+ -- 95th percentile @@ -43,7 +43,7 @@ SELECT approx_percentile_cont(0.95) WITHIN GROUP (ORDER BY i) FROM approx_test; +-----------------------------------------------------------------------------------+ | approx_percentile_cont(Float64(0.95)) WITHIN GROUP [approx_test.i ASC NULLS LAST] | +-----------------------------------------------------------------------------------+ -| 949 | +| 949.6607142857143 | +-----------------------------------------------------------------------------------+ -- Test approx_percentile_cont DESC @@ -53,7 +53,7 @@ SELECT approx_percentile_cont(0.5) WITHIN GROUP (ORDER BY i DESC) FROM approx_te +------------------------------------------------------------------------------------+ | approx_percentile_cont(Float64(0.5)) WITHIN GROUP [approx_test.i DESC NULLS FIRST] | +------------------------------------------------------------------------------------+ -| 499 | +| 499.25 | +------------------------------------------------------------------------------------+ -- first quartile @@ -62,7 +62,7 @@ SELECT approx_percentile_cont(0.25) WITHIN GROUP (ORDER BY i DESC) FROM approx_t +-------------------------------------------------------------------------------------+ | approx_percentile_cont(Float64(0.25)) WITHIN GROUP [approx_test.i DESC NULLS FIRST] | +-------------------------------------------------------------------------------------+ -| 749 | +| 749.375 | +-------------------------------------------------------------------------------------+ -- third quartile @@ -71,7 +71,7 @@ SELECT approx_percentile_cont(0.75) WITHIN GROUP (ORDER BY i DESC) FROM approx_t +-------------------------------------------------------------------------------------+ | approx_percentile_cont(Float64(0.75)) WITHIN GROUP [approx_test.i DESC NULLS FIRST] | +-------------------------------------------------------------------------------------+ -| 249 | +| 249.575 | +-------------------------------------------------------------------------------------+ -- 95th percentile @@ -80,7 +80,7 @@ SELECT approx_percentile_cont(0.95) WITHIN GROUP (ORDER BY i DESC) FROM approx_t +-------------------------------------------------------------------------------------+ | approx_percentile_cont(Float64(0.95)) WITHIN GROUP [approx_test.i DESC NULLS FIRST] | +-------------------------------------------------------------------------------------+ -| 49 | +| 49.50000000000004 | +-------------------------------------------------------------------------------------+ -- Test with different data types @@ -127,9 +127,9 @@ FROM approx_groups GROUP BY grp ORDER BY grp; +-----+--------------------------------------------------------------------------------------+ | grp | approx_percentile_cont(Float64(0.5)) WITHIN GROUP [approx_groups.val ASC NULLS LAST] | +-----+--------------------------------------------------------------------------------------+ -| 0 | 148 | -| 1 | 149 | -| 2 | 150 | +| 0 | 148.125 | +| 1 | 149.125 | +| 2 | 150.125 | +-----+--------------------------------------------------------------------------------------+ -- Test with NULL values @@ -142,7 +142,7 @@ SELECT approx_percentile_cont(0.5) WITHIN GROUP (ORDER BY i) FROM approx_test; +----------------------------------------------------------------------------------+ | approx_percentile_cont(Float64(0.5)) WITHIN GROUP [approx_test.i ASC NULLS LAST] | +----------------------------------------------------------------------------------+ -| 499 | +| 499.25 | +----------------------------------------------------------------------------------+ -- Test edge cases @@ -152,7 +152,7 @@ SELECT approx_percentile_cont(0.0) WITHIN GROUP (ORDER BY i) FROM approx_test; +--------------------------------------------------------------------------------+ | approx_percentile_cont(Float64(0)) WITHIN GROUP [approx_test.i ASC NULLS LAST] | +--------------------------------------------------------------------------------+ -| 0 | +| 0.0 | +--------------------------------------------------------------------------------+ SELECT approx_percentile_cont(1.0) WITHIN GROUP (ORDER BY i DESC) FROM approx_test; @@ -160,7 +160,7 @@ SELECT approx_percentile_cont(1.0) WITHIN GROUP (ORDER BY i DESC) FROM approx_te +----------------------------------------------------------------------------------+ | approx_percentile_cont(Float64(1)) WITHIN GROUP [approx_test.i DESC NULLS FIRST] | +----------------------------------------------------------------------------------+ -| 0 | +| 0.0 | +----------------------------------------------------------------------------------+ -- should be close to max @@ -169,7 +169,7 @@ SELECT approx_percentile_cont(1.0) WITHIN GROUP (ORDER BY i) FROM approx_test; +--------------------------------------------------------------------------------+ | approx_percentile_cont(Float64(1)) WITHIN GROUP [approx_test.i ASC NULLS LAST] | +--------------------------------------------------------------------------------+ -| 999 | +| 999.0 | +--------------------------------------------------------------------------------+ SELECT approx_percentile_cont(0.0) WITHIN GROUP (ORDER BY i DESC) FROM approx_test; @@ -177,7 +177,7 @@ SELECT approx_percentile_cont(0.0) WITHIN GROUP (ORDER BY i DESC) FROM approx_te +----------------------------------------------------------------------------------+ | approx_percentile_cont(Float64(0)) WITHIN GROUP [approx_test.i DESC NULLS FIRST] | +----------------------------------------------------------------------------------+ -| 999 | +| 999.0 | +----------------------------------------------------------------------------------+ DROP TABLE approx_test; diff --git a/tests/cases/standalone/common/aggregate/approx_percentile_cont_with_weight.result b/tests/cases/standalone/common/aggregate/approx_percentile_cont_with_weight.result index 91c9b77519..5bdab7b060 100644 --- a/tests/cases/standalone/common/aggregate/approx_percentile_cont_with_weight.result +++ b/tests/cases/standalone/common/aggregate/approx_percentile_cont_with_weight.result @@ -16,7 +16,7 @@ SELECT approx_percentile_cont_with_weight(weight, 0.5) WITHIN GROUP (ORDER BY "v +---------------------------------------------------------------------------------------------------------------------+ | approx_percentile_cont_with_weight(weight_test.weight,Float64(0.5)) WITHIN GROUP [weight_test.value ASC NULLS LAST] | +---------------------------------------------------------------------------------------------------------------------+ -| 33 | +| 33.333333333333336 | +---------------------------------------------------------------------------------------------------------------------+ -- Test different percentiles @@ -25,7 +25,7 @@ SELECT approx_percentile_cont_with_weight(weight, 0.25) WITHIN GROUP (ORDER BY " +----------------------------------------------------------------------------------------------------------------------+ | approx_percentile_cont_with_weight(weight_test.weight,Float64(0.25)) WITHIN GROUP [weight_test.value ASC NULLS LAST] | +----------------------------------------------------------------------------------------------------------------------+ -| 23 | +| 23.75 | +----------------------------------------------------------------------------------------------------------------------+ SELECT approx_percentile_cont_with_weight(weight, 0.75) WITHIN GROUP (ORDER BY "value") FROM weight_test; @@ -33,7 +33,7 @@ SELECT approx_percentile_cont_with_weight(weight, 0.75) WITHIN GROUP (ORDER BY " +----------------------------------------------------------------------------------------------------------------------+ | approx_percentile_cont_with_weight(weight_test.weight,Float64(0.75)) WITHIN GROUP [weight_test.value ASC NULLS LAST] | +----------------------------------------------------------------------------------------------------------------------+ -| 40 | +| 40.625 | +----------------------------------------------------------------------------------------------------------------------+ -- Test with groups @@ -53,8 +53,8 @@ FROM weight_groups GROUP BY grp ORDER BY grp; +-----+-------------------------------------------------------------------------------------------------------------------------+ | grp | approx_percentile_cont_with_weight(weight_groups.weight,Float64(0.5)) WITHIN GROUP [weight_groups.value ASC NULLS LAST] | +-----+-------------------------------------------------------------------------------------------------------------------------+ -| 1 | 18 | -| 2 | 212 | +| 1 | 18.333333333333332 | +| 2 | 212.5 | +-----+-------------------------------------------------------------------------------------------------------------------------+ -- Test with double values and weights @@ -82,7 +82,7 @@ SELECT approx_percentile_cont_with_weight("weight", 0.0) WITHIN GROUP (ORDER BY +-------------------------------------------------------------------------------------------------------------------+ | approx_percentile_cont_with_weight(weight_test.weight,Float64(0)) WITHIN GROUP [weight_test.value ASC NULLS LAST] | +-------------------------------------------------------------------------------------------------------------------+ -| 10 | +| 10.0 | +-------------------------------------------------------------------------------------------------------------------+ -- max @@ -91,7 +91,7 @@ SELECT approx_percentile_cont_with_weight("weight", 1.0) WITHIN GROUP (ORDER BY +-------------------------------------------------------------------------------------------------------------------+ | approx_percentile_cont_with_weight(weight_test.weight,Float64(1)) WITHIN GROUP [weight_test.value ASC NULLS LAST] | +-------------------------------------------------------------------------------------------------------------------+ -| 50 | +| 50.0 | +-------------------------------------------------------------------------------------------------------------------+ -- Test with zero weights @@ -116,7 +116,7 @@ SELECT approx_percentile_cont_with_weight(weight, 0.5) WITHIN GROUP (ORDER BY "v +---------------------------------------------------------------------------------------------------------------------+ | approx_percentile_cont_with_weight(weight_test.weight,Float64(0.5)) WITHIN GROUP [weight_test.value ASC NULLS LAST] | +---------------------------------------------------------------------------------------------------------------------+ -| 33 | +| 33.333333333333336 | +---------------------------------------------------------------------------------------------------------------------+ -- Test empty result @@ -143,7 +143,7 @@ SELECT approx_percentile_cont_with_weight(weight, 0.5) WITHIN GROUP (ORDER BY "v +-------------------------------------------------------------------------------------------------------------------------+ | approx_percentile_cont_with_weight(single_weight.weight,Float64(0.5)) WITHIN GROUP [single_weight.value ASC NULLS LAST] | +-------------------------------------------------------------------------------------------------------------------------+ -| 42 | +| 42.0 | +-------------------------------------------------------------------------------------------------------------------------+ -- Test equal weights (should behave like regular percentile) @@ -164,7 +164,7 @@ FROM equal_weight; +-----------------------------------------------------------------------------------------------------------------------+---------------------------------------------------------------------------------------+ | approx_percentile_cont_with_weight(equal_weight.weight,Float64(0.5)) WITHIN GROUP [equal_weight.value ASC NULLS LAST] | approx_percentile_cont(Float64(0.5)) WITHIN GROUP [equal_weight.value ASC NULLS LAST] | +-----------------------------------------------------------------------------------------------------------------------+---------------------------------------------------------------------------------------+ -| 25 | 25 | +| 25.0 | 25.0 | +-----------------------------------------------------------------------------------------------------------------------+---------------------------------------------------------------------------------------+ -- cleanup diff --git a/tests/cases/standalone/common/aggregate/median.result b/tests/cases/standalone/common/aggregate/median.result index 98af8787cc..777952ff40 100644 --- a/tests/cases/standalone/common/aggregate/median.result +++ b/tests/cases/standalone/common/aggregate/median.result @@ -6,7 +6,7 @@ SELECT median(NULL), median(1); +--------------+------------------+ | median(NULL) | median(Int64(1)) | +--------------+------------------+ -| | 1 | +| | 1.0 | +--------------+------------------+ -- test with simple table @@ -26,7 +26,7 @@ SELECT median(r)::VARCHAR FROM quantile; +--------------------+ | median(quantile.r) | +--------------------+ -| 4 | +| 4.5 | +--------------------+ SELECT median(r::FLOAT)::VARCHAR FROM quantile; @@ -50,7 +50,7 @@ SELECT median(r::SMALLINT)::VARCHAR FROM quantile WHERE r < 100; +--------------------+ | median(quantile.r) | +--------------------+ -| 4 | +| 4.5 | +--------------------+ SELECT median(r::INTEGER)::VARCHAR FROM quantile; @@ -58,7 +58,7 @@ SELECT median(r::INTEGER)::VARCHAR FROM quantile; +--------------------+ | median(quantile.r) | +--------------------+ -| 4 | +| 4.5 | +--------------------+ SELECT median(r::BIGINT)::VARCHAR FROM quantile; @@ -66,7 +66,7 @@ SELECT median(r::BIGINT)::VARCHAR FROM quantile; +--------------------+ | median(quantile.r) | +--------------------+ -| 4 | +| 4.5 | +--------------------+ -- test with NULL values @@ -83,7 +83,7 @@ SELECT median(42) FROM quantile; +-------------------+ | median(Int64(42)) | +-------------------+ -| 42 | +| 42.0 | +-------------------+ -- test with grouped data @@ -103,8 +103,8 @@ SELECT grp, median(val) FROM median_groups GROUP BY grp ORDER BY grp; +-----+---------------------------+ | grp | median(median_groups.val) | +-----+---------------------------+ -| 1 | 3 | -| 2 | 30 | +| 1 | 3.0 | +| 2 | 30.0 | | 3 | | +-----+---------------------------+ diff --git a/tests/cases/standalone/common/aggregate/multi_regions.result b/tests/cases/standalone/common/aggregate/multi_regions.result index 8627b83c15..fa0bf92931 100644 --- a/tests/cases/standalone/common/aggregate/multi_regions.result +++ b/tests/cases/standalone/common/aggregate/multi_regions.result @@ -59,9 +59,9 @@ select sum(val) from t; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[sum(t.val)] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__sum_merge(__sum_state(t.val)) as sum(t.val)] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[sum(t.val)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__sum_merge(__sum_state(t.val)) as sum(t.val)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -94,9 +94,9 @@ select sum(val) from t group by idc; | stage | node | plan_| +-+-+-+ | 0_| 0_|_ProjectionExec: expr=[sum(t.val)@1 as sum(t.val)] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[idc@0 as idc], aggr=[sum(t.val)] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[idc@0 as idc], aggr=[__sum_merge(__sum_state(t.val)) as sum(t.val)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[idc@0 as idc], aggr=[sum(t.val)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[idc@0 as idc], aggr=[__sum_merge(__sum_state(t.val)) as sum(t.val)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| diff --git a/tests/cases/standalone/common/alter/add_col_default.result b/tests/cases/standalone/common/alter/add_col_default.result index 5a9baf7186..c54df938c5 100644 --- a/tests/cases/standalone/common/alter/add_col_default.result +++ b/tests/cases/standalone/common/alter/add_col_default.result @@ -39,8 +39,10 @@ SELECT * FROM test order by j; SELECT * FROM test where k != 3; -++ -++ ++---+---+---+ +| i | j | k | ++---+---+---+ ++---+---+---+ ALTER TABLE test ADD COLUMN host STRING DEFAULT '' PRIMARY KEY; @@ -48,13 +50,17 @@ Affected Rows: 0 SELECT * FROM test where host != ''; -++ -++ ++---+---+---+------+ +| i | j | k | host | ++---+---+---+------+ ++---+---+---+------+ SELECT * FROM test where host != '' AND i = 3; -++ -++ ++---+---+---+------+ +| i | j | k | host | ++---+---+---+------+ ++---+---+---+------+ DROP TABLE test; diff --git a/tests/cases/standalone/common/alter/change_col_type.result b/tests/cases/standalone/common/alter/change_col_type.result index 9cdfa1c399..348d3b2145 100644 --- a/tests/cases/standalone/common/alter/change_col_type.result +++ b/tests/cases/standalone/common/alter/change_col_type.result @@ -1,32 +1,32 @@ -CREATE TABLE test(`id` INTEGER PRIMARY KEY, i INTEGER NULL, j TIMESTAMP TIME INDEX, k BOOLEAN); +CREATE TABLE change_col_type_test(`id` INTEGER PRIMARY KEY, i INTEGER NULL, j TIMESTAMP TIME INDEX, k BOOLEAN); Affected Rows: 0 -INSERT INTO test VALUES (1, 1, 1, false), (2, 2, 2, true); +INSERT INTO change_col_type_test VALUES (1, 1, 1, false), (2, 2, 2, true); Affected Rows: 2 -ALTER TABLE test MODIFY COLUMN "I" STRING; +ALTER TABLE change_col_type_test MODIFY COLUMN "I" STRING; -Error: 4002(TableColumnNotFound), Column I not exists in table test +Error: 4002(TableColumnNotFound), Column I not exists in table change_col_type_test -ALTER TABLE test MODIFY COLUMN k DATE; +ALTER TABLE change_col_type_test MODIFY COLUMN k DATE; -Error: 1004(InvalidArguments), Invalid alter table(test) request: column 'k' cannot be cast automatically to type 'Date' +Error: 1004(InvalidArguments), Invalid alter table(change_col_type_test) request: column 'k' cannot be cast automatically to type 'Date' -ALTER TABLE test MODIFY COLUMN id STRING; +ALTER TABLE change_col_type_test MODIFY COLUMN id STRING; -Error: 1004(InvalidArguments), Invalid alter table(test) request: Not allowed to change primary key index column 'id' +Error: 1004(InvalidArguments), Invalid alter table(change_col_type_test) request: Not allowed to change primary key index column 'id' -ALTER TABLE test MODIFY COLUMN j STRING; +ALTER TABLE change_col_type_test MODIFY COLUMN j STRING; -Error: 1004(InvalidArguments), Invalid alter table(test) request: time index column 'j' only supports widening its timestamp unit, cannot change type from 'TimestampMillisecond' to 'String' +Error: 1004(InvalidArguments), Invalid alter table(change_col_type_test) request: time index column 'j' only supports widening its timestamp unit, cannot change type from 'TimestampMillisecond' to 'String' -ALTER TABLE test MODIFY COLUMN I STRING; +ALTER TABLE change_col_type_test MODIFY COLUMN I STRING; Affected Rows: 0 -SELECT * FROM test; +SELECT * FROM change_col_type_test; +----+---+-------------------------+-------+ | id | i | j | k | @@ -35,12 +35,12 @@ SELECT * FROM test; | 2 | 2 | 1970-01-01T00:00:00.002 | true | +----+---+-------------------------+-------+ -INSERT INTO test VALUES (3, "greptime", 3, true); +INSERT INTO change_col_type_test VALUES (3, "greptime", 3, true); Affected Rows: 1 -- SQLNESS SORT_RESULT 3 1 -SELECT * FROM test; +SELECT * FROM change_col_type_test; +----+----------+-------------------------+-------+ | id | i | j | k | @@ -50,7 +50,7 @@ SELECT * FROM test; | 3 | greptime | 1970-01-01T00:00:00.003 | true | +----+----------+-------------------------+-------+ -DESCRIBE test; +DESCRIBE change_col_type_test; +--------+----------------------+-----+------+---------+---------------+ | Column | Type | Key | Null | Default | Semantic Type | @@ -61,12 +61,12 @@ DESCRIBE test; | k | Boolean | | YES | | FIELD | +--------+----------------------+-----+------+---------+---------------+ -ALTER TABLE test MODIFY COLUMN I INTEGER; +ALTER TABLE change_col_type_test MODIFY COLUMN I INTEGER; Affected Rows: 0 -- SQLNESS SORT_RESULT 3 1 -SELECT * FROM test; +SELECT * FROM change_col_type_test; +----+---+-------------------------+-------+ | id | i | j | k | @@ -76,7 +76,7 @@ SELECT * FROM test; | 3 | | 1970-01-01T00:00:00.003 | true | +----+---+-------------------------+-------+ -DESCRIBE test; +DESCRIBE change_col_type_test; +--------+----------------------+-----+------+---------+---------------+ | Column | Type | Key | Null | Default | Semantic Type | @@ -87,7 +87,7 @@ DESCRIBE test; | k | Boolean | | YES | | FIELD | +--------+----------------------+-----+------+---------+---------------+ -DROP TABLE test; +DROP TABLE change_col_type_test; Affected Rows: 0 @@ -125,8 +125,10 @@ SELECT * FROM ts_widen ORDER BY ts; -- predicate on the widened time index reads old-unit SST data correctly SELECT * FROM ts_widen WHERE ts > '2024-01-01 00:00:01' ORDER BY ts; -++ -++ ++------+----+ +| host | ts | ++------+----+ ++------+----+ SELECT * FROM ts_widen WHERE ts < '2024-01-01 00:00:01.000300' ORDER BY ts; diff --git a/tests/cases/standalone/common/alter/change_col_type.sql b/tests/cases/standalone/common/alter/change_col_type.sql index 3356112cd0..bb16af8847 100644 --- a/tests/cases/standalone/common/alter/change_col_type.sql +++ b/tests/cases/standalone/common/alter/change_col_type.sql @@ -1,34 +1,34 @@ -CREATE TABLE test(`id` INTEGER PRIMARY KEY, i INTEGER NULL, j TIMESTAMP TIME INDEX, k BOOLEAN); +CREATE TABLE change_col_type_test(`id` INTEGER PRIMARY KEY, i INTEGER NULL, j TIMESTAMP TIME INDEX, k BOOLEAN); -INSERT INTO test VALUES (1, 1, 1, false), (2, 2, 2, true); +INSERT INTO change_col_type_test VALUES (1, 1, 1, false), (2, 2, 2, true); -ALTER TABLE test MODIFY COLUMN "I" STRING; +ALTER TABLE change_col_type_test MODIFY COLUMN "I" STRING; -ALTER TABLE test MODIFY COLUMN k DATE; +ALTER TABLE change_col_type_test MODIFY COLUMN k DATE; -ALTER TABLE test MODIFY COLUMN id STRING; +ALTER TABLE change_col_type_test MODIFY COLUMN id STRING; -ALTER TABLE test MODIFY COLUMN j STRING; +ALTER TABLE change_col_type_test MODIFY COLUMN j STRING; -ALTER TABLE test MODIFY COLUMN I STRING; +ALTER TABLE change_col_type_test MODIFY COLUMN I STRING; -SELECT * FROM test; +SELECT * FROM change_col_type_test; -INSERT INTO test VALUES (3, "greptime", 3, true); +INSERT INTO change_col_type_test VALUES (3, "greptime", 3, true); -- SQLNESS SORT_RESULT 3 1 -SELECT * FROM test; +SELECT * FROM change_col_type_test; -DESCRIBE test; +DESCRIBE change_col_type_test; -ALTER TABLE test MODIFY COLUMN I INTEGER; +ALTER TABLE change_col_type_test MODIFY COLUMN I INTEGER; -- SQLNESS SORT_RESULT 3 1 -SELECT * FROM test; +SELECT * FROM change_col_type_test; -DESCRIBE test; +DESCRIBE change_col_type_test; -DROP TABLE test; +DROP TABLE change_col_type_test; CREATE TABLE ts_widen (host STRING, ts TIMESTAMP TIME INDEX); INSERT INTO ts_widen VALUES ("a", "2024-01-01 00:00:01"); diff --git a/tests/cases/standalone/common/alter/rename_table.result b/tests/cases/standalone/common/alter/rename_table.result index ace030e54c..6bf01e6eec 100644 --- a/tests/cases/standalone/common/alter/rename_table.result +++ b/tests/cases/standalone/common/alter/rename_table.result @@ -100,8 +100,10 @@ DESC TABLE "fGhI"; SELECT * FROM "fGhI"; -++ -++ ++------+------+ +| CoLa | cOlB | ++------+------+ ++------+------+ ALTER TABLE "fGhI" RENAME JkLmN; diff --git a/tests/cases/standalone/common/basic.result b/tests/cases/standalone/common/basic.result index f1589a33ec..37faa575fc 100644 --- a/tests/cases/standalone/common/basic.result +++ b/tests/cases/standalone/common/basic.result @@ -122,8 +122,10 @@ Affected Rows: 0 SELECT * from t2; -++ -++ ++-----+----+-----+ +| job | ts | val | ++-----+----+-----+ ++-----+----+-----+ INSERT INTO t2 VALUES ('job1', 0, 0), ('job2', 1, 1); @@ -152,8 +154,10 @@ select * from foo order by host asc; SELECT * from t1 order by ts desc; -++ -++ ++------+----+-----+ +| host | ts | val | ++------+----+-----+ ++------+----+-----+ SELECT * from t2 order by ts desc; diff --git a/tests/cases/standalone/common/catalog/schema.result b/tests/cases/standalone/common/catalog/schema.result index 759fadc04e..f0ff4b1e40 100644 --- a/tests/cases/standalone/common/catalog/schema.result +++ b/tests/cases/standalone/common/catalog/schema.result @@ -99,8 +99,10 @@ Error: 4001(TableNotFound), Table not found: greptime.test_public_schema.hello SHOW TABLES FROM test_public_schema; -++ -++ ++------------------------------+ +| Tables_in_test_public_schema | ++------------------------------+ ++------------------------------+ SHOW TABLES FROM public; diff --git a/tests/cases/standalone/common/create/metric_engine_partition.result b/tests/cases/standalone/common/create/metric_engine_partition.result index ff3fd40e94..d45f6adf32 100644 --- a/tests/cases/standalone/common/create/metric_engine_partition.result +++ b/tests/cases/standalone/common/create/metric_engine_partition.result @@ -140,17 +140,17 @@ select host, count(*) from logical_table_2 GROUP BY host ORDER BY host; +-+-+ | logical_plan_| Sort: logical_table_2.host ASC NULLS LAST_| |_|_Projection: logical_table_2.host, count(Int64(1)) AS count(*)_| -|_|_Aggregate: groupBy=[[logical_table_2.host]], aggr=[[__count_merge(__count_state(logical_table_2.ts)) AS count(Int64(1))]] | +|_|_Aggregate: groupBy=[[logical_table_2.host]], aggr=[[__count_merge(__count_state(logical_table_2.ts)) AS count(Int64(1))]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[logical_table_2.host]], aggr=[[__count_state(logical_table_2.ts)]]_| |_|_TableScan: logical_table_2_| |_| ]]_| | physical_plan | SortPreservingMergeExec: [host@0 ASC NULLS LAST]_| -|_|_SortExec: expr=[host@0 ASC NULLS LAST], preserve_partitioning=[true]_| |_|_ProjectionExec: expr=[host@0 as host, count(Int64(1))@1 as count(*)]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[host@0 as host], aggr=[count(Int64(1))]_| +|_|_SortExec: expr=[host@0 ASC NULLS LAST], preserve_partitioning=[true]_| +|_|_AggregateExec: mode=FinalPartitioned, gby=[host@0 as host], aggr=[__count_merge(__count_state(logical_table_2.ts)) as count(Int64(1))] | |_|_RepartitionExec: REDACTED -|_|_AggregateExec: mode=Partial, gby=[host@0 as host], aggr=[count(Int64(1))]_| +|_|_AggregateExec: mode=Partial, gby=[host@0 as host], aggr=[__count_merge(__count_state(logical_table_2.ts)) as count(Int64(1))]_| |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -180,17 +180,17 @@ select ts, count(*) from logical_table_2 GROUP BY ts ORDER BY ts; +-+-+ | logical_plan_| Sort: logical_table_2.ts ASC NULLS LAST_| |_|_Projection: logical_table_2.ts, count(Int64(1)) AS count(*)_| -|_|_Aggregate: groupBy=[[logical_table_2.ts]], aggr=[[__count_merge(__count_state(logical_table_2.ts)) AS count(Int64(1))]] | +|_|_Aggregate: groupBy=[[logical_table_2.ts]], aggr=[[__count_merge(__count_state(logical_table_2.ts)) AS count(Int64(1))]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[logical_table_2.ts]], aggr=[[__count_state(logical_table_2.ts)]]_| |_|_TableScan: logical_table_2_| |_| ]]_| | physical_plan | SortPreservingMergeExec: [ts@0 ASC NULLS LAST]_| -|_|_SortExec: expr=[ts@0 ASC NULLS LAST], preserve_partitioning=[true]_| |_|_ProjectionExec: expr=[ts@0 as ts, count(Int64(1))@1 as count(*)]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(Int64(1))]_| +|_|_SortExec: expr=[ts@0 ASC NULLS LAST], preserve_partitioning=[true]_| +|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[__count_merge(__count_state(logical_table_2.ts)) as count(Int64(1))] | |_|_RepartitionExec: REDACTED -|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(Int64(1))]_| +|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[__count_merge(__count_state(logical_table_2.ts)) as count(Int64(1))]_| |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -276,17 +276,17 @@ select a, count(*) from logical_table_3 GROUP BY a ORDER BY a; +-+-+ | logical_plan_| Sort: logical_table_3.a ASC NULLS LAST_| |_|_Projection: logical_table_3.a, count(Int64(1)) AS count(*)_| -|_|_Aggregate: groupBy=[[logical_table_3.a]], aggr=[[__count_merge(__count_state(logical_table_3.ts)) AS count(Int64(1))]] | +|_|_Aggregate: groupBy=[[logical_table_3.a]], aggr=[[__count_merge(__count_state(logical_table_3.ts)) AS count(Int64(1))]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[logical_table_3.a]], aggr=[[__count_state(logical_table_3.ts)]]_| |_|_TableScan: logical_table_3_| |_| ]]_| | physical_plan | SortPreservingMergeExec: [a@0 ASC NULLS LAST]_| -|_|_SortExec: expr=[a@0 ASC NULLS LAST], preserve_partitioning=[true]_| |_|_ProjectionExec: expr=[a@0 as a, count(Int64(1))@1 as count(*)]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count(Int64(1))]_| +|_|_SortExec: expr=[a@0 ASC NULLS LAST], preserve_partitioning=[true]_| +|_|_AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[__count_merge(__count_state(logical_table_3.ts)) as count(Int64(1))] | |_|_RepartitionExec: REDACTED -|_|_AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count(Int64(1))]_| +|_|_AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[__count_merge(__count_state(logical_table_3.ts)) as count(Int64(1))]_| |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -362,9 +362,9 @@ EXPLAIN select count(*) from logical_table_4; |_|_TableScan: logical_table_4_| |_| ]]_| | physical_plan | ProjectionExec: expr=[count(Int64(1))@0 as count(*)]_| -|_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))]_| +|_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(logical_table_4.ts)) as count(Int64(1))]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(logical_table_4.ts)) as count(Int64(1))] | |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -394,17 +394,17 @@ select ts, count(*) from logical_table_4 GROUP BY ts ORDER BY ts; +-+-+ | logical_plan_| Sort: logical_table_4.ts ASC NULLS LAST_| |_|_Projection: logical_table_4.ts, count(Int64(1)) AS count(*)_| -|_|_Aggregate: groupBy=[[logical_table_4.ts]], aggr=[[__count_merge(__count_state(logical_table_4.ts)) AS count(Int64(1))]] | +|_|_Aggregate: groupBy=[[logical_table_4.ts]], aggr=[[__count_merge(__count_state(logical_table_4.ts)) AS count(Int64(1))]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[logical_table_4.ts]], aggr=[[__count_state(logical_table_4.ts)]]_| |_|_TableScan: logical_table_4_| |_| ]]_| | physical_plan | SortPreservingMergeExec: [ts@0 ASC NULLS LAST]_| -|_|_SortExec: expr=[ts@0 ASC NULLS LAST], preserve_partitioning=[true]_| |_|_ProjectionExec: expr=[ts@0 as ts, count(Int64(1))@1 as count(*)]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(Int64(1))]_| +|_|_SortExec: expr=[ts@0 ASC NULLS LAST], preserve_partitioning=[true]_| +|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[__count_merge(__count_state(logical_table_4.ts)) as count(Int64(1))] | |_|_RepartitionExec: REDACTED -|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(Int64(1))]_| +|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[__count_merge(__count_state(logical_table_4.ts)) as count(Int64(1))]_| |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| diff --git a/tests/cases/standalone/common/cte/cte.result b/tests/cases/standalone/common/cte/cte.result index 1ea77cae68..4d936f5aaa 100644 --- a/tests/cases/standalone/common/cte/cte.result +++ b/tests/cases/standalone/common/cte/cte.result @@ -146,7 +146,8 @@ select from cte where alias2 > 0; -Error: 3000(PlanQuery), Failed to plan SQL: No field named alias2. Valid fields are cte.a. +Error: 3000(PlanQuery), Failed to plan SQL: No field named alias2. +Valid fields are cte.a. drop table a; diff --git a/tests/cases/standalone/common/cte/cte_join_build_side.result b/tests/cases/standalone/common/cte/cte_join_build_side.result index 51ee4d98a4..d549b21c24 100644 --- a/tests/cases/standalone/common/cte/cte_join_build_side.result +++ b/tests/cases/standalone/common/cte/cte_join_build_side.result @@ -59,16 +59,15 @@ GROUP BY t."db" ORDER BY c DESC, t."db"; | logical_plan | Sort: c DESC NULLS FIRST, t.db ASC NULLS LAST | | | Projection: t.db, count(Int64(1)) AS count(*) AS c | | | Aggregate: groupBy=[[t.db]], aggr=[[count(Int64(1))]] | -| | Projection: t.db | -| | Inner Join: t.db = td.db | -| | Projection: t.db | -| | MergeScan [is_placeholder=false, remote_input=[ | +| | LeftSemi Join: t.db = td.db | +| | Projection: t.db | +| | MergeScan [is_placeholder=false, remote_input=[ | | | SubqueryAlias: t | | | Filter: cte_join_logs.db IS NOT NULL | | | TableScan: cte_join_logs, partial_filters=[cte_join_logs.db IS NOT NULL] | | | ]] | -| | Projection: td.db | -| | MergeScan [is_placeholder=false, remote_input=[ | +| | Projection: td.db | +| | MergeScan [is_placeholder=false, remote_input=[ | | | SubqueryAlias: td | | | SubqueryAlias: top_dbs | | | Filter: cte_join_logs.db IS NOT NULL | @@ -78,13 +77,13 @@ GROUP BY t."db" ORDER BY c DESC, t."db"; | | Aggregate: groupBy=[[cte_join_logs.db]], aggr=[[count(cte_join_logs.ts) AS count(Int64(1))]] | | | TableScan: cte_join_logs | | | ]] | -| physical_plan | SortPreservingMergeExec: [c@1 DESC, db@0 ASC NULLS LAST] | -| | SortExec: expr=[c@1 DESC, db@0 ASC NULLS LAST], preserve_REDACTED -| | ProjectionExec: expr=[db@0 as db, count(Int64(1))@1 as c] | +| physical_plan | ProjectionExec: expr=[db@0 as db, count(Int64(1))@1 as c] | +| | SortPreservingMergeExec: [count(Int64(1))@1 DESC, db@0 ASC NULLS LAST] | +| | SortExec: expr=[count(Int64(1))@1 DESC, db@0 ASC NULLS LAST], preserve_REDACTED | | AggregateExec: mode=FinalPartitioned, gby=[db@0 as db], aggr=[count(Int64(1))] | | | RepartitionExec: REDACTED | | AggregateExec: mode=Partial, gby=[db@0 as db], aggr=[count(Int64(1))] | -| | HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(db@0, db@0)], projection=[db@1] | +| | HashJoinExec: mode=CollectLeft, join_type=RightSemi, on=[(db@0, db@0)] | | | ProjectionExec: expr=[db@0 as db] | | | MergeScanExec: REDACTED | | RepartitionExec: REDACTED diff --git a/tests/cases/standalone/common/delete/delete.result b/tests/cases/standalone/common/delete/delete.result index 2dc2264630..c5d9b779d9 100644 --- a/tests/cases/standalone/common/delete/delete.result +++ b/tests/cases/standalone/common/delete/delete.result @@ -88,8 +88,10 @@ ADMIN flush_table('monitor'); SELECT ts, host, cpu, memory FROM monitor WHERE cpu = 66.6 ORDER BY ts; -++ -++ ++----+------+-----+--------+ +| ts | host | cpu | memory | ++----+------+-----+--------+ ++----+------+-----+--------+ SELECT ts, host, cpu, memory FROM monitor ORDER BY ts; @@ -108,8 +110,10 @@ Affected Rows: 2 SELECT ts, host, cpu, memory FROM monitor WHERE memory > 2048 ORDER BY ts; -++ -++ ++----+------+-----+--------+ +| ts | host | cpu | memory | ++----+------+-----+--------+ ++----+------+-----+--------+ SELECT ts, host, cpu, memory FROM monitor ORDER BY ts; diff --git a/tests/cases/standalone/common/error/incorrect_sql.result b/tests/cases/standalone/common/error/incorrect_sql.result index 5e8f34273b..1dde80265c 100644 --- a/tests/cases/standalone/common/error/incorrect_sql.result +++ b/tests/cases/standalone/common/error/incorrect_sql.result @@ -7,7 +7,8 @@ Error: 1001(Unsupported), SQL statement is not supported, keyword: SELEC -- Unrecognized column SELECT x FROM (SELECT 1 as y); -Error: 3000(PlanQuery), Failed to plan SQL: No field named x. Valid fields are y. +Error: 3000(PlanQuery), Failed to plan SQL: No field named x. +Valid fields are y. -- Unrecognized function SELECT FUNFUNFUN(); @@ -18,18 +19,19 @@ Did you mean 'range_fn'? -- Wrong aggregate parameters SELECT SUM(42, 84, 11, 'hello'); -Error: 3000(PlanQuery), Failed to plan SQL: Error during planning: Execution error: Function 'sum' failed to match any signature, errors: Error during planning: Function 'sum' expects 1 arguments but received 4,Error during planning: Function 'sum' expects 1 arguments but received 4,Error during planning: Function 'sum' expects 1 arguments but received 4,Error during planning: Function 'sum' expects 1 arguments but received 4,Error during planning: Function 'sum' expects 1 arguments but received 4 No function matches the given name and argument types 'sum(Int64, Int64, Int64, Utf8)'. You might need to add explicit type casts. +Error: 3000(PlanQuery), Failed to plan SQL: Error during planning: Execution error: Function 'sum' failed to match any signature, errors: Error during planning: Function 'sum' expects 1 arguments but received 4,Error during planning: Function 'sum' expects 1 arguments but received 4,Error during planning: Function 'sum' expects 1 arguments but received 4,Error during planning: Function 'sum' expects 1 arguments but received 4,Error during planning: Function 'sum' expects 1 arguments but received 4,Error during planning: Function 'sum' expects 1 arguments but received 4. No function matches the given name and argument types 'sum(Int64, Int64, Int64, Utf8)'. You might need to add explicit type casts. Candidate functions: - sum(Coercion(TypeSignatureClass::Decimal)) - sum(Coercion(TypeSignatureClass::Native(LogicalType(Native(UInt64), UInt64)), implicit_coercion=ImplicitCoercion([Native(LogicalType(Native(UInt8), UInt8)), Native(LogicalType(Native(UInt16), UInt16)), Native(LogicalType(Native(UInt32), UInt32))], default_type=UInt64)) - sum(Coercion(TypeSignatureClass::Native(LogicalType(Native(Int64), Int64)), implicit_coercion=ImplicitCoercion([Native(LogicalType(Native(Int8), Int8)), Native(LogicalType(Native(Int16), Int16)), Native(LogicalType(Native(Int32), Int32))], default_type=Int64)) - sum(Coercion(TypeSignatureClass::Native(LogicalType(Native(Float64), Float64)), implicit_coercion=ImplicitCoercion([Float], default_type=Float64)) - sum(Coercion(TypeSignatureClass::Duration)) + sum(Decimal) + sum(UInt64) + sum(Int64) + sum(Float64) + sum(Duration) + sum(Interval) -- No matching function signature SELECT cos(0, 1, 2, 3); -Error: 3000(PlanQuery), Failed to plan SQL: Error during planning: Failed to coerce arguments to satisfy a call to 'cos' function: coercion from Int64, Int64, Int64, Int64 to the signature Uniform(1, [Float64, Float32]) failed No function matches the given name and argument types 'cos(Int64, Int64, Int64, Int64)'. You might need to add explicit type casts. +Error: 3000(PlanQuery), Failed to plan SQL: Error during planning: Failed to coerce arguments to satisfy a call to 'cos' function: coercion from Int64, Int64, Int64, Int64 to the signature Uniform(1, [Float64, Float32]) failed. No function matches the given name and argument types 'cos(Int64, Int64, Int64, Int64)'. You might need to add explicit type casts. Candidate functions: cos(Float64/Float32) @@ -69,12 +71,14 @@ Affected Rows: 0 -- Non-existent column SELECT feathe FROM chickens; -Error: 3000(PlanQuery), Failed to plan SQL: No field named feathe. Valid fields are chickens.feather, chickens.beak, chickens.ts. +Error: 3000(PlanQuery), Failed to plan SQL: No field named feathe. Did you mean 'chickens.feather'? +Valid fields are chickens.feather, chickens.beak, chickens.ts. -- Non-existent column with multiple tables SELECT feathe FROM chickens, integers, strings; -Error: 3000(PlanQuery), Failed to plan SQL: No field named feathe. Valid fields are chickens.feather, chickens.beak, chickens.ts, integers.integ, integers.ts, strings.str, strings.ts. +Error: 3000(PlanQuery), Failed to plan SQL: No field named feathe. Did you mean 'chickens.feather'? +Valid fields are chickens.feather, chickens.beak, chickens.ts, integers.integ, integers.ts, strings.str, strings.ts. -- Ambiguous column reference SELECT ts FROM chickens, integers; diff --git a/tests/cases/standalone/common/filter/cast_preimage.result b/tests/cases/standalone/common/filter/cast_preimage.result index 3209a842e0..42e73ea103 100644 --- a/tests/cases/standalone/common/filter/cast_preimage.result +++ b/tests/cases/standalone/common/filter/cast_preimage.result @@ -345,17 +345,21 @@ Affected Rows: 0 INSERT INTO cast_preimage_ts_ms VALUES ('host1', 0, 1), ('host2', 5000, 2), - ('host3', 5001, 3); + ('host3', 5001, 3), + ('safe_neg', -9223372036854, 4), + ('safe_pos', 9223372036854, 5); -Affected Rows: 3 +Affected Rows: 5 --- Timestamp widening equality is exact at millisecond precision. +-- Widening TIMESTAMP(3) to TIMESTAMP(9) can overflow at extreme timestamp +-- values. We accept this full-domain semantic tradeoff to retain native +-- millisecond pruning for aligned normal-range equality and IN predicates. -- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED -- SQLNESS REPLACE (peers.*) REDACTED -- SQLNESS REPLACE (Hash.*) REDACTED -- SQLNESS REPLACE (RepartitionExec:.*) RepartitionExec: REDACTED EXPLAIN SELECT host, v FROM cast_preimage_ts_ms -WHERE CAST(ts AS TIMESTAMP(9)) = '1970-01-01 00:00:05'::TIMESTAMP(9) +WHERE CAST(ts AS TIMESTAMP(9)) = 5000000000::TIMESTAMP(9) ORDER BY host; +---------------+-------------------------------------------------------------------------------------------------------------------+ @@ -372,13 +376,91 @@ ORDER BY host; | | | +---------------+-------------------------------------------------------------------------------------------------------------------+ --- Non-exact nanosecond literal should remain semantically correct. -SELECT host, v FROM cast_preimage_ts_ms -WHERE CAST(ts AS TIMESTAMP(9)) = '1970-01-01 00:00:05.000000001'::TIMESTAMP(9) +-- The aligned IN-list must likewise become bare millisecond scan filters. +-- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED +-- SQLNESS REPLACE (peers.*) REDACTED +-- SQLNESS REPLACE (Hash.*) REDACTED +-- SQLNESS REPLACE (RepartitionExec:.*) RepartitionExec: REDACTED +EXPLAIN SELECT host, v FROM cast_preimage_ts_ms +WHERE CAST(ts AS TIMESTAMP(9)) IN (0::TIMESTAMP(9), 5000000000::TIMESTAMP(9)) ORDER BY host; -++ -++ ++---------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ +| plan_type | plan | ++---------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ +| logical_plan | MergeScan [is_placeholder=false, remote_input=[ | +| | Sort: cast_preimage_ts_ms.host ASC NULLS LAST | +| | Projection: cast_preimage_ts_ms.host, cast_preimage_ts_ms.v | +| | Filter: cast_preimage_ts_ms.ts = TimestampMillisecond(0, None) OR cast_preimage_ts_ms.ts = TimestampMillisecond(5000, None) | +| | TableScan: cast_preimage_ts_ms, partial_filters=[cast_preimage_ts_ms.ts = TimestampMillisecond(0, None) OR cast_preimage_ts_ms.ts = TimestampMillisecond(5000, None)] | +| | ]] | +| physical_plan | CooperativeExec | +| | MergeScanExec: REDACTED +| | | ++---------------+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ + +SELECT host, v FROM cast_preimage_ts_ms +WHERE CAST(ts AS TIMESTAMP(9)) = 5000000000::TIMESTAMP(9) +ORDER BY host; + ++-------+---+ +| host | v | ++-------+---+ +| host2 | 2 | ++-------+---+ + +SELECT host, v FROM cast_preimage_ts_ms +WHERE CAST(ts AS TIMESTAMP(9)) = 5000000001::TIMESTAMP(9) +ORDER BY host; + ++------+---+ +| host | v | ++------+---+ ++------+---+ + +-- The safe millisecond values widen to -9223372036854000000 and +-- 9223372036854000000 nanoseconds without rendering extreme dates. +SELECT v, CAST(CAST(ts AS TIMESTAMP(9)) AS BIGINT) AS ts_ns +FROM cast_preimage_ts_ms +WHERE v IN (4, 5) +ORDER BY v; + ++---+----------------------+ +| v | ts_ns | ++---+----------------------+ +| 4 | -9223372036854000000 | +| 5 | 9223372036854000000 | ++---+----------------------+ + +INSERT INTO cast_preimage_ts_ms VALUES ('overflow_pos', 9223372036855, 7); + +Affected Rows: 1 + +-- An ordinary projection retains its overflow error. Under the accepted +-- pruning policy, aligned equality excludes the overflow row instead. +SELECT v, CAST(CAST(ts AS TIMESTAMP(9)) AS BIGINT) AS ts_ns +FROM cast_preimage_ts_ms WHERE host = 'overflow_pos'; + +Error: 3001(EngineExecuteQuery), Execution error: Cannot cast Timestamp(ms) value 9223372036855 to Timestamp(ns): converted value exceeds the representable i64 range + +SELECT v FROM cast_preimage_ts_ms +WHERE host = 'overflow_pos' AND CAST(ts AS TIMESTAMP(9)) = 5000000000::TIMESTAMP(9); + ++---+ +| v | ++---+ ++---+ + +-- Direct safe cast returns NULL, avoiding the existing SQL timestamp-precision +-- lowering limitation for TRY_CAST. +SELECT v, CAST(arrow_try_cast(ts, 'Timestamp(Nanosecond, None)') AS BIGINT) AS ts_ns +FROM cast_preimage_ts_ms WHERE host = 'overflow_pos'; + ++---+-------+ +| v | ts_ns | ++---+-------+ +| 7 | | ++---+-------+ DROP TABLE cast_preimage_ts; diff --git a/tests/cases/standalone/common/filter/cast_preimage.sql b/tests/cases/standalone/common/filter/cast_preimage.sql index 7ee7e8198f..988c71fa69 100644 --- a/tests/cases/standalone/common/filter/cast_preimage.sql +++ b/tests/cases/standalone/common/filter/cast_preimage.sql @@ -152,22 +152,60 @@ CREATE TABLE cast_preimage_ts_ms ( INSERT INTO cast_preimage_ts_ms VALUES ('host1', 0, 1), ('host2', 5000, 2), - ('host3', 5001, 3); + ('host3', 5001, 3), + ('safe_neg', -9223372036854, 4), + ('safe_pos', 9223372036854, 5); --- Timestamp widening equality is exact at millisecond precision. +-- Widening TIMESTAMP(3) to TIMESTAMP(9) can overflow at extreme timestamp +-- values. We accept this full-domain semantic tradeoff to retain native +-- millisecond pruning for aligned normal-range equality and IN predicates. -- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED -- SQLNESS REPLACE (peers.*) REDACTED -- SQLNESS REPLACE (Hash.*) REDACTED -- SQLNESS REPLACE (RepartitionExec:.*) RepartitionExec: REDACTED EXPLAIN SELECT host, v FROM cast_preimage_ts_ms -WHERE CAST(ts AS TIMESTAMP(9)) = '1970-01-01 00:00:05'::TIMESTAMP(9) +WHERE CAST(ts AS TIMESTAMP(9)) = 5000000000::TIMESTAMP(9) ORDER BY host; --- Non-exact nanosecond literal should remain semantically correct. -SELECT host, v FROM cast_preimage_ts_ms -WHERE CAST(ts AS TIMESTAMP(9)) = '1970-01-01 00:00:05.000000001'::TIMESTAMP(9) +-- The aligned IN-list must likewise become bare millisecond scan filters. +-- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED +-- SQLNESS REPLACE (peers.*) REDACTED +-- SQLNESS REPLACE (Hash.*) REDACTED +-- SQLNESS REPLACE (RepartitionExec:.*) RepartitionExec: REDACTED +EXPLAIN SELECT host, v FROM cast_preimage_ts_ms +WHERE CAST(ts AS TIMESTAMP(9)) IN (0::TIMESTAMP(9), 5000000000::TIMESTAMP(9)) ORDER BY host; +SELECT host, v FROM cast_preimage_ts_ms +WHERE CAST(ts AS TIMESTAMP(9)) = 5000000000::TIMESTAMP(9) +ORDER BY host; + +SELECT host, v FROM cast_preimage_ts_ms +WHERE CAST(ts AS TIMESTAMP(9)) = 5000000001::TIMESTAMP(9) +ORDER BY host; + +-- The safe millisecond values widen to -9223372036854000000 and +-- 9223372036854000000 nanoseconds without rendering extreme dates. +SELECT v, CAST(CAST(ts AS TIMESTAMP(9)) AS BIGINT) AS ts_ns +FROM cast_preimage_ts_ms +WHERE v IN (4, 5) +ORDER BY v; + +INSERT INTO cast_preimage_ts_ms VALUES ('overflow_pos', 9223372036855, 7); + +-- An ordinary projection retains its overflow error. Under the accepted +-- pruning policy, aligned equality excludes the overflow row instead. +SELECT v, CAST(CAST(ts AS TIMESTAMP(9)) AS BIGINT) AS ts_ns +FROM cast_preimage_ts_ms WHERE host = 'overflow_pos'; + +SELECT v FROM cast_preimage_ts_ms +WHERE host = 'overflow_pos' AND CAST(ts AS TIMESTAMP(9)) = 5000000000::TIMESTAMP(9); + +-- Direct safe cast returns NULL, avoiding the existing SQL timestamp-precision +-- lowering limitation for TRY_CAST. +SELECT v, CAST(arrow_try_cast(ts, 'Timestamp(Nanosecond, None)') AS BIGINT) AS ts_ns +FROM cast_preimage_ts_ms WHERE host = 'overflow_pos'; + DROP TABLE cast_preimage_ts; DROP TABLE cast_preimage_int; diff --git a/tests/cases/standalone/common/filter/constant_comparisons.result b/tests/cases/standalone/common/filter/constant_comparisons.result index 9dba691e91..6dbc81a86d 100644 --- a/tests/cases/standalone/common/filter/constant_comparisons.result +++ b/tests/cases/standalone/common/filter/constant_comparisons.result @@ -19,8 +19,10 @@ SELECT * FROM integers WHERE 2=2; SELECT * FROM integers WHERE 2=3; -++ -++ ++---+---+----+ +| a | b | ts | ++---+---+----+ ++---+---+----+ SELECT * FROM integers WHERE 2<>3; @@ -32,8 +34,10 @@ SELECT * FROM integers WHERE 2<>3; SELECT * FROM integers WHERE 2<>2; -++ -++ ++---+---+----+ +| a | b | ts | ++---+---+----+ ++---+---+----+ SELECT * FROM integers WHERE 2>1; @@ -45,8 +49,10 @@ SELECT * FROM integers WHERE 2>1; SELECT * FROM integers WHERE 2>2; -++ -++ ++---+---+----+ +| a | b | ts | ++---+---+----+ ++---+---+----+ SELECT * FROM integers WHERE 2>=2; @@ -58,8 +64,10 @@ SELECT * FROM integers WHERE 2>=2; SELECT * FROM integers WHERE 2>=3; -++ -++ ++---+---+----+ +| a | b | ts | ++---+---+----+ ++---+---+----+ SELECT * FROM integers WHERE 2<3; @@ -71,8 +79,10 @@ SELECT * FROM integers WHERE 2<3; SELECT * FROM integers WHERE 2<2; -++ -++ ++---+---+----+ +| a | b | ts | ++---+---+----+ ++---+---+----+ SELECT * FROM integers WHERE 2<=2; @@ -84,8 +94,10 @@ SELECT * FROM integers WHERE 2<=2; SELECT * FROM integers WHERE 2<=1; -++ -++ ++---+---+----+ +| a | b | ts | ++---+---+----+ ++---+---+----+ -- NULL comparisons SELECT a=NULL FROM integers; @@ -115,8 +127,10 @@ SELECT * FROM integers WHERE 2 IN (2, 3, 4, 5); SELECT * FROM integers WHERE 2 IN (1, 3, 4, 5); -++ -++ ++---+---+----+ +| a | b | ts | ++---+---+----+ ++---+---+----+ -- Clean up DROP TABLE integers; diff --git a/tests/cases/standalone/common/filter/hash_join_dyn_filter.result b/tests/cases/standalone/common/filter/hash_join_dyn_filter.result index 05c313de3c..b1402bd4d4 100644 --- a/tests/cases/standalone/common/filter/hash_join_dyn_filter.result +++ b/tests/cases/standalone/common/filter/hash_join_dyn_filter.result @@ -49,7 +49,7 @@ WHERE c.tier = 'gold'; | | MergeScan [is_placeholder=false, remote_input=[ | | | SubqueryAlias: c | | | Filter: customers.customer_id IS NOT NULL AND customers.tier = Utf8("gold") | -| | TableScan: customers, partial_filters=[customers.tier = Utf8("gold"), customers.customer_id IS NOT NULL] | +| | TableScan: customers, partial_filters=[customers.customer_id IS NOT NULL, customers.tier = Utf8("gold")] | | | ]] | | physical_plan | HashJoinExec: mode=Partitioned, join_type=Inner, on=[(customer_id@1, customer_id@0)], projection=[id@0, amount@2, name@4, tier@5] | | | RepartitionExec: REDACTED @@ -100,7 +100,7 @@ WHERE c.tier = 'gold'; |_|_|_| | 1_| 0_|_FilterExec: customer_id@0 IS NOT NULL AND tier@2 = gold metrics=REDACTED_| |_|_|_CooperativeExec metrics=REDACTED_| -|_|_|_SeqScan: region=REDACTED, {"partition_count":REDACTED, "projection": ["customer_id", "name", "tier", "ts"], "filters": ["tier = Utf8(\"gold\")", "customer_id IS NOT NULL"], "dyn_filters": ["DynamicFilter [ REDACTED ]"], "flat_format":REDACTED, "metrics_per_partition": REDACTED metrics=REDACTED_| +|_|_|_SeqScan: region=REDACTED, {"partition_count":REDACTED, "projection": ["customer_id", "name", "tier", "ts"], "filters": ["customer_id IS NOT NULL", "tier = Utf8(\"gold\")"], "dyn_filters": ["DynamicFilter [ REDACTED ]"], "flat_format":REDACTED, "metrics_per_partition": REDACTED metrics=REDACTED_| |_|_|_| |_|_| Total rows: REDACTED_| +-+-+-+ @@ -177,7 +177,7 @@ WHERE c.tier IN ('gold', 'silver'); | | MergeScan [is_placeholder=false, remote_input=[ | | | SubqueryAlias: c | | | Filter: customers.customer_id IS NOT NULL AND (customers.tier = Utf8("gold") OR customers.tier = Utf8("silver")) | -| | TableScan: customers, partial_filters=[customers.tier = Utf8("gold") OR customers.tier = Utf8("silver"), customers.customer_id IS NOT NULL] | +| | TableScan: customers, partial_filters=[customers.customer_id IS NOT NULL, customers.tier = Utf8("gold") OR customers.tier = Utf8("silver")] | | | ]] | | physical_plan | HashJoinExec: mode=Partitioned, join_type=Inner, on=[(cid@1, customer_id@0)], projection=[id@0, amount@2, name@4, tier@5] | | | RepartitionExec: REDACTED @@ -229,7 +229,7 @@ WHERE c.tier IN ('gold', 'silver'); |_|_|_| | 1_| 0_|_FilterExec: customer_id@0 IS NOT NULL AND (tier@2 = gold OR tier@2 = silver) metrics=REDACTED_| |_|_|_CooperativeExec metrics=REDACTED_| -|_|_|_SeqScan: region=REDACTED, {"partition_count":REDACTED, "projection": ["customer_id", "name", "tier", "ts"], "filters": ["tier = Utf8(\"gold\") OR tier = Utf8(\"silver\")", "customer_id IS NOT NULL"], "dyn_filters": ["DynamicFilter [ REDACTED ]"], "flat_format":REDACTED, "metrics_per_partition": REDACTED metrics=REDACTED_| +|_|_|_SeqScan: region=REDACTED, {"partition_count":REDACTED, "projection": ["customer_id", "name", "tier", "ts"], "filters": ["customer_id IS NOT NULL", "tier = Utf8(\"gold\") OR tier = Utf8(\"silver\")"], "dyn_filters": ["DynamicFilter [ REDACTED ]"], "flat_format":REDACTED, "metrics_per_partition": REDACTED metrics=REDACTED_| |_|_|_| |_|_| Total rows: REDACTED_| +-+-+-+ diff --git a/tests/cases/standalone/common/filter/hash_join_topk_dyn_filter.result b/tests/cases/standalone/common/filter/hash_join_topk_dyn_filter.result index 5798ddd07c..0ba0f3c4e0 100644 --- a/tests/cases/standalone/common/filter/hash_join_topk_dyn_filter.result +++ b/tests/cases/standalone/common/filter/hash_join_topk_dyn_filter.result @@ -65,7 +65,7 @@ WHERE c.tier IN ('gold', 'bronze'); | | MergeScan [is_placeholder=false, remote_input=[ | | | SubqueryAlias: c | | | Filter: customers.customer_id IS NOT NULL AND (customers.tier = Utf8("gold") OR customers.tier = Utf8("bronze")) | -| | TableScan: customers, partial_filters=[customers.tier = Utf8("gold") OR customers.tier = Utf8("bronze"), customers.customer_id IS NOT NULL] | +| | TableScan: customers, partial_filters=[customers.customer_id IS NOT NULL, customers.tier = Utf8("gold") OR customers.tier = Utf8("bronze")] | | | ]] | | physical_plan | HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(customer_id@1, customer_id@0)], projection=[id@0, amount@2, name@4, tier@5] | | | ProjectionExec: expr=[id@0 as id, customer_id@1 as customer_id, amount@2 as amount] | @@ -123,7 +123,7 @@ WHERE c.tier IN ('gold', 'bronze'); |_|_|_| | 1_| 0_|_FilterExec: customer_id@0 IS NOT NULL AND (tier@2 = gold OR tier@2 = bronze) metrics=REDACTED_| |_|_|_CooperativeExec metrics=REDACTED_| -|_|_|_SeqScan: region=REDACTED, {"partition_count":REDACTED, "projection": ["customer_id", "name", "tier", "ts"], "filters": ["tier = Utf8(\"gold\") OR tier = Utf8(\"bronze\")", "customer_id IS NOT NULL"], "dyn_filters": ["DynamicFilter [ REDACTED ]"], "flat_format":REDACTED, "metrics_per_partition": REDACTED metrics=REDACTED_| +|_|_|_SeqScan: region=REDACTED, {"partition_count":REDACTED, "projection": ["customer_id", "name", "tier", "ts"], "filters": ["customer_id IS NOT NULL", "tier = Utf8(\"gold\") OR tier = Utf8(\"bronze\")"], "dyn_filters": ["DynamicFilter [ REDACTED ]"], "flat_format":REDACTED, "metrics_per_partition": REDACTED metrics=REDACTED_| |_|_|_| |_|_| Total rows: REDACTED_| +-+-+-+ @@ -204,36 +204,35 @@ FROM ( ORDER BY amount DESC LIMIT 4; -+---------------+--------------------------------------------------------------------------------------------------------------------------------------------------------+ -| plan_type | plan | -+---------------+--------------------------------------------------------------------------------------------------------------------------------------------------------+ -| logical_plan | Sort: o.amount DESC NULLS FIRST, fetch=4 | -| | Projection: o.id, o.customer_id, c.name, c.tier, o.amount | -| | Inner Join: o.customer_id = c.customer_id | -| | Projection: o.id, o.customer_id, o.amount | -| | MergeScan [is_placeholder=false, remote_input=[ | -| | SubqueryAlias: o | -| | Filter: orders.customer_id IS NOT NULL | -| | TableScan: orders, partial_filters=[orders.customer_id IS NOT NULL] | -| | ]] | -| | Projection: c.customer_id, c.name, c.tier | -| | MergeScan [is_placeholder=false, remote_input=[ | -| | SubqueryAlias: c | -| | Filter: customers.customer_id IS NOT NULL AND (customers.tier = Utf8("gold") OR customers.tier = Utf8("silver")) | -| | TableScan: customers, partial_filters=[customers.tier = Utf8("gold") OR customers.tier = Utf8("silver"), customers.customer_id IS NOT NULL] | -| | ]] | -| physical_plan | SortPreservingMergeExec: [amount@4 DESC], fetch=4 | -| | SortExec: TopK(fetch=4), expr=[amount@4 DESC], preserve_partitioning=[true] | -| | ProjectionExec: expr=[id@0 as id, customer_id@1 as customer_id, name@3 as name, tier@4 as tier, amount@2 as amount] | -| | HashJoinExec: mode=Partitioned, join_type=Inner, on=[(customer_id@1, customer_id@0)], projection=[id@0, customer_id@1, amount@2, name@4, tier@5] | -| | RepartitionExec: partitioning=REDACTED -| | ProjectionExec: expr=[id@0 as id, customer_id@1 as customer_id, amount@2 as amount] | -| | MergeScanExec: REDACTED -| | RepartitionExec: partitioning=REDACTED -| | ProjectionExec: expr=[customer_id@0 as customer_id, name@1 as name, tier@2 as tier] | -| | MergeScanExec: REDACTED -| | | -+---------------+--------------------------------------------------------------------------------------------------------------------------------------------------------+ ++---------------+------------------------------------------------------------------------------------------------------------------------------------------------------+ +| plan_type | plan | ++---------------+------------------------------------------------------------------------------------------------------------------------------------------------------+ +| logical_plan | Sort: o.amount DESC NULLS FIRST, fetch=4 | +| | Projection: o.id, o.customer_id, c.name, c.tier, o.amount | +| | Inner Join: o.customer_id = c.customer_id | +| | Projection: o.id, o.customer_id, o.amount | +| | MergeScan [is_placeholder=false, remote_input=[ | +| | SubqueryAlias: o | +| | Filter: orders.customer_id IS NOT NULL | +| | TableScan: orders, partial_filters=[orders.customer_id IS NOT NULL] | +| | ]] | +| | Projection: c.customer_id, c.name, c.tier | +| | MergeScan [is_placeholder=false, remote_input=[ | +| | SubqueryAlias: c | +| | Filter: customers.customer_id IS NOT NULL AND (customers.tier = Utf8("gold") OR customers.tier = Utf8("silver")) | +| | TableScan: customers, partial_filters=[customers.customer_id IS NOT NULL, customers.tier = Utf8("gold") OR customers.tier = Utf8("silver")] | +| | ]] | +| physical_plan | SortPreservingMergeExec: [amount@4 DESC], fetch=4 | +| | SortExec: TopK(fetch=4), expr=[amount@4 DESC], preserve_partitioning=[true] | +| | HashJoinExec: mode=Partitioned, join_type=Inner, on=[(customer_id@1, customer_id@0)], projection=[id@0, customer_id@1, name@4, tier@5, amount@2] | +| | RepartitionExec: partitioning=REDACTED +| | ProjectionExec: expr=[id@0 as id, customer_id@1 as customer_id, amount@2 as amount] | +| | MergeScanExec: REDACTED +| | RepartitionExec: partitioning=REDACTED +| | ProjectionExec: expr=[customer_id@0 as customer_id, name@1 as name, tier@2 as tier] | +| | MergeScanExec: REDACTED +| | | ++---------------+------------------------------------------------------------------------------------------------------------------------------------------------------+ -- SQLNESS REPLACE ("metrics_per_partition":\s*.*metrics=) "metrics_per_partition": REDACTED metrics= -- SQLNESS REPLACE (metrics=\{.*\}) metrics=REDACTED @@ -267,8 +266,7 @@ LIMIT 4; +-+-+-+ | 0_| 0_|_SortPreservingMergeExec: [amount@4 DESC], fetch=4 metrics=REDACTED_| |_|_|_SortExec: TopK(fetch=4), expr=[amount@4 DESC], preserve_partitioning=[true] metrics=REDACTED_| -|_|_|_ProjectionExec: expr=[id@0 as id, customer_id@1 as customer_id, name@3 as name, tier@4 as tier, amount@2 as amount] metrics=REDACTED_| -|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(customer_id@1, customer_id@0)], projection=[id@0, customer_id@1, amount@2, name@4, tier@5] metrics=REDACTED_| +|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(customer_id@1, customer_id@0)], projection=[id@0, customer_id@1, name@4, tier@5, amount@2] metrics=REDACTED_| |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_ProjectionExec: expr=[id@0 as id, customer_id@1 as customer_id, amount@2 as amount] metrics=REDACTED_| |_|_|_MergeScanExec: REDACTED @@ -282,7 +280,7 @@ LIMIT 4; |_|_|_| | 1_| 0_|_FilterExec: customer_id@0 IS NOT NULL AND (tier@2 = gold OR tier@2 = silver) metrics=REDACTED_| |_|_|_CooperativeExec metrics=REDACTED_| -|_|_|_SeqScan: region=REDACTED, {"partition_count":REDACTED, "projection": ["customer_id", "name", "tier", "ts"], "filters": ["tier = Utf8(\"gold\") OR tier = Utf8(\"silver\")", "customer_id IS NOT NULL"], "dyn_filters": ["DynamicFilter [ REDACTED ]"], "flat_format":REDACTED, "metrics_per_partition": REDACTED metrics=REDACTED_| +|_|_|_SeqScan: region=REDACTED, {"partition_count":REDACTED, "projection": ["customer_id", "name", "tier", "ts"], "filters": ["customer_id IS NOT NULL", "tier = Utf8(\"gold\") OR tier = Utf8(\"silver\")"], "dyn_filters": ["DynamicFilter [ REDACTED ]"], "flat_format":REDACTED, "metrics_per_partition": REDACTED metrics=REDACTED_| |_|_|_| |_|_| Total rows: REDACTED_| +-+-+-+ @@ -380,20 +378,20 @@ WHERE c.tier IN ('gold', 'silver') | | Projection: o.id, o.customer_id, o.product_id, o.amount | | | MergeScan [is_placeholder=false, remote_input=[ | | | SubqueryAlias: o | -| | Filter: orders.product_id IS NOT NULL AND orders.customer_id IS NOT NULL | -| | TableScan: orders, partial_filters=[orders.product_id IS NOT NULL, orders.customer_id IS NOT NULL] | +| | Filter: orders.customer_id IS NOT NULL AND orders.product_id IS NOT NULL | +| | TableScan: orders, partial_filters=[orders.customer_id IS NOT NULL, orders.product_id IS NOT NULL] | | | ]] | | | Projection: c.customer_id, c.name, c.tier | | | MergeScan [is_placeholder=false, remote_input=[ | | | SubqueryAlias: c | | | Filter: customers.customer_id IS NOT NULL AND (customers.tier = Utf8("gold") OR customers.tier = Utf8("silver")) | -| | TableScan: customers, partial_filters=[customers.tier = Utf8("gold") OR customers.tier = Utf8("silver"), customers.customer_id IS NOT NULL] | +| | TableScan: customers, partial_filters=[customers.customer_id IS NOT NULL, customers.tier = Utf8("gold") OR customers.tier = Utf8("silver")] | | | ]] | | | Projection: p.product_id, p.name, p.category | | | MergeScan [is_placeholder=false, remote_input=[ | | | SubqueryAlias: p | | | Filter: products.product_id IS NOT NULL AND products.category = Utf8("electronics") | -| | TableScan: products, partial_filters=[products.category = Utf8("electronics"), products.product_id IS NOT NULL] | +| | TableScan: products, partial_filters=[products.product_id IS NOT NULL, products.category = Utf8("electronics")] | | | ]] | | physical_plan | ProjectionExec: expr=[id@0 as id, amount@1 as amount, name@2 as name, tier@3 as tier, name@4 as product_name, category@5 as category] | | | HashJoinExec: mode=Partitioned, join_type=Inner, on=[(product_id@1, product_id@0)], projection=[id@0, amount@2, name@3, tier@4, name@6, category@7] | @@ -452,17 +450,17 @@ WHERE c.tier IN ('gold', 'silver') |_|_|_ProjectionExec: expr=[product_id@0 as product_id, name@1 as name, category@2 as category] metrics=REDACTED_| |_|_|_MergeScanExec: REDACTED |_|_|_| -| 1_| 0_|_FilterExec: product_id@2 IS NOT NULL AND customer_id@1 IS NOT NULL metrics=REDACTED_| +| 1_| 0_|_FilterExec: customer_id@1 IS NOT NULL AND product_id@2 IS NOT NULL metrics=REDACTED_| |_|_|_CooperativeExec metrics=REDACTED_| -|_|_|_SeqScan: region=REDACTED, {"partition_count":REDACTED, "projection": ["id", "customer_id", "product_id", "amount", "ts"], "filters": ["product_id IS NOT NULL", "customer_id IS NOT NULL"], "flat_format":REDACTED, "metrics_per_partition": REDACTED metrics=REDACTED_| +|_|_|_SeqScan: region=REDACTED, {"partition_count":REDACTED, "projection": ["id", "customer_id", "product_id", "amount", "ts"], "filters": ["customer_id IS NOT NULL", "product_id IS NOT NULL"], "flat_format":REDACTED, "metrics_per_partition": REDACTED metrics=REDACTED_| |_|_|_| | 1_| 0_|_FilterExec: customer_id@0 IS NOT NULL AND (tier@2 = gold OR tier@2 = silver) metrics=REDACTED_| |_|_|_CooperativeExec metrics=REDACTED_| -|_|_|_SeqScan: region=REDACTED, {"partition_count":REDACTED, "projection": ["customer_id", "name", "tier", "ts"], "filters": ["tier = Utf8(\"gold\") OR tier = Utf8(\"silver\")", "customer_id IS NOT NULL"], "dyn_filters": ["DynamicFilter [ REDACTED ]"], "flat_format":REDACTED, "metrics_per_partition": REDACTED metrics=REDACTED_| +|_|_|_SeqScan: region=REDACTED, {"partition_count":REDACTED, "projection": ["customer_id", "name", "tier", "ts"], "filters": ["customer_id IS NOT NULL", "tier = Utf8(\"gold\") OR tier = Utf8(\"silver\")"], "dyn_filters": ["DynamicFilter [ REDACTED ]"], "flat_format":REDACTED, "metrics_per_partition": REDACTED metrics=REDACTED_| |_|_|_| | 1_| 0_|_FilterExec: product_id@0 IS NOT NULL AND category@2 = electronics metrics=REDACTED_| |_|_|_CooperativeExec metrics=REDACTED_| -|_|_|_SeqScan: region=REDACTED, {"partition_count":REDACTED, "projection": ["product_id", "name", "category", "ts"], "filters": ["category = Utf8(\"electronics\")", "product_id IS NOT NULL"], "dyn_filters": ["DynamicFilter [ REDACTED ]"], "flat_format":REDACTED, "metrics_per_partition": REDACTED metrics=REDACTED_| +|_|_|_SeqScan: region=REDACTED, {"partition_count":REDACTED, "projection": ["product_id", "name", "category", "ts"], "filters": ["product_id IS NOT NULL", "category = Utf8(\"electronics\")"], "dyn_filters": ["DynamicFilter [ REDACTED ]"], "flat_format":REDACTED, "metrics_per_partition": REDACTED metrics=REDACTED_| |_|_|_| |_|_| Total rows: REDACTED_| +-+-+-+ diff --git a/tests/cases/standalone/common/flow/flow_advance_ttl.result b/tests/cases/standalone/common/flow/flow_advance_ttl.result index 6b085390aa..d482d9cbac 100644 --- a/tests/cases/standalone/common/flow/flow_advance_ttl.result +++ b/tests/cases/standalone/common/flow/flow_advance_ttl.result @@ -99,8 +99,10 @@ FROM SELECT number FROM distinct_basic; -++ -++ ++--------+ +| number | ++--------+ ++--------+ -- SQLNESS SLEEP 6s ADMIN FLUSH_TABLE('distinct_basic'); @@ -142,8 +144,10 @@ FROM SELECT number FROM distinct_basic; -++ -++ ++--------+ +| number | ++--------+ ++--------+ DROP FLOW test_distinct_basic; diff --git a/tests/cases/standalone/common/flow/flow_flush.result b/tests/cases/standalone/common/flow/flow_flush.result index c94963488e..d6625767b3 100644 --- a/tests/cases/standalone/common/flow/flow_flush.result +++ b/tests/cases/standalone/common/flow/flow_flush.result @@ -42,8 +42,10 @@ SELECT FROM out_num_cnt_basic; -++ -++ ++---------------------------------+-------------+ +| sum(numbers_input_basic.number) | time_window | ++---------------------------------+-------------+ ++---------------------------------+-------------+ DROP FLOW test_numbers_basic; diff --git a/tests/cases/standalone/common/flow/flow_pending.result b/tests/cases/standalone/common/flow/flow_pending.result index d6fe01b38a..ca74021b5c 100644 --- a/tests/cases/standalone/common/flow/flow_pending.result +++ b/tests/cases/standalone/common/flow/flow_pending.result @@ -47,6 +47,8 @@ Affected Rows: 0 SELECT flow_name FROM INFORMATION_SCHEMA.FLOWS WHERE flow_name = 'pending_with_defer'; -++ -++ ++-----------+ +| flow_name | ++-----------+ ++-----------+ diff --git a/tests/cases/standalone/common/flow/flow_status.result b/tests/cases/standalone/common/flow/flow_status.result index 7b7ff8a1f2..c58277ecef 100644 --- a/tests/cases/standalone/common/flow/flow_status.result +++ b/tests/cases/standalone/common/flow/flow_status.result @@ -59,8 +59,10 @@ SELECT flow_id, flow_name FROM information_schema.flow_statistics WHERE flow_nam -- the like filter matches against flow_name; no matching flow returns empty. SHOW FLOW STATUS LIKE 'no_such_flow'; -++ -++ ++---------+-----------+------------+---------------------+----------------+------------+ +| flow_id | flow_name | start_time | last_execution_time | uptime_seconds | state_size | ++---------+-----------+------------+---------------------+----------------+------------+ ++---------+-----------+------------+---------------------+----------------+------------+ DROP FLOW test_flow_status; diff --git a/tests/cases/standalone/common/flow/flow_user_guide.result b/tests/cases/standalone/common/flow/flow_user_guide.result index 8bd9cab926..a6be36e9b1 100644 --- a/tests/cases/standalone/common/flow/flow_user_guide.result +++ b/tests/cases/standalone/common/flow/flow_user_guide.result @@ -444,8 +444,10 @@ SELECT FROM temp_alerts; -++ -++ ++-----------+-----+----------+----------+ +| sensor_id | loc | max_temp | event_ts | ++-----------+-----+----------+----------+ ++-----------+-----+----------+----------+ INSERT INTO temp_sensor_data diff --git a/tests/cases/standalone/common/flow/show_create_flow.result b/tests/cases/standalone/common/flow/show_create_flow.result index 431d1dfbb5..9499618ec6 100644 --- a/tests/cases/standalone/common/flow/show_create_flow.result +++ b/tests/cases/standalone/common/flow/show_create_flow.result @@ -17,13 +17,17 @@ Affected Rows: 0 SELECT flow_name, table_catalog, flow_definition, source_table_names FROM INFORMATION_SCHEMA.FLOWS WHERE flow_name='filter_numbers_show'; -++ -++ ++-----------+---------------+-----------------+--------------------+ +| flow_name | table_catalog | flow_definition | source_table_names | ++-----------+---------------+-----------------+--------------------+ ++-----------+---------------+-----------------+--------------------+ SHOW FLOWS LIKE 'filter_numbers_show'; -++ -++ ++-------+ +| Flows | ++-------+ ++-------+ CREATE FLOW filter_numbers_show SINK TO out_num_cnt_show AS SELECT number FROM numbers_input_show where number > 10; @@ -63,13 +67,17 @@ Affected Rows: 0 SELECT flow_name, table_catalog, flow_definition, source_table_names FROM INFORMATION_SCHEMA.FLOWS WHERE flow_name='filter_numbers_show'; -++ -++ ++-----------+---------------+-----------------+--------------------+ +| flow_name | table_catalog | flow_definition | source_table_names | ++-----------+---------------+-----------------+--------------------+ ++-----------+---------------+-----------------+--------------------+ SHOW FLOWS LIKE 'filter_numbers_show'; -++ -++ ++-------+ +| Flows | ++-------+ ++-------+ -- also test `CREATE OR REPLACE` and `IF NOT EXISTS` -- (flow exists, replace, if not exists)=(false, false, false) @@ -235,8 +243,10 @@ Error: 1001(Unsupported), Unsupported operation Create flow with both `IF NOT EX SELECT flow_name, table_catalog, flow_definition, source_table_names FROM INFORMATION_SCHEMA.FLOWS WHERE flow_name='filter_numbers_show'; -++ -++ ++-----------+---------------+-----------------+--------------------+ +| flow_name | table_catalog | flow_definition | source_table_names | ++-----------+---------------+-----------------+--------------------+ ++-----------+---------------+-----------------+--------------------+ DROP FLOW filter_numbers_show; diff --git a/tests/cases/standalone/common/function/geo.result b/tests/cases/standalone/common/function/geo.result index 65ba42e6ce..31efbf401d 100644 --- a/tests/cases/standalone/common/function/geo.result +++ b/tests/cases/standalone/common/function/geo.result @@ -360,11 +360,11 @@ FROM( SELECT UNNEST(geo_path(37.76938, -122.3889, 1728083375::TimestampSecond)); -+-----------------------------------------------------------------------------------------------------------------------------+-----------------------------------------------------------------------------------------------------------------------------+ -| __unnest_placeholder(geo_path(Float64(37.76938),Float64(-122.3889),arrow_cast(Int64(1728083375),Utf8("Timestamp(s)")))).lat | __unnest_placeholder(geo_path(Float64(37.76938),Float64(-122.3889),arrow_cast(Int64(1728083375),Utf8("Timestamp(s)")))).lng | -+-----------------------------------------------------------------------------------------------------------------------------+-----------------------------------------------------------------------------------------------------------------------------+ -| [37.76938] | [-122.3889] | -+-----------------------------------------------------------------------------------------------------------------------------+-----------------------------------------------------------------------------------------------------------------------------+ ++-------------------------------------------------------------------------------------------------------+-------------------------------------------------------------------------------------------------------+ +| geo_path(Float64(37.76938),Float64(-122.3889),arrow_cast(Int64(1728083375),Utf8("Timestamp(s)"))).lat | geo_path(Float64(37.76938),Float64(-122.3889),arrow_cast(Int64(1728083375),Utf8("Timestamp(s)"))).lng | ++-------------------------------------------------------------------------------------------------------+-------------------------------------------------------------------------------------------------------+ +| [37.76938] | [-122.3889] | ++-------------------------------------------------------------------------------------------------------+-------------------------------------------------------------------------------------------------------+ SELECT UNNEST(geo_path(lat, lon, ts)) FROM( @@ -377,11 +377,11 @@ FROM( SELECT 37.77001 AS lat, -122.3888 AS lon, 1728083372::TimestampSecond AS ts ); -+------------------------------------------------+------------------------------------------------+ -| __unnest_placeholder(geo_path(lat,lon,ts)).lat | __unnest_placeholder(geo_path(lat,lon,ts)).lng | -+------------------------------------------------+------------------------------------------------+ -| [37.77001, 37.76928, 37.76938, 37.7693] | [-122.3888, -122.3839, -122.3889, -122.382] | -+------------------------------------------------+------------------------------------------------+ ++-----------------------------------------+---------------------------------------------+ +| geo_path(lat,lon,ts).lat | geo_path(lat,lon,ts).lng | ++-----------------------------------------+---------------------------------------------+ +| [37.77001, 37.76928, 37.76938, 37.7693] | [-122.3888, -122.3839, -122.3889, -122.382] | ++-----------------------------------------+---------------------------------------------+ SELECT wkt_point_from_latlng(37.76938, -122.3889) AS point; diff --git a/tests/cases/standalone/common/function/json/json_get.result b/tests/cases/standalone/common/function/json/json_get.result index 830d4e73b9..3c58ca91b9 100644 --- a/tests/cases/standalone/common/function/json/json_get.result +++ b/tests/cases/standalone/common/function/json/json_get.result @@ -450,8 +450,10 @@ SELECT json_to_string(j) FROM jsons WHERE json_to_string(json_get_object(j, 'a.b SELECT json_to_string(j) FROM jsons WHERE json_get_string(json_get_object(j, 'a.x'), 'c') == 'foo'; -++ -++ ++-------------------------+ +| json_to_string(jsons.j) | ++-------------------------+ ++-------------------------+ DROP TABLE jsons; diff --git a/tests/cases/standalone/common/function/matches_term.result b/tests/cases/standalone/common/function/matches_term.result index 37ecf5a55f..d14bcf7ac2 100644 --- a/tests/cases/standalone/common/function/matches_term.result +++ b/tests/cases/standalone/common/function/matches_term.result @@ -488,13 +488,17 @@ SELECT * FROM zh_logs where `log_message` @@ 'ship_'; SELECT * FROM zh_logs where `log_message` @@ '登录_id'; -++ -++ ++----+-------------+ +| id | log_message | ++----+-------------+ ++----+-------------+ SELECT * FROM zh_logs where `log_message` @@ '手机号_trace'; -++ -++ ++----+-------------+ +| id | log_message | ++----+-------------+ ++----+-------------+ SELECT * FROM zh_logs where `log_message` @@ '手机'; diff --git a/tests/cases/standalone/common/function/string/replace.result b/tests/cases/standalone/common/function/string/replace.result index a4e1790d34..d829016f86 100644 --- a/tests/cases/standalone/common/function/string/replace.result +++ b/tests/cases/standalone/common/function/string/replace.result @@ -38,7 +38,7 @@ SELECT REPLACE('hello world', '', 'xyz'); +---------------------------------------------------+ | replace(Utf8("hello world"),Utf8(""),Utf8("xyz")) | +---------------------------------------------------+ -| xyzhxyzexyzlxyzlxyzoxyz xyzwxyzoxyzrxyzlxyzdxyz | +| hello world | +---------------------------------------------------+ SELECT REPLACE('', 'xyz', 'abc'); diff --git a/tests/cases/standalone/common/insert/append_mode.result b/tests/cases/standalone/common/insert/append_mode.result index 1e715ec004..cf8f4b3378 100644 --- a/tests/cases/standalone/common/insert/append_mode.result +++ b/tests/cases/standalone/common/insert/append_mode.result @@ -12,8 +12,10 @@ Affected Rows: 0 SELECT host, ts from append_mode_on ORDER BY host, ts; -++ -++ ++------+----+ +| host | ts | ++------+----+ ++------+----+ INSERT INTO append_mode_on VALUES ('host1',0, 0), ('host2', 1, 1,); diff --git a/tests/cases/standalone/common/insert/logical_metric_table.result b/tests/cases/standalone/common/insert/logical_metric_table.result index fe35ce6aa8..4ea332c2f5 100644 --- a/tests/cases/standalone/common/insert/logical_metric_table.result +++ b/tests/cases/standalone/common/insert/logical_metric_table.result @@ -25,8 +25,10 @@ Affected Rows: 0 SELECT * from t2; -++ -++ ++-----+----+-----+ +| job | ts | val | ++-----+----+-----+ ++-----+----+-----+ INSERT INTO t2 VALUES ('job1', 0, 0), ('job2', 1, 1); @@ -145,8 +147,10 @@ Affected Rows: 0 SELECT * from t2; -++ -++ ++-----+----+-----+ +| job | ts | val | ++-----+----+-----+ ++-----+----+-----+ INSERT INTO t2 VALUES ('job1', 0, 0), ('job2', 1, 1); diff --git a/tests/cases/standalone/common/insert/merge_mode.result b/tests/cases/standalone/common/insert/merge_mode.result index a5e03ae8b9..ff2d111a1d 100644 --- a/tests/cases/standalone/common/insert/merge_mode.result +++ b/tests/cases/standalone/common/insert/merge_mode.result @@ -157,6 +157,7 @@ DROP TABLE `delete_between`; Affected Rows: 0 +-- SQLNESS REPLACE (\sat\sline\s\d+\scolumn\s\d+) create table if not exists invalid_merge_mode( host string, ts timestamp, @@ -168,7 +169,7 @@ create table if not exists invalid_merge_mode( engine=mito with('merge_mode'='first_row'); -Error: 1004(InvalidArguments), Invalid options: Matching variant not found at line 1 column 25 +Error: 1004(InvalidArguments), Invalid options: Matching variant not found create table if not exists invalid_merge_mode( host string, diff --git a/tests/cases/standalone/common/insert/merge_mode.sql b/tests/cases/standalone/common/insert/merge_mode.sql index 9d22cc13d6..c66354f018 100644 --- a/tests/cases/standalone/common/insert/merge_mode.sql +++ b/tests/cases/standalone/common/insert/merge_mode.sql @@ -71,6 +71,7 @@ SELECT * FROM `delete_between`; DROP TABLE `delete_between`; +-- SQLNESS REPLACE (\sat\sline\s\d+\scolumn\s\d+) create table if not exists invalid_merge_mode( host string, ts timestamp, diff --git a/tests/cases/standalone/common/join/join_with_nulls.result b/tests/cases/standalone/common/join/join_with_nulls.result index 81fceba0ff..df2f429093 100644 --- a/tests/cases/standalone/common/join/join_with_nulls.result +++ b/tests/cases/standalone/common/join/join_with_nulls.result @@ -41,8 +41,10 @@ SELECT * FROM null_left l LEFT JOIN null_right r ON l."id" = r."id" ORDER BY l.t -- JOIN on string columns with NULLs SELECT * FROM null_left l INNER JOIN null_right r ON l.val = r.val ORDER BY l.ts; -++ -++ ++----+-----+----+----+-----+----+ +| id | val | ts | id | val | ts | ++----+-----+----+----+-----+----+ ++----+-----+----+----+-----+----+ -- JOIN with IS NOT DISTINCT FROM (treats NULL=NULL as true) SELECT * FROM null_left l INNER JOIN null_right r ON l."id" IS NOT DISTINCT FROM r."id" ORDER BY l.ts; diff --git a/tests/cases/standalone/common/join/self_join.result b/tests/cases/standalone/common/join/self_join.result index 4c6825a836..4720402615 100644 --- a/tests/cases/standalone/common/join/self_join.result +++ b/tests/cases/standalone/common/join/self_join.result @@ -44,8 +44,10 @@ JOIN employees_self m ON e.manager_id = m."id" WHERE e.salary > m.salary ORDER BY e."name"; -++ -++ ++----------+--------+---------+----------------+ +| employee | salary | manager | manager_salary | ++----------+--------+---------+----------------+ ++----------+--------+---------+----------------+ -- Self join to find colleagues (same manager) SELECT e1."name" as employee1, e2."name" as employee2, m."name" as shared_manager diff --git a/tests/cases/standalone/common/order/limit.result b/tests/cases/standalone/common/order/limit.result index 54735fac2d..558f8f6a0f 100644 --- a/tests/cases/standalone/common/order/limit.result +++ b/tests/cases/standalone/common/order/limit.result @@ -45,11 +45,11 @@ Error: 3000(PlanQuery), Failed to plan SQL: No field named a. SELECT a FROM test LIMIT SUM(42); -Error: 1001(Unsupported), This feature is not implemented: Unsupported LIMIT expression: Some(AggregateFunction(AggregateFunction { func: AggregateUDF { inner: Sum { signature: Signature { type_signature: OneOf([Coercible([Exact { desired_type: Decimal }]), Coercible([Implicit { desired_type: Native(LogicalType(Native(UInt64), UInt64)), implicit_coercion: ImplicitCoercion { allowed_source_types: [Native(LogicalType(Native(UInt8), UInt8)), Native(LogicalType(Native(UInt16), UInt16)), Native(LogicalType(Native(UInt32), UInt32))], default_casted_type: UInt64 } }]), Coercible([Implicit { desired_type: Native(LogicalType(Native(Int64), Int64)), implicit_coercion: ImplicitCoercion { allowed_source_types: [Native(LogicalType(Native(Int8), Int8)), Native(LogicalType(Native(Int16), Int16)), Native(LogicalType(Native(Int32), Int32))], default_casted_type: Int64 } }]), Coercible([Implicit { desired_type: Native(LogicalType(Native(Float64), Float64)), implicit_coercion: ImplicitCoercion { allowed_source_types: [Float], default_casted_type: Float64 } }]), Coercible([Exact { desired_type: Duration }])]), volatility: Immutable, parameter_names: None } } }, params: AggregateFunctionParams { args: [Literal(Int64(42), None)], distinct: false, filter: None, order_by: [], null_treatment: None } })) +Error: 1001(Unsupported), This feature is not implemented: Unsupported LIMIT expression: Some(AggregateFunction(AggregateFunction { func: AggregateUDF { inner: Sum { signature: Signature { type_signature: OneOf([Coercible([Exact { desired_type: Decimal, encoding_preservation: EncodingPreservation { preserve_dictionary: false } }]), Coercible([Implicit { desired_type: Native(LogicalType(Native(UInt64), UInt64)), implicit_coercion: ImplicitCoercion { allowed_source_types: [Native(LogicalType(Native(UInt8), UInt8)), Native(LogicalType(Native(UInt16), UInt16)), Native(LogicalType(Native(UInt32), UInt32))], default_casted_type: UInt64 }, encoding_preservation: EncodingPreservation { preserve_dictionary: false } }]), Coercible([Implicit { desired_type: Native(LogicalType(Native(Int64), Int64)), implicit_coercion: ImplicitCoercion { allowed_source_types: [Native(LogicalType(Native(Int8), Int8)), Native(LogicalType(Native(Int16), Int16)), Native(LogicalType(Native(Int32), Int32))], default_casted_type: Int64 }, encoding_preservation: EncodingPreservation { preserve_dictionary: false } }]), Coercible([Implicit { desired_type: Native(LogicalType(Native(Float64), Float64)), implicit_coercion: ImplicitCoercion { allowed_source_types: [Float], default_casted_type: Float64 }, encoding_preservation: EncodingPreservation { preserve_dictionary: false } }]), Coercible([Exact { desired_type: Duration, encoding_preservation: EncodingPreservation { preserve_dictionary: false } }]), Coercible([Exact { desired_type: Interval, encoding_preservation: EncodingPreservation { preserve_dictionary: false } }])]), volatility: Immutable, parameter_names: None } } }, params: AggregateFunctionParams { args: [Literal(Int64(42), None)], distinct: false, filter: None, order_by: [], null_treatment: None } })) SELECT a FROM test LIMIT row_number() OVER (); -Error: 3001(EngineExecuteQuery), This feature is not implemented: Unsupported LIMIT expression: Some(Cast(Cast { expr: WindowFunction(WindowFunction { fun: WindowUDF(WindowUDF { inner: RowNumber { signature: Signature { type_signature: Nullary, volatility: Immutable, parameter_names: None } } }), params: WindowFunctionParams { args: [], partition_by: [], order_by: [], window_frame: WindowFrame { units: Rows, start_bound: Preceding(UInt64(NULL)), end_bound: Following(UInt64(NULL)), is_causal: false }, filter: None, null_treatment: None, distinct: false } }), data_type: Int64 })) +Error: 3001(EngineExecuteQuery), This feature is not implemented: Unsupported LIMIT expression: Some(Cast(Cast { expr: WindowFunction(WindowFunction { fun: WindowUDF(WindowUDF { inner: RowNumber { signature: Signature { type_signature: Nullary, volatility: Immutable, parameter_names: None } } }), params: WindowFunctionParams { args: [], partition_by: [], order_by: [], window_frame: WindowFrame { units: Rows, start_bound: Preceding(UInt64(NULL)), end_bound: Following(UInt64(NULL)), is_causal: false }, filter: None, null_treatment: None, distinct: false } }), field: Field { name: "", data_type: Int64, nullable: true } })) CREATE TABLE test2 (a STRING, ts TIMESTAMP TIME INDEX); diff --git a/tests/cases/standalone/common/order/limit_zero.result b/tests/cases/standalone/common/order/limit_zero.result index 8460464b4f..a520881614 100644 --- a/tests/cases/standalone/common/order/limit_zero.result +++ b/tests/cases/standalone/common/order/limit_zero.result @@ -11,20 +11,26 @@ Affected Rows: 3 -- Test LIMIT 0 returns empty result SELECT * FROM test_data LIMIT 0; -++ -++ ++---+----+ +| i | ts | ++---+----+ ++---+----+ -- Test LIMIT 0 with aggregation SELECT SUM(i) FROM test_data LIMIT 0; -++ -++ ++------------------+ +| sum(test_data.i) | ++------------------+ ++------------------+ -- Test LIMIT 0 with WHERE clause SELECT * FROM test_data WHERE i > 0 LIMIT 0; -++ -++ ++---+----+ +| i | ts | ++---+----+ ++---+----+ -- Clean up DROP TABLE test_data; diff --git a/tests/cases/standalone/common/order/order_by.result b/tests/cases/standalone/common/order/order_by.result index aed1a491c2..da47fa663c 100644 --- a/tests/cases/standalone/common/order/order_by.result +++ b/tests/cases/standalone/common/order/order_by.result @@ -196,7 +196,8 @@ SELECT a-10 AS k FROM test UNION SELECT a-10 AS l FROM test ORDER BY k; -- CONTROVERSIAL: SQLite allows both "k" and "l" to be referenced here, Postgres and MonetDB give an error. SELECT a-10 AS k FROM test UNION SELECT a-10 AS l FROM test ORDER BY l; -Error: 3000(PlanQuery), Failed to plan SQL: No field named l. Valid fields are k. +Error: 3000(PlanQuery), Failed to plan SQL: No field named l. +Valid fields are k. -- Not compatible with duckdb, work in gretimedb SELECT a-10 AS k FROM test UNION SELECT a-10 AS l FROM test ORDER BY 1-k; diff --git a/tests/cases/standalone/common/order/order_by_expressions.result b/tests/cases/standalone/common/order/order_by_expressions.result index f121fac188..5c37867148 100644 --- a/tests/cases/standalone/common/order/order_by_expressions.result +++ b/tests/cases/standalone/common/order/order_by_expressions.result @@ -129,7 +129,14 @@ FROM test WHERE a IS NOT NULL ORDER BY a - (SELECT MIN(a) FROM test WHERE a IS NOT NULL); -Error: 1001(Unsupported), This feature is not implemented: Physical plan does not support logical expression ScalarSubquery() ++---+----+---------------+ +| a | b | diff_from_min | ++---+----+---------------+ +| 1 | 10 | 0 | +| 2 | 20 | 1 | +| 3 | 15 | 2 | +| 4 | 25 | 3 | ++---+----+---------------+ DROP TABLE test; diff --git a/tests/cases/standalone/common/partition.result b/tests/cases/standalone/common/partition.result index 9f512c33b9..fe6f99df59 100644 --- a/tests/cases/standalone/common/partition.result +++ b/tests/cases/standalone/common/partition.result @@ -104,8 +104,10 @@ Affected Rows: 5 SELECT * FROM my_table; -++ -++ ++---+---+----+ +| a | b | ts | ++---+---+----+ ++---+---+----+ DROP TABLE my_table; diff --git a/tests/cases/standalone/common/promql/absent.result b/tests/cases/standalone/common/promql/absent.result index 9c4fb4a2f7..468e173c77 100644 --- a/tests/cases/standalone/common/promql/absent.result +++ b/tests/cases/standalone/common/promql/absent.result @@ -23,14 +23,18 @@ Affected Rows: 8 -- SQLNESS SORT_RESULT 3 1 tql eval (0, 15, '5s') absent(t{job="job1"}); -++ -++ ++----+-----+-----+ +| ts | val | job | ++----+-----+-----+ ++----+-----+-----+ -- SQLNESS SORT_RESULT 3 1 tql eval (0, 15, '5s') absent(t{job="job2"}); -++ -++ ++----+-----+-----+ +| ts | val | job | ++----+-----+-----+ ++----+-----+-----+ -- SQLNESS SORT_RESULT 3 1 tql eval (0, 15, '5s') absent(t{job="job3"}); diff --git a/tests/cases/standalone/common/promql/encode_substrait.result b/tests/cases/standalone/common/promql/encode_substrait.result index bfc675b1ee..230eadb02f 100644 --- a/tests/cases/standalone/common/promql/encode_substrait.result +++ b/tests/cases/standalone/common/promql/encode_substrait.result @@ -42,8 +42,10 @@ tql eval (0, 100, '1s') tag_a="ffa", }[1h])[12h:1h]; -++ -++ ++----+-----------------------------------------------+-------+-------+----------+ +| ts | prom_increase(ts_range,val,ts,Int64(3600000)) | tag_a | tag_b | ts_range | ++----+-----------------------------------------------+-------+-------+----------+ ++----+-----------------------------------------------+-------+-------+----------+ drop table count_total; diff --git a/tests/cases/standalone/common/promql/histogram_multi_partition.result b/tests/cases/standalone/common/promql/histogram_multi_partition.result index 3fba2ba846..64e2f1e855 100644 --- a/tests/cases/standalone/common/promql/histogram_multi_partition.result +++ b/tests/cases/standalone/common/promql/histogram_multi_partition.result @@ -44,9 +44,9 @@ tql analyze (0, 10, '10s') histogram_quantile(0.5, sum by (le) (histogram_gap_bu | 0_| 0_|_HistogramFoldExec: le=@0, field=@2, quantile=0.5 REDACTED |_|_|_SortExec: expr=[ts@1 ASC NULLS LAST, TRY_CAST(le@0 AS Float64) ASC NULLS LAST], preserve_partitioning=[true] REDACTED |_|_|_RepartitionExec: partitioning=Hash([ts@1],REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[le@0 as le, ts@1 as ts], aggr=[sum(histogram_gap_bucket.val)] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[le@0 as le, ts@1 as ts], aggr=[__sum_merge(__sum_state(histogram_gap_bucket.val)) as sum(histogram_gap_bucket.val)] REDACTED |_|_|_RepartitionExec: partitioning=Hash([le@0, ts@1],REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[le@0 as le, ts@1 as ts], aggr=[sum(histogram_gap_bucket.val)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[le@0 as le, ts@1 as ts], aggr=[__sum_merge(__sum_state(histogram_gap_bucket.val)) as sum(histogram_gap_bucket.val)] REDACTED |_|_|_RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=2 REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| diff --git a/tests/cases/standalone/common/promql/instant_last_row.result b/tests/cases/standalone/common/promql/instant_last_row.result index 5093a357ea..16739197fb 100644 --- a/tests/cases/standalone/common/promql/instant_last_row.result +++ b/tests/cases/standalone/common/promql/instant_last_row.result @@ -52,8 +52,10 @@ Affected Rows: 0 -- SQLNESS SORT_RESULT 3 1 TQL EVAL (1000, 1000, '1s') instant_last_empty; -++ -++ ++----+-----+--------+ +| ts | val | series | ++----+-----+--------+ ++----+-----+--------+ DROP TABLE instant_last_empty; @@ -444,8 +446,10 @@ Affected Rows: 1 -- SQLNESS SORT_RESULT 3 1 TQL EVAL (1, 1, '1s') instant_last_field_filter{__field__="val"} > 5; -++ -++ ++-----+------+----------+----+ +| val | host | instance | ts | ++-----+------+----------+----+ ++-----+------+----------+----+ -- A value matcher filters the scan before selecting a sample: the older 10 -- remains eligible, unlike the post-selection comparison above. diff --git a/tests/cases/standalone/common/promql/label.result b/tests/cases/standalone/common/promql/label.result index bd9228a324..112b626171 100644 --- a/tests/cases/standalone/common/promql/label.result +++ b/tests/cases/standalone/common/promql/label.result @@ -103,8 +103,10 @@ TQL EVAL (0, 15, '5s') label_join(test{host="host1"}, "new_host", "-", "idc", "h -- Should return empty result instead of error tql eval label_join(demo_num_cpus, "new_label", "-", "instance", "job"); -++ -++ ++------+-------+-----------+ +| time | value | new_label | ++------+-------+-----------+ ++------+-------+-----------+ -- SQLNESS SORT_RESULT 3 1 TQL EVAL (0, 15, '5s') label_replace(test{host="host1"}, "new_idc", "$2", "idc", "(.*):(.*)"); @@ -202,14 +204,18 @@ TQL EVAL(0, 15, '5s') label_replace(vector(1), "host", "host1", "", ""); -- SQLNESS SORT_RESULT 3 1 TQL EVAL(0, 15, '5s') {__name__="test",host="host1"} * label_replace(vector(1), "host", "host1", "", ""); -++ -++ ++------+------+----------------------------+ +| host | time | test.val * .greptime_value | ++------+------+----------------------------+ ++------+------+----------------------------+ -- SQLNESS SORT_RESULT 3 1 TQL EVAL(0, 15, '5s') {__name__="test",host="host1"} + label_replace(vector(1), "host", "host1", "", ""); -++ -++ ++------+------+----------------------------+ +| host | time | test.val + .greptime_value | ++------+------+----------------------------+ ++------+------+----------------------------+ -- Empty regex with existing source label -- SQLNESS SORT_RESULT 3 1 @@ -272,15 +278,19 @@ TQL EVAL(0, 15, '5s') label_replace(test{host="host1"}, "host2", "", "instance", -- SQLNESS SORT_RESULT 3 1 TQL EVAL(0, 15, '5s') {__name__="test",host="host1"} * label_replace(vector(1), "host", "host2", "host", ""); -++ -++ ++------+------+----------------------------+ +| host | time | test.val * .greptime_value | ++------+------+----------------------------+ ++------+------+----------------------------+ -- Empty regex and not existing label in left expression -- SQLNESS SORT_RESULT 3 1 TQL EVAL(0, 15, '5s') {__name__="test",host="host1"} * label_replace(vector(1), "addr", "host1", "instance", ""); -++ -++ ++------+------+----------------------------+ +| addr | time | test.val * .greptime_value | ++------+------+----------------------------+ ++------+------+----------------------------+ TQL EVAL label_replace(demo_num_cpus, "~invalid", "", "src", "(.*)"); @@ -362,8 +372,10 @@ SELECT * FROM test; -- test the non-existent matchers -- TQL EVAL (0, 1, '5s') test{job=~"host1|host3"}; -++ -++ ++----+------+-----+ +| ts | host | val | ++----+------+-----+ ++----+------+-----+ -- SQLNESS SORT_RESULT 3 1 TQL EVAL (0, 1, '5s') test{job=~".*"}; @@ -377,8 +389,10 @@ TQL EVAL (0, 1, '5s') test{job=~".*"}; TQL EVAL (0, 1, '5s') test{job=~".+"}; -++ -++ ++----+------+-----+ +| ts | host | val | ++----+------+-----+ ++----+------+-----+ -- SQLNESS SORT_RESULT 3 1 TQL EVAL (0, 1, '5s') test{job=""}; @@ -392,8 +406,10 @@ TQL EVAL (0, 1, '5s') test{job=""}; TQL EVAL (0, 1, '5s') test{job!=""}; -++ -++ ++----+------+-----+ +| ts | host | val | ++----+------+-----+ ++----+------+-----+ DROP TABLE test; diff --git a/tests/cases/standalone/common/promql/native_time_selection.result b/tests/cases/standalone/common/promql/native_time_selection.result index 80d18319ad..2c0e20847c 100644 --- a/tests/cases/standalone/common/promql/native_time_selection.result +++ b/tests/cases/standalone/common/promql/native_time_selection.result @@ -94,8 +94,10 @@ TQL EVAL (1, 1, '1s', '1ms') native_time_us{series="exact"}; -- Future-only selection is empty before flushing, exercising the memtable path. TQL EVAL (1, 1, '1s', '300s') native_time_us{series="future"}; -++ -++ ++-----+--------+----+ +| val | series | ts | ++-----+--------+----+ ++-----+--------+----+ ADMIN FLUSH_TABLE('native_time_us'); @@ -116,8 +118,10 @@ TQL EVAL (1, 1, '1s', '1ms') native_time_us{series="exact"}; TQL EVAL (1, 1, '1s', '300s') timestamp(native_time_us{series="future"}); -++ -++ ++----+-------+--------+ +| ts | value | series | ++----+-------+--------+ ++----+-------+--------+ TQL EVAL (1, 1, '1s', '300s') timestamp(native_time_us{series="exact"}); @@ -337,8 +341,10 @@ TQL EVAL (1, 1, '1s', '1ms') native_time_ns{series="exact"}; -- Future-only selection is empty before flushing, exercising the memtable path. TQL EVAL (1, 1, '1s', '300s') native_time_ns{series="future"}; -++ -++ ++-----+--------+----+ +| val | series | ts | ++-----+--------+----+ ++-----+--------+----+ ADMIN FLUSH_TABLE('native_time_ns'); @@ -359,8 +365,10 @@ TQL EVAL (1, 1, '1s', '1ms') native_time_ns{series="exact"}; TQL EVAL (1, 1, '1s', '300s') timestamp(native_time_ns{series="future"}); -++ -++ ++----+-------+--------+ +| ts | value | series | ++----+-------+--------+ ++----+-------+--------+ TQL EVAL (1, 1, '1s', '300s') timestamp(native_time_ns{series="exact"}); diff --git a/tests/cases/standalone/common/promql/null_samples.result b/tests/cases/standalone/common/promql/null_samples.result index 8e0f159fe1..89270227a3 100644 --- a/tests/cases/standalone/common/promql/null_samples.result +++ b/tests/cases/standalone/common/promql/null_samples.result @@ -72,18 +72,24 @@ TQL EVAL (2, 7, '1s') absent_over_time(null_samples{host="a"}[4s]); -- All-NULL windows have no samples: only absent_over_time returns 1. TQL EVAL (3, 3, '1s') count_over_time(null_samples{host="b"}[4s]); -++ -++ ++----+------------------------------------+------+ +| ts | prom_count_over_time(ts_range,val) | host | ++----+------------------------------------+------+ ++----+------------------------------------+------+ TQL EVAL (3, 3, '1s') last_over_time(null_samples{host="b"}[4s]); -++ -++ ++----+-----------------------------------+------+ +| ts | prom_last_over_time(ts_range,val) | host | ++----+-----------------------------------+------+ ++----+-----------------------------------+------+ TQL EVAL (3, 3, '1s') present_over_time(null_samples{host="b"}[4s]); -++ -++ ++----+--------------------------------------+------+ +| ts | prom_present_over_time(ts_range,val) | host | ++----+--------------------------------------+------+ ++----+--------------------------------------+------+ TQL EVAL (3, 3, '1s') absent_over_time(null_samples{host="b"}[4s]); @@ -185,14 +191,18 @@ TQL EVAL (3, 3, '1s') quantile_over_time(0.5, null_samples{host="a"}[4s]); -- A window whose only slot is NULL holds no sample, like a window with no row. TQL EVAL (1, 1, '1s') rate(null_samples{host="a"}[1s]); -++ -++ ++----+----------------------------------------+------+ +| ts | prom_rate(ts_range,val,ts,Int64(1000)) | host | ++----+----------------------------------------+------+ ++----+----------------------------------------+------+ -- Prometheus returns an empty vector for a range without samples, not NaN. TQL EVAL (1, 1, '1s') quantile_over_time(0.5, null_samples{host="a"}[1s]); -++ -++ ++----+----------------------------------------------------+------+ +| ts | prom_quantile_over_time(ts_range,val,Float64(0.5)) | host | ++----+----------------------------------------------------+------+ ++----+----------------------------------------------------+------+ -- `b` never has a sample, so it must not reach the aggregation. -- SQLNESS SORT_RESULT 3 1 diff --git a/tests/cases/standalone/common/promql/offset.result b/tests/cases/standalone/common/promql/offset.result index 760e1ac333..c62bdaee1b 100644 --- a/tests/cases/standalone/common/promql/offset.result +++ b/tests/cases/standalone/common/promql/offset.result @@ -66,8 +66,10 @@ tql eval (1500, 1500, '1s') calculate_rate_offset_total offset -10m; -- SQLNESS SORT_RESULT 3 1 tql eval (0, 0, '1s') calculate_rate_offset_total offset 10m; -++ -++ ++----+-----+---+ +| ts | val | x | ++----+-----+---+ ++----+-----+---+ -- SQLNESS SORT_RESULT 3 1 tql eval (0, 0, '1s') calculate_rate_offset_total offset -10m; @@ -92,14 +94,18 @@ tql eval (3000, 3000, '1s') calculate_rate_offset_total offset 10m; -- SQLNESS SORT_RESULT 3 1 tql eval (3000, 3000, '1s') calculate_rate_offset_total offset -10m; -++ -++ ++----+-----+---+ +| ts | val | x | ++----+-----+---+ ++----+-----+---+ -- SQLNESS SORT_RESULT 3 1 tql eval (3000, 3000, '1s') rate(calculate_rate_window_total[10m]); -++ -++ ++------+------------------------------------------------+ +| time | prom_rate(time_range,value,time,Int64(600000)) | ++------+------------------------------------------------+ ++------+------------------------------------------------+ -- SQLNESS SORT_RESULT 3 1 tql eval (3000, 3000, '1s') rate(calculate_rate_offset_total[10m] offset 5m); diff --git a/tests/cases/standalone/common/promql/or_operation.result b/tests/cases/standalone/common/promql/or_operation.result index 29dbfd1f36..d79df294e8 100644 --- a/tests/cases/standalone/common/promql/or_operation.result +++ b/tests/cases/standalone/common/promql/or_operation.result @@ -31,8 +31,10 @@ tql eval (3000, 3000, '1s') http_requests{job="api",instance="0",env="production tql eval (3000, 3000, '1s') http_requests{job="web",instance="1",env="production"} * 5; -++ -++ ++-----+----------+-----+----+-----------------------------+ +| job | instance | env | ts | greptime_value * Float64(5) | ++-----+----------+-----+----+-----------------------------+ ++-----+----------+-----+----+-----------------------------+ tql eval (3000, 3000, '1s') http_requests{job="web",instance="1",env="production"} * 5 or http_requests{job="api",instance="0",env="production"} + 5; diff --git a/tests/cases/standalone/common/promql/regex.result b/tests/cases/standalone/common/promql/regex.result index bccbd0f340..7e0132c2f4 100644 --- a/tests/cases/standalone/common/promql/regex.result +++ b/tests/cases/standalone/common/promql/regex.result @@ -38,8 +38,10 @@ TQL EVAL (0, 100, '15s') test{host=~"(10.0.160.237:8080|10.0.160.237:9090)"}; TQL EVAL (0, 100, '15s') test{host=~"10\\.0\\.160\\.237:808|nonexistence"}; -++ -++ ++----+------+-----+ +| ts | host | val | ++----+------+-----+ ++----+------+-----+ TQL EVAL (0, 100, '15s') test{host=~"(10\\.0\\.160\\.237:8080|10\\.0\\.160\\.237:9090)"}; diff --git a/tests/cases/standalone/common/promql/set_operation.result b/tests/cases/standalone/common/promql/set_operation.result index 2c56567f08..2f0041f1a3 100644 --- a/tests/cases/standalone/common/promql/set_operation.result +++ b/tests/cases/standalone/common/promql/set_operation.result @@ -225,8 +225,10 @@ tql eval (3000, 3000, '1s') http_requests{g="canary"} unless http_requests{insta -- eval instant at 50m http_requests{group="canary"} unless on(job) http_requests{instance="0"} tql eval (3000, 3000, '1s') http_requests{g="canary"} unless on(job) http_requests{instance="0"}; -++ -++ ++----+-----+----------+---+----------------+ +| ts | job | instance | g | greptime_value | ++----+-----+----------+---+----------------+ ++----+-----+----------+---+----------------+ -- eval instant at 50m http_requests{group="canary"} unless on(job, instance) http_requests{instance="0"} -- http_requests{group="canary", instance="1", job="api-server"} 400 @@ -244,8 +246,10 @@ tql eval (3000, 3000, '1s') http_requests{g="canary"} unless on(job, instance) h -- eval instant at 50m http_requests{group="canary"} unless ignoring(group, instance) http_requests{instance="0"} tql eval (3000, 3000, '1s') http_requests{g="canary"} unless ignoring(g, instance) http_requests{instance="0"}; -++ -++ ++----+-----+----------+---+----------------+ +| ts | job | instance | g | greptime_value | ++----+-----+----------+---+----------------+ ++----+-----+----------+---+----------------+ -- eval instant at 50m http_requests{group="canary"} unless ignoring(group) http_requests{instance="0"} -- http_requests{group="canary", instance="1", job="api-server"} 400 diff --git a/tests/cases/standalone/common/promql/simple_histogram.result b/tests/cases/standalone/common/promql/simple_histogram.result index e02e138f80..a794a30e8b 100644 --- a/tests/cases/standalone/common/promql/simple_histogram.result +++ b/tests/cases/standalone/common/promql/simple_histogram.result @@ -134,8 +134,10 @@ tql eval (3000, 3000, '1s') label_replace(histogram_quantile(0.8, histogram_buck -- SQLNESS SORT_RESULT 3 1 tql eval (3000, 3000, '1s') histogram_quantile(0.2, rate(histogram_bucket[10m])); -++ -++ ++----+------------------------------------------+---+ +| ts | prom_rate(ts_range,val,ts,Int64(600000)) | s | ++----+------------------------------------------+---+ ++----+------------------------------------------+---+ drop table histogram_bucket; @@ -347,8 +349,10 @@ Affected Rows: 0 -- SQLNESS SORT_RESULT 3 1 tql eval(0, 10, '10s') histogram_quantile(0.99, sum by(pod,instance, fff) (rate(greptime_servers_postgres_query_elapsed_bucket{instance=~"xxx"}[1m]))); -++ -++ ++------+----------------------------------------------------+ +| time | sum(prom_rate(time_range,value,time,Int64(60000))) | ++------+----------------------------------------------------+ ++------+----------------------------------------------------+ -- test case where table exists but doesn't have 'le' column should raise error CREATE TABLE greptime_servers_postgres_query_elapsed_no_le ( @@ -365,14 +369,18 @@ Affected Rows: 0 -- SQLNESS SORT_RESULT 3 1 tql eval(0, 10, '10s') histogram_quantile(0.99, sum by(pod,instance, le) (rate(greptime_servers_postgres_query_elapsed_no_le{instance=~"xxx"}[1m]))); -++ -++ ++-----+----------+---+------------------------------------------+ +| pod | instance | t | sum(prom_rate(t_range,v,t,Int64(60000))) | ++-----+----------+---+------------------------------------------+ ++-----+----------+---+------------------------------------------+ -- SQLNESS SORT_RESULT 3 1 tql eval(0, 10, '10s') histogram_quantile(0.99, sum by(pod,instance, fbf) (rate(greptime_servers_postgres_query_elapsed_no_le{instance=~"xxx"}[1m]))); -++ -++ ++-----+----------+---+------------------------------------------+ +| pod | instance | t | sum(prom_rate(t_range,v,t,Int64(60000))) | ++-----+----------+---+------------------------------------------+ ++-----+----------+---+------------------------------------------+ drop table greptime_servers_postgres_query_elapsed_no_le; diff --git a/tests/cases/standalone/common/promql/time_fn.result b/tests/cases/standalone/common/promql/time_fn.result index 92ee594150..25e0cd1103 100644 --- a/tests/cases/standalone/common/promql/time_fn.result +++ b/tests/cases/standalone/common/promql/time_fn.result @@ -206,8 +206,10 @@ tql eval (1701413023, 1701413023, '1s') hour(); tql eval (1701413023, 1701413023, '1s') hour(metrics); -++ -++ ++----+----------------------------+ +| ts | date_part(Utf8("hour"),ts) | ++----+----------------------------+ ++----+----------------------------+ tql eval (1701413023, 1701413023, '1s') minute(); diff --git a/tests/cases/standalone/common/promql/timestamp_fn.result b/tests/cases/standalone/common/promql/timestamp_fn.result index 7cc2b24484..1f5682649a 100644 --- a/tests/cases/standalone/common/promql/timestamp_fn.result +++ b/tests/cases/standalone/common/promql/timestamp_fn.result @@ -105,19 +105,25 @@ tql eval (0, 60, '30s') timestamp(timestamp_test) - time(); -- Test timestamp() with other functions tql eval (0, 60, '30s') abs(timestamp(timestamp_test) - avg(timestamp(timestamp_test))) > 20; -++ -++ ++----+---------------------------------+ +| ts | abs(lhs.value - rhs.avg(value)) | ++----+---------------------------------+ ++----+---------------------------------+ -- Test Issue 6707 tql eval timestamp(demo_memory_usage_bytes * 1); -++ -++ ++------+--------------------+ +| time | value * Float64(1) | ++------+--------------------+ ++------+--------------------+ tql eval timestamp(-demo_memory_usage_bytes); -++ -++ ++------+-----------+ +| time | (- value) | ++------+-----------+ ++------+-----------+ tql eval (0, 60, '30s') timestamp(timestamp_test) == 60; diff --git a/tests/cases/standalone/common/promql/tsid_binary_join_regression.result b/tests/cases/standalone/common/promql/tsid_binary_join_regression.result index a3983260e3..d80d29ce00 100644 --- a/tests/cases/standalone/common/promql/tsid_binary_join_regression.result +++ b/tests/cases/standalone/common/promql/tsid_binary_join_regression.result @@ -115,8 +115,8 @@ TQL ANALYZE (0, 5, '5s') tsid_binary_join_left / tsid_binary_join_right; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_ProjectionExec: expr=[host@2 as host, job@3 as job, ts@5 as ts, __tsid@4 as __tsid, greptime_value@0 / greptime_value@1 as tsid_binary_join_left.greptime_value / tsid_binary_join_right.greptime_value] REDACTED -|_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@1, __tsid@3), (ts@2, ts@4)], projection=[greptime_value@0, greptime_value@3, host@4, job@5, __tsid@6, ts@7], NullsEqual: true REDACTED +| 0_| 0_|_ProjectionExec: expr=[host@0 as host, job@1 as job, ts@2 as ts, __tsid@3 as __tsid, greptime_value@4 / greptime_value@5 as tsid_binary_join_left.greptime_value / tsid_binary_join_right.greptime_value] REDACTED +|_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@1, __tsid@3), (ts@2, ts@4)], projection=[host@4, job@5, ts@7, __tsid@6, greptime_value@0, greptime_value@3], NullsEqual: true REDACTED |_|_|_ProjectionExec: expr=[greptime_value@0 as greptime_value, __tsid@3 as __tsid, ts@4 as ts] REDACTED |_|_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED @@ -155,8 +155,8 @@ TQL ANALYZE (0, 5, '5s') (tsid_binary_join_left + tsid_binary_join_right) / tsid +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_ProjectionExec: expr=[host@1 as host, job@2 as job, ts@4 as ts, __tsid@3 as __tsid, (greptime_value@0 + greptime_value@5) / greptime_value@0 as tsid_binary_join_left.greptime_value + tsid_binary_join_right.greptime_value / tsid_binary_join_left.greptime_value] REDACTED -|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(__tsid@3, __tsid@1), (ts@4, ts@2)], projection=[greptime_value@0, host@1, job@2, __tsid@3, ts@4, greptime_value@5], NullsEqual: true REDACTED +| 0_| 0_|_ProjectionExec: expr=[host@0 as host, job@1 as job, ts@2 as ts, __tsid@3 as __tsid, (greptime_value@4 + greptime_value@5) / greptime_value@4 as tsid_binary_join_left.greptime_value + tsid_binary_join_right.greptime_value / tsid_binary_join_left.greptime_value] REDACTED +|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(__tsid@3, __tsid@1), (ts@4, ts@2)], projection=[host@1, job@2, ts@4, __tsid@3, greptime_value@0, greptime_value@5], NullsEqual: true REDACTED |_|_|_RepartitionExec: partitioning=Hash([__tsid@3, ts@4],REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_RepartitionExec: partitioning=Hash([__tsid@1, ts@2],REDACTED @@ -194,8 +194,8 @@ TQL ANALYZE (0, 5, '5s') ((tsid_binary_join_left + tsid_binary_join_right) * (ts +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_ProjectionExec: expr=[host@1 as host, job@2 as job, ts@4 as ts, __tsid@3 as __tsid, (greptime_value@0 + greptime_value@5) * (greptime_value@0 - greptime_value@6) / (greptime_value@0 + 2) as tsid_binary_join_left.greptime_value + tsid_binary_join_right.greptime_value * tsid_binary_join_left.greptime_value - tsid_binary_join_third.greptime_value / tsid_binary_join_left.greptime_value + Float64(2)] REDACTED -|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(__tsid@3, __tsid@1), (ts@4, ts@2)], projection=[greptime_value@0, host@1, job@2, __tsid@3, ts@4, greptime_value@5, greptime_value@6], NullsEqual: true REDACTED +| 0_| 0_|_ProjectionExec: expr=[host@0 as host, job@1 as job, ts@2 as ts, __tsid@3 as __tsid, (greptime_value@4 + greptime_value@5) * (greptime_value@4 - greptime_value@6) / (greptime_value@4 + 2) as tsid_binary_join_left.greptime_value + tsid_binary_join_right.greptime_value * tsid_binary_join_left.greptime_value - tsid_binary_join_third.greptime_value / tsid_binary_join_left.greptime_value + Float64(2)] REDACTED +|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(__tsid@3, __tsid@1), (ts@4, ts@2)], projection=[host@1, job@2, ts@4, __tsid@3, greptime_value@0, greptime_value@5, greptime_value@6], NullsEqual: true REDACTED |_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(__tsid@3, __tsid@1), (ts@4, ts@2)], projection=[greptime_value@0, host@1, job@2, __tsid@3, ts@4, greptime_value@5], NullsEqual: true REDACTED |_|_|_RepartitionExec: partitioning=Hash([REDACTED |_|_|_MergeScanExec: REDACTED @@ -243,8 +243,8 @@ TQL ANALYZE (0, 5, '5s') tsid_binary_join_left / ignoring(host) tsid_binary_join +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_ProjectionExec: expr=[host@2 as host, job@3 as job, ts@5 as ts, __tsid@4 as __tsid, greptime_value@0 / greptime_value@1 as tsid_binary_join_left.greptime_value / tsid_binary_join_right.greptime_value] REDACTED -|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(job@1, job@2), (ts@2, ts@4)], projection=[greptime_value@0, greptime_value@3, host@4, job@5, __tsid@6, ts@7], NullsEqual: true REDACTED +| 0_| 0_|_ProjectionExec: expr=[host@0 as host, job@1 as job, ts@2 as ts, __tsid@3 as __tsid, greptime_value@4 / greptime_value@5 as tsid_binary_join_left.greptime_value / tsid_binary_join_right.greptime_value] REDACTED +|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(job@1, job@2), (ts@2, ts@4)], projection=[host@4, job@5, ts@7, __tsid@6, greptime_value@0, greptime_value@3], NullsEqual: true REDACTED |_|_|_RepartitionExec: partitioning=Hash([job@1, ts@2],REDACTED |_|_|_ProjectionExec: expr=[greptime_value@0 as greptime_value, job@2 as job, ts@4 as ts] REDACTED |_|_|_MergeScanExec: REDACTED @@ -282,8 +282,8 @@ TQL ANALYZE (0, 5, '5s') tsid_binary_join_left / on(job) tsid_binary_join_right_ +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_ProjectionExec: expr=[job@2 as job, ts@4 as ts, __tsid@3 as __tsid, greptime_value@0 / greptime_value@1 as tsid_binary_join_left.greptime_value / tsid_binary_join_right_by_job.greptime_value] REDACTED -|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(job@1, job@1), (ts@2, ts@3)], projection=[greptime_value@0, greptime_value@3, job@4, __tsid@5, ts@6], NullsEqual: true REDACTED +| 0_| 0_|_ProjectionExec: expr=[job@0 as job, ts@1 as ts, __tsid@2 as __tsid, greptime_value@3 / greptime_value@4 as tsid_binary_join_left.greptime_value / tsid_binary_join_right_by_job.greptime_value] REDACTED +|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(job@1, job@1), (ts@2, ts@3)], projection=[job@4, ts@6, __tsid@5, greptime_value@0, greptime_value@3], NullsEqual: true REDACTED |_|_|_RepartitionExec: partitioning=Hash([job@1, ts@2],REDACTED |_|_|_ProjectionExec: expr=[greptime_value@0 as greptime_value, job@2 as job, ts@4 as ts] REDACTED |_|_|_MergeScanExec: REDACTED @@ -322,8 +322,7 @@ TQL ANALYZE (0, 5, '5s') tsid_binary_join_left > tsid_binary_join_right; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_ProjectionExec: expr=[ts@4 as ts, greptime_value@0 as greptime_value, host@1 as host, job@2 as job, __tsid@3 as __tsid] REDACTED -|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(__tsid@3, __tsid@1), (ts@4, ts@2)], filter=greptime_value@1 < greptime_value@0, projection=[greptime_value@0, host@1, job@2, __tsid@3, ts@4], NullsEqual: true REDACTED +| 0_| 0_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(__tsid@3, __tsid@1), (ts@4, ts@2)], filter=greptime_value@1 < greptime_value@0, projection=[ts@4, greptime_value@0, host@1, job@2, __tsid@3], NullsEqual: true REDACTED |_|_|_RepartitionExec: partitioning=Hash([__tsid@3, ts@4],REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_RepartitionExec: partitioning=Hash([__tsid@1, ts@2],REDACTED @@ -361,8 +360,8 @@ TQL ANALYZE (0, 5, '5s') tsid_binary_join_left > bool tsid_binary_join_right; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_ProjectionExec: expr=[host@2 as host, job@3 as job, ts@5 as ts, __tsid@4 as __tsid, CAST(greptime_value@1 < greptime_value@0 AS Float64) as tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value] REDACTED -|_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@1, __tsid@3), (ts@2, ts@4)], projection=[greptime_value@0, greptime_value@3, host@4, job@5, __tsid@6, ts@7], NullsEqual: true REDACTED +| 0_| 0_|_ProjectionExec: expr=[host@0 as host, job@1 as job, ts@2 as ts, __tsid@3 as __tsid, CAST(greptime_value@4 < greptime_value@5 AS Float64) as tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value] REDACTED +|_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@1, __tsid@3), (ts@2, ts@4)], projection=[host@4, job@5, ts@7, __tsid@6, greptime_value@3, greptime_value@0], NullsEqual: true REDACTED |_|_|_ProjectionExec: expr=[greptime_value@0 as greptime_value, __tsid@3 as __tsid, ts@4 as ts] REDACTED |_|_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED @@ -400,11 +399,10 @@ TQL ANALYZE (0, 5, '5s') ((tsid_binary_join_left > tsid_binary_join_right) / tsi +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_ProjectionExec: expr=[host@2 as host, job@3 as job, ts@5 as ts, __tsid@4 as __tsid, greptime_value@0 / greptime_value@1 * 100 as .greptime_value / tsid_binary_join_left.greptime_value * Float64(100)] REDACTED -|_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@2, __tsid@3), (ts@0, ts@4)], projection=[greptime_value@1, greptime_value@3, host@4, job@5, __tsid@6, ts@7], NullsEqual: true REDACTED +| 0_| 0_|_ProjectionExec: expr=[host@0 as host, job@1 as job, ts@2 as ts, __tsid@3 as __tsid, greptime_value@4 / greptime_value@5 * 100 as .greptime_value / tsid_binary_join_left.greptime_value * Float64(100)] REDACTED +|_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@2, __tsid@3), (ts@0, ts@4)], projection=[host@4, job@5, ts@7, __tsid@6, greptime_value@1, greptime_value@3], NullsEqual: true REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_ProjectionExec: expr=[ts@2 as ts, greptime_value@0 as greptime_value, __tsid@1 as __tsid] REDACTED -|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(__tsid@1, __tsid@1), (ts@2, ts@2)], filter=greptime_value@1 < greptime_value@0, projection=[greptime_value@0, __tsid@1, ts@2], NullsEqual: true REDACTED +|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(__tsid@1, __tsid@1), (ts@2, ts@2)], filter=greptime_value@1 < greptime_value@0, projection=[ts@2, greptime_value@0, __tsid@1], NullsEqual: true REDACTED |_|_|_RepartitionExec: partitioning=Hash([REDACTED |_|_|_ProjectionExec: expr=[greptime_value@0 as greptime_value, __tsid@3 as __tsid, ts@4 as ts] REDACTED |_|_|_MergeScanExec: REDACTED @@ -450,13 +448,13 @@ TQL ANALYZE (0, 5, '5s') ((tsid_binary_join_left > bool tsid_binary_join_right) +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_ProjectionExec: expr=[host@2 as host, job@3 as job, ts@5 as ts, __tsid@4 as __tsid, tsid_binary_join_right.tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value + tsid_binary_join_left.greptime_value@0 / greptime_value@1 as lhs.tsid_binary_join_right.tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value + tsid_binary_join_left.greptime_value / rhs.greptime_value] REDACTED -|_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@1, __tsid@3), (ts@0, ts@4)], projection=[tsid_binary_join_right.tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value + tsid_binary_join_left.greptime_value@2, greptime_value@3, host@4, job@5, __tsid@6, ts@7], NullsEqual: true REDACTED +| 0_| 0_|_ProjectionExec: expr=[host@0 as host, job@1 as job, ts@2 as ts, __tsid@3 as __tsid, tsid_binary_join_right.tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value + tsid_binary_join_left.greptime_value@4 / greptime_value@5 as lhs.tsid_binary_join_right.tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value + tsid_binary_join_left.greptime_value / rhs.greptime_value] REDACTED +|_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@1, __tsid@3), (ts@0, ts@4)], projection=[host@4, job@5, ts@7, __tsid@6, tsid_binary_join_right.tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value + tsid_binary_join_left.greptime_value@2, greptime_value@3], NullsEqual: true REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_ProjectionExec: expr=[ts@3 as ts, __tsid@2 as __tsid, tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value@0 + greptime_value@1 as tsid_binary_join_right.tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value + tsid_binary_join_left.greptime_value] REDACTED -|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(__tsid@1, __tsid@1), (ts@0, ts@2)], projection=[tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value@2, greptime_value@3, __tsid@4, ts@5], NullsEqual: true REDACTED -|_|_|_ProjectionExec: expr=[ts@3 as ts, __tsid@2 as __tsid, CAST(greptime_value@1 < greptime_value@0 AS Float64) as tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value] REDACTED -|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(__tsid@1, __tsid@1), (ts@2, ts@2)], projection=[greptime_value@0, greptime_value@3, __tsid@4, ts@5], NullsEqual: true REDACTED +|_|_|_ProjectionExec: expr=[ts@0 as ts, __tsid@1 as __tsid, tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value@2 + greptime_value@3 as tsid_binary_join_right.tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value + tsid_binary_join_left.greptime_value] REDACTED +|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(__tsid@1, __tsid@1), (ts@0, ts@2)], projection=[ts@5, __tsid@4, tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value@2, greptime_value@3], NullsEqual: true REDACTED +|_|_|_ProjectionExec: expr=[ts@0 as ts, __tsid@1 as __tsid, CAST(greptime_value@2 < greptime_value@3 AS Float64) as tsid_binary_join_left.greptime_value > tsid_binary_join_right.greptime_value] REDACTED +|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(__tsid@1, __tsid@1), (ts@2, ts@2)], projection=[ts@5, __tsid@4, greptime_value@3, greptime_value@0], NullsEqual: true REDACTED |_|_|_RepartitionExec: partitioning=Hash([REDACTED |_|_|_ProjectionExec: expr=[greptime_value@0 as greptime_value, __tsid@3 as __tsid, ts@4 as ts] REDACTED |_|_|_MergeScanExec: REDACTED @@ -512,8 +510,8 @@ TQL ANALYZE (0, 5, '5s') (tsid_binary_join_left or tsid_binary_join_right) / tsi +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_ProjectionExec: expr=[host@2 as host, job@3 as job, ts@5 as ts, __tsid@4 as __tsid, greptime_value@0 / greptime_value@1 as lhs.greptime_value / rhs.greptime_value] REDACTED -|_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@1, __tsid@3), (ts@0, ts@4)], projection=[greptime_value@2, greptime_value@3, host@4, job@5, __tsid@6, ts@7], NullsEqual: true REDACTED +| 0_| 0_|_ProjectionExec: expr=[host@0 as host, job@1 as job, ts@2 as ts, __tsid@3 as __tsid, greptime_value@4 / greptime_value@5 as lhs.greptime_value / rhs.greptime_value] REDACTED +|_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@1, __tsid@3), (ts@0, ts@4)], projection=[host@4, job@5, ts@7, __tsid@6, greptime_value@2, greptime_value@3], NullsEqual: true REDACTED |_|_|_ProjectionExec: expr=[ts@0 as ts, __tsid@1 as __tsid, greptime_value@2 as greptime_value] REDACTED |_|_|_UnionDistinctOnExec: on col=[5, 6], ts_col=0 REDACTED |_|_|_CooperativeExec REDACTED @@ -564,11 +562,11 @@ TQL ANALYZE (0, 5, '5s') (tsid_binary_join_left / ignoring(host) group_left tsid +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_ProjectionExec: expr=[host@2 as host, job@3 as job, ts@5 as ts, __tsid@4 as __tsid, tsid_binary_join_left.greptime_value / tsid_binary_join_right.greptime_value@0 / greptime_value@1 as tsid_binary_join_right.tsid_binary_join_left.greptime_value / tsid_binary_join_right.greptime_value / tsid_binary_join_left.greptime_value] REDACTED -|_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@1, __tsid@3), (ts@0, ts@4)], projection=[tsid_binary_join_left.greptime_value / tsid_binary_join_right.greptime_value@2, greptime_value@3, host@4, job@5, __tsid@6, ts@7], NullsEqual: true REDACTED +| 0_| 0_|_ProjectionExec: expr=[host@0 as host, job@1 as job, ts@2 as ts, __tsid@3 as __tsid, tsid_binary_join_left.greptime_value / tsid_binary_join_right.greptime_value@4 / greptime_value@5 as tsid_binary_join_right.tsid_binary_join_left.greptime_value / tsid_binary_join_right.greptime_value / tsid_binary_join_left.greptime_value] REDACTED +|_|_|_HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(__tsid@1, __tsid@3), (ts@0, ts@4)], projection=[host@4, job@5, ts@7, __tsid@6, tsid_binary_join_left.greptime_value / tsid_binary_join_right.greptime_value@2, greptime_value@3], NullsEqual: true REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_ProjectionExec: expr=[ts@3 as ts, __tsid@2 as __tsid, greptime_value@0 / greptime_value@1 as tsid_binary_join_left.greptime_value / tsid_binary_join_right.greptime_value] REDACTED -|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(job@1, job@1), (ts@2, ts@3)], projection=[greptime_value@0, greptime_value@3, __tsid@5, ts@6], NullsEqual: true REDACTED +|_|_|_ProjectionExec: expr=[ts@0 as ts, __tsid@1 as __tsid, greptime_value@2 / greptime_value@3 as tsid_binary_join_left.greptime_value / tsid_binary_join_right.greptime_value] REDACTED +|_|_|_HashJoinExec: mode=Partitioned, join_type=Inner, on=[(job@1, job@1), (ts@2, ts@3)], projection=[ts@6, __tsid@5, greptime_value@0, greptime_value@3], NullsEqual: true REDACTED |_|_|_RepartitionExec: partitioning=Hash([REDACTED |_|_|_ProjectionExec: expr=[greptime_value@0 as greptime_value, job@2 as job, ts@4 as ts] REDACTED |_|_|_MergeScanExec: REDACTED diff --git a/tests/cases/standalone/common/range/by.result b/tests/cases/standalone/common/range/by.result index fd2feb1da4..ef842f3078 100644 --- a/tests/cases/standalone/common/range/by.result +++ b/tests/cases/standalone/common/range/by.result @@ -63,7 +63,8 @@ SELECT ts, length(host)::INT64 + 2, max(val) RANGE '5s' FROM host ALIGN '20s' BY -- project non-aggregation key SELECT ts, host, max(val) RANGE '5s' FROM host ALIGN '20s' BY () ORDER BY ts; -Error: 3001(EngineExecuteQuery), No field named host.host. Did you mean 'host.ts'?. +Error: 3001(EngineExecuteQuery), No field named host.host. Did you mean 'host.ts'? +Valid fields are "max(host.val) RANGE 5s", host.ts, "Int64(1)". DROP TABLE host; diff --git a/tests/cases/standalone/common/range/calculate.result b/tests/cases/standalone/common/range/calculate.result index f27cbef398..7037854f9e 100644 --- a/tests/cases/standalone/common/range/calculate.result +++ b/tests/cases/standalone/common/range/calculate.result @@ -126,16 +126,16 @@ SELECT ts, host, approx_percentile_cont(0.5) WITHIN GROUP (ORDER BY val) RANGE ' +---------------------+-------+--------------------------------------------------------------------------------------+ | ts | host | approx_percentile_cont(Float64(0.5)) WITHIN GROUP [host.val ASC NULLS LAST] RANGE 5s | +---------------------+-------+--------------------------------------------------------------------------------------+ -| 1970-01-01T00:00:00 | host1 | 0 | +| 1970-01-01T00:00:00 | host1 | 0.0 | | 1970-01-01T00:00:05 | host1 | | -| 1970-01-01T00:00:10 | host1 | 1 | +| 1970-01-01T00:00:10 | host1 | 1.0 | | 1970-01-01T00:00:15 | host1 | | -| 1970-01-01T00:00:20 | host1 | 2 | -| 1970-01-01T00:00:00 | host2 | 3 | +| 1970-01-01T00:00:20 | host1 | 2.0 | +| 1970-01-01T00:00:00 | host2 | 3.0 | | 1970-01-01T00:00:05 | host2 | | -| 1970-01-01T00:00:10 | host2 | 4 | +| 1970-01-01T00:00:10 | host2 | 4.0 | | 1970-01-01T00:00:15 | host2 | | -| 1970-01-01T00:00:20 | host2 | 5 | +| 1970-01-01T00:00:20 | host2 | 5.0 | +---------------------+-------+--------------------------------------------------------------------------------------+ -- Test complex range expr calculate diff --git a/tests/cases/standalone/common/range/error.result b/tests/cases/standalone/common/range/error.result index ff0795f450..feaee54d1a 100644 --- a/tests/cases/standalone/common/range/error.result +++ b/tests/cases/standalone/common/range/error.result @@ -33,23 +33,24 @@ Error: 2000(InvalidSyntax), Invalid SQL syntax: sql parser error: not a valid du -- 2.1 no range param SELECT min(val) FROM host ALIGN '5s'; -Error: 2000(InvalidSyntax), Invalid SQL syntax: sql parser error: Illegal Range select, no RANGE keyword found in any SelectItem +Error: 2000(InvalidSyntax), Invalid SQL syntax: sql parser error: Illegal range select: no RANGE keyword found in any SELECT item SELECT 1 FROM host ALIGN '5s'; -Error: 2000(InvalidSyntax), Invalid SQL syntax: sql parser error: Illegal Range select, no RANGE keyword found in any SelectItem +Error: 2000(InvalidSyntax), Invalid SQL syntax: sql parser error: Illegal range select: no RANGE keyword found in any SELECT item SELECT min(val) RANGE '10s', max(val) FROM host ALIGN '5s'; -Error: 3001(EngineExecuteQuery), No field named "max(host.val)". Valid fields are "min(host.val) RANGE 10s", host.ts, host.host. +Error: 3001(EngineExecuteQuery), No field named "max(host.val)". +Valid fields are "min(host.val) RANGE 10s", host.ts, host.host. SELECT min(val) * 2 RANGE '10s' FROM host ALIGN '5s'; -Error: 2000(InvalidSyntax), Invalid SQL syntax: sql parser error: Can't use the RANGE keyword in Expr 2 without function +Error: 2000(InvalidSyntax), Invalid SQL syntax: sql parser error: Can't use RANGE in expression 2 without a function SELECT 1 RANGE '10s' FILL NULL FROM host ALIGN '1h' FILL NULL; -Error: 2000(InvalidSyntax), Invalid SQL syntax: sql parser error: Can't use the RANGE keyword in Expr 1 without function +Error: 2000(InvalidSyntax), Invalid SQL syntax: sql parser error: Can't use RANGE in expression 1 without a function -- 2.2 no align param SELECT min(val) RANGE '5s' FROM host; @@ -85,7 +86,7 @@ SELECT covar(ceil(val), floor(val)) RANGE '20s' FROM host ALIGN '10s'; -- 2.4 nest query SELECT min(max(val) RANGE '20s') RANGE '20s' FROM host ALIGN '10s'; -Error: 2000(InvalidSyntax), Range Query: Nest Range Query is not allowed +Error: 3000(PlanQuery), Failed to plan SQL: Error during planning: Aggregate function calls cannot be nested: 'max(host.val)' is nested inside 'min(range_fn(max(host.val), Utf8("20s"), Utf8("")))' -- 2.5 wrong Aggregate SELECT rank() OVER (PARTITION BY host ORDER BY ts DESC) RANGE '10s' FROM host ALIGN '5s'; diff --git a/tests/cases/standalone/common/range/nest.result b/tests/cases/standalone/common/range/nest.result index 184b27545d..e283c45752 100644 --- a/tests/cases/standalone/common/range/nest.result +++ b/tests/cases/standalone/common/range/nest.result @@ -132,7 +132,6 @@ EXPLAIN SELECT ts, host, min(val) RANGE '5s' FROM host ALIGN '5s'; | plan_type_| plan_| +-+-+ | logical_plan_| RangeSelect: range_exprs=[min(host.val) RANGE 5s], align=5000ms, align_to=0ms, align_by=[host.host], time_index=ts | -|_|_Projection: host.ts, host.host, host.val_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Projection: host.ts, host.host, host.val_| |_|_TableScan: host_| diff --git a/tests/cases/standalone/common/range/special_aggr.result b/tests/cases/standalone/common/range/special_aggr.result index f6659eaa80..a4071349c9 100644 --- a/tests/cases/standalone/common/range/special_aggr.result +++ b/tests/cases/standalone/common/range/special_aggr.result @@ -231,7 +231,7 @@ SELECT ts, host, count(distinct *) RANGE '5s' FROM host ALIGN '5s' ORDER BY host -- Test error first_value/last_value SELECT ts, host, first_value(val, val) RANGE '5s' FROM host ALIGN '5s' ORDER BY host, ts; -Error: 3000(PlanQuery), Failed to plan SQL: Error during planning: The function 'first_value' expected 1 arguments but received 2 No function matches the given name and argument types 'first_value(Int64, Int64)'. You might need to add explicit type casts. +Error: 3000(PlanQuery), Failed to plan SQL: Error during planning: The function 'first_value' expected 1 arguments but received 2. No function matches the given name and argument types 'first_value(Int64, Int64)'. You might need to add explicit type casts. Candidate functions: first_value(Any) diff --git a/tests/cases/standalone/common/select/matches.result b/tests/cases/standalone/common/select/matches.result index 084cec8cd8..6a7db78ae3 100644 --- a/tests/cases/standalone/common/select/matches.result +++ b/tests/cases/standalone/common/select/matches.result @@ -80,13 +80,17 @@ select fox from fox where matches(fox, 'fox AND lazy') order by ts; select fox from fox where matches(fox, '-over -lazy') order by ts; -++ -++ ++-----+ +| fox | ++-----+ ++-----+ select fox from fox where matches(fox, '-over AND -lazy') order by ts; -++ -++ ++-----+ +| fox | ++-----+ ++-----+ select fox from fox where matches(fox, 'fox AND jumps OR over') order by ts; @@ -228,8 +232,10 @@ select fox from fox where matches(fox, '+(fox jumps) AND over') order by ts; select fox from fox where matches(fox, 'over -(fox jumps)') order by ts; -++ -++ ++-----+ +| fox | ++-----+ ++-----+ select fox from fox where matches(fox, 'over -(fox AND jumps)') order by ts; diff --git a/tests/cases/standalone/common/select/prune.result b/tests/cases/standalone/common/select/prune.result index d5e2b5c3de..5e219a195d 100644 --- a/tests/cases/standalone/common/select/prune.result +++ b/tests/cases/standalone/common/select/prune.result @@ -48,8 +48,10 @@ select * from demo where host='test2' and idc='idc1' and collector='disk'; select * from demo where host='test2' and idc='idc2'; -++ -++ ++----+-------+------+-----+-----------+ +| ts | value | host | idc | collector | ++----+-------+------+-----+-----------+ ++----+-------+------+-----+-----------+ select * from demo where host='test3' and idc>'idc1'; diff --git a/tests/cases/standalone/common/select/schema_reference.result b/tests/cases/standalone/common/select/schema_reference.result index 37d2de4c8c..9ae3ebcac5 100644 --- a/tests/cases/standalone/common/select/schema_reference.result +++ b/tests/cases/standalone/common/select/schema_reference.result @@ -25,7 +25,8 @@ SELECT s1.tbl.i FROM s1.tbl ORDER BY i; -- Test schema mismatch error - should fail SELECT s2.tbl.i FROM s1.tbl; -Error: 3000(PlanQuery), Failed to plan SQL: No field named s2.tbl.i. Valid fields are s1.tbl.i, s1.tbl.ts. +Error: 3000(PlanQuery), Failed to plan SQL: No field named s2.tbl.i. Did you mean 's1.tbl.i'? +Valid fields are s1.tbl.i, s1.tbl.ts. -- Clean up DROP TABLE s1.tbl; diff --git a/tests/cases/standalone/common/select/tql_filter.result b/tests/cases/standalone/common/select/tql_filter.result index b4fbfdb586..45f85ead00 100644 --- a/tests/cases/standalone/common/select/tql_filter.result +++ b/tests/cases/standalone/common/select/tql_filter.result @@ -95,8 +95,10 @@ Affected Rows: 5 -- SQLNESS SORT_RESULT 3 1 tql eval (0, 0, '1s') t2{a=~"10"}; -++ -++ ++---+---+---+ +| a | b | c | ++---+---+---+ ++---+---+---+ -- SQLNESS SORT_RESULT 3 1 tql eval (0, 0, '1s') t2{a=~"10.*"}; diff --git a/tests/cases/standalone/common/select/unnest.result b/tests/cases/standalone/common/select/unnest.result index e082a8a6c1..8555a49506 100644 --- a/tests/cases/standalone/common/select/unnest.result +++ b/tests/cases/standalone/common/select/unnest.result @@ -29,11 +29,11 @@ SELECT unnest([1,2,3]); SELECT unnest(struct(1,2,3)); -+-------------------------------------------------------------+-------------------------------------------------------------+-------------------------------------------------------------+ -| __unnest_placeholder(struct(Int64(1),Int64(2),Int64(3))).c0 | __unnest_placeholder(struct(Int64(1),Int64(2),Int64(3))).c1 | __unnest_placeholder(struct(Int64(1),Int64(2),Int64(3))).c2 | -+-------------------------------------------------------------+-------------------------------------------------------------+-------------------------------------------------------------+ -| 1 | 2 | 3 | -+-------------------------------------------------------------+-------------------------------------------------------------+-------------------------------------------------------------+ ++---------------------------------------+---------------------------------------+---------------------------------------+ +| struct(Int64(1),Int64(2),Int64(3)).c0 | struct(Int64(1),Int64(2),Int64(3)).c1 | struct(Int64(1),Int64(2),Int64(3)).c2 | ++---------------------------------------+---------------------------------------+---------------------------------------+ +| 1 | 2 | 3 | ++---------------------------------------+---------------------------------------+---------------------------------------+ -- Table function is not supported for now -- SELECT * FROM unnest([1,2,3]); diff --git a/tests/cases/standalone/common/setops/basic_setops.result b/tests/cases/standalone/common/setops/basic_setops.result index 2bee6caa0f..c1b6dd14ec 100644 --- a/tests/cases/standalone/common/setops/basic_setops.result +++ b/tests/cases/standalone/common/setops/basic_setops.result @@ -49,8 +49,10 @@ SELECT NULL UNION SELECT NULL; SELECT NULL EXCEPT SELECT NULL; -++ -++ ++------+ +| NULL | ++------+ ++------+ SELECT NULL INTERSECT SELECT NULL; @@ -120,8 +122,10 @@ SELECT 1 EXCEPT SELECT 2; SELECT 1 EXCEPT SELECT 1; -++ -++ ++----------+ +| Int64(1) | ++----------+ ++----------+ SELECT 1 INTERSECT SELECT 1; @@ -133,8 +137,10 @@ SELECT 1 INTERSECT SELECT 1; SELECT 1 INTERSECT SELECT 2; -++ -++ ++----------+ +| Int64(1) | ++----------+ ++----------+ DROP TABLE test; diff --git a/tests/cases/standalone/common/show/show_charset.result b/tests/cases/standalone/common/show/show_charset.result index 10b25864d3..abd9924e68 100644 --- a/tests/cases/standalone/common/show/show_charset.result +++ b/tests/cases/standalone/common/show/show_charset.result @@ -24,8 +24,10 @@ SHOW CHARACTER SET LIKE 'utf8'; SHOW CHARACTER SET LIKE 'latin1'; -++ -++ ++---------+-------------+-------------------+--------+ +| Charset | Description | Default collation | Maxlen | ++---------+-------------+-------------------+--------+ ++---------+-------------+-------------------+--------+ SHOW CHARSET LIKE 'utf8'; @@ -53,6 +55,8 @@ SHOW CHARSET WHERE Charset = 'utf8'; SHOW CHARSET WHERE Charset = 'latin1'; -++ -++ ++---------+-------------+-------------------+--------+ +| Charset | Description | Default collation | Maxlen | ++---------+-------------+-------------------+--------+ ++---------+-------------+-------------------+--------+ diff --git a/tests/cases/standalone/common/show/show_collation.result b/tests/cases/standalone/common/show/show_collation.result index 4dcf953059..99bd07346c 100644 --- a/tests/cases/standalone/common/show/show_collation.result +++ b/tests/cases/standalone/common/show/show_collation.result @@ -8,8 +8,10 @@ SHOW COLLATION; SHOW COLLATION LIKE 'utf8'; -++ -++ ++-----------+---------+----+---------+----------+---------+ +| Collation | Charset | Id | Default | Compiled | Sortlen | ++-----------+---------+----+---------+----------+---------+ ++-----------+---------+----+---------+----------+---------+ SHOW COLLATION WHERE Charset = 'utf8'; @@ -21,11 +23,15 @@ SHOW COLLATION WHERE Charset = 'utf8'; SHOW COLLATION WHERE Charset = 'latin1'; -++ -++ ++-----------+---------+----+---------+----------+---------+ +| Collation | Charset | Id | Default | Compiled | Sortlen | ++-----------+---------+----+---------+----------+---------+ ++-----------+---------+----+---------+----------+---------+ SHOW COLLATION LIKE 'latin1'; -++ -++ ++-----------+---------+----+---------+----------+---------+ +| Collation | Charset | Id | Default | Compiled | Sortlen | ++-----------+---------+----+---------+----------+---------+ ++-----------+---------+----+---------+----------+---------+ diff --git a/tests/cases/standalone/common/show/show_region.result b/tests/cases/standalone/common/show/show_region.result index 17d36c7438..914c480f7a 100644 --- a/tests/cases/standalone/common/show/show_region.result +++ b/tests/cases/standalone/common/show/show_region.result @@ -45,8 +45,10 @@ SHOW REGION FROM another_table in public; -- SQLNESS REPLACE (\d{1}) PEER_ID SHOW REGION FROM another_table WHERE Leader = 'No'; -++ -++ ++-------+--------+------+--------+ +| Table | Region | Peer | Leader | ++-------+--------+------+--------+ ++-------+--------+------+--------+ DROP TABLE my_table; diff --git a/tests/cases/standalone/common/skip_wal.result b/tests/cases/standalone/common/skip_wal.result index 5f3ae29f52..a3074779d3 100644 --- a/tests/cases/standalone/common/skip_wal.result +++ b/tests/cases/standalone/common/skip_wal.result @@ -22,8 +22,10 @@ Affected Rows: 3 -- SQLNESS ARG restart=true SELECT * FROM system_metrics; -++ -++ ++------+-----+----------+-------------+-----------+----+ +| host | idc | cpu_util | memory_util | disk_util | ts | ++------+-----+----------+-------------+-----------+----+ ++------+-----+----------+-------------+-----------+----+ INSERT INTO system_metrics VALUES diff --git a/tests/cases/standalone/common/subquery/neumann.result b/tests/cases/standalone/common/subquery/neumann.result index e575b1694c..66d167bfbc 100644 --- a/tests/cases/standalone/common/subquery/neumann.result +++ b/tests/cases/standalone/common/subquery/neumann.result @@ -55,7 +55,7 @@ WHERE s."id"=e.sid AND e.grade <= (SELECT AVG(e2.grade) - 1 FROM exams e2 WHERE s."id"=e2.sid OR (e2.curriculum=s.major AND s."year">=e2."year")) ORDER BY "name", course; -Error: 3001(EngineExecuteQuery), Error during planning: Correlated scalar subquery can only be used in Projection, Filter, Aggregate plan nodes +Error: 1001(Unsupported), This feature is not implemented: Physical plan does not support logical expression ScalarSubquery() -- Test 3: EXISTS subquery SELECT "name", major diff --git a/tests/cases/standalone/common/subquery/offset.result b/tests/cases/standalone/common/subquery/offset.result index aa5398b38e..7c9d67651f 100644 --- a/tests/cases/standalone/common/subquery/offset.result +++ b/tests/cases/standalone/common/subquery/offset.result @@ -11,8 +11,11 @@ Affected Rows: 1 SELECT (SELECT c0 FROM temp_values OFFSET 1) as result; -++ -++ ++--------+ +| result | ++--------+ +| | ++--------+ -- Test with actual data SELECT (SELECT c0 FROM temp_values OFFSET 0) as result; diff --git a/tests/cases/standalone/common/system/information_schema.result b/tests/cases/standalone/common/system/information_schema.result index bb2c4413c3..7fdfed0def 100644 --- a/tests/cases/standalone/common/system/information_schema.result +++ b/tests/cases/standalone/common/system/information_schema.result @@ -726,8 +726,10 @@ Affected Rows: 0 -- test query filter for key_column_usage -- select * from KEY_COLUMN_USAGE where CONSTRAINT_NAME = 'TIME INDEX' and TABLE_SCHEMA != 'greptime_private'; -++ -++ ++--------------------+-------------------+-----------------+---------------+--------------------+--------------+------------+-------------+------------------+-------------------------------+-------------------------+-----------------------+------------------------+---------------------+ +| constraint_catalog | constraint_schema | constraint_name | table_catalog | real_table_catalog | table_schema | table_name | column_name | ordinal_position | position_in_unique_constraint | referenced_table_schema | referenced_table_name | referenced_column_name | greptime_index_type | ++--------------------+-------------------+-----------------+---------------+--------------------+--------------+------------+-------------+------------------+-------------------------------+-------------------------+-----------------------+------------------------+---------------------+ ++--------------------+-------------------+-----------------+---------------+--------------------+--------------+------------+-------------+------------------+-------------------------------+-------------------------+-----------------------+------------------------+---------------------+ select * from KEY_COLUMN_USAGE where CONSTRAINT_NAME != 'TIME INDEX' and TABLE_SCHEMA != 'greptime_private'; @@ -739,8 +741,10 @@ select * from KEY_COLUMN_USAGE where CONSTRAINT_NAME != 'TIME INDEX' and TABLE_S select * from KEY_COLUMN_USAGE where CONSTRAINT_NAME LIKE '%INDEX' and TABLE_SCHEMA != 'greptime_private'; -++ -++ ++--------------------+-------------------+-----------------+---------------+--------------------+--------------+------------+-------------+------------------+-------------------------------+-------------------------+-----------------------+------------------------+---------------------+ +| constraint_catalog | constraint_schema | constraint_name | table_catalog | real_table_catalog | table_schema | table_name | column_name | ordinal_position | position_in_unique_constraint | referenced_table_schema | referenced_table_name | referenced_column_name | greptime_index_type | ++--------------------+-------------------+-----------------+---------------+--------------------+--------------+------------+-------------+------------------+-------------------------------+-------------------------+-----------------------+------------------------+---------------------+ ++--------------------+-------------------+-----------------+---------------+--------------------+--------------+------------+-------------+------------------+-------------------------------+-------------------------+-----------------------+------------------------+---------------------+ select * from KEY_COLUMN_USAGE where CONSTRAINT_NAME NOT LIKE '%INDEX' and TABLE_SCHEMA != 'greptime_private'; @@ -752,8 +756,10 @@ select * from KEY_COLUMN_USAGE where CONSTRAINT_NAME NOT LIKE '%INDEX' and TABLE select * from KEY_COLUMN_USAGE where CONSTRAINT_NAME == 'TIME INDEX' AND CONSTRAINT_SCHEMA != 'my_db' and TABLE_SCHEMA != 'greptime_private'; -++ -++ ++--------------------+-------------------+-----------------+---------------+--------------------+--------------+------------+-------------+------------------+-------------------------------+-------------------------+-----------------------+------------------------+---------------------+ +| constraint_catalog | constraint_schema | constraint_name | table_catalog | real_table_catalog | table_schema | table_name | column_name | ordinal_position | position_in_unique_constraint | referenced_table_schema | referenced_table_name | referenced_column_name | greptime_index_type | ++--------------------+-------------------+-----------------+---------------+--------------------+--------------+------------+-------------+------------------+-------------------------------+-------------------------+-----------------------+------------------------+---------------------+ ++--------------------+-------------------+-----------------+---------------+--------------------+--------------+------------+-------------+------------------+-------------------------------+-------------------------+-----------------------+------------------------+---------------------+ -- schemata -- desc table schemata; @@ -854,8 +860,10 @@ DESC TABLE COLUMN_PRIVILEGES; SELECT * FROM COLUMN_PRIVILEGES; -++ -++ ++---------+---------------+--------------+------------+-------------+----------------+--------------+ +| grantee | table_catalog | table_schema | table_name | column_name | privilege_type | is_grantable | ++---------+---------------+--------------+------------+-------------+----------------+--------------+ ++---------+---------------+--------------+------------+-------------+----------------+--------------+ DESC TABLE COLUMN_STATISTICS; @@ -870,8 +878,10 @@ DESC TABLE COLUMN_STATISTICS; SELECT * FROM COLUMN_STATISTICS; -++ -++ ++-------------+------------+-------------+-----------+ +| schema_name | table_name | column_name | histogram | ++-------------+------------+-------------+-----------+ ++-------------+------------+-------------+-----------+ SELECT * FROM CHARACTER_SETS; @@ -910,8 +920,10 @@ DESC TABLE CHECK_CONSTRAINTS; SELECT * FROM CHECK_CONSTRAINTS; -++ -++ ++--------------------+-------------------+-----------------+--------------+ +| constraint_catalog | constraint_schema | constraint_name | check_clause | ++--------------------+-------------------+-----------------+--------------+ ++--------------------+-------------------+-----------------+--------------+ DESC TABLE REGION_PEERS; diff --git a/tests/cases/standalone/common/system/pg_catalog.result b/tests/cases/standalone/common/system/pg_catalog.result index a36f0fe606..5a582fe7fe 100644 --- a/tests/cases/standalone/common/system/pg_catalog.result +++ b/tests/cases/standalone/common/system/pg_catalog.result @@ -1112,21 +1112,70 @@ CREATE table foo Affected Rows: 0 -- SQLNESS PROTOCOL POSTGRES -SELECT attname, atttypid FROM pg_catalog.pg_class AS cls INNER JOIN +SELECT attr.attnum, attname, atttypid FROM pg_catalog.pg_class AS cls INNER JOIN pg_catalog.pg_attribute AS attr ON cls.oid = attr.attrelid INNER JOIN pg_catalog.pg_type AS typ ON attr.atttypid = typ.oid WHERE attr.attnum >= 0 AND cls.oid = 'foo'::regclass::oid ORDER BY attr.attnum; -+-----------+----------+ -| attname | atttypid | -+-----------+----------+ -| ts | 1114 | -| log_data | 25 | -| count_num | 20 | -+-----------+----------+ ++--------+-----------+----------+ +| attnum | attname | atttypid | ++--------+-----------+----------+ +| 1 | ts | 1114 | +| 2 | log_data | 25 | +| 3 | count_num | 20 | ++--------+-----------+----------+ -- SQLNESS PROTOCOL POSTGRES DROP TABLE foo; Affected Rows: 0 +-- array_upper / array_lower UDFs (DataFusion ships array_length but not these) +SELECT array_upper(ARRAY[1,2,3], 1), array_lower(ARRAY[5,6,7], 1); + ++--------------------------------------------------------------+--------------------------------------------------------------+ +| array_upper(make_array(Int64(1),Int64(2),Int64(3)),Int64(1)) | array_lower(make_array(Int64(5),Int64(6),Int64(7)),Int64(1)) | ++--------------------------------------------------------------+--------------------------------------------------------------+ +| 3 | 1 | ++--------------------------------------------------------------+--------------------------------------------------------------+ + +-- NULL semantics: out-of-range dim (dim < 1) yields NULL +SELECT array_upper(ARRAY[1,2], 0), array_lower(ARRAY[1,2], 0); + ++-----------------------------------------------------+-----------------------------------------------------+ +| array_upper(make_array(Int64(1),Int64(2)),Int64(0)) | array_lower(make_array(Int64(1),Int64(2)),Int64(0)) | ++-----------------------------------------------------+-----------------------------------------------------+ +| | | ++-----------------------------------------------------+-----------------------------------------------------+ + +-- generate_series with int4 bounds from array_upper must execute (widened to int8) +SELECT n FROM generate_series(1, array_upper(ARRAY[10,20,30], 1)) AS t(n); + ++---+ +| n | ++---+ +| 1 | +| 2 | +| 3 | ++---+ + +-- ADBC type-info predicate: retain bool, excluding zero receivers and arrays. +-- SQLNESS PROTOCOL POSTGRES +WITH type_info AS ( + SELECT oid, typname, typreceive, typbasetype, typrelid, typarray + FROM pg_catalog.pg_type + WHERE (typreceive != 0 OR typsend != 0) + AND typtype != 'r' + AND typreceive::TEXT != 'array_recv' +) +SELECT oid, typname, typreceive +FROM type_info +WHERE oid IN (16, 269, 1000) +ORDER BY oid; + ++-----+---------+------------+ +| oid | typname | typreceive | ++-----+---------+------------+ +| 16 | bool | boolrecv | ++-----+---------+------------+ + diff --git a/tests/cases/standalone/common/system/pg_catalog.sql b/tests/cases/standalone/common/system/pg_catalog.sql index 1dd97c9dc3..89f45d24bd 100644 --- a/tests/cases/standalone/common/system/pg_catalog.sql +++ b/tests/cases/standalone/common/system/pg_catalog.sql @@ -274,10 +274,33 @@ CREATE table foo ); -- SQLNESS PROTOCOL POSTGRES -SELECT attname, atttypid FROM pg_catalog.pg_class AS cls INNER JOIN +SELECT attr.attnum, attname, atttypid FROM pg_catalog.pg_class AS cls INNER JOIN pg_catalog.pg_attribute AS attr ON cls.oid = attr.attrelid INNER JOIN pg_catalog.pg_type AS typ ON attr.atttypid = typ.oid WHERE attr.attnum >= 0 AND cls.oid = 'foo'::regclass::oid ORDER BY attr.attnum; -- SQLNESS PROTOCOL POSTGRES DROP TABLE foo; + +-- array_upper / array_lower UDFs (DataFusion ships array_length but not these) +SELECT array_upper(ARRAY[1,2,3], 1), array_lower(ARRAY[5,6,7], 1); + +-- NULL semantics: out-of-range dim (dim < 1) yields NULL +SELECT array_upper(ARRAY[1,2], 0), array_lower(ARRAY[1,2], 0); + +-- generate_series with int4 bounds from array_upper must execute (widened to int8) +SELECT n FROM generate_series(1, array_upper(ARRAY[10,20,30], 1)) AS t(n); + +-- ADBC type-info predicate: retain bool, excluding zero receivers and arrays. +-- SQLNESS PROTOCOL POSTGRES +WITH type_info AS ( + SELECT oid, typname, typreceive, typbasetype, typrelid, typarray + FROM pg_catalog.pg_type + WHERE (typreceive != 0 OR typsend != 0) + AND typtype != 'r' + AND typreceive::TEXT != 'array_recv' +) +SELECT oid, typname, typreceive +FROM type_info +WHERE oid IN (16, 269, 1000) +ORDER BY oid; diff --git a/tests/cases/standalone/common/system/semantic_graph.result b/tests/cases/standalone/common/system/semantic_graph.result index e1728c4ae0..9b23c4a506 100644 --- a/tests/cases/standalone/common/system/semantic_graph.result +++ b/tests/cases/standalone/common/system/semantic_graph.result @@ -2,13 +2,17 @@ -- virtual tables: readable, but rejecting every DDL/DML path. select observed_at, entity_type, entity_id, scope from greptime_private.semantic_entities; -++ -++ ++-------------+-------------+-----------+-------+ +| observed_at | entity_type | entity_id | scope | ++-------------+-------------+-----------+-------+ ++-------------+-------------+-----------+-------+ select observed_at, src_id, dst_id, rel_type from greptime_private.semantic_relationships; -++ -++ ++-------------+--------+--------+----------+ +| observed_at | src_id | dst_id | rel_type | ++-------------+--------+--------+----------+ ++-------------+--------+--------+----------+ insert into greptime_private.semantic_entities (observed_at, entity_type, entity_id) values (now(), 'service', 'svc-a'); @@ -137,8 +141,10 @@ Affected Rows: 1 select entity_type, entity_id from greptime_private.semantic_entities order by entity_type, entity_id; -++ -++ ++-------------+-----------+ +| entity_type | entity_id | ++-------------+-----------+ ++-------------+-----------+ alter table graph_late_metrics set 'greptime.semantic.entity.service.id' = 'svc', 'greptime.semantic.entity.service.scope' = 'env'; @@ -167,8 +173,10 @@ Affected Rows: 0 select entity_type, entity_id from greptime_private.semantic_entities order by entity_type, entity_id; -++ -++ ++-------------+-----------+ +| entity_type | entity_id | ++-------------+-----------+ ++-------------+-----------+ drop table graph_late_metrics; @@ -362,8 +370,10 @@ Affected Rows: 4 select src_id from greptime_private.semantic_relationships order by src_id; -++ -++ ++--------+ +| src_id | ++--------+ ++--------+ -- DROP is allowed (nothing structural is lost: the next INSERT recreates the -- canonical table) and cleans up after this test. diff --git a/tests/cases/standalone/common/tql-explain-analyze/explain.result b/tests/cases/standalone/common/tql-explain-analyze/explain.result index 6b03513a11..fed75f0448 100644 --- a/tests/cases/standalone/common/tql-explain-analyze/explain.result +++ b/tests/cases/standalone/common/tql-explain-analyze/explain.result @@ -105,6 +105,7 @@ TQL EXPLAIN VERBOSE (0, 10, '5s') test; | analyzed_logical_plan_| SAME TEXT AS ABOVE_| | logical_plan after rewrite_set_comparison_| SAME TEXT AS ABOVE_| | logical_plan after optimize_unions_| SAME TEXT AS ABOVE_| +| logical_plan after unions_to_filter_| SAME TEXT AS ABOVE_| | logical_plan after simplify_expressions_| SAME TEXT AS ABOVE_| | logical_plan after replace_distinct_aggregate_| SAME TEXT AS ABOVE_| | logical_plan after eliminate_join_| SAME TEXT AS ABOVE_| @@ -137,6 +138,7 @@ TQL EXPLAIN VERBOSE (0, 10, '5s') test; | logical_plan after JsonTypeConcretizeRule_| SAME TEXT AS ABOVE_| | logical_plan after rewrite_set_comparison_| SAME TEXT AS ABOVE_| | logical_plan after optimize_unions_| SAME TEXT AS ABOVE_| +| logical_plan after unions_to_filter_| SAME TEXT AS ABOVE_| | logical_plan after simplify_expressions_| SAME TEXT AS ABOVE_| | logical_plan after replace_distinct_aggregate_| SAME TEXT AS ABOVE_| | logical_plan after eliminate_join_| SAME TEXT AS ABOVE_| @@ -185,16 +187,18 @@ TQL EXPLAIN VERBOSE (0, 10, '5s') test; | physical_plan after PassDistributionRule_| SAME TEXT AS ABOVE_| | physical_plan after PromqlTsidNarrowJoin_| SAME TEXT AS ABOVE_| | physical_plan after EnforceSorting_| SAME TEXT AS ABOVE_| -| physical_plan after EnforceDistribution_| SAME TEXT AS ABOVE_| +| physical_plan after WindowTopN_| SAME TEXT AS ABOVE_| +| physical_plan after EnsureRequirements_| SAME TEXT AS ABOVE_| | physical_plan after CombinePartialFinalAggregate_| SAME TEXT AS ABOVE_| -| physical_plan after EnforceSorting_| SAME TEXT AS ABOVE_| | physical_plan after OptimizeAggregateOrder_| SAME TEXT AS ABOVE_| | physical_plan after ProjectionPushdown_| SAME TEXT AS ABOVE_| | physical_plan after OutputRequirements_| MergeScanExec: REDACTED |_|_| | physical_plan after LimitAggregation_| SAME TEXT AS ABOVE_| | physical_plan after LimitPushPastWindows_| SAME TEXT AS ABOVE_| +| physical_plan after HashJoinBuffering_| SAME TEXT AS ABOVE_| | physical_plan after LimitPushdown_| SAME TEXT AS ABOVE_| +| physical_plan after TopKRepartition_| SAME TEXT AS ABOVE_| | physical_plan after ProjectionPushdown_| SAME TEXT AS ABOVE_| | physical_plan after PushdownSort_| SAME TEXT AS ABOVE_| | physical_plan after EnsureCooperative_| CooperativeExec_| @@ -256,6 +260,7 @@ TQL EXPLAIN VERBOSE (0, 10, '5s') test AS series; | analyzed_logical_plan_| SAME TEXT AS ABOVE_| | logical_plan after rewrite_set_comparison_| SAME TEXT AS ABOVE_| | logical_plan after optimize_unions_| SAME TEXT AS ABOVE_| +| logical_plan after unions_to_filter_| SAME TEXT AS ABOVE_| | logical_plan after simplify_expressions_| SAME TEXT AS ABOVE_| | logical_plan after replace_distinct_aggregate_| SAME TEXT AS ABOVE_| | logical_plan after eliminate_join_| SAME TEXT AS ABOVE_| @@ -289,6 +294,7 @@ TQL EXPLAIN VERBOSE (0, 10, '5s') test AS series; | logical_plan after JsonTypeConcretizeRule_| SAME TEXT AS ABOVE_| | logical_plan after rewrite_set_comparison_| SAME TEXT AS ABOVE_| | logical_plan after optimize_unions_| SAME TEXT AS ABOVE_| +| logical_plan after unions_to_filter_| SAME TEXT AS ABOVE_| | logical_plan after simplify_expressions_| SAME TEXT AS ABOVE_| | logical_plan after replace_distinct_aggregate_| SAME TEXT AS ABOVE_| | logical_plan after eliminate_join_| SAME TEXT AS ABOVE_| @@ -338,16 +344,18 @@ TQL EXPLAIN VERBOSE (0, 10, '5s') test AS series; | physical_plan after PassDistributionRule_| SAME TEXT AS ABOVE_| | physical_plan after PromqlTsidNarrowJoin_| SAME TEXT AS ABOVE_| | physical_plan after EnforceSorting_| SAME TEXT AS ABOVE_| -| physical_plan after EnforceDistribution_| SAME TEXT AS ABOVE_| +| physical_plan after WindowTopN_| SAME TEXT AS ABOVE_| +| physical_plan after EnsureRequirements_| SAME TEXT AS ABOVE_| | physical_plan after CombinePartialFinalAggregate_| SAME TEXT AS ABOVE_| -| physical_plan after EnforceSorting_| SAME TEXT AS ABOVE_| | physical_plan after OptimizeAggregateOrder_| SAME TEXT AS ABOVE_| | physical_plan after ProjectionPushdown_| SAME TEXT AS ABOVE_| | physical_plan after OutputRequirements_| MergeScanExec: REDACTED |_|_| | physical_plan after LimitAggregation_| SAME TEXT AS ABOVE_| | physical_plan after LimitPushPastWindows_| SAME TEXT AS ABOVE_| +| physical_plan after HashJoinBuffering_| SAME TEXT AS ABOVE_| | physical_plan after LimitPushdown_| SAME TEXT AS ABOVE_| +| physical_plan after TopKRepartition_| SAME TEXT AS ABOVE_| | physical_plan after ProjectionPushdown_| SAME TEXT AS ABOVE_| | physical_plan after PushdownSort_| SAME TEXT AS ABOVE_| | physical_plan after EnsureCooperative_| CooperativeExec_| @@ -439,6 +447,7 @@ TQL EXPLAIN VERBOSE (0, 10, '5s') test_nano; | analyzed_logical_plan_| SAME TEXT AS ABOVE_| | logical_plan after rewrite_set_comparison_| SAME TEXT AS ABOVE_| | logical_plan after optimize_unions_| SAME TEXT AS ABOVE_| +| logical_plan after unions_to_filter_| SAME TEXT AS ABOVE_| | logical_plan after simplify_expressions_| SAME TEXT AS ABOVE_| | logical_plan after replace_distinct_aggregate_| SAME TEXT AS ABOVE_| | logical_plan after eliminate_join_| SAME TEXT AS ABOVE_| @@ -472,6 +481,7 @@ TQL EXPLAIN VERBOSE (0, 10, '5s') test_nano; | logical_plan after JsonTypeConcretizeRule_| SAME TEXT AS ABOVE_| | logical_plan after rewrite_set_comparison_| SAME TEXT AS ABOVE_| | logical_plan after optimize_unions_| SAME TEXT AS ABOVE_| +| logical_plan after unions_to_filter_| SAME TEXT AS ABOVE_| | logical_plan after simplify_expressions_| SAME TEXT AS ABOVE_| | logical_plan after replace_distinct_aggregate_| SAME TEXT AS ABOVE_| | logical_plan after eliminate_join_| SAME TEXT AS ABOVE_| @@ -521,16 +531,18 @@ TQL EXPLAIN VERBOSE (0, 10, '5s') test_nano; | physical_plan after PassDistributionRule_| SAME TEXT AS ABOVE_| | physical_plan after PromqlTsidNarrowJoin_| SAME TEXT AS ABOVE_| | physical_plan after EnforceSorting_| SAME TEXT AS ABOVE_| -| physical_plan after EnforceDistribution_| SAME TEXT AS ABOVE_| +| physical_plan after WindowTopN_| SAME TEXT AS ABOVE_| +| physical_plan after EnsureRequirements_| SAME TEXT AS ABOVE_| | physical_plan after CombinePartialFinalAggregate_| SAME TEXT AS ABOVE_| -| physical_plan after EnforceSorting_| SAME TEXT AS ABOVE_| | physical_plan after OptimizeAggregateOrder_| SAME TEXT AS ABOVE_| | physical_plan after ProjectionPushdown_| SAME TEXT AS ABOVE_| | physical_plan after OutputRequirements_| MergeScanExec: REDACTED |_|_| | physical_plan after LimitAggregation_| SAME TEXT AS ABOVE_| | physical_plan after LimitPushPastWindows_| SAME TEXT AS ABOVE_| +| physical_plan after HashJoinBuffering_| SAME TEXT AS ABOVE_| | physical_plan after LimitPushdown_| SAME TEXT AS ABOVE_| +| physical_plan after TopKRepartition_| SAME TEXT AS ABOVE_| | physical_plan after ProjectionPushdown_| SAME TEXT AS ABOVE_| | physical_plan after PushdownSort_| SAME TEXT AS ABOVE_| | physical_plan after EnsureCooperative_| CooperativeExec_| diff --git a/tests/cases/standalone/common/tql/binary_operator.result b/tests/cases/standalone/common/tql/binary_operator.result index 47d524c952..9952615e6b 100644 --- a/tests/cases/standalone/common/tql/binary_operator.result +++ b/tests/cases/standalone/common/tql/binary_operator.result @@ -8,8 +8,10 @@ Affected Rows: 3 tql eval (0, 30, '10s'), data < 1; -++ -++ ++----+-----+ +| ts | val | ++----+-----+ ++----+-----+ tql eval (0, 30, '10s'), data + (1 < bool 2); diff --git a/tests/cases/standalone/common/tql/case_sensitive.result b/tests/cases/standalone/common/tql/case_sensitive.result index 2da2473690..de3ef60a40 100644 --- a/tests/cases/standalone/common/tql/case_sensitive.result +++ b/tests/cases/standalone/common/tql/case_sensitive.result @@ -60,14 +60,18 @@ Affected Rows: 0 tql eval (0,10,'5s') sum(MemAvailable / 4) + sum(MemTotal / 4); -++ -++ ++------+---------------------------------------------------------------+ +| time | MemAvailable.sum(val / Float64(4)) + .sum(value / Float64(4)) | ++------+---------------------------------------------------------------+ ++------+---------------------------------------------------------------+ -- Cross schema is not supported tql eval (0,10,'5s') sum(MemAvailable / 4) + sum({__name__="AnotherSchema.MemTotal"} / 4); -++ -++ ++------+---------------------------------------------------------------+ +| time | MemAvailable.sum(val / Float64(4)) + .sum(value / Float64(4)) | ++------+---------------------------------------------------------------+ ++------+---------------------------------------------------------------+ drop table "MemAvailable"; diff --git a/tests/cases/standalone/common/tql/partition.result b/tests/cases/standalone/common/tql/partition.result index e32d13043b..ac3072342d 100644 --- a/tests/cases/standalone/common/tql/partition.result +++ b/tests/cases/standalone/common/tql/partition.result @@ -68,9 +68,9 @@ tql analyze (0, 10, '1s') 100 - (avg by (k) (irate(t[1m])) * 100); |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_SortPreservingMergeExec: [k@0 ASC NULLS LAST, j@1 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[k@0 ASC NULLS LAST, j@1 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[k@0 as k, j@1 as j], aggr=[avg(prom_irate(j_range,i))] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[k@0 as k, j@1 as j], aggr=[__avg_merge(__avg_state(prom_irate(j_range,i))) as avg(prom_irate(j_range,i))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[k@0 as k, j@1 as j], aggr=[avg(prom_irate(j_range,i))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[k@0 as k, j@1 as j], aggr=[__avg_merge(__avg_state(prom_irate(j_range,i))) as avg(prom_irate(j_range,i))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| diff --git a/tests/cases/standalone/common/tql/range.result b/tests/cases/standalone/common/tql/range.result index c650222448..4eab8d6dc5 100644 --- a/tests/cases/standalone/common/tql/range.result +++ b/tests/cases/standalone/common/tql/range.result @@ -177,44 +177,58 @@ TQL EVAL (60, 180, '60s') sum by(host) (rate(metrics[1m])) * 60; -- Test querying non-existent table TQL EVAL (60, 180, '60s') sum(rate(non_existent_table[1m])); -++ -++ ++------+----------------------------------------------------+ +| time | sum(prom_rate(time_range,value,time,Int64(60000))) | ++------+----------------------------------------------------+ ++------+----------------------------------------------------+ -- Test querying non-existent label TQL EVAL (60, 180, '60s') sum(rate(metrics{non_existent_label="value"}[1m])); -++ -++ ++----+----------------------------------------------+ +| ts | sum(prom_rate(ts_range,val,ts,Int64(60000))) | ++----+----------------------------------------------+ ++----+----------------------------------------------+ -- Test querying non-existent label value TQL EVAL (60, 180, '60s') sum(rate(metrics{host="non_existent_host"}[1m])); -++ -++ ++----+----------------------------------------------+ +| ts | sum(prom_rate(ts_range,val,ts,Int64(60000))) | ++----+----------------------------------------------+ ++----+----------------------------------------------+ -- Test querying multiple non-existent labels TQL EVAL (60, 180, '60s') sum(rate(metrics{non_existent_label1="value1", non_existent_label2="value2"}[1m])); -++ -++ ++----+----------------------------------------------+ +| ts | sum(prom_rate(ts_range,val,ts,Int64(60000))) | ++----+----------------------------------------------+ ++----+----------------------------------------------+ -- Test querying mix of existing and non-existent labels TQL EVAL (60, 180, '60s') sum(rate(metrics{host="host1", non_existent_label="value"}[1m])); -++ -++ ++----+----------------------------------------------+ +| ts | sum(prom_rate(ts_range,val,ts,Int64(60000))) | ++----+----------------------------------------------+ ++----+----------------------------------------------+ -- Test querying non-existent table with non-existent labels TQL EVAL (60, 180, '60s') sum(rate(non_existent_table{non_existent_label="value"}[1m])); -++ -++ ++------+----------------------------------------------------+ +| time | sum(prom_rate(time_range,value,time,Int64(60000))) | ++------+----------------------------------------------------+ ++------+----------------------------------------------------+ -- Test querying non-existent table with multiple non-existent labels TQL EVAL (60, 180, '60s') sum(rate(non_existent_table{label1="value1", label2="value2"}[1m])); -++ -++ ++------+----------------------------------------------------+ +| time | sum(prom_rate(time_range,value,time,Int64(60000))) | ++------+----------------------------------------------------+ ++------+----------------------------------------------------+ DROP TABLE metrics; diff --git a/tests/cases/standalone/common/tql/tql-cte.result b/tests/cases/standalone/common/tql/tql-cte.result index 8efe983763..7fbc621319 100644 --- a/tests/cases/standalone/common/tql/tql-cte.result +++ b/tests/cases/standalone/common/tql/tql-cte.result @@ -201,7 +201,7 @@ SELECT sum(val) FROM filtered; | | Projection: tql_data.ts, tql_data.val | | | SubqueryAlias: tql_data | | | Projection: metric.ts AS ts, prom_rate(ts_range,val,ts,Int64(20000)) AS val | -| | Filter: prom_rate(ts_range,val,ts,Int64(20000)) > Float64(0) AND prom_rate(ts_range,val,ts,Int64(20000)) IS NOT NULL | +| | Filter: prom_rate(ts_range,val,ts,Int64(20000)) IS NOT NULL AND prom_rate(ts_range,val,ts,Int64(20000)) > Float64(0) | | | Projection: metric.ts, prom_rate(ts_range, val, metric.ts, Int64(20000)) AS prom_rate(ts_range,val,ts,Int64(20000)) | | | PromRangeManipulate: req range=[0..40000], interval=[10000], eval range=[20000], time index=[ts], values=["val"] | | | PromSeriesNormalize: offset=[0], time index=[ts], filter NaN: [true] | @@ -244,8 +244,8 @@ SELECT round(avg(summary)) as avg_sum FROM tql_agg; | | Aggregate: groupBy=[[labels.ts]], aggr=[[sum(labels.cpu)]] | | | PromInstantManipulate: range=[0..40000], lookback=[300000], interval=[10000], time index=[ts] | | | PromSeriesDivide: tags=["host"] | -| | Filter: labels.host ~ Utf8("^(?:host.*)$") AND labels.ts >= TimestampMillisecond(-299999, None) AND labels.ts <= TimestampMillisecond(40000, None) | -| | TableScan: labels, partial_filters=[labels.host ~ Utf8("^(?:host.*)$"), labels.ts >= TimestampMillisecond(-299999, None), labels.ts <= TimestampMillisecond(40000, None)] | +| | Filter: labels.ts >= TimestampMillisecond(-299999, None) AND labels.ts <= TimestampMillisecond(40000, None) AND labels.host ~ Utf8("^(?:host.*)$") | +| | TableScan: labels, partial_filters=[labels.ts >= TimestampMillisecond(-299999, None), labels.ts <= TimestampMillisecond(40000, None), labels.host ~ Utf8("^(?:host.*)$")] | | | ]] | | physical_plan | CooperativeExec | | | MergeScanExec: REDACTED @@ -389,8 +389,8 @@ LIMIT 3; | physical_plan | ProjectionExec: expr=[metric_val@0 as metric_val, label_val@1 as label_val] | | | SortPreservingMergeExec: [ts@2 ASC NULLS LAST], fetch=3 | | | SortExec: TopK(fetch=3), expr=[ts@2 ASC NULLS LAST], preserve_REDACTED -| | ProjectionExec: expr=[val@1 as metric_val, cpu@2 as label_val, ts@0 as ts] | -| | HashJoinExec: mode=Partitioned, join_type=Inner, on=[(ts@0, ts@0)], projection=[ts@0, val@1, cpu@3] | +| | ProjectionExec: expr=[val@0 as metric_val, cpu@1 as label_val, ts@2 as ts] | +| | HashJoinExec: mode=Partitioned, join_type=Inner, on=[(ts@0, ts@0)], projection=[val@1, cpu@3, ts@0] | | | RepartitionExec: REDACTED | | MergeScanExec: REDACTED | | RepartitionExec: REDACTED @@ -651,7 +651,7 @@ SELECT count(*) as high_values FROM final; | | Projection: base_tql.ts, base_tql.val * Float64(100) AS percent | | | SubqueryAlias: base_tql | | | Projection: metric.ts AS ts, metric.val AS val | -| | Filter: metric.val * Float64(100) > Float64(200) AND metric.val > Float64(0) | +| | Filter: metric.val > Float64(0) AND metric.val * Float64(100) > Float64(200) | | | PromInstantManipulate: range=[0..40000], lookback=[300000], interval=[10000], time index=[ts] | | | PromSeriesDivide: tags=[] | | | Filter: metric.ts >= TimestampMillisecond(-299999, None) AND metric.ts <= TimestampMillisecond(40000, None) | @@ -668,8 +668,10 @@ WITH time_shifted AS ( ) SELECT * FROM time_shifted; -++ -++ ++----+-----+ +| ts | val | ++----+-----+ ++----+-----+ -- SQLNESS REPLACE (peers.*) REDACTED -- SQLNESS REPLACE (partitioning.*) REDACTED @@ -844,8 +846,8 @@ LIMIT 5; | | ]] | | physical_plan | SortPreservingMergeExec: [ts@0 ASC NULLS LAST, host@2 ASC NULLS LAST, avg_value@1 ASC NULLS LAST], fetch=5 | | | SortExec: TopK(fetch=5), expr=[ts@0 ASC NULLS LAST, host@2 ASC NULLS LAST, avg_value@1 ASC NULLS LAST], preserve_REDACTED -| | ProjectionExec: expr=[ts@1 as ts, cpu@0 as avg_value, host@2 as host] | -| | HashJoinExec: mode=Partitioned, join_type=Inner, on=[(date_trunc(Utf8("second"),t.ts)@2, date_trunc(Utf8("second"),l.ts)@2)], projection=[cpu@0, ts@1, host@4] | +| | ProjectionExec: expr=[ts@0 as ts, cpu@1 as avg_value, host@2 as host] | +| | HashJoinExec: mode=Partitioned, join_type=Inner, on=[(date_trunc(Utf8("second"),t.ts)@2, date_trunc(Utf8("second"),l.ts)@2)], projection=[ts@1, cpu@0, host@4] | | | RepartitionExec: REDACTED | | ProjectionExec: expr=[cpu@0 as cpu, ts@1 as ts, date_trunc(second, ts@1) as date_trunc(Utf8("second"),t.ts)] | | | RepartitionExec: REDACTED diff --git a/tests/cases/standalone/common/truncate/truncate.result b/tests/cases/standalone/common/truncate/truncate.result index 6cd490e26d..27fcbd9026 100644 --- a/tests/cases/standalone/common/truncate/truncate.result +++ b/tests/cases/standalone/common/truncate/truncate.result @@ -41,8 +41,10 @@ Affected Rows: 0 SELECT ts, host, cpu, memory FROM monitor ORDER BY ts; -++ -++ ++----+------+-----+--------+ +| ts | host | cpu | memory | ++----+------+-----+--------+ ++----+------+-----+--------+ -- truncate with time range INSERT INTO monitor(ts, host, cpu, memory) VALUES @@ -102,8 +104,10 @@ Affected Rows: 0 SELECT ts, host, cpu, memory FROM monitor ORDER BY ts; -++ -++ ++----+------+-----+--------+ +| ts | host | cpu | memory | ++----+------+-----+--------+ ++----+------+-----+--------+ INSERT INTO monitor(ts, host, cpu, memory) VALUES (1695217660000, 'host1', 88.8, 4096), @@ -128,8 +132,10 @@ Affected Rows: 0 SELECT ts, host, cpu, memory FROM monitor ORDER BY ts; -++ -++ ++----+------+-----+--------+ +| ts | host | cpu | memory | ++----+------+-----+--------+ ++----+------+-----+--------+ DROP TABLE monitor; diff --git a/tests/cases/standalone/common/ttl/alter_table_ttl.result b/tests/cases/standalone/common/ttl/alter_table_ttl.result index c93610907d..ac7c671064 100644 --- a/tests/cases/standalone/common/ttl/alter_table_ttl.result +++ b/tests/cases/standalone/common/ttl/alter_table_ttl.result @@ -61,8 +61,10 @@ ADMIN compact_table('test_ttl'); SELECT val from test_ttl; -++ -++ ++-----+ +| val | ++-----+ ++-----+ ALTER TABLE test_ttl SET ttl = '1 minute'; diff --git a/tests/cases/standalone/common/ttl/basic.result b/tests/cases/standalone/common/ttl/basic.result index b381b97075..1b24cf16a9 100644 --- a/tests/cases/standalone/common/ttl/basic.result +++ b/tests/cases/standalone/common/ttl/basic.result @@ -34,8 +34,10 @@ ADMIN compact_table('test_ttl'); SELECT val from test_ttl; -++ -++ ++-----+ +| val | ++-----+ ++-----+ DROP TABLE test_ttl; diff --git a/tests/cases/standalone/common/ttl/database_ttl.result b/tests/cases/standalone/common/ttl/database_ttl.result index 81d1edef78..cce95a9f15 100644 --- a/tests/cases/standalone/common/ttl/database_ttl.result +++ b/tests/cases/standalone/common/ttl/database_ttl.result @@ -44,8 +44,10 @@ ADMIN compact_table('test_ttl'); -- Must be expired -- SELECT val from test_ttl; -++ -++ ++-----+ +| val | ++-----+ ++-----+ ALTER DATABASE test_ttl_db SET ttl = '1 day'; diff --git a/tests/cases/standalone/common/ttl/database_ttl_with_metric_engine.result b/tests/cases/standalone/common/ttl/database_ttl_with_metric_engine.result index ad59b7bd1a..0eacfccf46 100644 --- a/tests/cases/standalone/common/ttl/database_ttl_with_metric_engine.result +++ b/tests/cases/standalone/common/ttl/database_ttl_with_metric_engine.result @@ -52,8 +52,10 @@ ADMIN compact_table('phy'); --- should be expired -- SELECT val, host FROM test_ttl; -++ -++ ++-----+------+ +| val | host | ++-----+------+ ++-----+------+ ALTER DATABASE test_ttl_db SET ttl = '1 day'; diff --git a/tests/cases/standalone/common/ttl/metric_engine_ttl.result b/tests/cases/standalone/common/ttl/metric_engine_ttl.result index 6152c0cd58..3f0ac57d87 100644 --- a/tests/cases/standalone/common/ttl/metric_engine_ttl.result +++ b/tests/cases/standalone/common/ttl/metric_engine_ttl.result @@ -43,8 +43,10 @@ ADMIN compact_table('phy'); --- should be expired -- SELECT val, host FROM test_ttl; -++ -++ ++-----+------+ +| val | host | ++-----+------+ ++-----+------+ ALTER TABLE phy SET ttl = '1 day'; diff --git a/tests/cases/standalone/common/ttl/ttl_instant.result b/tests/cases/standalone/common/ttl/ttl_instant.result index 49913fd90e..110fd92d92 100644 --- a/tests/cases/standalone/common/ttl/ttl_instant.result +++ b/tests/cases/standalone/common/ttl/ttl_instant.result @@ -40,8 +40,10 @@ from ORDER BY val; -++ -++ ++-----+ +| val | ++-----+ ++-----+ -- SQLNESS SLEEP 2s ADMIN flush_table('test_ttl'); @@ -67,8 +69,10 @@ from ORDER BY val; -++ -++ ++-----+ +| val | ++-----+ ++-----+ ALTER TABLE test_ttl UNSET 'ttl'; @@ -215,8 +219,10 @@ from ORDER BY val; -++ -++ ++-----+ +| val | ++-----+ ++-----+ ALTER TABLE test_ttl @@ -279,8 +285,10 @@ from ORDER BY val; -++ -++ ++-----+ +| val | ++-----+ ++-----+ -- to make sure alter back and forth from duration to/from instant wouldn't break anything ALTER TABLE @@ -345,8 +353,10 @@ from ORDER BY val; -++ -++ ++-----+ +| val | ++-----+ ++-----+ DROP TABLE test_ttl; diff --git a/tests/cases/standalone/common/types/float/ieee_floating_points.result b/tests/cases/standalone/common/types/float/ieee_floating_points.result index 69198d490e..0432048dd5 100644 --- a/tests/cases/standalone/common/types/float/ieee_floating_points.result +++ b/tests/cases/standalone/common/types/float/ieee_floating_points.result @@ -58,8 +58,10 @@ SELECT d, d < -1000000 FROM float_special ORDER BY ts; -- NaN != NaN SELECT f, f = f FROM float_special WHERE f != f ORDER BY ts; -++ -++ ++---+-----------------------------------+ +| f | float_special.f = float_special.f | ++---+-----------------------------------+ ++---+-----------------------------------+ SELECT d, d IS NULL FROM float_special ORDER BY ts; diff --git a/tests/cases/standalone/common/types/float/infinity.result b/tests/cases/standalone/common/types/float/infinity.result index c35287f1cd..c27c9a53ee 100644 --- a/tests/cases/standalone/common/types/float/infinity.result +++ b/tests/cases/standalone/common/types/float/infinity.result @@ -127,8 +127,10 @@ SELECT f FROM floats WHERE f>'-inf'::FLOAT ORDER BY 1; SELECT f FROM floats WHERE f>'inf'::FLOAT; -++ -++ ++---+ +| f | ++---+ ++---+ -- >= SELECT f FROM floats WHERE f>=1 ORDER BY f; @@ -178,8 +180,10 @@ SELECT f FROM floats WHERE f<'inf'::FLOAT ORDER BY f; SELECT f FROM floats WHERE f<'-inf'::FLOAT; -++ -++ ++---+ +| f | ++---+ ++---+ -- <= SELECT f FROM floats WHERE f<=1 ORDER BY f; @@ -341,8 +345,10 @@ SELECT d FROM doubles WHERE d>'-inf'::DOUBLE ORDER BY 1; SELECT d FROM doubles WHERE d>'inf'::DOUBLE; -++ -++ ++---+ +| d | ++---+ ++---+ -- >= SELECT d FROM doubles WHERE d>=1 ORDER BY d; @@ -392,8 +398,10 @@ SELECT d FROM doubles WHERE d<'inf'::DOUBLE ORDER BY d; SELECT d FROM doubles WHERE d<'-inf'::DOUBLE; -++ -++ ++---+ +| d | ++---+ ++---+ -- <= SELECT d FROM doubles WHERE d<=1 ORDER BY d; diff --git a/tests/cases/standalone/common/types/float/nan.result b/tests/cases/standalone/common/types/float/nan.result index 9ab48fe88e..e7472e7da6 100644 --- a/tests/cases/standalone/common/types/float/nan.result +++ b/tests/cases/standalone/common/types/float/nan.result @@ -100,8 +100,10 @@ SELECT f FROM floats WHERE f>0; SELECT f FROM floats WHERE f>'nan'::FLOAT; -++ -++ ++---+ +| f | ++---+ ++---+ -- >= SELECT f FROM floats WHERE f>=1; @@ -124,8 +126,10 @@ SELECT f FROM floats WHERE f>='nan'::FLOAT; -- < SELECT f FROM floats WHERE f<1; -++ -++ ++---+ +| f | ++---+ ++---+ SELECT f FROM floats WHERE f<'nan'::FLOAT; @@ -257,8 +261,10 @@ SELECT d FROM doubles WHERE d>0; SELECT d FROM doubles WHERE d>'nan'::DOUBLE; -++ -++ ++---+ +| d | ++---+ ++---+ -- >= SELECT d FROM doubles WHERE d>=1; @@ -281,8 +287,10 @@ SELECT d FROM doubles WHERE d>='nan'::DOUBLE; -- < SELECT d FROM doubles WHERE d<1; -++ -++ ++---+ +| d | ++---+ ++---+ SELECT d FROM doubles WHERE d<'nan'::DOUBLE; diff --git a/tests/cases/standalone/common/types/json/json.result b/tests/cases/standalone/common/types/json/json.result index 8fad9632b1..541606a3ed 100644 --- a/tests/cases/standalone/common/types/json/json.result +++ b/tests/cases/standalone/common/types/json/json.result @@ -131,8 +131,10 @@ Error: 3001(EngineExecuteQuery), Execution error: cannot parse 'Morning my frien SELECT json_to_string(j), t FROM jsons; -++ -++ ++-------------------------+---+ +| json_to_string(jsons.j) | t | ++-------------------------+---+ ++-------------------------+---+ CREATE TABLE json_empty (j JSON, t timestamp time index); diff --git a/tests/cases/standalone/common/types/timestamp/timestamp.result b/tests/cases/standalone/common/types/timestamp/timestamp.result index 50cba0eac4..9ce5c3d853 100644 --- a/tests/cases/standalone/common/types/timestamp/timestamp.result +++ b/tests/cases/standalone/common/types/timestamp/timestamp.result @@ -75,21 +75,22 @@ SELECT MAX(t) FROM timestamp; SELECT SUM(t) FROM timestamp; -Error: 3000(PlanQuery), Failed to plan SQL: Error during planning: Execution error: Function 'sum' failed to match any signature, errors: Error during planning: Function 'sum' requires TypeSignatureClass::Decimal, but received Timestamp(ms) (DataType: Timestamp(ms)).,Error during planning: Function 'sum' requires TypeSignatureClass::Native(LogicalType(Native(UInt64), UInt64)), but received Timestamp(ms) (DataType: Timestamp(ms)).,Error during planning: Function 'sum' requires TypeSignatureClass::Native(LogicalType(Native(Int64), Int64)), but received Timestamp(ms) (DataType: Timestamp(ms)).,Error during planning: Function 'sum' requires TypeSignatureClass::Native(LogicalType(Native(Float64), Float64)), but received Timestamp(ms) (DataType: Timestamp(ms)).,Error during planning: Function 'sum' requires TypeSignatureClass::Duration, but received Timestamp(ms) (DataType: Timestamp(ms)). No function matches the given name and argument types 'sum(Timestamp(ms))'. You might need to add explicit type casts. +Error: 3000(PlanQuery), Failed to plan SQL: Error during planning: Execution error: Function 'sum' failed to match any signature, errors: Error during planning: Function 'sum' requires Decimal, but received Timestamp(ms) (DataType: Timestamp(ms)).,Error during planning: Function 'sum' requires UInt64, but received Timestamp(ms) (DataType: Timestamp(ms)).,Error during planning: Function 'sum' requires Int64, but received Timestamp(ms) (DataType: Timestamp(ms)).,Error during planning: Function 'sum' requires Float64, but received Timestamp(ms) (DataType: Timestamp(ms)).,Error during planning: Function 'sum' requires Duration, but received Timestamp(ms) (DataType: Timestamp(ms)).,Error during planning: Function 'sum' requires Interval, but received Timestamp(ms) (DataType: Timestamp(ms)).. No function matches the given name and argument types 'sum(Timestamp(ms))'. You might need to add explicit type casts. Candidate functions: - sum(Coercion(TypeSignatureClass::Decimal)) - sum(Coercion(TypeSignatureClass::Native(LogicalType(Native(UInt64), UInt64)), implicit_coercion=ImplicitCoercion([Native(LogicalType(Native(UInt8), UInt8)), Native(LogicalType(Native(UInt16), UInt16)), Native(LogicalType(Native(UInt32), UInt32))], default_type=UInt64)) - sum(Coercion(TypeSignatureClass::Native(LogicalType(Native(Int64), Int64)), implicit_coercion=ImplicitCoercion([Native(LogicalType(Native(Int8), Int8)), Native(LogicalType(Native(Int16), Int16)), Native(LogicalType(Native(Int32), Int32))], default_type=Int64)) - sum(Coercion(TypeSignatureClass::Native(LogicalType(Native(Float64), Float64)), implicit_coercion=ImplicitCoercion([Float], default_type=Float64)) - sum(Coercion(TypeSignatureClass::Duration)) + sum(Decimal) + sum(UInt64) + sum(Int64) + sum(Float64) + sum(Duration) + sum(Interval) SELECT AVG(t) FROM timestamp; -Error: 3000(PlanQuery), Failed to plan SQL: Error during planning: Execution error: Function 'avg' failed to match any signature, errors: Error during planning: Function 'avg' requires TypeSignatureClass::Decimal, but received Timestamp(ms) (DataType: Timestamp(ms)).,Error during planning: Function 'avg' requires TypeSignatureClass::Duration, but received Timestamp(ms) (DataType: Timestamp(ms)).,Error during planning: Function 'avg' requires TypeSignatureClass::Native(LogicalType(Native(Float64), Float64)), but received Timestamp(ms) (DataType: Timestamp(ms)). No function matches the given name and argument types 'avg(Timestamp(ms))'. You might need to add explicit type casts. +Error: 3000(PlanQuery), Failed to plan SQL: Error during planning: Execution error: Function 'avg' failed to match any signature, errors: Error during planning: Function 'avg' requires Decimal, but received Timestamp(ms) (DataType: Timestamp(ms)).,Error during planning: Function 'avg' requires Duration, but received Timestamp(ms) (DataType: Timestamp(ms)).,Error during planning: Function 'avg' requires Float64, but received Timestamp(ms) (DataType: Timestamp(ms)).. No function matches the given name and argument types 'avg(Timestamp(ms))'. You might need to add explicit type casts. Candidate functions: - avg(Coercion(TypeSignatureClass::Decimal)) - avg(Coercion(TypeSignatureClass::Duration)) - avg(Coercion(TypeSignatureClass::Native(LogicalType(Native(Float64), Float64)), implicit_coercion=ImplicitCoercion([Integer, Float], default_type=Float64)) + avg(Decimal) + avg(Duration) + avg(Float64) SELECT t+t FROM timestamp; diff --git a/tests/cases/standalone/common/view/columns.result b/tests/cases/standalone/common/view/columns.result index 7184cd3da1..b9e9f819b3 100644 --- a/tests/cases/standalone/common/view/columns.result +++ b/tests/cases/standalone/common/view/columns.result @@ -58,7 +58,8 @@ SELECT a FROM v1; SELECT n FROM v1; -Error: 3000(PlanQuery), Failed to plan SQL: No field named n. Valid fields are v1.a. +Error: 3000(PlanQuery), Failed to plan SQL: No field named n. +Valid fields are v1.a. CREATE OR REPLACE VIEW v1 (a, b) AS SELECT n, n+1 FROM t1; @@ -167,11 +168,13 @@ SELECT a,b FROM v1; SELECT n FROM v1; -Error: 3000(PlanQuery), Failed to plan SQL: No field named n. Valid fields are v1.a, v1.b. +Error: 3000(PlanQuery), Failed to plan SQL: No field named n. +Valid fields are v1.a, v1.b. SELECT * FROM v1 WHERE n > 5; -Error: 3000(PlanQuery), Failed to plan SQL: No field named n. Valid fields are v1.a, v1.b. +Error: 3000(PlanQuery), Failed to plan SQL: No field named n. +Valid fields are v1.a, v1.b. -- test view after altering table t1 -- CREATE OR REPLACE VIEW v1 AS SELECT n, ts FROM t1 LIMIT 5; @@ -212,7 +215,8 @@ Affected Rows: 0 SELECT * FROM v1; -Error: 1002(Unexpected), Failed to decode DataFusion plan: No field named n. Valid fields are greptime.public.t1.ts, greptime.public.t1.s. +Error: 1002(Unexpected), Failed to decode DataFusion plan: No field named n. +Valid fields are greptime.public.t1.ts, greptime.public.t1.s. DROP VIEW v1; diff --git a/tests/cases/standalone/common/view/create.result b/tests/cases/standalone/common/view/create.result index c1e49fb3ae..577583515e 100644 --- a/tests/cases/standalone/common/view/create.result +++ b/tests/cases/standalone/common/view/create.result @@ -157,18 +157,24 @@ SELECT * FROM INFORMATION_SCHEMA.TABLES WHERE TABLE_TYPE = 'VIEW' ORDER BY TABLE SHOW COLUMNS FROM test_view; -++ -++ ++-------+------+------+-----+---------+-------+---------------+ +| Field | Type | Null | Key | Default | Extra | Greptime_type | ++-------+------+------+-----+---------+-------+---------------+ ++-------+------+------+-----+---------+-------+---------------+ SHOW FULL COLUMNS FROM test_view; -++ -++ ++-------+------+-----------+------+-----+---------+---------+------------+-------+---------------+ +| Field | Type | Collation | Null | Key | Default | Comment | Privileges | Extra | Greptime_type | ++-------+------+-----------+------+-----+---------+---------+------------+-------+---------------+ ++-------+------+-----------+------+-----+---------+---------+------------+-------+---------------+ SELECT * FROM INFORMATION_SCHEMA.COLUMNS WHERE TABLE_NAME = 'test_view'; -++ -++ ++---------------+--------------+------------+-------------+------------------+--------------------------+------------------------+-------------------+---------------+--------------------+--------------------+----------------+------------+-------+------------+-----------------------+--------------------+-----------+---------------+----------------+-------------+-------------+----------------+--------+ +| table_catalog | table_schema | table_name | column_name | ordinal_position | character_maximum_length | character_octet_length | numeric_precision | numeric_scale | datetime_precision | character_set_name | collation_name | column_key | extra | privileges | generation_expression | greptime_data_type | data_type | semantic_type | column_default | is_nullable | column_type | column_comment | srs_id | ++---------------+--------------+------------+-------------+------------------+--------------------------+------------------------+-------------------+---------------+--------------------+--------------------+----------------+------------+-------+------------+-----------------------+--------------------+-----------+---------------+----------------+-------------+-------------+----------------+--------+ ++---------------+--------------+------------+-------------+------------------+--------------------------+------------------------+-------------------+---------------+--------------------+--------------------+----------------+------------+-------+------------+-----------------------+--------------------+-----------+---------------+----------------+-------------+-------------+----------------+--------+ SELECT * FROM test_view LIMIT 10; diff --git a/tests/cases/standalone/common/view/view.result b/tests/cases/standalone/common/view/view.result index 21d674e54e..701a5d0abd 100644 --- a/tests/cases/standalone/common/view/view.result +++ b/tests/cases/standalone/common/view/view.result @@ -58,8 +58,10 @@ Error: 4001(TableNotFound), Failed to plan SQL: Table not found: greptime.public SHOW VIEWS; -++ -++ ++-------+ +| Views | ++-------+ ++-------+ DROP VIEW v1; @@ -92,6 +94,8 @@ SHOW TABLES; SHOW VIEWS; -++ -++ ++-------+ +| Views | ++-------+ ++-------+ diff --git a/tests/cases/standalone/copy/copy_database_from_fs_parquet.result b/tests/cases/standalone/copy/copy_database_from_fs_parquet.result index 3ec38aa7ca..662355099c 100644 --- a/tests/cases/standalone/copy/copy_database_from_fs_parquet.result +++ b/tests/cases/standalone/copy/copy_database_from_fs_parquet.result @@ -20,8 +20,10 @@ Affected Rows: 2 SELECT * FROM demo ORDER BY ts; -++ -++ ++------+-----+--------+----+ +| host | cpu | memory | ts | ++------+-----+--------+----+ ++------+-----+--------+----+ COPY DATABASE public FROM '${SQLNESS_HOME}/demo/export/parquet/'; diff --git a/tests/cases/standalone/flow-tql/flow_tql.result b/tests/cases/standalone/flow-tql/flow_tql.result index 50e1d000bc..de93fb7c51 100644 --- a/tests/cases/standalone/flow-tql/flow_tql.result +++ b/tests/cases/standalone/flow-tql/flow_tql.result @@ -43,8 +43,10 @@ SHOW CREATE TABLE cnt_reqs; -- test if sink table is tql queryable TQL EVAL (now() - '1m'::interval, now(), '5s') count_values("status_code", cnt_reqs); -++ -++ ++------------------------------------------+----+-------------+ +| count(cnt_reqs.count(http_requests.val)) | ts | status_code | ++------------------------------------------+----+-------------+ ++------------------------------------------+----+-------------+ INSERT INTO TABLE http_requests VALUES (now() - '17s'::interval, 'host1', 'idc1', 200), @@ -187,8 +189,10 @@ SHOW CREATE TABLE cnt_reqs; -- test if sink table is tql queryable TQL EVAL (now() - '1m'::interval, now(), '5s') count_values("status_code", cnt_reqs); -++ -++ ++------------------------------------------+----+-------------+ +| count(cnt_reqs.count(http_requests.val)) | ts | status_code | ++------------------------------------------+----+-------------+ ++------------------------------------------+----+-------------+ INSERT INTO TABLE http_requests VALUES (0::Timestamp, 'host1', 'idc1', 200), @@ -278,8 +282,10 @@ SHOW CREATE TABLE rate_reqs; -- test if sink table is tql queryable TQL EVAL (now() - '1m'::interval, now(), '5s') count_values("status_code", rate_reqs); -++ -++ ++-----------------------------------------------------------+----+-------------+ +| count(rate_reqs.prom_rate(ts_range,val,ts,Int64(300000))) | ts | status_code | ++-----------------------------------------------------------+----+-------------+ ++-----------------------------------------------------------+----+-------------+ INSERT INTO TABLE http_requests VALUES (now() - '1m'::interval, 0), @@ -359,8 +365,10 @@ SHOW CREATE TABLE rate_reqs; -- test if sink table is tql queryable TQL EVAL (now() - '1m'::interval, now(), '5s') count_values("status_code", rate_reqs); -++ -++ ++-----------------------------------------------------------+----+-------------+ +| count(rate_reqs.prom_rate(ts_range,byte,ts,Int64(60000))) | ts | status_code | ++-----------------------------------------------------------+----+-------------+ ++-----------------------------------------------------------+----+-------------+ INSERT INTO TABLE http_requests_total VALUES ('localhost', 'my_service', 'instance1', 100, now() - '1min'::interval), diff --git a/tests/cases/standalone/information_schema/cluster_info.result b/tests/cases/standalone/information_schema/cluster_info.result index 04567ff721..9fe338fbb0 100644 --- a/tests/cases/standalone/information_schema/cluster_info.result +++ b/tests/cases/standalone/information_schema/cluster_info.result @@ -45,8 +45,10 @@ SELECT peer_id, peer_type, peer_addr, version, git_commit, start_time, uptime, a SELECT peer_id, peer_type, peer_addr, version, git_commit, start_time, uptime, active_time FROM CLUSTER_INFO WHERE PEER_TYPE != 'STANDALONE'; -++ -++ ++---------+-----------+-----------+---------+------------+------------+--------+-------------+ +| peer_id | peer_type | peer_addr | version | git_commit | start_time | uptime | active_time | ++---------+-----------+-----------+---------+------------+------------+--------+-------------+ ++---------+-----------+-----------+---------+------------+------------+--------+-------------+ -- SQLNESS REPLACE version node_version -- SQLNESS REPLACE (\s[\-0-9T:\.]{15,}) Start_time @@ -60,8 +62,10 @@ SELECT peer_id, peer_type, peer_addr, version, git_commit, start_time, uptime, a SELECT peer_id, peer_type, peer_addr, version, git_commit, start_time, uptime, active_time FROM CLUSTER_INFO WHERE PEER_ID > 0; -++ -++ ++---------+-----------+-----------+---------+------------+------------+--------+-------------+ +| peer_id | peer_type | peer_addr | version | git_commit | start_time | uptime | active_time | ++---------+-----------+-----------+---------+------------+------------+--------+-------------+ ++---------+-----------+-----------+---------+------------+------------+--------+-------------+ SELECT peer_type, total_cpu_millicores!=0, total_memory_bytes!=0 FROM CLUSTER_INFO ORDER BY peer_type; diff --git a/tests/cases/standalone/limit/limit.result b/tests/cases/standalone/limit/limit.result index 41738535a2..860637de2e 100644 --- a/tests/cases/standalone/limit/limit.result +++ b/tests/cases/standalone/limit/limit.result @@ -1,7 +1,9 @@ SELECT * FROM (SELECT SUM(number) FROM numbers LIMIT 100000000000) LIMIT 0; -++ -++ ++---------------------+ +| sum(numbers.number) | ++---------------------+ ++---------------------+ EXPLAIN SELECT * FROM (SELECT SUM(number) FROM numbers LIMIT 100000000000) LIMIT 0; diff --git a/tests/cases/standalone/optimizer/count.result b/tests/cases/standalone/optimizer/count.result index 8a128a9200..925e89eeb2 100644 --- a/tests/cases/standalone/optimizer/count.result +++ b/tests/cases/standalone/optimizer/count.result @@ -135,9 +135,9 @@ select count(1) from count_where_bug where `tag` = 'b'; | 0_| 0_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| -| 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +| 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(count_where_bug.ts) as count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(count_where_bug.ts) as count(Int64(1))] REDACTED |_|_|_UnorderedScan: region=REDACTED, "partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0} REDACTED |_|_|_| |_|_| Total rows: 1_| @@ -176,9 +176,9 @@ select count(1) from count_where_bug where ts > '2024-09-06T06:00:04Z'; | 0_| 0_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| -| 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +| 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(count_where_bug.ts) as count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(count_where_bug.ts) as count(Int64(1))] REDACTED |_|_|_UnorderedScan: region=REDACTED, "partition_count":REDACTED REDACTED |_|_|_| |_|_| Total rows: 1_| @@ -215,9 +215,9 @@ select count(1) from count_where_bug where num != 3; | 0_| 0_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| -| 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +| 1_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(count_where_bug.ts) as count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(count_where_bug.ts) as count(Int64(1))] REDACTED |_|_|_FilterExec: num@1 != 3, projection=[ts@0] REDACTED |_|_|_UnorderedScan: region=REDACTED, "partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0} REDACTED |_|_|_| @@ -275,9 +275,9 @@ select count(1) from count_where_bug; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -311,9 +311,9 @@ select count(1) from count_where_bug where `tag` = 'b'; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -338,9 +338,9 @@ select count(1) from count_where_bug where `tag` = 'b'; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -375,9 +375,9 @@ select count(1) from count_where_bug where ts > '2024-09-06T06:00:04Z'; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -415,9 +415,9 @@ select count(1) from count_where_bug where num != 3; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(count_where_bug.ts)) as count(Int64(1))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| diff --git a/tests/cases/standalone/optimizer/filter_column_pruning.result b/tests/cases/standalone/optimizer/filter_column_pruning.result index 9cff5fb695..3a16bef50d 100644 --- a/tests/cases/standalone/optimizer/filter_column_pruning.result +++ b/tests/cases/standalone/optimizer/filter_column_pruning.result @@ -54,9 +54,9 @@ EXPLAIN ANALYZE VERBOSE SELECT cpu_usage FROM filter_prune_test WHERE host = 'ho |_|_|_| | 1_| 0_|_ProjectionExec: expr=[cpu_usage@0 as cpu_usage] REDACTED |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED -|_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[cpu_usage@1 as cpu_usage, ts@0 as ts] REDACTED +|_|_|_WindowedSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED +|_|_|_PartSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0}, "projection": ["ts", "cpu_usage"], "filters": ["host = Dictionary(UInt32, Utf8(\"host1\"))"], "flat_format": REDACTED, "REDACTED |_|_|_| |_|_| Total rows: 3_| @@ -90,9 +90,9 @@ EXPLAIN ANALYZE VERBOSE SELECT mem_usage, cpu_usage FROM filter_prune_test WHERE |_|_|_| | 1_| 0_|_ProjectionExec: expr=[mem_usage@0 as mem_usage, cpu_usage@1 as cpu_usage] REDACTED |_|_|_SortPreservingMergeExec: [ts@2 ASC NULLS LAST] REDACTED -|_|_|_WindowedSortExec: expr=ts@2 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_PartSortExec: expr=ts@2 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[mem_usage@2 as mem_usage, cpu_usage@1 as cpu_usage, ts@0 as ts] REDACTED +|_|_|_WindowedSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED +|_|_|_PartSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0}, "projection": ["ts", "cpu_usage", "mem_usage"], "filters": ["host = Dictionary(UInt32, Utf8(\"host1\"))", "region = Dictionary(UInt32, Utf8(\"us-east\"))"], "flat_format": REDACTED, "REDACTED |_|_|_| |_|_| Total rows: 2_| @@ -130,8 +130,7 @@ EXPLAIN ANALYZE VERBOSE SELECT host, `region` FROM filter_prune_test WHERE cpu_u |_|_|_SortPreservingMergeExec: [ts@2 ASC NULLS LAST] REDACTED |_|_|_WindowedSortExec: expr=ts@2 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_PartSortExec: expr=ts@2 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_ProjectionExec: expr=[host@1 as host, region@2 as region, ts@0 as ts] REDACTED -|_|_|_FilterExec: cpu_usage@3 > 20, projection=[ts@0, host@1, region@2] REDACTED +|_|_|_FilterExec: cpu_usage@3 > 20, projection=[host@1, region@2, ts@0] REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0}, "projection": ["ts", "host", "region", "cpu_usage"], "filters": ["cpu_usage > Float64(20)"], "flat_format": REDACTED, "REDACTED |_|_|_| |_|_| Total rows: 4_| @@ -167,9 +166,9 @@ EXPLAIN ANALYZE VERBOSE SELECT host, cpu_usage FROM filter_prune_test WHERE ts > |_|_|_| | 1_| 0_|_ProjectionExec: expr=[host@0 as host, cpu_usage@1 as cpu_usage] REDACTED |_|_|_SortPreservingMergeExec: [ts@2 ASC NULLS LAST] REDACTED -|_|_|_WindowedSortExec: expr=ts@2 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_PartSortExec: expr=ts@2 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[host@1 as host, cpu_usage@2 as cpu_usage, ts@0 as ts] REDACTED +|_|_|_WindowedSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED +|_|_|_PartSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0}, "projection": ["ts", "host", "cpu_usage"], "filters": ["ts > TimestampMillisecond(2000, None)"], "flat_format": REDACTED, "REDACTED |_|_|_| |_|_| Total rows: 4_| @@ -205,8 +204,7 @@ EXPLAIN ANALYZE VERBOSE SELECT mem_usage FROM filter_prune_test WHERE host = 'ho |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED |_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_ProjectionExec: expr=[mem_usage@1 as mem_usage, ts@0 as ts] REDACTED -|_|_|_FilterExec: cpu_usage@1 > 10, projection=[ts@0, mem_usage@2] REDACTED +|_|_|_FilterExec: cpu_usage@1 > 10, projection=[mem_usage@2, ts@0] REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0}, "projection": ["ts", "cpu_usage", "mem_usage"], "filters": ["host = Dictionary(UInt32, Utf8(\"host1\"))", "cpu_usage > Float64(10)", "region = Dictionary(UInt32, Utf8(\"us-east\"))"], "flat_format": REDACTED, "REDACTED |_|_|_| |_|_| Total rows: 2_| @@ -274,8 +272,8 @@ EXPLAIN ANALYZE VERBOSE SELECT `region`, AVG(cpu_usage) as avg_cpu FROM filter_p |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_SortPreservingMergeExec: [region@0 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[region@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED |_|_|_ProjectionExec: expr=[region@0 as region, avg(filter_prune_test.cpu_usage)@1 as avg_cpu] REDACTED +|_|_|_SortExec: expr=[region@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED |_|_|_AggregateExec: mode=FinalPartitioned, gby=[region@0 as region], aggr=[avg(filter_prune_test.cpu_usage)] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_AggregateExec: mode=Partial, gby=[region@0 as region], aggr=[avg(filter_prune_test.cpu_usage)] REDACTED @@ -316,9 +314,9 @@ EXPLAIN ANALYZE VERBOSE SELECT cpu_usage FROM filter_prune_test WHERE host IN (' |_|_|_| | 1_| 0_|_ProjectionExec: expr=[cpu_usage@0 as cpu_usage] REDACTED |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED -|_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[cpu_usage@1 as cpu_usage, ts@0 as ts] REDACTED +|_|_|_WindowedSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED +|_|_|_PartSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0}, "projection": ["ts", "cpu_usage"], "filters": ["host = Dictionary(UInt32, Utf8(\"host1\")) OR host = Dictionary(UInt32, Utf8(\"host2\"))"], "flat_format": REDACTED, "REDACTED |_|_|_| |_|_| Total rows: 5_| @@ -355,8 +353,7 @@ EXPLAIN ANALYZE VERBOSE SELECT host FROM filter_prune_test WHERE cpu_usage BETWE |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED |_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_ProjectionExec: expr=[host@1 as host, ts@0 as ts] REDACTED -|_|_|_FilterExec: cpu_usage@2 >= 15 AND cpu_usage@2 <= 30, projection=[ts@0, host@1] REDACTED +|_|_|_FilterExec: cpu_usage@2 >= 15 AND cpu_usage@2 <= 30, projection=[host@1, ts@0] REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0}, "projection": ["ts", "host", "cpu_usage"], "filters": ["cpu_usage >= Float64(15)", "cpu_usage <= Float64(30)"], "flat_format": REDACTED, "REDACTED |_|_|_| |_|_| Total rows: 3_| @@ -395,8 +392,7 @@ EXPLAIN ANALYZE VERBOSE SELECT cpu_usage FROM filter_prune_test WHERE host LIKE |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED |_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_ProjectionExec: expr=[cpu_usage@1 as cpu_usage, ts@0 as ts] REDACTED -|_|_|_FilterExec: host@1 LIKE host% AND region@2 LIKE us-%, projection=[ts@0, cpu_usage@3] REDACTED +|_|_|_FilterExec: host@1 LIKE host% AND region@2 LIKE us-%, projection=[cpu_usage@3, ts@0] REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0}, "projection": ["ts", "host", "region", "cpu_usage"], "filters": ["host LIKE Dictionary(UInt32, Utf8(\"host%\"))", "region LIKE Dictionary(UInt32, Utf8(\"us-%\"))"], "flat_format": REDACTED, "REDACTED |_|_|_| |_|_| Total rows: 5_| @@ -434,8 +430,7 @@ EXPLAIN ANALYZE VERBOSE SELECT cpu_usage FROM filter_prune_test WHERE host = 'ho |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED |_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_ProjectionExec: expr=[cpu_usage@1 as cpu_usage, ts@0 as ts] REDACTED -|_|_|_FilterExec: host@1 = host1 OR region@2 = eu-west, projection=[ts@0, cpu_usage@3] REDACTED +|_|_|_FilterExec: host@1 = host1 OR region@2 = eu-west, projection=[cpu_usage@3, ts@0] REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0}, "projection": ["ts", "host", "region", "cpu_usage"], "filters": ["host = Dictionary(UInt32, Utf8(\"host1\")) OR region = Dictionary(UInt32, Utf8(\"eu-west\"))"], "flat_format": REDACTED, "REDACTED |_|_|_| |_|_| Total rows: 4_| @@ -469,10 +464,10 @@ EXPLAIN ANALYZE VERBOSE SELECT cpu_usage FROM (SELECT * FROM filter_prune_test W |_|_|_| | 1_| 0_|_ProjectionExec: expr=[cpu_usage@0 as cpu_usage] REDACTED |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED -|_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[cpu_usage@1 as cpu_usage, ts@0 as ts] REDACTED -|_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0}, "projection": ["ts", "cpu_usage"], "filters": ["region = Dictionary(UInt32, Utf8(\"us-east\"))", "host = Dictionary(UInt32, Utf8(\"host1\"))"], "flat_format": REDACTED, "REDACTED +|_|_|_WindowedSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED +|_|_|_PartSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED +|_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0}, "projection": ["ts", "cpu_usage"], "filters": ["host = Dictionary(UInt32, Utf8(\"host1\"))", "region = Dictionary(UInt32, Utf8(\"us-east\"))"], "flat_format": REDACTED, "REDACTED |_|_|_| |_|_| Total rows: 2_| +-+-+-+ @@ -516,9 +511,9 @@ EXPLAIN ANALYZE VERBOSE SELECT cpu_usage FROM filter_prune_test WHERE host = 'ho |_|_|_| | 1_| 0_|_ProjectionExec: expr=[cpu_usage@0 as cpu_usage] REDACTED |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED -|_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[cpu_usage@1 as cpu_usage, ts@0 as ts] REDACTED +|_|_|_WindowedSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED +|_|_|_PartSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":0, "files":1, "file_ranges":1}, "projection": ["ts", "cpu_usage"], "filters": ["host = Dictionary(UInt32, Utf8(\"host1\"))"], \"file\":REDACTED |_|_|_| |_|_| Total rows: 3_| @@ -553,9 +548,9 @@ EXPLAIN ANALYZE VERBOSE SELECT mem_usage, cpu_usage FROM filter_prune_test WHERE |_|_|_| | 1_| 0_|_ProjectionExec: expr=[mem_usage@0 as mem_usage, cpu_usage@1 as cpu_usage] REDACTED |_|_|_SortPreservingMergeExec: [ts@2 ASC NULLS LAST] REDACTED -|_|_|_WindowedSortExec: expr=ts@2 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_PartSortExec: expr=ts@2 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[mem_usage@2 as mem_usage, cpu_usage@1 as cpu_usage, ts@0 as ts] REDACTED +|_|_|_WindowedSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED +|_|_|_PartSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":0, "files":1, "file_ranges":1}, "projection": ["ts", "cpu_usage", "mem_usage"], "filters": ["host = Dictionary(UInt32, Utf8(\"host1\"))", "region = Dictionary(UInt32, Utf8(\"us-east\"))"], \"file\":REDACTED |_|_|_| |_|_| Total rows: 2_| @@ -594,8 +589,7 @@ EXPLAIN ANALYZE VERBOSE SELECT host, `region` FROM filter_prune_test WHERE cpu_u |_|_|_SortPreservingMergeExec: [ts@2 ASC NULLS LAST] REDACTED |_|_|_WindowedSortExec: expr=ts@2 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_PartSortExec: expr=ts@2 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_ProjectionExec: expr=[host@1 as host, region@2 as region, ts@0 as ts] REDACTED -|_|_|_FilterExec: cpu_usage@3 > 20, projection=[ts@0, host@1, region@2] REDACTED +|_|_|_FilterExec: cpu_usage@3 > 20, projection=[host@1, region@2, ts@0] REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":0, "files":1, "file_ranges":1}, "projection": ["ts", "host", "region", "cpu_usage"], "filters": ["cpu_usage > Float64(20)"], \"file\":REDACTED |_|_|_| |_|_| Total rows: 4_| @@ -632,9 +626,9 @@ EXPLAIN ANALYZE VERBOSE SELECT host, cpu_usage FROM filter_prune_test WHERE ts > |_|_|_| | 1_| 0_|_ProjectionExec: expr=[host@0 as host, cpu_usage@1 as cpu_usage] REDACTED |_|_|_SortPreservingMergeExec: [ts@2 ASC NULLS LAST] REDACTED -|_|_|_WindowedSortExec: expr=ts@2 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_PartSortExec: expr=ts@2 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[host@1 as host, cpu_usage@2 as cpu_usage, ts@0 as ts] REDACTED +|_|_|_WindowedSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED +|_|_|_PartSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":0, "files":1, "file_ranges":1}, "projection": ["ts", "host", "cpu_usage"], "filters": ["ts > TimestampMillisecond(2000, None)"], \"file\":REDACTED |_|_|_| |_|_| Total rows: 4_| @@ -671,8 +665,7 @@ EXPLAIN ANALYZE VERBOSE SELECT mem_usage FROM filter_prune_test WHERE host = 'ho |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED |_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_ProjectionExec: expr=[mem_usage@1 as mem_usage, ts@0 as ts] REDACTED -|_|_|_FilterExec: cpu_usage@1 > 10, projection=[ts@0, mem_usage@2] REDACTED +|_|_|_FilterExec: cpu_usage@1 > 10, projection=[mem_usage@2, ts@0] REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":0, "files":1, "file_ranges":1}, "projection": ["ts", "cpu_usage", "mem_usage"], "filters": ["host = Dictionary(UInt32, Utf8(\"host1\"))", "cpu_usage > Float64(10)", "region = Dictionary(UInt32, Utf8(\"us-east\"))"], \"file\":REDACTED |_|_|_| |_|_| Total rows: 2_| @@ -742,8 +735,8 @@ EXPLAIN ANALYZE VERBOSE SELECT `region`, AVG(cpu_usage) as avg_cpu FROM filter_p |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_SortPreservingMergeExec: [region@0 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[region@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED |_|_|_ProjectionExec: expr=[region@0 as region, avg(filter_prune_test.cpu_usage)@1 as avg_cpu] REDACTED +|_|_|_SortExec: expr=[region@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED |_|_|_AggregateExec: mode=FinalPartitioned, gby=[region@0 as region], aggr=[avg(filter_prune_test.cpu_usage)] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_AggregateExec: mode=Partial, gby=[region@0 as region], aggr=[avg(filter_prune_test.cpu_usage)] REDACTED @@ -785,9 +778,9 @@ EXPLAIN ANALYZE VERBOSE SELECT cpu_usage FROM filter_prune_test WHERE host IN (' |_|_|_| | 1_| 0_|_ProjectionExec: expr=[cpu_usage@0 as cpu_usage] REDACTED |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED -|_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[cpu_usage@1 as cpu_usage, ts@0 as ts] REDACTED +|_|_|_WindowedSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED +|_|_|_PartSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":0, "files":1, "file_ranges":1}, "projection": ["ts", "cpu_usage"], "filters": ["host = Dictionary(UInt32, Utf8(\"host1\")) OR host = Dictionary(UInt32, Utf8(\"host2\"))"], \"file\":REDACTED |_|_|_| |_|_| Total rows: 5_| @@ -825,8 +818,7 @@ EXPLAIN ANALYZE VERBOSE SELECT host FROM filter_prune_test WHERE cpu_usage BETWE |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED |_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_ProjectionExec: expr=[host@1 as host, ts@0 as ts] REDACTED -|_|_|_FilterExec: cpu_usage@2 >= 15 AND cpu_usage@2 <= 30, projection=[ts@0, host@1] REDACTED +|_|_|_FilterExec: cpu_usage@2 >= 15 AND cpu_usage@2 <= 30, projection=[host@1, ts@0] REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":0, "files":1, "file_ranges":1}, "projection": ["ts", "host", "cpu_usage"], "filters": ["cpu_usage >= Float64(15)", "cpu_usage <= Float64(30)"], \"file\":REDACTED |_|_|_| |_|_| Total rows: 3_| @@ -866,8 +858,7 @@ EXPLAIN ANALYZE VERBOSE SELECT cpu_usage FROM filter_prune_test WHERE host LIKE |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED |_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_ProjectionExec: expr=[cpu_usage@1 as cpu_usage, ts@0 as ts] REDACTED -|_|_|_FilterExec: host@1 LIKE host% AND region@2 LIKE us-%, projection=[ts@0, cpu_usage@3] REDACTED +|_|_|_FilterExec: host@1 LIKE host% AND region@2 LIKE us-%, projection=[cpu_usage@3, ts@0] REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":0, "files":1, "file_ranges":1}, "projection": ["ts", "host", "region", "cpu_usage"], "filters": ["host LIKE Dictionary(UInt32, Utf8(\"host%\"))", "region LIKE Dictionary(UInt32, Utf8(\"us-%\"))"], \"file\":REDACTED |_|_|_| |_|_| Total rows: 5_| @@ -906,8 +897,7 @@ EXPLAIN ANALYZE VERBOSE SELECT cpu_usage FROM filter_prune_test WHERE host = 'ho |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED |_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_ProjectionExec: expr=[cpu_usage@1 as cpu_usage, ts@0 as ts] REDACTED -|_|_|_FilterExec: host@1 = host1 OR region@2 = eu-west, projection=[ts@0, cpu_usage@3] REDACTED +|_|_|_FilterExec: host@1 = host1 OR region@2 = eu-west, projection=[cpu_usage@3, ts@0] REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":0, "files":1, "file_ranges":1}, "projection": ["ts", "host", "region", "cpu_usage"], "filters": ["host = Dictionary(UInt32, Utf8(\"host1\")) OR region = Dictionary(UInt32, Utf8(\"eu-west\"))"], \"file\":REDACTED |_|_|_| |_|_| Total rows: 4_| @@ -942,10 +932,10 @@ EXPLAIN ANALYZE VERBOSE SELECT cpu_usage FROM (SELECT * FROM filter_prune_test W |_|_|_| | 1_| 0_|_ProjectionExec: expr=[cpu_usage@0 as cpu_usage] REDACTED |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED -|_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[cpu_usage@1 as cpu_usage, ts@0 as ts] REDACTED -|_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":0, "files":1, "file_ranges":1}, "projection": ["ts", "cpu_usage"], "filters": ["region = Dictionary(UInt32, Utf8(\"us-east\"))", "host = Dictionary(UInt32, Utf8(\"host1\"))"], \"file\":REDACTED +|_|_|_WindowedSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED +|_|_|_PartSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED +|_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":1, "mem_ranges":0, "files":1, "file_ranges":1}, "projection": ["ts", "cpu_usage"], "filters": ["host = Dictionary(UInt32, Utf8(\"host1\"))", "region = Dictionary(UInt32, Utf8(\"us-east\"))"], \"file\":REDACTED |_|_|_| |_|_| Total rows: 2_| +-+-+-+ @@ -1031,9 +1021,9 @@ EXPLAIN ANALYZE VERBOSE SELECT field1 FROM filter_prune_files WHERE tag_key = 'a |_|_|_| | 1_| 0_|_ProjectionExec: expr=[field1@0 as field1] REDACTED |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED -|_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[field1@1 as field1, ts@0 as ts] REDACTED +|_|_|_WindowedSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED +|_|_|_PartSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":3, "mem_ranges":1, "files":2, "file_ranges":2}, "projection": ["ts", "field1"], "filters": ["tag_key = Dictionary(UInt32, Utf8(\"a\"))"], \"file\":REDACTED |_|_|_| |_|_| Total rows: 2_| @@ -1073,8 +1063,7 @@ EXPLAIN ANALYZE VERBOSE SELECT field1 FROM filter_prune_files WHERE field2 > 5.0 |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED |_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_ProjectionExec: expr=[field1@1 as field1, ts@0 as ts] REDACTED -|_|_|_FilterExec: field2@2 > 5, projection=[ts@0, field1@1] REDACTED +|_|_|_FilterExec: field2@2 > 5, projection=[field1@1, ts@0] REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":3, "mem_ranges":1, "files":2, "file_ranges":2}, "projection": ["ts", "field1", "field2"], "filters": ["field2 > Float64(5)"], \"file\":REDACTED |_|_|_| |_|_| Total rows: 5_| @@ -1111,8 +1100,7 @@ EXPLAIN ANALYZE VERBOSE SELECT field3 FROM filter_prune_files WHERE tag_key = 'b |_|_|_SortPreservingMergeExec: [ts@1 ASC NULLS LAST] REDACTED |_|_|_WindowedSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_PartSortExec: expr=ts@1 ASC NULLS LAST num_ranges=REDACTED REDACTED -|_|_|_ProjectionExec: expr=[field3@1 as field3, ts@0 as ts] REDACTED -|_|_|_FilterExec: field2@1 > 7, projection=[ts@0, field3@2] REDACTED +|_|_|_FilterExec: field2@1 > 7, projection=[field3@2, ts@0] REDACTED |_|_|_SeqScan: region=REDACTED, {"partition_count":{"count":3, "mem_ranges":1, "files":2, "file_ranges":2}, "projection": ["ts", "field2", "field3"], "filters": ["tag_key = Dictionary(UInt32, Utf8(\"b\"))", "field2 > Float64(7)"], \"file\":REDACTED |_|_|_| |_|_| Total rows: 2_| diff --git a/tests/cases/standalone/optimizer/filter_push_down.result b/tests/cases/standalone/optimizer/filter_push_down.result index b959e063f0..6705bb3c83 100644 --- a/tests/cases/standalone/optimizer/filter_push_down.result +++ b/tests/cases/standalone/optimizer/filter_push_down.result @@ -190,14 +190,18 @@ SELECT * FROM (SELECT i1.i AS a, i2.i AS b, row_number() OVER (ORDER BY i1.i, i2 -- Align the result to PostgreSQL: empty. SELECT * FROM (SELECT 0=1 AS cond FROM integers i1, integers i2) a1 WHERE cond ORDER BY 1; -++ -++ ++------+ +| cond | ++------+ ++------+ -- Align the result to PostgreSQL: empty. SELECT * FROM (SELECT 0=1 AS cond FROM integers i1, integers i2 GROUP BY 1) a1 WHERE cond ORDER BY 1; -++ -++ ++------+ +| cond | ++------+ ++------+ DROP TABLE integers; diff --git a/tests/cases/standalone/optimizer/first_value_advance.result b/tests/cases/standalone/optimizer/first_value_advance.result index c543a8df2a..f3b24db6d4 100644 --- a/tests/cases/standalone/optimizer/first_value_advance.result +++ b/tests/cases/standalone/optimizer/first_value_advance.result @@ -314,14 +314,14 @@ explain select first_value(ts order by ts) from t; +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]] | +| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[]], aggr=[[__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_TableScan: t_| |_| ]]_| -| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[__first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -343,9 +343,9 @@ explain analyze +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -400,19 +400,19 @@ order by time_window, ordered_host; +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Sort: time_window ASC NULLS LAST, ordered_host ASC NULLS LAST_| +| logical_plan_| Sort: time_window ASC NULLS LAST_| |_|_Projection: first_value(t.host) ORDER BY [t.ts ASC NULLS LAST] AS ordered_host, first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t.ts) AS time_window_| -|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t.ts)]], aggr=[[__first_value_merge(__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]] | +|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t.ts)]], aggr=[[__first_value_merge(__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 5000000 }"), t.ts) AS date_bin(Utf8("5 milliseconds"),t.ts)]], aggr=[[__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_TableScan: t_| |_| ]]_| -| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST]_| -|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST]_| |_|_ProjectionExec: expr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST]@1 as ordered_host, first_value(t.val) ORDER BY [t.ts ASC NULLS LAST]@2 as first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]@3 as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t.ts)@0 as time_window]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__first_value_merge(__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED -|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__first_value_merge(__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -441,12 +441,12 @@ order by time_window, ordered_host; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST] REDACTED |_|_|_ProjectionExec: expr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST]@1 as ordered_host, first_value(t.val) ORDER BY [t.ts ASC NULLS LAST]@2 as first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]@3 as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t.ts)@0 as time_window] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__first_value_merge(__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__first_value_merge(__first_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as first_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -518,14 +518,14 @@ explain +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]] | +| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[]], aggr=[[__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_TableScan: t1_| |_| ]]_| -| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[__first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -547,9 +547,9 @@ explain analyze +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -698,19 +698,19 @@ order by time_window, ordered_host; +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Sort: time_window ASC NULLS LAST, ordered_host ASC NULLS LAST_| +| logical_plan_| Sort: time_window ASC NULLS LAST_| |_|_Projection: first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST] AS ordered_host, first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t1.ts) AS time_window_| -|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t1.ts)]], aggr=[[__first_value_merge(__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]] | +|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t1.ts)]], aggr=[[__first_value_merge(__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 5000000 }"), t1.ts) AS date_bin(Utf8("5 milliseconds"),t1.ts)]], aggr=[[__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_TableScan: t1_| |_| ]]_| -| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST]_| -|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST]_| |_|_ProjectionExec: expr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST]@1 as ordered_host, first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST]@2 as first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]@3 as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t1.ts)@0 as time_window]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__first_value_merge(__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED -|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__first_value_merge(__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -739,12 +739,12 @@ order by time_window, ordered_host; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST] REDACTED |_|_|_ProjectionExec: expr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST]@1 as ordered_host, first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST]@2 as first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]@3 as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t1.ts)@0 as time_window] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__first_value_merge(__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__first_value_merge(__first_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __first_value_merge(__first_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as first_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| diff --git a/tests/cases/standalone/optimizer/join_filter_pushdown_edge.result b/tests/cases/standalone/optimizer/join_filter_pushdown_edge.result index e12ea22f4a..c9b1f37c9a 100644 --- a/tests/cases/standalone/optimizer/join_filter_pushdown_edge.result +++ b/tests/cases/standalone/optimizer/join_filter_pushdown_edge.result @@ -100,7 +100,7 @@ EXPLAIN SELECT f.k, f.val FROM fact f | | MergeScan [is_placeholder=false, remote_input=[ | | | SubqueryAlias: f | | | Filter: CAST(fact.val AS Int32) IS NOT NULL AND fact.ts >= TimestampMillisecond(1706572800000, None) AND fact.val > Float64(1) | -| | TableScan: fact, partial_filters=[fact.ts >= TimestampMillisecond(1706572800000, None), fact.val > Float64(1), CAST(fact.val AS Int32) IS NOT NULL] | +| | TableScan: fact, partial_filters=[CAST(fact.val AS Int32) IS NOT NULL, fact.ts >= TimestampMillisecond(1706572800000, None), fact.val > Float64(1)] | | | ]] | | | Projection: d.k | | | MergeScan [is_placeholder=false, remote_input=[ | diff --git a/tests/cases/standalone/optimizer/last_value_advance.result b/tests/cases/standalone/optimizer/last_value_advance.result index 15e5ca80cb..6b4960be6a 100644 --- a/tests/cases/standalone/optimizer/last_value_advance.result +++ b/tests/cases/standalone/optimizer/last_value_advance.result @@ -314,14 +314,14 @@ explain select last_value(ts order by ts) from t; +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]] | +| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[]], aggr=[[__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_TableScan: t_| |_| ]]_| -| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[__last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -343,9 +343,9 @@ explain analyze +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -400,19 +400,19 @@ order by time_window, ordered_host; +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Sort: time_window ASC NULLS LAST, ordered_host ASC NULLS LAST_| +| logical_plan_| Sort: time_window ASC NULLS LAST_| |_|_Projection: last_value(t.host) ORDER BY [t.ts ASC NULLS LAST] AS ordered_host, last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t.ts) AS time_window_| -|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t.ts)]], aggr=[[__last_value_merge(__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]] | +|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t.ts)]], aggr=[[__last_value_merge(__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) AS last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 5000000 }"), t.ts) AS date_bin(Utf8("5 milliseconds"),t.ts)]], aggr=[[__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]]]_| |_|_TableScan: t_| |_| ]]_| -| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST]_| -|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST]_| |_|_ProjectionExec: expr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST]@1 as ordered_host, last_value(t.val) ORDER BY [t.ts ASC NULLS LAST]@2 as last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]@3 as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t.ts)@0 as time_window]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__last_value_merge(__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED -|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__last_value_merge(__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -441,12 +441,12 @@ order by time_window, ordered_host; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST] REDACTED |_|_|_ProjectionExec: expr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST]@1 as ordered_host, last_value(t.val) ORDER BY [t.ts ASC NULLS LAST]@2 as last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]@3 as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t.ts)@0 as time_window] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__last_value_merge(__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t.ts)@0 as date_bin(Utf8("5 milliseconds"),t.ts)], aggr=[__last_value_merge(__last_value_state(t.host) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.host) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.val) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.val) ORDER BY [t.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t.ts) ORDER BY [t.ts ASC NULLS LAST]) as last_value(t.ts) ORDER BY [t.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -518,14 +518,14 @@ explain +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]] | +| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[]], aggr=[[__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_TableScan: t1_| |_| ]]_| -| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[__last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_CoalescePartitionsExec_| -|_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[], aggr=[__last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -547,9 +547,9 @@ explain analyze +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +| 0_| 0_|_AggregateExec: mode=Final, gby=[], aggr=[__last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| @@ -698,19 +698,19 @@ order by time_window, ordered_host; +-+-+ | plan_type_| plan_| +-+-+ -| logical_plan_| Sort: time_window ASC NULLS LAST, ordered_host ASC NULLS LAST_| +| logical_plan_| Sort: time_window ASC NULLS LAST_| |_|_Projection: last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST] AS ordered_host, last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t1.ts) AS time_window_| -|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t1.ts)]], aggr=[[__last_value_merge(__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]] | +|_|_Aggregate: groupBy=[[date_bin(Utf8("5 milliseconds"),t1.ts)]], aggr=[[__last_value_merge(__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) AS last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_MergeScan [is_placeholder=false, remote_input=[_| |_| Aggregate: groupBy=[[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 5000000 }"), t1.ts) AS date_bin(Utf8("5 milliseconds"),t1.ts)]], aggr=[[__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]]_| |_|_TableScan: t1_| |_| ]]_| -| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST]_| -|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| physical_plan | SortPreservingMergeExec: [time_window@3 ASC NULLS LAST]_| |_|_ProjectionExec: expr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST]@1 as ordered_host, last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST]@2 as last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]@3 as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t1.ts)@0 as time_window]_| -|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__last_value_merge(__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] | |_|_RepartitionExec: REDACTED -|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| +|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__last_value_merge(__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]]_| |_|_RepartitionExec: REDACTED |_|_MergeScanExec: REDACTED |_|_| @@ -739,12 +739,12 @@ order by time_window, ordered_host; +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[time_window@3 ASC NULLS LAST, ordered_host@0 ASC NULLS LAST], preserve_REDACTED +| 0_| 0_|_SortPreservingMergeExec: [time_window@3 ASC NULLS LAST] REDACTED |_|_|_ProjectionExec: expr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST]@1 as ordered_host, last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST]@2 as last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]@3 as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST], date_bin(Utf8("5 milliseconds"),t1.ts)@0 as time_window] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_SortExec: expr=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 ASC NULLS LAST], preserve_REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__last_value_merge(__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[date_bin(Utf8("5 milliseconds"),t1.ts)@0 as date_bin(Utf8("5 milliseconds"),t1.ts)], aggr=[__last_value_merge(__last_value_state(t1.host) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.host) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.val) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.val) ORDER BY [t1.ts ASC NULLS LAST], __last_value_merge(__last_value_state(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]) as last_value(t1.ts) ORDER BY [t1.ts ASC NULLS LAST]] REDACTED |_|_|_RepartitionExec: REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| diff --git a/tests/cases/standalone/optimizer/lateral_join_guard.result b/tests/cases/standalone/optimizer/lateral_join_guard.result index e14ab5c9a7..d2a565089e 100644 --- a/tests/cases/standalone/optimizer/lateral_join_guard.result +++ b/tests/cases/standalone/optimizer/lateral_join_guard.result @@ -1,9 +1,7 @@ --- Document the current aliased SQL LATERAL limitation and guard the remote --- scan boundary. DataFusion's DecorrelateLateralJoin does not currently match --- the SubqueryAlias(Subquery) shape produced by `LATERAL (...) d`, so this query --- is still expected to fail physical planning with an outer_ref expression. The --- important regression assertion is that the remaining outer_ref predicate must --- NOT be advertised as a remote TableScan.partial_filters predicate. +-- Guard the remote scan boundary for an aliased SQL LATERAL query. +-- DataFusion 55 decorrelates this shape into an inner join. The resulting +-- scan-local IS NOT NULL predicates may be pushed down, but no outer_ref +-- expression may be advertised as a remote TableScan.partial_filters predicate. CREATE TABLE lateral_fact ( ts TIMESTAMP(3) TIME INDEX, k STRING, @@ -52,6 +50,9 @@ ADMIN FLUSH_TABLE('lateral_dim'); -- SQLNESS REPLACE region=\d+\(\d+,\s+\d+\) region=REDACTED -- SQLNESS REPLACE (peers.*) REDACTED +-- SQLNESS REPLACE partitioning=Hash\(\[k@(\d+)\],\s*\d+\) partitioning=Hash([k@$1], REDACTED) +-- SQLNESS REPLACE input_partitions=\d+ input_partitions=REDACTED +-- SQLNESS REPLACE (input_partitions=REDACTED)(\s+)\| $1| EXPLAIN SELECT f.k, d.threshold FROM lateral_fact f, LATERAL ( @@ -60,28 +61,35 @@ LATERAL ( WHERE f.val > d.threshold ORDER BY f.k; -+---------------------+----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ -| plan_type | plan | -+---------------------+----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ -| logical_plan | Sort: f.k ASC NULLS LAST | -| | Projection: f.k, d.threshold | -| | Inner Join: Filter: f.val > d.threshold | -| | Projection: f.k, f.val | -| | MergeScan [is_placeholder=false, remote_input=[ | -| | SubqueryAlias: f | -| | TableScan: lateral_fact | -| | ]] | -| | SubqueryAlias: d | -| | Subquery: | -| | SubqueryAlias: d | -| | Projection: lateral_dim.threshold | -| | Filter: lateral_dim.k = outer_ref(f.k) | -| | Projection: lateral_dim.k, lateral_dim.threshold | -| | MergeScan [is_placeholder=false, remote_input=[ | -| | TableScan: lateral_dim | -| | ]] | -| physical_plan_error | This feature is not implemented: Physical plan does not support logical expression OuterReferenceColumn(Field { name: "k", data_type: Dictionary(UInt32, Utf8), nullable: true }, Column { relation: Some(Bare { table: "f" }), name: "k" }) | -+---------------------+----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------+ ++---------------+---------------------------------------------------------------------------------------------------------------------------------+ +| plan_type | plan | ++---------------+---------------------------------------------------------------------------------------------------------------------------------+ +| logical_plan | Sort: f.k ASC NULLS LAST | +| | Projection: f.k, d.threshold | +| | Inner Join: f.k = d.k Filter: f.val > d.threshold | +| | Projection: f.k, f.val | +| | MergeScan [is_placeholder=false, remote_input=[ | +| | SubqueryAlias: f | +| | Filter: lateral_fact.k IS NOT NULL | +| | TableScan: lateral_fact, partial_filters=[lateral_fact.k IS NOT NULL] | +| | ]] | +| | MergeScan [is_placeholder=false, remote_input=[ | +| | SubqueryAlias: d | +| | Projection: d.threshold, d.k | +| | SubqueryAlias: d | +| | Filter: lateral_dim.k IS NOT NULL | +| | TableScan: lateral_dim, partial_filters=[lateral_dim.k IS NOT NULL] | +| | ]] | +| physical_plan | SortPreservingMergeExec: [k@0 ASC NULLS LAST] | +| | SortExec: expr=[k@0 ASC NULLS LAST], preserve_partitioning=[true] | +| | HashJoinExec: mode=Partitioned, join_type=Inner, on=[(k@0, k@1)], filter=val@0 > threshold@1, projection=[k@0, threshold@2] | +| | RepartitionExec: partitioning=Hash([k@0], REDACTED), input_partitions=REDACTED| +| | ProjectionExec: expr=[k@1 as k, val@2 as val] | +| | MergeScanExec: REDACTED +| | RepartitionExec: partitioning=Hash([k@1], REDACTED), input_partitions=REDACTED| +| | MergeScanExec: REDACTED +| | | ++---------------+---------------------------------------------------------------------------------------------------------------------------------+ DROP TABLE lateral_fact; diff --git a/tests/cases/standalone/optimizer/lateral_join_guard.sql b/tests/cases/standalone/optimizer/lateral_join_guard.sql index 20ed6b5cef..9500475661 100644 --- a/tests/cases/standalone/optimizer/lateral_join_guard.sql +++ b/tests/cases/standalone/optimizer/lateral_join_guard.sql @@ -1,9 +1,7 @@ --- Document the current aliased SQL LATERAL limitation and guard the remote --- scan boundary. DataFusion's DecorrelateLateralJoin does not currently match --- the SubqueryAlias(Subquery) shape produced by `LATERAL (...) d`, so this query --- is still expected to fail physical planning with an outer_ref expression. The --- important regression assertion is that the remaining outer_ref predicate must --- NOT be advertised as a remote TableScan.partial_filters predicate. +-- Guard the remote scan boundary for an aliased SQL LATERAL query. +-- DataFusion 55 decorrelates this shape into an inner join. The resulting +-- scan-local IS NOT NULL predicates may be pushed down, but no outer_ref +-- expression may be advertised as a remote TableScan.partial_filters predicate. CREATE TABLE lateral_fact ( ts TIMESTAMP(3) TIME INDEX, @@ -32,6 +30,9 @@ ADMIN FLUSH_TABLE('lateral_dim'); -- SQLNESS REPLACE region=\d+\(\d+,\s+\d+\) region=REDACTED -- SQLNESS REPLACE (peers.*) REDACTED +-- SQLNESS REPLACE partitioning=Hash\(\[k@(\d+)\],\s*\d+\) partitioning=Hash([k@$1], REDACTED) +-- SQLNESS REPLACE input_partitions=\d+ input_partitions=REDACTED +-- SQLNESS REPLACE (input_partitions=REDACTED)(\s+)\| $1| EXPLAIN SELECT f.k, d.threshold FROM lateral_fact f, LATERAL ( diff --git a/tests/cases/standalone/optimizer/order_by.result b/tests/cases/standalone/optimizer/order_by.result index 06b06ae442..020e1dc28a 100644 --- a/tests/cases/standalone/optimizer/order_by.result +++ b/tests/cases/standalone/optimizer/order_by.result @@ -140,10 +140,9 @@ EXPLAIN ANALYZE SELECT i, t AS alias_ts FROM test_pk ORDER BY t DESC LIMIT 5; | 0_| 0_|_CooperativeExec REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| -| 1_| 0_|_ProjectionExec: expr=[i@0 as i, alias_ts@1 as alias_ts] REDACTED -|_|_|_SortPreservingMergeExec: [t@2 DESC], fetch=5 REDACTED -|_|_|_SortExec: TopK(fetch=5), expr=[alias_ts@1 DESC], preserve_partitioning=[true], filter=[alias_ts@1 IS NULL OR alias_ts@1 > 2] REDACTED -|_|_|_ProjectionExec: expr=[i@0 as i, t@1 as alias_ts, t@1 as t] REDACTED +| 1_| 0_|_SortPreservingMergeExec: [alias_ts@1 DESC], fetch=5 REDACTED +|_|_|_ProjectionExec: expr=[i@0 as i, t@1 as alias_ts] REDACTED +|_|_|_SortExec: TopK(fetch=5), expr=[t@1 DESC], preserve_partitioning=[true], filter=[t@1 IS NULL OR t@1 > 2] REDACTED |_|_|_SeqScan: region=REDACTED, "partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0} REDACTED |_|_|_| |_|_| Total rows: 5_| @@ -164,8 +163,8 @@ EXPLAIN ANALYZE SELECT i, t AS alias_ts FROM test_pk ORDER BY alias_ts DESC LIMI |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_SortPreservingMergeExec: [alias_ts@1 DESC], fetch=5 REDACTED -|_|_|_SortExec: TopK(fetch=5), expr=[alias_ts@1 DESC], preserve_partitioning=[true], filter=[alias_ts@1 IS NULL OR alias_ts@1 > 2] REDACTED |_|_|_ProjectionExec: expr=[i@0 as i, t@1 as alias_ts] REDACTED +|_|_|_SortExec: TopK(fetch=5), expr=[t@1 DESC], preserve_partitioning=[true], filter=[t@1 IS NULL OR t@1 > 2] REDACTED |_|_|_SeqScan: region=REDACTED, "partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0} REDACTED |_|_|_| |_|_| Total rows: 5_| diff --git a/tests/cases/standalone/optimizer/rewrite_set_comparison.result b/tests/cases/standalone/optimizer/rewrite_set_comparison.result index 6d9e04458c..f7463099de 100644 --- a/tests/cases/standalone/optimizer/rewrite_set_comparison.result +++ b/tests/cases/standalone/optimizer/rewrite_set_comparison.result @@ -147,8 +147,10 @@ EXPLAIN SELECT v FROM sc_t WHERE v != ALL(SELECT v FROM sc_s) ORDER BY v; SELECT v FROM sc_t WHERE v != ALL(SELECT v FROM sc_s) ORDER BY v; -++ -++ ++---+ +| v | ++---+ ++---+ DROP TABLE sc_t; diff --git a/tests/cases/standalone/optimizer/windowed_sort.result b/tests/cases/standalone/optimizer/windowed_sort.result index 3be0dd1c3e..9bffb3d58d 100644 --- a/tests/cases/standalone/optimizer/windowed_sort.result +++ b/tests/cases/standalone/optimizer/windowed_sort.result @@ -71,8 +71,8 @@ ORDER BY |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_SortPreservingMergeExec: [collect_time@0 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[collect_time@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED |_|_|_ProjectionExec: expr=[collect_time_utc@0 as collect_time, peak_current@1 as peak_current] REDACTED +|_|_|_SortExec: expr=[collect_time_utc@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED |_|_|_SeqScan: region=REDACTED, "partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0} REDACTED |_|_|_| |_|_| Total rows: 8_| @@ -120,8 +120,8 @@ ORDER BY |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_SortPreservingMergeExec: [collect_time_0@0 ASC NULLS LAST] REDACTED -|_|_|_SortExec: expr=[collect_time_0@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED |_|_|_ProjectionExec: expr=[collect_time_utc@0 as collect_time_0, peak_current@1 as peak_current] REDACTED +|_|_|_SortExec: expr=[collect_time_utc@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED |_|_|_SeqScan: region=REDACTED, "partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0} REDACTED |_|_|_| |_|_| Total rows: 8_| @@ -172,9 +172,9 @@ ORDER BY |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_SortPreservingMergeExec: [true_collect_time@0 DESC] REDACTED -|_|_|_WindowedSortExec: expr=true_collect_time@0 DESC num_ranges=REDACTED REDACTED -|_|_|_PartSortExec: expr=true_collect_time@0 DESC num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[collect_time@0 as true_collect_time, collect_time_utc@1 as collect_time, peak_current@2 as peak_current] REDACTED +|_|_|_WindowedSortExec: expr=collect_time@0 DESC num_ranges=REDACTED REDACTED +|_|_|_PartSortExec: expr=collect_time@0 DESC num_ranges=REDACTED REDACTED |_|_|_SeqScan: region=REDACTED, "partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0} REDACTED |_|_|_| |_|_| Total rows: 8_| @@ -225,9 +225,9 @@ ORDER BY |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_SortPreservingMergeExec: [true_collect_time@1 DESC] REDACTED -|_|_|_WindowedSortExec: expr=true_collect_time@1 DESC num_ranges=REDACTED REDACTED -|_|_|_PartSortExec: expr=true_collect_time@1 DESC num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[collect_time_utc@1 as collect_time, collect_time@0 as true_collect_time, peak_current@2 as peak_current] REDACTED +|_|_|_WindowedSortExec: expr=collect_time@0 DESC num_ranges=REDACTED REDACTED +|_|_|_PartSortExec: expr=collect_time@0 DESC num_ranges=REDACTED REDACTED |_|_|_SeqScan: region=REDACTED, "partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0} REDACTED |_|_|_| |_|_| Total rows: 8_| diff --git a/tests/cases/standalone/optimizer/windowed_sort_advance.result b/tests/cases/standalone/optimizer/windowed_sort_advance.result index c0be1bece9..46ff90a69f 100644 --- a/tests/cases/standalone/optimizer/windowed_sort_advance.result +++ b/tests/cases/standalone/optimizer/windowed_sort_advance.result @@ -65,8 +65,8 @@ EXPLAIN ANALYZE select ts as ts, status, value from `a` where ts >= '2026-03-12T |_|_|_MergeScanExec: REDACTED |_|_|_| | 1_| 0_|_SortPreservingMergeExec: [ts@0 ASC NULLS LAST] REDACTED -|_|_|_WindowedSortExec: expr=ts@0 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_ProjectionExec: expr=[ts@2 as ts, status@1 as status, value@0 as value] REDACTED +|_|_|_WindowedSortExec: expr=ts@2 ASC NULLS LAST num_ranges=REDACTED REDACTED |_|_|_SeqScan: region=REDACTED, "partition_count":{"count":1, "mem_ranges":1, "files":0, "file_ranges":0} REDACTED |_|_|_| |_|_| Total rows: 10_| diff --git a/tests/cases/standalone/tql-explain-analyze/analyze.result b/tests/cases/standalone/tql-explain-analyze/analyze.result index cbbce8eaf1..7c0b9b16e6 100644 --- a/tests/cases/standalone/tql-explain-analyze/analyze.result +++ b/tests/cases/standalone/tql-explain-analyze/analyze.result @@ -268,8 +268,10 @@ Affected Rows: 0 TQL EVAL sum(test2); -++ -++ ++--------------------+---------------------------+ +| greptime_timestamp | sum(test2.greptime_value) | ++--------------------+---------------------------+ ++--------------------+---------------------------+ -- SQLNESS REPLACE (metrics.*) REDACTED -- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED @@ -285,9 +287,9 @@ TQL ANALYZE sum(test2); +-+-+-+ | 0_| 0_|_SortPreservingMergeExec: [greptime_timestamp@0 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[greptime_timestamp@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[sum(test2.greptime_value)] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[__sum_merge(__sum_state(test2.greptime_value)) as sum(test2.greptime_value)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[sum(test2.greptime_value)] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[greptime_timestamp@0 as greptime_timestamp], aggr=[__sum_merge(__sum_state(test2.greptime_value)) as sum(test2.greptime_value)] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| diff --git a/tests/cases/standalone/tql-explain-analyze/tsid_column.result b/tests/cases/standalone/tql-explain-analyze/tsid_column.result index cf2deb7664..c761267bd5 100644 --- a/tests/cases/standalone/tql-explain-analyze/tsid_column.result +++ b/tests/cases/standalone/tql-explain-analyze/tsid_column.result @@ -76,9 +76,9 @@ TQL ANALYZE (0, 10, '5s') sum by (job, instance) (tsid_metric); |_|_|_| | 1_| 0_|_SortPreservingMergeExec: [job@0 ASC NULLS LAST, instance@1 ASC NULLS LAST, ts@2 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[job@0 ASC NULLS LAST, instance@1 ASC NULLS LAST, ts@2 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[job@0 as job, instance@1 as instance, ts@2 as ts], aggr=[sum(tsid_metric.val), __tsid] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[job@0 as job, instance@1 as instance, ts@2 as ts], aggr=[sum(tsid_metric.val), first_value(tsid_metric.__tsid) as __tsid] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[job@2 as job, instance@1 as instance, ts@4 as ts], aggr=[sum(tsid_metric.val), __tsid] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[job@2 as job, instance@1 as instance, ts@4 as ts], aggr=[sum(tsid_metric.val), first_value(tsid_metric.__tsid) as __tsid] REDACTED |_|_|_PromInstantManipulateExec: range=[0..10000], lookback=[300000], interval=[5000], time index=[ts] REDACTED |_|_|_PromSeriesDivideExec: tags=["__tsid"] REDACTED |_|_|_ProjectionExec: expr=[val@1 as val, instance@3 as instance, job@4 as job, __tsid@2 as __tsid, ts@0 as ts] REDACTED @@ -99,7 +99,7 @@ TQL ANALYZE (0, 10, '5s') sum(irate(tsid_metric[1h])) / scalar(count(count(tsid +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_ProjectionExec: expr=[ts@1 as ts, sum(prom_irate(ts_range,val))@2 / scalar(count(count(tsid_metric.val)))@0 as lhs.sum(prom_irate(ts_range,val)) / rhs.scalar(count(count(tsid_metric.val)))] REDACTED +| 0_| 0_|_ProjectionExec: expr=[ts@0 as ts, sum(prom_irate(ts_range,val))@1 / scalar(count(count(tsid_metric.val)))@2 as lhs.sum(prom_irate(ts_range,val)) / rhs.scalar(count(count(tsid_metric.val)))] REDACTED |_|_|_REDACTED |_|_|_ScalarCalculateExec: tags=[] REDACTED |_|_|_CooperativeExec REDACTED @@ -109,9 +109,9 @@ TQL ANALYZE (0, 10, '5s') sum(irate(tsid_metric[1h])) / scalar(count(count(tsid |_|_|_| | 1_| 0_|_SortPreservingMergeExec: [ts@0 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[ts@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(count(tsid_metric.val))] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(tsid_metric.ts) as count(count(tsid_metric.val))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(count(tsid_metric.val))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(tsid_metric.ts) as count(count(tsid_metric.val))] REDACTED |_|_|_ProjectionExec: expr=[ts@0 as ts] REDACTED |_|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts, job@1 as job], aggr=[] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED @@ -150,7 +150,7 @@ TQL ANALYZE (0, 10, '5s') sum(irate(tsid_metric[1h])) / scalar(count(sum(tsid_m +-+-+-+ | stage | node | plan_| +-+-+-+ -| 0_| 0_|_ProjectionExec: expr=[ts@1 as ts, sum(prom_irate(ts_range,val))@2 / scalar(count(sum(tsid_metric.val)))@0 as lhs.sum(prom_irate(ts_range,val)) / rhs.scalar(count(sum(tsid_metric.val)))] REDACTED +| 0_| 0_|_ProjectionExec: expr=[ts@0 as ts, sum(prom_irate(ts_range,val))@1 / scalar(count(sum(tsid_metric.val)))@2 as lhs.sum(prom_irate(ts_range,val)) / rhs.scalar(count(sum(tsid_metric.val)))] REDACTED |_|_|_REDACTED |_|_|_ScalarCalculateExec: tags=[] REDACTED |_|_|_CooperativeExec REDACTED @@ -160,15 +160,14 @@ TQL ANALYZE (0, 10, '5s') sum(irate(tsid_metric[1h])) / scalar(count(sum(tsid_m |_|_|_| | 1_| 0_|_SortPreservingMergeExec: [ts@0 ASC NULLS LAST] REDACTED |_|_|_SortExec: expr=[ts@0 ASC NULLS LAST], preserve_partitioning=[true] REDACTED -|_|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(sum(tsid_metric.val))] REDACTED +|_|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts], aggr=[count(tsid_metric.ts) as count(sum(tsid_metric.val))] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(sum(tsid_metric.val))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts], aggr=[count(tsid_metric.ts) as count(sum(tsid_metric.val))] REDACTED |_|_|_ProjectionExec: expr=[ts@0 as ts] REDACTED |_|_|_AggregateExec: mode=FinalPartitioned, gby=[ts@0 as ts, job@1 as job], aggr=[] REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_AggregateExec: mode=Partial, gby=[ts@0 as ts, job@1 as job], aggr=[] REDACTED -|_|_|_ProjectionExec: expr=[ts@1 as ts, job@0 as job] REDACTED -|_|_|_FilterExec: val@0 IS NOT NULL, projection=[job@1, ts@2] REDACTED +|_|_|_FilterExec: val@0 IS NOT NULL, projection=[ts@2, job@1] REDACTED |_|_|_ProjectionExec: expr=[val@0 as val, job@1 as job, ts@3 as ts] REDACTED |_|_|_PromInstantManipulateExec: range=[0..10000], lookback=[300000], interval=[5000], time index=[ts] REDACTED |_|_|_PromSeriesDivideExec: tags=["__tsid"] REDACTED diff --git a/tests/cases/standalone/tql/general_table.result b/tests/cases/standalone/tql/general_table.result new file mode 100644 index 0000000000..e4add60c60 --- /dev/null +++ b/tests/cases/standalone/tql/general_table.result @@ -0,0 +1,53 @@ +-- run PromQL on non-typical prometheus table schema +CREATE TABLE IF NOT EXISTS `cpu_usage` ( + `job` STRING NULL, + `value` DOUBLE NULL, + `ts` TIMESTAMP(9) NOT NULL, + TIME INDEX (`ts`), + PRIMARY KEY (`job`) +) +ENGINE=mito +WITH( + merge_mode = 'last_non_null' +); + +Affected Rows: 0 + +-- SQLNESS REPLACE (metrics.*) REDACTED +-- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED +-- SQLNESS REPLACE (Hash.*) REDACTED +-- SQLNESS REPLACE (-+) - +-- SQLNESS REPLACE (\s\s+) _ +-- SQLNESS REPLACE (cpu_usage\.ts_range) ts_range +-- SQLNESS REPLACE (cpu_usage\.ts) ts +-- SQLNESS REPLACE (SeriesScan:.*|SeqScan:.*) ScanExec: REDACTED +-- SQLNESS REPLACE (peers.*) REDACTED +-- SQLNESS REPLACE region=\d+\(\d+,\s+\d+\) region=REDACTED +TQL analyze (0, 10, '1s') sum by(job) (irate(cpu_usage{job="fire"}[5s])) / 1e9; + ++-+-+-+ +| stage | node | plan_| ++-+-+-+ +| 0_| 0_|_CooperativeExec REDACTED +|_|_|_MergeScanExec: REDACTED +|_|_|_| +| 1_| 0_|_ProjectionExec: expr=[job@0 as job, ts@1 as ts, sum(prom_irate(ts_range,value))@2 / 1000000000 as sum(prom_irate(ts_range,value)) / Float64(1000000000)] REDACTED +|_|_|_RepartitionExec: partitioning=REDACTED +|_|_|_SortPreservingMergeExec: [job@0 ASC NULLS LAST, ts@1 ASC NULLS LAST] REDACTED +|_|_|_SortExec: expr=[job@0 ASC NULLS LAST, ts@1 ASC NULLS LAST], preserve_partitioning=[true] REDACTED +|_|_|_AggregateExec: mode=SinglePartitioned, gby=[job@2 as job, ts@0 as ts], aggr=[sum(prom_irate(ts_range,value))] REDACTED +|_|_|_FilterExec: prom_irate(ts_range,value)@1 IS NOT NULL REDACTED +|_|_|_ProjectionExec: expr=[ts@2 as ts, prom_irate(ts_range@3, value@0) as prom_irate(ts_range,value), job@1 as job] REDACTED +|_|_|_PromRangeManipulateExec: req range=[0..10000], interval=[1000], eval range=[5000], time index=[ts] REDACTED +|_|_|_PromSeriesNormalizeExec: offset=[0], time index=[ts], filter NaN: [true] REDACTED +|_|_|_PromSeriesDivideExec: tags=["job"] REDACTED +|_|_|_ProjectionExec: expr=[value@1 as value, job@0 as job, ts@2 as ts] REDACTED +|_|_|_ScanExec: REDACTED +|_|_|_| +|_|_| Total rows: 0_| ++-+-+-+ + +drop table `cpu_usage`; + +Affected Rows: 0 + diff --git a/tests/cases/standalone/tql/general_table.sql b/tests/cases/standalone/tql/general_table.sql new file mode 100644 index 0000000000..53dd5b4e7f --- /dev/null +++ b/tests/cases/standalone/tql/general_table.sql @@ -0,0 +1,26 @@ +-- run PromQL on non-typical prometheus table schema +CREATE TABLE IF NOT EXISTS `cpu_usage` ( + `job` STRING NULL, + `value` DOUBLE NULL, + `ts` TIMESTAMP(9) NOT NULL, + TIME INDEX (`ts`), + PRIMARY KEY (`job`) +) +ENGINE=mito +WITH( + merge_mode = 'last_non_null' +); + +-- SQLNESS REPLACE (metrics.*) REDACTED +-- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED +-- SQLNESS REPLACE (Hash.*) REDACTED +-- SQLNESS REPLACE (-+) - +-- SQLNESS REPLACE (\s\s+) _ +-- SQLNESS REPLACE (cpu_usage\.ts_range) ts_range +-- SQLNESS REPLACE (cpu_usage\.ts) ts +-- SQLNESS REPLACE (SeriesScan:.*|SeqScan:.*) ScanExec: REDACTED +-- SQLNESS REPLACE (peers.*) REDACTED +-- SQLNESS REPLACE region=\d+\(\d+,\s+\d+\) region=REDACTED +TQL analyze (0, 10, '1s') sum by(job) (irate(cpu_usage{job="fire"}[5s])) / 1e9; + +drop table `cpu_usage`; diff --git a/tests/compatibility/cases/analyze_verbose_remote_metrics_extension/verify.result b/tests/compatibility/cases/analyze_verbose_remote_metrics_extension/verify.result index 45aac8d9c3..7394ca8fae 100644 --- a/tests/compatibility/cases/analyze_verbose_remote_metrics_extension/verify.result +++ b/tests/compatibility/cases/analyze_verbose_remote_metrics_extension/verify.result @@ -14,9 +14,9 @@ EXPLAIN ANALYZE VERBOSE SELECT count(*) FROM t_analyze_verbose_remote_metrics_ex | stage | node | plan_| +-+-+-+ | 0_| 0_|_ProjectionExec: expr=[count(Int64(1))@0 as count(*)] REDACTED -|_|_|_AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Final, gby=[], aggr=[__count_merge(__count_state(t_analyze_verbose_remote_REDACTED |_|_|_CoalescePartitionsExec REDACTED -|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] REDACTED +|_|_|_AggregateExec: mode=Partial, gby=[], aggr=[__count_merge(__count_state(t_analyze_verbose_remote_REDACTED |_|_|_RepartitionExec: partitioning=REDACTED |_|_|_MergeScanExec: REDACTED |_|_|_| diff --git a/tests/compatibility/cases/datafusion_substrait_view/case.toml b/tests/compatibility/cases/datafusion_substrait_view/case.toml new file mode 100644 index 0000000000..3794292b67 --- /dev/null +++ b/tests/compatibility/cases/datafusion_substrait_view/case.toml @@ -0,0 +1,8 @@ +name = "datafusion_substrait_view" +reason = "Verify persisted CREATE VIEW Substrait plans remain executable across DataFusion upgrades." +introduced_by = "DataFusion 54 upgrade" +topologies = ["distributed"] +from_range = ["*"] +to_range = ["*"] +features = ["view", "substrait", "table", "query"] +owner = "query" diff --git a/tests/compatibility/cases/datafusion_substrait_view/setup.sql b/tests/compatibility/cases/datafusion_substrait_view/setup.sql new file mode 100644 index 0000000000..43f9aa43f4 --- /dev/null +++ b/tests/compatibility/cases/datafusion_substrait_view/setup.sql @@ -0,0 +1,80 @@ +CREATE TABLE view_metrics ( + host STRING, + "region" STRING, + "value" INT, + event_ts TIMESTAMP TIME INDEX, + PRIMARY KEY (host) +); + +INSERT INTO view_metrics VALUES + ('alpha', 'east', 10, '2024-01-01 00:00:00'), + ('beta', 'east', NULL, '2024-01-01 00:01:00'), + ('gamma', 'west', 30, '2024-01-01 00:02:00'), + ('delta', 'west', 5, '2024-01-01 00:03:00'); + +CREATE TABLE view_labels ( + host STRING, + label STRING, + enabled BOOLEAN, + label_ts TIMESTAMP TIME INDEX, + PRIMARY KEY (host) +); + +INSERT INTO view_labels VALUES + ('alpha', 'production', TRUE, '2024-01-01 00:00:00'), + ('beta', 'staging', NULL, '2024-01-01 00:01:00'), + ('gamma', NULL, FALSE, '2024-01-01 00:02:00'), + ('orphan', 'unused', TRUE, '2024-01-01 00:03:00'); + +CREATE VIEW persisted_scan_filter AS +SELECT + host AS host_name, + value AS metric_value, + CAST(event_ts AS TIMESTAMP(3)) AS event_time +FROM view_metrics +WHERE value IS NOT NULL AND value >= 10; + +CREATE VIEW persisted_grouped_aggregate AS +SELECT + "region" AS region_name, + COUNT(*) AS row_count, + SUM(value) AS value_sum, + AVG(value) AS value_avg +FROM view_metrics +GROUP BY "region"; + +CREATE VIEW persisted_union AS +SELECT host AS host_name, "region" AS region_name +FROM view_metrics +WHERE "region" = 'east' +UNION +SELECT host AS host_name, "region" AS region_name +FROM view_metrics +WHERE host = 'beta' +UNION ALL +SELECT host AS host_name, "region" AS region_name +FROM view_metrics +WHERE host = 'beta'; + +CREATE VIEW persisted_inner_join AS +SELECT + metrics.host AS host_name, + metrics.value AS metric_value, + labels.label AS label_name +FROM view_metrics AS metrics +INNER JOIN view_labels AS labels ON metrics.host = labels.host +WHERE metrics.value IS NOT NULL OR labels.enabled IS NULL; + +CREATE VIEW persisted_window AS +SELECT + "region" AS region_name, + host AS host_name, + event_ts AS event_time, + ROW_NUMBER() OVER (PARTITION BY "region" ORDER BY event_ts) AS row_num +FROM view_metrics; + +CREATE VIEW persisted_date_format AS +SELECT + host AS host_name, + date_format(event_ts, '%Y-%m-%d') AS event_day +FROM view_metrics; diff --git a/tests/compatibility/cases/datafusion_substrait_view/verify.result b/tests/compatibility/cases/datafusion_substrait_view/verify.result new file mode 100644 index 0000000000..28dabc3fd2 --- /dev/null +++ b/tests/compatibility/cases/datafusion_substrait_view/verify.result @@ -0,0 +1,71 @@ +SELECT host_name, metric_value, event_time +FROM persisted_scan_filter +ORDER BY host_name; + ++-----------+--------------+---------------------+ +| host_name | metric_value | event_time | ++-----------+--------------+---------------------+ +| alpha | 10 | 2024-01-01T00:00:00 | +| gamma | 30 | 2024-01-01T00:02:00 | ++-----------+--------------+---------------------+ + +SELECT region_name, row_count, value_sum, value_avg +FROM persisted_grouped_aggregate +ORDER BY region_name; + ++-------------+-----------+-----------+-----------+ +| region_name | row_count | value_sum | value_avg | ++-------------+-----------+-----------+-----------+ +| east | 2 | 10 | 10.0 | +| west | 2 | 35 | 17.5 | ++-------------+-----------+-----------+-----------+ + +SELECT host_name, region_name +FROM persisted_union +ORDER BY host_name, region_name; + ++-----------+-------------+ +| host_name | region_name | ++-----------+-------------+ +| alpha | east | +| beta | east | +| beta | east | ++-----------+-------------+ + +SELECT host_name, metric_value, label_name +FROM persisted_inner_join +ORDER BY host_name; + ++-----------+--------------+------------+ +| host_name | metric_value | label_name | ++-----------+--------------+------------+ +| alpha | 10 | production | +| beta | | staging | +| gamma | 30 | | ++-----------+--------------+------------+ + +SELECT region_name, host_name, event_time, row_num +FROM persisted_window +ORDER BY region_name, row_num; + ++-------------+-----------+---------------------+---------+ +| region_name | host_name | event_time | row_num | ++-------------+-----------+---------------------+---------+ +| east | alpha | 2024-01-01T00:00:00 | 1 | +| east | beta | 2024-01-01T00:01:00 | 2 | +| west | gamma | 2024-01-01T00:02:00 | 1 | +| west | delta | 2024-01-01T00:03:00 | 2 | ++-------------+-----------+---------------------+---------+ + +SELECT host_name, event_day +FROM persisted_date_format +ORDER BY host_name; + ++-----------+------------+ +| host_name | event_day | ++-----------+------------+ +| alpha | 2024-01-01 | +| beta | 2024-01-01 | +| delta | 2024-01-01 | +| gamma | 2024-01-01 | ++-----------+------------+ diff --git a/tests/compatibility/cases/datafusion_substrait_view/verify.sql b/tests/compatibility/cases/datafusion_substrait_view/verify.sql new file mode 100644 index 0000000000..0f2f39de48 --- /dev/null +++ b/tests/compatibility/cases/datafusion_substrait_view/verify.sql @@ -0,0 +1,23 @@ +SELECT host_name, metric_value, event_time +FROM persisted_scan_filter +ORDER BY host_name; + +SELECT region_name, row_count, value_sum, value_avg +FROM persisted_grouped_aggregate +ORDER BY region_name; + +SELECT host_name, region_name +FROM persisted_union +ORDER BY host_name, region_name; + +SELECT host_name, metric_value, label_name +FROM persisted_inner_join +ORDER BY host_name; + +SELECT region_name, host_name, event_time, row_num +FROM persisted_window +ORDER BY region_name, row_num; + +SELECT host_name, event_day +FROM persisted_date_format +ORDER BY host_name;