Files
greptimedb/tests/cases/standalone/common/aggregate/array_agg.result
T
dennis zhuang affc0a1b1d fix: align ordered aggregate state type with the accumulator output (#9340)
* fix: align ordered aggregate state type with the accumulator output

Signed-off-by: Dennis Zhuang <killme2008@gmail.com>

* fix: reject mismatched aggregate states and keep hard-ordered aggregates unsplit

Signed-off-by: Dennis Zhuang <killme2008@gmail.com>

* fix: keep WITHIN GROUP aggregates splittable and relabel state fields by position

Signed-off-by: Dennis Zhuang <killme2008@gmail.com>

---------

Signed-off-by: Dennis Zhuang <killme2008@gmail.com>
2026-09-24 07:02:14 +00:00

238 lines
11 KiB
Plaintext

-- Migrated from DuckDB test style: test array aggregation
-- Test ARRAY_AGG function
-- Test with integers
CREATE TABLE integers(i INTEGER, g INTEGER, ts TIMESTAMP TIME INDEX);
Affected Rows: 0
INSERT INTO integers VALUES (1, 1, 1000), (2, 1, 2000), (3, 1, 3000), (4, 2, 4000), (5, 2, 5000);
Affected Rows: 5
-- Basic array aggregation
SELECT array_agg(i) FROM integers;
+-----------------------+
| array_agg(integers.i) |
+-----------------------+
| [1, 2, 3, 4, 5] |
+-----------------------+
-- Array aggregation with GROUP BY
SELECT g, array_agg(i) FROM integers GROUP BY g ORDER BY g;
+---+-----------------------+
| g | array_agg(integers.i) |
+---+-----------------------+
| 1 | [1, 2, 3] |
| 2 | [4, 5] |
+---+-----------------------+
-- Test with ORDER BY
SELECT array_agg(i ORDER BY i DESC) FROM integers;
+--------------------------------------------------------------+
| array_agg(integers.i) ORDER BY [integers.i DESC NULLS FIRST] |
+--------------------------------------------------------------+
| [5, 4, 3, 2, 1] |
+--------------------------------------------------------------+
SELECT g, array_agg(i ORDER BY i DESC) FROM integers GROUP BY g ORDER BY g;
+---+--------------------------------------------------------------+
| g | array_agg(integers.i) ORDER BY [integers.i DESC NULLS FIRST] |
+---+--------------------------------------------------------------+
| 1 | [3, 2, 1] |
| 2 | [5, 4] |
+---+--------------------------------------------------------------+
-- Test with strings
CREATE TABLE strings(s VARCHAR, g INTEGER, ts TIMESTAMP TIME INDEX);
Affected Rows: 0
INSERT INTO strings VALUES
('apple', 1, 1000), ('banana', 1, 2000), ('cherry', 2, 3000),
('date', 2, 4000), ('elderberry', 1, 5000);
Affected Rows: 5
SELECT array_agg(s) FROM strings;
+-------------------------------------------+
| array_agg(strings.s) |
+-------------------------------------------+
| [apple, banana, cherry, date, elderberry] |
+-------------------------------------------+
SELECT g, array_agg(s ORDER BY s) FROM strings GROUP BY g ORDER BY g;
+---+----------------------------------------------------------+
| g | array_agg(strings.s) ORDER BY [strings.s ASC NULLS LAST] |
+---+----------------------------------------------------------+
| 1 | [apple, banana, elderberry] |
| 2 | [cherry, date] |
+---+----------------------------------------------------------+
-- Test with NULL values
INSERT INTO strings VALUES (NULL, 1, 6000), ('fig', NULL, 7000);
Affected Rows: 2
SELECT array_agg(s) FROM strings WHERE s IS NOT NULL;
+------------------------------------------------+
| array_agg(strings.s) |
+------------------------------------------------+
| [apple, banana, cherry, date, elderberry, fig] |
+------------------------------------------------+
SELECT g, array_agg(s) FROM strings WHERE g IS NOT NULL GROUP BY g ORDER BY g;
+---+-------------------------------+
| g | array_agg(strings.s) |
+---+-------------------------------+
| 1 | [apple, banana, elderberry, ] |
| 2 | [cherry, date] |
+---+-------------------------------+
-- Test with DISTINCT
SELECT array_agg(DISTINCT s ORDER BY s) FROM strings WHERE s IS NOT NULL;
+-------------------------------------------------------------------+
| array_agg(DISTINCT strings.s) ORDER BY [strings.s ASC NULLS LAST] |
+-------------------------------------------------------------------+
| [apple, banana, cherry, date, elderberry, fig] |
+-------------------------------------------------------------------+
-- Test empty result
SELECT array_agg(i) FROM integers WHERE i > 100;
+-----------------------+
| array_agg(integers.i) |
+-----------------------+
| |
+-----------------------+
-- Test with doubles
CREATE TABLE doubles(d DOUBLE, ts TIMESTAMP TIME INDEX);
Affected Rows: 0
INSERT INTO doubles VALUES (1.1, 1000), (2.2, 2000), (3.3, 3000), (4.4, 4000);
Affected Rows: 4
SELECT array_agg(d ORDER BY d) FROM doubles;
+----------------------------------------------------------+
| array_agg(doubles.d) ORDER BY [doubles.d ASC NULLS LAST] |
+----------------------------------------------------------+
| [1.1, 2.2, 3.3, 4.4] |
+----------------------------------------------------------+
-- cleanup
DROP TABLE integers;
Affected Rows: 0
DROP TABLE strings;
Affected Rows: 0
DROP TABLE doubles;
Affected Rows: 0
-- On partitioned tables the aggregate is split into partial state and merge
CREATE TABLE array_agg_partitioned (
ts TIMESTAMP TIME INDEX,
k INT,
lat DOUBLE,
PRIMARY KEY(k)
)
PARTITION ON COLUMNS (k) (k < 10, k >= 10 AND k < 20, k >= 20);
Affected Rows: 0
INSERT INTO array_agg_partitioned VALUES
(1000, 1, 1),
(2000, 11, 2),
(3000, 21, 3),
(4000, 2, 4),
(5000, 12, 5),
(6000, 22, 6),
(7000, 3, 7);
Affected Rows: 7
SELECT array_agg(lat ORDER BY ts) FROM array_agg_partitioned;
+-----------------------------------------------------------------------------------------+
| array_agg(array_agg_partitioned.lat) ORDER BY [array_agg_partitioned.ts ASC NULLS LAST] |
+-----------------------------------------------------------------------------------------+
| [1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0] |
+-----------------------------------------------------------------------------------------+
SELECT lat > 3 AS g, array_agg(lat ORDER BY ts DESC) FROM array_agg_partitioned GROUP BY g ORDER BY g;
+-------+-------------------------------------------------------------------------------------------+
| g | array_agg(array_agg_partitioned.lat) ORDER BY [array_agg_partitioned.ts DESC NULLS FIRST] |
+-------+-------------------------------------------------------------------------------------------+
| false | [3.0, 2.0, 1.0] |
| true | [7.0, 6.0, 5.0, 4.0] |
+-------+-------------------------------------------------------------------------------------------+
SELECT array_agg(k ORDER BY k % 10, ts DESC) FROM array_agg_partitioned;
+---------------------------------------------------------------------------------------------------------------------------------------------+
| array_agg(array_agg_partitioned.k) ORDER BY [array_agg_partitioned.k % Int64(10) ASC NULLS LAST, array_agg_partitioned.ts DESC NULLS FIRST] |
+---------------------------------------------------------------------------------------------------------------------------------------------+
| [21, 11, 1, 22, 12, 2, 3] |
+---------------------------------------------------------------------------------------------------------------------------------------------+
-- nth_value needs sorted input, so it isn't split into partial state and merge
SELECT nth_value(lat, 2 ORDER BY ts), nth_value(lat, 3 ORDER BY ts DESC) FROM array_agg_partitioned;
+--------------------------------------------------------------------------------------------------+----------------------------------------------------------------------------------------------------+
| nth_value(array_agg_partitioned.lat,Int64(2)) ORDER BY [array_agg_partitioned.ts ASC NULLS LAST] | nth_value(array_agg_partitioned.lat,Int64(3)) ORDER BY [array_agg_partitioned.ts DESC NULLS FIRST] |
+--------------------------------------------------------------------------------------------------+----------------------------------------------------------------------------------------------------+
| 2.0 | 5.0 |
+--------------------------------------------------------------------------------------------------+----------------------------------------------------------------------------------------------------+
-- WITHIN GROUP aggregates don't need sorted input and are still split
-- SQLNESS REPLACE (peers.*) REDACTED
-- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED
-- SQLNESS REPLACE (-+) -
-- SQLNESS REPLACE (\s\s+) _
EXPLAIN SELECT sum(lat), approx_percentile_cont(0.75) WITHIN GROUP (ORDER BY lat DESC) FROM array_agg_partitioned;
+-+-+
| plan_type_| plan_|
+-+-+
| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__sum_merge(__sum_state(array_agg_partitioned.lat)) AS sum(array_agg_partitioned.lat), __approx_percentile_cont_merge(__approx_percentile_cont_state(array_agg_partitioned.lat,Float64(0.75)) ORDER BY [array_agg_partitioned.lat DESC NULLS FIRST]) AS approx_percentile_cont(Float64(0.75)) WITHIN GROUP [array_agg_partitioned.lat DESC NULLS FIRST]]]_|
|_|_MergeScan [is_placeholder=false, remote_input=[_|
|_| Aggregate: groupBy=[[]], aggr=[[__sum_state(array_agg_partitioned.lat), __approx_percentile_cont_state(array_agg_partitioned.lat, Float64(0.75)) ORDER BY [array_agg_partitioned.lat DESC NULLS FIRST]]]_|
|_|_TableScan: array_agg_partitioned_|
|_| ]]_|
| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[__sum_merge(__sum_state(array_agg_partitioned.lat)) as sum(array_agg_partitioned.lat), __approx_percentile_cont_merge(__approx_percentile_cont_state(array_agg_partitioned.lat,Float64(0.75)) ORDER BY [array_agg_partitioned.lat DESC NULLS FIRST]) as approx_percentile_cont(Float64(0.75)) WITHIN GROUP [array_agg_partitioned.lat DESC NULLS FIRST]]_|
|_|_CoalescePartitionsExec_|
|_|_AggregateExec: mode=Partial, gby=[], aggr=[__sum_merge(__sum_state(array_agg_partitioned.lat)) as sum(array_agg_partitioned.lat), __approx_percentile_cont_merge(__approx_percentile_cont_state(array_agg_partitioned.lat,Float64(0.75)) ORDER BY [array_agg_partitioned.lat DESC NULLS FIRST]) as approx_percentile_cont(Float64(0.75)) WITHIN GROUP [array_agg_partitioned.lat DESC NULLS FIRST]] |
|_|_RepartitionExec: partitioning=REDACTED
|_|_MergeScanExec: REDACTED
|_|_|
+-+-+
SELECT sum(lat), approx_percentile_cont(0.75) WITHIN GROUP (ORDER BY lat DESC) FROM array_agg_partitioned;
+--------------------------------+-------------------------------------------------------------------------------------------------+
| sum(array_agg_partitioned.lat) | approx_percentile_cont(Float64(0.75)) WITHIN GROUP [array_agg_partitioned.lat DESC NULLS FIRST] |
+--------------------------------+-------------------------------------------------------------------------------------------------+
| 28.0 | 2.25 |
+--------------------------------+-------------------------------------------------------------------------------------------------+
DROP TABLE array_agg_partitioned;
Affected Rows: 0