mirror of
https://github.com/GreptimeTeam/greptimedb.git
synced 2026-10-03 10:35:35 +00:00
* fix: align ordered aggregate state type with the accumulator output Signed-off-by: Dennis Zhuang <killme2008@gmail.com> * fix: reject mismatched aggregate states and keep hard-ordered aggregates unsplit Signed-off-by: Dennis Zhuang <killme2008@gmail.com> * fix: keep WITHIN GROUP aggregates splittable and relabel state fields by position Signed-off-by: Dennis Zhuang <killme2008@gmail.com> --------- Signed-off-by: Dennis Zhuang <killme2008@gmail.com>
238 lines
11 KiB
Plaintext
238 lines
11 KiB
Plaintext
-- Migrated from DuckDB test style: test array aggregation
|
|
-- Test ARRAY_AGG function
|
|
-- Test with integers
|
|
CREATE TABLE integers(i INTEGER, g INTEGER, ts TIMESTAMP TIME INDEX);
|
|
|
|
Affected Rows: 0
|
|
|
|
INSERT INTO integers VALUES (1, 1, 1000), (2, 1, 2000), (3, 1, 3000), (4, 2, 4000), (5, 2, 5000);
|
|
|
|
Affected Rows: 5
|
|
|
|
-- Basic array aggregation
|
|
SELECT array_agg(i) FROM integers;
|
|
|
|
+-----------------------+
|
|
| array_agg(integers.i) |
|
|
+-----------------------+
|
|
| [1, 2, 3, 4, 5] |
|
|
+-----------------------+
|
|
|
|
-- Array aggregation with GROUP BY
|
|
SELECT g, array_agg(i) FROM integers GROUP BY g ORDER BY g;
|
|
|
|
+---+-----------------------+
|
|
| g | array_agg(integers.i) |
|
|
+---+-----------------------+
|
|
| 1 | [1, 2, 3] |
|
|
| 2 | [4, 5] |
|
|
+---+-----------------------+
|
|
|
|
-- Test with ORDER BY
|
|
SELECT array_agg(i ORDER BY i DESC) FROM integers;
|
|
|
|
+--------------------------------------------------------------+
|
|
| array_agg(integers.i) ORDER BY [integers.i DESC NULLS FIRST] |
|
|
+--------------------------------------------------------------+
|
|
| [5, 4, 3, 2, 1] |
|
|
+--------------------------------------------------------------+
|
|
|
|
SELECT g, array_agg(i ORDER BY i DESC) FROM integers GROUP BY g ORDER BY g;
|
|
|
|
+---+--------------------------------------------------------------+
|
|
| g | array_agg(integers.i) ORDER BY [integers.i DESC NULLS FIRST] |
|
|
+---+--------------------------------------------------------------+
|
|
| 1 | [3, 2, 1] |
|
|
| 2 | [5, 4] |
|
|
+---+--------------------------------------------------------------+
|
|
|
|
-- Test with strings
|
|
CREATE TABLE strings(s VARCHAR, g INTEGER, ts TIMESTAMP TIME INDEX);
|
|
|
|
Affected Rows: 0
|
|
|
|
INSERT INTO strings VALUES
|
|
('apple', 1, 1000), ('banana', 1, 2000), ('cherry', 2, 3000),
|
|
('date', 2, 4000), ('elderberry', 1, 5000);
|
|
|
|
Affected Rows: 5
|
|
|
|
SELECT array_agg(s) FROM strings;
|
|
|
|
+-------------------------------------------+
|
|
| array_agg(strings.s) |
|
|
+-------------------------------------------+
|
|
| [apple, banana, cherry, date, elderberry] |
|
|
+-------------------------------------------+
|
|
|
|
SELECT g, array_agg(s ORDER BY s) FROM strings GROUP BY g ORDER BY g;
|
|
|
|
+---+----------------------------------------------------------+
|
|
| g | array_agg(strings.s) ORDER BY [strings.s ASC NULLS LAST] |
|
|
+---+----------------------------------------------------------+
|
|
| 1 | [apple, banana, elderberry] |
|
|
| 2 | [cherry, date] |
|
|
+---+----------------------------------------------------------+
|
|
|
|
-- Test with NULL values
|
|
INSERT INTO strings VALUES (NULL, 1, 6000), ('fig', NULL, 7000);
|
|
|
|
Affected Rows: 2
|
|
|
|
SELECT array_agg(s) FROM strings WHERE s IS NOT NULL;
|
|
|
|
+------------------------------------------------+
|
|
| array_agg(strings.s) |
|
|
+------------------------------------------------+
|
|
| [apple, banana, cherry, date, elderberry, fig] |
|
|
+------------------------------------------------+
|
|
|
|
SELECT g, array_agg(s) FROM strings WHERE g IS NOT NULL GROUP BY g ORDER BY g;
|
|
|
|
+---+-------------------------------+
|
|
| g | array_agg(strings.s) |
|
|
+---+-------------------------------+
|
|
| 1 | [apple, banana, elderberry, ] |
|
|
| 2 | [cherry, date] |
|
|
+---+-------------------------------+
|
|
|
|
-- Test with DISTINCT
|
|
SELECT array_agg(DISTINCT s ORDER BY s) FROM strings WHERE s IS NOT NULL;
|
|
|
|
+-------------------------------------------------------------------+
|
|
| array_agg(DISTINCT strings.s) ORDER BY [strings.s ASC NULLS LAST] |
|
|
+-------------------------------------------------------------------+
|
|
| [apple, banana, cherry, date, elderberry, fig] |
|
|
+-------------------------------------------------------------------+
|
|
|
|
-- Test empty result
|
|
SELECT array_agg(i) FROM integers WHERE i > 100;
|
|
|
|
+-----------------------+
|
|
| array_agg(integers.i) |
|
|
+-----------------------+
|
|
| |
|
|
+-----------------------+
|
|
|
|
-- Test with doubles
|
|
CREATE TABLE doubles(d DOUBLE, ts TIMESTAMP TIME INDEX);
|
|
|
|
Affected Rows: 0
|
|
|
|
INSERT INTO doubles VALUES (1.1, 1000), (2.2, 2000), (3.3, 3000), (4.4, 4000);
|
|
|
|
Affected Rows: 4
|
|
|
|
SELECT array_agg(d ORDER BY d) FROM doubles;
|
|
|
|
+----------------------------------------------------------+
|
|
| array_agg(doubles.d) ORDER BY [doubles.d ASC NULLS LAST] |
|
|
+----------------------------------------------------------+
|
|
| [1.1, 2.2, 3.3, 4.4] |
|
|
+----------------------------------------------------------+
|
|
|
|
-- cleanup
|
|
DROP TABLE integers;
|
|
|
|
Affected Rows: 0
|
|
|
|
DROP TABLE strings;
|
|
|
|
Affected Rows: 0
|
|
|
|
DROP TABLE doubles;
|
|
|
|
Affected Rows: 0
|
|
|
|
-- On partitioned tables the aggregate is split into partial state and merge
|
|
CREATE TABLE array_agg_partitioned (
|
|
ts TIMESTAMP TIME INDEX,
|
|
k INT,
|
|
lat DOUBLE,
|
|
PRIMARY KEY(k)
|
|
)
|
|
PARTITION ON COLUMNS (k) (k < 10, k >= 10 AND k < 20, k >= 20);
|
|
|
|
Affected Rows: 0
|
|
|
|
INSERT INTO array_agg_partitioned VALUES
|
|
(1000, 1, 1),
|
|
(2000, 11, 2),
|
|
(3000, 21, 3),
|
|
(4000, 2, 4),
|
|
(5000, 12, 5),
|
|
(6000, 22, 6),
|
|
(7000, 3, 7);
|
|
|
|
Affected Rows: 7
|
|
|
|
SELECT array_agg(lat ORDER BY ts) FROM array_agg_partitioned;
|
|
|
|
+-----------------------------------------------------------------------------------------+
|
|
| array_agg(array_agg_partitioned.lat) ORDER BY [array_agg_partitioned.ts ASC NULLS LAST] |
|
|
+-----------------------------------------------------------------------------------------+
|
|
| [1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0] |
|
|
+-----------------------------------------------------------------------------------------+
|
|
|
|
SELECT lat > 3 AS g, array_agg(lat ORDER BY ts DESC) FROM array_agg_partitioned GROUP BY g ORDER BY g;
|
|
|
|
+-------+-------------------------------------------------------------------------------------------+
|
|
| g | array_agg(array_agg_partitioned.lat) ORDER BY [array_agg_partitioned.ts DESC NULLS FIRST] |
|
|
+-------+-------------------------------------------------------------------------------------------+
|
|
| false | [3.0, 2.0, 1.0] |
|
|
| true | [7.0, 6.0, 5.0, 4.0] |
|
|
+-------+-------------------------------------------------------------------------------------------+
|
|
|
|
SELECT array_agg(k ORDER BY k % 10, ts DESC) FROM array_agg_partitioned;
|
|
|
|
+---------------------------------------------------------------------------------------------------------------------------------------------+
|
|
| array_agg(array_agg_partitioned.k) ORDER BY [array_agg_partitioned.k % Int64(10) ASC NULLS LAST, array_agg_partitioned.ts DESC NULLS FIRST] |
|
|
+---------------------------------------------------------------------------------------------------------------------------------------------+
|
|
| [21, 11, 1, 22, 12, 2, 3] |
|
|
+---------------------------------------------------------------------------------------------------------------------------------------------+
|
|
|
|
-- nth_value needs sorted input, so it isn't split into partial state and merge
|
|
SELECT nth_value(lat, 2 ORDER BY ts), nth_value(lat, 3 ORDER BY ts DESC) FROM array_agg_partitioned;
|
|
|
|
+--------------------------------------------------------------------------------------------------+----------------------------------------------------------------------------------------------------+
|
|
| nth_value(array_agg_partitioned.lat,Int64(2)) ORDER BY [array_agg_partitioned.ts ASC NULLS LAST] | nth_value(array_agg_partitioned.lat,Int64(3)) ORDER BY [array_agg_partitioned.ts DESC NULLS FIRST] |
|
|
+--------------------------------------------------------------------------------------------------+----------------------------------------------------------------------------------------------------+
|
|
| 2.0 | 5.0 |
|
|
+--------------------------------------------------------------------------------------------------+----------------------------------------------------------------------------------------------------+
|
|
|
|
-- WITHIN GROUP aggregates don't need sorted input and are still split
|
|
-- SQLNESS REPLACE (peers.*) REDACTED
|
|
-- SQLNESS REPLACE (RoundRobinBatch.*) REDACTED
|
|
-- SQLNESS REPLACE (-+) -
|
|
-- SQLNESS REPLACE (\s\s+) _
|
|
EXPLAIN SELECT sum(lat), approx_percentile_cont(0.75) WITHIN GROUP (ORDER BY lat DESC) FROM array_agg_partitioned;
|
|
|
|
+-+-+
|
|
| plan_type_| plan_|
|
|
+-+-+
|
|
| logical_plan_| Aggregate: groupBy=[[]], aggr=[[__sum_merge(__sum_state(array_agg_partitioned.lat)) AS sum(array_agg_partitioned.lat), __approx_percentile_cont_merge(__approx_percentile_cont_state(array_agg_partitioned.lat,Float64(0.75)) ORDER BY [array_agg_partitioned.lat DESC NULLS FIRST]) AS approx_percentile_cont(Float64(0.75)) WITHIN GROUP [array_agg_partitioned.lat DESC NULLS FIRST]]]_|
|
|
|_|_MergeScan [is_placeholder=false, remote_input=[_|
|
|
|_| Aggregate: groupBy=[[]], aggr=[[__sum_state(array_agg_partitioned.lat), __approx_percentile_cont_state(array_agg_partitioned.lat, Float64(0.75)) ORDER BY [array_agg_partitioned.lat DESC NULLS FIRST]]]_|
|
|
|_|_TableScan: array_agg_partitioned_|
|
|
|_| ]]_|
|
|
| physical_plan | AggregateExec: mode=Final, gby=[], aggr=[__sum_merge(__sum_state(array_agg_partitioned.lat)) as sum(array_agg_partitioned.lat), __approx_percentile_cont_merge(__approx_percentile_cont_state(array_agg_partitioned.lat,Float64(0.75)) ORDER BY [array_agg_partitioned.lat DESC NULLS FIRST]) as approx_percentile_cont(Float64(0.75)) WITHIN GROUP [array_agg_partitioned.lat DESC NULLS FIRST]]_|
|
|
|_|_CoalescePartitionsExec_|
|
|
|_|_AggregateExec: mode=Partial, gby=[], aggr=[__sum_merge(__sum_state(array_agg_partitioned.lat)) as sum(array_agg_partitioned.lat), __approx_percentile_cont_merge(__approx_percentile_cont_state(array_agg_partitioned.lat,Float64(0.75)) ORDER BY [array_agg_partitioned.lat DESC NULLS FIRST]) as approx_percentile_cont(Float64(0.75)) WITHIN GROUP [array_agg_partitioned.lat DESC NULLS FIRST]] |
|
|
|_|_RepartitionExec: partitioning=REDACTED
|
|
|_|_MergeScanExec: REDACTED
|
|
|_|_|
|
|
+-+-+
|
|
|
|
SELECT sum(lat), approx_percentile_cont(0.75) WITHIN GROUP (ORDER BY lat DESC) FROM array_agg_partitioned;
|
|
|
|
+--------------------------------+-------------------------------------------------------------------------------------------------+
|
|
| sum(array_agg_partitioned.lat) | approx_percentile_cont(Float64(0.75)) WITHIN GROUP [array_agg_partitioned.lat DESC NULLS FIRST] |
|
|
+--------------------------------+-------------------------------------------------------------------------------------------------+
|
|
| 28.0 | 2.25 |
|
|
+--------------------------------+-------------------------------------------------------------------------------------------------+
|
|
|
|
DROP TABLE array_agg_partitioned;
|
|
|
|
Affected Rows: 0
|
|
|