Files
greptimedb/tests-integration/tests/repartition.rs
T
dennis zhuang aa36f74feb fix(ci): build tests-integration lib with meta-srv/mock (#9299)
* fix(ci): build tests-integration lib with meta-srv/mock

tests-integration's lib code (src/cluster.rs) uses meta_srv::mocks, but
the dependency carrying the mock feature sits in [dev-dependencies].
Builds that only touch the lib, such as the apidoc job's cargo doc
--workspace, resolve meta-srv without mock and fail with E0432.
--all-targets builds unify dev-dependency features, which is why check,
clippy and nextest stayed green.

Move the mock-enabled meta-srv entry back to [dependencies]. The other
testing features moved out in #9072 are not needed by the lib and stay
in [dev-dependencies].

Signed-off-by: Dennis Zhuang <killme2008@gmail.com>

* test(repartition): split per-case repartition tests

test_repartition_metric ran four format/primary-key-encoding cases in a
single test function, and test_repartition_mito ran two format cases.
Each case builds its own 3-datanode cluster and runs a full repartition
plus GC cycle, so on S3 the metric test took 165-178s against the 180s
nextest terminate-after. Merge queue runs failed on it at random.

Split each case into its own test. Cases were already independent, so
they now run in parallel and each stays far inside the timeout, and a
failure points at one encoding instead of four.

Signed-off-by: Dennis Zhuang <killme2008@gmail.com>

---------

Signed-off-by: Dennis Zhuang <killme2008@gmail.com>
2026-09-22 14:38:22 +00:00

2164 lines
77 KiB
Rust

// Copyright 2023 Greptime Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use std::collections::BTreeSet;
use std::sync::Arc;
use std::time::Duration;
use client::{DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
use common_error::root_source;
use common_event_recorder::{PersistentEventContext, TriggerReason};
use common_meta::key::table_name::TableNameKey;
use common_procedure::{ProcedureContext, ProcedureWithId, watcher};
use common_query::Output;
use common_telemetry::info;
use common_test_util::recordbatch::check_output_stream;
use common_test_util::temp_dir::{TempDir, create_temp_dir};
use common_wal::config::DatanodeWalConfig;
use frontend::instance::Instance;
use meta_srv::gc::{self, BatchGcProcedure, GcSchedulerOptions, GcTickerRef};
use meta_srv::metasrv::Metasrv;
use mito2::gc::GcConfig;
use servers::error::Result as ServerResult;
use servers::query_handler::sql::SqlQueryHandler;
use session::context::{QueryContext, QueryContextRef};
use store_api::codec::PrimaryKeyEncoding;
use store_api::storage::{RegionId, TableId};
use tests_integration::cluster::{GreptimeDbCluster, GreptimeDbClusterBuilder};
use tests_integration::test_util::{StorageType, get_test_store_config};
use tokio::sync::oneshot;
#[macro_export]
macro_rules! repartition_tests {
($($service:ident),*) => {
$(
paste::item! {
mod [<integration_repartition_ $service:lower _test>] {
// Every case below builds its own cluster and runs a full repartition plus
// GC cycle, which takes around a minute on object-store backends. Keep one
// case per test so a case stays well inside the nextest slow timeout and
// cases run in parallel; do not fold several cases back into one test.
#[tokio::test(flavor = "multi_thread")]
async fn [< test_repartition_mito_flat >]() {
let store_type = tests_integration::test_util::StorageType::$service;
if store_type.test_on() {
common_telemetry::init_default_ut_logging();
$crate::repartition::test_repartition_mito(store_type, true).await;
}
}
#[tokio::test(flavor = "multi_thread")]
async fn [< test_repartition_mito_primary_key >]() {
let store_type = tests_integration::test_util::StorageType::$service;
if store_type.test_on() {
common_telemetry::init_default_ut_logging();
$crate::repartition::test_repartition_mito(store_type, false).await;
}
}
#[tokio::test(flavor = "multi_thread")]
async fn [< test_partition_unpartitioned_mito >]() {
let store_type = tests_integration::test_util::StorageType::$service;
if store_type.test_on() {
common_telemetry::init_default_ut_logging();
$crate::repartition::test_partition_unpartitioned_mito(store_type).await;
}
}
#[tokio::test(flavor = "multi_thread")]
async fn [< test_repartition_on_columns_metadata_mito >]() {
let store_type = tests_integration::test_util::StorageType::$service;
if store_type.test_on() {
common_telemetry::init_default_ut_logging();
$crate::repartition::test_repartition_on_columns_metadata_mito(store_type).await;
}
}
#[tokio::test(flavor = "multi_thread")]
async fn [< test_repartition_on_columns_data_correctness_mito >]() {
let store_type = tests_integration::test_util::StorageType::$service;
if store_type.test_on() {
common_telemetry::init_default_ut_logging();
$crate::repartition::test_repartition_on_columns_data_correctness_mito(store_type).await;
}
}
#[tokio::test(flavor = "multi_thread")]
async fn [< test_partition_unpartitioned_metric >]() {
let store_type = tests_integration::test_util::StorageType::$service;
if store_type.test_on() {
common_telemetry::init_default_ut_logging();
$crate::repartition::test_partition_unpartitioned_metric(store_type).await;
}
}
#[tokio::test(flavor = "multi_thread")]
async fn [< test_repartition_metric_flat_sparse >]() {
let store_type = tests_integration::test_util::StorageType::$service;
if store_type.test_on() {
use store_api::codec::PrimaryKeyEncoding;
common_telemetry::init_default_ut_logging();
$crate::repartition::test_repartition_metric(store_type, true, PrimaryKeyEncoding::Sparse).await;
}
}
#[tokio::test(flavor = "multi_thread")]
async fn [< test_repartition_metric_flat_dense >]() {
let store_type = tests_integration::test_util::StorageType::$service;
if store_type.test_on() {
use store_api::codec::PrimaryKeyEncoding;
common_telemetry::init_default_ut_logging();
$crate::repartition::test_repartition_metric(store_type, true, PrimaryKeyEncoding::Dense).await;
}
}
#[tokio::test(flavor = "multi_thread")]
async fn [< test_repartition_metric_primary_key_sparse >]() {
let store_type = tests_integration::test_util::StorageType::$service;
if store_type.test_on() {
use store_api::codec::PrimaryKeyEncoding;
common_telemetry::init_default_ut_logging();
$crate::repartition::test_repartition_metric(store_type, false, PrimaryKeyEncoding::Sparse).await;
}
}
#[tokio::test(flavor = "multi_thread")]
async fn [< test_repartition_metric_primary_key_dense >]() {
let store_type = tests_integration::test_util::StorageType::$service;
if store_type.test_on() {
use store_api::codec::PrimaryKeyEncoding;
common_telemetry::init_default_ut_logging();
$crate::repartition::test_repartition_metric(store_type, false, PrimaryKeyEncoding::Dense).await;
}
}
}
}
)*
};
}
#[tokio::test(flavor = "multi_thread")]
async fn test_repartition_physical_metric_without_logical_table() {
common_telemetry::init_default_ut_logging();
let (store_config, _guard) = get_test_store_config(&StorageType::File);
let home_dir = create_temp_dir("repartition_physical_metric_without_logical_table");
let cluster =
GreptimeDbClusterBuilder::new("test_repartition_physical_metric_without_logical_table")
.await
.with_datanodes(3)
.with_store_config(store_config)
.with_shared_home_dir(Arc::new(home_dir))
.with_datanode_wal_config(DatanodeWalConfig::Noop)
.with_metasrv_gc_config(GcSchedulerOptions {
enable: true,
gc_cooldown_period: Duration::from_nanos(1),
..Default::default()
})
.with_datanode_gc_config(GcConfig {
enable: true,
lingering_time: Some(Duration::from_secs(0)),
unknown_file_lingering_time: Duration::from_secs(0),
..Default::default()
})
.build(true)
.await;
let instance = cluster.fe_instance();
let query_ctx = QueryContext::arc();
let sql = r#"
CREATE TABLE `physical_metric_without_logical`(
`ts` TIMESTAMP TIME INDEX,
`val` DOUBLE,
`host` STRING PRIMARY KEY
) PARTITION ON COLUMNS (`host`) (
`host` < 'm',
`host` >= 'm'
) ENGINE = metric
WITH (
"physical_metric_table" = "true"
);
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let table_id = get_table_id(&cluster.metasrv, "physical_metric_without_logical").await;
let table_info = cluster
.metasrv
.table_metadata_manager()
.table_info_manager()
.get(table_id)
.await
.unwrap()
.unwrap();
let column_ids_before_split = table_info.table_info.meta.column_ids.clone();
assert_eq!(column_ids_before_split, vec![0, 1, 2]);
let sql = r#"
ALTER TABLE `physical_metric_without_logical` SPLIT PARTITION (
`host` < 'm'
) INTO (
`host` < 'g',
`host` >= 'g' AND `host` < 'm'
);
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let table_info = cluster
.metasrv
.table_metadata_manager()
.table_info_manager()
.get(table_id)
.await
.unwrap()
.unwrap();
assert_eq!(
table_info.table_info.meta.column_ids,
column_ids_before_split
);
let result = run_sql(
instance,
&query_partitions_sql("physical_metric_without_logical"),
query_ctx,
)
.await
.unwrap();
let expected = r#"+---------------+--------------+---------------------------------+----------------+----------------------+------------------------+-----------------------+----------------------------+
| table_catalog | table_schema | table_name | partition_name | partition_expression | partition_description | greptime_partition_id | partition_ordinal_position |
+---------------+--------------+---------------------------------+----------------+----------------------+------------------------+-----------------------+----------------------------+
| greptime | public | physical_metric_without_logical | p0 | host | host < g | 4398046511104 | 1 |
| greptime | public | physical_metric_without_logical | p1 | host | host >= m | 4398046511105 | 2 |
| greptime | public | physical_metric_without_logical | p2 | host | host >= g AND host < m | 4398046511106 | 3 |
+---------------+--------------+---------------------------------+----------------+----------------------+------------------------+-----------------------+----------------------------+"#;
check_output_stream(result.data, expected).await;
}
/// COUNT must use visible rows rather than the full row counts of shared SSTs.
#[tokio::test(flavor = "multi_thread")]
async fn test_repartition_append_count_file() {
let (cluster, _home_guard) = append_count_cluster("repartition_append_count").await;
let instance = cluster.fe_instance();
let table = "count_repartition";
prepare_append_count_table(instance, table, "").await;
assert_append_count(instance, table, 100, 1).await;
flush_append_count_table(instance, table).await;
assert_append_count(instance, table, 100, 1).await;
let sql = "ALTER TABLE count_repartition PARTITION ON COLUMNS (device_id) \
(device_id < 50, device_id >= 50)";
run_sql(instance, sql, QueryContext::arc()).await.unwrap();
wait_for_append_count_regions(instance, table, 2).await;
assert_append_count(instance, table, 100, 0).await;
check_append_count_after_write(instance, table, 0).await;
}
#[tokio::test(flavor = "multi_thread")]
async fn test_split_append_count_file() {
let (cluster, _home_guard) = append_count_cluster("split_append_count").await;
let instance = cluster.fe_instance();
let table = "count_split";
prepare_append_count_table(
instance,
table,
"PARTITION ON COLUMNS (device_id) (device_id < 50, device_id >= 50)",
)
.await;
assert_append_count(instance, table, 100, 2).await;
flush_append_count_table(instance, table).await;
assert_append_count(instance, table, 100, 2).await;
let sql = "ALTER TABLE count_split SPLIT PARTITION (device_id < 50) INTO \
(device_id < 25, device_id >= 25 AND device_id < 50)";
run_sql(instance, sql, QueryContext::arc()).await.unwrap();
wait_for_append_count_regions(instance, table, 3).await;
assert_append_count(instance, table, 100, 1).await;
check_append_count_after_write(instance, table, 1).await;
}
#[tokio::test(flavor = "multi_thread")]
async fn test_repartition_append_count_memtable_file() {
let (cluster, home_guard) = append_count_cluster("repartition_append_count_memtable").await;
assert!(home_guard.path().is_dir());
let instance = cluster.fe_instance();
let table = "count_repartition_memtable";
prepare_append_count_table(instance, table, "").await;
assert_append_count(instance, table, 100, 1).await;
// Entering staging must flush the populated memtable before changing partitions.
let sql = "ALTER TABLE count_repartition_memtable PARTITION ON COLUMNS (device_id) \
(device_id < 50, device_id >= 50)";
run_sql(instance, sql, QueryContext::arc()).await.unwrap();
wait_for_append_count_regions(instance, table, 2).await;
assert_append_count(instance, table, 100, 0).await;
check_append_count_after_write(instance, table, 0).await;
assert!(home_guard.path().is_dir());
}
async fn append_count_cluster(name: &str) -> (GreptimeDbCluster, Arc<TempDir>) {
common_telemetry::init_default_ut_logging();
let (store_config, _guard) = get_test_store_config(&StorageType::File);
let home_dir = Arc::new(create_temp_dir(name));
let cluster = GreptimeDbClusterBuilder::new(name)
.await
.with_shared_home_dir(Arc::clone(&home_dir))
.with_datanodes(3)
.with_store_config(store_config)
.with_datanode_wal_config(DatanodeWalConfig::Noop)
.with_metasrv_gc_config(GcSchedulerOptions {
enable: true,
..Default::default()
})
.with_datanode_gc_config(GcConfig {
enable: true,
..Default::default()
})
.build(true)
.await;
(cluster, home_dir)
}
async fn prepare_append_count_table(instance: &Arc<Instance>, table: &str, partitions: &str) {
let sql = format!(
"CREATE TABLE {table} (ts TIMESTAMP TIME INDEX, device_id INT) \
{partitions} ENGINE=mito WITH(append_mode='true')"
);
run_sql(instance, &sql, QueryContext::arc()).await.unwrap();
let sql = format!(
"INSERT INTO {table} SELECT to_timestamp_millis(value), CAST(value AS INT) \
FROM generate_series(0, 99)"
);
run_sql(instance, &sql, QueryContext::arc()).await.unwrap();
}
async fn flush_append_count_table(instance: &Arc<Instance>, table: &str) {
run_sql(
instance,
&format!("ADMIN flush_table('{table}')"),
QueryContext::arc(),
)
.await
.unwrap();
}
async fn check_append_count_after_write(
instance: &Arc<Instance>,
table: &str,
statistics_regions: usize,
) {
run_sql(
instance,
&format!("INSERT INTO {table} VALUES (to_timestamp_millis(100), 100)"),
QueryContext::arc(),
)
.await
.unwrap();
assert_append_count(instance, table, 101, statistics_regions).await;
flush_append_count_table(instance, table).await;
assert_append_count(instance, table, 101, statistics_regions).await;
}
async fn wait_for_append_count_regions(instance: &Arc<Instance>, table: &str, expected: usize) {
// Wait for frontend routing, not for COUNT to become correct.
tokio::time::timeout(Duration::from_secs(10), async {
loop {
let plan =
append_count_query(instance, &format!("EXPLAIN ANALYZE SELECT * FROM {table}"))
.await;
if plan.matches("UnorderedScan: region=").count() == expected {
return;
}
tokio::time::sleep(Duration::from_millis(50)).await;
}
})
.await
.expect("new partition routes must become visible");
}
async fn append_count_query(instance: &Arc<Instance>, sql: &str) -> String {
let output = run_sql(instance, sql, QueryContext::arc()).await.unwrap();
let batches = match output.data {
common_query::OutputData::Stream(stream) => {
common_recordbatch::RecordBatches::try_collect(stream)
.await
.unwrap()
}
common_query::OutputData::RecordBatches(batches) => batches,
_ => panic!("expected query output: {sql}"),
};
batches.pretty_print().unwrap()
}
async fn assert_append_count(
instance: &Arc<Instance>,
table: &str,
rows: usize,
statistics_regions: usize,
) {
let expected = format!("+-----+\n| n |\n+-----+\n| {rows:<3} |\n+-----+");
let count = append_count_query(instance, &format!("SELECT count(*) AS n FROM {table}")).await;
assert_eq!(count, expected, "unbounded COUNT for {table}");
let scanned_count = append_count_query(
instance,
&format!("SELECT count(*) AS n FROM (SELECT * FROM {table} LIMIT 1000)"),
)
.await;
assert_eq!(scanned_count, expected, "scanned COUNT for {table}");
// Only regions whose source row counts are exact may use statistics.
let plan = append_count_query(
instance,
&format!("EXPLAIN ANALYZE SELECT count(*) FROM {table}"),
)
.await;
assert_eq!(
plan.matches("PlaceholderRowExec").count(),
statistics_regions,
"{plan}"
);
}
#[tokio::test(flavor = "multi_thread")]
async fn test_repartition_multi_hop_fulltext_index_gc_file() {
if StorageType::File.test_on() {
common_telemetry::init_default_ut_logging();
test_repartition_multi_hop_fulltext_index_gc().await;
}
}
async fn test_repartition_multi_hop_fulltext_index_gc() {
let store_type = StorageType::File;
let cluster_name = "test_repartition_multi_hop_fulltext_index_gc";
let table_name = "repartition_multi_hop_index_table";
let (store_config, _guard) = get_test_store_config(&store_type);
let datanodes = 3u64;
let home_dir = create_temp_dir("test_repartition_multi_hop_fulltext_index_gc_data_home");
let cluster = GreptimeDbClusterBuilder::new(cluster_name)
.await
.with_shared_home_dir(Arc::new(home_dir))
.with_datanodes(datanodes as u32)
.with_store_config(store_config)
.with_datanode_wal_config(DatanodeWalConfig::Noop)
.with_metasrv_gc_config(GcSchedulerOptions {
enable: true,
gc_cooldown_period: Duration::from_nanos(1),
..Default::default()
})
.with_datanode_gc_config(GcConfig {
enable: true,
lingering_time: Some(Duration::from_secs(0)),
unknown_file_lingering_time: Duration::from_secs(0),
..Default::default()
})
.build(true)
.await;
let metasrv = &cluster.metasrv;
let ticker = metasrv.gc_ticker().unwrap();
let query_ctx = QueryContext::arc();
let instance = cluster.fe_instance();
run_sql(
instance,
r#"
CREATE TABLE `repartition_multi_hop_index_table`(
`id` INT,
`msg` STRING FULLTEXT INDEX,
`ts` TIMESTAMP TIME INDEX,
PRIMARY KEY(`id`)
) PARTITION ON COLUMNS (`id`) (
`id` < 100,
`id` >= 100
) ENGINE = mito
WITH ('sst_format' = 'flat');
"#,
query_ctx.clone(),
)
.await
.unwrap();
run_sql(
instance,
r#"
INSERT INTO `repartition_multi_hop_index_table` VALUES
(10, 'quick fox in low partition', '2022-01-01 00:00:00'),
(60, 'quick fox in middle partition', '2022-01-01 00:00:01'),
(90, 'quick fox in high partition', '2022-01-01 00:00:02'),
(120, 'quick fox in outer partition', '2022-01-01 00:00:03');
"#,
query_ctx.clone(),
)
.await
.unwrap();
run_sql(
instance,
"ADMIN FLUSH_TABLE('repartition_multi_hop_index_table')",
query_ctx.clone(),
)
.await
.unwrap();
run_sql(
instance,
r#"
ALTER TABLE `repartition_multi_hop_index_table` SPLIT PARTITION (
`id` < 100
) INTO (
`id` < 50,
`id` >= 50 AND `id` < 100
);
"#,
query_ctx.clone(),
)
.await
.unwrap();
tokio::time::sleep(Duration::from_millis(500)).await;
run_sql(
instance,
r#"
ALTER TABLE `repartition_multi_hop_index_table` SPLIT PARTITION (
`id` >= 50 AND `id` < 100
) INTO (
`id` >= 50 AND `id` < 75,
`id` >= 75 AND `id` < 100
);
"#,
query_ctx.clone(),
)
.await
.unwrap();
tokio::time::sleep(Duration::from_millis(500)).await;
run_sql(
instance,
r#"
INSERT INTO `repartition_multi_hop_index_table` VALUES
(80, 'fresh fox in soon dropped source', '2022-01-01 00:00:04');
"#,
query_ctx.clone(),
)
.await
.unwrap();
run_sql(
instance,
"ADMIN FLUSH_TABLE('repartition_multi_hop_index_table')",
query_ctx.clone(),
)
.await
.unwrap();
let fox_query = "SELECT id, msg FROM `repartition_multi_hop_index_table` WHERE MATCHES(msg, 'fox') ORDER BY id";
let expected = "\
+-----+----------------------------------+
| id | msg |
+-----+----------------------------------+
| 10 | quick fox in low partition |
| 60 | quick fox in middle partition |
| 80 | fresh fox in soon dropped source |
| 90 | quick fox in high partition |
| 120 | quick fox in outer partition |
+-----+----------------------------------+";
let result = run_sql(instance, fox_query, query_ctx.clone())
.await
.unwrap();
check_output_stream(result.data, expected).await;
let manifest_entries = cluster_manifest_index_entries(&cluster).await;
assert!(
manifest_entries
.iter()
.any(|(_, origin_region_id, region_id)| origin_region_id != region_id),
"expected at least one cross-origin indexed SST before GC, got {manifest_entries:?}"
);
run_sql(
instance,
r#"
ALTER TABLE `repartition_multi_hop_index_table` MERGE PARTITION (
`id` >= 50 AND `id` < 75,
`id` >= 75 AND `id` < 100
);
"#,
query_ctx.clone(),
)
.await
.unwrap();
tokio::time::sleep(Duration::from_millis(500)).await;
trigger_full_gc(&ticker).await;
trigger_table_gc(metasrv, table_name).await;
let result = run_sql(instance, fox_query, query_ctx.clone())
.await
.unwrap();
check_output_stream(result.data, expected).await;
let manifest_index_paths = cluster_manifest_index_paths(&cluster).await;
assert!(
!manifest_index_paths.is_empty(),
"expected manifest index paths after GC"
);
let storage_paths = cluster.list_sst_files_from_all_datanodes().await;
assert!(
manifest_index_paths.is_subset(&storage_paths),
"manifest index paths should exist in storage after GC, missing: {:?}",
manifest_index_paths
.difference(&storage_paths)
.collect::<Vec<_>>()
);
let cross_origin_index_paths = cluster_manifest_index_entries(&cluster)
.await
.into_iter()
.filter_map(|(index_file_path, origin_region_id, region_id)| {
(origin_region_id != region_id).then_some(index_file_path)
})
.collect::<BTreeSet<_>>();
assert!(
!cross_origin_index_paths.is_empty(),
"expected cross-origin indexed SSTs after GC"
);
assert!(
cross_origin_index_paths.is_subset(&storage_paths),
"cross-origin manifest index paths should exist in storage after GC, missing: {:?}",
cross_origin_index_paths
.difference(&storage_paths)
.collect::<Vec<_>>()
);
}
pub async fn test_partition_unpartitioned_mito(store_type: StorageType) {
info!(
"test_partition_unpartitioned_mito: store_type: {:?}",
store_type
);
let cluster_name = "test_partition_unpartitioned_mito";
let (store_config, _guard) = get_test_store_config(&store_type);
let datanodes = 3u64;
let mut builder = GreptimeDbClusterBuilder::new(cluster_name).await;
if matches!(store_type, StorageType::File) {
let home_dir = create_temp_dir("test_partition_unpartitioned_mito_data_home");
builder = builder.with_shared_home_dir(Arc::new(home_dir));
}
let cluster = builder
.with_datanodes(datanodes as u32)
.with_store_config(store_config)
.with_datanode_wal_config(DatanodeWalConfig::Noop)
.with_metasrv_gc_config(GcSchedulerOptions {
enable: true,
gc_cooldown_period: Duration::from_nanos(1),
..Default::default()
})
.with_datanode_gc_config(GcConfig {
enable: true,
lingering_time: Some(Duration::from_secs(0)),
unknown_file_lingering_time: Duration::from_secs(0),
..Default::default()
})
.build(true)
.await;
let query_ctx = QueryContext::arc();
let instance = cluster.fe_instance();
let sql = r#"
CREATE TABLE `partition_unpartitioned_mito_table`(
`id` INT,
`city` STRING,
`ts` TIMESTAMP TIME INDEX,
PRIMARY KEY(`id`, `city`)
) ENGINE = mito;
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let sql = r#"
INSERT INTO `partition_unpartitioned_mito_table` VALUES
(1, 'New York', '2022-01-01 00:00:00'),
(10, 'Paris', '2022-01-01 00:00:00'),
(20, 'Beijing', '2022-01-01 00:00:00');
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let sql = r#"
ALTER TABLE `partition_unpartitioned_mito_table` PARTITION ON COLUMNS (`id`) (
`id` < 10,
`id` >= 10 AND `id` < 20,
`id` >= 20
);
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
// Wait for cache invalidation.
tokio::time::sleep(Duration::from_millis(500)).await;
let result = run_sql(
instance,
"SELECT * FROM `partition_unpartitioned_mito_table` ORDER BY `id`",
query_ctx.clone(),
)
.await
.unwrap();
let expected = "\
+----+----------+---------------------+
| id | city | ts |
+----+----------+---------------------+
| 1 | New York | 2022-01-01T00:00:00 |
| 10 | Paris | 2022-01-01T00:00:00 |
| 20 | Beijing | 2022-01-01T00:00:00 |
+----+----------+---------------------+";
check_output_stream(result.data, expected).await;
let result = run_sql(
instance,
"\
SELECT partition_expression, partition_description \
FROM information_schema.partitions \
WHERE table_name = 'partition_unpartitioned_mito_table' \
ORDER BY partition_ordinal_position;",
query_ctx.clone(),
)
.await
.unwrap();
let expected_partitions = r#"+----------------------+-----------------------+
| partition_expression | partition_description |
+----------------------+-----------------------+
| id | id < 10 |
| id | id >= 10 AND id < 20 |
| id | id >= 20 |
+----------------------+-----------------------+"#;
check_output_stream(result.data, expected_partitions).await;
let sql = r#"
INSERT INTO `partition_unpartitioned_mito_table` VALUES
(5, 'London', '2022-01-02 00:00:00'),
(15, 'Tokyo', '2022-01-02 00:00:00'),
(25, 'Shanghai', '2022-01-02 00:00:00');
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let result = run_sql(
instance,
"SELECT * FROM `partition_unpartitioned_mito_table` ORDER BY `id`",
query_ctx.clone(),
)
.await
.unwrap();
let expected = "\
+----+----------+---------------------+
| id | city | ts |
+----+----------+---------------------+
| 1 | New York | 2022-01-01T00:00:00 |
| 5 | London | 2022-01-02T00:00:00 |
| 10 | Paris | 2022-01-01T00:00:00 |
| 15 | Tokyo | 2022-01-02T00:00:00 |
| 20 | Beijing | 2022-01-01T00:00:00 |
| 25 | Shanghai | 2022-01-02T00:00:00 |
+----+----------+---------------------+";
check_output_stream(result.data, expected).await;
run_sql(
instance,
"DROP TABLE `partition_unpartitioned_mito_table`",
query_ctx.clone(),
)
.await
.unwrap();
}
pub async fn test_repartition_on_columns_metadata_mito(store_type: StorageType) {
info!(
"test_repartition_on_columns_metadata_mito: store_type: {:?}",
store_type
);
let cluster_name = "test_repartition_on_columns_metadata_mito";
let (store_config, _guard) = get_test_store_config(&store_type);
let datanodes = 3u64;
let mut builder = GreptimeDbClusterBuilder::new(cluster_name).await;
if matches!(store_type, StorageType::File) {
let home_dir = create_temp_dir("test_repartition_on_columns_metadata_mito_data_home");
builder = builder.with_shared_home_dir(Arc::new(home_dir));
}
let cluster = builder
.with_datanodes(datanodes as u32)
.with_store_config(store_config)
.with_datanode_wal_config(DatanodeWalConfig::Noop)
.with_metasrv_gc_config(GcSchedulerOptions {
enable: true,
gc_cooldown_period: Duration::from_nanos(1),
..Default::default()
})
.with_datanode_gc_config(GcConfig {
enable: true,
lingering_time: Some(Duration::from_secs(0)),
unknown_file_lingering_time: Duration::from_secs(0),
..Default::default()
})
.build(true)
.await;
let query_ctx = QueryContext::arc();
let instance = cluster.fe_instance();
assert_on_columns_metadata_overwrite(
instance,
"repartition_on_columns_metadata_table",
&repartition_on_columns_sql("repartition_on_columns_metadata_table"),
query_ctx.clone(),
)
.await;
assert_on_columns_metadata_overwrite(
instance,
"split_on_columns_metadata_table",
&split_on_columns_sql("split_on_columns_metadata_table"),
query_ctx.clone(),
)
.await;
assert_on_columns_rejects_removed_remaining_column(
instance,
"repartition_on_columns_error_table",
&repartition_on_columns_removed_column_sql("repartition_on_columns_error_table"),
query_ctx.clone(),
)
.await;
assert_on_columns_rejects_removed_remaining_column(
instance,
"split_on_columns_error_table",
&split_on_columns_removed_column_sql("split_on_columns_error_table"),
query_ctx.clone(),
)
.await;
run_sql(
instance,
"DROP TABLE `repartition_on_columns_metadata_table`",
query_ctx.clone(),
)
.await
.unwrap();
run_sql(
instance,
"DROP TABLE `split_on_columns_metadata_table`",
query_ctx.clone(),
)
.await
.unwrap();
run_sql(
instance,
"DROP TABLE `repartition_on_columns_error_table`",
query_ctx.clone(),
)
.await
.unwrap();
run_sql(
instance,
"DROP TABLE `split_on_columns_error_table`",
query_ctx.clone(),
)
.await
.unwrap();
}
pub async fn test_repartition_on_columns_data_correctness_mito(store_type: StorageType) {
info!(
"test_repartition_on_columns_data_correctness_mito: store_type: {:?}",
store_type
);
let cluster_name = "test_repartition_on_columns_data_correctness_mito";
let (store_config, _guard) = get_test_store_config(&store_type);
let datanodes = 3u64;
let mut builder = GreptimeDbClusterBuilder::new(cluster_name).await;
if matches!(store_type, StorageType::File) {
let home_dir =
create_temp_dir("test_repartition_on_columns_data_correctness_mito_data_home");
builder = builder.with_shared_home_dir(Arc::new(home_dir));
}
let cluster = builder
.with_datanodes(datanodes as u32)
.with_store_config(store_config)
.with_datanode_wal_config(DatanodeWalConfig::Noop)
.with_metasrv_gc_config(GcSchedulerOptions {
enable: true,
gc_cooldown_period: Duration::from_nanos(1),
..Default::default()
})
.with_datanode_gc_config(GcConfig {
enable: true,
lingering_time: Some(Duration::from_secs(0)),
unknown_file_lingering_time: Duration::from_secs(0),
..Default::default()
})
.build(true)
.await;
let query_ctx = QueryContext::arc();
let instance = cluster.fe_instance();
assert_on_columns_data_correctness(
instance,
"repartition_on_columns_data_table",
&repartition_on_columns_sql("repartition_on_columns_data_table"),
query_ctx.clone(),
)
.await;
assert_on_columns_data_correctness(
instance,
"split_on_columns_data_table",
&split_on_columns_sql("split_on_columns_data_table"),
query_ctx.clone(),
)
.await;
run_sql(
instance,
"DROP TABLE `repartition_on_columns_data_table`",
query_ctx.clone(),
)
.await
.unwrap();
run_sql(
instance,
"DROP TABLE `split_on_columns_data_table`",
query_ctx.clone(),
)
.await
.unwrap();
}
pub async fn test_partition_unpartitioned_metric(store_type: StorageType) {
info!(
"test_partition_unpartitioned_metric: store_type: {:?}",
store_type
);
let cluster_name = "test_partition_unpartitioned_metric";
let (store_config, _guard) = get_test_store_config(&store_type);
let datanodes = 3u64;
let mut builder = GreptimeDbClusterBuilder::new(cluster_name).await;
if matches!(store_type, StorageType::File) {
let home_dir = create_temp_dir("test_partition_unpartitioned_metric_data_home");
builder = builder.with_shared_home_dir(Arc::new(home_dir));
}
let cluster = builder
.with_datanodes(datanodes as u32)
.with_store_config(store_config)
.with_datanode_wal_config(DatanodeWalConfig::Noop)
.with_metasrv_gc_config(GcSchedulerOptions {
enable: true,
gc_cooldown_period: Duration::from_nanos(1),
..Default::default()
})
.with_datanode_gc_config(GcConfig {
enable: true,
lingering_time: Some(Duration::from_secs(0)),
unknown_file_lingering_time: Duration::from_secs(0),
..Default::default()
})
.build(true)
.await;
let query_ctx = QueryContext::arc();
let instance = cluster.fe_instance();
let sql = r#"
CREATE TABLE `partition_unpartitioned_metric_phy`(
`ts` TIMESTAMP TIME INDEX,
`val` DOUBLE,
`host` STRING PRIMARY KEY
) ENGINE = metric
WITH (
"physical_metric_table" = "true"
);
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let sql = r#"
CREATE TABLE `partition_unpartitioned_metric_log`(
`ts` TIMESTAMP TIME INDEX,
`val` DOUBLE,
`host` STRING PRIMARY KEY
) ENGINE = metric
WITH (
"on_physical_table" = "partition_unpartitioned_metric_phy"
);
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let sql = r#"
INSERT INTO `partition_unpartitioned_metric_log` (`host`, `ts`, `val`) VALUES
('a_host', '2022-01-01 00:00:00', 1),
('z_host', '2022-01-01 00:00:00', 2);
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let sql = r#"
ALTER TABLE `partition_unpartitioned_metric_phy` PARTITION ON COLUMNS (`host`) (
`host` < 'm',
`host` >= 'm'
);
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
// Wait for cache invalidation.
tokio::time::sleep(Duration::from_millis(500)).await;
let result = run_sql(
instance,
"SELECT * FROM `partition_unpartitioned_metric_log` ORDER BY `host`",
query_ctx.clone(),
)
.await
.unwrap();
let expected = "\
+--------+---------------------+-----+
| host | ts | val |
+--------+---------------------+-----+
| a_host | 2022-01-01T00:00:00 | 1.0 |
| z_host | 2022-01-01T00:00:00 | 2.0 |
+--------+---------------------+-----+";
check_output_stream(result.data, expected).await;
let result = run_sql(
instance,
"\
SELECT partition_expression, partition_description \
FROM information_schema.partitions \
WHERE table_name = 'partition_unpartitioned_metric_phy' \
ORDER BY partition_ordinal_position;",
query_ctx.clone(),
)
.await
.unwrap();
let expected_partitions = r#"+----------------------+-----------------------+
| partition_expression | partition_description |
+----------------------+-----------------------+
| host | host < m |
| host | host >= m |
+----------------------+-----------------------+"#;
check_output_stream(result.data, expected_partitions).await;
let sql = r#"
INSERT INTO `partition_unpartitioned_metric_log` (`host`, `ts`, `val`) VALUES
('b_host', '2022-01-02 00:00:00', 3),
('x_host', '2022-01-02 00:00:00', 4);
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let result = run_sql(
instance,
"SELECT * FROM `partition_unpartitioned_metric_log` ORDER BY `host`",
query_ctx.clone(),
)
.await
.unwrap();
let expected = "\
+--------+---------------------+-----+
| host | ts | val |
+--------+---------------------+-----+
| a_host | 2022-01-01T00:00:00 | 1.0 |
| b_host | 2022-01-02T00:00:00 | 3.0 |
| x_host | 2022-01-02T00:00:00 | 4.0 |
| z_host | 2022-01-01T00:00:00 | 2.0 |
+--------+---------------------+-----+";
check_output_stream(result.data, expected).await;
run_sql(
instance,
"DROP TABLE `partition_unpartitioned_metric_log`",
query_ctx.clone(),
)
.await
.unwrap();
run_sql(
instance,
"DROP TABLE `partition_unpartitioned_metric_phy`",
query_ctx.clone(),
)
.await
.unwrap();
}
async fn get_table_id(metasrv: &Arc<Metasrv>, table_name: &str) -> TableId {
metasrv
.table_metadata_manager()
.table_name_manager()
.get(TableNameKey::new(
DEFAULT_CATALOG_NAME,
DEFAULT_SCHEMA_NAME,
table_name,
))
.await
.unwrap()
.unwrap()
.table_id()
}
async fn trigger_table_gc(metasrv: &Arc<Metasrv>, table_name: &str) {
info!("triggering table gc for table: {}", table_name);
let table_metadata_manager = metasrv.table_metadata_manager();
let table_id = get_table_id(metasrv, table_name).await;
let (_, table_route_value) = table_metadata_manager
.table_route_manager()
.get_physical_table_route(table_id)
.await
.unwrap();
let region_ids = table_route_value
.region_routes
.iter()
.map(|r| r.region.id)
.collect::<Vec<_>>();
let procedure = BatchGcProcedure::new(
metasrv.mailbox().clone(),
metasrv.table_metadata_manager().clone(),
metasrv.runtime_switch_manager().clone(),
metasrv.options().grpc.server_addr.clone(),
region_ids.clone(),
false, // full_file_listing
Duration::from_secs(10), // timeout
Default::default(),
);
// Submit the procedure to the procedure manager
let procedure_with_id = ProcedureWithId::with_random_id(Box::new(procedure));
let mut watcher = metasrv
.procedure_manager()
.submit(procedure_with_id)
.await
.unwrap();
watcher::wait(&mut watcher).await.unwrap();
}
async fn assert_table_sst_files_match_manifests(cluster: &GreptimeDbCluster, table_name: &str) {
// Other tables may retain compacted SSTs until their own GC runs.
let table_id = get_table_id(&cluster.metasrv, table_name).await;
let table_dir = format!("/{table_id}/");
let retain_table_files = |files: BTreeSet<String>| {
files
.into_iter()
.filter(|path| path.contains(&table_dir))
.collect::<BTreeSet<_>>()
};
let storage_files = retain_table_files(cluster.list_sst_files_from_all_datanodes().await);
let manifest_files = retain_table_files(cluster.list_sst_files_from_manifests().await);
assert_eq!(storage_files, manifest_files);
}
async fn trigger_full_gc(ticker: &GcTickerRef) {
info!("triggering full gc");
let (tx, rx) = oneshot::channel();
ticker
.sender
.send(gc::Event::Manually {
sender: tx,
region_ids: None,
full_file_listing: None,
timeout: None,
procedure_context: ProcedureContext::from_event_context(PersistentEventContext::new(
TriggerReason::Manual,
)),
})
.await
.unwrap();
let _ = rx.await.unwrap().unwrap();
}
async fn cluster_manifest_index_entries(
cluster: &GreptimeDbCluster,
) -> Vec<(String, RegionId, RegionId)> {
let mut entries = Vec::new();
for datanode in cluster.datanode_instances.values() {
let region_server = datanode.region_server();
let mito = region_server.mito_engine().unwrap();
entries.extend(
mito.all_ssts_from_manifest()
.await
.into_iter()
.filter_map(|entry| {
entry.index_file_path.map(|index_file_path| {
(index_file_path, entry.origin_region_id, entry.region_id)
})
}),
);
}
entries
}
async fn cluster_manifest_index_paths(cluster: &GreptimeDbCluster) -> BTreeSet<String> {
cluster_manifest_index_entries(cluster)
.await
.into_iter()
.map(|(index_file_path, _, _)| index_file_path)
.collect()
}
fn query_partitions_sql(table_name: &str) -> String {
// We query information_schema.partitions to assert repartition results across engines,
// rather than relying on SHOW CREATE TABLE formatting differences.
format!(
"\
SELECT table_catalog, table_schema, table_name, partition_name, partition_expression, \
partition_description, greptime_partition_id, partition_ordinal_position \
FROM information_schema.partitions \
WHERE table_name = '{}' \
ORDER BY partition_ordinal_position;",
table_name
)
}
pub async fn test_repartition_mito(store_type: StorageType, flat_format: bool) {
info!(
"test_repartition_mito: store_type: {:?}, flat_format: {:?}",
store_type, flat_format
);
let cluster_name = "test_repartition_mito";
let (store_config, _guard) = get_test_store_config(&store_type);
let datanodes = 3u64;
let mut builder = GreptimeDbClusterBuilder::new(cluster_name).await;
if matches!(store_type, StorageType::File) {
let home_dir = create_temp_dir("test_repartition_mito_data_home");
builder = builder.with_shared_home_dir(Arc::new(home_dir));
}
let cluster = builder
.with_datanodes(datanodes as u32)
.with_store_config(store_config)
.with_datanode_wal_config(DatanodeWalConfig::Noop)
.with_metasrv_gc_config(GcSchedulerOptions {
enable: true,
gc_cooldown_period: Duration::from_nanos(1),
..Default::default()
})
.with_datanode_gc_config(GcConfig {
enable: true,
lingering_time: Some(Duration::from_secs(0)),
unknown_file_lingering_time: Duration::from_secs(0),
..Default::default()
})
.build(true)
.await;
let metasrv = &cluster.metasrv;
let ticker = metasrv.gc_ticker().unwrap();
let query_ctx = QueryContext::arc();
let instance = cluster.fe_instance();
// 1. Setup: Create a table with partitions (format varies by test case)
let sql = if flat_format {
r#"
CREATE TABLE `repartition_mito_table`(
`id` INT,
`city` STRING,
`ts` TIMESTAMP TIME INDEX,
PRIMARY KEY(`id`, `city`)
) PARTITION ON COLUMNS (`id`) (
`id` < 10,
`id` >= 10 AND `id` < 20,
`id` >= 20
) ENGINE = mito
WITH (
'sst_format' = 'flat'
);
"#
} else {
r#"
CREATE TABLE `repartition_mito_table`(
`id` INT,
`city` STRING,
`ts` TIMESTAMP TIME INDEX,
PRIMARY KEY(`id`, `city`)
) PARTITION ON COLUMNS (`id`) (
`id` < 10,
`id` >= 10 AND `id` < 20,
`id` >= 20
) ENGINE = mito;
"#
};
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let sql = r#"
INSERT INTO `repartition_mito_table` VALUES
(1, 'New York', '2022-01-01 00:00:00'),
(5, 'London', '2022-01-01 00:00:00'),
(10, 'Paris', '2022-01-01 00:00:00'),
(15, 'Tokyo', '2022-01-01 00:00:00'),
(20, 'Beijing', '2022-01-01 00:00:00'),
(25, 'Shanghai', '2022-01-01 00:00:00');
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let result = run_sql(
instance,
"SELECT * FROM `repartition_mito_table` ORDER BY `id`",
query_ctx.clone(),
)
.await
.unwrap();
let expected = "\
+----+----------+---------------------+
| id | city | ts |
+----+----------+---------------------+
| 1 | New York | 2022-01-01T00:00:00 |
| 5 | London | 2022-01-01T00:00:00 |
| 10 | Paris | 2022-01-01T00:00:00 |
| 15 | Tokyo | 2022-01-01T00:00:00 |
| 20 | Beijing | 2022-01-01T00:00:00 |
| 25 | Shanghai | 2022-01-01T00:00:00 |
+----+----------+---------------------+";
check_output_stream(result.data, expected).await;
// 2. Split Partition
let sql = r#"
ALTER TABLE `repartition_mito_table` SPLIT PARTITION (
`id` < 10
) INTO (
`id` < 5,
`id` >= 5 AND `id` < 10
);
"#;
let _result = run_sql(instance, sql, query_ctx.clone()).await.unwrap();
// Wait for cache invalidation
tokio::time::sleep(Duration::from_millis(500)).await;
let result = run_sql(
instance,
"SELECT * FROM `repartition_mito_table` ORDER BY `id`",
query_ctx.clone(),
)
.await
.unwrap();
let expected = "\
+----+----------+---------------------+
| id | city | ts |
+----+----------+---------------------+
| 1 | New York | 2022-01-01T00:00:00 |
| 5 | London | 2022-01-01T00:00:00 |
| 10 | Paris | 2022-01-01T00:00:00 |
| 15 | Tokyo | 2022-01-01T00:00:00 |
| 20 | Beijing | 2022-01-01T00:00:00 |
| 25 | Shanghai | 2022-01-01T00:00:00 |
+----+----------+---------------------+";
check_output_stream(result.data, expected).await;
trigger_table_gc(metasrv, "repartition_mito_table").await;
// Should be ok before compact.
let result = run_sql(
instance,
"SELECT * FROM `repartition_mito_table` ORDER BY `id`",
query_ctx.clone(),
)
.await
.unwrap();
check_output_stream(result.data, expected).await;
assert_table_sst_files_match_manifests(&cluster, "repartition_mito_table").await;
// It should be ok, if we try to compact the table after split partition.
let compact_sql = "ADMIN COMPACT_TABLE('repartition_mito_table', 'swcs', '3600')";
let _result = run_sql(instance, compact_sql, query_ctx.clone())
.await
.unwrap();
// Should be no change after compact.
let result = run_sql(
instance,
"SELECT * FROM `repartition_mito_table` ORDER BY `id`",
query_ctx.clone(),
)
.await
.unwrap();
check_output_stream(result.data, expected).await;
// Trigger GC to clean up the compacted files.
trigger_table_gc(metasrv, "repartition_mito_table").await;
// Should be no change after GC.
let result = run_sql(
instance,
"SELECT * FROM `repartition_mito_table` ORDER BY `id`",
query_ctx.clone(),
)
.await
.unwrap();
check_output_stream(result.data, expected).await;
assert_table_sst_files_match_manifests(&cluster, "repartition_mito_table").await;
let result = run_sql(
instance,
&query_partitions_sql("repartition_mito_table"),
query_ctx.clone(),
)
.await
.unwrap();
let expected_create_table_after_split = r#"+---------------+--------------+------------------------+----------------+----------------------+-----------------------+-----------------------+----------------------------+
| table_catalog | table_schema | table_name | partition_name | partition_expression | partition_description | greptime_partition_id | partition_ordinal_position |
+---------------+--------------+------------------------+----------------+----------------------+-----------------------+-----------------------+----------------------------+
| greptime | public | repartition_mito_table | p0 | id | id < 5 | 4398046511104 | 1 |
| greptime | public | repartition_mito_table | p1 | id | id >= 10 AND id < 20 | 4398046511105 | 2 |
| greptime | public | repartition_mito_table | p2 | id | id >= 20 | 4398046511106 | 3 |
| greptime | public | repartition_mito_table | p3 | id | id >= 5 AND id < 10 | 4398046511107 | 4 |
+---------------+--------------+------------------------+----------------+----------------------+-----------------------+-----------------------+----------------------------+"#;
check_output_stream(result.data, expected_create_table_after_split).await;
let sql =
r#"INSERT INTO `repartition_mito_table` VALUES (2, 'Split1', '2022-01-02 00:00:00');"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let sql =
r#"INSERT INTO `repartition_mito_table` VALUES (7, 'Split2', '2022-01-02 00:00:00');"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let result = run_sql(
instance,
"SELECT * FROM `repartition_mito_table` WHERE `id` IN (2, 7) ORDER BY `id`",
query_ctx.clone(),
)
.await
.unwrap();
let expected_split_inserts = "\
+----+--------+---------------------+
| id | city | ts |
+----+--------+---------------------+
| 2 | Split1 | 2022-01-02T00:00:00 |
| 7 | Split2 | 2022-01-02T00:00:00 |
+----+--------+---------------------+";
check_output_stream(result.data, expected_split_inserts).await;
// 3. Merge Partition
let sql = r#"
ALTER TABLE `repartition_mito_table` MERGE PARTITION (
`id` >= 10 AND `id` < 20,
`id` >= 20
);
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
// Wait for cache invalidation
tokio::time::sleep(Duration::from_millis(500)).await;
let result = run_sql(
instance,
"SELECT * FROM `repartition_mito_table` ORDER BY `id`",
query_ctx.clone(),
)
.await
.unwrap();
let expected_all = "\
+----+----------+---------------------+
| id | city | ts |
+----+----------+---------------------+
| 1 | New York | 2022-01-01T00:00:00 |
| 2 | Split1 | 2022-01-02T00:00:00 |
| 5 | London | 2022-01-01T00:00:00 |
| 7 | Split2 | 2022-01-02T00:00:00 |
| 10 | Paris | 2022-01-01T00:00:00 |
| 15 | Tokyo | 2022-01-01T00:00:00 |
| 20 | Beijing | 2022-01-01T00:00:00 |
| 25 | Shanghai | 2022-01-01T00:00:00 |
+----+----------+---------------------+";
check_output_stream(result.data, expected_all).await;
trigger_table_gc(metasrv, "repartition_mito_table").await;
// Trigger GC to clean up the compacted files.
trigger_full_gc(&ticker).await;
let result = run_sql(
instance,
"SELECT * FROM `repartition_mito_table` ORDER BY `id`",
query_ctx.clone(),
)
.await
.unwrap();
check_output_stream(result.data, expected_all).await;
assert_table_sst_files_match_manifests(&cluster, "repartition_mito_table").await;
// It should be ok, if we try to compact the table after merge partition.
let compact_sql = "ADMIN COMPACT_TABLE('repartition_mito_table', 'swcs', '3600')";
let _result = run_sql(instance, compact_sql, query_ctx.clone())
.await
.unwrap();
// Should be no change after compact.
let result = run_sql(
instance,
"SELECT * FROM `repartition_mito_table` ORDER BY `id`",
query_ctx.clone(),
)
.await
.unwrap();
check_output_stream(result.data, expected_all).await;
trigger_table_gc(metasrv, "repartition_mito_table").await;
trigger_full_gc(&ticker).await;
// Should be no change after GC.
let result = run_sql(
instance,
"SELECT * FROM `repartition_mito_table` ORDER BY `id`",
query_ctx.clone(),
)
.await
.unwrap();
check_output_stream(result.data, expected_all).await;
assert_table_sst_files_match_manifests(&cluster, "repartition_mito_table").await;
let result = run_sql(
instance,
&query_partitions_sql("repartition_mito_table"),
query_ctx.clone(),
)
.await
.unwrap();
let expected_create_table_after_merge = r#"+---------------+--------------+------------------------+----------------+----------------------+-----------------------+-----------------------+----------------------------+
| table_catalog | table_schema | table_name | partition_name | partition_expression | partition_description | greptime_partition_id | partition_ordinal_position |
+---------------+--------------+------------------------+----------------+----------------------+-----------------------+-----------------------+----------------------------+
| greptime | public | repartition_mito_table | p0 | id | id < 5 | 4398046511104 | 1 |
| greptime | public | repartition_mito_table | p1 | id | id >= 10 | 4398046511105 | 2 |
| greptime | public | repartition_mito_table | p2 | id | id >= 5 AND id < 10 | 4398046511107 | 3 |
+---------------+--------------+------------------------+----------------+----------------------+-----------------------+-----------------------+----------------------------+"#;
check_output_stream(result.data, expected_create_table_after_merge).await;
let sql =
r#"INSERT INTO `repartition_mito_table` VALUES (12, 'Merge1', '2022-01-03 00:00:00');"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let sql =
r#"INSERT INTO `repartition_mito_table` VALUES (30, 'Merge2', '2022-01-03 00:00:00');"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let result = run_sql(
instance,
"SELECT * FROM `repartition_mito_table` WHERE `id` IN (12, 30) ORDER BY `id`",
query_ctx.clone(),
)
.await
.unwrap();
let expected_merge_inserts = "\
+----+--------+---------------------+
| id | city | ts |
+----+--------+---------------------+
| 12 | Merge1 | 2022-01-03T00:00:00 |
| 30 | Merge2 | 2022-01-03T00:00:00 |
+----+--------+---------------------+";
check_output_stream(result.data, expected_merge_inserts).await;
run_sql(
instance,
"DROP TABLE `repartition_mito_table`",
query_ctx.clone(),
)
.await
.unwrap();
}
pub async fn test_repartition_metric(
store_type: StorageType,
flat_format: bool,
primary_key_encoding: PrimaryKeyEncoding,
) {
info!(
"test_repartition_metric: store_type: {:?}, flat_format: {:?}, primary_key_encoding: {:?}",
store_type, flat_format, primary_key_encoding
);
let cluster_name = "test_repartition_metric";
let (store_config, _guard) = get_test_store_config(&store_type);
let datanodes = 3u64;
let mut builder = GreptimeDbClusterBuilder::new(cluster_name).await;
if matches!(store_type, StorageType::File) {
let home_dir = create_temp_dir("test_repartition_metric_data_home");
builder = builder.with_shared_home_dir(Arc::new(home_dir));
}
let cluster = builder
.with_datanodes(datanodes as u32)
.with_store_config(store_config)
.with_datanode_wal_config(DatanodeWalConfig::Noop)
.with_metasrv_gc_config(GcSchedulerOptions {
enable: true,
gc_cooldown_period: Duration::from_nanos(1),
..Default::default()
})
.with_datanode_gc_config(GcConfig {
enable: true,
lingering_time: Some(Duration::from_secs(0)),
unknown_file_lingering_time: Duration::from_secs(0),
..Default::default()
})
.build(true)
.await;
let metasrv = &cluster.metasrv;
let ticker = metasrv.gc_ticker().unwrap();
let query_ctx = QueryContext::arc();
let instance = cluster.fe_instance();
// Explicitly configure sst_format and primary key encoding to cover the matrix.
let sst_format = if flat_format { "flat" } else { "primary_key" };
let primary_key_encoding = match primary_key_encoding {
PrimaryKeyEncoding::Dense => "dense",
PrimaryKeyEncoding::Sparse => "sparse",
};
let sql = format!(
r#"
CREATE TABLE `repart_phy_metric`(
`ts` TIMESTAMP TIME INDEX,
`val` DOUBLE,
`host` STRING PRIMARY KEY
) PARTITION ON COLUMNS (`host`) (
`host` < 'm',
`host` >= 'm'
) ENGINE = metric
WITH (
"physical_metric_table" = "",
'sst_format' = '{sst_format}',
"primary_key_encoding" = "{primary_key_encoding}",
"index.type" = "inverted",
);
"#
);
run_sql(instance, &sql, query_ctx.clone()).await.unwrap();
// A second logical table exercises repartition behavior across multiple logical tables
// sharing the same physical metric table.
let sql = r#"
CREATE TABLE `repart_log_metric`(
`ts` TIMESTAMP TIME INDEX,
`val` DOUBLE,
`host` STRING PRIMARY KEY
) ENGINE = metric WITH ("on_physical_table" = "repart_phy_metric");
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let sql = r#"
CREATE TABLE `repart_log_metric_job`(
`ts` TIMESTAMP TIME INDEX,
`val` DOUBLE,
`job` STRING PRIMARY KEY
) ENGINE = metric WITH ("on_physical_table" = "repart_phy_metric");
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let sql = r#"
INSERT INTO `repart_log_metric` (`host`, `ts`, `val`) VALUES
('a_host', '2022-01-01 00:00:00', 1),
('z_host', '2022-01-01 00:00:00', 2);
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let result = run_sql(
instance,
"SELECT * FROM `repart_log_metric` ORDER BY `host`",
query_ctx.clone(),
)
.await
.unwrap();
let expected = "\
+--------+---------------------+-----+
| host | ts | val |
+--------+---------------------+-----+
| a_host | 2022-01-01T00:00:00 | 1.0 |
| z_host | 2022-01-01T00:00:00 | 2.0 |
+--------+---------------------+-----+";
check_output_stream(result.data, expected).await;
// Split physical table partition
let sql = r#"
ALTER TABLE `repart_phy_metric` SPLIT PARTITION (
`host` < 'm'
) INTO (
`host` < 'g',
`host` >= 'g' AND `host` < 'm'
);
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
// Wait for cache invalidation
tokio::time::sleep(Duration::from_millis(500)).await;
let result = run_sql(
instance,
&query_partitions_sql("repart_phy_metric"),
query_ctx.clone(),
)
.await
.unwrap();
// Partition ids and order are expected to be stable within a single test run.
let expected_create_table_after_split = r#"+---------------+--------------+-------------------+----------------+----------------------+------------------------+-----------------------+----------------------------+
| table_catalog | table_schema | table_name | partition_name | partition_expression | partition_description | greptime_partition_id | partition_ordinal_position |
+---------------+--------------+-------------------+----------------+----------------------+------------------------+-----------------------+----------------------------+
| greptime | public | repart_phy_metric | p0 | host | host < g | 4398046511104 | 1 |
| greptime | public | repart_phy_metric | p1 | host | host >= m | 4398046511105 | 2 |
| greptime | public | repart_phy_metric | p2 | host | host >= g AND host < m | 4398046511106 | 3 |
+---------------+--------------+-------------------+----------------+----------------------+------------------------+-----------------------+----------------------------+"#;
check_output_stream(result.data, expected_create_table_after_split).await;
let regions = cluster.list_all_regions().await;
let region0 = regions.get(&RegionId::new(1024, 0)).unwrap();
let region2 = regions.get(&RegionId::new(1024, 2)).unwrap();
let primary_keys_in_region_0 = region0
.metadata()
.primary_key_columns()
.cloned()
.collect::<Vec<_>>();
let primary_keys_in_region_2 = region2
.metadata()
.primary_key_columns()
.cloned()
.collect::<Vec<_>>();
info!("primary_keys_in_region_0: {:?}", primary_keys_in_region_0);
info!("primary_keys_in_region_2: {:?}", primary_keys_in_region_2);
assert_eq!(primary_keys_in_region_0, primary_keys_in_region_2);
let sql = r#"
ALTER TABLE `repart_log_metric_job` ADD COLUMN `cpu` STRING PRIMARY KEY;
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let result = run_sql(
instance,
"SELECT * FROM `repart_log_metric` ORDER BY `host`",
query_ctx.clone(),
)
.await
.unwrap();
let expected = "\
+--------+---------------------+-----+
| host | ts | val |
+--------+---------------------+-----+
| a_host | 2022-01-01T00:00:00 | 1.0 |
| z_host | 2022-01-01T00:00:00 | 2.0 |
+--------+---------------------+-----+";
check_output_stream(result.data, expected).await;
trigger_table_gc(metasrv, "repart_phy_metric").await;
// Should be ok before compact.
let result = run_sql(
instance,
"SELECT * FROM `repart_log_metric` ORDER BY `host`",
query_ctx.clone(),
)
.await
.unwrap();
check_output_stream(result.data, expected).await;
assert_table_sst_files_match_manifests(&cluster, "repart_phy_metric").await;
// It should be ok, if we try to compact the table after split partition.
let compact_sql = "ADMIN COMPACT_TABLE('repart_phy_metric', 'swcs', '3600')";
let _result = run_sql(instance, compact_sql, query_ctx.clone())
.await
.unwrap();
// Should be no change after compact.
let result = run_sql(
instance,
"SELECT * FROM `repart_log_metric` ORDER BY `host`",
query_ctx.clone(),
)
.await
.unwrap();
check_output_stream(result.data, expected).await;
// Trigger GC to clean up the compacted files.
trigger_table_gc(metasrv, "repart_phy_metric").await;
// Should be no change after GC.
let result = run_sql(
instance,
"SELECT * FROM `repart_log_metric` ORDER BY `host`",
query_ctx.clone(),
)
.await
.unwrap();
check_output_stream(result.data, expected).await;
assert_table_sst_files_match_manifests(&cluster, "repart_phy_metric").await;
let sql = r#"INSERT INTO `repart_log_metric` (`host`, `ts`, `val`) VALUES ('b_host', '2022-01-02 00:00:00', 3.0);"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let sql = r#"INSERT INTO `repart_log_metric` (`host`, `ts`, `val`) VALUES ('h_host', '2022-01-02 00:00:00', 4.0);"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let result = run_sql(
instance,
"SELECT * FROM `repart_log_metric` WHERE `host` IN ('b_host', 'h_host') ORDER BY `host`",
query_ctx.clone(),
)
.await
.unwrap();
let expected = "\
+--------+---------------------+-----+
| host | ts | val |
+--------+---------------------+-----+
| b_host | 2022-01-02T00:00:00 | 3.0 |
| h_host | 2022-01-02T00:00:00 | 4.0 |
+--------+---------------------+-----+";
check_output_stream(result.data, expected).await;
let sql = r#"
ALTER TABLE `repart_phy_metric` MERGE PARTITION (
`host` < 'g',
`host` >= 'g' AND `host` < 'm'
);
"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
// Wait for cache invalidation
tokio::time::sleep(Duration::from_millis(500)).await;
let result = run_sql(
instance,
&query_partitions_sql("repart_phy_metric"),
query_ctx.clone(),
)
.await
.unwrap();
let expected_create_table_after_merge = r#"+---------------+--------------+-------------------+----------------+----------------------+-----------------------+-----------------------+----------------------------+
| table_catalog | table_schema | table_name | partition_name | partition_expression | partition_description | greptime_partition_id | partition_ordinal_position |
+---------------+--------------+-------------------+----------------+----------------------+-----------------------+-----------------------+----------------------------+
| greptime | public | repart_phy_metric | p0 | host | host < m | 4398046511104 | 1 |
| greptime | public | repart_phy_metric | p1 | host | host >= m | 4398046511105 | 2 |
+---------------+--------------+-------------------+----------------+----------------------+-----------------------+-----------------------+----------------------------+"#;
check_output_stream(result.data, expected_create_table_after_merge).await;
let result = run_sql(
instance,
"SELECT * FROM `repart_log_metric` ORDER BY `host`",
query_ctx.clone(),
)
.await
.unwrap();
let expected = "\
+--------+---------------------+-----+
| host | ts | val |
+--------+---------------------+-----+
| a_host | 2022-01-01T00:00:00 | 1.0 |
| b_host | 2022-01-02T00:00:00 | 3.0 |
| h_host | 2022-01-02T00:00:00 | 4.0 |
| z_host | 2022-01-01T00:00:00 | 2.0 |
+--------+---------------------+-----+";
check_output_stream(result.data, expected).await;
trigger_table_gc(metasrv, "repart_phy_metric").await;
trigger_full_gc(&ticker).await;
// Should be no change after GC.
let result = run_sql(
instance,
"SELECT * FROM `repart_log_metric` ORDER BY `host`",
query_ctx.clone(),
)
.await
.unwrap();
check_output_stream(result.data, expected).await;
assert_table_sst_files_match_manifests(&cluster, "repart_phy_metric").await;
// It should be ok, if we try to compact the table after merge partition.
let compact_sql = "ADMIN COMPACT_TABLE('repart_phy_metric', 'swcs', '3600')";
let _result = run_sql(instance, compact_sql, query_ctx.clone())
.await
.unwrap();
// Should be no change after compact.
let result = run_sql(
instance,
"SELECT * FROM `repart_log_metric` ORDER BY `host`",
query_ctx.clone(),
)
.await
.unwrap();
check_output_stream(result.data, expected).await;
// Trigger GC to clean up the compacted files.
trigger_table_gc(metasrv, "repart_phy_metric").await;
trigger_full_gc(&ticker).await;
// Should be no change after GC.
let result = run_sql(
instance,
"SELECT * FROM `repart_log_metric` ORDER BY `host`",
query_ctx.clone(),
)
.await
.unwrap();
check_output_stream(result.data, expected).await;
assert_table_sst_files_match_manifests(&cluster, "repart_phy_metric").await;
let sql = r#"INSERT INTO `repart_log_metric` (`host`, `ts`, `val`) VALUES ('c_host', '2022-01-03 00:00:00', 5.0);"#;
run_sql(instance, sql, query_ctx.clone()).await.unwrap();
let result = run_sql(
instance,
"SELECT * FROM `repart_log_metric` WHERE `host` = 'c_host'",
query_ctx.clone(),
)
.await
.unwrap();
let expected = "\
+--------+---------------------+-----+
| host | ts | val |
+--------+---------------------+-----+
| c_host | 2022-01-03T00:00:00 | 5.0 |
+--------+---------------------+-----+";
check_output_stream(result.data, expected).await;
run_sql(
instance,
"DROP TABLE `repart_log_metric_job`",
query_ctx.clone(),
)
.await
.unwrap();
run_sql(
instance,
"DROP TABLE `repart_log_metric`",
query_ctx.clone(),
)
.await
.unwrap();
run_sql(
instance,
"DROP TABLE `repart_phy_metric`",
query_ctx.clone(),
)
.await
.unwrap();
}
async fn create_on_columns_test_table(
instance: &Arc<Instance>,
table_name: &str,
query_ctx: QueryContextRef,
) {
let sql = format!(
r#"
CREATE TABLE `{table_name}`(
`id` INT,
`city` STRING,
`ts` TIMESTAMP TIME INDEX,
PRIMARY KEY(`id`, `city`)
) PARTITION ON COLUMNS (`id`) (
`id` < 10,
`id` >= 10 AND `id` < 20,
`id` >= 20
) ENGINE = mito;
"#
);
run_sql(instance, &sql, query_ctx).await.unwrap();
}
async fn insert_on_columns_old_rows(
instance: &Arc<Instance>,
table_name: &str,
query_ctx: QueryContextRef,
) {
let sql = format!(
r#"
INSERT INTO `{table_name}` VALUES
(1, 'London', '2022-01-01 00:00:00'),
(5, 'New York', '2022-01-01 00:00:00'),
(10, 'Paris', '2022-01-01 00:00:00'),
(15, 'Tokyo', '2022-01-01 00:00:00'),
(20, 'Beijing', '2022-01-01 00:00:00');
"#
);
run_sql(instance, &sql, query_ctx).await.unwrap();
}
async fn assert_on_columns_metadata_overwrite(
instance: &Arc<Instance>,
table_name: &str,
alter_sql: &str,
query_ctx: QueryContextRef,
) {
create_on_columns_test_table(instance, table_name, query_ctx.clone()).await;
run_sql(instance, alter_sql, query_ctx.clone())
.await
.unwrap();
tokio::time::sleep(Duration::from_millis(500)).await;
assert_partition_expression(instance, table_name, "id, city", query_ctx).await;
}
async fn assert_on_columns_rejects_removed_remaining_column(
instance: &Arc<Instance>,
table_name: &str,
alter_sql: &str,
query_ctx: QueryContextRef,
) {
create_on_columns_test_table(instance, table_name, query_ctx.clone()).await;
assert_sql_err_eq(
instance,
alter_sql,
"Invalid partition rule: partition expression references column 'id' that is not in target partition columns",
query_ctx,
)
.await;
}
async fn assert_on_columns_data_correctness(
instance: &Arc<Instance>,
table_name: &str,
alter_sql: &str,
query_ctx: QueryContextRef,
) {
create_on_columns_test_table(instance, table_name, query_ctx.clone()).await;
insert_on_columns_old_rows(instance, table_name, query_ctx.clone()).await;
run_sql(instance, alter_sql, query_ctx.clone())
.await
.unwrap();
tokio::time::sleep(Duration::from_millis(500)).await;
let result = run_sql(
instance,
&format!("SELECT * FROM `{table_name}` ORDER BY `id`, `city`"),
query_ctx.clone(),
)
.await
.unwrap();
let expected = "\
+----+----------+---------------------+
| id | city | ts |
+----+----------+---------------------+
| 1 | London | 2022-01-01T00:00:00 |
| 5 | New York | 2022-01-01T00:00:00 |
| 10 | Paris | 2022-01-01T00:00:00 |
| 15 | Tokyo | 2022-01-01T00:00:00 |
| 20 | Beijing | 2022-01-01T00:00:00 |
+----+----------+---------------------+";
check_output_stream(result.data, expected).await;
let sql = format!(
r#"
INSERT INTO `{table_name}` VALUES
(2, 'Amsterdam', '2022-01-02 00:00:00'),
(7, 'Zurich', '2022-01-02 00:00:00'),
(12, 'Berlin', '2022-01-02 00:00:00');
"#
);
run_sql(instance, &sql, query_ctx.clone()).await.unwrap();
let result = run_sql(
instance,
&format!("SELECT * FROM `{table_name}` ORDER BY `id`, `city`"),
query_ctx,
)
.await
.unwrap();
let expected = "\
+----+-----------+---------------------+
| id | city | ts |
+----+-----------+---------------------+
| 1 | London | 2022-01-01T00:00:00 |
| 2 | Amsterdam | 2022-01-02T00:00:00 |
| 5 | New York | 2022-01-01T00:00:00 |
| 7 | Zurich | 2022-01-02T00:00:00 |
| 10 | Paris | 2022-01-01T00:00:00 |
| 12 | Berlin | 2022-01-02T00:00:00 |
| 15 | Tokyo | 2022-01-01T00:00:00 |
| 20 | Beijing | 2022-01-01T00:00:00 |
+----+-----------+---------------------+";
check_output_stream(result.data, expected).await;
}
fn repartition_on_columns_sql(table_name: &str) -> String {
format!(
r#"
ALTER TABLE `{table_name}` REPARTITION (
`id` < 10
) ON COLUMNS (`id`, `city`) INTO (
`id` < 10 AND `city` < 'M',
`id` < 10 AND `city` >= 'M'
);
"#
)
}
fn split_on_columns_sql(table_name: &str) -> String {
format!(
r#"
ALTER TABLE `{table_name}` SPLIT PARTITION (
`id` < 10
) ON COLUMNS (`id`, `city`) INTO (
`id` < 10 AND `city` < 'M',
`id` < 10 AND `city` >= 'M'
);
"#
)
}
fn repartition_on_columns_removed_column_sql(table_name: &str) -> String {
format!(
r#"
ALTER TABLE `{table_name}` REPARTITION (
`id` < 10
) ON COLUMNS (`city`) INTO (
`city` < 'M',
`city` >= 'M'
);
"#
)
}
fn split_on_columns_removed_column_sql(table_name: &str) -> String {
format!(
r#"
ALTER TABLE `{table_name}` SPLIT PARTITION (
`id` < 10
) ON COLUMNS (`city`) INTO (
`city` < 'M',
`city` >= 'M'
);
"#
)
}
async fn assert_partition_expression(
instance: &Arc<Instance>,
table_name: &str,
expected_partition_expression: &str,
query_ctx: QueryContextRef,
) {
let result = run_sql(
instance,
&format!(
"SELECT DISTINCT partition_expression FROM information_schema.partitions WHERE table_name = '{table_name}' ORDER BY partition_expression"
),
query_ctx,
)
.await
.unwrap();
let expected = format!(
"\
+----------------------+
| partition_expression |
+----------------------+
| {expected_partition_expression:<20} |
+----------------------+"
);
check_output_stream(result.data, &expected).await;
}
async fn assert_sql_err_eq(
instance: &Arc<Instance>,
sql: &str,
expected: &str,
query_ctx: QueryContextRef,
) {
let err = run_sql(instance, sql, query_ctx).await.unwrap_err();
let root = root_source(&err).unwrap_or(&err);
assert_eq!(root.to_string(), expected);
}
async fn run_sql(
instance: &Arc<Instance>,
sql: &str,
query_ctx: QueryContextRef,
) -> ServerResult<Output> {
info!("Run SQL: {sql}");
instance.do_query(sql, query_ctx).await.remove(0)
}