Files
greptimedb/tests-integration/tests/gc_event.rs
T
dennis zhuang f3eb8e6c72 test: cut integration test time and make the storage matrix meaningful (#9308)
* test: cut integration test time and make the storage matrix meaningful

tests-integration is ~85% of workspace test CPU, and 81% of that is the
S3/S3WithCache variants of the HTTP and gRPC suites. Those suites do not
touch the object store: of the 70 matrix HTTP tests only one flushed and
read back an SST, so the matrix was paying real AWS round trips to
re-prove protocol parsing.

- Point the PR CI object-store matrix at the MinIO already started by
  tests-integration/fixtures. Three GT_S3_* consumers did not read
  GT_S3_ENDPOINT_URL and would have hit real AWS with MinIO credentials;
  they now do.
- Add a nightly Linux job against real AWS S3, and pass GT_S3_* into the
  release integration-test container. The release previously ran every
  remote-backend case as a skip and only exercised the file backend.
- Give each S3WithCache test its own read cache directory. They shared
  /tmp/greptimedb_cache, which the datanode wipes on startup, so a
  starting test deleted the read cache of a running one.
- Add flush -> read-back assertions to the tests whose columns have a
  non-trivial SST representation: JSON/JSON2 columns, native histograms,
  metric-engine logical tables, and tables carrying fulltext or skipping
  indexes whose puffin files only exist after a flush.
- Move eight tests that create no table out of the storage matrix.
- Make the event recorder flush interval a constructor parameter and
  shorten it in the event tests, which otherwise wait a 5s window per DDL
  they assert on. It is skipped by serde and never reaches config files.
- Drop duplicates: test_grpc_zstd_compression was a verbatim copy of
  test_grpc_message_size_ok and is now rewritten to assert the negotiated
  grpc-encoding; test_execute_copy_to_{s3,oss,gcs,azblob} were strict
  prefixes of their copy_from siblings; two standalone/distributed event
  test pairs shared one assertion body.
- Fix and un-ignore stddev_by_label. stddev_pop merges partial aggregates
  in a parallelism-dependent order, so its last digits are unstable; the
  test now compares values with a tolerance.
- Rebase the jaeger v1 fixture on the current instant. It carries
  ttl=7d with 2025 timestamps, so its rows were only readable as long as
  they stayed in the memtable.

Signed-off-by: Dennis Zhuang <killme2008@gmail.com>

* test: address review — wire nightly real-S3 job into check-status, keep the short event interval

The nightly `check-status` job did not depend on the new real-S3 job, so a
failure there would not have reached the status or Slack notification.

In database_ddl_event the short interval was set by a first
`with_event_recorder_options` call and then overwritten by the pre-existing
one, which carries `..Default::default()`. Merged into a single call.

Signed-off-by: Dennis Zhuang <killme2008@gmail.com>

---------

Signed-off-by: Dennis Zhuang <killme2008@gmail.com>
2026-09-23 01:52:52 +00:00

210 lines
6.7 KiB
Rust

// Copyright 2023 Greptime Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use std::sync::Arc;
use std::time::Duration;
use common_test_util::temp_dir::create_temp_dir;
use meta_srv::gc::GcSchedulerOptions;
use mito2::gc::GcConfig;
use servers::query_handler::sql::SqlQueryHandler;
use session::context::QueryContext;
use tests_integration::cluster::{GreptimeDbCluster, GreptimeDbClusterBuilder};
use tests_integration::test_util::{
StorageType, get_test_store_config, setup_authenticated_grpc_database,
};
use crate::event_recorder_test_util::{assert_procedure_actor, find_eventually_string};
const TABLE_NAME: &str = "batch_gc_events";
const PROCEDURE_ACTOR: &str = "procedure_actor";
const PROCEDURE_ACTOR_PASSWORD: &str = "procedure_actor_pwd";
const MAX_ATTEMPTS: usize = 60;
const POLL_INTERVAL: Duration = Duration::from_millis(250);
#[tokio::test(flavor = "multi_thread")]
async fn test_batch_gc_event() {
let store_type = StorageType::File;
if !store_type.test_on() {
return;
}
common_telemetry::init_default_ut_logging();
let (store_config, _guard) = get_test_store_config(&store_type);
let home_dir = create_temp_dir("test_batch_gc_event_data_home");
// Keeps the production event-recorder flush interval: `assert_sst_count`
// counts SST files across every table, so eagerly flushed event rows would
// show up in the count.
let cluster = GreptimeDbClusterBuilder::new("test_batch_gc_event")
.await
.with_datanodes(1)
.with_store_config(store_config)
.with_shared_home_dir(Arc::new(home_dir))
.with_metasrv_gc_config(GcSchedulerOptions {
enable: true,
..Default::default()
})
.with_datanode_gc_config(GcConfig {
enable: true,
lingering_time: Some(Duration::ZERO),
unknown_file_lingering_time: Duration::ZERO,
..Default::default()
})
.build(true)
.await;
let instance = cluster.fe_instance();
execute(
instance,
&format!(
"CREATE TABLE {TABLE_NAME} (\
ts TIMESTAMP TIME INDEX, \
val DOUBLE, \
host STRING\
) WITH (append_mode = 'true')"
),
)
.await;
for day in 1..=4 {
execute(
instance,
&format!(
"INSERT INTO {TABLE_NAME} VALUES \
('2023-01-{day:02} 10:00:00', {day}.0, 'host{day}')"
),
)
.await;
execute(instance, &format!("ADMIN FLUSH_TABLE('{TABLE_NAME}')")).await;
}
assert_sst_count(&cluster, 4).await;
let mut deleted_file_ids = cluster
.list_sst_files_from_all_datanodes()
.await
.into_iter()
.map(|path| {
path.rsplit('/')
.next()
.unwrap()
.strip_suffix(".parquet")
.unwrap()
.to_string()
})
.collect::<Vec<_>>();
execute(instance, &format!("ADMIN COMPACT_TABLE('{TABLE_NAME}')")).await;
assert_sst_count(&cluster, 5).await;
let table = instance
.catalog_manager()
.table("greptime", "public", TABLE_NAME, None)
.await
.unwrap()
.unwrap();
let table_id = table.table_info().table_id();
let (_, route) = cluster
.metasrv
.table_metadata_manager()
.table_route_manager()
.get_physical_table_route(table_id)
.await
.unwrap();
let regions = route
.region_routes
.iter()
.map(|route| route.region.id)
.collect::<Vec<_>>();
assert_eq!(regions.len(), 1);
let (actor_db, _actor_grpc_server) = setup_authenticated_grpc_database(
instance.clone(),
PROCEDURE_ACTOR,
PROCEDURE_ACTOR_PASSWORD,
)
.await;
actor_db
.sql(format!("ADMIN GC_REGIONS({})", regions[0].as_u64()))
.await
.unwrap();
assert_sst_count(&cluster, 1).await;
let procedure_id = find_eventually_string(
instance,
&format!(
"SELECT procedure_id FROM greptime_private.events WHERE type = 'batch_gc' AND actor = '{PROCEDURE_ACTOR}' AND json_get_string(procedure_trigger, 'type') = 'Submitted'"
),
"procedure_id",
)
.await;
assert_procedure_actor(instance, &procedure_id, Some(PROCEDURE_ACTOR)).await;
let submitted = format!(
r#"SELECT json_to_string(event_context) AS event_context
FROM greptime_private.events
WHERE type = 'batch_gc'
AND procedure_id = '{procedure_id}'
AND procedure_state = 'Running'
AND json_get_string(procedure_trigger, 'type') = 'Submitted'"#,
);
assert_eq!(
r#"{"protocol":"grpc","reason":"manual"}"#,
find_eventually_string(instance, &submitted, "event_context").await
);
let succeeded = format!(
r#"SELECT json_to_string(gc_report) AS gc_report
FROM greptime_private.events
WHERE type = 'batch_gc'
AND procedure_id = '{procedure_id}'
AND procedure_state = 'Done'
AND json_path_match(procedure_trigger, '$.type == "Succeeded"')
AND json_is_null(payload)
AND event_context IS NULL"#,
);
let actual_report: serde_json::Value =
serde_json::from_str(&find_eventually_string(instance, &succeeded, "gc_report").await)
.unwrap();
assert_eq!(actual_report["need_retry"], false);
let mut actual_file_ids = actual_report["deleted_files"]
.as_array()
.unwrap()
.iter()
.map(|file_id| file_id.as_str().unwrap())
.collect::<Vec<_>>();
actual_file_ids.sort_unstable();
deleted_file_ids.sort_unstable();
assert_eq!(actual_file_ids, deleted_file_ids);
}
async fn execute(instance: &Arc<frontend::instance::Instance>, sql: &str) {
instance
.do_query(sql, QueryContext::arc())
.await
.remove(0)
.unwrap();
}
async fn assert_sst_count(cluster: &GreptimeDbCluster, expected: usize) {
let mut last_actual = 0;
for _ in 0..MAX_ATTEMPTS {
last_actual = cluster.list_sst_files_from_all_datanodes().await.len();
if last_actual == expected {
return;
}
tokio::time::sleep(POLL_INTERVAL).await;
}
panic!("timed out waiting for {expected} SST files, found {last_actual}");
}