From 00eeff9b8d170c925821d051a32f9d38fba0ca69 Mon Sep 17 00:00:00 2001 From: Erik Grinaker Date: Wed, 16 Apr 2025 16:41:02 +0200 Subject: [PATCH] pageserver: add `compaction_shard_ancestor` to disable shard ancestor compaction (#11608) ## Problem Splits of large tenants (several TB) can cause a huge amount of shard ancestor compaction work, which can overload Pageservers. Touches https://github.com/neondatabase/cloud/issues/22532. ## Summary of changes Add a setting `compaction_shard_ancestor` (default `true`) to disable shard ancestor compaction on a per-tenant basis. --- control_plane/src/pageserver.rs | 5 +++++ libs/pageserver_api/src/config.rs | 4 ++++ libs/pageserver_api/src/models.rs | 13 +++++++++++++ pageserver/src/tenant/timeline.rs | 8 ++++++++ pageserver/src/tenant/timeline/compaction.rs | 3 +-- test_runner/regress/test_attach_tenant_config.py | 1 + 6 files changed, 32 insertions(+), 2 deletions(-) diff --git a/control_plane/src/pageserver.rs b/control_plane/src/pageserver.rs index 5c985e6dc8..b9257a27bf 100644 --- a/control_plane/src/pageserver.rs +++ b/control_plane/src/pageserver.rs @@ -413,6 +413,11 @@ impl PageServerNode { .map(serde_json::from_str) .transpose() .context("Failed to parse 'compaction_algorithm' json")?, + compaction_shard_ancestor: settings + .remove("compaction_shard_ancestor") + .map(|x| x.parse::()) + .transpose() + .context("Failed to parse 'compaction_shard_ancestor' as a bool")?, compaction_l0_first: settings .remove("compaction_l0_first") .map(|x| x.parse::()) diff --git a/libs/pageserver_api/src/config.rs b/libs/pageserver_api/src/config.rs index 53b68afb0f..e734b07c38 100644 --- a/libs/pageserver_api/src/config.rs +++ b/libs/pageserver_api/src/config.rs @@ -379,6 +379,8 @@ pub struct TenantConfigToml { /// size exceeds `compaction_upper_limit * checkpoint_distance`. pub compaction_upper_limit: usize, pub compaction_algorithm: crate::models::CompactionAlgorithmSettings, + /// If true, enable shard ancestor compaction (enabled by default). + pub compaction_shard_ancestor: bool, /// If true, compact down L0 across all tenant timelines before doing regular compaction. L0 /// compaction must be responsive to avoid read amp during heavy ingestion. Defaults to true. pub compaction_l0_first: bool, @@ -677,6 +679,7 @@ pub mod tenant_conf_defaults { pub const DEFAULT_COMPACTION_PERIOD: &str = "20 s"; pub const DEFAULT_COMPACTION_THRESHOLD: usize = 10; + pub const DEFAULT_COMPACTION_SHARD_ANCESTOR: bool = true; // This value needs to be tuned to avoid OOM. We have 3/4*CPUs threads for L0 compaction, that's // 3/4*16=9 on most of our pageservers. Compacting 20 layers requires about 1 GB memory (could @@ -734,6 +737,7 @@ impl Default for TenantConfigToml { compaction_algorithm: crate::models::CompactionAlgorithmSettings { kind: DEFAULT_COMPACTION_ALGORITHM, }, + compaction_shard_ancestor: DEFAULT_COMPACTION_SHARD_ANCESTOR, compaction_l0_first: DEFAULT_COMPACTION_L0_FIRST, compaction_l0_semaphore: DEFAULT_COMPACTION_L0_SEMAPHORE, l0_flush_delay_threshold: None, diff --git a/libs/pageserver_api/src/models.rs b/libs/pageserver_api/src/models.rs index f491ed10e1..ea5456e04b 100644 --- a/libs/pageserver_api/src/models.rs +++ b/libs/pageserver_api/src/models.rs @@ -526,6 +526,8 @@ pub struct TenantConfigPatch { #[serde(skip_serializing_if = "FieldPatch::is_noop")] pub compaction_algorithm: FieldPatch, #[serde(skip_serializing_if = "FieldPatch::is_noop")] + pub compaction_shard_ancestor: FieldPatch, + #[serde(skip_serializing_if = "FieldPatch::is_noop")] pub compaction_l0_first: FieldPatch, #[serde(skip_serializing_if = "FieldPatch::is_noop")] pub compaction_l0_semaphore: FieldPatch, @@ -615,6 +617,9 @@ pub struct TenantConfig { #[serde(skip_serializing_if = "Option::is_none")] pub compaction_algorithm: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub compaction_shard_ancestor: Option, + #[serde(skip_serializing_if = "Option::is_none")] pub compaction_l0_first: Option, @@ -724,6 +729,7 @@ impl TenantConfig { mut compaction_threshold, mut compaction_upper_limit, mut compaction_algorithm, + mut compaction_shard_ancestor, mut compaction_l0_first, mut compaction_l0_semaphore, mut l0_flush_delay_threshold, @@ -772,6 +778,9 @@ impl TenantConfig { .compaction_upper_limit .apply(&mut compaction_upper_limit); patch.compaction_algorithm.apply(&mut compaction_algorithm); + patch + .compaction_shard_ancestor + .apply(&mut compaction_shard_ancestor); patch.compaction_l0_first.apply(&mut compaction_l0_first); patch .compaction_l0_semaphore @@ -860,6 +869,7 @@ impl TenantConfig { compaction_threshold, compaction_upper_limit, compaction_algorithm, + compaction_shard_ancestor, compaction_l0_first, compaction_l0_semaphore, l0_flush_delay_threshold, @@ -920,6 +930,9 @@ impl TenantConfig { .as_ref() .unwrap_or(&global_conf.compaction_algorithm) .clone(), + compaction_shard_ancestor: self + .compaction_shard_ancestor + .unwrap_or(global_conf.compaction_shard_ancestor), compaction_l0_first: self .compaction_l0_first .unwrap_or(global_conf.compaction_l0_first), diff --git a/pageserver/src/tenant/timeline.rs b/pageserver/src/tenant/timeline.rs index 613834dc88..bc54c85119 100644 --- a/pageserver/src/tenant/timeline.rs +++ b/pageserver/src/tenant/timeline.rs @@ -2702,6 +2702,14 @@ impl Timeline { .clone() } + pub fn get_compaction_shard_ancestor(&self) -> bool { + let tenant_conf = self.tenant_conf.load(); + tenant_conf + .tenant_conf + .compaction_shard_ancestor + .unwrap_or(self.conf.default_tenant_conf.compaction_shard_ancestor) + } + fn get_eviction_policy(&self) -> EvictionPolicy { let tenant_conf = self.tenant_conf.load(); tenant_conf diff --git a/pageserver/src/tenant/timeline/compaction.rs b/pageserver/src/tenant/timeline/compaction.rs index 76c153d60f..92b24a73c9 100644 --- a/pageserver/src/tenant/timeline/compaction.rs +++ b/pageserver/src/tenant/timeline/compaction.rs @@ -1239,8 +1239,7 @@ impl Timeline { let partition_count = self.partitioning.read().0.0.parts.len(); // 4. Shard ancestor compaction - - if self.shard_identity.count >= ShardCount::new(2) { + if self.get_compaction_shard_ancestor() && self.shard_identity.count >= ShardCount::new(2) { // Limit the number of layer rewrites to the number of partitions: this means its // runtime should be comparable to a full round of image layer creations, rather than // being potentially much longer. diff --git a/test_runner/regress/test_attach_tenant_config.py b/test_runner/regress/test_attach_tenant_config.py index 9b6930695c..ee408e3c65 100644 --- a/test_runner/regress/test_attach_tenant_config.py +++ b/test_runner/regress/test_attach_tenant_config.py @@ -155,6 +155,7 @@ def test_fully_custom_config(positive_env: NeonEnv): "compaction_algorithm": { "kind": "tiered", }, + "compaction_shard_ancestor": False, "eviction_policy": { "kind": "LayerAccessThreshold", "period": "20s",